@cspeach/cli 1.1.19 → 1.1.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/agent/anthropic-provider.js +30 -10
  2. package/dist/agent/cache-keepalive.js +162 -0
  3. package/dist/agent/cold-prune.js +116 -0
  4. package/dist/agent/loop.js +675 -148
  5. package/dist/agent/provider-shape.js +263 -0
  6. package/dist/agent/providers/ai-hub-provider.js +17 -2
  7. package/dist/agent/providers/byok-provider.js +33 -3
  8. package/dist/agent/providers/local-provider.js +8 -1
  9. package/dist/agent/repair-partial.js +66 -5
  10. package/dist/agent/summarise-via-provider.js +6 -1
  11. package/dist/agent/system-prompt.js +38 -0
  12. package/dist/agent/tool-dispatch.js +8 -0
  13. package/dist/agent/tool-loading-pin.js +100 -0
  14. package/dist/cli.js +11 -0
  15. package/dist/commands/auto-compact.js +33 -16
  16. package/dist/commands/compact.js +37 -2
  17. package/dist/commands/config-set.js +10 -1
  18. package/dist/commands/config-show.js +11 -0
  19. package/dist/commands/cost.js +14 -2
  20. package/dist/commands/plan-audit.js +1 -0
  21. package/dist/config/loader.js +55 -2
  22. package/dist/cost/cost-log.js +62 -2
  23. package/dist/cost/pricing.js +6 -2
  24. package/dist/lib/spill-labels.js +13 -0
  25. package/dist/models/resolve.js +93 -2
  26. package/dist/models/server-config.js +158 -3
  27. package/dist/one-shot.js +15 -5
  28. package/dist/projects/image-attachments.js +15 -2
  29. package/dist/renderer/footer-line.js +6 -2
  30. package/dist/renderer/startup-lines.js +5 -3
  31. package/dist/renderer/tool-labels.js +33 -2
  32. package/dist/renderer/ui-width.js +13 -0
  33. package/dist/repl/current-transport.js +13 -0
  34. package/dist/repl/post-turn-status.js +8 -1
  35. package/dist/repl.js +71 -10
  36. package/dist/session/repin-model.js +18 -0
  37. package/dist/session/store.js +16 -2
  38. package/dist/skills/bundled-skills.js +1 -1
  39. package/dist/skills/preamble.js +75 -0
  40. package/dist/skills/source-manifest.js +11 -1
  41. package/dist/tools/filesystem/file-read.js +11 -1
  42. package/dist/tools/result-spill.js +238 -0
  43. package/dist/tools/sap-read.js +58 -14
  44. package/dist/tools/shell/shell_exec.js +9 -0
  45. package/dist/tools/subagent/adt-serial.js +33 -0
  46. package/dist/tools/subagent/agent_run.js +2 -0
  47. package/dist/tools/subagent/read_agent.js +178 -0
  48. package/dist/tools/subagent/reader-prompt.js +48 -0
  49. package/dist/tools/todo.js +3 -1
  50. package/dist/tools/tool-loading.js +255 -0
  51. package/dist/tools/tool-output-read.js +117 -0
  52. package/dist/tools/transport.js +6 -1
  53. package/dist/ui/footer.js +5 -5
  54. package/dist/ui/sap-state-store.js +1 -1
  55. package/dist/ui/turn-status-emitter.js +37 -0
  56. package/dist/ui/turn-status.js +1 -1
  57. package/package.json +2 -1
@@ -0,0 +1,263 @@
1
+ import { TOOL_CHANGES_BETA, isToolSearchEntry } from '../tools/tool-loading.js';
2
+ /** Betas the proxy sends on every managed turn (messages.ts `betas: [...]`). */
3
+ export const BASE_BETAS = ['compact-2026-01-12'];
4
+ /** Task 6 — beta that enables `fallbacks: 'default'` (server-side fallback). */
5
+ export const FALLBACK_BETA = 'server-side-fallback-2026-07-01';
6
+ /**
7
+ * Controller ruling (Task 6): an ALLOW-list of the families probe P3 proved
8
+ * accept `fallbacks: 'default'`. Exact match after lowercasing and stripping a
9
+ * trailing -YYYYMMDD — so claude-opus-4-8, a future claude-opus-5-9, or a local
10
+ * alias get no fallbacks and no fallback beta. Mirrors the proxy's
11
+ * models/config.ts modelAcceptsFallbacks.
12
+ */
13
+ export const FALLBACK_MODELS = new Set([
14
+ 'claude-opus-5-5', 'claude-opus-5', 'claude-sonnet-4-6', 'claude-haiku-4-5',
15
+ ]);
16
+ export function modelAcceptsFallbacks(model) {
17
+ return FALLBACK_MODELS.has(model.trim().toLowerCase().replace(/-\d{8}$/, ''));
18
+ }
19
+ /**
20
+ * Task 7 — beta that enables `thinking.block_binding` (preserved thinking:
21
+ * `prefix_mismatch_behavior: 'drop_block'` drops a thinking block whose prefix
22
+ * changed instead of rejecting the request).
23
+ */
24
+ export const BINDING_BETA = 'thinking-binding-controls-2026-08-01';
25
+ /**
26
+ * Task 7 — ALLOW-list of the families probe P4 proved accept
27
+ * `thinking.block_binding` with BINDING_BETA. Same matching as
28
+ * FALLBACK_MODELS (lowercased, trailing -YYYYMMDD stripped): claude-haiku-4-5*
29
+ * and claude-opus-4-8 were not probed for it, so they get neither the field
30
+ * nor the beta. Mirrors the proxy's models/config.ts modelAcceptsBinding.
31
+ */
32
+ export const BINDING_MODELS = new Set([
33
+ 'claude-opus-5-5', 'claude-opus-5', 'claude-sonnet-4-6',
34
+ ]);
35
+ export function modelAcceptsBinding(model) {
36
+ return BINDING_MODELS.has(model.trim().toLowerCase().replace(/-\d{8}$/, ''));
37
+ }
38
+ /**
39
+ * Task 8 (ruling F5, spec D9) — ALLOW-list of the families probe P2 proved
40
+ * accept `max_tokens: 65536` (claude-opus-5, claude-opus-5-5). Same matching
41
+ * as FALLBACK_MODELS (lowercased, trailing -YYYYMMDD stripped), so a future
42
+ * claude-opus-5-9 keeps the 32768 default until it is probed.
43
+ */
44
+ export const MAX_TOKENS_65K_MODELS = new Set([
45
+ 'claude-opus-5-5', 'claude-opus-5',
46
+ ]);
47
+ export const SESSION_MAX_TOKENS_DEFAULT = 32768;
48
+ export const SESSION_MAX_TOKENS_OPUS5 = 65536;
49
+ export function sessionMaxTokens(model) {
50
+ return MAX_TOKENS_65K_MODELS.has(model.trim().toLowerCase().replace(/-\d{8}$/, ''))
51
+ ? SESSION_MAX_TOKENS_OPUS5
52
+ : SESSION_MAX_TOKENS_DEFAULT;
53
+ }
54
+ /**
55
+ * Integration finding I1 (2026-09-28) — the per-model OUTPUT CEILING every
56
+ * request's max_tokens is clamped to, maxTokensOverride included (agent_run
57
+ * children default to 100000). Only the 65k allow-list above is proven;
58
+ * every other model (claude-opus-4-8, claude-sonnet-4-6, claude-haiku-4-5,
59
+ * unknown ids) is capped at 32768. Dated ids are stripped. Without this an
60
+ * enforced sonnet/haiku child sent 100000 and the API answered 400.
61
+ */
62
+ export function modelOutputCeiling(model) {
63
+ return sessionMaxTokens(model);
64
+ }
65
+ export function clampMaxTokens(model, requested) {
66
+ return Math.min(requested, modelOutputCeiling(model));
67
+ }
68
+ /**
69
+ * Task 16 — the beta for `role: 'system'` `tool_addition` messages (sent
70
+ * whenever a request declares a deferred tool or carries a system message)
71
+ * and the tool search predicate: one definition each, in tools/tool-loading.
72
+ */
73
+ export { TOOL_CHANGES_BETA, isToolSearchEntry };
74
+ function isDeferredEntry(t) {
75
+ return !!t && typeof t === 'object' && t.defer_loading === true;
76
+ }
77
+ /** Task 16 — a tool the request loads in full (neither deferred nor the search tool). */
78
+ export function isLoadedToolEntry(t) {
79
+ return !isDeferredEntry(t) && !isToolSearchEntry(t);
80
+ }
81
+ function hasSystemMessage(messages) {
82
+ return (messages ?? []).some((m) => m?.role === 'system');
83
+ }
84
+ /**
85
+ * Betas matching the fields present in a shaped request. The beta follows the
86
+ * field: the model is gated where the field is set (`fallbacks` in BYOK
87
+ * shaping, `thinking.block_binding` in the loop), not here.
88
+ */
89
+ export function requiredBetas(params) {
90
+ const betas = [...BASE_BETAS];
91
+ if (params.fallbacks)
92
+ betas.push(FALLBACK_BETA);
93
+ if (params.thinking?.block_binding)
94
+ betas.push(BINDING_BETA);
95
+ if ((params.tools ?? []).some(isDeferredEntry) || hasSystemMessage(params.messages))
96
+ betas.push(TOOL_CHANGES_BETA);
97
+ return betas;
98
+ }
99
+ const EPHEMERAL = { type: 'ephemeral' };
100
+ /**
101
+ * Add the system prompt and (optionally) the proxy's two remaining cache
102
+ * breakpoints: the last tool and the last block of the last message. Never
103
+ * mutates the input — the loop's session messages are reused next turn.
104
+ * A caller-supplied `system` is replaced, as the proxy ignores body.system.
105
+ */
106
+ export function withSkillPrompt(params, system, opts) {
107
+ const out = { ...params, system };
108
+ if (!opts.cacheMarkers)
109
+ return out;
110
+ // Task 16 — the marker goes on the last LOADED tool (loaded entries come
111
+ // first, deferred ones and the search tool after); with no deferred entries
112
+ // that is the last tool, as before.
113
+ if (Array.isArray(params.tools) && params.tools.length > 0) {
114
+ const tools = [...params.tools];
115
+ const markIdx = lastIndexWhere(tools, isLoadedToolEntry);
116
+ if (markIdx >= 0) {
117
+ tools[markIdx] = { ...tools[markIdx], cache_control: EPHEMERAL };
118
+ out.tools = tools;
119
+ }
120
+ }
121
+ // Task 16 — the marker goes on the last NON-system message: a trailing
122
+ // `role: 'system'` tool_addition is never marked (probe P7 proved this
123
+ // shape; a marker on a tool_addition block is unproven).
124
+ const lastIdx = lastIndexWhere(params.messages, (m) => m?.role !== 'system');
125
+ if (lastIdx >= 0) {
126
+ const messages = [...params.messages];
127
+ const last = { ...messages[lastIdx] };
128
+ if (typeof last.content === 'string') {
129
+ last.content = [{ type: 'text', text: last.content, cache_control: EPHEMERAL }];
130
+ }
131
+ else if (Array.isArray(last.content) && last.content.length > 0) {
132
+ const blocks = [...last.content];
133
+ blocks[blocks.length - 1] = { ...blocks[blocks.length - 1], cache_control: EPHEMERAL };
134
+ last.content = blocks;
135
+ }
136
+ messages[lastIdx] = last;
137
+ out.messages = messages;
138
+ }
139
+ return out;
140
+ }
141
+ function lastIndexWhere(arr, pred) {
142
+ for (let i = arr.length - 1; i >= 0; i--)
143
+ if (pred(arr[i]))
144
+ return i;
145
+ return -1;
146
+ }
147
+ const LEGACY_TOP_LEVEL = ['model', 'max_tokens', 'messages', 'system', 'tools', 'tool_choice', 'metadata'];
148
+ const LEGACY_THINKING = ['type', 'display'];
149
+ /** Task 16 — server tool blocks (tool search) the legacy wires never saw. */
150
+ const SERVER_TOOL_BLOCKS = new Set(['server_tool_use', 'tool_search_tool_result']);
151
+ /** A shallow copy of an object block without `cache_control`. */
152
+ function withoutCacheControl(b) {
153
+ if (!b || typeof b !== 'object' || !('cache_control' in b))
154
+ return b;
155
+ const { cache_control: _drop, ...rest } = b;
156
+ return rest;
157
+ }
158
+ /** Task 16 — legacy tools: every tool loads (no defer_loading), no search entry, no markers. */
159
+ function legacyTools(tools) {
160
+ return tools
161
+ .filter((t) => !isToolSearchEntry(t))
162
+ .map((t) => {
163
+ const copy = withoutCacheControl(t);
164
+ if (!isDeferredEntry(copy))
165
+ return copy;
166
+ const { defer_loading: _drop, ...rest } = copy;
167
+ return rest;
168
+ });
169
+ }
170
+ /**
171
+ * Task 16 — legacy messages: no `role: 'system'` messages, no server tool
172
+ * blocks (an assistant message left empty by that is dropped), no nested
173
+ * `cache_control` (Task 5 carry).
174
+ */
175
+ function legacyMessages(messages) {
176
+ const out = [];
177
+ for (const m of messages) {
178
+ if (m.role === 'system')
179
+ continue;
180
+ if (!Array.isArray(m.content)) {
181
+ out.push(m);
182
+ continue;
183
+ }
184
+ const content = m.content
185
+ .filter((b) => !SERVER_TOOL_BLOCKS.has(b?.type ?? ''))
186
+ .map(withoutCacheControl);
187
+ if (content.length === 0)
188
+ continue;
189
+ out.push({ ...m, content });
190
+ }
191
+ return out;
192
+ }
193
+ /**
194
+ * Whitelist copy of the fields AI-hub and local accepted before this plan.
195
+ * Every newer field (output_config, fallbacks, betas, thinking.block_binding,
196
+ * …) is dropped, and so are the Task 16 shapes (deferred tools load in full,
197
+ * the search tool, system messages, server tool blocks) and any nested
198
+ * cache_control. Never deletes from the input.
199
+ */
200
+ export function toLegacyShape(params) {
201
+ const src = params;
202
+ const out = {};
203
+ for (const k of LEGACY_TOP_LEVEL) {
204
+ if (src[k] !== undefined)
205
+ out[k] = src[k];
206
+ }
207
+ if (Array.isArray(params.tools))
208
+ out.tools = legacyTools(params.tools);
209
+ if (Array.isArray(params.messages))
210
+ out.messages = legacyMessages(params.messages);
211
+ if (Array.isArray(params.system))
212
+ out.system = params.system.map(withoutCacheControl);
213
+ if (params.thinking !== undefined) {
214
+ const t = params.thinking;
215
+ const thinking = {};
216
+ for (const k of LEGACY_THINKING) {
217
+ if (t[k] !== undefined)
218
+ thinking[k] = t[k];
219
+ }
220
+ out.thinking = thinking;
221
+ }
222
+ return out;
223
+ }
224
+ /**
225
+ * Task 9 (Task 5 carry) — headers only our proxy understands. On a direct
226
+ * path (BYOK talks to the public API) they would leak CSPeach plumbing to a
227
+ * third party, so they are removed before the SDK sees them. Matching is
228
+ * case-insensitive: X-CSForge-* and X-CSPeach-*.
229
+ */
230
+ export function withoutInternalHeaders(headers) {
231
+ const out = {};
232
+ for (const [k, v] of Object.entries(headers ?? {})) {
233
+ if (/^x-cs(forge|peach)-/i.test(k))
234
+ continue;
235
+ out[k] = v;
236
+ }
237
+ return out;
238
+ }
239
+ /**
240
+ * Task 9 — the role header is new proxy plumbing; AI-hub keeps today's header
241
+ * set, so only X-CSPeach-Model-Role is removed there.
242
+ */
243
+ export function withoutRoleHeader(headers) {
244
+ const out = {};
245
+ for (const [k, v] of Object.entries(headers ?? {})) {
246
+ if (k.toLowerCase() === 'x-cspeach-model-role')
247
+ continue;
248
+ out[k] = v;
249
+ }
250
+ return out;
251
+ }
252
+ /**
253
+ * The active skill travels as the X-CSForge-Skill header (coordination §1).
254
+ * Direct providers need it to fetch the skill body; without it the model
255
+ * would run with no skill prompt, so fail loudly instead.
256
+ */
257
+ export function requireSkillName(options, mode) {
258
+ const name = options.headers?.['X-CSForge-Skill'];
259
+ if (!name) {
260
+ throw new Error(`${mode} request is missing the X-CSForge-Skill header — cannot load the skill prompt for this turn.`);
261
+ }
262
+ return name;
263
+ }
@@ -2,6 +2,8 @@ import Anthropic from '@anthropic-ai/sdk';
2
2
  import keychain from '../../auth/keychain.js';
3
3
  import { createManifestSkillSource } from '../../skills/source-manifest.js';
4
4
  import { DEFAULT_MANIFEST_URL } from '../../skills/manifest-client.js';
5
+ import { requireSkillName, toLegacyShape, withoutRoleHeader } from '../provider-shape.js';
6
+ import { systemPromptForSkill } from '../system-prompt.js';
5
7
  const KEYCHAIN_SERVICE = 'cspeach.ai-hub';
6
8
  const KEYCHAIN_ACCOUNT = 'token';
7
9
  export function createAiHubProvider(cfg) {
@@ -16,14 +18,27 @@ export function createAiHubProvider(cfg) {
16
18
  mode: 'ai-hub',
17
19
  skillSource,
18
20
  async createStream(params, options = {}) {
21
+ // Task 5 (ruling F3) — AI-hub keeps today's field set only (newer
22
+ // fields such as output_config are stripped) and gains the preamble +
23
+ // skill body as `system`. No cache markers: the hub has no flag saying
24
+ // it accepts cache_control, so none is sent.
25
+ const skillName = requireSkillName(options, 'AI Hub');
19
26
  const token = await getToken();
27
+ // Task 18 — the `_reader` header uses the pinned reader prompt, no lookup.
28
+ const system = await systemPromptForSkill(skillName, skillSource, { cacheMarkers: false });
29
+ const request = {
30
+ ...toLegacyShape(params),
31
+ system,
32
+ model: modelAlias,
33
+ };
20
34
  const sdk = new Anthropic({
21
35
  apiKey: token,
22
36
  baseURL,
23
- defaultHeaders: { ...options.headers },
37
+ // Task 9 — the proxy-only role header is not forwarded to the hub.
38
+ defaultHeaders: withoutRoleHeader(options.headers),
24
39
  });
25
40
  // v0.6 — pass AbortSignal so hard-interrupt tears down the stream.
26
- return sdk.messages.stream({ ...params, model: modelAlias }, options.signal ? { signal: options.signal } : undefined);
41
+ return sdk.messages.stream(request, options.signal ? { signal: options.signal } : undefined);
27
42
  },
28
43
  async healthCheck() {
29
44
  const token = await keychain.getPassword(KEYCHAIN_SERVICE, KEYCHAIN_ACCOUNT);
@@ -2,23 +2,53 @@ import Anthropic from '@anthropic-ai/sdk';
2
2
  import keychain from '../../auth/keychain.js';
3
3
  import { createManifestSkillSource } from '../../skills/source-manifest.js';
4
4
  import { DEFAULT_MANIFEST_URL } from '../../skills/manifest-client.js';
5
+ import { modelAcceptsFallbacks, requiredBetas, requireSkillName, withoutInternalHeaders, withSkillPrompt } from '../provider-shape.js';
6
+ import { systemPromptForSkill } from '../system-prompt.js';
5
7
  const KEYCHAIN_SERVICE = 'cspeach.anthropic';
6
8
  const KEYCHAIN_ACCOUNT = 'api-key';
7
9
  // Cheapest Anthropic model available at doctor-probe time.
8
10
  const HEALTH_PROBE_MODEL = 'claude-haiku-4-6-latest';
9
11
  export function createByokProvider(_cfg) {
10
12
  const skillSource = createManifestSkillSource(DEFAULT_MANIFEST_URL);
13
+ // Task 5 (ruling F3) — the loop's request is provider-neutral; BYOK talks to
14
+ // the model directly, so it adds what the proxy adds for managed: preamble +
15
+ // skill body as `system`, cache markers, and the betas matching the fields
16
+ // present. Shared by createStream and keepalive (Task 15) so a ping warms
17
+ // exactly the prefix the real call wrote.
18
+ async function shapeRequest(params, options) {
19
+ const skillName = requireSkillName(options, 'BYOK');
20
+ const key = await getKey();
21
+ // Task 18 — the `_reader` header uses the pinned reader prompt, no lookup.
22
+ const system = await systemPromptForSkill(skillName, skillSource, { cacheMarkers: true });
23
+ // Task 6 — opt into server-side fallback, only for the families probe
24
+ // P3 proved (allow-list); requiredBetas adds its beta.
25
+ const prompted = withSkillPrompt(params, system, { cacheMarkers: true });
26
+ const shaped = modelAcceptsFallbacks(params.model)
27
+ ? { ...prompted, fallbacks: 'default' }
28
+ : prompted;
29
+ const betas = requiredBetas(shaped);
30
+ // Task 9 — CSPeach-internal headers never reach the public API.
31
+ const sdk = new Anthropic({ apiKey: key, defaultHeaders: withoutInternalHeaders(options.headers) });
32
+ return { sdk, body: { ...shaped, betas } };
33
+ }
11
34
  return {
12
35
  name: 'anthropic-byok',
13
36
  mode: 'byok',
14
37
  skillSource,
15
38
  async createStream(params, options = {}) {
16
- const key = await getKey();
17
- const sdk = new Anthropic({ apiKey: key, defaultHeaders: { ...options.headers } });
39
+ const { sdk, body } = await shapeRequest(params, options);
18
40
  // v0.6 — pass AbortSignal so the agent loop's per-turn AbortController
19
41
  // can tear down the stream on Esc / SIGINT-during-turn. The Anthropic
20
42
  // SDK's `signal` option propagates to the underlying fetch.
21
- return sdk.messages.stream(params, options.signal ? { signal: options.signal } : undefined);
43
+ return sdk.beta.messages.stream(body, options.signal ? { signal: options.signal } : undefined);
44
+ },
45
+ // Task 15 — keepalive ping, direct to the API: the SAME shaped body as
46
+ // createStream (system, cache markers, betas, fallbacks) with
47
+ // max_tokens 0 and no stream (probe P5: accepted, reads the cache). No
48
+ // SDK retries — a failed ping stops that wait, it must never storm.
49
+ async keepalive(params, options = {}) {
50
+ const { sdk, body } = await shapeRequest(params, options);
51
+ await sdk.beta.messages.create({ ...body, max_tokens: 0 }, { maxRetries: 0, timeout: 20_000 });
22
52
  },
23
53
  async healthCheck() {
24
54
  const key = await keychain.getPassword(KEYCHAIN_SERVICE, KEYCHAIN_ACCOUNT);
@@ -5,6 +5,8 @@ import { sha256 } from '@noble/hashes/sha256';
5
5
  import { bytesToHex } from '@noble/hashes/utils';
6
6
  import { createBundledSkillSource } from '../../skills/source-bundled.js';
7
7
  import { BUNDLED_SKILLS } from '../../skills/bundled-skills.js';
8
+ import { requireSkillName } from '../provider-shape.js';
9
+ import { buildSystemPrompt } from '../system-prompt.js';
8
10
  export function createLocalProvider(cfg) {
9
11
  if (!cfg.llm.local_base_url) {
10
12
  throw new Error('local mode requires llm.local_base_url — run: cspeach config set llm.local_base_url http://localhost:11434/v1');
@@ -20,9 +22,14 @@ export function createLocalProvider(cfg) {
20
22
  if (params.tools && params.tools.length > 0) {
21
23
  throw new Error('Tool calling not supported in local mode (Plan A). Switch to managed/byok/ai-hub, or wait for the local-mode tool-calling follow-up PR.');
22
24
  }
25
+ // Task 5 (ruling F3) — preamble + bundled skill body as the system
26
+ // message (no cache markers). anthropicParamsToOpenAI copies only
27
+ // max_tokens / messages / system, so newer fields never reach the endpoint.
28
+ const skillName = requireSkillName(options, 'Local');
29
+ const skill = await skillSource.getSkillBody(skillName);
23
30
  const OpenAI = await loadOpenAI();
24
31
  const sdk = new OpenAI({ apiKey: 'local-ignored', baseURL });
25
- const openAiParams = anthropicParamsToOpenAI(params, model);
32
+ const openAiParams = anthropicParamsToOpenAI({ ...params, system: buildSystemPrompt(skill.body, { cacheMarkers: false }) }, model);
26
33
  // stream: true widens via object spread; the literal-true overload cannot be
27
34
  // selected by structural inference, so cast to the streaming return shape.
28
35
  // v0.6 — pass AbortSignal through OpenAI SDK request options.
@@ -1,3 +1,30 @@
1
+ /**
2
+ * Repair partial assistant content blocks before persisting them to a session.
3
+ *
4
+ * When a provider stream is interrupted mid-flight (HTTP socket dropped,
5
+ * AbortController fired, network blip), the agent loop's accumulated
6
+ * `currentAssistantContent` may include `tool_use` blocks whose `input` JSON
7
+ * never finished streaming — typically the SDK has buffered the bytes into
8
+ * `partial_json` but `JSON.parse` was never called, so `input` is undefined.
9
+ *
10
+ * Persisting an assistant message that contains an incomplete tool_use block
11
+ * causes the next API call to reject the message with:
12
+ * 400 each `tool_use` block must contain `input` (or similar)
13
+ *
14
+ * Drop incomplete tool_use blocks. The model will re-emit them on the next
15
+ * round (the prior tool round's results are already on session.messages,
16
+ * giving the model the same context). Text blocks are kept even if
17
+ * truncated — partial prose is still useful for the user to read, and a
18
+ * short paragraph fragment is a valid assistant message. Thinking blocks are
19
+ * kept only when complete and signed (see isUnsignedThinking).
20
+ */
21
+ /**
22
+ * Block types whose input streams as `input_json_delta` (and so may carry a
23
+ * residual `partial_json`). Task 16 (F6): the tool search tool's
24
+ * `server_tool_use` streams its input the same way as a client `tool_use`
25
+ * (probe P6: 5 and 8 input_json_delta events).
26
+ */
27
+ export const STREAMED_INPUT_BLOCKS = new Set(['tool_use', 'server_tool_use']);
1
28
  /**
2
29
  * 2026-09-27 — a thinking block is only valid to send back to the API with its
3
30
  * complete signature. The signature arrives last (`signature_delta`, right
@@ -17,7 +44,7 @@ export function repairPartialBlocks(blocks, opts = {}) {
17
44
  return blocks.filter((b) => {
18
45
  if (b == null || typeof b !== 'object')
19
46
  return false;
20
- if (b.type === 'tool_use') {
47
+ if (STREAMED_INPUT_BLOCKS.has(b.type)) {
21
48
  // Complete only when input was successfully parsed (object, not undefined).
22
49
  if (b.input === undefined)
23
50
  return false;
@@ -44,6 +71,32 @@ export function repairPartialBlocks(blocks, opts = {}) {
44
71
  return true;
45
72
  });
46
73
  }
74
+ /**
75
+ * Task 6 (2026-09-27) — server-side fallback. When the first model hands off
76
+ * to a fallback model mid-output, the message carries a `fallback` block. The
77
+ * API rule for echoing it back: omit the `thinking`, `redacted_thinking` and
78
+ * `tool_use` blocks that precede the LAST `fallback` block (they belong to
79
+ * the model that did not finish). Everything else is kept in order.
80
+ *
81
+ * No `fallback` block ⇒ the same array reference is returned (no copy).
82
+ */
83
+ // Task 16 review — also the tool search tool's server blocks (conservative,
84
+ // same rule as tool_use: a server tool call of the model that did not finish).
85
+ const PRE_FALLBACK_DROP = new Set([
86
+ 'thinking', 'redacted_thinking', 'tool_use', 'server_tool_use', 'tool_search_tool_result',
87
+ ]);
88
+ export function stripPreFallbackBlocks(blocks) {
89
+ let last = -1;
90
+ for (let i = blocks.length - 1; i >= 0; i--) {
91
+ if (blocks[i]?.type === 'fallback') {
92
+ last = i;
93
+ break;
94
+ }
95
+ }
96
+ if (last < 0)
97
+ return blocks;
98
+ return blocks.filter((b, i) => i >= last || !PRE_FALLBACK_DROP.has(b?.type));
99
+ }
47
100
  /**
48
101
  * Heal a session already poisoned by an unsigned thinking block (saved by a
49
102
  * CLI before 2026-09-27): drop every thinking / redacted_thinking block that
@@ -96,7 +149,7 @@ export function healSessionMessagesInPlace(messages) {
96
149
  if (!Array.isArray(msg.content))
97
150
  continue;
98
151
  for (const block of msg.content) {
99
- if (block?.type === 'tool_use' && typeof block.partial_json === 'string') {
152
+ if (STREAMED_INPUT_BLOCKS.has(block?.type) && typeof block.partial_json === 'string') {
100
153
  if (block.input === undefined) {
101
154
  try {
102
155
  block.input = JSON.parse(block.partial_json);
@@ -199,10 +252,18 @@ export function healOrphanToolUses(messages) {
199
252
  }
200
253
  return added;
201
254
  }
202
- /** Review M2 — true when every block is thinking / redacted_thinking (and there is one). */
255
+ /**
256
+ * Review M2 — true when the message holds at least one thinking /
257
+ * redacted_thinking block and nothing else readable. Integration M1 — a
258
+ * `fallback` block (Task 6) is neutral: [fallback, thinking] is still
259
+ * thinking-only; [fallback] alone is not.
260
+ */
203
261
  export function isThinkingOnly(blocks) {
204
- return Array.isArray(blocks) && blocks.length > 0
205
- && blocks.every((b) => b?.type === 'thinking' || b?.type === 'redacted_thinking');
262
+ if (!Array.isArray(blocks))
263
+ return false;
264
+ const rest = blocks.filter((b) => b?.type !== 'fallback');
265
+ return rest.length > 0
266
+ && rest.every((b) => b?.type === 'thinking' || b?.type === 'redacted_thinking');
206
267
  }
207
268
  /**
208
269
  * Review M2 — a turn cut right after a signed thinking block (before any
@@ -29,7 +29,12 @@ export async function summariseViaProvider(provider, system, user, opts) {
29
29
  system,
30
30
  messages: [{ role: 'user', content: user }],
31
31
  }, {
32
- headers: opts.headers,
32
+ // Task 9 — every caller is /compact, so the role defaults to 'compact'
33
+ // (the proxy never enforces the session model on it). A caller-set
34
+ // X-CSPeach-Model-Role wins. No headers ⇒ unchanged (undefined).
35
+ headers: opts.headers
36
+ ? { 'X-CSPeach-Model-Role': 'compact', ...opts.headers }
37
+ : opts.headers,
33
38
  signal: opts.signal,
34
39
  });
35
40
  let accumulated = '';
@@ -0,0 +1,38 @@
1
+ /**
2
+ * Client-side system prompt for providers that talk to a model directly
3
+ * (BYOK, AI-hub, local). Task 5 (2026-09-27, ruling F3).
4
+ *
5
+ * Mirrors the proxy's managed system prompt (cspeach-proxy/src/routes/
6
+ * messages.ts): two text blocks — the CSPeach principles preamble, then the
7
+ * active skill's SKILL.md body — each carrying an ephemeral cache marker when
8
+ * the provider supports them.
9
+ */
10
+ import { FORGE_PRINCIPLES_PREAMBLE } from '../skills/preamble.js';
11
+ import { READER_SKILL, READER_SYSTEM_PROMPT } from '../tools/subagent/reader-prompt.js';
12
+ export function buildSystemPrompt(skillBody, opts) {
13
+ const marker = opts.cacheMarkers ? { cache_control: { type: 'ephemeral' } } : {};
14
+ return [
15
+ { type: 'text', text: FORGE_PRINCIPLES_PREAMBLE, ...marker },
16
+ { type: 'text', text: skillBody, ...marker },
17
+ ];
18
+ }
19
+ /**
20
+ * Task 18 (ruling F8) — a reader subagent's system prompt: the pinned
21
+ * READER_SYSTEM_PROMPT alone (no preamble, no skill body), as the proxy sends
22
+ * it for the `_reader` skill header.
23
+ */
24
+ export function buildReaderSystemPrompt(opts) {
25
+ const marker = opts.cacheMarkers ? { cache_control: { type: 'ephemeral' } } : {};
26
+ return [{ type: 'text', text: READER_SYSTEM_PROMPT, ...marker }];
27
+ }
28
+ /**
29
+ * The system prompt for the skill named in the X-CSForge-Skill header. The
30
+ * internal `_reader` skill has no SKILL.md, so it never reaches the skill
31
+ * source (a lookup would fail).
32
+ */
33
+ export async function systemPromptForSkill(skillName, skillSource, opts) {
34
+ if (skillName === READER_SKILL)
35
+ return buildReaderSystemPrompt(opts);
36
+ const skill = await skillSource.getSkillBody(skillName);
37
+ return buildSystemPrompt(skill.body, opts);
38
+ }
@@ -4,11 +4,19 @@ import { NOT_CONNECTED_MESSAGE } from '../sap/offline-adt-client.js';
4
4
  import { presentSafetyConfirmation } from '../repl/safety-confirm.js';
5
5
  import { shouldGateRule8, recordWriteOp, renderPlanSummary, getWriteOpsThisTurn, } from '../repl/rule8-detector.js';
6
6
  import { isCredentialHandoverApproved, markCredentialHandoverApproved, } from '../repl/credential-handover-gate.js';
7
+ import { READER_TOOLS } from '../tools/subagent/reader-prompt.js';
8
+ const READER_TOOL_SET = new Set(READER_TOOLS);
7
9
  export async function dispatchTool(name, args, ctx) {
8
10
  const tool = getTool(name);
9
11
  if (!tool) {
10
12
  return { content: JSON.stringify({ error: `unknown_tool: ${name}` }), is_error: true };
11
13
  }
14
+ // Task 18 (ruling F22) — a reader subagent is read-only and depth 1: only
15
+ // its fixed tool list runs, whatever the model calls (no writes, no
16
+ // questions or approvals, no nested read_agent / agent_run, no shell).
17
+ if (ctx.readerMode && !READER_TOOL_SET.has(name)) {
18
+ return { content: `error: read-only reader — ${name} is not allowed here`, is_error: true };
19
+ }
12
20
  // Standalone safety net (2026-09-16). SAP-dependent tools are already
13
21
  // filtered out of the list the model sees (agent/loop.ts), but a model can
14
22
  // still call a tool it remembers from an earlier turn or from training. Fail
@@ -0,0 +1,100 @@
1
+ /**
2
+ * Final review I2 (controller ruling, 2026-09-28) — the tool-loading mode is
3
+ * pinned on the SESSION, not resolved per process.
4
+ *
5
+ * The first request pins `session.tool_loading` to the mode that request
6
+ * uses (the process-resolved mode, gated by the model allow-list). A resumed
7
+ * session keeps its pinned mode: a 'static' session never turns deferred
8
+ * (its `tools[]` stays byte-stable), and a 'deferred' session stays deferred
9
+ * while the call allows it.
10
+ *
11
+ * A 'deferred' session must become 'static' when a call cannot take the
12
+ * deferred shapes: the admin kill switch, a failed startup config fetch
13
+ * (static until restart), `CSPEACH_TOOL_LOADING=static`, or a call on a model
14
+ * outside the allow-list (a plan-tier sonnet turn, an adopted opus-4-8). Its
15
+ * history then still holds deferred artefacts — `server_tool_use` /
16
+ * `tool_search_tool_result` blocks and `role: 'system'` tool_addition
17
+ * messages — which a request with no search tool and no deferred entries
18
+ * cannot carry. They are stripped ONCE, before that request, and the session
19
+ * is re-pinned 'static' so the strip never repeats. This is a deliberate
20
+ * one-time prefix rewrite, like /compact: every assistant message from the
21
+ * first edit on loses its thinking blocks (the API binds thinking to the exact
22
+ * prefix that produced it), the same stripThinking /compact and the cold
23
+ * prune apply.
24
+ */
25
+ import { stripThinking } from '../commands/compact.js';
26
+ import { modelAcceptsDeferredTools } from '../tools/tool-loading.js';
27
+ const SERVER_TOOL_BLOCK_TYPES = new Set(['server_tool_use', 'tool_search_tool_result']);
28
+ /** True when the history holds a shape only a deferred-mode request may carry. */
29
+ export function hasDeferredArtefacts(messages) {
30
+ for (const m of messages ?? []) {
31
+ if (m?.role === 'system')
32
+ return true;
33
+ if (Array.isArray(m?.content) && m.content.some((b) => SERVER_TOOL_BLOCK_TYPES.has(b?.type)))
34
+ return true;
35
+ }
36
+ return false;
37
+ }
38
+ /**
39
+ * Remove every deferred artefact from `messages` in place: `role: 'system'`
40
+ * messages, and `server_tool_use` / `tool_search_tool_result` blocks (a
41
+ * message left with no content is dropped). Every assistant message from the
42
+ * first edit on loses its thinking blocks (a thinking-only message keeps a
43
+ * placeholder text). Returns the number of messages and blocks removed; 0 ⇒
44
+ * untouched. Idempotent.
45
+ */
46
+ export function stripDeferredArtefacts(messages) {
47
+ let removed = 0;
48
+ let firstEdit = -1;
49
+ const out = [];
50
+ for (const m of messages) {
51
+ if (m?.role === 'system') {
52
+ removed++;
53
+ if (firstEdit < 0)
54
+ firstEdit = out.length;
55
+ continue;
56
+ }
57
+ if (Array.isArray(m?.content)) {
58
+ const kept = m.content.filter((b) => !SERVER_TOOL_BLOCK_TYPES.has(b?.type));
59
+ if (kept.length !== m.content.length) {
60
+ removed += m.content.length - kept.length;
61
+ if (firstEdit < 0)
62
+ firstEdit = out.length;
63
+ if (kept.length > 0)
64
+ out.push({ ...m, content: kept });
65
+ continue;
66
+ }
67
+ }
68
+ out.push(m);
69
+ }
70
+ if (removed === 0)
71
+ return 0;
72
+ for (let i = firstEdit; i < out.length; i++)
73
+ out[i] = stripThinking(out[i]);
74
+ messages.splice(0, messages.length, ...out);
75
+ return removed;
76
+ }
77
+ /**
78
+ * Decide the tool-loading mode for one call and keep the session consistent
79
+ * with it. `resolved` is the process-resolved mode (resolveToolLoading);
80
+ * `model` is the model this call sends. Mutates `session.tool_loading` and,
81
+ * on a deferred → static change, `session.messages` (see the file comment).
82
+ */
83
+ export function settleSessionToolLoading(session, i) {
84
+ const allowed = i.resolved === 'deferred' && modelAcceptsDeferredTools(i.model) ? 'deferred' : 'static';
85
+ // A session saved before the pin existed: its history tells how it ran.
86
+ const pinned = session.tool_loading
87
+ ?? (hasDeferredArtefacts(session.messages) ? 'deferred' : undefined);
88
+ if (pinned === undefined) {
89
+ session.tool_loading = allowed;
90
+ return { mode: allowed, stripped: 0 };
91
+ }
92
+ if (pinned === 'deferred' && allowed === 'deferred') {
93
+ session.tool_loading = 'deferred';
94
+ return { mode: 'deferred', stripped: 0 };
95
+ }
96
+ // Static for this call (and from now on): strip once, pin 'static'.
97
+ session.tool_loading = 'static';
98
+ const stripped = pinned === 'deferred' ? stripDeferredArtefacts(session.messages) : 0;
99
+ return { mode: 'static', stripped };
100
+ }