@cspeach/cli 1.1.19 → 1.1.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/agent/anthropic-provider.js +30 -10
  2. package/dist/agent/cache-keepalive.js +162 -0
  3. package/dist/agent/cold-prune.js +116 -0
  4. package/dist/agent/loop.js +675 -148
  5. package/dist/agent/provider-shape.js +263 -0
  6. package/dist/agent/providers/ai-hub-provider.js +17 -2
  7. package/dist/agent/providers/byok-provider.js +33 -3
  8. package/dist/agent/providers/local-provider.js +8 -1
  9. package/dist/agent/repair-partial.js +66 -5
  10. package/dist/agent/summarise-via-provider.js +6 -1
  11. package/dist/agent/system-prompt.js +38 -0
  12. package/dist/agent/tool-dispatch.js +8 -0
  13. package/dist/agent/tool-loading-pin.js +100 -0
  14. package/dist/cli.js +11 -0
  15. package/dist/commands/auto-compact.js +33 -16
  16. package/dist/commands/compact.js +37 -2
  17. package/dist/commands/config-set.js +10 -1
  18. package/dist/commands/config-show.js +11 -0
  19. package/dist/commands/cost.js +14 -2
  20. package/dist/commands/plan-audit.js +1 -0
  21. package/dist/config/loader.js +55 -2
  22. package/dist/cost/cost-log.js +62 -2
  23. package/dist/cost/pricing.js +6 -2
  24. package/dist/lib/spill-labels.js +13 -0
  25. package/dist/models/resolve.js +93 -2
  26. package/dist/models/server-config.js +158 -3
  27. package/dist/one-shot.js +15 -5
  28. package/dist/projects/image-attachments.js +15 -2
  29. package/dist/renderer/footer-line.js +6 -2
  30. package/dist/renderer/startup-lines.js +5 -3
  31. package/dist/renderer/tool-labels.js +33 -2
  32. package/dist/renderer/ui-width.js +13 -0
  33. package/dist/repl/current-transport.js +13 -0
  34. package/dist/repl/post-turn-status.js +8 -1
  35. package/dist/repl.js +71 -10
  36. package/dist/session/repin-model.js +18 -0
  37. package/dist/session/store.js +16 -2
  38. package/dist/skills/bundled-skills.js +1 -1
  39. package/dist/skills/preamble.js +75 -0
  40. package/dist/skills/source-manifest.js +11 -1
  41. package/dist/tools/filesystem/file-read.js +11 -1
  42. package/dist/tools/result-spill.js +238 -0
  43. package/dist/tools/sap-read.js +58 -14
  44. package/dist/tools/shell/shell_exec.js +9 -0
  45. package/dist/tools/subagent/adt-serial.js +33 -0
  46. package/dist/tools/subagent/agent_run.js +2 -0
  47. package/dist/tools/subagent/read_agent.js +178 -0
  48. package/dist/tools/subagent/reader-prompt.js +48 -0
  49. package/dist/tools/todo.js +3 -1
  50. package/dist/tools/tool-loading.js +255 -0
  51. package/dist/tools/tool-output-read.js +117 -0
  52. package/dist/tools/transport.js +6 -1
  53. package/dist/ui/footer.js +5 -5
  54. package/dist/ui/sap-state-store.js +1 -1
  55. package/dist/ui/turn-status-emitter.js +37 -0
  56. package/dist/ui/turn-status.js +1 -1
  57. package/package.json +2 -1
@@ -1,11 +1,15 @@
1
1
  import chalk from 'chalk';
2
2
  import { basename } from 'node:path';
3
3
  import { dispatchTool } from './tool-dispatch.js';
4
- import { listTools, getTool, toAnthropicTools, isStandalone, toolsForContext } from '../tools/index.js';
4
+ import { listTools, getTool, isStandalone, toolsForContext, toAnthropicTools } from '../tools/index.js';
5
5
  import { saveSession } from '../session/store.js';
6
6
  import { recordCompletedToolCall } from '../session/pending.js';
7
7
  import { loadConfig } from '../config/loader.js';
8
- import { resolveModelRole } from '../models/resolve.js';
8
+ import { resolveModelRole, resolveEffort, modelAcceptsEffort, resolveAutoCompactThreshold, callModelFor } from '../models/resolve.js';
9
+ import { getCustomerModelConfig, managedProxyServesNewShapes, servedToolLoading } from '../models/server-config.js';
10
+ import { buildToolList, buildSkillToolAddition, placeSystemMessagesInPlace, resolveToolLoading, } from '../tools/tool-loading.js';
11
+ import { settleSessionToolLoading } from './tool-loading-pin.js';
12
+ import { resolveSkillAlias } from '../skill-catalog.js';
9
13
  import { retryWithBackoff } from './retry.js';
10
14
  import { renderChunk, initialBufferState, render } from '../renderer/pipeline.js';
11
15
  import { nextWritingActivity } from '../renderer/fence-state.js';
@@ -23,7 +27,8 @@ import { turnStatusEmitter } from '../ui/turn-status-emitter.js';
23
27
  // (progress-chatter import removed 2026-05-01 — superseded by CC-style
24
28
  // two-line dispatch renderer; re-add if a future in-place spinner returns)
25
29
  import { buildRetryCapPausePayload, buildSkippedSiblingResults } from './retry-cap.js';
26
- import { appendCostLine, buildEntry as buildCostEntry } from '../cost/cost-log.js';
30
+ import { appendCostLine, buildEntry as buildCostEntry, summariseInputTransformations } from '../cost/cost-log.js';
31
+ import { modelAcceptsBinding, sessionMaxTokens, clampMaxTokens } from './provider-shape.js';
27
32
  import { computeSessionCost } from '../cost/session-cost.js';
28
33
  import { refreshCreditsAfterTurn, creditsForCostLog } from '../cost/credits-wire.js';
29
34
  import { globalStore } from '../ui/sap-state-store.js';
@@ -43,7 +48,7 @@ import { renderStandaloneContextBlock } from '../sap/standalone-profile.js';
43
48
  import { renderStandardsBlock } from '../standards/standards-file.js';
44
49
  import { composeSessionContextBlock } from './session-context.js';
45
50
  import { objectKeyFromInput } from './retry-key.js';
46
- import { repairPartialBlocks, healSessionMessagesInPlace, healOrphanToolUses, isPartialJsonApiError, stripUnsignedThinking, isThinkingOnly, dropTrailingThinkingOnly } from './repair-partial.js';
51
+ import { repairPartialBlocks, stripPreFallbackBlocks, healSessionMessagesInPlace, healOrphanToolUses, isPartialJsonApiError, stripUnsignedThinking, isThinkingOnly, dropTrailingThinkingOnly } from './repair-partial.js';
47
52
  import { drainSteering } from './steering-queue.js';
48
53
  import { TurnStreamWriter } from './turn-stream.js';
49
54
  import { applyToolResultCheckpoint } from './skill-checkpoint.js';
@@ -54,6 +59,10 @@ import { getCurrentTransport } from '../repl/current-transport.js';
54
59
  import { collectTurnAssistantText } from './turn-assistant-text.js';
55
60
  import { checkAndMark, newGuardState } from './parallel-write-guard.js';
56
61
  import { userPromptText } from '../session/user-prompt.js';
62
+ import { CacheKeepalive, snapshotParams, recordLastRequest, isCacheablePrompt, noteCacheTouched } from './cache-keepalive.js';
63
+ import { maybeColdPrune } from './cold-prune.js';
64
+ import { createAdtLock, serializeAdt } from '../tools/subagent/adt-serial.js';
65
+ import { hasReaderReadTools, INTERRUPTED_LINE } from '../tools/subagent/reader-prompt.js';
57
66
  /**
58
67
  * Author identity for project-file metadata. Reads CSPEACH_AUTHOR_NAME first,
59
68
  * then platform USER/USERNAME, then a generic fallback. Role is fixed to
@@ -324,8 +333,61 @@ export async function maybeOfferSave(p) {
324
333
  p.emit('\n' + formatAttention({ text: 'Not saved as a project file', fact: String(err?.message ?? err) }).trimEnd());
325
334
  }
326
335
  }
336
+ /**
337
+ * Task 12 — sessions that have already shown the "set by your account admin"
338
+ * line. The REPL rebuilds `params.ctx` every turn but keeps the same
339
+ * `ctx.session` object, so the flag is keyed on it (session-scoped, never
340
+ * persisted: a WeakSet, not a session field).
341
+ */
342
+ const adminModelLineShown = new WeakSet();
343
+ /**
344
+ * Task 18 — a reader child must not repaint the parent's turn-status row
345
+ * (several readers run at once; the parent shows one "N readers running…").
346
+ */
347
+ const SILENT_STATUS = {
348
+ activity: (_label) => undefined,
349
+ pause: () => undefined,
350
+ resume: () => undefined,
351
+ };
352
+ /** Task 18 — at most this many read_agent calls run at the same time. */
353
+ export const MAX_PARALLEL_READERS = 4;
354
+ const READ_AGENT = 'read_agent';
355
+ /**
356
+ * Task 18 — the read_agent blocks of the run that starts at `start`: every
357
+ * following tool_use block up to the first one that is not read_agent
358
+ * (non-tool blocks such as text in between are skipped).
359
+ */
360
+ function readerRunFrom(content, start) {
361
+ const out = [];
362
+ for (let i = start; i < content.length; i++) {
363
+ const b = content[i];
364
+ if (b?.type !== 'tool_use')
365
+ continue;
366
+ if (b.name !== READ_AGENT)
367
+ break;
368
+ out.push(b);
369
+ }
370
+ return out;
371
+ }
372
+ /** Drop a trailing -YYYYMMDD date suffix from a model id (for equality checks only). */
373
+ function stripModelDate(model) {
374
+ return model?.replace(/-\d{8}$/, '');
375
+ }
327
376
  export async function runTurn(params) {
328
- resetRule8State(); // Rule 8 — fresh batch counter per LLM turn (= per user prompt)
377
+ // Task 18 — a reader subagent runs INSIDE the parent's turn: it must not
378
+ // reset the parent's Rule 8 batch counter, drain the parent's steering
379
+ // queue or repaint the parent's status row.
380
+ const readerChild = params.ctx.readerMode === true;
381
+ if (!readerChild)
382
+ resetRule8State(); // Rule 8 — fresh batch counter per LLM turn (= per user prompt)
383
+ const status = readerChild ? SILENT_STATUS : turnStatusEmitter;
384
+ // Task 18 (F24) — reader spend of THIS turn; read by every cost line below.
385
+ params.ctx.readerSpend = { cost: 0, calls: 0 };
386
+ // Fix I1 — the turn's abort signal travels on the ctx to reader clones and
387
+ // into their child turns.
388
+ params.ctx.signal = params.signal;
389
+ // Fix M3 — reader calls already carried by a written cost line.
390
+ let readerCallsLogged = 0;
329
391
  // Piece 2 / P2 — fresh "plan changed this turn" flag per TOP-LEVEL turn.
330
392
  // Not every path ends in pushPostTurnStatus (the Ink /reroute turn does
331
393
  // not), so a mark left by such a turn must not earn the NEXT turn a close
@@ -346,6 +408,26 @@ export async function runTurn(params) {
346
408
  const toolRowEmitter = isDeclaredOneShot() ? (params.ctx.chunkEmitter ?? params.chunkEmitter) : params.chunkEmitter;
347
409
  const cfg = await loadConfig();
348
410
  const session = params.ctx.session;
411
+ // Task 15 — prompt-cache keepalive while this turn waits (a slow tool, an
412
+ // approval / ask_question prompt, the save hook). One instance per turn, so
413
+ // `in_turn_pings` caps the whole turn. Managed and BYOK only (AI-hub and
414
+ // local have no `keepalive`). `lastRequest` is the SNAPSHOT of the last
415
+ // request this turn sent (ruling F4), never the live messages array.
416
+ const keepaliveOn = cfg.keepalive?.enabled !== false && typeof params.provider.keepalive === 'function';
417
+ const keepalive = new CacheKeepalive({
418
+ provider: params.provider,
419
+ maxPings: keepaliveOn ? (cfg.keepalive?.in_turn_pings ?? 11) : 0,
420
+ // Task 19 — a successful ping keeps the cache warm (read by the cold prune).
421
+ onPing: (_n, err) => { if (err === undefined)
422
+ noteCacheTouched(session); },
423
+ });
424
+ let lastRequest;
425
+ // Never ping before a request exists, or when the API cached nothing
426
+ // (prompt under the minimum cacheable size).
427
+ const armKeepalive = () => {
428
+ if (keepaliveOn && lastRequest?.cacheable)
429
+ keepalive.arm(lastRequest.params, lastRequest.options);
430
+ };
349
431
  // Define `emit` early so the Phase J promote dispatch (below) can stream
350
432
  // its prompt + status lines through the same Ink-aware sink the rest of
351
433
  // runTurn uses. Originally defined later in the function — hoisted in v0.5
@@ -365,7 +447,10 @@ export async function runTurn(params) {
365
447
  let promotedFromForSave = null;
366
448
  let userMessageForLLM = params.userMessage;
367
449
  let userMessageForSave = params.userMessage;
368
- {
450
+ // Task 18 fix M2 — a reader's brief is plain text from the model: no
451
+ // --from promotion, no image or @file expansion (lean prefix, and a child
452
+ // must never open an interactive prompt).
453
+ if (!readerChild) {
369
454
  // Forgiveness: `--from @<document>` (a .docx/.pdf/.txt/.md, not a saved
370
455
  // .cspeach.json) means "attach this file", not "chain a result". Rewrite it
371
456
  // to a plain @<file> attachment and let Phase D ingest it — never error out
@@ -412,7 +497,9 @@ export async function runTurn(params) {
412
497
  // or too many images → the turn is not sent (same early exit as --from).
413
498
  // Pasted clipboard images arrive as [Image #N] chips — swap each for its
414
499
  // saved file path first, so the step below attaches it like a typed path.
415
- const imageExpansion = await expandImageAttachments(expandImageChips(userMessageForLLM), params.ctx.cwd, (line) => emit(chalk.dim(line)));
500
+ const imageExpansion = readerChild
501
+ ? { text: userMessageForLLM, images: [] }
502
+ : await expandImageAttachments(expandImageChips(userMessageForLLM), params.ctx.cwd, (line) => emit(chalk.dim(line)));
416
503
  if (imageExpansion.error) {
417
504
  emit(chalk.yellow(`
418
505
  ${imageExpansion.error}`));
@@ -429,7 +516,9 @@ ${imageExpansion.error}`));
429
516
  // The save copy keeps the original `@<filename>` token rather than
430
517
  // the expanded body so the saved envelope's source.input stays
431
518
  // human-readable; only the LLM sees the inflated prose.
432
- userMessageForLLM = await expandTextFileAttachments(userMessageForLLM, params.ctx.cwd, (line) => emit(chalk.dim(line)));
519
+ if (!readerChild) {
520
+ userMessageForLLM = await expandTextFileAttachments(userMessageForLLM, params.ctx.cwd, (line) => emit(chalk.dim(line)));
521
+ }
433
522
  // 2026-05-06: alongside the deterministic prompt-fact extraction
434
523
  // (package / transport / etc.), inject the connected SAP system's
435
524
  // release + platform + ABAP version. One CVERS query the first time
@@ -527,6 +616,90 @@ ${imageExpansion.error}`));
527
616
  emit(chalk.dim(`(${droppedImages} older image(s) removed from the conversation to keep it under the size limit)`));
528
617
  }
529
618
  session.skill = params.skill;
619
+ // Task 16 (spec D11) — deferred tool loading. The tools array is built ONE
620
+ // way for the whole session: the core set plus the starters of the skill
621
+ // pinned on `session.tool_set_skill` are loaded, every other tool is
622
+ // declared with defer_loading, the tool search tool goes last ('deferred');
623
+ // or today's full list ('static'). A changed array would break the prompt
624
+ // cache and the preserved-thinking prefix, so a skill switch APPENDS a
625
+ // `role: 'system'` tool_addition message instead (after a user message).
626
+ // Standalone hides the SAP tools (toolsForContext); the agent_run
627
+ // read-only filter (toolFilter) applies before the split, so a filtered-out
628
+ // tool is neither loaded nor deferred (cannot be found by search either).
629
+ session.tool_set_skill ??= params.skill;
630
+ const toolLoadingMode = resolveToolLoading({
631
+ providerMode: params.provider.mode,
632
+ env: process.env,
633
+ served: servedToolLoading(),
634
+ });
635
+ const contextTools = toolsForContext(listTools(), params.ctx);
636
+ // Task 18 fixes M4 / M5 — no read_agent for an agent_run child, and none
637
+ // when a reader would have no real read tool (standalone, file tools off).
638
+ const visibleTools = params.ctx.subagent || !hasReaderReadTools(new Set(contextTools.map((t) => t.name)))
639
+ ? contextTools.filter((t) => t.name !== READ_AGENT)
640
+ : contextTools;
641
+ // Final review C1 (controller ruling) — deferred shapes only for a model on
642
+ // the probe-proven allow-list, decided by the model actually sent on the
643
+ // call (a plan-tier sonnet override, a haiku child or claude-opus-4-8 ⇒
644
+ // static). Integration finding N1 (b): decided ONCE, at turn start — a
645
+ // model change mid-turn (an adopted served model) never strips deferred
646
+ // shapes in the middle of a turn (that would edit the prefix the in-flight
647
+ // thinking is bound to): the turn finishes on its own mode and the
648
+ // session settles at the next turn start. The proxy converts a deferred
649
+ // request for a model off its allow-list to static upstream.
650
+ // Final review I2 (controller ruling) — the mode is pinned on the session
651
+ // (session.tool_loading) at the first request and kept on resume; a
652
+ // deferred session that must turn static (kill switch, failed startup
653
+ // fetch, a model off the allow-list) has its deferred shapes stripped ONCE
654
+ // (a deliberate rewrite like /compact) and is re-pinned static.
655
+ const callToolMode = (model) => {
656
+ const settled = settleSessionToolLoading(session, { resolved: toolLoadingMode, model });
657
+ if (settled.stripped > 0) {
658
+ emit(chalk.dim('(this session now sends the full tool list — earlier tool-search steps were removed from the conversation)'));
659
+ }
660
+ return settled.mode;
661
+ };
662
+ const filteredTools = params.toolFilter ? visibleTools.filter(params.toolFilter) : visibleTools;
663
+ const builtTools = new Map();
664
+ const toolsFor = (mode) => {
665
+ let list = builtTools.get(mode);
666
+ if (!list) {
667
+ list = buildToolList(filteredTools, { skill: session.tool_set_skill, mode });
668
+ builtTools.set(mode, list);
669
+ }
670
+ return list;
671
+ };
672
+ // Task 18 — a reader's fixed list bypasses buildToolList entirely.
673
+ const readerTools = params.toolsOverride
674
+ ? toAnthropicTools(params.toolsOverride
675
+ .map((n) => visibleTools.find((t) => t.name === n))
676
+ .filter((t) => t !== undefined))
677
+ : undefined;
678
+ // Integration finding N1 (a) — the model each call is SENT with: an
679
+ // enforced customer model (managed, session role) replaces the pinned one,
680
+ // so call 0 already takes the static list when enforcement serves a model
681
+ // off the deferred allow-list.
682
+ const sentModel = () => callModelFor(params.modelOverride ?? session.model, { providerMode: params.provider.mode, role: params.modelRole });
683
+ const turnTools = readerTools ?? toolsFor(callToolMode(sentModel()));
684
+ /**
685
+ * Append a tool_addition for a skill's starters not yet loaded or surfaced.
686
+ * N1 (c) — built from the turn's list (the list every request of this turn
687
+ * carries) and never for a session pinned static.
688
+ */
689
+ const surfaceStarters = (skill) => {
690
+ const msg = buildSkillToolAddition({
691
+ skill,
692
+ toolLoading: session.tool_loading,
693
+ toolList: turnTools,
694
+ messages: session.messages,
695
+ });
696
+ if (msg)
697
+ session.messages.push(msg);
698
+ };
699
+ // A turn on a different skill than the one the array was built for: the
700
+ // user message was just pushed, so the system message follows it.
701
+ if (params.skill !== session.tool_set_skill)
702
+ surfaceStarters(params.skill);
530
703
  /**
531
704
  * key: `${tool_name}:${object}` — consecutive failures per (tool, target).
532
705
  * For object-identifying tools (sap_set_source / sap_activate / etc.) the
@@ -554,6 +727,10 @@ ${imageExpansion.error}`));
554
727
  };
555
728
  // Turn number for the cost log — count of prior user messages + this one.
556
729
  const turnNumber = session.messages.filter((m) => m.role === 'user').length + 1;
730
+ // Final review I4 — the monotonic per-session turn counter the auto-compact
731
+ // throttle compares (turnNumber above shrinks after a compaction). An older
732
+ // session seeds it so it never starts below its stored marker.
733
+ session.turn_seq = (session.turn_seq ?? Math.max(turnNumber - 1, session.lastAutoCompactTurn ?? 0)) + 1;
557
734
  // Snapshot session.messages.length at turn start so the save hook can
558
735
  // walk every assistant message added during this turn — not just the
559
736
  // final one. Skills like /abap-cca emit their manifest in an early
@@ -579,8 +756,24 @@ ${imageExpansion.error}`));
579
756
  // long phase (e.g. /abap-plan c2.behavior) that overruns the output cap
580
757
  // mid-response finishes instead of dying with "unexpected stop reason".
581
758
  let maxTokenContinuations = 0;
759
+ // Task 16 (F6) — bounded re-sends on stop_reason 'pause_turn' (a server
760
+ // tool paused a long turn). Consecutive only: reset by any other round.
761
+ let pauseTurnContinuations = 0;
762
+ // Task 3 (2026-09-27) — 0-based index of each model call within this turn,
763
+ // recorded on the cost line as `call_index`.
764
+ let roundIndex = 0;
582
765
  // (emit was hoisted to the top of the function in v0.5 so the Phase J
583
766
  // promote dispatch could share it. Original location was here.)
767
+ // Task 19 (owner decision 7) — cold-cache prune, default OFF
768
+ // (CSPEACH_COLD_PRUNE=on). Once, before the first createStream: when the
769
+ // cache has expired (> 300 s since the last call / ping) the next call
770
+ // re-writes the history anyway, so old tool results are elided first.
771
+ {
772
+ const pruned = maybeColdPrune(session);
773
+ if (pruned > 0) {
774
+ emit(chalk.dim(`(cache was cold — ${pruned} old tool result(s) elided to keep the rewrite small)`));
775
+ }
776
+ }
584
777
  while (true) {
585
778
  // Mid-turn steering (2026-06-08): inject any user corrections typed since the
586
779
  // last round so the next createStream sees them. Drains at the top of EVERY
@@ -589,7 +782,7 @@ ${imageExpansion.error}`));
589
782
  // Anthropic API concatenates consecutive user messages, so injecting after a
590
783
  // tool_result user message is valid.
591
784
  {
592
- const steers = drainSteering();
785
+ const steers = readerChild ? [] : drainSteering();
593
786
  if (steers.length > 0) {
594
787
  const text = steers.join('\n');
595
788
  session.messages.push({ role: 'user', content: text });
@@ -615,7 +808,7 @@ ${imageExpansion.error}`));
615
808
  // routed) covers liveness feedback. Classic mode (no chunkEmitter)
616
809
  // keeps the stdout spinner unchanged.
617
810
  // Guard: test/unit/ink-cursor-invariant.test.ts.
618
- turnStatusEmitter.activity('thinking');
811
+ status.activity('thinking');
619
812
  const thinkingSpinner = startThinkingSpinner({ chunkEmitter: livenessEmitter });
620
813
  // v0.6 — guaranteed-visible heartbeat alongside the in-place spinner.
621
814
  // The spinner self-disables on non-TTY (Windows PowerShell sometimes
@@ -627,6 +820,9 @@ ${imageExpansion.error}`));
627
820
  // being raw-written to stdout (which Ink overdraws on its next
628
821
  // render, briefly flashing the line then making it disappear).
629
822
  const thinkingHeartbeat = startThinkingHeartbeat({ chunkEmitter: livenessEmitter });
823
+ // Task 15 — the request actually sent by the attempt that succeeded.
824
+ let sentParams;
825
+ let sentHeaders;
630
826
  try {
631
827
  stream = await retryWithBackoff(() => (async () => {
632
828
  // W2.5 — apply optional tool filter (used by agent_run with
@@ -634,9 +830,11 @@ ${imageExpansion.error}`));
634
830
  // the subagent). Mirrors CC's Tq2() pattern.
635
831
  // Standalone (2026-09-16): hide every SAP-dependent tool from the
636
832
  // model — calling one could only fail. Identity when connected.
637
- const allTools = toolsForContext(listTools(), params.ctx);
638
- const filteredTools = params.toolFilter ? allTools.filter(params.toolFilter) : allTools;
639
- const tools = toAnthropicTools(filteredTools);
833
+ // Task 16 — built once per mode above (byte-stable per session).
834
+ // Final review C1/I2 + N1 (b) — the turn's mode, settled at turn
835
+ // start (any one-time deferred → static strip ran there).
836
+ const model = sentModel();
837
+ const tools = turnTools;
640
838
  // Bug 11a (proactive heal, 2026-06-07) — strip any residual
641
839
  // `partial_json` from tool_use blocks on EVERY outgoing request.
642
840
  // `content_block_stop` normally deletes it (see below), but any miss —
@@ -650,10 +848,22 @@ ${imageExpansion.error}`));
650
848
  // the single chokepoint every request passes through, retries included —
651
849
  // makes the 400 categorically impossible. Idempotent and O(messages).
652
850
  healSessionMessagesInPlace(session.messages);
653
- // 2026-09-27 — same chokepoint: a thinking block without its
654
- // signature (a turn cancelled mid-thinking, saved by an older CLI)
655
- // 400s every later request with "Invalid `signature` in `thinking`
656
- // block". Drop it; the model simply thinks again.
851
+ // Task 7 (spec D8) — preserved thinking binds each thinking block to
852
+ // the exact prefix that produced it, so history must be append-only.
853
+ // The heals below are no-ops on a history the API accepted (they only
854
+ // touch blocks the API would have rejected, or drop the LAST assistant
855
+ // message when it is thinking-only — no later block is bound to a
856
+ // prefix containing it), so they are not prefix edits. Order (merge
857
+ // with the UI branch's round-4 heal, cost finding core-I5): tool-mode
858
+ // settle (turn start) → partial_json → unsigned thinking →
859
+ // trailing thinking-only → orphan tool results → system placement
860
+ // LAST, because the earlier heals may remove the last assistant
861
+ // message and so change the tail placeSystemMessagesInPlace works on.
862
+ //
863
+ // 2026-09-27 — a thinking block without its signature (a turn
864
+ // cancelled mid-thinking, saved by an older CLI) 400s every later
865
+ // request with "Invalid `signature` in `thinking` block". Drop it;
866
+ // the model simply thinks again.
657
867
  stripUnsignedThinking(session.messages);
658
868
  dropTrailingThinkingOnly(session.messages);
659
869
  // 2026-09-24 — same chokepoint: a tool call cut off before its
@@ -661,37 +871,87 @@ ${imageExpansion.error}`));
661
871
  // `continue` could never succeed). Give each orphan an error result
662
872
  // so the model sees the call did not complete and redoes it.
663
873
  const healedOrphans = healOrphanToolUses(session.messages);
874
+ // Task 16 (probe D2) — a system tool_addition must precede an
875
+ // assistant message or end the array ([system, user] is a 400).
876
+ // Same chokepoint: covers steering, the watchdog nudge and a turn
877
+ // that ended before the model answered the tool_addition.
878
+ placeSystemMessagesInPlace(session.messages);
664
879
  if (healedOrphans > 0) {
665
880
  emit(chalk.dim(`(repaired ${healedOrphans} tool call(s) cut off by an interruption — the model will redo them)`));
666
881
  }
882
+ // Task 4 — effort only from an explicit act (ruling F2); never to a
883
+ // model that rejects it (probe P1: claude-haiku-4-5* ⇒ 400).
884
+ const effort = params.effortOverride ?? resolveEffort(cfg);
885
+ const sendEffort = effort !== undefined && modelAcceptsEffort(model);
667
886
  const streamParams = {
668
887
  // A4 — per-turn override (plan model tiering) wins over the
669
- // configured default; absent on every non-plan turn. model-governance
670
- // step 2d — the session-default model now resolves env > local >
671
- // server > built-in (byte-identical to cfg.default_model when nothing set).
672
- model: params.modelOverride ?? resolveModelRole('session_default', cfg),
888
+ // session model; absent on every non-plan turn. Task 3 (2026-09-27)
889
+ // — the session's PINNED model, resolved once at newSession
890
+ // (env > local > server > built-in) after the served config was
891
+ // awaited. A per-call resolve let call 0 run on the built-in and
892
+ // call 1 on the late-landing served model: a full cache rewrite.
893
+ model,
673
894
  // v0.3.1 — was 8192. Bumped because code-gen skills (abap-generate
674
895
  // etc.) kept running out of budget mid-turn: adaptive thinking +
675
896
  // multiple tool-result prompts + TL;DR mandate + summary prose
676
897
  // + the final question widget didn't fit in 8K. Symptom: stream
677
898
  // ended with stop_reason='length' right before the widget, and
678
899
  // the user saw the skill "stuck" after the intro prose.
679
- // 32768 is Opus 4.7's recommended ceiling for agentic turns and
680
- // gives comfortable headroom for the investigate-first pattern.
681
- max_tokens: params.maxTokensOverride ?? 32768,
900
+ // Task 8 (2026-09-27, ruling F5) — thinking counts toward
901
+ // max_tokens. 32768 truncated a turn once in the 2026-09 profile,
902
+ // and Opus 5.x thinks more per turn, so claude-opus-5* session
903
+ // turns get 65536 (the guide's starting point for long agentic
904
+ // turns; probe P2 accepted it on claude-opus-5 and
905
+ // claude-opus-5-5). Every other model keeps 32768: its output
906
+ // limit is not verified here. Changing max_tokens breaks neither
907
+ // the prompt cache nor the thinking block binding. Follows the
908
+ // model actually sent (plan-tier sonnet turns stay at 32768);
909
+ // maxTokensOverride (subagents) still wins below the ceiling.
910
+ // I1 (2026-09-28) — clamped to the sent model's output ceiling:
911
+ // an agent_run child asks for 100000, which an enforced
912
+ // sonnet/haiku model rejects with a 400.
913
+ max_tokens: clampMaxTokens(model, params.maxTokensOverride ?? sessionMaxTokens(model)),
682
914
  messages: session.messages,
683
915
  tools,
684
- thinking: { type: 'adaptive', display: 'summarized' },
916
+ // Task 7 — preserved thinking. Opus 5.5 binds each thinking block
917
+ // to the exact prefix that produced it; `drop_block` makes the API
918
+ // drop a block whose prefix changed (reported on message_start's
919
+ // input_transformations) instead of rejecting the request. Only to
920
+ // the families probe P4 proved (allow-list, dated ids stripped).
921
+ // Task 21 fix round (controller ruling) — on the managed path only
922
+ // when the proxy serves tool_loading (an old proxy adds no binding
923
+ // beta); BYOK adds its own beta; AI-hub/local strip the field.
924
+ thinking: modelAcceptsBinding(model) && (params.provider.mode !== 'managed' || managedProxyServesNewShapes())
925
+ ? { type: 'adaptive', display: 'summarized', block_binding: { prefix_mismatch_behavior: 'drop_block' } }
926
+ : { type: 'adaptive', display: 'summarized' },
927
+ ...(sendEffort ? { output_config: { effort } } : {}),
685
928
  };
686
929
  // Skill travels as a header per coordination doc §1.
687
930
  // Do NOT add `skill` to streamParams — it is not part of the Anthropic SDK body schema.
688
931
  // v0.6 — forward the per-turn abort signal so Esc / SIGINT-during-turn
689
932
  // tears down the in-flight stream cleanly.
933
+ const headers = { 'X-CSForge-Skill': params.skill, 'X-CSPeach-Model-Role': params.modelRole ?? 'session_default' };
934
+ sentParams = streamParams;
935
+ sentHeaders = headers;
690
936
  return params.provider.createStream(streamParams, {
691
- headers: { 'X-CSForge-Skill': params.skill },
937
+ // Task 9 — the role header tells the proxy which pipeline role
938
+ // this call runs under (body unchanged).
939
+ headers,
692
940
  signal: params.signal,
693
941
  });
694
942
  })());
943
+ // Task 15 (F14) — stamp every successful createStream; Task 19 reads it.
944
+ session.last_request_at = new Date().toISOString();
945
+ // Task 15 (F4) — snapshot NOW, before the loop pushes this round's
946
+ // assistant message: the snapshot ends on a user message. Also left on
947
+ // the session (WeakMap, never persisted) for the REPL's idle keepalive.
948
+ if (keepaliveOn && sentParams) {
949
+ lastRequest = recordLastRequest(session, {
950
+ params: snapshotParams(sentParams),
951
+ options: { headers: { ...(sentHeaders ?? {}) } },
952
+ cacheable: false, // set from message_start usage below
953
+ });
954
+ }
695
955
  }
696
956
  catch (err) {
697
957
  // Stop the thinking spinner before printing any error — leaving
@@ -734,12 +994,26 @@ ${imageExpansion.error}`));
734
994
  throw err;
735
995
  }
736
996
  let currentAssistantContent = [];
997
+ // Task 3 — this call's index within the turn, and the model the server
998
+ // reported in `message_start` (may differ from the requested id).
999
+ const callIndex = roundIndex++;
1000
+ let lastServedModel;
737
1001
  // 2026-09-27 — blocks whose content_block_stop arrived. A thinking block
738
1002
  // not in here was cut mid-stream and must never be persisted.
739
1003
  const completedBlocks = new WeakSet();
740
1004
  let sawEndTurn = false;
741
1005
  let sawToolUse = false;
742
1006
  let sawMaxTokens = false;
1007
+ let sawPauseTurn = false;
1008
+ // Task 6 — a safety classifier declined the request (HTTP 200,
1009
+ // stop_reason 'refusal'). Branch on stop_reason only; stop_details is
1010
+ // captured for the cost log (its category), never shown to the user.
1011
+ let sawRefusal = false;
1012
+ let refusalCategory;
1013
+ // Task 7 — thinking blocks the API dropped (prefix binding mismatch), from
1014
+ // message_start.message.input_transformations. Cost line only; the user
1015
+ // sees nothing.
1016
+ let inputTransformations;
743
1017
  // M10 — ensure session.usage is present (shipped session schema may omit it
744
1018
  // for older saved sessions; we default to zero on first turn).
745
1019
  if (!session.usage) {
@@ -782,7 +1056,7 @@ ${imageExpansion.error}`));
782
1056
  // content. Long emissions (a 4k-token plan manifest) can look
783
1057
  // silent if the renderer buffers the block — the row keeps
784
1058
  // ticking "writing response · 90s" regardless.
785
- turnStatusEmitter.activity('writing response');
1059
+ status.activity('writing response');
786
1060
  writingCode = false;
787
1061
  const block = event.content_block;
788
1062
  currentAssistantContent.push(block);
@@ -803,7 +1077,7 @@ ${imageExpansion.error}`));
803
1077
  const wa = nextWritingActivity(writingCode, buffer.pending);
804
1078
  writingCode = wa.writingCode;
805
1079
  if (wa.label)
806
- turnStatusEmitter.activity(wa.label);
1080
+ status.activity(wa.label);
807
1081
  if (output) {
808
1082
  if (params.chunkEmitter)
809
1083
  params.chunkEmitter.emit('chunk', output);
@@ -838,7 +1112,10 @@ ${imageExpansion.error}`));
838
1112
  const last = currentAssistantContent[currentAssistantContent.length - 1];
839
1113
  if (last && typeof last === 'object')
840
1114
  completedBlocks.add(last);
841
- if (last?.type === 'tool_use' && last.partial_json) {
1115
+ // Task 16 (F6) — the tool search tool's server_tool_use streams its
1116
+ // input as input_json_delta too (probe P6); parse and strip it the
1117
+ // same way, or the next request 400s on partial_json.
1118
+ if ((last?.type === 'tool_use' || last?.type === 'server_tool_use') && last.partial_json) {
842
1119
  try {
843
1120
  last.input = JSON.parse(last.partial_json);
844
1121
  }
@@ -849,9 +1126,26 @@ ${imageExpansion.error}`));
849
1126
  }
850
1127
  }
851
1128
  else if (type === 'message_start') {
1129
+ // Task 3 — the model the server actually ran (adopted after the
1130
+ // stream in managed mode only; see below).
1131
+ const servedModel = event.message?.model;
1132
+ if (typeof servedModel === 'string' && servedModel)
1133
+ lastServedModel = servedModel;
1134
+ // Task 7 — present only when the binding beta was sent; [] when
1135
+ // nothing was dropped (summarised to undefined ⇒ no cost-line key).
1136
+ inputTransformations = summariseInputTransformations(event.message?.input_transformations);
852
1137
  // message_start carries the prompt usage (input tokens + any cache reads).
853
1138
  const msgUsage = event.message?.usage;
1139
+ // Task 15 — ping only a prompt the API actually cached.
1140
+ if (lastRequest && isCacheablePrompt(msgUsage))
1141
+ lastRequest.cacheable = true;
854
1142
  if (msgUsage) {
1143
+ // Task 19 (F12) — this call's full prompt size, read by the
1144
+ // end-of-turn auto-compact gate (the last call of the turn wins).
1145
+ session.last_context_tokens =
1146
+ (msgUsage.input_tokens ?? 0)
1147
+ + (msgUsage.cache_read_input_tokens ?? 0)
1148
+ + (msgUsage.cache_creation_input_tokens ?? 0);
855
1149
  session.usage.input_tokens += msgUsage.input_tokens ?? 0;
856
1150
  session.usage.cache_read_input_tokens += msgUsage.cache_read_input_tokens ?? 0;
857
1151
  // Phase 1 cost tracking — cache writes are billed at 1.25× input.
@@ -877,6 +1171,13 @@ ${imageExpansion.error}`));
877
1171
  sawToolUse = true;
878
1172
  if (stop_reason === 'max_tokens' || stop_reason === 'length')
879
1173
  sawMaxTokens = true;
1174
+ if (stop_reason === 'pause_turn')
1175
+ sawPauseTurn = true;
1176
+ if (stop_reason === 'refusal') {
1177
+ sawRefusal = true;
1178
+ const category = event.delta?.stop_details?.category;
1179
+ refusalCategory = typeof category === 'string' && category ? category : 'unknown';
1180
+ }
880
1181
  // H2 — `deltaUsage.output_tokens` is the cumulative running total for
881
1182
  // THIS message, NOT a per-event delta. Compute the increment vs the
882
1183
  // last-seen cumulative and add only that. Guard against `null` /
@@ -924,11 +1225,57 @@ ${imageExpansion.error}`));
924
1225
  // only DROPS a block when its `input` is undefined — which never happens on
925
1226
  // a clean round (content_block_stop always sets `input`), so this is a
926
1227
  // no-op-except-strip on the happy path. See ./repair-partial.ts.
927
- currentAssistantContent = repairPartialBlocks(currentAssistantContent, { completed: completedBlocks });
928
- // Review M2 — a cut right after the thinking block leaves a thinking-only
929
- // turn: nothing the user can read, and nothing the next request needs.
1228
+ // Task 6 — after a server-side fallback, the first model's thinking /
1229
+ // tool_use blocks before the last `fallback` block must not be echoed back.
1230
+ // The `fallback` block itself stays (the adoption guard below reads it).
1231
+ currentAssistantContent = stripPreFallbackBlocks(repairPartialBlocks(currentAssistantContent, { completed: completedBlocks }));
1232
+ // Review M2 (UI branch) — a cut right after the thinking block leaves a
1233
+ // thinking-only turn: nothing the user can read, and nothing the next
1234
+ // request needs.
1235
+ // Integration M1 — a `fallback` block is neutral to that test, so
1236
+ // [fallback, thinking] is dropped too; remember the fallback first so
1237
+ // the adoption guard below still refuses to pin the fallback model.
1238
+ const carriedFallback = currentAssistantContent.some((b) => b?.type === 'fallback');
930
1239
  if (isThinkingOnly(currentAssistantContent))
931
1240
  currentAssistantContent = [];
1241
+ // Task 3 (F10) — adopt the served model. Managed mode only: BYOK / AI-hub
1242
+ // / local streams report a local alias, which would leave the session
1243
+ // unpriced. Never on a per-turn override, and never on a message carrying
1244
+ // a `fallback` block (a refusal fallback must not pin the fallback model
1245
+ // for the rest of the session). Otherwise the served id is only logged.
1246
+ if (params.provider.mode === 'managed'
1247
+ && !params.modelOverride
1248
+ && typeof lastServedModel === 'string'
1249
+ && lastServedModel
1250
+ && !carriedFallback
1251
+ // Task 6 — a refused message is discarded, so it proves nothing about
1252
+ // the model (and may have been served by a fallback).
1253
+ && !sawRefusal
1254
+ // Probe ruling C1 — served ids can carry a -YYYYMMDD suffix (haiku-4-5
1255
+ // is served as claude-haiku-4-5-20251001): a dated id equal to the
1256
+ // pinned id is not a model switch.
1257
+ && stripModelDate(session.model) !== stripModelDate(lastServedModel)) {
1258
+ session.model = lastServedModel;
1259
+ session.served_model_source = 'served';
1260
+ // Task 12 — explain the switch, once per session, only when the proxy
1261
+ // enforces the account admin's choice (/v1/me/model-config enforced)
1262
+ // AND the served model IS that choice (dated suffix ignored) — any
1263
+ // other switch (e.g. a version-gate downgrade) is not the admin's doing.
1264
+ const customer = getCustomerModelConfig();
1265
+ if (customer?.enforced === true
1266
+ && typeof customer.model === 'string'
1267
+ && stripModelDate(customer.model) === stripModelDate(lastServedModel)
1268
+ && !adminModelLineShown.has(session)) {
1269
+ adminModelLineShown.add(session);
1270
+ emit(chalk.dim(`Using ${lastServedModel} (set by your account admin)`));
1271
+ }
1272
+ }
1273
+ // Task 6 — a refusal (before any output or mid-stream) is not an answer:
1274
+ // discard whatever partial content streamed so nothing is persisted as a
1275
+ // complete assistant message. The cost line below is still written.
1276
+ if (sawRefusal) {
1277
+ currentAssistantContent = [];
1278
+ }
932
1279
  // Persist the assistant message — even if it's partial. Empty content
933
1280
  // arrays are skipped so we don't push a meaningless empty assistant
934
1281
  // message that would confuse the next round.
@@ -976,7 +1323,11 @@ ${imageExpansion.error}`));
976
1323
  // the cost line so the JSONL records what the LEDGER charged alongside
977
1324
  // our COGS, and so the status footer the REPL prints next reads a fresh
978
1325
  // figure. No-op + no request outside credits mode; capped at 2.5 s.
979
- await refreshCreditsAfterTurn();
1326
+ // Task 18 — a reader child never samples the balance: its charge then
1327
+ // lands in the parent's next credits delta, and parallel readers never
1328
+ // race on the shared sample.
1329
+ if (!readerChild)
1330
+ await refreshCreditsAfterTurn();
980
1331
  const turnTokens = {
981
1332
  input: (session.usage?.input_tokens ?? 0) - turnStartTokens.input,
982
1333
  output: (session.usage?.output_tokens ?? 0) - turnStartTokens.output,
@@ -994,8 +1345,23 @@ ${imageExpansion.error}`));
994
1345
  duration_ms: turnDurationMs,
995
1346
  // Additive: present only in credits mode (undefined in USD mode, so
996
1347
  // the emitted line is byte-identical to a pre-credits one).
997
- credits: creditsForCostLog(),
1348
+ credits: readerChild ? undefined : creditsForCostLog(),
1349
+ // Task 3 — what the server reported, which call of the turn, and
1350
+ // the call kind. served_model is omitted when the stream had none.
1351
+ served_model: lastServedModel ?? undefined,
1352
+ call_index: callIndex,
1353
+ // Task 18 — a reader child's lines (its own reader-*-cost.jsonl) say so.
1354
+ kind: readerChild ? 'reader' : 'main',
1355
+ // Task 6 — present only on a refusal ('unknown' when stop_details
1356
+ // carried no category). Never the explanation text.
1357
+ refusal_category: sawRefusal ? refusalCategory : undefined,
1358
+ // Task 7 — present only when the API dropped thinking blocks.
1359
+ input_transformations: inputTransformations,
1360
+ // Task 18 (F24) — running reader totals of this turn (absent until a
1361
+ // reader has run).
1362
+ reader: params.ctx.readerSpend,
998
1363
  });
1364
+ readerCallsLogged = params.ctx.readerSpend?.calls ?? 0;
999
1365
  void appendCostLine(session.id, entry);
1000
1366
  }
1001
1367
  // Close the stream-to-disk mirror with a footer (lightweight diagnostics).
@@ -1008,6 +1374,18 @@ ${imageExpansion.error}`));
1008
1374
  // graceful error UX with full session-id + log-path context.
1009
1375
  if (interruptedError !== null)
1010
1376
  throw interruptedError;
1377
+ // Task 6 — the model declined. One plain line (no vendor name, no
1378
+ // category, no policy text), then end the turn. Nothing was persisted.
1379
+ if (sawRefusal) {
1380
+ emit('\n' + formatAttention({
1381
+ text: 'The model declined this request',
1382
+ fact: 'a safety filter stopped it',
1383
+ next: 'rephrase it, or split it into smaller steps',
1384
+ }).trimEnd());
1385
+ session.awaitingSkillAnswer = false;
1386
+ await saveSession(session);
1387
+ return;
1388
+ }
1011
1389
  if (sawEndTurn) {
1012
1390
  emit('');
1013
1391
  // v0.6 Layer 3 — stream-end watchdog warning. The between-rounds
@@ -1052,7 +1430,7 @@ ${imageExpansion.error}`));
1052
1430
  // their manifest in an early message before a closing ask_question
1053
1431
  // widget; the final wrap-up message ends the turn but doesn't
1054
1432
  // carry the manifest. See ./turn-assistant-text.ts for details.
1055
- turnStatusEmitter.activity('finishing up');
1433
+ status.activity('finishing up');
1056
1434
  const assistantText = collectTurnAssistantText(session.messages, turnStartMessageCount);
1057
1435
  // 2026-09-17 — the save hook below asks the user things (save y/N,
1058
1436
  // blocker pickers). Those are not tool dispatches, so nothing paused
@@ -1060,9 +1438,11 @@ ${imageExpansion.error}`));
1060
1438
  // while the user was the one being waited on (Laeeq's dry run). Pause
1061
1439
  // for the whole hook — the row hides while paused — and resume after,
1062
1440
  // so any real work that follows (auto-compact) shows again.
1063
- turnStatusEmitter.pause();
1441
+ status.pause();
1064
1442
  try {
1065
1443
  if (!params.suppressSaveHook) {
1444
+ // Task 15 — the save prompt waits on the user; keep the cache warm.
1445
+ armKeepalive();
1066
1446
  await maybeOfferSave({
1067
1447
  skill: params.skill,
1068
1448
  assistantText,
@@ -1078,7 +1458,8 @@ ${imageExpansion.error}`));
1078
1458
  }
1079
1459
  }
1080
1460
  finally {
1081
- turnStatusEmitter.resume();
1461
+ keepalive.disarm();
1462
+ status.resume();
1082
1463
  }
1083
1464
  // Phase 2 #9 (2026-05-16) — auto-compact end-of-turn hook.
1084
1465
  // Runs only on the success path (after maybeOfferSave) so an interrupted
@@ -1086,30 +1467,68 @@ ${imageExpansion.error}`));
1086
1467
  // helper internally swallows non-fatal errors and emits a yellow note so
1087
1468
  // a Haiku blip never regresses a turn that just succeeded. Threshold and
1088
1469
  // throttle live in config.compact — see config/loader.ts:CompactConfig.
1089
- try {
1090
- const { maybeAutoCompact } = await import('../commands/auto-compact.js');
1091
- const { buildSummarisationPrompt, serialiseForSummariser } = await import('../commands/compact.js');
1092
- const { summariseViaProvider } = await import('./summarise-via-provider.js');
1093
- await maybeAutoCompact({
1094
- session,
1095
- config: cfg.compact,
1096
- turnNumber,
1097
- emit,
1098
- summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise),
1099
- // model-governance step 2d — compact model resolves env > local >
1100
- // server > built-in (== COMPACTION_MODEL when nothing set).
1101
- { model: resolveModelRole('compact', cfg), maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact' } }),
1102
- });
1103
- }
1104
- catch (err) {
1105
- // Belt-and-braces — maybeAutoCompact already swallows internally,
1106
- // but if its own dynamic-import or wiring blows up we still must
1107
- // not regress the success path.
1108
- emit(chalk.gray(`[auto-compact] skipped: ${err instanceof Error ? err.message : String(err)}`));
1470
+ // Task 19 — top-level REPL turns only (never one-shot, subagents or
1471
+ // readers); the gate reads session.last_context_tokens against the
1472
+ // resolved threshold (file > served auto_compact_tokens > 150 000).
1473
+ if (params.autoCompact === true) {
1474
+ try {
1475
+ const { maybeAutoCompact } = await import('../commands/auto-compact.js');
1476
+ const { buildSummarisationPrompt, serialiseForSummariser } = await import('../commands/compact.js');
1477
+ const { summariseViaProvider } = await import('./summarise-via-provider.js');
1478
+ await maybeAutoCompact({
1479
+ session,
1480
+ config: { ...cfg.compact, auto_threshold_tokens: resolveAutoCompactThreshold(cfg) },
1481
+ turnNumber: session.turn_seq,
1482
+ emit,
1483
+ summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise),
1484
+ // model-governance step 2d — compact model resolves env > local >
1485
+ // server > built-in (== COMPACTION_MODEL when nothing set).
1486
+ { model: resolveModelRole('compact', cfg), maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact', 'X-CSPeach-Model-Role': 'compact' } }),
1487
+ });
1488
+ }
1489
+ catch (err) {
1490
+ // Belt-and-braces — maybeAutoCompact already swallows internally,
1491
+ // but if its own dynamic-import or wiring blows up we still must
1492
+ // not regress the success path.
1493
+ emit(chalk.gray(`[auto-compact] skipped: ${err instanceof Error ? err.message : String(err)}`));
1494
+ }
1109
1495
  }
1110
1496
  return;
1111
1497
  }
1112
- if (!sawToolUse) {
1498
+ // Task 6 fix round 1 — stop_reason 'tool_use' but no tool_use block left
1499
+ // (e.g. the only one preceded a `fallback` block and was stripped). There
1500
+ // is nothing to dispatch; looping would re-send the same history. End the
1501
+ // turn (the message is already persisted).
1502
+ if (sawToolUse && !currentAssistantContent.some((b) => b?.type === 'tool_use')) {
1503
+ emit(chalk.yellow('\n[the model stopped for a tool call but sent none — ending turn]'));
1504
+ session.awaitingSkillAnswer = false;
1505
+ await saveSession(session);
1506
+ return;
1507
+ }
1508
+ if (!sawPauseTurn)
1509
+ pauseTurnContinuations = 0;
1510
+ // Task 16 review — a paused message that already carries a client
1511
+ // tool_use is dispatched like a tool_use stop (its result is the next
1512
+ // user message), not re-sent blind.
1513
+ const pausedWithToolUse = sawPauseTurn && currentAssistantContent.some((b) => b?.type === 'tool_use');
1514
+ if (pausedWithToolUse)
1515
+ pauseTurnContinuations = 0;
1516
+ if (!sawToolUse && !pausedWithToolUse) {
1517
+ // Task 16 (F6) — 'pause_turn': a server tool (tool search) paused the
1518
+ // turn. The paused assistant message is already persisted; re-send
1519
+ // with it last and NO new user message, so the model resumes. At most
1520
+ // 3 in a row (never observed in probe P6, handled anyway).
1521
+ const MAX_PAUSE_CONTINUATIONS = 3;
1522
+ if (sawPauseTurn) {
1523
+ if (pauseTurnContinuations < MAX_PAUSE_CONTINUATIONS) {
1524
+ pauseTurnContinuations += 1;
1525
+ continue;
1526
+ }
1527
+ emit(chalk.yellow(`\n[the model paused ${MAX_PAUSE_CONTINUATIONS + 1} times in a row — stopping. Type "continue" to resume.]`));
1528
+ session.awaitingSkillAnswer = false;
1529
+ await saveSession(session);
1530
+ return;
1531
+ }
1113
1532
  // The response was cut off at the output-token cap (stop_reason
1114
1533
  // 'max_tokens'/'length') mid-generation — NOT a real end of turn. The
1115
1534
  // truncated assistant message is already persisted (line ~733), so
@@ -1146,102 +1565,186 @@ ${imageExpansion.error}`));
1146
1565
  // continue to run normally. See ./parallel-write-guard.ts.
1147
1566
  const writeGuard = newGuardState();
1148
1567
  const toolResults = [];
1568
+ // Task 18 (spec D18, rulings F21, F27) — concurrent reader group. The
1569
+ // ⏺ lines print first, then one spinner / heartbeat / status label
1570
+ // ("N readers running…"). At most MAX_PARALLEL_READERS run at once; each
1571
+ // gets a shallow ctx clone with its own toolUseId (the shared slot is
1572
+ // never used) and an adt whose calls go through ONE lock (one ADT request
1573
+ // at a time). The keepalive is armed once for the group.
1574
+ const readerResults = new Map();
1575
+ const runReaderGroup = async (group) => {
1576
+ const label = `${group.length} reader${group.length === 1 ? '' : 's'} running…`;
1577
+ for (const b of group) {
1578
+ renderToolCallTop({ name: b.name, args: (b.input ?? {}), chunkEmitter: toolRowEmitter });
1579
+ }
1580
+ const groupSpinner = startToolSpinner({ chunkEmitter: livenessEmitter, label });
1581
+ status.activity(label);
1582
+ const groupHeartbeat = startThinkingHeartbeat({ chunkEmitter: livenessEmitter, label });
1583
+ const adtLock = createAdtLock();
1584
+ armKeepalive();
1585
+ try {
1586
+ for (let i = 0; i < group.length; i += MAX_PARALLEL_READERS) {
1587
+ // Fix I1 — Esc / Ctrl+C: no further chunk starts; every reader not
1588
+ // started gets an interrupted result so each tool_use stays paired.
1589
+ if (params.signal?.aborted) {
1590
+ for (const b of group.slice(i)) {
1591
+ readerResults.set(b.id, { result: { content: INTERRUPTED_LINE, is_error: true }, durationMs: 0 });
1592
+ }
1593
+ break;
1594
+ }
1595
+ await Promise.all(group.slice(i, i + MAX_PARALLEL_READERS).map(async (b) => {
1596
+ const started = Date.now();
1597
+ const clone = {
1598
+ ...params.ctx,
1599
+ toolUseId: b.id,
1600
+ adt: params.ctx.adt ? serializeAdt(params.ctx.adt, adtLock) : params.ctx.adt,
1601
+ };
1602
+ let r;
1603
+ try {
1604
+ r = await dispatchTool(b.name, b.input, clone);
1605
+ }
1606
+ catch (err) {
1607
+ r = { content: JSON.stringify({ error: String(err) }), is_error: true };
1608
+ }
1609
+ readerResults.set(b.id, { result: r, durationMs: Date.now() - started });
1610
+ }));
1611
+ }
1612
+ }
1613
+ finally {
1614
+ keepalive.disarm();
1615
+ groupHeartbeat.stop();
1616
+ groupSpinner.stop();
1617
+ }
1618
+ // Fix M3 — an interrupted turn writes no further model-call line, so
1619
+ // the readers' spend is recorded now on a zero-token rollup line.
1620
+ const spend = params.ctx.readerSpend;
1621
+ if (params.signal?.aborted && spend && spend.calls > readerCallsLogged) {
1622
+ void appendCostLine(session.id, buildCostEntry({
1623
+ turn: turnNumber,
1624
+ model: params.modelOverride ?? session.model,
1625
+ tokens: { input: 0, output: 0, cacheRead: 0, cacheCreate: 0 },
1626
+ kind: 'reader_rollup',
1627
+ reader: spend,
1628
+ }));
1629
+ readerCallsLogged = spend.calls;
1630
+ }
1631
+ };
1149
1632
  for (const block of currentAssistantContent) {
1150
1633
  if (block.type === 'tool_use') {
1151
- // Phase 1: print the dispatch line (CC-style: ⏺ name(args)).
1152
- // No-op for widget-suppressed tools (ask_question) — the form/modal
1153
- // is the visible representation; see tool-widget.ts.
1154
- renderToolCallTop({
1155
- name: block.name,
1156
- args: (block.input ?? {}),
1157
- chunkEmitter: toolRowEmitter,
1158
- });
1159
- // Phase 2a: start the peach-themed spinner. No-op outside TTY / Ink
1160
- // / CI — the result line still prints, just without the in-place
1161
- // animation. Spinner is purely a "still working" cue, not
1162
- // load-bearing for output. Skipped entirely for widget-suppressed
1163
- // tools (ask_question): with the ⏺ line gone the spinner would
1164
- // orphan on its own row above the form.
1165
- const spinner = isWidgetSuppressedTool(block.name)
1166
- ? { stop: () => undefined }
1167
- : startToolSpinner({ chunkEmitter: livenessEmitter, label: toolActivityText(block.name, (block.input ?? {})) });
1168
- // 2026-06-06 (turn-liveness, B5 smoke feedback) — the in-place tool
1169
- // spinner self-disables under Ink (phantom cursor #25), which left
1170
- // slow tool calls (SAP over VPN: 10-60s) as DEAD AIR between the ⏺
1171
- // dispatch line and the ⎿ result line. Labelled heartbeat prints
1172
- // fresh "<tool> running… (10s)" lines at thresholds — visible on
1173
- // every terminal, Ink included.
1174
- // Interactive tools wait on the USER, not the system — ticking
1175
- // "ask_question running… (30s)" while they think is noise.
1176
- // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
1177
- // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
1178
- // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
1179
- // BEFORE the op runs. file_write / shell_exec are otherwise classified
1180
- // non-interactive, so without this the loop would start the per-second
1181
- // heartbeat + keep the turn-status row ticking UNDER the modal — the
1182
- // observed live bug (doubled card, lost Enter, history-replay leak,
1183
- // ~10-min wedge with the tool spinner ticking under the modal). Detect
1184
- // the gate the SAME way tool-dispatch does (tool.isMutating &&
1185
- // shouldGateRule8) and treat it as user-blocking: pause the status row,
1186
- // skip the heartbeat. The gate's own clearActiveSpinner +
1187
- // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
1188
- // is the outer one — together no live render source contends with the
1189
- // modal for the Ink frame or raw-mode stdin.
1190
- const willTripBatchGate = (() => {
1191
- const t = getTool(block.name);
1192
- return !!t?.isMutating && shouldGateRule8();
1193
- })();
1194
- const isInteractiveTool = block.name === 'ask_question' ||
1195
- block.name === 'request_approval' ||
1196
- willTripBatchGate;
1197
- // Interactive tools block on the USER. PAUSE the turn-status row (don't
1198
- // just relabel it): a live 250ms tick repaints the dynamic frame and
1199
- // overdraws the inquirer approval picker / churns the Ink ask_question
1200
- // modal — the hidden-question + stacked-border + lost-Enter bug
1201
- // (2026-06-07). resume() in the finally brings it back the moment the
1202
- // user answers. Non-interactive tools keep the ticking label + heartbeat.
1203
- if (isInteractiveTool)
1204
- turnStatusEmitter.pause();
1205
- else
1206
- turnStatusEmitter.activity(toolActivityText(block.name, (block.input ?? {})));
1207
- const toolHeartbeat = isInteractiveTool
1208
- ? { stop: () => undefined }
1209
- : startThinkingHeartbeat({
1210
- chunkEmitter: livenessEmitter,
1211
- label: toolHeartbeatLabel(block.name, (block.input ?? {})),
1212
- });
1213
- // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
1214
- const dispatchStart = Date.now();
1215
- // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1216
- // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1217
- // id the loop records below. Dispatch is sequential, so a single slot
1218
- // on the shared ctx is safe.
1219
- params.ctx.toolUseId = block.id;
1220
- // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
1221
- const guardDecision = checkAndMark(block.name, writeGuard);
1634
+ // Task 18 (spec D18) — a run of consecutive read_agent blocks runs as
1635
+ // ONE concurrent group the first time the loop reaches it; each block
1636
+ // then takes its stored result below. Every other tool keeps today's
1637
+ // sequential dispatch.
1638
+ if (block.name === READ_AGENT && !readerResults.has(block.id)) {
1639
+ await runReaderGroup(readerRunFrom(currentAssistantContent, currentAssistantContent.indexOf(block)));
1640
+ }
1641
+ const ranInGroup = readerResults.get(block.id);
1222
1642
  let result;
1223
- try {
1224
- result = guardDecision.allow
1225
- ? await dispatchTool(block.name, block.input, params.ctx)
1226
- : { content: guardDecision.errorContent, is_error: true };
1643
+ let durationMs;
1644
+ if (ranInGroup) {
1645
+ result = ranInGroup.result;
1646
+ durationMs = ranInGroup.durationMs;
1227
1647
  }
1228
- finally {
1229
- // D19: clear the slot so a future non-loop invocation on this ctx
1230
- // can't inherit a stale block id.
1231
- params.ctx.toolUseId = undefined;
1232
- // Stop on the error path too — the interval is unref'd but would
1233
- // otherwise keep printing "<tool> running…" into the NEXT prompt
1234
- // after a dispatch throw.
1235
- toolHeartbeat.stop();
1236
- // Resume the turn-status row now that the user has answered (or the
1237
- // interactive dispatch threw). No-op for non-interactive tools.
1648
+ else {
1649
+ // Phase 1: print the dispatch line (CC-style: ⏺ name(args)).
1650
+ // No-op for widget-suppressed tools (ask_question) — the form/modal
1651
+ // is the visible representation; see tool-widget.ts.
1652
+ renderToolCallTop({
1653
+ name: block.name,
1654
+ args: (block.input ?? {}),
1655
+ chunkEmitter: toolRowEmitter,
1656
+ });
1657
+ // Phase 2a: start the peach-themed spinner. No-op outside TTY / Ink
1658
+ // / CI — the result line still prints, just without the in-place
1659
+ // animation. Spinner is purely a "still working" cue, not
1660
+ // load-bearing for output. Skipped entirely for widget-suppressed
1661
+ // tools (ask_question): with the ⏺ line gone the spinner would
1662
+ // orphan on its own row above the form.
1663
+ const spinner = isWidgetSuppressedTool(block.name)
1664
+ ? { stop: () => undefined }
1665
+ : startToolSpinner({ chunkEmitter: livenessEmitter, label: toolActivityText(block.name, (block.input ?? {})) });
1666
+ // 2026-06-06 (turn-liveness, B5 smoke feedback) — the in-place tool
1667
+ // spinner self-disables under Ink (phantom cursor #25), which left
1668
+ // slow tool calls (SAP over VPN: 10-60s) as DEAD AIR between the ⏺
1669
+ // dispatch line and the ⎿ result line. Labelled heartbeat prints
1670
+ // fresh "<tool> running… (10s)" lines at thresholds — visible on
1671
+ // every terminal, Ink included.
1672
+ // Interactive tools wait on the USER, not the system — ticking
1673
+ // "ask_question running… (30s)" while they think is noise.
1674
+ // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
1675
+ // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
1676
+ // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
1677
+ // BEFORE the op runs. file_write / shell_exec are otherwise classified
1678
+ // non-interactive, so without this the loop would start the per-second
1679
+ // heartbeat + keep the turn-status row ticking UNDER the modal — the
1680
+ // observed live bug (doubled card, lost Enter, history-replay leak,
1681
+ // ~10-min wedge with the tool spinner ticking under the modal). Detect
1682
+ // the gate the SAME way tool-dispatch does (tool.isMutating &&
1683
+ // shouldGateRule8) and treat it as user-blocking: pause the status row,
1684
+ // skip the heartbeat. The gate's own clearActiveSpinner +
1685
+ // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
1686
+ // is the outer one — together no live render source contends with the
1687
+ // modal for the Ink frame or raw-mode stdin.
1688
+ const willTripBatchGate = (() => {
1689
+ const t = getTool(block.name);
1690
+ return !!t?.isMutating && shouldGateRule8();
1691
+ })();
1692
+ const isInteractiveTool = block.name === 'ask_question' ||
1693
+ block.name === 'request_approval' ||
1694
+ willTripBatchGate;
1695
+ // Interactive tools block on the USER. PAUSE the turn-status row (don't
1696
+ // just relabel it): a live 250ms tick repaints the dynamic frame and
1697
+ // overdraws the inquirer approval picker / churns the Ink ask_question
1698
+ // modal — the hidden-question + stacked-border + lost-Enter bug
1699
+ // (2026-06-07). resume() in the finally brings it back the moment the
1700
+ // user answers. Non-interactive tools keep the ticking label + heartbeat.
1238
1701
  if (isInteractiveTool)
1239
- turnStatusEmitter.resume();
1702
+ status.pause();
1703
+ else
1704
+ status.activity(toolActivityText(block.name, (block.input ?? {})));
1705
+ const toolHeartbeat = isInteractiveTool
1706
+ ? { stop: () => undefined }
1707
+ : startThinkingHeartbeat({
1708
+ chunkEmitter: livenessEmitter,
1709
+ label: toolHeartbeatLabel(block.name, (block.input ?? {})),
1710
+ });
1711
+ // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
1712
+ const dispatchStart = Date.now();
1713
+ // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1714
+ // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1715
+ // id the loop records below. Dispatch is sequential, so a single slot
1716
+ // on the shared ctx is safe.
1717
+ params.ctx.toolUseId = block.id;
1718
+ // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
1719
+ const guardDecision = checkAndMark(block.name, writeGuard);
1720
+ // Task 15 — keep the prompt cache warm while the tool (or the user
1721
+ // answering its prompt) takes time; disarmed in the finally.
1722
+ if (guardDecision.allow)
1723
+ armKeepalive();
1724
+ try {
1725
+ result = guardDecision.allow
1726
+ ? await dispatchTool(block.name, block.input, params.ctx)
1727
+ : { content: guardDecision.errorContent, is_error: true };
1728
+ }
1729
+ finally {
1730
+ keepalive.disarm();
1731
+ // D19: clear the slot so a future non-loop invocation on this ctx
1732
+ // can't inherit a stale block id.
1733
+ params.ctx.toolUseId = undefined;
1734
+ // Stop on the error path too — the interval is unref'd but would
1735
+ // otherwise keep printing "<tool> running…" into the NEXT prompt
1736
+ // after a dispatch throw.
1737
+ toolHeartbeat.stop();
1738
+ // Resume the turn-status row now that the user has answered (or the
1739
+ // interactive dispatch threw). No-op for non-interactive tools.
1740
+ if (isInteractiveTool)
1741
+ status.resume();
1742
+ }
1743
+ durationMs = Date.now() - dispatchStart;
1744
+ // Phase 2c: stop the spinner, which erases its line so the result
1745
+ // row paints in place of it (no scrollback artifacts).
1746
+ spinner.stop();
1240
1747
  }
1241
- const durationMs = Date.now() - dispatchStart;
1242
- // Phase 2c: stop the spinner, which erases its line so the result
1243
- // row paints in place of it (no scrollback artifacts).
1244
- spinner.stop();
1245
1748
  // Phase 3: print result line ( ⎿ ✓ summary · timing).
1246
1749
  // 2026-05-15 (bug 1): the old `slice(0, 80)` clipped to the JSON
1247
1750
  // preamble (e.g. `{"error":"write_failed","detail":"HTTP 409: <?xml v…`)
@@ -1367,6 +1870,11 @@ ${imageExpansion.error}`));
1367
1870
  }
1368
1871
  }
1369
1872
  session.messages.push({ role: 'user', content: toolResults });
1873
+ // Task 16 — a successful dispatch_skill in this round moves the session
1874
+ // to another skill: surface that skill's starters now, right after the
1875
+ // tool-result user message (append-only; tools[] is unchanged).
1876
+ for (const skill of dispatchedSkills(currentAssistantContent, toolResults))
1877
+ surfaceStarters(skill);
1370
1878
  await saveSession(session);
1371
1879
  // v0.6 Layer 3 — between-rounds watchdog check. If the turn has been
1372
1880
  // running too long (wall-clock or token budget), inject a synthetic
@@ -1390,3 +1898,22 @@ ${imageExpansion.error}`));
1390
1898
  // Loop continues → next messages.create with the tool results.
1391
1899
  }
1392
1900
  }
1901
+ /**
1902
+ * Task 16 — skills a successful `dispatch_skill` call in this round queued
1903
+ * (`/abap-fiori-build --from @x` ⇒ `abap-fiori-build`, aliases resolved).
1904
+ */
1905
+ function dispatchedSkills(assistantContent, toolResults) {
1906
+ const out = [];
1907
+ for (const b of assistantContent) {
1908
+ if (b?.type !== 'tool_use' || b.name !== 'dispatch_skill')
1909
+ continue;
1910
+ const r = toolResults.find((t) => t?.tool_use_id === b.id);
1911
+ if (!r || r.is_error)
1912
+ continue;
1913
+ const command = typeof b.input?.command === 'string' ? b.input.command.trim() : '';
1914
+ const head = command.startsWith('/') ? command.slice(1).split(/\s+/)[0] : '';
1915
+ if (head)
1916
+ out.push(resolveSkillAlias(head));
1917
+ }
1918
+ return out;
1919
+ }