@cspeach/cli 1.1.18 → 1.1.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (252) hide show
  1. package/README.md +3 -4
  2. package/dist/agent/anthropic-provider.js +30 -10
  3. package/dist/agent/cache-keepalive.js +162 -0
  4. package/dist/agent/cold-prune.js +116 -0
  5. package/dist/agent/loop.js +804 -159
  6. package/dist/agent/provider-shape.js +263 -0
  7. package/dist/agent/providers/ai-hub-provider.js +17 -2
  8. package/dist/agent/providers/byok-provider.js +33 -3
  9. package/dist/agent/providers/local-provider.js +12 -2
  10. package/dist/agent/repair-partial.js +125 -7
  11. package/dist/agent/summarise-via-provider.js +6 -1
  12. package/dist/agent/system-prompt.js +38 -0
  13. package/dist/agent/tool-dispatch.js +66 -21
  14. package/dist/agent/tool-loading-pin.js +100 -0
  15. package/dist/agent/turn-error-ux.js +77 -26
  16. package/dist/agent/turn-stream.js +25 -3
  17. package/dist/approvals/adt-type.js +27 -0
  18. package/dist/approvals/advisory-prompt.js +48 -0
  19. package/dist/approvals/advisory-render.js +25 -11
  20. package/dist/approvals/approval-prompt.js +74 -24
  21. package/dist/approvals/jwt.js +2 -1
  22. package/dist/approvals/render.js +64 -36
  23. package/dist/approvals/risk-floor.js +53 -2
  24. package/dist/auth/api-key.js +30 -16
  25. package/dist/auth/keychain.js +0 -0
  26. package/dist/auth/me.js +25 -7
  27. package/dist/cli.js +36 -6
  28. package/dist/commands/auto-compact.js +33 -16
  29. package/dist/commands/compact.js +49 -11
  30. package/dist/commands/config-set.js +16 -3
  31. package/dist/commands/config-show.js +18 -5
  32. package/dist/commands/cost.js +54 -35
  33. package/dist/commands/help.js +88 -50
  34. package/dist/commands/login.js +28 -19
  35. package/dist/commands/logout.js +7 -3
  36. package/dist/commands/plan-audit.js +1 -0
  37. package/dist/commands/plan-chain.js +71 -77
  38. package/dist/commands/plan-continue.js +1 -0
  39. package/dist/commands/plan-resume.js +20 -12
  40. package/dist/commands/whoami.js +23 -9
  41. package/dist/config/loader.js +70 -5
  42. package/dist/cost/cost-log.js +62 -2
  43. package/dist/cost/pricing.js +11 -6
  44. package/dist/doctor/checks/forge-rules.js +13 -3
  45. package/dist/doctor/checks/keychain.js +6 -4
  46. package/dist/doctor/run.js +101 -20
  47. package/dist/index.js +5 -2
  48. package/dist/lib/graceful-exit.js +56 -0
  49. package/dist/lib/piped-prompt.js +65 -0
  50. package/dist/lib/spill-labels.js +13 -0
  51. package/dist/lock-contention.js +2 -2
  52. package/dist/models/resolve.js +93 -2
  53. package/dist/models/server-config.js +158 -3
  54. package/dist/one-shot.js +150 -45
  55. package/dist/projects/answer-blockers.js +6 -1
  56. package/dist/projects/handover-md.js +15 -14
  57. package/dist/projects/image-attachments.js +15 -2
  58. package/dist/projects/plan-run.js +97 -54
  59. package/dist/projects/promote-command.js +11 -2
  60. package/dist/projects/save-command.js +3 -2
  61. package/dist/projects/status.js +2 -1
  62. package/dist/projects/yes-no.js +12 -0
  63. package/dist/renderer/abap-inline.js +18 -14
  64. package/dist/renderer/answer-trim.js +69 -0
  65. package/dist/renderer/banners.js +8 -8
  66. package/dist/renderer/brand-settled.js +19 -0
  67. package/dist/renderer/change-card.js +156 -0
  68. package/dist/renderer/checklist-format.js +130 -0
  69. package/dist/renderer/color-mode.js +45 -0
  70. package/dist/renderer/error-detail.js +104 -0
  71. package/dist/renderer/fence-state.js +46 -0
  72. package/dist/renderer/footer-line.js +100 -0
  73. package/dist/renderer/glyphs.js +25 -0
  74. package/dist/renderer/goodbye.js +38 -0
  75. package/dist/renderer/legacy-palette.js +41 -0
  76. package/dist/renderer/look.js +30 -0
  77. package/dist/renderer/markdown.js +299 -92
  78. package/dist/renderer/notice-log.js +72 -0
  79. package/dist/renderer/notice-shape.js +48 -0
  80. package/dist/renderer/notices.js +40 -8
  81. package/dist/renderer/panel-rows.js +16 -0
  82. package/dist/renderer/pipeline.js +66 -0
  83. package/dist/renderer/progress-chatter.js +23 -17
  84. package/dist/renderer/rendering-mode.js +34 -0
  85. package/dist/renderer/routing-line.js +14 -0
  86. package/dist/renderer/sanitize.js +12 -0
  87. package/dist/renderer/severity.js +2 -7
  88. package/dist/renderer/startup-lines.js +154 -0
  89. package/dist/renderer/status-footer.js +52 -34
  90. package/dist/renderer/steering-echo.js +41 -0
  91. package/dist/renderer/syntax.js +31 -12
  92. package/dist/renderer/tables.js +5 -1
  93. package/dist/renderer/theme.js +44 -0
  94. package/dist/renderer/thinking-heartbeat.js +33 -2
  95. package/dist/renderer/todo-block.js +10 -49
  96. package/dist/renderer/tool-labels.js +269 -0
  97. package/dist/renderer/tool-widget.js +245 -51
  98. package/dist/renderer/trace.js +42 -0
  99. package/dist/renderer/transcript-flow.js +132 -0
  100. package/dist/renderer/tty.js +34 -1
  101. package/dist/renderer/ui-width.js +55 -0
  102. package/dist/renderer/verify-chain.js +4 -2
  103. package/dist/renderer/widget-fallback.js +58 -65
  104. package/dist/repl/bracketed-paste.js +7 -1
  105. package/dist/repl/builtin-commands.js +19 -8
  106. package/dist/repl/credential-handover-gate.js +28 -0
  107. package/dist/repl/current-transport.js +13 -0
  108. package/dist/repl/early-line-buffer.js +5 -2
  109. package/dist/repl/file-picker.js +54 -12
  110. package/dist/repl/ink-stdin-guard.js +66 -0
  111. package/dist/repl/inquirer-guard.js +59 -16
  112. package/dist/repl/inquirer-theme.js +27 -33
  113. package/dist/repl/paste-marker.js +19 -0
  114. package/dist/repl/plan-turn-end.js +20 -0
  115. package/dist/repl/post-turn-status.js +43 -11
  116. package/dist/repl/reroute-turn.js +21 -0
  117. package/dist/repl/reset-tty-stdin.js +44 -0
  118. package/dist/repl/restore-guard.js +22 -0
  119. package/dist/repl/resume-standalone.js +63 -0
  120. package/dist/repl/rule8-detector.js +11 -3
  121. package/dist/repl/safety-confirm.js +161 -93
  122. package/dist/repl/session-spend-line.js +8 -4
  123. package/dist/repl/themed-prompts.js +19 -0
  124. package/dist/repl/ui-look-command.js +75 -0
  125. package/dist/repl/update-method-preview-hook.js +48 -5
  126. package/dist/repl.js +863 -339
  127. package/dist/rewind/cli.js +3 -2
  128. package/dist/rewind/format.js +14 -9
  129. package/dist/rewind/restore.js +11 -0
  130. package/dist/router/classifier.js +47 -4
  131. package/dist/router/intent-extractor.js +24 -8
  132. package/dist/router/piped-routing.js +23 -0
  133. package/dist/router/resume-routing.js +16 -0
  134. package/dist/router/routing-failure.js +57 -0
  135. package/dist/sap/first-run-choice.js +1 -1
  136. package/dist/sap/onboarding.js +3 -2
  137. package/dist/sap/standalone-onboarding.js +1 -1
  138. package/dist/sap/system-info.js +4 -2
  139. package/dist/sap/unreachable.js +28 -0
  140. package/dist/sap-errors/clean-error-text.js +182 -0
  141. package/dist/session/interrupt-reason.js +39 -0
  142. package/dist/session/recap.js +36 -24
  143. package/dist/session/repin-model.js +18 -0
  144. package/dist/session/resume.js +104 -36
  145. package/dist/session/store.js +84 -28
  146. package/dist/session/user-prompt.js +21 -0
  147. package/dist/skill-catalog.js +7 -0
  148. package/dist/skills/bundled-skills.js +76 -69
  149. package/dist/skills/preamble.js +75 -0
  150. package/dist/skills/source-manifest.js +11 -1
  151. package/dist/standards/standards-init.js +1 -1
  152. package/dist/test-helpers/answer-any.js +46 -0
  153. package/dist/tools/approval.js +27 -11
  154. package/dist/tools/ask-question.js +47 -5
  155. package/dist/tools/filesystem/file-read.js +11 -1
  156. package/dist/tools/filesystem/file-write.js +19 -2
  157. package/dist/tools/fiori/fe-extend.js +16 -1
  158. package/dist/tools/fiori/fe-scaffold.js +105 -3
  159. package/dist/tools/fiori/html-escapes.js +28 -0
  160. package/dist/tools/fiori/metadata/audit.js +605 -0
  161. package/dist/tools/fiori/metadata/references.js +104 -0
  162. package/dist/tools/fiori/metadata/types.js +8 -0
  163. package/dist/tools/fiori/preview/app-guard.js +215 -0
  164. package/dist/tools/fiori/preview/env.js +193 -0
  165. package/dist/tools/fiori/preview/readiness.js +127 -0
  166. package/dist/tools/fiori/preview/registry.js +412 -0
  167. package/dist/tools/fiori/preview/start.js +469 -0
  168. package/dist/tools/fiori/samples/loader.js +20 -5
  169. package/dist/tools/fiori/scaffold.js +18 -6
  170. package/dist/tools/fiori/smoke/app-driver-page.js +372 -0
  171. package/dist/tools/fiori/smoke/app-driver.js +100 -0
  172. package/dist/tools/fiori/smoke/assertions.js +161 -24
  173. package/dist/tools/fiori/smoke/audit-columns.js +21 -0
  174. package/dist/tools/fiori/smoke/batch.js +273 -0
  175. package/dist/tools/fiori/smoke/draft-safety.js +241 -0
  176. package/dist/tools/fiori/smoke/draft-smoke.js +473 -0
  177. package/dist/tools/fiori/smoke/driver.js +171 -4
  178. package/dist/tools/fiori/smoke/evidence.js +35 -0
  179. package/dist/tools/fiori/smoke/labels-codes.js +207 -0
  180. package/dist/tools/fiori/smoke/nav-error.js +26 -0
  181. package/dist/tools/fiori/smoke/run-smoke.js +155 -12
  182. package/dist/tools/fiori/tools.js +651 -7
  183. package/dist/tools/fiori/ui5-version.js +16 -0
  184. package/dist/tools/index.js +8 -0
  185. package/dist/tools/local-build.js +6 -0
  186. package/dist/tools/result-spill.js +238 -0
  187. package/dist/tools/sap-read.js +74 -15
  188. package/dist/tools/sap-write.js +122 -10
  189. package/dist/tools/shell/shell_exec.js +39 -12
  190. package/dist/tools/subagent/adt-serial.js +33 -0
  191. package/dist/tools/subagent/agent_run.js +2 -0
  192. package/dist/tools/subagent/read_agent.js +178 -0
  193. package/dist/tools/subagent/reader-prompt.js +48 -0
  194. package/dist/tools/syntax-state.js +67 -0
  195. package/dist/tools/todo.js +26 -19
  196. package/dist/tools/tool-loading.js +255 -0
  197. package/dist/tools/tool-output-read.js +117 -0
  198. package/dist/tools/transport-fit.js +462 -0
  199. package/dist/tools/transport-resolution.js +3 -1
  200. package/dist/tools/transport.js +39 -8
  201. package/dist/tools/verify.js +184 -8
  202. package/dist/ui/app.js +284 -146
  203. package/dist/ui/approval-modal.js +41 -15
  204. package/dist/ui/ask-answer-text.js +17 -0
  205. package/dist/ui/body.js +25 -6
  206. package/dist/ui/card-slot.js +98 -0
  207. package/dist/ui/checklist.js +9 -0
  208. package/dist/ui/coaching-picker-classic.js +8 -5
  209. package/dist/ui/command-palette.js +6 -4
  210. package/dist/ui/confirm-request.js +18 -0
  211. package/dist/ui/context-grid.js +2 -1
  212. package/dist/ui/file-palette.js +6 -4
  213. package/dist/ui/files-card-emitter.js +19 -0
  214. package/dist/ui/footer-line.js +16 -0
  215. package/dist/ui/footer.js +105 -104
  216. package/dist/ui/input-wrap.js +122 -0
  217. package/dist/ui/interrupt.js +103 -0
  218. package/dist/ui/key-burst.js +117 -0
  219. package/dist/ui/line-resolution.js +4 -3
  220. package/dist/ui/live-area.js +71 -0
  221. package/dist/ui/login-banner.js +54 -43
  222. package/dist/ui/rewind-panel.js +76 -24
  223. package/dist/ui/role-props.js +47 -0
  224. package/dist/ui/sap-state-store.js +2 -1
  225. package/dist/ui/session-timeline.js +49 -3
  226. package/dist/ui/skill-picker.js +5 -47
  227. package/dist/ui/text-input.js +128 -24
  228. package/dist/ui/todo-emitter.js +19 -0
  229. package/dist/ui/turn-status-emitter.js +65 -2
  230. package/dist/ui/turn-status.js +16 -19
  231. package/dist/ui/use-card-request.js +44 -0
  232. package/dist/ui/widgets/ask-card.js +521 -0
  233. package/dist/ui/widgets/bar-chart.js +11 -5
  234. package/dist/ui/widgets/coaching-picker.js +50 -55
  235. package/dist/ui/widgets/confirm-card.js +193 -0
  236. package/dist/ui/widgets/dep-graph.js +3 -2
  237. package/dist/ui/widgets/diff-viewer.js +6 -5
  238. package/dist/ui/widgets/files-card.js +97 -0
  239. package/dist/ui/widgets/question-card.js +2 -1
  240. package/dist/ui/widgets/reclaim-raw-stdin.js +24 -0
  241. package/dist/ui/widgets/shortcuts-sheet.js +25 -0
  242. package/dist/ui/widgets/stack-frames.js +2 -1
  243. package/dist/upgrade-check.js +39 -8
  244. package/package.json +12 -2
  245. package/test-harness/fe-draft/app.ts +188 -0
  246. package/test-harness/fe-draft/fe-draft.e2e.test.ts +461 -0
  247. package/test-harness/fe-draft/manifest-enhancer.unit.test.ts +51 -0
  248. package/test-harness/fe-draft/routes.ts +224 -0
  249. package/test-harness/fe-draft/routes.unit.test.ts +45 -0
  250. package/test-harness/fe-draft/server.ts +225 -0
  251. package/test-harness/fe-draft/variants.ts +196 -0
  252. package/vitest.smoke-e2e.config.ts +20 -0
@@ -1,16 +1,25 @@
1
1
  import chalk from 'chalk';
2
2
  import { basename } from 'node:path';
3
3
  import { dispatchTool } from './tool-dispatch.js';
4
- import { listTools, getTool, toAnthropicTools, isStandalone, toolsForContext } from '../tools/index.js';
4
+ import { listTools, getTool, isStandalone, toolsForContext, toAnthropicTools } from '../tools/index.js';
5
5
  import { saveSession } from '../session/store.js';
6
6
  import { recordCompletedToolCall } from '../session/pending.js';
7
7
  import { loadConfig } from '../config/loader.js';
8
- import { resolveModelRole } from '../models/resolve.js';
8
+ import { resolveModelRole, resolveEffort, modelAcceptsEffort, resolveAutoCompactThreshold, callModelFor } from '../models/resolve.js';
9
+ import { getCustomerModelConfig, managedProxyServesNewShapes, servedToolLoading } from '../models/server-config.js';
10
+ import { buildToolList, buildSkillToolAddition, placeSystemMessagesInPlace, resolveToolLoading, } from '../tools/tool-loading.js';
11
+ import { settleSessionToolLoading } from './tool-loading-pin.js';
12
+ import { resolveSkillAlias } from '../skill-catalog.js';
9
13
  import { retryWithBackoff } from './retry.js';
10
14
  import { renderChunk, initialBufferState, render } from '../renderer/pipeline.js';
11
- import { renderToolCallTop, renderToolCallBottom, startToolSpinner, startThinkingSpinner, isWidgetSuppressedTool } from '../renderer/tool-widget.js';
15
+ import { nextWritingActivity } from '../renderer/fence-state.js';
16
+ import { renderToolCallTop, renderToolCallBottom, startToolSpinner, startThinkingSpinner, isWidgetSuppressedTool, toolActivityText, toolHeartbeatLabel } from '../renderer/tool-widget.js';
12
17
  import { formatToolErrorSummary } from '../sap-errors/parse-adt-exception.js';
13
18
  import { startThinkingHeartbeat } from '../renderer/thinking-heartbeat.js';
19
+ import { isDeclaredOneShot } from '../renderer/tty.js';
20
+ import { USER_STOP_REASON } from '../session/interrupt-reason.js';
21
+ import { writeStdout, stdoutFlow } from '../renderer/transcript-flow.js';
22
+ import { formatFailure, formatAttention, formatInfo } from '../renderer/notice-shape.js';
14
23
  // 2026-06-07 — status-row narration. Safe to import in every mode: the
15
24
  // emitter has zero listeners outside Ink (classic / one-shot / subagent),
16
25
  // so activity() calls are no-ops there.
@@ -18,13 +27,15 @@ import { turnStatusEmitter } from '../ui/turn-status-emitter.js';
18
27
  // (progress-chatter import removed 2026-05-01 — superseded by CC-style
19
28
  // two-line dispatch renderer; re-add if a future in-place spinner returns)
20
29
  import { buildRetryCapPausePayload, buildSkippedSiblingResults } from './retry-cap.js';
21
- import { appendCostLine, buildEntry as buildCostEntry } from '../cost/cost-log.js';
30
+ import { appendCostLine, buildEntry as buildCostEntry, summariseInputTransformations } from '../cost/cost-log.js';
31
+ import { modelAcceptsBinding, sessionMaxTokens, clampMaxTokens } from './provider-shape.js';
22
32
  import { computeSessionCost } from '../cost/session-cost.js';
23
33
  import { refreshCreditsAfterTurn, creditsForCostLog } from '../cost/credits-wire.js';
24
34
  import { globalStore } from '../ui/sap-state-store.js';
25
35
  import { lastAssistantIsAwaitingAnswer } from '../session/awaiting-answer.js';
26
36
  import { enrichUserMessage } from '../router/intent-extractor.js';
27
37
  import { resetRule8State, shouldGateRule8 } from '../repl/rule8-detector.js';
38
+ import { consumeTodoSetThisTurn } from '../ui/todo-emitter.js';
28
39
  import { runSaveCommand, runPromoteCommand, detectArtifactSkill, offerAnswerBlockers } from '../projects/index.js';
29
40
  import { parseFromFlag, coerceDocFromFlagToAttachment } from '../skills/promotion-dispatch.js';
30
41
  import { input } from '@inquirer/prompts';
@@ -37,7 +48,7 @@ import { renderStandaloneContextBlock } from '../sap/standalone-profile.js';
37
48
  import { renderStandardsBlock } from '../standards/standards-file.js';
38
49
  import { composeSessionContextBlock } from './session-context.js';
39
50
  import { objectKeyFromInput } from './retry-key.js';
40
- import { repairPartialBlocks, healSessionMessagesInPlace, healOrphanToolUses, isPartialJsonApiError } from './repair-partial.js';
51
+ import { repairPartialBlocks, stripPreFallbackBlocks, healSessionMessagesInPlace, healOrphanToolUses, isPartialJsonApiError, stripUnsignedThinking, isThinkingOnly, dropTrailingThinkingOnly } from './repair-partial.js';
41
52
  import { drainSteering } from './steering-queue.js';
42
53
  import { TurnStreamWriter } from './turn-stream.js';
43
54
  import { applyToolResultCheckpoint } from './skill-checkpoint.js';
@@ -48,6 +59,10 @@ import { getCurrentTransport } from '../repl/current-transport.js';
48
59
  import { collectTurnAssistantText } from './turn-assistant-text.js';
49
60
  import { checkAndMark, newGuardState } from './parallel-write-guard.js';
50
61
  import { userPromptText } from '../session/user-prompt.js';
62
+ import { CacheKeepalive, snapshotParams, recordLastRequest, isCacheablePrompt, noteCacheTouched } from './cache-keepalive.js';
63
+ import { maybeColdPrune } from './cold-prune.js';
64
+ import { createAdtLock, serializeAdt } from '../tools/subagent/adt-serial.js';
65
+ import { hasReaderReadTools, INTERRUPTED_LINE } from '../tools/subagent/reader-prompt.js';
51
66
  /**
52
67
  * Author identity for project-file metadata. Reads CSPEACH_AUTHOR_NAME first,
53
68
  * then platform USER/USERNAME, then a generic fallback. Role is fixed to
@@ -155,9 +170,52 @@ export async function inkAwarePrompt(q) {
155
170
  }
156
171
  return withInquirer(() => input({ message: q }));
157
172
  }
173
+ /**
174
+ * Piece 2 / Q3 (spec §3 "[y/N] prompts") — the Yes/No question for the six
175
+ * [y/N] sites (via projects/yes-no.ts askYesNo). Returns 'y' | 'n' | ''.
176
+ *
177
+ * - headless → '' (never blocks), same log line as inkAwarePrompt. The save
178
+ * hook keeps its blanket 'y' by NOT passing this in headless (maybeOfferSave).
179
+ * - Ink → an AskCard Yes/No (v1 choice 'y'/'n') under the card-slot focus
180
+ * claim: cursor on the default, marked "(default)"; y/n keys answer at
181
+ * once; Esc / no listener / anything else → '' (today's
182
+ * `result.answer ?? ''`). Leaves one decision trace in scrollback.
183
+ * - classic → inquirer input with today's exact "<q> [y/N]: " text.
184
+ */
185
+ export async function inkAwareConfirm(q, opts) {
186
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
187
+ if (isHeadless()) {
188
+ console.error(chalk.dim(`headless: skipped prompt '${q.trim()}' → '' (default)`));
189
+ return '';
190
+ }
191
+ if (shouldUseInk()) {
192
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
193
+ const { withFocusClaim } = await import('../ui/card-slot.js');
194
+ const { emitTrace, formatDecisionTrace } = await import('../renderer/trace.js');
195
+ const { uiWidth } = await import('../renderer/ui-width.js');
196
+ const res = await withFocusClaim(() => askQuestionEmitter.request({
197
+ id: 'agent-confirm',
198
+ question: q,
199
+ kind: 'choice',
200
+ choices: [{ value: 'y', label: 'Yes' }, { value: 'n', label: 'No' }],
201
+ defaultValue: opts.default,
202
+ keys: { y: 'y', n: 'n' },
203
+ }));
204
+ const answer = res.answer === 'y' || res.answer === 'n' ? res.answer : '';
205
+ emitTrace(formatDecisionTrace(q, answer === 'y' ? 'Yes' : answer === 'n' ? 'No' : 'cancelled', uiWidth()));
206
+ return answer;
207
+ }
208
+ const hint = opts.default === 'y' ? '[Y/n]' : '[y/N]';
209
+ const raw = (await withInquirer(() => input({ message: `${q} ${hint}: ` }))).trim().toLowerCase();
210
+ if (raw === 'y' || raw === 'yes')
211
+ return 'y';
212
+ if (raw === '')
213
+ return opts.default;
214
+ return 'n';
215
+ }
158
216
  /**
159
217
  * Choices picker for an answer-blocker that carries `options`. Renders the
160
- * native Ink AskQuestionModal (the same CC-style picker the ask_question tool
218
+ * native Ink AskCard (the same CC-style picker the ask_question tool
161
219
  * uses) with the candidate answers plus "Write my own" and "Skip" escapes.
162
220
  *
163
221
  * - headless → skip (never blocks on stdin)
@@ -220,7 +278,7 @@ export async function maybeOfferSave(p) {
220
278
  if (!detected)
221
279
  return;
222
280
  saveSkill = detected;
223
- p.emit(chalk.dim(`[save] ${detected} artifact manifest detected in this turn's output (turn was labelled '${p.skill}') — offering save.`));
281
+ p.emit(formatInfo(`${detected} output found in this turn (routed as ${p.skill}) · offering to save it`).trimEnd());
224
282
  }
225
283
  // B5 (2026-06-11) — defect D2: in headless runs the artifact IS the point
226
284
  // of the run, so the "Save as project file? [y/N]" prompt defaults YES
@@ -249,6 +307,9 @@ export async function maybeOfferSave(p) {
249
307
  author: getAuthorIdentity(),
250
308
  cwd: process.cwd(),
251
309
  prompt,
310
+ // Piece 2 / Q3 — interactive saves ask a Yes/No card; headless passes
311
+ // no confirm so the blanket-'y' prompt above still answers (B5 contract).
312
+ confirm: isHeadless() ? undefined : inkAwareConfirm,
252
313
  log: (...lines) => lines.forEach((l) => p.emit(l)),
253
314
  promotedFrom: p.promotedFrom ?? null,
254
315
  });
@@ -262,19 +323,111 @@ export async function maybeOfferSave(p) {
262
323
  path: savedPath,
263
324
  identity: getAuthorIdentity(),
264
325
  prompt: inkAwarePrompt,
326
+ confirm: inkAwareConfirm,
265
327
  choose: inkAwareChoose,
266
328
  log: (...lines) => lines.forEach((l) => p.emit(l)),
267
329
  });
268
330
  }
269
331
  }
270
332
  catch (err) {
271
- p.emit(chalk.yellow(`\n[save] skipped: ${err?.message ?? err}`));
333
+ p.emit('\n' + formatAttention({ text: 'Not saved as a project file', fact: String(err?.message ?? err) }).trimEnd());
272
334
  }
273
335
  }
336
+ /**
337
+ * Task 12 — sessions that have already shown the "set by your account admin"
338
+ * line. The REPL rebuilds `params.ctx` every turn but keeps the same
339
+ * `ctx.session` object, so the flag is keyed on it (session-scoped, never
340
+ * persisted: a WeakSet, not a session field).
341
+ */
342
+ const adminModelLineShown = new WeakSet();
343
+ /**
344
+ * Task 18 — a reader child must not repaint the parent's turn-status row
345
+ * (several readers run at once; the parent shows one "N readers running…").
346
+ */
347
+ const SILENT_STATUS = {
348
+ activity: (_label) => undefined,
349
+ pause: () => undefined,
350
+ resume: () => undefined,
351
+ };
352
+ /** Task 18 — at most this many read_agent calls run at the same time. */
353
+ export const MAX_PARALLEL_READERS = 4;
354
+ const READ_AGENT = 'read_agent';
355
+ /**
356
+ * Task 18 — the read_agent blocks of the run that starts at `start`: every
357
+ * following tool_use block up to the first one that is not read_agent
358
+ * (non-tool blocks such as text in between are skipped).
359
+ */
360
+ function readerRunFrom(content, start) {
361
+ const out = [];
362
+ for (let i = start; i < content.length; i++) {
363
+ const b = content[i];
364
+ if (b?.type !== 'tool_use')
365
+ continue;
366
+ if (b.name !== READ_AGENT)
367
+ break;
368
+ out.push(b);
369
+ }
370
+ return out;
371
+ }
372
+ /** Drop a trailing -YYYYMMDD date suffix from a model id (for equality checks only). */
373
+ function stripModelDate(model) {
374
+ return model?.replace(/-\d{8}$/, '');
375
+ }
274
376
  export async function runTurn(params) {
275
- resetRule8State(); // Rule 8 — fresh batch counter per LLM turn (= per user prompt)
377
+ // Task 18 — a reader subagent runs INSIDE the parent's turn: it must not
378
+ // reset the parent's Rule 8 batch counter, drain the parent's steering
379
+ // queue or repaint the parent's status row.
380
+ const readerChild = params.ctx.readerMode === true;
381
+ if (!readerChild)
382
+ resetRule8State(); // Rule 8 — fresh batch counter per LLM turn (= per user prompt)
383
+ const status = readerChild ? SILENT_STATUS : turnStatusEmitter;
384
+ // Task 18 (F24) — reader spend of THIS turn; read by every cost line below.
385
+ params.ctx.readerSpend = { cost: 0, calls: 0 };
386
+ // Fix I1 — the turn's abort signal travels on the ctx to reader clones and
387
+ // into their child turns.
388
+ params.ctx.signal = params.signal;
389
+ // Fix M3 — reader calls already carried by a written cost line.
390
+ let readerCallsLogged = 0;
391
+ // Piece 2 / P2 — fresh "plan changed this turn" flag per TOP-LEVEL turn.
392
+ // Not every path ends in pushPostTurnStatus (the Ink /reroute turn does
393
+ // not), so a mark left by such a turn must not earn the NEXT turn a close
394
+ // line. Subagent ctxs omit todoEmitter: a child turn keeps the parent's mark.
395
+ if (params.ctx.todoEmitter)
396
+ consumeTodoSetThisTurn();
397
+ // Panels spacing (polish P1): the REPL printed the prompt etc. outside the
398
+ // stdout transcript flow, so a stdout turn starts from "fresh line".
399
+ if (!params.chunkEmitter)
400
+ stdoutFlow.reset();
401
+ // Fix round 7 item 2 — one-shot passes its answer emitter (stdout, trimmed)
402
+ // as chunkEmitter. Liveness (spinners, heartbeats) must never enter the
403
+ // answer: in one-shot it keeps its own sinks (stderr heartbeat, in-place
404
+ // spinner only on a real terminal) exactly as before.
405
+ const livenessEmitter = isDeclaredOneShot() ? undefined : params.chunkEmitter;
406
+ // Owner ruling (r3 M11) — in one-shot the tool rows go with the tool
407
+ // notices (ctx.chunkEmitter → stderr), never into the answer on stdout.
408
+ const toolRowEmitter = isDeclaredOneShot() ? (params.ctx.chunkEmitter ?? params.chunkEmitter) : params.chunkEmitter;
276
409
  const cfg = await loadConfig();
277
410
  const session = params.ctx.session;
411
+ // Task 15 — prompt-cache keepalive while this turn waits (a slow tool, an
412
+ // approval / ask_question prompt, the save hook). One instance per turn, so
413
+ // `in_turn_pings` caps the whole turn. Managed and BYOK only (AI-hub and
414
+ // local have no `keepalive`). `lastRequest` is the SNAPSHOT of the last
415
+ // request this turn sent (ruling F4), never the live messages array.
416
+ const keepaliveOn = cfg.keepalive?.enabled !== false && typeof params.provider.keepalive === 'function';
417
+ const keepalive = new CacheKeepalive({
418
+ provider: params.provider,
419
+ maxPings: keepaliveOn ? (cfg.keepalive?.in_turn_pings ?? 11) : 0,
420
+ // Task 19 — a successful ping keeps the cache warm (read by the cold prune).
421
+ onPing: (_n, err) => { if (err === undefined)
422
+ noteCacheTouched(session); },
423
+ });
424
+ let lastRequest;
425
+ // Never ping before a request exists, or when the API cached nothing
426
+ // (prompt under the minimum cacheable size).
427
+ const armKeepalive = () => {
428
+ if (keepaliveOn && lastRequest?.cacheable)
429
+ keepalive.arm(lastRequest.params, lastRequest.options);
430
+ };
278
431
  // Define `emit` early so the Phase J promote dispatch (below) can stream
279
432
  // its prompt + status lines through the same Ink-aware sink the rest of
280
433
  // runTurn uses. Originally defined later in the function — hoisted in v0.5
@@ -294,7 +447,10 @@ export async function runTurn(params) {
294
447
  let promotedFromForSave = null;
295
448
  let userMessageForLLM = params.userMessage;
296
449
  let userMessageForSave = params.userMessage;
297
- {
450
+ // Task 18 fix M2 — a reader's brief is plain text from the model: no
451
+ // --from promotion, no image or @file expansion (lean prefix, and a child
452
+ // must never open an interactive prompt).
453
+ if (!readerChild) {
298
454
  // Forgiveness: `--from @<document>` (a .docx/.pdf/.txt/.md, not a saved
299
455
  // .cspeach.json) means "attach this file", not "chain a result". Rewrite it
300
456
  // to a plain @<file> attachment and let Phase D ingest it — never error out
@@ -311,6 +467,7 @@ export async function runTurn(params) {
311
467
  sourcePath: fromPath,
312
468
  targetSkill: params.skill,
313
469
  prompt: inkAwarePrompt,
470
+ confirm: inkAwareConfirm,
314
471
  log: (...lines) => lines.forEach((l) => emit(l)),
315
472
  });
316
473
  if (!result) {
@@ -340,7 +497,9 @@ export async function runTurn(params) {
340
497
  // or too many images → the turn is not sent (same early exit as --from).
341
498
  // Pasted clipboard images arrive as [Image #N] chips — swap each for its
342
499
  // saved file path first, so the step below attaches it like a typed path.
343
- const imageExpansion = await expandImageAttachments(expandImageChips(userMessageForLLM), params.ctx.cwd, (line) => emit(chalk.dim(line)));
500
+ const imageExpansion = readerChild
501
+ ? { text: userMessageForLLM, images: [] }
502
+ : await expandImageAttachments(expandImageChips(userMessageForLLM), params.ctx.cwd, (line) => emit(chalk.dim(line)));
344
503
  if (imageExpansion.error) {
345
504
  emit(chalk.yellow(`
346
505
  ${imageExpansion.error}`));
@@ -357,7 +516,9 @@ ${imageExpansion.error}`));
357
516
  // The save copy keeps the original `@<filename>` token rather than
358
517
  // the expanded body so the saved envelope's source.input stays
359
518
  // human-readable; only the LLM sees the inflated prose.
360
- userMessageForLLM = await expandTextFileAttachments(userMessageForLLM, params.ctx.cwd, (line) => emit(chalk.dim(line)));
519
+ if (!readerChild) {
520
+ userMessageForLLM = await expandTextFileAttachments(userMessageForLLM, params.ctx.cwd, (line) => emit(chalk.dim(line)));
521
+ }
361
522
  // 2026-05-06: alongside the deterministic prompt-fact extraction
362
523
  // (package / transport / etc.), inject the connected SAP system's
363
524
  // release + platform + ABAP version. One CVERS query the first time
@@ -385,7 +546,12 @@ ${imageExpansion.error}`));
385
546
  // user has no standards file, so the composed block (and therefore the whole
386
547
  // prompt) stays byte-identical to pre-feature output for existing users.
387
548
  const standardsBlock = await renderStandardsBlock(params.ctx.cwd);
388
- const sessionContextBlock = composeSessionContextBlock([sapSystemBlock, standardsBlock]);
549
+ // Round 6 review I4 — a resume moved this session to another system: tell
550
+ // the model once, on this turn, then drop the note.
551
+ const switchNote = session.systemSwitchNote
552
+ ? `<system_switch>${session.systemSwitchNote}</system_switch>`
553
+ : '';
554
+ const sessionContextBlock = composeSessionContextBlock([sapSystemBlock, standardsBlock, switchNote]);
389
555
  // Deterministically extract structured facts from the user's prompt
390
556
  // (package, transport, object names, environment hint, create intent)
391
557
  // and prepend them as an XML `<session_context>` block. The model reads
@@ -398,9 +564,12 @@ ${imageExpansion.error}`));
398
564
  // here, which would defeat a "first turn only" check. Skip only when the
399
565
  // message already begins with <session_context> (caller enriched it) or
400
566
  // when extraction yields nothing (enrichUserMessage returns original).
567
+ // Final batch 1 — a caller-enriched message still gets the switch note,
568
+ // inside its own <session_context> block.
401
569
  const enriched = userMessageForLLM.startsWith('<session_context>')
402
- ? userMessageForLLM
570
+ ? (switchNote ? userMessageForLLM.replace('<session_context>', `<session_context>\n ${switchNote}`) : userMessageForLLM)
403
571
  : enrichUserMessage(userMessageForLLM, sessionContextBlock);
572
+ const switchNoteUsed = switchNote !== '' && enriched.includes(switchNote);
404
573
  // Phase 2 of Track B — opt-in project-context enrichment, gated by
405
574
  // CSPEACH_PROJECT_CONTEXT=on. Default OFF: production behavior is
406
575
  // identical to today. When on, the rendered <project_context> block
@@ -434,6 +603,10 @@ ${imageExpansion.error}`));
434
603
  // Images (if any) ride as blocks ahead of the text; with none this is the
435
604
  // same plain string as before.
436
605
  session.messages.push({ role: 'user', content: buildUserContent(finalUserContent, imageExpansion.images) });
606
+ // Round 6 review I4 / final batch 1 — the switch note is dropped only once
607
+ // it is in a pushed user message.
608
+ if (switchNoteUsed)
609
+ delete session.systemSwitchNote;
437
610
  if (imageExpansion.images.length > 0 && params.provider.name === 'local-openai-compat') {
438
611
  // The local (OpenAI-compatible) provider keeps text blocks only.
439
612
  emit(chalk.dim('(local model: the image is not sent — the model gets the text only)'));
@@ -443,6 +616,90 @@ ${imageExpansion.error}`));
443
616
  emit(chalk.dim(`(${droppedImages} older image(s) removed from the conversation to keep it under the size limit)`));
444
617
  }
445
618
  session.skill = params.skill;
619
+ // Task 16 (spec D11) — deferred tool loading. The tools array is built ONE
620
+ // way for the whole session: the core set plus the starters of the skill
621
+ // pinned on `session.tool_set_skill` are loaded, every other tool is
622
+ // declared with defer_loading, the tool search tool goes last ('deferred');
623
+ // or today's full list ('static'). A changed array would break the prompt
624
+ // cache and the preserved-thinking prefix, so a skill switch APPENDS a
625
+ // `role: 'system'` tool_addition message instead (after a user message).
626
+ // Standalone hides the SAP tools (toolsForContext); the agent_run
627
+ // read-only filter (toolFilter) applies before the split, so a filtered-out
628
+ // tool is neither loaded nor deferred (cannot be found by search either).
629
+ session.tool_set_skill ??= params.skill;
630
+ const toolLoadingMode = resolveToolLoading({
631
+ providerMode: params.provider.mode,
632
+ env: process.env,
633
+ served: servedToolLoading(),
634
+ });
635
+ const contextTools = toolsForContext(listTools(), params.ctx);
636
+ // Task 18 fixes M4 / M5 — no read_agent for an agent_run child, and none
637
+ // when a reader would have no real read tool (standalone, file tools off).
638
+ const visibleTools = params.ctx.subagent || !hasReaderReadTools(new Set(contextTools.map((t) => t.name)))
639
+ ? contextTools.filter((t) => t.name !== READ_AGENT)
640
+ : contextTools;
641
+ // Final review C1 (controller ruling) — deferred shapes only for a model on
642
+ // the probe-proven allow-list, decided by the model actually sent on the
643
+ // call (a plan-tier sonnet override, a haiku child or claude-opus-4-8 ⇒
644
+ // static). Integration finding N1 (b): decided ONCE, at turn start — a
645
+ // model change mid-turn (an adopted served model) never strips deferred
646
+ // shapes in the middle of a turn (that would edit the prefix the in-flight
647
+ // thinking is bound to): the turn finishes on its own mode and the
648
+ // session settles at the next turn start. The proxy converts a deferred
649
+ // request for a model off its allow-list to static upstream.
650
+ // Final review I2 (controller ruling) — the mode is pinned on the session
651
+ // (session.tool_loading) at the first request and kept on resume; a
652
+ // deferred session that must turn static (kill switch, failed startup
653
+ // fetch, a model off the allow-list) has its deferred shapes stripped ONCE
654
+ // (a deliberate rewrite like /compact) and is re-pinned static.
655
+ const callToolMode = (model) => {
656
+ const settled = settleSessionToolLoading(session, { resolved: toolLoadingMode, model });
657
+ if (settled.stripped > 0) {
658
+ emit(chalk.dim('(this session now sends the full tool list — earlier tool-search steps were removed from the conversation)'));
659
+ }
660
+ return settled.mode;
661
+ };
662
+ const filteredTools = params.toolFilter ? visibleTools.filter(params.toolFilter) : visibleTools;
663
+ const builtTools = new Map();
664
+ const toolsFor = (mode) => {
665
+ let list = builtTools.get(mode);
666
+ if (!list) {
667
+ list = buildToolList(filteredTools, { skill: session.tool_set_skill, mode });
668
+ builtTools.set(mode, list);
669
+ }
670
+ return list;
671
+ };
672
+ // Task 18 — a reader's fixed list bypasses buildToolList entirely.
673
+ const readerTools = params.toolsOverride
674
+ ? toAnthropicTools(params.toolsOverride
675
+ .map((n) => visibleTools.find((t) => t.name === n))
676
+ .filter((t) => t !== undefined))
677
+ : undefined;
678
+ // Integration finding N1 (a) — the model each call is SENT with: an
679
+ // enforced customer model (managed, session role) replaces the pinned one,
680
+ // so call 0 already takes the static list when enforcement serves a model
681
+ // off the deferred allow-list.
682
+ const sentModel = () => callModelFor(params.modelOverride ?? session.model, { providerMode: params.provider.mode, role: params.modelRole });
683
+ const turnTools = readerTools ?? toolsFor(callToolMode(sentModel()));
684
+ /**
685
+ * Append a tool_addition for a skill's starters not yet loaded or surfaced.
686
+ * N1 (c) — built from the turn's list (the list every request of this turn
687
+ * carries) and never for a session pinned static.
688
+ */
689
+ const surfaceStarters = (skill) => {
690
+ const msg = buildSkillToolAddition({
691
+ skill,
692
+ toolLoading: session.tool_loading,
693
+ toolList: turnTools,
694
+ messages: session.messages,
695
+ });
696
+ if (msg)
697
+ session.messages.push(msg);
698
+ };
699
+ // A turn on a different skill than the one the array was built for: the
700
+ // user message was just pushed, so the system message follows it.
701
+ if (params.skill !== session.tool_set_skill)
702
+ surfaceStarters(params.skill);
446
703
  /**
447
704
  * key: `${tool_name}:${object}` — consecutive failures per (tool, target).
448
705
  * For object-identifying tools (sap_set_source / sap_activate / etc.) the
@@ -470,6 +727,10 @@ ${imageExpansion.error}`));
470
727
  };
471
728
  // Turn number for the cost log — count of prior user messages + this one.
472
729
  const turnNumber = session.messages.filter((m) => m.role === 'user').length + 1;
730
+ // Final review I4 — the monotonic per-session turn counter the auto-compact
731
+ // throttle compares (turnNumber above shrinks after a compaction). An older
732
+ // session seeds it so it never starts below its stored marker.
733
+ session.turn_seq = (session.turn_seq ?? Math.max(turnNumber - 1, session.lastAutoCompactTurn ?? 0)) + 1;
473
734
  // Snapshot session.messages.length at turn start so the save hook can
474
735
  // walk every assistant message added during this turn — not just the
475
736
  // final one. Skills like /abap-cca emit their manifest in an early
@@ -495,8 +756,24 @@ ${imageExpansion.error}`));
495
756
  // long phase (e.g. /abap-plan c2.behavior) that overruns the output cap
496
757
  // mid-response finishes instead of dying with "unexpected stop reason".
497
758
  let maxTokenContinuations = 0;
759
+ // Task 16 (F6) — bounded re-sends on stop_reason 'pause_turn' (a server
760
+ // tool paused a long turn). Consecutive only: reset by any other round.
761
+ let pauseTurnContinuations = 0;
762
+ // Task 3 (2026-09-27) — 0-based index of each model call within this turn,
763
+ // recorded on the cost line as `call_index`.
764
+ let roundIndex = 0;
498
765
  // (emit was hoisted to the top of the function in v0.5 so the Phase J
499
766
  // promote dispatch could share it. Original location was here.)
767
+ // Task 19 (owner decision 7) — cold-cache prune, default OFF
768
+ // (CSPEACH_COLD_PRUNE=on). Once, before the first createStream: when the
769
+ // cache has expired (> 300 s since the last call / ping) the next call
770
+ // re-writes the history anyway, so old tool results are elided first.
771
+ {
772
+ const pruned = maybeColdPrune(session);
773
+ if (pruned > 0) {
774
+ emit(chalk.dim(`(cache was cold — ${pruned} old tool result(s) elided to keep the rewrite small)`));
775
+ }
776
+ }
500
777
  while (true) {
501
778
  // Mid-turn steering (2026-06-08): inject any user corrections typed since the
502
779
  // last round so the next createStream sees them. Drains at the top of EVERY
@@ -505,7 +782,7 @@ ${imageExpansion.error}`));
505
782
  // Anthropic API concatenates consecutive user messages, so injecting after a
506
783
  // tool_result user message is valid.
507
784
  {
508
- const steers = drainSteering();
785
+ const steers = readerChild ? [] : drainSteering();
509
786
  if (steers.length > 0) {
510
787
  const text = steers.join('\n');
511
788
  session.messages.push({ role: 'user', content: text });
@@ -513,6 +790,7 @@ ${imageExpansion.error}`));
513
790
  }
514
791
  }
515
792
  let buffer = initialBufferState();
793
+ let writingCode = false; // piece 2 §11.2 — "writing code…" while a fence is held back
516
794
  let stream;
517
795
  // 2026-05-06: paint a "thinking" spinner during the LLM round-trip
518
796
  // wait so the user sees activity instead of staring at a silent
@@ -530,8 +808,8 @@ ${imageExpansion.error}`));
530
808
  // routed) covers liveness feedback. Classic mode (no chunkEmitter)
531
809
  // keeps the stdout spinner unchanged.
532
810
  // Guard: test/unit/ink-cursor-invariant.test.ts.
533
- turnStatusEmitter.activity('thinking');
534
- const thinkingSpinner = startThinkingSpinner({ chunkEmitter: params.chunkEmitter });
811
+ status.activity('thinking');
812
+ const thinkingSpinner = startThinkingSpinner({ chunkEmitter: livenessEmitter });
535
813
  // v0.6 — guaranteed-visible heartbeat alongside the in-place spinner.
536
814
  // The spinner self-disables on non-TTY (Windows PowerShell sometimes
537
815
  // reports isTTY=false; resume mode also lost spinner visibility) so a
@@ -541,7 +819,10 @@ ${imageExpansion.error}`));
541
819
  // heartbeat lines land in the Body region (via React) instead of
542
820
  // being raw-written to stdout (which Ink overdraws on its next
543
821
  // render, briefly flashing the line then making it disappear).
544
- const thinkingHeartbeat = startThinkingHeartbeat({ chunkEmitter: params.chunkEmitter });
822
+ const thinkingHeartbeat = startThinkingHeartbeat({ chunkEmitter: livenessEmitter });
823
+ // Task 15 — the request actually sent by the attempt that succeeded.
824
+ let sentParams;
825
+ let sentHeaders;
545
826
  try {
546
827
  stream = await retryWithBackoff(() => (async () => {
547
828
  // W2.5 — apply optional tool filter (used by agent_run with
@@ -549,9 +830,11 @@ ${imageExpansion.error}`));
549
830
  // the subagent). Mirrors CC's Tq2() pattern.
550
831
  // Standalone (2026-09-16): hide every SAP-dependent tool from the
551
832
  // model — calling one could only fail. Identity when connected.
552
- const allTools = toolsForContext(listTools(), params.ctx);
553
- const filteredTools = params.toolFilter ? allTools.filter(params.toolFilter) : allTools;
554
- const tools = toAnthropicTools(filteredTools);
833
+ // Task 16 — built once per mode above (byte-stable per session).
834
+ // Final review C1/I2 + N1 (b) — the turn's mode, settled at turn
835
+ // start (any one-time deferred → static strip ran there).
836
+ const model = sentModel();
837
+ const tools = turnTools;
555
838
  // Bug 11a (proactive heal, 2026-06-07) — strip any residual
556
839
  // `partial_json` from tool_use blocks on EVERY outgoing request.
557
840
  // `content_block_stop` normally deletes it (see below), but any miss —
@@ -565,42 +848,110 @@ ${imageExpansion.error}`));
565
848
  // the single chokepoint every request passes through, retries included —
566
849
  // makes the 400 categorically impossible. Idempotent and O(messages).
567
850
  healSessionMessagesInPlace(session.messages);
851
+ // Task 7 (spec D8) — preserved thinking binds each thinking block to
852
+ // the exact prefix that produced it, so history must be append-only.
853
+ // The heals below are no-ops on a history the API accepted (they only
854
+ // touch blocks the API would have rejected, or drop the LAST assistant
855
+ // message when it is thinking-only — no later block is bound to a
856
+ // prefix containing it), so they are not prefix edits. Order (merge
857
+ // with the UI branch's round-4 heal, cost finding core-I5): tool-mode
858
+ // settle (turn start) → partial_json → unsigned thinking →
859
+ // trailing thinking-only → orphan tool results → system placement
860
+ // LAST, because the earlier heals may remove the last assistant
861
+ // message and so change the tail placeSystemMessagesInPlace works on.
862
+ //
863
+ // 2026-09-27 — a thinking block without its signature (a turn
864
+ // cancelled mid-thinking, saved by an older CLI) 400s every later
865
+ // request with "Invalid `signature` in `thinking` block". Drop it;
866
+ // the model simply thinks again.
867
+ stripUnsignedThinking(session.messages);
868
+ dropTrailingThinkingOnly(session.messages);
568
869
  // 2026-09-24 — same chokepoint: a tool call cut off before its
569
870
  // result was recorded 400s every later request (a customer's
570
871
  // `continue` could never succeed). Give each orphan an error result
571
872
  // so the model sees the call did not complete and redoes it.
572
873
  const healedOrphans = healOrphanToolUses(session.messages);
874
+ // Task 16 (probe D2) — a system tool_addition must precede an
875
+ // assistant message or end the array ([system, user] is a 400).
876
+ // Same chokepoint: covers steering, the watchdog nudge and a turn
877
+ // that ended before the model answered the tool_addition.
878
+ placeSystemMessagesInPlace(session.messages);
573
879
  if (healedOrphans > 0) {
574
880
  emit(chalk.dim(`(repaired ${healedOrphans} tool call(s) cut off by an interruption — the model will redo them)`));
575
881
  }
882
+ // Task 4 — effort only from an explicit act (ruling F2); never to a
883
+ // model that rejects it (probe P1: claude-haiku-4-5* ⇒ 400).
884
+ const effort = params.effortOverride ?? resolveEffort(cfg);
885
+ const sendEffort = effort !== undefined && modelAcceptsEffort(model);
576
886
  const streamParams = {
577
887
  // A4 — per-turn override (plan model tiering) wins over the
578
- // configured default; absent on every non-plan turn. model-governance
579
- // step 2d — the session-default model now resolves env > local >
580
- // server > built-in (byte-identical to cfg.default_model when nothing set).
581
- model: params.modelOverride ?? resolveModelRole('session_default', cfg),
888
+ // session model; absent on every non-plan turn. Task 3 (2026-09-27)
889
+ // — the session's PINNED model, resolved once at newSession
890
+ // (env > local > server > built-in) after the served config was
891
+ // awaited. A per-call resolve let call 0 run on the built-in and
892
+ // call 1 on the late-landing served model: a full cache rewrite.
893
+ model,
582
894
  // v0.3.1 — was 8192. Bumped because code-gen skills (abap-generate
583
895
  // etc.) kept running out of budget mid-turn: adaptive thinking +
584
896
  // multiple tool-result prompts + TL;DR mandate + summary prose
585
897
  // + the final question widget didn't fit in 8K. Symptom: stream
586
898
  // ended with stop_reason='length' right before the widget, and
587
899
  // the user saw the skill "stuck" after the intro prose.
588
- // 32768 is Opus 4.7's recommended ceiling for agentic turns and
589
- // gives comfortable headroom for the investigate-first pattern.
590
- max_tokens: params.maxTokensOverride ?? 32768,
900
+ // Task 8 (2026-09-27, ruling F5) — thinking counts toward
901
+ // max_tokens. 32768 truncated a turn once in the 2026-09 profile,
902
+ // and Opus 5.x thinks more per turn, so claude-opus-5* session
903
+ // turns get 65536 (the guide's starting point for long agentic
904
+ // turns; probe P2 accepted it on claude-opus-5 and
905
+ // claude-opus-5-5). Every other model keeps 32768: its output
906
+ // limit is not verified here. Changing max_tokens breaks neither
907
+ // the prompt cache nor the thinking block binding. Follows the
908
+ // model actually sent (plan-tier sonnet turns stay at 32768);
909
+ // maxTokensOverride (subagents) still wins below the ceiling.
910
+ // I1 (2026-09-28) — clamped to the sent model's output ceiling:
911
+ // an agent_run child asks for 100000, which an enforced
912
+ // sonnet/haiku model rejects with a 400.
913
+ max_tokens: clampMaxTokens(model, params.maxTokensOverride ?? sessionMaxTokens(model)),
591
914
  messages: session.messages,
592
915
  tools,
593
- thinking: { type: 'adaptive', display: 'summarized' },
916
+ // Task 7 — preserved thinking. Opus 5.5 binds each thinking block
917
+ // to the exact prefix that produced it; `drop_block` makes the API
918
+ // drop a block whose prefix changed (reported on message_start's
919
+ // input_transformations) instead of rejecting the request. Only to
920
+ // the families probe P4 proved (allow-list, dated ids stripped).
921
+ // Task 21 fix round (controller ruling) — on the managed path only
922
+ // when the proxy serves tool_loading (an old proxy adds no binding
923
+ // beta); BYOK adds its own beta; AI-hub/local strip the field.
924
+ thinking: modelAcceptsBinding(model) && (params.provider.mode !== 'managed' || managedProxyServesNewShapes())
925
+ ? { type: 'adaptive', display: 'summarized', block_binding: { prefix_mismatch_behavior: 'drop_block' } }
926
+ : { type: 'adaptive', display: 'summarized' },
927
+ ...(sendEffort ? { output_config: { effort } } : {}),
594
928
  };
595
929
  // Skill travels as a header per coordination doc §1.
596
930
  // Do NOT add `skill` to streamParams — it is not part of the Anthropic SDK body schema.
597
931
  // v0.6 — forward the per-turn abort signal so Esc / SIGINT-during-turn
598
932
  // tears down the in-flight stream cleanly.
933
+ const headers = { 'X-CSForge-Skill': params.skill, 'X-CSPeach-Model-Role': params.modelRole ?? 'session_default' };
934
+ sentParams = streamParams;
935
+ sentHeaders = headers;
599
936
  return params.provider.createStream(streamParams, {
600
- headers: { 'X-CSForge-Skill': params.skill },
937
+ // Task 9 — the role header tells the proxy which pipeline role
938
+ // this call runs under (body unchanged).
939
+ headers,
601
940
  signal: params.signal,
602
941
  });
603
942
  })());
943
+ // Task 15 (F14) — stamp every successful createStream; Task 19 reads it.
944
+ session.last_request_at = new Date().toISOString();
945
+ // Task 15 (F4) — snapshot NOW, before the loop pushes this round's
946
+ // assistant message: the snapshot ends on a user message. Also left on
947
+ // the session (WeakMap, never persisted) for the REPL's idle keepalive.
948
+ if (keepaliveOn && sentParams) {
949
+ lastRequest = recordLastRequest(session, {
950
+ params: snapshotParams(sentParams),
951
+ options: { headers: { ...(sentHeaders ?? {}) } },
952
+ cacheable: false, // set from message_start usage below
953
+ });
954
+ }
604
955
  }
605
956
  catch (err) {
606
957
  // Stop the thinking spinner before printing any error — leaving
@@ -621,7 +972,11 @@ ${imageExpansion.error}`));
621
972
  if (isPartialJsonApiError(err)) {
622
973
  const scrubbed = healSessionMessagesInPlace(session.messages);
623
974
  await saveSession(session);
624
- emit(chalk.yellow(`\n⚠ Session was poisoned by a partial-json tool_use block; healed in-memory${scrubbed > 0 ? ` (${scrubbed} block${scrubbed === 1 ? '' : 's'} cleaned)` : ''}. Please retry your last message.`));
975
+ emit('\n' + formatAttention({
976
+ text: 'Session repaired',
977
+ fact: `a broken tool call was cleaned up${scrubbed > 0 ? ` (${scrubbed} block${scrubbed === 1 ? '' : 's'})` : ''}`,
978
+ next: 'send your last message again',
979
+ }).trimEnd());
625
980
  return;
626
981
  }
627
982
  // Proxy returns 409 upgrade_required when CLI is older than skill's min_cli_version.
@@ -629,16 +984,36 @@ ${imageExpansion.error}`));
629
984
  const bodyRaw = err?.error ?? err?.response?.data ?? err?.body;
630
985
  const body = typeof bodyRaw === 'string' ? JSON.parse(bodyRaw) : bodyRaw;
631
986
  if (status === 409 && body?.error === 'upgrade_required') {
632
- emit(chalk.yellow.bold(`\n⚠ This skill requires cspeach >= ${body.min_cli_version} (you have ${body.current ?? 'unknown'}).`));
633
- emit(chalk.yellow(` Run: cspeach --upgrade`));
987
+ emit('\n' + formatFailure({
988
+ what: 'This skill needs a newer CSPeach',
989
+ detail: [`needs ${body.min_cli_version} or later · you have ${body.current ?? 'unknown'}`],
990
+ next: 'cspeach --upgrade',
991
+ }).trimEnd());
634
992
  return;
635
993
  }
636
994
  throw err;
637
995
  }
638
996
  let currentAssistantContent = [];
997
+ // Task 3 — this call's index within the turn, and the model the server
998
+ // reported in `message_start` (may differ from the requested id).
999
+ const callIndex = roundIndex++;
1000
+ let lastServedModel;
1001
+ // 2026-09-27 — blocks whose content_block_stop arrived. A thinking block
1002
+ // not in here was cut mid-stream and must never be persisted.
1003
+ const completedBlocks = new WeakSet();
639
1004
  let sawEndTurn = false;
640
1005
  let sawToolUse = false;
641
1006
  let sawMaxTokens = false;
1007
+ let sawPauseTurn = false;
1008
+ // Task 6 — a safety classifier declined the request (HTTP 200,
1009
+ // stop_reason 'refusal'). Branch on stop_reason only; stop_details is
1010
+ // captured for the cost log (its category), never shown to the user.
1011
+ let sawRefusal = false;
1012
+ let refusalCategory;
1013
+ // Task 7 — thinking blocks the API dropped (prefix binding mismatch), from
1014
+ // message_start.message.input_transformations. Cost line only; the user
1015
+ // sees nothing.
1016
+ let inputTransformations;
642
1017
  // M10 — ensure session.usage is present (shipped session schema may omit it
643
1018
  // for older saved sessions; we default to zero on first turn).
644
1019
  if (!session.usage) {
@@ -681,7 +1056,8 @@ ${imageExpansion.error}`));
681
1056
  // content. Long emissions (a 4k-token plan manifest) can look
682
1057
  // silent if the renderer buffers the block — the row keeps
683
1058
  // ticking "writing response · 90s" regardless.
684
- turnStatusEmitter.activity('writing response');
1059
+ status.activity('writing response');
1060
+ writingCode = false;
685
1061
  const block = event.content_block;
686
1062
  currentAssistantContent.push(block);
687
1063
  if (block.type === 'text') {
@@ -689,7 +1065,7 @@ ${imageExpansion.error}`));
689
1065
  if (params.chunkEmitter)
690
1066
  params.chunkEmitter.emit('chunk', sep);
691
1067
  else
692
- process.stdout.write(sep);
1068
+ writeStdout(sep);
693
1069
  }
694
1070
  }
695
1071
  else if (type === 'content_block_delta') {
@@ -698,11 +1074,15 @@ ${imageExpansion.error}`));
698
1074
  if (delta.type === 'text_delta') {
699
1075
  const { output, state } = renderChunk(delta.text, buffer);
700
1076
  buffer = state;
1077
+ const wa = nextWritingActivity(writingCode, buffer.pending);
1078
+ writingCode = wa.writingCode;
1079
+ if (wa.label)
1080
+ status.activity(wa.label);
701
1081
  if (output) {
702
1082
  if (params.chunkEmitter)
703
1083
  params.chunkEmitter.emit('chunk', output);
704
1084
  else
705
- process.stdout.write(output);
1085
+ writeStdout(output);
706
1086
  }
707
1087
  if (last?.type === 'text')
708
1088
  last.text = (last.text ?? '') + delta.text;
@@ -730,7 +1110,12 @@ ${imageExpansion.error}`));
730
1110
  }
731
1111
  else if (type === 'content_block_stop') {
732
1112
  const last = currentAssistantContent[currentAssistantContent.length - 1];
733
- if (last?.type === 'tool_use' && last.partial_json) {
1113
+ if (last && typeof last === 'object')
1114
+ completedBlocks.add(last);
1115
+ // Task 16 (F6) — the tool search tool's server_tool_use streams its
1116
+ // input as input_json_delta too (probe P6); parse and strip it the
1117
+ // same way, or the next request 400s on partial_json.
1118
+ if ((last?.type === 'tool_use' || last?.type === 'server_tool_use') && last.partial_json) {
734
1119
  try {
735
1120
  last.input = JSON.parse(last.partial_json);
736
1121
  }
@@ -741,9 +1126,26 @@ ${imageExpansion.error}`));
741
1126
  }
742
1127
  }
743
1128
  else if (type === 'message_start') {
1129
+ // Task 3 — the model the server actually ran (adopted after the
1130
+ // stream in managed mode only; see below).
1131
+ const servedModel = event.message?.model;
1132
+ if (typeof servedModel === 'string' && servedModel)
1133
+ lastServedModel = servedModel;
1134
+ // Task 7 — present only when the binding beta was sent; [] when
1135
+ // nothing was dropped (summarised to undefined ⇒ no cost-line key).
1136
+ inputTransformations = summariseInputTransformations(event.message?.input_transformations);
744
1137
  // message_start carries the prompt usage (input tokens + any cache reads).
745
1138
  const msgUsage = event.message?.usage;
1139
+ // Task 15 — ping only a prompt the API actually cached.
1140
+ if (lastRequest && isCacheablePrompt(msgUsage))
1141
+ lastRequest.cacheable = true;
746
1142
  if (msgUsage) {
1143
+ // Task 19 (F12) — this call's full prompt size, read by the
1144
+ // end-of-turn auto-compact gate (the last call of the turn wins).
1145
+ session.last_context_tokens =
1146
+ (msgUsage.input_tokens ?? 0)
1147
+ + (msgUsage.cache_read_input_tokens ?? 0)
1148
+ + (msgUsage.cache_creation_input_tokens ?? 0);
747
1149
  session.usage.input_tokens += msgUsage.input_tokens ?? 0;
748
1150
  session.usage.cache_read_input_tokens += msgUsage.cache_read_input_tokens ?? 0;
749
1151
  // Phase 1 cost tracking — cache writes are billed at 1.25× input.
@@ -769,6 +1171,13 @@ ${imageExpansion.error}`));
769
1171
  sawToolUse = true;
770
1172
  if (stop_reason === 'max_tokens' || stop_reason === 'length')
771
1173
  sawMaxTokens = true;
1174
+ if (stop_reason === 'pause_turn')
1175
+ sawPauseTurn = true;
1176
+ if (stop_reason === 'refusal') {
1177
+ sawRefusal = true;
1178
+ const category = event.delta?.stop_details?.category;
1179
+ refusalCategory = typeof category === 'string' && category ? category : 'unknown';
1180
+ }
772
1181
  // H2 — `deltaUsage.output_tokens` is the cumulative running total for
773
1182
  // THIS message, NOT a per-event delta. Compute the increment vs the
774
1183
  // last-seen cumulative and add only that. Guard against `null` /
@@ -804,8 +1213,9 @@ ${imageExpansion.error}`));
804
1213
  if (params.chunkEmitter)
805
1214
  params.chunkEmitter.emit('chunk', flushed);
806
1215
  else
807
- process.stdout.write(flushed);
1216
+ writeStdout(flushed);
808
1217
  buffer = initialBufferState();
1218
+ writingCode = false;
809
1219
  }
810
1220
  // Repair partial blocks before persistence. ALWAYS run, not just on
811
1221
  // interruption: on a clean tool-use round the model's tool_use blocks can
@@ -815,7 +1225,57 @@ ${imageExpansion.error}`));
815
1225
  // only DROPS a block when its `input` is undefined — which never happens on
816
1226
  // a clean round (content_block_stop always sets `input`), so this is a
817
1227
  // no-op-except-strip on the happy path. See ./repair-partial.ts.
818
- currentAssistantContent = repairPartialBlocks(currentAssistantContent);
1228
+ // Task 6 — after a server-side fallback, the first model's thinking /
1229
+ // tool_use blocks before the last `fallback` block must not be echoed back.
1230
+ // The `fallback` block itself stays (the adoption guard below reads it).
1231
+ currentAssistantContent = stripPreFallbackBlocks(repairPartialBlocks(currentAssistantContent, { completed: completedBlocks }));
1232
+ // Review M2 (UI branch) — a cut right after the thinking block leaves a
1233
+ // thinking-only turn: nothing the user can read, and nothing the next
1234
+ // request needs.
1235
+ // Integration M1 — a `fallback` block is neutral to that test, so
1236
+ // [fallback, thinking] is dropped too; remember the fallback first so
1237
+ // the adoption guard below still refuses to pin the fallback model.
1238
+ const carriedFallback = currentAssistantContent.some((b) => b?.type === 'fallback');
1239
+ if (isThinkingOnly(currentAssistantContent))
1240
+ currentAssistantContent = [];
1241
+ // Task 3 (F10) — adopt the served model. Managed mode only: BYOK / AI-hub
1242
+ // / local streams report a local alias, which would leave the session
1243
+ // unpriced. Never on a per-turn override, and never on a message carrying
1244
+ // a `fallback` block (a refusal fallback must not pin the fallback model
1245
+ // for the rest of the session). Otherwise the served id is only logged.
1246
+ if (params.provider.mode === 'managed'
1247
+ && !params.modelOverride
1248
+ && typeof lastServedModel === 'string'
1249
+ && lastServedModel
1250
+ && !carriedFallback
1251
+ // Task 6 — a refused message is discarded, so it proves nothing about
1252
+ // the model (and may have been served by a fallback).
1253
+ && !sawRefusal
1254
+ // Probe ruling C1 — served ids can carry a -YYYYMMDD suffix (haiku-4-5
1255
+ // is served as claude-haiku-4-5-20251001): a dated id equal to the
1256
+ // pinned id is not a model switch.
1257
+ && stripModelDate(session.model) !== stripModelDate(lastServedModel)) {
1258
+ session.model = lastServedModel;
1259
+ session.served_model_source = 'served';
1260
+ // Task 12 — explain the switch, once per session, only when the proxy
1261
+ // enforces the account admin's choice (/v1/me/model-config enforced)
1262
+ // AND the served model IS that choice (dated suffix ignored) — any
1263
+ // other switch (e.g. a version-gate downgrade) is not the admin's doing.
1264
+ const customer = getCustomerModelConfig();
1265
+ if (customer?.enforced === true
1266
+ && typeof customer.model === 'string'
1267
+ && stripModelDate(customer.model) === stripModelDate(lastServedModel)
1268
+ && !adminModelLineShown.has(session)) {
1269
+ adminModelLineShown.add(session);
1270
+ emit(chalk.dim(`Using ${lastServedModel} (set by your account admin)`));
1271
+ }
1272
+ }
1273
+ // Task 6 — a refusal (before any output or mid-stream) is not an answer:
1274
+ // discard whatever partial content streamed so nothing is persisted as a
1275
+ // complete assistant message. The cost line below is still written.
1276
+ if (sawRefusal) {
1277
+ currentAssistantContent = [];
1278
+ }
819
1279
  // Persist the assistant message — even if it's partial. Empty content
820
1280
  // arrays are skipped so we don't push a meaningless empty assistant
821
1281
  // message that would confuse the next round.
@@ -842,7 +1302,9 @@ ${imageExpansion.error}`));
842
1302
  if (interruptedError !== null) {
843
1303
  session.turnInterrupted = true;
844
1304
  const msg = interruptedError instanceof Error ? interruptedError.message : String(interruptedError);
845
- session.turnInterruptedReason = msg.slice(0, 200);
1305
+ // Round 7 item 8d — a stop the user asked for (Ctrl+C, /exit mid-turn)
1306
+ // is saved as such, never as the raw "Request was aborted.".
1307
+ session.turnInterruptedReason = params.signal?.aborted ? USER_STOP_REASON : msg.slice(0, 200);
846
1308
  }
847
1309
  else {
848
1310
  // Clean completion — clear any stale interrupted flag from a prior turn.
@@ -861,7 +1323,11 @@ ${imageExpansion.error}`));
861
1323
  // the cost line so the JSONL records what the LEDGER charged alongside
862
1324
  // our COGS, and so the status footer the REPL prints next reads a fresh
863
1325
  // figure. No-op + no request outside credits mode; capped at 2.5 s.
864
- await refreshCreditsAfterTurn();
1326
+ // Task 18 — a reader child never samples the balance: its charge then
1327
+ // lands in the parent's next credits delta, and parallel readers never
1328
+ // race on the shared sample.
1329
+ if (!readerChild)
1330
+ await refreshCreditsAfterTurn();
865
1331
  const turnTokens = {
866
1332
  input: (session.usage?.input_tokens ?? 0) - turnStartTokens.input,
867
1333
  output: (session.usage?.output_tokens ?? 0) - turnStartTokens.output,
@@ -879,19 +1345,47 @@ ${imageExpansion.error}`));
879
1345
  duration_ms: turnDurationMs,
880
1346
  // Additive: present only in credits mode (undefined in USD mode, so
881
1347
  // the emitted line is byte-identical to a pre-credits one).
882
- credits: creditsForCostLog(),
1348
+ credits: readerChild ? undefined : creditsForCostLog(),
1349
+ // Task 3 — what the server reported, which call of the turn, and
1350
+ // the call kind. served_model is omitted when the stream had none.
1351
+ served_model: lastServedModel ?? undefined,
1352
+ call_index: callIndex,
1353
+ // Task 18 — a reader child's lines (its own reader-*-cost.jsonl) say so.
1354
+ kind: readerChild ? 'reader' : 'main',
1355
+ // Task 6 — present only on a refusal ('unknown' when stop_details
1356
+ // carried no category). Never the explanation text.
1357
+ refusal_category: sawRefusal ? refusalCategory : undefined,
1358
+ // Task 7 — present only when the API dropped thinking blocks.
1359
+ input_transformations: inputTransformations,
1360
+ // Task 18 (F24) — running reader totals of this turn (absent until a
1361
+ // reader has run).
1362
+ reader: params.ctx.readerSpend,
883
1363
  });
1364
+ readerCallsLogged = params.ctx.readerSpend?.calls ?? 0;
884
1365
  void appendCostLine(session.id, entry);
885
1366
  }
886
1367
  // Close the stream-to-disk mirror with a footer (lightweight diagnostics).
887
- // Best-effort — failure here is invisible to the turn flow.
888
- void turnStreamWriter.close(interruptedError !== null
1368
+ // Best-effort — failure here is invisible to the turn flow. Awaited
1369
+ // (round 6 item 4e) so an exit right after the turn finds the log.
1370
+ await turnStreamWriter.close(interruptedError !== null
889
1371
  ? `_interrupted: ${session.turnInterruptedReason ?? 'unknown'}_`
890
1372
  : undefined);
891
1373
  // Re-throw AFTER persistence so the outer REPL catch can render the
892
1374
  // graceful error UX with full session-id + log-path context.
893
1375
  if (interruptedError !== null)
894
1376
  throw interruptedError;
1377
+ // Task 6 — the model declined. One plain line (no vendor name, no
1378
+ // category, no policy text), then end the turn. Nothing was persisted.
1379
+ if (sawRefusal) {
1380
+ emit('\n' + formatAttention({
1381
+ text: 'The model declined this request',
1382
+ fact: 'a safety filter stopped it',
1383
+ next: 'rephrase it, or split it into smaller steps',
1384
+ }).trimEnd());
1385
+ session.awaitingSkillAnswer = false;
1386
+ await saveSession(session);
1387
+ return;
1388
+ }
895
1389
  if (sawEndTurn) {
896
1390
  emit('');
897
1391
  // v0.6 Layer 3 — stream-end watchdog warning. The between-rounds
@@ -936,7 +1430,7 @@ ${imageExpansion.error}`));
936
1430
  // their manifest in an early message before a closing ask_question
937
1431
  // widget; the final wrap-up message ends the turn but doesn't
938
1432
  // carry the manifest. See ./turn-assistant-text.ts for details.
939
- turnStatusEmitter.activity('finishing up');
1433
+ status.activity('finishing up');
940
1434
  const assistantText = collectTurnAssistantText(session.messages, turnStartMessageCount);
941
1435
  // 2026-09-17 — the save hook below asks the user things (save y/N,
942
1436
  // blocker pickers). Those are not tool dispatches, so nothing paused
@@ -944,9 +1438,11 @@ ${imageExpansion.error}`));
944
1438
  // while the user was the one being waited on (Laeeq's dry run). Pause
945
1439
  // for the whole hook — the row hides while paused — and resume after,
946
1440
  // so any real work that follows (auto-compact) shows again.
947
- turnStatusEmitter.pause();
1441
+ status.pause();
948
1442
  try {
949
1443
  if (!params.suppressSaveHook) {
1444
+ // Task 15 — the save prompt waits on the user; keep the cache warm.
1445
+ armKeepalive();
950
1446
  await maybeOfferSave({
951
1447
  skill: params.skill,
952
1448
  assistantText,
@@ -962,7 +1458,8 @@ ${imageExpansion.error}`));
962
1458
  }
963
1459
  }
964
1460
  finally {
965
- turnStatusEmitter.resume();
1461
+ keepalive.disarm();
1462
+ status.resume();
966
1463
  }
967
1464
  // Phase 2 #9 (2026-05-16) — auto-compact end-of-turn hook.
968
1465
  // Runs only on the success path (after maybeOfferSave) so an interrupted
@@ -970,30 +1467,68 @@ ${imageExpansion.error}`));
970
1467
  // helper internally swallows non-fatal errors and emits a yellow note so
971
1468
  // a Haiku blip never regresses a turn that just succeeded. Threshold and
972
1469
  // throttle live in config.compact — see config/loader.ts:CompactConfig.
973
- try {
974
- const { maybeAutoCompact } = await import('../commands/auto-compact.js');
975
- const { buildSummarisationPrompt, serialiseForSummariser } = await import('../commands/compact.js');
976
- const { summariseViaProvider } = await import('./summarise-via-provider.js');
977
- await maybeAutoCompact({
978
- session,
979
- config: cfg.compact,
980
- turnNumber,
981
- emit,
982
- summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise),
983
- // model-governance step 2d — compact model resolves env > local >
984
- // server > built-in (== COMPACTION_MODEL when nothing set).
985
- { model: resolveModelRole('compact', cfg), maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact' } }),
986
- });
987
- }
988
- catch (err) {
989
- // Belt-and-braces — maybeAutoCompact already swallows internally,
990
- // but if its own dynamic-import or wiring blows up we still must
991
- // not regress the success path.
992
- emit(chalk.gray(`[auto-compact] skipped: ${err instanceof Error ? err.message : String(err)}`));
1470
+ // Task 19 — top-level REPL turns only (never one-shot, subagents or
1471
+ // readers); the gate reads session.last_context_tokens against the
1472
+ // resolved threshold (file > served auto_compact_tokens > 150 000).
1473
+ if (params.autoCompact === true) {
1474
+ try {
1475
+ const { maybeAutoCompact } = await import('../commands/auto-compact.js');
1476
+ const { buildSummarisationPrompt, serialiseForSummariser } = await import('../commands/compact.js');
1477
+ const { summariseViaProvider } = await import('./summarise-via-provider.js');
1478
+ await maybeAutoCompact({
1479
+ session,
1480
+ config: { ...cfg.compact, auto_threshold_tokens: resolveAutoCompactThreshold(cfg) },
1481
+ turnNumber: session.turn_seq,
1482
+ emit,
1483
+ summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise),
1484
+ // model-governance step 2d — compact model resolves env > local >
1485
+ // server > built-in (== COMPACTION_MODEL when nothing set).
1486
+ { model: resolveModelRole('compact', cfg), maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact', 'X-CSPeach-Model-Role': 'compact' } }),
1487
+ });
1488
+ }
1489
+ catch (err) {
1490
+ // Belt-and-braces — maybeAutoCompact already swallows internally,
1491
+ // but if its own dynamic-import or wiring blows up we still must
1492
+ // not regress the success path.
1493
+ emit(chalk.gray(`[auto-compact] skipped: ${err instanceof Error ? err.message : String(err)}`));
1494
+ }
993
1495
  }
994
1496
  return;
995
1497
  }
996
- if (!sawToolUse) {
1498
+ // Task 6 fix round 1 — stop_reason 'tool_use' but no tool_use block left
1499
+ // (e.g. the only one preceded a `fallback` block and was stripped). There
1500
+ // is nothing to dispatch; looping would re-send the same history. End the
1501
+ // turn (the message is already persisted).
1502
+ if (sawToolUse && !currentAssistantContent.some((b) => b?.type === 'tool_use')) {
1503
+ emit(chalk.yellow('\n[the model stopped for a tool call but sent none — ending turn]'));
1504
+ session.awaitingSkillAnswer = false;
1505
+ await saveSession(session);
1506
+ return;
1507
+ }
1508
+ if (!sawPauseTurn)
1509
+ pauseTurnContinuations = 0;
1510
+ // Task 16 review — a paused message that already carries a client
1511
+ // tool_use is dispatched like a tool_use stop (its result is the next
1512
+ // user message), not re-sent blind.
1513
+ const pausedWithToolUse = sawPauseTurn && currentAssistantContent.some((b) => b?.type === 'tool_use');
1514
+ if (pausedWithToolUse)
1515
+ pauseTurnContinuations = 0;
1516
+ if (!sawToolUse && !pausedWithToolUse) {
1517
+ // Task 16 (F6) — 'pause_turn': a server tool (tool search) paused the
1518
+ // turn. The paused assistant message is already persisted; re-send
1519
+ // with it last and NO new user message, so the model resumes. At most
1520
+ // 3 in a row (never observed in probe P6, handled anyway).
1521
+ const MAX_PAUSE_CONTINUATIONS = 3;
1522
+ if (sawPauseTurn) {
1523
+ if (pauseTurnContinuations < MAX_PAUSE_CONTINUATIONS) {
1524
+ pauseTurnContinuations += 1;
1525
+ continue;
1526
+ }
1527
+ emit(chalk.yellow(`\n[the model paused ${MAX_PAUSE_CONTINUATIONS + 1} times in a row — stopping. Type "continue" to resume.]`));
1528
+ session.awaitingSkillAnswer = false;
1529
+ await saveSession(session);
1530
+ return;
1531
+ }
997
1532
  // The response was cut off at the output-token cap (stop_reason
998
1533
  // 'max_tokens'/'length') mid-generation — NOT a real end of turn. The
999
1534
  // truncated assistant message is already persisted (line ~733), so
@@ -1030,102 +1565,186 @@ ${imageExpansion.error}`));
1030
1565
  // continue to run normally. See ./parallel-write-guard.ts.
1031
1566
  const writeGuard = newGuardState();
1032
1567
  const toolResults = [];
1568
+ // Task 18 (spec D18, rulings F21, F27) — concurrent reader group. The
1569
+ // ⏺ lines print first, then one spinner / heartbeat / status label
1570
+ // ("N readers running…"). At most MAX_PARALLEL_READERS run at once; each
1571
+ // gets a shallow ctx clone with its own toolUseId (the shared slot is
1572
+ // never used) and an adt whose calls go through ONE lock (one ADT request
1573
+ // at a time). The keepalive is armed once for the group.
1574
+ const readerResults = new Map();
1575
+ const runReaderGroup = async (group) => {
1576
+ const label = `${group.length} reader${group.length === 1 ? '' : 's'} running…`;
1577
+ for (const b of group) {
1578
+ renderToolCallTop({ name: b.name, args: (b.input ?? {}), chunkEmitter: toolRowEmitter });
1579
+ }
1580
+ const groupSpinner = startToolSpinner({ chunkEmitter: livenessEmitter, label });
1581
+ status.activity(label);
1582
+ const groupHeartbeat = startThinkingHeartbeat({ chunkEmitter: livenessEmitter, label });
1583
+ const adtLock = createAdtLock();
1584
+ armKeepalive();
1585
+ try {
1586
+ for (let i = 0; i < group.length; i += MAX_PARALLEL_READERS) {
1587
+ // Fix I1 — Esc / Ctrl+C: no further chunk starts; every reader not
1588
+ // started gets an interrupted result so each tool_use stays paired.
1589
+ if (params.signal?.aborted) {
1590
+ for (const b of group.slice(i)) {
1591
+ readerResults.set(b.id, { result: { content: INTERRUPTED_LINE, is_error: true }, durationMs: 0 });
1592
+ }
1593
+ break;
1594
+ }
1595
+ await Promise.all(group.slice(i, i + MAX_PARALLEL_READERS).map(async (b) => {
1596
+ const started = Date.now();
1597
+ const clone = {
1598
+ ...params.ctx,
1599
+ toolUseId: b.id,
1600
+ adt: params.ctx.adt ? serializeAdt(params.ctx.adt, adtLock) : params.ctx.adt,
1601
+ };
1602
+ let r;
1603
+ try {
1604
+ r = await dispatchTool(b.name, b.input, clone);
1605
+ }
1606
+ catch (err) {
1607
+ r = { content: JSON.stringify({ error: String(err) }), is_error: true };
1608
+ }
1609
+ readerResults.set(b.id, { result: r, durationMs: Date.now() - started });
1610
+ }));
1611
+ }
1612
+ }
1613
+ finally {
1614
+ keepalive.disarm();
1615
+ groupHeartbeat.stop();
1616
+ groupSpinner.stop();
1617
+ }
1618
+ // Fix M3 — an interrupted turn writes no further model-call line, so
1619
+ // the readers' spend is recorded now on a zero-token rollup line.
1620
+ const spend = params.ctx.readerSpend;
1621
+ if (params.signal?.aborted && spend && spend.calls > readerCallsLogged) {
1622
+ void appendCostLine(session.id, buildCostEntry({
1623
+ turn: turnNumber,
1624
+ model: params.modelOverride ?? session.model,
1625
+ tokens: { input: 0, output: 0, cacheRead: 0, cacheCreate: 0 },
1626
+ kind: 'reader_rollup',
1627
+ reader: spend,
1628
+ }));
1629
+ readerCallsLogged = spend.calls;
1630
+ }
1631
+ };
1033
1632
  for (const block of currentAssistantContent) {
1034
1633
  if (block.type === 'tool_use') {
1035
- // Phase 1: print the dispatch line (CC-style: ⏺ name(args)).
1036
- // No-op for widget-suppressed tools (ask_question) — the form/modal
1037
- // is the visible representation; see tool-widget.ts.
1038
- renderToolCallTop({
1039
- name: block.name,
1040
- args: (block.input ?? {}),
1041
- chunkEmitter: params.chunkEmitter,
1042
- });
1043
- // Phase 2a: start the peach-themed spinner. No-op outside TTY / Ink
1044
- // / CI — the result line still prints, just without the in-place
1045
- // animation. Spinner is purely a "still working" cue, not
1046
- // load-bearing for output. Skipped entirely for widget-suppressed
1047
- // tools (ask_question): with the ⏺ line gone the spinner would
1048
- // orphan on its own row above the form.
1049
- const spinner = isWidgetSuppressedTool(block.name)
1050
- ? { stop: () => undefined }
1051
- : startToolSpinner({ chunkEmitter: params.chunkEmitter });
1052
- // 2026-06-06 (turn-liveness, B5 smoke feedback) — the in-place tool
1053
- // spinner self-disables under Ink (phantom cursor #25), which left
1054
- // slow tool calls (SAP over VPN: 10-60s) as DEAD AIR between the ⏺
1055
- // dispatch line and the ⎿ result line. Labelled heartbeat prints
1056
- // fresh "<tool> running… (10s)" lines at thresholds — visible on
1057
- // every terminal, Ink included.
1058
- // Interactive tools wait on the USER, not the system — ticking
1059
- // "ask_question running… (30s)" while they think is noise.
1060
- // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
1061
- // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
1062
- // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
1063
- // BEFORE the op runs. file_write / shell_exec are otherwise classified
1064
- // non-interactive, so without this the loop would start the per-second
1065
- // heartbeat + keep the turn-status row ticking UNDER the modal — the
1066
- // observed live bug (doubled card, lost Enter, history-replay leak,
1067
- // ~10-min wedge with the tool spinner ticking under the modal). Detect
1068
- // the gate the SAME way tool-dispatch does (tool.isMutating &&
1069
- // shouldGateRule8) and treat it as user-blocking: pause the status row,
1070
- // skip the heartbeat. The gate's own clearActiveSpinner +
1071
- // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
1072
- // is the outer one — together no live render source contends with the
1073
- // modal for the Ink frame or raw-mode stdin.
1074
- const willTripBatchGate = (() => {
1075
- const t = getTool(block.name);
1076
- return !!t?.isMutating && shouldGateRule8();
1077
- })();
1078
- const isInteractiveTool = block.name === 'ask_question' ||
1079
- block.name === 'request_approval' ||
1080
- willTripBatchGate;
1081
- // Interactive tools block on the USER. PAUSE the turn-status row (don't
1082
- // just relabel it): a live 250ms tick repaints the dynamic frame and
1083
- // overdraws the inquirer approval picker / churns the Ink ask_question
1084
- // modal — the hidden-question + stacked-border + lost-Enter bug
1085
- // (2026-06-07). resume() in the finally brings it back the moment the
1086
- // user answers. Non-interactive tools keep the ticking label + heartbeat.
1087
- if (isInteractiveTool)
1088
- turnStatusEmitter.pause();
1089
- else
1090
- turnStatusEmitter.activity(block.name);
1091
- const toolHeartbeat = isInteractiveTool
1092
- ? { stop: () => undefined }
1093
- : startThinkingHeartbeat({
1094
- chunkEmitter: params.chunkEmitter,
1095
- label: `${block.name} running…`,
1096
- });
1097
- // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
1098
- const dispatchStart = Date.now();
1099
- // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1100
- // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1101
- // id the loop records below. Dispatch is sequential, so a single slot
1102
- // on the shared ctx is safe.
1103
- params.ctx.toolUseId = block.id;
1104
- // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
1105
- const guardDecision = checkAndMark(block.name, writeGuard);
1634
+ // Task 18 (spec D18) — a run of consecutive read_agent blocks runs as
1635
+ // ONE concurrent group the first time the loop reaches it; each block
1636
+ // then takes its stored result below. Every other tool keeps today's
1637
+ // sequential dispatch.
1638
+ if (block.name === READ_AGENT && !readerResults.has(block.id)) {
1639
+ await runReaderGroup(readerRunFrom(currentAssistantContent, currentAssistantContent.indexOf(block)));
1640
+ }
1641
+ const ranInGroup = readerResults.get(block.id);
1106
1642
  let result;
1107
- try {
1108
- result = guardDecision.allow
1109
- ? await dispatchTool(block.name, block.input, params.ctx)
1110
- : { content: guardDecision.errorContent, is_error: true };
1643
+ let durationMs;
1644
+ if (ranInGroup) {
1645
+ result = ranInGroup.result;
1646
+ durationMs = ranInGroup.durationMs;
1111
1647
  }
1112
- finally {
1113
- // D19: clear the slot so a future non-loop invocation on this ctx
1114
- // can't inherit a stale block id.
1115
- params.ctx.toolUseId = undefined;
1116
- // Stop on the error path too — the interval is unref'd but would
1117
- // otherwise keep printing "<tool> running…" into the NEXT prompt
1118
- // after a dispatch throw.
1119
- toolHeartbeat.stop();
1120
- // Resume the turn-status row now that the user has answered (or the
1121
- // interactive dispatch threw). No-op for non-interactive tools.
1648
+ else {
1649
+ // Phase 1: print the dispatch line (CC-style: ⏺ name(args)).
1650
+ // No-op for widget-suppressed tools (ask_question) — the form/modal
1651
+ // is the visible representation; see tool-widget.ts.
1652
+ renderToolCallTop({
1653
+ name: block.name,
1654
+ args: (block.input ?? {}),
1655
+ chunkEmitter: toolRowEmitter,
1656
+ });
1657
+ // Phase 2a: start the peach-themed spinner. No-op outside TTY / Ink
1658
+ // / CI — the result line still prints, just without the in-place
1659
+ // animation. Spinner is purely a "still working" cue, not
1660
+ // load-bearing for output. Skipped entirely for widget-suppressed
1661
+ // tools (ask_question): with the ⏺ line gone the spinner would
1662
+ // orphan on its own row above the form.
1663
+ const spinner = isWidgetSuppressedTool(block.name)
1664
+ ? { stop: () => undefined }
1665
+ : startToolSpinner({ chunkEmitter: livenessEmitter, label: toolActivityText(block.name, (block.input ?? {})) });
1666
+ // 2026-06-06 (turn-liveness, B5 smoke feedback) — the in-place tool
1667
+ // spinner self-disables under Ink (phantom cursor #25), which left
1668
+ // slow tool calls (SAP over VPN: 10-60s) as DEAD AIR between the ⏺
1669
+ // dispatch line and the ⎿ result line. Labelled heartbeat prints
1670
+ // fresh "<tool> running… (10s)" lines at thresholds — visible on
1671
+ // every terminal, Ink included.
1672
+ // Interactive tools wait on the USER, not the system — ticking
1673
+ // "ask_question running… (30s)" while they think is noise.
1674
+ // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
1675
+ // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
1676
+ // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
1677
+ // BEFORE the op runs. file_write / shell_exec are otherwise classified
1678
+ // non-interactive, so without this the loop would start the per-second
1679
+ // heartbeat + keep the turn-status row ticking UNDER the modal — the
1680
+ // observed live bug (doubled card, lost Enter, history-replay leak,
1681
+ // ~10-min wedge with the tool spinner ticking under the modal). Detect
1682
+ // the gate the SAME way tool-dispatch does (tool.isMutating &&
1683
+ // shouldGateRule8) and treat it as user-blocking: pause the status row,
1684
+ // skip the heartbeat. The gate's own clearActiveSpinner +
1685
+ // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
1686
+ // is the outer one — together no live render source contends with the
1687
+ // modal for the Ink frame or raw-mode stdin.
1688
+ const willTripBatchGate = (() => {
1689
+ const t = getTool(block.name);
1690
+ return !!t?.isMutating && shouldGateRule8();
1691
+ })();
1692
+ const isInteractiveTool = block.name === 'ask_question' ||
1693
+ block.name === 'request_approval' ||
1694
+ willTripBatchGate;
1695
+ // Interactive tools block on the USER. PAUSE the turn-status row (don't
1696
+ // just relabel it): a live 250ms tick repaints the dynamic frame and
1697
+ // overdraws the inquirer approval picker / churns the Ink ask_question
1698
+ // modal — the hidden-question + stacked-border + lost-Enter bug
1699
+ // (2026-06-07). resume() in the finally brings it back the moment the
1700
+ // user answers. Non-interactive tools keep the ticking label + heartbeat.
1122
1701
  if (isInteractiveTool)
1123
- turnStatusEmitter.resume();
1702
+ status.pause();
1703
+ else
1704
+ status.activity(toolActivityText(block.name, (block.input ?? {})));
1705
+ const toolHeartbeat = isInteractiveTool
1706
+ ? { stop: () => undefined }
1707
+ : startThinkingHeartbeat({
1708
+ chunkEmitter: livenessEmitter,
1709
+ label: toolHeartbeatLabel(block.name, (block.input ?? {})),
1710
+ });
1711
+ // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
1712
+ const dispatchStart = Date.now();
1713
+ // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1714
+ // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1715
+ // id the loop records below. Dispatch is sequential, so a single slot
1716
+ // on the shared ctx is safe.
1717
+ params.ctx.toolUseId = block.id;
1718
+ // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
1719
+ const guardDecision = checkAndMark(block.name, writeGuard);
1720
+ // Task 15 — keep the prompt cache warm while the tool (or the user
1721
+ // answering its prompt) takes time; disarmed in the finally.
1722
+ if (guardDecision.allow)
1723
+ armKeepalive();
1724
+ try {
1725
+ result = guardDecision.allow
1726
+ ? await dispatchTool(block.name, block.input, params.ctx)
1727
+ : { content: guardDecision.errorContent, is_error: true };
1728
+ }
1729
+ finally {
1730
+ keepalive.disarm();
1731
+ // D19: clear the slot so a future non-loop invocation on this ctx
1732
+ // can't inherit a stale block id.
1733
+ params.ctx.toolUseId = undefined;
1734
+ // Stop on the error path too — the interval is unref'd but would
1735
+ // otherwise keep printing "<tool> running…" into the NEXT prompt
1736
+ // after a dispatch throw.
1737
+ toolHeartbeat.stop();
1738
+ // Resume the turn-status row now that the user has answered (or the
1739
+ // interactive dispatch threw). No-op for non-interactive tools.
1740
+ if (isInteractiveTool)
1741
+ status.resume();
1742
+ }
1743
+ durationMs = Date.now() - dispatchStart;
1744
+ // Phase 2c: stop the spinner, which erases its line so the result
1745
+ // row paints in place of it (no scrollback artifacts).
1746
+ spinner.stop();
1124
1747
  }
1125
- const durationMs = Date.now() - dispatchStart;
1126
- // Phase 2c: stop the spinner, which erases its line so the result
1127
- // row paints in place of it (no scrollback artifacts).
1128
- spinner.stop();
1129
1748
  // Phase 3: print result line ( ⎿ ✓ summary · timing).
1130
1749
  // 2026-05-15 (bug 1): the old `slice(0, 80)` clipped to the JSON
1131
1750
  // preamble (e.g. `{"error":"write_failed","detail":"HTTP 409: <?xml v…`)
@@ -1139,13 +1758,15 @@ ${imageExpansion.error}`));
1139
1758
  durationMs,
1140
1759
  isError: result.is_error ?? false,
1141
1760
  resultSummary,
1142
- chunkEmitter: params.chunkEmitter,
1761
+ chunkEmitter: toolRowEmitter,
1143
1762
  // D29 (2026-06-12): self-identifying result row. Heartbeat lines,
1144
1763
  // sap-client warns, and notice lines legitimately print between
1145
1764
  // the ⏺ top line and this row — without the name here those rows
1146
1765
  // read as anonymous `⎿ ✓ 364ms` orphans in the transcript.
1147
1766
  name: block.name,
1148
1767
  args: (block.input ?? {}),
1768
+ // Panels look: counts ("17 lines") and folded command output.
1769
+ resultContent: typeof result.content === 'string' ? result.content : undefined,
1149
1770
  });
1150
1771
  // v0.3 (Step 28.0): populate session.toolCalls so the SessionTimeline
1151
1772
  // overlay can render per-write history. Cap per-entry payload size at
@@ -1249,6 +1870,11 @@ ${imageExpansion.error}`));
1249
1870
  }
1250
1871
  }
1251
1872
  session.messages.push({ role: 'user', content: toolResults });
1873
+ // Task 16 — a successful dispatch_skill in this round moves the session
1874
+ // to another skill: surface that skill's starters now, right after the
1875
+ // tool-result user message (append-only; tools[] is unchanged).
1876
+ for (const skill of dispatchedSkills(currentAssistantContent, toolResults))
1877
+ surfaceStarters(skill);
1252
1878
  await saveSession(session);
1253
1879
  // v0.6 Layer 3 — between-rounds watchdog check. If the turn has been
1254
1880
  // running too long (wall-clock or token budget), inject a synthetic
@@ -1272,3 +1898,22 @@ ${imageExpansion.error}`));
1272
1898
  // Loop continues → next messages.create with the tool results.
1273
1899
  }
1274
1900
  }
1901
+ /**
1902
+ * Task 16 — skills a successful `dispatch_skill` call in this round queued
1903
+ * (`/abap-fiori-build --from @x` ⇒ `abap-fiori-build`, aliases resolved).
1904
+ */
1905
+ function dispatchedSkills(assistantContent, toolResults) {
1906
+ const out = [];
1907
+ for (const b of assistantContent) {
1908
+ if (b?.type !== 'tool_use' || b.name !== 'dispatch_skill')
1909
+ continue;
1910
+ const r = toolResults.find((t) => t?.tool_use_id === b.id);
1911
+ if (!r || r.is_error)
1912
+ continue;
1913
+ const command = typeof b.input?.command === 'string' ? b.input.command.trim() : '';
1914
+ const head = command.startsWith('/') ? command.slice(1).split(/\s+/)[0] : '';
1915
+ if (head)
1916
+ out.push(resolveSkillAlias(head));
1917
+ }
1918
+ return out;
1919
+ }