@cspeach/cli 0.8.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/README.md +1 -1
  2. package/dist/agent/intent-system-prompt.js +1 -1
  3. package/dist/agent/loop.js +330 -36
  4. package/dist/agent/providers/license-gate.js +44 -0
  5. package/dist/agent/skill-checkpoint.js +1 -1
  6. package/dist/agent/steering-queue.js +27 -0
  7. package/dist/agent/tool-dispatch.js +15 -0
  8. package/dist/approvals/canonical.js +91 -0
  9. package/dist/approvals/jwt.js +84 -7
  10. package/dist/approvals/render.js +38 -0
  11. package/dist/auth/org-anthropic-key.js +25 -0
  12. package/dist/classifier/client.js +24 -5
  13. package/dist/commands/config-set.js +95 -0
  14. package/dist/commands/login.js +31 -14
  15. package/dist/commands/plan-model-tier.js +83 -0
  16. package/dist/commands/plan-resume.js +435 -0
  17. package/dist/config/loader.js +102 -3
  18. package/dist/cost/pricing.js +14 -5
  19. package/dist/doctor/checks/_http-probe.js +1 -0
  20. package/dist/doctor/checks/cert.js +14 -3
  21. package/dist/doctor/checks/sap.js +30 -8
  22. package/dist/doctor/checks/zcspeach.js +19 -4
  23. package/dist/one-shot.js +52 -4
  24. package/dist/projects/answer-blockers.js +137 -0
  25. package/dist/projects/email-template.js +2 -0
  26. package/dist/projects/extract-cca.js +108 -16
  27. package/dist/projects/extract-modernize.js +1 -1
  28. package/dist/projects/extract-plan.js +178 -0
  29. package/dist/projects/extract-spec-gap.js +34 -7
  30. package/dist/projects/extract-test-coverage.js +1 -1
  31. package/dist/projects/extract-upgrade.js +113 -22
  32. package/dist/projects/index.js +7 -1
  33. package/dist/projects/merge-cca.js +292 -0
  34. package/dist/projects/merge-upgrade.js +173 -0
  35. package/dist/projects/migration.js +103 -1
  36. package/dist/projects/output-paths.js +27 -0
  37. package/dist/projects/plan-run.js +254 -0
  38. package/dist/projects/plan-schema.js +210 -0
  39. package/dist/projects/promote-command.js +26 -2
  40. package/dist/projects/promote.js +128 -0
  41. package/dist/projects/save-command.js +263 -20
  42. package/dist/projects/status.js +22 -0
  43. package/dist/projects/validate.js +2 -0
  44. package/dist/projects/workspace.js +164 -20
  45. package/dist/renderer/notices.js +64 -0
  46. package/dist/renderer/progress-chatter.js +8 -0
  47. package/dist/renderer/syntax.js +16 -1
  48. package/dist/renderer/thinking-heartbeat.js +13 -1
  49. package/dist/renderer/tool-widget.js +18 -4
  50. package/dist/renderer/tty.js +43 -4
  51. package/dist/renderer/verify-chain.js +77 -0
  52. package/dist/repl/at-picker.js +93 -21
  53. package/dist/repl/builtin-commands.js +37 -0
  54. package/dist/repl/early-line-buffer.js +68 -0
  55. package/dist/repl/inquirer-guard.js +70 -5
  56. package/dist/repl/numbered-menu.js +131 -0
  57. package/dist/repl/post-turn-status.js +2 -2
  58. package/dist/repl/rule8-detector.js +17 -2
  59. package/dist/repl/safety-confirm.js +111 -2
  60. package/dist/repl/safety-mode-state.js +19 -3
  61. package/dist/repl/slash-picker.js +25 -19
  62. package/dist/repl.js +470 -22
  63. package/dist/router/classifier.js +150 -6
  64. package/dist/sap/capability-matrix.js +20 -0
  65. package/dist/sap/capability-matrix.json +11236 -0
  66. package/dist/sap/capability.js +146 -0
  67. package/dist/sap/connection-manager.js +19 -1
  68. package/dist/sap/onboarding.js +42 -4
  69. package/dist/session/pending.js +27 -0
  70. package/dist/skill-catalog.js +54 -43
  71. package/dist/skills/bundled-skills.js +279 -1
  72. package/dist/skills/promotion-dispatch.js +23 -0
  73. package/dist/tools/_command-shared.js +36 -12
  74. package/dist/tools/_filesystem-shared.js +139 -4
  75. package/dist/tools/_flag.js +25 -0
  76. package/dist/tools/approval.js +64 -21
  77. package/dist/tools/ask-question.js +96 -4
  78. package/dist/tools/capability/tool.js +74 -0
  79. package/dist/tools/dispatch-skill.js +22 -1
  80. package/dist/tools/extend-model/anchored-insert.js +810 -0
  81. package/dist/tools/extend-model/tool.js +188 -0
  82. package/dist/tools/filesystem/extract-document.js +57 -0
  83. package/dist/tools/filesystem/file-edit.js +12 -2
  84. package/dist/tools/filesystem/file-read.js +2 -2
  85. package/dist/tools/filesystem/file-write.js +11 -2
  86. package/dist/tools/filesystem/glob.js +11 -0
  87. package/dist/tools/filesystem/grep.js +10 -0
  88. package/dist/tools/filesystem/read-document.js +107 -0
  89. package/dist/tools/fiori/apply.js +50 -0
  90. package/dist/tools/fiori/bin.js +3 -0
  91. package/dist/tools/fiori/catalog/index.js +27 -0
  92. package/dist/tools/fiori/catalog/value-help.js +230 -0
  93. package/dist/tools/fiori/catalog/viz-chart.js +177 -0
  94. package/dist/tools/fiori/cli.js +71 -0
  95. package/dist/tools/fiori/deploy-config.js +73 -0
  96. package/dist/tools/fiori/fe-scaffold.js +45 -0
  97. package/dist/tools/fiori/i18n.js +39 -0
  98. package/dist/tools/fiori/manifest.js +70 -0
  99. package/dist/tools/fiori/render.js +77 -0
  100. package/dist/tools/fiori/scaffold.js +39 -0
  101. package/dist/tools/fiori/tools.js +356 -0
  102. package/dist/tools/fiori/types.js +1 -0
  103. package/dist/tools/local-build.js +76 -0
  104. package/dist/tools/local-files.js +31 -0
  105. package/dist/tools/project/_merge-shared.js +68 -0
  106. package/dist/tools/project/cca_merge.js +164 -0
  107. package/dist/tools/project/playbook_get.js +1 -1
  108. package/dist/tools/project/upgrade_merge_progress.js +206 -0
  109. package/dist/tools/sap-read.js +53 -9
  110. package/dist/tools/sap-write.js +530 -21
  111. package/dist/tools/shell/shell_exec.js +41 -6
  112. package/dist/tools/snapshot.js +37 -14
  113. package/dist/tools/subagent/background_run.js +17 -1
  114. package/dist/tools/transport-resolution.js +86 -0
  115. package/dist/tools/transport.js +224 -5
  116. package/dist/tools/write-mode.js +4 -0
  117. package/dist/ui/app.js +84 -11
  118. package/dist/ui/body.js +13 -0
  119. package/dist/ui/command-palette.js +46 -10
  120. package/dist/ui/file-palette.js +44 -0
  121. package/dist/ui/footer.js +28 -11
  122. package/dist/ui/line-resolution.js +92 -0
  123. package/dist/ui/session-timeline.js +1 -0
  124. package/dist/ui/text-input.js +150 -0
  125. package/dist/ui/turn-status-emitter.js +52 -0
  126. package/dist/ui/turn-status.js +59 -0
  127. package/dist/ui/widgets/ask-question-modal.js +30 -2
  128. package/package.json +23 -4
  129. package/bench/README.md +0 -78
  130. package/bench/prompts/abap-document-cds.md +0 -44
  131. package/bench/prompts/abap-explain-bdef-handler.md +0 -57
  132. package/bench/prompts/abap-test-method.md +0 -42
  133. package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
  134. package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
  135. package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
  136. package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
  137. package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
  138. package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
  139. package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
  140. package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
  141. package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # CSPeach
2
2
 
3
- AI-assisted ABAP development CLI. 34 skills, direct SAP access, safety
3
+ AI-assisted ABAP development CLI. 35 skills, direct SAP access, safety
4
4
  gates the model cannot bypass.
5
5
 
6
6
  ## Install
@@ -19,7 +19,7 @@ You help senior SAP consultants and ABAP developers plan, build, modernize, migr
19
19
  - Live access to the user's SAP system via the sap_* tool family (~30 tools — get_source, set_source, atc_run, transport_create, sql_query, snapshot_take, etc.).
20
20
  - Filesystem + shell + web access to the project the user is working in (file_read, file_edit, file_write, glob, grep, shell_exec, web_fetch, web_search).
21
21
  - Subagent dispatch (agent_run) for focused sub-tasks; background_run + monitor_emit for non-blocking processes; schedule_draft_create for persisting schedule envelopes (no runner yet — see tool description).
22
- - A library of 34 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
22
+ - A library of 35 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
23
23
  - The project context for this session (CDS views, tables, packages, conventions) — appended below if available.
24
24
 
25
25
  When the user states an intent, your job is:
@@ -1,29 +1,35 @@
1
1
  import chalk from 'chalk';
2
2
  import { basename } from 'node:path';
3
3
  import { dispatchTool } from './tool-dispatch.js';
4
- import { listTools, toAnthropicTools } from '../tools/index.js';
4
+ import { listTools, getTool, toAnthropicTools } from '../tools/index.js';
5
5
  import { saveSession } from '../session/store.js';
6
+ import { recordCompletedToolCall } from '../session/pending.js';
6
7
  import { loadConfig } from '../config/loader.js';
7
8
  import { retryWithBackoff } from './retry.js';
8
9
  import { renderChunk, initialBufferState, render } from '../renderer/pipeline.js';
9
10
  import { renderToolCallTop, renderToolCallBottom, startToolSpinner, startThinkingSpinner } from '../renderer/tool-widget.js';
10
11
  import { formatToolErrorSummary } from '../sap-errors/parse-adt-exception.js';
11
12
  import { startThinkingHeartbeat } from '../renderer/thinking-heartbeat.js';
13
+ // 2026-06-07 — status-row narration. Safe to import in every mode: the
14
+ // emitter has zero listeners outside Ink (classic / one-shot / subagent),
15
+ // so activity() calls are no-ops there.
16
+ import { turnStatusEmitter } from '../ui/turn-status-emitter.js';
12
17
  // (progress-chatter import removed 2026-05-01 — superseded by CC-style
13
18
  // two-line dispatch renderer; re-add if a future in-place spinner returns)
14
19
  import { buildRetryCapPausePayload, buildSkippedSiblingResults } from './retry-cap.js';
15
20
  import { appendCostLine, buildEntry as buildCostEntry } from '../cost/cost-log.js';
16
21
  import { lastAssistantIsAwaitingAnswer } from '../session/awaiting-answer.js';
17
22
  import { enrichUserMessage } from '../router/intent-extractor.js';
18
- import { resetRule8State } from '../repl/rule8-detector.js';
19
- import { runSaveCommand, runPromoteCommand } from '../projects/index.js';
20
- import { parseFromFlag } from '../skills/promotion-dispatch.js';
23
+ import { resetRule8State, shouldGateRule8 } from '../repl/rule8-detector.js';
24
+ import { runSaveCommand, runPromoteCommand, detectArtifactSkill, offerAnswerBlockers } from '../projects/index.js';
25
+ import { parseFromFlag, coerceDocFromFlagToAttachment } from '../skills/promotion-dispatch.js';
21
26
  import { input } from '@inquirer/prompts';
22
27
  import { withInquirer } from '../repl/inquirer-guard.js';
23
28
  import { expandTextFileAttachments } from '../projects/workspace.js';
24
29
  import { getSapSystemInfo, renderSapSystemBlock } from '../sap/system-info.js';
25
30
  import { objectKeyFromInput } from './retry-key.js';
26
31
  import { repairPartialBlocks, healSessionMessagesInPlace, isPartialJsonApiError } from './repair-partial.js';
32
+ import { drainSteering } from './steering-queue.js';
27
33
  import { TurnStreamWriter } from './turn-stream.js';
28
34
  import { applyToolResultCheckpoint } from './skill-checkpoint.js';
29
35
  import { loadWatchdogConfig, evaluateWatchdog } from './turn-watchdog.js';
@@ -56,7 +62,7 @@ const SAVE_HOOK_SKILLS = new Set([
56
62
  // 2026-05-08: upgrade pipeline manifest envelopes. Each emits a
57
63
  // <!-- csforge:upgrade-manifest --> block that the matching
58
64
  // extract-upgrade.* parser turns into the manifest envelope. The
59
- // heavy detail (per-finding data) stays in `.abapforge/upgrades/`,
65
+ // heavy detail (per-finding data) stays in `.cspeach/upgrades/`,
60
66
  // referenced by detail_path inside the envelope.
61
67
  'abap-upgrade-scan',
62
68
  'abap-upgrade-fix',
@@ -64,7 +70,7 @@ const SAVE_HOOK_SKILLS = new Set([
64
70
  'abap-upgrade-merge',
65
71
  // 2026-05-09: CCA pipeline manifest envelopes. /abap-cca emits a
66
72
  // <!-- csforge:cca-manifest --> block wrapping the heavy project.json
67
- // detail at .abapforge/cca/projects/<slug>/. /abap-cca-merge produces
73
+ // detail at .cspeach/cca/projects/<slug>/. /abap-cca-merge produces
68
74
  // a consolidated cca-assessment from N parallel consultant slices.
69
75
  'abap-cca',
70
76
  'abap-cca-merge',
@@ -73,6 +79,10 @@ const SAVE_HOOK_SKILLS = new Set([
73
79
  // and can fan out into further chains after their own envelope saves.
74
80
  'abap-modernize',
75
81
  'abap-test',
82
+ // 2026-06-06 B3: /abap-plan create mode emits a csforge:plan-manifest
83
+ // JSON block → plan envelope. Resume turns set suppressSaveHook so the
84
+ // revision path in commands/plan-resume.ts persists instead.
85
+ 'abap-plan',
76
86
  ]);
77
87
  /**
78
88
  * True when any user message in the session already carries the rendered
@@ -110,9 +120,21 @@ export function hasProjectContextBlock(messages) {
110
120
  *
111
121
  * Single shape `(q) => Promise<string>` so it slots straight into the
112
122
  * existing `prompt:` callback in runSaveCommand + runPromoteCommand.
123
+ * Exported for A1 (2026-06-10): repl.tsx reuses it as the consent prompt
124
+ * for harness-owned plan auto-run (offerNextPhaseAutoRun).
113
125
  */
114
- async function inkAwarePrompt(q) {
115
- const { shouldUseInk } = await import('../renderer/tty.js');
126
+ export async function inkAwarePrompt(q) {
127
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
128
+ // B5 (2026-06-11) — headless runs must NEVER block on stdin (defect D1
129
+ // family). Any prompt that reaches the generic helper in headless mode
130
+ // takes the conservative default: empty answer (callers treat '' as
131
+ // decline / skip). Prompts that need a different headless default (e.g.
132
+ // the artifact-save hook, which defaults YES) install their own
133
+ // headless-aware prompt BEFORE reaching this helper — see maybeOfferSave.
134
+ if (isHeadless()) {
135
+ console.error(chalk.dim(`headless: skipped prompt '${q.trim()}' → '' (default)`));
136
+ return '';
137
+ }
116
138
  if (shouldUseInk()) {
117
139
  const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
118
140
  const result = await askQuestionEmitter.request({
@@ -125,25 +147,117 @@ async function inkAwarePrompt(q) {
125
147
  }
126
148
  return withInquirer(() => input({ message: q }));
127
149
  }
150
+ /**
151
+ * Choices picker for an answer-blocker that carries `options`. Renders the
152
+ * native Ink AskQuestionModal (the same CC-style picker the ask_question tool
153
+ * uses) with the candidate answers plus "Write my own" and "Skip" escapes.
154
+ *
155
+ * - headless → skip (never blocks on stdin)
156
+ * - classic (non-Ink) → 'write' so the caller falls back to the free-text box
157
+ * - Ink → choice/multi modal; Esc or empty selection → skip
158
+ *
159
+ * The two sentinels are filtered out of any real picked answer; for multi-select
160
+ * the modal returns comma-joined values, so Write/Skip win over any co-selected
161
+ * options.
162
+ */
163
+ const CHOOSE_WRITE = '__cspeach_write__';
164
+ const CHOOSE_SKIP = '__cspeach_skip__';
165
+ export async function inkAwareChoose(a) {
166
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
167
+ if (isHeadless())
168
+ return { kind: 'skip' };
169
+ if (!shouldUseInk())
170
+ return { kind: 'write' }; // classic REPL → free-text fallback
171
+ const choices = [
172
+ ...a.options.map((o) => ({ value: o, label: o })),
173
+ { value: CHOOSE_WRITE, label: '✏️ Write my own…' },
174
+ { value: CHOOSE_SKIP, label: '↷ Skip (leave for the viewer)' },
175
+ ];
176
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
177
+ const res = await askQuestionEmitter.request({
178
+ id: 'answer-blocker',
179
+ question: `Blocker ${a.index + 1}/${a.total}: ${a.question}`,
180
+ context: a.multiSelect ? 'Space to toggle one or more · Enter to submit' : undefined,
181
+ kind: a.multiSelect ? 'multi' : 'choice',
182
+ choices,
183
+ });
184
+ if (res.cancelled || res.answer == null)
185
+ return { kind: 'skip' };
186
+ const values = res.answer.split(',');
187
+ if (values.includes(CHOOSE_WRITE))
188
+ return { kind: 'write' };
189
+ if (values.includes(CHOOSE_SKIP))
190
+ return { kind: 'skip' };
191
+ const real = values.filter((v) => v !== CHOOSE_WRITE && v !== CHOOSE_SKIP);
192
+ if (real.length === 0)
193
+ return { kind: 'skip' };
194
+ return { kind: 'picked', answer: real.join(', ') };
195
+ }
128
196
  export async function maybeOfferSave(p) {
129
- if (!SAVE_HOOK_SKILLS.has(p.skill))
130
- return;
131
197
  if (p.assistantText.trim().length === 0)
132
198
  return;
199
+ // C1 (2026-06-11, defect D22) — the hook fires on artifact EVIDENCE, not
200
+ // only on the routed-skill label. The seed-session plan save was lost
201
+ // because the turn was misrouted (label 'abap-rap', not in the set) while
202
+ // the output carried a complete csforge:plan-manifest block. The label
203
+ // stays the primary trigger — it is authoritative for prose-extracted
204
+ // artefacts with no inline marker (spec-gap / design / estimate) and keeps
205
+ // merge-skill provenance exact (abap-cca-merge saves as abap-cca-merge).
206
+ // When the label would NOT fire, a recognizable manifest block in the
207
+ // output resolves the registry skill instead (detectArtifactSkill), so a
208
+ // misroute can no longer silently lose the artifact.
209
+ let saveSkill = p.skill;
210
+ if (!SAVE_HOOK_SKILLS.has(p.skill)) {
211
+ const detected = detectArtifactSkill(p.assistantText);
212
+ if (!detected)
213
+ return;
214
+ saveSkill = detected;
215
+ p.emit(chalk.dim(`[save] ${detected} artifact manifest detected in this turn's output (turn was labelled '${p.skill}') — offering save.`));
216
+ }
217
+ // B5 (2026-06-11) — defect D2: in headless runs the artifact IS the point
218
+ // of the run, so the "Save as project file? [y/N]" prompt defaults YES
219
+ // instead of blocking on stdin that will never answer. The auto-answer is
220
+ // logged so the transcript stays honest about who said yes.
221
+ //
222
+ // CONTRACT: this headless prompt blanket-answers 'y' to ANY question it
223
+ // is asked, so runSaveCommand must ask at most ONE question (the save
224
+ // confirm). See the matching note at the prompt site in
225
+ // src/projects/save-command.ts before adding further prompts there.
226
+ const { isHeadless } = await import('../renderer/tty.js');
227
+ const prompt = isHeadless()
228
+ ? async (q) => {
229
+ p.emit(chalk.dim(`headless: auto-answered '${q.trim()}' → 'y' (artifact save defaults YES)`));
230
+ return 'y';
231
+ }
232
+ : inkAwarePrompt;
133
233
  try {
134
- await runSaveCommand({
234
+ const savedPath = await runSaveCommand({
135
235
  skillOutput: p.assistantText,
136
- skillName: p.skill,
236
+ skillName: saveSkill,
137
237
  skillVersion: '1.0',
138
238
  skillInput: p.userMessage,
139
239
  tokensUsed: p.tokensUsed,
140
240
  model: p.model,
141
241
  author: getAuthorIdentity(),
142
242
  cwd: process.cwd(),
143
- prompt: inkAwarePrompt,
243
+ prompt,
144
244
  log: (...lines) => lines.forEach((l) => p.emit(l)),
145
245
  promotedFrom: p.promotedFrom ?? null,
146
246
  });
247
+ // After a spec-gap file is saved, let a developer answer the open BLOCKER
248
+ // questions inline (recorded as `answered` items in the SAME file) — the
249
+ // business questions stay open for a consultant to answer in the viewer.
250
+ // Uses inkAwarePrompt directly (NOT the save hook's blanket-'y' headless
251
+ // prompt): in headless it returns '' → declined, so nothing is written.
252
+ if (savedPath && saveSkill === 'abap-spec-gap') {
253
+ await offerAnswerBlockers({
254
+ path: savedPath,
255
+ identity: getAuthorIdentity(),
256
+ prompt: inkAwarePrompt,
257
+ choose: inkAwareChoose,
258
+ log: (...lines) => lines.forEach((l) => p.emit(l)),
259
+ });
260
+ }
147
261
  }
148
262
  catch (err) {
149
263
  p.emit(chalk.yellow(`\n[save] skipped: ${err?.message ?? err}`));
@@ -173,7 +287,17 @@ export async function runTurn(params) {
173
287
  let userMessageForLLM = params.userMessage;
174
288
  let userMessageForSave = params.userMessage;
175
289
  {
176
- const { fromPath, rest } = parseFromFlag(params.userMessage);
290
+ // Forgiveness: `--from @<document>` (a .docx/.pdf/.txt/.md, not a saved
291
+ // .cspeach.json) means "attach this file", not "chain a result". Rewrite it
292
+ // to a plain @<file> attachment and let Phase D ingest it — never error out
293
+ // and cancel the turn on this very natural slip.
294
+ const docAttach = coerceDocFromFlagToAttachment(params.userMessage);
295
+ if (docAttach) {
296
+ emit(chalk.dim(`[--from] ${docAttach.filename} is a document, not a saved result — attaching it as input instead.`));
297
+ userMessageForLLM = docAttach.rewritten;
298
+ userMessageForSave = docAttach.rewritten;
299
+ }
300
+ const { fromPath, rest } = parseFromFlag(docAttach ? '' : params.userMessage);
177
301
  if (fromPath) {
178
302
  const result = await runPromoteCommand({
179
303
  sourcePath: fromPath,
@@ -318,9 +442,31 @@ export async function runTurn(params) {
318
442
  const watchdogConfig = loadWatchdogConfig();
319
443
  const turnStartedAt = Date.now();
320
444
  let watchdogWarned = false;
445
+ // 2026-06-08 — manifest-on-interrupt: track the last plan-manifest block we
446
+ // already handed to onPlanManifest, so we fire once per DISTINCT manifest
447
+ // (a continue-in-session run emits one per phase) and not once per round.
448
+ let lastFiredManifest = null;
449
+ // Bounded auto-continue across max_tokens truncations within one turn, so a
450
+ // long phase (e.g. /abap-plan c2.behavior) that overruns the output cap
451
+ // mid-response finishes instead of dying with "unexpected stop reason".
452
+ let maxTokenContinuations = 0;
321
453
  // (emit was hoisted to the top of the function in v0.5 so the Phase J
322
454
  // promote dispatch could share it. Original location was here.)
323
455
  while (true) {
456
+ // Mid-turn steering (2026-06-08): inject any user corrections typed since the
457
+ // last round so the next createStream sees them. Drains at the top of EVERY
458
+ // round — round 1 normally drains an empty queue (no-op); a steer typed
459
+ // during round N lands at the start of round N+1 (between rounds). The
460
+ // Anthropic API concatenates consecutive user messages, so injecting after a
461
+ // tool_result user message is valid.
462
+ {
463
+ const steers = drainSteering();
464
+ if (steers.length > 0) {
465
+ const text = steers.join('\n');
466
+ session.messages.push({ role: 'user', content: text });
467
+ emit(chalk.cyan(`↳ steering: ${steers.join(' / ')}`));
468
+ }
469
+ }
324
470
  let buffer = initialBufferState();
325
471
  let stream;
326
472
  // 2026-05-06: paint a "thinking" spinner during the LLM round-trip
@@ -339,6 +485,7 @@ export async function runTurn(params) {
339
485
  // routed) covers liveness feedback. Classic mode (no chunkEmitter)
340
486
  // keeps the stdout spinner unchanged.
341
487
  // Guard: test/unit/ink-cursor-invariant.test.ts.
488
+ turnStatusEmitter.activity('thinking');
342
489
  const thinkingSpinner = startThinkingSpinner({ chunkEmitter: params.chunkEmitter });
343
490
  // v0.6 — guaranteed-visible heartbeat alongside the in-place spinner.
344
491
  // The spinner self-disables on non-TTY (Windows PowerShell sometimes
@@ -358,8 +505,23 @@ export async function runTurn(params) {
358
505
  const allTools = listTools();
359
506
  const filteredTools = params.toolFilter ? allTools.filter(params.toolFilter) : allTools;
360
507
  const tools = toAnthropicTools(filteredTools);
508
+ // Bug 11a (proactive heal, 2026-06-07) — strip any residual
509
+ // `partial_json` from tool_use blocks on EVERY outgoing request.
510
+ // `content_block_stop` normally deletes it (see below), but any miss —
511
+ // a multi-tool grounding round, an SDK edge case, poison rehydrated
512
+ // from disk — leaves a block carrying both `input` AND `partial_json`,
513
+ // and the NEXT createStream 400s with:
514
+ // messages.N.content.M.tool_use.partial_json: Extra inputs are not permitted
515
+ // which interrupts the turn and (in a /abap-plan resume) discards the
516
+ // expensive grounding work before any manifest is written. The reactive
517
+ // heal in the catch below only fires AFTER that damage. Sanitising here —
518
+ // the single chokepoint every request passes through, retries included —
519
+ // makes the 400 categorically impossible. Idempotent and O(messages).
520
+ healSessionMessagesInPlace(session.messages);
361
521
  const streamParams = {
362
- model: cfg.default_model,
522
+ // A4 — per-turn override (plan model tiering) wins over the
523
+ // configured default; absent on every non-plan turn.
524
+ model: params.modelOverride ?? cfg.default_model,
363
525
  // v0.3.1 — was 8192. Bumped because code-gen skills (abap-generate
364
526
  // etc.) kept running out of budget mid-turn: adaptive thinking +
365
527
  // multiple tool-result prompts + TL;DR mandate + summary prose
@@ -419,6 +581,7 @@ export async function runTurn(params) {
419
581
  let currentAssistantContent = [];
420
582
  let sawEndTurn = false;
421
583
  let sawToolUse = false;
584
+ let sawMaxTokens = false;
422
585
  // M10 — ensure session.usage is present (shipped session schema may omit it
423
586
  // for older saved sessions; we default to zero on first turn).
424
587
  if (!session.usage) {
@@ -457,6 +620,11 @@ export async function runTurn(params) {
457
620
  // call stop on an already-stopped handle, which is a no-op.
458
621
  thinkingSpinner.stop();
459
622
  thinkingHeartbeat.stop();
623
+ // 2026-06-07 — status-row narration: the model is now emitting
624
+ // content. Long emissions (a 4k-token plan manifest) can look
625
+ // silent if the renderer buffers the block — the row keeps
626
+ // ticking "writing response · 90s" regardless.
627
+ turnStatusEmitter.activity('writing response');
460
628
  const block = event.content_block;
461
629
  currentAssistantContent.push(block);
462
630
  if (block.type === 'text') {
@@ -537,6 +705,8 @@ export async function runTurn(params) {
537
705
  sawEndTurn = true;
538
706
  if (stop_reason === 'tool_use')
539
707
  sawToolUse = true;
708
+ if (stop_reason === 'max_tokens' || stop_reason === 'length')
709
+ sawMaxTokens = true;
540
710
  // H2 — `deltaUsage.output_tokens` is the cumulative running total for
541
711
  // THIS message, NOT a per-event delta. Compute the increment vs the
542
712
  // last-seen cumulative and add only that. Guard against `null` /
@@ -571,18 +741,36 @@ export async function runTurn(params) {
571
741
  process.stdout.write(flushed);
572
742
  buffer = initialBufferState();
573
743
  }
574
- // Repair partial blocks before persistence — drop incomplete tool_use
575
- // fragments so the next API call doesn't reject the assistant message.
576
- // See ./repair-partial.ts for the full rationale.
577
- if (interruptedError !== null) {
578
- currentAssistantContent = repairPartialBlocks(currentAssistantContent);
579
- }
744
+ // Repair partial blocks before persistence. ALWAYS run, not just on
745
+ // interruption: on a clean tool-use round the model's tool_use blocks can
746
+ // still carry a leftover `partial_json` field, and persisting that poisons
747
+ // the on-disk session (the next createStream 400s — see the proactive heal
748
+ // above). `repairPartialBlocks` strips `partial_json` unconditionally and
749
+ // only DROPS a block when its `input` is undefined — which never happens on
750
+ // a clean round (content_block_stop always sets `input`), so this is a
751
+ // no-op-except-strip on the happy path. See ./repair-partial.ts.
752
+ currentAssistantContent = repairPartialBlocks(currentAssistantContent);
580
753
  // Persist the assistant message — even if it's partial. Empty content
581
754
  // arrays are skipped so we don't push a meaningless empty assistant
582
755
  // message that would confuse the next round.
583
756
  if (currentAssistantContent.length > 0) {
584
757
  session.messages.push({ role: 'assistant', content: currentAssistantContent });
585
758
  }
759
+ // 2026-06-08 — manifest-on-interrupt. The assistant message just pushed may
760
+ // carry a finalised csforge:plan-manifest block (the /abap-plan resume emits
761
+ // it BEFORE the continuation ask_question, in this same message). Persist it
762
+ // NOW — before the round's tool loop dispatches that modal — so a Ctrl+C /
763
+ // exit during the modal can't lose the phase. Fire once per distinct block;
764
+ // never let a save error crash the turn.
765
+ if (params.onPlanManifest) {
766
+ const turnText = collectTurnAssistantText(session.messages, turnStartMessageCount);
767
+ const blocks = turnText.match(/<!--\s*csforge:plan-manifest\s*\n[\s\S]*?\n\s*-->/g);
768
+ const latest = blocks ? blocks[blocks.length - 1] : null;
769
+ if (latest && latest !== lastFiredManifest) {
770
+ lastFiredManifest = latest;
771
+ void Promise.resolve(params.onPlanManifest(turnText)).catch(() => { });
772
+ }
773
+ }
586
774
  session.last_turn_at = new Date().toISOString();
587
775
  if (interruptedError !== null) {
588
776
  session.turnInterrupted = true;
@@ -611,7 +799,10 @@ export async function runTurn(params) {
611
799
  const turnDurationMs = Date.now() - turnStartedAt;
612
800
  const entry = buildCostEntry({
613
801
  turn: turnNumber,
614
- model: session.model,
802
+ // A4 — when a per-turn override ran the stream on a different model
803
+ // (plan model tiering), the cost line must record the ACTUAL model
804
+ // or the Sonnet-priced turn would be billed at session-model rates.
805
+ model: params.modelOverride ?? session.model,
615
806
  tokens: turnTokens,
616
807
  duration_ms: turnDurationMs,
617
808
  });
@@ -670,16 +861,22 @@ export async function runTurn(params) {
670
861
  // their manifest in an early message before a closing ask_question
671
862
  // widget; the final wrap-up message ends the turn but doesn't
672
863
  // carry the manifest. See ./turn-assistant-text.ts for details.
864
+ turnStatusEmitter.activity('finishing up');
673
865
  const assistantText = collectTurnAssistantText(session.messages, turnStartMessageCount);
674
- await maybeOfferSave({
675
- skill: params.skill,
676
- assistantText,
677
- userMessage: userMessageForSave,
678
- tokensUsed: Math.max(0, (session.usage?.output_tokens ?? 0) - turnStartOutputTokens),
679
- model: cfg.default_model ?? session.model ?? 'unknown',
680
- emit,
681
- promotedFrom: promotedFromForSave,
682
- });
866
+ if (!params.suppressSaveHook) {
867
+ await maybeOfferSave({
868
+ skill: params.skill,
869
+ assistantText,
870
+ userMessage: userMessageForSave,
871
+ tokensUsed: Math.max(0, (session.usage?.output_tokens ?? 0) - turnStartOutputTokens),
872
+ // A4 — record the actual turn model if an override ran the stream
873
+ // (unreachable today: plan-resume turns suppress this hook, and
874
+ // they are the only modelOverride caller — kept honest anyway).
875
+ model: params.modelOverride ?? cfg.default_model ?? session.model ?? 'unknown',
876
+ emit,
877
+ promotedFrom: promotedFromForSave,
878
+ });
879
+ }
683
880
  // Phase 2 #9 (2026-05-16) — auto-compact end-of-turn hook.
684
881
  // Runs only on the success path (after maybeOfferSave) so an interrupted
685
882
  // turn never triggers a summariser call on partial state. The maybeAutoCompact
@@ -707,7 +904,29 @@ export async function runTurn(params) {
707
904
  return;
708
905
  }
709
906
  if (!sawToolUse) {
710
- // Unexpected stop reason — bail to avoid infinite loop.
907
+ // The response was cut off at the output-token cap (stop_reason
908
+ // 'max_tokens'/'length') mid-generation — NOT a real end of turn. The
909
+ // truncated assistant message is already persisted (line ~733), so
910
+ // continue the turn and let the model finish (and, for /abap-plan, still
911
+ // emit its manifest). Bounded so a pathologically long response can't
912
+ // loop forever.
913
+ const MAX_TOKEN_CONTINUATIONS = 4;
914
+ if (sawMaxTokens && maxTokenContinuations < MAX_TOKEN_CONTINUATIONS) {
915
+ maxTokenContinuations += 1;
916
+ emit(chalk.yellow(`\n[output limit reached — auto-continuing (${maxTokenContinuations}/${MAX_TOKEN_CONTINUATIONS})]`));
917
+ session.messages.push({
918
+ role: 'user',
919
+ content: 'Your previous response was cut off at the output-token limit. Continue exactly where you left off — do NOT repeat what you already wrote. '
920
+ + 'If you were mid-way through a tool call, or (for /abap-plan) had not yet emitted the closing csforge:plan-manifest block, complete it now. '
921
+ + 'Keep narration brief to stay within the limit.',
922
+ });
923
+ continue;
924
+ }
925
+ if (sawMaxTokens) {
926
+ emit(chalk.yellow(`\n[output limit hit ${maxTokenContinuations}× this turn — stopping. Type "continue" to resume.]`));
927
+ return;
928
+ }
929
+ // Truly unexpected stop reason — bail to avoid an infinite loop.
711
930
  emit(chalk.yellow('\n[unexpected stop reason — ending turn]'));
712
931
  return;
713
932
  }
@@ -734,13 +953,79 @@ export async function runTurn(params) {
734
953
  // animation. Spinner is purely a "still working" cue, not
735
954
  // load-bearing for output.
736
955
  const spinner = startToolSpinner({ chunkEmitter: params.chunkEmitter });
956
+ // 2026-06-06 (turn-liveness, B5 smoke feedback) — the in-place tool
957
+ // spinner self-disables under Ink (phantom cursor #25), which left
958
+ // slow tool calls (SAP over VPN: 10-60s) as DEAD AIR between the ⏺
959
+ // dispatch line and the ⎿ result line. Labelled heartbeat prints
960
+ // fresh "<tool> running… (10s)" lines at thresholds — visible on
961
+ // every terminal, Ink included.
962
+ // Interactive tools wait on the USER, not the system — ticking
963
+ // "ask_question running… (30s)" while they think is noise.
964
+ // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
965
+ // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
966
+ // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
967
+ // BEFORE the op runs. file_write / shell_exec are otherwise classified
968
+ // non-interactive, so without this the loop would start the per-second
969
+ // heartbeat + keep the turn-status row ticking UNDER the modal — the
970
+ // observed live bug (doubled card, lost Enter, history-replay leak,
971
+ // ~10-min wedge with the tool spinner ticking under the modal). Detect
972
+ // the gate the SAME way tool-dispatch does (tool.isMutating &&
973
+ // shouldGateRule8) and treat it as user-blocking: pause the status row,
974
+ // skip the heartbeat. The gate's own clearActiveSpinner +
975
+ // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
976
+ // is the outer one — together no live render source contends with the
977
+ // modal for the Ink frame or raw-mode stdin.
978
+ const willTripBatchGate = (() => {
979
+ const t = getTool(block.name);
980
+ return !!t?.isMutating && shouldGateRule8();
981
+ })();
982
+ const isInteractiveTool = block.name === 'ask_question' ||
983
+ block.name === 'request_approval' ||
984
+ willTripBatchGate;
985
+ // Interactive tools block on the USER. PAUSE the turn-status row (don't
986
+ // just relabel it): a live 250ms tick repaints the dynamic frame and
987
+ // overdraws the inquirer approval picker / churns the Ink ask_question
988
+ // modal — the hidden-question + stacked-border + lost-Enter bug
989
+ // (2026-06-07). resume() in the finally brings it back the moment the
990
+ // user answers. Non-interactive tools keep the ticking label + heartbeat.
991
+ if (isInteractiveTool)
992
+ turnStatusEmitter.pause();
993
+ else
994
+ turnStatusEmitter.activity(block.name);
995
+ const toolHeartbeat = isInteractiveTool
996
+ ? { stop: () => undefined }
997
+ : startThinkingHeartbeat({
998
+ chunkEmitter: params.chunkEmitter,
999
+ label: `${block.name} running…`,
1000
+ });
737
1001
  // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
738
1002
  const dispatchStart = Date.now();
1003
+ // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1004
+ // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1005
+ // id the loop records below. Dispatch is sequential, so a single slot
1006
+ // on the shared ctx is safe.
1007
+ params.ctx.toolUseId = block.id;
739
1008
  // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
740
1009
  const guardDecision = checkAndMark(block.name, writeGuard);
741
- const result = guardDecision.allow
742
- ? await dispatchTool(block.name, block.input, params.ctx)
743
- : { content: guardDecision.errorContent, is_error: true };
1010
+ let result;
1011
+ try {
1012
+ result = guardDecision.allow
1013
+ ? await dispatchTool(block.name, block.input, params.ctx)
1014
+ : { content: guardDecision.errorContent, is_error: true };
1015
+ }
1016
+ finally {
1017
+ // D19: clear the slot so a future non-loop invocation on this ctx
1018
+ // can't inherit a stale block id.
1019
+ params.ctx.toolUseId = undefined;
1020
+ // Stop on the error path too — the interval is unref'd but would
1021
+ // otherwise keep printing "<tool> running…" into the NEXT prompt
1022
+ // after a dispatch throw.
1023
+ toolHeartbeat.stop();
1024
+ // Resume the turn-status row now that the user has answered (or the
1025
+ // interactive dispatch threw). No-op for non-interactive tools.
1026
+ if (isInteractiveTool)
1027
+ turnStatusEmitter.resume();
1028
+ }
744
1029
  const durationMs = Date.now() - dispatchStart;
745
1030
  // Phase 2c: stop the spinner, which erases its line so the result
746
1031
  // row paints in place of it (no scrollback artifacts).
@@ -759,16 +1044,25 @@ export async function runTurn(params) {
759
1044
  isError: result.is_error ?? false,
760
1045
  resultSummary,
761
1046
  chunkEmitter: params.chunkEmitter,
1047
+ // D29 (2026-06-12): self-identifying result row. Heartbeat lines,
1048
+ // sap-client warns, and notice lines legitimately print between
1049
+ // the ⏺ top line and this row — without the name here those rows
1050
+ // read as anonymous `⎿ ✓ 364ms` orphans in the transcript.
1051
+ name: block.name,
1052
+ args: (block.input ?? {}),
762
1053
  });
763
1054
  // v0.3 (Step 28.0): populate session.toolCalls so the SessionTimeline
764
1055
  // overlay can render per-write history. Cap per-entry payload size at
765
1056
  // 4kB to bound session.json growth.
1057
+ // D19 (2026-06-11): upsert, not push — write handlers journal the same
1058
+ // call through the WAL under this block.id; pushing unconditionally
1059
+ // produced two ledger entries per approval-gated write.
766
1060
  const completedAt = new Date().toISOString();
767
1061
  {
768
1062
  const resultText = typeof result.content === 'string'
769
1063
  ? result.content
770
1064
  : JSON.stringify(result.content);
771
- session.toolCalls.push({
1065
+ recordCompletedToolCall(session, {
772
1066
  tool_use_id: block.id,
773
1067
  tool: block.name,
774
1068
  args: block.input ?? {},
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Thrown when a non-managed LLM mode is used without a valid CSPeach login.
3
+ * Carries a user-facing message; callers print it and exit rather than dumping
4
+ * a stack trace.
5
+ */
6
+ export class CspeachLicenseError extends Error {
7
+ mode;
8
+ constructor(mode) {
9
+ super(`CSPeach requires a license to run in '${mode}' mode.\n` +
10
+ `In this mode your prompts go straight to your own LLM — but the CSPeach skills\n` +
11
+ `and CLI are licensed software. Run \`cspeach login\` to activate your license.\n` +
12
+ `Need an access key? Email laeeq.siddique@cremencing.com.`);
13
+ this.name = 'CspeachLicenseError';
14
+ this.mode = mode;
15
+ }
16
+ }
17
+ /**
18
+ * Gate every NON-managed mode (byok / local / ai-hub) on a valid CSPeach login.
19
+ *
20
+ * Why: managed mode authenticates against the proxy on every call, so it is
21
+ * already gated. The other modes bypass our proxy for inference — and the
22
+ * skills (our IP) are bundled into the package — so without this check a fresh
23
+ * `npm install` is fully usable by anyone who never logged in, untracked. A
24
+ * CSPeach key is admin-issued (no open self-signup), so requiring one means we
25
+ * know who is using the product and can revoke access.
26
+ *
27
+ * Scope (Part A): require a CSPeach key to be PRESENT. Server-side validation +
28
+ * revocation (so a forged key fails) is the Part-B follow-up — see
29
+ * docs/byok-portal-handover.md §0.
30
+ */
31
+ export async function assertModeLicensed(mode, getBearer) {
32
+ if (mode === 'managed')
33
+ return;
34
+ let key = '';
35
+ try {
36
+ key = (await getBearer()) ?? '';
37
+ }
38
+ catch {
39
+ key = '';
40
+ }
41
+ if (key.trim().length === 0) {
42
+ throw new CspeachLicenseError(mode);
43
+ }
44
+ }
@@ -7,7 +7,7 @@
7
7
  * with full conversation context.
8
8
  *
9
9
  * What Layer 1 does NOT cover: skill-internal state files. /abap-cca, for
10
- * example, maintains `.abapforge/cca/projects/<id>/project.json` with an
10
+ * example, maintains `.cspeach/cca/projects/<id>/project.json` with an
11
11
  * inventory of objects, classifications, and per-package state. The skill
12
12
  * writes this file at PHASE BOUNDARIES (end of DISCOVER, end of INVENTORY,
13
13
  * etc.). A crash mid-DISCOVER means the partial inventory in project.json