@cspeach/cli 0.9.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/README.md +1 -1
  2. package/dist/agent/intent-system-prompt.js +1 -1
  3. package/dist/agent/loop.js +209 -20
  4. package/dist/agent/providers/license-gate.js +44 -0
  5. package/dist/agent/skill-checkpoint.js +1 -1
  6. package/dist/agent/tool-dispatch.js +15 -0
  7. package/dist/approvals/canonical.js +91 -0
  8. package/dist/approvals/jwt.js +39 -2
  9. package/dist/auth/org-anthropic-key.js +25 -0
  10. package/dist/classifier/client.js +18 -3
  11. package/dist/commands/config-set.js +95 -0
  12. package/dist/commands/login.js +31 -14
  13. package/dist/commands/plan-model-tier.js +83 -0
  14. package/dist/commands/plan-resume.js +148 -21
  15. package/dist/config/loader.js +95 -1
  16. package/dist/doctor/checks/_http-probe.js +1 -0
  17. package/dist/doctor/checks/cert.js +14 -3
  18. package/dist/doctor/checks/sap.js +30 -8
  19. package/dist/doctor/checks/zcspeach.js +19 -4
  20. package/dist/one-shot.js +52 -4
  21. package/dist/projects/answer-blockers.js +137 -0
  22. package/dist/projects/extract-cca.js +108 -16
  23. package/dist/projects/extract-modernize.js +1 -1
  24. package/dist/projects/extract-plan.js +130 -37
  25. package/dist/projects/extract-spec-gap.js +34 -7
  26. package/dist/projects/extract-test-coverage.js +1 -1
  27. package/dist/projects/extract-upgrade.js +113 -22
  28. package/dist/projects/index.js +5 -2
  29. package/dist/projects/merge-cca.js +292 -0
  30. package/dist/projects/merge-upgrade.js +173 -0
  31. package/dist/projects/migration.js +103 -1
  32. package/dist/projects/output-paths.js +27 -0
  33. package/dist/projects/plan-run.js +159 -25
  34. package/dist/projects/plan-schema.js +63 -3
  35. package/dist/projects/promote-command.js +25 -2
  36. package/dist/projects/promote.js +128 -0
  37. package/dist/projects/save-command.js +247 -20
  38. package/dist/projects/status.js +3 -1
  39. package/dist/projects/validate.js +1 -1
  40. package/dist/projects/workspace.js +164 -20
  41. package/dist/renderer/notices.js +64 -0
  42. package/dist/renderer/progress-chatter.js +8 -0
  43. package/dist/renderer/tool-widget.js +18 -4
  44. package/dist/renderer/tty.js +43 -4
  45. package/dist/renderer/verify-chain.js +77 -0
  46. package/dist/repl/at-picker.js +60 -7
  47. package/dist/repl/builtin-commands.js +37 -0
  48. package/dist/repl/early-line-buffer.js +68 -0
  49. package/dist/repl/inquirer-guard.js +70 -5
  50. package/dist/repl/numbered-menu.js +131 -0
  51. package/dist/repl/post-turn-status.js +2 -2
  52. package/dist/repl/rule8-detector.js +17 -2
  53. package/dist/repl/safety-confirm.js +111 -2
  54. package/dist/repl/safety-mode-state.js +19 -3
  55. package/dist/repl/slash-picker.js +10 -15
  56. package/dist/repl.js +301 -35
  57. package/dist/router/classifier.js +150 -6
  58. package/dist/sap/capability-matrix.js +20 -0
  59. package/dist/sap/capability-matrix.json +11236 -0
  60. package/dist/sap/capability.js +146 -0
  61. package/dist/sap/connection-manager.js +19 -1
  62. package/dist/sap/onboarding.js +42 -4
  63. package/dist/session/pending.js +27 -0
  64. package/dist/skill-catalog.js +48 -43
  65. package/dist/skills/bundled-skills.js +279 -1
  66. package/dist/skills/promotion-dispatch.js +23 -0
  67. package/dist/tools/_command-shared.js +36 -12
  68. package/dist/tools/_filesystem-shared.js +139 -4
  69. package/dist/tools/_flag.js +25 -0
  70. package/dist/tools/approval.js +64 -21
  71. package/dist/tools/ask-question.js +96 -4
  72. package/dist/tools/capability/tool.js +74 -0
  73. package/dist/tools/dispatch-skill.js +22 -1
  74. package/dist/tools/extend-model/anchored-insert.js +810 -0
  75. package/dist/tools/extend-model/tool.js +188 -0
  76. package/dist/tools/filesystem/extract-document.js +57 -0
  77. package/dist/tools/filesystem/file-edit.js +12 -2
  78. package/dist/tools/filesystem/file-read.js +2 -2
  79. package/dist/tools/filesystem/file-write.js +11 -2
  80. package/dist/tools/filesystem/glob.js +11 -0
  81. package/dist/tools/filesystem/grep.js +10 -0
  82. package/dist/tools/filesystem/read-document.js +107 -0
  83. package/dist/tools/fiori/apply.js +50 -0
  84. package/dist/tools/fiori/bin.js +3 -0
  85. package/dist/tools/fiori/catalog/index.js +27 -0
  86. package/dist/tools/fiori/catalog/value-help.js +230 -0
  87. package/dist/tools/fiori/catalog/viz-chart.js +177 -0
  88. package/dist/tools/fiori/cli.js +71 -0
  89. package/dist/tools/fiori/deploy-config.js +73 -0
  90. package/dist/tools/fiori/fe-scaffold.js +45 -0
  91. package/dist/tools/fiori/i18n.js +39 -0
  92. package/dist/tools/fiori/manifest.js +70 -0
  93. package/dist/tools/fiori/render.js +77 -0
  94. package/dist/tools/fiori/scaffold.js +39 -0
  95. package/dist/tools/fiori/tools.js +356 -0
  96. package/dist/tools/fiori/types.js +1 -0
  97. package/dist/tools/local-build.js +76 -0
  98. package/dist/tools/local-files.js +31 -0
  99. package/dist/tools/project/_merge-shared.js +68 -0
  100. package/dist/tools/project/cca_merge.js +164 -0
  101. package/dist/tools/project/playbook_get.js +1 -1
  102. package/dist/tools/project/upgrade_merge_progress.js +206 -0
  103. package/dist/tools/sap-read.js +53 -9
  104. package/dist/tools/sap-write.js +530 -21
  105. package/dist/tools/shell/shell_exec.js +41 -6
  106. package/dist/tools/snapshot.js +37 -14
  107. package/dist/tools/subagent/background_run.js +17 -1
  108. package/dist/tools/transport-resolution.js +86 -0
  109. package/dist/tools/transport.js +224 -5
  110. package/dist/tools/write-mode.js +4 -0
  111. package/dist/ui/app.js +6 -2
  112. package/dist/ui/body.js +13 -0
  113. package/dist/ui/footer.js +20 -6
  114. package/dist/ui/line-resolution.js +17 -6
  115. package/dist/ui/session-timeline.js +1 -0
  116. package/dist/ui/text-input.js +150 -0
  117. package/dist/ui/widgets/ask-question-modal.js +4 -1
  118. package/package.json +19 -3
  119. package/bench/README.md +0 -78
  120. package/bench/prompts/abap-document-cds.md +0 -44
  121. package/bench/prompts/abap-explain-bdef-handler.md +0 -57
  122. package/bench/prompts/abap-test-method.md +0 -42
  123. package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
  124. package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
  125. package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
  126. package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
  127. package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
  128. package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
  129. package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
  130. package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
  131. package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # CSPeach
2
2
 
3
- AI-assisted ABAP development CLI. 34 skills, direct SAP access, safety
3
+ AI-assisted ABAP development CLI. 35 skills, direct SAP access, safety
4
4
  gates the model cannot bypass.
5
5
 
6
6
  ## Install
@@ -19,7 +19,7 @@ You help senior SAP consultants and ABAP developers plan, build, modernize, migr
19
19
  - Live access to the user's SAP system via the sap_* tool family (~30 tools — get_source, set_source, atc_run, transport_create, sql_query, snapshot_take, etc.).
20
20
  - Filesystem + shell + web access to the project the user is working in (file_read, file_edit, file_write, glob, grep, shell_exec, web_fetch, web_search).
21
21
  - Subagent dispatch (agent_run) for focused sub-tasks; background_run + monitor_emit for non-blocking processes; schedule_draft_create for persisting schedule envelopes (no runner yet — see tool description).
22
- - A library of 34 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
22
+ - A library of 35 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
23
23
  - The project context for this session (CDS views, tables, packages, conventions) — appended below if available.
24
24
 
25
25
  When the user states an intent, your job is:
@@ -1,8 +1,9 @@
1
1
  import chalk from 'chalk';
2
2
  import { basename } from 'node:path';
3
3
  import { dispatchTool } from './tool-dispatch.js';
4
- import { listTools, toAnthropicTools } from '../tools/index.js';
4
+ import { listTools, getTool, toAnthropicTools } from '../tools/index.js';
5
5
  import { saveSession } from '../session/store.js';
6
+ import { recordCompletedToolCall } from '../session/pending.js';
6
7
  import { loadConfig } from '../config/loader.js';
7
8
  import { retryWithBackoff } from './retry.js';
8
9
  import { renderChunk, initialBufferState, render } from '../renderer/pipeline.js';
@@ -19,9 +20,9 @@ import { buildRetryCapPausePayload, buildSkippedSiblingResults } from './retry-c
19
20
  import { appendCostLine, buildEntry as buildCostEntry } from '../cost/cost-log.js';
20
21
  import { lastAssistantIsAwaitingAnswer } from '../session/awaiting-answer.js';
21
22
  import { enrichUserMessage } from '../router/intent-extractor.js';
22
- import { resetRule8State } from '../repl/rule8-detector.js';
23
- import { runSaveCommand, runPromoteCommand } from '../projects/index.js';
24
- import { parseFromFlag } from '../skills/promotion-dispatch.js';
23
+ import { resetRule8State, shouldGateRule8 } from '../repl/rule8-detector.js';
24
+ import { runSaveCommand, runPromoteCommand, detectArtifactSkill, offerAnswerBlockers } from '../projects/index.js';
25
+ import { parseFromFlag, coerceDocFromFlagToAttachment } from '../skills/promotion-dispatch.js';
25
26
  import { input } from '@inquirer/prompts';
26
27
  import { withInquirer } from '../repl/inquirer-guard.js';
27
28
  import { expandTextFileAttachments } from '../projects/workspace.js';
@@ -61,7 +62,7 @@ const SAVE_HOOK_SKILLS = new Set([
61
62
  // 2026-05-08: upgrade pipeline manifest envelopes. Each emits a
62
63
  // <!-- csforge:upgrade-manifest --> block that the matching
63
64
  // extract-upgrade.* parser turns into the manifest envelope. The
64
- // heavy detail (per-finding data) stays in `.abapforge/upgrades/`,
65
+ // heavy detail (per-finding data) stays in `.cspeach/upgrades/`,
65
66
  // referenced by detail_path inside the envelope.
66
67
  'abap-upgrade-scan',
67
68
  'abap-upgrade-fix',
@@ -69,7 +70,7 @@ const SAVE_HOOK_SKILLS = new Set([
69
70
  'abap-upgrade-merge',
70
71
  // 2026-05-09: CCA pipeline manifest envelopes. /abap-cca emits a
71
72
  // <!-- csforge:cca-manifest --> block wrapping the heavy project.json
72
- // detail at .abapforge/cca/projects/<slug>/. /abap-cca-merge produces
73
+ // detail at .cspeach/cca/projects/<slug>/. /abap-cca-merge produces
73
74
  // a consolidated cca-assessment from N parallel consultant slices.
74
75
  'abap-cca',
75
76
  'abap-cca-merge',
@@ -119,9 +120,21 @@ export function hasProjectContextBlock(messages) {
119
120
  *
120
121
  * Single shape `(q) => Promise<string>` so it slots straight into the
121
122
  * existing `prompt:` callback in runSaveCommand + runPromoteCommand.
123
+ * Exported for A1 (2026-06-10): repl.tsx reuses it as the consent prompt
124
+ * for harness-owned plan auto-run (offerNextPhaseAutoRun).
122
125
  */
123
- async function inkAwarePrompt(q) {
124
- const { shouldUseInk } = await import('../renderer/tty.js');
126
+ export async function inkAwarePrompt(q) {
127
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
128
+ // B5 (2026-06-11) — headless runs must NEVER block on stdin (defect D1
129
+ // family). Any prompt that reaches the generic helper in headless mode
130
+ // takes the conservative default: empty answer (callers treat '' as
131
+ // decline / skip). Prompts that need a different headless default (e.g.
132
+ // the artifact-save hook, which defaults YES) install their own
133
+ // headless-aware prompt BEFORE reaching this helper — see maybeOfferSave.
134
+ if (isHeadless()) {
135
+ console.error(chalk.dim(`headless: skipped prompt '${q.trim()}' → '' (default)`));
136
+ return '';
137
+ }
125
138
  if (shouldUseInk()) {
126
139
  const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
127
140
  const result = await askQuestionEmitter.request({
@@ -134,25 +147,117 @@ async function inkAwarePrompt(q) {
134
147
  }
135
148
  return withInquirer(() => input({ message: q }));
136
149
  }
150
+ /**
151
+ * Choices picker for an answer-blocker that carries `options`. Renders the
152
+ * native Ink AskQuestionModal (the same CC-style picker the ask_question tool
153
+ * uses) with the candidate answers plus "Write my own" and "Skip" escapes.
154
+ *
155
+ * - headless → skip (never blocks on stdin)
156
+ * - classic (non-Ink) → 'write' so the caller falls back to the free-text box
157
+ * - Ink → choice/multi modal; Esc or empty selection → skip
158
+ *
159
+ * The two sentinels are filtered out of any real picked answer; for multi-select
160
+ * the modal returns comma-joined values, so Write/Skip win over any co-selected
161
+ * options.
162
+ */
163
+ const CHOOSE_WRITE = '__cspeach_write__';
164
+ const CHOOSE_SKIP = '__cspeach_skip__';
165
+ export async function inkAwareChoose(a) {
166
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
167
+ if (isHeadless())
168
+ return { kind: 'skip' };
169
+ if (!shouldUseInk())
170
+ return { kind: 'write' }; // classic REPL → free-text fallback
171
+ const choices = [
172
+ ...a.options.map((o) => ({ value: o, label: o })),
173
+ { value: CHOOSE_WRITE, label: '✏️ Write my own…' },
174
+ { value: CHOOSE_SKIP, label: '↷ Skip (leave for the viewer)' },
175
+ ];
176
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
177
+ const res = await askQuestionEmitter.request({
178
+ id: 'answer-blocker',
179
+ question: `Blocker ${a.index + 1}/${a.total}: ${a.question}`,
180
+ context: a.multiSelect ? 'Space to toggle one or more · Enter to submit' : undefined,
181
+ kind: a.multiSelect ? 'multi' : 'choice',
182
+ choices,
183
+ });
184
+ if (res.cancelled || res.answer == null)
185
+ return { kind: 'skip' };
186
+ const values = res.answer.split(',');
187
+ if (values.includes(CHOOSE_WRITE))
188
+ return { kind: 'write' };
189
+ if (values.includes(CHOOSE_SKIP))
190
+ return { kind: 'skip' };
191
+ const real = values.filter((v) => v !== CHOOSE_WRITE && v !== CHOOSE_SKIP);
192
+ if (real.length === 0)
193
+ return { kind: 'skip' };
194
+ return { kind: 'picked', answer: real.join(', ') };
195
+ }
137
196
  export async function maybeOfferSave(p) {
138
- if (!SAVE_HOOK_SKILLS.has(p.skill))
139
- return;
140
197
  if (p.assistantText.trim().length === 0)
141
198
  return;
199
+ // C1 (2026-06-11, defect D22) — the hook fires on artifact EVIDENCE, not
200
+ // only on the routed-skill label. The seed-session plan save was lost
201
+ // because the turn was misrouted (label 'abap-rap', not in the set) while
202
+ // the output carried a complete csforge:plan-manifest block. The label
203
+ // stays the primary trigger — it is authoritative for prose-extracted
204
+ // artefacts with no inline marker (spec-gap / design / estimate) and keeps
205
+ // merge-skill provenance exact (abap-cca-merge saves as abap-cca-merge).
206
+ // When the label would NOT fire, a recognizable manifest block in the
207
+ // output resolves the registry skill instead (detectArtifactSkill), so a
208
+ // misroute can no longer silently lose the artifact.
209
+ let saveSkill = p.skill;
210
+ if (!SAVE_HOOK_SKILLS.has(p.skill)) {
211
+ const detected = detectArtifactSkill(p.assistantText);
212
+ if (!detected)
213
+ return;
214
+ saveSkill = detected;
215
+ p.emit(chalk.dim(`[save] ${detected} artifact manifest detected in this turn's output (turn was labelled '${p.skill}') — offering save.`));
216
+ }
217
+ // B5 (2026-06-11) — defect D2: in headless runs the artifact IS the point
218
+ // of the run, so the "Save as project file? [y/N]" prompt defaults YES
219
+ // instead of blocking on stdin that will never answer. The auto-answer is
220
+ // logged so the transcript stays honest about who said yes.
221
+ //
222
+ // CONTRACT: this headless prompt blanket-answers 'y' to ANY question it
223
+ // is asked, so runSaveCommand must ask at most ONE question (the save
224
+ // confirm). See the matching note at the prompt site in
225
+ // src/projects/save-command.ts before adding further prompts there.
226
+ const { isHeadless } = await import('../renderer/tty.js');
227
+ const prompt = isHeadless()
228
+ ? async (q) => {
229
+ p.emit(chalk.dim(`headless: auto-answered '${q.trim()}' → 'y' (artifact save defaults YES)`));
230
+ return 'y';
231
+ }
232
+ : inkAwarePrompt;
142
233
  try {
143
- await runSaveCommand({
234
+ const savedPath = await runSaveCommand({
144
235
  skillOutput: p.assistantText,
145
- skillName: p.skill,
236
+ skillName: saveSkill,
146
237
  skillVersion: '1.0',
147
238
  skillInput: p.userMessage,
148
239
  tokensUsed: p.tokensUsed,
149
240
  model: p.model,
150
241
  author: getAuthorIdentity(),
151
242
  cwd: process.cwd(),
152
- prompt: inkAwarePrompt,
243
+ prompt,
153
244
  log: (...lines) => lines.forEach((l) => p.emit(l)),
154
245
  promotedFrom: p.promotedFrom ?? null,
155
246
  });
247
+ // After a spec-gap file is saved, let a developer answer the open BLOCKER
248
+ // questions inline (recorded as `answered` items in the SAME file) — the
249
+ // business questions stay open for a consultant to answer in the viewer.
250
+ // Uses inkAwarePrompt directly (NOT the save hook's blanket-'y' headless
251
+ // prompt): in headless it returns '' → declined, so nothing is written.
252
+ if (savedPath && saveSkill === 'abap-spec-gap') {
253
+ await offerAnswerBlockers({
254
+ path: savedPath,
255
+ identity: getAuthorIdentity(),
256
+ prompt: inkAwarePrompt,
257
+ choose: inkAwareChoose,
258
+ log: (...lines) => lines.forEach((l) => p.emit(l)),
259
+ });
260
+ }
156
261
  }
157
262
  catch (err) {
158
263
  p.emit(chalk.yellow(`\n[save] skipped: ${err?.message ?? err}`));
@@ -182,7 +287,17 @@ export async function runTurn(params) {
182
287
  let userMessageForLLM = params.userMessage;
183
288
  let userMessageForSave = params.userMessage;
184
289
  {
185
- const { fromPath, rest } = parseFromFlag(params.userMessage);
290
+ // Forgiveness: `--from @<document>` (a .docx/.pdf/.txt/.md, not a saved
291
+ // .cspeach.json) means "attach this file", not "chain a result". Rewrite it
292
+ // to a plain @<file> attachment and let Phase D ingest it — never error out
293
+ // and cancel the turn on this very natural slip.
294
+ const docAttach = coerceDocFromFlagToAttachment(params.userMessage);
295
+ if (docAttach) {
296
+ emit(chalk.dim(`[--from] ${docAttach.filename} is a document, not a saved result — attaching it as input instead.`));
297
+ userMessageForLLM = docAttach.rewritten;
298
+ userMessageForSave = docAttach.rewritten;
299
+ }
300
+ const { fromPath, rest } = parseFromFlag(docAttach ? '' : params.userMessage);
186
301
  if (fromPath) {
187
302
  const result = await runPromoteCommand({
188
303
  sourcePath: fromPath,
@@ -331,6 +446,10 @@ export async function runTurn(params) {
331
446
  // already handed to onPlanManifest, so we fire once per DISTINCT manifest
332
447
  // (a continue-in-session run emits one per phase) and not once per round.
333
448
  let lastFiredManifest = null;
449
+ // Bounded auto-continue across max_tokens truncations within one turn, so a
450
+ // long phase (e.g. /abap-plan c2.behavior) that overruns the output cap
451
+ // mid-response finishes instead of dying with "unexpected stop reason".
452
+ let maxTokenContinuations = 0;
334
453
  // (emit was hoisted to the top of the function in v0.5 so the Phase J
335
454
  // promote dispatch could share it. Original location was here.)
336
455
  while (true) {
@@ -400,7 +519,9 @@ export async function runTurn(params) {
400
519
  // makes the 400 categorically impossible. Idempotent and O(messages).
401
520
  healSessionMessagesInPlace(session.messages);
402
521
  const streamParams = {
403
- model: cfg.default_model,
522
+ // A4 — per-turn override (plan model tiering) wins over the
523
+ // configured default; absent on every non-plan turn.
524
+ model: params.modelOverride ?? cfg.default_model,
404
525
  // v0.3.1 — was 8192. Bumped because code-gen skills (abap-generate
405
526
  // etc.) kept running out of budget mid-turn: adaptive thinking +
406
527
  // multiple tool-result prompts + TL;DR mandate + summary prose
@@ -460,6 +581,7 @@ export async function runTurn(params) {
460
581
  let currentAssistantContent = [];
461
582
  let sawEndTurn = false;
462
583
  let sawToolUse = false;
584
+ let sawMaxTokens = false;
463
585
  // M10 — ensure session.usage is present (shipped session schema may omit it
464
586
  // for older saved sessions; we default to zero on first turn).
465
587
  if (!session.usage) {
@@ -583,6 +705,8 @@ export async function runTurn(params) {
583
705
  sawEndTurn = true;
584
706
  if (stop_reason === 'tool_use')
585
707
  sawToolUse = true;
708
+ if (stop_reason === 'max_tokens' || stop_reason === 'length')
709
+ sawMaxTokens = true;
586
710
  // H2 — `deltaUsage.output_tokens` is the cumulative running total for
587
711
  // THIS message, NOT a per-event delta. Compute the increment vs the
588
712
  // last-seen cumulative and add only that. Guard against `null` /
@@ -675,7 +799,10 @@ export async function runTurn(params) {
675
799
  const turnDurationMs = Date.now() - turnStartedAt;
676
800
  const entry = buildCostEntry({
677
801
  turn: turnNumber,
678
- model: session.model,
802
+ // A4 — when a per-turn override ran the stream on a different model
803
+ // (plan model tiering), the cost line must record the ACTUAL model
804
+ // or the Sonnet-priced turn would be billed at session-model rates.
805
+ model: params.modelOverride ?? session.model,
679
806
  tokens: turnTokens,
680
807
  duration_ms: turnDurationMs,
681
808
  });
@@ -742,7 +869,10 @@ export async function runTurn(params) {
742
869
  assistantText,
743
870
  userMessage: userMessageForSave,
744
871
  tokensUsed: Math.max(0, (session.usage?.output_tokens ?? 0) - turnStartOutputTokens),
745
- model: cfg.default_model ?? session.model ?? 'unknown',
872
+ // A4 — record the actual turn model if an override ran the stream
873
+ // (unreachable today: plan-resume turns suppress this hook, and
874
+ // they are the only modelOverride caller — kept honest anyway).
875
+ model: params.modelOverride ?? cfg.default_model ?? session.model ?? 'unknown',
746
876
  emit,
747
877
  promotedFrom: promotedFromForSave,
748
878
  });
@@ -774,7 +904,29 @@ export async function runTurn(params) {
774
904
  return;
775
905
  }
776
906
  if (!sawToolUse) {
777
- // Unexpected stop reason — bail to avoid infinite loop.
907
+ // The response was cut off at the output-token cap (stop_reason
908
+ // 'max_tokens'/'length') mid-generation — NOT a real end of turn. The
909
+ // truncated assistant message is already persisted (line ~733), so
910
+ // continue the turn and let the model finish (and, for /abap-plan, still
911
+ // emit its manifest). Bounded so a pathologically long response can't
912
+ // loop forever.
913
+ const MAX_TOKEN_CONTINUATIONS = 4;
914
+ if (sawMaxTokens && maxTokenContinuations < MAX_TOKEN_CONTINUATIONS) {
915
+ maxTokenContinuations += 1;
916
+ emit(chalk.yellow(`\n[output limit reached — auto-continuing (${maxTokenContinuations}/${MAX_TOKEN_CONTINUATIONS})]`));
917
+ session.messages.push({
918
+ role: 'user',
919
+ content: 'Your previous response was cut off at the output-token limit. Continue exactly where you left off — do NOT repeat what you already wrote. '
920
+ + 'If you were mid-way through a tool call, or (for /abap-plan) had not yet emitted the closing csforge:plan-manifest block, complete it now. '
921
+ + 'Keep narration brief to stay within the limit.',
922
+ });
923
+ continue;
924
+ }
925
+ if (sawMaxTokens) {
926
+ emit(chalk.yellow(`\n[output limit hit ${maxTokenContinuations}× this turn — stopping. Type "continue" to resume.]`));
927
+ return;
928
+ }
929
+ // Truly unexpected stop reason — bail to avoid an infinite loop.
778
930
  emit(chalk.yellow('\n[unexpected stop reason — ending turn]'));
779
931
  return;
780
932
  }
@@ -809,7 +961,27 @@ export async function runTurn(params) {
809
961
  // every terminal, Ink included.
810
962
  // Interactive tools wait on the USER, not the system — ticking
811
963
  // "ask_question running… (30s)" while they think is noise.
812
- const isInteractiveTool = block.name === 'ask_question' || block.name === 'request_approval';
964
+ // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
965
+ // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
966
+ // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
967
+ // BEFORE the op runs. file_write / shell_exec are otherwise classified
968
+ // non-interactive, so without this the loop would start the per-second
969
+ // heartbeat + keep the turn-status row ticking UNDER the modal — the
970
+ // observed live bug (doubled card, lost Enter, history-replay leak,
971
+ // ~10-min wedge with the tool spinner ticking under the modal). Detect
972
+ // the gate the SAME way tool-dispatch does (tool.isMutating &&
973
+ // shouldGateRule8) and treat it as user-blocking: pause the status row,
974
+ // skip the heartbeat. The gate's own clearActiveSpinner +
975
+ // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
976
+ // is the outer one — together no live render source contends with the
977
+ // modal for the Ink frame or raw-mode stdin.
978
+ const willTripBatchGate = (() => {
979
+ const t = getTool(block.name);
980
+ return !!t?.isMutating && shouldGateRule8();
981
+ })();
982
+ const isInteractiveTool = block.name === 'ask_question' ||
983
+ block.name === 'request_approval' ||
984
+ willTripBatchGate;
813
985
  // Interactive tools block on the USER. PAUSE the turn-status row (don't
814
986
  // just relabel it): a live 250ms tick repaints the dynamic frame and
815
987
  // overdraws the inquirer approval picker / churns the Ink ask_question
@@ -828,6 +1000,11 @@ export async function runTurn(params) {
828
1000
  });
829
1001
  // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
830
1002
  const dispatchStart = Date.now();
1003
+ // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1004
+ // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1005
+ // id the loop records below. Dispatch is sequential, so a single slot
1006
+ // on the shared ctx is safe.
1007
+ params.ctx.toolUseId = block.id;
831
1008
  // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
832
1009
  const guardDecision = checkAndMark(block.name, writeGuard);
833
1010
  let result;
@@ -837,6 +1014,9 @@ export async function runTurn(params) {
837
1014
  : { content: guardDecision.errorContent, is_error: true };
838
1015
  }
839
1016
  finally {
1017
+ // D19: clear the slot so a future non-loop invocation on this ctx
1018
+ // can't inherit a stale block id.
1019
+ params.ctx.toolUseId = undefined;
840
1020
  // Stop on the error path too — the interval is unref'd but would
841
1021
  // otherwise keep printing "<tool> running…" into the NEXT prompt
842
1022
  // after a dispatch throw.
@@ -864,16 +1044,25 @@ export async function runTurn(params) {
864
1044
  isError: result.is_error ?? false,
865
1045
  resultSummary,
866
1046
  chunkEmitter: params.chunkEmitter,
1047
+ // D29 (2026-06-12): self-identifying result row. Heartbeat lines,
1048
+ // sap-client warns, and notice lines legitimately print between
1049
+ // the ⏺ top line and this row — without the name here those rows
1050
+ // read as anonymous `⎿ ✓ 364ms` orphans in the transcript.
1051
+ name: block.name,
1052
+ args: (block.input ?? {}),
867
1053
  });
868
1054
  // v0.3 (Step 28.0): populate session.toolCalls so the SessionTimeline
869
1055
  // overlay can render per-write history. Cap per-entry payload size at
870
1056
  // 4kB to bound session.json growth.
1057
+ // D19 (2026-06-11): upsert, not push — write handlers journal the same
1058
+ // call through the WAL under this block.id; pushing unconditionally
1059
+ // produced two ledger entries per approval-gated write.
871
1060
  const completedAt = new Date().toISOString();
872
1061
  {
873
1062
  const resultText = typeof result.content === 'string'
874
1063
  ? result.content
875
1064
  : JSON.stringify(result.content);
876
- session.toolCalls.push({
1065
+ recordCompletedToolCall(session, {
877
1066
  tool_use_id: block.id,
878
1067
  tool: block.name,
879
1068
  args: block.input ?? {},
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Thrown when a non-managed LLM mode is used without a valid CSPeach login.
3
+ * Carries a user-facing message; callers print it and exit rather than dumping
4
+ * a stack trace.
5
+ */
6
+ export class CspeachLicenseError extends Error {
7
+ mode;
8
+ constructor(mode) {
9
+ super(`CSPeach requires a license to run in '${mode}' mode.\n` +
10
+ `In this mode your prompts go straight to your own LLM — but the CSPeach skills\n` +
11
+ `and CLI are licensed software. Run \`cspeach login\` to activate your license.\n` +
12
+ `Need an access key? Email laeeq.siddique@cremencing.com.`);
13
+ this.name = 'CspeachLicenseError';
14
+ this.mode = mode;
15
+ }
16
+ }
17
+ /**
18
+ * Gate every NON-managed mode (byok / local / ai-hub) on a valid CSPeach login.
19
+ *
20
+ * Why: managed mode authenticates against the proxy on every call, so it is
21
+ * already gated. The other modes bypass our proxy for inference — and the
22
+ * skills (our IP) are bundled into the package — so without this check a fresh
23
+ * `npm install` is fully usable by anyone who never logged in, untracked. A
24
+ * CSPeach key is admin-issued (no open self-signup), so requiring one means we
25
+ * know who is using the product and can revoke access.
26
+ *
27
+ * Scope (Part A): require a CSPeach key to be PRESENT. Server-side validation +
28
+ * revocation (so a forged key fails) is the Part-B follow-up — see
29
+ * docs/byok-portal-handover.md §0.
30
+ */
31
+ export async function assertModeLicensed(mode, getBearer) {
32
+ if (mode === 'managed')
33
+ return;
34
+ let key = '';
35
+ try {
36
+ key = (await getBearer()) ?? '';
37
+ }
38
+ catch {
39
+ key = '';
40
+ }
41
+ if (key.trim().length === 0) {
42
+ throw new CspeachLicenseError(mode);
43
+ }
44
+ }
@@ -7,7 +7,7 @@
7
7
  * with full conversation context.
8
8
  *
9
9
  * What Layer 1 does NOT cover: skill-internal state files. /abap-cca, for
10
- * example, maintains `.abapforge/cca/projects/<id>/project.json` with an
10
+ * example, maintains `.cspeach/cca/projects/<id>/project.json` with an
11
11
  * inventory of objects, classifications, and per-package state. The skill
12
12
  * writes this file at PHASE BOUNDARIES (end of DISCOVER, end of INVENTORY,
13
13
  * etc.). A crash mid-DISCOVER means the partial inventory in project.json
@@ -23,6 +23,21 @@ export async function dispatchTool(name, args, ctx) {
23
23
  batch_count: getWriteOpsThisTurn().length + 1,
24
24
  });
25
25
  if (!r.confirmed) {
26
+ // B5 — discriminate WHO declined. A headless fail-fast (nobody ever
27
+ // saw the card) must not masquerade as a user decision: the model
28
+ // (and any transcript reader) reacts differently to "the user said
29
+ // no" vs "no user was available to say yes". Uniform `headless: true`
30
+ // marker matches the other headless result payloads.
31
+ if (r.headless) {
32
+ return {
33
+ content: JSON.stringify({
34
+ error: 'headless_safety_decline',
35
+ headless: true,
36
+ reason: r.reason,
37
+ }),
38
+ is_error: true,
39
+ };
40
+ }
26
41
  return {
27
42
  content: JSON.stringify({ error: 'cancelled_by_user', reason: r.reason }),
28
43
  is_error: true,
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Canonical approval-object strings (Task A3 — defects D17/D21/D28).
3
+ *
4
+ * The battery exposed an APPROVAL_INVALID:object_mismatch epidemic (~8 wasted
5
+ * approval round-trips): request_approval minted whatever free-text object
6
+ * string the model wrote in the change row, while every write tool validated
7
+ * against a structured argument (args.name / args.className / args.description
8
+ * / objects[0].name). Any decoration the model added at mint time — and models
9
+ * reliably add decoration — broke the byte-equality check in
10
+ * verifyAndSpendApprovalId.
11
+ *
12
+ * Observed mismatch classes (exact strings from the battery sessions):
13
+ * D17 "ZBP_I_DOWNTIMELOG (CCIMP / lhc_DowntimeLog.validateEndAfterStart)"
14
+ * vs "ZBP_I_DOWNTIMELOG" → trailing parenthetical
15
+ * D21 "Transport: Magic fix" vs "Magic fix" → leading transport label
16
+ * D28 "ZR_PM_MAINTREQ,ZBP_R_PM_MAINTREQ,…" vs "ZR_PM_MAINTREQ"
17
+ * → one batch activate approval, per-object activate calls
18
+ *
19
+ * The fix is ONE code path for both sides:
20
+ * - mint side (jwt.ts#mintChangeApproval, used by request_approval) stores
21
+ * canonicalApprovalObject(change.object) in the JWT;
22
+ * - spend side (jwt.ts#verifyAndSpendApprovalId) compares via
23
+ * approvalObjectMatches, which canonicalizes BOTH the JWT payload object
24
+ * and the tool's expected object before comparing.
25
+ * The spend side canonicalizes too because (a) the expected side is always the
26
+ * raw structured tool argument, never pre-canonicalized, and (b) it is
27
+ * defense-in-depth against any future mint site that bypasses
28
+ * mintChangeApproval and stores a raw string.
29
+ */
30
+ /** Leading transport-request labels the model prepends to CTS descriptions. */
31
+ const LEADING_LABEL = /^(?:transport(?:\s+request)?|tr|cts|request)\s*:\s*/i;
32
+ /** A trailing parenthetical decoration, e.g. "(testclasses include)". */
33
+ const TRAILING_PAREN = /\s*\([^()]*\)\s*$/;
34
+ /** What an ABAP repository object name looks like once canonicalized. */
35
+ const OBJECT_NAME = /^[A-Z0-9_/]+$/;
36
+ /**
37
+ * Normalize an approval object string so mint and spend sides can never
38
+ * disagree on decoration, case, or whitespace:
39
+ *
40
+ * 1. trim
41
+ * 2. strip a leading "Transport:" / "TR:" / "Request:" label (D21)
42
+ * 3. strip trailing parenthetical decorations, stacked or single (D17)
43
+ * 4. collapse internal whitespace
44
+ * 5. uppercase
45
+ *
46
+ * All steps are applied symmetrically to both sides. Note that step 3 strips
47
+ * trailing parentheticals from free text too: transport descriptions
48
+ * "Fix dump (urgent)" and "Fix dump (rollback)" both canonicalize to
49
+ * "FIX DUMP" and therefore cross-match. This collision class is accepted
50
+ * because descriptions are display labels, not object identities, and every
51
+ * approval is session-scoped, TTL-bounded, and user-confirmed — the user saw
52
+ * the specific change row that minted the JWT.
53
+ */
54
+ export function canonicalApprovalObject(raw) {
55
+ let s = String(raw ?? '').trim();
56
+ s = s.replace(LEADING_LABEL, '');
57
+ for (let prev = ''; prev !== s;) {
58
+ prev = s;
59
+ s = s.replace(TRAILING_PAREN, '');
60
+ }
61
+ return s.replace(/\s+/g, ' ').trim().toUpperCase();
62
+ }
63
+ /**
64
+ * Spend-side comparison: does the object string stored in the approval JWT
65
+ * authorize an operation on `expected` (the structured argument the write
66
+ * tool validates against)? Returns the match kind, or `false` for no match.
67
+ *
68
+ * - byte equality ('exact'), OR
69
+ * - canonical equality ('canonical'), OR
70
+ * - the minted string is a comma-joined list of ABAP object names and
71
+ * `expected` is one of them ('list_member', D28 — batch activate approval
72
+ * spent by per-object sap_activate calls).
73
+ *
74
+ * The list rule only applies when EVERY part looks like an object name
75
+ * (no spaces / free text), so a transport description containing commas can
76
+ * never partially match.
77
+ */
78
+ export function approvalObjectMatches(minted, expected) {
79
+ if (minted === expected)
80
+ return 'exact';
81
+ const m = canonicalApprovalObject(minted);
82
+ const e = canonicalApprovalObject(expected);
83
+ if (m === e)
84
+ return 'canonical';
85
+ const parts = m.split(',').map((p) => p.trim()).filter((p) => p.length > 0);
86
+ if (parts.length < 2)
87
+ return false;
88
+ if (!parts.every((p) => OBJECT_NAME.test(p)))
89
+ return false;
90
+ return parts.includes(e) ? 'list_member' : false;
91
+ }
@@ -1,5 +1,7 @@
1
1
  import { SignJWT, jwtVerify } from 'jose';
2
2
  import crypto from 'node:crypto';
3
+ import { auditLog } from '@cspeach/sap-client';
4
+ import { canonicalApprovalObject, approvalObjectMatches } from './canonical.js';
3
5
  // Session-scoped HMAC key — generated per CLI start, never persisted.
4
6
  const sessionKey = crypto.randomBytes(32);
5
7
  const usedNonces = new Set();
@@ -43,6 +45,24 @@ export async function mintApprovalId(payload) {
43
45
  .setExpirationTime('30m')
44
46
  .sign(sessionKey);
45
47
  }
48
+ /**
49
+ * THE minting path for request_approval changes (Task A3 — D17/D21/D28).
50
+ *
51
+ * The object string stored in the JWT is canonicalized by the same helper the
52
+ * spend-side comparison in verifyAndSpendApprovalId uses, so mint and spend
53
+ * can never disagree on decoration ("Transport: …", "(testclasses include)"),
54
+ * case, or whitespace. request_approval and any future minting site MUST go
55
+ * through this function instead of calling mintApprovalId with a free-text
56
+ * object string.
57
+ */
58
+ export async function mintChangeApproval(change, sessionId) {
59
+ return mintApprovalId({
60
+ object: canonicalApprovalObject(change.object),
61
+ type: change.type,
62
+ op: change.op,
63
+ session_id: sessionId,
64
+ });
65
+ }
46
66
  export async function verifyAndSpendApprovalId(jwt, expectedObject, expectedOp) {
47
67
  // 2026-05-15 (bug 7): the previous catch lumped EVERY thrown failure
48
68
  // into `'invalid_signature'`, including expirations (which used to be
@@ -75,8 +95,19 @@ export async function verifyAndSpendApprovalId(jwt, expectedObject, expectedOp)
75
95
  // taxonomy short. Mismatch / spent-nonce paths NEVER reach this catch.
76
96
  return { ok: false, reason: 'invalid_signature' };
77
97
  }
78
- if (payload.object !== expectedObject)
98
+ // Canonical comparison (Task A3 — D17/D21/D28): both sides are normalized
99
+ // by the same helper that mintChangeApproval used at mint time, and a
100
+ // comma-joined batch approval matches each of its member objects. The spend
101
+ // side canonicalizes as well because (a) `expectedObject` is always the raw
102
+ // structured tool argument, never pre-canonicalized, and (b) it is
103
+ // defense-in-depth against any future mint site that bypasses
104
+ // mintChangeApproval and stores a raw string. (It is NOT back-compat for
105
+ // old JWTs: the HMAC key is per-process and never persisted, so no
106
+ // pre-canonicalization token can ever reach this verifier.)
107
+ const matchKind = approvalObjectMatches(payload.object, expectedObject);
108
+ if (!matchKind) {
79
109
  return { ok: false, reason: 'object_mismatch' };
110
+ }
80
111
  // Op-group coverage: a create approval also authorizes the modify+activate
81
112
  // that complete the create; a modify approval authorizes its activate.
82
113
  const covered = OP_COVERAGE[payload.op];
@@ -90,5 +121,11 @@ export async function verifyAndSpendApprovalId(jwt, expectedObject, expectedOp)
90
121
  return { ok: false, reason: 'nonce_spent' };
91
122
  usedNonces.add(payload.nonce);
92
123
  }
93
- return { ok: true, payload };
124
+ // Auditability: a lenient (non-byte-equal) match is an authorization
125
+ // decision worth a trace — record WHICH fold let the spend through, in the
126
+ // same audit log the write tools use.
127
+ if (matchKind !== 'exact') {
128
+ await auditLog('approvalSpend', payload.type, expectedObject, 'success', 0, `matchKind=${matchKind} minted="${payload.object}"`);
129
+ }
130
+ return { ok: true, payload, matchKind };
94
131
  }
@@ -0,0 +1,25 @@
1
+ /**
2
+ * Fetch the organisation's Anthropic API key from the CSPeach proxy.
3
+ *
4
+ * Returns the key string on success, or null on any failure.
5
+ * Non-fatal by design — callers must fall back to the manual prompt when null
6
+ * is returned. The key is NEVER logged.
7
+ */
8
+ export async function fetchOrgAnthropicKey(proxyUrl, bearer, deps) {
9
+ const fetchFn = deps?.fetchImpl ?? fetch;
10
+ const url = `${proxyUrl.replace(/\/$/, '')}/v1/me/anthropic-key`;
11
+ try {
12
+ const r = await fetchFn(url, {
13
+ headers: { Authorization: `Bearer ${bearer}` },
14
+ });
15
+ if (!r.ok)
16
+ return null;
17
+ const body = await r.json();
18
+ if (typeof body.key === 'string' && body.key.length > 0)
19
+ return body.key;
20
+ return null;
21
+ }
22
+ catch {
23
+ return null;
24
+ }
25
+ }