@cspeach/cli 0.9.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/README.md +1 -1
  2. package/dist/agent/intent-system-prompt.js +1 -1
  3. package/dist/agent/loop.js +228 -26
  4. package/dist/agent/providers/license-gate.js +44 -0
  5. package/dist/agent/skill-checkpoint.js +1 -1
  6. package/dist/agent/tool-dispatch.js +15 -0
  7. package/dist/approvals/canonical.js +91 -0
  8. package/dist/approvals/jwt.js +39 -2
  9. package/dist/approvals/op-labels.js +124 -0
  10. package/dist/approvals/render.js +42 -36
  11. package/dist/auth/org-anthropic-key.js +25 -0
  12. package/dist/classifier/client.js +18 -3
  13. package/dist/cli.js +15 -0
  14. package/dist/commands/compact.js +28 -2
  15. package/dist/commands/config-set.js +284 -0
  16. package/dist/commands/config-show.js +20 -0
  17. package/dist/commands/export-audit.js +43 -0
  18. package/dist/commands/help.js +5 -0
  19. package/dist/commands/login.js +31 -14
  20. package/dist/commands/plan-audit-evidence.js +266 -0
  21. package/dist/commands/plan-audit.js +692 -0
  22. package/dist/commands/plan-chain.js +671 -0
  23. package/dist/commands/plan-continue.js +179 -0
  24. package/dist/commands/plan-gate.js +154 -0
  25. package/dist/commands/plan-model-tier.js +83 -0
  26. package/dist/commands/plan-resume.js +728 -46
  27. package/dist/config/loader.js +223 -5
  28. package/dist/config/model-defaults.js +14 -0
  29. package/dist/cost/pricing.js +27 -1
  30. package/dist/doctor/checks/_http-probe.js +1 -0
  31. package/dist/doctor/checks/cert.js +14 -3
  32. package/dist/doctor/checks/sap.js +30 -8
  33. package/dist/doctor/checks/system-roles.js +41 -0
  34. package/dist/doctor/checks/zcspeach.js +19 -4
  35. package/dist/doctor/run.js +2 -0
  36. package/dist/models/resolve.js +61 -0
  37. package/dist/models/server-config.js +155 -0
  38. package/dist/one-shot.js +76 -6
  39. package/dist/projects/answer-blockers.js +137 -0
  40. package/dist/projects/extract-cca.js +111 -17
  41. package/dist/projects/extract-modernize.js +4 -2
  42. package/dist/projects/extract-plan.js +184 -37
  43. package/dist/projects/extract-spec-gap.js +34 -7
  44. package/dist/projects/extract-test-coverage.js +4 -2
  45. package/dist/projects/extract-upgrade.js +116 -23
  46. package/dist/projects/handover-md.js +195 -0
  47. package/dist/projects/index.js +5 -2
  48. package/dist/projects/merge-cca.js +292 -0
  49. package/dist/projects/merge-upgrade.js +173 -0
  50. package/dist/projects/migration.js +103 -1
  51. package/dist/projects/output-paths.js +27 -0
  52. package/dist/projects/plan-run.js +285 -27
  53. package/dist/projects/plan-schema.js +136 -3
  54. package/dist/projects/promote-command.js +25 -2
  55. package/dist/projects/promote.js +128 -0
  56. package/dist/projects/run-lease.js +157 -0
  57. package/dist/projects/save-command.js +259 -21
  58. package/dist/projects/status.js +3 -1
  59. package/dist/projects/validate.js +1 -1
  60. package/dist/projects/workspace.js +164 -20
  61. package/dist/renderer/notices.js +64 -0
  62. package/dist/renderer/progress-chatter.js +8 -0
  63. package/dist/renderer/status-footer.js +22 -12
  64. package/dist/renderer/thinking-heartbeat.js +64 -8
  65. package/dist/renderer/todo-block.js +51 -0
  66. package/dist/renderer/tool-widget.js +55 -4
  67. package/dist/renderer/tty.js +43 -4
  68. package/dist/renderer/verify-chain.js +77 -0
  69. package/dist/repl/at-picker.js +60 -7
  70. package/dist/repl/bracketed-paste.js +28 -19
  71. package/dist/repl/builtin-commands.js +42 -0
  72. package/dist/repl/current-transport.js +10 -0
  73. package/dist/repl/early-line-buffer.js +68 -0
  74. package/dist/repl/history.js +86 -0
  75. package/dist/repl/ink-stdin-guard.js +64 -0
  76. package/dist/repl/inquirer-guard.js +70 -5
  77. package/dist/repl/mode-ceiling.js +16 -0
  78. package/dist/repl/mode-cycle.js +104 -0
  79. package/dist/repl/numbered-menu.js +131 -0
  80. package/dist/repl/post-turn-status.js +26 -6
  81. package/dist/repl/rule8-detector.js +17 -2
  82. package/dist/repl/safety-confirm.js +111 -2
  83. package/dist/repl/safety-mode-state.js +19 -3
  84. package/dist/repl/slash-completer.js +5 -0
  85. package/dist/repl/slash-picker.js +10 -15
  86. package/dist/repl.js +1232 -95
  87. package/dist/rewind/candidates.js +194 -0
  88. package/dist/rewind/cli.js +137 -0
  89. package/dist/rewind/format.js +27 -0
  90. package/dist/rewind/restore.js +245 -0
  91. package/dist/router/classifier.js +150 -6
  92. package/dist/sap/capability-matrix.js +20 -0
  93. package/dist/sap/capability-matrix.json +11236 -0
  94. package/dist/sap/capability.js +146 -0
  95. package/dist/sap/connection-manager.js +19 -1
  96. package/dist/sap/onboarding.js +42 -4
  97. package/dist/session/audit-export.js +459 -0
  98. package/dist/session/context-report.js +163 -0
  99. package/dist/session/pending.js +27 -0
  100. package/dist/session/recap.js +160 -0
  101. package/dist/skill-catalog.js +51 -40
  102. package/dist/skills/bundled-skills.js +272 -1
  103. package/dist/skills/promotion-dispatch.js +23 -0
  104. package/dist/tools/_command-shared.js +36 -12
  105. package/dist/tools/_filesystem-shared.js +139 -4
  106. package/dist/tools/_flag.js +25 -0
  107. package/dist/tools/approval.js +177 -26
  108. package/dist/tools/ask-question.js +400 -7
  109. package/dist/tools/capability/tool.js +74 -0
  110. package/dist/tools/dispatch-skill.js +22 -1
  111. package/dist/tools/extend-model/anchored-insert.js +1414 -0
  112. package/dist/tools/extend-model/tool.js +340 -0
  113. package/dist/tools/filesystem/extract-document.js +57 -0
  114. package/dist/tools/filesystem/file-edit.js +12 -2
  115. package/dist/tools/filesystem/file-read.js +2 -2
  116. package/dist/tools/filesystem/file-write.js +11 -2
  117. package/dist/tools/filesystem/glob.js +11 -0
  118. package/dist/tools/filesystem/grep.js +10 -0
  119. package/dist/tools/filesystem/read-document.js +107 -0
  120. package/dist/tools/fiori/apply.js +50 -0
  121. package/dist/tools/fiori/bin.js +3 -0
  122. package/dist/tools/fiori/catalog/index.js +27 -0
  123. package/dist/tools/fiori/catalog/value-help.js +230 -0
  124. package/dist/tools/fiori/catalog/viz-chart.js +177 -0
  125. package/dist/tools/fiori/cli.js +71 -0
  126. package/dist/tools/fiori/deploy-config.js +73 -0
  127. package/dist/tools/fiori/fe-extend.js +76 -0
  128. package/dist/tools/fiori/fe-scaffold.js +71 -0
  129. package/dist/tools/fiori/floorplan-map.js +19 -0
  130. package/dist/tools/fiori/i18n.js +39 -0
  131. package/dist/tools/fiori/manifest.js +70 -0
  132. package/dist/tools/fiori/render.js +77 -0
  133. package/dist/tools/fiori/samples/data/index.json +13602 -0
  134. package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
  135. package/dist/tools/fiori/samples/loader.js +248 -0
  136. package/dist/tools/fiori/samples/search.js +63 -0
  137. package/dist/tools/fiori/samples/types.js +2 -0
  138. package/dist/tools/fiori/scaffold.js +39 -0
  139. package/dist/tools/fiori/smoke/assertions.js +74 -0
  140. package/dist/tools/fiori/smoke/browser.js +52 -0
  141. package/dist/tools/fiori/smoke/driver.js +89 -0
  142. package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
  143. package/dist/tools/fiori/smoke/run-smoke.js +149 -0
  144. package/dist/tools/fiori/tools.js +681 -0
  145. package/dist/tools/fiori/types.js +1 -0
  146. package/dist/tools/local-build.js +86 -0
  147. package/dist/tools/local-files.js +31 -0
  148. package/dist/tools/project/_merge-shared.js +68 -0
  149. package/dist/tools/project/cca_merge.js +164 -0
  150. package/dist/tools/project/playbook_get.js +1 -1
  151. package/dist/tools/project/upgrade_merge_progress.js +206 -0
  152. package/dist/tools/sap-read.js +132 -20
  153. package/dist/tools/sap-write.js +550 -21
  154. package/dist/tools/shell/shell_exec.js +41 -6
  155. package/dist/tools/snapshot.js +63 -14
  156. package/dist/tools/subagent/agent_run.js +27 -3
  157. package/dist/tools/subagent/background_run.js +17 -1
  158. package/dist/tools/todo.js +144 -0
  159. package/dist/tools/transport-resolution.js +86 -0
  160. package/dist/tools/transport.js +224 -5
  161. package/dist/tools/write-mode.js +4 -0
  162. package/dist/ui/app.js +378 -21
  163. package/dist/ui/approval-modal.js +49 -16
  164. package/dist/ui/ask-question-emitter.js +14 -0
  165. package/dist/ui/body.js +13 -0
  166. package/dist/ui/context-grid.js +108 -0
  167. package/dist/ui/footer.js +120 -27
  168. package/dist/ui/header.js +7 -0
  169. package/dist/ui/line-resolution.js +35 -8
  170. package/dist/ui/rewind-emitter.js +10 -0
  171. package/dist/ui/rewind-panel.js +81 -0
  172. package/dist/ui/sap-state-store.js +1 -0
  173. package/dist/ui/session-timeline.js +1 -0
  174. package/dist/ui/status-line.js +43 -0
  175. package/dist/ui/text-input.js +214 -0
  176. package/dist/ui/todo-emitter.js +25 -0
  177. package/dist/ui/todo-panel.js +64 -0
  178. package/dist/ui/turn-status-emitter.js +50 -4
  179. package/dist/ui/turn-status.js +18 -3
  180. package/dist/ui/widgets/ask-form.js +242 -0
  181. package/dist/ui/widgets/ask-question-modal.js +21 -8
  182. package/package.json +22 -3
  183. package/bench/README.md +0 -78
  184. package/bench/prompts/abap-document-cds.md +0 -44
  185. package/bench/prompts/abap-explain-bdef-handler.md +0 -57
  186. package/bench/prompts/abap-test-method.md +0 -42
  187. package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
  188. package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
  189. package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
  190. package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
  191. package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
  192. package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
  193. package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
  194. package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
  195. package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # CSPeach
2
2
 
3
- AI-assisted ABAP development CLI. 34 skills, direct SAP access, safety
3
+ AI-assisted ABAP development CLI. 35 skills, direct SAP access, safety
4
4
  gates the model cannot bypass.
5
5
 
6
6
  ## Install
@@ -19,7 +19,7 @@ You help senior SAP consultants and ABAP developers plan, build, modernize, migr
19
19
  - Live access to the user's SAP system via the sap_* tool family (~30 tools — get_source, set_source, atc_run, transport_create, sql_query, snapshot_take, etc.).
20
20
  - Filesystem + shell + web access to the project the user is working in (file_read, file_edit, file_write, glob, grep, shell_exec, web_fetch, web_search).
21
21
  - Subagent dispatch (agent_run) for focused sub-tasks; background_run + monitor_emit for non-blocking processes; schedule_draft_create for persisting schedule envelopes (no runner yet — see tool description).
22
- - A library of 34 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
22
+ - A library of 35 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
23
23
  - The project context for this session (CDS views, tables, packages, conventions) — appended below if available.
24
24
 
25
25
  When the user states an intent, your job is:
@@ -1,12 +1,14 @@
1
1
  import chalk from 'chalk';
2
2
  import { basename } from 'node:path';
3
3
  import { dispatchTool } from './tool-dispatch.js';
4
- import { listTools, toAnthropicTools } from '../tools/index.js';
4
+ import { listTools, getTool, toAnthropicTools } from '../tools/index.js';
5
5
  import { saveSession } from '../session/store.js';
6
+ import { recordCompletedToolCall } from '../session/pending.js';
6
7
  import { loadConfig } from '../config/loader.js';
8
+ import { resolveModelRole } from '../models/resolve.js';
7
9
  import { retryWithBackoff } from './retry.js';
8
10
  import { renderChunk, initialBufferState, render } from '../renderer/pipeline.js';
9
- import { renderToolCallTop, renderToolCallBottom, startToolSpinner, startThinkingSpinner } from '../renderer/tool-widget.js';
11
+ import { renderToolCallTop, renderToolCallBottom, startToolSpinner, startThinkingSpinner, isWidgetSuppressedTool } from '../renderer/tool-widget.js';
10
12
  import { formatToolErrorSummary } from '../sap-errors/parse-adt-exception.js';
11
13
  import { startThinkingHeartbeat } from '../renderer/thinking-heartbeat.js';
12
14
  // 2026-06-07 — status-row narration. Safe to import in every mode: the
@@ -19,9 +21,9 @@ import { buildRetryCapPausePayload, buildSkippedSiblingResults } from './retry-c
19
21
  import { appendCostLine, buildEntry as buildCostEntry } from '../cost/cost-log.js';
20
22
  import { lastAssistantIsAwaitingAnswer } from '../session/awaiting-answer.js';
21
23
  import { enrichUserMessage } from '../router/intent-extractor.js';
22
- import { resetRule8State } from '../repl/rule8-detector.js';
23
- import { runSaveCommand, runPromoteCommand } from '../projects/index.js';
24
- import { parseFromFlag } from '../skills/promotion-dispatch.js';
24
+ import { resetRule8State, shouldGateRule8 } from '../repl/rule8-detector.js';
25
+ import { runSaveCommand, runPromoteCommand, detectArtifactSkill, offerAnswerBlockers } from '../projects/index.js';
26
+ import { parseFromFlag, coerceDocFromFlagToAttachment } from '../skills/promotion-dispatch.js';
25
27
  import { input } from '@inquirer/prompts';
26
28
  import { withInquirer } from '../repl/inquirer-guard.js';
27
29
  import { expandTextFileAttachments } from '../projects/workspace.js';
@@ -61,7 +63,7 @@ const SAVE_HOOK_SKILLS = new Set([
61
63
  // 2026-05-08: upgrade pipeline manifest envelopes. Each emits a
62
64
  // <!-- csforge:upgrade-manifest --> block that the matching
63
65
  // extract-upgrade.* parser turns into the manifest envelope. The
64
- // heavy detail (per-finding data) stays in `.abapforge/upgrades/`,
66
+ // heavy detail (per-finding data) stays in `.cspeach/upgrades/`,
65
67
  // referenced by detail_path inside the envelope.
66
68
  'abap-upgrade-scan',
67
69
  'abap-upgrade-fix',
@@ -69,7 +71,7 @@ const SAVE_HOOK_SKILLS = new Set([
69
71
  'abap-upgrade-merge',
70
72
  // 2026-05-09: CCA pipeline manifest envelopes. /abap-cca emits a
71
73
  // <!-- csforge:cca-manifest --> block wrapping the heavy project.json
72
- // detail at .abapforge/cca/projects/<slug>/. /abap-cca-merge produces
74
+ // detail at .cspeach/cca/projects/<slug>/. /abap-cca-merge produces
73
75
  // a consolidated cca-assessment from N parallel consultant slices.
74
76
  'abap-cca',
75
77
  'abap-cca-merge',
@@ -119,9 +121,21 @@ export function hasProjectContextBlock(messages) {
119
121
  *
120
122
  * Single shape `(q) => Promise<string>` so it slots straight into the
121
123
  * existing `prompt:` callback in runSaveCommand + runPromoteCommand.
124
+ * Exported for A1 (2026-06-10): repl.tsx reuses it as the consent prompt
125
+ * for harness-owned plan auto-run (offerNextPhaseAutoRun).
122
126
  */
123
- async function inkAwarePrompt(q) {
124
- const { shouldUseInk } = await import('../renderer/tty.js');
127
+ export async function inkAwarePrompt(q) {
128
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
129
+ // B5 (2026-06-11) — headless runs must NEVER block on stdin (defect D1
130
+ // family). Any prompt that reaches the generic helper in headless mode
131
+ // takes the conservative default: empty answer (callers treat '' as
132
+ // decline / skip). Prompts that need a different headless default (e.g.
133
+ // the artifact-save hook, which defaults YES) install their own
134
+ // headless-aware prompt BEFORE reaching this helper — see maybeOfferSave.
135
+ if (isHeadless()) {
136
+ console.error(chalk.dim(`headless: skipped prompt '${q.trim()}' → '' (default)`));
137
+ return '';
138
+ }
125
139
  if (shouldUseInk()) {
126
140
  const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
127
141
  const result = await askQuestionEmitter.request({
@@ -134,25 +148,117 @@ async function inkAwarePrompt(q) {
134
148
  }
135
149
  return withInquirer(() => input({ message: q }));
136
150
  }
151
+ /**
152
+ * Choices picker for an answer-blocker that carries `options`. Renders the
153
+ * native Ink AskQuestionModal (the same CC-style picker the ask_question tool
154
+ * uses) with the candidate answers plus "Write my own" and "Skip" escapes.
155
+ *
156
+ * - headless → skip (never blocks on stdin)
157
+ * - classic (non-Ink) → 'write' so the caller falls back to the free-text box
158
+ * - Ink → choice/multi modal; Esc or empty selection → skip
159
+ *
160
+ * The two sentinels are filtered out of any real picked answer; for multi-select
161
+ * the modal returns comma-joined values, so Write/Skip win over any co-selected
162
+ * options.
163
+ */
164
+ const CHOOSE_WRITE = '__cspeach_write__';
165
+ const CHOOSE_SKIP = '__cspeach_skip__';
166
+ export async function inkAwareChoose(a) {
167
+ const { shouldUseInk, isHeadless } = await import('../renderer/tty.js');
168
+ if (isHeadless())
169
+ return { kind: 'skip' };
170
+ if (!shouldUseInk())
171
+ return { kind: 'write' }; // classic REPL → free-text fallback
172
+ const choices = [
173
+ ...a.options.map((o) => ({ value: o, label: o })),
174
+ { value: CHOOSE_WRITE, label: '✏️ Write my own…' },
175
+ { value: CHOOSE_SKIP, label: '↷ Skip (leave for the viewer)' },
176
+ ];
177
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
178
+ const res = await askQuestionEmitter.request({
179
+ id: 'answer-blocker',
180
+ question: `Blocker ${a.index + 1}/${a.total}: ${a.question}`,
181
+ context: a.multiSelect ? 'Space to toggle one or more · Enter to submit' : undefined,
182
+ kind: a.multiSelect ? 'multi' : 'choice',
183
+ choices,
184
+ });
185
+ if (res.cancelled || res.answer == null)
186
+ return { kind: 'skip' };
187
+ const values = res.answer.split(',');
188
+ if (values.includes(CHOOSE_WRITE))
189
+ return { kind: 'write' };
190
+ if (values.includes(CHOOSE_SKIP))
191
+ return { kind: 'skip' };
192
+ const real = values.filter((v) => v !== CHOOSE_WRITE && v !== CHOOSE_SKIP);
193
+ if (real.length === 0)
194
+ return { kind: 'skip' };
195
+ return { kind: 'picked', answer: real.join(', ') };
196
+ }
137
197
  export async function maybeOfferSave(p) {
138
- if (!SAVE_HOOK_SKILLS.has(p.skill))
139
- return;
140
198
  if (p.assistantText.trim().length === 0)
141
199
  return;
200
+ // C1 (2026-06-11, defect D22) — the hook fires on artifact EVIDENCE, not
201
+ // only on the routed-skill label. The seed-session plan save was lost
202
+ // because the turn was misrouted (label 'abap-rap', not in the set) while
203
+ // the output carried a complete csforge:plan-manifest block. The label
204
+ // stays the primary trigger — it is authoritative for prose-extracted
205
+ // artefacts with no inline marker (spec-gap / design / estimate) and keeps
206
+ // merge-skill provenance exact (abap-cca-merge saves as abap-cca-merge).
207
+ // When the label would NOT fire, a recognizable manifest block in the
208
+ // output resolves the registry skill instead (detectArtifactSkill), so a
209
+ // misroute can no longer silently lose the artifact.
210
+ let saveSkill = p.skill;
211
+ if (!SAVE_HOOK_SKILLS.has(p.skill)) {
212
+ const detected = detectArtifactSkill(p.assistantText);
213
+ if (!detected)
214
+ return;
215
+ saveSkill = detected;
216
+ p.emit(chalk.dim(`[save] ${detected} artifact manifest detected in this turn's output (turn was labelled '${p.skill}') — offering save.`));
217
+ }
218
+ // B5 (2026-06-11) — defect D2: in headless runs the artifact IS the point
219
+ // of the run, so the "Save as project file? [y/N]" prompt defaults YES
220
+ // instead of blocking on stdin that will never answer. The auto-answer is
221
+ // logged so the transcript stays honest about who said yes.
222
+ //
223
+ // CONTRACT: this headless prompt blanket-answers 'y' to ANY question it
224
+ // is asked, so runSaveCommand must ask at most ONE question (the save
225
+ // confirm). See the matching note at the prompt site in
226
+ // src/projects/save-command.ts before adding further prompts there.
227
+ const { isHeadless } = await import('../renderer/tty.js');
228
+ const prompt = isHeadless()
229
+ ? async (q) => {
230
+ p.emit(chalk.dim(`headless: auto-answered '${q.trim()}' → 'y' (artifact save defaults YES)`));
231
+ return 'y';
232
+ }
233
+ : inkAwarePrompt;
142
234
  try {
143
- await runSaveCommand({
235
+ const savedPath = await runSaveCommand({
144
236
  skillOutput: p.assistantText,
145
- skillName: p.skill,
237
+ skillName: saveSkill,
146
238
  skillVersion: '1.0',
147
239
  skillInput: p.userMessage,
148
240
  tokensUsed: p.tokensUsed,
149
241
  model: p.model,
150
242
  author: getAuthorIdentity(),
151
243
  cwd: process.cwd(),
152
- prompt: inkAwarePrompt,
244
+ prompt,
153
245
  log: (...lines) => lines.forEach((l) => p.emit(l)),
154
246
  promotedFrom: p.promotedFrom ?? null,
155
247
  });
248
+ // After a spec-gap file is saved, let a developer answer the open BLOCKER
249
+ // questions inline (recorded as `answered` items in the SAME file) — the
250
+ // business questions stay open for a consultant to answer in the viewer.
251
+ // Uses inkAwarePrompt directly (NOT the save hook's blanket-'y' headless
252
+ // prompt): in headless it returns '' → declined, so nothing is written.
253
+ if (savedPath && saveSkill === 'abap-spec-gap') {
254
+ await offerAnswerBlockers({
255
+ path: savedPath,
256
+ identity: getAuthorIdentity(),
257
+ prompt: inkAwarePrompt,
258
+ choose: inkAwareChoose,
259
+ log: (...lines) => lines.forEach((l) => p.emit(l)),
260
+ });
261
+ }
156
262
  }
157
263
  catch (err) {
158
264
  p.emit(chalk.yellow(`\n[save] skipped: ${err?.message ?? err}`));
@@ -182,7 +288,17 @@ export async function runTurn(params) {
182
288
  let userMessageForLLM = params.userMessage;
183
289
  let userMessageForSave = params.userMessage;
184
290
  {
185
- const { fromPath, rest } = parseFromFlag(params.userMessage);
291
+ // Forgiveness: `--from @<document>` (a .docx/.pdf/.txt/.md, not a saved
292
+ // .cspeach.json) means "attach this file", not "chain a result". Rewrite it
293
+ // to a plain @<file> attachment and let Phase D ingest it — never error out
294
+ // and cancel the turn on this very natural slip.
295
+ const docAttach = coerceDocFromFlagToAttachment(params.userMessage);
296
+ if (docAttach) {
297
+ emit(chalk.dim(`[--from] ${docAttach.filename} is a document, not a saved result — attaching it as input instead.`));
298
+ userMessageForLLM = docAttach.rewritten;
299
+ userMessageForSave = docAttach.rewritten;
300
+ }
301
+ const { fromPath, rest } = parseFromFlag(docAttach ? '' : params.userMessage);
186
302
  if (fromPath) {
187
303
  const result = await runPromoteCommand({
188
304
  sourcePath: fromPath,
@@ -331,6 +447,10 @@ export async function runTurn(params) {
331
447
  // already handed to onPlanManifest, so we fire once per DISTINCT manifest
332
448
  // (a continue-in-session run emits one per phase) and not once per round.
333
449
  let lastFiredManifest = null;
450
+ // Bounded auto-continue across max_tokens truncations within one turn, so a
451
+ // long phase (e.g. /abap-plan c2.behavior) that overruns the output cap
452
+ // mid-response finishes instead of dying with "unexpected stop reason".
453
+ let maxTokenContinuations = 0;
334
454
  // (emit was hoisted to the top of the function in v0.5 so the Phase J
335
455
  // promote dispatch could share it. Original location was here.)
336
456
  while (true) {
@@ -400,7 +520,11 @@ export async function runTurn(params) {
400
520
  // makes the 400 categorically impossible. Idempotent and O(messages).
401
521
  healSessionMessagesInPlace(session.messages);
402
522
  const streamParams = {
403
- model: cfg.default_model,
523
+ // A4 — per-turn override (plan model tiering) wins over the
524
+ // configured default; absent on every non-plan turn. model-governance
525
+ // step 2d — the session-default model now resolves env > local >
526
+ // server > built-in (byte-identical to cfg.default_model when nothing set).
527
+ model: params.modelOverride ?? resolveModelRole('session_default', cfg),
404
528
  // v0.3.1 — was 8192. Bumped because code-gen skills (abap-generate
405
529
  // etc.) kept running out of budget mid-turn: adaptive thinking +
406
530
  // multiple tool-result prompts + TL;DR mandate + summary prose
@@ -460,6 +584,7 @@ export async function runTurn(params) {
460
584
  let currentAssistantContent = [];
461
585
  let sawEndTurn = false;
462
586
  let sawToolUse = false;
587
+ let sawMaxTokens = false;
463
588
  // M10 — ensure session.usage is present (shipped session schema may omit it
464
589
  // for older saved sessions; we default to zero on first turn).
465
590
  if (!session.usage) {
@@ -583,6 +708,8 @@ export async function runTurn(params) {
583
708
  sawEndTurn = true;
584
709
  if (stop_reason === 'tool_use')
585
710
  sawToolUse = true;
711
+ if (stop_reason === 'max_tokens' || stop_reason === 'length')
712
+ sawMaxTokens = true;
586
713
  // H2 — `deltaUsage.output_tokens` is the cumulative running total for
587
714
  // THIS message, NOT a per-event delta. Compute the increment vs the
588
715
  // last-seen cumulative and add only that. Guard against `null` /
@@ -640,7 +767,8 @@ export async function runTurn(params) {
640
767
  // never let a save error crash the turn.
641
768
  if (params.onPlanManifest) {
642
769
  const turnText = collectTurnAssistantText(session.messages, turnStartMessageCount);
643
- const blocks = turnText.match(/<!--\s*csforge:plan-manifest\s*\n[\s\S]*?\n\s*-->/g);
770
+ // E2 dual-read: accept both the legacy `csforge:` and current `cspeach:` prefixes.
771
+ const blocks = turnText.match(/<!--\s*(?:csforge|cspeach):plan-manifest\s*\n[\s\S]*?\n\s*-->/g);
644
772
  const latest = blocks ? blocks[blocks.length - 1] : null;
645
773
  if (latest && latest !== lastFiredManifest) {
646
774
  lastFiredManifest = latest;
@@ -675,7 +803,10 @@ export async function runTurn(params) {
675
803
  const turnDurationMs = Date.now() - turnStartedAt;
676
804
  const entry = buildCostEntry({
677
805
  turn: turnNumber,
678
- model: session.model,
806
+ // A4 — when a per-turn override ran the stream on a different model
807
+ // (plan model tiering), the cost line must record the ACTUAL model
808
+ // or the Sonnet-priced turn would be billed at session-model rates.
809
+ model: params.modelOverride ?? session.model,
679
810
  tokens: turnTokens,
680
811
  duration_ms: turnDurationMs,
681
812
  });
@@ -742,7 +873,10 @@ export async function runTurn(params) {
742
873
  assistantText,
743
874
  userMessage: userMessageForSave,
744
875
  tokensUsed: Math.max(0, (session.usage?.output_tokens ?? 0) - turnStartOutputTokens),
745
- model: cfg.default_model ?? session.model ?? 'unknown',
876
+ // A4 — record the actual turn model if an override ran the stream
877
+ // (unreachable today: plan-resume turns suppress this hook, and
878
+ // they are the only modelOverride caller — kept honest anyway).
879
+ model: params.modelOverride ?? resolveModelRole('session_default', cfg) ?? session.model ?? 'unknown',
746
880
  emit,
747
881
  promotedFrom: promotedFromForSave,
748
882
  });
@@ -755,14 +889,17 @@ export async function runTurn(params) {
755
889
  // throttle live in config.compact — see config/loader.ts:CompactConfig.
756
890
  try {
757
891
  const { maybeAutoCompact } = await import('../commands/auto-compact.js');
758
- const { buildSummarisationPrompt, serialiseForSummariser, COMPACTION_MODEL } = await import('../commands/compact.js');
892
+ const { buildSummarisationPrompt, serialiseForSummariser } = await import('../commands/compact.js');
759
893
  const { summariseViaProvider } = await import('./summarise-via-provider.js');
760
894
  await maybeAutoCompact({
761
895
  session,
762
896
  config: cfg.compact,
763
897
  turnNumber,
764
898
  emit,
765
- summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise), { model: COMPACTION_MODEL, maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact' } }),
899
+ summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise),
900
+ // model-governance step 2d — compact model resolves env > local >
901
+ // server > built-in (== COMPACTION_MODEL when nothing set).
902
+ { model: resolveModelRole('compact', cfg), maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact' } }),
766
903
  });
767
904
  }
768
905
  catch (err) {
@@ -774,7 +911,29 @@ export async function runTurn(params) {
774
911
  return;
775
912
  }
776
913
  if (!sawToolUse) {
777
- // Unexpected stop reason — bail to avoid infinite loop.
914
+ // The response was cut off at the output-token cap (stop_reason
915
+ // 'max_tokens'/'length') mid-generation — NOT a real end of turn. The
916
+ // truncated assistant message is already persisted (line ~733), so
917
+ // continue the turn and let the model finish (and, for /abap-plan, still
918
+ // emit its manifest). Bounded so a pathologically long response can't
919
+ // loop forever.
920
+ const MAX_TOKEN_CONTINUATIONS = 4;
921
+ if (sawMaxTokens && maxTokenContinuations < MAX_TOKEN_CONTINUATIONS) {
922
+ maxTokenContinuations += 1;
923
+ emit(chalk.yellow(`\n[output limit reached — auto-continuing (${maxTokenContinuations}/${MAX_TOKEN_CONTINUATIONS})]`));
924
+ session.messages.push({
925
+ role: 'user',
926
+ content: 'Your previous response was cut off at the output-token limit. Continue exactly where you left off — do NOT repeat what you already wrote. '
927
+ + 'If you were mid-way through a tool call, or (for /abap-plan) had not yet emitted the closing csforge:plan-manifest block, complete it now. '
928
+ + 'Keep narration brief to stay within the limit.',
929
+ });
930
+ continue;
931
+ }
932
+ if (sawMaxTokens) {
933
+ emit(chalk.yellow(`\n[output limit hit ${maxTokenContinuations}× this turn — stopping. Type "continue" to resume.]`));
934
+ return;
935
+ }
936
+ // Truly unexpected stop reason — bail to avoid an infinite loop.
778
937
  emit(chalk.yellow('\n[unexpected stop reason — ending turn]'));
779
938
  return;
780
939
  }
@@ -791,6 +950,8 @@ export async function runTurn(params) {
791
950
  for (const block of currentAssistantContent) {
792
951
  if (block.type === 'tool_use') {
793
952
  // Phase 1: print the dispatch line (CC-style: ⏺ name(args)).
953
+ // No-op for widget-suppressed tools (ask_question) — the form/modal
954
+ // is the visible representation; see tool-widget.ts.
794
955
  renderToolCallTop({
795
956
  name: block.name,
796
957
  args: (block.input ?? {}),
@@ -799,8 +960,12 @@ export async function runTurn(params) {
799
960
  // Phase 2a: start the peach-themed spinner. No-op outside TTY / Ink
800
961
  // / CI — the result line still prints, just without the in-place
801
962
  // animation. Spinner is purely a "still working" cue, not
802
- // load-bearing for output.
803
- const spinner = startToolSpinner({ chunkEmitter: params.chunkEmitter });
963
+ // load-bearing for output. Skipped entirely for widget-suppressed
964
+ // tools (ask_question): with the ⏺ line gone the spinner would
965
+ // orphan on its own row above the form.
966
+ const spinner = isWidgetSuppressedTool(block.name)
967
+ ? { stop: () => undefined }
968
+ : startToolSpinner({ chunkEmitter: params.chunkEmitter });
804
969
  // 2026-06-06 (turn-liveness, B5 smoke feedback) — the in-place tool
805
970
  // spinner self-disables under Ink (phantom cursor #25), which left
806
971
  // slow tool calls (SAP over VPN: 10-60s) as DEAD AIR between the ⏺
@@ -809,7 +974,27 @@ export async function runTurn(params) {
809
974
  // every terminal, Ink included.
810
975
  // Interactive tools wait on the USER, not the system — ticking
811
976
  // "ask_question running… (30s)" while they think is noise.
812
- const isInteractiveTool = block.name === 'ask_question' || block.name === 'request_approval';
977
+ // CRITICAL race fix (2026-06-13) — a mutating tool that will trip the
978
+ // Rule 8 batch gate (2nd+ write of the turn) ALSO blocks on the user:
979
+ // presentSafetyConfirmation opens an Ink modal from inside dispatchTool
980
+ // BEFORE the op runs. file_write / shell_exec are otherwise classified
981
+ // non-interactive, so without this the loop would start the per-second
982
+ // heartbeat + keep the turn-status row ticking UNDER the modal — the
983
+ // observed live bug (doubled card, lost Enter, history-replay leak,
984
+ // ~10-min wedge with the tool spinner ticking under the modal). Detect
985
+ // the gate the SAME way tool-dispatch does (tool.isMutating &&
986
+ // shouldGateRule8) and treat it as user-blocking: pause the status row,
987
+ // skip the heartbeat. The gate's own clearActiveSpinner +
988
+ // turnStatusEmitter.pause (safety-confirm.ts) is the inner belt; this
989
+ // is the outer one — together no live render source contends with the
990
+ // modal for the Ink frame or raw-mode stdin.
991
+ const willTripBatchGate = (() => {
992
+ const t = getTool(block.name);
993
+ return !!t?.isMutating && shouldGateRule8();
994
+ })();
995
+ const isInteractiveTool = block.name === 'ask_question' ||
996
+ block.name === 'request_approval' ||
997
+ willTripBatchGate;
813
998
  // Interactive tools block on the USER. PAUSE the turn-status row (don't
814
999
  // just relabel it): a live 250ms tick repaints the dynamic frame and
815
1000
  // overdraws the inquirer approval picker / churns the Ink ask_question
@@ -828,6 +1013,11 @@ export async function runTurn(params) {
828
1013
  });
829
1014
  // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
830
1015
  const dispatchStart = Date.now();
1016
+ // D19 (2026-06-11) — expose the LLM tool_use id to the handler so the
1017
+ // write-tool WAL (appendPending/finalizeToolCall) is keyed by the SAME
1018
+ // id the loop records below. Dispatch is sequential, so a single slot
1019
+ // on the shared ctx is safe.
1020
+ params.ctx.toolUseId = block.id;
831
1021
  // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
832
1022
  const guardDecision = checkAndMark(block.name, writeGuard);
833
1023
  let result;
@@ -837,6 +1027,9 @@ export async function runTurn(params) {
837
1027
  : { content: guardDecision.errorContent, is_error: true };
838
1028
  }
839
1029
  finally {
1030
+ // D19: clear the slot so a future non-loop invocation on this ctx
1031
+ // can't inherit a stale block id.
1032
+ params.ctx.toolUseId = undefined;
840
1033
  // Stop on the error path too — the interval is unref'd but would
841
1034
  // otherwise keep printing "<tool> running…" into the NEXT prompt
842
1035
  // after a dispatch throw.
@@ -864,16 +1057,25 @@ export async function runTurn(params) {
864
1057
  isError: result.is_error ?? false,
865
1058
  resultSummary,
866
1059
  chunkEmitter: params.chunkEmitter,
1060
+ // D29 (2026-06-12): self-identifying result row. Heartbeat lines,
1061
+ // sap-client warns, and notice lines legitimately print between
1062
+ // the ⏺ top line and this row — without the name here those rows
1063
+ // read as anonymous `⎿ ✓ 364ms` orphans in the transcript.
1064
+ name: block.name,
1065
+ args: (block.input ?? {}),
867
1066
  });
868
1067
  // v0.3 (Step 28.0): populate session.toolCalls so the SessionTimeline
869
1068
  // overlay can render per-write history. Cap per-entry payload size at
870
1069
  // 4kB to bound session.json growth.
1070
+ // D19 (2026-06-11): upsert, not push — write handlers journal the same
1071
+ // call through the WAL under this block.id; pushing unconditionally
1072
+ // produced two ledger entries per approval-gated write.
871
1073
  const completedAt = new Date().toISOString();
872
1074
  {
873
1075
  const resultText = typeof result.content === 'string'
874
1076
  ? result.content
875
1077
  : JSON.stringify(result.content);
876
- session.toolCalls.push({
1078
+ recordCompletedToolCall(session, {
877
1079
  tool_use_id: block.id,
878
1080
  tool: block.name,
879
1081
  args: block.input ?? {},
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Thrown when a non-managed LLM mode is used without a valid CSPeach login.
3
+ * Carries a user-facing message; callers print it and exit rather than dumping
4
+ * a stack trace.
5
+ */
6
+ export class CspeachLicenseError extends Error {
7
+ mode;
8
+ constructor(mode) {
9
+ super(`CSPeach requires a license to run in '${mode}' mode.\n` +
10
+ `In this mode your prompts go straight to your own LLM — but the CSPeach skills\n` +
11
+ `and CLI are licensed software. Run \`cspeach login\` to activate your license.\n` +
12
+ `Need an access key? Email laeeq.siddique@cremencing.com.`);
13
+ this.name = 'CspeachLicenseError';
14
+ this.mode = mode;
15
+ }
16
+ }
17
+ /**
18
+ * Gate every NON-managed mode (byok / local / ai-hub) on a valid CSPeach login.
19
+ *
20
+ * Why: managed mode authenticates against the proxy on every call, so it is
21
+ * already gated. The other modes bypass our proxy for inference — and the
22
+ * skills (our IP) are bundled into the package — so without this check a fresh
23
+ * `npm install` is fully usable by anyone who never logged in, untracked. A
24
+ * CSPeach key is admin-issued (no open self-signup), so requiring one means we
25
+ * know who is using the product and can revoke access.
26
+ *
27
+ * Scope (Part A): require a CSPeach key to be PRESENT. Server-side validation +
28
+ * revocation (so a forged key fails) is the Part-B follow-up — see
29
+ * docs/byok-portal-handover.md §0.
30
+ */
31
+ export async function assertModeLicensed(mode, getBearer) {
32
+ if (mode === 'managed')
33
+ return;
34
+ let key = '';
35
+ try {
36
+ key = (await getBearer()) ?? '';
37
+ }
38
+ catch {
39
+ key = '';
40
+ }
41
+ if (key.trim().length === 0) {
42
+ throw new CspeachLicenseError(mode);
43
+ }
44
+ }
@@ -7,7 +7,7 @@
7
7
  * with full conversation context.
8
8
  *
9
9
  * What Layer 1 does NOT cover: skill-internal state files. /abap-cca, for
10
- * example, maintains `.abapforge/cca/projects/<id>/project.json` with an
10
+ * example, maintains `.cspeach/cca/projects/<id>/project.json` with an
11
11
  * inventory of objects, classifications, and per-package state. The skill
12
12
  * writes this file at PHASE BOUNDARIES (end of DISCOVER, end of INVENTORY,
13
13
  * etc.). A crash mid-DISCOVER means the partial inventory in project.json
@@ -23,6 +23,21 @@ export async function dispatchTool(name, args, ctx) {
23
23
  batch_count: getWriteOpsThisTurn().length + 1,
24
24
  });
25
25
  if (!r.confirmed) {
26
+ // B5 — discriminate WHO declined. A headless fail-fast (nobody ever
27
+ // saw the card) must not masquerade as a user decision: the model
28
+ // (and any transcript reader) reacts differently to "the user said
29
+ // no" vs "no user was available to say yes". Uniform `headless: true`
30
+ // marker matches the other headless result payloads.
31
+ if (r.headless) {
32
+ return {
33
+ content: JSON.stringify({
34
+ error: 'headless_safety_decline',
35
+ headless: true,
36
+ reason: r.reason,
37
+ }),
38
+ is_error: true,
39
+ };
40
+ }
26
41
  return {
27
42
  content: JSON.stringify({ error: 'cancelled_by_user', reason: r.reason }),
28
43
  is_error: true,
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Canonical approval-object strings (Task A3 — defects D17/D21/D28).
3
+ *
4
+ * The battery exposed an APPROVAL_INVALID:object_mismatch epidemic (~8 wasted
5
+ * approval round-trips): request_approval minted whatever free-text object
6
+ * string the model wrote in the change row, while every write tool validated
7
+ * against a structured argument (args.name / args.className / args.description
8
+ * / objects[0].name). Any decoration the model added at mint time — and models
9
+ * reliably add decoration — broke the byte-equality check in
10
+ * verifyAndSpendApprovalId.
11
+ *
12
+ * Observed mismatch classes (exact strings from the battery sessions):
13
+ * D17 "ZBP_I_DOWNTIMELOG (CCIMP / lhc_DowntimeLog.validateEndAfterStart)"
14
+ * vs "ZBP_I_DOWNTIMELOG" → trailing parenthetical
15
+ * D21 "Transport: Magic fix" vs "Magic fix" → leading transport label
16
+ * D28 "ZR_PM_MAINTREQ,ZBP_R_PM_MAINTREQ,…" vs "ZR_PM_MAINTREQ"
17
+ * → one batch activate approval, per-object activate calls
18
+ *
19
+ * The fix is ONE code path for both sides:
20
+ * - mint side (jwt.ts#mintChangeApproval, used by request_approval) stores
21
+ * canonicalApprovalObject(change.object) in the JWT;
22
+ * - spend side (jwt.ts#verifyAndSpendApprovalId) compares via
23
+ * approvalObjectMatches, which canonicalizes BOTH the JWT payload object
24
+ * and the tool's expected object before comparing.
25
+ * The spend side canonicalizes too because (a) the expected side is always the
26
+ * raw structured tool argument, never pre-canonicalized, and (b) it is
27
+ * defense-in-depth against any future mint site that bypasses
28
+ * mintChangeApproval and stores a raw string.
29
+ */
30
+ /** Leading transport-request labels the model prepends to CTS descriptions. */
31
+ const LEADING_LABEL = /^(?:transport(?:\s+request)?|tr|cts|request)\s*:\s*/i;
32
+ /** A trailing parenthetical decoration, e.g. "(testclasses include)". */
33
+ const TRAILING_PAREN = /\s*\([^()]*\)\s*$/;
34
+ /** What an ABAP repository object name looks like once canonicalized. */
35
+ const OBJECT_NAME = /^[A-Z0-9_/]+$/;
36
+ /**
37
+ * Normalize an approval object string so mint and spend sides can never
38
+ * disagree on decoration, case, or whitespace:
39
+ *
40
+ * 1. trim
41
+ * 2. strip a leading "Transport:" / "TR:" / "Request:" label (D21)
42
+ * 3. strip trailing parenthetical decorations, stacked or single (D17)
43
+ * 4. collapse internal whitespace
44
+ * 5. uppercase
45
+ *
46
+ * All steps are applied symmetrically to both sides. Note that step 3 strips
47
+ * trailing parentheticals from free text too: transport descriptions
48
+ * "Fix dump (urgent)" and "Fix dump (rollback)" both canonicalize to
49
+ * "FIX DUMP" and therefore cross-match. This collision class is accepted
50
+ * because descriptions are display labels, not object identities, and every
51
+ * approval is session-scoped, TTL-bounded, and user-confirmed — the user saw
52
+ * the specific change row that minted the JWT.
53
+ */
54
+ export function canonicalApprovalObject(raw) {
55
+ let s = String(raw ?? '').trim();
56
+ s = s.replace(LEADING_LABEL, '');
57
+ for (let prev = ''; prev !== s;) {
58
+ prev = s;
59
+ s = s.replace(TRAILING_PAREN, '');
60
+ }
61
+ return s.replace(/\s+/g, ' ').trim().toUpperCase();
62
+ }
63
+ /**
64
+ * Spend-side comparison: does the object string stored in the approval JWT
65
+ * authorize an operation on `expected` (the structured argument the write
66
+ * tool validates against)? Returns the match kind, or `false` for no match.
67
+ *
68
+ * - byte equality ('exact'), OR
69
+ * - canonical equality ('canonical'), OR
70
+ * - the minted string is a comma-joined list of ABAP object names and
71
+ * `expected` is one of them ('list_member', D28 — batch activate approval
72
+ * spent by per-object sap_activate calls).
73
+ *
74
+ * The list rule only applies when EVERY part looks like an object name
75
+ * (no spaces / free text), so a transport description containing commas can
76
+ * never partially match.
77
+ */
78
+ export function approvalObjectMatches(minted, expected) {
79
+ if (minted === expected)
80
+ return 'exact';
81
+ const m = canonicalApprovalObject(minted);
82
+ const e = canonicalApprovalObject(expected);
83
+ if (m === e)
84
+ return 'canonical';
85
+ const parts = m.split(',').map((p) => p.trim()).filter((p) => p.length > 0);
86
+ if (parts.length < 2)
87
+ return false;
88
+ if (!parts.every((p) => OBJECT_NAME.test(p)))
89
+ return false;
90
+ return parts.includes(e) ? 'list_member' : false;
91
+ }