@cspeach/cli 0.6.5 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +26 -0
  2. package/dist/agent/intent-system-prompt.js +46 -0
  3. package/dist/agent/loop.js +180 -18
  4. package/dist/agent/parallel-write-guard.js +71 -0
  5. package/dist/agent/repair-partial.js +88 -1
  6. package/dist/agent/retry-cap.js +121 -0
  7. package/dist/agent/summarise-via-provider.js +51 -0
  8. package/dist/approvals/jwt.js +33 -12
  9. package/dist/commands/auto-compact.js +94 -0
  10. package/dist/commands/compact.js +265 -0
  11. package/dist/commands/cost.js +56 -0
  12. package/dist/commands/help.js +21 -14
  13. package/dist/config/loader.js +39 -0
  14. package/dist/cost/cost-log.js +113 -0
  15. package/dist/cost/pricing.js +91 -0
  16. package/dist/one-shot.js +1 -1
  17. package/dist/renderer/markdown.js +7 -1
  18. package/dist/renderer/status-footer.js +37 -14
  19. package/dist/renderer/thinking-heartbeat.js +15 -3
  20. package/dist/renderer/tool-widget.js +6 -1
  21. package/dist/renderer/tty.js +27 -0
  22. package/dist/repl/cspeach-shell-detect.js +33 -0
  23. package/dist/repl/post-turn-status.js +68 -0
  24. package/dist/repl/slash-completer.js +2 -0
  25. package/dist/repl/slash-picker.js +2 -0
  26. package/dist/repl.js +371 -46
  27. package/dist/sap-errors/parse-adt-exception.js +149 -0
  28. package/dist/session/schema.js +2 -1
  29. package/dist/session/store.js +19 -0
  30. package/dist/tools/_filesystem-shared.js +9 -1
  31. package/dist/tools/ask-question.js +19 -0
  32. package/dist/tools/sap-read.js +56 -4
  33. package/dist/tools/sap-write.js +19 -0
  34. package/dist/tools/subagent/schedule_draft_create.js +137 -0
  35. package/dist/ui/alt-screen.js +120 -0
  36. package/dist/ui/app.js +47 -15
  37. package/dist/ui/ask-question-emitter.js +13 -0
  38. package/dist/ui/body.js +39 -23
  39. package/dist/ui/coaching-picker-classic.js +8 -1
  40. package/dist/ui/footer.js +10 -2
  41. package/dist/ui/header.js +19 -4
  42. package/dist/ui/sap-state-store.js +13 -1
  43. package/dist/ui/sidebar.js +5 -3
  44. package/dist/ui/status-emitter.js +19 -0
  45. package/dist/ui/status-row.js +30 -4
  46. package/dist/ui/widgets/ask-question-modal.js +132 -0
  47. package/dist/ui/widgets/coaching-picker.js +55 -8
  48. package/package.json +1 -1
package/README.md CHANGED
@@ -60,6 +60,32 @@ machine.
60
60
  Ten Forge Rules are non-negotiable. Full philosophy in the `CSPeach Principles`
61
61
  preamble shown to the model every turn.
62
62
 
63
+ ## Cost transparency (since 0.7.0)
64
+
65
+ Every turn shows what it cost. After each model call you see:
66
+
67
+ ```
68
+ session abc123 │ S4H │ /abap-explain │ 2 turns │ 487 tok │ $0.07 │ session: $0.18 │ 3.6s
69
+ ```
70
+
71
+ - `tok` = tokens billed this turn (input + output + cache)
72
+ - `$` = this turn's cost
73
+ - `session:` = cumulative across the open session
74
+ - `/cost` shows a per-skill / per-tool breakdown
75
+ - `/compact` summarises older turns when context gets large (cuts subsequent-turn cost ~80-90%)
76
+ - `auto_compact` in `~/.cspeach/config.toml` fires `/compact` automatically over a configurable threshold
77
+
78
+ On `/exit` you get a session summary + the `cspeach --resume <id>` command in plain text so you can pick up where you left off.
79
+
80
+ ## UI modes
81
+
82
+ Two REPL modes, both supported. **Default is `classic`.** Switch with `/ui ink` or `/ui classic`; saves to `~/.cspeach/config.toml` under `[ui]`.
83
+
84
+ - **`classic`** — chalk + readline. The supported, demo-tested mode. Use this unless you have a reason not to. Native terminal scroll, mouse wheel, search — everything your terminal does naturally just works.
85
+ - **`ink`** — React-based TUI. **PREVIEW.** Some rendering quirks remain (occasional phantom cursor, picker fossil in scrollback). Functional for daily use but not yet polished enough to recommend as default. Promoted to default in a later release.
86
+
87
+ If `ink` is misbehaving for you, run `/ui classic` once — change applies after relaunch.
88
+
63
89
  ## Configuration
64
90
 
65
91
  `~/.cspeach/config.toml` is created on first `cspeach config add-sap`:
@@ -0,0 +1,46 @@
1
+ /**
2
+ * Phase 4 — neutral system prompt for intent-driven agent loop.
3
+ *
4
+ * Intent-mode is the user-states-intent / model-composes-plan dispatch
5
+ * pattern (parallel to skill-mode where a /skill scaffolds the flow).
6
+ * The prompt is intentionally short — the playbook library + tool
7
+ * descriptions carry domain knowledge; the prompt only frames role +
8
+ * safety + how to fetch playbooks.
9
+ *
10
+ * Project-context block (if available) is appended by runTurnIntent
11
+ * via the existing `maybeBuildProjectContext()` path; it is NOT part
12
+ * of this constant.
13
+ */
14
+ export function getIntentSystemPrompt() {
15
+ return `You are CSPeach, an AI development agent for SAP / ABAP work.
16
+
17
+ You help senior SAP consultants and ABAP developers plan, build, modernize, migrate, test, review, and ship SAP solutions. You have:
18
+
19
+ - Live access to the user's SAP system via the sap_* tool family (~30 tools — get_source, set_source, atc_run, transport_create, sql_query, snapshot_take, etc.).
20
+ - Filesystem + shell + web access to the project the user is working in (file_read, file_edit, file_write, glob, grep, shell_exec, web_fetch, web_search).
21
+ - Subagent dispatch (agent_run) for focused sub-tasks; background_run + monitor_emit for non-blocking processes; schedule_draft_create for persisting schedule envelopes (no runner yet — see tool description).
22
+ - A library of 34 SAP-development playbooks you can read on demand via playbook_get. These cover RAP scaffolding, S/4HANA upgrade remediation, CCA estate analysis, modernization, code review, performance triage, and more. When you need expertise on a specific workflow, fetch the relevant playbook BEFORE acting.
23
+ - The project context for this session (CDS views, tables, packages, conventions) — appended below if available.
24
+
25
+ When the user states an intent, your job is:
26
+
27
+ 1. **Confirm the intent** — repeat what they want in one sentence; ask clarifying questions only if essential.
28
+ 2. **Pick the right approach** — fetch playbooks if relevant via playbook_get; otherwise compose tools from first principles.
29
+ 3. **Execute** — use tools as needed. Snapshot before any SAP write (Forge Rule 7); use sap_update_method for method-only changes (Rule 7a); never overwrite without a transport (Rule 9); verify after every write (Rule 10).
30
+ 4. **Surface artefacts** — for substantial work, save a .cspeach.json envelope per the existing artefact-type convention (cca-assessment, upgrade-progress, spec-gap, etc.).
31
+ 5. **Hand back** — when the work is done OR when human decision is needed.
32
+
33
+ Follow the 10 Forge Rules. They are binding:
34
+ 1. No blind code generation — ask missing questions first
35
+ 2. No code change without impact thinking
36
+ 3. No cloud readiness claims without evidence
37
+ 4. No review without quality criteria
38
+ 5. No done without preflight
39
+ 6. Read only by default
40
+ 7. No write without snapshot
41
+ 8. No batch without plan (any 2+ mutating tools in one turn pauses for explicit approval — Rule 8 is active and you cannot disable it)
42
+ 9. Transport isolation
43
+ 10. Verify after every write
44
+
45
+ Tone: senior consultant talking to peer. No marketing prose. Specific verifiable claims. Plain writing.`;
46
+ }
@@ -7,10 +7,12 @@ import { loadConfig } from '../config/loader.js';
7
7
  import { retryWithBackoff } from './retry.js';
8
8
  import { renderChunk, initialBufferState, render } from '../renderer/pipeline.js';
9
9
  import { renderToolCallTop, renderToolCallBottom, startToolSpinner, startThinkingSpinner } from '../renderer/tool-widget.js';
10
+ import { formatToolErrorSummary } from '../sap-errors/parse-adt-exception.js';
10
11
  import { startThinkingHeartbeat } from '../renderer/thinking-heartbeat.js';
11
12
  // (progress-chatter import removed 2026-05-01 — superseded by CC-style
12
13
  // two-line dispatch renderer; re-add if a future in-place spinner returns)
13
- import { ERR } from '../errors/codes.js';
14
+ import { buildRetryCapPausePayload, buildSkippedSiblingResults } from './retry-cap.js';
15
+ import { appendCostLine, buildEntry as buildCostEntry } from '../cost/cost-log.js';
14
16
  import { lastAssistantIsAwaitingAnswer } from '../session/awaiting-answer.js';
15
17
  import { enrichUserMessage } from '../router/intent-extractor.js';
16
18
  import { resetRule8State } from '../repl/rule8-detector.js';
@@ -21,7 +23,7 @@ import { withInquirer } from '../repl/inquirer-guard.js';
21
23
  import { expandTextFileAttachments } from '../projects/workspace.js';
22
24
  import { getSapSystemInfo, renderSapSystemBlock } from '../sap/system-info.js';
23
25
  import { objectKeyFromInput } from './retry-key.js';
24
- import { repairPartialBlocks } from './repair-partial.js';
26
+ import { repairPartialBlocks, healSessionMessagesInPlace, isPartialJsonApiError } from './repair-partial.js';
25
27
  import { TurnStreamWriter } from './turn-stream.js';
26
28
  import { applyToolResultCheckpoint } from './skill-checkpoint.js';
27
29
  import { loadWatchdogConfig, evaluateWatchdog } from './turn-watchdog.js';
@@ -29,6 +31,7 @@ import { maybeBuildProjectContext } from './maybe-build-project-context.js';
29
31
  import { buildSapConnectionContext } from './sap-connection-adapter.js';
30
32
  import { getCurrentTransport } from '../repl/current-transport.js';
31
33
  import { collectTurnAssistantText } from './turn-assistant-text.js';
34
+ import { checkAndMark, newGuardState } from './parallel-write-guard.js';
32
35
  /**
33
36
  * Author identity for project-file metadata. Reads CSPEACH_AUTHOR_NAME first,
34
37
  * then platform USER/USERNAME, then a generic fallback. Role is fixed to
@@ -99,6 +102,29 @@ export function hasProjectContextBlock(messages) {
99
102
  }
100
103
  return false;
101
104
  }
105
+ /**
106
+ * Critical #3 (2026-05-17) — shared prompt helper that routes through the
107
+ * Ink-native ask_question modal when running under Ink, falls back to
108
+ * inquirer in classic mode. Prior code path called withInquirer() directly
109
+ * which corrupted the Ink frame on every save-offer / --from prompt.
110
+ *
111
+ * Single shape `(q) => Promise<string>` so it slots straight into the
112
+ * existing `prompt:` callback in runSaveCommand + runPromoteCommand.
113
+ */
114
+ async function inkAwarePrompt(q) {
115
+ const { shouldUseInk } = await import('../renderer/tty.js');
116
+ if (shouldUseInk()) {
117
+ const { askQuestionEmitter } = await import('../ui/ask-question-emitter.js');
118
+ const result = await askQuestionEmitter.request({
119
+ id: 'agent-prompt',
120
+ question: q,
121
+ kind: 'text',
122
+ choices: [],
123
+ });
124
+ return result.answer ?? '';
125
+ }
126
+ return withInquirer(() => input({ message: q }));
127
+ }
102
128
  export async function maybeOfferSave(p) {
103
129
  if (!SAVE_HOOK_SKILLS.has(p.skill))
104
130
  return;
@@ -114,7 +140,7 @@ export async function maybeOfferSave(p) {
114
140
  model: p.model,
115
141
  author: getAuthorIdentity(),
116
142
  cwd: process.cwd(),
117
- prompt: async (q) => withInquirer(() => input({ message: q })),
143
+ prompt: inkAwarePrompt,
118
144
  log: (...lines) => lines.forEach((l) => p.emit(l)),
119
145
  promotedFrom: p.promotedFrom ?? null,
120
146
  });
@@ -152,7 +178,7 @@ export async function runTurn(params) {
152
178
  const result = await runPromoteCommand({
153
179
  sourcePath: fromPath,
154
180
  targetSkill: params.skill,
155
- prompt: async (q) => withInquirer(() => input({ message: q })),
181
+ prompt: inkAwarePrompt,
156
182
  log: (...lines) => lines.forEach((l) => emit(l)),
157
183
  });
158
184
  if (!result) {
@@ -263,6 +289,18 @@ export async function runTurn(params) {
263
289
  // missing on older session schemas; the loop initializes it before the
264
290
  // first stream event, so 0 is a safe pre-init baseline.
265
291
  const turnStartOutputTokens = session.usage?.output_tokens ?? 0;
292
+ // Phase 1 cost tracking (2026-05-16) — snapshot ALL four counters at
293
+ // turn start so we can compute a per-turn $ cost at turn end. The
294
+ // existing turnStartOutputTokens stays for the watchdog (it only cares
295
+ // about output tokens), but the cost calculator needs the full set.
296
+ const turnStartTokens = {
297
+ input: session.usage?.input_tokens ?? 0,
298
+ output: session.usage?.output_tokens ?? 0,
299
+ cacheRead: session.usage?.cache_read_input_tokens ?? 0,
300
+ cacheCreate: session.usage?.cache_creation_input_tokens ?? 0,
301
+ };
302
+ // Turn number for the cost log — count of prior user messages + this one.
303
+ const turnNumber = session.messages.filter((m) => m.role === 'user').length + 1;
266
304
  // Snapshot session.messages.length at turn start so the save hook can
267
305
  // walk every assistant message added during this turn — not just the
268
306
  // final one. Skills like /abap-cca emit their manifest in an early
@@ -299,7 +337,11 @@ export async function runTurn(params) {
299
337
  // reports isTTY=false; resume mode also lost spinner visibility) so a
300
338
  // 45-second wait looked like a hang. The heartbeat prints fresh lines
301
339
  // at 3s/10s/30s/60s/90s/2m/3m/5m thresholds — works on any terminal.
302
- const thinkingHeartbeat = startThinkingHeartbeat();
340
+ // Phase A #4 (2026-05-16) — pass chunkEmitter so under Ink the
341
+ // heartbeat lines land in the Body region (via React) instead of
342
+ // being raw-written to stdout (which Ink overdraws on its next
343
+ // render, briefly flashing the line then making it disappear).
344
+ const thinkingHeartbeat = startThinkingHeartbeat({ chunkEmitter: params.chunkEmitter });
303
345
  try {
304
346
  stream = await retryWithBackoff(() => (async () => {
305
347
  const tools = toAnthropicTools(listTools());
@@ -333,6 +375,23 @@ export async function runTurn(params) {
333
375
  // it spinning while an error message is emitted looks broken.
334
376
  thinkingSpinner.stop();
335
377
  thinkingHeartbeat.stop();
378
+ // Bug 11a (2026-05-15) — in-process heal for partial_json poisoning.
379
+ // When a prior turn persisted a tool_use block carrying both `input`
380
+ // AND a leftover `partial_json` field, the next createStream call
381
+ // fails with HTTP 400:
382
+ // messages.N.content.M.tool_use.partial_json:
383
+ // Extra inputs are not permitted
384
+ // The on-disk session was healed by loadSession's strip pass, but the
385
+ // in-memory session.messages array we just sent is what poisoned the
386
+ // call. Scrub it in place, save, tell the user the session was healed,
387
+ // and return cleanly so the REPL prompt comes back — they retype the
388
+ // last prompt without having to /exit + --resume.
389
+ if (isPartialJsonApiError(err)) {
390
+ const scrubbed = healSessionMessagesInPlace(session.messages);
391
+ await saveSession(session);
392
+ emit(chalk.yellow(`\n⚠ Session was poisoned by a partial-json tool_use block; healed in-memory${scrubbed > 0 ? ` (${scrubbed} block${scrubbed === 1 ? '' : 's'} cleaned)` : ''}. Please retry your last message.`));
393
+ return;
394
+ }
336
395
  // Proxy returns 409 upgrade_required when CLI is older than skill's min_cli_version.
337
396
  const status = err?.status ?? err?.response?.status;
338
397
  const bodyRaw = err?.error ?? err?.response?.data ?? err?.body;
@@ -350,7 +409,12 @@ export async function runTurn(params) {
350
409
  // M10 — ensure session.usage is present (shipped session schema may omit it
351
410
  // for older saved sessions; we default to zero on first turn).
352
411
  if (!session.usage) {
353
- session.usage = { input_tokens: 0, output_tokens: 0, cache_read_input_tokens: 0 };
412
+ session.usage = { input_tokens: 0, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 };
413
+ }
414
+ // Phase 1 cost tracking — backfill cache_creation if loaded from an
415
+ // older schema (store.ts:validateSchema also does this; belt + braces).
416
+ if (session.usage.cache_creation_input_tokens == null) {
417
+ session.usage.cache_creation_input_tokens = 0;
354
418
  }
355
419
  // H2 — Anthropic's `message_delta.usage.output_tokens` is CUMULATIVE for the
356
420
  // current message (it grows monotonically as the model streams). Naively
@@ -444,6 +508,12 @@ export async function runTurn(params) {
444
508
  if (msgUsage) {
445
509
  session.usage.input_tokens += msgUsage.input_tokens ?? 0;
446
510
  session.usage.cache_read_input_tokens += msgUsage.cache_read_input_tokens ?? 0;
511
+ // Phase 1 cost tracking — cache writes are billed at 1.25× input.
512
+ // Anthropic emits cache_creation_input_tokens alongside cache_read
513
+ // when this message wrote new content to the cache.
514
+ session.usage.cache_creation_input_tokens =
515
+ (session.usage.cache_creation_input_tokens ?? 0)
516
+ + (msgUsage.cache_creation_input_tokens ?? 0);
447
517
  }
448
518
  // New message begins — reset the per-message cumulative-output scratch.
449
519
  messageOutputTokensSoFar = 0;
@@ -512,6 +582,28 @@ export async function runTurn(params) {
512
582
  session.turnInterruptedReason = undefined;
513
583
  }
514
584
  await saveSession(session);
585
+ // Phase 1 cost tracking (2026-05-16) — compute this turn's $ cost from
586
+ // the deltas (turn-end counters minus turn-start snapshot) and append a
587
+ // single JSONL line to ~/.cspeach/sessions/<id>-cost.jsonl. Best-effort:
588
+ // failures inside appendCostLine are swallowed so cost-tracking can
589
+ // never crash a turn. Runs for both happy and interrupted paths so
590
+ // the user's spend is recorded even when a turn dies mid-stream.
591
+ {
592
+ const turnTokens = {
593
+ input: (session.usage?.input_tokens ?? 0) - turnStartTokens.input,
594
+ output: (session.usage?.output_tokens ?? 0) - turnStartTokens.output,
595
+ cacheRead: (session.usage?.cache_read_input_tokens ?? 0) - turnStartTokens.cacheRead,
596
+ cacheCreate: (session.usage?.cache_creation_input_tokens ?? 0) - turnStartTokens.cacheCreate,
597
+ };
598
+ const turnDurationMs = Date.now() - turnStartedAt;
599
+ const entry = buildCostEntry({
600
+ turn: turnNumber,
601
+ model: session.model,
602
+ tokens: turnTokens,
603
+ duration_ms: turnDurationMs,
604
+ });
605
+ void appendCostLine(session.id, entry);
606
+ }
515
607
  // Close the stream-to-disk mirror with a footer (lightweight diagnostics).
516
608
  // Best-effort — failure here is invisible to the turn flow.
517
609
  void turnStreamWriter.close(interruptedError !== null
@@ -575,6 +667,30 @@ export async function runTurn(params) {
575
667
  emit,
576
668
  promotedFrom: promotedFromForSave,
577
669
  });
670
+ // Phase 2 #9 (2026-05-16) — auto-compact end-of-turn hook.
671
+ // Runs only on the success path (after maybeOfferSave) so an interrupted
672
+ // turn never triggers a summariser call on partial state. The maybeAutoCompact
673
+ // helper internally swallows non-fatal errors and emits a yellow note so
674
+ // a Haiku blip never regresses a turn that just succeeded. Threshold and
675
+ // throttle live in config.compact — see config/loader.ts:CompactConfig.
676
+ try {
677
+ const { maybeAutoCompact } = await import('../commands/auto-compact.js');
678
+ const { buildSummarisationPrompt, serialiseForSummariser, COMPACTION_MODEL } = await import('../commands/compact.js');
679
+ const { summariseViaProvider } = await import('./summarise-via-provider.js');
680
+ await maybeAutoCompact({
681
+ session,
682
+ config: cfg.compact,
683
+ turnNumber,
684
+ emit,
685
+ summarise: async (toSummarise) => summariseViaProvider(params.provider, buildSummarisationPrompt(), serialiseForSummariser(toSummarise), { model: COMPACTION_MODEL, maxTokens: 4000, headers: { 'X-CSForge-Skill': 'compact' } }),
686
+ });
687
+ }
688
+ catch (err) {
689
+ // Belt-and-braces — maybeAutoCompact already swallows internally,
690
+ // but if its own dynamic-import or wiring blows up we still must
691
+ // not regress the success path.
692
+ emit(chalk.gray(`[auto-compact] skipped: ${err instanceof Error ? err.message : String(err)}`));
693
+ }
578
694
  return;
579
695
  }
580
696
  if (!sawToolUse) {
@@ -583,6 +699,14 @@ export async function runTurn(params) {
583
699
  return;
584
700
  }
585
701
  // Dispatch each tool_use block.
702
+ //
703
+ // Bug 8 — parallel-write guard: when the model emits multiple write-class
704
+ // tool_use blocks in a single round (sap_set_source, sap_update_method,
705
+ // sap_delete_object, sap_create_object), execute only the FIRST. Reject
706
+ // the rest with a structured envelope so the model reissues them
707
+ // sequentially on the next round. Read-only tools and approval requests
708
+ // continue to run normally. See ./parallel-write-guard.ts.
709
+ const writeGuard = newGuardState();
586
710
  const toolResults = [];
587
711
  for (const block of currentAssistantContent) {
588
712
  if (block.type === 'tool_use') {
@@ -599,14 +723,23 @@ export async function runTurn(params) {
599
723
  const spinner = startToolSpinner({ chunkEmitter: params.chunkEmitter });
600
724
  // Phase 2b: dispatch (may take 100ms–several seconds for write tools).
601
725
  const dispatchStart = Date.now();
602
- const result = await dispatchTool(block.name, block.input, params.ctx);
726
+ // Bug 8 — short-circuit for 2nd+ write-class tool in the same round.
727
+ const guardDecision = checkAndMark(block.name, writeGuard);
728
+ const result = guardDecision.allow
729
+ ? await dispatchTool(block.name, block.input, params.ctx)
730
+ : { content: guardDecision.errorContent, is_error: true };
603
731
  const durationMs = Date.now() - dispatchStart;
604
732
  // Phase 2c: stop the spinner, which erases its line so the result
605
733
  // row paints in place of it (no scrollback artifacts).
606
734
  spinner.stop();
607
735
  // Phase 3: print result line ( ⎿ ✓ summary · timing).
736
+ // 2026-05-15 (bug 1): the old `slice(0, 80)` clipped to the JSON
737
+ // preamble (e.g. `{"error":"write_failed","detail":"HTTP 409: <?xml v…`)
738
+ // and hid the actual SAP error message behind XML noise. We now
739
+ // unwrap the JSON, pull the ADT exception message if there is one,
740
+ // and let renderToolCallBottom apply its own (wider) width clamp.
608
741
  const resultSummary = result.is_error
609
- ? (typeof result.content === 'string' ? result.content.slice(0, 80) : 'error')
742
+ ? (typeof result.content === 'string' ? formatToolErrorSummary(result.content) : 'error')
610
743
  : undefined;
611
744
  renderToolCallBottom({
612
745
  durationMs,
@@ -654,19 +787,48 @@ export async function runTurn(params) {
654
787
  const count = (failMap.get(objKey) ?? 0) + 1;
655
788
  failMap.set(objKey, count);
656
789
  if (count >= RETRY_CAP) {
657
- // Cap hit — push the error result, save session, and abort the turn.
658
- toolResults.push({
659
- type: 'tool_result',
660
- tool_use_id: block.id,
661
- is_error: true,
662
- content: JSON.stringify({
663
- error: ERR.APPROVAL_THRASH,
664
- detail: `Retry cap hit after ${RETRY_CAP} consecutive failures on ${objKey}. Aborting turn.`,
665
- }),
790
+ // Bug 2 (2026-05-15) — pause the turn instead of aborting it.
791
+ // The old behaviour returned with the model's last context being
792
+ // an error tool_result, so the next user message resumed from a
793
+ // confused state and the user had no clear "retry" affordance.
794
+ //
795
+ // New behaviour: synthesise a single user-readable tool_result
796
+ // explaining the cap and quoting the last error so the model on
797
+ // the next turn knows the context, then emit a yellow [paused]
798
+ // line telling the human exactly what to do, reset the failMap
799
+ // entry to zero so the next retry has full runway, save, and
800
+ // return cleanly. The REPL prompt comes back and the user's
801
+ // next message routes normally.
802
+ const pause = buildRetryCapPausePayload({
803
+ toolUseId: block.id,
804
+ objKey,
805
+ retryCap: RETRY_CAP,
806
+ lastErrorContent: result.content,
666
807
  });
808
+ toolResults.push(pause.toolResult);
809
+ // Reviewer #2 fix (2026-05-15): if the assistant message had
810
+ // sibling tool_use blocks AFTER the one that tripped the cap,
811
+ // they were never dispatched. Without paired tool_results the
812
+ // next createStream call rejects with `tool_use_id was not
813
+ // found`. Push a synthetic skipped result for every remaining
814
+ // tool_use so the message is well-formed for the next round.
815
+ const skipped = buildSkippedSiblingResults({
816
+ assistantContent: currentAssistantContent,
817
+ currentBlockIndex: currentAssistantContent.indexOf(block),
818
+ objKey,
819
+ });
820
+ for (const sk of skipped)
821
+ toolResults.push(sk);
667
822
  session.messages.push({ role: 'user', content: toolResults });
823
+ // Reset the counter so the next retry has full runway. Without
824
+ // this reset the next attempt by the model would re-trip the cap
825
+ // on its first failure and pause again.
826
+ failMap.set(objKey, 0);
668
827
  await saveSession(session);
669
- emit(chalk.red(`\n[retry cap] ${objKey} failed ${RETRY_CAP} times — aborting turn.`));
828
+ emit(chalk.yellow(`\n${pause.pauseLine}`));
829
+ if (pause.lastErrorSummary) {
830
+ emit(chalk.gray(` last error: ${pause.lastErrorSummary}`));
831
+ }
670
832
  return;
671
833
  }
672
834
  }
@@ -0,0 +1,71 @@
1
+ /**
2
+ * Bug 8 — parallel-write guard.
3
+ *
4
+ * The `/abap-generate` skill (and others) instruct the model to issue writes
5
+ * "one at a time". The model nevertheless emits multiple write-class
6
+ * `tool_use` blocks in a single assistant round, which causes:
7
+ *
8
+ * 1. Approval JWTs being spent in parallel — the second+ write fails with
9
+ * "approval already spent / invalid".
10
+ * 2. Read-modify-write corruption (Bug 13) when two writes race against the
11
+ * same class.
12
+ * 3. Snapshot/audit ordering becomes non-deterministic.
13
+ *
14
+ * Fix: within a single LLM round, allow at most ONE write-class tool to
15
+ * execute. Reject subsequent write-class blocks with a structured envelope
16
+ * the model can recover from on the next round (sequential reissue).
17
+ *
18
+ * Read-only tools and approval requests still run normally, even after a
19
+ * write-class tool has already executed in the same round — the round
20
+ * boundary is a one-shot guard for *writes*, not a general serialiser.
21
+ *
22
+ * Scope of "write-class" matches the spec for this bug fix and is narrower
23
+ * than tools/write-mode.ts WRITE_TOOLS: only the four tools that perform
24
+ * net-new source mutation are gated here.
25
+ */
26
+ /** Tools that mutate ABAP source. Narrower than WRITE_TOOLS — excludes
27
+ * activate / publish / transport_release / message_maintain. The bug
28
+ * specifically targets source-write parallelism. */
29
+ export const WRITE_CLASS_TOOLS = new Set([
30
+ 'sap_set_source',
31
+ 'sap_update_method',
32
+ 'sap_delete_object',
33
+ 'sap_create_object',
34
+ ]);
35
+ export function isWriteClassTool(name) {
36
+ return WRITE_CLASS_TOOLS.has(name);
37
+ }
38
+ export function newGuardState() {
39
+ return { writeIssued: false };
40
+ }
41
+ /**
42
+ * Decide whether the given tool may dispatch in the current round.
43
+ *
44
+ * Returns `{ allow: true }` to proceed.
45
+ * Returns `{ allow: false, errorContent }` when the tool is a write-class
46
+ * tool and another write-class tool has already issued in this round.
47
+ * The caller should NOT dispatch — emit `errorContent` as the tool
48
+ * result so the model sees the rejection and reissues sequentially on
49
+ * the next round.
50
+ *
51
+ * Mutates `state.writeIssued = true` when allowing a write-class tool.
52
+ */
53
+ export function checkAndMark(toolName, state) {
54
+ if (!isWriteClassTool(toolName)) {
55
+ return { allow: true };
56
+ }
57
+ if (state.writeIssued) {
58
+ return {
59
+ allow: false,
60
+ errorContent: JSON.stringify({
61
+ error: 'parallel_writes_forbidden',
62
+ detail: 'Issue write-class tools one at a time. ' +
63
+ 'Tool N+ in this round was not executed. ' +
64
+ 'Reissue this call on the next round, after the first write completes.',
65
+ tool: toolName,
66
+ }),
67
+ };
68
+ }
69
+ state.writeIssued = true;
70
+ return { allow: true };
71
+ }
@@ -23,9 +23,96 @@ export function repairPartialBlocks(blocks) {
23
23
  return false;
24
24
  if (b.type === 'tool_use') {
25
25
  // Complete only when input was successfully parsed (object, not undefined).
26
- return b.input !== undefined;
26
+ if (b.input === undefined)
27
+ return false;
28
+ // Strip residual `partial_json` left over by interrupted streaming.
29
+ // Even with `input` parsed, an extra `partial_json` field on the block
30
+ // gets persisted to session.messages and the next API call fails with
31
+ // 400 messages.N.content.M.tool_use.partial_json:
32
+ // Extra inputs are not permitted
33
+ // Mirrors the heal in session/store.ts:loadSession.
34
+ if (typeof b.partial_json === 'string')
35
+ delete b.partial_json;
36
+ return true;
27
37
  }
28
38
  // text, thinking, redacted_thinking, etc. — keep as-is.
29
39
  return true;
30
40
  });
31
41
  }
42
+ /**
43
+ * In-memory heal for a poisoned session.messages array.
44
+ *
45
+ * Counterpart to the heal in session/store.ts:loadSession that runs on every
46
+ * disk load. When the agent loop catches an API 400 whose message contains
47
+ * `partial_json: Extra inputs are not permitted`, the live in-memory messages
48
+ * array is the culprit — a previous turn persisted a tool_use block with both
49
+ * `input` parsed AND a leftover `partial_json` string. The session on disk
50
+ * was already healed by loadSession, but the in-process array is what the
51
+ * SDK call serialises, so we need to scrub it here too.
52
+ *
53
+ * Before this heal existed, the only path back was `/exit` then
54
+ * `cspeach --resume <id>` (which goes through loadSession's heal). Users
55
+ * had to bounce the REPL after every poisoned turn.
56
+ *
57
+ * Returns the number of blocks scrubbed so the caller can decide whether
58
+ * the heal actually changed anything (no-op heals shouldn't print a yellow
59
+ * note).
60
+ */
61
+ export function healSessionMessagesInPlace(messages) {
62
+ let scrubbed = 0;
63
+ for (const msg of messages ?? []) {
64
+ if (!Array.isArray(msg.content))
65
+ continue;
66
+ for (const block of msg.content) {
67
+ if (block?.type === 'tool_use' && typeof block.partial_json === 'string') {
68
+ if (block.input === undefined) {
69
+ try {
70
+ block.input = JSON.parse(block.partial_json);
71
+ }
72
+ catch {
73
+ block.input = {};
74
+ }
75
+ }
76
+ delete block.partial_json;
77
+ scrubbed++;
78
+ }
79
+ }
80
+ }
81
+ return scrubbed;
82
+ }
83
+ /**
84
+ * Heuristic — recognises the Anthropic API 400 that fires when a persisted
85
+ * tool_use block still carries `partial_json` alongside the parsed `input`.
86
+ * Matches both the raw SDK error message and a stringified body.
87
+ *
88
+ * Narrow fingerprint: `partial_json` token AND a phrase from the documented
89
+ * Anthropic 400 (`Extra inputs are not permitted` / `not permitted`). The
90
+ * substring `partial_json` alone is broad enough to false-positive on
91
+ * unrelated 4xx whose error body echoes the offending request — we'd then
92
+ * silently scrub a healthy session. Requiring the second phrase keeps the
93
+ * heal scoped to the actual fingerprint without burning correctness when
94
+ * Anthropic varies wording slightly (`/not permitted|Extra inputs/i`
95
+ * matches both observed forms).
96
+ */
97
+ export function isPartialJsonApiError(err) {
98
+ if (err == null)
99
+ return false;
100
+ const candidates = [];
101
+ if (err instanceof Error)
102
+ candidates.push(err.message ?? '');
103
+ if (typeof err === 'object') {
104
+ const e = err;
105
+ const body = e.body ?? e.error ?? e.response?.data;
106
+ if (typeof body === 'string')
107
+ candidates.push(body);
108
+ else if (body && typeof body === 'object') {
109
+ try {
110
+ candidates.push(JSON.stringify(body));
111
+ }
112
+ catch { /* ignore */ }
113
+ }
114
+ }
115
+ return candidates.some((s) => typeof s === 'string'
116
+ && s.includes('partial_json')
117
+ && /not permitted|Extra inputs/i.test(s));
118
+ }