@cspeach/cli 0.9.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/README.md +1 -1
  2. package/dist/agent/intent-system-prompt.js +1 -1
  3. package/dist/agent/loop.js +228 -26
  4. package/dist/agent/providers/license-gate.js +44 -0
  5. package/dist/agent/skill-checkpoint.js +1 -1
  6. package/dist/agent/tool-dispatch.js +15 -0
  7. package/dist/approvals/canonical.js +91 -0
  8. package/dist/approvals/jwt.js +39 -2
  9. package/dist/approvals/op-labels.js +124 -0
  10. package/dist/approvals/render.js +42 -36
  11. package/dist/auth/org-anthropic-key.js +25 -0
  12. package/dist/classifier/client.js +18 -3
  13. package/dist/cli.js +15 -0
  14. package/dist/commands/compact.js +28 -2
  15. package/dist/commands/config-set.js +284 -0
  16. package/dist/commands/config-show.js +20 -0
  17. package/dist/commands/export-audit.js +43 -0
  18. package/dist/commands/help.js +5 -0
  19. package/dist/commands/login.js +31 -14
  20. package/dist/commands/plan-audit-evidence.js +266 -0
  21. package/dist/commands/plan-audit.js +692 -0
  22. package/dist/commands/plan-chain.js +671 -0
  23. package/dist/commands/plan-continue.js +179 -0
  24. package/dist/commands/plan-gate.js +154 -0
  25. package/dist/commands/plan-model-tier.js +83 -0
  26. package/dist/commands/plan-resume.js +728 -46
  27. package/dist/config/loader.js +223 -5
  28. package/dist/config/model-defaults.js +14 -0
  29. package/dist/cost/pricing.js +27 -1
  30. package/dist/doctor/checks/_http-probe.js +1 -0
  31. package/dist/doctor/checks/cert.js +14 -3
  32. package/dist/doctor/checks/sap.js +30 -8
  33. package/dist/doctor/checks/system-roles.js +41 -0
  34. package/dist/doctor/checks/zcspeach.js +19 -4
  35. package/dist/doctor/run.js +2 -0
  36. package/dist/models/resolve.js +61 -0
  37. package/dist/models/server-config.js +155 -0
  38. package/dist/one-shot.js +76 -6
  39. package/dist/projects/answer-blockers.js +137 -0
  40. package/dist/projects/extract-cca.js +111 -17
  41. package/dist/projects/extract-modernize.js +4 -2
  42. package/dist/projects/extract-plan.js +184 -37
  43. package/dist/projects/extract-spec-gap.js +34 -7
  44. package/dist/projects/extract-test-coverage.js +4 -2
  45. package/dist/projects/extract-upgrade.js +116 -23
  46. package/dist/projects/handover-md.js +195 -0
  47. package/dist/projects/index.js +5 -2
  48. package/dist/projects/merge-cca.js +292 -0
  49. package/dist/projects/merge-upgrade.js +173 -0
  50. package/dist/projects/migration.js +103 -1
  51. package/dist/projects/output-paths.js +27 -0
  52. package/dist/projects/plan-run.js +285 -27
  53. package/dist/projects/plan-schema.js +136 -3
  54. package/dist/projects/promote-command.js +25 -2
  55. package/dist/projects/promote.js +128 -0
  56. package/dist/projects/run-lease.js +157 -0
  57. package/dist/projects/save-command.js +259 -21
  58. package/dist/projects/status.js +3 -1
  59. package/dist/projects/validate.js +1 -1
  60. package/dist/projects/workspace.js +164 -20
  61. package/dist/renderer/notices.js +64 -0
  62. package/dist/renderer/progress-chatter.js +8 -0
  63. package/dist/renderer/status-footer.js +22 -12
  64. package/dist/renderer/thinking-heartbeat.js +64 -8
  65. package/dist/renderer/todo-block.js +51 -0
  66. package/dist/renderer/tool-widget.js +55 -4
  67. package/dist/renderer/tty.js +43 -4
  68. package/dist/renderer/verify-chain.js +77 -0
  69. package/dist/repl/at-picker.js +60 -7
  70. package/dist/repl/bracketed-paste.js +28 -19
  71. package/dist/repl/builtin-commands.js +42 -0
  72. package/dist/repl/current-transport.js +10 -0
  73. package/dist/repl/early-line-buffer.js +68 -0
  74. package/dist/repl/history.js +86 -0
  75. package/dist/repl/ink-stdin-guard.js +64 -0
  76. package/dist/repl/inquirer-guard.js +70 -5
  77. package/dist/repl/mode-ceiling.js +16 -0
  78. package/dist/repl/mode-cycle.js +104 -0
  79. package/dist/repl/numbered-menu.js +131 -0
  80. package/dist/repl/post-turn-status.js +26 -6
  81. package/dist/repl/rule8-detector.js +17 -2
  82. package/dist/repl/safety-confirm.js +111 -2
  83. package/dist/repl/safety-mode-state.js +19 -3
  84. package/dist/repl/slash-completer.js +5 -0
  85. package/dist/repl/slash-picker.js +10 -15
  86. package/dist/repl.js +1232 -95
  87. package/dist/rewind/candidates.js +194 -0
  88. package/dist/rewind/cli.js +137 -0
  89. package/dist/rewind/format.js +27 -0
  90. package/dist/rewind/restore.js +245 -0
  91. package/dist/router/classifier.js +150 -6
  92. package/dist/sap/capability-matrix.js +20 -0
  93. package/dist/sap/capability-matrix.json +11236 -0
  94. package/dist/sap/capability.js +146 -0
  95. package/dist/sap/connection-manager.js +19 -1
  96. package/dist/sap/onboarding.js +42 -4
  97. package/dist/session/audit-export.js +459 -0
  98. package/dist/session/context-report.js +163 -0
  99. package/dist/session/pending.js +27 -0
  100. package/dist/session/recap.js +160 -0
  101. package/dist/skill-catalog.js +51 -40
  102. package/dist/skills/bundled-skills.js +272 -1
  103. package/dist/skills/promotion-dispatch.js +23 -0
  104. package/dist/tools/_command-shared.js +36 -12
  105. package/dist/tools/_filesystem-shared.js +139 -4
  106. package/dist/tools/_flag.js +25 -0
  107. package/dist/tools/approval.js +177 -26
  108. package/dist/tools/ask-question.js +400 -7
  109. package/dist/tools/capability/tool.js +74 -0
  110. package/dist/tools/dispatch-skill.js +22 -1
  111. package/dist/tools/extend-model/anchored-insert.js +1414 -0
  112. package/dist/tools/extend-model/tool.js +340 -0
  113. package/dist/tools/filesystem/extract-document.js +57 -0
  114. package/dist/tools/filesystem/file-edit.js +12 -2
  115. package/dist/tools/filesystem/file-read.js +2 -2
  116. package/dist/tools/filesystem/file-write.js +11 -2
  117. package/dist/tools/filesystem/glob.js +11 -0
  118. package/dist/tools/filesystem/grep.js +10 -0
  119. package/dist/tools/filesystem/read-document.js +107 -0
  120. package/dist/tools/fiori/apply.js +50 -0
  121. package/dist/tools/fiori/bin.js +3 -0
  122. package/dist/tools/fiori/catalog/index.js +27 -0
  123. package/dist/tools/fiori/catalog/value-help.js +230 -0
  124. package/dist/tools/fiori/catalog/viz-chart.js +177 -0
  125. package/dist/tools/fiori/cli.js +71 -0
  126. package/dist/tools/fiori/deploy-config.js +73 -0
  127. package/dist/tools/fiori/fe-extend.js +76 -0
  128. package/dist/tools/fiori/fe-scaffold.js +71 -0
  129. package/dist/tools/fiori/floorplan-map.js +19 -0
  130. package/dist/tools/fiori/i18n.js +39 -0
  131. package/dist/tools/fiori/manifest.js +70 -0
  132. package/dist/tools/fiori/render.js +77 -0
  133. package/dist/tools/fiori/samples/data/index.json +13602 -0
  134. package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
  135. package/dist/tools/fiori/samples/loader.js +248 -0
  136. package/dist/tools/fiori/samples/search.js +63 -0
  137. package/dist/tools/fiori/samples/types.js +2 -0
  138. package/dist/tools/fiori/scaffold.js +39 -0
  139. package/dist/tools/fiori/smoke/assertions.js +74 -0
  140. package/dist/tools/fiori/smoke/browser.js +52 -0
  141. package/dist/tools/fiori/smoke/driver.js +89 -0
  142. package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
  143. package/dist/tools/fiori/smoke/run-smoke.js +149 -0
  144. package/dist/tools/fiori/tools.js +681 -0
  145. package/dist/tools/fiori/types.js +1 -0
  146. package/dist/tools/local-build.js +86 -0
  147. package/dist/tools/local-files.js +31 -0
  148. package/dist/tools/project/_merge-shared.js +68 -0
  149. package/dist/tools/project/cca_merge.js +164 -0
  150. package/dist/tools/project/playbook_get.js +1 -1
  151. package/dist/tools/project/upgrade_merge_progress.js +206 -0
  152. package/dist/tools/sap-read.js +132 -20
  153. package/dist/tools/sap-write.js +550 -21
  154. package/dist/tools/shell/shell_exec.js +41 -6
  155. package/dist/tools/snapshot.js +63 -14
  156. package/dist/tools/subagent/agent_run.js +27 -3
  157. package/dist/tools/subagent/background_run.js +17 -1
  158. package/dist/tools/todo.js +144 -0
  159. package/dist/tools/transport-resolution.js +86 -0
  160. package/dist/tools/transport.js +224 -5
  161. package/dist/tools/write-mode.js +4 -0
  162. package/dist/ui/app.js +378 -21
  163. package/dist/ui/approval-modal.js +49 -16
  164. package/dist/ui/ask-question-emitter.js +14 -0
  165. package/dist/ui/body.js +13 -0
  166. package/dist/ui/context-grid.js +108 -0
  167. package/dist/ui/footer.js +120 -27
  168. package/dist/ui/header.js +7 -0
  169. package/dist/ui/line-resolution.js +35 -8
  170. package/dist/ui/rewind-emitter.js +10 -0
  171. package/dist/ui/rewind-panel.js +81 -0
  172. package/dist/ui/sap-state-store.js +1 -0
  173. package/dist/ui/session-timeline.js +1 -0
  174. package/dist/ui/status-line.js +43 -0
  175. package/dist/ui/text-input.js +214 -0
  176. package/dist/ui/todo-emitter.js +25 -0
  177. package/dist/ui/todo-panel.js +64 -0
  178. package/dist/ui/turn-status-emitter.js +50 -4
  179. package/dist/ui/turn-status.js +18 -3
  180. package/dist/ui/widgets/ask-form.js +242 -0
  181. package/dist/ui/widgets/ask-question-modal.js +21 -8
  182. package/package.json +22 -3
  183. package/bench/README.md +0 -78
  184. package/bench/prompts/abap-document-cds.md +0 -44
  185. package/bench/prompts/abap-explain-bdef-handler.md +0 -57
  186. package/bench/prompts/abap-test-method.md +0 -42
  187. package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
  188. package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
  189. package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
  190. package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
  191. package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
  192. package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
  193. package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
  194. package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
  195. package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
@@ -8,11 +8,26 @@
8
8
  * allow = ["docker", "make"]
9
9
  * block in ~/.cspeach/config.toml.
10
10
  *
11
- * Cross-platform: on Windows resolves the .cmd / .bat / .exe shim
12
- * via PATH walk before spawning (Node's spawn with shell:false can't
13
- * directly invoke .cmd files; using shell:true reopens shell-injection
14
- * + CVE-2024-27980). On Unix the bare name is passed straight through
15
- * — spawn finds it on PATH.
11
+ * Cross-platform spawn via cross-spawn (the library npm itself uses).
12
+ * The actual spawn is handed the ABSOLUTE PATH-resolved executable path
13
+ * (from resolveExecutable's PATH-only walk) + argv ARRAY. cross-spawn
14
+ * still inspects the file extension and, on Windows, invokes .cmd / .bat
15
+ * through cmd.exe with correct per-arg escaping. This satisfies Node's
16
+ * CVE-2024-27980 restriction (spawn with shell:false refuses to launch
17
+ * .cmd/.bat directly) WITHOUT ever setting shell:true — args stay an
18
+ * array, so shell metacharacters are inert and there is no injection
19
+ * surface. On Unix cross-spawn is a near-passthrough to
20
+ * child_process.spawn (unchanged behavior).
21
+ *
22
+ * SECURITY — why the absolute path, not the bare name: cross-spawn
23
+ * resolves a BARE command name cwd-FIRST on Windows (node_modules/which
24
+ * checks process.cwd() before PATH on win32). A model with an in-tree
25
+ * write primitive could plant `<safelisted>.cmd` in the project tree and
26
+ * have cross-spawn pick it up ahead of the real PATH binary → arbitrary
27
+ * code execution, safelist bypassed. Spawning the absolute PATH-resolved
28
+ * path (which never consults cwd/PATH for a path-bearing command) closes
29
+ * that vector. The pre-flight resolveExecutable() PATH walk and the spawn
30
+ * now use the SAME binary — no pre-flight/spawn divergence.
16
31
  *
17
32
  * Operational caps:
18
33
  * - 30 s wall-clock timeout (overridable up to 300 s via timeout_ms arg)
@@ -24,7 +39,7 @@
24
39
  * Flag-gated: invisible to listTools() unless CSPEACH_TOOL_SHELL_EXEC=on.
25
40
  * isMutating: true — hooks into the existing approval gate.
26
41
  */
27
- import { spawn } from 'node:child_process';
42
+ import spawn from 'cross-spawn';
28
43
  import { promises as fs } from 'node:fs';
29
44
  import * as path from 'node:path';
30
45
  import { registerTool } from '../index.js';
@@ -80,6 +95,9 @@ export async function shellExecHandler(args, ctx) {
80
95
  catch {
81
96
  return { content: `error: cwd "${args.cwd ?? '.'}" does not exist`, is_error: true };
82
97
  }
98
+ // Resolve the bare safelisted name to an absolute executable path via a
99
+ // PATH-only walk (no cwd-first). This is BOTH the friendly "not installed"
100
+ // pre-flight check AND the spawn target below — same binary, no divergence.
83
101
  const resolved = await resolveExecutable(args.command);
84
102
  if (resolved === null) {
85
103
  return {
@@ -101,6 +119,23 @@ export async function shellExecHandler(args, ctx) {
101
119
  settled = true;
102
120
  resolve(result);
103
121
  };
122
+ // Pass the already-validated absolute `resolved` path (from
123
+ // resolveExecutable's PATH-only walk) — NOT the bare command name.
124
+ //
125
+ // SECURITY: cross-spawn resolves a BARE name cwd-first on Windows
126
+ // (node_modules/which checks process.cwd() before PATH on win32, and
127
+ // it chdir's into options.cwd first). A model with a write primitive
128
+ // could plant `<safelisted>.cmd` (e.g. git.cmd) in the project tree and
129
+ // have cross-spawn pick it up cwd-first → arbitrary code execution,
130
+ // safelist bypassed. Handing cross-spawn the absolute PATH-resolved
131
+ // path closes that: `which` short-circuits to the literal file when the
132
+ // command contains a path separator (pathEnv=['']), so cwd/PATH are
133
+ // never consulted. cross-spawn STILL inspects the file extension and
134
+ // wraps an absolute `.cmd`/`.bat` through cmd.exe with proper per-arg
135
+ // escaping (parse.js isExecutableRegExp only matches .com/.exe), so the
136
+ // Windows shim handling + CVE-2024-27980 .cmd restriction handling are
137
+ // preserved. shell:false stays (argv-only; metacharacters inert). The
138
+ // existence check above and this spawn now target the SAME binary.
104
139
  const child = spawn(resolved, argv, {
105
140
  cwd,
106
141
  env: buildSafeEnv(),
@@ -14,7 +14,18 @@
14
14
  * cleanup(retentionHours?) → Promise<number>
15
15
  */
16
16
  import { registerTool } from './index.js';
17
- import { snapshots } from '@cspeach/sap-client';
17
+ import { snapshots, SapError } from '@cspeach/sap-client';
18
+ /** True only for a genuine ADT 404 — the object does not exist. Any other
19
+ * failure (network, auth, 5xx, timeout) is a READ failure, not proof of
20
+ * absence, and must never be labelled "object not found". */
21
+ function isNotFound(err) {
22
+ return err instanceof SapError && err.httpStatus === 404;
23
+ }
24
+ /** Short single-line failure message for transcripts / error envelopes. */
25
+ function shortErrorMessage(err) {
26
+ const msg = err instanceof Error ? err.message : String(err);
27
+ return msg.replace(/\s+/g, ' ').trim();
28
+ }
18
29
  registerTool({
19
30
  name: 'sap_snapshot_take',
20
31
  description: "Take a snapshot of an object's current source before modification. Usually invoked automatically by the CLI before any write — rarely called directly by the model.",
@@ -28,25 +39,63 @@ registerTool({
28
39
  required: ['name', 'type'],
29
40
  },
30
41
  handler: async (args, ctx) => {
31
- const source = await ctx.adt.getSource(args.type, args.name).catch(() => null);
32
- if (source === null) {
33
- return { content: JSON.stringify({ skipped: true, reason: 'object_not_found' }) };
42
+ let source;
43
+ try {
44
+ source = await ctx.adt.getSource(args.type, args.name);
45
+ }
46
+ catch (err) {
47
+ if (isNotFound(err)) {
48
+ return { content: JSON.stringify({ skipped: true, reason: 'object_not_found' }) };
49
+ }
50
+ // Read failed for some OTHER reason — the object may well exist.
51
+ // Claiming "object_not_found" here would be a lie; surface the failure.
52
+ return {
53
+ content: JSON.stringify({ error: 'source_read_failed', detail: shortErrorMessage(err) }),
54
+ is_error: true,
55
+ };
34
56
  }
35
57
  const entry = await snapshots.take(args.type, args.name, source, 'manual');
36
58
  return { content: JSON.stringify({ snapshot: entry, size: source.length }) };
37
59
  },
38
60
  });
39
- /**
40
- * Auto-snapshot helper called by write tools before mutating an object.
41
- * Silently skips if the object does not exist yet (first write after create).
42
- *
43
- * snapshots.take expects (type, name, source, reason).
44
- */
45
61
  export async function autoSnapshot(objectName, objectType, adt) {
46
- const src = await adt.getSource(objectType, objectName).catch(() => null);
47
- if (src === null)
48
- return; // new object — nothing to snapshot
49
- await snapshots.take(objectType, objectName, src, 'before_write');
62
+ let src;
63
+ try {
64
+ src = await adt.getSource(objectType, objectName);
65
+ }
66
+ catch (err) {
67
+ if (isNotFound(err))
68
+ return { taken: false, reason: 'object_not_found' }; // new object
69
+ // The DEFAULT read failed for a non-404 reason. Before treating this as a
70
+ // hard read failure (which blocks the write, Rule 7), distinguish an
71
+ // empty/never-activated SHELL from a real object whose source we couldn't
72
+ // read. A freshly-created behaviour-pool CLAS (ZBP … FOR BEHAVIOR OF …)
73
+ // EXISTS but has no committed source yet, so the default read returns HTTP
74
+ // 400 — not 404. Such a shell has NO committed ACTIVE version. Probe for one:
75
+ // - active read 404 → no active version ⇒ fresh shell, nothing to
76
+ // snapshot ⇒ proceed ('new_shell').
77
+ // - active read succeeds → there IS committed source to protect ⇒
78
+ // snapshot THAT active source and proceed.
79
+ // - active read fails otherwise → can't confirm emptiness ⇒ conservative
80
+ // block ('source_read_failed', unchanged).
81
+ let activeSrc;
82
+ try {
83
+ activeSrc = await adt.getSource(objectType, objectName, 'active');
84
+ }
85
+ catch (activeErr) {
86
+ if (isNotFound(activeErr))
87
+ return { taken: false, reason: 'new_shell' };
88
+ // Could not confirm the object is empty — keep the safety property: a
89
+ // real object with source we couldn't read must still block the write.
90
+ return { taken: false, reason: 'source_read_failed', detail: shortErrorMessage(err) };
91
+ }
92
+ // Committed active version exists — snapshot it. A snapshots.take failure
93
+ // still THROWS (Rule 7 hard-fail), same as the clean-read path below.
94
+ const activeEntry = await snapshots.take(objectType, objectName, activeSrc, 'before_write');
95
+ return { taken: true, entry: activeEntry };
96
+ }
97
+ const entry = await snapshots.take(objectType, objectName, src, 'before_write');
98
+ return { taken: true, entry };
50
99
  }
51
100
  // ── sap_snapshot_list ─────────────────────────────────────────────────────────
52
101
  registerTool({
@@ -28,6 +28,7 @@
28
28
  */
29
29
  import { EventEmitter } from 'node:events';
30
30
  import { registerTool } from '../index.js';
31
+ import { isPlanPhaseActive, notePlanPhaseSubagentDispatch, } from '../../commands/plan-gate.js';
31
32
  import { runTurn } from '../../agent/loop.js';
32
33
  import { collectTurnAssistantText } from '../../agent/turn-assistant-text.js';
33
34
  export const MAX_AGENT_DEPTH = 3;
@@ -98,12 +99,20 @@ export async function agentRunHandler(args, ctx) {
98
99
  // not parent's start time. Avoids surprising file timestamps.
99
100
  started_at: new Date().toISOString(),
100
101
  last_turn_at: new Date().toISOString(),
102
+ // Task 5 (ux-wave2): a fresh subagent has no task list — without this
103
+ // the spread above would leak the PARENT's todos into the child's
104
+ // session file.
105
+ todos: undefined,
101
106
  };
102
107
  // Build an isolated child ctx. Crucial differences vs parent ctx:
103
108
  // - session: the fresh child session above (independent usage counters)
104
109
  // - previewHook: undefined — subagent must NOT auto-approve writes
105
110
  // - pendingDispatch / currentTransport: cleared — subagent doesn't
106
111
  // mutate parent REPL state
112
+ // - todoEmitter: omitted (Task 5) — todo_set emits via
113
+ // ctx.todoEmitter?.emit, so a child's todo_set updates only
114
+ // childSession.todos and never repaints the PARENT's Ctrl+T panel /
115
+ // turn-status strip. Same seam as chunkEmitter suppression.
107
116
  // chunkEmitter is passed via the RunTurn params (not the ctx) as a
108
117
  // null-sink EventEmitter — see I2 fix below.
109
118
  const childCtx = {
@@ -113,7 +122,8 @@ export async function agentRunHandler(args, ctx) {
113
122
  cwd: ctx.cwd,
114
123
  provider: ctx.provider,
115
124
  skillSource: ctx.skillSource,
116
- // previewHook + pendingDispatch + currentTransport intentionally omitted.
125
+ // previewHook + pendingDispatch + currentTransport + todoEmitter
126
+ // intentionally omitted.
117
127
  };
118
128
  // I2 fix: a null-sink EventEmitter suppresses all subagent output. Every
119
129
  // stdout-write site in loop.ts is guarded by `if (params.chunkEmitter)`,
@@ -128,9 +138,23 @@ export async function agentRunHandler(args, ctx) {
128
138
  // mutating tool from the subagent's tool list. Saves tokens on the
129
139
  // sub-agent invocation (smaller tools array = less input cost) AND
130
140
  // hard-prevents accidental writes from the subagent.
131
- const toolFilter = args.read_only
141
+ //
142
+ // Task 7 (agentic-flow, 2026-07-03) — inside a plan phase the filter is
143
+ // FORCED regardless of args.read_only: subagents can read, never write,
144
+ // and the model cannot opt out (read_only:false is ignored). Outside plan
145
+ // phases the opt-in behaviour above is byte-identical to before.
146
+ const planPhase = isPlanPhaseActive();
147
+ const readOnly = planPhase || Boolean(args.read_only);
148
+ const toolFilter = readOnly
132
149
  ? (t) => !t.isMutating
133
150
  : undefined;
151
+ if (planPhase)
152
+ notePlanPhaseSubagentDispatch();
153
+ // Task 7 — per-dispatch announce on the PARENT's emitter (childCtx
154
+ // deliberately omits chunkEmitter, so this is the one line the user sees
155
+ // for each dispatch). 'info' renders dim in both the Ink and classic REPLs.
156
+ const taskPreview = args.task.length > 60 ? `${args.task.slice(0, 60)}…` : args.task;
157
+ ctx.chunkEmitter?.emit('info', `↳ subagent: ${args.skill}${readOnly ? ' — read-only' : ''} — ${taskPreview}`);
134
158
  try {
135
159
  await runTurn({
136
160
  provider: ctx.provider,
@@ -189,7 +213,7 @@ registerTool({
189
213
  },
190
214
  read_only: {
191
215
  type: 'boolean',
192
- description: 'When true, the subagent only sees read-only tools (no sap_set_source, no file-write, no shell_exec, etc.). Use this for research / analysis subagents that should not make changes — e.g. spawning a subagent to read code, run greps, or summarise findings. Smaller tools array = lower per-call cost; also a hard guard against accidental writes.',
216
+ description: 'When true, the subagent only sees read-only tools (no sap_set_source, no file-write, no shell_exec, etc.). Use this for research / analysis subagents that should not make changes — e.g. spawning a subagent to read code, run greps, or summarise findings. Smaller tools array = lower per-call cost; also a hard guard against accidental writes. NOTE: during plan phases the read-only filter is ALWAYS enforced regardless of this argument.',
193
217
  },
194
218
  },
195
219
  required: ['skill', 'task'],
@@ -16,7 +16,7 @@
16
16
  * CSPEACH_TOOL_BACKGROUND_RUN=on. isMutating: true — hooks the existing
17
17
  * approval gate (same UX as shell_exec).
18
18
  */
19
- import { spawn } from 'node:child_process';
19
+ import spawn from 'cross-spawn';
20
20
  import { promises as fs } from 'node:fs';
21
21
  import * as path from 'node:path';
22
22
  import { randomUUID } from 'node:crypto';
@@ -69,6 +69,9 @@ export async function backgroundRunHandler(args, ctx) {
69
69
  catch {
70
70
  return { content: `error: cwd "${args.cwd ?? '.'}" does not exist`, is_error: true };
71
71
  }
72
+ // Resolve the bare safelisted name to an absolute executable path via a
73
+ // PATH-only walk (no cwd-first). Both the friendly "not installed"
74
+ // pre-flight check AND the spawn target below — same binary, no divergence.
72
75
  const resolved = await resolveExecutable(args.command);
73
76
  if (resolved === null) {
74
77
  return {
@@ -78,6 +81,19 @@ export async function backgroundRunHandler(args, ctx) {
78
81
  }
79
82
  let child;
80
83
  try {
84
+ // Absolute PATH-resolved `resolved` (NOT the bare command name).
85
+ //
86
+ // SECURITY: cross-spawn resolves a BARE name cwd-first on Windows
87
+ // (node_modules/which checks process.cwd() before PATH and chdir's into
88
+ // options.cwd first). A model with a write primitive could plant
89
+ // `<safelisted>.cmd` in the project tree and have it picked up cwd-first
90
+ // → arbitrary code execution, safelist bypassed. Handing cross-spawn the
91
+ // absolute PATH-resolved path closes that (`which` short-circuits to the
92
+ // literal file for path-bearing commands, pathEnv=['']). cross-spawn
93
+ // still wraps an absolute `.cmd`/`.bat` through cmd.exe with per-arg
94
+ // escaping, so Windows shim + CVE-2024-27980 handling stay. shell:false
95
+ // → metacharacters inert. Existence check + spawn now target the SAME
96
+ // binary. See shell_exec.ts for the full rationale.
81
97
  child = spawn(resolved, argv, {
82
98
  cwd,
83
99
  env: buildSafeEnv(),
@@ -0,0 +1,144 @@
1
+ /**
2
+ * todo_set — UX Wave 2 / Task 4. CC's TodoWrite adapted for CSPeach: the
3
+ * model maintains a visible task list on long multi-step jobs (RAP stacks,
4
+ * upgrade batches, plan phases). Task 5 renders it (Ctrl+T panel +
5
+ * turn-status strip) via todoEmitter; /compact appends it to the summary
6
+ * block so the plan survives compaction (commands/compact.ts).
7
+ *
8
+ * Semantics: FULL-LIST REPLACE on every call — never a delta. 1..20 items.
9
+ *
10
+ * Transcript render (CC-style, 2026-07): on every successful set the
11
+ * handler also emits the FULL checklist as a scrollback block via
12
+ * ctx.chunkEmitter (renderer/todo-block.ts). The raw ⏺/⎿ todo_set rows are
13
+ * widget-suppressed (renderer/tool-widget.ts) so the block is the sole
14
+ * visible representation — the transcript now shows the whole plan
15
+ * advancing, not a raw `⏺ todo_set({"todos":[…])` JSON line. This is the
16
+ * Claude-Code parity the owner asked for; the turn-status strip (single ▶
17
+ * row) and Ctrl+T panel stay exactly as they were.
18
+ *
19
+ * Session-write route (chosen with evidence, 2026-07-05): the tool mutates
20
+ * `ctx.session.todos` directly. ToolContext.session IS the live
21
+ * SessionState object (tools/index.ts) — the same instance repl.tsx /
22
+ * one-shot.ts construct the ctx from — and agent/loop.ts already calls
23
+ * saveSession(session) end-of-turn (and session/pending.ts saves per
24
+ * mutating-tool WAL entry), so no extra save call is needed here. The
25
+ * emitter-plus-repl-side-subscriber alternative was rejected: it would
26
+ * add a second writer for state the tool already owns, and headless /
27
+ * one-shot paths (no subscriber mounted) would silently lose persistence.
28
+ *
29
+ * Validation discipline: structured error results (is_error: true), never
30
+ * throws — caps at 20 items, rejects empty lists, empty text, unknown
31
+ * status. MULTIPLE in_progress items are a WARNING in the result, not an
32
+ * error: blocking the model mid-job over a presentation nit would stall
33
+ * long workflows, and the list is still coherent and renderable — the
34
+ * warning text steers the model to fix it on its next call instead.
35
+ *
36
+ * isMutating: false — writes CLI-side session state only, never SAP.
37
+ * Headless: identical behaviour (session write + emitter absent/no-op;
38
+ * no prompt involved).
39
+ *
40
+ * Emitter seam (Task 5, 2026-07-05): the panel channel arrives via
41
+ * `ctx.todoEmitter` — NOT a module-level import — mirroring how
42
+ * chunkEmitter is ctx-carried. agent_run's child ctx omits it, so a
43
+ * subagent's todo_set updates only the CHILD session and can never
44
+ * clobber the parent's Ctrl+T panel.
45
+ */
46
+ import { registerTool } from './index.js';
47
+ import { formatTodoChecklistBlock } from '../renderer/todo-block.js';
48
+ /** Hard cap on list length — past this the list stops being a glanceable plan. */
49
+ export const TODO_MAX_ITEMS = 20;
50
+ const VALID_STATUSES = new Set(['pending', 'in_progress', 'completed']);
51
+ function errorResult(message) {
52
+ return { content: `error: ${message}`, is_error: true };
53
+ }
54
+ export async function todoSetHandler(args, ctx) {
55
+ const raw = args?.todos;
56
+ if (!Array.isArray(raw)) {
57
+ return errorResult("todos must be an array of { text, status } items (status: 'pending' | 'in_progress' | 'completed').");
58
+ }
59
+ if (raw.length === 0) {
60
+ return errorResult('todos must contain at least 1 item — this is a full-list replace, never call it with an empty list.');
61
+ }
62
+ if (raw.length > TODO_MAX_ITEMS) {
63
+ return errorResult(`todos exceeds the ${TODO_MAX_ITEMS}-item cap (got ${raw.length}). Collapse completed work or group related steps into one item.`);
64
+ }
65
+ // Validate + normalise BEFORE touching session state — a rejected call
66
+ // must leave the previous list fully intact.
67
+ const todos = [];
68
+ for (let i = 0; i < raw.length; i++) {
69
+ const item = raw[i];
70
+ const text = typeof item?.text === 'string' ? item.text.trim() : '';
71
+ if (!text) {
72
+ return errorResult(`todos[${i}].text must be a non-empty string.`);
73
+ }
74
+ if (typeof item?.status !== 'string' || !VALID_STATUSES.has(item.status)) {
75
+ return errorResult(`todos[${i}].status must be one of 'pending' | 'in_progress' | 'completed' (got ${JSON.stringify(item?.status ?? null)}).`);
76
+ }
77
+ todos.push({ text, status: item.status });
78
+ }
79
+ const inProgress = todos.filter((t) => t.status === 'in_progress');
80
+ // Full-list replace on the LIVE session object — persisted by the loop's
81
+ // existing end-of-turn saveSession (see module doc for the evidence).
82
+ ctx.session.todos = todos;
83
+ // UI fan-out (Task 5) — via the ctx-carried emitter (absent on subagent
84
+ // ctxs → suppressed; absent listener → harmless no-op). Emit a defensive
85
+ // copy so a consumer mutating the payload can never corrupt the
86
+ // persisted session state.
87
+ ctx.todoEmitter?.emit('update', todos.map((t) => ({ ...t })));
88
+ // Transcript render (CC-style, 2026-07): drop the FULL checklist into
89
+ // scrollback on every successful set. The raw ⏺/⎿ todo_set rows are
90
+ // widget-suppressed (renderer/tool-widget.ts), so this block is the sole
91
+ // visible representation — mirroring ask_question. ctx.chunkEmitter routes
92
+ // to Ink Static scrollback AND the classic 'chunk'→stdout path
93
+ // (renderer/notices.ts). Subagent ctxs deliberately omit chunkEmitter
94
+ // (tools/subagent/agent_run.ts), so a child turn's todo_set can never
95
+ // print its checklist into the PARENT transcript — same seam as the
96
+ // todoEmitter suppression above.
97
+ ctx.chunkEmitter?.emit('chunk', formatTodoChecklistBlock(todos));
98
+ const result = {
99
+ count: todos.length,
100
+ in_progress: inProgress[0]?.text ?? null,
101
+ };
102
+ if (inProgress.length > 1) {
103
+ result.warning =
104
+ `${inProgress.length} items are in_progress — exactly one item should be in_progress at a time. ` +
105
+ 'Mark the others pending or completed on your next todo_set call.';
106
+ }
107
+ return { content: JSON.stringify(result) };
108
+ }
109
+ registerTool({
110
+ name: 'todo_set',
111
+ // Model-behavior lever — this exact framing is binding (task-4 brief).
112
+ description: 'Maintain your task list for multi-step work. Call with the FULL updated list whenever you ' +
113
+ 'complete a step, start a new one, or discover new work. Exactly one item should be ' +
114
+ 'in_progress at a time. The user sees this list — keep items short and outcome-shaped.',
115
+ isMutating: false,
116
+ category: 'session',
117
+ input_schema: {
118
+ type: 'object',
119
+ properties: {
120
+ todos: {
121
+ type: 'array',
122
+ minItems: 1,
123
+ maxItems: TODO_MAX_ITEMS,
124
+ description: 'The full task list — replaces the previous list entirely.',
125
+ items: {
126
+ type: 'object',
127
+ properties: {
128
+ text: {
129
+ type: 'string',
130
+ description: 'Short, outcome-shaped item, e.g. "Activate ZCL_ORDER_API".',
131
+ },
132
+ status: {
133
+ type: 'string',
134
+ enum: ['pending', 'in_progress', 'completed'],
135
+ },
136
+ },
137
+ required: ['text', 'status'],
138
+ },
139
+ },
140
+ },
141
+ required: ['todos'],
142
+ },
143
+ handler: (args, ctx) => todoSetHandler(args, ctx),
144
+ });
@@ -0,0 +1,86 @@
1
+ /**
2
+ * transport-resolution — owning-transport check before SAP writes (Rule 9,
3
+ * battery defects D15/D20).
4
+ *
5
+ * The trap this kills: an object (or one of its LIMU pieces) is already
6
+ * locked in an open transport request. Writing under ANY other transport is
7
+ * guaranteed to fail with a CTS lock conflict (CTS_WBO_API 020 — observed
8
+ * live: "Object LIMU CINC ZBP_I_DOWNTIMELOG…CCIMP is already locked in
9
+ * request S4HK903431"). Asking the human to pick from N open transports
10
+ * (D15: 89 of them) or creating a fresh junk transport (D20) cannot succeed.
11
+ *
12
+ * Correct flow, now structural instead of prose-only:
13
+ * - object locked in exactly one open request → USE that transport
14
+ * silently, overriding a conflicting supplied transport (with a notice);
15
+ * - object locked in several open requests → ambiguous, never force —
16
+ * keep the supplied transport and surface the candidates;
17
+ * - object free / lookup failed → current behaviour (supplied transport).
18
+ *
19
+ * The lookup is one E071+E070 query pair (AdtClient.transportForObject) and
20
+ * is strictly best-effort: a failed lookup (VPN drop, data-preview auth,
21
+ * older mocks without the method) must NEVER block the write.
22
+ */
23
+ import { wasTransportCreatedThisSession } from './transport.js';
24
+ /**
25
+ * Resolve the transport for a write against the owning-transport ledger.
26
+ *
27
+ * @param ctx tool context (uses ctx.adt + ctx.chunkEmitter)
28
+ * @param objectType ABAP object type of the write target (CLAS, PROG, …)
29
+ * @param objectName ABAP object name of the write target
30
+ * @param supplied the transport the tool would use today (each tool keeps
31
+ * its own args/session fallback semantics and passes the
32
+ * outcome here)
33
+ */
34
+ export async function resolveWriteTransport(ctx, objectType, objectName, supplied) {
35
+ // TODO(perf): per-turn memo. Every write funnels through this lookup
36
+ // (2 datapreview SQL queries inside transportForObject), so a multi-method
37
+ // turn — N sap_update_method calls against the same class — amplifies to
38
+ // N×2 queries for an answer that cannot change mid-turn. Memoise per turn,
39
+ // keyed `${objectType}:${objectName}`, before the battery grows
40
+ // write-heavy scenarios. No implementation yet.
41
+ let owning = null;
42
+ try {
43
+ owning = await ctx.adt.transportForObject(objectType, objectName);
44
+ }
45
+ catch (err) {
46
+ // Best-effort only — never block or delay the write on a failed lookup,
47
+ // but leave a dim trace so "why didn't it auto-resolve?" is diagnosable
48
+ // ('info' channel = non-fatal diagnostics; 'warn' is reserved for the
49
+ // conflicting-override case below).
50
+ ctx.chunkEmitter?.emit('info', `owning-transport lookup failed for ${objectType} ${objectName} — `
51
+ + `proceeding with ${supplied ?? 'no transport'} `
52
+ + `(${err instanceof Error ? err.message : String(err)})`);
53
+ return { transport: supplied };
54
+ }
55
+ if (owning?.transport) {
56
+ const conflicting = !!supplied && supplied !== owning.transport;
57
+ const note = conflicting
58
+ ? `${objectName} is already locked in ${owning.transport} — writing under the owning transport and ignoring transport=${supplied} (SAP would refuse with a CTS lock conflict)`
59
+ : `object locked in ${owning.transport} — writing under owning transport`;
60
+ // warn only when we OVERRIDE what the caller asked for; the happy match
61
+ // (supplied === owning, or nothing supplied) is informational.
62
+ ctx.chunkEmitter?.emit(conflicting ? 'warn' : 'info', note);
63
+ // Orphan hint (third junk TR this month, 2026-06-12): the session's
64
+ // current transport was created by sap_transport_create THIS session, yet
65
+ // this write was forced into a different (owning) request — the created
66
+ // TR is probably an empty orphan. Warn only; deleting a TR is destructive
67
+ // and stays with the human.
68
+ const sessionCurrent = ctx.currentTransport?.get() ?? null;
69
+ if (sessionCurrent
70
+ && sessionCurrent !== owning.transport
71
+ && wasTransportCreatedThisSession(sessionCurrent)) {
72
+ ctx.chunkEmitter?.emit('warn', `transport ${sessionCurrent} created this session was not used by this write — `
73
+ + 'it may be an empty orphan (delete via SE09 or ask me)');
74
+ }
75
+ return { transport: owning.transport, owning: owning.transport, note };
76
+ }
77
+ if (owning && owning.candidates.length > 1) {
78
+ const note = `${objectName} pieces are locked in ${owning.candidates.length} open transports `
79
+ + `(${owning.candidates.join(', ')}) — cannot auto-resolve the owning request; `
80
+ + `using ${supplied ?? 'no transport'}`;
81
+ ctx.chunkEmitter?.emit('warn', note);
82
+ return { transport: supplied, note };
83
+ }
84
+ // Object is free — current behaviour.
85
+ return { transport: supplied };
86
+ }