@cspeach/cli 0.9.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/agent/intent-system-prompt.js +1 -1
- package/dist/agent/loop.js +228 -26
- package/dist/agent/providers/license-gate.js +44 -0
- package/dist/agent/skill-checkpoint.js +1 -1
- package/dist/agent/tool-dispatch.js +15 -0
- package/dist/approvals/canonical.js +91 -0
- package/dist/approvals/jwt.js +39 -2
- package/dist/approvals/op-labels.js +124 -0
- package/dist/approvals/render.js +42 -36
- package/dist/auth/org-anthropic-key.js +25 -0
- package/dist/classifier/client.js +18 -3
- package/dist/cli.js +15 -0
- package/dist/commands/compact.js +28 -2
- package/dist/commands/config-set.js +284 -0
- package/dist/commands/config-show.js +20 -0
- package/dist/commands/export-audit.js +43 -0
- package/dist/commands/help.js +5 -0
- package/dist/commands/login.js +31 -14
- package/dist/commands/plan-audit-evidence.js +266 -0
- package/dist/commands/plan-audit.js +692 -0
- package/dist/commands/plan-chain.js +671 -0
- package/dist/commands/plan-continue.js +179 -0
- package/dist/commands/plan-gate.js +154 -0
- package/dist/commands/plan-model-tier.js +83 -0
- package/dist/commands/plan-resume.js +728 -46
- package/dist/config/loader.js +223 -5
- package/dist/config/model-defaults.js +14 -0
- package/dist/cost/pricing.js +27 -1
- package/dist/doctor/checks/_http-probe.js +1 -0
- package/dist/doctor/checks/cert.js +14 -3
- package/dist/doctor/checks/sap.js +30 -8
- package/dist/doctor/checks/system-roles.js +41 -0
- package/dist/doctor/checks/zcspeach.js +19 -4
- package/dist/doctor/run.js +2 -0
- package/dist/models/resolve.js +61 -0
- package/dist/models/server-config.js +155 -0
- package/dist/one-shot.js +76 -6
- package/dist/projects/answer-blockers.js +137 -0
- package/dist/projects/extract-cca.js +111 -17
- package/dist/projects/extract-modernize.js +4 -2
- package/dist/projects/extract-plan.js +184 -37
- package/dist/projects/extract-spec-gap.js +34 -7
- package/dist/projects/extract-test-coverage.js +4 -2
- package/dist/projects/extract-upgrade.js +116 -23
- package/dist/projects/handover-md.js +195 -0
- package/dist/projects/index.js +5 -2
- package/dist/projects/merge-cca.js +292 -0
- package/dist/projects/merge-upgrade.js +173 -0
- package/dist/projects/migration.js +103 -1
- package/dist/projects/output-paths.js +27 -0
- package/dist/projects/plan-run.js +285 -27
- package/dist/projects/plan-schema.js +136 -3
- package/dist/projects/promote-command.js +25 -2
- package/dist/projects/promote.js +128 -0
- package/dist/projects/run-lease.js +157 -0
- package/dist/projects/save-command.js +259 -21
- package/dist/projects/status.js +3 -1
- package/dist/projects/validate.js +1 -1
- package/dist/projects/workspace.js +164 -20
- package/dist/renderer/notices.js +64 -0
- package/dist/renderer/progress-chatter.js +8 -0
- package/dist/renderer/status-footer.js +22 -12
- package/dist/renderer/thinking-heartbeat.js +64 -8
- package/dist/renderer/todo-block.js +51 -0
- package/dist/renderer/tool-widget.js +55 -4
- package/dist/renderer/tty.js +43 -4
- package/dist/renderer/verify-chain.js +77 -0
- package/dist/repl/at-picker.js +60 -7
- package/dist/repl/bracketed-paste.js +28 -19
- package/dist/repl/builtin-commands.js +42 -0
- package/dist/repl/current-transport.js +10 -0
- package/dist/repl/early-line-buffer.js +68 -0
- package/dist/repl/history.js +86 -0
- package/dist/repl/ink-stdin-guard.js +64 -0
- package/dist/repl/inquirer-guard.js +70 -5
- package/dist/repl/mode-ceiling.js +16 -0
- package/dist/repl/mode-cycle.js +104 -0
- package/dist/repl/numbered-menu.js +131 -0
- package/dist/repl/post-turn-status.js +26 -6
- package/dist/repl/rule8-detector.js +17 -2
- package/dist/repl/safety-confirm.js +111 -2
- package/dist/repl/safety-mode-state.js +19 -3
- package/dist/repl/slash-completer.js +5 -0
- package/dist/repl/slash-picker.js +10 -15
- package/dist/repl.js +1232 -95
- package/dist/rewind/candidates.js +194 -0
- package/dist/rewind/cli.js +137 -0
- package/dist/rewind/format.js +27 -0
- package/dist/rewind/restore.js +245 -0
- package/dist/router/classifier.js +150 -6
- package/dist/sap/capability-matrix.js +20 -0
- package/dist/sap/capability-matrix.json +11236 -0
- package/dist/sap/capability.js +146 -0
- package/dist/sap/connection-manager.js +19 -1
- package/dist/sap/onboarding.js +42 -4
- package/dist/session/audit-export.js +459 -0
- package/dist/session/context-report.js +163 -0
- package/dist/session/pending.js +27 -0
- package/dist/session/recap.js +160 -0
- package/dist/skill-catalog.js +51 -40
- package/dist/skills/bundled-skills.js +272 -1
- package/dist/skills/promotion-dispatch.js +23 -0
- package/dist/tools/_command-shared.js +36 -12
- package/dist/tools/_filesystem-shared.js +139 -4
- package/dist/tools/_flag.js +25 -0
- package/dist/tools/approval.js +177 -26
- package/dist/tools/ask-question.js +400 -7
- package/dist/tools/capability/tool.js +74 -0
- package/dist/tools/dispatch-skill.js +22 -1
- package/dist/tools/extend-model/anchored-insert.js +1414 -0
- package/dist/tools/extend-model/tool.js +340 -0
- package/dist/tools/filesystem/extract-document.js +57 -0
- package/dist/tools/filesystem/file-edit.js +12 -2
- package/dist/tools/filesystem/file-read.js +2 -2
- package/dist/tools/filesystem/file-write.js +11 -2
- package/dist/tools/filesystem/glob.js +11 -0
- package/dist/tools/filesystem/grep.js +10 -0
- package/dist/tools/filesystem/read-document.js +107 -0
- package/dist/tools/fiori/apply.js +50 -0
- package/dist/tools/fiori/bin.js +3 -0
- package/dist/tools/fiori/catalog/index.js +27 -0
- package/dist/tools/fiori/catalog/value-help.js +230 -0
- package/dist/tools/fiori/catalog/viz-chart.js +177 -0
- package/dist/tools/fiori/cli.js +71 -0
- package/dist/tools/fiori/deploy-config.js +73 -0
- package/dist/tools/fiori/fe-extend.js +76 -0
- package/dist/tools/fiori/fe-scaffold.js +71 -0
- package/dist/tools/fiori/floorplan-map.js +19 -0
- package/dist/tools/fiori/i18n.js +39 -0
- package/dist/tools/fiori/manifest.js +70 -0
- package/dist/tools/fiori/render.js +77 -0
- package/dist/tools/fiori/samples/data/index.json +13602 -0
- package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
- package/dist/tools/fiori/samples/loader.js +248 -0
- package/dist/tools/fiori/samples/search.js +63 -0
- package/dist/tools/fiori/samples/types.js +2 -0
- package/dist/tools/fiori/scaffold.js +39 -0
- package/dist/tools/fiori/smoke/assertions.js +74 -0
- package/dist/tools/fiori/smoke/browser.js +52 -0
- package/dist/tools/fiori/smoke/driver.js +89 -0
- package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
- package/dist/tools/fiori/smoke/run-smoke.js +149 -0
- package/dist/tools/fiori/tools.js +681 -0
- package/dist/tools/fiori/types.js +1 -0
- package/dist/tools/local-build.js +86 -0
- package/dist/tools/local-files.js +31 -0
- package/dist/tools/project/_merge-shared.js +68 -0
- package/dist/tools/project/cca_merge.js +164 -0
- package/dist/tools/project/playbook_get.js +1 -1
- package/dist/tools/project/upgrade_merge_progress.js +206 -0
- package/dist/tools/sap-read.js +132 -20
- package/dist/tools/sap-write.js +550 -21
- package/dist/tools/shell/shell_exec.js +41 -6
- package/dist/tools/snapshot.js +63 -14
- package/dist/tools/subagent/agent_run.js +27 -3
- package/dist/tools/subagent/background_run.js +17 -1
- package/dist/tools/todo.js +144 -0
- package/dist/tools/transport-resolution.js +86 -0
- package/dist/tools/transport.js +224 -5
- package/dist/tools/write-mode.js +4 -0
- package/dist/ui/app.js +378 -21
- package/dist/ui/approval-modal.js +49 -16
- package/dist/ui/ask-question-emitter.js +14 -0
- package/dist/ui/body.js +13 -0
- package/dist/ui/context-grid.js +108 -0
- package/dist/ui/footer.js +120 -27
- package/dist/ui/header.js +7 -0
- package/dist/ui/line-resolution.js +35 -8
- package/dist/ui/rewind-emitter.js +10 -0
- package/dist/ui/rewind-panel.js +81 -0
- package/dist/ui/sap-state-store.js +1 -0
- package/dist/ui/session-timeline.js +1 -0
- package/dist/ui/status-line.js +43 -0
- package/dist/ui/text-input.js +214 -0
- package/dist/ui/todo-emitter.js +25 -0
- package/dist/ui/todo-panel.js +64 -0
- package/dist/ui/turn-status-emitter.js +50 -4
- package/dist/ui/turn-status.js +18 -3
- package/dist/ui/widgets/ask-form.js +242 -0
- package/dist/ui/widgets/ask-question-modal.js +21 -8
- package/package.json +22 -3
- package/bench/README.md +0 -78
- package/bench/prompts/abap-document-cds.md +0 -44
- package/bench/prompts/abap-explain-bdef-handler.md +0 -57
- package/bench/prompts/abap-test-method.md +0 -42
- package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
- package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
- package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
- package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
- package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
- package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
- package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
- package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
- package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
|
@@ -8,11 +8,26 @@
|
|
|
8
8
|
* allow = ["docker", "make"]
|
|
9
9
|
* block in ~/.cspeach/config.toml.
|
|
10
10
|
*
|
|
11
|
-
* Cross-platform
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
11
|
+
* Cross-platform spawn via cross-spawn (the library npm itself uses).
|
|
12
|
+
* The actual spawn is handed the ABSOLUTE PATH-resolved executable path
|
|
13
|
+
* (from resolveExecutable's PATH-only walk) + argv ARRAY. cross-spawn
|
|
14
|
+
* still inspects the file extension and, on Windows, invokes .cmd / .bat
|
|
15
|
+
* through cmd.exe with correct per-arg escaping. This satisfies Node's
|
|
16
|
+
* CVE-2024-27980 restriction (spawn with shell:false refuses to launch
|
|
17
|
+
* .cmd/.bat directly) WITHOUT ever setting shell:true — args stay an
|
|
18
|
+
* array, so shell metacharacters are inert and there is no injection
|
|
19
|
+
* surface. On Unix cross-spawn is a near-passthrough to
|
|
20
|
+
* child_process.spawn (unchanged behavior).
|
|
21
|
+
*
|
|
22
|
+
* SECURITY — why the absolute path, not the bare name: cross-spawn
|
|
23
|
+
* resolves a BARE command name cwd-FIRST on Windows (node_modules/which
|
|
24
|
+
* checks process.cwd() before PATH on win32). A model with an in-tree
|
|
25
|
+
* write primitive could plant `<safelisted>.cmd` in the project tree and
|
|
26
|
+
* have cross-spawn pick it up ahead of the real PATH binary → arbitrary
|
|
27
|
+
* code execution, safelist bypassed. Spawning the absolute PATH-resolved
|
|
28
|
+
* path (which never consults cwd/PATH for a path-bearing command) closes
|
|
29
|
+
* that vector. The pre-flight resolveExecutable() PATH walk and the spawn
|
|
30
|
+
* now use the SAME binary — no pre-flight/spawn divergence.
|
|
16
31
|
*
|
|
17
32
|
* Operational caps:
|
|
18
33
|
* - 30 s wall-clock timeout (overridable up to 300 s via timeout_ms arg)
|
|
@@ -24,7 +39,7 @@
|
|
|
24
39
|
* Flag-gated: invisible to listTools() unless CSPEACH_TOOL_SHELL_EXEC=on.
|
|
25
40
|
* isMutating: true — hooks into the existing approval gate.
|
|
26
41
|
*/
|
|
27
|
-
import
|
|
42
|
+
import spawn from 'cross-spawn';
|
|
28
43
|
import { promises as fs } from 'node:fs';
|
|
29
44
|
import * as path from 'node:path';
|
|
30
45
|
import { registerTool } from '../index.js';
|
|
@@ -80,6 +95,9 @@ export async function shellExecHandler(args, ctx) {
|
|
|
80
95
|
catch {
|
|
81
96
|
return { content: `error: cwd "${args.cwd ?? '.'}" does not exist`, is_error: true };
|
|
82
97
|
}
|
|
98
|
+
// Resolve the bare safelisted name to an absolute executable path via a
|
|
99
|
+
// PATH-only walk (no cwd-first). This is BOTH the friendly "not installed"
|
|
100
|
+
// pre-flight check AND the spawn target below — same binary, no divergence.
|
|
83
101
|
const resolved = await resolveExecutable(args.command);
|
|
84
102
|
if (resolved === null) {
|
|
85
103
|
return {
|
|
@@ -101,6 +119,23 @@ export async function shellExecHandler(args, ctx) {
|
|
|
101
119
|
settled = true;
|
|
102
120
|
resolve(result);
|
|
103
121
|
};
|
|
122
|
+
// Pass the already-validated absolute `resolved` path (from
|
|
123
|
+
// resolveExecutable's PATH-only walk) — NOT the bare command name.
|
|
124
|
+
//
|
|
125
|
+
// SECURITY: cross-spawn resolves a BARE name cwd-first on Windows
|
|
126
|
+
// (node_modules/which checks process.cwd() before PATH on win32, and
|
|
127
|
+
// it chdir's into options.cwd first). A model with a write primitive
|
|
128
|
+
// could plant `<safelisted>.cmd` (e.g. git.cmd) in the project tree and
|
|
129
|
+
// have cross-spawn pick it up cwd-first → arbitrary code execution,
|
|
130
|
+
// safelist bypassed. Handing cross-spawn the absolute PATH-resolved
|
|
131
|
+
// path closes that: `which` short-circuits to the literal file when the
|
|
132
|
+
// command contains a path separator (pathEnv=['']), so cwd/PATH are
|
|
133
|
+
// never consulted. cross-spawn STILL inspects the file extension and
|
|
134
|
+
// wraps an absolute `.cmd`/`.bat` through cmd.exe with proper per-arg
|
|
135
|
+
// escaping (parse.js isExecutableRegExp only matches .com/.exe), so the
|
|
136
|
+
// Windows shim handling + CVE-2024-27980 .cmd restriction handling are
|
|
137
|
+
// preserved. shell:false stays (argv-only; metacharacters inert). The
|
|
138
|
+
// existence check above and this spawn now target the SAME binary.
|
|
104
139
|
const child = spawn(resolved, argv, {
|
|
105
140
|
cwd,
|
|
106
141
|
env: buildSafeEnv(),
|
package/dist/tools/snapshot.js
CHANGED
|
@@ -14,7 +14,18 @@
|
|
|
14
14
|
* cleanup(retentionHours?) → Promise<number>
|
|
15
15
|
*/
|
|
16
16
|
import { registerTool } from './index.js';
|
|
17
|
-
import { snapshots } from '@cspeach/sap-client';
|
|
17
|
+
import { snapshots, SapError } from '@cspeach/sap-client';
|
|
18
|
+
/** True only for a genuine ADT 404 — the object does not exist. Any other
|
|
19
|
+
* failure (network, auth, 5xx, timeout) is a READ failure, not proof of
|
|
20
|
+
* absence, and must never be labelled "object not found". */
|
|
21
|
+
function isNotFound(err) {
|
|
22
|
+
return err instanceof SapError && err.httpStatus === 404;
|
|
23
|
+
}
|
|
24
|
+
/** Short single-line failure message for transcripts / error envelopes. */
|
|
25
|
+
function shortErrorMessage(err) {
|
|
26
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
27
|
+
return msg.replace(/\s+/g, ' ').trim();
|
|
28
|
+
}
|
|
18
29
|
registerTool({
|
|
19
30
|
name: 'sap_snapshot_take',
|
|
20
31
|
description: "Take a snapshot of an object's current source before modification. Usually invoked automatically by the CLI before any write — rarely called directly by the model.",
|
|
@@ -28,25 +39,63 @@ registerTool({
|
|
|
28
39
|
required: ['name', 'type'],
|
|
29
40
|
},
|
|
30
41
|
handler: async (args, ctx) => {
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
42
|
+
let source;
|
|
43
|
+
try {
|
|
44
|
+
source = await ctx.adt.getSource(args.type, args.name);
|
|
45
|
+
}
|
|
46
|
+
catch (err) {
|
|
47
|
+
if (isNotFound(err)) {
|
|
48
|
+
return { content: JSON.stringify({ skipped: true, reason: 'object_not_found' }) };
|
|
49
|
+
}
|
|
50
|
+
// Read failed for some OTHER reason — the object may well exist.
|
|
51
|
+
// Claiming "object_not_found" here would be a lie; surface the failure.
|
|
52
|
+
return {
|
|
53
|
+
content: JSON.stringify({ error: 'source_read_failed', detail: shortErrorMessage(err) }),
|
|
54
|
+
is_error: true,
|
|
55
|
+
};
|
|
34
56
|
}
|
|
35
57
|
const entry = await snapshots.take(args.type, args.name, source, 'manual');
|
|
36
58
|
return { content: JSON.stringify({ snapshot: entry, size: source.length }) };
|
|
37
59
|
},
|
|
38
60
|
});
|
|
39
|
-
/**
|
|
40
|
-
* Auto-snapshot helper called by write tools before mutating an object.
|
|
41
|
-
* Silently skips if the object does not exist yet (first write after create).
|
|
42
|
-
*
|
|
43
|
-
* snapshots.take expects (type, name, source, reason).
|
|
44
|
-
*/
|
|
45
61
|
export async function autoSnapshot(objectName, objectType, adt) {
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
62
|
+
let src;
|
|
63
|
+
try {
|
|
64
|
+
src = await adt.getSource(objectType, objectName);
|
|
65
|
+
}
|
|
66
|
+
catch (err) {
|
|
67
|
+
if (isNotFound(err))
|
|
68
|
+
return { taken: false, reason: 'object_not_found' }; // new object
|
|
69
|
+
// The DEFAULT read failed for a non-404 reason. Before treating this as a
|
|
70
|
+
// hard read failure (which blocks the write, Rule 7), distinguish an
|
|
71
|
+
// empty/never-activated SHELL from a real object whose source we couldn't
|
|
72
|
+
// read. A freshly-created behaviour-pool CLAS (ZBP … FOR BEHAVIOR OF …)
|
|
73
|
+
// EXISTS but has no committed source yet, so the default read returns HTTP
|
|
74
|
+
// 400 — not 404. Such a shell has NO committed ACTIVE version. Probe for one:
|
|
75
|
+
// - active read 404 → no active version ⇒ fresh shell, nothing to
|
|
76
|
+
// snapshot ⇒ proceed ('new_shell').
|
|
77
|
+
// - active read succeeds → there IS committed source to protect ⇒
|
|
78
|
+
// snapshot THAT active source and proceed.
|
|
79
|
+
// - active read fails otherwise → can't confirm emptiness ⇒ conservative
|
|
80
|
+
// block ('source_read_failed', unchanged).
|
|
81
|
+
let activeSrc;
|
|
82
|
+
try {
|
|
83
|
+
activeSrc = await adt.getSource(objectType, objectName, 'active');
|
|
84
|
+
}
|
|
85
|
+
catch (activeErr) {
|
|
86
|
+
if (isNotFound(activeErr))
|
|
87
|
+
return { taken: false, reason: 'new_shell' };
|
|
88
|
+
// Could not confirm the object is empty — keep the safety property: a
|
|
89
|
+
// real object with source we couldn't read must still block the write.
|
|
90
|
+
return { taken: false, reason: 'source_read_failed', detail: shortErrorMessage(err) };
|
|
91
|
+
}
|
|
92
|
+
// Committed active version exists — snapshot it. A snapshots.take failure
|
|
93
|
+
// still THROWS (Rule 7 hard-fail), same as the clean-read path below.
|
|
94
|
+
const activeEntry = await snapshots.take(objectType, objectName, activeSrc, 'before_write');
|
|
95
|
+
return { taken: true, entry: activeEntry };
|
|
96
|
+
}
|
|
97
|
+
const entry = await snapshots.take(objectType, objectName, src, 'before_write');
|
|
98
|
+
return { taken: true, entry };
|
|
50
99
|
}
|
|
51
100
|
// ── sap_snapshot_list ─────────────────────────────────────────────────────────
|
|
52
101
|
registerTool({
|
|
@@ -28,6 +28,7 @@
|
|
|
28
28
|
*/
|
|
29
29
|
import { EventEmitter } from 'node:events';
|
|
30
30
|
import { registerTool } from '../index.js';
|
|
31
|
+
import { isPlanPhaseActive, notePlanPhaseSubagentDispatch, } from '../../commands/plan-gate.js';
|
|
31
32
|
import { runTurn } from '../../agent/loop.js';
|
|
32
33
|
import { collectTurnAssistantText } from '../../agent/turn-assistant-text.js';
|
|
33
34
|
export const MAX_AGENT_DEPTH = 3;
|
|
@@ -98,12 +99,20 @@ export async function agentRunHandler(args, ctx) {
|
|
|
98
99
|
// not parent's start time. Avoids surprising file timestamps.
|
|
99
100
|
started_at: new Date().toISOString(),
|
|
100
101
|
last_turn_at: new Date().toISOString(),
|
|
102
|
+
// Task 5 (ux-wave2): a fresh subagent has no task list — without this
|
|
103
|
+
// the spread above would leak the PARENT's todos into the child's
|
|
104
|
+
// session file.
|
|
105
|
+
todos: undefined,
|
|
101
106
|
};
|
|
102
107
|
// Build an isolated child ctx. Crucial differences vs parent ctx:
|
|
103
108
|
// - session: the fresh child session above (independent usage counters)
|
|
104
109
|
// - previewHook: undefined — subagent must NOT auto-approve writes
|
|
105
110
|
// - pendingDispatch / currentTransport: cleared — subagent doesn't
|
|
106
111
|
// mutate parent REPL state
|
|
112
|
+
// - todoEmitter: omitted (Task 5) — todo_set emits via
|
|
113
|
+
// ctx.todoEmitter?.emit, so a child's todo_set updates only
|
|
114
|
+
// childSession.todos and never repaints the PARENT's Ctrl+T panel /
|
|
115
|
+
// turn-status strip. Same seam as chunkEmitter suppression.
|
|
107
116
|
// chunkEmitter is passed via the RunTurn params (not the ctx) as a
|
|
108
117
|
// null-sink EventEmitter — see I2 fix below.
|
|
109
118
|
const childCtx = {
|
|
@@ -113,7 +122,8 @@ export async function agentRunHandler(args, ctx) {
|
|
|
113
122
|
cwd: ctx.cwd,
|
|
114
123
|
provider: ctx.provider,
|
|
115
124
|
skillSource: ctx.skillSource,
|
|
116
|
-
// previewHook + pendingDispatch + currentTransport
|
|
125
|
+
// previewHook + pendingDispatch + currentTransport + todoEmitter
|
|
126
|
+
// intentionally omitted.
|
|
117
127
|
};
|
|
118
128
|
// I2 fix: a null-sink EventEmitter suppresses all subagent output. Every
|
|
119
129
|
// stdout-write site in loop.ts is guarded by `if (params.chunkEmitter)`,
|
|
@@ -128,9 +138,23 @@ export async function agentRunHandler(args, ctx) {
|
|
|
128
138
|
// mutating tool from the subagent's tool list. Saves tokens on the
|
|
129
139
|
// sub-agent invocation (smaller tools array = less input cost) AND
|
|
130
140
|
// hard-prevents accidental writes from the subagent.
|
|
131
|
-
|
|
141
|
+
//
|
|
142
|
+
// Task 7 (agentic-flow, 2026-07-03) — inside a plan phase the filter is
|
|
143
|
+
// FORCED regardless of args.read_only: subagents can read, never write,
|
|
144
|
+
// and the model cannot opt out (read_only:false is ignored). Outside plan
|
|
145
|
+
// phases the opt-in behaviour above is byte-identical to before.
|
|
146
|
+
const planPhase = isPlanPhaseActive();
|
|
147
|
+
const readOnly = planPhase || Boolean(args.read_only);
|
|
148
|
+
const toolFilter = readOnly
|
|
132
149
|
? (t) => !t.isMutating
|
|
133
150
|
: undefined;
|
|
151
|
+
if (planPhase)
|
|
152
|
+
notePlanPhaseSubagentDispatch();
|
|
153
|
+
// Task 7 — per-dispatch announce on the PARENT's emitter (childCtx
|
|
154
|
+
// deliberately omits chunkEmitter, so this is the one line the user sees
|
|
155
|
+
// for each dispatch). 'info' renders dim in both the Ink and classic REPLs.
|
|
156
|
+
const taskPreview = args.task.length > 60 ? `${args.task.slice(0, 60)}…` : args.task;
|
|
157
|
+
ctx.chunkEmitter?.emit('info', `↳ subagent: ${args.skill}${readOnly ? ' — read-only' : ''} — ${taskPreview}`);
|
|
134
158
|
try {
|
|
135
159
|
await runTurn({
|
|
136
160
|
provider: ctx.provider,
|
|
@@ -189,7 +213,7 @@ registerTool({
|
|
|
189
213
|
},
|
|
190
214
|
read_only: {
|
|
191
215
|
type: 'boolean',
|
|
192
|
-
description: 'When true, the subagent only sees read-only tools (no sap_set_source, no file-write, no shell_exec, etc.). Use this for research / analysis subagents that should not make changes — e.g. spawning a subagent to read code, run greps, or summarise findings. Smaller tools array = lower per-call cost; also a hard guard against accidental writes.',
|
|
216
|
+
description: 'When true, the subagent only sees read-only tools (no sap_set_source, no file-write, no shell_exec, etc.). Use this for research / analysis subagents that should not make changes — e.g. spawning a subagent to read code, run greps, or summarise findings. Smaller tools array = lower per-call cost; also a hard guard against accidental writes. NOTE: during plan phases the read-only filter is ALWAYS enforced regardless of this argument.',
|
|
193
217
|
},
|
|
194
218
|
},
|
|
195
219
|
required: ['skill', 'task'],
|
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
* CSPEACH_TOOL_BACKGROUND_RUN=on. isMutating: true — hooks the existing
|
|
17
17
|
* approval gate (same UX as shell_exec).
|
|
18
18
|
*/
|
|
19
|
-
import
|
|
19
|
+
import spawn from 'cross-spawn';
|
|
20
20
|
import { promises as fs } from 'node:fs';
|
|
21
21
|
import * as path from 'node:path';
|
|
22
22
|
import { randomUUID } from 'node:crypto';
|
|
@@ -69,6 +69,9 @@ export async function backgroundRunHandler(args, ctx) {
|
|
|
69
69
|
catch {
|
|
70
70
|
return { content: `error: cwd "${args.cwd ?? '.'}" does not exist`, is_error: true };
|
|
71
71
|
}
|
|
72
|
+
// Resolve the bare safelisted name to an absolute executable path via a
|
|
73
|
+
// PATH-only walk (no cwd-first). Both the friendly "not installed"
|
|
74
|
+
// pre-flight check AND the spawn target below — same binary, no divergence.
|
|
72
75
|
const resolved = await resolveExecutable(args.command);
|
|
73
76
|
if (resolved === null) {
|
|
74
77
|
return {
|
|
@@ -78,6 +81,19 @@ export async function backgroundRunHandler(args, ctx) {
|
|
|
78
81
|
}
|
|
79
82
|
let child;
|
|
80
83
|
try {
|
|
84
|
+
// Absolute PATH-resolved `resolved` (NOT the bare command name).
|
|
85
|
+
//
|
|
86
|
+
// SECURITY: cross-spawn resolves a BARE name cwd-first on Windows
|
|
87
|
+
// (node_modules/which checks process.cwd() before PATH and chdir's into
|
|
88
|
+
// options.cwd first). A model with a write primitive could plant
|
|
89
|
+
// `<safelisted>.cmd` in the project tree and have it picked up cwd-first
|
|
90
|
+
// → arbitrary code execution, safelist bypassed. Handing cross-spawn the
|
|
91
|
+
// absolute PATH-resolved path closes that (`which` short-circuits to the
|
|
92
|
+
// literal file for path-bearing commands, pathEnv=['']). cross-spawn
|
|
93
|
+
// still wraps an absolute `.cmd`/`.bat` through cmd.exe with per-arg
|
|
94
|
+
// escaping, so Windows shim + CVE-2024-27980 handling stay. shell:false
|
|
95
|
+
// → metacharacters inert. Existence check + spawn now target the SAME
|
|
96
|
+
// binary. See shell_exec.ts for the full rationale.
|
|
81
97
|
child = spawn(resolved, argv, {
|
|
82
98
|
cwd,
|
|
83
99
|
env: buildSafeEnv(),
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* todo_set — UX Wave 2 / Task 4. CC's TodoWrite adapted for CSPeach: the
|
|
3
|
+
* model maintains a visible task list on long multi-step jobs (RAP stacks,
|
|
4
|
+
* upgrade batches, plan phases). Task 5 renders it (Ctrl+T panel +
|
|
5
|
+
* turn-status strip) via todoEmitter; /compact appends it to the summary
|
|
6
|
+
* block so the plan survives compaction (commands/compact.ts).
|
|
7
|
+
*
|
|
8
|
+
* Semantics: FULL-LIST REPLACE on every call — never a delta. 1..20 items.
|
|
9
|
+
*
|
|
10
|
+
* Transcript render (CC-style, 2026-07): on every successful set the
|
|
11
|
+
* handler also emits the FULL checklist as a scrollback block via
|
|
12
|
+
* ctx.chunkEmitter (renderer/todo-block.ts). The raw ⏺/⎿ todo_set rows are
|
|
13
|
+
* widget-suppressed (renderer/tool-widget.ts) so the block is the sole
|
|
14
|
+
* visible representation — the transcript now shows the whole plan
|
|
15
|
+
* advancing, not a raw `⏺ todo_set({"todos":[…])` JSON line. This is the
|
|
16
|
+
* Claude-Code parity the owner asked for; the turn-status strip (single ▶
|
|
17
|
+
* row) and Ctrl+T panel stay exactly as they were.
|
|
18
|
+
*
|
|
19
|
+
* Session-write route (chosen with evidence, 2026-07-05): the tool mutates
|
|
20
|
+
* `ctx.session.todos` directly. ToolContext.session IS the live
|
|
21
|
+
* SessionState object (tools/index.ts) — the same instance repl.tsx /
|
|
22
|
+
* one-shot.ts construct the ctx from — and agent/loop.ts already calls
|
|
23
|
+
* saveSession(session) end-of-turn (and session/pending.ts saves per
|
|
24
|
+
* mutating-tool WAL entry), so no extra save call is needed here. The
|
|
25
|
+
* emitter-plus-repl-side-subscriber alternative was rejected: it would
|
|
26
|
+
* add a second writer for state the tool already owns, and headless /
|
|
27
|
+
* one-shot paths (no subscriber mounted) would silently lose persistence.
|
|
28
|
+
*
|
|
29
|
+
* Validation discipline: structured error results (is_error: true), never
|
|
30
|
+
* throws — caps at 20 items, rejects empty lists, empty text, unknown
|
|
31
|
+
* status. MULTIPLE in_progress items are a WARNING in the result, not an
|
|
32
|
+
* error: blocking the model mid-job over a presentation nit would stall
|
|
33
|
+
* long workflows, and the list is still coherent and renderable — the
|
|
34
|
+
* warning text steers the model to fix it on its next call instead.
|
|
35
|
+
*
|
|
36
|
+
* isMutating: false — writes CLI-side session state only, never SAP.
|
|
37
|
+
* Headless: identical behaviour (session write + emitter absent/no-op;
|
|
38
|
+
* no prompt involved).
|
|
39
|
+
*
|
|
40
|
+
* Emitter seam (Task 5, 2026-07-05): the panel channel arrives via
|
|
41
|
+
* `ctx.todoEmitter` — NOT a module-level import — mirroring how
|
|
42
|
+
* chunkEmitter is ctx-carried. agent_run's child ctx omits it, so a
|
|
43
|
+
* subagent's todo_set updates only the CHILD session and can never
|
|
44
|
+
* clobber the parent's Ctrl+T panel.
|
|
45
|
+
*/
|
|
46
|
+
import { registerTool } from './index.js';
|
|
47
|
+
import { formatTodoChecklistBlock } from '../renderer/todo-block.js';
|
|
48
|
+
/** Hard cap on list length — past this the list stops being a glanceable plan. */
|
|
49
|
+
export const TODO_MAX_ITEMS = 20;
|
|
50
|
+
const VALID_STATUSES = new Set(['pending', 'in_progress', 'completed']);
|
|
51
|
+
function errorResult(message) {
|
|
52
|
+
return { content: `error: ${message}`, is_error: true };
|
|
53
|
+
}
|
|
54
|
+
export async function todoSetHandler(args, ctx) {
|
|
55
|
+
const raw = args?.todos;
|
|
56
|
+
if (!Array.isArray(raw)) {
|
|
57
|
+
return errorResult("todos must be an array of { text, status } items (status: 'pending' | 'in_progress' | 'completed').");
|
|
58
|
+
}
|
|
59
|
+
if (raw.length === 0) {
|
|
60
|
+
return errorResult('todos must contain at least 1 item — this is a full-list replace, never call it with an empty list.');
|
|
61
|
+
}
|
|
62
|
+
if (raw.length > TODO_MAX_ITEMS) {
|
|
63
|
+
return errorResult(`todos exceeds the ${TODO_MAX_ITEMS}-item cap (got ${raw.length}). Collapse completed work or group related steps into one item.`);
|
|
64
|
+
}
|
|
65
|
+
// Validate + normalise BEFORE touching session state — a rejected call
|
|
66
|
+
// must leave the previous list fully intact.
|
|
67
|
+
const todos = [];
|
|
68
|
+
for (let i = 0; i < raw.length; i++) {
|
|
69
|
+
const item = raw[i];
|
|
70
|
+
const text = typeof item?.text === 'string' ? item.text.trim() : '';
|
|
71
|
+
if (!text) {
|
|
72
|
+
return errorResult(`todos[${i}].text must be a non-empty string.`);
|
|
73
|
+
}
|
|
74
|
+
if (typeof item?.status !== 'string' || !VALID_STATUSES.has(item.status)) {
|
|
75
|
+
return errorResult(`todos[${i}].status must be one of 'pending' | 'in_progress' | 'completed' (got ${JSON.stringify(item?.status ?? null)}).`);
|
|
76
|
+
}
|
|
77
|
+
todos.push({ text, status: item.status });
|
|
78
|
+
}
|
|
79
|
+
const inProgress = todos.filter((t) => t.status === 'in_progress');
|
|
80
|
+
// Full-list replace on the LIVE session object — persisted by the loop's
|
|
81
|
+
// existing end-of-turn saveSession (see module doc for the evidence).
|
|
82
|
+
ctx.session.todos = todos;
|
|
83
|
+
// UI fan-out (Task 5) — via the ctx-carried emitter (absent on subagent
|
|
84
|
+
// ctxs → suppressed; absent listener → harmless no-op). Emit a defensive
|
|
85
|
+
// copy so a consumer mutating the payload can never corrupt the
|
|
86
|
+
// persisted session state.
|
|
87
|
+
ctx.todoEmitter?.emit('update', todos.map((t) => ({ ...t })));
|
|
88
|
+
// Transcript render (CC-style, 2026-07): drop the FULL checklist into
|
|
89
|
+
// scrollback on every successful set. The raw ⏺/⎿ todo_set rows are
|
|
90
|
+
// widget-suppressed (renderer/tool-widget.ts), so this block is the sole
|
|
91
|
+
// visible representation — mirroring ask_question. ctx.chunkEmitter routes
|
|
92
|
+
// to Ink Static scrollback AND the classic 'chunk'→stdout path
|
|
93
|
+
// (renderer/notices.ts). Subagent ctxs deliberately omit chunkEmitter
|
|
94
|
+
// (tools/subagent/agent_run.ts), so a child turn's todo_set can never
|
|
95
|
+
// print its checklist into the PARENT transcript — same seam as the
|
|
96
|
+
// todoEmitter suppression above.
|
|
97
|
+
ctx.chunkEmitter?.emit('chunk', formatTodoChecklistBlock(todos));
|
|
98
|
+
const result = {
|
|
99
|
+
count: todos.length,
|
|
100
|
+
in_progress: inProgress[0]?.text ?? null,
|
|
101
|
+
};
|
|
102
|
+
if (inProgress.length > 1) {
|
|
103
|
+
result.warning =
|
|
104
|
+
`${inProgress.length} items are in_progress — exactly one item should be in_progress at a time. ` +
|
|
105
|
+
'Mark the others pending or completed on your next todo_set call.';
|
|
106
|
+
}
|
|
107
|
+
return { content: JSON.stringify(result) };
|
|
108
|
+
}
|
|
109
|
+
registerTool({
|
|
110
|
+
name: 'todo_set',
|
|
111
|
+
// Model-behavior lever — this exact framing is binding (task-4 brief).
|
|
112
|
+
description: 'Maintain your task list for multi-step work. Call with the FULL updated list whenever you ' +
|
|
113
|
+
'complete a step, start a new one, or discover new work. Exactly one item should be ' +
|
|
114
|
+
'in_progress at a time. The user sees this list — keep items short and outcome-shaped.',
|
|
115
|
+
isMutating: false,
|
|
116
|
+
category: 'session',
|
|
117
|
+
input_schema: {
|
|
118
|
+
type: 'object',
|
|
119
|
+
properties: {
|
|
120
|
+
todos: {
|
|
121
|
+
type: 'array',
|
|
122
|
+
minItems: 1,
|
|
123
|
+
maxItems: TODO_MAX_ITEMS,
|
|
124
|
+
description: 'The full task list — replaces the previous list entirely.',
|
|
125
|
+
items: {
|
|
126
|
+
type: 'object',
|
|
127
|
+
properties: {
|
|
128
|
+
text: {
|
|
129
|
+
type: 'string',
|
|
130
|
+
description: 'Short, outcome-shaped item, e.g. "Activate ZCL_ORDER_API".',
|
|
131
|
+
},
|
|
132
|
+
status: {
|
|
133
|
+
type: 'string',
|
|
134
|
+
enum: ['pending', 'in_progress', 'completed'],
|
|
135
|
+
},
|
|
136
|
+
},
|
|
137
|
+
required: ['text', 'status'],
|
|
138
|
+
},
|
|
139
|
+
},
|
|
140
|
+
},
|
|
141
|
+
required: ['todos'],
|
|
142
|
+
},
|
|
143
|
+
handler: (args, ctx) => todoSetHandler(args, ctx),
|
|
144
|
+
});
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* transport-resolution — owning-transport check before SAP writes (Rule 9,
|
|
3
|
+
* battery defects D15/D20).
|
|
4
|
+
*
|
|
5
|
+
* The trap this kills: an object (or one of its LIMU pieces) is already
|
|
6
|
+
* locked in an open transport request. Writing under ANY other transport is
|
|
7
|
+
* guaranteed to fail with a CTS lock conflict (CTS_WBO_API 020 — observed
|
|
8
|
+
* live: "Object LIMU CINC ZBP_I_DOWNTIMELOG…CCIMP is already locked in
|
|
9
|
+
* request S4HK903431"). Asking the human to pick from N open transports
|
|
10
|
+
* (D15: 89 of them) or creating a fresh junk transport (D20) cannot succeed.
|
|
11
|
+
*
|
|
12
|
+
* Correct flow, now structural instead of prose-only:
|
|
13
|
+
* - object locked in exactly one open request → USE that transport
|
|
14
|
+
* silently, overriding a conflicting supplied transport (with a notice);
|
|
15
|
+
* - object locked in several open requests → ambiguous, never force —
|
|
16
|
+
* keep the supplied transport and surface the candidates;
|
|
17
|
+
* - object free / lookup failed → current behaviour (supplied transport).
|
|
18
|
+
*
|
|
19
|
+
* The lookup is one E071+E070 query pair (AdtClient.transportForObject) and
|
|
20
|
+
* is strictly best-effort: a failed lookup (VPN drop, data-preview auth,
|
|
21
|
+
* older mocks without the method) must NEVER block the write.
|
|
22
|
+
*/
|
|
23
|
+
import { wasTransportCreatedThisSession } from './transport.js';
|
|
24
|
+
/**
|
|
25
|
+
* Resolve the transport for a write against the owning-transport ledger.
|
|
26
|
+
*
|
|
27
|
+
* @param ctx tool context (uses ctx.adt + ctx.chunkEmitter)
|
|
28
|
+
* @param objectType ABAP object type of the write target (CLAS, PROG, …)
|
|
29
|
+
* @param objectName ABAP object name of the write target
|
|
30
|
+
* @param supplied the transport the tool would use today (each tool keeps
|
|
31
|
+
* its own args/session fallback semantics and passes the
|
|
32
|
+
* outcome here)
|
|
33
|
+
*/
|
|
34
|
+
export async function resolveWriteTransport(ctx, objectType, objectName, supplied) {
|
|
35
|
+
// TODO(perf): per-turn memo. Every write funnels through this lookup
|
|
36
|
+
// (2 datapreview SQL queries inside transportForObject), so a multi-method
|
|
37
|
+
// turn — N sap_update_method calls against the same class — amplifies to
|
|
38
|
+
// N×2 queries for an answer that cannot change mid-turn. Memoise per turn,
|
|
39
|
+
// keyed `${objectType}:${objectName}`, before the battery grows
|
|
40
|
+
// write-heavy scenarios. No implementation yet.
|
|
41
|
+
let owning = null;
|
|
42
|
+
try {
|
|
43
|
+
owning = await ctx.adt.transportForObject(objectType, objectName);
|
|
44
|
+
}
|
|
45
|
+
catch (err) {
|
|
46
|
+
// Best-effort only — never block or delay the write on a failed lookup,
|
|
47
|
+
// but leave a dim trace so "why didn't it auto-resolve?" is diagnosable
|
|
48
|
+
// ('info' channel = non-fatal diagnostics; 'warn' is reserved for the
|
|
49
|
+
// conflicting-override case below).
|
|
50
|
+
ctx.chunkEmitter?.emit('info', `owning-transport lookup failed for ${objectType} ${objectName} — `
|
|
51
|
+
+ `proceeding with ${supplied ?? 'no transport'} `
|
|
52
|
+
+ `(${err instanceof Error ? err.message : String(err)})`);
|
|
53
|
+
return { transport: supplied };
|
|
54
|
+
}
|
|
55
|
+
if (owning?.transport) {
|
|
56
|
+
const conflicting = !!supplied && supplied !== owning.transport;
|
|
57
|
+
const note = conflicting
|
|
58
|
+
? `${objectName} is already locked in ${owning.transport} — writing under the owning transport and ignoring transport=${supplied} (SAP would refuse with a CTS lock conflict)`
|
|
59
|
+
: `object locked in ${owning.transport} — writing under owning transport`;
|
|
60
|
+
// warn only when we OVERRIDE what the caller asked for; the happy match
|
|
61
|
+
// (supplied === owning, or nothing supplied) is informational.
|
|
62
|
+
ctx.chunkEmitter?.emit(conflicting ? 'warn' : 'info', note);
|
|
63
|
+
// Orphan hint (third junk TR this month, 2026-06-12): the session's
|
|
64
|
+
// current transport was created by sap_transport_create THIS session, yet
|
|
65
|
+
// this write was forced into a different (owning) request — the created
|
|
66
|
+
// TR is probably an empty orphan. Warn only; deleting a TR is destructive
|
|
67
|
+
// and stays with the human.
|
|
68
|
+
const sessionCurrent = ctx.currentTransport?.get() ?? null;
|
|
69
|
+
if (sessionCurrent
|
|
70
|
+
&& sessionCurrent !== owning.transport
|
|
71
|
+
&& wasTransportCreatedThisSession(sessionCurrent)) {
|
|
72
|
+
ctx.chunkEmitter?.emit('warn', `transport ${sessionCurrent} created this session was not used by this write — `
|
|
73
|
+
+ 'it may be an empty orphan (delete via SE09 or ask me)');
|
|
74
|
+
}
|
|
75
|
+
return { transport: owning.transport, owning: owning.transport, note };
|
|
76
|
+
}
|
|
77
|
+
if (owning && owning.candidates.length > 1) {
|
|
78
|
+
const note = `${objectName} pieces are locked in ${owning.candidates.length} open transports `
|
|
79
|
+
+ `(${owning.candidates.join(', ')}) — cannot auto-resolve the owning request; `
|
|
80
|
+
+ `using ${supplied ?? 'no transport'}`;
|
|
81
|
+
ctx.chunkEmitter?.emit('warn', note);
|
|
82
|
+
return { transport: supplied, note };
|
|
83
|
+
}
|
|
84
|
+
// Object is free — current behaviour.
|
|
85
|
+
return { transport: supplied };
|
|
86
|
+
}
|