@cspeach/cli 1.1.19 → 1.1.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/anthropic-provider.js +30 -10
- package/dist/agent/cache-keepalive.js +162 -0
- package/dist/agent/cold-prune.js +116 -0
- package/dist/agent/loop.js +675 -148
- package/dist/agent/provider-shape.js +263 -0
- package/dist/agent/providers/ai-hub-provider.js +17 -2
- package/dist/agent/providers/byok-provider.js +33 -3
- package/dist/agent/providers/local-provider.js +8 -1
- package/dist/agent/repair-partial.js +66 -5
- package/dist/agent/summarise-via-provider.js +6 -1
- package/dist/agent/system-prompt.js +38 -0
- package/dist/agent/tool-dispatch.js +8 -0
- package/dist/agent/tool-loading-pin.js +100 -0
- package/dist/cli.js +11 -0
- package/dist/commands/auto-compact.js +33 -16
- package/dist/commands/compact.js +37 -2
- package/dist/commands/config-set.js +10 -1
- package/dist/commands/config-show.js +11 -0
- package/dist/commands/cost.js +14 -2
- package/dist/commands/plan-audit.js +1 -0
- package/dist/config/loader.js +55 -2
- package/dist/cost/cost-log.js +62 -2
- package/dist/cost/pricing.js +6 -2
- package/dist/lib/spill-labels.js +13 -0
- package/dist/models/resolve.js +93 -2
- package/dist/models/server-config.js +158 -3
- package/dist/one-shot.js +15 -5
- package/dist/projects/image-attachments.js +15 -2
- package/dist/renderer/footer-line.js +6 -2
- package/dist/renderer/startup-lines.js +5 -3
- package/dist/renderer/tool-labels.js +33 -2
- package/dist/renderer/ui-width.js +13 -0
- package/dist/repl/current-transport.js +13 -0
- package/dist/repl/post-turn-status.js +8 -1
- package/dist/repl.js +71 -10
- package/dist/session/repin-model.js +18 -0
- package/dist/session/store.js +16 -2
- package/dist/skills/bundled-skills.js +1 -1
- package/dist/skills/preamble.js +75 -0
- package/dist/skills/source-manifest.js +11 -1
- package/dist/tools/filesystem/file-read.js +11 -1
- package/dist/tools/result-spill.js +238 -0
- package/dist/tools/sap-read.js +58 -14
- package/dist/tools/shell/shell_exec.js +9 -0
- package/dist/tools/subagent/adt-serial.js +33 -0
- package/dist/tools/subagent/agent_run.js +2 -0
- package/dist/tools/subagent/read_agent.js +178 -0
- package/dist/tools/subagent/reader-prompt.js +48 -0
- package/dist/tools/todo.js +3 -1
- package/dist/tools/tool-loading.js +255 -0
- package/dist/tools/tool-output-read.js +117 -0
- package/dist/tools/transport.js +6 -1
- package/dist/ui/footer.js +5 -5
- package/dist/ui/sap-state-store.js +1 -1
- package/dist/ui/turn-status-emitter.js +37 -0
- package/dist/ui/turn-status.js +1 -1
- package/package.json +2 -1
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Task 17 (model control, 2026-09-27, spec D12) — large tool results go to a
|
|
3
|
+
* file; the history keeps a preview and a path.
|
|
4
|
+
*
|
|
5
|
+
* Claude Code's "output too large, saved to file" pattern. Above
|
|
6
|
+
* SPILL_THRESHOLD_CHARS the full text is written to
|
|
7
|
+
* ~/.cspeach/streams/<sessionId>/tool-<toolUseId>.txt and the tool_result
|
|
8
|
+
* carries the first 60 lines (plus the last 20 for source, SQL, shell and file windows),
|
|
9
|
+
* then one pointer line naming tool_output_read.
|
|
10
|
+
*
|
|
11
|
+
* Sizing happens ONCE, when the tool_result is created (the handler's return
|
|
12
|
+
* value) — never afterwards, so it is not a history edit and the cached
|
|
13
|
+
* prefix / preserved thinking stay append-only.
|
|
14
|
+
*
|
|
15
|
+
* The spill file holds exactly the text the tool would have returned — no
|
|
16
|
+
* more (no secrets beyond what the tool already produced), and it lives in the
|
|
17
|
+
* per-session area under ~/.cspeach, never in the user's project.
|
|
18
|
+
*/
|
|
19
|
+
import * as fs from 'node:fs';
|
|
20
|
+
import * as path from 'node:path';
|
|
21
|
+
import * as crypto from 'node:crypto';
|
|
22
|
+
import { streamsRoot } from '../agent/turn-stream.js';
|
|
23
|
+
import { rememberSpillLabel } from '../lib/spill-labels.js';
|
|
24
|
+
export const SPILL_THRESHOLD_CHARS = 8_000;
|
|
25
|
+
export const PREVIEW_HEAD_LINES = 60;
|
|
26
|
+
export const PREVIEW_TAIL_LINES = 20;
|
|
27
|
+
/** One preview line never carries more than this (a minified JSON line must not defeat the preview). */
|
|
28
|
+
export const PREVIEW_LINE_MAX_CHARS = 500;
|
|
29
|
+
/** Character budgets for the head and tail blocks of the preview. */
|
|
30
|
+
export const PREVIEW_HEAD_BUDGET_CHARS = 6_000;
|
|
31
|
+
export const PREVIEW_TAIL_BUDGET_CHARS = 2_000;
|
|
32
|
+
/**
|
|
33
|
+
* Kinds whose preview also keeps the last lines (the end matters: totals,
|
|
34
|
+
* exit code, ENDCLASS; for file_read the window's last line). Only search
|
|
35
|
+
* hits and reports are head-only.
|
|
36
|
+
*/
|
|
37
|
+
const TAIL_KINDS = new Set(['source', 'sql', 'shell', 'file']);
|
|
38
|
+
/** Session ids / tool_use ids become file names: keep them path-inert. */
|
|
39
|
+
function safeSegment(s) {
|
|
40
|
+
const cleaned = s.replace(/[^A-Za-z0-9_-]/g, '_');
|
|
41
|
+
return cleaned.length > 0 ? cleaned : '_';
|
|
42
|
+
}
|
|
43
|
+
/** The directory that holds one session's spill files. */
|
|
44
|
+
export function spillDirFor(sessionId) {
|
|
45
|
+
return path.join(streamsRoot(), safeSegment(sessionId));
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* Session id as the tools see it, or undefined when the ctx carries none
|
|
49
|
+
* (some tests and non-loop call paths). Without a session there is no
|
|
50
|
+
* private spill dir, so nothing is spilled and tool_output_read refuses —
|
|
51
|
+
* never a shared fallback directory (fix round 1, minor 3).
|
|
52
|
+
*/
|
|
53
|
+
export function sessionIdOf(ctx) {
|
|
54
|
+
const id = ctx.session?.id;
|
|
55
|
+
return typeof id === 'string' && id.length > 0 ? id : undefined;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Spill file names: `tool-<id>.txt`, or `tool-<id>.n.txt` for text that is
|
|
59
|
+
* already line-numbered (file_read windows keep the FILE's line numbers, so
|
|
60
|
+
* tool_output_read shows those lines as they are instead of numbering again —
|
|
61
|
+
* one numbering space). tool_output_read accepts only these shapes.
|
|
62
|
+
*/
|
|
63
|
+
export const SPILL_FILE_RE = /^tool-[A-Za-z0-9_-]+(\.n)?\.txt$/;
|
|
64
|
+
export const NUMBERED_SPILL_RE = /\.n\.txt$/;
|
|
65
|
+
/** Spill files older than this are deleted at REPL start (like the notice logs). */
|
|
66
|
+
export const SPILL_MAX_AGE_DAYS = 14;
|
|
67
|
+
function clipLine(line) {
|
|
68
|
+
if (line.length <= PREVIEW_LINE_MAX_CHARS)
|
|
69
|
+
return line;
|
|
70
|
+
return `${line.slice(0, PREVIEW_LINE_MAX_CHARS)}… [+${line.length - PREVIEW_LINE_MAX_CHARS} chars]`;
|
|
71
|
+
}
|
|
72
|
+
export function pointerLine(lines, chars, filePath) {
|
|
73
|
+
return `[${lines} lines / ${chars} chars — full output saved to ${filePath}; read slices with tool_output_read(path, offset, limit)]`;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Build the preview text for `text` (already known to be large). Pure. Line
|
|
77
|
+
* numbers in the omitted marker are the spill file's own (1-based, =
|
|
78
|
+
* tool_output_read offset + 1); for an already-numbered spill the marker names
|
|
79
|
+
* the offset instead, since the visible numbers belong to the source file.
|
|
80
|
+
*/
|
|
81
|
+
export function buildPreview(text, kind, filePath, opts = {}) {
|
|
82
|
+
const lines = text.split('\n');
|
|
83
|
+
const total = lines.length;
|
|
84
|
+
const head = [];
|
|
85
|
+
let used = 0;
|
|
86
|
+
for (let i = 0; i < total && head.length < PREVIEW_HEAD_LINES; i++) {
|
|
87
|
+
const l = clipLine(lines[i]);
|
|
88
|
+
if (head.length > 0 && used + l.length + 1 > PREVIEW_HEAD_BUDGET_CHARS)
|
|
89
|
+
break;
|
|
90
|
+
head.push(l);
|
|
91
|
+
used += l.length + 1;
|
|
92
|
+
}
|
|
93
|
+
const tail = [];
|
|
94
|
+
if (TAIL_KINDS.has(kind)) {
|
|
95
|
+
let tailUsed = 0;
|
|
96
|
+
for (let i = total - 1; i >= head.length && tail.length < PREVIEW_TAIL_LINES; i--) {
|
|
97
|
+
const l = clipLine(lines[i]);
|
|
98
|
+
if (tail.length > 0 && tailUsed + l.length + 1 > PREVIEW_TAIL_BUDGET_CHARS)
|
|
99
|
+
break;
|
|
100
|
+
tail.unshift(l);
|
|
101
|
+
tailUsed += l.length + 1;
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
const out = [...head];
|
|
105
|
+
const omittedFrom = head.length + 1; // 1-based first omitted line
|
|
106
|
+
const omittedTo = total - tail.length; // 1-based last omitted line
|
|
107
|
+
if (omittedTo >= omittedFrom) {
|
|
108
|
+
out.push(opts.numbered
|
|
109
|
+
? `[… ${omittedTo - omittedFrom + 1} lines omitted — tool_output_read offset=${omittedFrom - 1} …]`
|
|
110
|
+
: `[… lines ${omittedFrom}–${omittedTo} omitted …]`);
|
|
111
|
+
}
|
|
112
|
+
out.push(...tail);
|
|
113
|
+
out.push(pointerLine(total, text.length, filePath));
|
|
114
|
+
return out.join('\n');
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Size a tool result at ingestion. At or below the threshold → unchanged.
|
|
118
|
+
* Above it → the full text goes to the session spill file and a preview comes
|
|
119
|
+
* back. A write failure fails OPEN (full text returned, nothing lost).
|
|
120
|
+
*
|
|
121
|
+
* `header` (sap_get_source's version line) counts toward the threshold and
|
|
122
|
+
* leads the preview, but is NOT written to the file: the file is the body
|
|
123
|
+
* alone, so tool_output_read line k is body line k (fix round 1, I1), and the
|
|
124
|
+
* pointer's counts describe the body.
|
|
125
|
+
*/
|
|
126
|
+
export function spillIfLarge(i) {
|
|
127
|
+
const whole = i.header !== undefined ? `${i.header}\n${i.text}` : i.text;
|
|
128
|
+
if (whole.length <= (i.threshold ?? SPILL_THRESHOLD_CHARS))
|
|
129
|
+
return { spilled: false, text: whole };
|
|
130
|
+
const dir = spillDirFor(i.sessionId);
|
|
131
|
+
const id = safeSegment(i.toolUseId && i.toolUseId.length > 0 ? i.toolUseId : crypto.randomUUID());
|
|
132
|
+
const filePath = path.join(dir, `tool-${id}${i.numbered ? '.n' : ''}.txt`);
|
|
133
|
+
try {
|
|
134
|
+
fs.mkdirSync(dir, { recursive: true, mode: 0o700 });
|
|
135
|
+
fs.writeFileSync(filePath, i.text, { encoding: 'utf-8', mode: 0o600 });
|
|
136
|
+
}
|
|
137
|
+
catch {
|
|
138
|
+
return { spilled: false, text: whole };
|
|
139
|
+
}
|
|
140
|
+
const preview = buildPreview(i.text, i.kind, filePath, { numbered: i.numbered });
|
|
141
|
+
return { spilled: true, text: i.header !== undefined ? `${i.header}\n${preview}` : preview, path: filePath };
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* Convenience for handlers: size `result.content` with the ctx's session and
|
|
145
|
+
* tool_use id. With a header, `result.content` is the body and the returned
|
|
146
|
+
* content is `header\n…`. No session id → never spilled.
|
|
147
|
+
*/
|
|
148
|
+
export function spillToolResult(result, ctx, kind, opts = {}) {
|
|
149
|
+
const sessionId = sessionIdOf(ctx);
|
|
150
|
+
const whole = opts.header !== undefined ? `${opts.header}\n${result.content}` : result.content;
|
|
151
|
+
if (sessionId === undefined)
|
|
152
|
+
return whole === result.content ? result : { ...result, content: whole };
|
|
153
|
+
const { label, ...spillOpts } = opts;
|
|
154
|
+
const s = spillIfLarge({ text: result.content, sessionId, toolUseId: ctx.toolUseId, kind, ...spillOpts });
|
|
155
|
+
if (s.spilled && s.path && label)
|
|
156
|
+
rememberSpillLabel(s.path, label);
|
|
157
|
+
return s.text === result.content ? result : { ...result, content: s.text };
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Delete spill files (tool-*.txt in every session dir under the spill root)
|
|
161
|
+
* last modified more than SPILL_MAX_AGE_DAYS ago, then any session dir left
|
|
162
|
+
* empty. Other files (the turn mirrors) are left alone; symlinks and
|
|
163
|
+
* junctions (root, dirs, files) are never followed. Best effort: never
|
|
164
|
+
* throws. Returns how many files were deleted.
|
|
165
|
+
*/
|
|
166
|
+
export function pruneSpillFiles(now = Date.now()) {
|
|
167
|
+
const cutoff = now - SPILL_MAX_AGE_DAYS * 24 * 60 * 60 * 1000;
|
|
168
|
+
let removed = 0;
|
|
169
|
+
let dirs;
|
|
170
|
+
try {
|
|
171
|
+
// Final review tools-I1 — never follow links: a symlinked / junctioned
|
|
172
|
+
// root, session dir or file is skipped (lstat), so the prune can never
|
|
173
|
+
// delete outside the CLI's own area.
|
|
174
|
+
if (!fs.lstatSync(streamsRoot()).isDirectory())
|
|
175
|
+
return 0;
|
|
176
|
+
dirs = fs.readdirSync(streamsRoot());
|
|
177
|
+
}
|
|
178
|
+
catch {
|
|
179
|
+
return 0;
|
|
180
|
+
}
|
|
181
|
+
for (const d of dirs) {
|
|
182
|
+
const dir = path.join(streamsRoot(), d);
|
|
183
|
+
let names;
|
|
184
|
+
try {
|
|
185
|
+
const st = fs.lstatSync(dir);
|
|
186
|
+
if (st.isSymbolicLink() || !st.isDirectory())
|
|
187
|
+
continue;
|
|
188
|
+
names = fs.readdirSync(dir);
|
|
189
|
+
}
|
|
190
|
+
catch {
|
|
191
|
+
continue;
|
|
192
|
+
}
|
|
193
|
+
for (const name of names) {
|
|
194
|
+
if (!SPILL_FILE_RE.test(name))
|
|
195
|
+
continue;
|
|
196
|
+
const p = path.join(dir, name);
|
|
197
|
+
try {
|
|
198
|
+
const st = fs.lstatSync(p);
|
|
199
|
+
if (st.isSymbolicLink() || !st.isFile())
|
|
200
|
+
continue;
|
|
201
|
+
if (st.mtimeMs < cutoff) {
|
|
202
|
+
fs.unlinkSync(p);
|
|
203
|
+
removed++;
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
catch {
|
|
207
|
+
/* best effort */
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
try {
|
|
211
|
+
if (fs.readdirSync(dir).length === 0)
|
|
212
|
+
fs.rmdirSync(dir);
|
|
213
|
+
}
|
|
214
|
+
catch {
|
|
215
|
+
/* best effort */
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
return removed;
|
|
219
|
+
}
|
|
220
|
+
/** True when `child` lies strictly inside `dir`. */
|
|
221
|
+
export function isInside(dir, child) {
|
|
222
|
+
const rel = path.relative(dir, child);
|
|
223
|
+
return rel.length > 0 && !rel.startsWith('..') && !path.isAbsolute(rel);
|
|
224
|
+
}
|
|
225
|
+
/** `p` and, when it exists, its realpath (8.3 names, /var → /private/var). */
|
|
226
|
+
export function withRealpath(p) {
|
|
227
|
+
try {
|
|
228
|
+
const real = fs.realpathSync.native(p);
|
|
229
|
+
return real === p ? [p] : [p, real];
|
|
230
|
+
}
|
|
231
|
+
catch {
|
|
232
|
+
return [p];
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
/** True when `absPath` (already a realpath) lies under the spill root (any session). */
|
|
236
|
+
export function isUnderSpillRoot(absPath) {
|
|
237
|
+
return withRealpath(streamsRoot()).some((root) => isInside(root, absPath));
|
|
238
|
+
}
|
package/dist/tools/sap-read.js
CHANGED
|
@@ -24,6 +24,48 @@
|
|
|
24
24
|
import { registerTool } from './index.js';
|
|
25
25
|
import { normaliseSyntaxResult, checkSyntax } from './verify.js';
|
|
26
26
|
import { recordSyntaxCheck } from './syntax-state.js';
|
|
27
|
+
import { spillToolResult } from './result-spill.js';
|
|
28
|
+
// -----------------------------------------------------------------------
|
|
29
|
+
// Task 17 (spec D12, ruling F18) — model-facing formats.
|
|
30
|
+
// -----------------------------------------------------------------------
|
|
31
|
+
/**
|
|
32
|
+
* sap_get_source's one-line header: `[version: active]`,
|
|
33
|
+
* `[version: inactive | <warning>]` (any version may carry the split-state
|
|
34
|
+
* warning), or `[version: unknown]`. Plan-audit evidence reads the first
|
|
35
|
+
* 200 chars of the result, so the version leads it.
|
|
36
|
+
*/
|
|
37
|
+
export function formatSourceHeader(version, warning) {
|
|
38
|
+
return warning ? `[version: ${version} | ${warning.replace(/\s*\n\s*/g, ' ')}]` : `[version: ${version}]`;
|
|
39
|
+
}
|
|
40
|
+
/** Inverse of formatSourceHeader: the version (and warning) back from a result. */
|
|
41
|
+
export function parseSourceHeader(text) {
|
|
42
|
+
const first = text.split('\n', 1)[0] ?? '';
|
|
43
|
+
const m = /^\[version: ([a-z]+)(?: \| (.*))?\]$/.exec(first);
|
|
44
|
+
if (!m)
|
|
45
|
+
return { version: 'unknown' };
|
|
46
|
+
return m[2] !== undefined ? { version: m[1], warning: m[2] } : { version: m[1] };
|
|
47
|
+
}
|
|
48
|
+
/** One TSV cell: tabs / newlines inside a value cannot break the grid. */
|
|
49
|
+
const tsvCell = (v) => (v === null || v === undefined ? '' : String(v)).replace(/[\t\r\n]+/g, ' ');
|
|
50
|
+
export function formatSqlTsv(result) {
|
|
51
|
+
const cols = result.columns ?? [];
|
|
52
|
+
const rows = result.rows ?? [];
|
|
53
|
+
const lines = [cols.map(tsvCell).join('\t'), ...rows.map((r) => cols.map((c) => tsvCell(r[c])).join('\t'))];
|
|
54
|
+
const n = rows.length;
|
|
55
|
+
const total = typeof result.totalRows === 'number' && result.totalRows > n ? ` of ${result.totalRows}` : '';
|
|
56
|
+
lines.push(`(${n} ${n === 1 ? 'row' : 'rows'}${total})`);
|
|
57
|
+
return lines.join('\n');
|
|
58
|
+
}
|
|
59
|
+
/** Search hits, one per line: NAME TYPE PACKAGE DESCRIPTION URI ("-" for an empty field). */
|
|
60
|
+
export function formatSearchHits(hits) {
|
|
61
|
+
if (hits.length === 0)
|
|
62
|
+
return '(no hits)';
|
|
63
|
+
const f = (v) => {
|
|
64
|
+
const t = typeof v === 'string' ? v.replace(/\s+/g, ' ').trim() : '';
|
|
65
|
+
return t.length > 0 ? t : '-';
|
|
66
|
+
};
|
|
67
|
+
return hits.map((h) => [h.name, h.type, h.packageName ?? h.package, h.description, h.uri].map(f).join(' ')).join('\n');
|
|
68
|
+
}
|
|
27
69
|
// -----------------------------------------------------------------------
|
|
28
70
|
// 1. sap_get_source
|
|
29
71
|
// -----------------------------------------------------------------------
|
|
@@ -36,7 +78,8 @@ registerTool({
|
|
|
36
78
|
+ 'the response is the active source AND a warning is emitted so the '
|
|
37
79
|
+ 'caller knows the inactive edits are not visible. For a freshly '
|
|
38
80
|
+ 'created object with NO active version yet, the inactive (working) '
|
|
39
|
-
+ 'source is returned with version="inactive" and an honest note.'
|
|
81
|
+
+ 'source is returned with version="inactive" and an honest note. '
|
|
82
|
+
+ 'The first line is a [version: …] header, not part of the source.',
|
|
40
83
|
isMutating: false,
|
|
41
84
|
requiresSap: true,
|
|
42
85
|
input_schema: {
|
|
@@ -117,13 +160,13 @@ registerTool({
|
|
|
117
160
|
// probe failed — leave resolvedVersion as 'unknown'
|
|
118
161
|
}
|
|
119
162
|
}
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
};
|
|
163
|
+
// Fix round 1 (I1): the header leads the result and the preview, but a
|
|
164
|
+
// spill file holds the source alone — tool_output_read line k is source
|
|
165
|
+
// line k, matching syntax-check / ATC line numbers.
|
|
166
|
+
return spillToolResult({ content: String(source) }, ctx, 'source', {
|
|
167
|
+
header: formatSourceHeader(resolvedVersion, splitStateWarning),
|
|
168
|
+
label: String(args.name).toUpperCase(),
|
|
169
|
+
});
|
|
127
170
|
},
|
|
128
171
|
});
|
|
129
172
|
// -----------------------------------------------------------------------
|
|
@@ -155,10 +198,11 @@ registerTool({
|
|
|
155
198
|
// `version` right after name/type puts it inside the excerpt window.
|
|
156
199
|
const versionMatch = /\badtcore:version="([^"]*)"/.exec(result.rawXml ?? '');
|
|
157
200
|
const version = versionMatch?.[1] || 'unknown';
|
|
158
|
-
// name/type/version lead the serialized JSON
|
|
159
|
-
//
|
|
160
|
-
//
|
|
161
|
-
const { name, type, ...rest } = result;
|
|
201
|
+
// name/type/version lead the serialized JSON, so `version` sits inside
|
|
202
|
+
// the excerpt window. Task 17: rawXml is parsed above and never sent to
|
|
203
|
+
// the model (it was the bulk of the result).
|
|
204
|
+
const { name, type, rawXml, ...rest } = result;
|
|
205
|
+
void rawXml;
|
|
162
206
|
return { content: JSON.stringify({ name, type, version, ...rest }, null, 2) };
|
|
163
207
|
},
|
|
164
208
|
});
|
|
@@ -187,7 +231,7 @@ registerTool({
|
|
|
187
231
|
},
|
|
188
232
|
handler: async (args, ctx) => {
|
|
189
233
|
const results = await ctx.adt.searchObject(args.query, args.type, args.maxResults ?? 100);
|
|
190
|
-
return { content:
|
|
234
|
+
return spillToolResult({ content: formatSearchHits(results) }, ctx, 'search', { label: `search "${args.query}"` });
|
|
191
235
|
},
|
|
192
236
|
});
|
|
193
237
|
// -----------------------------------------------------------------------
|
|
@@ -231,7 +275,7 @@ registerTool({
|
|
|
231
275
|
},
|
|
232
276
|
handler: async (args, ctx) => {
|
|
233
277
|
const result = await ctx.adt.sqlQuery(args.query, args.maxRows ?? 100);
|
|
234
|
-
return { content:
|
|
278
|
+
return spillToolResult({ content: formatSqlTsv(result) }, ctx, 'sql', { label: 'query result' });
|
|
235
279
|
},
|
|
236
280
|
});
|
|
237
281
|
// -----------------------------------------------------------------------
|
|
@@ -46,12 +46,21 @@ import { registerTool } from '../index.js';
|
|
|
46
46
|
import { resolveSafePath, PathOutsideRootError } from '../_filesystem-shared.js';
|
|
47
47
|
import { loadSafelist, buildSafeEnv, resolveExecutable, validateArgv, rejectPathSeparator, DEFAULT_SAFELIST, KILL_GRACE_MS, } from '../_command-shared.js';
|
|
48
48
|
import { killProcessTree } from '../fiori/preview/registry.js';
|
|
49
|
+
import { spillToolResult } from '../result-spill.js';
|
|
49
50
|
const DEFAULT_TIMEOUT_MS = 30_000;
|
|
50
51
|
/** After the kill's grace: how long to wait for the pipes before returning anyway. */
|
|
51
52
|
const ORPHAN_WAIT_MS = 3_000;
|
|
52
53
|
const MAX_TIMEOUT_MS = 300_000;
|
|
53
54
|
const MAX_OUTPUT_BYTES = 1_000_000; // per stream
|
|
55
|
+
/**
|
|
56
|
+
* Task 17 (spec D12) — output above 8 000 chars is saved to the session spill
|
|
57
|
+
* file; the result carries the first 60 + last 20 lines (the exit code stays
|
|
58
|
+
* visible) and a pointer. Sized once, here, when the result is created.
|
|
59
|
+
*/
|
|
54
60
|
export async function shellExecHandler(args, ctx) {
|
|
61
|
+
return spillToolResult(await runShellExec(args, ctx), ctx, 'shell', { label: 'command output' });
|
|
62
|
+
}
|
|
63
|
+
async function runShellExec(args, ctx) {
|
|
55
64
|
if (!args.command || typeof args.command !== 'string') {
|
|
56
65
|
return { content: 'error: command is required (non-empty string)', is_error: true };
|
|
57
66
|
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Task 18 (ruling F21) — parallel readers share one AdtClient, and concurrent
|
|
3
|
+
* ADT requests through it are unverified. Readers run their MODEL work in
|
|
4
|
+
* parallel, but their SAP calls go through one lock: one ADT request at a
|
|
5
|
+
* time for the whole reader group.
|
|
6
|
+
*/
|
|
7
|
+
/** A FIFO mutex. A call that fails releases the lock like one that succeeds. */
|
|
8
|
+
export function createAdtLock() {
|
|
9
|
+
let tail = Promise.resolve();
|
|
10
|
+
return {
|
|
11
|
+
run(fn) {
|
|
12
|
+
const result = tail.then(() => fn());
|
|
13
|
+
tail = result.then(() => undefined, () => undefined);
|
|
14
|
+
return result;
|
|
15
|
+
},
|
|
16
|
+
};
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* A view of `adt` whose method calls run through `lock`. Methods run on the
|
|
20
|
+
* real client (`this` is the client, so its internal calls are not re-queued);
|
|
21
|
+
* non-function properties pass through. Every AdtClient method a reader tool
|
|
22
|
+
* calls is async, so returning a promise changes nothing for the caller.
|
|
23
|
+
*/
|
|
24
|
+
export function serializeAdt(adt, lock) {
|
|
25
|
+
return new Proxy(adt, {
|
|
26
|
+
get(target, prop, _receiver) {
|
|
27
|
+
const value = Reflect.get(target, prop, target);
|
|
28
|
+
if (typeof value !== 'function')
|
|
29
|
+
return value;
|
|
30
|
+
return (...args) => lock.run(() => value.apply(target, args));
|
|
31
|
+
},
|
|
32
|
+
});
|
|
33
|
+
}
|
|
@@ -122,6 +122,8 @@ export async function agentRunHandler(args, ctx) {
|
|
|
122
122
|
cwd: ctx.cwd,
|
|
123
123
|
provider: ctx.provider,
|
|
124
124
|
skillSource: ctx.skillSource,
|
|
125
|
+
// Task 18 fix M4 — an agent_run child never starts reader subagents.
|
|
126
|
+
subagent: true,
|
|
125
127
|
// previewHook + pendingDispatch + currentTransport + todoEmitter
|
|
126
128
|
// intentionally omitted.
|
|
127
129
|
};
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* read_agent — Task 18 (model control, 2026-09-27, spec D18).
|
|
3
|
+
*
|
|
4
|
+
* Heavy reading (many sources, a where-used walk, ATC-finding triage, a
|
|
5
|
+
* package inventory) runs in a lean child conversation and only a short
|
|
6
|
+
* report returns to the parent's history. Unlike agent_run (flag-gated, a
|
|
7
|
+
* full skill prefix per child), a reader carries:
|
|
8
|
+
* - a ~300-token system prompt owned by the proxy (skill header `_reader`,
|
|
9
|
+
* ruling F8; BYOK / AI-hub use the pinned copy in reader-prompt.ts),
|
|
10
|
+
* - a fixed small tool list (READER_TOOLS: read-only, no tool search),
|
|
11
|
+
* - the session model at effort `low`, role `reader` (never enforced, F7),
|
|
12
|
+
* - a capped report: overflow is saved to a file and the tail is cut.
|
|
13
|
+
*
|
|
14
|
+
* Read-only and depth 1 are enforced by `ctx.readerMode` at dispatch (F22).
|
|
15
|
+
* The loop runs consecutive read_agent calls concurrently, each with its own
|
|
16
|
+
* ctx clone whose `adt` is serialized through one lock (F21). The child's
|
|
17
|
+
* spend is added to the parent's per-turn `readerSpend` (F24); the child also
|
|
18
|
+
* writes its own reader-*-cost.jsonl.
|
|
19
|
+
*/
|
|
20
|
+
import { EventEmitter } from 'node:events';
|
|
21
|
+
import { listTools, registerTool, toolsForContext } from '../index.js';
|
|
22
|
+
import { runTurn } from '../../agent/loop.js';
|
|
23
|
+
import { collectTurnAssistantText } from '../../agent/turn-assistant-text.js';
|
|
24
|
+
import { newSession } from '../../session/schema.js';
|
|
25
|
+
import { computeSessionCost } from '../../cost/session-cost.js';
|
|
26
|
+
import { sessionIdOf, spillIfLarge } from '../result-spill.js';
|
|
27
|
+
import { INTERRUPTED_LINE, READER_SKILL, READER_SYSTEM_PROMPT, READER_TOOLS, hasReaderReadTools } from './reader-prompt.js';
|
|
28
|
+
export { INTERRUPTED_LINE, READER_SKILL, READER_SYSTEM_PROMPT, READER_TOOLS, hasReaderReadTools };
|
|
29
|
+
export const BRIEF_MAX_CHARS = 4_000;
|
|
30
|
+
export const DEFAULT_REPORT_MAX_CHARS = 3_000;
|
|
31
|
+
export const HARD_REPORT_MAX_CHARS = 6_000;
|
|
32
|
+
/**
|
|
33
|
+
* Final review tools-I4 — told to every reader. The proxy owns the reader's
|
|
34
|
+
* system prompt (pinned byte-equal), so the CLI says this in the child's
|
|
35
|
+
* user message.
|
|
36
|
+
*/
|
|
37
|
+
export const PARENT_OUTPUT_LINE = 'A saved-output path in this brief (a scan or ATC result the main session already ran) can be read with ' +
|
|
38
|
+
'tool_output_read. Read it instead of repeating the scan: never re-run ATC or a scan whose output you were given.';
|
|
39
|
+
/** The result when the account's proxy refuses the `_reader` skill. */
|
|
40
|
+
export const NOT_ENABLED_LINE = 'readers are not enabled for this account — read the sources directly';
|
|
41
|
+
function reportCap(raw) {
|
|
42
|
+
if (typeof raw !== 'number' || !Number.isFinite(raw) || raw < 1)
|
|
43
|
+
return DEFAULT_REPORT_MAX_CHARS;
|
|
44
|
+
return Math.min(Math.floor(raw), HARD_REPORT_MAX_CHARS);
|
|
45
|
+
}
|
|
46
|
+
function isNotEntitled(err) {
|
|
47
|
+
const e = err;
|
|
48
|
+
if (e?.error?.error === 'skill_not_in_entitlements')
|
|
49
|
+
return true;
|
|
50
|
+
return typeof e?.message === 'string' && e.message.includes('skill_not_in_entitlements');
|
|
51
|
+
}
|
|
52
|
+
export async function readAgentHandler(args, ctx) {
|
|
53
|
+
const brief = args?.brief;
|
|
54
|
+
if (typeof brief !== 'string' || brief.trim().length === 0) {
|
|
55
|
+
return { content: "error: 'brief' is required: say what to read and what to report.", is_error: true };
|
|
56
|
+
}
|
|
57
|
+
if (brief.length > BRIEF_MAX_CHARS) {
|
|
58
|
+
return { content: `error: 'brief' is ${brief.length} chars; keep it under ${BRIEF_MAX_CHARS}.`, is_error: true };
|
|
59
|
+
}
|
|
60
|
+
// Depth 1 (dispatch refuses this too; belt and braces for direct calls).
|
|
61
|
+
if (ctx.readerMode) {
|
|
62
|
+
return { content: 'error: a reader cannot start another reader.', is_error: true };
|
|
63
|
+
}
|
|
64
|
+
// Fix M4 — an agent_run child does its own reading.
|
|
65
|
+
if (ctx.subagent) {
|
|
66
|
+
return { content: 'error: a subagent cannot start a reader — read the sources directly.', is_error: true };
|
|
67
|
+
}
|
|
68
|
+
// Fix M5 — nothing a reader could read here (standalone, file tools off).
|
|
69
|
+
if (!hasReaderReadTools(new Set(toolsForContext(listTools(), ctx).map((t) => t.name)))) {
|
|
70
|
+
return { content: 'error: no read tools are available to a reader here — read the sources directly.', is_error: true };
|
|
71
|
+
}
|
|
72
|
+
if (ctx.signal?.aborted)
|
|
73
|
+
return { content: INTERRUPTED_LINE, is_error: true };
|
|
74
|
+
if (!ctx.provider) {
|
|
75
|
+
return { content: 'error: read_agent requires ctx.provider — not wired into this ToolContext.', is_error: true };
|
|
76
|
+
}
|
|
77
|
+
const cap = reportCap(args.report_max_chars);
|
|
78
|
+
// The session model, never a cheaper one (owner decision 11); effort low.
|
|
79
|
+
const model = ctx.session.model;
|
|
80
|
+
const childSession = newSession(`reader-${Date.now()}-${Math.floor(Math.random() * 1e9).toString(16)}`, ctx.session.sap_system ?? null, READER_SKILL, model);
|
|
81
|
+
// Only what a read needs. No previewHook (no writes), no chunkEmitter /
|
|
82
|
+
// todoEmitter (the child never paints the parent UI), no pendingDispatch /
|
|
83
|
+
// currentTransport (the child never changes parent REPL state).
|
|
84
|
+
const parentSessionId = sessionIdOf(ctx);
|
|
85
|
+
const childCtx = {
|
|
86
|
+
adt: ctx.adt,
|
|
87
|
+
sapAlias: ctx.sapAlias,
|
|
88
|
+
session: childSession,
|
|
89
|
+
cwd: ctx.cwd,
|
|
90
|
+
provider: ctx.provider,
|
|
91
|
+
skillSource: ctx.skillSource,
|
|
92
|
+
readerMode: true,
|
|
93
|
+
// Final review tools-I4 — the reader may open the parent's saved outputs.
|
|
94
|
+
...(parentSessionId !== undefined ? { spillParentSessionId: parentSessionId } : {}),
|
|
95
|
+
};
|
|
96
|
+
const userMessage = `${brief}\n\n${PARENT_OUTPUT_LINE}\n\nReport limit: ${cap} characters.`;
|
|
97
|
+
let failure = null;
|
|
98
|
+
try {
|
|
99
|
+
await runTurn({
|
|
100
|
+
provider: ctx.provider,
|
|
101
|
+
userMessage,
|
|
102
|
+
skill: READER_SKILL,
|
|
103
|
+
ctx: childCtx,
|
|
104
|
+
// Null sink: the child's text lands in childSession.messages only.
|
|
105
|
+
chunkEmitter: new EventEmitter(),
|
|
106
|
+
modelRole: 'reader',
|
|
107
|
+
toolsOverride: READER_TOOLS,
|
|
108
|
+
effortOverride: 'low',
|
|
109
|
+
modelOverride: model,
|
|
110
|
+
suppressSaveHook: true,
|
|
111
|
+
// Fix I1 — Esc / Ctrl+C on the parent turn stops the reader's stream.
|
|
112
|
+
signal: ctx.signal,
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
catch (err) {
|
|
116
|
+
failure = err;
|
|
117
|
+
}
|
|
118
|
+
finally {
|
|
119
|
+
// Spend counts whether or not the reader finished.
|
|
120
|
+
if (ctx.readerSpend) {
|
|
121
|
+
ctx.readerSpend.cost += computeSessionCost(childSession);
|
|
122
|
+
ctx.readerSpend.calls += 1;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
if (failure !== null) {
|
|
126
|
+
if (ctx.signal?.aborted)
|
|
127
|
+
return { content: INTERRUPTED_LINE, is_error: true };
|
|
128
|
+
if (isNotEntitled(failure))
|
|
129
|
+
return { content: NOT_ENABLED_LINE, is_error: true };
|
|
130
|
+
const msg = failure instanceof Error ? failure.message : String(failure);
|
|
131
|
+
return { content: `error: the reader failed — ${msg.slice(0, 300)}`, is_error: true };
|
|
132
|
+
}
|
|
133
|
+
const text = collectTurnAssistantText(childSession.messages, 0).trim();
|
|
134
|
+
if (text.length === 0) {
|
|
135
|
+
return {
|
|
136
|
+
content: 'error: the reader returned no report — read the sources directly.',
|
|
137
|
+
is_error: true,
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
if (text.length <= cap)
|
|
141
|
+
return { content: text };
|
|
142
|
+
// Over the cap: the full report goes to the PARENT's spill area under the
|
|
143
|
+
// parent's tool_use id, where tool_output_read can open it.
|
|
144
|
+
const head = text.slice(0, cap);
|
|
145
|
+
const sessionId = sessionIdOf(ctx);
|
|
146
|
+
const spilled = sessionId !== undefined
|
|
147
|
+
? spillIfLarge({ text, sessionId, toolUseId: ctx.toolUseId, kind: 'report', threshold: cap })
|
|
148
|
+
: null;
|
|
149
|
+
const where = spilled?.spilled ? ` — full report saved to ${spilled.path}` : '';
|
|
150
|
+
return { content: `${head}\n[report cut at ${cap} chars${where}]` };
|
|
151
|
+
}
|
|
152
|
+
registerTool({
|
|
153
|
+
name: 'read_agent',
|
|
154
|
+
description: 'Hand heavy reading to a lean read-only reader; only its short report comes back. Use it for more ' +
|
|
155
|
+
'than two source reads, a where-used scan, ATC-finding triage or a package inventory when the turn ' +
|
|
156
|
+
'will keep working afterwards. The reader has read-only SAP and file tools; it cannot write, ask the ' +
|
|
157
|
+
'user or start another reader. Give a precise brief: what to read and what to report. To hand over ' +
|
|
158
|
+
'a result you already have (ATC findings), put its saved-output path in the brief: the reader reads it ' +
|
|
159
|
+
'and never re-runs the scan. Send ' +
|
|
160
|
+
'independent readers in one response; they run at the same time.',
|
|
161
|
+
isMutating: false,
|
|
162
|
+
category: 'subagent',
|
|
163
|
+
input_schema: {
|
|
164
|
+
type: 'object',
|
|
165
|
+
properties: {
|
|
166
|
+
brief: {
|
|
167
|
+
type: 'string',
|
|
168
|
+
description: 'What to read (objects, includes, packages) and what to report. Max 4000 characters.',
|
|
169
|
+
},
|
|
170
|
+
report_max_chars: {
|
|
171
|
+
type: 'number',
|
|
172
|
+
description: 'Report length limit. Default 3000, max 6000. Longer reports are cut and saved to a file.',
|
|
173
|
+
},
|
|
174
|
+
},
|
|
175
|
+
required: ['brief'],
|
|
176
|
+
},
|
|
177
|
+
handler: (args, ctx) => readAgentHandler(args, ctx),
|
|
178
|
+
});
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Task 18 (model control, 2026-09-27, spec D18, ruling F8) — the reader
|
|
3
|
+
* subagent's system prompt and tool list.
|
|
4
|
+
*
|
|
5
|
+
* The PROXY owns READER_SYSTEM_PROMPT (cspeach-proxy/src/skills/
|
|
6
|
+
* reader-prompt.ts): for the skill header `_reader` it sends that constant
|
|
7
|
+
* and ignores the client's `system`. This is the CLI's pinned copy, used only
|
|
8
|
+
* where the CLI talks to a model directly (BYOK, AI-hub, local).
|
|
9
|
+
* src/tools/__tests__/reader-prompt-pin.test.ts asserts the two template
|
|
10
|
+
* literals stay byte-equal — edit both files together.
|
|
11
|
+
*
|
|
12
|
+
* No imports on purpose: tool-dispatch.ts reads READER_TOOLS, and importing
|
|
13
|
+
* read_agent.ts there would close a cycle through agent/loop.ts.
|
|
14
|
+
*/
|
|
15
|
+
/** The skill header a reader's model calls carry. */
|
|
16
|
+
export const READER_SKILL = '_reader';
|
|
17
|
+
export const READER_SYSTEM_PROMPT = `You are a read-only investigator inside CSPeach. Read exactly what the brief asks for with the tools you have, then answer the brief. Report facts with object names, includes and line numbers. Say what you could not find. No recommendations unless the brief asks. Stay under the report limit.`;
|
|
18
|
+
/**
|
|
19
|
+
* The reader's fixed, small, static tool list (no deferred entries, no tool
|
|
20
|
+
* search): the ten read-only tools of spec D18 plus tool_output_read (Task
|
|
21
|
+
* 17), so a reader can open a result that was saved to a file. Every entry is
|
|
22
|
+
* `isMutating: false`. Flag-gated entries (file_read, grep, glob) reach the
|
|
23
|
+
* reader only when the session has them on, and standalone hides the SAP ones.
|
|
24
|
+
* tool-dispatch.ts refuses every other tool while `ctx.readerMode` is set.
|
|
25
|
+
*/
|
|
26
|
+
export const READER_TOOLS = Object.freeze([
|
|
27
|
+
'sap_get_source',
|
|
28
|
+
'sap_object_structure',
|
|
29
|
+
'sap_search_object',
|
|
30
|
+
'sap_sql_query',
|
|
31
|
+
'sap_usage_references',
|
|
32
|
+
'sap_class_includes',
|
|
33
|
+
'sap_atc_run',
|
|
34
|
+
'file_read',
|
|
35
|
+
'grep',
|
|
36
|
+
'glob',
|
|
37
|
+
'tool_output_read',
|
|
38
|
+
]);
|
|
39
|
+
/** Task 18 fix I1 — the result of a reader stopped by Esc / Ctrl+C (or never started). */
|
|
40
|
+
export const INTERRUPTED_LINE = 'interrupted: the reader was stopped before it finished';
|
|
41
|
+
/**
|
|
42
|
+
* Task 18 fix M5 — true when a reader would have at least one real read tool
|
|
43
|
+
* among `visibleNames` (tool_output_read alone reads nothing new: standalone
|
|
44
|
+
* with the file tools off). The loop hides read_agent otherwise.
|
|
45
|
+
*/
|
|
46
|
+
export function hasReaderReadTools(visibleNames) {
|
|
47
|
+
return READER_TOOLS.some((n) => n !== 'tool_output_read' && visibleNames.has(n));
|
|
48
|
+
}
|
package/dist/tools/todo.js
CHANGED
|
@@ -124,7 +124,9 @@ registerTool({
|
|
|
124
124
|
'in_progress at a time. The user sees this list — keep items short and outcome-shaped. ' +
|
|
125
125
|
'Update at EVERY step transition: when you finish a step, immediately call todo_set ' +
|
|
126
126
|
'marking it completed and the next step in_progress. Never batch several finished steps ' +
|
|
127
|
-
'into one later call — the user watches the ▶ indicator live.'
|
|
127
|
+
'into one later call — the user watches the ▶ indicator live. ' +
|
|
128
|
+
// Task 16 (cost profile lever 7): a lone todo_set round re-reads the whole context.
|
|
129
|
+
'Call todo_set in the same response as your next action, never as the only tool call in a response.',
|
|
128
130
|
isMutating: false,
|
|
129
131
|
category: 'session',
|
|
130
132
|
input_schema: {
|