blun-king-cli 9.1.536 → 9.1.561
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LIESMICH.txt +13 -869
- package/README.md +41 -833
- package/bin/assistant-message-offload-policy.cjs +3 -1
- package/bin/compaction-transaction-policy.cjs +122 -0
- package/bin/context-performance-policy.cjs +2 -5
- package/bin/context-pressure-policy.cjs +20 -0
- package/bin/cron-run-output.cjs +45 -0
- package/bin/cron-run-store.cjs +145 -0
- package/bin/default-model-output-budget-policy.cjs +28 -0
- package/bin/durable-task-resume-policy.cjs +130 -0
- package/bin/durable-task-resume-runtime.cjs +117 -0
- package/bin/durable-task-resume-store.cjs +88 -0
- package/bin/editable-tool-approval-policy.cjs +540 -0
- package/bin/editable-tool-approval-runtime.cjs +99 -0
- package/bin/file-observation-policy.cjs +133 -0
- package/bin/html-to-research-markdown.cjs +146 -0
- package/bin/launcher-runtime.js +0 -1
- package/bin/micro-compaction-policy.cjs +64 -0
- package/bin/mnemo-connect-heartbeat.cjs +1 -3
- package/bin/programmatic-tool-runtime.mjs +330 -4
- package/bin/read-continuation-policy.cjs +36 -5
- package/bin/retry-checkpoint-policy.cjs +13 -0
- package/bin/scoped-cron-run-policy.cjs +358 -0
- package/bin/session-checkpoint-policy.cjs +25 -0
- package/bin/startup-preferences.cjs +4 -3
- package/bin/structured-agent-swarm-output.cjs +325 -0
- package/bin/subagent-context-fork-policy.cjs +155 -0
- package/bin/subagent-skill-policy.cjs +204 -0
- package/bin/telegram-approval-relay.cjs +2 -1
- package/bin/tool-file-persistence.cjs +141 -0
- package/bin/tool-result-offload-policy.cjs +25 -33
- package/bin/turn-thinking-policy.cjs +2 -26
- package/bin/update-notice.js +30 -18
- package/bin/user-message-offload-policy.cjs +3 -1
- package/blun.mjs +1564 -640
- package/codebase-index/README.md +12 -0
- package/codebase-index/codebase_index.py +129 -18
- package/package.json +23 -58
- package/telegram-plugin/bin/telegram-mnemo-capture.cjs +1 -3
- package/telegram-plugin/bin/telegram-typing-keepalive.cjs +89 -0
- package/telegram-plugin/dist/bridge.mjs +8 -1
- package/CHANGELOG.md +0 -321
- package/agent-spine-plugin/CHANGELOG.md +0 -406
- package/agent-spine-plugin/CONTRIBUTING.md +0 -52
- package/agent-spine-plugin/README.md +0 -344
- package/agent-spine-plugin/SECURITY.md +0 -47
- package/agent-spine-plugin/docs/acceptance.md +0 -61
- package/agent-spine-plugin/docs/architecture.md +0 -183
- package/agent-spine-plugin/docs/attention.md +0 -121
- package/agent-spine-plugin/docs/automatic-continuity.md +0 -79
- package/agent-spine-plugin/docs/channel-runtime.md +0 -92
- package/agent-spine-plugin/docs/coordination.md +0 -138
- package/agent-spine-plugin/docs/feed-transport.md +0 -99
- package/agent-spine-plugin/docs/gateway-runtime.md +0 -116
- package/agent-spine-plugin/docs/harness-reference.md +0 -45
- package/agent-spine-plugin/docs/host-integration.md +0 -129
- package/agent-spine-plugin/docs/https-transport.md +0 -116
- package/agent-spine-plugin/docs/learning.md +0 -133
- package/agent-spine-plugin/docs/object-transport.md +0 -93
- package/agent-spine-plugin/docs/peer-transport.md +0 -88
- package/agent-spine-plugin/docs/preflight-recall.md +0 -69
- package/agent-spine-plugin/docs/preservation-contract.md +0 -53
- package/agent-spine-plugin/docs/quality-gates.md +0 -50
- package/agent-spine-plugin/docs/relationships.md +0 -73
- package/agent-spine-plugin/docs/releasing.md +0 -83
- package/agent-spine-plugin/docs/roadmap.md +0 -307
- package/agent-spine-plugin/docs/selfstarter.md +0 -88
- package/agent-spine-plugin/docs/session-briefing.md +0 -74
- package/agent-spine-plugin/docs/shared-memory.md +0 -259
- package/agent-spine-plugin/docs/source-roots.md +0 -86
- package/agent-spine-plugin/docs/sqlite-transport.md +0 -76
- package/agent-spine-plugin/scripts/check-hosts.js +0 -195
- package/agent-spine-plugin/scripts/check-install.js +0 -569
- package/agent-spine-plugin/scripts/check-syntax.js +0 -29
- package/agent-spine-plugin/scripts/github-actions.js +0 -11
- package/agent-spine-plugin/scripts/release-check.js +0 -128
- package/agent-spine-plugin/scripts/run-acceptance.js +0 -19
- package/agent-spine-plugin/scripts/run-checks.js +0 -46
- package/agent-spine-plugin/scripts/run-tests-hermetic.js +0 -73
- package/agent-spine-plugin/spine-example/1-identity.md +0 -12
- package/agent-spine-plugin/spine-example/2-voice.md +0 -6
- package/agent-spine-plugin/spine-example/3-conduct.md +0 -8
- package/agent-spine-plugin/spine-example/4-history.md +0 -4
- package/bin/empty-response-retry-policy.cjs +0 -29
- package/bin/fredrik-glm-provider.cjs +0 -256
- package/bin/package-regression-policy.cjs +0 -77
- package/fredrik-glm-profile.toml.example +0 -26
- package/release-planned-removals.json +0 -15
- package/scripts/check-active-profile-plugin-startup.js +0 -36
- package/scripts/check-active-work-steer-regression.js +0 -46
- package/scripts/check-approval-observability-regression.js +0 -111
- package/scripts/check-approval-queue-shortcuts-regression.js +0 -65
- package/scripts/check-bundled-agent-spine-regression.js +0 -48
- package/scripts/check-codebase-search-packaging-regression.js +0 -92
- package/scripts/check-copy-command-regression.js +0 -74
- package/scripts/check-current-turn-read-pin-mutation-regression.js +0 -72
- package/scripts/check-current-turn-read-pin-regression.js +0 -94
- package/scripts/check-deepseek-native-max-regression.js +0 -49
- package/scripts/check-empty-response-effort-downgrade-regression.js +0 -48
- package/scripts/check-fredrik-glm-mutation-regression.js +0 -18
- package/scripts/check-fredrik-glm-regression.js +0 -169
- package/scripts/check-historical-tool-result-preview-regression.js +0 -77
- package/scripts/check-history-pressure-offload-regression.js +0 -77
- package/scripts/check-mcp-startup-wait-budget.js +0 -48
- package/scripts/check-package-regression.js +0 -38
- package/scripts/check-plugin-startup-regression.js +0 -53
- package/scripts/check-programmatic-context-isolation-regression.js +0 -193
- package/scripts/check-programmatic-tool-regression.js +0 -294
- package/scripts/check-queue-controls-regression.js +0 -189
- package/scripts/check-release-metadata.js +0 -103
- package/scripts/check-reload-agent-spine-regression.js +0 -76
- package/scripts/check-resume-replay-regression.js +0 -102
- package/scripts/check-session-cancel-regression.js +0 -43
- package/scripts/check-session-picker-resume-metrics-regression.js +0 -97
- package/scripts/check-session-start-hook-context-regression.js +0 -228
- package/scripts/check-shell-terminal-isolation-regression.js +0 -81
- package/scripts/check-slash-escape-regression.js +0 -89
- package/scripts/check-startup-swarm-command-regression.js +0 -24
- package/scripts/check-structured-subagent-output-regression.js +0 -331
- package/scripts/check-telegram-bridge-watchdog.js +0 -60
- package/scripts/check-telegram-direct-work-resume-regression.js +0 -53
- package/scripts/check-telegram-loop-exactly-once-regression.js +0 -71
- package/scripts/check-todo-loop-regression.js +0 -78
- package/scripts/check-todo-progress-regression.js +0 -416
- package/scripts/check-todo-recovery-catalog-regression.js +0 -50
- package/scripts/check-tool-schema-capacity-regression.js +0 -40
- package/scripts/programmatic-tool-runtime.test.mjs +0 -365
- package/scripts/structured-subagent-output.test.cjs +0 -170
- /package/{scripts → bin}/fix-node-pty-perms.js +0 -0
|
@@ -1,365 +0,0 @@
|
|
|
1
|
-
import assert from 'node:assert/strict';
|
|
2
|
-
import test from 'node:test';
|
|
3
|
-
|
|
4
|
-
import { createProgrammaticToolRuntime } from '../bin/programmatic-tool-runtime.mjs';
|
|
5
|
-
|
|
6
|
-
test('runs deterministic transforms and returns bounded state', async () => {
|
|
7
|
-
const runtime = createProgrammaticToolRuntime({
|
|
8
|
-
allowedTools: [],
|
|
9
|
-
invokeTool: async () => assert.fail('no tool call expected'),
|
|
10
|
-
});
|
|
11
|
-
|
|
12
|
-
const outcome = await runtime.run({
|
|
13
|
-
code: 'state.total = [3, 1, 2].sort().reduce((a, b) => a + b, 0); return state.total;',
|
|
14
|
-
state: {},
|
|
15
|
-
});
|
|
16
|
-
|
|
17
|
-
assert.equal(outcome.result, 6);
|
|
18
|
-
assert.deepEqual(outcome.state, { total: 6 });
|
|
19
|
-
assert.equal(outcome.callCount, 0);
|
|
20
|
-
});
|
|
21
|
-
|
|
22
|
-
test('exposes only explicitly allowed tools through one host invocation boundary', async () => {
|
|
23
|
-
const calls = [];
|
|
24
|
-
const runtime = createProgrammaticToolRuntime({
|
|
25
|
-
allowedTools: [{ name: 'lookup_record', alias: 'lookupRecord' }],
|
|
26
|
-
invokeTool: async (call) => {
|
|
27
|
-
calls.push(call);
|
|
28
|
-
return { id: call.args.id, label: `record-${call.args.id}` };
|
|
29
|
-
},
|
|
30
|
-
});
|
|
31
|
-
|
|
32
|
-
const outcome = await runtime.run({
|
|
33
|
-
code: 'return tools.lookupRecord({ id: 7 });',
|
|
34
|
-
state: {},
|
|
35
|
-
});
|
|
36
|
-
|
|
37
|
-
assert.deepEqual(outcome.result, { id: 7, label: 'record-7' });
|
|
38
|
-
assert.equal(calls.length, 1);
|
|
39
|
-
assert.equal(calls[0].name, 'lookup_record');
|
|
40
|
-
assert.deepEqual(calls[0].args, { id: 7 });
|
|
41
|
-
assert.match(calls[0].callId, /^ptc_lookup_record_/u);
|
|
42
|
-
});
|
|
43
|
-
|
|
44
|
-
test('does not expose host filesystem, network, process, clock, or dynamic eval', async () => {
|
|
45
|
-
const runtime = createProgrammaticToolRuntime({
|
|
46
|
-
allowedTools: [],
|
|
47
|
-
invokeTool: async () => assert.fail('no tool call expected'),
|
|
48
|
-
});
|
|
49
|
-
|
|
50
|
-
const outcome = await runtime.run({
|
|
51
|
-
code: `return {
|
|
52
|
-
process: typeof process,
|
|
53
|
-
require: typeof require,
|
|
54
|
-
fetch: typeof fetch,
|
|
55
|
-
date: typeof Date,
|
|
56
|
-
performance: typeof performance,
|
|
57
|
-
eval: typeof eval,
|
|
58
|
-
fn: typeof Function,
|
|
59
|
-
};`,
|
|
60
|
-
state: {},
|
|
61
|
-
});
|
|
62
|
-
|
|
63
|
-
assert.deepEqual(outcome.result, {
|
|
64
|
-
process: 'undefined',
|
|
65
|
-
require: 'undefined',
|
|
66
|
-
fetch: 'undefined',
|
|
67
|
-
date: 'undefined',
|
|
68
|
-
performance: 'undefined',
|
|
69
|
-
eval: 'undefined',
|
|
70
|
-
fn: 'undefined',
|
|
71
|
-
});
|
|
72
|
-
});
|
|
73
|
-
|
|
74
|
-
test('rejects unknown or invalid tool aliases before execution', () => {
|
|
75
|
-
assert.throws(
|
|
76
|
-
() => createProgrammaticToolRuntime({
|
|
77
|
-
allowedTools: [{ name: 'Read', alias: 'bad-alias' }],
|
|
78
|
-
invokeTool: async () => ({}),
|
|
79
|
-
}),
|
|
80
|
-
/valid JavaScript identifier/u,
|
|
81
|
-
);
|
|
82
|
-
|
|
83
|
-
assert.throws(
|
|
84
|
-
() => createProgrammaticToolRuntime({
|
|
85
|
-
allowedTools: [
|
|
86
|
-
{ name: 'Read', alias: 'read' },
|
|
87
|
-
{ name: 'OtherRead', alias: 'read' },
|
|
88
|
-
],
|
|
89
|
-
invokeTool: async () => ({}),
|
|
90
|
-
}),
|
|
91
|
-
/duplicate tool alias/u,
|
|
92
|
-
);
|
|
93
|
-
});
|
|
94
|
-
|
|
95
|
-
test('fails closed when the program exceeds the tool-call limit', async () => {
|
|
96
|
-
let calls = 0;
|
|
97
|
-
const runtime = createProgrammaticToolRuntime({
|
|
98
|
-
allowedTools: [{ name: 'counter', alias: 'counter' }],
|
|
99
|
-
invokeTool: async () => ++calls,
|
|
100
|
-
limits: { maxToolCalls: 2 },
|
|
101
|
-
});
|
|
102
|
-
|
|
103
|
-
await assert.rejects(
|
|
104
|
-
runtime.run({
|
|
105
|
-
code: 'tools.counter({}); tools.counter({}); tools.counter({}); return 3;',
|
|
106
|
-
state: {},
|
|
107
|
-
}),
|
|
108
|
-
/tool-call limit of 2/u,
|
|
109
|
-
);
|
|
110
|
-
assert.equal(calls, 2);
|
|
111
|
-
});
|
|
112
|
-
|
|
113
|
-
test('interrupts non-terminating code at the execution deadline', async () => {
|
|
114
|
-
const runtime = createProgrammaticToolRuntime({
|
|
115
|
-
allowedTools: [],
|
|
116
|
-
invokeTool: async () => assert.fail('no tool call expected'),
|
|
117
|
-
limits: { guestExecutionMs: 40, timeoutMs: 1_000 },
|
|
118
|
-
});
|
|
119
|
-
|
|
120
|
-
await assert.rejects(
|
|
121
|
-
runtime.run({ code: 'while (true) {}', state: {} }),
|
|
122
|
-
/execution deadline/u,
|
|
123
|
-
);
|
|
124
|
-
});
|
|
125
|
-
|
|
126
|
-
test('does not charge host waiting time against the guest execution budget', async () => {
|
|
127
|
-
const runtime = createProgrammaticToolRuntime({
|
|
128
|
-
allowedTools: [{ name: 'slow', alias: 'slow' }],
|
|
129
|
-
invokeTool: async ({ args }) => {
|
|
130
|
-
await new Promise((resolve) => setTimeout(resolve, 75));
|
|
131
|
-
return args.value;
|
|
132
|
-
},
|
|
133
|
-
limits: { guestExecutionMs: 30, timeoutMs: 250 },
|
|
134
|
-
});
|
|
135
|
-
|
|
136
|
-
const outcome = await runtime.run({
|
|
137
|
-
code: 'const value = tools.slow({ value: 8 }); return value + 1;',
|
|
138
|
-
state: {},
|
|
139
|
-
});
|
|
140
|
-
|
|
141
|
-
assert.equal(outcome.result, 9);
|
|
142
|
-
});
|
|
143
|
-
|
|
144
|
-
test('interrupts a pending host call at the wall deadline', async () => {
|
|
145
|
-
const runtime = createProgrammaticToolRuntime({
|
|
146
|
-
allowedTools: [{ name: 'wait', alias: 'wait' }],
|
|
147
|
-
invokeTool: async () => new Promise(() => {}),
|
|
148
|
-
limits: { guestExecutionMs: 1_000, timeoutMs: 40 },
|
|
149
|
-
});
|
|
150
|
-
|
|
151
|
-
await assert.rejects(
|
|
152
|
-
runtime.run({ code: 'return tools.wait({});', state: {} }),
|
|
153
|
-
/execution deadline exceeded/u,
|
|
154
|
-
);
|
|
155
|
-
});
|
|
156
|
-
|
|
157
|
-
test('rejects oversized result and state payloads', async () => {
|
|
158
|
-
const runtime = createProgrammaticToolRuntime({
|
|
159
|
-
allowedTools: [],
|
|
160
|
-
invokeTool: async () => assert.fail('no tool call expected'),
|
|
161
|
-
limits: { maxOutputChars: 32, maxStateChars: 32 },
|
|
162
|
-
});
|
|
163
|
-
|
|
164
|
-
await assert.rejects(
|
|
165
|
-
runtime.run({ code: 'return "x".repeat(100);', state: {} }),
|
|
166
|
-
/result exceeds 32 characters/u,
|
|
167
|
-
);
|
|
168
|
-
await assert.rejects(
|
|
169
|
-
runtime.run({ code: 'state.value = "x".repeat(100); return 1;', state: {} }),
|
|
170
|
-
/state exceeds 32 characters/u,
|
|
171
|
-
);
|
|
172
|
-
});
|
|
173
|
-
|
|
174
|
-
test('emits observable start, nested tool, and completion events', async () => {
|
|
175
|
-
const events = [];
|
|
176
|
-
const runtime = createProgrammaticToolRuntime({
|
|
177
|
-
allowedTools: [{ name: 'echo', alias: 'echo' }],
|
|
178
|
-
invokeTool: async ({ args }) => args,
|
|
179
|
-
onEvent: (event) => events.push(event),
|
|
180
|
-
});
|
|
181
|
-
|
|
182
|
-
await runtime.run({ code: 'return tools.echo({ value: 9 });', state: {} });
|
|
183
|
-
|
|
184
|
-
assert.deepEqual(events.map((event) => event.type), [
|
|
185
|
-
'interpreter.started',
|
|
186
|
-
'interpreter.tool_call',
|
|
187
|
-
'interpreter.tool_result',
|
|
188
|
-
'interpreter.completed',
|
|
189
|
-
]);
|
|
190
|
-
assert.equal(events[1].toolName, 'echo');
|
|
191
|
-
assert.equal(events[2].isError, false);
|
|
192
|
-
});
|
|
193
|
-
|
|
194
|
-
test('rejects a tool failure without hiding it from observability', async () => {
|
|
195
|
-
const events = [];
|
|
196
|
-
const runtime = createProgrammaticToolRuntime({
|
|
197
|
-
allowedTools: [{ name: 'fail', alias: 'fail' }],
|
|
198
|
-
invokeTool: async () => {
|
|
199
|
-
throw new Error('approval denied');
|
|
200
|
-
},
|
|
201
|
-
onEvent: (event) => events.push(event),
|
|
202
|
-
});
|
|
203
|
-
|
|
204
|
-
await assert.rejects(
|
|
205
|
-
runtime.run({ code: 'return tools.fail({});', state: {} }),
|
|
206
|
-
/approval denied/u,
|
|
207
|
-
);
|
|
208
|
-
assert.deepEqual(events.map((event) => event.type), [
|
|
209
|
-
'interpreter.started',
|
|
210
|
-
'interpreter.tool_call',
|
|
211
|
-
'interpreter.tool_result',
|
|
212
|
-
'interpreter.failed',
|
|
213
|
-
]);
|
|
214
|
-
assert.equal(events[2].isError, true);
|
|
215
|
-
});
|
|
216
|
-
|
|
217
|
-
test('preserves an external abort reason while cancelling a pending host call', async () => {
|
|
218
|
-
const controller = new AbortController();
|
|
219
|
-
let hostSignal;
|
|
220
|
-
const runtime = createProgrammaticToolRuntime({
|
|
221
|
-
allowedTools: [{ name: 'wait', alias: 'wait' }],
|
|
222
|
-
invokeTool: ({ signal }) => {
|
|
223
|
-
hostSignal = signal;
|
|
224
|
-
return new Promise(() => {});
|
|
225
|
-
},
|
|
226
|
-
limits: { timeoutMs: 1_000 },
|
|
227
|
-
});
|
|
228
|
-
|
|
229
|
-
const running = runtime.run({
|
|
230
|
-
code: 'return tools.wait({});',
|
|
231
|
-
state: {},
|
|
232
|
-
signal: controller.signal,
|
|
233
|
-
});
|
|
234
|
-
setTimeout(() => controller.abort(new Error('operator cancelled')), 10);
|
|
235
|
-
|
|
236
|
-
await assert.rejects(running, /operator cancelled/u);
|
|
237
|
-
assert.equal(hostSignal.aborted, true);
|
|
238
|
-
});
|
|
239
|
-
|
|
240
|
-
test('rejects oversized and cyclic tool results at the host boundary', async () => {
|
|
241
|
-
const oversized = createProgrammaticToolRuntime({
|
|
242
|
-
allowedTools: [{ name: 'large', alias: 'large' }],
|
|
243
|
-
invokeTool: async () => ({ value: 'x'.repeat(100) }),
|
|
244
|
-
limits: { maxToolResultChars: 32 },
|
|
245
|
-
});
|
|
246
|
-
await assert.rejects(
|
|
247
|
-
oversized.run({ code: 'return tools.large({});', state: {} }),
|
|
248
|
-
/tool result exceeds 32 characters/u,
|
|
249
|
-
);
|
|
250
|
-
|
|
251
|
-
const cyclic = {};
|
|
252
|
-
cyclic.self = cyclic;
|
|
253
|
-
const nonJson = createProgrammaticToolRuntime({
|
|
254
|
-
allowedTools: [{ name: 'cyclic', alias: 'cyclic' }],
|
|
255
|
-
invokeTool: async () => cyclic,
|
|
256
|
-
});
|
|
257
|
-
await assert.rejects(
|
|
258
|
-
nonJson.run({ code: 'return tools.cyclic({});', state: {} }),
|
|
259
|
-
/tool cyclic result is not JSON serializable/u,
|
|
260
|
-
);
|
|
261
|
-
});
|
|
262
|
-
|
|
263
|
-
test('rejects oversized code, initial state, and tool arguments before crossing limits', async () => {
|
|
264
|
-
const runtime = createProgrammaticToolRuntime({
|
|
265
|
-
allowedTools: [{ name: 'echo', alias: 'echo' }],
|
|
266
|
-
invokeTool: async ({ args }) => args,
|
|
267
|
-
limits: {
|
|
268
|
-
maxCodeChars: 24,
|
|
269
|
-
maxStateChars: 24,
|
|
270
|
-
maxToolArgumentChars: 24,
|
|
271
|
-
},
|
|
272
|
-
});
|
|
273
|
-
|
|
274
|
-
await assert.rejects(
|
|
275
|
-
runtime.run({ code: 'return "0123456789012345678901234";', state: {} }),
|
|
276
|
-
/program exceeds 24 characters/u,
|
|
277
|
-
);
|
|
278
|
-
await assert.rejects(
|
|
279
|
-
runtime.run({ code: 'return 1;', state: { value: 'x'.repeat(30) } }),
|
|
280
|
-
/state exceeds 24 characters/u,
|
|
281
|
-
);
|
|
282
|
-
await assert.rejects(
|
|
283
|
-
runtime.run({ code: 'return tools.echo({value:"xxxxxxxxxxxxxxxxxxxxxxxx"});', state: {} }),
|
|
284
|
-
/program exceeds 24 characters/u,
|
|
285
|
-
);
|
|
286
|
-
|
|
287
|
-
const argumentRuntime = createProgrammaticToolRuntime({
|
|
288
|
-
allowedTools: [{ name: 'echo', alias: 'echo' }],
|
|
289
|
-
invokeTool: async ({ args }) => args,
|
|
290
|
-
limits: { maxToolArgumentChars: 24 },
|
|
291
|
-
});
|
|
292
|
-
await assert.rejects(
|
|
293
|
-
argumentRuntime.run({ code: 'return tools.echo({ value: "x".repeat(30) });', state: {} }),
|
|
294
|
-
/tool arguments exceed 24 characters/u,
|
|
295
|
-
);
|
|
296
|
-
});
|
|
297
|
-
|
|
298
|
-
test('does not leave the private host bridge reachable from guest code', async () => {
|
|
299
|
-
const runtime = createProgrammaticToolRuntime({
|
|
300
|
-
allowedTools: [{ name: 'echo', alias: 'echo' }],
|
|
301
|
-
invokeTool: async ({ args }) => args,
|
|
302
|
-
});
|
|
303
|
-
|
|
304
|
-
const outcome = await runtime.run({
|
|
305
|
-
code: `return {
|
|
306
|
-
direct: typeof globalThis.__blun_ptc_0,
|
|
307
|
-
leakedKeys: Object.keys(globalThis).filter((key) => key.startsWith('__blun_ptc_')),
|
|
308
|
-
normalCall: tools.echo({ value: 4 }),
|
|
309
|
-
};`,
|
|
310
|
-
state: {},
|
|
311
|
-
});
|
|
312
|
-
|
|
313
|
-
assert.deepEqual(outcome.result, {
|
|
314
|
-
direct: 'undefined',
|
|
315
|
-
leakedKeys: [],
|
|
316
|
-
normalCall: { value: 4 },
|
|
317
|
-
});
|
|
318
|
-
});
|
|
319
|
-
|
|
320
|
-
test('continues only through an explicit JSON state snapshot', async () => {
|
|
321
|
-
const runtime = createProgrammaticToolRuntime({
|
|
322
|
-
allowedTools: [],
|
|
323
|
-
invokeTool: async () => assert.fail('no tool call expected'),
|
|
324
|
-
});
|
|
325
|
-
|
|
326
|
-
const first = await runtime.run({
|
|
327
|
-
code: 'state.count = (state.count ?? 0) + 1; return state.count;',
|
|
328
|
-
state: {},
|
|
329
|
-
});
|
|
330
|
-
const second = await runtime.run({
|
|
331
|
-
code: 'state.count += 1; return state.count;',
|
|
332
|
-
state: first.state,
|
|
333
|
-
});
|
|
334
|
-
|
|
335
|
-
assert.deepEqual(first, { result: 1, state: { count: 1 }, callCount: 0 });
|
|
336
|
-
assert.deepEqual(second, { result: 2, state: { count: 2 }, callCount: 0 });
|
|
337
|
-
});
|
|
338
|
-
|
|
339
|
-
test('enforces the guest memory ceiling', async () => {
|
|
340
|
-
const runtime = createProgrammaticToolRuntime({
|
|
341
|
-
allowedTools: [],
|
|
342
|
-
invokeTool: async () => assert.fail('no tool call expected'),
|
|
343
|
-
limits: { memoryBytes: 1024 * 1024, timeoutMs: 1_000 },
|
|
344
|
-
});
|
|
345
|
-
|
|
346
|
-
await assert.rejects(
|
|
347
|
-
runtime.run({ code: 'return "x".repeat(10_000_000);', state: {} }),
|
|
348
|
-
/memory|allocation|null/iu,
|
|
349
|
-
);
|
|
350
|
-
});
|
|
351
|
-
|
|
352
|
-
test('releases isolated contexts across repeated async tool runs', async () => {
|
|
353
|
-
const runtime = createProgrammaticToolRuntime({
|
|
354
|
-
allowedTools: [{ name: 'increment', alias: 'increment' }],
|
|
355
|
-
invokeTool: async ({ args }) => args.value + 1,
|
|
356
|
-
});
|
|
357
|
-
|
|
358
|
-
for (let index = 0; index < 20; index += 1) {
|
|
359
|
-
const outcome = await runtime.run({
|
|
360
|
-
code: `return tools.increment({ value: ${index} });`,
|
|
361
|
-
state: {},
|
|
362
|
-
});
|
|
363
|
-
assert.equal(outcome.result, index + 1);
|
|
364
|
-
}
|
|
365
|
-
});
|
|
@@ -1,170 +0,0 @@
|
|
|
1
|
-
'use strict';
|
|
2
|
-
|
|
3
|
-
const assert = require('node:assert/strict');
|
|
4
|
-
const test = require('node:test');
|
|
5
|
-
const {
|
|
6
|
-
MAX_SCHEMA_BYTES,
|
|
7
|
-
MAX_SCHEMA_DEPTH,
|
|
8
|
-
MAX_SCHEMA_NODES,
|
|
9
|
-
prepareStructuredResponseFormat,
|
|
10
|
-
validateStructuredSubagentOutput,
|
|
11
|
-
buildStructuredResponseFormatReminder,
|
|
12
|
-
buildStructuredResponseRepairPrompt,
|
|
13
|
-
buildStructuredSubagentEnvelope,
|
|
14
|
-
} = require('../bin/structured-subagent-output.cjs');
|
|
15
|
-
|
|
16
|
-
const reviewFormat = {
|
|
17
|
-
name: 'review_result',
|
|
18
|
-
schema: {
|
|
19
|
-
type: 'object',
|
|
20
|
-
properties: {
|
|
21
|
-
status: { type: 'string', enum: ['pass', 'fail'] },
|
|
22
|
-
findings: { type: 'array', items: { type: 'string' }, maxItems: 20 },
|
|
23
|
-
},
|
|
24
|
-
required: ['status', 'findings'],
|
|
25
|
-
additionalProperties: false,
|
|
26
|
-
},
|
|
27
|
-
};
|
|
28
|
-
|
|
29
|
-
function compileReviewValidator(schema) {
|
|
30
|
-
assert.deepEqual(schema, reviewFormat.schema);
|
|
31
|
-
const validator = (value) => {
|
|
32
|
-
const errors = [];
|
|
33
|
-
if (value === null || typeof value !== 'object' || Array.isArray(value)) {
|
|
34
|
-
errors.push({ instancePath: '', message: 'must be object' });
|
|
35
|
-
} else {
|
|
36
|
-
for (const required of schema.required) {
|
|
37
|
-
if (!(required in value)) errors.push({ keyword: 'required', instancePath: '', params: { missingProperty: required } });
|
|
38
|
-
}
|
|
39
|
-
for (const key of Object.keys(value)) {
|
|
40
|
-
if (!(key in schema.properties)) errors.push({ keyword: 'additionalProperties', instancePath: '', params: { additionalProperty: key } });
|
|
41
|
-
}
|
|
42
|
-
if ('status' in value && !schema.properties.status.enum.includes(value.status)) {
|
|
43
|
-
errors.push({ instancePath: '/status', message: 'must be equal to one of the allowed values' });
|
|
44
|
-
}
|
|
45
|
-
if ('findings' in value && (!Array.isArray(value.findings) || value.findings.some((item) => typeof item !== 'string'))) {
|
|
46
|
-
errors.push({ instancePath: '/findings', message: 'must be an array of strings' });
|
|
47
|
-
}
|
|
48
|
-
}
|
|
49
|
-
validator.errors = errors;
|
|
50
|
-
return errors.length === 0;
|
|
51
|
-
};
|
|
52
|
-
validator.errors = [];
|
|
53
|
-
return validator;
|
|
54
|
-
}
|
|
55
|
-
|
|
56
|
-
test('valid output returns parsed object and deterministic envelope', () => {
|
|
57
|
-
const prepared = prepareStructuredResponseFormat(reviewFormat, compileReviewValidator);
|
|
58
|
-
const validation = validateStructuredSubagentOutput('{"status":"pass","findings":[]}', prepared);
|
|
59
|
-
assert.equal(validation.ok, true);
|
|
60
|
-
assert.deepEqual(validation.value, { status: 'pass', findings: [] });
|
|
61
|
-
assert.equal(buildStructuredSubagentEnvelope({
|
|
62
|
-
agentId: 'agent-1',
|
|
63
|
-
profileName: 'coder',
|
|
64
|
-
responseFormat: prepared,
|
|
65
|
-
result: validation.value,
|
|
66
|
-
usage: { inputOther: 7, output: 5, inputCacheRead: 3, inputCacheCreation: 2 },
|
|
67
|
-
}), '{"agent_id":"agent-1","actual_subagent_type":"coder","status":"completed","response_format":"review_result","result":{"status":"pass","findings":[]},"usage":{"input":7,"output":5,"cache_read":3,"cache_write":2}}');
|
|
68
|
-
});
|
|
69
|
-
|
|
70
|
-
test('invalid JSON, leading prose, and fenced JSON fail exact parsing', () => {
|
|
71
|
-
const prepared = prepareStructuredResponseFormat(reviewFormat, compileReviewValidator);
|
|
72
|
-
for (const text of [
|
|
73
|
-
'{bad',
|
|
74
|
-
'Here is the result: {"status":"pass","findings":[]}',
|
|
75
|
-
'```json\n{"status":"pass","findings":[]}\n```',
|
|
76
|
-
]) {
|
|
77
|
-
const result = validateStructuredSubagentOutput(text, prepared);
|
|
78
|
-
assert.equal(result.ok, false);
|
|
79
|
-
assert.match(result.error, /not exactly one JSON value/u);
|
|
80
|
-
}
|
|
81
|
-
});
|
|
82
|
-
|
|
83
|
-
test('validator errors are concrete and bounded', () => {
|
|
84
|
-
const prepared = prepareStructuredResponseFormat(reviewFormat, compileReviewValidator);
|
|
85
|
-
const missing = validateStructuredSubagentOutput('{"status":"pass"}', prepared);
|
|
86
|
-
assert.equal(missing.ok, false);
|
|
87
|
-
assert.match(missing.error, /required property "findings"/u);
|
|
88
|
-
const extra = validateStructuredSubagentOutput('{"status":"pass","findings":[],"extra":1}', prepared);
|
|
89
|
-
assert.equal(extra.ok, false);
|
|
90
|
-
assert.match(extra.error, /additional property "extra"/u);
|
|
91
|
-
});
|
|
92
|
-
|
|
93
|
-
test('validator exceptions fail closed and remain distinguishable', () => {
|
|
94
|
-
const prepared = prepareStructuredResponseFormat(reviewFormat, () => () => {
|
|
95
|
-
throw new Error('compiler runtime exploded');
|
|
96
|
-
});
|
|
97
|
-
const result = validateStructuredSubagentOutput('{"status":"pass","findings":[]}', prepared);
|
|
98
|
-
assert.equal(result.ok, false);
|
|
99
|
-
assert.equal(result.validatorException, true);
|
|
100
|
-
assert.match(result.error, /validator exception/u);
|
|
101
|
-
});
|
|
102
|
-
|
|
103
|
-
test('reminder and repair prompt carry only the bounded schema contract', () => {
|
|
104
|
-
const prepared = prepareStructuredResponseFormat(reviewFormat, compileReviewValidator);
|
|
105
|
-
const reminder = buildStructuredResponseFormatReminder(prepared);
|
|
106
|
-
assert.match(reminder, /exactly one JSON object/u);
|
|
107
|
-
assert.match(reminder, /grants no additional capability/u);
|
|
108
|
-
assert.match(reminder, /review_result/u);
|
|
109
|
-
const repair = buildStructuredResponseRepairPrompt(prepared, 'missing findings');
|
|
110
|
-
assert.match(repair, /missing findings/u);
|
|
111
|
-
assert.match(repair, /nothing else/u);
|
|
112
|
-
});
|
|
113
|
-
|
|
114
|
-
test('malformed and unsupported schemas fail before compilation', () => {
|
|
115
|
-
let compileCalls = 0;
|
|
116
|
-
const compile = () => {
|
|
117
|
-
compileCalls += 1;
|
|
118
|
-
return () => true;
|
|
119
|
-
};
|
|
120
|
-
const rejected = [
|
|
121
|
-
{ name: 'x', schema: { type: 'array', items: { type: 'string' } } },
|
|
122
|
-
{ name: 'x', schema: { type: 'object', $ref: 'https://example.test/schema.json' } },
|
|
123
|
-
{ name: 'x', schema: { type: 'object', properties: { value: { type: 'string', pattern: '.*' } } } },
|
|
124
|
-
{ name: 'bad name', schema: { type: 'object' } },
|
|
125
|
-
];
|
|
126
|
-
for (const responseFormat of rejected) {
|
|
127
|
-
assert.throws(() => prepareStructuredResponseFormat(responseFormat, compile));
|
|
128
|
-
}
|
|
129
|
-
assert.equal(compileCalls, 0);
|
|
130
|
-
});
|
|
131
|
-
|
|
132
|
-
test('oversized schema fails before compilation', () => {
|
|
133
|
-
const huge = {
|
|
134
|
-
name: 'huge',
|
|
135
|
-
schema: { type: 'object', description: 'x'.repeat(MAX_SCHEMA_BYTES) },
|
|
136
|
-
};
|
|
137
|
-
assert.throws(() => prepareStructuredResponseFormat(huge, () => assert.fail('must not compile')), /exceeds/u);
|
|
138
|
-
});
|
|
139
|
-
|
|
140
|
-
test('deep schema fails before compilation', () => {
|
|
141
|
-
let node = { type: 'string' };
|
|
142
|
-
for (let index = 0; index < MAX_SCHEMA_DEPTH; index += 1) {
|
|
143
|
-
node = { type: 'array', items: node };
|
|
144
|
-
}
|
|
145
|
-
assert.throws(() => prepareStructuredResponseFormat({
|
|
146
|
-
name: 'deep',
|
|
147
|
-
schema: { type: 'object', properties: { value: node } },
|
|
148
|
-
}, () => assert.fail('must not compile')), /maximum depth/u);
|
|
149
|
-
});
|
|
150
|
-
|
|
151
|
-
test('node-heavy schema fails before compilation', () => {
|
|
152
|
-
const properties = {};
|
|
153
|
-
for (let index = 0; index < MAX_SCHEMA_NODES; index += 1) properties[`p${index}`] = { type: 'string' };
|
|
154
|
-
assert.throws(() => prepareStructuredResponseFormat({
|
|
155
|
-
name: 'wide',
|
|
156
|
-
schema: { type: 'object', properties },
|
|
157
|
-
}, () => assert.fail('must not compile')), /maximum node count/u);
|
|
158
|
-
});
|
|
159
|
-
|
|
160
|
-
test('validator is compiled exactly once and schema is cloned', () => {
|
|
161
|
-
let compileCalls = 0;
|
|
162
|
-
const input = structuredClone(reviewFormat);
|
|
163
|
-
const prepared = prepareStructuredResponseFormat(input, (schema) => {
|
|
164
|
-
compileCalls += 1;
|
|
165
|
-
return compileReviewValidator(schema);
|
|
166
|
-
});
|
|
167
|
-
input.schema.properties.status.enum.push('later');
|
|
168
|
-
assert.equal(compileCalls, 1);
|
|
169
|
-
assert.deepEqual(prepared.schema.properties.status.enum, ['pass', 'fail']);
|
|
170
|
-
});
|
|
File without changes
|