@basein/runner 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +276 -0
- package/dist/auth/client.d.ts +85 -0
- package/dist/auth/client.js +284 -0
- package/dist/bin/bir-hooks.d.ts +48 -0
- package/dist/bin/bir-hooks.js +201 -0
- package/dist/bin/bir-proxy.d.ts +45 -0
- package/dist/bin/bir-proxy.js +207 -0
- package/dist/bin/bir-scenario.d.ts +24 -0
- package/dist/bin/bir-scenario.js +177 -0
- package/dist/bin/bir.d.ts +21 -0
- package/dist/bin/bir.js +876 -0
- package/dist/config/adapters/claude-code.d.ts +76 -0
- package/dist/config/adapters/claude-code.js +181 -0
- package/dist/config/adapters/generic.d.ts +17 -0
- package/dist/config/adapters/generic.js +36 -0
- package/dist/config/generate.d.ts +127 -0
- package/dist/config/generate.js +114 -0
- package/dist/config/resolve.d.ts +68 -0
- package/dist/config/resolve.js +132 -0
- package/dist/control/client.d.ts +56 -0
- package/dist/control/client.js +86 -0
- package/dist/control/correlation.d.ts +86 -0
- package/dist/control/correlation.js +0 -0
- package/dist/control/discovery.d.ts +50 -0
- package/dist/control/discovery.js +123 -0
- package/dist/control/ordering.d.ts +38 -0
- package/dist/control/ordering.js +44 -0
- package/dist/control/paths.d.ts +32 -0
- package/dist/control/paths.js +56 -0
- package/dist/control/server.d.ts +272 -0
- package/dist/control/server.js +1131 -0
- package/dist/control/transcript.d.ts +75 -0
- package/dist/control/transcript.js +241 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +32 -0
- package/dist/jsonrpc/framing.d.ts +49 -0
- package/dist/jsonrpc/framing.js +143 -0
- package/dist/jsonrpc/types.d.ts +52 -0
- package/dist/jsonrpc/types.js +46 -0
- package/dist/proxy/intercept.d.ts +55 -0
- package/dist/proxy/intercept.js +147 -0
- package/dist/proxy/relay.d.ts +97 -0
- package/dist/proxy/relay.js +166 -0
- package/dist/proxy/session.d.ts +116 -0
- package/dist/proxy/session.js +319 -0
- package/dist/record/housekeeping.d.ts +34 -0
- package/dist/record/housekeeping.js +39 -0
- package/dist/record/queue.d.ts +48 -0
- package/dist/record/queue.js +96 -0
- package/dist/record/recorder.d.ts +111 -0
- package/dist/record/recorder.js +39 -0
- package/dist/record/redact.d.ts +37 -0
- package/dist/record/redact.js +119 -0
- package/dist/record/remote-recorder.d.ts +110 -0
- package/dist/record/remote-recorder.js +301 -0
- package/dist/record/truncate.d.ts +36 -0
- package/dist/record/truncate.js +85 -0
- package/dist/replay/bundle.d.ts +36 -0
- package/dist/replay/bundle.js +89 -0
- package/dist/replay/controller.d.ts +300 -0
- package/dist/replay/controller.js +807 -0
- package/dist/replay/coverage.d.ts +41 -0
- package/dist/replay/coverage.js +56 -0
- package/dist/replay/derive.d.ts +58 -0
- package/dist/replay/derive.js +166 -0
- package/dist/replay/executor.d.ts +78 -0
- package/dist/replay/executor.js +233 -0
- package/dist/replay/logic.d.ts +31 -0
- package/dist/replay/logic.js +50 -0
- package/dist/replay/plan.d.ts +181 -0
- package/dist/replay/plan.js +397 -0
- package/dist/replay/pricing.d.ts +41 -0
- package/dist/replay/pricing.js +76 -0
- package/dist/replay/source-run.d.ts +50 -0
- package/dist/replay/source-run.js +98 -0
- package/dist/replay/tool-error.d.ts +22 -0
- package/dist/replay/tool-error.js +60 -0
- package/dist/replay/types.d.ts +116 -0
- package/dist/replay/types.js +35 -0
- package/dist/upstream/client.d.ts +78 -0
- package/dist/upstream/client.js +114 -0
- package/dist/upstream/http-client.d.ts +78 -0
- package/dist/upstream/http-client.js +261 -0
- package/dist/upstream/lazy-client.d.ts +31 -0
- package/dist/upstream/lazy-client.js +53 -0
- package/dist/upstream/stdio-client.d.ts +57 -0
- package/dist/upstream/stdio-client.js +203 -0
- package/dist/util/log.d.ts +27 -0
- package/dist/util/log.js +51 -0
- package/dist/util/version.d.ts +2 -0
- package/dist/util/version.js +40 -0
- package/docs/BaseInstRunner.md +621 -0
- package/docs/calculatedReplay.md +1185 -0
- package/docs/calculatedReplayGuide.md +448 -0
- package/docs/installRun.md +413 -0
- package/docs/mcpmark.md +752 -0
- package/docs/quickstart.md +201 -0
- package/docs/t-bench.md +394 -0
- package/package.json +56 -0
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* proxy/session — tier negotiation and the proxy's recording lifecycle (§0.1).
|
|
3
|
+
*
|
|
4
|
+
* D5 says the hook binary owns run identity. D8 says support any MCP client.
|
|
5
|
+
* Only Claude Code has hooks, so both cannot hold universally, and ownership is
|
|
6
|
+
* negotiated here at startup:
|
|
7
|
+
*
|
|
8
|
+
* **Tier 1 — Bound.** A control server is discoverable for this cwd. It owns
|
|
9
|
+
* the run; this proxy only reports steps. Built-ins and MCP land in one
|
|
10
|
+
* ordered stream, with the prompt and the final answer.
|
|
11
|
+
*
|
|
12
|
+
* **Tier 2 — Standalone.** No control server inside the discovery window. The
|
|
13
|
+
* proxy owns a run of its own and records MCP calls only — no prompt, no final
|
|
14
|
+
* answer, no built-in steps. That is not a degraded bug; it is the honest
|
|
15
|
+
* ceiling of what an MCP proxy can observe, and the run says so in
|
|
16
|
+
* `metadata.tier` so nothing downstream mistakes it for a complete trace.
|
|
17
|
+
*
|
|
18
|
+
* THE DISCOVERY WINDOW EXISTS BECAUSE OF A RACE: the host may spawn MCP servers
|
|
19
|
+
* before `SessionStart` fires, and the ordering between the two is not
|
|
20
|
+
* guaranteed. So steps are **buffered in memory** while the search runs, and
|
|
21
|
+
* flushed into whichever owner wins. Nothing is lost to the race, and nothing
|
|
22
|
+
* blocks: relaying never waits on this.
|
|
23
|
+
*/
|
|
24
|
+
import { hostname } from "node:os";
|
|
25
|
+
import { ControlClient } from "../control/client.js";
|
|
26
|
+
import { resolveControl } from "../control/discovery.js";
|
|
27
|
+
import { qualifyToolName } from "../control/correlation.js";
|
|
28
|
+
import { authenticate } from "../auth/client.js";
|
|
29
|
+
import { NullRecorder } from "../record/recorder.js";
|
|
30
|
+
import { RemoteRecorder } from "../record/remote-recorder.js";
|
|
31
|
+
import { StepQueue } from "../record/queue.js";
|
|
32
|
+
import { StepIndexAllocator } from "../control/ordering.js";
|
|
33
|
+
import { serializeCapped } from "../record/truncate.js";
|
|
34
|
+
import { redact } from "../record/redact.js";
|
|
35
|
+
import { logLine, logDetail, errText } from "../util/log.js";
|
|
36
|
+
import { packageVersion } from "../util/version.js";
|
|
37
|
+
/** How long a proxy waits for a control server before falling to Tier 2. */
|
|
38
|
+
export const DISCOVERY_WINDOW_MS = 5_000;
|
|
39
|
+
export class ProxySession {
|
|
40
|
+
serverName;
|
|
41
|
+
opts;
|
|
42
|
+
tierValue = "pending";
|
|
43
|
+
control;
|
|
44
|
+
/** Tier 2 only. */
|
|
45
|
+
recorder;
|
|
46
|
+
runId;
|
|
47
|
+
ordering = new StepIndexAllocator();
|
|
48
|
+
queue = new StepQueue({
|
|
49
|
+
onDrop: (dropped) => {
|
|
50
|
+
if (!this.lossyValue)
|
|
51
|
+
logLine("run.lossy", { why: "step queue overflowed", dropped });
|
|
52
|
+
this.lossyValue = true;
|
|
53
|
+
},
|
|
54
|
+
onError: (err) => logLine("proxy.step_failed", { error: errText(err) }),
|
|
55
|
+
});
|
|
56
|
+
/** Steps that arrived before the tier was known. */
|
|
57
|
+
buffer = [];
|
|
58
|
+
lossyValue = false;
|
|
59
|
+
negotiation;
|
|
60
|
+
startedAt = Date.now();
|
|
61
|
+
/** The replay work loop, when one was started. */
|
|
62
|
+
workLoop;
|
|
63
|
+
closing = false;
|
|
64
|
+
constructor(opts) {
|
|
65
|
+
this.opts = opts;
|
|
66
|
+
this.serverName = opts.serverName;
|
|
67
|
+
}
|
|
68
|
+
get tier() {
|
|
69
|
+
return this.tierValue;
|
|
70
|
+
}
|
|
71
|
+
get lossy() {
|
|
72
|
+
return this.lossyValue;
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* Whether the schema-relaxation half of correlation should be applied to
|
|
76
|
+
* `tools/list` (Phase 6). Only ever true in Tier 1 with correlation enabled:
|
|
77
|
+
* relaxation is visible to the model on every call, so it is never paid for
|
|
78
|
+
* unless it is actually buying a merge.
|
|
79
|
+
*/
|
|
80
|
+
get correlationActive() {
|
|
81
|
+
return this.tierValue === "bound" && !this.opts.noCorrelation;
|
|
82
|
+
}
|
|
83
|
+
/** Begin tier negotiation. Returns immediately; relaying never waits on it. */
|
|
84
|
+
start() {
|
|
85
|
+
if (this.negotiation)
|
|
86
|
+
return;
|
|
87
|
+
this.negotiation = this.negotiate().catch((err) => {
|
|
88
|
+
logLine("proxy.negotiation_failed", { server: this.serverName, error: errText(err) });
|
|
89
|
+
this.tierValue = "standalone";
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
async negotiate() {
|
|
93
|
+
if (this.opts.standalone) {
|
|
94
|
+
await this.becomeStandalone("--standalone was requested");
|
|
95
|
+
return;
|
|
96
|
+
}
|
|
97
|
+
const window = this.opts.discoveryWindowMs ?? DISCOVERY_WINDOW_MS;
|
|
98
|
+
const found = await resolveControl(this.opts.cwd, window);
|
|
99
|
+
if (!found?.url) {
|
|
100
|
+
await this.becomeStandalone(`no control server within ${window}ms`);
|
|
101
|
+
return;
|
|
102
|
+
}
|
|
103
|
+
const client = new ControlClient(found.url, found.token);
|
|
104
|
+
const registered = await client.register({
|
|
105
|
+
serverName: this.serverName,
|
|
106
|
+
pid: process.pid,
|
|
107
|
+
cwd: this.opts.cwd,
|
|
108
|
+
version: packageVersion(),
|
|
109
|
+
});
|
|
110
|
+
if (!registered) {
|
|
111
|
+
await this.becomeStandalone("control server did not answer /proxy/register");
|
|
112
|
+
return;
|
|
113
|
+
}
|
|
114
|
+
this.control = client;
|
|
115
|
+
this.tierValue = "bound";
|
|
116
|
+
logLine("tier.bound", {
|
|
117
|
+
server: this.serverName,
|
|
118
|
+
control: found.url,
|
|
119
|
+
sess: registered.sessionId,
|
|
120
|
+
correlation: this.opts.noCorrelation ? "fingerprint" : "injected",
|
|
121
|
+
replay: registered.replay ? "on" : undefined,
|
|
122
|
+
});
|
|
123
|
+
this.flushBuffer();
|
|
124
|
+
if (registered.replay)
|
|
125
|
+
this.startWorkLoop(client, registered.pollHoldMs ?? 25_000);
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Take calculated-scenario steps from the control server and run them on the
|
|
129
|
+
* upstream this proxy already owns (docs/calculatedReplay.md §16.2).
|
|
130
|
+
*
|
|
131
|
+
* The whole point of the design lives in three lines below: `upstream.request`
|
|
132
|
+
* allocates an id in the reserved `bir:` space and the transport consumes its
|
|
133
|
+
* own response in `accept()` before the relay's handlers ever see it — so this
|
|
134
|
+
* costs **zero tokens**, the model is not involved, and `RecordingInterceptor`
|
|
135
|
+
* cannot double-record it. That interface comment says `request()` is "never
|
|
136
|
+
* used by the relay"; it was built for `bir doctor` and tests, and this is its
|
|
137
|
+
* second, load-bearing user.
|
|
138
|
+
*/
|
|
139
|
+
startWorkLoop(client, holdMs) {
|
|
140
|
+
const upstream = this.opts.upstream;
|
|
141
|
+
if (!upstream || this.workLoop)
|
|
142
|
+
return;
|
|
143
|
+
this.workLoop = (async () => {
|
|
144
|
+
logLine("replay.worker", { server: this.serverName, why: "polling for scenario steps" });
|
|
145
|
+
while (!this.closing) {
|
|
146
|
+
let work;
|
|
147
|
+
try {
|
|
148
|
+
work = await client.poll(this.serverName, holdMs);
|
|
149
|
+
}
|
|
150
|
+
catch {
|
|
151
|
+
work = undefined;
|
|
152
|
+
}
|
|
153
|
+
if (this.closing)
|
|
154
|
+
break;
|
|
155
|
+
if (!work) {
|
|
156
|
+
// Either an empty hold (normal) or an unreachable server. A short
|
|
157
|
+
// pause keeps the second case from becoming a busy loop.
|
|
158
|
+
await new Promise((resolve) => {
|
|
159
|
+
const t = setTimeout(resolve, 250);
|
|
160
|
+
t.unref?.();
|
|
161
|
+
});
|
|
162
|
+
continue;
|
|
163
|
+
}
|
|
164
|
+
try {
|
|
165
|
+
const result = await upstream.request("tools/call", { name: work.toolName, arguments: work.arguments }, AbortSignal.timeout(work.timeoutMs));
|
|
166
|
+
await client.result(work.workId, result);
|
|
167
|
+
}
|
|
168
|
+
catch (err) {
|
|
169
|
+
// A rejection here means the step could not be RUN — which is exactly
|
|
170
|
+
// the condition under which the control server may substitute the
|
|
171
|
+
// step's recorded output. A tool that ran and failed resolves with its
|
|
172
|
+
// failure as the result, and is not this branch.
|
|
173
|
+
await client.result(work.workId, undefined, errText(err));
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
})().catch((err) => logLine("replay.worker_failed", { server: this.serverName, error: errText(err) }));
|
|
177
|
+
}
|
|
178
|
+
/**
|
|
179
|
+
* Take ownership of a run ourselves.
|
|
180
|
+
*
|
|
181
|
+
* ORDER MATTERS HERE. The recorder is built (which may mean an auth round
|
|
182
|
+
* trip) *before* the tier is published, because `dispatch` routes on the tier:
|
|
183
|
+
* publishing "standalone" first opens a window in which a step is neither
|
|
184
|
+
* buffered nor recordable, and every `tools/call` in that window is lost.
|
|
185
|
+
* `dispatch` re-buffers defensively too — belt and braces on a race whose
|
|
186
|
+
* failure mode is silent data loss.
|
|
187
|
+
*/
|
|
188
|
+
async becomeStandalone(why) {
|
|
189
|
+
logLine("tier.standalone", {
|
|
190
|
+
server: this.serverName,
|
|
191
|
+
why,
|
|
192
|
+
records: "MCP calls only — no prompt, no final answer, no built-in steps",
|
|
193
|
+
});
|
|
194
|
+
this.recorder = await this.buildRecorder();
|
|
195
|
+
this.runId = this.recorder.startRun("", {
|
|
196
|
+
recorder: "baseinstrunner",
|
|
197
|
+
tier: "standalone",
|
|
198
|
+
host: { app: process.env.BIR_HOST_APP ?? "unknown", version: process.env.BIR_HOST_VERSION },
|
|
199
|
+
wrappedServers: [this.serverName],
|
|
200
|
+
lossy: this.lossyValue,
|
|
201
|
+
machine: hostname(),
|
|
202
|
+
cwd: this.opts.cwd,
|
|
203
|
+
});
|
|
204
|
+
this.tierValue = "standalone";
|
|
205
|
+
logLine("run.start", { run: this.runId, tier: "standalone", server: this.serverName });
|
|
206
|
+
this.flushBuffer();
|
|
207
|
+
}
|
|
208
|
+
async buildRecorder() {
|
|
209
|
+
if (this.opts.recorderFactory) {
|
|
210
|
+
return (await this.opts.recorderFactory()) ?? new NullRecorder();
|
|
211
|
+
}
|
|
212
|
+
const baseUrl = (process.env.BIR_AUTH_URL ?? "").replace(/\/+$/, "");
|
|
213
|
+
if (!baseUrl) {
|
|
214
|
+
logLine("recorder.disabled", { why: "BIR_AUTH_URL is not set" });
|
|
215
|
+
return new NullRecorder();
|
|
216
|
+
}
|
|
217
|
+
// Non-interactive: a proxy runs inside the host's process tree with its stdio
|
|
218
|
+
// bound to the JSON-RPC stream. There is nowhere to prompt, and blocking on
|
|
219
|
+
// one would hang the host's server startup.
|
|
220
|
+
const session = await authenticate({ authUrl: baseUrl, nonInteractive: true });
|
|
221
|
+
if (!session) {
|
|
222
|
+
logLine("recorder.disabled", { why: "no BaseIn session — run `bir login`" });
|
|
223
|
+
return new NullRecorder();
|
|
224
|
+
}
|
|
225
|
+
return new RemoteRecorder({ baseUrl, session });
|
|
226
|
+
}
|
|
227
|
+
/**
|
|
228
|
+
* Record one completed `tools/call`. Fire-and-forget by contract: it is called
|
|
229
|
+
* *after* the result is already on its way to the host, and it never awaits.
|
|
230
|
+
*/
|
|
231
|
+
reportStep(step) {
|
|
232
|
+
if (this.tierValue === "pending") {
|
|
233
|
+
this.bufferStep(step);
|
|
234
|
+
return;
|
|
235
|
+
}
|
|
236
|
+
this.dispatch(step);
|
|
237
|
+
}
|
|
238
|
+
/** Hold a step until an owner exists. Bounded, and honest when it overflows. */
|
|
239
|
+
bufferStep(step) {
|
|
240
|
+
if (this.buffer.length >= 1000) {
|
|
241
|
+
this.buffer.shift();
|
|
242
|
+
if (!this.lossyValue) {
|
|
243
|
+
logLine("run.lossy", { why: "discovery buffer overflowed", server: this.serverName });
|
|
244
|
+
}
|
|
245
|
+
this.lossyValue = true;
|
|
246
|
+
}
|
|
247
|
+
this.buffer.push(step);
|
|
248
|
+
}
|
|
249
|
+
flushBuffer() {
|
|
250
|
+
const buffered = this.buffer.splice(0, this.buffer.length);
|
|
251
|
+
for (const step of buffered)
|
|
252
|
+
this.dispatch(step);
|
|
253
|
+
if (buffered.length) {
|
|
254
|
+
logDetail("proxy.buffer_flushed", { server: this.serverName, steps: buffered.length });
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
dispatch(step) {
|
|
258
|
+
if (this.tierValue === "bound" && this.control) {
|
|
259
|
+
const control = this.control;
|
|
260
|
+
this.queue.push(async () => {
|
|
261
|
+
const ok = await control.report(step);
|
|
262
|
+
if (!ok) {
|
|
263
|
+
// §10: the control server died mid-session. Everything after this is
|
|
264
|
+
// Tier 2, and the run is flagged lossy — the steps already sent live in
|
|
265
|
+
// the bound run, the rest in ours.
|
|
266
|
+
logLine("tier.downgrade", {
|
|
267
|
+
server: this.serverName,
|
|
268
|
+
why: "control server stopped answering",
|
|
269
|
+
});
|
|
270
|
+
this.control = undefined;
|
|
271
|
+
this.lossyValue = true;
|
|
272
|
+
await this.becomeStandalone("control server stopped answering");
|
|
273
|
+
this.dispatch(step);
|
|
274
|
+
}
|
|
275
|
+
});
|
|
276
|
+
return;
|
|
277
|
+
}
|
|
278
|
+
const recorder = this.recorder;
|
|
279
|
+
const runId = this.runId;
|
|
280
|
+
if (!recorder || !runId) {
|
|
281
|
+
// No owner yet — mid-downgrade, or the recorder is still being built.
|
|
282
|
+
// Buffering costs a moment; dropping costs the step.
|
|
283
|
+
this.bufferStep(step);
|
|
284
|
+
return;
|
|
285
|
+
}
|
|
286
|
+
const pair = this.ordering.allocatePair();
|
|
287
|
+
const toolName = step.qualifiedName || qualifyToolName(step.serverName, step.toolName);
|
|
288
|
+
const toolInput = serializeCapped(redact(step.args));
|
|
289
|
+
const toolOutput = step.isError ? undefined : serializeCapped(redact(step.result));
|
|
290
|
+
const toolError = step.isError
|
|
291
|
+
? (step.errorMessage ?? serializeCapped(redact(step.result)))
|
|
292
|
+
: undefined;
|
|
293
|
+
this.queue.push(() => {
|
|
294
|
+
recorder.recordToolSelected(runId, pair.selected, { toolName, toolInput });
|
|
295
|
+
recorder.recordToolResponse(runId, pair.response, { toolName, toolOutput, toolError });
|
|
296
|
+
});
|
|
297
|
+
}
|
|
298
|
+
/** Close the Tier 2 run, if we own one, and drain the queue. */
|
|
299
|
+
async close() {
|
|
300
|
+
// Stops the work loop at its next turn. It is not awaited: it may be parked
|
|
301
|
+
// in a 25 s long-poll, and shutdown does not wait on a poll that will be
|
|
302
|
+
// answered by the control server closing anyway.
|
|
303
|
+
this.closing = true;
|
|
304
|
+
await this.negotiation?.catch(() => undefined);
|
|
305
|
+
this.flushBuffer();
|
|
306
|
+
await this.queue.flush();
|
|
307
|
+
if (this.recorder && this.runId) {
|
|
308
|
+
this.recorder.finishRun(this.runId, undefined, { durationMs: Date.now() - this.startedAt });
|
|
309
|
+
await this.recorder.flush(this.runId).catch(() => undefined);
|
|
310
|
+
logLine("run.finish", {
|
|
311
|
+
run: this.runId,
|
|
312
|
+
tier: "standalone",
|
|
313
|
+
steps: this.ordering.next,
|
|
314
|
+
lossy: this.lossyValue,
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
}
|
|
319
|
+
//# sourceMappingURL=session.js.map
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* housekeeping — host tools that are not work, and must not be treated as it.
|
|
3
|
+
*
|
|
4
|
+
* A host calls tools of its own that have nothing to do with the user's task.
|
|
5
|
+
* Claude Code opens a turn with `ToolSearch` to load deferred tool definitions,
|
|
6
|
+
* and writes a `TodoWrite` list while it plans. Neither reads or changes
|
|
7
|
+
* anything the user asked about.
|
|
8
|
+
*
|
|
9
|
+
* They must be excluded in TWO places, and missing either one is expensive:
|
|
10
|
+
*
|
|
11
|
+
* 1. **Recording.** A `ToolSearch` recorded as a step becomes a step of the
|
|
12
|
+
* calculated scenario — observed in production as a five-step scenario
|
|
13
|
+
* whose first step was `ToolSearch`. That is not merely untidy: the step's
|
|
14
|
+
* generated `reasoning` goes into the steering directive the model reads on
|
|
15
|
+
* every matched turn (~150 chars of it), it is not an MCP tool so it drags
|
|
16
|
+
* the whole scenario from `direct` mode into `steer`, and the analyser
|
|
17
|
+
* spends a Claude call writing input/output logic for a tool that has no
|
|
18
|
+
* output worth threading.
|
|
19
|
+
*
|
|
20
|
+
* 2. **Replay** (`ReplayController.preTool`). A divergence retires the plan,
|
|
21
|
+
* so treating the host's first housekeeping call as "the model went
|
|
22
|
+
* off-script" kills the replay before the model has had a chance to follow
|
|
23
|
+
* it — observed as `replay.diverge ... called=ToolSearch step=0/4`, after
|
|
24
|
+
* which the turn cost full price.
|
|
25
|
+
*
|
|
26
|
+
* DELIBERATELY TINY. Every entry must be pure host bookkeeping that cannot
|
|
27
|
+
* possibly be the user's task. `Read`, `Bash` and `Grep` are NOT here: they do
|
|
28
|
+
* real work, they belong in a recording, and reaching for one mid-replay
|
|
29
|
+
* usually *is* the model doing the task another way.
|
|
30
|
+
*/
|
|
31
|
+
/** Host tools that are never recorded as steps and never count as divergence. */
|
|
32
|
+
export declare const HOUSEKEEPING_TOOLS: ReadonlySet<string>;
|
|
33
|
+
export declare function isHousekeeping(toolName: string): boolean;
|
|
34
|
+
//# sourceMappingURL=housekeeping.d.ts.map
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* housekeeping — host tools that are not work, and must not be treated as it.
|
|
3
|
+
*
|
|
4
|
+
* A host calls tools of its own that have nothing to do with the user's task.
|
|
5
|
+
* Claude Code opens a turn with `ToolSearch` to load deferred tool definitions,
|
|
6
|
+
* and writes a `TodoWrite` list while it plans. Neither reads or changes
|
|
7
|
+
* anything the user asked about.
|
|
8
|
+
*
|
|
9
|
+
* They must be excluded in TWO places, and missing either one is expensive:
|
|
10
|
+
*
|
|
11
|
+
* 1. **Recording.** A `ToolSearch` recorded as a step becomes a step of the
|
|
12
|
+
* calculated scenario — observed in production as a five-step scenario
|
|
13
|
+
* whose first step was `ToolSearch`. That is not merely untidy: the step's
|
|
14
|
+
* generated `reasoning` goes into the steering directive the model reads on
|
|
15
|
+
* every matched turn (~150 chars of it), it is not an MCP tool so it drags
|
|
16
|
+
* the whole scenario from `direct` mode into `steer`, and the analyser
|
|
17
|
+
* spends a Claude call writing input/output logic for a tool that has no
|
|
18
|
+
* output worth threading.
|
|
19
|
+
*
|
|
20
|
+
* 2. **Replay** (`ReplayController.preTool`). A divergence retires the plan,
|
|
21
|
+
* so treating the host's first housekeeping call as "the model went
|
|
22
|
+
* off-script" kills the replay before the model has had a chance to follow
|
|
23
|
+
* it — observed as `replay.diverge ... called=ToolSearch step=0/4`, after
|
|
24
|
+
* which the turn cost full price.
|
|
25
|
+
*
|
|
26
|
+
* DELIBERATELY TINY. Every entry must be pure host bookkeeping that cannot
|
|
27
|
+
* possibly be the user's task. `Read`, `Bash` and `Grep` are NOT here: they do
|
|
28
|
+
* real work, they belong in a recording, and reaching for one mid-replay
|
|
29
|
+
* usually *is* the model doing the task another way.
|
|
30
|
+
*/
|
|
31
|
+
/** Host tools that are never recorded as steps and never count as divergence. */
|
|
32
|
+
export const HOUSEKEEPING_TOOLS = new Set([
|
|
33
|
+
"ToolSearch",
|
|
34
|
+
"TodoWrite",
|
|
35
|
+
]);
|
|
36
|
+
export function isHousekeeping(toolName) {
|
|
37
|
+
return HOUSEKEEPING_TOOLS.has(toolName);
|
|
38
|
+
}
|
|
39
|
+
//# sourceMappingURL=housekeeping.js.map
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* queue — the bounded, fire-and-forget step queue (Phase 8, backpressure).
|
|
3
|
+
*
|
|
4
|
+
* Two rules from §3 and §10 meet here:
|
|
5
|
+
*
|
|
6
|
+
* - **Never let recording block a tool call.** Work is enqueued after the
|
|
7
|
+
* result is already on its way to the host, and `push` is synchronous.
|
|
8
|
+
* - **Never grow without bound.** A recorder that is slow or down must not turn
|
|
9
|
+
* into a memory leak in the host's own process tree. Past `capacity` the
|
|
10
|
+
* *oldest* item is dropped and the run is flagged `lossy` — losing the start
|
|
11
|
+
* of a long run is strictly better than losing its end, which is where the
|
|
12
|
+
* final answer lives.
|
|
13
|
+
*
|
|
14
|
+
* A dropped item is reported through {@link onDrop} so the flag reaches the run's
|
|
15
|
+
* metadata rather than dying in a log line.
|
|
16
|
+
*/
|
|
17
|
+
export interface StepQueueOptions {
|
|
18
|
+
/** Max queued items before the oldest is dropped. Default 1000. */
|
|
19
|
+
capacity?: number;
|
|
20
|
+
/** Called once per dropped item, with the running drop total. */
|
|
21
|
+
onDrop?: (dropped: number) => void;
|
|
22
|
+
/** Called when a task throws. Sends are best-effort; failure is logged, not retried. */
|
|
23
|
+
onError?: (err: unknown) => void;
|
|
24
|
+
}
|
|
25
|
+
export declare const DEFAULT_CAPACITY = 1000;
|
|
26
|
+
/** A FIFO of async tasks, drained one at a time, with a hard size bound. */
|
|
27
|
+
export declare class StepQueue {
|
|
28
|
+
private readonly items;
|
|
29
|
+
private readonly capacity;
|
|
30
|
+
private readonly onDrop?;
|
|
31
|
+
private readonly onError?;
|
|
32
|
+
/** Resolved (and cleared) every time the queue goes idle. */
|
|
33
|
+
private readonly waiters;
|
|
34
|
+
private draining;
|
|
35
|
+
private droppedCount;
|
|
36
|
+
constructor(opts?: StepQueueOptions);
|
|
37
|
+
/** True once anything has been dropped — the run's `lossy` flag. */
|
|
38
|
+
get lossy(): boolean;
|
|
39
|
+
get dropped(): number;
|
|
40
|
+
get size(): number;
|
|
41
|
+
get busy(): boolean;
|
|
42
|
+
/** Enqueue work. Synchronous and never throws. */
|
|
43
|
+
push(task: () => Promise<void> | void): void;
|
|
44
|
+
private drain;
|
|
45
|
+
/** Await the queue draining. Resolves immediately when already idle. */
|
|
46
|
+
flush(): Promise<void>;
|
|
47
|
+
}
|
|
48
|
+
//# sourceMappingURL=queue.d.ts.map
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* queue — the bounded, fire-and-forget step queue (Phase 8, backpressure).
|
|
3
|
+
*
|
|
4
|
+
* Two rules from §3 and §10 meet here:
|
|
5
|
+
*
|
|
6
|
+
* - **Never let recording block a tool call.** Work is enqueued after the
|
|
7
|
+
* result is already on its way to the host, and `push` is synchronous.
|
|
8
|
+
* - **Never grow without bound.** A recorder that is slow or down must not turn
|
|
9
|
+
* into a memory leak in the host's own process tree. Past `capacity` the
|
|
10
|
+
* *oldest* item is dropped and the run is flagged `lossy` — losing the start
|
|
11
|
+
* of a long run is strictly better than losing its end, which is where the
|
|
12
|
+
* final answer lives.
|
|
13
|
+
*
|
|
14
|
+
* A dropped item is reported through {@link onDrop} so the flag reaches the run's
|
|
15
|
+
* metadata rather than dying in a log line.
|
|
16
|
+
*/
|
|
17
|
+
export const DEFAULT_CAPACITY = 1000;
|
|
18
|
+
/** A FIFO of async tasks, drained one at a time, with a hard size bound. */
|
|
19
|
+
export class StepQueue {
|
|
20
|
+
items = [];
|
|
21
|
+
capacity;
|
|
22
|
+
onDrop;
|
|
23
|
+
onError;
|
|
24
|
+
/** Resolved (and cleared) every time the queue goes idle. */
|
|
25
|
+
waiters = [];
|
|
26
|
+
draining = false;
|
|
27
|
+
droppedCount = 0;
|
|
28
|
+
constructor(opts = {}) {
|
|
29
|
+
this.capacity = opts.capacity ?? DEFAULT_CAPACITY;
|
|
30
|
+
this.onDrop = opts.onDrop;
|
|
31
|
+
this.onError = opts.onError;
|
|
32
|
+
}
|
|
33
|
+
/** True once anything has been dropped — the run's `lossy` flag. */
|
|
34
|
+
get lossy() {
|
|
35
|
+
return this.droppedCount > 0;
|
|
36
|
+
}
|
|
37
|
+
get dropped() {
|
|
38
|
+
return this.droppedCount;
|
|
39
|
+
}
|
|
40
|
+
get size() {
|
|
41
|
+
return this.items.length;
|
|
42
|
+
}
|
|
43
|
+
get busy() {
|
|
44
|
+
return this.draining || this.items.length > 0;
|
|
45
|
+
}
|
|
46
|
+
/** Enqueue work. Synchronous and never throws. */
|
|
47
|
+
push(task) {
|
|
48
|
+
if (this.items.length >= this.capacity) {
|
|
49
|
+
this.items.shift();
|
|
50
|
+
this.droppedCount += 1;
|
|
51
|
+
this.onDrop?.(this.droppedCount);
|
|
52
|
+
}
|
|
53
|
+
this.items.push(task);
|
|
54
|
+
void this.drain();
|
|
55
|
+
}
|
|
56
|
+
async drain() {
|
|
57
|
+
if (this.draining)
|
|
58
|
+
return;
|
|
59
|
+
this.draining = true;
|
|
60
|
+
try {
|
|
61
|
+
for (;;) {
|
|
62
|
+
const task = this.items.shift();
|
|
63
|
+
if (!task)
|
|
64
|
+
break;
|
|
65
|
+
try {
|
|
66
|
+
await task();
|
|
67
|
+
}
|
|
68
|
+
catch (err) {
|
|
69
|
+
this.onError?.(err);
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
finally {
|
|
74
|
+
this.draining = false;
|
|
75
|
+
// A task enqueued during the last await is still in `items`; restart
|
|
76
|
+
// rather than waking waiters on a queue that is not actually empty.
|
|
77
|
+
if (this.items.length > 0) {
|
|
78
|
+
void this.drain();
|
|
79
|
+
}
|
|
80
|
+
else {
|
|
81
|
+
const waiting = this.waiters.splice(0, this.waiters.length);
|
|
82
|
+
for (const resolve of waiting)
|
|
83
|
+
resolve();
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
/** Await the queue draining. Resolves immediately when already idle. */
|
|
88
|
+
flush() {
|
|
89
|
+
if (!this.busy)
|
|
90
|
+
return Promise.resolve();
|
|
91
|
+
return new Promise((resolve) => {
|
|
92
|
+
this.waiters.push(resolve);
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
//# sourceMappingURL=queue.js.map
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recorder — the surface both tiers use (§5).
|
|
3
|
+
*
|
|
4
|
+
* Deliberately identical in shape to RRepeat's `RemoteRecorder`, so the BaseIn
|
|
5
|
+
* service needs no new endpoints for v1: a BaseInstRunner run is the same row
|
|
6
|
+
* shape as an RRepeat one, discriminated only by `metadata.recorder`.
|
|
7
|
+
*/
|
|
8
|
+
import type { ExecutionStage, ExecutionStepResult } from "../replay/types.js";
|
|
9
|
+
export interface RunMetrics {
|
|
10
|
+
costUsd?: number;
|
|
11
|
+
durationMs?: number;
|
|
12
|
+
}
|
|
13
|
+
export interface Recorder {
|
|
14
|
+
/** Create the run; returns the client-generated runId synchronously. */
|
|
15
|
+
startRun(input: string, metadata?: Record<string, unknown>): string;
|
|
16
|
+
recordToolSelected(runId: string, stepIndex: number, d: {
|
|
17
|
+
toolName: string;
|
|
18
|
+
toolInput: string;
|
|
19
|
+
context?: string;
|
|
20
|
+
}): string;
|
|
21
|
+
recordToolResponse(runId: string, stepIndex: number, d: {
|
|
22
|
+
toolName: string;
|
|
23
|
+
toolOutput?: string;
|
|
24
|
+
toolError?: string;
|
|
25
|
+
}): string;
|
|
26
|
+
recordFinalAnswer(runId: string, stepIndex: number, d: {
|
|
27
|
+
answer: string;
|
|
28
|
+
}): string;
|
|
29
|
+
finishRun(runId: string, finalOutput?: string, metrics?: RunMetrics): void;
|
|
30
|
+
/** Await every queued send for a run. Called once at session end. */
|
|
31
|
+
flush(runId: string): Promise<void>;
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* A similar-prompt hit reported by `POST /recordings/runs`. On a hit the server
|
|
35
|
+
* does **not** create a run, so every later step post for that id would 404 —
|
|
36
|
+
* callers must stop recording it. What the caller does *instead* of recording is
|
|
37
|
+
* v2's subject (docs/calculatedReplay.md); stopping is mandatory either way.
|
|
38
|
+
*/
|
|
39
|
+
export interface RunMatch {
|
|
40
|
+
runId: string;
|
|
41
|
+
scenarioId: string | null;
|
|
42
|
+
similarity: number;
|
|
43
|
+
scenario: Record<string, unknown> | null;
|
|
44
|
+
executionTicket?: string;
|
|
45
|
+
}
|
|
46
|
+
/** Optional capability: `null` means "fresh run, keep recording". */
|
|
47
|
+
export interface MatchAware {
|
|
48
|
+
getMatch(runId: string): Promise<RunMatch | null>;
|
|
49
|
+
}
|
|
50
|
+
export declare function isMatchAware(r: Recorder): r is Recorder & MatchAware;
|
|
51
|
+
/**
|
|
52
|
+
* What a matched turn actually cost, reported once per match
|
|
53
|
+
* (docs/calculatedReplay.md §11).
|
|
54
|
+
*
|
|
55
|
+
* `not_steered` and `failed` mean the agent ran the task the ordinary way; the
|
|
56
|
+
* service files those as **baseline samples** rather than savings, which is what
|
|
57
|
+
* keeps the ledger's denominator honest. `steered_full` and `diverged` book a
|
|
58
|
+
* saving against that baseline.
|
|
59
|
+
*/
|
|
60
|
+
export interface ExecutionReport {
|
|
61
|
+
scenarioId: string;
|
|
62
|
+
/** The match's claim token. It *is* the execution row's id, so a doubled report books once. */
|
|
63
|
+
ticket?: string;
|
|
64
|
+
outcome: "steered_full" | "diverged" | "not_steered" | "failed";
|
|
65
|
+
/** The derivation call — the only tokens replay itself spends. */
|
|
66
|
+
deriveCostUsd: number;
|
|
67
|
+
/** The live turn, from the transcript usage delta. */
|
|
68
|
+
sessionCostUsd: number;
|
|
69
|
+
/** Divergence recovery. Zero here by construction: the proxies run for free. */
|
|
70
|
+
fallbackCostUsd: number;
|
|
71
|
+
durationMs: number;
|
|
72
|
+
stepsPlanned: number;
|
|
73
|
+
stepsPinned: number;
|
|
74
|
+
/** False when the usage watermark was missing — "unmeasured" beats a wrong baseline. */
|
|
75
|
+
measured: boolean;
|
|
76
|
+
pricingVersion: string;
|
|
77
|
+
prompt?: string;
|
|
78
|
+
/**
|
|
79
|
+
* One verdict per step this run reached (errorshandling.md). Absent when the
|
|
80
|
+
* run reached none — a decline, or a turn that died before the first step.
|
|
81
|
+
*
|
|
82
|
+
* This is the only channel through which "step 2 of your scenario throws"
|
|
83
|
+
* reaches the person who owns the scenario: the log line is on their machine,
|
|
84
|
+
* the recording page is not.
|
|
85
|
+
*/
|
|
86
|
+
steps?: ExecutionStepResult[];
|
|
87
|
+
/** The headline failure, lifted from the first step that broke. */
|
|
88
|
+
error?: string;
|
|
89
|
+
errorStage?: ExecutionStage;
|
|
90
|
+
errorStepIndex?: number;
|
|
91
|
+
errorToolName?: string;
|
|
92
|
+
}
|
|
93
|
+
/** Optional capability: reporting needs a service, and a NullRecorder has none. */
|
|
94
|
+
export interface ScenarioReporter {
|
|
95
|
+
reportExecution(report: ExecutionReport): void;
|
|
96
|
+
}
|
|
97
|
+
export declare function isScenarioReporter(r: Recorder): r is Recorder & ScenarioReporter;
|
|
98
|
+
/**
|
|
99
|
+
* A recorder that drops everything. Used when auth is unavailable: recording is
|
|
100
|
+
* best-effort, and a session must never fail because BaseIn is unreachable (§10).
|
|
101
|
+
*/
|
|
102
|
+
export declare class NullRecorder implements Recorder {
|
|
103
|
+
private n;
|
|
104
|
+
startRun(): string;
|
|
105
|
+
recordToolSelected(): string;
|
|
106
|
+
recordToolResponse(): string;
|
|
107
|
+
recordFinalAnswer(): string;
|
|
108
|
+
finishRun(): void;
|
|
109
|
+
flush(): Promise<void>;
|
|
110
|
+
}
|
|
111
|
+
//# sourceMappingURL=recorder.d.ts.map
|