@basein/runner 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +276 -0
  3. package/dist/auth/client.d.ts +85 -0
  4. package/dist/auth/client.js +284 -0
  5. package/dist/bin/bir-hooks.d.ts +48 -0
  6. package/dist/bin/bir-hooks.js +201 -0
  7. package/dist/bin/bir-proxy.d.ts +45 -0
  8. package/dist/bin/bir-proxy.js +207 -0
  9. package/dist/bin/bir-scenario.d.ts +24 -0
  10. package/dist/bin/bir-scenario.js +177 -0
  11. package/dist/bin/bir.d.ts +21 -0
  12. package/dist/bin/bir.js +876 -0
  13. package/dist/config/adapters/claude-code.d.ts +76 -0
  14. package/dist/config/adapters/claude-code.js +181 -0
  15. package/dist/config/adapters/generic.d.ts +17 -0
  16. package/dist/config/adapters/generic.js +36 -0
  17. package/dist/config/generate.d.ts +127 -0
  18. package/dist/config/generate.js +114 -0
  19. package/dist/config/resolve.d.ts +68 -0
  20. package/dist/config/resolve.js +132 -0
  21. package/dist/control/client.d.ts +56 -0
  22. package/dist/control/client.js +86 -0
  23. package/dist/control/correlation.d.ts +86 -0
  24. package/dist/control/correlation.js +0 -0
  25. package/dist/control/discovery.d.ts +50 -0
  26. package/dist/control/discovery.js +123 -0
  27. package/dist/control/ordering.d.ts +38 -0
  28. package/dist/control/ordering.js +44 -0
  29. package/dist/control/paths.d.ts +32 -0
  30. package/dist/control/paths.js +56 -0
  31. package/dist/control/server.d.ts +272 -0
  32. package/dist/control/server.js +1131 -0
  33. package/dist/control/transcript.d.ts +75 -0
  34. package/dist/control/transcript.js +241 -0
  35. package/dist/index.d.ts +37 -0
  36. package/dist/index.js +32 -0
  37. package/dist/jsonrpc/framing.d.ts +49 -0
  38. package/dist/jsonrpc/framing.js +143 -0
  39. package/dist/jsonrpc/types.d.ts +52 -0
  40. package/dist/jsonrpc/types.js +46 -0
  41. package/dist/proxy/intercept.d.ts +55 -0
  42. package/dist/proxy/intercept.js +147 -0
  43. package/dist/proxy/relay.d.ts +97 -0
  44. package/dist/proxy/relay.js +166 -0
  45. package/dist/proxy/session.d.ts +116 -0
  46. package/dist/proxy/session.js +319 -0
  47. package/dist/record/housekeeping.d.ts +34 -0
  48. package/dist/record/housekeeping.js +39 -0
  49. package/dist/record/queue.d.ts +48 -0
  50. package/dist/record/queue.js +96 -0
  51. package/dist/record/recorder.d.ts +111 -0
  52. package/dist/record/recorder.js +39 -0
  53. package/dist/record/redact.d.ts +37 -0
  54. package/dist/record/redact.js +119 -0
  55. package/dist/record/remote-recorder.d.ts +110 -0
  56. package/dist/record/remote-recorder.js +301 -0
  57. package/dist/record/truncate.d.ts +36 -0
  58. package/dist/record/truncate.js +85 -0
  59. package/dist/replay/bundle.d.ts +36 -0
  60. package/dist/replay/bundle.js +89 -0
  61. package/dist/replay/controller.d.ts +300 -0
  62. package/dist/replay/controller.js +807 -0
  63. package/dist/replay/coverage.d.ts +41 -0
  64. package/dist/replay/coverage.js +56 -0
  65. package/dist/replay/derive.d.ts +58 -0
  66. package/dist/replay/derive.js +166 -0
  67. package/dist/replay/executor.d.ts +78 -0
  68. package/dist/replay/executor.js +233 -0
  69. package/dist/replay/logic.d.ts +31 -0
  70. package/dist/replay/logic.js +50 -0
  71. package/dist/replay/plan.d.ts +181 -0
  72. package/dist/replay/plan.js +397 -0
  73. package/dist/replay/pricing.d.ts +41 -0
  74. package/dist/replay/pricing.js +76 -0
  75. package/dist/replay/source-run.d.ts +50 -0
  76. package/dist/replay/source-run.js +98 -0
  77. package/dist/replay/tool-error.d.ts +22 -0
  78. package/dist/replay/tool-error.js +60 -0
  79. package/dist/replay/types.d.ts +116 -0
  80. package/dist/replay/types.js +35 -0
  81. package/dist/upstream/client.d.ts +78 -0
  82. package/dist/upstream/client.js +114 -0
  83. package/dist/upstream/http-client.d.ts +78 -0
  84. package/dist/upstream/http-client.js +261 -0
  85. package/dist/upstream/lazy-client.d.ts +31 -0
  86. package/dist/upstream/lazy-client.js +53 -0
  87. package/dist/upstream/stdio-client.d.ts +57 -0
  88. package/dist/upstream/stdio-client.js +203 -0
  89. package/dist/util/log.d.ts +27 -0
  90. package/dist/util/log.js +51 -0
  91. package/dist/util/version.d.ts +2 -0
  92. package/dist/util/version.js +40 -0
  93. package/docs/BaseInstRunner.md +621 -0
  94. package/docs/calculatedReplay.md +1185 -0
  95. package/docs/calculatedReplayGuide.md +448 -0
  96. package/docs/installRun.md +413 -0
  97. package/docs/mcpmark.md +752 -0
  98. package/docs/quickstart.md +201 -0
  99. package/docs/t-bench.md +394 -0
  100. package/package.json +56 -0
@@ -0,0 +1,319 @@
1
+ /**
2
+ * proxy/session — tier negotiation and the proxy's recording lifecycle (§0.1).
3
+ *
4
+ * D5 says the hook binary owns run identity. D8 says support any MCP client.
5
+ * Only Claude Code has hooks, so both cannot hold universally, and ownership is
6
+ * negotiated here at startup:
7
+ *
8
+ * **Tier 1 — Bound.** A control server is discoverable for this cwd. It owns
9
+ * the run; this proxy only reports steps. Built-ins and MCP land in one
10
+ * ordered stream, with the prompt and the final answer.
11
+ *
12
+ * **Tier 2 — Standalone.** No control server inside the discovery window. The
13
+ * proxy owns a run of its own and records MCP calls only — no prompt, no final
14
+ * answer, no built-in steps. That is not a degraded bug; it is the honest
15
+ * ceiling of what an MCP proxy can observe, and the run says so in
16
+ * `metadata.tier` so nothing downstream mistakes it for a complete trace.
17
+ *
18
+ * THE DISCOVERY WINDOW EXISTS BECAUSE OF A RACE: the host may spawn MCP servers
19
+ * before `SessionStart` fires, and the ordering between the two is not
20
+ * guaranteed. So steps are **buffered in memory** while the search runs, and
21
+ * flushed into whichever owner wins. Nothing is lost to the race, and nothing
22
+ * blocks: relaying never waits on this.
23
+ */
24
+ import { hostname } from "node:os";
25
+ import { ControlClient } from "../control/client.js";
26
+ import { resolveControl } from "../control/discovery.js";
27
+ import { qualifyToolName } from "../control/correlation.js";
28
+ import { authenticate } from "../auth/client.js";
29
+ import { NullRecorder } from "../record/recorder.js";
30
+ import { RemoteRecorder } from "../record/remote-recorder.js";
31
+ import { StepQueue } from "../record/queue.js";
32
+ import { StepIndexAllocator } from "../control/ordering.js";
33
+ import { serializeCapped } from "../record/truncate.js";
34
+ import { redact } from "../record/redact.js";
35
+ import { logLine, logDetail, errText } from "../util/log.js";
36
+ import { packageVersion } from "../util/version.js";
37
+ /** How long a proxy waits for a control server before falling to Tier 2. */
38
+ export const DISCOVERY_WINDOW_MS = 5_000;
39
+ export class ProxySession {
40
+ serverName;
41
+ opts;
42
+ tierValue = "pending";
43
+ control;
44
+ /** Tier 2 only. */
45
+ recorder;
46
+ runId;
47
+ ordering = new StepIndexAllocator();
48
+ queue = new StepQueue({
49
+ onDrop: (dropped) => {
50
+ if (!this.lossyValue)
51
+ logLine("run.lossy", { why: "step queue overflowed", dropped });
52
+ this.lossyValue = true;
53
+ },
54
+ onError: (err) => logLine("proxy.step_failed", { error: errText(err) }),
55
+ });
56
+ /** Steps that arrived before the tier was known. */
57
+ buffer = [];
58
+ lossyValue = false;
59
+ negotiation;
60
+ startedAt = Date.now();
61
+ /** The replay work loop, when one was started. */
62
+ workLoop;
63
+ closing = false;
64
+ constructor(opts) {
65
+ this.opts = opts;
66
+ this.serverName = opts.serverName;
67
+ }
68
+ get tier() {
69
+ return this.tierValue;
70
+ }
71
+ get lossy() {
72
+ return this.lossyValue;
73
+ }
74
+ /**
75
+ * Whether the schema-relaxation half of correlation should be applied to
76
+ * `tools/list` (Phase 6). Only ever true in Tier 1 with correlation enabled:
77
+ * relaxation is visible to the model on every call, so it is never paid for
78
+ * unless it is actually buying a merge.
79
+ */
80
+ get correlationActive() {
81
+ return this.tierValue === "bound" && !this.opts.noCorrelation;
82
+ }
83
+ /** Begin tier negotiation. Returns immediately; relaying never waits on it. */
84
+ start() {
85
+ if (this.negotiation)
86
+ return;
87
+ this.negotiation = this.negotiate().catch((err) => {
88
+ logLine("proxy.negotiation_failed", { server: this.serverName, error: errText(err) });
89
+ this.tierValue = "standalone";
90
+ });
91
+ }
92
+ async negotiate() {
93
+ if (this.opts.standalone) {
94
+ await this.becomeStandalone("--standalone was requested");
95
+ return;
96
+ }
97
+ const window = this.opts.discoveryWindowMs ?? DISCOVERY_WINDOW_MS;
98
+ const found = await resolveControl(this.opts.cwd, window);
99
+ if (!found?.url) {
100
+ await this.becomeStandalone(`no control server within ${window}ms`);
101
+ return;
102
+ }
103
+ const client = new ControlClient(found.url, found.token);
104
+ const registered = await client.register({
105
+ serverName: this.serverName,
106
+ pid: process.pid,
107
+ cwd: this.opts.cwd,
108
+ version: packageVersion(),
109
+ });
110
+ if (!registered) {
111
+ await this.becomeStandalone("control server did not answer /proxy/register");
112
+ return;
113
+ }
114
+ this.control = client;
115
+ this.tierValue = "bound";
116
+ logLine("tier.bound", {
117
+ server: this.serverName,
118
+ control: found.url,
119
+ sess: registered.sessionId,
120
+ correlation: this.opts.noCorrelation ? "fingerprint" : "injected",
121
+ replay: registered.replay ? "on" : undefined,
122
+ });
123
+ this.flushBuffer();
124
+ if (registered.replay)
125
+ this.startWorkLoop(client, registered.pollHoldMs ?? 25_000);
126
+ }
127
+ /**
128
+ * Take calculated-scenario steps from the control server and run them on the
129
+ * upstream this proxy already owns (docs/calculatedReplay.md §16.2).
130
+ *
131
+ * The whole point of the design lives in three lines below: `upstream.request`
132
+ * allocates an id in the reserved `bir:` space and the transport consumes its
133
+ * own response in `accept()` before the relay's handlers ever see it — so this
134
+ * costs **zero tokens**, the model is not involved, and `RecordingInterceptor`
135
+ * cannot double-record it. That interface comment says `request()` is "never
136
+ * used by the relay"; it was built for `bir doctor` and tests, and this is its
137
+ * second, load-bearing user.
138
+ */
139
+ startWorkLoop(client, holdMs) {
140
+ const upstream = this.opts.upstream;
141
+ if (!upstream || this.workLoop)
142
+ return;
143
+ this.workLoop = (async () => {
144
+ logLine("replay.worker", { server: this.serverName, why: "polling for scenario steps" });
145
+ while (!this.closing) {
146
+ let work;
147
+ try {
148
+ work = await client.poll(this.serverName, holdMs);
149
+ }
150
+ catch {
151
+ work = undefined;
152
+ }
153
+ if (this.closing)
154
+ break;
155
+ if (!work) {
156
+ // Either an empty hold (normal) or an unreachable server. A short
157
+ // pause keeps the second case from becoming a busy loop.
158
+ await new Promise((resolve) => {
159
+ const t = setTimeout(resolve, 250);
160
+ t.unref?.();
161
+ });
162
+ continue;
163
+ }
164
+ try {
165
+ const result = await upstream.request("tools/call", { name: work.toolName, arguments: work.arguments }, AbortSignal.timeout(work.timeoutMs));
166
+ await client.result(work.workId, result);
167
+ }
168
+ catch (err) {
169
+ // A rejection here means the step could not be RUN — which is exactly
170
+ // the condition under which the control server may substitute the
171
+ // step's recorded output. A tool that ran and failed resolves with its
172
+ // failure as the result, and is not this branch.
173
+ await client.result(work.workId, undefined, errText(err));
174
+ }
175
+ }
176
+ })().catch((err) => logLine("replay.worker_failed", { server: this.serverName, error: errText(err) }));
177
+ }
178
+ /**
179
+ * Take ownership of a run ourselves.
180
+ *
181
+ * ORDER MATTERS HERE. The recorder is built (which may mean an auth round
182
+ * trip) *before* the tier is published, because `dispatch` routes on the tier:
183
+ * publishing "standalone" first opens a window in which a step is neither
184
+ * buffered nor recordable, and every `tools/call` in that window is lost.
185
+ * `dispatch` re-buffers defensively too — belt and braces on a race whose
186
+ * failure mode is silent data loss.
187
+ */
188
+ async becomeStandalone(why) {
189
+ logLine("tier.standalone", {
190
+ server: this.serverName,
191
+ why,
192
+ records: "MCP calls only — no prompt, no final answer, no built-in steps",
193
+ });
194
+ this.recorder = await this.buildRecorder();
195
+ this.runId = this.recorder.startRun("", {
196
+ recorder: "baseinstrunner",
197
+ tier: "standalone",
198
+ host: { app: process.env.BIR_HOST_APP ?? "unknown", version: process.env.BIR_HOST_VERSION },
199
+ wrappedServers: [this.serverName],
200
+ lossy: this.lossyValue,
201
+ machine: hostname(),
202
+ cwd: this.opts.cwd,
203
+ });
204
+ this.tierValue = "standalone";
205
+ logLine("run.start", { run: this.runId, tier: "standalone", server: this.serverName });
206
+ this.flushBuffer();
207
+ }
208
+ async buildRecorder() {
209
+ if (this.opts.recorderFactory) {
210
+ return (await this.opts.recorderFactory()) ?? new NullRecorder();
211
+ }
212
+ const baseUrl = (process.env.BIR_AUTH_URL ?? "").replace(/\/+$/, "");
213
+ if (!baseUrl) {
214
+ logLine("recorder.disabled", { why: "BIR_AUTH_URL is not set" });
215
+ return new NullRecorder();
216
+ }
217
+ // Non-interactive: a proxy runs inside the host's process tree with its stdio
218
+ // bound to the JSON-RPC stream. There is nowhere to prompt, and blocking on
219
+ // one would hang the host's server startup.
220
+ const session = await authenticate({ authUrl: baseUrl, nonInteractive: true });
221
+ if (!session) {
222
+ logLine("recorder.disabled", { why: "no BaseIn session — run `bir login`" });
223
+ return new NullRecorder();
224
+ }
225
+ return new RemoteRecorder({ baseUrl, session });
226
+ }
227
+ /**
228
+ * Record one completed `tools/call`. Fire-and-forget by contract: it is called
229
+ * *after* the result is already on its way to the host, and it never awaits.
230
+ */
231
+ reportStep(step) {
232
+ if (this.tierValue === "pending") {
233
+ this.bufferStep(step);
234
+ return;
235
+ }
236
+ this.dispatch(step);
237
+ }
238
+ /** Hold a step until an owner exists. Bounded, and honest when it overflows. */
239
+ bufferStep(step) {
240
+ if (this.buffer.length >= 1000) {
241
+ this.buffer.shift();
242
+ if (!this.lossyValue) {
243
+ logLine("run.lossy", { why: "discovery buffer overflowed", server: this.serverName });
244
+ }
245
+ this.lossyValue = true;
246
+ }
247
+ this.buffer.push(step);
248
+ }
249
+ flushBuffer() {
250
+ const buffered = this.buffer.splice(0, this.buffer.length);
251
+ for (const step of buffered)
252
+ this.dispatch(step);
253
+ if (buffered.length) {
254
+ logDetail("proxy.buffer_flushed", { server: this.serverName, steps: buffered.length });
255
+ }
256
+ }
257
+ dispatch(step) {
258
+ if (this.tierValue === "bound" && this.control) {
259
+ const control = this.control;
260
+ this.queue.push(async () => {
261
+ const ok = await control.report(step);
262
+ if (!ok) {
263
+ // §10: the control server died mid-session. Everything after this is
264
+ // Tier 2, and the run is flagged lossy — the steps already sent live in
265
+ // the bound run, the rest in ours.
266
+ logLine("tier.downgrade", {
267
+ server: this.serverName,
268
+ why: "control server stopped answering",
269
+ });
270
+ this.control = undefined;
271
+ this.lossyValue = true;
272
+ await this.becomeStandalone("control server stopped answering");
273
+ this.dispatch(step);
274
+ }
275
+ });
276
+ return;
277
+ }
278
+ const recorder = this.recorder;
279
+ const runId = this.runId;
280
+ if (!recorder || !runId) {
281
+ // No owner yet — mid-downgrade, or the recorder is still being built.
282
+ // Buffering costs a moment; dropping costs the step.
283
+ this.bufferStep(step);
284
+ return;
285
+ }
286
+ const pair = this.ordering.allocatePair();
287
+ const toolName = step.qualifiedName || qualifyToolName(step.serverName, step.toolName);
288
+ const toolInput = serializeCapped(redact(step.args));
289
+ const toolOutput = step.isError ? undefined : serializeCapped(redact(step.result));
290
+ const toolError = step.isError
291
+ ? (step.errorMessage ?? serializeCapped(redact(step.result)))
292
+ : undefined;
293
+ this.queue.push(() => {
294
+ recorder.recordToolSelected(runId, pair.selected, { toolName, toolInput });
295
+ recorder.recordToolResponse(runId, pair.response, { toolName, toolOutput, toolError });
296
+ });
297
+ }
298
+ /** Close the Tier 2 run, if we own one, and drain the queue. */
299
+ async close() {
300
+ // Stops the work loop at its next turn. It is not awaited: it may be parked
301
+ // in a 25 s long-poll, and shutdown does not wait on a poll that will be
302
+ // answered by the control server closing anyway.
303
+ this.closing = true;
304
+ await this.negotiation?.catch(() => undefined);
305
+ this.flushBuffer();
306
+ await this.queue.flush();
307
+ if (this.recorder && this.runId) {
308
+ this.recorder.finishRun(this.runId, undefined, { durationMs: Date.now() - this.startedAt });
309
+ await this.recorder.flush(this.runId).catch(() => undefined);
310
+ logLine("run.finish", {
311
+ run: this.runId,
312
+ tier: "standalone",
313
+ steps: this.ordering.next,
314
+ lossy: this.lossyValue,
315
+ });
316
+ }
317
+ }
318
+ }
319
+ //# sourceMappingURL=session.js.map
@@ -0,0 +1,34 @@
1
+ /**
2
+ * housekeeping — host tools that are not work, and must not be treated as it.
3
+ *
4
+ * A host calls tools of its own that have nothing to do with the user's task.
5
+ * Claude Code opens a turn with `ToolSearch` to load deferred tool definitions,
6
+ * and writes a `TodoWrite` list while it plans. Neither reads or changes
7
+ * anything the user asked about.
8
+ *
9
+ * They must be excluded in TWO places, and missing either one is expensive:
10
+ *
11
+ * 1. **Recording.** A `ToolSearch` recorded as a step becomes a step of the
12
+ * calculated scenario — observed in production as a five-step scenario
13
+ * whose first step was `ToolSearch`. That is not merely untidy: the step's
14
+ * generated `reasoning` goes into the steering directive the model reads on
15
+ * every matched turn (~150 chars of it), it is not an MCP tool so it drags
16
+ * the whole scenario from `direct` mode into `steer`, and the analyser
17
+ * spends a Claude call writing input/output logic for a tool that has no
18
+ * output worth threading.
19
+ *
20
+ * 2. **Replay** (`ReplayController.preTool`). A divergence retires the plan,
21
+ * so treating the host's first housekeeping call as "the model went
22
+ * off-script" kills the replay before the model has had a chance to follow
23
+ * it — observed as `replay.diverge ... called=ToolSearch step=0/4`, after
24
+ * which the turn cost full price.
25
+ *
26
+ * DELIBERATELY TINY. Every entry must be pure host bookkeeping that cannot
27
+ * possibly be the user's task. `Read`, `Bash` and `Grep` are NOT here: they do
28
+ * real work, they belong in a recording, and reaching for one mid-replay
29
+ * usually *is* the model doing the task another way.
30
+ */
31
+ /** Host tools that are never recorded as steps and never count as divergence. */
32
+ export declare const HOUSEKEEPING_TOOLS: ReadonlySet<string>;
33
+ export declare function isHousekeeping(toolName: string): boolean;
34
+ //# sourceMappingURL=housekeeping.d.ts.map
@@ -0,0 +1,39 @@
1
+ /**
2
+ * housekeeping — host tools that are not work, and must not be treated as it.
3
+ *
4
+ * A host calls tools of its own that have nothing to do with the user's task.
5
+ * Claude Code opens a turn with `ToolSearch` to load deferred tool definitions,
6
+ * and writes a `TodoWrite` list while it plans. Neither reads or changes
7
+ * anything the user asked about.
8
+ *
9
+ * They must be excluded in TWO places, and missing either one is expensive:
10
+ *
11
+ * 1. **Recording.** A `ToolSearch` recorded as a step becomes a step of the
12
+ * calculated scenario — observed in production as a five-step scenario
13
+ * whose first step was `ToolSearch`. That is not merely untidy: the step's
14
+ * generated `reasoning` goes into the steering directive the model reads on
15
+ * every matched turn (~150 chars of it), it is not an MCP tool so it drags
16
+ * the whole scenario from `direct` mode into `steer`, and the analyser
17
+ * spends a Claude call writing input/output logic for a tool that has no
18
+ * output worth threading.
19
+ *
20
+ * 2. **Replay** (`ReplayController.preTool`). A divergence retires the plan,
21
+ * so treating the host's first housekeeping call as "the model went
22
+ * off-script" kills the replay before the model has had a chance to follow
23
+ * it — observed as `replay.diverge ... called=ToolSearch step=0/4`, after
24
+ * which the turn cost full price.
25
+ *
26
+ * DELIBERATELY TINY. Every entry must be pure host bookkeeping that cannot
27
+ * possibly be the user's task. `Read`, `Bash` and `Grep` are NOT here: they do
28
+ * real work, they belong in a recording, and reaching for one mid-replay
29
+ * usually *is* the model doing the task another way.
30
+ */
31
+ /** Host tools that are never recorded as steps and never count as divergence. */
32
+ export const HOUSEKEEPING_TOOLS = new Set([
33
+ "ToolSearch",
34
+ "TodoWrite",
35
+ ]);
36
+ export function isHousekeeping(toolName) {
37
+ return HOUSEKEEPING_TOOLS.has(toolName);
38
+ }
39
+ //# sourceMappingURL=housekeeping.js.map
@@ -0,0 +1,48 @@
1
+ /**
2
+ * queue — the bounded, fire-and-forget step queue (Phase 8, backpressure).
3
+ *
4
+ * Two rules from §3 and §10 meet here:
5
+ *
6
+ * - **Never let recording block a tool call.** Work is enqueued after the
7
+ * result is already on its way to the host, and `push` is synchronous.
8
+ * - **Never grow without bound.** A recorder that is slow or down must not turn
9
+ * into a memory leak in the host's own process tree. Past `capacity` the
10
+ * *oldest* item is dropped and the run is flagged `lossy` — losing the start
11
+ * of a long run is strictly better than losing its end, which is where the
12
+ * final answer lives.
13
+ *
14
+ * A dropped item is reported through {@link onDrop} so the flag reaches the run's
15
+ * metadata rather than dying in a log line.
16
+ */
17
+ export interface StepQueueOptions {
18
+ /** Max queued items before the oldest is dropped. Default 1000. */
19
+ capacity?: number;
20
+ /** Called once per dropped item, with the running drop total. */
21
+ onDrop?: (dropped: number) => void;
22
+ /** Called when a task throws. Sends are best-effort; failure is logged, not retried. */
23
+ onError?: (err: unknown) => void;
24
+ }
25
+ export declare const DEFAULT_CAPACITY = 1000;
26
+ /** A FIFO of async tasks, drained one at a time, with a hard size bound. */
27
+ export declare class StepQueue {
28
+ private readonly items;
29
+ private readonly capacity;
30
+ private readonly onDrop?;
31
+ private readonly onError?;
32
+ /** Resolved (and cleared) every time the queue goes idle. */
33
+ private readonly waiters;
34
+ private draining;
35
+ private droppedCount;
36
+ constructor(opts?: StepQueueOptions);
37
+ /** True once anything has been dropped — the run's `lossy` flag. */
38
+ get lossy(): boolean;
39
+ get dropped(): number;
40
+ get size(): number;
41
+ get busy(): boolean;
42
+ /** Enqueue work. Synchronous and never throws. */
43
+ push(task: () => Promise<void> | void): void;
44
+ private drain;
45
+ /** Await the queue draining. Resolves immediately when already idle. */
46
+ flush(): Promise<void>;
47
+ }
48
+ //# sourceMappingURL=queue.d.ts.map
@@ -0,0 +1,96 @@
1
+ /**
2
+ * queue — the bounded, fire-and-forget step queue (Phase 8, backpressure).
3
+ *
4
+ * Two rules from §3 and §10 meet here:
5
+ *
6
+ * - **Never let recording block a tool call.** Work is enqueued after the
7
+ * result is already on its way to the host, and `push` is synchronous.
8
+ * - **Never grow without bound.** A recorder that is slow or down must not turn
9
+ * into a memory leak in the host's own process tree. Past `capacity` the
10
+ * *oldest* item is dropped and the run is flagged `lossy` — losing the start
11
+ * of a long run is strictly better than losing its end, which is where the
12
+ * final answer lives.
13
+ *
14
+ * A dropped item is reported through {@link onDrop} so the flag reaches the run's
15
+ * metadata rather than dying in a log line.
16
+ */
17
+ export const DEFAULT_CAPACITY = 1000;
18
+ /** A FIFO of async tasks, drained one at a time, with a hard size bound. */
19
+ export class StepQueue {
20
+ items = [];
21
+ capacity;
22
+ onDrop;
23
+ onError;
24
+ /** Resolved (and cleared) every time the queue goes idle. */
25
+ waiters = [];
26
+ draining = false;
27
+ droppedCount = 0;
28
+ constructor(opts = {}) {
29
+ this.capacity = opts.capacity ?? DEFAULT_CAPACITY;
30
+ this.onDrop = opts.onDrop;
31
+ this.onError = opts.onError;
32
+ }
33
+ /** True once anything has been dropped — the run's `lossy` flag. */
34
+ get lossy() {
35
+ return this.droppedCount > 0;
36
+ }
37
+ get dropped() {
38
+ return this.droppedCount;
39
+ }
40
+ get size() {
41
+ return this.items.length;
42
+ }
43
+ get busy() {
44
+ return this.draining || this.items.length > 0;
45
+ }
46
+ /** Enqueue work. Synchronous and never throws. */
47
+ push(task) {
48
+ if (this.items.length >= this.capacity) {
49
+ this.items.shift();
50
+ this.droppedCount += 1;
51
+ this.onDrop?.(this.droppedCount);
52
+ }
53
+ this.items.push(task);
54
+ void this.drain();
55
+ }
56
+ async drain() {
57
+ if (this.draining)
58
+ return;
59
+ this.draining = true;
60
+ try {
61
+ for (;;) {
62
+ const task = this.items.shift();
63
+ if (!task)
64
+ break;
65
+ try {
66
+ await task();
67
+ }
68
+ catch (err) {
69
+ this.onError?.(err);
70
+ }
71
+ }
72
+ }
73
+ finally {
74
+ this.draining = false;
75
+ // A task enqueued during the last await is still in `items`; restart
76
+ // rather than waking waiters on a queue that is not actually empty.
77
+ if (this.items.length > 0) {
78
+ void this.drain();
79
+ }
80
+ else {
81
+ const waiting = this.waiters.splice(0, this.waiters.length);
82
+ for (const resolve of waiting)
83
+ resolve();
84
+ }
85
+ }
86
+ }
87
+ /** Await the queue draining. Resolves immediately when already idle. */
88
+ flush() {
89
+ if (!this.busy)
90
+ return Promise.resolve();
91
+ return new Promise((resolve) => {
92
+ this.waiters.push(resolve);
93
+ });
94
+ }
95
+ }
96
+ //# sourceMappingURL=queue.js.map
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Recorder — the surface both tiers use (§5).
3
+ *
4
+ * Deliberately identical in shape to RRepeat's `RemoteRecorder`, so the BaseIn
5
+ * service needs no new endpoints for v1: a BaseInstRunner run is the same row
6
+ * shape as an RRepeat one, discriminated only by `metadata.recorder`.
7
+ */
8
+ import type { ExecutionStage, ExecutionStepResult } from "../replay/types.js";
9
+ export interface RunMetrics {
10
+ costUsd?: number;
11
+ durationMs?: number;
12
+ }
13
+ export interface Recorder {
14
+ /** Create the run; returns the client-generated runId synchronously. */
15
+ startRun(input: string, metadata?: Record<string, unknown>): string;
16
+ recordToolSelected(runId: string, stepIndex: number, d: {
17
+ toolName: string;
18
+ toolInput: string;
19
+ context?: string;
20
+ }): string;
21
+ recordToolResponse(runId: string, stepIndex: number, d: {
22
+ toolName: string;
23
+ toolOutput?: string;
24
+ toolError?: string;
25
+ }): string;
26
+ recordFinalAnswer(runId: string, stepIndex: number, d: {
27
+ answer: string;
28
+ }): string;
29
+ finishRun(runId: string, finalOutput?: string, metrics?: RunMetrics): void;
30
+ /** Await every queued send for a run. Called once at session end. */
31
+ flush(runId: string): Promise<void>;
32
+ }
33
+ /**
34
+ * A similar-prompt hit reported by `POST /recordings/runs`. On a hit the server
35
+ * does **not** create a run, so every later step post for that id would 404 —
36
+ * callers must stop recording it. What the caller does *instead* of recording is
37
+ * v2's subject (docs/calculatedReplay.md); stopping is mandatory either way.
38
+ */
39
+ export interface RunMatch {
40
+ runId: string;
41
+ scenarioId: string | null;
42
+ similarity: number;
43
+ scenario: Record<string, unknown> | null;
44
+ executionTicket?: string;
45
+ }
46
+ /** Optional capability: `null` means "fresh run, keep recording". */
47
+ export interface MatchAware {
48
+ getMatch(runId: string): Promise<RunMatch | null>;
49
+ }
50
+ export declare function isMatchAware(r: Recorder): r is Recorder & MatchAware;
51
+ /**
52
+ * What a matched turn actually cost, reported once per match
53
+ * (docs/calculatedReplay.md §11).
54
+ *
55
+ * `not_steered` and `failed` mean the agent ran the task the ordinary way; the
56
+ * service files those as **baseline samples** rather than savings, which is what
57
+ * keeps the ledger's denominator honest. `steered_full` and `diverged` book a
58
+ * saving against that baseline.
59
+ */
60
+ export interface ExecutionReport {
61
+ scenarioId: string;
62
+ /** The match's claim token. It *is* the execution row's id, so a doubled report books once. */
63
+ ticket?: string;
64
+ outcome: "steered_full" | "diverged" | "not_steered" | "failed";
65
+ /** The derivation call — the only tokens replay itself spends. */
66
+ deriveCostUsd: number;
67
+ /** The live turn, from the transcript usage delta. */
68
+ sessionCostUsd: number;
69
+ /** Divergence recovery. Zero here by construction: the proxies run for free. */
70
+ fallbackCostUsd: number;
71
+ durationMs: number;
72
+ stepsPlanned: number;
73
+ stepsPinned: number;
74
+ /** False when the usage watermark was missing — "unmeasured" beats a wrong baseline. */
75
+ measured: boolean;
76
+ pricingVersion: string;
77
+ prompt?: string;
78
+ /**
79
+ * One verdict per step this run reached (errorshandling.md). Absent when the
80
+ * run reached none — a decline, or a turn that died before the first step.
81
+ *
82
+ * This is the only channel through which "step 2 of your scenario throws"
83
+ * reaches the person who owns the scenario: the log line is on their machine,
84
+ * the recording page is not.
85
+ */
86
+ steps?: ExecutionStepResult[];
87
+ /** The headline failure, lifted from the first step that broke. */
88
+ error?: string;
89
+ errorStage?: ExecutionStage;
90
+ errorStepIndex?: number;
91
+ errorToolName?: string;
92
+ }
93
+ /** Optional capability: reporting needs a service, and a NullRecorder has none. */
94
+ export interface ScenarioReporter {
95
+ reportExecution(report: ExecutionReport): void;
96
+ }
97
+ export declare function isScenarioReporter(r: Recorder): r is Recorder & ScenarioReporter;
98
+ /**
99
+ * A recorder that drops everything. Used when auth is unavailable: recording is
100
+ * best-effort, and a session must never fail because BaseIn is unreachable (§10).
101
+ */
102
+ export declare class NullRecorder implements Recorder {
103
+ private n;
104
+ startRun(): string;
105
+ recordToolSelected(): string;
106
+ recordToolResponse(): string;
107
+ recordFinalAnswer(): string;
108
+ finishRun(): void;
109
+ flush(): Promise<void>;
110
+ }
111
+ //# sourceMappingURL=recorder.d.ts.map