@basein/runner 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +276 -0
  3. package/dist/auth/client.d.ts +85 -0
  4. package/dist/auth/client.js +284 -0
  5. package/dist/bin/bir-hooks.d.ts +48 -0
  6. package/dist/bin/bir-hooks.js +201 -0
  7. package/dist/bin/bir-proxy.d.ts +45 -0
  8. package/dist/bin/bir-proxy.js +207 -0
  9. package/dist/bin/bir-scenario.d.ts +24 -0
  10. package/dist/bin/bir-scenario.js +177 -0
  11. package/dist/bin/bir.d.ts +21 -0
  12. package/dist/bin/bir.js +876 -0
  13. package/dist/config/adapters/claude-code.d.ts +76 -0
  14. package/dist/config/adapters/claude-code.js +181 -0
  15. package/dist/config/adapters/generic.d.ts +17 -0
  16. package/dist/config/adapters/generic.js +36 -0
  17. package/dist/config/generate.d.ts +127 -0
  18. package/dist/config/generate.js +114 -0
  19. package/dist/config/resolve.d.ts +68 -0
  20. package/dist/config/resolve.js +132 -0
  21. package/dist/control/client.d.ts +56 -0
  22. package/dist/control/client.js +86 -0
  23. package/dist/control/correlation.d.ts +86 -0
  24. package/dist/control/correlation.js +0 -0
  25. package/dist/control/discovery.d.ts +50 -0
  26. package/dist/control/discovery.js +123 -0
  27. package/dist/control/ordering.d.ts +38 -0
  28. package/dist/control/ordering.js +44 -0
  29. package/dist/control/paths.d.ts +32 -0
  30. package/dist/control/paths.js +56 -0
  31. package/dist/control/server.d.ts +272 -0
  32. package/dist/control/server.js +1131 -0
  33. package/dist/control/transcript.d.ts +75 -0
  34. package/dist/control/transcript.js +241 -0
  35. package/dist/index.d.ts +37 -0
  36. package/dist/index.js +32 -0
  37. package/dist/jsonrpc/framing.d.ts +49 -0
  38. package/dist/jsonrpc/framing.js +143 -0
  39. package/dist/jsonrpc/types.d.ts +52 -0
  40. package/dist/jsonrpc/types.js +46 -0
  41. package/dist/proxy/intercept.d.ts +55 -0
  42. package/dist/proxy/intercept.js +147 -0
  43. package/dist/proxy/relay.d.ts +97 -0
  44. package/dist/proxy/relay.js +166 -0
  45. package/dist/proxy/session.d.ts +116 -0
  46. package/dist/proxy/session.js +319 -0
  47. package/dist/record/housekeeping.d.ts +34 -0
  48. package/dist/record/housekeeping.js +39 -0
  49. package/dist/record/queue.d.ts +48 -0
  50. package/dist/record/queue.js +96 -0
  51. package/dist/record/recorder.d.ts +111 -0
  52. package/dist/record/recorder.js +39 -0
  53. package/dist/record/redact.d.ts +37 -0
  54. package/dist/record/redact.js +119 -0
  55. package/dist/record/remote-recorder.d.ts +110 -0
  56. package/dist/record/remote-recorder.js +301 -0
  57. package/dist/record/truncate.d.ts +36 -0
  58. package/dist/record/truncate.js +85 -0
  59. package/dist/replay/bundle.d.ts +36 -0
  60. package/dist/replay/bundle.js +89 -0
  61. package/dist/replay/controller.d.ts +300 -0
  62. package/dist/replay/controller.js +807 -0
  63. package/dist/replay/coverage.d.ts +41 -0
  64. package/dist/replay/coverage.js +56 -0
  65. package/dist/replay/derive.d.ts +58 -0
  66. package/dist/replay/derive.js +166 -0
  67. package/dist/replay/executor.d.ts +78 -0
  68. package/dist/replay/executor.js +233 -0
  69. package/dist/replay/logic.d.ts +31 -0
  70. package/dist/replay/logic.js +50 -0
  71. package/dist/replay/plan.d.ts +181 -0
  72. package/dist/replay/plan.js +397 -0
  73. package/dist/replay/pricing.d.ts +41 -0
  74. package/dist/replay/pricing.js +76 -0
  75. package/dist/replay/source-run.d.ts +50 -0
  76. package/dist/replay/source-run.js +98 -0
  77. package/dist/replay/tool-error.d.ts +22 -0
  78. package/dist/replay/tool-error.js +60 -0
  79. package/dist/replay/types.d.ts +116 -0
  80. package/dist/replay/types.js +35 -0
  81. package/dist/upstream/client.d.ts +78 -0
  82. package/dist/upstream/client.js +114 -0
  83. package/dist/upstream/http-client.d.ts +78 -0
  84. package/dist/upstream/http-client.js +261 -0
  85. package/dist/upstream/lazy-client.d.ts +31 -0
  86. package/dist/upstream/lazy-client.js +53 -0
  87. package/dist/upstream/stdio-client.d.ts +57 -0
  88. package/dist/upstream/stdio-client.js +203 -0
  89. package/dist/util/log.d.ts +27 -0
  90. package/dist/util/log.js +51 -0
  91. package/dist/util/version.d.ts +2 -0
  92. package/dist/util/version.js +40 -0
  93. package/docs/BaseInstRunner.md +621 -0
  94. package/docs/calculatedReplay.md +1185 -0
  95. package/docs/calculatedReplayGuide.md +448 -0
  96. package/docs/installRun.md +413 -0
  97. package/docs/mcpmark.md +752 -0
  98. package/docs/quickstart.md +201 -0
  99. package/docs/t-bench.md +394 -0
  100. package/package.json +56 -0
@@ -0,0 +1,807 @@
1
+ /**
2
+ * controller — everything replay decides, in one place (docs/calculatedReplay.md).
3
+ *
4
+ * The control server owns runs, ordering and recording. This owns the *plan*:
5
+ * the gate ladder, the mode choice, arming, pinning, threading, direct
6
+ * execution, divergence recovery and the execution report. `ControlServer`
7
+ * delegates to it and stores one {@link ReplayState} per run.
8
+ *
9
+ * THE GOVERNING RULE IS UNCHANGED and outranks every optimisation here: the host
10
+ * session must never fail because of BaseInstRunner. Every method below either
11
+ * degrades to "run the turn normally" or returns something the session can
12
+ * ignore. Nothing throws at a hook.
13
+ */
14
+ import { parseQualifiedName } from "../control/correlation.js";
15
+ import { isHousekeeping } from "../record/housekeeping.js";
16
+ import { redact } from "../record/redact.js";
17
+ import { serializeCapped } from "../record/truncate.js";
18
+ import { logDetail, logLine, errText } from "../util/log.js";
19
+ import { MAX_REPLAY_REASON } from "./bundle.js";
20
+ import { coverageOf, modeFor, reachOf } from "./coverage.js";
21
+ import { deriveParameters } from "./derive.js";
22
+ import { ProxyWorkQueue } from "./executor.js";
23
+ import { ScenarioReplayPlan } from "./plan.js";
24
+ import { PRICING_VERSION } from "./pricing.js";
25
+ import { SourceRunOutputs } from "./source-run.js";
26
+ import { toolResultError } from "./tool-error.js";
27
+ import { OUTCOME_RANK, isReadyScenario, } from "./types.js";
28
+ /** The first-party tool a `direct` plan is delivered through (§6.3). */
29
+ export const DIRECT_TOOL_NAME = "mcp__bir__run_scenario";
30
+ /** How long a `/proxy/poll` is held open before it answers empty. */
31
+ export const POLL_HOLD_MS = 25_000;
32
+ export const DEFAULT_BUDGETS = {
33
+ matchMs: 2_500,
34
+ deriveMs: 8_000,
35
+ planMs: 120_000,
36
+ stepMs: 60_000,
37
+ };
38
+ /** How many step verdicts one report carries. A chain longer than this is not
39
+ * a chain anybody is reading step by step, and the report has to stay small. */
40
+ const MAX_STEP_RESULTS = 200;
41
+ /** Per-step error budget. The console shows these inline, under the step. */
42
+ const MAX_STEP_ERROR_CHARS = 500;
43
+ export class ReplayController {
44
+ enabled;
45
+ work = new ProxyWorkQueue();
46
+ budgets;
47
+ opts;
48
+ constructor(opts) {
49
+ this.opts = opts;
50
+ this.enabled = opts.enabled;
51
+ this.budgets = { ...DEFAULT_BUDGETS, ...(opts.budgets ?? {}) };
52
+ }
53
+ /** The plan's own delivery vehicle is never a scenario step. */
54
+ isDirectTool(toolName) {
55
+ return toolName === DIRECT_TOOL_NAME;
56
+ }
57
+ /**
58
+ * Await a match under {@link ReplayBudgets.matchMs}.
59
+ *
60
+ * Budget expiry is not an error and never blocks the user: it means this turn
61
+ * runs normally, and the one `replay.decision` line says so. Resolving `null`
62
+ * covers both "no match" and "too slow", which are the same thing from here.
63
+ */
64
+ async awaitMatch(pending) {
65
+ let timer;
66
+ const budget = new Promise((resolve) => {
67
+ timer = setTimeout(() => resolve(null), this.budgets.matchMs);
68
+ timer.unref?.();
69
+ });
70
+ try {
71
+ const match = (await Promise.race([pending, budget]));
72
+ return match ?? null;
73
+ }
74
+ catch {
75
+ return null;
76
+ }
77
+ finally {
78
+ if (timer)
79
+ clearTimeout(timer);
80
+ }
81
+ }
82
+ /**
83
+ * Run the gate ladder and, if every gate passes, arm a plan.
84
+ *
85
+ * Always returns a state — a declined match still has a ticket to redeem and a
86
+ * baseline sample to contribute, and losing that is how a savings ledger ends
87
+ * up with a denominator nobody measured.
88
+ */
89
+ arm(match, prompt, wrapped) {
90
+ const state = {
91
+ scenarioId: match.scenarioId,
92
+ ticket: match.executionTicket,
93
+ similarity: match.similarity,
94
+ matchedRunId: match.runId,
95
+ mode: "none",
96
+ wrapped,
97
+ pinned: new Map(),
98
+ outcome: "not_steered",
99
+ deriveCostUsd: 0,
100
+ fallbackCostUsd: 0,
101
+ stepsPlanned: 0,
102
+ stepsPinned: 0,
103
+ stepResults: new Map(),
104
+ armedAt: Date.now(),
105
+ reported: false,
106
+ retired: false,
107
+ };
108
+ const decline = (why) => {
109
+ logLine("replay.decision", {
110
+ verdict: "no-steer",
111
+ run: match.runId,
112
+ scenario: match.scenarioId ?? undefined,
113
+ similarity: match.similarity.toFixed(3),
114
+ threshold: this.opts.minSimilarity,
115
+ why,
116
+ });
117
+ return state;
118
+ };
119
+ // Gate 2 — replay enabled.
120
+ if (!this.enabled)
121
+ return decline("BIR_REPLAY is not set");
122
+ // Gate 3 — a ready scenario with steps.
123
+ const scenario = match.scenario;
124
+ if (!match.scenarioId || !isReadyScenario(scenario)) {
125
+ return decline(match.scenarioId ? "scenario is not ready" : "no scenario for the matched run");
126
+ }
127
+ // Gate 4 — similarity. Detection decides "don't record this again"; steering
128
+ // decides "don't think about this again", which is a much stronger claim and
129
+ // deserves a stronger threshold.
130
+ if (match.similarity < this.opts.minSimilarity) {
131
+ return decline("similarity below threshold");
132
+ }
133
+ // Gate 5 — tool coverage.
134
+ const mode = modeFor(scenario.steps, wrapped, this.opts.allowServers);
135
+ if (mode === "none")
136
+ return decline("no step is executable");
137
+ const params = this.startDerivation(scenario, prompt, state);
138
+ state.plan = new ScenarioReplayPlan({ scenario, params, mode });
139
+ state.mode = mode;
140
+ state.stepsPlanned = scenario.steps.length;
141
+ if (this.opts.authUrl && this.opts.authToken) {
142
+ state.sourceRun = new SourceRunOutputs({
143
+ baseUrl: this.opts.authUrl,
144
+ runId: scenario.runId,
145
+ token: this.opts.authToken,
146
+ fetchImpl: this.opts.fetchImpl,
147
+ });
148
+ }
149
+ // The audit line precedes the action, always (§13.2, mitigation 4).
150
+ logLine("plan.armed", {
151
+ run: match.runId,
152
+ scenario: scenario.id,
153
+ mode,
154
+ steps: scenario.steps.length,
155
+ similarity: match.similarity.toFixed(3),
156
+ coverage: coverageOf(scenario.steps, wrapped, this.opts.allowServers).join(","),
157
+ tools: scenario.steps.map((s) => s.toolName).join(","),
158
+ });
159
+ return state;
160
+ }
161
+ /** The directive to inject via `additionalContext`, or undefined when declined. */
162
+ directiveFor(state) {
163
+ return state.plan?.steeringDirective(DIRECT_TOOL_NAME);
164
+ }
165
+ /**
166
+ * `PreToolUse`, while a plan is active.
167
+ *
168
+ * Three tiers (§8): pin the expected call; on divergence execute the remainder
169
+ * and hand it back through the best channel available; if even that fails,
170
+ * abort to an ordinary turn.
171
+ */
172
+ async preTool(state, toolName, toolUseId) {
173
+ const plan = state.plan;
174
+ if (!plan || state.retired)
175
+ return { kind: "passthrough" };
176
+ // The delivery vehicle for a direct plan is not a step in it.
177
+ if (this.isDirectTool(toolName))
178
+ return { kind: "passthrough" };
179
+ try {
180
+ await this.withBudget(plan.ready(), this.budgets.deriveMs, "derivation");
181
+ }
182
+ catch (err) {
183
+ logLine("replay.derive_failed", {
184
+ scenario: state.scenarioId ?? undefined,
185
+ why: "continuing with the scenario's recorded sample values",
186
+ error: errText(err),
187
+ });
188
+ }
189
+ // A direct plan does not steer individual calls: the model was asked for one
190
+ // tool call and made a different one. That is divergence.
191
+ const expected = plan.expectedTool();
192
+ if (state.mode === "steer" && toolName === expected && toolUseId) {
193
+ try {
194
+ const step = plan.currentStep();
195
+ const input = plan.toolInputForCurrentStep();
196
+ state.pinned.set(toolUseId, {
197
+ stepIndex: plan.currentStepIndex,
198
+ reach: this.reachFor(state, step),
199
+ toolName,
200
+ pinnedAt: Date.now(),
201
+ });
202
+ logDetail("replay.pin", { n: plan.currentStepIndex, tool: toolName });
203
+ return { kind: "pin", input, stepIndex: plan.currentStepIndex };
204
+ }
205
+ catch (err) {
206
+ // `toolInputLogic` threw: the pin is impossible, but the scenario may
207
+ // still be recoverable by executing the rest.
208
+ //
209
+ // Recorded before diverging, because `composeBundle` starts at this same
210
+ // step and will throw on the same logic — and its own report would then
211
+ // be the only trace, attributed to a re-execution rather than to the
212
+ // steered call that actually hit it first.
213
+ const step = plan.currentStep();
214
+ if (step) {
215
+ this.recordStep(state, {
216
+ step,
217
+ outcome: "failed",
218
+ stage: "tool_input_logic",
219
+ error: errText(err),
220
+ });
221
+ }
222
+ return await this.diverge(state, toolName, `toolInputLogic threw: ${errText(err)}`);
223
+ }
224
+ }
225
+ // HOUSEKEEPING IS NOT DIVERGENCE. See `record/housekeeping.ts`.
226
+ if (isHousekeeping(toolName)) {
227
+ logDetail("replay.housekeeping", {
228
+ tool: toolName,
229
+ step: `${plan.currentStepIndex}/${plan.stepCount}`,
230
+ why: "host bookkeeping, not task work — plan stays armed",
231
+ });
232
+ return { kind: "passthrough" };
233
+ }
234
+ return await this.diverge(state, toolName, `expected ${expected ?? "no more tools"}, model called ${toolName}`);
235
+ }
236
+ /**
237
+ * `PostToolUse` for a call this plan pinned.
238
+ *
239
+ * `output` must be serialized the way the step's output was **recorded**: for a
240
+ * wrapped MCP step that is the proxy's whole `CallToolResult`, which is why the
241
+ * server threads those from `/proxy/step` rather than from the hook (§7.2).
242
+ * Returns true when the plan is now complete.
243
+ */
244
+ postTool(state, toolUseId, output) {
245
+ const plan = state.plan;
246
+ const pin = state.pinned.get(toolUseId);
247
+ if (!plan || !pin || state.retired)
248
+ return false;
249
+ state.pinned.delete(toolUseId);
250
+ // The pin's own step, not `currentStep()`: `applyOutput` moves the cursor, so
251
+ // reading it after the call would judge the *next* step, and reading it
252
+ // before assumes no other pin ever moved it.
253
+ const step = plan.allSteps()[pin.stepIndex];
254
+ let derivedKeys = [];
255
+ try {
256
+ derivedKeys = plan.applyOutput(output);
257
+ }
258
+ catch (err) {
259
+ logLine("replay.thread_failed", {
260
+ scenario: state.scenarioId ?? undefined,
261
+ step: pin.stepIndex,
262
+ why: "toolOutputLogic threw — continuing as a normal run",
263
+ error: errText(err),
264
+ });
265
+ if (step) {
266
+ this.recordStep(state, {
267
+ step,
268
+ outcome: "failed",
269
+ stage: "tool_output_logic",
270
+ error: errText(err),
271
+ ms: Date.now() - pin.pinnedAt,
272
+ });
273
+ }
274
+ this.retire(state, "failed");
275
+ return false;
276
+ }
277
+ // One more step ran under the plan. Counted here, as each pin threads, and
278
+ // not from `pinned.size` at pin time: a sequential chain never has more than
279
+ // one pin outstanding, so that size reported 1 whatever the chain's length.
280
+ state.stepsPinned += 1;
281
+ // The step ran in the live session and threaded. This is the only place a
282
+ // *steered* step is judged: `plan` never sees the call, the host did. A tool
283
+ // that ran and reported an error did not do the step's work, so the plan is
284
+ // retired as `failed` — a baseline sample, never a saving.
285
+ const toolError = step ? toolResultError(output) : undefined;
286
+ if (step && toolError) {
287
+ const info = {
288
+ step,
289
+ outcome: "failed",
290
+ stage: "tool_call",
291
+ error: toolError,
292
+ ms: Date.now() - pin.pinnedAt,
293
+ };
294
+ this.logStep(info);
295
+ this.recordStep(state, info);
296
+ this.retire(state, "failed");
297
+ return false;
298
+ }
299
+ if (step) {
300
+ this.recordStep(state, {
301
+ step,
302
+ outcome: "executed",
303
+ ms: Date.now() - pin.pinnedAt,
304
+ });
305
+ }
306
+ const done = plan.isDone();
307
+ logDetail("replay.thread", {
308
+ n: pin.stepIndex,
309
+ via: pin.reach === "direct" ? "proxy" : "hook",
310
+ emitted: derivedKeys.join(",") || undefined,
311
+ done,
312
+ });
313
+ if (done) {
314
+ this.upgrade(state, "steered_full");
315
+ this.retire(state, undefined);
316
+ logLine("replay.done", {
317
+ scenario: state.scenarioId ?? undefined,
318
+ mode: state.mode,
319
+ steps: `${plan.stepCount}/${plan.stepCount}`,
320
+ outcome: state.outcome,
321
+ ms: Date.now() - state.armedAt,
322
+ });
323
+ }
324
+ return done;
325
+ }
326
+ /** Whether a pinned call's output is threaded from the proxy's report (§7.2). */
327
+ threadsFromProxy(state, toolUseId) {
328
+ return state.pinned.get(toolUseId)?.reach === "direct";
329
+ }
330
+ /** True when this call belongs to the active plan. */
331
+ isPinned(state, toolUseId) {
332
+ return state.pinned.has(toolUseId);
333
+ }
334
+ /**
335
+ * `POST /scenario/run` — the whole of a `direct` plan, executed here.
336
+ *
337
+ * Every step runs through the proxy that already owns its upstream, so the
338
+ * model spends nothing beyond the turn that reads the results.
339
+ */
340
+ async runArmed(state) {
341
+ const plan = state.plan;
342
+ if (!plan || state.retired)
343
+ return { ok: false, why: "no_plan" };
344
+ try {
345
+ await this.withBudget(plan.ready(), this.budgets.deriveMs, "derivation");
346
+ }
347
+ catch (err) {
348
+ logLine("replay.derive_failed", {
349
+ scenario: state.scenarioId ?? undefined,
350
+ why: "continuing with the scenario's recorded sample values",
351
+ error: errText(err),
352
+ });
353
+ }
354
+ const deadline = Date.now() + this.budgets.planMs;
355
+ let result;
356
+ try {
357
+ result = await plan.runToCompletion(this.executeStep(), state.sourceRun ? (step) => state.sourceRun.outputFor(step) : undefined, this.observeStep(state), deadline);
358
+ }
359
+ catch (err) {
360
+ // The scenario's own logic failed. Retire and let the model do the work.
361
+ logLine("replay.failed", {
362
+ scenario: state.scenarioId ?? undefined,
363
+ why: "scenario logic threw — continuing as a normal run",
364
+ error: errText(err),
365
+ });
366
+ this.retire(state, "failed");
367
+ return { ok: false, why: "not_ready" };
368
+ }
369
+ state.stepsPinned += result.executed + result.recorded;
370
+ let responseModel = {};
371
+ try {
372
+ responseModel = plan.responseModel();
373
+ }
374
+ catch (err) {
375
+ logDetail("replay.response_model_failed", { error: errText(err) });
376
+ }
377
+ // A plan whose tools errored ran to the end and did nothing. `steered_full`
378
+ // would book the largest saving available for it; `failed` books a baseline
379
+ // sample, which is what a replay that produced nothing is worth (mcpmark.md §13).
380
+ this.upgrade(state, result.errored > 0
381
+ ? "failed"
382
+ : result.partial || result.skipped > 0
383
+ ? "diverged"
384
+ : "steered_full");
385
+ this.retire(state, undefined);
386
+ logLine("replay.done", {
387
+ scenario: state.scenarioId ?? undefined,
388
+ mode: state.mode,
389
+ steps: `${result.executed + result.recorded}/${plan.stepCount}`,
390
+ recorded: result.recorded || undefined,
391
+ skipped: result.skipped || undefined,
392
+ errored: result.errored || undefined,
393
+ outcome: state.outcome,
394
+ partial: result.partial || undefined,
395
+ ms: Date.now() - state.armedAt,
396
+ });
397
+ return {
398
+ ok: true,
399
+ text: result.text,
400
+ responseModel,
401
+ steps: result.executed + result.recorded,
402
+ partial: result.partial,
403
+ };
404
+ }
405
+ /**
406
+ * Build the execution report for a sealed run, or undefined when there is
407
+ * nothing to report.
408
+ *
409
+ * `costUsd <= 0` on a decline is not worth sending: the service answers
410
+ * `202 { recorded: false }` for a baseline sample with no measured cost, and a
411
+ * report that books nothing is noise on both sides.
412
+ */
413
+ buildReport(state, d) {
414
+ if (state.reported || !state.scenarioId)
415
+ return undefined;
416
+ state.reported = true;
417
+ const total = state.deriveCostUsd + d.sessionCostUsd + state.fallbackCostUsd;
418
+ const steps = this.stepResultsOf(state);
419
+ const isBaseline = state.outcome === "not_steered" || state.outcome === "failed";
420
+ // A costless decline is noise on both sides — *unless* a step actually broke,
421
+ // which is the one thing the recording page cannot learn any other way. The
422
+ // service accepts a costless report that says why (its errorshandling.md).
423
+ if (isBaseline && total <= 0 && !steps?.some((s) => s.status === "failed")) {
424
+ return undefined;
425
+ }
426
+ const blame = this.blameStep(steps);
427
+ return {
428
+ scenarioId: state.scenarioId,
429
+ ticket: state.ticket,
430
+ outcome: state.outcome,
431
+ deriveCostUsd: state.deriveCostUsd,
432
+ sessionCostUsd: d.sessionCostUsd,
433
+ fallbackCostUsd: state.fallbackCostUsd,
434
+ durationMs: d.durationMs,
435
+ stepsPlanned: state.stepsPlanned,
436
+ stepsPinned: state.stepsPinned,
437
+ measured: d.measured,
438
+ pricingVersion: PRICING_VERSION,
439
+ prompt: d.prompt,
440
+ steps,
441
+ // The headline, lifted from the first step that broke. The console leads
442
+ // with this and shows the per-step verdicts underneath, so the two must
443
+ // name the same failure rather than being assembled independently.
444
+ error: blame?.error,
445
+ errorStage: blame?.stage,
446
+ errorStepIndex: blame?.stepIndex,
447
+ errorToolName: blame?.toolName,
448
+ };
449
+ }
450
+ /**
451
+ * The step a failed replay is fairly blamed on: the first whose own logic
452
+ * threw, or failing that the first that could not run at all.
453
+ *
454
+ * A `failed` step outranks a `skipped` one however early the skip came — a
455
+ * scenario whose logic throws is broken for everybody, while a step that could
456
+ * not run here says something about this machine.
457
+ */
458
+ blameStep(steps) {
459
+ if (!steps)
460
+ return undefined;
461
+ return (steps.find((s) => s.status === "failed") ?? steps.find((s) => s.status === "skipped"));
462
+ }
463
+ /**
464
+ * Run a scenario that no match armed — `bir replay` (§12.1).
465
+ *
466
+ * The same plan, the same executor, the same live proxies; only the trigger
467
+ * differs. It is how you test a scenario without a session, how a Tier 2 or
468
+ * non-Claude-Code client gets any replay at all, and the first thing to reach
469
+ * for when a steered turn behaves oddly.
470
+ *
471
+ * Deliberately outside the gate ladder: the operator typed the scenario id, so
472
+ * there is nothing to be similar *to* and nothing to decline. It books no
473
+ * execution either — there is no ticket, and inventing a saving for a manual
474
+ * invocation is exactly the kind of number a ledger must never contain.
475
+ */
476
+ async runAdHoc(scenario, prompt, wrapped) {
477
+ if (!isReadyScenario(scenario))
478
+ return { ok: false, why: "scenario is not ready" };
479
+ const mode = modeFor(scenario.steps, wrapped, this.opts.allowServers);
480
+ if (mode === "none")
481
+ return { ok: false, why: "no step is executable" };
482
+ const state = {
483
+ scenarioId: scenario.id,
484
+ similarity: 1,
485
+ matchedRunId: scenario.runId,
486
+ mode,
487
+ wrapped,
488
+ pinned: new Map(),
489
+ outcome: "not_steered",
490
+ deriveCostUsd: 0,
491
+ fallbackCostUsd: 0,
492
+ stepsPlanned: scenario.steps.length,
493
+ stepsPinned: 0,
494
+ stepResults: new Map(),
495
+ armedAt: Date.now(),
496
+ // No ticket, so nothing to redeem — and `buildReport` is never called for
497
+ // an ad-hoc run anyway.
498
+ reported: true,
499
+ retired: false,
500
+ };
501
+ const params = this.startDerivation(scenario, prompt, state);
502
+ const plan = new ScenarioReplayPlan({ scenario, params, mode });
503
+ state.plan = plan;
504
+ if (this.opts.authUrl && this.opts.authToken) {
505
+ state.sourceRun = new SourceRunOutputs({
506
+ baseUrl: this.opts.authUrl,
507
+ runId: scenario.runId,
508
+ token: this.opts.authToken,
509
+ fetchImpl: this.opts.fetchImpl,
510
+ });
511
+ }
512
+ logLine("plan.armed", {
513
+ scenario: scenario.id,
514
+ mode,
515
+ steps: scenario.steps.length,
516
+ coverage: coverageOf(scenario.steps, wrapped, this.opts.allowServers).join(","),
517
+ why: "bir replay — no match, an operator asked for it",
518
+ });
519
+ try {
520
+ await this.withBudget(plan.ready(), this.budgets.deriveMs, "derivation");
521
+ }
522
+ catch (err) {
523
+ logLine("replay.derive_failed", { scenario: scenario.id, error: errText(err) });
524
+ }
525
+ const trace = [];
526
+ try {
527
+ const result = await plan.runToCompletion(this.executeStep(), state.sourceRun ? (step) => state.sourceRun.outputFor(step) : undefined, (info) => {
528
+ this.observeStep(state)(info);
529
+ trace.push({
530
+ step: info.step.stepIndex,
531
+ tool: info.step.toolName,
532
+ input: info.input,
533
+ outcome: info.outcome,
534
+ ms: info.ms,
535
+ });
536
+ }, Date.now() + this.budgets.planMs);
537
+ let responseModel = {};
538
+ try {
539
+ responseModel = plan.responseModel();
540
+ }
541
+ catch (err) {
542
+ logDetail("replay.response_model_failed", { error: errText(err) });
543
+ }
544
+ return {
545
+ ok: true,
546
+ text: result.text,
547
+ responseModel,
548
+ steps: result.executed + result.recorded,
549
+ partial: result.partial,
550
+ trace,
551
+ };
552
+ }
553
+ catch (err) {
554
+ return { ok: false, why: errText(err), trace };
555
+ }
556
+ }
557
+ /** Release every parked poller. Called at control-server shutdown. */
558
+ close() {
559
+ this.work.close();
560
+ }
561
+ // ── internals ────────────────────────────────────────────────────────────
562
+ /**
563
+ * Kick off derivation without awaiting it (§10). The prompt hook returns the
564
+ * directive the moment the match lands; the first `PreToolUse` — or
565
+ * `/scenario/run`, which has no hook timeout at all — is where the wait lands.
566
+ */
567
+ startDerivation(scenario, prompt, state) {
568
+ const derive = this.opts.deriveImpl ?? deriveParameters;
569
+ return derive({
570
+ prompt,
571
+ intent: scenario.intent ?? "",
572
+ paramsObject: scenario.paramsObject,
573
+ apiKey: this.opts.apiKey ?? process.env.ANTHROPIC_API_KEY,
574
+ fetchImpl: this.opts.fetchImpl,
575
+ })
576
+ .then((r) => {
577
+ state.deriveCostUsd += r.costUsd;
578
+ logLine("replay.derived", {
579
+ scenario: scenario.id,
580
+ params: Object.keys(r.params).length,
581
+ costUsd: r.costUsd.toFixed(4),
582
+ source: r.derived ? "prompt" : "recorded samples",
583
+ });
584
+ return r.params;
585
+ })
586
+ .catch((err) => {
587
+ logLine("replay.derive_failed", {
588
+ scenario: scenario.id,
589
+ why: "falling back to the scenario's recorded sample values",
590
+ error: errText(err),
591
+ });
592
+ const fallback = {};
593
+ for (const [key, entry] of Object.entries(scenario.paramsObject ?? {})) {
594
+ fallback[key] = entry.sampleValue;
595
+ }
596
+ return fallback;
597
+ });
598
+ }
599
+ /**
600
+ * Divergence (§8). Execute the remaining steps for real, then deliver.
601
+ *
602
+ * The plan is retired first: whatever happens next, this turn neither steers
603
+ * nor re-injects again.
604
+ */
605
+ async diverge(state, toolName, why) {
606
+ const plan = state.plan;
607
+ if (!plan)
608
+ return { kind: "passthrough" };
609
+ logLine("replay.diverge", {
610
+ scenario: state.scenarioId ?? undefined,
611
+ expected: plan.expectedTool() ?? "(none)",
612
+ called: toolName,
613
+ step: `${plan.currentStepIndex}/${plan.stepCount}`,
614
+ why,
615
+ });
616
+ this.retire(state, undefined);
617
+ let composed;
618
+ try {
619
+ composed = await plan.composeBundle(MAX_REPLAY_REASON, this.executeStep(), state.sourceRun ? (step) => state.sourceRun.outputFor(step) : undefined, this.observeStep(state));
620
+ }
621
+ catch (err) {
622
+ this.upgrade(state, "failed");
623
+ logLine("replay.compose_failed", {
624
+ scenario: state.scenarioId ?? undefined,
625
+ why: "re-execution threw — continuing as a normal run",
626
+ error: errText(err),
627
+ });
628
+ return { kind: "abort" };
629
+ }
630
+ // Same rule as `runArmed`: a recovered chain whose tools errored is `failed`.
631
+ this.upgrade(state, composed.errored > 0 ? "failed" : "diverged");
632
+ state.stepsPinned += composed.executed + composed.recorded;
633
+ // The headline number of this whole design: recovering from a divergence
634
+ // costs nothing, because the proxies were already connected.
635
+ logLine("replay.compose", {
636
+ scenario: state.scenarioId ?? undefined,
637
+ remaining: composed.executed + composed.recorded + composed.skipped,
638
+ executed: composed.executed,
639
+ recorded: composed.recorded,
640
+ skipped: composed.skipped,
641
+ errored: composed.errored || undefined,
642
+ bytes: composed.text.length,
643
+ costUsd: state.fallbackCostUsd.toFixed(2),
644
+ });
645
+ logLine("replay.done", {
646
+ scenario: state.scenarioId ?? undefined,
647
+ mode: state.mode,
648
+ steps: `${composed.executed + composed.recorded}/${plan.stepCount}`,
649
+ outcome: state.outcome,
650
+ ms: Date.now() - state.armedAt,
651
+ });
652
+ if (composed.executed + composed.recorded === 0) {
653
+ // Nothing survived; a bundle of nothing is worse than no bundle.
654
+ return { kind: "abort" };
655
+ }
656
+ // Bash-clean delivery: the model reaches for `Bash` when it cannot call a
657
+ // scripted tool, and a genuine command output is trusted where a `deny`
658
+ // reason is read as adversarial interception. Neutralise the delimiter first.
659
+ if (toolName === "Bash") {
660
+ const delimiter = "BIR_EOF";
661
+ const safe = composed.text.split(delimiter).join("BIR_EOF_");
662
+ logDetail("replay.inject", { channel: "bash", bytes: composed.text.length });
663
+ return { kind: "bash", command: `cat <<'${delimiter}'\n${safe}\n${delimiter}` };
664
+ }
665
+ logDetail("replay.inject", { channel: "deny", bytes: composed.text.length });
666
+ return { kind: "deny", reason: composed.text };
667
+ }
668
+ /**
669
+ * The executor handed to a plan: dispatch a step to the proxy that owns its
670
+ * upstream.
671
+ *
672
+ * Rejecting means "could not be run **here**" — which is the only condition
673
+ * under which a recorded output may stand in. A tool that ran and failed
674
+ * resolves with its failure as the response, exactly as it would in a session.
675
+ */
676
+ executeStep() {
677
+ return async (step, input) => {
678
+ const mcp = parseQualifiedName(step.toolName);
679
+ if (!mcp)
680
+ throw new Error(`${step.toolName} is not an MCP tool — it can only run in the session`);
681
+ const result = await this.work.call(mcp.serverName, mcp.toolName, input, this.budgets.stepMs);
682
+ // Serialize exactly as the proxy records it, or `toolOutputLogic` — which
683
+ // was authored against that shape — silently derives nothing (§7.2).
684
+ return serializeCapped(redact(result));
685
+ };
686
+ }
687
+ /**
688
+ * Where a pinned step's output will arrive from — the proxy's report, or the
689
+ * hook's `tool_response` (§7.2).
690
+ *
691
+ * NOT the same question as {@link reachOf}'s, and conflating them is a silent
692
+ * bug. `reachOf` answers "may *we* execute this step ourselves", which the
693
+ * allowlist restricts. This answers "will a `bir-proxy` see this call and
694
+ * report it", which depends only on whether the server is **wrapped** — a
695
+ * wrapped server the allowlist excludes is still proxied, still reported, and
696
+ * still recorded in the proxy's serialization. Threading such a step from the
697
+ * hook's differently-shaped view would derive nothing at all.
698
+ */
699
+ reachFor(state, step) {
700
+ return reachOf(step.toolName, state.wrapped);
701
+ }
702
+ /**
703
+ * Log a step and remember its verdict, in that order.
704
+ *
705
+ * Every driver passes this — direct, divergence recovery and ad-hoc alike — so
706
+ * there is exactly one place a step's fate is decided, and the console can
707
+ * never be told something the log does not also say.
708
+ */
709
+ observeStep(state) {
710
+ return (info) => {
711
+ this.logStep(info);
712
+ this.recordStep(state, info);
713
+ };
714
+ }
715
+ /** Upsert one step's verdict. Later news about a step replaces earlier news. */
716
+ recordStep(state, info) {
717
+ const status = info.outcome === "executed" ? "ok" : info.outcome;
718
+ if (state.stepResults.size >= MAX_STEP_RESULTS && !state.stepResults.has(info.step.stepIndex)) {
719
+ return;
720
+ }
721
+ const error = info.error
722
+ ? info.error.length > MAX_STEP_ERROR_CHARS
723
+ ? `${info.error.slice(0, MAX_STEP_ERROR_CHARS)}…`
724
+ : info.error
725
+ : undefined;
726
+ state.stepResults.set(info.step.stepIndex, {
727
+ stepIndex: info.step.stepIndex,
728
+ toolName: info.step.toolName,
729
+ status,
730
+ // A stage is only meaningful when something went wrong. `skipped` broke at
731
+ // the call itself — its tool could not run here.
732
+ stage: info.stage ?? (status === "skipped" ? "tool_call" : undefined),
733
+ error,
734
+ durationMs: info.ms,
735
+ });
736
+ }
737
+ /** The step verdicts, in chain order, for the execution report. */
738
+ stepResultsOf(state) {
739
+ if (state.stepResults.size === 0)
740
+ return undefined;
741
+ return [...state.stepResults.values()].sort((a, b) => a.stepIndex - b.stepIndex);
742
+ }
743
+ logStep(info) {
744
+ const mcp = parseQualifiedName(info.step.toolName);
745
+ if (info.outcome === "failed") {
746
+ logLine("replay.step_failed", {
747
+ n: info.step.stepIndex,
748
+ tool: info.step.toolName,
749
+ stage: info.stage,
750
+ why: info.stage === "tool_call"
751
+ ? "its tool ran and reported an error — the step's work did not happen"
752
+ : "the scenario's own logic threw — this step cannot work until it is recalculated",
753
+ error: info.error,
754
+ });
755
+ return;
756
+ }
757
+ if (info.outcome === "skipped") {
758
+ logLine("replay.step_skipped", {
759
+ n: info.step.stepIndex,
760
+ tool: info.step.toolName,
761
+ why: "its tool could not run here and it has no recorded output",
762
+ error: info.error,
763
+ });
764
+ return;
765
+ }
766
+ if (info.outcome === "recorded") {
767
+ logLine("replay.step_recorded", {
768
+ n: info.step.stepIndex,
769
+ tool: info.step.toolName,
770
+ why: "served its recorded output — its tool could not run here",
771
+ error: info.error,
772
+ });
773
+ return;
774
+ }
775
+ logLine("replay.step", {
776
+ n: info.step.stepIndex,
777
+ tool: mcp?.toolName ?? info.step.toolName,
778
+ server: mcp?.serverName,
779
+ ms: info.ms,
780
+ ok: true,
781
+ emitted: info.derivedKeys?.join(",") || undefined,
782
+ });
783
+ }
784
+ /** Outcomes only move forward; `not_steered` is the floor (§11.1). */
785
+ upgrade(state, outcome) {
786
+ if (OUTCOME_RANK[outcome] > OUTCOME_RANK[state.outcome])
787
+ state.outcome = outcome;
788
+ }
789
+ retire(state, outcome) {
790
+ if (outcome)
791
+ this.upgrade(state, outcome);
792
+ state.retired = true;
793
+ state.pinned.clear();
794
+ }
795
+ withBudget(p, ms, what) {
796
+ let timer;
797
+ const budget = new Promise((_, reject) => {
798
+ timer = setTimeout(() => reject(new Error(`${what} exceeded ${ms}ms`)), ms);
799
+ timer.unref?.();
800
+ });
801
+ return Promise.race([p, budget]).finally(() => {
802
+ if (timer)
803
+ clearTimeout(timer);
804
+ });
805
+ }
806
+ }
807
+ //# sourceMappingURL=controller.js.map