@basein/runner 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +276 -0
- package/dist/auth/client.d.ts +85 -0
- package/dist/auth/client.js +284 -0
- package/dist/bin/bir-hooks.d.ts +48 -0
- package/dist/bin/bir-hooks.js +201 -0
- package/dist/bin/bir-proxy.d.ts +45 -0
- package/dist/bin/bir-proxy.js +207 -0
- package/dist/bin/bir-scenario.d.ts +24 -0
- package/dist/bin/bir-scenario.js +177 -0
- package/dist/bin/bir.d.ts +21 -0
- package/dist/bin/bir.js +876 -0
- package/dist/config/adapters/claude-code.d.ts +76 -0
- package/dist/config/adapters/claude-code.js +181 -0
- package/dist/config/adapters/generic.d.ts +17 -0
- package/dist/config/adapters/generic.js +36 -0
- package/dist/config/generate.d.ts +127 -0
- package/dist/config/generate.js +114 -0
- package/dist/config/resolve.d.ts +68 -0
- package/dist/config/resolve.js +132 -0
- package/dist/control/client.d.ts +56 -0
- package/dist/control/client.js +86 -0
- package/dist/control/correlation.d.ts +86 -0
- package/dist/control/correlation.js +0 -0
- package/dist/control/discovery.d.ts +50 -0
- package/dist/control/discovery.js +123 -0
- package/dist/control/ordering.d.ts +38 -0
- package/dist/control/ordering.js +44 -0
- package/dist/control/paths.d.ts +32 -0
- package/dist/control/paths.js +56 -0
- package/dist/control/server.d.ts +272 -0
- package/dist/control/server.js +1131 -0
- package/dist/control/transcript.d.ts +75 -0
- package/dist/control/transcript.js +241 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +32 -0
- package/dist/jsonrpc/framing.d.ts +49 -0
- package/dist/jsonrpc/framing.js +143 -0
- package/dist/jsonrpc/types.d.ts +52 -0
- package/dist/jsonrpc/types.js +46 -0
- package/dist/proxy/intercept.d.ts +55 -0
- package/dist/proxy/intercept.js +147 -0
- package/dist/proxy/relay.d.ts +97 -0
- package/dist/proxy/relay.js +166 -0
- package/dist/proxy/session.d.ts +116 -0
- package/dist/proxy/session.js +319 -0
- package/dist/record/housekeeping.d.ts +34 -0
- package/dist/record/housekeeping.js +39 -0
- package/dist/record/queue.d.ts +48 -0
- package/dist/record/queue.js +96 -0
- package/dist/record/recorder.d.ts +111 -0
- package/dist/record/recorder.js +39 -0
- package/dist/record/redact.d.ts +37 -0
- package/dist/record/redact.js +119 -0
- package/dist/record/remote-recorder.d.ts +110 -0
- package/dist/record/remote-recorder.js +301 -0
- package/dist/record/truncate.d.ts +36 -0
- package/dist/record/truncate.js +85 -0
- package/dist/replay/bundle.d.ts +36 -0
- package/dist/replay/bundle.js +89 -0
- package/dist/replay/controller.d.ts +300 -0
- package/dist/replay/controller.js +807 -0
- package/dist/replay/coverage.d.ts +41 -0
- package/dist/replay/coverage.js +56 -0
- package/dist/replay/derive.d.ts +58 -0
- package/dist/replay/derive.js +166 -0
- package/dist/replay/executor.d.ts +78 -0
- package/dist/replay/executor.js +233 -0
- package/dist/replay/logic.d.ts +31 -0
- package/dist/replay/logic.js +50 -0
- package/dist/replay/plan.d.ts +181 -0
- package/dist/replay/plan.js +397 -0
- package/dist/replay/pricing.d.ts +41 -0
- package/dist/replay/pricing.js +76 -0
- package/dist/replay/source-run.d.ts +50 -0
- package/dist/replay/source-run.js +98 -0
- package/dist/replay/tool-error.d.ts +22 -0
- package/dist/replay/tool-error.js +60 -0
- package/dist/replay/types.d.ts +116 -0
- package/dist/replay/types.js +35 -0
- package/dist/upstream/client.d.ts +78 -0
- package/dist/upstream/client.js +114 -0
- package/dist/upstream/http-client.d.ts +78 -0
- package/dist/upstream/http-client.js +261 -0
- package/dist/upstream/lazy-client.d.ts +31 -0
- package/dist/upstream/lazy-client.js +53 -0
- package/dist/upstream/stdio-client.d.ts +57 -0
- package/dist/upstream/stdio-client.js +203 -0
- package/dist/util/log.d.ts +27 -0
- package/dist/util/log.js +51 -0
- package/dist/util/version.d.ts +2 -0
- package/dist/util/version.js +40 -0
- package/docs/BaseInstRunner.md +621 -0
- package/docs/calculatedReplay.md +1185 -0
- package/docs/calculatedReplayGuide.md +448 -0
- package/docs/installRun.md +413 -0
- package/docs/mcpmark.md +752 -0
- package/docs/quickstart.md +201 -0
- package/docs/t-bench.md +394 -0
- package/package.json +56 -0
|
@@ -0,0 +1,807 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* controller — everything replay decides, in one place (docs/calculatedReplay.md).
|
|
3
|
+
*
|
|
4
|
+
* The control server owns runs, ordering and recording. This owns the *plan*:
|
|
5
|
+
* the gate ladder, the mode choice, arming, pinning, threading, direct
|
|
6
|
+
* execution, divergence recovery and the execution report. `ControlServer`
|
|
7
|
+
* delegates to it and stores one {@link ReplayState} per run.
|
|
8
|
+
*
|
|
9
|
+
* THE GOVERNING RULE IS UNCHANGED and outranks every optimisation here: the host
|
|
10
|
+
* session must never fail because of BaseInstRunner. Every method below either
|
|
11
|
+
* degrades to "run the turn normally" or returns something the session can
|
|
12
|
+
* ignore. Nothing throws at a hook.
|
|
13
|
+
*/
|
|
14
|
+
import { parseQualifiedName } from "../control/correlation.js";
|
|
15
|
+
import { isHousekeeping } from "../record/housekeeping.js";
|
|
16
|
+
import { redact } from "../record/redact.js";
|
|
17
|
+
import { serializeCapped } from "../record/truncate.js";
|
|
18
|
+
import { logDetail, logLine, errText } from "../util/log.js";
|
|
19
|
+
import { MAX_REPLAY_REASON } from "./bundle.js";
|
|
20
|
+
import { coverageOf, modeFor, reachOf } from "./coverage.js";
|
|
21
|
+
import { deriveParameters } from "./derive.js";
|
|
22
|
+
import { ProxyWorkQueue } from "./executor.js";
|
|
23
|
+
import { ScenarioReplayPlan } from "./plan.js";
|
|
24
|
+
import { PRICING_VERSION } from "./pricing.js";
|
|
25
|
+
import { SourceRunOutputs } from "./source-run.js";
|
|
26
|
+
import { toolResultError } from "./tool-error.js";
|
|
27
|
+
import { OUTCOME_RANK, isReadyScenario, } from "./types.js";
|
|
28
|
+
/** The first-party tool a `direct` plan is delivered through (§6.3). */
|
|
29
|
+
export const DIRECT_TOOL_NAME = "mcp__bir__run_scenario";
|
|
30
|
+
/** How long a `/proxy/poll` is held open before it answers empty. */
|
|
31
|
+
export const POLL_HOLD_MS = 25_000;
|
|
32
|
+
export const DEFAULT_BUDGETS = {
|
|
33
|
+
matchMs: 2_500,
|
|
34
|
+
deriveMs: 8_000,
|
|
35
|
+
planMs: 120_000,
|
|
36
|
+
stepMs: 60_000,
|
|
37
|
+
};
|
|
38
|
+
/** How many step verdicts one report carries. A chain longer than this is not
|
|
39
|
+
* a chain anybody is reading step by step, and the report has to stay small. */
|
|
40
|
+
const MAX_STEP_RESULTS = 200;
|
|
41
|
+
/** Per-step error budget. The console shows these inline, under the step. */
|
|
42
|
+
const MAX_STEP_ERROR_CHARS = 500;
|
|
43
|
+
export class ReplayController {
|
|
44
|
+
enabled;
|
|
45
|
+
work = new ProxyWorkQueue();
|
|
46
|
+
budgets;
|
|
47
|
+
opts;
|
|
48
|
+
constructor(opts) {
|
|
49
|
+
this.opts = opts;
|
|
50
|
+
this.enabled = opts.enabled;
|
|
51
|
+
this.budgets = { ...DEFAULT_BUDGETS, ...(opts.budgets ?? {}) };
|
|
52
|
+
}
|
|
53
|
+
/** The plan's own delivery vehicle is never a scenario step. */
|
|
54
|
+
isDirectTool(toolName) {
|
|
55
|
+
return toolName === DIRECT_TOOL_NAME;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Await a match under {@link ReplayBudgets.matchMs}.
|
|
59
|
+
*
|
|
60
|
+
* Budget expiry is not an error and never blocks the user: it means this turn
|
|
61
|
+
* runs normally, and the one `replay.decision` line says so. Resolving `null`
|
|
62
|
+
* covers both "no match" and "too slow", which are the same thing from here.
|
|
63
|
+
*/
|
|
64
|
+
async awaitMatch(pending) {
|
|
65
|
+
let timer;
|
|
66
|
+
const budget = new Promise((resolve) => {
|
|
67
|
+
timer = setTimeout(() => resolve(null), this.budgets.matchMs);
|
|
68
|
+
timer.unref?.();
|
|
69
|
+
});
|
|
70
|
+
try {
|
|
71
|
+
const match = (await Promise.race([pending, budget]));
|
|
72
|
+
return match ?? null;
|
|
73
|
+
}
|
|
74
|
+
catch {
|
|
75
|
+
return null;
|
|
76
|
+
}
|
|
77
|
+
finally {
|
|
78
|
+
if (timer)
|
|
79
|
+
clearTimeout(timer);
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* Run the gate ladder and, if every gate passes, arm a plan.
|
|
84
|
+
*
|
|
85
|
+
* Always returns a state — a declined match still has a ticket to redeem and a
|
|
86
|
+
* baseline sample to contribute, and losing that is how a savings ledger ends
|
|
87
|
+
* up with a denominator nobody measured.
|
|
88
|
+
*/
|
|
89
|
+
arm(match, prompt, wrapped) {
|
|
90
|
+
const state = {
|
|
91
|
+
scenarioId: match.scenarioId,
|
|
92
|
+
ticket: match.executionTicket,
|
|
93
|
+
similarity: match.similarity,
|
|
94
|
+
matchedRunId: match.runId,
|
|
95
|
+
mode: "none",
|
|
96
|
+
wrapped,
|
|
97
|
+
pinned: new Map(),
|
|
98
|
+
outcome: "not_steered",
|
|
99
|
+
deriveCostUsd: 0,
|
|
100
|
+
fallbackCostUsd: 0,
|
|
101
|
+
stepsPlanned: 0,
|
|
102
|
+
stepsPinned: 0,
|
|
103
|
+
stepResults: new Map(),
|
|
104
|
+
armedAt: Date.now(),
|
|
105
|
+
reported: false,
|
|
106
|
+
retired: false,
|
|
107
|
+
};
|
|
108
|
+
const decline = (why) => {
|
|
109
|
+
logLine("replay.decision", {
|
|
110
|
+
verdict: "no-steer",
|
|
111
|
+
run: match.runId,
|
|
112
|
+
scenario: match.scenarioId ?? undefined,
|
|
113
|
+
similarity: match.similarity.toFixed(3),
|
|
114
|
+
threshold: this.opts.minSimilarity,
|
|
115
|
+
why,
|
|
116
|
+
});
|
|
117
|
+
return state;
|
|
118
|
+
};
|
|
119
|
+
// Gate 2 — replay enabled.
|
|
120
|
+
if (!this.enabled)
|
|
121
|
+
return decline("BIR_REPLAY is not set");
|
|
122
|
+
// Gate 3 — a ready scenario with steps.
|
|
123
|
+
const scenario = match.scenario;
|
|
124
|
+
if (!match.scenarioId || !isReadyScenario(scenario)) {
|
|
125
|
+
return decline(match.scenarioId ? "scenario is not ready" : "no scenario for the matched run");
|
|
126
|
+
}
|
|
127
|
+
// Gate 4 — similarity. Detection decides "don't record this again"; steering
|
|
128
|
+
// decides "don't think about this again", which is a much stronger claim and
|
|
129
|
+
// deserves a stronger threshold.
|
|
130
|
+
if (match.similarity < this.opts.minSimilarity) {
|
|
131
|
+
return decline("similarity below threshold");
|
|
132
|
+
}
|
|
133
|
+
// Gate 5 — tool coverage.
|
|
134
|
+
const mode = modeFor(scenario.steps, wrapped, this.opts.allowServers);
|
|
135
|
+
if (mode === "none")
|
|
136
|
+
return decline("no step is executable");
|
|
137
|
+
const params = this.startDerivation(scenario, prompt, state);
|
|
138
|
+
state.plan = new ScenarioReplayPlan({ scenario, params, mode });
|
|
139
|
+
state.mode = mode;
|
|
140
|
+
state.stepsPlanned = scenario.steps.length;
|
|
141
|
+
if (this.opts.authUrl && this.opts.authToken) {
|
|
142
|
+
state.sourceRun = new SourceRunOutputs({
|
|
143
|
+
baseUrl: this.opts.authUrl,
|
|
144
|
+
runId: scenario.runId,
|
|
145
|
+
token: this.opts.authToken,
|
|
146
|
+
fetchImpl: this.opts.fetchImpl,
|
|
147
|
+
});
|
|
148
|
+
}
|
|
149
|
+
// The audit line precedes the action, always (§13.2, mitigation 4).
|
|
150
|
+
logLine("plan.armed", {
|
|
151
|
+
run: match.runId,
|
|
152
|
+
scenario: scenario.id,
|
|
153
|
+
mode,
|
|
154
|
+
steps: scenario.steps.length,
|
|
155
|
+
similarity: match.similarity.toFixed(3),
|
|
156
|
+
coverage: coverageOf(scenario.steps, wrapped, this.opts.allowServers).join(","),
|
|
157
|
+
tools: scenario.steps.map((s) => s.toolName).join(","),
|
|
158
|
+
});
|
|
159
|
+
return state;
|
|
160
|
+
}
|
|
161
|
+
/** The directive to inject via `additionalContext`, or undefined when declined. */
|
|
162
|
+
directiveFor(state) {
|
|
163
|
+
return state.plan?.steeringDirective(DIRECT_TOOL_NAME);
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* `PreToolUse`, while a plan is active.
|
|
167
|
+
*
|
|
168
|
+
* Three tiers (§8): pin the expected call; on divergence execute the remainder
|
|
169
|
+
* and hand it back through the best channel available; if even that fails,
|
|
170
|
+
* abort to an ordinary turn.
|
|
171
|
+
*/
|
|
172
|
+
async preTool(state, toolName, toolUseId) {
|
|
173
|
+
const plan = state.plan;
|
|
174
|
+
if (!plan || state.retired)
|
|
175
|
+
return { kind: "passthrough" };
|
|
176
|
+
// The delivery vehicle for a direct plan is not a step in it.
|
|
177
|
+
if (this.isDirectTool(toolName))
|
|
178
|
+
return { kind: "passthrough" };
|
|
179
|
+
try {
|
|
180
|
+
await this.withBudget(plan.ready(), this.budgets.deriveMs, "derivation");
|
|
181
|
+
}
|
|
182
|
+
catch (err) {
|
|
183
|
+
logLine("replay.derive_failed", {
|
|
184
|
+
scenario: state.scenarioId ?? undefined,
|
|
185
|
+
why: "continuing with the scenario's recorded sample values",
|
|
186
|
+
error: errText(err),
|
|
187
|
+
});
|
|
188
|
+
}
|
|
189
|
+
// A direct plan does not steer individual calls: the model was asked for one
|
|
190
|
+
// tool call and made a different one. That is divergence.
|
|
191
|
+
const expected = plan.expectedTool();
|
|
192
|
+
if (state.mode === "steer" && toolName === expected && toolUseId) {
|
|
193
|
+
try {
|
|
194
|
+
const step = plan.currentStep();
|
|
195
|
+
const input = plan.toolInputForCurrentStep();
|
|
196
|
+
state.pinned.set(toolUseId, {
|
|
197
|
+
stepIndex: plan.currentStepIndex,
|
|
198
|
+
reach: this.reachFor(state, step),
|
|
199
|
+
toolName,
|
|
200
|
+
pinnedAt: Date.now(),
|
|
201
|
+
});
|
|
202
|
+
logDetail("replay.pin", { n: plan.currentStepIndex, tool: toolName });
|
|
203
|
+
return { kind: "pin", input, stepIndex: plan.currentStepIndex };
|
|
204
|
+
}
|
|
205
|
+
catch (err) {
|
|
206
|
+
// `toolInputLogic` threw: the pin is impossible, but the scenario may
|
|
207
|
+
// still be recoverable by executing the rest.
|
|
208
|
+
//
|
|
209
|
+
// Recorded before diverging, because `composeBundle` starts at this same
|
|
210
|
+
// step and will throw on the same logic — and its own report would then
|
|
211
|
+
// be the only trace, attributed to a re-execution rather than to the
|
|
212
|
+
// steered call that actually hit it first.
|
|
213
|
+
const step = plan.currentStep();
|
|
214
|
+
if (step) {
|
|
215
|
+
this.recordStep(state, {
|
|
216
|
+
step,
|
|
217
|
+
outcome: "failed",
|
|
218
|
+
stage: "tool_input_logic",
|
|
219
|
+
error: errText(err),
|
|
220
|
+
});
|
|
221
|
+
}
|
|
222
|
+
return await this.diverge(state, toolName, `toolInputLogic threw: ${errText(err)}`);
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
// HOUSEKEEPING IS NOT DIVERGENCE. See `record/housekeeping.ts`.
|
|
226
|
+
if (isHousekeeping(toolName)) {
|
|
227
|
+
logDetail("replay.housekeeping", {
|
|
228
|
+
tool: toolName,
|
|
229
|
+
step: `${plan.currentStepIndex}/${plan.stepCount}`,
|
|
230
|
+
why: "host bookkeeping, not task work — plan stays armed",
|
|
231
|
+
});
|
|
232
|
+
return { kind: "passthrough" };
|
|
233
|
+
}
|
|
234
|
+
return await this.diverge(state, toolName, `expected ${expected ?? "no more tools"}, model called ${toolName}`);
|
|
235
|
+
}
|
|
236
|
+
/**
|
|
237
|
+
* `PostToolUse` for a call this plan pinned.
|
|
238
|
+
*
|
|
239
|
+
* `output` must be serialized the way the step's output was **recorded**: for a
|
|
240
|
+
* wrapped MCP step that is the proxy's whole `CallToolResult`, which is why the
|
|
241
|
+
* server threads those from `/proxy/step` rather than from the hook (§7.2).
|
|
242
|
+
* Returns true when the plan is now complete.
|
|
243
|
+
*/
|
|
244
|
+
postTool(state, toolUseId, output) {
|
|
245
|
+
const plan = state.plan;
|
|
246
|
+
const pin = state.pinned.get(toolUseId);
|
|
247
|
+
if (!plan || !pin || state.retired)
|
|
248
|
+
return false;
|
|
249
|
+
state.pinned.delete(toolUseId);
|
|
250
|
+
// The pin's own step, not `currentStep()`: `applyOutput` moves the cursor, so
|
|
251
|
+
// reading it after the call would judge the *next* step, and reading it
|
|
252
|
+
// before assumes no other pin ever moved it.
|
|
253
|
+
const step = plan.allSteps()[pin.stepIndex];
|
|
254
|
+
let derivedKeys = [];
|
|
255
|
+
try {
|
|
256
|
+
derivedKeys = plan.applyOutput(output);
|
|
257
|
+
}
|
|
258
|
+
catch (err) {
|
|
259
|
+
logLine("replay.thread_failed", {
|
|
260
|
+
scenario: state.scenarioId ?? undefined,
|
|
261
|
+
step: pin.stepIndex,
|
|
262
|
+
why: "toolOutputLogic threw — continuing as a normal run",
|
|
263
|
+
error: errText(err),
|
|
264
|
+
});
|
|
265
|
+
if (step) {
|
|
266
|
+
this.recordStep(state, {
|
|
267
|
+
step,
|
|
268
|
+
outcome: "failed",
|
|
269
|
+
stage: "tool_output_logic",
|
|
270
|
+
error: errText(err),
|
|
271
|
+
ms: Date.now() - pin.pinnedAt,
|
|
272
|
+
});
|
|
273
|
+
}
|
|
274
|
+
this.retire(state, "failed");
|
|
275
|
+
return false;
|
|
276
|
+
}
|
|
277
|
+
// One more step ran under the plan. Counted here, as each pin threads, and
|
|
278
|
+
// not from `pinned.size` at pin time: a sequential chain never has more than
|
|
279
|
+
// one pin outstanding, so that size reported 1 whatever the chain's length.
|
|
280
|
+
state.stepsPinned += 1;
|
|
281
|
+
// The step ran in the live session and threaded. This is the only place a
|
|
282
|
+
// *steered* step is judged: `plan` never sees the call, the host did. A tool
|
|
283
|
+
// that ran and reported an error did not do the step's work, so the plan is
|
|
284
|
+
// retired as `failed` — a baseline sample, never a saving.
|
|
285
|
+
const toolError = step ? toolResultError(output) : undefined;
|
|
286
|
+
if (step && toolError) {
|
|
287
|
+
const info = {
|
|
288
|
+
step,
|
|
289
|
+
outcome: "failed",
|
|
290
|
+
stage: "tool_call",
|
|
291
|
+
error: toolError,
|
|
292
|
+
ms: Date.now() - pin.pinnedAt,
|
|
293
|
+
};
|
|
294
|
+
this.logStep(info);
|
|
295
|
+
this.recordStep(state, info);
|
|
296
|
+
this.retire(state, "failed");
|
|
297
|
+
return false;
|
|
298
|
+
}
|
|
299
|
+
if (step) {
|
|
300
|
+
this.recordStep(state, {
|
|
301
|
+
step,
|
|
302
|
+
outcome: "executed",
|
|
303
|
+
ms: Date.now() - pin.pinnedAt,
|
|
304
|
+
});
|
|
305
|
+
}
|
|
306
|
+
const done = plan.isDone();
|
|
307
|
+
logDetail("replay.thread", {
|
|
308
|
+
n: pin.stepIndex,
|
|
309
|
+
via: pin.reach === "direct" ? "proxy" : "hook",
|
|
310
|
+
emitted: derivedKeys.join(",") || undefined,
|
|
311
|
+
done,
|
|
312
|
+
});
|
|
313
|
+
if (done) {
|
|
314
|
+
this.upgrade(state, "steered_full");
|
|
315
|
+
this.retire(state, undefined);
|
|
316
|
+
logLine("replay.done", {
|
|
317
|
+
scenario: state.scenarioId ?? undefined,
|
|
318
|
+
mode: state.mode,
|
|
319
|
+
steps: `${plan.stepCount}/${plan.stepCount}`,
|
|
320
|
+
outcome: state.outcome,
|
|
321
|
+
ms: Date.now() - state.armedAt,
|
|
322
|
+
});
|
|
323
|
+
}
|
|
324
|
+
return done;
|
|
325
|
+
}
|
|
326
|
+
/** Whether a pinned call's output is threaded from the proxy's report (§7.2). */
|
|
327
|
+
threadsFromProxy(state, toolUseId) {
|
|
328
|
+
return state.pinned.get(toolUseId)?.reach === "direct";
|
|
329
|
+
}
|
|
330
|
+
/** True when this call belongs to the active plan. */
|
|
331
|
+
isPinned(state, toolUseId) {
|
|
332
|
+
return state.pinned.has(toolUseId);
|
|
333
|
+
}
|
|
334
|
+
/**
|
|
335
|
+
* `POST /scenario/run` — the whole of a `direct` plan, executed here.
|
|
336
|
+
*
|
|
337
|
+
* Every step runs through the proxy that already owns its upstream, so the
|
|
338
|
+
* model spends nothing beyond the turn that reads the results.
|
|
339
|
+
*/
|
|
340
|
+
async runArmed(state) {
|
|
341
|
+
const plan = state.plan;
|
|
342
|
+
if (!plan || state.retired)
|
|
343
|
+
return { ok: false, why: "no_plan" };
|
|
344
|
+
try {
|
|
345
|
+
await this.withBudget(plan.ready(), this.budgets.deriveMs, "derivation");
|
|
346
|
+
}
|
|
347
|
+
catch (err) {
|
|
348
|
+
logLine("replay.derive_failed", {
|
|
349
|
+
scenario: state.scenarioId ?? undefined,
|
|
350
|
+
why: "continuing with the scenario's recorded sample values",
|
|
351
|
+
error: errText(err),
|
|
352
|
+
});
|
|
353
|
+
}
|
|
354
|
+
const deadline = Date.now() + this.budgets.planMs;
|
|
355
|
+
let result;
|
|
356
|
+
try {
|
|
357
|
+
result = await plan.runToCompletion(this.executeStep(), state.sourceRun ? (step) => state.sourceRun.outputFor(step) : undefined, this.observeStep(state), deadline);
|
|
358
|
+
}
|
|
359
|
+
catch (err) {
|
|
360
|
+
// The scenario's own logic failed. Retire and let the model do the work.
|
|
361
|
+
logLine("replay.failed", {
|
|
362
|
+
scenario: state.scenarioId ?? undefined,
|
|
363
|
+
why: "scenario logic threw — continuing as a normal run",
|
|
364
|
+
error: errText(err),
|
|
365
|
+
});
|
|
366
|
+
this.retire(state, "failed");
|
|
367
|
+
return { ok: false, why: "not_ready" };
|
|
368
|
+
}
|
|
369
|
+
state.stepsPinned += result.executed + result.recorded;
|
|
370
|
+
let responseModel = {};
|
|
371
|
+
try {
|
|
372
|
+
responseModel = plan.responseModel();
|
|
373
|
+
}
|
|
374
|
+
catch (err) {
|
|
375
|
+
logDetail("replay.response_model_failed", { error: errText(err) });
|
|
376
|
+
}
|
|
377
|
+
// A plan whose tools errored ran to the end and did nothing. `steered_full`
|
|
378
|
+
// would book the largest saving available for it; `failed` books a baseline
|
|
379
|
+
// sample, which is what a replay that produced nothing is worth (mcpmark.md §13).
|
|
380
|
+
this.upgrade(state, result.errored > 0
|
|
381
|
+
? "failed"
|
|
382
|
+
: result.partial || result.skipped > 0
|
|
383
|
+
? "diverged"
|
|
384
|
+
: "steered_full");
|
|
385
|
+
this.retire(state, undefined);
|
|
386
|
+
logLine("replay.done", {
|
|
387
|
+
scenario: state.scenarioId ?? undefined,
|
|
388
|
+
mode: state.mode,
|
|
389
|
+
steps: `${result.executed + result.recorded}/${plan.stepCount}`,
|
|
390
|
+
recorded: result.recorded || undefined,
|
|
391
|
+
skipped: result.skipped || undefined,
|
|
392
|
+
errored: result.errored || undefined,
|
|
393
|
+
outcome: state.outcome,
|
|
394
|
+
partial: result.partial || undefined,
|
|
395
|
+
ms: Date.now() - state.armedAt,
|
|
396
|
+
});
|
|
397
|
+
return {
|
|
398
|
+
ok: true,
|
|
399
|
+
text: result.text,
|
|
400
|
+
responseModel,
|
|
401
|
+
steps: result.executed + result.recorded,
|
|
402
|
+
partial: result.partial,
|
|
403
|
+
};
|
|
404
|
+
}
|
|
405
|
+
/**
|
|
406
|
+
* Build the execution report for a sealed run, or undefined when there is
|
|
407
|
+
* nothing to report.
|
|
408
|
+
*
|
|
409
|
+
* `costUsd <= 0` on a decline is not worth sending: the service answers
|
|
410
|
+
* `202 { recorded: false }` for a baseline sample with no measured cost, and a
|
|
411
|
+
* report that books nothing is noise on both sides.
|
|
412
|
+
*/
|
|
413
|
+
buildReport(state, d) {
|
|
414
|
+
if (state.reported || !state.scenarioId)
|
|
415
|
+
return undefined;
|
|
416
|
+
state.reported = true;
|
|
417
|
+
const total = state.deriveCostUsd + d.sessionCostUsd + state.fallbackCostUsd;
|
|
418
|
+
const steps = this.stepResultsOf(state);
|
|
419
|
+
const isBaseline = state.outcome === "not_steered" || state.outcome === "failed";
|
|
420
|
+
// A costless decline is noise on both sides — *unless* a step actually broke,
|
|
421
|
+
// which is the one thing the recording page cannot learn any other way. The
|
|
422
|
+
// service accepts a costless report that says why (its errorshandling.md).
|
|
423
|
+
if (isBaseline && total <= 0 && !steps?.some((s) => s.status === "failed")) {
|
|
424
|
+
return undefined;
|
|
425
|
+
}
|
|
426
|
+
const blame = this.blameStep(steps);
|
|
427
|
+
return {
|
|
428
|
+
scenarioId: state.scenarioId,
|
|
429
|
+
ticket: state.ticket,
|
|
430
|
+
outcome: state.outcome,
|
|
431
|
+
deriveCostUsd: state.deriveCostUsd,
|
|
432
|
+
sessionCostUsd: d.sessionCostUsd,
|
|
433
|
+
fallbackCostUsd: state.fallbackCostUsd,
|
|
434
|
+
durationMs: d.durationMs,
|
|
435
|
+
stepsPlanned: state.stepsPlanned,
|
|
436
|
+
stepsPinned: state.stepsPinned,
|
|
437
|
+
measured: d.measured,
|
|
438
|
+
pricingVersion: PRICING_VERSION,
|
|
439
|
+
prompt: d.prompt,
|
|
440
|
+
steps,
|
|
441
|
+
// The headline, lifted from the first step that broke. The console leads
|
|
442
|
+
// with this and shows the per-step verdicts underneath, so the two must
|
|
443
|
+
// name the same failure rather than being assembled independently.
|
|
444
|
+
error: blame?.error,
|
|
445
|
+
errorStage: blame?.stage,
|
|
446
|
+
errorStepIndex: blame?.stepIndex,
|
|
447
|
+
errorToolName: blame?.toolName,
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
/**
|
|
451
|
+
* The step a failed replay is fairly blamed on: the first whose own logic
|
|
452
|
+
* threw, or failing that the first that could not run at all.
|
|
453
|
+
*
|
|
454
|
+
* A `failed` step outranks a `skipped` one however early the skip came — a
|
|
455
|
+
* scenario whose logic throws is broken for everybody, while a step that could
|
|
456
|
+
* not run here says something about this machine.
|
|
457
|
+
*/
|
|
458
|
+
blameStep(steps) {
|
|
459
|
+
if (!steps)
|
|
460
|
+
return undefined;
|
|
461
|
+
return (steps.find((s) => s.status === "failed") ?? steps.find((s) => s.status === "skipped"));
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* Run a scenario that no match armed — `bir replay` (§12.1).
|
|
465
|
+
*
|
|
466
|
+
* The same plan, the same executor, the same live proxies; only the trigger
|
|
467
|
+
* differs. It is how you test a scenario without a session, how a Tier 2 or
|
|
468
|
+
* non-Claude-Code client gets any replay at all, and the first thing to reach
|
|
469
|
+
* for when a steered turn behaves oddly.
|
|
470
|
+
*
|
|
471
|
+
* Deliberately outside the gate ladder: the operator typed the scenario id, so
|
|
472
|
+
* there is nothing to be similar *to* and nothing to decline. It books no
|
|
473
|
+
* execution either — there is no ticket, and inventing a saving for a manual
|
|
474
|
+
* invocation is exactly the kind of number a ledger must never contain.
|
|
475
|
+
*/
|
|
476
|
+
async runAdHoc(scenario, prompt, wrapped) {
|
|
477
|
+
if (!isReadyScenario(scenario))
|
|
478
|
+
return { ok: false, why: "scenario is not ready" };
|
|
479
|
+
const mode = modeFor(scenario.steps, wrapped, this.opts.allowServers);
|
|
480
|
+
if (mode === "none")
|
|
481
|
+
return { ok: false, why: "no step is executable" };
|
|
482
|
+
const state = {
|
|
483
|
+
scenarioId: scenario.id,
|
|
484
|
+
similarity: 1,
|
|
485
|
+
matchedRunId: scenario.runId,
|
|
486
|
+
mode,
|
|
487
|
+
wrapped,
|
|
488
|
+
pinned: new Map(),
|
|
489
|
+
outcome: "not_steered",
|
|
490
|
+
deriveCostUsd: 0,
|
|
491
|
+
fallbackCostUsd: 0,
|
|
492
|
+
stepsPlanned: scenario.steps.length,
|
|
493
|
+
stepsPinned: 0,
|
|
494
|
+
stepResults: new Map(),
|
|
495
|
+
armedAt: Date.now(),
|
|
496
|
+
// No ticket, so nothing to redeem — and `buildReport` is never called for
|
|
497
|
+
// an ad-hoc run anyway.
|
|
498
|
+
reported: true,
|
|
499
|
+
retired: false,
|
|
500
|
+
};
|
|
501
|
+
const params = this.startDerivation(scenario, prompt, state);
|
|
502
|
+
const plan = new ScenarioReplayPlan({ scenario, params, mode });
|
|
503
|
+
state.plan = plan;
|
|
504
|
+
if (this.opts.authUrl && this.opts.authToken) {
|
|
505
|
+
state.sourceRun = new SourceRunOutputs({
|
|
506
|
+
baseUrl: this.opts.authUrl,
|
|
507
|
+
runId: scenario.runId,
|
|
508
|
+
token: this.opts.authToken,
|
|
509
|
+
fetchImpl: this.opts.fetchImpl,
|
|
510
|
+
});
|
|
511
|
+
}
|
|
512
|
+
logLine("plan.armed", {
|
|
513
|
+
scenario: scenario.id,
|
|
514
|
+
mode,
|
|
515
|
+
steps: scenario.steps.length,
|
|
516
|
+
coverage: coverageOf(scenario.steps, wrapped, this.opts.allowServers).join(","),
|
|
517
|
+
why: "bir replay — no match, an operator asked for it",
|
|
518
|
+
});
|
|
519
|
+
try {
|
|
520
|
+
await this.withBudget(plan.ready(), this.budgets.deriveMs, "derivation");
|
|
521
|
+
}
|
|
522
|
+
catch (err) {
|
|
523
|
+
logLine("replay.derive_failed", { scenario: scenario.id, error: errText(err) });
|
|
524
|
+
}
|
|
525
|
+
const trace = [];
|
|
526
|
+
try {
|
|
527
|
+
const result = await plan.runToCompletion(this.executeStep(), state.sourceRun ? (step) => state.sourceRun.outputFor(step) : undefined, (info) => {
|
|
528
|
+
this.observeStep(state)(info);
|
|
529
|
+
trace.push({
|
|
530
|
+
step: info.step.stepIndex,
|
|
531
|
+
tool: info.step.toolName,
|
|
532
|
+
input: info.input,
|
|
533
|
+
outcome: info.outcome,
|
|
534
|
+
ms: info.ms,
|
|
535
|
+
});
|
|
536
|
+
}, Date.now() + this.budgets.planMs);
|
|
537
|
+
let responseModel = {};
|
|
538
|
+
try {
|
|
539
|
+
responseModel = plan.responseModel();
|
|
540
|
+
}
|
|
541
|
+
catch (err) {
|
|
542
|
+
logDetail("replay.response_model_failed", { error: errText(err) });
|
|
543
|
+
}
|
|
544
|
+
return {
|
|
545
|
+
ok: true,
|
|
546
|
+
text: result.text,
|
|
547
|
+
responseModel,
|
|
548
|
+
steps: result.executed + result.recorded,
|
|
549
|
+
partial: result.partial,
|
|
550
|
+
trace,
|
|
551
|
+
};
|
|
552
|
+
}
|
|
553
|
+
catch (err) {
|
|
554
|
+
return { ok: false, why: errText(err), trace };
|
|
555
|
+
}
|
|
556
|
+
}
|
|
557
|
+
/** Release every parked poller. Called at control-server shutdown. */
|
|
558
|
+
close() {
|
|
559
|
+
this.work.close();
|
|
560
|
+
}
|
|
561
|
+
// ── internals ────────────────────────────────────────────────────────────
|
|
562
|
+
/**
|
|
563
|
+
* Kick off derivation without awaiting it (§10). The prompt hook returns the
|
|
564
|
+
* directive the moment the match lands; the first `PreToolUse` — or
|
|
565
|
+
* `/scenario/run`, which has no hook timeout at all — is where the wait lands.
|
|
566
|
+
*/
|
|
567
|
+
startDerivation(scenario, prompt, state) {
|
|
568
|
+
const derive = this.opts.deriveImpl ?? deriveParameters;
|
|
569
|
+
return derive({
|
|
570
|
+
prompt,
|
|
571
|
+
intent: scenario.intent ?? "",
|
|
572
|
+
paramsObject: scenario.paramsObject,
|
|
573
|
+
apiKey: this.opts.apiKey ?? process.env.ANTHROPIC_API_KEY,
|
|
574
|
+
fetchImpl: this.opts.fetchImpl,
|
|
575
|
+
})
|
|
576
|
+
.then((r) => {
|
|
577
|
+
state.deriveCostUsd += r.costUsd;
|
|
578
|
+
logLine("replay.derived", {
|
|
579
|
+
scenario: scenario.id,
|
|
580
|
+
params: Object.keys(r.params).length,
|
|
581
|
+
costUsd: r.costUsd.toFixed(4),
|
|
582
|
+
source: r.derived ? "prompt" : "recorded samples",
|
|
583
|
+
});
|
|
584
|
+
return r.params;
|
|
585
|
+
})
|
|
586
|
+
.catch((err) => {
|
|
587
|
+
logLine("replay.derive_failed", {
|
|
588
|
+
scenario: scenario.id,
|
|
589
|
+
why: "falling back to the scenario's recorded sample values",
|
|
590
|
+
error: errText(err),
|
|
591
|
+
});
|
|
592
|
+
const fallback = {};
|
|
593
|
+
for (const [key, entry] of Object.entries(scenario.paramsObject ?? {})) {
|
|
594
|
+
fallback[key] = entry.sampleValue;
|
|
595
|
+
}
|
|
596
|
+
return fallback;
|
|
597
|
+
});
|
|
598
|
+
}
|
|
599
|
+
/**
|
|
600
|
+
* Divergence (§8). Execute the remaining steps for real, then deliver.
|
|
601
|
+
*
|
|
602
|
+
* The plan is retired first: whatever happens next, this turn neither steers
|
|
603
|
+
* nor re-injects again.
|
|
604
|
+
*/
|
|
605
|
+
async diverge(state, toolName, why) {
|
|
606
|
+
const plan = state.plan;
|
|
607
|
+
if (!plan)
|
|
608
|
+
return { kind: "passthrough" };
|
|
609
|
+
logLine("replay.diverge", {
|
|
610
|
+
scenario: state.scenarioId ?? undefined,
|
|
611
|
+
expected: plan.expectedTool() ?? "(none)",
|
|
612
|
+
called: toolName,
|
|
613
|
+
step: `${plan.currentStepIndex}/${plan.stepCount}`,
|
|
614
|
+
why,
|
|
615
|
+
});
|
|
616
|
+
this.retire(state, undefined);
|
|
617
|
+
let composed;
|
|
618
|
+
try {
|
|
619
|
+
composed = await plan.composeBundle(MAX_REPLAY_REASON, this.executeStep(), state.sourceRun ? (step) => state.sourceRun.outputFor(step) : undefined, this.observeStep(state));
|
|
620
|
+
}
|
|
621
|
+
catch (err) {
|
|
622
|
+
this.upgrade(state, "failed");
|
|
623
|
+
logLine("replay.compose_failed", {
|
|
624
|
+
scenario: state.scenarioId ?? undefined,
|
|
625
|
+
why: "re-execution threw — continuing as a normal run",
|
|
626
|
+
error: errText(err),
|
|
627
|
+
});
|
|
628
|
+
return { kind: "abort" };
|
|
629
|
+
}
|
|
630
|
+
// Same rule as `runArmed`: a recovered chain whose tools errored is `failed`.
|
|
631
|
+
this.upgrade(state, composed.errored > 0 ? "failed" : "diverged");
|
|
632
|
+
state.stepsPinned += composed.executed + composed.recorded;
|
|
633
|
+
// The headline number of this whole design: recovering from a divergence
|
|
634
|
+
// costs nothing, because the proxies were already connected.
|
|
635
|
+
logLine("replay.compose", {
|
|
636
|
+
scenario: state.scenarioId ?? undefined,
|
|
637
|
+
remaining: composed.executed + composed.recorded + composed.skipped,
|
|
638
|
+
executed: composed.executed,
|
|
639
|
+
recorded: composed.recorded,
|
|
640
|
+
skipped: composed.skipped,
|
|
641
|
+
errored: composed.errored || undefined,
|
|
642
|
+
bytes: composed.text.length,
|
|
643
|
+
costUsd: state.fallbackCostUsd.toFixed(2),
|
|
644
|
+
});
|
|
645
|
+
logLine("replay.done", {
|
|
646
|
+
scenario: state.scenarioId ?? undefined,
|
|
647
|
+
mode: state.mode,
|
|
648
|
+
steps: `${composed.executed + composed.recorded}/${plan.stepCount}`,
|
|
649
|
+
outcome: state.outcome,
|
|
650
|
+
ms: Date.now() - state.armedAt,
|
|
651
|
+
});
|
|
652
|
+
if (composed.executed + composed.recorded === 0) {
|
|
653
|
+
// Nothing survived; a bundle of nothing is worse than no bundle.
|
|
654
|
+
return { kind: "abort" };
|
|
655
|
+
}
|
|
656
|
+
// Bash-clean delivery: the model reaches for `Bash` when it cannot call a
|
|
657
|
+
// scripted tool, and a genuine command output is trusted where a `deny`
|
|
658
|
+
// reason is read as adversarial interception. Neutralise the delimiter first.
|
|
659
|
+
if (toolName === "Bash") {
|
|
660
|
+
const delimiter = "BIR_EOF";
|
|
661
|
+
const safe = composed.text.split(delimiter).join("BIR_EOF_");
|
|
662
|
+
logDetail("replay.inject", { channel: "bash", bytes: composed.text.length });
|
|
663
|
+
return { kind: "bash", command: `cat <<'${delimiter}'\n${safe}\n${delimiter}` };
|
|
664
|
+
}
|
|
665
|
+
logDetail("replay.inject", { channel: "deny", bytes: composed.text.length });
|
|
666
|
+
return { kind: "deny", reason: composed.text };
|
|
667
|
+
}
|
|
668
|
+
/**
|
|
669
|
+
* The executor handed to a plan: dispatch a step to the proxy that owns its
|
|
670
|
+
* upstream.
|
|
671
|
+
*
|
|
672
|
+
* Rejecting means "could not be run **here**" — which is the only condition
|
|
673
|
+
* under which a recorded output may stand in. A tool that ran and failed
|
|
674
|
+
* resolves with its failure as the response, exactly as it would in a session.
|
|
675
|
+
*/
|
|
676
|
+
executeStep() {
|
|
677
|
+
return async (step, input) => {
|
|
678
|
+
const mcp = parseQualifiedName(step.toolName);
|
|
679
|
+
if (!mcp)
|
|
680
|
+
throw new Error(`${step.toolName} is not an MCP tool — it can only run in the session`);
|
|
681
|
+
const result = await this.work.call(mcp.serverName, mcp.toolName, input, this.budgets.stepMs);
|
|
682
|
+
// Serialize exactly as the proxy records it, or `toolOutputLogic` — which
|
|
683
|
+
// was authored against that shape — silently derives nothing (§7.2).
|
|
684
|
+
return serializeCapped(redact(result));
|
|
685
|
+
};
|
|
686
|
+
}
|
|
687
|
+
/**
|
|
688
|
+
* Where a pinned step's output will arrive from — the proxy's report, or the
|
|
689
|
+
* hook's `tool_response` (§7.2).
|
|
690
|
+
*
|
|
691
|
+
* NOT the same question as {@link reachOf}'s, and conflating them is a silent
|
|
692
|
+
* bug. `reachOf` answers "may *we* execute this step ourselves", which the
|
|
693
|
+
* allowlist restricts. This answers "will a `bir-proxy` see this call and
|
|
694
|
+
* report it", which depends only on whether the server is **wrapped** — a
|
|
695
|
+
* wrapped server the allowlist excludes is still proxied, still reported, and
|
|
696
|
+
* still recorded in the proxy's serialization. Threading such a step from the
|
|
697
|
+
* hook's differently-shaped view would derive nothing at all.
|
|
698
|
+
*/
|
|
699
|
+
reachFor(state, step) {
|
|
700
|
+
return reachOf(step.toolName, state.wrapped);
|
|
701
|
+
}
|
|
702
|
+
/**
|
|
703
|
+
* Log a step and remember its verdict, in that order.
|
|
704
|
+
*
|
|
705
|
+
* Every driver passes this — direct, divergence recovery and ad-hoc alike — so
|
|
706
|
+
* there is exactly one place a step's fate is decided, and the console can
|
|
707
|
+
* never be told something the log does not also say.
|
|
708
|
+
*/
|
|
709
|
+
observeStep(state) {
|
|
710
|
+
return (info) => {
|
|
711
|
+
this.logStep(info);
|
|
712
|
+
this.recordStep(state, info);
|
|
713
|
+
};
|
|
714
|
+
}
|
|
715
|
+
/** Upsert one step's verdict. Later news about a step replaces earlier news. */
|
|
716
|
+
recordStep(state, info) {
|
|
717
|
+
const status = info.outcome === "executed" ? "ok" : info.outcome;
|
|
718
|
+
if (state.stepResults.size >= MAX_STEP_RESULTS && !state.stepResults.has(info.step.stepIndex)) {
|
|
719
|
+
return;
|
|
720
|
+
}
|
|
721
|
+
const error = info.error
|
|
722
|
+
? info.error.length > MAX_STEP_ERROR_CHARS
|
|
723
|
+
? `${info.error.slice(0, MAX_STEP_ERROR_CHARS)}…`
|
|
724
|
+
: info.error
|
|
725
|
+
: undefined;
|
|
726
|
+
state.stepResults.set(info.step.stepIndex, {
|
|
727
|
+
stepIndex: info.step.stepIndex,
|
|
728
|
+
toolName: info.step.toolName,
|
|
729
|
+
status,
|
|
730
|
+
// A stage is only meaningful when something went wrong. `skipped` broke at
|
|
731
|
+
// the call itself — its tool could not run here.
|
|
732
|
+
stage: info.stage ?? (status === "skipped" ? "tool_call" : undefined),
|
|
733
|
+
error,
|
|
734
|
+
durationMs: info.ms,
|
|
735
|
+
});
|
|
736
|
+
}
|
|
737
|
+
/** The step verdicts, in chain order, for the execution report. */
|
|
738
|
+
stepResultsOf(state) {
|
|
739
|
+
if (state.stepResults.size === 0)
|
|
740
|
+
return undefined;
|
|
741
|
+
return [...state.stepResults.values()].sort((a, b) => a.stepIndex - b.stepIndex);
|
|
742
|
+
}
|
|
743
|
+
logStep(info) {
|
|
744
|
+
const mcp = parseQualifiedName(info.step.toolName);
|
|
745
|
+
if (info.outcome === "failed") {
|
|
746
|
+
logLine("replay.step_failed", {
|
|
747
|
+
n: info.step.stepIndex,
|
|
748
|
+
tool: info.step.toolName,
|
|
749
|
+
stage: info.stage,
|
|
750
|
+
why: info.stage === "tool_call"
|
|
751
|
+
? "its tool ran and reported an error — the step's work did not happen"
|
|
752
|
+
: "the scenario's own logic threw — this step cannot work until it is recalculated",
|
|
753
|
+
error: info.error,
|
|
754
|
+
});
|
|
755
|
+
return;
|
|
756
|
+
}
|
|
757
|
+
if (info.outcome === "skipped") {
|
|
758
|
+
logLine("replay.step_skipped", {
|
|
759
|
+
n: info.step.stepIndex,
|
|
760
|
+
tool: info.step.toolName,
|
|
761
|
+
why: "its tool could not run here and it has no recorded output",
|
|
762
|
+
error: info.error,
|
|
763
|
+
});
|
|
764
|
+
return;
|
|
765
|
+
}
|
|
766
|
+
if (info.outcome === "recorded") {
|
|
767
|
+
logLine("replay.step_recorded", {
|
|
768
|
+
n: info.step.stepIndex,
|
|
769
|
+
tool: info.step.toolName,
|
|
770
|
+
why: "served its recorded output — its tool could not run here",
|
|
771
|
+
error: info.error,
|
|
772
|
+
});
|
|
773
|
+
return;
|
|
774
|
+
}
|
|
775
|
+
logLine("replay.step", {
|
|
776
|
+
n: info.step.stepIndex,
|
|
777
|
+
tool: mcp?.toolName ?? info.step.toolName,
|
|
778
|
+
server: mcp?.serverName,
|
|
779
|
+
ms: info.ms,
|
|
780
|
+
ok: true,
|
|
781
|
+
emitted: info.derivedKeys?.join(",") || undefined,
|
|
782
|
+
});
|
|
783
|
+
}
|
|
784
|
+
/** Outcomes only move forward; `not_steered` is the floor (§11.1). */
|
|
785
|
+
upgrade(state, outcome) {
|
|
786
|
+
if (OUTCOME_RANK[outcome] > OUTCOME_RANK[state.outcome])
|
|
787
|
+
state.outcome = outcome;
|
|
788
|
+
}
|
|
789
|
+
retire(state, outcome) {
|
|
790
|
+
if (outcome)
|
|
791
|
+
this.upgrade(state, outcome);
|
|
792
|
+
state.retired = true;
|
|
793
|
+
state.pinned.clear();
|
|
794
|
+
}
|
|
795
|
+
withBudget(p, ms, what) {
|
|
796
|
+
let timer;
|
|
797
|
+
const budget = new Promise((_, reject) => {
|
|
798
|
+
timer = setTimeout(() => reject(new Error(`${what} exceeded ${ms}ms`)), ms);
|
|
799
|
+
timer.unref?.();
|
|
800
|
+
});
|
|
801
|
+
return Promise.race([p, budget]).finally(() => {
|
|
802
|
+
if (timer)
|
|
803
|
+
clearTimeout(timer);
|
|
804
|
+
});
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
//# sourceMappingURL=controller.js.map
|