@basein/runner 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +276 -0
- package/dist/auth/client.d.ts +85 -0
- package/dist/auth/client.js +284 -0
- package/dist/bin/bir-hooks.d.ts +48 -0
- package/dist/bin/bir-hooks.js +201 -0
- package/dist/bin/bir-proxy.d.ts +45 -0
- package/dist/bin/bir-proxy.js +207 -0
- package/dist/bin/bir-scenario.d.ts +24 -0
- package/dist/bin/bir-scenario.js +177 -0
- package/dist/bin/bir.d.ts +21 -0
- package/dist/bin/bir.js +876 -0
- package/dist/config/adapters/claude-code.d.ts +76 -0
- package/dist/config/adapters/claude-code.js +181 -0
- package/dist/config/adapters/generic.d.ts +17 -0
- package/dist/config/adapters/generic.js +36 -0
- package/dist/config/generate.d.ts +127 -0
- package/dist/config/generate.js +114 -0
- package/dist/config/resolve.d.ts +68 -0
- package/dist/config/resolve.js +132 -0
- package/dist/control/client.d.ts +56 -0
- package/dist/control/client.js +86 -0
- package/dist/control/correlation.d.ts +86 -0
- package/dist/control/correlation.js +0 -0
- package/dist/control/discovery.d.ts +50 -0
- package/dist/control/discovery.js +123 -0
- package/dist/control/ordering.d.ts +38 -0
- package/dist/control/ordering.js +44 -0
- package/dist/control/paths.d.ts +32 -0
- package/dist/control/paths.js +56 -0
- package/dist/control/server.d.ts +272 -0
- package/dist/control/server.js +1131 -0
- package/dist/control/transcript.d.ts +75 -0
- package/dist/control/transcript.js +241 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +32 -0
- package/dist/jsonrpc/framing.d.ts +49 -0
- package/dist/jsonrpc/framing.js +143 -0
- package/dist/jsonrpc/types.d.ts +52 -0
- package/dist/jsonrpc/types.js +46 -0
- package/dist/proxy/intercept.d.ts +55 -0
- package/dist/proxy/intercept.js +147 -0
- package/dist/proxy/relay.d.ts +97 -0
- package/dist/proxy/relay.js +166 -0
- package/dist/proxy/session.d.ts +116 -0
- package/dist/proxy/session.js +319 -0
- package/dist/record/housekeeping.d.ts +34 -0
- package/dist/record/housekeeping.js +39 -0
- package/dist/record/queue.d.ts +48 -0
- package/dist/record/queue.js +96 -0
- package/dist/record/recorder.d.ts +111 -0
- package/dist/record/recorder.js +39 -0
- package/dist/record/redact.d.ts +37 -0
- package/dist/record/redact.js +119 -0
- package/dist/record/remote-recorder.d.ts +110 -0
- package/dist/record/remote-recorder.js +301 -0
- package/dist/record/truncate.d.ts +36 -0
- package/dist/record/truncate.js +85 -0
- package/dist/replay/bundle.d.ts +36 -0
- package/dist/replay/bundle.js +89 -0
- package/dist/replay/controller.d.ts +300 -0
- package/dist/replay/controller.js +807 -0
- package/dist/replay/coverage.d.ts +41 -0
- package/dist/replay/coverage.js +56 -0
- package/dist/replay/derive.d.ts +58 -0
- package/dist/replay/derive.js +166 -0
- package/dist/replay/executor.d.ts +78 -0
- package/dist/replay/executor.js +233 -0
- package/dist/replay/logic.d.ts +31 -0
- package/dist/replay/logic.js +50 -0
- package/dist/replay/plan.d.ts +181 -0
- package/dist/replay/plan.js +397 -0
- package/dist/replay/pricing.d.ts +41 -0
- package/dist/replay/pricing.js +76 -0
- package/dist/replay/source-run.d.ts +50 -0
- package/dist/replay/source-run.js +98 -0
- package/dist/replay/tool-error.d.ts +22 -0
- package/dist/replay/tool-error.js +60 -0
- package/dist/replay/types.d.ts +116 -0
- package/dist/replay/types.js +35 -0
- package/dist/upstream/client.d.ts +78 -0
- package/dist/upstream/client.js +114 -0
- package/dist/upstream/http-client.d.ts +78 -0
- package/dist/upstream/http-client.js +261 -0
- package/dist/upstream/lazy-client.d.ts +31 -0
- package/dist/upstream/lazy-client.js +53 -0
- package/dist/upstream/stdio-client.d.ts +57 -0
- package/dist/upstream/stdio-client.js +203 -0
- package/dist/util/log.d.ts +27 -0
- package/dist/util/log.js +51 -0
- package/dist/util/version.d.ts +2 -0
- package/dist/util/version.js +40 -0
- package/docs/BaseInstRunner.md +621 -0
- package/docs/calculatedReplay.md +1185 -0
- package/docs/calculatedReplayGuide.md +448 -0
- package/docs/installRun.md +413 -0
- package/docs/mcpmark.md +752 -0
- package/docs/quickstart.md +201 -0
- package/docs/t-bench.md +394 -0
- package/package.json +56 -0
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* truncate — payload size caps (Phase 4.4).
|
|
3
|
+
*
|
|
4
|
+
* A single `chrome-devtools` screenshot is a megabyte of base64. Left alone it
|
|
5
|
+
* dominates the run, the request body and the corpus, and it is worth nothing to
|
|
6
|
+
* a scenario. So every recorded payload is capped twice:
|
|
7
|
+
*
|
|
8
|
+
* 1. **Per string**, so one enormous leaf cannot hide inside an otherwise small
|
|
9
|
+
* structure and so the rest of that structure still survives intact.
|
|
10
|
+
* 2. **Per payload**, on the serialized form, as the actual guarantee. If the
|
|
11
|
+
* whole thing still exceeds the cap after step 1 it is replaced by an
|
|
12
|
+
* explicit marker rather than silently clipped mid-JSON.
|
|
13
|
+
*
|
|
14
|
+
* Truncation is always **visible**: a truncated string keeps a suffix naming the
|
|
15
|
+
* bytes dropped, and a truncated payload serialises to
|
|
16
|
+
* `{ truncated: true, originalBytes, preview }`. A consumer can tell a short
|
|
17
|
+
* result from a clipped one, which is the whole point.
|
|
18
|
+
*/
|
|
19
|
+
/** 64 KiB per payload, as suggested in Phase 4.4. */
|
|
20
|
+
export const DEFAULT_MAX_PAYLOAD_BYTES = 64 * 1024;
|
|
21
|
+
/** 8 KiB per individual string leaf. */
|
|
22
|
+
export const DEFAULT_MAX_STRING_BYTES = 8 * 1024;
|
|
23
|
+
const utf8 = (s) => Buffer.byteLength(s, "utf8");
|
|
24
|
+
/** Clip one string to `max` bytes, appending a marker naming what was dropped. */
|
|
25
|
+
export function truncateString(text, max) {
|
|
26
|
+
const bytes = utf8(text);
|
|
27
|
+
if (bytes <= max)
|
|
28
|
+
return text;
|
|
29
|
+
// Slice on a byte boundary, then repair any split multi-byte character.
|
|
30
|
+
const head = Buffer.from(text, "utf8").subarray(0, max).toString("utf8").replace(/�$/, "");
|
|
31
|
+
return `${head}… [truncated ${bytes - utf8(head)} of ${bytes} bytes]`;
|
|
32
|
+
}
|
|
33
|
+
/** Deep-copy `value`, clipping every string leaf. Input is never mutated. */
|
|
34
|
+
export function truncateStrings(value, max, depth = 0) {
|
|
35
|
+
if (depth > 32)
|
|
36
|
+
return "[max depth]";
|
|
37
|
+
if (typeof value === "string")
|
|
38
|
+
return truncateString(value, max);
|
|
39
|
+
if (Array.isArray(value))
|
|
40
|
+
return value.map((v) => truncateStrings(v, max, depth + 1));
|
|
41
|
+
if (value && typeof value === "object") {
|
|
42
|
+
const out = {};
|
|
43
|
+
for (const [k, v] of Object.entries(value)) {
|
|
44
|
+
out[k] = truncateStrings(v, max, depth + 1);
|
|
45
|
+
}
|
|
46
|
+
return out;
|
|
47
|
+
}
|
|
48
|
+
return value;
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Serialize `value` for recording, capped and with truncation made explicit.
|
|
52
|
+
* Never throws: an unserialisable payload becomes a marker string.
|
|
53
|
+
*/
|
|
54
|
+
export function serializeCapped(value, opts = {}) {
|
|
55
|
+
const maxPayload = opts.maxPayloadBytes ?? DEFAULT_MAX_PAYLOAD_BYTES;
|
|
56
|
+
const maxString = opts.maxStringBytes ?? DEFAULT_MAX_STRING_BYTES;
|
|
57
|
+
let original;
|
|
58
|
+
try {
|
|
59
|
+
original = JSON.stringify(value) ?? "null";
|
|
60
|
+
}
|
|
61
|
+
catch {
|
|
62
|
+
return JSON.stringify({ truncated: true, reason: "unserializable" });
|
|
63
|
+
}
|
|
64
|
+
const originalBytes = utf8(original);
|
|
65
|
+
if (originalBytes <= maxPayload)
|
|
66
|
+
return original;
|
|
67
|
+
let clipped;
|
|
68
|
+
try {
|
|
69
|
+
clipped = JSON.stringify(truncateStrings(value, maxString)) ?? "null";
|
|
70
|
+
}
|
|
71
|
+
catch {
|
|
72
|
+
clipped = "null";
|
|
73
|
+
}
|
|
74
|
+
if (utf8(clipped) <= maxPayload) {
|
|
75
|
+
return JSON.stringify({ truncated: true, originalBytes, value: JSON.parse(clipped) });
|
|
76
|
+
}
|
|
77
|
+
// Still too big: the structure itself is the problem (thousands of nodes, not
|
|
78
|
+
// one fat leaf). Keep a readable preview and say so.
|
|
79
|
+
return JSON.stringify({
|
|
80
|
+
truncated: true,
|
|
81
|
+
originalBytes,
|
|
82
|
+
preview: truncateString(clipped, Math.max(1024, Math.floor(maxPayload / 2))),
|
|
83
|
+
});
|
|
84
|
+
}
|
|
85
|
+
//# sourceMappingURL=truncate.js.map
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* bundle — laying executed steps out as text the model can answer from
|
|
3
|
+
* (docs/calculatedReplay.md §8).
|
|
4
|
+
*
|
|
5
|
+
* Ported from RRepeat's `payload-executor.ts`, whose fair-share rule is the
|
|
6
|
+
* right one and is worth restating: a naive "truncate the whole thing at N"
|
|
7
|
+
* drops the *last* steps entirely, and the last steps are the ones carrying the
|
|
8
|
+
* answer. So labels and computed inputs get a fixed budget, and whatever remains
|
|
9
|
+
* is split evenly across the N responses — every step stays represented, and the
|
|
10
|
+
* ones that were shortened say so.
|
|
11
|
+
*
|
|
12
|
+
* `recorded: true` marks a response that came from the source run rather than a
|
|
13
|
+
* fresh execution (§8.1). It is surfaced in the header *and* per entry, because
|
|
14
|
+
* a bundle that quietly passed off a stale result as a live one would be the
|
|
15
|
+
* worst outcome available here.
|
|
16
|
+
*/
|
|
17
|
+
/** `permissionDecisionReason` is fed to the model as text; keep denials bounded. */
|
|
18
|
+
export declare const MAX_REPLAY_REASON = 60000;
|
|
19
|
+
/** One executed step, pre-stringified and ready to lay out. */
|
|
20
|
+
export interface BundleEntry {
|
|
21
|
+
toolName: string;
|
|
22
|
+
/** JSON of the computed tool input, already capped. */
|
|
23
|
+
input: string;
|
|
24
|
+
response: string;
|
|
25
|
+
/** True when `response` is a recorded output, not a fresh run. */
|
|
26
|
+
recorded?: boolean;
|
|
27
|
+
}
|
|
28
|
+
/** Cap a computed input for inclusion in a bundle. */
|
|
29
|
+
export declare function bundleInput(value: unknown): string;
|
|
30
|
+
/**
|
|
31
|
+
* Compose the bundle. `maxChars` is a hard ceiling on the returned string.
|
|
32
|
+
*/
|
|
33
|
+
export declare function assembleBundle(entries: readonly BundleEntry[], maxChars: number): string;
|
|
34
|
+
/** Truncate to `max` chars, appending a marker naming how much was dropped. */
|
|
35
|
+
export declare function truncate(s: string, max: number): string;
|
|
36
|
+
//# sourceMappingURL=bundle.d.ts.map
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* bundle — laying executed steps out as text the model can answer from
|
|
3
|
+
* (docs/calculatedReplay.md §8).
|
|
4
|
+
*
|
|
5
|
+
* Ported from RRepeat's `payload-executor.ts`, whose fair-share rule is the
|
|
6
|
+
* right one and is worth restating: a naive "truncate the whole thing at N"
|
|
7
|
+
* drops the *last* steps entirely, and the last steps are the ones carrying the
|
|
8
|
+
* answer. So labels and computed inputs get a fixed budget, and whatever remains
|
|
9
|
+
* is split evenly across the N responses — every step stays represented, and the
|
|
10
|
+
* ones that were shortened say so.
|
|
11
|
+
*
|
|
12
|
+
* `recorded: true` marks a response that came from the source run rather than a
|
|
13
|
+
* fresh execution (§8.1). It is surfaced in the header *and* per entry, because
|
|
14
|
+
* a bundle that quietly passed off a stale result as a live one would be the
|
|
15
|
+
* worst outcome available here.
|
|
16
|
+
*/
|
|
17
|
+
/** `permissionDecisionReason` is fed to the model as text; keep denials bounded. */
|
|
18
|
+
export const MAX_REPLAY_REASON = 60_000;
|
|
19
|
+
/** Cap each computed input so one huge argument object cannot starve responses. */
|
|
20
|
+
const MAX_INPUT_CHARS = 2_000;
|
|
21
|
+
/** Fixed structural overhead reserved per step (labels, separators). */
|
|
22
|
+
const PER_STEP_OVERHEAD = 80;
|
|
23
|
+
/** Cap a computed input for inclusion in a bundle. */
|
|
24
|
+
export function bundleInput(value) {
|
|
25
|
+
let text;
|
|
26
|
+
try {
|
|
27
|
+
text = JSON.stringify(value) ?? "null";
|
|
28
|
+
}
|
|
29
|
+
catch {
|
|
30
|
+
text = "[unserializable]";
|
|
31
|
+
}
|
|
32
|
+
return truncate(text, MAX_INPUT_CHARS);
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* The header must say what actually happened. Every entry not marked `recorded`
|
|
36
|
+
* was executed for real, moments ago, against the live system — and a model told
|
|
37
|
+
* the opposite ("previously recorded") cannot verify a write without calling a
|
|
38
|
+
* tool, so it reasonably redoes the whole task. On a CRUD scenario that measured
|
|
39
|
+
* at +40% over baseline with perfect plumbing (docs/mcpmark.md §12).
|
|
40
|
+
*/
|
|
41
|
+
const LIVE_HEADER = "[BaseInstRunner calculated replay] The tool sequence below has JUST been executed " +
|
|
42
|
+
"for real, moments ago, against the live system, on the user's behalf. Every result " +
|
|
43
|
+
"is that call's actual output, and every side effect — rows written, tables created, " +
|
|
44
|
+
"files changed — is already in place. Do NOT repeat, redo or re-verify this work, and " +
|
|
45
|
+
"do NOT call any tools. Answer the user's request from these results.";
|
|
46
|
+
const LIVE_FOOTER = "\n\nThe work above is already done. Answer the user's request from these results. " +
|
|
47
|
+
"Do not call any tools.";
|
|
48
|
+
/** Some entry stood in from the source run: say which, and keep the caveat. */
|
|
49
|
+
const MIXED_HEADER = "[BaseInstRunner calculated replay] A known-good tool sequence was executed for this " +
|
|
50
|
+
"request. Below are the tool calls and their results, in order. Steps not marked " +
|
|
51
|
+
"(recorded) ran for real just now and their side effects are already in place. " +
|
|
52
|
+
"Results marked (recorded) come from an earlier recording rather than a fresh run, " +
|
|
53
|
+
"because their tool cannot be executed here — treat them as possibly out of date. " +
|
|
54
|
+
"Use them to answer the user's request now — do NOT call any tools.";
|
|
55
|
+
const MIXED_FOOTER = "\n\nAnswer the user's request using the results above. Do not call any tools.";
|
|
56
|
+
/**
|
|
57
|
+
* Compose the bundle. `maxChars` is a hard ceiling on the returned string.
|
|
58
|
+
*/
|
|
59
|
+
export function assembleBundle(entries, maxChars) {
|
|
60
|
+
const n = entries.length;
|
|
61
|
+
const anyRecorded = entries.some((e) => e.recorded);
|
|
62
|
+
const header = anyRecorded ? MIXED_HEADER : LIVE_HEADER;
|
|
63
|
+
const footer = anyRecorded ? MIXED_FOOTER : LIVE_FOOTER;
|
|
64
|
+
const fixedCost = header.length +
|
|
65
|
+
footer.length +
|
|
66
|
+
entries.reduce((sum, e) => sum + PER_STEP_OVERHEAD + e.toolName.length + e.input.length, 0);
|
|
67
|
+
const responseBudget = Math.max(0, maxChars - fixedCost);
|
|
68
|
+
const perResponse = n > 0 ? Math.floor(responseBudget / n) : responseBudget;
|
|
69
|
+
const sections = entries.map((e, i) => {
|
|
70
|
+
const response = truncate(e.response, perResponse);
|
|
71
|
+
return (`\n\n--- Step ${i + 1}: ${e.toolName} ---\n` +
|
|
72
|
+
`Call: ${e.input}\n` +
|
|
73
|
+
`Result${e.recorded ? " (recorded)" : ""}: ${response}`);
|
|
74
|
+
});
|
|
75
|
+
return truncate(header + sections.join("") + footer, maxChars);
|
|
76
|
+
}
|
|
77
|
+
/** Truncate to `max` chars, appending a marker naming how much was dropped. */
|
|
78
|
+
export function truncate(s, max) {
|
|
79
|
+
if (max <= 0)
|
|
80
|
+
return "";
|
|
81
|
+
if (s.length <= max)
|
|
82
|
+
return s;
|
|
83
|
+
const marker = (dropped) => `…[truncated ${dropped} chars]`;
|
|
84
|
+
// Reserve room for the marker so the total stays within `max`.
|
|
85
|
+
const reserve = marker(s.length).length;
|
|
86
|
+
const keep = Math.max(0, max - reserve);
|
|
87
|
+
return s.slice(0, keep) + marker(s.length - keep);
|
|
88
|
+
}
|
|
89
|
+
//# sourceMappingURL=bundle.js.map
|
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* controller — everything replay decides, in one place (docs/calculatedReplay.md).
|
|
3
|
+
*
|
|
4
|
+
* The control server owns runs, ordering and recording. This owns the *plan*:
|
|
5
|
+
* the gate ladder, the mode choice, arming, pinning, threading, direct
|
|
6
|
+
* execution, divergence recovery and the execution report. `ControlServer`
|
|
7
|
+
* delegates to it and stores one {@link ReplayState} per run.
|
|
8
|
+
*
|
|
9
|
+
* THE GOVERNING RULE IS UNCHANGED and outranks every optimisation here: the host
|
|
10
|
+
* session must never fail because of BaseInstRunner. Every method below either
|
|
11
|
+
* degrades to "run the turn normally" or returns something the session can
|
|
12
|
+
* ignore. Nothing throws at a hook.
|
|
13
|
+
*/
|
|
14
|
+
import type { ExecutionReport } from "../record/recorder.js";
|
|
15
|
+
import { type ReplayMode, type StepReach } from "./coverage.js";
|
|
16
|
+
import { deriveParameters } from "./derive.js";
|
|
17
|
+
import { ProxyWorkQueue } from "./executor.js";
|
|
18
|
+
import { ScenarioReplayPlan } from "./plan.js";
|
|
19
|
+
import { SourceRunOutputs } from "./source-run.js";
|
|
20
|
+
import { type ExecutionOutcome, type ExecutionStepResult, type SerializedScenario } from "./types.js";
|
|
21
|
+
/** The first-party tool a `direct` plan is delivered through (§6.3). */
|
|
22
|
+
export declare const DIRECT_TOOL_NAME = "mcp__bir__run_scenario";
|
|
23
|
+
/** How long a `/proxy/poll` is held open before it answers empty. */
|
|
24
|
+
export declare const POLL_HOLD_MS = 25000;
|
|
25
|
+
export interface ReplayBudgets {
|
|
26
|
+
/** How long `/session/prompt` waits for the match. */
|
|
27
|
+
matchMs: number;
|
|
28
|
+
/** How long the first `PreToolUse` waits for derivation. */
|
|
29
|
+
deriveMs: number;
|
|
30
|
+
/** Ceiling on a whole direct plan. */
|
|
31
|
+
planMs: number;
|
|
32
|
+
/** Ceiling on one direct `tools/call`. */
|
|
33
|
+
stepMs: number;
|
|
34
|
+
}
|
|
35
|
+
export declare const DEFAULT_BUDGETS: ReplayBudgets;
|
|
36
|
+
export interface ReplayOptions {
|
|
37
|
+
/** `BIR_REPLAY=1`. Everything here is inert when false. */
|
|
38
|
+
enabled: boolean;
|
|
39
|
+
/** Minimum similarity to steer — deliberately above the service's detection threshold. */
|
|
40
|
+
minSimilarity: number;
|
|
41
|
+
/** `BIR_REPLAY_ALLOW_SERVERS`; undefined means every wrapped server. */
|
|
42
|
+
allowServers?: ReadonlySet<string>;
|
|
43
|
+
budgets?: Partial<ReplayBudgets>;
|
|
44
|
+
/** For the lazy source-run fetch and nothing else. */
|
|
45
|
+
authUrl?: string;
|
|
46
|
+
authToken?: () => string;
|
|
47
|
+
/** Injected by tests. */
|
|
48
|
+
deriveImpl?: typeof deriveParameters;
|
|
49
|
+
fetchImpl?: typeof fetch;
|
|
50
|
+
/** Anthropic key for derivation. Absent → recorded sample values, free. */
|
|
51
|
+
apiKey?: string;
|
|
52
|
+
}
|
|
53
|
+
/** One pinned call, remembered so its output is threaded from the right source. */
|
|
54
|
+
interface PinnedCall {
|
|
55
|
+
stepIndex: number;
|
|
56
|
+
reach: StepReach;
|
|
57
|
+
toolName: string;
|
|
58
|
+
/** When it was pinned — a steered step's duration is measured across hooks. */
|
|
59
|
+
pinnedAt: number;
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* Per-run replay state. Created on every match — including declined ones,
|
|
63
|
+
* because a decline still has to be reported as a baseline sample (§11.1).
|
|
64
|
+
*/
|
|
65
|
+
export interface ReplayState {
|
|
66
|
+
scenarioId: string | null;
|
|
67
|
+
ticket?: string;
|
|
68
|
+
similarity: number;
|
|
69
|
+
matchedRunId: string;
|
|
70
|
+
mode: ReplayMode;
|
|
71
|
+
/** Server keys a proxy wraps — decides where an output is threaded from (§7.2). */
|
|
72
|
+
wrapped: ReadonlySet<string>;
|
|
73
|
+
/** Undefined when a gate declined. */
|
|
74
|
+
plan?: ScenarioReplayPlan;
|
|
75
|
+
/** `tool_use_id` → the step it pinned. */
|
|
76
|
+
pinned: Map<string, PinnedCall>;
|
|
77
|
+
outcome: ExecutionOutcome;
|
|
78
|
+
deriveCostUsd: number;
|
|
79
|
+
fallbackCostUsd: number;
|
|
80
|
+
stepsPlanned: number;
|
|
81
|
+
stepsPinned: number;
|
|
82
|
+
/**
|
|
83
|
+
* One verdict per step this run reached, by `stepIndex`. Reported with the
|
|
84
|
+
* execution so the recording page can say which step of the chain failed and
|
|
85
|
+
* why, instead of showing a healthy-looking scenario that never works.
|
|
86
|
+
*
|
|
87
|
+
* A map, not an array: a steered step is decided across two hooks (pinned in
|
|
88
|
+
* `PreToolUse`, threaded in `PostToolUse`), and the second word on a step
|
|
89
|
+
* replaces the first rather than appending a second verdict for it.
|
|
90
|
+
*/
|
|
91
|
+
stepResults: Map<number, ExecutionStepResult>;
|
|
92
|
+
armedAt: number;
|
|
93
|
+
reported: boolean;
|
|
94
|
+
sourceRun?: SourceRunOutputs;
|
|
95
|
+
/** Set once a plan has been retired, so nothing re-arms mid-turn. */
|
|
96
|
+
retired: boolean;
|
|
97
|
+
}
|
|
98
|
+
/** What `PreToolUse` should do about this call. */
|
|
99
|
+
export type PreToolAction =
|
|
100
|
+
/** Pin the arguments and let the real tool run. */
|
|
101
|
+
{
|
|
102
|
+
kind: "pin";
|
|
103
|
+
input: Record<string, unknown>;
|
|
104
|
+
stepIndex: number;
|
|
105
|
+
}
|
|
106
|
+
/** Not ours — the server handles it normally. */
|
|
107
|
+
| {
|
|
108
|
+
kind: "passthrough";
|
|
109
|
+
}
|
|
110
|
+
/** Rewrite a `Bash` call to `cat` the bundle, so the model reads real output. */
|
|
111
|
+
| {
|
|
112
|
+
kind: "bash";
|
|
113
|
+
command: string;
|
|
114
|
+
}
|
|
115
|
+
/** Deny this call, handing the bundle back as the reason. */
|
|
116
|
+
| {
|
|
117
|
+
kind: "deny";
|
|
118
|
+
reason: string;
|
|
119
|
+
}
|
|
120
|
+
/** Give up; continue as an ordinary turn. */
|
|
121
|
+
| {
|
|
122
|
+
kind: "abort";
|
|
123
|
+
};
|
|
124
|
+
export declare class ReplayController {
|
|
125
|
+
readonly enabled: boolean;
|
|
126
|
+
readonly work: ProxyWorkQueue;
|
|
127
|
+
readonly budgets: ReplayBudgets;
|
|
128
|
+
private readonly opts;
|
|
129
|
+
constructor(opts: ReplayOptions);
|
|
130
|
+
/** The plan's own delivery vehicle is never a scenario step. */
|
|
131
|
+
isDirectTool(toolName: string): boolean;
|
|
132
|
+
/**
|
|
133
|
+
* Await a match under {@link ReplayBudgets.matchMs}.
|
|
134
|
+
*
|
|
135
|
+
* Budget expiry is not an error and never blocks the user: it means this turn
|
|
136
|
+
* runs normally, and the one `replay.decision` line says so. Resolving `null`
|
|
137
|
+
* covers both "no match" and "too slow", which are the same thing from here.
|
|
138
|
+
*/
|
|
139
|
+
awaitMatch(pending: Promise<unknown>): Promise<RunMatchLike | null>;
|
|
140
|
+
/**
|
|
141
|
+
* Run the gate ladder and, if every gate passes, arm a plan.
|
|
142
|
+
*
|
|
143
|
+
* Always returns a state — a declined match still has a ticket to redeem and a
|
|
144
|
+
* baseline sample to contribute, and losing that is how a savings ledger ends
|
|
145
|
+
* up with a denominator nobody measured.
|
|
146
|
+
*/
|
|
147
|
+
arm(match: RunMatchLike, prompt: string, wrapped: ReadonlySet<string>): ReplayState;
|
|
148
|
+
/** The directive to inject via `additionalContext`, or undefined when declined. */
|
|
149
|
+
directiveFor(state: ReplayState): string | undefined;
|
|
150
|
+
/**
|
|
151
|
+
* `PreToolUse`, while a plan is active.
|
|
152
|
+
*
|
|
153
|
+
* Three tiers (§8): pin the expected call; on divergence execute the remainder
|
|
154
|
+
* and hand it back through the best channel available; if even that fails,
|
|
155
|
+
* abort to an ordinary turn.
|
|
156
|
+
*/
|
|
157
|
+
preTool(state: ReplayState, toolName: string, toolUseId: string): Promise<PreToolAction>;
|
|
158
|
+
/**
|
|
159
|
+
* `PostToolUse` for a call this plan pinned.
|
|
160
|
+
*
|
|
161
|
+
* `output` must be serialized the way the step's output was **recorded**: for a
|
|
162
|
+
* wrapped MCP step that is the proxy's whole `CallToolResult`, which is why the
|
|
163
|
+
* server threads those from `/proxy/step` rather than from the hook (§7.2).
|
|
164
|
+
* Returns true when the plan is now complete.
|
|
165
|
+
*/
|
|
166
|
+
postTool(state: ReplayState, toolUseId: string, output: string): boolean;
|
|
167
|
+
/** Whether a pinned call's output is threaded from the proxy's report (§7.2). */
|
|
168
|
+
threadsFromProxy(state: ReplayState, toolUseId: string): boolean;
|
|
169
|
+
/** True when this call belongs to the active plan. */
|
|
170
|
+
isPinned(state: ReplayState, toolUseId: string): boolean;
|
|
171
|
+
/**
|
|
172
|
+
* `POST /scenario/run` — the whole of a `direct` plan, executed here.
|
|
173
|
+
*
|
|
174
|
+
* Every step runs through the proxy that already owns its upstream, so the
|
|
175
|
+
* model spends nothing beyond the turn that reads the results.
|
|
176
|
+
*/
|
|
177
|
+
runArmed(state: ReplayState): Promise<{
|
|
178
|
+
ok: boolean;
|
|
179
|
+
why?: string;
|
|
180
|
+
text?: string;
|
|
181
|
+
responseModel?: Record<string, unknown>;
|
|
182
|
+
steps?: number;
|
|
183
|
+
partial?: boolean;
|
|
184
|
+
}>;
|
|
185
|
+
/**
|
|
186
|
+
* Build the execution report for a sealed run, or undefined when there is
|
|
187
|
+
* nothing to report.
|
|
188
|
+
*
|
|
189
|
+
* `costUsd <= 0` on a decline is not worth sending: the service answers
|
|
190
|
+
* `202 { recorded: false }` for a baseline sample with no measured cost, and a
|
|
191
|
+
* report that books nothing is noise on both sides.
|
|
192
|
+
*/
|
|
193
|
+
buildReport(state: ReplayState, d: {
|
|
194
|
+
sessionCostUsd: number;
|
|
195
|
+
measured: boolean;
|
|
196
|
+
durationMs: number;
|
|
197
|
+
prompt?: string;
|
|
198
|
+
}): ExecutionReport | undefined;
|
|
199
|
+
/**
|
|
200
|
+
* The step a failed replay is fairly blamed on: the first whose own logic
|
|
201
|
+
* threw, or failing that the first that could not run at all.
|
|
202
|
+
*
|
|
203
|
+
* A `failed` step outranks a `skipped` one however early the skip came — a
|
|
204
|
+
* scenario whose logic throws is broken for everybody, while a step that could
|
|
205
|
+
* not run here says something about this machine.
|
|
206
|
+
*/
|
|
207
|
+
private blameStep;
|
|
208
|
+
/**
|
|
209
|
+
* Run a scenario that no match armed — `bir replay` (§12.1).
|
|
210
|
+
*
|
|
211
|
+
* The same plan, the same executor, the same live proxies; only the trigger
|
|
212
|
+
* differs. It is how you test a scenario without a session, how a Tier 2 or
|
|
213
|
+
* non-Claude-Code client gets any replay at all, and the first thing to reach
|
|
214
|
+
* for when a steered turn behaves oddly.
|
|
215
|
+
*
|
|
216
|
+
* Deliberately outside the gate ladder: the operator typed the scenario id, so
|
|
217
|
+
* there is nothing to be similar *to* and nothing to decline. It books no
|
|
218
|
+
* execution either — there is no ticket, and inventing a saving for a manual
|
|
219
|
+
* invocation is exactly the kind of number a ledger must never contain.
|
|
220
|
+
*/
|
|
221
|
+
runAdHoc(scenario: SerializedScenario, prompt: string, wrapped: ReadonlySet<string>): Promise<{
|
|
222
|
+
ok: boolean;
|
|
223
|
+
why?: string;
|
|
224
|
+
text?: string;
|
|
225
|
+
responseModel?: Record<string, unknown>;
|
|
226
|
+
steps?: number;
|
|
227
|
+
partial?: boolean;
|
|
228
|
+
trace?: Array<{
|
|
229
|
+
step: number;
|
|
230
|
+
tool: string;
|
|
231
|
+
input: unknown;
|
|
232
|
+
outcome: string;
|
|
233
|
+
ms: number;
|
|
234
|
+
}>;
|
|
235
|
+
}>;
|
|
236
|
+
/** Release every parked poller. Called at control-server shutdown. */
|
|
237
|
+
close(): void;
|
|
238
|
+
/**
|
|
239
|
+
* Kick off derivation without awaiting it (§10). The prompt hook returns the
|
|
240
|
+
* directive the moment the match lands; the first `PreToolUse` — or
|
|
241
|
+
* `/scenario/run`, which has no hook timeout at all — is where the wait lands.
|
|
242
|
+
*/
|
|
243
|
+
private startDerivation;
|
|
244
|
+
/**
|
|
245
|
+
* Divergence (§8). Execute the remaining steps for real, then deliver.
|
|
246
|
+
*
|
|
247
|
+
* The plan is retired first: whatever happens next, this turn neither steers
|
|
248
|
+
* nor re-injects again.
|
|
249
|
+
*/
|
|
250
|
+
private diverge;
|
|
251
|
+
/**
|
|
252
|
+
* The executor handed to a plan: dispatch a step to the proxy that owns its
|
|
253
|
+
* upstream.
|
|
254
|
+
*
|
|
255
|
+
* Rejecting means "could not be run **here**" — which is the only condition
|
|
256
|
+
* under which a recorded output may stand in. A tool that ran and failed
|
|
257
|
+
* resolves with its failure as the response, exactly as it would in a session.
|
|
258
|
+
*/
|
|
259
|
+
private executeStep;
|
|
260
|
+
/**
|
|
261
|
+
* Where a pinned step's output will arrive from — the proxy's report, or the
|
|
262
|
+
* hook's `tool_response` (§7.2).
|
|
263
|
+
*
|
|
264
|
+
* NOT the same question as {@link reachOf}'s, and conflating them is a silent
|
|
265
|
+
* bug. `reachOf` answers "may *we* execute this step ourselves", which the
|
|
266
|
+
* allowlist restricts. This answers "will a `bir-proxy` see this call and
|
|
267
|
+
* report it", which depends only on whether the server is **wrapped** — a
|
|
268
|
+
* wrapped server the allowlist excludes is still proxied, still reported, and
|
|
269
|
+
* still recorded in the proxy's serialization. Threading such a step from the
|
|
270
|
+
* hook's differently-shaped view would derive nothing at all.
|
|
271
|
+
*/
|
|
272
|
+
private reachFor;
|
|
273
|
+
/**
|
|
274
|
+
* Log a step and remember its verdict, in that order.
|
|
275
|
+
*
|
|
276
|
+
* Every driver passes this — direct, divergence recovery and ad-hoc alike — so
|
|
277
|
+
* there is exactly one place a step's fate is decided, and the console can
|
|
278
|
+
* never be told something the log does not also say.
|
|
279
|
+
*/
|
|
280
|
+
private observeStep;
|
|
281
|
+
/** Upsert one step's verdict. Later news about a step replaces earlier news. */
|
|
282
|
+
private recordStep;
|
|
283
|
+
/** The step verdicts, in chain order, for the execution report. */
|
|
284
|
+
private stepResultsOf;
|
|
285
|
+
private logStep;
|
|
286
|
+
/** Outcomes only move forward; `not_steered` is the floor (§11.1). */
|
|
287
|
+
private upgrade;
|
|
288
|
+
private retire;
|
|
289
|
+
private withBudget;
|
|
290
|
+
}
|
|
291
|
+
/** The subset of `RunMatch` replay consumes. Structural, so tests need no recorder. */
|
|
292
|
+
export interface RunMatchLike {
|
|
293
|
+
runId: string;
|
|
294
|
+
scenarioId: string | null;
|
|
295
|
+
similarity: number;
|
|
296
|
+
scenario: Record<string, unknown> | null;
|
|
297
|
+
executionTicket?: string;
|
|
298
|
+
}
|
|
299
|
+
export {};
|
|
300
|
+
//# sourceMappingURL=controller.d.ts.map
|