@basein/runner 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +276 -0
- package/dist/auth/client.d.ts +85 -0
- package/dist/auth/client.js +284 -0
- package/dist/bin/bir-hooks.d.ts +48 -0
- package/dist/bin/bir-hooks.js +201 -0
- package/dist/bin/bir-proxy.d.ts +45 -0
- package/dist/bin/bir-proxy.js +207 -0
- package/dist/bin/bir-scenario.d.ts +24 -0
- package/dist/bin/bir-scenario.js +177 -0
- package/dist/bin/bir.d.ts +21 -0
- package/dist/bin/bir.js +876 -0
- package/dist/config/adapters/claude-code.d.ts +76 -0
- package/dist/config/adapters/claude-code.js +181 -0
- package/dist/config/adapters/generic.d.ts +17 -0
- package/dist/config/adapters/generic.js +36 -0
- package/dist/config/generate.d.ts +127 -0
- package/dist/config/generate.js +114 -0
- package/dist/config/resolve.d.ts +68 -0
- package/dist/config/resolve.js +132 -0
- package/dist/control/client.d.ts +56 -0
- package/dist/control/client.js +86 -0
- package/dist/control/correlation.d.ts +86 -0
- package/dist/control/correlation.js +0 -0
- package/dist/control/discovery.d.ts +50 -0
- package/dist/control/discovery.js +123 -0
- package/dist/control/ordering.d.ts +38 -0
- package/dist/control/ordering.js +44 -0
- package/dist/control/paths.d.ts +32 -0
- package/dist/control/paths.js +56 -0
- package/dist/control/server.d.ts +272 -0
- package/dist/control/server.js +1131 -0
- package/dist/control/transcript.d.ts +75 -0
- package/dist/control/transcript.js +241 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +32 -0
- package/dist/jsonrpc/framing.d.ts +49 -0
- package/dist/jsonrpc/framing.js +143 -0
- package/dist/jsonrpc/types.d.ts +52 -0
- package/dist/jsonrpc/types.js +46 -0
- package/dist/proxy/intercept.d.ts +55 -0
- package/dist/proxy/intercept.js +147 -0
- package/dist/proxy/relay.d.ts +97 -0
- package/dist/proxy/relay.js +166 -0
- package/dist/proxy/session.d.ts +116 -0
- package/dist/proxy/session.js +319 -0
- package/dist/record/housekeeping.d.ts +34 -0
- package/dist/record/housekeeping.js +39 -0
- package/dist/record/queue.d.ts +48 -0
- package/dist/record/queue.js +96 -0
- package/dist/record/recorder.d.ts +111 -0
- package/dist/record/recorder.js +39 -0
- package/dist/record/redact.d.ts +37 -0
- package/dist/record/redact.js +119 -0
- package/dist/record/remote-recorder.d.ts +110 -0
- package/dist/record/remote-recorder.js +301 -0
- package/dist/record/truncate.d.ts +36 -0
- package/dist/record/truncate.js +85 -0
- package/dist/replay/bundle.d.ts +36 -0
- package/dist/replay/bundle.js +89 -0
- package/dist/replay/controller.d.ts +300 -0
- package/dist/replay/controller.js +807 -0
- package/dist/replay/coverage.d.ts +41 -0
- package/dist/replay/coverage.js +56 -0
- package/dist/replay/derive.d.ts +58 -0
- package/dist/replay/derive.js +166 -0
- package/dist/replay/executor.d.ts +78 -0
- package/dist/replay/executor.js +233 -0
- package/dist/replay/logic.d.ts +31 -0
- package/dist/replay/logic.js +50 -0
- package/dist/replay/plan.d.ts +181 -0
- package/dist/replay/plan.js +397 -0
- package/dist/replay/pricing.d.ts +41 -0
- package/dist/replay/pricing.js +76 -0
- package/dist/replay/source-run.d.ts +50 -0
- package/dist/replay/source-run.js +98 -0
- package/dist/replay/tool-error.d.ts +22 -0
- package/dist/replay/tool-error.js +60 -0
- package/dist/replay/types.d.ts +116 -0
- package/dist/replay/types.js +35 -0
- package/dist/upstream/client.d.ts +78 -0
- package/dist/upstream/client.js +114 -0
- package/dist/upstream/http-client.d.ts +78 -0
- package/dist/upstream/http-client.js +261 -0
- package/dist/upstream/lazy-client.d.ts +31 -0
- package/dist/upstream/lazy-client.js +53 -0
- package/dist/upstream/stdio-client.d.ts +57 -0
- package/dist/upstream/stdio-client.js +203 -0
- package/dist/util/log.d.ts +27 -0
- package/dist/util/log.js +51 -0
- package/dist/util/version.d.ts +2 -0
- package/dist/util/version.js +40 -0
- package/docs/BaseInstRunner.md +621 -0
- package/docs/calculatedReplay.md +1185 -0
- package/docs/calculatedReplayGuide.md +448 -0
- package/docs/installRun.md +413 -0
- package/docs/mcpmark.md +752 -0
- package/docs/quickstart.md +201 -0
- package/docs/t-bench.md +394 -0
- package/package.json +56 -0
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recorder — the surface both tiers use (§5).
|
|
3
|
+
*
|
|
4
|
+
* Deliberately identical in shape to RRepeat's `RemoteRecorder`, so the BaseIn
|
|
5
|
+
* service needs no new endpoints for v1: a BaseInstRunner run is the same row
|
|
6
|
+
* shape as an RRepeat one, discriminated only by `metadata.recorder`.
|
|
7
|
+
*/
|
|
8
|
+
export function isMatchAware(r) {
|
|
9
|
+
return typeof r.getMatch === "function";
|
|
10
|
+
}
|
|
11
|
+
export function isScenarioReporter(r) {
|
|
12
|
+
return typeof r.reportExecution === "function";
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* A recorder that drops everything. Used when auth is unavailable: recording is
|
|
16
|
+
* best-effort, and a session must never fail because BaseIn is unreachable (§10).
|
|
17
|
+
*/
|
|
18
|
+
export class NullRecorder {
|
|
19
|
+
n = 0;
|
|
20
|
+
startRun() {
|
|
21
|
+
return `run_local_${++this.n}`;
|
|
22
|
+
}
|
|
23
|
+
recordToolSelected() {
|
|
24
|
+
return `step_local_${++this.n}`;
|
|
25
|
+
}
|
|
26
|
+
recordToolResponse() {
|
|
27
|
+
return `step_local_${++this.n}`;
|
|
28
|
+
}
|
|
29
|
+
recordFinalAnswer() {
|
|
30
|
+
return `step_local_${++this.n}`;
|
|
31
|
+
}
|
|
32
|
+
finishRun() {
|
|
33
|
+
/* nothing to do */
|
|
34
|
+
}
|
|
35
|
+
async flush() {
|
|
36
|
+
/* nothing to await */
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
//# sourceMappingURL=recorder.js.map
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* redact — scrub secret material before anything leaves the process (Phase 4.3).
|
|
3
|
+
*
|
|
4
|
+
* The proxy sits on the credential path for every wrapped server (§9), and every
|
|
5
|
+
* argument and result it sees crosses a process boundary and then the network.
|
|
6
|
+
* So redaction runs on the recording copy of both arguments *and* results,
|
|
7
|
+
* before the step report is even queued.
|
|
8
|
+
*
|
|
9
|
+
* Three rules, in the order they fire:
|
|
10
|
+
* 1. **By key.** A value under a key matching {@link SECRET_KEY} is replaced
|
|
11
|
+
* wholesale, whatever its type. `authorization: { scheme, token }` loses
|
|
12
|
+
* both halves, because the object as a whole is the secret.
|
|
13
|
+
* 2. **`env` blocks.** An object under an `env` key has *every* value replaced.
|
|
14
|
+
* MCP server configs put credentials there under arbitrary names, so no key
|
|
15
|
+
* pattern would catch them.
|
|
16
|
+
* 3. **By shape.** A string that looks like a bearer token, a JWT, an
|
|
17
|
+
* `sk-`/`ghp_`-style key or an AWS access key id is replaced wherever it
|
|
18
|
+
* appears — including inside free text, where key-matching cannot reach.
|
|
19
|
+
*
|
|
20
|
+
* Redaction never throws and never mutates its input: a cycle, a `BigInt`, a
|
|
21
|
+
* getter that explodes — each degrades to a placeholder rather than losing the
|
|
22
|
+
* step or, worse, taking the session down.
|
|
23
|
+
*/
|
|
24
|
+
export declare const REDACTED = "[redacted]";
|
|
25
|
+
/** Keys whose values are secret regardless of shape. */
|
|
26
|
+
export declare const SECRET_KEY: RegExp;
|
|
27
|
+
/** Replace every secret-shaped run inside a string. */
|
|
28
|
+
export declare function redactText(text: string): string;
|
|
29
|
+
/**
|
|
30
|
+
* Deep-clone `value` with secrets removed. The result is always JSON-safe, so
|
|
31
|
+
* the caller can serialise it without a second failure mode.
|
|
32
|
+
*
|
|
33
|
+
* @param value anything, including cyclic structures.
|
|
34
|
+
* @param maxDepth guard against pathological nesting (default 24).
|
|
35
|
+
*/
|
|
36
|
+
export declare function redact(value: unknown, maxDepth?: number): unknown;
|
|
37
|
+
//# sourceMappingURL=redact.d.ts.map
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* redact — scrub secret material before anything leaves the process (Phase 4.3).
|
|
3
|
+
*
|
|
4
|
+
* The proxy sits on the credential path for every wrapped server (§9), and every
|
|
5
|
+
* argument and result it sees crosses a process boundary and then the network.
|
|
6
|
+
* So redaction runs on the recording copy of both arguments *and* results,
|
|
7
|
+
* before the step report is even queued.
|
|
8
|
+
*
|
|
9
|
+
* Three rules, in the order they fire:
|
|
10
|
+
* 1. **By key.** A value under a key matching {@link SECRET_KEY} is replaced
|
|
11
|
+
* wholesale, whatever its type. `authorization: { scheme, token }` loses
|
|
12
|
+
* both halves, because the object as a whole is the secret.
|
|
13
|
+
* 2. **`env` blocks.** An object under an `env` key has *every* value replaced.
|
|
14
|
+
* MCP server configs put credentials there under arbitrary names, so no key
|
|
15
|
+
* pattern would catch them.
|
|
16
|
+
* 3. **By shape.** A string that looks like a bearer token, a JWT, an
|
|
17
|
+
* `sk-`/`ghp_`-style key or an AWS access key id is replaced wherever it
|
|
18
|
+
* appears — including inside free text, where key-matching cannot reach.
|
|
19
|
+
*
|
|
20
|
+
* Redaction never throws and never mutates its input: a cycle, a `BigInt`, a
|
|
21
|
+
* getter that explodes — each degrades to a placeholder rather than losing the
|
|
22
|
+
* step or, worse, taking the session down.
|
|
23
|
+
*/
|
|
24
|
+
export const REDACTED = "[redacted]";
|
|
25
|
+
/** Keys whose values are secret regardless of shape. */
|
|
26
|
+
export const SECRET_KEY = /(token|secret|password|passwd|api[_-]?key|apikey|authorization|auth|cookie|session[_-]?id|credential|private[_-]?key|access[_-]?key|client[_-]?secret)/i;
|
|
27
|
+
/** Keys whose *children* are all secret (values only — the names stay). */
|
|
28
|
+
const ENV_KEY = /^env(ironment)?$/i;
|
|
29
|
+
/** Value shapes that are secret wherever they appear, including inside prose. */
|
|
30
|
+
const SECRET_SHAPES = [
|
|
31
|
+
/\bBearer\s+[A-Za-z0-9\-._~+/]+=*/gi, // Authorization header value
|
|
32
|
+
/\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]*/g, // JWT
|
|
33
|
+
/\bsk-[A-Za-z0-9_-]{16,}/g, // OpenAI / Anthropic style
|
|
34
|
+
/\bgh[pousr]_[A-Za-z0-9]{16,}/g, // GitHub tokens
|
|
35
|
+
/\bxox[abposr]-[A-Za-z0-9-]{10,}/g, // Slack tokens
|
|
36
|
+
/\bAKIA[0-9A-Z]{16}\b/g, // AWS access key id
|
|
37
|
+
/\b(?:AIza)[0-9A-Za-z_-]{20,}/g, // Google API key
|
|
38
|
+
];
|
|
39
|
+
/** Replace every secret-shaped run inside a string. */
|
|
40
|
+
export function redactText(text) {
|
|
41
|
+
let out = text;
|
|
42
|
+
for (const shape of SECRET_SHAPES) {
|
|
43
|
+
shape.lastIndex = 0;
|
|
44
|
+
out = out.replace(shape, REDACTED);
|
|
45
|
+
}
|
|
46
|
+
return out;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Deep-clone `value` with secrets removed. The result is always JSON-safe, so
|
|
50
|
+
* the caller can serialise it without a second failure mode.
|
|
51
|
+
*
|
|
52
|
+
* @param value anything, including cyclic structures.
|
|
53
|
+
* @param maxDepth guard against pathological nesting (default 24).
|
|
54
|
+
*/
|
|
55
|
+
export function redact(value, maxDepth = 24) {
|
|
56
|
+
return walk(value, undefined, 0, maxDepth, new WeakSet());
|
|
57
|
+
}
|
|
58
|
+
function walk(value, key, depth, maxDepth, seen) {
|
|
59
|
+
if (key !== undefined && SECRET_KEY.test(key))
|
|
60
|
+
return REDACTED;
|
|
61
|
+
if (value === null || value === undefined)
|
|
62
|
+
return value ?? null;
|
|
63
|
+
switch (typeof value) {
|
|
64
|
+
case "string":
|
|
65
|
+
return redactText(value);
|
|
66
|
+
case "number":
|
|
67
|
+
return Number.isFinite(value) ? value : String(value);
|
|
68
|
+
case "boolean":
|
|
69
|
+
return value;
|
|
70
|
+
case "bigint":
|
|
71
|
+
return value.toString();
|
|
72
|
+
case "function":
|
|
73
|
+
case "symbol":
|
|
74
|
+
return `[${typeof value}]`;
|
|
75
|
+
default:
|
|
76
|
+
break;
|
|
77
|
+
}
|
|
78
|
+
if (depth >= maxDepth)
|
|
79
|
+
return "[max depth]";
|
|
80
|
+
const obj = value;
|
|
81
|
+
if (seen.has(obj))
|
|
82
|
+
return "[circular]";
|
|
83
|
+
seen.add(obj);
|
|
84
|
+
try {
|
|
85
|
+
if (Array.isArray(obj)) {
|
|
86
|
+
return obj.map((item) => walk(item, undefined, depth + 1, maxDepth, seen));
|
|
87
|
+
}
|
|
88
|
+
if (obj instanceof Date)
|
|
89
|
+
return obj.toISOString();
|
|
90
|
+
if (obj instanceof Error)
|
|
91
|
+
return { name: obj.name, message: redactText(obj.message) };
|
|
92
|
+
const out = {};
|
|
93
|
+
for (const [k, v] of Object.entries(obj)) {
|
|
94
|
+
if (SECRET_KEY.test(k)) {
|
|
95
|
+
out[k] = REDACTED;
|
|
96
|
+
}
|
|
97
|
+
else if (ENV_KEY.test(k) && v && typeof v === "object" && !Array.isArray(v)) {
|
|
98
|
+
// Rule 2: an env block's names are useful, its values never are.
|
|
99
|
+
const env = {};
|
|
100
|
+
for (const name of Object.keys(v))
|
|
101
|
+
env[name] = REDACTED;
|
|
102
|
+
out[k] = env;
|
|
103
|
+
}
|
|
104
|
+
else {
|
|
105
|
+
out[k] = walk(v, k, depth + 1, maxDepth, seen);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
return out;
|
|
109
|
+
}
|
|
110
|
+
catch {
|
|
111
|
+
// A throwing getter or an exotic proxy. One step's fidelity is not worth a
|
|
112
|
+
// crash on the recording path.
|
|
113
|
+
return "[unreadable]";
|
|
114
|
+
}
|
|
115
|
+
finally {
|
|
116
|
+
seen.delete(obj);
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
//# sourceMappingURL=redact.js.map
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* RemoteRecorder — streams each step to the BaseIn auth-service (Phase 4.2).
|
|
3
|
+
*
|
|
4
|
+
* Ported unchanged in shape from RRepeat's `server/remote-recorder.ts`, so the
|
|
5
|
+
* already-live endpoints need no v1 changes:
|
|
6
|
+
* POST /recordings/runs
|
|
7
|
+
* POST /recordings/runs/:id/steps
|
|
8
|
+
* PATCH /recordings/runs/:id/steps/:stepId
|
|
9
|
+
* POST /recordings/runs/:id/finish
|
|
10
|
+
*
|
|
11
|
+
* Four properties matter, and each is load-bearing:
|
|
12
|
+
* - **Client-generated ids** (`run_`/`step_` uuids) so callers get an id
|
|
13
|
+
* synchronously and never await the network on the hot path.
|
|
14
|
+
* - **One ordered promise chain per run**, so steps arrive in the order they
|
|
15
|
+
* were recorded even though every send is fire-and-forget.
|
|
16
|
+
* - **Best-effort.** A failed send is logged and dropped, never retried inline
|
|
17
|
+
* and never thrown at the caller (§10).
|
|
18
|
+
* - **One token refresh on 401**, then one retry. Sessions outlive access tokens.
|
|
19
|
+
*
|
|
20
|
+
* SIMILAR-PROMPT HITS. `POST /recordings/runs` may answer `{ id: null, matched }`,
|
|
21
|
+
* meaning the server recognised the prompt and did **not** create a run. Every
|
|
22
|
+
* later step post for that id would then 404, so callers must stop recording —
|
|
23
|
+
* {@link RemoteRecorder.getMatch} is how they find out. v1 only stops; replay is v2.
|
|
24
|
+
*/
|
|
25
|
+
import { type AuthSession } from "../auth/client.js";
|
|
26
|
+
import type { ExecutionReport, Recorder, RunMatch, RunMetrics } from "./recorder.js";
|
|
27
|
+
export interface RemoteRecorderOptions {
|
|
28
|
+
baseUrl: string;
|
|
29
|
+
session: AuthSession;
|
|
30
|
+
/** Value for the run's `framework` column. Defaults to `baseinstrunner`. */
|
|
31
|
+
framework?: string;
|
|
32
|
+
}
|
|
33
|
+
export declare class RemoteRecorder implements Recorder {
|
|
34
|
+
private readonly baseUrl;
|
|
35
|
+
private readonly framework;
|
|
36
|
+
private session;
|
|
37
|
+
/** runId → tail of that run's ordered send chain. */
|
|
38
|
+
private readonly chains;
|
|
39
|
+
/** runId → the similar-prompt outcome of its create call. */
|
|
40
|
+
private readonly matches;
|
|
41
|
+
/**
|
|
42
|
+
* Execution reports ride their own chain: a matched turn has no run on the
|
|
43
|
+
* service (that is what a match means), so there is no run chain to append to.
|
|
44
|
+
*/
|
|
45
|
+
private readonly executionChainKey;
|
|
46
|
+
constructor(opts: RemoteRecorderOptions);
|
|
47
|
+
startRun(input: string, metadata?: Record<string, unknown>): string;
|
|
48
|
+
/** Await the similar-prompt match for a run's create call (null = fresh run). */
|
|
49
|
+
getMatch(runId: string): Promise<RunMatch | null>;
|
|
50
|
+
recordToolSelected(runId: string, stepIndex: number, data: {
|
|
51
|
+
toolName: string;
|
|
52
|
+
toolInput: string;
|
|
53
|
+
context?: string;
|
|
54
|
+
}): string;
|
|
55
|
+
recordToolResponse(runId: string, stepIndex: number, data: {
|
|
56
|
+
toolName: string;
|
|
57
|
+
toolOutput?: string;
|
|
58
|
+
toolError?: string;
|
|
59
|
+
}): string;
|
|
60
|
+
recordFinalAnswer(runId: string, stepIndex: number, data: {
|
|
61
|
+
answer: string;
|
|
62
|
+
}): string;
|
|
63
|
+
/** Backfill a step's `context` (the model's reasoning) after the fact. */
|
|
64
|
+
backfillContext(runId: string, stepId: string, context: string): void;
|
|
65
|
+
finishRun(runId: string, finalOutput?: string, metrics?: RunMetrics): void;
|
|
66
|
+
/**
|
|
67
|
+
* Report what a matched turn cost (docs/calculatedReplay.md §11.2).
|
|
68
|
+
*
|
|
69
|
+
* Queued on the run's own chain so it lands after that run's last step and so
|
|
70
|
+
* `flush` waits for it — a report is the last thing a matched turn produces.
|
|
71
|
+
* Best-effort like everything else here: an unredeemed ticket leaves no
|
|
72
|
+
* execution row at all, which loses a saving but can never invent one.
|
|
73
|
+
*
|
|
74
|
+
* A doubled report (a retry, `Stop` racing `SessionEnd`) is safe by
|
|
75
|
+
* construction: the ticket *is* the row's id, so the service books once and
|
|
76
|
+
* answers `200 { duplicate: true }`.
|
|
77
|
+
*/
|
|
78
|
+
reportExecution(report: ExecutionReport): void;
|
|
79
|
+
/** Await every queued send for a run, including its execution report. */
|
|
80
|
+
flush(runId: string): Promise<void>;
|
|
81
|
+
/** Append to this run's ordered chain; failures are logged, not thrown. */
|
|
82
|
+
private enqueue;
|
|
83
|
+
private post;
|
|
84
|
+
private patch;
|
|
85
|
+
private send;
|
|
86
|
+
/**
|
|
87
|
+
* Adopt a newer session written by a sibling process, if there is one.
|
|
88
|
+
*
|
|
89
|
+
* Returns true when this recorder now holds a usable token it did not have to
|
|
90
|
+
* spend a refresh on.
|
|
91
|
+
*/
|
|
92
|
+
private adoptStoredSession;
|
|
93
|
+
/**
|
|
94
|
+
* Silent token refresh.
|
|
95
|
+
*
|
|
96
|
+
* WHY THIS IS NOT JUST A POST. Several processes share one credentials file —
|
|
97
|
+
* `bir-hooks` and one `bir-proxy` per wrapped server — and the service rotates
|
|
98
|
+
* the refresh token on every use. So when two of them hit a 401 together, both
|
|
99
|
+
* refresh, one wins, and the loser's token is now invalid *forever*: it fails,
|
|
100
|
+
* gives up, and silently records nothing for the rest of the session while
|
|
101
|
+
* logging one warning per dropped step. Observed, not hypothetical.
|
|
102
|
+
*
|
|
103
|
+
* The fix is to treat the credentials file as the shared source of truth:
|
|
104
|
+
* check it before spending our token, and check it again if our refresh is
|
|
105
|
+
* rejected, because the process that beat us has already written a good one.
|
|
106
|
+
*/
|
|
107
|
+
private refresh;
|
|
108
|
+
private refreshOverNetwork;
|
|
109
|
+
}
|
|
110
|
+
//# sourceMappingURL=remote-recorder.d.ts.map
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* RemoteRecorder — streams each step to the BaseIn auth-service (Phase 4.2).
|
|
3
|
+
*
|
|
4
|
+
* Ported unchanged in shape from RRepeat's `server/remote-recorder.ts`, so the
|
|
5
|
+
* already-live endpoints need no v1 changes:
|
|
6
|
+
* POST /recordings/runs
|
|
7
|
+
* POST /recordings/runs/:id/steps
|
|
8
|
+
* PATCH /recordings/runs/:id/steps/:stepId
|
|
9
|
+
* POST /recordings/runs/:id/finish
|
|
10
|
+
*
|
|
11
|
+
* Four properties matter, and each is load-bearing:
|
|
12
|
+
* - **Client-generated ids** (`run_`/`step_` uuids) so callers get an id
|
|
13
|
+
* synchronously and never await the network on the hot path.
|
|
14
|
+
* - **One ordered promise chain per run**, so steps arrive in the order they
|
|
15
|
+
* were recorded even though every send is fire-and-forget.
|
|
16
|
+
* - **Best-effort.** A failed send is logged and dropped, never retried inline
|
|
17
|
+
* and never thrown at the caller (§10).
|
|
18
|
+
* - **One token refresh on 401**, then one retry. Sessions outlive access tokens.
|
|
19
|
+
*
|
|
20
|
+
* SIMILAR-PROMPT HITS. `POST /recordings/runs` may answer `{ id: null, matched }`,
|
|
21
|
+
* meaning the server recognised the prompt and did **not** create a run. Every
|
|
22
|
+
* later step post for that id would then 404, so callers must stop recording —
|
|
23
|
+
* {@link RemoteRecorder.getMatch} is how they find out. v1 only stops; replay is v2.
|
|
24
|
+
*/
|
|
25
|
+
import { randomUUID } from "node:crypto";
|
|
26
|
+
import { loadCredentials, saveCredentials } from "../auth/client.js";
|
|
27
|
+
import { logLine, errText } from "../util/log.js";
|
|
28
|
+
export class RemoteRecorder {
|
|
29
|
+
baseUrl;
|
|
30
|
+
framework;
|
|
31
|
+
session;
|
|
32
|
+
/** runId → tail of that run's ordered send chain. */
|
|
33
|
+
chains = new Map();
|
|
34
|
+
/** runId → the similar-prompt outcome of its create call. */
|
|
35
|
+
matches = new Map();
|
|
36
|
+
/**
|
|
37
|
+
* Execution reports ride their own chain: a matched turn has no run on the
|
|
38
|
+
* service (that is what a match means), so there is no run chain to append to.
|
|
39
|
+
*/
|
|
40
|
+
executionChainKey = "__executions__";
|
|
41
|
+
constructor(opts) {
|
|
42
|
+
this.baseUrl = opts.baseUrl.replace(/\/+$/, "");
|
|
43
|
+
this.session = opts.session;
|
|
44
|
+
this.framework = opts.framework ?? "baseinstrunner";
|
|
45
|
+
}
|
|
46
|
+
startRun(input, metadata) {
|
|
47
|
+
const id = "run_" + randomUUID();
|
|
48
|
+
let resolveMatch;
|
|
49
|
+
this.matches.set(id, new Promise((r) => {
|
|
50
|
+
resolveMatch = r;
|
|
51
|
+
}));
|
|
52
|
+
this.enqueue(id, async () => {
|
|
53
|
+
try {
|
|
54
|
+
const body = await this.post("/recordings/runs", {
|
|
55
|
+
id,
|
|
56
|
+
framework: this.framework,
|
|
57
|
+
input,
|
|
58
|
+
metadata,
|
|
59
|
+
});
|
|
60
|
+
const matched = body && typeof body === "object"
|
|
61
|
+
? (body.matched ?? null)
|
|
62
|
+
: null;
|
|
63
|
+
resolveMatch(matched);
|
|
64
|
+
}
|
|
65
|
+
catch (err) {
|
|
66
|
+
resolveMatch(null); // fail-safe: on a create error, record normally
|
|
67
|
+
throw err; // preserve enqueue's dropped-send logging
|
|
68
|
+
}
|
|
69
|
+
});
|
|
70
|
+
return id;
|
|
71
|
+
}
|
|
72
|
+
/** Await the similar-prompt match for a run's create call (null = fresh run). */
|
|
73
|
+
async getMatch(runId) {
|
|
74
|
+
const pending = this.matches.get(runId);
|
|
75
|
+
return pending ? await pending : null;
|
|
76
|
+
}
|
|
77
|
+
recordToolSelected(runId, stepIndex, data) {
|
|
78
|
+
const id = "step_" + randomUUID();
|
|
79
|
+
this.enqueue(runId, () => this.post(`/recordings/runs/${runId}/steps`, {
|
|
80
|
+
id,
|
|
81
|
+
stepIndex,
|
|
82
|
+
type: "tool_selected",
|
|
83
|
+
toolName: data.toolName,
|
|
84
|
+
toolInput: data.toolInput,
|
|
85
|
+
context: data.context,
|
|
86
|
+
}));
|
|
87
|
+
return id;
|
|
88
|
+
}
|
|
89
|
+
recordToolResponse(runId, stepIndex, data) {
|
|
90
|
+
const id = "step_" + randomUUID();
|
|
91
|
+
this.enqueue(runId, () => this.post(`/recordings/runs/${runId}/steps`, {
|
|
92
|
+
id,
|
|
93
|
+
stepIndex,
|
|
94
|
+
type: "tool_response",
|
|
95
|
+
toolName: data.toolName,
|
|
96
|
+
toolOutput: data.toolOutput,
|
|
97
|
+
toolError: data.toolError,
|
|
98
|
+
}));
|
|
99
|
+
return id;
|
|
100
|
+
}
|
|
101
|
+
recordFinalAnswer(runId, stepIndex, data) {
|
|
102
|
+
const id = "step_" + randomUUID();
|
|
103
|
+
this.enqueue(runId, () => this.post(`/recordings/runs/${runId}/steps`, {
|
|
104
|
+
id,
|
|
105
|
+
stepIndex,
|
|
106
|
+
type: "final_answer",
|
|
107
|
+
answer: data.answer,
|
|
108
|
+
}));
|
|
109
|
+
return id;
|
|
110
|
+
}
|
|
111
|
+
/** Backfill a step's `context` (the model's reasoning) after the fact. */
|
|
112
|
+
backfillContext(runId, stepId, context) {
|
|
113
|
+
this.enqueue(runId, () => this.patch(`/recordings/runs/${runId}/steps/${stepId}`, { context }));
|
|
114
|
+
}
|
|
115
|
+
finishRun(runId, finalOutput, metrics) {
|
|
116
|
+
this.enqueue(runId, () => this.post(`/recordings/runs/${runId}/finish`, {
|
|
117
|
+
finalOutput,
|
|
118
|
+
originalCostUsd: metrics?.costUsd,
|
|
119
|
+
originalMs: metrics?.durationMs,
|
|
120
|
+
}));
|
|
121
|
+
}
|
|
122
|
+
/**
|
|
123
|
+
* Report what a matched turn cost (docs/calculatedReplay.md §11.2).
|
|
124
|
+
*
|
|
125
|
+
* Queued on the run's own chain so it lands after that run's last step and so
|
|
126
|
+
* `flush` waits for it — a report is the last thing a matched turn produces.
|
|
127
|
+
* Best-effort like everything else here: an unredeemed ticket leaves no
|
|
128
|
+
* execution row at all, which loses a saving but can never invent one.
|
|
129
|
+
*
|
|
130
|
+
* A doubled report (a retry, `Stop` racing `SessionEnd`) is safe by
|
|
131
|
+
* construction: the ticket *is* the row's id, so the service books once and
|
|
132
|
+
* answers `200 { duplicate: true }`.
|
|
133
|
+
*/
|
|
134
|
+
reportExecution(report) {
|
|
135
|
+
const costUsd = report.deriveCostUsd + report.sessionCostUsd + report.fallbackCostUsd;
|
|
136
|
+
this.enqueue(this.executionChainKey, async () => {
|
|
137
|
+
const body = await this.post(`/scenarios/${report.scenarioId}/executions`, {
|
|
138
|
+
ticket: report.ticket,
|
|
139
|
+
outcome: report.outcome,
|
|
140
|
+
costUsd,
|
|
141
|
+
deriveCostUsd: report.deriveCostUsd,
|
|
142
|
+
sessionCostUsd: report.sessionCostUsd,
|
|
143
|
+
fallbackCostUsd: report.fallbackCostUsd,
|
|
144
|
+
durationMs: report.durationMs,
|
|
145
|
+
stepsPlanned: report.stepsPlanned,
|
|
146
|
+
stepsPinned: report.stepsPinned,
|
|
147
|
+
measured: report.measured,
|
|
148
|
+
pricingVersion: report.pricingVersion,
|
|
149
|
+
prompt: report.prompt,
|
|
150
|
+
// Per-step verdicts and the headline failure (errorshandling.md). All
|
|
151
|
+
// optional on the service side, so an older service ignores them rather
|
|
152
|
+
// than rejecting the report — and the costs still book.
|
|
153
|
+
steps: report.steps,
|
|
154
|
+
error: report.error,
|
|
155
|
+
errorStage: report.errorStage,
|
|
156
|
+
errorStepIndex: report.errorStepIndex,
|
|
157
|
+
errorToolName: report.errorToolName,
|
|
158
|
+
});
|
|
159
|
+
const r = (body ?? {});
|
|
160
|
+
const failed = report.steps?.filter((s) => s.status === "failed").length ?? 0;
|
|
161
|
+
logLine("execution.reported", {
|
|
162
|
+
scenario: report.scenarioId,
|
|
163
|
+
outcome: report.outcome,
|
|
164
|
+
derive: report.deriveCostUsd.toFixed(4),
|
|
165
|
+
session: report.sessionCostUsd.toFixed(4),
|
|
166
|
+
fallback: report.fallbackCostUsd.toFixed(4),
|
|
167
|
+
savedUsd: r.savedUsd ?? undefined,
|
|
168
|
+
duplicate: r.duplicate ? true : undefined,
|
|
169
|
+
measured: report.measured,
|
|
170
|
+
steps: report.steps?.length,
|
|
171
|
+
stepsFailed: failed || undefined,
|
|
172
|
+
});
|
|
173
|
+
});
|
|
174
|
+
}
|
|
175
|
+
/** Await every queued send for a run, including its execution report. */
|
|
176
|
+
async flush(runId) {
|
|
177
|
+
await (this.chains.get(runId) ?? Promise.resolve());
|
|
178
|
+
this.chains.delete(runId);
|
|
179
|
+
this.matches.delete(runId);
|
|
180
|
+
// Reports live on their own chain (a matched run has no run chain of its own
|
|
181
|
+
// — the service created no run), so draining that chain is a separate step.
|
|
182
|
+
await (this.chains.get(this.executionChainKey) ?? Promise.resolve());
|
|
183
|
+
}
|
|
184
|
+
// ── internals ────────────────────────────────────────────────────────────
|
|
185
|
+
/** Append to this run's ordered chain; failures are logged, not thrown. */
|
|
186
|
+
enqueue(runId, fn) {
|
|
187
|
+
const prev = this.chains.get(runId) ?? Promise.resolve();
|
|
188
|
+
const next = prev.then(fn).catch((err) => {
|
|
189
|
+
// A dropped send means this run's recording is incomplete — something an
|
|
190
|
+
// auditor has to be able to see, so it is never verbose-only.
|
|
191
|
+
logLine("recorder.send_failed", {
|
|
192
|
+
run: runId,
|
|
193
|
+
why: "this run's recording is incomplete",
|
|
194
|
+
error: errText(err),
|
|
195
|
+
});
|
|
196
|
+
});
|
|
197
|
+
this.chains.set(runId, next);
|
|
198
|
+
}
|
|
199
|
+
post(path, body) {
|
|
200
|
+
return this.send("POST", path, body);
|
|
201
|
+
}
|
|
202
|
+
patch(path, body) {
|
|
203
|
+
return this.send("PATCH", path, body);
|
|
204
|
+
}
|
|
205
|
+
async send(method, path, body, retried = false) {
|
|
206
|
+
const res = await fetch(`${this.baseUrl}${path}`, {
|
|
207
|
+
method,
|
|
208
|
+
headers: {
|
|
209
|
+
"content-type": "application/json",
|
|
210
|
+
authorization: `Bearer ${this.session.accessToken}`,
|
|
211
|
+
},
|
|
212
|
+
body: JSON.stringify(body),
|
|
213
|
+
});
|
|
214
|
+
if (res.status === 401 && !retried) {
|
|
215
|
+
await this.refresh();
|
|
216
|
+
return this.send(method, path, body, true);
|
|
217
|
+
}
|
|
218
|
+
const text = await res.text().catch(() => "");
|
|
219
|
+
if (!res.ok) {
|
|
220
|
+
throw new Error(`HTTP ${res.status} ${method} ${path}${text ? ` — ${text}` : ""}`);
|
|
221
|
+
}
|
|
222
|
+
if (!text)
|
|
223
|
+
return undefined;
|
|
224
|
+
try {
|
|
225
|
+
return JSON.parse(text);
|
|
226
|
+
}
|
|
227
|
+
catch {
|
|
228
|
+
return undefined;
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
/**
|
|
232
|
+
* Adopt a newer session written by a sibling process, if there is one.
|
|
233
|
+
*
|
|
234
|
+
* Returns true when this recorder now holds a usable token it did not have to
|
|
235
|
+
* spend a refresh on.
|
|
236
|
+
*/
|
|
237
|
+
adoptStoredSession() {
|
|
238
|
+
const stored = loadCredentials();
|
|
239
|
+
if (!stored)
|
|
240
|
+
return false;
|
|
241
|
+
if (stored.accessToken === this.session.accessToken)
|
|
242
|
+
return false;
|
|
243
|
+
if (stored.accessExpiresAt - 30_000 <= Date.now())
|
|
244
|
+
return false;
|
|
245
|
+
this.session = stored;
|
|
246
|
+
logLine("auth.adopted", { why: "another process had already refreshed" });
|
|
247
|
+
return true;
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* Silent token refresh.
|
|
251
|
+
*
|
|
252
|
+
* WHY THIS IS NOT JUST A POST. Several processes share one credentials file —
|
|
253
|
+
* `bir-hooks` and one `bir-proxy` per wrapped server — and the service rotates
|
|
254
|
+
* the refresh token on every use. So when two of them hit a 401 together, both
|
|
255
|
+
* refresh, one wins, and the loser's token is now invalid *forever*: it fails,
|
|
256
|
+
* gives up, and silently records nothing for the rest of the session while
|
|
257
|
+
* logging one warning per dropped step. Observed, not hypothetical.
|
|
258
|
+
*
|
|
259
|
+
* The fix is to treat the credentials file as the shared source of truth:
|
|
260
|
+
* check it before spending our token, and check it again if our refresh is
|
|
261
|
+
* rejected, because the process that beat us has already written a good one.
|
|
262
|
+
*/
|
|
263
|
+
async refresh() {
|
|
264
|
+
if (this.adoptStoredSession())
|
|
265
|
+
return;
|
|
266
|
+
try {
|
|
267
|
+
await this.refreshOverNetwork();
|
|
268
|
+
}
|
|
269
|
+
catch (err) {
|
|
270
|
+
// Someone may have refreshed while we were failing to.
|
|
271
|
+
if (this.adoptStoredSession())
|
|
272
|
+
return;
|
|
273
|
+
throw err;
|
|
274
|
+
}
|
|
275
|
+
}
|
|
276
|
+
async refreshOverNetwork() {
|
|
277
|
+
const res = await fetch(`${this.baseUrl}/auth/refresh`, {
|
|
278
|
+
method: "POST",
|
|
279
|
+
headers: { "content-type": "application/json" },
|
|
280
|
+
body: JSON.stringify({ refreshToken: this.session.refreshToken }),
|
|
281
|
+
});
|
|
282
|
+
if (!res.ok)
|
|
283
|
+
throw new Error(`refresh failed: HTTP ${res.status}`);
|
|
284
|
+
const r = (await res.json());
|
|
285
|
+
this.session = {
|
|
286
|
+
accessToken: r.accessToken,
|
|
287
|
+
refreshToken: r.refreshToken,
|
|
288
|
+
accessExpiresAt: Date.now() + r.expiresIn * 1000,
|
|
289
|
+
user: r.user,
|
|
290
|
+
};
|
|
291
|
+
try {
|
|
292
|
+
saveCredentials(this.session); // persist so the next run reuses it
|
|
293
|
+
}
|
|
294
|
+
catch (err) {
|
|
295
|
+
// Persisting is a convenience for the *next* process. Letting it throw
|
|
296
|
+
// here aborts the retry this refresh exists to enable, and loses the step.
|
|
297
|
+
logLine("auth.save_failed", { error: errText(err), why: "the token still works in-process" });
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
//# sourceMappingURL=remote-recorder.js.map
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* truncate — payload size caps (Phase 4.4).
|
|
3
|
+
*
|
|
4
|
+
* A single `chrome-devtools` screenshot is a megabyte of base64. Left alone it
|
|
5
|
+
* dominates the run, the request body and the corpus, and it is worth nothing to
|
|
6
|
+
* a scenario. So every recorded payload is capped twice:
|
|
7
|
+
*
|
|
8
|
+
* 1. **Per string**, so one enormous leaf cannot hide inside an otherwise small
|
|
9
|
+
* structure and so the rest of that structure still survives intact.
|
|
10
|
+
* 2. **Per payload**, on the serialized form, as the actual guarantee. If the
|
|
11
|
+
* whole thing still exceeds the cap after step 1 it is replaced by an
|
|
12
|
+
* explicit marker rather than silently clipped mid-JSON.
|
|
13
|
+
*
|
|
14
|
+
* Truncation is always **visible**: a truncated string keeps a suffix naming the
|
|
15
|
+
* bytes dropped, and a truncated payload serialises to
|
|
16
|
+
* `{ truncated: true, originalBytes, preview }`. A consumer can tell a short
|
|
17
|
+
* result from a clipped one, which is the whole point.
|
|
18
|
+
*/
|
|
19
|
+
/** 64 KiB per payload, as suggested in Phase 4.4. */
|
|
20
|
+
export declare const DEFAULT_MAX_PAYLOAD_BYTES: number;
|
|
21
|
+
/** 8 KiB per individual string leaf. */
|
|
22
|
+
export declare const DEFAULT_MAX_STRING_BYTES: number;
|
|
23
|
+
export interface TruncateOptions {
|
|
24
|
+
maxPayloadBytes?: number;
|
|
25
|
+
maxStringBytes?: number;
|
|
26
|
+
}
|
|
27
|
+
/** Clip one string to `max` bytes, appending a marker naming what was dropped. */
|
|
28
|
+
export declare function truncateString(text: string, max: number): string;
|
|
29
|
+
/** Deep-copy `value`, clipping every string leaf. Input is never mutated. */
|
|
30
|
+
export declare function truncateStrings(value: unknown, max: number, depth?: number): unknown;
|
|
31
|
+
/**
|
|
32
|
+
* Serialize `value` for recording, capped and with truncation made explicit.
|
|
33
|
+
* Never throws: an unserialisable payload becomes a marker string.
|
|
34
|
+
*/
|
|
35
|
+
export declare function serializeCapped(value: unknown, opts?: TruncateOptions): string;
|
|
36
|
+
//# sourceMappingURL=truncate.d.ts.map
|