@tangle-network/agent-eval 0.139.3 → 0.140.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/analyst/index.d.ts +15 -13
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +6 -6
- package/dist/{benchmark-DDVdWcwA.d.ts → benchmark-DxaZfy0w.d.ts} +3 -3
- package/dist/{benchmark-DDVdWcwA.d.ts.map → benchmark-DxaZfy0w.d.ts.map} +1 -1
- package/dist/{benchmark-command-xvi2liH7.js → benchmark-command-CK0UnXAD.js} +54 -33
- package/dist/benchmark-command-CK0UnXAD.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-zxhy1QV3.js → benchmarks-HwoBE32G.js} +4 -4
- package/dist/{benchmarks-zxhy1QV3.js.map → benchmarks-HwoBE32G.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-DrS6_hLd.js → campaign-BzjYNYVZ.js} +5 -5
- package/dist/{campaign-DrS6_hLd.js.map → campaign-BzjYNYVZ.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-BohnDFBq.d.ts → client-BoqGxEqx.d.ts} +4 -4
- package/dist/{client-BohnDFBq.d.ts.map → client-BoqGxEqx.d.ts.map} +1 -1
- package/dist/{completion-verifier-IPoP4fQO.d.ts → completion-verifier-D15NHYSk.d.ts} +5 -5
- package/dist/{completion-verifier-IPoP4fQO.d.ts.map → completion-verifier-D15NHYSk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +6 -6
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CZ9diLxY.js → cost-ledger-DMFxsLKr.js} +22 -8
- package/dist/cost-ledger-DMFxsLKr.js.map +1 -0
- package/dist/{cost-ledger-DKgyIWRj.d.ts → cost-ledger-FuQvHxPm.d.ts} +8 -2
- package/dist/{cost-ledger-DKgyIWRj.d.ts.map → cost-ledger-FuQvHxPm.d.ts.map} +1 -1
- package/dist/{default-registry-B8vf7Rmf.d.ts → default-registry-Ci7wAAR8.d.ts} +5 -5
- package/dist/{default-registry-B8vf7Rmf.d.ts.map → default-registry-Ci7wAAR8.d.ts.map} +1 -1
- package/dist/{default-registry-BgJJItGr.js → default-registry-DCp-6hc-.js} +3 -3
- package/dist/{default-registry-BgJJItGr.js.map → default-registry-DCp-6hc-.js.map} +1 -1
- package/dist/{dspy-rlm-engine-DTkVyDX-.js → dspy-rlm-engine-Bkak4nzo.js} +11 -4
- package/dist/dspy-rlm-engine-Bkak4nzo.js.map +1 -0
- package/dist/{eval-campaign-BmptJj50.js → eval-campaign-YdkpWWoT.js} +2 -2
- package/dist/{eval-campaign-BmptJj50.js.map → eval-campaign-YdkpWWoT.js.map} +1 -1
- package/dist/{exact-types-MaaFcllV.d.ts → exact-types-B0lJV3tu.d.ts} +2 -2
- package/dist/{exact-types-MaaFcllV.d.ts.map → exact-types-B0lJV3tu.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts → external-optimizer-contracts-nb7c_WAR.d.ts} +12 -2
- package/dist/{external-optimizer-contracts-BrxY2Sli.d.ts.map → external-optimizer-contracts-nb7c_WAR.d.ts.map} +1 -1
- package/dist/{extract-usage-DZs601Va.js → extract-usage-C5vMw-0R.js} +2 -2
- package/dist/{extract-usage-DZs601Va.js.map → extract-usage-C5vMw-0R.js.map} +1 -1
- package/dist/{feedback-trajectory-BJUWOkJM.d.ts → feedback-trajectory-BCHqzLh3.d.ts} +3 -3
- package/dist/{feedback-trajectory-BJUWOkJM.d.ts.map → feedback-trajectory-BCHqzLh3.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-_66rVpwN.d.ts → index-CKI1CXTL.d.ts} +5 -5
- package/dist/{index-_66rVpwN.d.ts.map → index-CKI1CXTL.d.ts.map} +1 -1
- package/dist/{index-CWOPCJiw.d.ts → index-D2enbqA0.d.ts} +2 -2
- package/dist/{index-CWOPCJiw.d.ts.map → index-D2enbqA0.d.ts.map} +1 -1
- package/dist/{index-BTm_P9aC.d.ts → index-DCP4I2Qx.d.ts} +10 -10
- package/dist/{index-BTm_P9aC.d.ts.map → index-DCP4I2Qx.d.ts.map} +1 -1
- package/dist/{index-CtR1xh4V.d.ts → index-DFLVtPZ9.d.ts} +3 -3
- package/dist/{index-CtR1xh4V.d.ts.map → index-DFLVtPZ9.d.ts.map} +1 -1
- package/dist/index.d.ts +23 -23
- package/dist/index.js +13 -13
- package/dist/{insight-report-Bu5Wi9tG.d.ts → insight-report-Bh_8ksel.d.ts} +4 -4
- package/dist/{insight-report-Bu5Wi9tG.d.ts.map → insight-report-Bh_8ksel.d.ts.map} +1 -1
- package/dist/{integrity-COTh3DTH.d.ts → integrity-DRXobPEs.d.ts} +2 -2
- package/dist/{integrity-COTh3DTH.d.ts.map → integrity-DRXobPEs.d.ts.map} +1 -1
- package/dist/{kind-factory-CFxA0JQX.js → kind-factory-DB7nIs35.js} +2 -2
- package/dist/{kind-factory-CFxA0JQX.js.map → kind-factory-DB7nIs35.js.map} +1 -1
- package/dist/{llm-client-bkztEfIx.js → llm-client-B3WXSH5Y.js} +2 -2
- package/dist/{llm-client-bkztEfIx.js.map → llm-client-B3WXSH5Y.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-fZarvIm-.d.ts → release-report-B_bQOHM-.d.ts} +3 -3
- package/dist/{release-report-fZarvIm-.d.ts.map → release-report-B_bQOHM-.d.ts.map} +1 -1
- package/dist/{replay-DjG4IG60.d.ts → replay-BqTgoioO.d.ts} +6 -6
- package/dist/{replay-DjG4IG60.d.ts.map → replay-BqTgoioO.d.ts.map} +1 -1
- package/dist/{replay-SA4OB7O7.js → replay-k2MsOmv5.js} +4 -4
- package/dist/{replay-SA4OB7O7.js.map → replay-k2MsOmv5.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BxhtGfKa.d.ts → researcher-C6lzl-rP.d.ts} +5 -5
- package/dist/{researcher-BxhtGfKa.d.ts.map → researcher-C6lzl-rP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CqSLiV51.d.ts → reward-hacking-CEVVmy3h.d.ts} +2 -2
- package/dist/{reward-hacking-CqSLiV51.d.ts.map → reward-hacking-CEVVmy3h.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts → rubric-predictive-validity-C2CthIfY.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DQBQj6uV.d.ts.map → rubric-predictive-validity-C2CthIfY.d.ts.map} +1 -1
- package/dist/{run-evidence-C4RcRQT5.d.ts → run-evidence-8Ou28QSa.d.ts} +3 -3
- package/dist/{run-evidence-C4RcRQT5.d.ts.map → run-evidence-8Ou28QSa.d.ts.map} +1 -1
- package/dist/{run-record-CztDMXVF.d.ts → run-record-Tb3TTtUn.d.ts} +2 -2
- package/dist/{run-record-CztDMXVF.d.ts.map → run-record-Tb3TTtUn.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-BuIJ9IfB.js → semantic-concept-judge-DJQtFr95.js} +3 -3
- package/dist/{semantic-concept-judge-BuIJ9IfB.js.map → semantic-concept-judge-DJQtFr95.js.map} +1 -1
- package/dist/{server-DaCpLfi0.js → server-Cu4M3NSO.js} +3 -3
- package/dist/{server-DaCpLfi0.js.map → server-Cu4M3NSO.js.map} +1 -1
- package/dist/{single-run-lock-BTTtPZ9N.js → single-run-lock-CiQThJxB.js} +22 -12
- package/dist/single-run-lock-CiQThJxB.js.map +1 -0
- package/dist/{skill-usage-B-BFS8M2.d.ts → skill-usage-CVVnoIx-.d.ts} +26 -10
- package/dist/skill-usage-CVVnoIx-.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-BbGnCC53.js → skillopt-optimization-method-CSBQ8Qma.js} +4 -4
- package/dist/{skillopt-optimization-method-BbGnCC53.js.map → skillopt-optimization-method-CSBQ8Qma.js.map} +1 -1
- package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts → skillopt-optimization-method-D1dqGzzH.d.ts} +11 -11
- package/dist/{skillopt-optimization-method-_s0Tub7Y.d.ts.map → skillopt-optimization-method-D1dqGzzH.d.ts.map} +1 -1
- package/dist/{statistics-B5d0Zd-z.d.ts → statistics-B4u_CiFd.d.ts} +2 -2
- package/dist/{statistics-B5d0Zd-z.d.ts.map → statistics-B4u_CiFd.d.ts.map} +1 -1
- package/dist/{store-otlp-DX4fGIcf.js → store-otlp-vRByAR6h.js} +2 -2
- package/dist/{store-otlp-DX4fGIcf.js.map → store-otlp-vRByAR6h.js.map} +1 -1
- package/dist/{summary-report-Cg7BifAM.d.ts → summary-report-o3eJ3gxG.d.ts} +3 -3
- package/dist/{summary-report-Cg7BifAM.d.ts.map → summary-report-o3eJ3gxG.d.ts.map} +1 -1
- package/dist/{tool-groups-CdYq22lX.d.ts → tool-groups-DVQTy9lq.d.ts} +8 -8
- package/dist/{tool-groups-CdYq22lX.d.ts.map → tool-groups-DVQTy9lq.d.ts.map} +1 -1
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +4 -4
- package/dist/{types-uPrS6mD-.d.ts → types-BjMFz88h.d.ts} +2 -2
- package/dist/{types-uPrS6mD-.d.ts.map → types-BjMFz88h.d.ts.map} +1 -1
- package/dist/{types-DoEYskCd.d.ts → types-D3jh6F98.d.ts} +4 -4
- package/dist/{types-DoEYskCd.d.ts.map → types-D3jh6F98.d.ts.map} +1 -1
- package/dist/{types-BBFNHxSK.d.ts → types-Dk7PB7vh.d.ts} +5 -5
- package/dist/{types-BBFNHxSK.d.ts.map → types-Dk7PB7vh.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/package.json +1 -1
- package/dist/benchmark-command-xvi2liH7.js.map +0 -1
- package/dist/cost-ledger-CZ9diLxY.js.map +0 -1
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +0 -1
- package/dist/single-run-lock-BTTtPZ9N.js.map +0 -1
- package/dist/skill-usage-B-BFS8M2.d.ts.map +0 -1
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { s as resolveModelPricing } from "./metrics-C9YY1OcL.js";
|
|
2
|
-
import { D as RawAnalystFindingSchema } from "./kind-factory-
|
|
3
|
-
import { S as removeCredentialEnvironment, d as runWithCleanup, f as startExternalOptimizerModelProxy, h as sendJson, l as runExternalOptimizerProcess, m as listenLocal, p as closeServer } from "./single-run-lock-
|
|
2
|
+
import { D as RawAnalystFindingSchema } from "./kind-factory-DB7nIs35.js";
|
|
3
|
+
import { S as removeCredentialEnvironment, d as runWithCleanup, f as startExternalOptimizerModelProxy, h as sendJson, l as runExternalOptimizerProcess, m as listenLocal, p as closeServer } from "./single-run-lock-CiQThJxB.js";
|
|
4
4
|
import { randomBytes } from "node:crypto";
|
|
5
5
|
import { createServer } from "node:http";
|
|
6
6
|
//#region src/analyst/trace-tool-callback.ts
|
|
@@ -142,7 +142,7 @@ function isRecord$1(value) {
|
|
|
142
142
|
//#endregion
|
|
143
143
|
//#region src/analyst/dspy-rlm-engine.ts
|
|
144
144
|
const DEFAULT_TIMEOUT_MS = 10 * 6e4;
|
|
145
|
-
const DEFAULT_MODEL_OUTPUT_TOKENS =
|
|
145
|
+
const DEFAULT_MODEL_OUTPUT_TOKENS = 16384;
|
|
146
146
|
const DEFAULT_MAX_COST_USD = 1;
|
|
147
147
|
const MAX_MODEL_REQUEST_BYTES = 16 * 1024 * 1024;
|
|
148
148
|
const MAX_MODEL_RESPONSE_BYTES = 4 * 1024 * 1024;
|
|
@@ -154,6 +154,9 @@ function createDspyRlmTraceEngine(options) {
|
|
|
154
154
|
assertOptions(options);
|
|
155
155
|
const maxOutputTokens = options.maxOutputTokens ?? DEFAULT_MODEL_OUTPUT_TOKENS;
|
|
156
156
|
if (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0) throw new TypeError("DSPy RLM maxOutputTokens must be a positive safe integer");
|
|
157
|
+
const controlAdapter = options.controlAdapter ?? "tolerant";
|
|
158
|
+
const maxReasoningTokens = options.maxReasoningTokens ?? maxOutputTokens * 4;
|
|
159
|
+
if (!Number.isSafeInteger(maxReasoningTokens) || maxReasoningTokens < 0) throw new TypeError("DSPy RLM maxReasoningTokens must be a non-negative safe integer");
|
|
157
160
|
const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
158
161
|
if (!Number.isSafeInteger(timeoutMs) || timeoutMs <= 0) throw new TypeError("DSPy RLM timeoutMs must be a positive safe integer");
|
|
159
162
|
const maxCostUsd = options.maxCostUsd ?? DEFAULT_MAX_COST_USD;
|
|
@@ -173,6 +176,8 @@ function createDspyRlmTraceEngine(options) {
|
|
|
173
176
|
pricing: { ...pricing },
|
|
174
177
|
max_cost_usd: maxCostUsd,
|
|
175
178
|
max_output_tokens: maxOutputTokens,
|
|
179
|
+
max_reasoning_tokens: maxReasoningTokens,
|
|
180
|
+
control_adapter: controlAdapter,
|
|
176
181
|
timeout_ms: timeoutMs,
|
|
177
182
|
max_request_bytes: MAX_MODEL_REQUEST_BYTES,
|
|
178
183
|
max_response_bytes: MAX_MODEL_RESPONSE_BYTES,
|
|
@@ -199,6 +204,7 @@ function createDspyRlmTraceEngine(options) {
|
|
|
199
204
|
maxRequestBytes: MAX_MODEL_REQUEST_BYTES,
|
|
200
205
|
maxResponseBytes: MAX_MODEL_RESPONSE_BYTES,
|
|
201
206
|
maxOutputTokensPerRequest: maxOutputTokens,
|
|
207
|
+
maxReasoningTokensPerRequest: maxReasoningTokens,
|
|
202
208
|
requestTimeoutMs: timeoutMs,
|
|
203
209
|
pricing
|
|
204
210
|
},
|
|
@@ -238,6 +244,7 @@ function createDspyRlmTraceEngine(options) {
|
|
|
238
244
|
description,
|
|
239
245
|
parameters
|
|
240
246
|
})),
|
|
247
|
+
controlAdapter,
|
|
241
248
|
limits: {
|
|
242
249
|
maxIterations: request.limits.maxIterations,
|
|
243
250
|
maxLlmCalls: request.limits.maxLlmCalls,
|
|
@@ -341,4 +348,4 @@ function isRecord(value) {
|
|
|
341
348
|
//#endregion
|
|
342
349
|
export { createDspyRlmTraceEngine as t };
|
|
343
350
|
|
|
344
|
-
//# sourceMappingURL=dspy-rlm-engine-
|
|
351
|
+
//# sourceMappingURL=dspy-rlm-engine-Bkak4nzo.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"dspy-rlm-engine-Bkak4nzo.js","names":["isRecord"],"sources":["../src/analyst/trace-tool-callback.ts","../src/analyst/dspy-rlm-engine.ts"],"sourcesContent":["import { randomBytes } from 'node:crypto'\nimport { createServer, type IncomingMessage, type ServerResponse } from 'node:http'\nimport { closeServer, listenLocal, sendJson } from '../campaign/external-optimizer-http'\nimport type { TraceAnalysisToolDescriptor } from '../trace-analyst/tools'\n\nconst MAX_REQUEST_BYTES = 1_000_000\nconst MAX_RESPONSE_BYTES = 4_000_000\n\nexport interface TraceToolCallback {\n url: string\n token: string\n calls: () => number\n close: () => Promise<void>\n}\n\n/** Expose one bounded trace-tool set only on an authenticated loopback socket. */\nexport async function startTraceToolCallback(args: {\n tools: readonly TraceAnalysisToolDescriptor[]\n maxCalls: number\n signal?: AbortSignal\n}): Promise<TraceToolCallback> {\n if (!Number.isSafeInteger(args.maxCalls) || args.maxCalls <= 0) {\n throw new TypeError('trace tool callback maxCalls must be a positive safe integer')\n }\n args.signal?.throwIfAborted()\n const byName = new Map(args.tools.map((tool) => [tool.name, tool]))\n if (byName.size !== args.tools.length) {\n throw new Error('trace tool callback received duplicate tool names')\n }\n\n const token = randomBytes(32).toString('hex')\n let calls = 0\n let accepting = true\n let closePromise: Promise<void> | undefined\n const activeControllers = new Set<AbortController>()\n const activeHandlers = new Set<Promise<void>>()\n const server = createServer((request, response) => {\n if (!accepting) {\n sendJsonIfOpen(response, 503, { error: 'trace tool callback is closing' })\n return\n }\n const controller = new AbortController()\n const abortRequest = (): void => {\n request.destroy()\n response.destroy()\n }\n activeControllers.add(controller)\n controller.signal.addEventListener('abort', abortRequest, { once: true })\n\n let handler!: Promise<void>\n handler = handleRequest(request, response, controller.signal).finally(() => {\n controller.signal.removeEventListener('abort', abortRequest)\n activeControllers.delete(controller)\n activeHandlers.delete(handler)\n })\n activeHandlers.add(handler)\n void handler.catch(() => undefined)\n })\n const port = await listenLocal(server)\n const close = (): Promise<void> => {\n closePromise ??= closeCallback()\n return closePromise\n }\n const onAbort = (): void => {\n void close().catch(() => undefined)\n }\n args.signal?.addEventListener('abort', onAbort, { once: true })\n if (args.signal?.aborted) onAbort()\n\n return {\n url: `http://127.0.0.1:${port}/call`,\n token,\n calls: () => calls,\n close,\n }\n\n async function handleRequest(\n request: IncomingMessage,\n response: ServerResponse,\n signal: AbortSignal,\n ): Promise<void> {\n try {\n if (request.method !== 'POST' || request.url !== '/call') {\n sendJsonIfOpen(response, 404, { error: 'not found' })\n return\n }\n if (request.headers.authorization !== `Bearer ${token}`) {\n sendJsonIfOpen(response, 401, { error: 'unauthorized' })\n return\n }\n if (calls >= args.maxCalls) {\n sendJsonIfOpen(response, 429, { error: 'trace tool call limit reached' })\n return\n }\n const body = await readJson(request)\n if (!isRecord(body) || typeof body.name !== 'string' || !('args' in body)) {\n sendJsonIfOpen(response, 400, { error: 'name and args are required' })\n return\n }\n const tool = byName.get(body.name)\n if (!tool) {\n sendJsonIfOpen(response, 404, { error: `unknown trace tool '${body.name}'` })\n return\n }\n calls += 1\n const result = await tool.handler(body.args, { signal })\n const encoded = JSON.stringify({ result })\n if (Buffer.byteLength(encoded) > MAX_RESPONSE_BYTES) {\n sendJsonIfOpen(response, 413, { error: 'trace tool response too large' })\n return\n }\n response.writeHead(200, {\n 'content-type': 'application/json; charset=utf-8',\n 'content-length': String(Buffer.byteLength(encoded)),\n })\n response.end(encoded)\n } catch (error) {\n sendJsonIfOpen(response, signal.aborted ? 499 : 400, {\n error: error instanceof Error ? error.message : String(error),\n })\n }\n }\n\n async function closeCallback(): Promise<void> {\n args.signal?.removeEventListener('abort', onAbort)\n accepting = false\n const closingServer = closeServer(server)\n server.closeIdleConnections?.()\n for (const controller of activeControllers) controller.abort()\n const [serverResult] = await Promise.allSettled([\n closingServer,\n waitForActiveHandlers(activeHandlers),\n ])\n if (activeControllers.size !== 0 || activeHandlers.size !== 0) {\n throw new Error('trace tool callback closed with active requests')\n }\n if (serverResult?.status === 'rejected') throw serverResult.reason\n }\n}\n\nasync function waitForActiveHandlers(activeHandlers: Set<Promise<void>>): Promise<void> {\n while (activeHandlers.size > 0) {\n await Promise.allSettled([...activeHandlers])\n }\n}\n\nfunction readJson(request: IncomingMessage): Promise<unknown> {\n return new Promise((resolve, reject) => {\n let size = 0\n const chunks: Buffer[] = []\n request.on('data', (chunk: Buffer) => {\n size += chunk.length\n if (size > MAX_REQUEST_BYTES) {\n reject(new Error('trace tool request too large'))\n request.destroy()\n return\n }\n chunks.push(chunk)\n })\n request.on('error', reject)\n request.on('end', () => {\n try {\n resolve(JSON.parse(Buffer.concat(chunks).toString('utf8')))\n } catch (error) {\n reject(error)\n }\n })\n })\n}\n\nfunction sendJsonIfOpen(response: ServerResponse, status: number, body: unknown): void {\n if (response.destroyed || response.writableEnded) return\n sendJson(response, status, body)\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n","import {\n type ExternalOptimizerModelProxy,\n type ExternalOptimizerRunnerCommand,\n removeCredentialEnvironment,\n} from '../campaign/external-optimizer-contracts'\nimport { startExternalOptimizerModelProxy } from '../campaign/external-optimizer-model-proxy'\nimport { runWithCleanup } from '../campaign/external-optimizer-resources'\nimport { runExternalOptimizerProcess } from '../campaign/external-optimizer-subprocess'\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport { resolveModelPricing } from '../metrics'\nimport type { TraceAnalysisEngine, TraceAnalysisEngineResult } from './engine'\nimport { type RawAnalystFinding, RawAnalystFindingSchema } from './finding-signature'\nimport { startTraceToolCallback } from './trace-tool-callback'\n\nconst DEFAULT_TIMEOUT_MS = 10 * 60_000\n// 4096 is below what current coding models emit for a full findings array:\n// glm-5.2 through an OpenAI-compatible gateway returns 8192 and the request is\n// rejected outright (`502 — provider reported 8192 completion tokens,\n// exceeding requested limit 4096`), so the default failed 2/2 smoke cases\n// before any analysis ran. The cap exists to bound spend, and `maxCostUsd`\n// already does that directly, so it starts above what a real completion needs.\nconst DEFAULT_MODEL_OUTPUT_TOKENS = 16_384\nconst DEFAULT_MAX_COST_USD = 1\nconst MAX_MODEL_REQUEST_BYTES = 16 * 1024 * 1024\nconst MAX_MODEL_RESPONSE_BYTES = 4 * 1024 * 1024\nconst BRIDGE_MODULE = 'agent_eval_rpc.dspy_rlm_bridge'\n/** Bumped whenever this engine's execution behavior changes. */\nconst DSPY_RLM_ENGINE_VERSION = '1.0.0'\n\nexport interface DspyRlmTraceEngineOptions {\n baseUrl: string\n apiKey: string\n model: string\n /** Exact provider rates. Required when the model is absent from the pricing table. */\n pricing?: CustomTokenPricing\n /** Maximum provider spend for one investigation. Default: 1 USD. */\n maxCostUsd?: number\n /** Controller response cap. Default: 16384. */\n maxOutputTokens?: number\n /**\n * Thinking tokens one controller turn may bill on top of its completion.\n * A reasoning model bills these beyond `maxOutputTokens`, so the cost\n * reservation must cover them. Default: four times the completion cap.\n */\n maxReasoningTokens?: number\n /**\n * How the controller's reasoning and code fields are obtained.\n *\n * `tolerant` parses marker output strictly first, then recovers the fields\n * deterministically from prose plus a fenced code block — the shape coding\n * models naturally emit — at no extra model cost. `two-step` extracts with a\n * second call per turn. `chat` accepts marker output only. Default:\n * `tolerant`.\n */\n controlAdapter?: 'chat' | 'two-step' | 'tolerant'\n /** Python command used to load agent-eval-rpc[dspy]. Default: python. */\n runner?: ExternalOptimizerRunnerCommand\n /** Whole investigation deadline. Default: 10 minutes. */\n timeoutMs?: number\n}\n\n/** Use the official DSPy RLM as a bounded recursive trace-analysis engine. */\nexport function createDspyRlmTraceEngine(options: DspyRlmTraceEngineOptions): TraceAnalysisEngine {\n assertOptions(options)\n const maxOutputTokens = options.maxOutputTokens ?? DEFAULT_MODEL_OUTPUT_TOKENS\n if (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0) {\n throw new TypeError('DSPy RLM maxOutputTokens must be a positive safe integer')\n }\n const controlAdapter = options.controlAdapter ?? 'tolerant'\n const maxReasoningTokens = options.maxReasoningTokens ?? maxOutputTokens * 4\n if (!Number.isSafeInteger(maxReasoningTokens) || maxReasoningTokens < 0) {\n throw new TypeError('DSPy RLM maxReasoningTokens must be a non-negative safe integer')\n }\n const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS\n if (!Number.isSafeInteger(timeoutMs) || timeoutMs <= 0) {\n throw new TypeError('DSPy RLM timeoutMs must be a positive safe integer')\n }\n const maxCostUsd = options.maxCostUsd ?? DEFAULT_MAX_COST_USD\n if (!Number.isFinite(maxCostUsd) || maxCostUsd <= 0) {\n throw new TypeError('DSPy RLM maxCostUsd must be positive and finite')\n }\n const pricing = options.pricing ?? pricingForModel(options.model)\n\n const runner = sanitizedRunner(options.runner)\n return {\n id: 'dspy-rlm',\n description: 'Official DSPy RLM with bounded trace tools and metered model calls.',\n model: options.model,\n version: DSPY_RLM_ENGINE_VERSION,\n executionConfig: {\n bridge_module: BRIDGE_MODULE,\n base_url: options.baseUrl,\n model: options.model,\n api_key_provided: true,\n pricing: { ...pricing },\n max_cost_usd: maxCostUsd,\n max_output_tokens: maxOutputTokens,\n max_reasoning_tokens: maxReasoningTokens,\n control_adapter: controlAdapter,\n timeout_ms: timeoutMs,\n max_request_bytes: MAX_MODEL_REQUEST_BYTES,\n max_response_bytes: MAX_MODEL_RESPONSE_BYTES,\n runner: runner ? 'caller-supplied' : 'default',\n runner_command: runner?.command ?? null,\n },\n async analyze(request) {\n const callback = await startTraceToolCallback({\n tools: request.tools,\n maxCalls: request.limits.maxToolCalls,\n ...(request.signal ? { signal: request.signal } : {}),\n })\n let modelProxy: ExternalOptimizerModelProxy | undefined\n const result = await runWithCleanup({\n label: 'DSPy RLM trace-analysis resources',\n run: async () => {\n modelProxy = await startExternalOptimizerModelProxy({\n upstreamBaseUrl: options.baseUrl,\n upstreamApiKey: options.apiKey,\n model: options.model,\n budget: {\n maxCostUsd,\n maxRequests: request.limits.maxIterations + request.limits.maxLlmCalls + 1,\n maxRequestBytes: MAX_MODEL_REQUEST_BYTES,\n maxResponseBytes: MAX_MODEL_RESPONSE_BYTES,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n requestTimeoutMs: timeoutMs,\n pricing,\n },\n costLedger: request.costLedger,\n channel: 'analyst',\n phase: request.costPhase,\n actor: request.analystId,\n ...(request.costTags ? { tags: request.costTags } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n request.log?.('trace analyst engine started', {\n engine: 'dspy-rlm',\n model: options.model,\n tools: request.tools.map((tool) => tool.name),\n limits: request.limits,\n })\n const raw = await runExternalOptimizerProcess<unknown>({\n label: 'DSPy RLM trace analysis',\n tempPrefix: 'agent-eval-dspy-rlm-',\n module: BRIDGE_MODULE,\n input: {\n operation: 'analyze',\n question: request.question,\n instructions: request.instructions,\n modelProxy: {\n baseUrl: modelProxy.baseUrl,\n apiKey: modelProxy.apiKey,\n model: options.model,\n maxOutputTokens,\n },\n toolCallback: {\n url: callback.url,\n token: callback.token,\n },\n toolSpecs: request.tools.map(({ name, description, parameters }) => ({\n name,\n description,\n parameters,\n })),\n controlAdapter,\n limits: {\n maxIterations: request.limits.maxIterations,\n maxLlmCalls: request.limits.maxLlmCalls,\n maxOutputChars: request.limits.maxOutputChars,\n },\n },\n ...(runner ? { runner } : {}),\n timeoutMs,\n ...(request.signal ? { signal: request.signal } : {}),\n })\n const parsed = parseBridgeOutput(raw, (index, reason) => {\n request.log?.('finding rejected: bridge row failed schema validation', {\n engine: 'dspy-rlm',\n index,\n reason,\n })\n })\n const successfulCompletions = modelProxy.successfulCompletions()\n const requestAttempts = modelProxy.requestAttempts()\n if (parsed.modelCalls !== successfulCompletions) {\n throw new Error(\n `DSPy RLM reported ${parsed.modelCalls} model calls, but the provider proxy recorded ${successfulCompletions}`,\n )\n }\n return {\n ...parsed,\n toolCalls: callback.calls(),\n runtime: {\n ...parsed.runtime,\n modelRequestAttempts: requestAttempts,\n modelSuccessfulCompletions: successfulCompletions,\n },\n } satisfies TraceAnalysisEngineResult\n },\n cleanup: async () => {\n const results = await Promise.allSettled([\n ...(modelProxy ? [modelProxy.close()] : []),\n callback.close(),\n ])\n const errors = results.flatMap((entry) =>\n entry.status === 'rejected' ? [entry.reason] : [],\n )\n if (errors.length > 0) {\n throw new AggregateError(errors, 'DSPy RLM resource cleanup failed')\n }\n },\n })\n request.log?.('trace analyst engine completed', {\n engine: 'dspy-rlm',\n model_calls: result.modelCalls,\n model_request_attempts: result.runtime.modelRequestAttempts,\n tool_calls: result.toolCalls,\n findings: result.findings.length,\n })\n return result\n },\n }\n}\n\nfunction parseBridgeOutput(\n value: unknown,\n onRejectedFinding: (index: number, reason: string) => void,\n): Omit<TraceAnalysisEngineResult, 'toolCalls'> {\n if (!isRecord(value)) throw new Error('DSPy RLM bridge output must be an object')\n if (typeof value.answer !== 'string' || !value.answer.trim()) {\n throw new Error('DSPy RLM bridge returned no answer')\n }\n if (!Array.isArray(value.findings)) {\n throw new Error('DSPy RLM bridge findings must be an array')\n }\n // Findings are model output: one malformed row is model noise, not a bridge\n // fault, and the rest of the paid investigation must survive it. Rejected\n // rows are logged per row and counted in runtime.rejectedFindings.\n let rejectedFindings = 0\n const findings: RawAnalystFinding[] = []\n value.findings.forEach((finding, index) => {\n const parsed = RawAnalystFindingSchema.safeParse(finding)\n if (!parsed.success) {\n rejectedFindings += 1\n onRejectedFinding(\n index,\n parsed.error.issues.map((issue) => `${issue.path.join('.')}: ${issue.message}`).join('; '),\n )\n return\n }\n findings.push(parsed.data)\n })\n if (!Array.isArray(value.trajectory)) {\n throw new Error('DSPy RLM bridge trajectory must be an array')\n }\n if (!Number.isSafeInteger(value.modelCalls) || (value.modelCalls as number) <= 0) {\n throw new Error('DSPy RLM bridge modelCalls must be a positive safe integer')\n }\n if (!isRecord(value.runtime)) {\n throw new Error('DSPy RLM bridge runtime must be an object')\n }\n return {\n answer: value.answer,\n findings,\n trajectory: value.trajectory,\n modelCalls: value.modelCalls as number,\n runtime: { ...value.runtime, rejectedFindings },\n }\n}\n\nfunction assertOptions(options: DspyRlmTraceEngineOptions): void {\n for (const [name, value] of [\n ['baseUrl', options.baseUrl],\n ['apiKey', options.apiKey],\n ['model', options.model],\n ] as const) {\n if (typeof value !== 'string' || !value.trim()) {\n throw new TypeError(`DSPy RLM ${name} must be a non-empty string`)\n }\n }\n}\n\nfunction pricingForModel(model: string): CustomTokenPricing {\n const pricing = resolveModelPricing(model)\n if (!pricing) {\n throw new Error(\n `no pricing is configured for '${model}'; provide DspyRlmTraceEngineOptions.pricing`,\n )\n }\n return {\n inputUsdPerMillion: pricing.input * 1_000,\n outputUsdPerMillion: pricing.output * 1_000,\n }\n}\n\nfunction sanitizedRunner(\n runner: ExternalOptimizerRunnerCommand | undefined,\n): ExternalOptimizerRunnerCommand | undefined {\n if (!runner) return undefined\n return {\n ...(runner.command ? { command: runner.command } : {}),\n ...(runner.args ? { args: runner.args } : {}),\n ...(runner.env ? { env: removeCredentialEnvironment(runner.env) } : {}),\n }\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n"],"mappings":";;;;;;AAKA,MAAM,oBAAoB;AAC1B,MAAM,qBAAqB;;AAU3B,eAAsB,uBAAuB,MAId;CAC7B,IAAI,CAAC,OAAO,cAAc,KAAK,QAAQ,KAAK,KAAK,YAAY,GAC3D,MAAM,IAAI,UAAU,8DAA8D;CAEpF,KAAK,QAAQ,eAAe;CAC5B,MAAM,SAAS,IAAI,IAAI,KAAK,MAAM,KAAK,SAAS,CAAC,KAAK,MAAM,IAAI,CAAC,CAAC;CAClE,IAAI,OAAO,SAAS,KAAK,MAAM,QAC7B,MAAM,IAAI,MAAM,mDAAmD;CAGrE,MAAM,QAAQ,YAAY,EAAE,CAAC,CAAC,SAAS,KAAK;CAC5C,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI;CACJ,MAAM,oCAAoB,IAAI,IAAqB;CACnD,MAAM,iCAAiB,IAAI,IAAmB;CAC9C,MAAM,SAAS,cAAc,SAAS,aAAa;EACjD,IAAI,CAAC,WAAW;GACd,eAAe,UAAU,KAAK,EAAE,OAAO,iCAAiC,CAAC;GACzE;EACF;EACA,MAAM,aAAa,IAAI,gBAAgB;EACvC,MAAM,qBAA2B;GAC/B,QAAQ,QAAQ;GAChB,SAAS,QAAQ;EACnB;EACA,kBAAkB,IAAI,UAAU;EAChC,WAAW,OAAO,iBAAiB,SAAS,cAAc,EAAE,MAAM,KAAK,CAAC;EAExE,IAAI;EACJ,UAAU,cAAc,SAAS,UAAU,WAAW,MAAM,CAAC,CAAC,cAAc;GAC1E,WAAW,OAAO,oBAAoB,SAAS,YAAY;GAC3D,kBAAkB,OAAO,UAAU;GACnC,eAAe,OAAO,OAAO;EAC/B,CAAC;EACD,eAAe,IAAI,OAAO;EAC1B,QAAa,YAAY,KAAA,CAAS;CACpC,CAAC;CACD,MAAM,OAAO,MAAM,YAAY,MAAM;CACrC,MAAM,cAA6B;EACjC,iBAAiB,cAAc;EAC/B,OAAO;CACT;CACA,MAAM,gBAAsB;EAC1B,MAAW,CAAC,CAAC,YAAY,KAAA,CAAS;CACpC;CACA,KAAK,QAAQ,iBAAiB,SAAS,SAAS,EAAE,MAAM,KAAK,CAAC;CAC9D,IAAI,KAAK,QAAQ,SAAS,QAAQ;CAElC,OAAO;EACL,KAAK,oBAAoB,KAAK;EAC9B;EACA,aAAa;EACb;CACF;CAEA,eAAe,cACb,SACA,UACA,QACe;EACf,IAAI;GACF,IAAI,QAAQ,WAAW,UAAU,QAAQ,QAAQ,SAAS;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,YAAY,CAAC;IACpD;GACF;GACA,IAAI,QAAQ,QAAQ,kBAAkB,UAAU,SAAS;IACvD,eAAe,UAAU,KAAK,EAAE,OAAO,eAAe,CAAC;IACvD;GACF;GACA,IAAI,SAAS,KAAK,UAAU;IAC1B,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,MAAM,OAAO,MAAM,SAAS,OAAO;GACnC,IAAI,CAACA,WAAS,IAAI,KAAK,OAAO,KAAK,SAAS,YAAY,EAAE,UAAU,OAAO;IACzE,eAAe,UAAU,KAAK,EAAE,OAAO,6BAA6B,CAAC;IACrE;GACF;GACA,MAAM,OAAO,OAAO,IAAI,KAAK,IAAI;GACjC,IAAI,CAAC,MAAM;IACT,eAAe,UAAU,KAAK,EAAE,OAAO,uBAAuB,KAAK,KAAK,GAAG,CAAC;IAC5E;GACF;GACA,SAAS;GACT,MAAM,SAAS,MAAM,KAAK,QAAQ,KAAK,MAAM,EAAE,OAAO,CAAC;GACvD,MAAM,UAAU,KAAK,UAAU,EAAE,OAAO,CAAC;GACzC,IAAI,OAAO,WAAW,OAAO,IAAI,oBAAoB;IACnD,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,SAAS,UAAU,KAAK;IACtB,gBAAgB;IAChB,kBAAkB,OAAO,OAAO,WAAW,OAAO,CAAC;GACrD,CAAC;GACD,SAAS,IAAI,OAAO;EACtB,SAAS,OAAO;GACd,eAAe,UAAU,OAAO,UAAU,MAAM,KAAK,EACnD,OAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,EAC9D,CAAC;EACH;CACF;CAEA,eAAe,gBAA+B;EAC5C,KAAK,QAAQ,oBAAoB,SAAS,OAAO;EACjD,YAAY;EACZ,MAAM,gBAAgB,YAAY,MAAM;EACxC,OAAO,uBAAuB;EAC9B,KAAK,MAAM,cAAc,mBAAmB,WAAW,MAAM;EAC7D,MAAM,CAAC,gBAAgB,MAAM,QAAQ,WAAW,CAC9C,eACA,sBAAsB,cAAc,CACtC,CAAC;EACD,IAAI,kBAAkB,SAAS,KAAK,eAAe,SAAS,GAC1D,MAAM,IAAI,MAAM,iDAAiD;EAEnE,IAAI,cAAc,WAAW,YAAY,MAAM,aAAa;CAC9D;AACF;AAEA,eAAe,sBAAsB,gBAAmD;CACtF,OAAO,eAAe,OAAO,GAC3B,MAAM,QAAQ,WAAW,CAAC,GAAG,cAAc,CAAC;AAEhD;AAEA,SAAS,SAAS,SAA4C;CAC5D,OAAO,IAAI,SAAS,SAAS,WAAW;EACtC,IAAI,OAAO;EACX,MAAM,SAAmB,CAAC;EAC1B,QAAQ,GAAG,SAAS,UAAkB;GACpC,QAAQ,MAAM;GACd,IAAI,OAAO,mBAAmB;IAC5B,uBAAO,IAAI,MAAM,8BAA8B,CAAC;IAChD,QAAQ,QAAQ;IAChB;GACF;GACA,OAAO,KAAK,KAAK;EACnB,CAAC;EACD,QAAQ,GAAG,SAAS,MAAM;EAC1B,QAAQ,GAAG,aAAa;GACtB,IAAI;IACF,QAAQ,KAAK,MAAM,OAAO,OAAO,MAAM,CAAC,CAAC,SAAS,MAAM,CAAC,CAAC;GAC5D,SAAS,OAAO;IACd,OAAO,KAAK;GACd;EACF,CAAC;CACH,CAAC;AACH;AAEA,SAAS,eAAe,UAA0B,QAAgB,MAAqB;CACrF,IAAI,SAAS,aAAa,SAAS,eAAe;CAClD,SAAS,UAAU,QAAQ,IAAI;AACjC;AAEA,SAASA,WAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E;;;ACnKA,MAAM,qBAAqB,KAAK;AAOhC,MAAM,8BAA8B;AACpC,MAAM,uBAAuB;AAC7B,MAAM,0BAA0B,KAAK,OAAO;AAC5C,MAAM,2BAA2B,IAAI,OAAO;AAC5C,MAAM,gBAAgB;;AAEtB,MAAM,0BAA0B;;AAmChC,SAAgB,yBAAyB,SAAyD;CAChG,cAAc,OAAO;CACrB,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,IAAI,CAAC,OAAO,cAAc,eAAe,KAAK,mBAAmB,GAC/D,MAAM,IAAI,UAAU,0DAA0D;CAEhF,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,qBAAqB,QAAQ,sBAAsB,kBAAkB;CAC3E,IAAI,CAAC,OAAO,cAAc,kBAAkB,KAAK,qBAAqB,GACpE,MAAM,IAAI,UAAU,iEAAiE;CAEvF,MAAM,YAAY,QAAQ,aAAa;CACvC,IAAI,CAAC,OAAO,cAAc,SAAS,KAAK,aAAa,GACnD,MAAM,IAAI,UAAU,oDAAoD;CAE1E,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,GAChD,MAAM,IAAI,UAAU,iDAAiD;CAEvE,MAAM,UAAU,QAAQ,WAAW,gBAAgB,QAAQ,KAAK;CAEhE,MAAM,SAAS,gBAAgB,QAAQ,MAAM;CAC7C,OAAO;EACL,IAAI;EACJ,aAAa;EACb,OAAO,QAAQ;EACf,SAAS;EACT,iBAAiB;GACf,eAAe;GACf,UAAU,QAAQ;GAClB,OAAO,QAAQ;GACf,kBAAkB;GAClB,SAAS,EAAE,GAAG,QAAQ;GACtB,cAAc;GACd,mBAAmB;GACnB,sBAAsB;GACtB,iBAAiB;GACjB,YAAY;GACZ,mBAAmB;GACnB,oBAAoB;GACpB,QAAQ,SAAS,oBAAoB;GACrC,gBAAgB,QAAQ,WAAW;EACrC;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,WAAW,MAAM,uBAAuB;IAC5C,OAAO,QAAQ;IACf,UAAU,QAAQ,OAAO;IACzB,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;GACrD,CAAC;GACD,IAAI;GACJ,MAAM,SAAS,MAAM,eAAe;IAClC,OAAO;IACP,KAAK,YAAY;KACf,aAAa,MAAM,iCAAiC;MAClD,iBAAiB,QAAQ;MACzB,gBAAgB,QAAQ;MACxB,OAAO,QAAQ;MACf,QAAQ;OACN;OACA,aAAa,QAAQ,OAAO,gBAAgB,QAAQ,OAAO,cAAc;OACzE,iBAAiB;OACjB,kBAAkB;OAClB,2BAA2B;OAC3B,8BAA8B;OAC9B,kBAAkB;OAClB;MACF;MACA,YAAY,QAAQ;MACpB,SAAS;MACT,OAAO,QAAQ;MACf,OAAO,QAAQ;MACf,GAAI,QAAQ,WAAW,EAAE,MAAM,QAAQ,SAAS,IAAI,CAAC;MACrD,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC;KACD,QAAQ,MAAM,gCAAgC;MAC5C,QAAQ;MACR,OAAO,QAAQ;MACf,OAAO,QAAQ,MAAM,KAAK,SAAS,KAAK,IAAI;MAC5C,QAAQ,QAAQ;KAClB,CAAC;KAmCD,MAAM,SAAS,kBAAkB,MAlCf,4BAAqC;MACrD,OAAO;MACP,YAAY;MACZ,QAAQ;MACR,OAAO;OACL,WAAW;OACX,UAAU,QAAQ;OAClB,cAAc,QAAQ;OACtB,YAAY;QACV,SAAS,WAAW;QACpB,QAAQ,WAAW;QACnB,OAAO,QAAQ;QACf;OACF;OACA,cAAc;QACZ,KAAK,SAAS;QACd,OAAO,SAAS;OAClB;OACA,WAAW,QAAQ,MAAM,KAAK,EAAE,MAAM,aAAa,kBAAkB;QACnE;QACA;QACA;OACF,EAAE;OACF;OACA,QAAQ;QACN,eAAe,QAAQ,OAAO;QAC9B,aAAa,QAAQ,OAAO;QAC5B,gBAAgB,QAAQ,OAAO;OACjC;MACF;MACA,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;MAC3B;MACA,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC,IACsC,OAAO,WAAW;MACvD,QAAQ,MAAM,yDAAyD;OACrE,QAAQ;OACR;OACA;MACF,CAAC;KACH,CAAC;KACD,MAAM,wBAAwB,WAAW,sBAAsB;KAC/D,MAAM,kBAAkB,WAAW,gBAAgB;KACnD,IAAI,OAAO,eAAe,uBACxB,MAAM,IAAI,MACR,qBAAqB,OAAO,WAAW,gDAAgD,uBACzF;KAEF,OAAO;MACL,GAAG;MACH,WAAW,SAAS,MAAM;MAC1B,SAAS;OACP,GAAG,OAAO;OACV,sBAAsB;OACtB,4BAA4B;MAC9B;KACF;IACF;IACA,SAAS,YAAY;KAKnB,MAAM,UAAS,MAJO,QAAQ,WAAW,CACvC,GAAI,aAAa,CAAC,WAAW,MAAM,CAAC,IAAI,CAAC,GACzC,SAAS,MAAM,CACjB,CAAC,EAAA,CACsB,SAAS,UAC9B,MAAM,WAAW,aAAa,CAAC,MAAM,MAAM,IAAI,CAAC,CAClD;KACA,IAAI,OAAO,SAAS,GAClB,MAAM,IAAI,eAAe,QAAQ,kCAAkC;IAEvE;GACF,CAAC;GACD,QAAQ,MAAM,kCAAkC;IAC9C,QAAQ;IACR,aAAa,OAAO;IACpB,wBAAwB,OAAO,QAAQ;IACvC,YAAY,OAAO;IACnB,UAAU,OAAO,SAAS;GAC5B,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,kBACP,OACA,mBAC8C;CAC9C,IAAI,CAAC,SAAS,KAAK,GAAG,MAAM,IAAI,MAAM,0CAA0C;CAChF,IAAI,OAAO,MAAM,WAAW,YAAY,CAAC,MAAM,OAAO,KAAK,GACzD,MAAM,IAAI,MAAM,oCAAoC;CAEtD,IAAI,CAAC,MAAM,QAAQ,MAAM,QAAQ,GAC/B,MAAM,IAAI,MAAM,2CAA2C;CAK7D,IAAI,mBAAmB;CACvB,MAAM,WAAgC,CAAC;CACvC,MAAM,SAAS,SAAS,SAAS,UAAU;EACzC,MAAM,SAAS,wBAAwB,UAAU,OAAO;EACxD,IAAI,CAAC,OAAO,SAAS;GACnB,oBAAoB;GACpB,kBACE,OACA,OAAO,MAAM,OAAO,KAAK,UAAU,GAAG,MAAM,KAAK,KAAK,GAAG,EAAE,IAAI,MAAM,SAAS,CAAC,CAAC,KAAK,IAAI,CAC3F;GACA;EACF;EACA,SAAS,KAAK,OAAO,IAAI;CAC3B,CAAC;CACD,IAAI,CAAC,MAAM,QAAQ,MAAM,UAAU,GACjC,MAAM,IAAI,MAAM,6CAA6C;CAE/D,IAAI,CAAC,OAAO,cAAc,MAAM,UAAU,KAAM,MAAM,cAAyB,GAC7E,MAAM,IAAI,MAAM,4DAA4D;CAE9E,IAAI,CAAC,SAAS,MAAM,OAAO,GACzB,MAAM,IAAI,MAAM,2CAA2C;CAE7D,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM;EAClB,YAAY,MAAM;EAClB,SAAS;GAAE,GAAG,MAAM;GAAS;EAAiB;CAChD;AACF;AAEA,SAAS,cAAc,SAA0C;CAC/D,KAAK,MAAM,CAAC,MAAM,UAAU;EAC1B,CAAC,WAAW,QAAQ,OAAO;EAC3B,CAAC,UAAU,QAAQ,MAAM;EACzB,CAAC,SAAS,QAAQ,KAAK;CACzB,GACE,IAAI,OAAO,UAAU,YAAY,CAAC,MAAM,KAAK,GAC3C,MAAM,IAAI,UAAU,YAAY,KAAK,4BAA4B;AAGvE;AAEA,SAAS,gBAAgB,OAAmC;CAC1D,MAAM,UAAU,oBAAoB,KAAK;CACzC,IAAI,CAAC,SACH,MAAM,IAAI,MACR,iCAAiC,MAAM,6CACzC;CAEF,OAAO;EACL,oBAAoB,QAAQ,QAAQ;EACpC,qBAAqB,QAAQ,SAAS;CACxC;AACF;AAEA,SAAS,gBACP,QAC4C;CAC5C,IAAI,CAAC,QAAQ,OAAO,KAAA;CACpB,OAAO;EACL,GAAI,OAAO,UAAU,EAAE,SAAS,OAAO,QAAQ,IAAI,CAAC;EACpD,GAAI,OAAO,OAAO,EAAE,MAAM,OAAO,KAAK,IAAI,CAAC;EAC3C,GAAI,OAAO,MAAM,EAAE,KAAK,4BAA4B,OAAO,GAAG,EAAE,IAAI,CAAC;CACvE;AACF;AAEA,SAAS,SAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { f as verifyAgentProfileCell, h as hashJson, p as canonicalize, s as buildAgentProfileCell } from "./agent-profile-cell-CbfBm2g6.js";
|
|
2
2
|
import { t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
3
|
-
import { a as assertLlmRoute } from "./llm-client-
|
|
3
|
+
import { a as assertLlmRoute } from "./llm-client-B3WXSH5Y.js";
|
|
4
4
|
import { t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
5
5
|
import { s as validateRunRecord } from "./run-record-vRgqWmJw.js";
|
|
6
6
|
import { i as researchReport } from "./summary-report-9A5y7EsK.js";
|
|
@@ -346,4 +346,4 @@ function defaultRunId(params) {
|
|
|
346
346
|
//#endregion
|
|
347
347
|
export { runEvalCampaign as t };
|
|
348
348
|
|
|
349
|
-
//# sourceMappingURL=eval-campaign-
|
|
349
|
+
//# sourceMappingURL=eval-campaign-YdkpWWoT.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"eval-campaign-BmptJj50.js","names":[],"sources":["../src/eval-campaign.ts"],"sourcesContent":["/**\n * EvalCampaign — opinionated matrix runner that wires the four\n * capture-integrity directives by construction.\n *\n * The canonical benchmark shape — matrix runner → for each\n * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the\n * run → analyze — has a bug class at the integration boundary: raw\n * events not captured, route silently wrong, integrity not asserted,\n * analyst never run. The directives in `SKILL.md § Capture integrity`\n * are the mitigations.\n *\n * `EvalCampaign` is the structural fix — consumers don't wire the\n * integrity surface themselves; the campaign owns it. Specifically:\n *\n * - calls `assertLlmRoute` once at preflight before any work runs\n * - constructs a per-run `TraceStore` and `RawProviderSink` via factories\n * - constructs the `TraceEmitter` with `onRunComplete: [analyst hook]`\n * - hands the runner an `LlmClientOptions` pre-wired with the sink and\n * trace context — the runner can't accidentally call an LLM without\n * capturing the raw HTTP envelope\n * - calls `assertRunCaptured` after every `endRun` and routes failures\n * through a configurable policy (`throw` / `mark_failed` / `log`)\n * - assembles per-run `RunRecord`s and runs `researchReport` at the end\n * so the campaign artifact is launch-decision-grade by default\n * - embeds the campaign fingerprint (a SHA-256 over the canonicalised\n * run set) and optional `preregistrationHash` in the report\n *\n * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`\n * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped\n * lives in the campaign. This is the inversion-of-control point — consumers\n * stop writing matrix runners and start writing scenario-runners.\n *\n * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):\n *\n * - Distributed/cluster execution (concurrency is local async)\n * - Adaptive sampling / sequential interim looks\n * - Resume from partial state across crashes\n * - LLM-call retry beyond what `LlmClient` already does\n */\n\nimport {\n type AgentProfileCell,\n type AgentProfileCellInput,\n buildAgentProfileCell,\n verifyAgentProfileCell,\n} from './agent-profile-cell'\nimport { assertLlmRoute, type LlmClientOptions, type LlmRouteRequirements } from './llm-client'\nimport { canonicalize, hashJson } from './pre-registration'\nimport type {\n JudgeScoresRecord,\n RunCostProvenance,\n RunJudgeMetadata,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTaskFailure,\n RunTokenUsage,\n} from './run-record'\nimport { validateRunRecord } from './run-record'\nimport { type ResearchReport, type ResearchReportOptions, researchReport } from './summary-report'\nimport type { RunCompleteHook } from './trace/emitter'\nimport { TraceEmitter } from './trace/emitter'\nimport {\n assertRunCaptured,\n RunIntegrityError,\n type RunIntegrityExpectations,\n type RunIntegrityReport,\n} from './trace/integrity'\nimport { FileSystemRawProviderSink, type RawProviderSink } from './trace/raw-provider-sink'\nimport type { TraceStore } from './trace/store'\n\n// ── Public types ─────────────────────────────────────────────────────────\n\nexport interface CampaignVariant<V> {\n id: string\n payload: V\n}\n\nexport interface CampaignScenario {\n scenarioId: string\n /** Free-form metadata propagated to runs and reports. */\n tags?: Record<string, string>\n}\n\nexport interface CampaignRunContext<V> {\n /** Stable run id. The campaign generates this; the runner does not. */\n runId: string\n /** Logical experiment id (campaignId by default; overridable per-run via opts). */\n experimentId: string\n variant: V\n variantId: string\n scenarioId: string\n scenarioTags: Record<string, string>\n seed: number\n splitTag: RunSplitTag\n /**\n * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired\n * (analyst auto-execution if configured, plus integrity check). The\n * runner MUST call `emitter.startRun` before doing any work and either\n * `emitter.endRun` or `emitter.abortRun` before returning.\n */\n emitter: TraceEmitter\n store: TraceStore\n rawSink: RawProviderSink\n /**\n * Pre-wired LLM client options — `rawSink` and `traceContext` are populated\n * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The\n * runner can spread additional fields if needed.\n */\n llmOpts: LlmClientOptions\n}\n\ninterface CampaignRunOutcomeFields {\n /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */\n pass: boolean\n /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */\n score: number\n /** Cost in USD, or null when the runner could not capture it. */\n costUsd: number | null\n /** Source of the cost amount. */\n costProvenance: RunCostProvenance\n tokenUsage: RunTokenUsage\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** sha256 of the effective prompt sent to the model. */\n promptHash: string\n /** sha256 of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Optional extra numeric metrics to land in `outcome.raw`. */\n raw?: Record<string, number>\n /** Optional judge metadata when a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /**\n * Optional per-judge / per-dim breakdown for ensemble-judged runs.\n * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.\n * Single-judge or scalar-only runs leave this unset.\n */\n judgeScores?: JudgeScoresRecord\n /**\n * Agent profile cell observed by the runner. When supplied, it overrides\n * `EvalCampaignOptions.agentProfile` for this run and must match the\n * outcome's `model` and `promptHash`.\n */\n agentProfile?: AgentProfileCell | AgentProfileCellInput\n}\n\n/** Campaign result with the same task-failure invariant as `RunRecord`. */\nexport type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure\n\nexport type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>\n\nexport type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log'\n\nexport interface EvalCampaignOptions<V> {\n /**\n * Stable id for the campaign. Used as the default `experimentId` on\n * every run, and folded into the campaign fingerprint.\n */\n campaignId: string\n variants: CampaignVariant<V>[]\n scenarios: CampaignScenario[]\n /** Default `[0, 1, 2]`. */\n seeds?: number[]\n /** Default `'holdout'` — the split that anchors a launch decision. */\n splitTag?: RunSplitTag\n /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */\n commitSha: string\n /**\n * LLM client config. Augmented per-run with `rawSink` and `traceContext`\n * before being passed to the runner. The campaign asserts this config\n * matches `routeRequirements` once at preflight.\n */\n llmOpts: LlmClientOptions\n /**\n * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail\n * loud if the campaign would silently fall back to the public router or\n * run unauthenticated. Override with an empty object to disable.\n */\n routeRequirements?: LlmRouteRequirements\n /**\n * Per-run TraceStore factory. Common shape: a fresh store per run keyed\n * on `runId`. Implementations that share a store across the campaign\n * are valid — the campaign only writes through `emitter`.\n */\n storeFactory: (params: CampaignFactoryParams) => TraceStore\n /**\n * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`\n * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;\n * otherwise required. Forensic capture is non-negotiable in a campaign\n * run — pass `NoopRawProviderSink` explicitly if you want to opt out.\n */\n rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink\n /**\n * Filesystem root for default `rawSinkFactory`. Ignored if\n * `rawSinkFactory` is supplied.\n */\n workDir?: string\n /**\n * Extra `onRunComplete` hooks the campaign appends (after its own\n * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.\n */\n onRunComplete?: RunCompleteHook[]\n /**\n * Per-run integrity expectations. Defaults to:\n * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.\n * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.\n */\n integrity?: RunIntegrityExpectations\n /** Behaviour when integrity fails. Default `'mark_failed'`. */\n onIntegrityFailure?: CampaignIntegrityPolicy\n /**\n * Per-run runner. Receives a fully-wired context; produces an outcome\n * the campaign converts into a `RunRecord`.\n */\n runner: CampaignRunner<V>\n /**\n * If set, the campaign computes `researchReport` at the end. `comparator`\n * is a `variantId`. Other fields are forwarded verbatim.\n */\n report?: { comparator?: string } & Omit<\n ResearchReportOptions,\n 'comparator' | 'preregistrationHash' | 'generatedAt'\n >\n /**\n * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).\n * Embedded in the campaign fingerprint and the research report.\n */\n preregistrationHash?: string\n /** Local concurrency. Default `1` (sequential). */\n concurrency?: number\n /**\n * Override the time source. Tests pass a mock to make wallMs deterministic.\n */\n now?: () => number\n /** Override the runId generator. Tests pin this. */\n runId?: (params: CampaignFactoryParams) => string\n /**\n * Agent profile cell for campaign runs. Static profiles can pass an object;\n * routers or variant-specific harnesses can pass a factory. The campaign\n * stamps the built cell onto every `RunRecord` and rejects profile/model or\n * profile/prompt contradictions.\n */\n agentProfile?:\n | AgentProfileCell\n | AgentProfileCellInput\n | ((\n params: CampaignFactoryParams & {\n variant: V\n scenarioTags: Record<string, string>\n },\n ) =>\n | AgentProfileCell\n | AgentProfileCellInput\n | Promise<AgentProfileCell | AgentProfileCellInput>)\n}\n\nexport interface CampaignFactoryParams {\n campaignId: string\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n}\n\nexport interface FailedRun {\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n reason: string\n error?: string\n}\n\nexport interface EvalCampaignResult {\n campaignId: string\n /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */\n campaignFingerprint: string\n preregistrationHash: string | null\n /** Successful runs only. Failed runs land in `failedRuns`. */\n runs: RunRecord[]\n /** Integrity reports for every successful run. */\n integrityReports: RunIntegrityReport[]\n failedRuns: FailedRun[]\n /** Computed when `report` is set on options. */\n report?: ResearchReport\n startedAt: string\n endedAt: string\n}\n\n// ── Implementation ───────────────────────────────────────────────────────\n\nconst DEFAULT_INTEGRITY: RunIntegrityExpectations = {\n llmSpansMin: 1,\n requireRawCoverageOfLlmSpans: true,\n requireOutcome: true,\n}\n\nconst DEFAULT_ROUTE: LlmRouteRequirements = {\n requireExplicitBaseUrl: true,\n requireAuth: true,\n}\n\nexport async function runEvalCampaign<V>(\n opts: EvalCampaignOptions<V>,\n): Promise<EvalCampaignResult> {\n // ── Preflight ──────────────────────────────────────────────────────\n assertLlmRoute(opts.llmOpts, opts.routeRequirements ?? DEFAULT_ROUTE)\n\n if (opts.variants.length === 0) {\n throw new Error('runEvalCampaign: variants must be non-empty.')\n }\n if (opts.scenarios.length === 0) {\n throw new Error('runEvalCampaign: scenarios must be non-empty.')\n }\n const variantIds = new Set<string>()\n for (const v of opts.variants) {\n if (variantIds.has(v.id)) {\n throw new Error(`runEvalCampaign: duplicate variant id \"${v.id}\".`)\n }\n variantIds.add(v.id)\n }\n const scenarioIds = new Set<string>()\n for (const s of opts.scenarios) {\n if (scenarioIds.has(s.scenarioId)) {\n throw new Error(`runEvalCampaign: duplicate scenarioId \"${s.scenarioId}\".`)\n }\n scenarioIds.add(s.scenarioId)\n }\n if (opts.report?.comparator && !variantIds.has(opts.report.comparator)) {\n throw new Error(\n `runEvalCampaign: report.comparator \"${opts.report.comparator}\" is not a configured variantId.`,\n )\n }\n if (!opts.commitSha) {\n throw new Error('runEvalCampaign: commitSha is required (every RunRecord needs it).')\n }\n\n const seeds = opts.seeds ?? [0, 1, 2]\n const splitTag: RunSplitTag = opts.splitTag ?? 'holdout'\n const concurrency = Math.max(1, opts.concurrency ?? 1)\n const integrity = { ...DEFAULT_INTEGRITY, ...(opts.integrity ?? {}) }\n const onIntegrityFailure: CampaignIntegrityPolicy = opts.onIntegrityFailure ?? 'mark_failed'\n const now = opts.now ?? (() => Date.now())\n const baseUrl = (opts.llmOpts.baseUrl ?? '').replace(/\\/+$/, '')\n const provider = opts.llmOpts.provider ?? null\n const preregistrationHash = opts.preregistrationHash ?? null\n\n const rawSinkFactory = opts.rawSinkFactory ?? defaultRawSinkFactory(opts.workDir)\n\n // ── Fingerprint ────────────────────────────────────────────────────\n const campaignFingerprint = await hashJson(\n canonicalize({\n campaignId: opts.campaignId,\n variants: opts.variants.map((v) => v.id).sort(),\n scenarios: opts.scenarios.map((s) => s.scenarioId).sort(),\n seeds: [...seeds].sort((a, b) => a - b),\n splitTag,\n comparator: opts.report?.comparator ?? null,\n baseUrl,\n provider,\n preregistrationHash,\n }),\n )\n\n // ── Plan the matrix ────────────────────────────────────────────────\n type Cell = { variant: CampaignVariant<V>; scenario: CampaignScenario; seed: number }\n const cells: Cell[] = []\n for (const variant of opts.variants) {\n for (const scenario of opts.scenarios) {\n for (const seed of seeds) {\n cells.push({ variant, scenario, seed })\n }\n }\n }\n\n const startedAt = new Date(now()).toISOString()\n const runs: RunRecord[] = []\n const integrityReports: RunIntegrityReport[] = []\n const failedRuns: FailedRun[] = []\n\n // ── Execute (bounded-concurrency worker pool) ──────────────────────\n // A genuine (non-CellExecutionError) error from any worker is a bug, not a\n // run-level failure. We must NOT let `Promise.all` reject mid-flight and\n // orphan the other workers' in-progress runs (emitters never finalized,\n // sinks/handles leak, partial work silently discarded). Instead: capture the\n // first genuine error, stop dispatching new cells so in-flight workers wind\n // down, finalize every still-open run, then re-throw the aggregated error.\n let cursor = 0\n let aborting = false\n const genuineErrors: unknown[] = []\n // Emitters for runs that are currently mid-flight, keyed by runId. A worker\n // registers its emitter before invoking the runner and removes it once the\n // run is finalized (via endRun/abortRun, success or failure). Anything left\n // here after the pool settles is an orphan we must abort.\n const openRuns = new Map<string, TraceEmitter>()\n\n async function worker(): Promise<void> {\n while (!aborting) {\n const i = cursor++\n if (i >= cells.length) return\n const cell = cells[i]!\n try {\n const result = await runOneCell(cell)\n runs.push(result.record)\n integrityReports.push(result.integrity)\n } catch (err) {\n if (err instanceof CellExecutionError) {\n failedRuns.push(err.failed)\n if (err.integrity) integrityReports.push(err.integrity)\n } else {\n // Genuine bug — not a runner failure, not an integrity failure.\n // Capture it and stop dispatching so peers wind down gracefully;\n // re-thrown after the pool settles. Do not surface here — that would\n // reject Promise.all and orphan the other workers' runs.\n genuineErrors.push(err)\n aborting = true\n return\n }\n }\n }\n }\n\n async function runOneCell(\n cell: Cell,\n ): Promise<{ record: RunRecord; integrity: RunIntegrityReport }> {\n const runId = (opts.runId ?? defaultRunId)({\n campaignId: opts.campaignId,\n runId: '', // unused by default generator\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n })\n const factoryParams: CampaignFactoryParams = {\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n }\n const store = opts.storeFactory(factoryParams)\n const rawSink = rawSinkFactory(factoryParams)\n\n const emitter = new TraceEmitter(store, {\n runId,\n now: opts.now,\n onRunComplete: opts.onRunComplete,\n })\n // Track this run as open so a genuine error elsewhere in the pool can\n // finalize it instead of orphaning it. Removed in the finally below.\n openRuns.set(runId, emitter)\n\n const llmOpts: LlmClientOptions = {\n ...opts.llmOpts,\n rawSink,\n traceContext: { runId },\n }\n\n const ctx: CampaignRunContext<V> = {\n runId,\n experimentId: opts.campaignId,\n variant: cell.variant.payload,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n scenarioTags: cell.scenario.tags ?? {},\n seed: cell.seed,\n splitTag,\n emitter,\n store,\n rawSink,\n llmOpts,\n }\n\n try {\n const wallStart = now()\n let outcome: CampaignRunOutcome\n try {\n outcome = await opts.runner(ctx)\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err)\n // The runner threw mid-execution. Abort the run so the emitter\n // finalizes. The only benign abortRun failure is \"the runner never\n // started the run\" (nothing to finalize). A store-write failure (disk\n // full, FS error) is a genuine diagnostic — surface it rather than\n // masking it as a plain runner failure.\n await finalizeAbort(emitter, runId, message)\n throw new CellExecutionError({\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'runner_threw',\n error: message,\n })\n }\n const wallMs = now() - wallStart\n\n const integrityReport = await assertRunCaptured(store, runId, { ...integrity, rawSink })\n if (!integrityReport.ok) {\n switch (onIntegrityFailure) {\n case 'throw':\n throw new RunIntegrityError(integrityReport)\n case 'mark_failed':\n throw new CellExecutionError(\n {\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'integrity_failed',\n error: integrityReport.issues.map((i) => i.code).join(', '),\n },\n integrityReport,\n )\n case 'log':\n // Caller wants the run admitted with a flagged report; fall through.\n break\n }\n }\n\n const recordOutcome: RunOutcome = {\n raw: outcome.raw ?? {},\n }\n if (splitTag === 'holdout') recordOutcome.holdoutScore = outcome.score\n else recordOutcome.searchScore = outcome.score\n if (outcome.judgeScores !== undefined) recordOutcome.judgeScores = outcome.judgeScores\n\n const record: RunRecord = {\n runId,\n experimentId: opts.campaignId,\n candidateId: cell.variant.id,\n seed: cell.seed,\n model: outcome.model,\n promptHash: outcome.promptHash,\n configHash: outcome.configHash,\n commitSha: opts.commitSha,\n wallMs,\n costUsd: outcome.costUsd,\n costProvenance: outcome.costProvenance,\n tokenUsage: outcome.tokenUsage,\n terminalOutcome: 'succeeded',\n judgeMetadata: outcome.judgeMetadata,\n outcome: recordOutcome,\n ...(outcome.failureClass ? { failureClass: outcome.failureClass } : {}),\n failureMode: outcome.failureMode,\n splitTag,\n scenarioId: cell.scenario.scenarioId,\n }\n const profileSource =\n outcome.agentProfile ??\n (typeof opts.agentProfile === 'function'\n ? await opts.agentProfile({\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n variant: cell.variant.payload,\n scenarioTags: cell.scenario.tags ?? {},\n })\n : opts.agentProfile)\n if (profileSource !== undefined) {\n const agentProfile = await resolveAgentProfileCell(profileSource)\n assertAgentProfileMatchesRun(agentProfile, outcome.model, outcome.promptHash)\n record.agentProfile = agentProfile\n }\n return { record: validateRunRecord(record), integrity: integrityReport }\n } finally {\n // This run's worker has finished with it (success, run-level failure, or\n // genuine error). It is no longer the pool's job to finalize — drop it so\n // the post-settle sweep doesn't double-abort a finalized run.\n openRuns.delete(runId)\n }\n }\n\n const workers = Array.from({ length: Math.min(concurrency, cells.length) }, () => worker())\n // allSettled (not all): a genuine error in one worker must not reject the\n // pool mid-flight and orphan the others. Each worker captures its own\n // genuine error into `genuineErrors` and returns; we re-throw below.\n await Promise.allSettled(workers)\n\n // Finalize any run still open after the pool wound down. With the\n // stop-dispatch flag these are the runs that were mid-flight in peer workers\n // when the first genuine error fired — abort them so their emitters finalize\n // (hooks fire, store records a terminal status) instead of leaking.\n for (const [runId, emitter] of openRuns) {\n await finalizeAbort(emitter, runId, 'campaign aborted: genuine error in a sibling run')\n }\n openRuns.clear()\n\n if (genuineErrors.length > 0) {\n throw genuineErrors.length === 1\n ? genuineErrors[0]\n : new AggregateError(\n genuineErrors,\n `runEvalCampaign: ${genuineErrors.length} runs failed with genuine (non-run-level) errors`,\n )\n }\n\n // ── Optional research report ───────────────────────────────────────\n let report: ResearchReport | undefined\n if (opts.report) {\n const reportOpts: ResearchReportOptions = {\n ...opts.report,\n comparator: opts.report.comparator,\n split: splitTag === 'dev' ? 'search' : splitTag,\n generatedAt: new Date(now()).toISOString(),\n preregistrationHash: preregistrationHash ?? undefined,\n }\n report = await researchReport(runs, reportOpts)\n }\n\n const endedAt = new Date(now()).toISOString()\n\n return {\n campaignId: opts.campaignId,\n campaignFingerprint,\n preregistrationHash,\n runs,\n integrityReports,\n failedRuns,\n report,\n startedAt,\n endedAt,\n }\n}\n\n// ── Internal ─────────────────────────────────────────────────────────────\n\nclass CellExecutionError extends Error {\n readonly failed: FailedRun\n readonly integrity?: RunIntegrityReport\n constructor(failed: FailedRun, integrity?: RunIntegrityReport) {\n super(`cell ${failed.variantId}/${failed.scenarioId}@${failed.seed} failed: ${failed.reason}`)\n this.failed = failed\n this.integrity = integrity\n }\n}\n\n/**\n * Abort a run whose owning work threw or was orphaned by a sibling's genuine\n * error, finalizing the emitter so hooks fire and the store records a terminal\n * status. Safe to call unconditionally: it only aborts runs that are still\n * `running`. Two no-op cases are intentional and benign:\n *\n * - the run was never started (absent from the store) — nothing to finalize\n * - the run is already terminal (`completed` / `failed` / `aborted`) —\n * finalizing again would overwrite the real outcome (e.g. flip a passed run\n * to `aborted` with `{ pass: false, notes: reason }`), so we leave it alone\n *\n * Any OTHER failure of the store read or the abort write (disk full, FS fault,\n * backend down) is a genuine diagnostic and propagates rather than being\n * swallowed.\n */\nexport async function finalizeAbort(\n emitter: TraceEmitter,\n runId: string,\n reason: string,\n): Promise<void> {\n const existing = await emitter.traceStore.getRun(runId)\n if (existing === undefined) return // run never started; nothing to abort\n if (existing.status !== 'running') return // already finalized; never overwrite a real outcome\n await emitter.abortRun(reason)\n}\n\nfunction defaultRawSinkFactory(workDir: string | undefined) {\n return (params: CampaignFactoryParams): RawProviderSink => {\n if (!workDir) {\n throw new Error(\n 'runEvalCampaign: rawSinkFactory not supplied and workDir not set. Pass either to enable raw provider capture, or pass `new NoopRawProviderSink()` via rawSinkFactory to opt out explicitly.',\n )\n }\n return new FileSystemRawProviderSink({\n dir: `${workDir}/raw-events/${params.runId}`,\n })\n }\n}\n\nasync function resolveAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n if (isAgentProfileCell(input)) {\n if (!(await verifyAgentProfileCell(input))) {\n throw new Error(`runEvalCampaign: agentProfile.cellId does not match its content`)\n }\n return input\n }\n return buildAgentProfileCell(input)\n}\n\nfunction isAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): input is AgentProfileCell {\n return 'schemaVersion' in input && 'cellId' in input\n}\n\nfunction assertAgentProfileMatchesRun(\n profile: AgentProfileCell,\n model: string,\n promptHash: string,\n): void {\n if (profile.model !== undefined && profile.model !== model) {\n throw new Error(\n `runEvalCampaign: agentProfile.model \"${profile.model}\" does not match outcome.model \"${model}\"`,\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== promptHash) {\n throw new Error(\n `runEvalCampaign: agentProfile.promptHash \"${profile.promptHash}\" does not match outcome.promptHash \"${promptHash}\"`,\n )\n }\n}\n\nfunction defaultRunId(params: CampaignFactoryParams): string {\n // Stable across re-runs: fingerprint of (campaignId, variantId, scenarioId, seed).\n // Caller can override via opts.runId for non-deterministic IDs.\n const base = `${params.campaignId}::${params.variantId}::${params.scenarioId}::${params.seed}`\n // Lightweight hex: we don't need crypto-grade here, just stability + uniqueness.\n let h1 = 0x811c9dc5\n let h2 = 0x12345678\n for (let i = 0; i < base.length; i++) {\n const c = base.charCodeAt(i)\n h1 = Math.imul(h1 ^ c, 0x01000193) >>> 0\n h2 = Math.imul(h2 ^ c, 0x9e3779b1) >>> 0\n }\n return `run-${h1.toString(16).padStart(8, '0')}${h2.toString(16).padStart(8, '0')}`\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmSA,MAAM,oBAA8C;CAClD,aAAa;CACb,8BAA8B;CAC9B,gBAAgB;AAClB;AAEA,MAAM,gBAAsC;CAC1C,wBAAwB;CACxB,aAAa;AACf;AAEA,eAAsB,gBACpB,MAC6B;CAE7B,eAAe,KAAK,SAAS,KAAK,qBAAqB,aAAa;CAEpE,IAAI,KAAK,SAAS,WAAW,GAC3B,MAAM,IAAI,MAAM,8CAA8C;CAEhE,IAAI,KAAK,UAAU,WAAW,GAC5B,MAAM,IAAI,MAAM,+CAA+C;CAEjE,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,KAAK,UAAU;EAC7B,IAAI,WAAW,IAAI,EAAE,EAAE,GACrB,MAAM,IAAI,MAAM,0CAA0C,EAAE,GAAG,GAAG;EAEpE,WAAW,IAAI,EAAE,EAAE;CACrB;CACA,MAAM,8BAAc,IAAI,IAAY;CACpC,KAAK,MAAM,KAAK,KAAK,WAAW;EAC9B,IAAI,YAAY,IAAI,EAAE,UAAU,GAC9B,MAAM,IAAI,MAAM,0CAA0C,EAAE,WAAW,GAAG;EAE5E,YAAY,IAAI,EAAE,UAAU;CAC9B;CACA,IAAI,KAAK,QAAQ,cAAc,CAAC,WAAW,IAAI,KAAK,OAAO,UAAU,GACnE,MAAM,IAAI,MACR,uCAAuC,KAAK,OAAO,WAAW,iCAChE;CAEF,IAAI,CAAC,KAAK,WACR,MAAM,IAAI,MAAM,oEAAoE;CAGtF,MAAM,QAAQ,KAAK,SAAS;EAAC;EAAG;EAAG;CAAC;CACpC,MAAM,WAAwB,KAAK,YAAY;CAC/C,MAAM,cAAc,KAAK,IAAI,GAAG,KAAK,eAAe,CAAC;CACrD,MAAM,YAAY;EAAE,GAAG;EAAmB,GAAI,KAAK,aAAa,CAAC;CAAG;CACpE,MAAM,qBAA8C,KAAK,sBAAsB;CAC/E,MAAM,MAAM,KAAK,cAAc,KAAK,IAAI;CACxC,MAAM,WAAW,KAAK,QAAQ,WAAW,GAAA,CAAI,QAAQ,QAAQ,EAAE;CAC/D,MAAM,WAAW,KAAK,QAAQ,YAAY;CAC1C,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,MAAM,iBAAiB,KAAK,kBAAkB,sBAAsB,KAAK,OAAO;CAGhF,MAAM,sBAAsB,MAAM,SAChC,aAAa;EACX,YAAY,KAAK;EACjB,UAAU,KAAK,SAAS,KAAK,MAAM,EAAE,EAAE,CAAC,CAAC,KAAK;EAC9C,WAAW,KAAK,UAAU,KAAK,MAAM,EAAE,UAAU,CAAC,CAAC,KAAK;EACxD,OAAO,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EACtC;EACA,YAAY,KAAK,QAAQ,cAAc;EACvC;EACA;EACA;CACF,CAAC,CACH;CAIA,MAAM,QAAgB,CAAC;CACvB,KAAK,MAAM,WAAW,KAAK,UACzB,KAAK,MAAM,YAAY,KAAK,WAC1B,KAAK,MAAM,QAAQ,OACjB,MAAM,KAAK;EAAE;EAAS;EAAU;CAAK,CAAC;CAK5C,MAAM,YAAY,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAC9C,MAAM,OAAoB,CAAC;CAC3B,MAAM,mBAAyC,CAAC;CAChD,MAAM,aAA0B,CAAC;CASjC,IAAI,SAAS;CACb,IAAI,WAAW;CACf,MAAM,gBAA2B,CAAC;CAKlC,MAAM,2BAAW,IAAI,IAA0B;CAE/C,eAAe,SAAwB;EACrC,OAAO,CAAC,UAAU;GAChB,MAAM,IAAI;GACV,IAAI,KAAK,MAAM,QAAQ;GACvB,MAAM,OAAO,MAAM;GACnB,IAAI;IACF,MAAM,SAAS,MAAM,WAAW,IAAI;IACpC,KAAK,KAAK,OAAO,MAAM;IACvB,iBAAiB,KAAK,OAAO,SAAS;GACxC,SAAS,KAAK;IACZ,IAAI,eAAe,oBAAoB;KACrC,WAAW,KAAK,IAAI,MAAM;KAC1B,IAAI,IAAI,WAAW,iBAAiB,KAAK,IAAI,SAAS;IACxD,OAAO;KAKL,cAAc,KAAK,GAAG;KACtB,WAAW;KACX;IACF;GACF;EACF;CACF;CAEA,eAAe,WACb,MAC+D;EAC/D,MAAM,SAAS,KAAK,SAAS,aAAA,CAAc;GACzC,YAAY,KAAK;GACjB,OAAO;GACP,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb,CAAC;EACD,MAAM,gBAAuC;GAC3C,YAAY,KAAK;GACjB;GACA,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb;EACA,MAAM,QAAQ,KAAK,aAAa,aAAa;EAC7C,MAAM,UAAU,eAAe,aAAa;EAE5C,MAAM,UAAU,IAAI,aAAa,OAAO;GACtC;GACA,KAAK,KAAK;GACV,eAAe,KAAK;EACtB,CAAC;EAGD,SAAS,IAAI,OAAO,OAAO;EAE3B,MAAM,UAA4B;GAChC,GAAG,KAAK;GACR;GACA,cAAc,EAAE,MAAM;EACxB;EAEA,MAAM,MAA6B;GACjC;GACA,cAAc,KAAK;GACnB,SAAS,KAAK,QAAQ;GACtB,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,cAAc,KAAK,SAAS,QAAQ,CAAC;GACrC,MAAM,KAAK;GACX;GACA;GACA;GACA;GACA;EACF;EAEA,IAAI;GACF,MAAM,YAAY,IAAI;GACtB,IAAI;GACJ,IAAI;IACF,UAAU,MAAM,KAAK,OAAO,GAAG;GACjC,SAAS,KAAK;IACZ,MAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAM/D,MAAM,cAAc,SAAS,OAAO,OAAO;IAC3C,MAAM,IAAI,mBAAmB;KAC3B;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO;IACT,CAAC;GACH;GACA,MAAM,SAAS,IAAI,IAAI;GAEvB,MAAM,kBAAkB,MAAM,kBAAkB,OAAO,OAAO;IAAE,GAAG;IAAW;GAAQ,CAAC;GACvF,IAAI,CAAC,gBAAgB,IACnB,QAAQ,oBAAR;IACE,KAAK,SACH,MAAM,IAAI,kBAAkB,eAAe;IAC7C,KAAK,eACH,MAAM,IAAI,mBACR;KACE;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO,gBAAgB,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,IAAI;IAC5D,GACA,eACF;IACF,KAAK,OAEH;GACJ;GAGF,MAAM,gBAA4B,EAChC,KAAK,QAAQ,OAAO,CAAC,EACvB;GACA,IAAI,aAAa,WAAW,cAAc,eAAe,QAAQ;QAC5D,cAAc,cAAc,QAAQ;GACzC,IAAI,QAAQ,gBAAgB,KAAA,GAAW,cAAc,cAAc,QAAQ;GAE3E,MAAM,SAAoB;IACxB;IACA,cAAc,KAAK;IACnB,aAAa,KAAK,QAAQ;IAC1B,MAAM,KAAK;IACX,OAAO,QAAQ;IACf,YAAY,QAAQ;IACpB,YAAY,QAAQ;IACpB,WAAW,KAAK;IAChB;IACA,SAAS,QAAQ;IACjB,gBAAgB,QAAQ;IACxB,YAAY,QAAQ;IACpB,iBAAiB;IACjB,eAAe,QAAQ;IACvB,SAAS;IACT,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;IACrE,aAAa,QAAQ;IACrB;IACA,YAAY,KAAK,SAAS;GAC5B;GACA,MAAM,gBACJ,QAAQ,iBACP,OAAO,KAAK,iBAAiB,aAC1B,MAAM,KAAK,aAAa;IACtB,YAAY,KAAK;IACjB;IACA,WAAW,KAAK,QAAQ;IACxB,YAAY,KAAK,SAAS;IAC1B,MAAM,KAAK;IACX,SAAS,KAAK,QAAQ;IACtB,cAAc,KAAK,SAAS,QAAQ,CAAC;GACvC,CAAC,IACD,KAAK;GACX,IAAI,kBAAkB,KAAA,GAAW;IAC/B,MAAM,eAAe,MAAM,wBAAwB,aAAa;IAChE,6BAA6B,cAAc,QAAQ,OAAO,QAAQ,UAAU;IAC5E,OAAO,eAAe;GACxB;GACA,OAAO;IAAE,QAAQ,kBAAkB,MAAM;IAAG,WAAW;GAAgB;EACzE,UAAU;GAIR,SAAS,OAAO,KAAK;EACvB;CACF;CAEA,MAAM,UAAU,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,MAAM,MAAM,EAAE,SAAS,OAAO,CAAC;CAI1F,MAAM,QAAQ,WAAW,OAAO;CAMhC,KAAK,MAAM,CAAC,OAAO,YAAY,UAC7B,MAAM,cAAc,SAAS,OAAO,kDAAkD;CAExF,SAAS,MAAM;CAEf,IAAI,cAAc,SAAS,GACzB,MAAM,cAAc,WAAW,IAC3B,cAAc,KACd,IAAI,eACF,eACA,oBAAoB,cAAc,OAAO,iDAC3C;CAIN,IAAI;CACJ,IAAI,KAAK,QAQP,SAAS,MAAM,eAAe,MAAM;EANlC,GAAG,KAAK;EACR,YAAY,KAAK,OAAO;EACxB,OAAO,aAAa,QAAQ,WAAW;EACvC,aAAa,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;EACzC,qBAAqB,uBAAuB,KAAA;CAED,CAAC;CAGhD,MAAM,UAAU,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAE5C,OAAO;EACL,YAAY,KAAK;EACjB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAIA,IAAM,qBAAN,cAAiC,MAAM;CACrC;CACA;CACA,YAAY,QAAmB,WAAgC;EAC7D,MAAM,QAAQ,OAAO,UAAU,GAAG,OAAO,WAAW,GAAG,OAAO,KAAK,WAAW,OAAO,QAAQ;EAC7F,KAAK,SAAS;EACd,KAAK,YAAY;CACnB;AACF;;;;;;;;;;;;;;;;AAiBA,eAAsB,cACpB,SACA,OACA,QACe;CACf,MAAM,WAAW,MAAM,QAAQ,WAAW,OAAO,KAAK;CACtD,IAAI,aAAa,KAAA,GAAW;CAC5B,IAAI,SAAS,WAAW,WAAW;CACnC,MAAM,QAAQ,SAAS,MAAM;AAC/B;AAEA,SAAS,sBAAsB,SAA6B;CAC1D,QAAQ,WAAmD;EACzD,IAAI,CAAC,SACH,MAAM,IAAI,MACR,6LACF;EAEF,OAAO,IAAI,0BAA0B,EACnC,KAAK,GAAG,QAAQ,cAAc,OAAO,QACvC,CAAC;CACH;AACF;AAEA,eAAe,wBACb,OAC2B;CAC3B,IAAI,mBAAmB,KAAK,GAAG;EAC7B,IAAI,CAAE,MAAM,uBAAuB,KAAK,GACtC,MAAM,IAAI,MAAM,iEAAiE;EAEnF,OAAO;CACT;CACA,OAAO,sBAAsB,KAAK;AACpC;AAEA,SAAS,mBACP,OAC2B;CAC3B,OAAO,mBAAmB,SAAS,YAAY;AACjD;AAEA,SAAS,6BACP,SACA,OACA,YACM;CACN,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OACnD,MAAM,IAAI,MACR,wCAAwC,QAAQ,MAAM,kCAAkC,MAAM,EAChG;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,YAC7D,MAAM,IAAI,MACR,6CAA6C,QAAQ,WAAW,uCAAuC,WAAW,EACpH;AAEJ;AAEA,SAAS,aAAa,QAAuC;CAG3D,MAAM,OAAO,GAAG,OAAO,WAAW,IAAI,OAAO,UAAU,IAAI,OAAO,WAAW,IAAI,OAAO;CAExF,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,KAAK,WAAW,CAAC;EAC3B,KAAK,KAAK,KAAK,KAAK,GAAG,QAAU,MAAM;EACvC,KAAK,KAAK,KAAK,KAAK,GAAG,UAAU,MAAM;CACzC;CACA,OAAO,OAAO,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG,IAAI,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG;AAClF"}
|
|
1
|
+
{"version":3,"file":"eval-campaign-YdkpWWoT.js","names":[],"sources":["../src/eval-campaign.ts"],"sourcesContent":["/**\n * EvalCampaign — opinionated matrix runner that wires the four\n * capture-integrity directives by construction.\n *\n * The canonical benchmark shape — matrix runner → for each\n * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the\n * run → analyze — has a bug class at the integration boundary: raw\n * events not captured, route silently wrong, integrity not asserted,\n * analyst never run. The directives in `SKILL.md § Capture integrity`\n * are the mitigations.\n *\n * `EvalCampaign` is the structural fix — consumers don't wire the\n * integrity surface themselves; the campaign owns it. Specifically:\n *\n * - calls `assertLlmRoute` once at preflight before any work runs\n * - constructs a per-run `TraceStore` and `RawProviderSink` via factories\n * - constructs the `TraceEmitter` with `onRunComplete: [analyst hook]`\n * - hands the runner an `LlmClientOptions` pre-wired with the sink and\n * trace context — the runner can't accidentally call an LLM without\n * capturing the raw HTTP envelope\n * - calls `assertRunCaptured` after every `endRun` and routes failures\n * through a configurable policy (`throw` / `mark_failed` / `log`)\n * - assembles per-run `RunRecord`s and runs `researchReport` at the end\n * so the campaign artifact is launch-decision-grade by default\n * - embeds the campaign fingerprint (a SHA-256 over the canonicalised\n * run set) and optional `preregistrationHash` in the report\n *\n * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`\n * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped\n * lives in the campaign. This is the inversion-of-control point — consumers\n * stop writing matrix runners and start writing scenario-runners.\n *\n * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):\n *\n * - Distributed/cluster execution (concurrency is local async)\n * - Adaptive sampling / sequential interim looks\n * - Resume from partial state across crashes\n * - LLM-call retry beyond what `LlmClient` already does\n */\n\nimport {\n type AgentProfileCell,\n type AgentProfileCellInput,\n buildAgentProfileCell,\n verifyAgentProfileCell,\n} from './agent-profile-cell'\nimport { assertLlmRoute, type LlmClientOptions, type LlmRouteRequirements } from './llm-client'\nimport { canonicalize, hashJson } from './pre-registration'\nimport type {\n JudgeScoresRecord,\n RunCostProvenance,\n RunJudgeMetadata,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTaskFailure,\n RunTokenUsage,\n} from './run-record'\nimport { validateRunRecord } from './run-record'\nimport { type ResearchReport, type ResearchReportOptions, researchReport } from './summary-report'\nimport type { RunCompleteHook } from './trace/emitter'\nimport { TraceEmitter } from './trace/emitter'\nimport {\n assertRunCaptured,\n RunIntegrityError,\n type RunIntegrityExpectations,\n type RunIntegrityReport,\n} from './trace/integrity'\nimport { FileSystemRawProviderSink, type RawProviderSink } from './trace/raw-provider-sink'\nimport type { TraceStore } from './trace/store'\n\n// ── Public types ─────────────────────────────────────────────────────────\n\nexport interface CampaignVariant<V> {\n id: string\n payload: V\n}\n\nexport interface CampaignScenario {\n scenarioId: string\n /** Free-form metadata propagated to runs and reports. */\n tags?: Record<string, string>\n}\n\nexport interface CampaignRunContext<V> {\n /** Stable run id. The campaign generates this; the runner does not. */\n runId: string\n /** Logical experiment id (campaignId by default; overridable per-run via opts). */\n experimentId: string\n variant: V\n variantId: string\n scenarioId: string\n scenarioTags: Record<string, string>\n seed: number\n splitTag: RunSplitTag\n /**\n * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired\n * (analyst auto-execution if configured, plus integrity check). The\n * runner MUST call `emitter.startRun` before doing any work and either\n * `emitter.endRun` or `emitter.abortRun` before returning.\n */\n emitter: TraceEmitter\n store: TraceStore\n rawSink: RawProviderSink\n /**\n * Pre-wired LLM client options — `rawSink` and `traceContext` are populated\n * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The\n * runner can spread additional fields if needed.\n */\n llmOpts: LlmClientOptions\n}\n\ninterface CampaignRunOutcomeFields {\n /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */\n pass: boolean\n /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */\n score: number\n /** Cost in USD, or null when the runner could not capture it. */\n costUsd: number | null\n /** Source of the cost amount. */\n costProvenance: RunCostProvenance\n tokenUsage: RunTokenUsage\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** sha256 of the effective prompt sent to the model. */\n promptHash: string\n /** sha256 of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Optional extra numeric metrics to land in `outcome.raw`. */\n raw?: Record<string, number>\n /** Optional judge metadata when a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /**\n * Optional per-judge / per-dim breakdown for ensemble-judged runs.\n * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.\n * Single-judge or scalar-only runs leave this unset.\n */\n judgeScores?: JudgeScoresRecord\n /**\n * Agent profile cell observed by the runner. When supplied, it overrides\n * `EvalCampaignOptions.agentProfile` for this run and must match the\n * outcome's `model` and `promptHash`.\n */\n agentProfile?: AgentProfileCell | AgentProfileCellInput\n}\n\n/** Campaign result with the same task-failure invariant as `RunRecord`. */\nexport type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure\n\nexport type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>\n\nexport type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log'\n\nexport interface EvalCampaignOptions<V> {\n /**\n * Stable id for the campaign. Used as the default `experimentId` on\n * every run, and folded into the campaign fingerprint.\n */\n campaignId: string\n variants: CampaignVariant<V>[]\n scenarios: CampaignScenario[]\n /** Default `[0, 1, 2]`. */\n seeds?: number[]\n /** Default `'holdout'` — the split that anchors a launch decision. */\n splitTag?: RunSplitTag\n /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */\n commitSha: string\n /**\n * LLM client config. Augmented per-run with `rawSink` and `traceContext`\n * before being passed to the runner. The campaign asserts this config\n * matches `routeRequirements` once at preflight.\n */\n llmOpts: LlmClientOptions\n /**\n * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail\n * loud if the campaign would silently fall back to the public router or\n * run unauthenticated. Override with an empty object to disable.\n */\n routeRequirements?: LlmRouteRequirements\n /**\n * Per-run TraceStore factory. Common shape: a fresh store per run keyed\n * on `runId`. Implementations that share a store across the campaign\n * are valid — the campaign only writes through `emitter`.\n */\n storeFactory: (params: CampaignFactoryParams) => TraceStore\n /**\n * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`\n * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;\n * otherwise required. Forensic capture is non-negotiable in a campaign\n * run — pass `NoopRawProviderSink` explicitly if you want to opt out.\n */\n rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink\n /**\n * Filesystem root for default `rawSinkFactory`. Ignored if\n * `rawSinkFactory` is supplied.\n */\n workDir?: string\n /**\n * Extra `onRunComplete` hooks the campaign appends (after its own\n * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.\n */\n onRunComplete?: RunCompleteHook[]\n /**\n * Per-run integrity expectations. Defaults to:\n * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.\n * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.\n */\n integrity?: RunIntegrityExpectations\n /** Behaviour when integrity fails. Default `'mark_failed'`. */\n onIntegrityFailure?: CampaignIntegrityPolicy\n /**\n * Per-run runner. Receives a fully-wired context; produces an outcome\n * the campaign converts into a `RunRecord`.\n */\n runner: CampaignRunner<V>\n /**\n * If set, the campaign computes `researchReport` at the end. `comparator`\n * is a `variantId`. Other fields are forwarded verbatim.\n */\n report?: { comparator?: string } & Omit<\n ResearchReportOptions,\n 'comparator' | 'preregistrationHash' | 'generatedAt'\n >\n /**\n * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).\n * Embedded in the campaign fingerprint and the research report.\n */\n preregistrationHash?: string\n /** Local concurrency. Default `1` (sequential). */\n concurrency?: number\n /**\n * Override the time source. Tests pass a mock to make wallMs deterministic.\n */\n now?: () => number\n /** Override the runId generator. Tests pin this. */\n runId?: (params: CampaignFactoryParams) => string\n /**\n * Agent profile cell for campaign runs. Static profiles can pass an object;\n * routers or variant-specific harnesses can pass a factory. The campaign\n * stamps the built cell onto every `RunRecord` and rejects profile/model or\n * profile/prompt contradictions.\n */\n agentProfile?:\n | AgentProfileCell\n | AgentProfileCellInput\n | ((\n params: CampaignFactoryParams & {\n variant: V\n scenarioTags: Record<string, string>\n },\n ) =>\n | AgentProfileCell\n | AgentProfileCellInput\n | Promise<AgentProfileCell | AgentProfileCellInput>)\n}\n\nexport interface CampaignFactoryParams {\n campaignId: string\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n}\n\nexport interface FailedRun {\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n reason: string\n error?: string\n}\n\nexport interface EvalCampaignResult {\n campaignId: string\n /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */\n campaignFingerprint: string\n preregistrationHash: string | null\n /** Successful runs only. Failed runs land in `failedRuns`. */\n runs: RunRecord[]\n /** Integrity reports for every successful run. */\n integrityReports: RunIntegrityReport[]\n failedRuns: FailedRun[]\n /** Computed when `report` is set on options. */\n report?: ResearchReport\n startedAt: string\n endedAt: string\n}\n\n// ── Implementation ───────────────────────────────────────────────────────\n\nconst DEFAULT_INTEGRITY: RunIntegrityExpectations = {\n llmSpansMin: 1,\n requireRawCoverageOfLlmSpans: true,\n requireOutcome: true,\n}\n\nconst DEFAULT_ROUTE: LlmRouteRequirements = {\n requireExplicitBaseUrl: true,\n requireAuth: true,\n}\n\nexport async function runEvalCampaign<V>(\n opts: EvalCampaignOptions<V>,\n): Promise<EvalCampaignResult> {\n // ── Preflight ──────────────────────────────────────────────────────\n assertLlmRoute(opts.llmOpts, opts.routeRequirements ?? DEFAULT_ROUTE)\n\n if (opts.variants.length === 0) {\n throw new Error('runEvalCampaign: variants must be non-empty.')\n }\n if (opts.scenarios.length === 0) {\n throw new Error('runEvalCampaign: scenarios must be non-empty.')\n }\n const variantIds = new Set<string>()\n for (const v of opts.variants) {\n if (variantIds.has(v.id)) {\n throw new Error(`runEvalCampaign: duplicate variant id \"${v.id}\".`)\n }\n variantIds.add(v.id)\n }\n const scenarioIds = new Set<string>()\n for (const s of opts.scenarios) {\n if (scenarioIds.has(s.scenarioId)) {\n throw new Error(`runEvalCampaign: duplicate scenarioId \"${s.scenarioId}\".`)\n }\n scenarioIds.add(s.scenarioId)\n }\n if (opts.report?.comparator && !variantIds.has(opts.report.comparator)) {\n throw new Error(\n `runEvalCampaign: report.comparator \"${opts.report.comparator}\" is not a configured variantId.`,\n )\n }\n if (!opts.commitSha) {\n throw new Error('runEvalCampaign: commitSha is required (every RunRecord needs it).')\n }\n\n const seeds = opts.seeds ?? [0, 1, 2]\n const splitTag: RunSplitTag = opts.splitTag ?? 'holdout'\n const concurrency = Math.max(1, opts.concurrency ?? 1)\n const integrity = { ...DEFAULT_INTEGRITY, ...(opts.integrity ?? {}) }\n const onIntegrityFailure: CampaignIntegrityPolicy = opts.onIntegrityFailure ?? 'mark_failed'\n const now = opts.now ?? (() => Date.now())\n const baseUrl = (opts.llmOpts.baseUrl ?? '').replace(/\\/+$/, '')\n const provider = opts.llmOpts.provider ?? null\n const preregistrationHash = opts.preregistrationHash ?? null\n\n const rawSinkFactory = opts.rawSinkFactory ?? defaultRawSinkFactory(opts.workDir)\n\n // ── Fingerprint ────────────────────────────────────────────────────\n const campaignFingerprint = await hashJson(\n canonicalize({\n campaignId: opts.campaignId,\n variants: opts.variants.map((v) => v.id).sort(),\n scenarios: opts.scenarios.map((s) => s.scenarioId).sort(),\n seeds: [...seeds].sort((a, b) => a - b),\n splitTag,\n comparator: opts.report?.comparator ?? null,\n baseUrl,\n provider,\n preregistrationHash,\n }),\n )\n\n // ── Plan the matrix ────────────────────────────────────────────────\n type Cell = { variant: CampaignVariant<V>; scenario: CampaignScenario; seed: number }\n const cells: Cell[] = []\n for (const variant of opts.variants) {\n for (const scenario of opts.scenarios) {\n for (const seed of seeds) {\n cells.push({ variant, scenario, seed })\n }\n }\n }\n\n const startedAt = new Date(now()).toISOString()\n const runs: RunRecord[] = []\n const integrityReports: RunIntegrityReport[] = []\n const failedRuns: FailedRun[] = []\n\n // ── Execute (bounded-concurrency worker pool) ──────────────────────\n // A genuine (non-CellExecutionError) error from any worker is a bug, not a\n // run-level failure. We must NOT let `Promise.all` reject mid-flight and\n // orphan the other workers' in-progress runs (emitters never finalized,\n // sinks/handles leak, partial work silently discarded). Instead: capture the\n // first genuine error, stop dispatching new cells so in-flight workers wind\n // down, finalize every still-open run, then re-throw the aggregated error.\n let cursor = 0\n let aborting = false\n const genuineErrors: unknown[] = []\n // Emitters for runs that are currently mid-flight, keyed by runId. A worker\n // registers its emitter before invoking the runner and removes it once the\n // run is finalized (via endRun/abortRun, success or failure). Anything left\n // here after the pool settles is an orphan we must abort.\n const openRuns = new Map<string, TraceEmitter>()\n\n async function worker(): Promise<void> {\n while (!aborting) {\n const i = cursor++\n if (i >= cells.length) return\n const cell = cells[i]!\n try {\n const result = await runOneCell(cell)\n runs.push(result.record)\n integrityReports.push(result.integrity)\n } catch (err) {\n if (err instanceof CellExecutionError) {\n failedRuns.push(err.failed)\n if (err.integrity) integrityReports.push(err.integrity)\n } else {\n // Genuine bug — not a runner failure, not an integrity failure.\n // Capture it and stop dispatching so peers wind down gracefully;\n // re-thrown after the pool settles. Do not surface here — that would\n // reject Promise.all and orphan the other workers' runs.\n genuineErrors.push(err)\n aborting = true\n return\n }\n }\n }\n }\n\n async function runOneCell(\n cell: Cell,\n ): Promise<{ record: RunRecord; integrity: RunIntegrityReport }> {\n const runId = (opts.runId ?? defaultRunId)({\n campaignId: opts.campaignId,\n runId: '', // unused by default generator\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n })\n const factoryParams: CampaignFactoryParams = {\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n }\n const store = opts.storeFactory(factoryParams)\n const rawSink = rawSinkFactory(factoryParams)\n\n const emitter = new TraceEmitter(store, {\n runId,\n now: opts.now,\n onRunComplete: opts.onRunComplete,\n })\n // Track this run as open so a genuine error elsewhere in the pool can\n // finalize it instead of orphaning it. Removed in the finally below.\n openRuns.set(runId, emitter)\n\n const llmOpts: LlmClientOptions = {\n ...opts.llmOpts,\n rawSink,\n traceContext: { runId },\n }\n\n const ctx: CampaignRunContext<V> = {\n runId,\n experimentId: opts.campaignId,\n variant: cell.variant.payload,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n scenarioTags: cell.scenario.tags ?? {},\n seed: cell.seed,\n splitTag,\n emitter,\n store,\n rawSink,\n llmOpts,\n }\n\n try {\n const wallStart = now()\n let outcome: CampaignRunOutcome\n try {\n outcome = await opts.runner(ctx)\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err)\n // The runner threw mid-execution. Abort the run so the emitter\n // finalizes. The only benign abortRun failure is \"the runner never\n // started the run\" (nothing to finalize). A store-write failure (disk\n // full, FS error) is a genuine diagnostic — surface it rather than\n // masking it as a plain runner failure.\n await finalizeAbort(emitter, runId, message)\n throw new CellExecutionError({\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'runner_threw',\n error: message,\n })\n }\n const wallMs = now() - wallStart\n\n const integrityReport = await assertRunCaptured(store, runId, { ...integrity, rawSink })\n if (!integrityReport.ok) {\n switch (onIntegrityFailure) {\n case 'throw':\n throw new RunIntegrityError(integrityReport)\n case 'mark_failed':\n throw new CellExecutionError(\n {\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'integrity_failed',\n error: integrityReport.issues.map((i) => i.code).join(', '),\n },\n integrityReport,\n )\n case 'log':\n // Caller wants the run admitted with a flagged report; fall through.\n break\n }\n }\n\n const recordOutcome: RunOutcome = {\n raw: outcome.raw ?? {},\n }\n if (splitTag === 'holdout') recordOutcome.holdoutScore = outcome.score\n else recordOutcome.searchScore = outcome.score\n if (outcome.judgeScores !== undefined) recordOutcome.judgeScores = outcome.judgeScores\n\n const record: RunRecord = {\n runId,\n experimentId: opts.campaignId,\n candidateId: cell.variant.id,\n seed: cell.seed,\n model: outcome.model,\n promptHash: outcome.promptHash,\n configHash: outcome.configHash,\n commitSha: opts.commitSha,\n wallMs,\n costUsd: outcome.costUsd,\n costProvenance: outcome.costProvenance,\n tokenUsage: outcome.tokenUsage,\n terminalOutcome: 'succeeded',\n judgeMetadata: outcome.judgeMetadata,\n outcome: recordOutcome,\n ...(outcome.failureClass ? { failureClass: outcome.failureClass } : {}),\n failureMode: outcome.failureMode,\n splitTag,\n scenarioId: cell.scenario.scenarioId,\n }\n const profileSource =\n outcome.agentProfile ??\n (typeof opts.agentProfile === 'function'\n ? await opts.agentProfile({\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n variant: cell.variant.payload,\n scenarioTags: cell.scenario.tags ?? {},\n })\n : opts.agentProfile)\n if (profileSource !== undefined) {\n const agentProfile = await resolveAgentProfileCell(profileSource)\n assertAgentProfileMatchesRun(agentProfile, outcome.model, outcome.promptHash)\n record.agentProfile = agentProfile\n }\n return { record: validateRunRecord(record), integrity: integrityReport }\n } finally {\n // This run's worker has finished with it (success, run-level failure, or\n // genuine error). It is no longer the pool's job to finalize — drop it so\n // the post-settle sweep doesn't double-abort a finalized run.\n openRuns.delete(runId)\n }\n }\n\n const workers = Array.from({ length: Math.min(concurrency, cells.length) }, () => worker())\n // allSettled (not all): a genuine error in one worker must not reject the\n // pool mid-flight and orphan the others. Each worker captures its own\n // genuine error into `genuineErrors` and returns; we re-throw below.\n await Promise.allSettled(workers)\n\n // Finalize any run still open after the pool wound down. With the\n // stop-dispatch flag these are the runs that were mid-flight in peer workers\n // when the first genuine error fired — abort them so their emitters finalize\n // (hooks fire, store records a terminal status) instead of leaking.\n for (const [runId, emitter] of openRuns) {\n await finalizeAbort(emitter, runId, 'campaign aborted: genuine error in a sibling run')\n }\n openRuns.clear()\n\n if (genuineErrors.length > 0) {\n throw genuineErrors.length === 1\n ? genuineErrors[0]\n : new AggregateError(\n genuineErrors,\n `runEvalCampaign: ${genuineErrors.length} runs failed with genuine (non-run-level) errors`,\n )\n }\n\n // ── Optional research report ───────────────────────────────────────\n let report: ResearchReport | undefined\n if (opts.report) {\n const reportOpts: ResearchReportOptions = {\n ...opts.report,\n comparator: opts.report.comparator,\n split: splitTag === 'dev' ? 'search' : splitTag,\n generatedAt: new Date(now()).toISOString(),\n preregistrationHash: preregistrationHash ?? undefined,\n }\n report = await researchReport(runs, reportOpts)\n }\n\n const endedAt = new Date(now()).toISOString()\n\n return {\n campaignId: opts.campaignId,\n campaignFingerprint,\n preregistrationHash,\n runs,\n integrityReports,\n failedRuns,\n report,\n startedAt,\n endedAt,\n }\n}\n\n// ── Internal ─────────────────────────────────────────────────────────────\n\nclass CellExecutionError extends Error {\n readonly failed: FailedRun\n readonly integrity?: RunIntegrityReport\n constructor(failed: FailedRun, integrity?: RunIntegrityReport) {\n super(`cell ${failed.variantId}/${failed.scenarioId}@${failed.seed} failed: ${failed.reason}`)\n this.failed = failed\n this.integrity = integrity\n }\n}\n\n/**\n * Abort a run whose owning work threw or was orphaned by a sibling's genuine\n * error, finalizing the emitter so hooks fire and the store records a terminal\n * status. Safe to call unconditionally: it only aborts runs that are still\n * `running`. Two no-op cases are intentional and benign:\n *\n * - the run was never started (absent from the store) — nothing to finalize\n * - the run is already terminal (`completed` / `failed` / `aborted`) —\n * finalizing again would overwrite the real outcome (e.g. flip a passed run\n * to `aborted` with `{ pass: false, notes: reason }`), so we leave it alone\n *\n * Any OTHER failure of the store read or the abort write (disk full, FS fault,\n * backend down) is a genuine diagnostic and propagates rather than being\n * swallowed.\n */\nexport async function finalizeAbort(\n emitter: TraceEmitter,\n runId: string,\n reason: string,\n): Promise<void> {\n const existing = await emitter.traceStore.getRun(runId)\n if (existing === undefined) return // run never started; nothing to abort\n if (existing.status !== 'running') return // already finalized; never overwrite a real outcome\n await emitter.abortRun(reason)\n}\n\nfunction defaultRawSinkFactory(workDir: string | undefined) {\n return (params: CampaignFactoryParams): RawProviderSink => {\n if (!workDir) {\n throw new Error(\n 'runEvalCampaign: rawSinkFactory not supplied and workDir not set. Pass either to enable raw provider capture, or pass `new NoopRawProviderSink()` via rawSinkFactory to opt out explicitly.',\n )\n }\n return new FileSystemRawProviderSink({\n dir: `${workDir}/raw-events/${params.runId}`,\n })\n }\n}\n\nasync function resolveAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n if (isAgentProfileCell(input)) {\n if (!(await verifyAgentProfileCell(input))) {\n throw new Error(`runEvalCampaign: agentProfile.cellId does not match its content`)\n }\n return input\n }\n return buildAgentProfileCell(input)\n}\n\nfunction isAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): input is AgentProfileCell {\n return 'schemaVersion' in input && 'cellId' in input\n}\n\nfunction assertAgentProfileMatchesRun(\n profile: AgentProfileCell,\n model: string,\n promptHash: string,\n): void {\n if (profile.model !== undefined && profile.model !== model) {\n throw new Error(\n `runEvalCampaign: agentProfile.model \"${profile.model}\" does not match outcome.model \"${model}\"`,\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== promptHash) {\n throw new Error(\n `runEvalCampaign: agentProfile.promptHash \"${profile.promptHash}\" does not match outcome.promptHash \"${promptHash}\"`,\n )\n }\n}\n\nfunction defaultRunId(params: CampaignFactoryParams): string {\n // Stable across re-runs: fingerprint of (campaignId, variantId, scenarioId, seed).\n // Caller can override via opts.runId for non-deterministic IDs.\n const base = `${params.campaignId}::${params.variantId}::${params.scenarioId}::${params.seed}`\n // Lightweight hex: we don't need crypto-grade here, just stability + uniqueness.\n let h1 = 0x811c9dc5\n let h2 = 0x12345678\n for (let i = 0; i < base.length; i++) {\n const c = base.charCodeAt(i)\n h1 = Math.imul(h1 ^ c, 0x01000193) >>> 0\n h2 = Math.imul(h2 ^ c, 0x9e3779b1) >>> 0\n }\n return `run-${h1.toString(16).padStart(8, '0')}${h2.toString(16).padStart(8, '0')}`\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmSA,MAAM,oBAA8C;CAClD,aAAa;CACb,8BAA8B;CAC9B,gBAAgB;AAClB;AAEA,MAAM,gBAAsC;CAC1C,wBAAwB;CACxB,aAAa;AACf;AAEA,eAAsB,gBACpB,MAC6B;CAE7B,eAAe,KAAK,SAAS,KAAK,qBAAqB,aAAa;CAEpE,IAAI,KAAK,SAAS,WAAW,GAC3B,MAAM,IAAI,MAAM,8CAA8C;CAEhE,IAAI,KAAK,UAAU,WAAW,GAC5B,MAAM,IAAI,MAAM,+CAA+C;CAEjE,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,KAAK,UAAU;EAC7B,IAAI,WAAW,IAAI,EAAE,EAAE,GACrB,MAAM,IAAI,MAAM,0CAA0C,EAAE,GAAG,GAAG;EAEpE,WAAW,IAAI,EAAE,EAAE;CACrB;CACA,MAAM,8BAAc,IAAI,IAAY;CACpC,KAAK,MAAM,KAAK,KAAK,WAAW;EAC9B,IAAI,YAAY,IAAI,EAAE,UAAU,GAC9B,MAAM,IAAI,MAAM,0CAA0C,EAAE,WAAW,GAAG;EAE5E,YAAY,IAAI,EAAE,UAAU;CAC9B;CACA,IAAI,KAAK,QAAQ,cAAc,CAAC,WAAW,IAAI,KAAK,OAAO,UAAU,GACnE,MAAM,IAAI,MACR,uCAAuC,KAAK,OAAO,WAAW,iCAChE;CAEF,IAAI,CAAC,KAAK,WACR,MAAM,IAAI,MAAM,oEAAoE;CAGtF,MAAM,QAAQ,KAAK,SAAS;EAAC;EAAG;EAAG;CAAC;CACpC,MAAM,WAAwB,KAAK,YAAY;CAC/C,MAAM,cAAc,KAAK,IAAI,GAAG,KAAK,eAAe,CAAC;CACrD,MAAM,YAAY;EAAE,GAAG;EAAmB,GAAI,KAAK,aAAa,CAAC;CAAG;CACpE,MAAM,qBAA8C,KAAK,sBAAsB;CAC/E,MAAM,MAAM,KAAK,cAAc,KAAK,IAAI;CACxC,MAAM,WAAW,KAAK,QAAQ,WAAW,GAAA,CAAI,QAAQ,QAAQ,EAAE;CAC/D,MAAM,WAAW,KAAK,QAAQ,YAAY;CAC1C,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,MAAM,iBAAiB,KAAK,kBAAkB,sBAAsB,KAAK,OAAO;CAGhF,MAAM,sBAAsB,MAAM,SAChC,aAAa;EACX,YAAY,KAAK;EACjB,UAAU,KAAK,SAAS,KAAK,MAAM,EAAE,EAAE,CAAC,CAAC,KAAK;EAC9C,WAAW,KAAK,UAAU,KAAK,MAAM,EAAE,UAAU,CAAC,CAAC,KAAK;EACxD,OAAO,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EACtC;EACA,YAAY,KAAK,QAAQ,cAAc;EACvC;EACA;EACA;CACF,CAAC,CACH;CAIA,MAAM,QAAgB,CAAC;CACvB,KAAK,MAAM,WAAW,KAAK,UACzB,KAAK,MAAM,YAAY,KAAK,WAC1B,KAAK,MAAM,QAAQ,OACjB,MAAM,KAAK;EAAE;EAAS;EAAU;CAAK,CAAC;CAK5C,MAAM,YAAY,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAC9C,MAAM,OAAoB,CAAC;CAC3B,MAAM,mBAAyC,CAAC;CAChD,MAAM,aAA0B,CAAC;CASjC,IAAI,SAAS;CACb,IAAI,WAAW;CACf,MAAM,gBAA2B,CAAC;CAKlC,MAAM,2BAAW,IAAI,IAA0B;CAE/C,eAAe,SAAwB;EACrC,OAAO,CAAC,UAAU;GAChB,MAAM,IAAI;GACV,IAAI,KAAK,MAAM,QAAQ;GACvB,MAAM,OAAO,MAAM;GACnB,IAAI;IACF,MAAM,SAAS,MAAM,WAAW,IAAI;IACpC,KAAK,KAAK,OAAO,MAAM;IACvB,iBAAiB,KAAK,OAAO,SAAS;GACxC,SAAS,KAAK;IACZ,IAAI,eAAe,oBAAoB;KACrC,WAAW,KAAK,IAAI,MAAM;KAC1B,IAAI,IAAI,WAAW,iBAAiB,KAAK,IAAI,SAAS;IACxD,OAAO;KAKL,cAAc,KAAK,GAAG;KACtB,WAAW;KACX;IACF;GACF;EACF;CACF;CAEA,eAAe,WACb,MAC+D;EAC/D,MAAM,SAAS,KAAK,SAAS,aAAA,CAAc;GACzC,YAAY,KAAK;GACjB,OAAO;GACP,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb,CAAC;EACD,MAAM,gBAAuC;GAC3C,YAAY,KAAK;GACjB;GACA,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb;EACA,MAAM,QAAQ,KAAK,aAAa,aAAa;EAC7C,MAAM,UAAU,eAAe,aAAa;EAE5C,MAAM,UAAU,IAAI,aAAa,OAAO;GACtC;GACA,KAAK,KAAK;GACV,eAAe,KAAK;EACtB,CAAC;EAGD,SAAS,IAAI,OAAO,OAAO;EAE3B,MAAM,UAA4B;GAChC,GAAG,KAAK;GACR;GACA,cAAc,EAAE,MAAM;EACxB;EAEA,MAAM,MAA6B;GACjC;GACA,cAAc,KAAK;GACnB,SAAS,KAAK,QAAQ;GACtB,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,cAAc,KAAK,SAAS,QAAQ,CAAC;GACrC,MAAM,KAAK;GACX;GACA;GACA;GACA;GACA;EACF;EAEA,IAAI;GACF,MAAM,YAAY,IAAI;GACtB,IAAI;GACJ,IAAI;IACF,UAAU,MAAM,KAAK,OAAO,GAAG;GACjC,SAAS,KAAK;IACZ,MAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAM/D,MAAM,cAAc,SAAS,OAAO,OAAO;IAC3C,MAAM,IAAI,mBAAmB;KAC3B;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO;IACT,CAAC;GACH;GACA,MAAM,SAAS,IAAI,IAAI;GAEvB,MAAM,kBAAkB,MAAM,kBAAkB,OAAO,OAAO;IAAE,GAAG;IAAW;GAAQ,CAAC;GACvF,IAAI,CAAC,gBAAgB,IACnB,QAAQ,oBAAR;IACE,KAAK,SACH,MAAM,IAAI,kBAAkB,eAAe;IAC7C,KAAK,eACH,MAAM,IAAI,mBACR;KACE;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO,gBAAgB,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,IAAI;IAC5D,GACA,eACF;IACF,KAAK,OAEH;GACJ;GAGF,MAAM,gBAA4B,EAChC,KAAK,QAAQ,OAAO,CAAC,EACvB;GACA,IAAI,aAAa,WAAW,cAAc,eAAe,QAAQ;QAC5D,cAAc,cAAc,QAAQ;GACzC,IAAI,QAAQ,gBAAgB,KAAA,GAAW,cAAc,cAAc,QAAQ;GAE3E,MAAM,SAAoB;IACxB;IACA,cAAc,KAAK;IACnB,aAAa,KAAK,QAAQ;IAC1B,MAAM,KAAK;IACX,OAAO,QAAQ;IACf,YAAY,QAAQ;IACpB,YAAY,QAAQ;IACpB,WAAW,KAAK;IAChB;IACA,SAAS,QAAQ;IACjB,gBAAgB,QAAQ;IACxB,YAAY,QAAQ;IACpB,iBAAiB;IACjB,eAAe,QAAQ;IACvB,SAAS;IACT,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;IACrE,aAAa,QAAQ;IACrB;IACA,YAAY,KAAK,SAAS;GAC5B;GACA,MAAM,gBACJ,QAAQ,iBACP,OAAO,KAAK,iBAAiB,aAC1B,MAAM,KAAK,aAAa;IACtB,YAAY,KAAK;IACjB;IACA,WAAW,KAAK,QAAQ;IACxB,YAAY,KAAK,SAAS;IAC1B,MAAM,KAAK;IACX,SAAS,KAAK,QAAQ;IACtB,cAAc,KAAK,SAAS,QAAQ,CAAC;GACvC,CAAC,IACD,KAAK;GACX,IAAI,kBAAkB,KAAA,GAAW;IAC/B,MAAM,eAAe,MAAM,wBAAwB,aAAa;IAChE,6BAA6B,cAAc,QAAQ,OAAO,QAAQ,UAAU;IAC5E,OAAO,eAAe;GACxB;GACA,OAAO;IAAE,QAAQ,kBAAkB,MAAM;IAAG,WAAW;GAAgB;EACzE,UAAU;GAIR,SAAS,OAAO,KAAK;EACvB;CACF;CAEA,MAAM,UAAU,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,MAAM,MAAM,EAAE,SAAS,OAAO,CAAC;CAI1F,MAAM,QAAQ,WAAW,OAAO;CAMhC,KAAK,MAAM,CAAC,OAAO,YAAY,UAC7B,MAAM,cAAc,SAAS,OAAO,kDAAkD;CAExF,SAAS,MAAM;CAEf,IAAI,cAAc,SAAS,GACzB,MAAM,cAAc,WAAW,IAC3B,cAAc,KACd,IAAI,eACF,eACA,oBAAoB,cAAc,OAAO,iDAC3C;CAIN,IAAI;CACJ,IAAI,KAAK,QAQP,SAAS,MAAM,eAAe,MAAM;EANlC,GAAG,KAAK;EACR,YAAY,KAAK,OAAO;EACxB,OAAO,aAAa,QAAQ,WAAW;EACvC,aAAa,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;EACzC,qBAAqB,uBAAuB,KAAA;CAED,CAAC;CAGhD,MAAM,UAAU,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAE5C,OAAO;EACL,YAAY,KAAK;EACjB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAIA,IAAM,qBAAN,cAAiC,MAAM;CACrC;CACA;CACA,YAAY,QAAmB,WAAgC;EAC7D,MAAM,QAAQ,OAAO,UAAU,GAAG,OAAO,WAAW,GAAG,OAAO,KAAK,WAAW,OAAO,QAAQ;EAC7F,KAAK,SAAS;EACd,KAAK,YAAY;CACnB;AACF;;;;;;;;;;;;;;;;AAiBA,eAAsB,cACpB,SACA,OACA,QACe;CACf,MAAM,WAAW,MAAM,QAAQ,WAAW,OAAO,KAAK;CACtD,IAAI,aAAa,KAAA,GAAW;CAC5B,IAAI,SAAS,WAAW,WAAW;CACnC,MAAM,QAAQ,SAAS,MAAM;AAC/B;AAEA,SAAS,sBAAsB,SAA6B;CAC1D,QAAQ,WAAmD;EACzD,IAAI,CAAC,SACH,MAAM,IAAI,MACR,6LACF;EAEF,OAAO,IAAI,0BAA0B,EACnC,KAAK,GAAG,QAAQ,cAAc,OAAO,QACvC,CAAC;CACH;AACF;AAEA,eAAe,wBACb,OAC2B;CAC3B,IAAI,mBAAmB,KAAK,GAAG;EAC7B,IAAI,CAAE,MAAM,uBAAuB,KAAK,GACtC,MAAM,IAAI,MAAM,iEAAiE;EAEnF,OAAO;CACT;CACA,OAAO,sBAAsB,KAAK;AACpC;AAEA,SAAS,mBACP,OAC2B;CAC3B,OAAO,mBAAmB,SAAS,YAAY;AACjD;AAEA,SAAS,6BACP,SACA,OACA,YACM;CACN,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OACnD,MAAM,IAAI,MACR,wCAAwC,QAAQ,MAAM,kCAAkC,MAAM,EAChG;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,YAC7D,MAAM,IAAI,MACR,6CAA6C,QAAQ,WAAW,uCAAuC,WAAW,EACpH;AAEJ;AAEA,SAAS,aAAa,QAAuC;CAG3D,MAAM,OAAO,GAAG,OAAO,WAAW,IAAI,OAAO,UAAU,IAAI,OAAO,WAAW,IAAI,OAAO;CAExF,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,KAAK,WAAW,CAAC;EAC3B,KAAK,KAAK,KAAK,KAAK,GAAG,QAAU,MAAM;EACvC,KAAK,KAAK,KAAK,KAAK,GAAG,UAAU,MAAM;CACzC;CACA,OAAO,OAAO,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG,IAAI,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG;AAClF"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { l as AnalystRunResult, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-
|
|
1
|
+
import { l as AnalystRunResult, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-D3jh6F98.js";
|
|
2
2
|
import { z } from "zod";
|
|
3
3
|
//#region src/analyst/exact-types.d.ts
|
|
4
4
|
/** Analyst metadata required before the exact registry path will execute it. */
|
|
@@ -231,4 +231,4 @@ declare const exactExecutionPlanSchema: z.ZodObject<{
|
|
|
231
231
|
}, z.core.$strict>;
|
|
232
232
|
//#endregion
|
|
233
233
|
export { ExactAnalystRunPolicySnapshot as a, ExactAnalystSnapshot as c, ExactExecutionComponentSnapshot as d, ExactAnalystRunEvent as i, ExactCapableAnalyst as l, ExactAnalystExecutionPlanSnapshot as n, ExactAnalystRunResult as o, ExactAnalystRunCompletion as r, ExactAnalystRunSummary as s, ExactAnalystBudgetSnapshot as t, ExactExecutionComponentIdentity as u };
|
|
234
|
-
//# sourceMappingURL=exact-types-
|
|
234
|
+
//# sourceMappingURL=exact-types-B0lJV3tu.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"exact-types-
|
|
1
|
+
{"version":3,"file":"exact-types-B0lJV3tu.d.ts","names":[],"sources":["../src/analyst/exact-types.ts"],"mappings":";;;;UAMiB,oBAAoB,0BAA0B,QAAQ;;WAE5D,iBAAiB,SAAS;;KAGhC,aAAa,KAAK,0BAA0B,mBACpC,aAAa,UACtB,+BACc,aAAa,IAAI,aAAa,EAAE,WAC5C;KAEM,uBAAuB,aAAa,EAAE,aAAa;KACnD,6BAA6B,aAAa,EAAE,aAAa;;UAGpD;EACf;EACA;EACA,QAAQ,SAAS;;;UAIF;EACf;EACA;EACA;;KAyBU,gCAAgC,aAAa,EAAE,aAAa;KAC5D,oCAAoC,aAC9C,EAAE,aAAa;KAGL,yBAAyB;;EAEnC;;KAGU;EACN;;EAEA;EACA;IAAS;IAAe;;;UAGb,8BAA8B;EAC7C,aAAa;EACb,gBAAgB;EAChB,YAAY;;;KAIF,wBACP,QAAQ;EAAmB;;EAC1B,gBAAgB;MAEjB,KAAK,QAAQ;EAAmB;;EAC/B,SAAS;KAEX,QAAQ;EAAmB;MAC1B,KAAK,QAAQ;EAAmB;;EAC/B,SAAS;MAEV,KAAK,QAAQ;EAAmB;;EAC/B,QAAQ;;cAwCR,uBAAqB,EAAA;;;;;;;;;;;;;;;;;;;;;;;;;GAOzB,EAAA,KAAA;cAII,sBAAoB,EAAA,uBAAA,EAAA;;;;;;;;;;;GAaxB,EAAA,KAAA;cAiBI,sBAAoB,EAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAYxB,EAAA,KAAA;cAEI,0BAAwB,EAAA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA+E1B,EAAA,KAAA"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { b as CustomTokenPricing } from "./cost-ledger-
|
|
1
|
+
import { b as CustomTokenPricing } from "./cost-ledger-FuQvHxPm.js";
|
|
2
2
|
//#region src/campaign/external-optimizer-contracts.d.ts
|
|
3
3
|
interface ExternalOptimizerRunnerCommand {
|
|
4
4
|
command?: string;
|
|
@@ -22,6 +22,16 @@ interface ExternalOptimizerModelBudget {
|
|
|
22
22
|
maxResponseBytes: number;
|
|
23
23
|
/** Reject a request asking the provider for more output tokens. */
|
|
24
24
|
maxOutputTokensPerRequest: number;
|
|
25
|
+
/**
|
|
26
|
+
* Reasoning tokens a single response may bill beyond its completion limit.
|
|
27
|
+
*
|
|
28
|
+
* A reasoning model bounds only the completion by `max_tokens` and bills
|
|
29
|
+
* thinking on top, so a reservation sized to the completion alone is always
|
|
30
|
+
* too small and the ledger refuses the real charge. Callers that route a
|
|
31
|
+
* reasoning model declare its thinking budget here; the reservation covers
|
|
32
|
+
* it and a response exceeding it still fails loudly. Default: 0.
|
|
33
|
+
*/
|
|
34
|
+
maxReasoningTokensPerRequest?: number;
|
|
25
35
|
/** Rates used to estimate cost when the provider omits a valid `usage.cost`. */
|
|
26
36
|
pricing: CustomTokenPricing;
|
|
27
37
|
/** Per-provider-request deadline. Default: 300,000 ms. */
|
|
@@ -29,4 +39,4 @@ interface ExternalOptimizerModelBudget {
|
|
|
29
39
|
}
|
|
30
40
|
//#endregion
|
|
31
41
|
export { ExternalTextEvaluationRequest as a, ExternalTextCandidate as i, ExternalOptimizerResumeMode as n, ExternalOptimizerRunnerCommand as r, ExternalOptimizerModelBudget as t };
|
|
32
|
-
//# sourceMappingURL=external-optimizer-contracts-
|
|
42
|
+
//# sourceMappingURL=external-optimizer-contracts-nb7c_WAR.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"external-optimizer-contracts-
|
|
1
|
+
{"version":3,"file":"external-optimizer-contracts-nb7c_WAR.d.ts","names":[],"sources":["../src/campaign/external-optimizer-contracts.ts"],"mappings":";;UAKiB;EACf;EACA;EACA,MAAM,OAAO;;KAGH;KAEA,iCAAiC;UAE5B;EACf,WAAW;EACX;;UAUe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;EAUA;;EAEA,SAAS;;EAET"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { c as ValidationError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKEN_ATTR_KEYS, RUN_COST_ATTR_KEYS } from "./trace-attributes.js";
|
|
3
|
-
import "./kind-factory-
|
|
3
|
+
import "./kind-factory-DB7nIs35.js";
|
|
4
4
|
import { t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
5
5
|
import { TOKEN_USAGE_INPUT_KEYS, TOKEN_USAGE_OUTPUT_KEYS, firstTokenCount, tokenUsageSource } from "@tangle-network/agent-core/telemetry";
|
|
6
6
|
import { SSEChunkParser } from "@tangle-network/agent-core/sse";
|
|
@@ -460,4 +460,4 @@ async function extractUsageFromResponse(response, sseOptions) {
|
|
|
460
460
|
//#endregion
|
|
461
461
|
export { recordAggregateMeasurements as a, readTaskFailureLabels as i, extractUsageFromResponse as n, summarizeExecutionMeasurements as o, extractUsageFromSse as r, summarizeTraceErrors as s, extractUsage as t };
|
|
462
462
|
|
|
463
|
-
//# sourceMappingURL=extract-usage-
|
|
463
|
+
//# sourceMappingURL=extract-usage-C5vMw-0R.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"extract-usage-DZs601Va.js","names":["obj"],"sources":["../src/trace/error-classification.ts","../src/trace/execution-measurements.ts","../src/trace/task-failure-attributes.ts","../src/trace/extract-usage.ts"],"sourcesContent":["import type { OtlpSpanRole } from './otlp-attributes'\n\nexport type TraceErrorRole = OtlpSpanRole\n\nexport interface TraceErrorSignal {\n id: string\n parentId?: string\n role: TraceErrorRole\n error: boolean\n processRoot: boolean\n}\n\nexport interface TraceErrorSummary {\n total: number\n execution: number\n process: number\n guardrail: number\n evaluation: number\n propagated: number\n unclassified: number\n}\n\n/**\n * Classify errored spans without counting a propagated parent status as a\n * second execution failure.\n */\nexport function summarizeTraceErrors(signals: readonly TraceErrorSignal[]): TraceErrorSummary {\n const byId = new Map<string, TraceErrorSignal>()\n for (const signal of signals) {\n if (byId.has(signal.id)) {\n throw new Error(`summarizeTraceErrors: duplicate span id '${signal.id}'`)\n }\n byId.set(signal.id, signal)\n }\n\n const propagated = new Set<string>()\n for (const signal of signals) {\n if (!signal.error) continue\n const visited = new Set<string>()\n let parentId = signal.parentId\n while (parentId && !visited.has(parentId)) {\n visited.add(parentId)\n const parent = byId.get(parentId)\n if (!parent) break\n if (parent.error) propagated.add(parent.id)\n parentId = parent.parentId\n }\n }\n\n const summary: TraceErrorSummary = {\n total: 0,\n execution: 0,\n process: 0,\n guardrail: 0,\n evaluation: 0,\n propagated: 0,\n unclassified: 0,\n }\n\n for (const signal of signals) {\n if (!signal.error) continue\n summary.total += 1\n if (signal.role === 'GUARDRAIL') {\n summary.guardrail += 1\n } else if (signal.role === 'EVALUATOR') {\n summary.evaluation += 1\n } else if (signal.processRoot) {\n summary.process += 1\n } else if (propagated.has(signal.id)) {\n summary.propagated += 1\n } else if (\n signal.role === 'AGENT' ||\n signal.role === 'CHAIN' ||\n signal.role === 'LLM' ||\n signal.role === 'TOOL'\n ) {\n summary.execution += 1\n } else {\n summary.unclassified += 1\n }\n }\n\n return summary\n}\n","import type { RunTokenUsage } from '../run-record'\nimport {\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_COST_ATTR_KEYS,\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n RUN_COST_ATTR_KEYS,\n} from './otlp-attributes'\n\nexport interface ExecutionMeasurementSpan {\n id: string\n parentId?: string\n attributes: Record<string, unknown>\n modelCall: boolean\n aggregate: boolean\n}\n\nexport interface MeasurementCoverage {\n value?: number\n reportingCalls: number\n complete: boolean\n}\n\nexport interface ExecutionMeasurements {\n tokenUsage: RunTokenUsage\n modelCallCount: number\n callSpanIds: string[]\n cost: MeasurementCoverage\n aggregate?: {\n tokenUsage: RunTokenUsage\n costUsd?: number\n }\n}\n\nconst TOKEN_MEASUREMENT_KEY_GROUPS = [\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n] as const\n\nconst EXECUTION_MEASUREMENT_KEY_GROUPS = [\n ...TOKEN_MEASUREMENT_KEY_GROUPS,\n LLM_COST_ATTR_KEYS,\n] as const\n\ninterface RetainedCallSummary {\n callCount: number\n measurements: Array<{\n total: number\n reportingCalls: number\n }>\n}\n\n/**\n * Reconcile execution measurements across nested telemetry wrappers.\n * A measured parent is used only when a descendant call does not report the\n * same field, so aggregate wrappers neither duplicate complete child data nor\n * erase complementary parent fields.\n */\nexport function summarizeExecutionMeasurements(\n spans: ExecutionMeasurementSpan[],\n): ExecutionMeasurements {\n const byId = new Map<string, ExecutionMeasurementSpan>()\n for (const span of spans) {\n if (byId.has(span.id)) {\n throw new Error(`summarizeExecutionMeasurements: duplicate span id \"${span.id}\"`)\n }\n byId.set(span.id, span)\n }\n const tokenMeasurementKeys = TOKEN_MEASUREMENT_KEY_GROUPS.flat()\n const candidates = spans.filter(\n (span) =>\n span.modelCall ||\n (!span.aggregate && readNumber(span.attributes, tokenMeasurementKeys) !== undefined),\n )\n const candidateIds = new Set(candidates.map((span) => span.id))\n const candidateChildren = new Map<string, ExecutionMeasurementSpan[]>()\n for (const candidate of candidates) {\n const parentId = nearestCandidateParent(candidate, byId, candidateIds)\n if (!parentId) continue\n const children = candidateChildren.get(parentId) ?? []\n children.push(candidate)\n candidateChildren.set(parentId, children)\n }\n const aggregateIds = classifyAggregateSpans(candidates, candidateChildren)\n const untypedRunCostIds = new Set(\n spans\n .filter(\n (span) =>\n !span.modelCall &&\n !span.aggregate &&\n !candidateIds.has(span.id) &&\n readNumber(span.attributes, RUN_COST_ATTR_KEYS) !== undefined,\n )\n .map((span) => span.id),\n )\n const aggregateSourceIds = new Set(\n spans\n .filter(\n (span) => span.aggregate || aggregateIds.has(span.id) || untypedRunCostIds.has(span.id),\n )\n .map((span) => span.id),\n )\n\n const calls = candidates.filter((span) => !aggregateIds.has(span.id))\n const input = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_INPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const output = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const cached = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n )\n const aggregate = summarizeAggregateMeasurements(\n spans,\n byId,\n new Set([...aggregateIds, ...untypedRunCostIds]),\n )\n\n return {\n tokenUsage: {\n input: input.value ?? 0,\n output: Math.max(output.value ?? 0, reasoning.value ?? 0),\n ...(reasoning.value !== undefined ? { reasoning: reasoning.value } : {}),\n ...(cached.value !== undefined ? { cached: cached.value } : {}),\n ...(cacheWrite.value !== undefined ? { cacheWrite: cacheWrite.value } : {}),\n },\n modelCallCount: calls.length,\n callSpanIds: calls.map((span) => span.id),\n cost: reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_COST_ATTR_KEYS),\n ...(aggregate ? { aggregate } : {}),\n }\n}\n\nexport function recordAggregateMeasurements(\n raw: Record<string, number>,\n aggregate: ExecutionMeasurements['aggregate'],\n): void {\n if (!aggregate) return\n raw.aggregate_prompt_tokens = aggregate.tokenUsage.input\n raw.aggregate_completion_tokens = aggregate.tokenUsage.output\n if (aggregate.tokenUsage.reasoning !== undefined)\n raw.aggregate_reasoning_tokens = aggregate.tokenUsage.reasoning\n if (aggregate.tokenUsage.cached !== undefined)\n raw.aggregate_cached_tokens = aggregate.tokenUsage.cached\n if (aggregate.tokenUsage.cacheWrite !== undefined)\n raw.aggregate_cache_write_tokens = aggregate.tokenUsage.cacheWrite\n if (aggregate.costUsd !== undefined) raw.aggregate_cost_usd = aggregate.costUsd\n}\n\nfunction summarizeAggregateMeasurements(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateIds: Set<string>,\n): ExecutionMeasurements['aggregate'] {\n const aggregates = spans.filter((span) => span.aggregate || aggregateIds.has(span.id))\n const input = reconcileTopLevelMeasurement(aggregates, byId, LLM_INPUT_TOKEN_ATTR_KEYS)\n const output = reconcileTopLevelMeasurement(aggregates, byId, LLM_OUTPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileTopLevelMeasurement(aggregates, byId, LLM_REASONING_TOKEN_ATTR_KEYS)\n const cached = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS)\n const costUsd = reconcileTopLevelMeasurement(aggregates, byId, LLM_COST_ATTR_KEYS)\n if (\n input === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined &&\n costUsd === undefined\n )\n return undefined\n return {\n tokenUsage: {\n input: input ?? 0,\n output: output ?? reasoning ?? 0,\n ...(reasoning !== undefined ? { reasoning } : {}),\n ...(cached !== undefined ? { cached } : {}),\n ...(cacheWrite !== undefined ? { cacheWrite } : {}),\n },\n ...(costUsd !== undefined ? { costUsd } : {}),\n }\n}\n\nfunction reconcileTopLevelMeasurement(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n keys: readonly string[],\n): number | undefined {\n const selected = new Map<string, number>()\n for (const span of spans) {\n const value = readNumber(span.attributes, keys)\n if (value !== undefined) selected.set(span.id, value)\n }\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (ancestorIds(span, byId).some((ancestorId) => selected.has(ancestorId))) {\n selected.delete(spanId)\n }\n }\n return selected.size > 0\n ? [...selected.values()].reduce((total, value) => total + value, 0)\n : undefined\n}\n\nfunction reconcileMeasurement(\n calls: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n fallbackKeys?: readonly string[],\n): MeasurementCoverage {\n const selected = new Map<string, number>()\n const callIds = new Set(calls.map((call) => call.id))\n let reportingCalls = 0\n\n for (const call of calls) {\n const primary = nearestMeasurement(call, byId, callIds, aggregateSourceIds, keys)\n const fallback = fallbackKeys\n ? nearestMeasurement(call, byId, callIds, aggregateSourceIds, fallbackKeys)\n : undefined\n const source =\n primary && fallback && primary.span.id === fallback.span.id\n ? { span: primary.span, value: Math.max(primary.value, fallback.value) }\n : (primary ?? fallback)\n if (!source) continue\n reportingCalls += 1\n selected.set(source.span.id, source.value)\n }\n\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (\n ancestorIds(span, byId).some(\n (ancestorId) => selected.has(ancestorId) && !callIds.has(ancestorId),\n )\n ) {\n selected.delete(spanId)\n }\n }\n\n return {\n ...(selected.size > 0\n ? { value: [...selected.values()].reduce((total, value) => total + value, 0) }\n : {}),\n reportingCalls,\n complete: calls.length > 0 && reportingCalls === calls.length,\n }\n}\n\nfunction nearestMeasurement(\n call: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n callIds: Set<string>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n): { span: ExecutionMeasurementSpan; value: number } | undefined {\n let current: ExecutionMeasurementSpan | undefined = call\n const seen = new Set<string>()\n while (current && !seen.has(current.id)) {\n seen.add(current.id)\n const value = readNumber(current.attributes, keys)\n if (\n value !== undefined &&\n (current.id === call.id || (!callIds.has(current.id) && aggregateSourceIds.has(current.id)))\n ) {\n return { span: current, value }\n }\n current = current.parentId ? byId.get(current.parentId) : undefined\n }\n return undefined\n}\n\nfunction classifyAggregateSpans(\n candidates: ExecutionMeasurementSpan[],\n childrenById: Map<string, ExecutionMeasurementSpan[]>,\n): Set<string> {\n const aggregateIds = new Set<string>()\n const summaries = new Map<string, RetainedCallSummary>()\n const visiting = new Set<string>()\n\n const visit = (span: ExecutionMeasurementSpan): RetainedCallSummary => {\n const cached = summaries.get(span.id)\n if (cached) return cached\n if (visiting.has(span.id)) return emptyRetainedCallSummary()\n visiting.add(span.id)\n\n const descendants = emptyRetainedCallSummary()\n for (const child of childrenById.get(span.id) ?? []) {\n const childDescendants = visit(child)\n if (!aggregateIds.has(child.id)) addRetainedCall(descendants, child)\n addRetainedCallSummary(descendants, childDescendants)\n }\n\n if (\n descendants.callCount > 0 &&\n (!span.modelCall || hasCompatibleDescendantMeasurements(span, descendants))\n ) {\n aggregateIds.add(span.id)\n }\n\n visiting.delete(span.id)\n summaries.set(span.id, descendants)\n return descendants\n }\n\n for (const candidate of candidates) visit(candidate)\n return aggregateIds\n}\n\nfunction emptyRetainedCallSummary(): RetainedCallSummary {\n return {\n callCount: 0,\n measurements: EXECUTION_MEASUREMENT_KEY_GROUPS.map(() => ({\n total: 0,\n reportingCalls: 0,\n })),\n }\n}\n\nfunction addRetainedCall(summary: RetainedCallSummary, span: ExecutionMeasurementSpan): void {\n summary.callCount += 1\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const value = readNumber(span.attributes, EXECUTION_MEASUREMENT_KEY_GROUPS[index]!)\n if (value === undefined) continue\n const measurement = summary.measurements[index]!\n measurement.total += value\n measurement.reportingCalls += 1\n }\n}\n\nfunction addRetainedCallSummary(target: RetainedCallSummary, source: RetainedCallSummary): void {\n target.callCount += source.callCount\n for (let index = 0; index < target.measurements.length; index += 1) {\n const measurement = target.measurements[index]!\n const sourceMeasurement = source.measurements[index]!\n measurement.total += sourceMeasurement.total\n measurement.reportingCalls += sourceMeasurement.reportingCalls\n }\n}\n\nfunction hasCompatibleDescendantMeasurements(\n span: ExecutionMeasurementSpan,\n descendants: RetainedCallSummary,\n): boolean {\n let parentMeasurements = 0\n let descendantMeasurements = 0\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const keys = EXECUTION_MEASUREMENT_KEY_GROUPS[index]!\n const parentValue = readNumber(span.attributes, keys)\n const measurement = descendants.measurements[index]!\n if (parentValue !== undefined) parentMeasurements += 1\n if (measurement.reportingCalls > 0) descendantMeasurements += 1\n if (parentValue === undefined || measurement.reportingCalls === 0) continue\n if (measurement.reportingCalls !== descendants.callCount) continue\n if (Math.abs(parentValue - measurement.total) > 1e-12) return false\n }\n return parentMeasurements === 0 || descendantMeasurements > 0\n}\n\nfunction ancestorIds(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n): string[] {\n const ids: string[] = []\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n ids.push(parentId)\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return ids\n}\n\nfunction nearestCandidateParent(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n candidateIds: Set<string>,\n): string | undefined {\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n if (candidateIds.has(parentId)) return parentId\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return undefined\n}\n\nfunction readNumber(\n attributes: Record<string, unknown>,\n keys: readonly string[],\n): number | undefined {\n for (const key of keys) {\n const value = attributes[key]\n const parsed =\n typeof value === 'number'\n ? value\n : typeof value === 'string' && value.length > 0\n ? Number(value)\n : Number.NaN\n if (Number.isFinite(parsed) && parsed >= 0) return parsed\n }\n return undefined\n}\n","import { ValidationError } from '../errors'\nimport { FAILURE_CLASSES, type FailureClass } from './schema'\n\nconst TASK_FAILURE_CLASS_ATTR = 'tangle.task.failure_class'\nconst TASK_FAILURE_MODE_ATTR = 'tangle.task.failure_mode'\n\ninterface AttributeCarrier {\n attributes: Record<string, unknown>\n}\n\nexport type TaskFailureLabels =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | { failureClass: Exclude<FailureClass, 'success'>; failureMode?: string }\n\nexport function readTaskFailureLabels(\n roots: readonly AttributeCarrier[],\n context: string,\n): TaskFailureLabels {\n const failureClass = readConsistentRootString(roots, TASK_FAILURE_CLASS_ATTR, context)\n const failureMode = readConsistentRootString(roots, TASK_FAILURE_MODE_ATTR, context)\n\n if (failureClass !== undefined && !FAILURE_CLASSES.includes(failureClass as FailureClass)) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_CLASS_ATTR} must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n if (failureMode !== undefined && (failureClass === undefined || failureClass === 'success')) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_MODE_ATTR} requires a non-success ${TASK_FAILURE_CLASS_ATTR}`,\n )\n }\n\n if (failureClass === undefined) return {}\n if (failureClass === 'success') return { failureClass }\n return {\n failureClass: failureClass as Exclude<FailureClass, 'success'>,\n ...(failureMode ? { failureMode } : {}),\n }\n}\n\nfunction readConsistentRootString(\n roots: readonly AttributeCarrier[],\n key: string,\n context: string,\n): string | undefined {\n const values = new Set<string>()\n for (const root of roots) {\n if (!Object.hasOwn(root.attributes, key)) continue\n const value = root.attributes[key]\n if (typeof value !== 'string' || value.trim().length === 0) {\n throw new ValidationError(`${context}: ${key} must be a non-empty string`)\n }\n values.add(value)\n }\n\n if (values.size > 1) {\n throw new ValidationError(\n `${context}: conflicting ${key} values: ${[...values].sort().join(', ')}`,\n )\n }\n return values.values().next().value\n}\n","/**\n * Provider response and SSE usage extraction.\n *\n * Missing usage returns `null`; reported zeroes and cache-only activity remain\n * distinguishable from absent telemetry.\n */\n\nimport { SSEChunkParser } from '@tangle-network/agent-core/sse'\nimport {\n firstTokenCount,\n TOKEN_USAGE_INPUT_KEYS,\n TOKEN_USAGE_OUTPUT_KEYS,\n tokenUsageSource,\n} from '@tangle-network/agent-core/telemetry'\nimport type { RunTokenUsage } from '../run-record'\n\nexport type ExtractedUsage = RunTokenUsage\nexport type SseUsageMode = 'cumulative' | 'delta'\n\nexport interface ExtractUsageFromSseOptions {\n /** Provider usage events are cumulative snapshots unless explicitly marked as deltas. */\n mode?: SseUsageMode\n}\n\nconst INPUT_OTHER_KEYS = ['input_other'] as const\nconst EXCLUSIVE_REASONING_KEYS = ['reasoning', 'reasoning_tokens', 'reasoningTokens'] as const\nconst INCLUSIVE_REASONING_KEYS = ['reasoning_output_tokens', 'reasoningOutputTokens'] as const\nconst INPUT_DETAIL_KEYS = ['prompt_tokens_details', 'input_tokens_details'] as const\nconst OUTPUT_DETAIL_KEYS = ['completion_tokens_details', 'output_tokens_details'] as const\nconst CACHED_KEYS = [\n 'cache',\n 'cached_tokens',\n 'cached_input_tokens',\n 'cache_read_tokens',\n 'cache_read_input_tokens',\n 'cachedTokens',\n 'cachedInputTokens',\n 'cacheReadTokens',\n 'cacheReadInputTokens',\n 'input_cache_read',\n] as const\nconst CACHE_WRITE_KEYS = [\n 'cache_creation_tokens',\n 'cache_creation_input_tokens',\n 'cacheCreationTokens',\n 'cacheCreationInputTokens',\n 'input_cache_creation',\n] as const\n\nfunction nestedRecord(\n source: Record<string, unknown>,\n key: string,\n): Record<string, unknown> | undefined {\n const value = source[key]\n return value && typeof value === 'object' && !Array.isArray(value)\n ? (value as Record<string, unknown>)\n : undefined\n}\n\nfunction nestedNumber(\n source: Record<string, unknown>,\n recordKeys: readonly string[],\n valueKeys: readonly string[],\n): number | undefined {\n for (const recordKey of recordKeys) {\n const value = firstTokenCount(nestedRecord(source, recordKey), valueKeys)\n if (value !== undefined) return value\n }\n return undefined\n}\n\n/**\n * Pull `{ input, output, cached?, cacheWrite? }` from a parsed response\n * body. Accepts a top-level `usage` object (the common case) or a body that IS\n * the usage object. Returns null only when none of those categories is present.\n */\nexport function extractUsage(body: unknown): ExtractedUsage | null {\n if (!body || typeof body !== 'object') return null\n const obj = body as Record<string, unknown>\n const usage = tokenUsageSource(obj)\n const input = firstTokenCount(usage, TOKEN_USAGE_INPUT_KEYS)\n const inputOther = firstTokenCount(usage, INPUT_OTHER_KEYS)\n const output = firstTokenCount(usage, TOKEN_USAGE_OUTPUT_KEYS)\n const exclusiveReasoning = firstTokenCount(usage, EXCLUSIVE_REASONING_KEYS)\n const inclusiveReasoning =\n firstTokenCount(usage, INCLUSIVE_REASONING_KEYS) ??\n nestedNumber(usage, OUTPUT_DETAIL_KEYS, ['reasoning_tokens', 'reasoningTokens'])\n const reasoning = exclusiveReasoning ?? inclusiveReasoning\n const nestedCache = nestedRecord(usage, 'cache')\n const cached =\n firstTokenCount(usage, CACHED_KEYS) ??\n firstTokenCount(nestedCache, ['read']) ??\n nestedNumber(usage, INPUT_DETAIL_KEYS, ['cached_tokens', 'cachedTokens'])\n const cacheWrite =\n firstTokenCount(usage, CACHE_WRITE_KEYS) ?? firstTokenCount(nestedCache, ['write'])\n if (\n input === undefined &&\n inputOther === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined\n )\n return null\n const result: ExtractedUsage = {\n input: (input ?? 0) + (inputOther ?? 0),\n output:\n (output ?? (inclusiveReasoning !== undefined ? inclusiveReasoning : 0)) +\n (exclusiveReasoning ?? 0),\n }\n if (reasoning !== undefined) result.reasoning = reasoning\n if (cached !== undefined) result.cached = cached\n if (cacheWrite !== undefined) result.cacheWrite = cacheWrite\n return result\n}\n\n/**\n * Extract token usage from a complete SSE response body using the shared SSE\n * frame parser. Cumulative snapshots are the fail-safe default because summing\n * them inflates billing; callers with explicit delta events opt into `mode: 'delta'`.\n */\nexport function extractUsageFromSse(\n text: string,\n options: ExtractUsageFromSseOptions = {},\n): ExtractedUsage | null {\n const mode = options.mode ?? 'cumulative'\n let input = 0\n let output = 0\n let reasoning = 0\n let sawReasoning = false\n let cached = 0\n let sawCached = false\n let cacheWrite = 0\n let sawCacheWrite = false\n let found = false\n const parser = new SSEChunkParser<unknown>({ transform: parseSseJson })\n const events = [...parser.push(text), ...parser.flush()]\n const merge = mode === 'delta' ? (current: number, next: number) => current + next : Math.max\n\n for (const event of events) {\n const usage = extractUsage(event.data)\n if (!usage) continue\n input = merge(input, usage.input)\n output = merge(output, usage.output)\n if (usage.reasoning !== undefined) {\n reasoning = merge(reasoning, usage.reasoning)\n sawReasoning = true\n }\n if (usage.cached !== undefined) {\n cached = merge(cached, usage.cached)\n sawCached = true\n }\n if (usage.cacheWrite !== undefined) {\n cacheWrite = merge(cacheWrite, usage.cacheWrite)\n sawCacheWrite = true\n }\n found = true\n }\n if (!found) return null\n return {\n input,\n output,\n ...(sawReasoning ? { reasoning } : {}),\n ...(sawCached ? { cached } : {}),\n ...(sawCacheWrite ? { cacheWrite } : {}),\n }\n}\n\nfunction parseSseJson(raw: string): unknown | null {\n const payload = raw.trim()\n if (!payload || payload === '[DONE]') return null\n try {\n return JSON.parse(payload)\n } catch {\n return null\n }\n}\n\n/**\n * Extract usage from an HTTP `Response` without consuming the caller's body:\n * clones, reads the text, and tries the JSON parser first, then the SSE\n * accumulator. Best-effort — returns null on any read/parse miss so a usage tee\n * never takes down the underlying call.\n */\nexport async function extractUsageFromResponse(\n response: Response,\n sseOptions?: ExtractUsageFromSseOptions,\n): Promise<ExtractedUsage | null> {\n let text: string\n try {\n text = await response.clone().text()\n } catch {\n return null\n }\n let json: unknown\n try {\n json = JSON.parse(text)\n } catch {\n json = undefined\n }\n return extractUsage(json) ?? extractUsageFromSse(text, sseOptions)\n}\n"],"mappings":";;;;;;;;;;;AA0BA,SAAgB,qBAAqB,SAAyD;CAC5F,MAAM,uBAAO,IAAI,IAA8B;CAC/C,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,KAAK,IAAI,OAAO,EAAE,GACpB,MAAM,IAAI,MAAM,4CAA4C,OAAO,GAAG,EAAE;EAE1E,KAAK,IAAI,OAAO,IAAI,MAAM;CAC5B;CAEA,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,MAAM,0BAAU,IAAI,IAAY;EAChC,IAAI,WAAW,OAAO;EACtB,OAAO,YAAY,CAAC,QAAQ,IAAI,QAAQ,GAAG;GACzC,QAAQ,IAAI,QAAQ;GACpB,MAAM,SAAS,KAAK,IAAI,QAAQ;GAChC,IAAI,CAAC,QAAQ;GACb,IAAI,OAAO,OAAO,WAAW,IAAI,OAAO,EAAE;GAC1C,WAAW,OAAO;EACpB;CACF;CAEA,MAAM,UAA6B;EACjC,OAAO;EACP,WAAW;EACX,SAAS;EACT,WAAW;EACX,YAAY;EACZ,YAAY;EACZ,cAAc;CAChB;CAEA,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,QAAQ,SAAS;EACjB,IAAI,OAAO,SAAS,aAClB,QAAQ,aAAa;OAChB,IAAI,OAAO,SAAS,aACzB,QAAQ,cAAc;OACjB,IAAI,OAAO,aAChB,QAAQ,WAAW;OACd,IAAI,WAAW,IAAI,OAAO,EAAE,GACjC,QAAQ,cAAc;OACjB,IACL,OAAO,SAAS,WAChB,OAAO,SAAS,WAChB,OAAO,SAAS,SAChB,OAAO,SAAS,QAEhB,QAAQ,aAAa;OAErB,QAAQ,gBAAgB;CAE5B;CAEA,OAAO;AACT;;;AC/CA,MAAM,+BAA+B;CACnC;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,mCAAmC,CACvC,GAAG,8BACH,kBACF;;;;;;;AAgBA,SAAgB,+BACd,OACuB;CACvB,MAAM,uBAAO,IAAI,IAAsC;CACvD,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,IAAI,KAAK,EAAE,GAClB,MAAM,IAAI,MAAM,sDAAsD,KAAK,GAAG,EAAE;EAElF,KAAK,IAAI,KAAK,IAAI,IAAI;CACxB;CACA,MAAM,uBAAuB,6BAA6B,KAAK;CAC/D,MAAM,aAAa,MAAM,QACtB,SACC,KAAK,aACJ,CAAC,KAAK,aAAa,WAAW,KAAK,YAAY,oBAAoB,MAAM,KAAA,CAC9E;CACA,MAAM,eAAe,IAAI,IAAI,WAAW,KAAK,SAAS,KAAK,EAAE,CAAC;CAC9D,MAAM,oCAAoB,IAAI,IAAwC;CACtE,KAAK,MAAM,aAAa,YAAY;EAClC,MAAM,WAAW,uBAAuB,WAAW,MAAM,YAAY;EACrE,IAAI,CAAC,UAAU;EACf,MAAM,WAAW,kBAAkB,IAAI,QAAQ,KAAK,CAAC;EACrD,SAAS,KAAK,SAAS;EACvB,kBAAkB,IAAI,UAAU,QAAQ;CAC1C;CACA,MAAM,eAAe,uBAAuB,YAAY,iBAAiB;CACzE,MAAM,oBAAoB,IAAI,IAC5B,MACG,QACE,SACC,CAAC,KAAK,aACN,CAAC,KAAK,aACN,CAAC,aAAa,IAAI,KAAK,EAAE,KACzB,WAAW,KAAK,YAAY,kBAAkB,MAAM,KAAA,CACxD,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CACA,MAAM,qBAAqB,IAAI,IAC7B,MACG,QACE,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,KAAK,kBAAkB,IAAI,KAAK,EAAE,CACxF,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CAEA,MAAM,QAAQ,WAAW,QAAQ,SAAS,CAAC,aAAa,IAAI,KAAK,EAAE,CAAC;CACpE,MAAM,QAAQ,qBAAqB,OAAO,MAAM,oBAAoB,yBAAyB;CAC7F,MAAM,YAAY,qBAChB,OACA,MACA,oBACA,6BACF;CACA,MAAM,SAAS,qBACb,OACA,MACA,oBACA,4BACA,6BACF;CACA,MAAM,SAAS,qBAAqB,OAAO,MAAM,oBAAoB,0BAA0B;CAC/F,MAAM,aAAa,qBACjB,OACA,MACA,oBACA,+BACF;CACA,MAAM,YAAY,+BAChB,OACA,sBACA,IAAI,IAAI,CAAC,GAAG,cAAc,GAAG,iBAAiB,CAAC,CACjD;CAEA,OAAO;EACL,YAAY;GACV,OAAO,MAAM,SAAS;GACtB,QAAQ,KAAK,IAAI,OAAO,SAAS,GAAG,UAAU,SAAS,CAAC;GACxD,GAAI,UAAU,UAAU,KAAA,IAAY,EAAE,WAAW,UAAU,MAAM,IAAI,CAAC;GACtE,GAAI,OAAO,UAAU,KAAA,IAAY,EAAE,QAAQ,OAAO,MAAM,IAAI,CAAC;GAC7D,GAAI,WAAW,UAAU,KAAA,IAAY,EAAE,YAAY,WAAW,MAAM,IAAI,CAAC;EAC3E;EACA,gBAAgB,MAAM;EACtB,aAAa,MAAM,KAAK,SAAS,KAAK,EAAE;EACxC,MAAM,qBAAqB,OAAO,MAAM,oBAAoB,kBAAkB;EAC9E,GAAI,YAAY,EAAE,UAAU,IAAI,CAAC;CACnC;AACF;AAEA,SAAgB,4BACd,KACA,WACM;CACN,IAAI,CAAC,WAAW;CAChB,IAAI,0BAA0B,UAAU,WAAW;CACnD,IAAI,8BAA8B,UAAU,WAAW;CACvD,IAAI,UAAU,WAAW,cAAc,KAAA,GACrC,IAAI,6BAA6B,UAAU,WAAW;CACxD,IAAI,UAAU,WAAW,WAAW,KAAA,GAClC,IAAI,0BAA0B,UAAU,WAAW;CACrD,IAAI,UAAU,WAAW,eAAe,KAAA,GACtC,IAAI,+BAA+B,UAAU,WAAW;CAC1D,IAAI,UAAU,YAAY,KAAA,GAAW,IAAI,qBAAqB,UAAU;AAC1E;AAEA,SAAS,+BACP,OACA,MACA,cACoC;CACpC,MAAM,aAAa,MAAM,QAAQ,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,CAAC;CACrF,MAAM,QAAQ,6BAA6B,YAAY,MAAM,yBAAyB;CACtF,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,YAAY,6BAA6B,YAAY,MAAM,6BAA6B;CAC9F,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,aAAa,6BAA6B,YAAY,MAAM,+BAA+B;CACjG,MAAM,UAAU,6BAA6B,YAAY,MAAM,kBAAkB;CACjF,IACE,UAAU,KAAA,KACV,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,KACf,YAAY,KAAA,GAEZ,OAAO,KAAA;CACT,OAAO;EACL,YAAY;GACV,OAAO,SAAS;GAChB,QAAQ,UAAU,aAAa;GAC/B,GAAI,cAAc,KAAA,IAAY,EAAE,UAAU,IAAI,CAAC;GAC/C,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;GACzC,GAAI,eAAe,KAAA,IAAY,EAAE,WAAW,IAAI,CAAC;EACnD;EACA,GAAI,YAAY,KAAA,IAAY,EAAE,QAAQ,IAAI,CAAC;CAC7C;AACF;AAEA,SAAS,6BACP,OACA,MACA,MACoB;CACpB,MAAM,2BAAW,IAAI,IAAoB;CACzC,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAAQ,WAAW,KAAK,YAAY,IAAI;EAC9C,IAAI,UAAU,KAAA,GAAW,SAAS,IAAI,KAAK,IAAI,KAAK;CACtD;CACA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IAAI,YAAY,MAAM,IAAI,CAAC,CAAC,MAAM,eAAe,SAAS,IAAI,UAAU,CAAC,GACvE,SAAS,OAAO,MAAM;CAE1B;CACA,OAAO,SAAS,OAAO,IACnB,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,IAChE,KAAA;AACN;AAEA,SAAS,qBACP,OACA,MACA,oBACA,MACA,cACqB;CACrB,MAAM,2BAAW,IAAI,IAAoB;CACzC,MAAM,UAAU,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,EAAE,CAAC;CACpD,IAAI,iBAAiB;CAErB,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,IAAI;EAChF,MAAM,WAAW,eACb,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,YAAY,IACxE,KAAA;EACJ,MAAM,SACJ,WAAW,YAAY,QAAQ,KAAK,OAAO,SAAS,KAAK,KACrD;GAAE,MAAM,QAAQ;GAAM,OAAO,KAAK,IAAI,QAAQ,OAAO,SAAS,KAAK;EAAE,IACpE,WAAW;EAClB,IAAI,CAAC,QAAQ;EACb,kBAAkB;EAClB,SAAS,IAAI,OAAO,KAAK,IAAI,OAAO,KAAK;CAC3C;CAEA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IACE,YAAY,MAAM,IAAI,CAAC,CAAC,MACrB,eAAe,SAAS,IAAI,UAAU,KAAK,CAAC,QAAQ,IAAI,UAAU,CACrE,GAEA,SAAS,OAAO,MAAM;CAE1B;CAEA,OAAO;EACL,GAAI,SAAS,OAAO,IAChB,EAAE,OAAO,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,EAAE,IAC3E,CAAC;EACL;EACA,UAAU,MAAM,SAAS,KAAK,mBAAmB,MAAM;CACzD;AACF;AAEA,SAAS,mBACP,MACA,MACA,SACA,oBACA,MAC+D;CAC/D,IAAI,UAAgD;CACpD,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,WAAW,CAAC,KAAK,IAAI,QAAQ,EAAE,GAAG;EACvC,KAAK,IAAI,QAAQ,EAAE;EACnB,MAAM,QAAQ,WAAW,QAAQ,YAAY,IAAI;EACjD,IACE,UAAU,KAAA,MACT,QAAQ,OAAO,KAAK,MAAO,CAAC,QAAQ,IAAI,QAAQ,EAAE,KAAK,mBAAmB,IAAI,QAAQ,EAAE,IAEzF,OAAO;GAAE,MAAM;GAAS;EAAM;EAEhC,UAAU,QAAQ,WAAW,KAAK,IAAI,QAAQ,QAAQ,IAAI,KAAA;CAC5D;AAEF;AAEA,SAAS,uBACP,YACA,cACa;CACb,MAAM,+BAAe,IAAI,IAAY;CACrC,MAAM,4BAAY,IAAI,IAAiC;CACvD,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,SAAS,SAAwD;EACrE,MAAM,SAAS,UAAU,IAAI,KAAK,EAAE;EACpC,IAAI,QAAQ,OAAO;EACnB,IAAI,SAAS,IAAI,KAAK,EAAE,GAAG,OAAO,yBAAyB;EAC3D,SAAS,IAAI,KAAK,EAAE;EAEpB,MAAM,cAAc,yBAAyB;EAC7C,KAAK,MAAM,SAAS,aAAa,IAAI,KAAK,EAAE,KAAK,CAAC,GAAG;GACnD,MAAM,mBAAmB,MAAM,KAAK;GACpC,IAAI,CAAC,aAAa,IAAI,MAAM,EAAE,GAAG,gBAAgB,aAAa,KAAK;GACnE,uBAAuB,aAAa,gBAAgB;EACtD;EAEA,IACE,YAAY,YAAY,MACvB,CAAC,KAAK,aAAa,oCAAoC,MAAM,WAAW,IAEzE,aAAa,IAAI,KAAK,EAAE;EAG1B,SAAS,OAAO,KAAK,EAAE;EACvB,UAAU,IAAI,KAAK,IAAI,WAAW;EAClC,OAAO;CACT;CAEA,KAAK,MAAM,aAAa,YAAY,MAAM,SAAS;CACnD,OAAO;AACT;AAEA,SAAS,2BAAgD;CACvD,OAAO;EACL,WAAW;EACX,cAAc,iCAAiC,WAAW;GACxD,OAAO;GACP,gBAAgB;EAClB,EAAE;CACJ;AACF;AAEA,SAAS,gBAAgB,SAA8B,MAAsC;CAC3F,QAAQ,aAAa;CACrB,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,QAAQ,WAAW,KAAK,YAAY,iCAAiC,MAAO;EAClF,IAAI,UAAU,KAAA,GAAW;EACzB,MAAM,cAAc,QAAQ,aAAa;EACzC,YAAY,SAAS;EACrB,YAAY,kBAAkB;CAChC;AACF;AAEA,SAAS,uBAAuB,QAA6B,QAAmC;CAC9F,OAAO,aAAa,OAAO;CAC3B,KAAK,IAAI,QAAQ,GAAG,QAAQ,OAAO,aAAa,QAAQ,SAAS,GAAG;EAClE,MAAM,cAAc,OAAO,aAAa;EACxC,MAAM,oBAAoB,OAAO,aAAa;EAC9C,YAAY,SAAS,kBAAkB;EACvC,YAAY,kBAAkB,kBAAkB;CAClD;AACF;AAEA,SAAS,oCACP,MACA,aACS;CACT,IAAI,qBAAqB;CACzB,IAAI,yBAAyB;CAC7B,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,OAAO,iCAAiC;EAC9C,MAAM,cAAc,WAAW,KAAK,YAAY,IAAI;EACpD,MAAM,cAAc,YAAY,aAAa;EAC7C,IAAI,gBAAgB,KAAA,GAAW,sBAAsB;EACrD,IAAI,YAAY,iBAAiB,GAAG,0BAA0B;EAC9D,IAAI,gBAAgB,KAAA,KAAa,YAAY,mBAAmB,GAAG;EACnE,IAAI,YAAY,mBAAmB,YAAY,WAAW;EAC1D,IAAI,KAAK,IAAI,cAAc,YAAY,KAAK,IAAI,OAAO,OAAO;CAChE;CACA,OAAO,uBAAuB,KAAK,yBAAyB;AAC9D;AAEA,SAAS,YACP,MACA,MACU;CACV,MAAM,MAAgB,CAAC;CACvB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,KAAK,QAAQ;EACjB,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;CACA,OAAO;AACT;AAEA,SAAS,uBACP,MACA,MACA,cACoB;CACpB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,aAAa,IAAI,QAAQ,GAAG,OAAO;EACvC,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;AAEF;AAEA,SAAS,WACP,YACA,MACoB;CACpB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,WAAW;EACzB,MAAM,SACJ,OAAO,UAAU,WACb,QACA,OAAO,UAAU,YAAY,MAAM,SAAS,IAC1C,OAAO,KAAK,IACZ;EACR,IAAI,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACrD;AAEF;;;ACpaA,MAAM,0BAA0B;AAChC,MAAM,yBAAyB;AAW/B,SAAgB,sBACd,OACA,SACmB;CACnB,MAAM,eAAe,yBAAyB,OAAO,yBAAyB,OAAO;CACrF,MAAM,cAAc,yBAAyB,OAAO,wBAAwB,OAAO;CAEnF,IAAI,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,YAA4B,GACtF,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,wBAAwB,kBAAkB,gBAAgB,KAAK,IAAI,GACpF;CAEF,IAAI,gBAAgB,KAAA,MAAc,iBAAiB,KAAA,KAAa,iBAAiB,YAC/E,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,uBAAuB,0BAA0B,yBAClE;CAGF,IAAI,iBAAiB,KAAA,GAAW,OAAO,CAAC;CACxC,IAAI,iBAAiB,WAAW,OAAO,EAAE,aAAa;CACtD,OAAO;EACS;EACd,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;CACvC;AACF;AAEA,SAAS,yBACP,OACA,KACA,SACoB;CACpB,MAAM,yBAAS,IAAI,IAAY;CAC/B,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,OAAO,OAAO,KAAK,YAAY,GAAG,GAAG;EAC1C,MAAM,QAAQ,KAAK,WAAW;EAC9B,IAAI,OAAO,UAAU,YAAY,MAAM,KAAK,CAAC,CAAC,WAAW,GACvD,MAAM,IAAI,gBAAgB,GAAG,QAAQ,IAAI,IAAI,4BAA4B;EAE3E,OAAO,IAAI,KAAK;CAClB;CAEA,IAAI,OAAO,OAAO,GAChB,MAAM,IAAI,gBACR,GAAG,QAAQ,gBAAgB,IAAI,WAAW,CAAC,GAAG,MAAM,CAAC,CAAC,KAAK,CAAC,CAAC,KAAK,IAAI,GACxE;CAEF,OAAO,OAAO,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;AAChC;;;;;;;;;ACtCA,MAAM,mBAAmB,CAAC,aAAa;AACvC,MAAM,2BAA2B;CAAC;CAAa;CAAoB;AAAiB;AACpF,MAAM,2BAA2B,CAAC,2BAA2B,uBAAuB;AACpF,MAAM,oBAAoB,CAAC,yBAAyB,sBAAsB;AAC1E,MAAM,qBAAqB,CAAC,6BAA6B,uBAAuB;AAChF,MAAM,cAAc;CAClB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AACA,MAAM,mBAAmB;CACvB;CACA;CACA;CACA;CACA;AACF;AAEA,SAAS,aACP,QACA,KACqC;CACrC,MAAM,QAAQ,OAAO;CACrB,OAAO,SAAS,OAAO,UAAU,YAAY,CAAC,MAAM,QAAQ,KAAK,IAC5D,QACD,KAAA;AACN;AAEA,SAAS,aACP,QACA,YACA,WACoB;CACpB,KAAK,MAAM,aAAa,YAAY;EAClC,MAAM,QAAQ,gBAAgB,aAAa,QAAQ,SAAS,GAAG,SAAS;EACxE,IAAI,UAAU,KAAA,GAAW,OAAO;CAClC;AAEF;;;;;;AAOA,SAAgB,aAAa,MAAsC;CACjE,IAAI,CAAC,QAAQ,OAAO,SAAS,UAAU,OAAO;CAE9C,MAAM,QAAQ,iBAAiBA,IAAG;CAClC,MAAM,QAAQ,gBAAgB,OAAO,sBAAsB;CAC3D,MAAM,aAAa,gBAAgB,OAAO,gBAAgB;CAC1D,MAAM,SAAS,gBAAgB,OAAO,uBAAuB;CAC7D,MAAM,qBAAqB,gBAAgB,OAAO,wBAAwB;CAC1E,MAAM,qBACJ,gBAAgB,OAAO,wBAAwB,KAC/C,aAAa,OAAO,oBAAoB,CAAC,oBAAoB,iBAAiB,CAAC;CACjF,MAAM,YAAY,sBAAsB;CACxC,MAAM,cAAc,aAAa,OAAO,OAAO;CAC/C,MAAM,SACJ,gBAAgB,OAAO,WAAW,KAClC,gBAAgB,aAAa,CAAC,MAAM,CAAC,KACrC,aAAa,OAAO,mBAAmB,CAAC,iBAAiB,cAAc,CAAC;CAC1E,MAAM,aACJ,gBAAgB,OAAO,gBAAgB,KAAK,gBAAgB,aAAa,CAAC,OAAO,CAAC;CACpF,IACE,UAAU,KAAA,KACV,eAAe,KAAA,KACf,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,GAEf,OAAO;CACT,MAAM,SAAyB;EAC7B,QAAQ,SAAS,MAAM,cAAc;EACrC,SACG,WAAW,uBAAuB,KAAA,IAAY,qBAAqB,OACnE,sBAAsB;CAC3B;CACA,IAAI,cAAc,KAAA,GAAW,OAAO,YAAY;CAChD,IAAI,WAAW,KAAA,GAAW,OAAO,SAAS;CAC1C,IAAI,eAAe,KAAA,GAAW,OAAO,aAAa;CAClD,OAAO;AACT;;;;;;AAOA,SAAgB,oBACd,MACA,UAAsC,CAAC,GAChB;CACvB,MAAM,OAAO,QAAQ,QAAQ;CAC7B,IAAI,QAAQ;CACZ,IAAI,SAAS;CACb,IAAI,YAAY;CAChB,IAAI,eAAe;CACnB,IAAI,SAAS;CACb,IAAI,YAAY;CAChB,IAAI,aAAa;CACjB,IAAI,gBAAgB;CACpB,IAAI,QAAQ;CACZ,MAAM,SAAS,IAAI,eAAwB,EAAE,WAAW,aAAa,CAAC;CACtE,MAAM,SAAS,CAAC,GAAG,OAAO,KAAK,IAAI,GAAG,GAAG,OAAO,MAAM,CAAC;CACvD,MAAM,QAAQ,SAAS,WAAW,SAAiB,SAAiB,UAAU,OAAO,KAAK;CAE1F,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,QAAQ,aAAa,MAAM,IAAI;EACrC,IAAI,CAAC,OAAO;EACZ,QAAQ,MAAM,OAAO,MAAM,KAAK;EAChC,SAAS,MAAM,QAAQ,MAAM,MAAM;EACnC,IAAI,MAAM,cAAc,KAAA,GAAW;GACjC,YAAY,MAAM,WAAW,MAAM,SAAS;GAC5C,eAAe;EACjB;EACA,IAAI,MAAM,WAAW,KAAA,GAAW;GAC9B,SAAS,MAAM,QAAQ,MAAM,MAAM;GACnC,YAAY;EACd;EACA,IAAI,MAAM,eAAe,KAAA,GAAW;GAClC,aAAa,MAAM,YAAY,MAAM,UAAU;GAC/C,gBAAgB;EAClB;EACA,QAAQ;CACV;CACA,IAAI,CAAC,OAAO,OAAO;CACnB,OAAO;EACL;EACA;EACA,GAAI,eAAe,EAAE,UAAU,IAAI,CAAC;EACpC,GAAI,YAAY,EAAE,OAAO,IAAI,CAAC;EAC9B,GAAI,gBAAgB,EAAE,WAAW,IAAI,CAAC;CACxC;AACF;AAEA,SAAS,aAAa,KAA6B;CACjD,MAAM,UAAU,IAAI,KAAK;CACzB,IAAI,CAAC,WAAW,YAAY,UAAU,OAAO;CAC7C,IAAI;EACF,OAAO,KAAK,MAAM,OAAO;CAC3B,QAAQ;EACN,OAAO;CACT;AACF;;;;;;;AAQA,eAAsB,yBACpB,UACA,YACgC;CAChC,IAAI;CACJ,IAAI;EACF,OAAO,MAAM,SAAS,MAAM,CAAC,CAAC,KAAK;CACrC,QAAQ;EACN,OAAO;CACT;CACA,IAAI;CACJ,IAAI;EACF,OAAO,KAAK,MAAM,IAAI;CACxB,QAAQ;EACN,OAAO,KAAA;CACT;CACA,OAAO,aAAa,IAAI,KAAK,oBAAoB,MAAM,UAAU;AACnE"}
|
|
1
|
+
{"version":3,"file":"extract-usage-C5vMw-0R.js","names":["obj"],"sources":["../src/trace/error-classification.ts","../src/trace/execution-measurements.ts","../src/trace/task-failure-attributes.ts","../src/trace/extract-usage.ts"],"sourcesContent":["import type { OtlpSpanRole } from './otlp-attributes'\n\nexport type TraceErrorRole = OtlpSpanRole\n\nexport interface TraceErrorSignal {\n id: string\n parentId?: string\n role: TraceErrorRole\n error: boolean\n processRoot: boolean\n}\n\nexport interface TraceErrorSummary {\n total: number\n execution: number\n process: number\n guardrail: number\n evaluation: number\n propagated: number\n unclassified: number\n}\n\n/**\n * Classify errored spans without counting a propagated parent status as a\n * second execution failure.\n */\nexport function summarizeTraceErrors(signals: readonly TraceErrorSignal[]): TraceErrorSummary {\n const byId = new Map<string, TraceErrorSignal>()\n for (const signal of signals) {\n if (byId.has(signal.id)) {\n throw new Error(`summarizeTraceErrors: duplicate span id '${signal.id}'`)\n }\n byId.set(signal.id, signal)\n }\n\n const propagated = new Set<string>()\n for (const signal of signals) {\n if (!signal.error) continue\n const visited = new Set<string>()\n let parentId = signal.parentId\n while (parentId && !visited.has(parentId)) {\n visited.add(parentId)\n const parent = byId.get(parentId)\n if (!parent) break\n if (parent.error) propagated.add(parent.id)\n parentId = parent.parentId\n }\n }\n\n const summary: TraceErrorSummary = {\n total: 0,\n execution: 0,\n process: 0,\n guardrail: 0,\n evaluation: 0,\n propagated: 0,\n unclassified: 0,\n }\n\n for (const signal of signals) {\n if (!signal.error) continue\n summary.total += 1\n if (signal.role === 'GUARDRAIL') {\n summary.guardrail += 1\n } else if (signal.role === 'EVALUATOR') {\n summary.evaluation += 1\n } else if (signal.processRoot) {\n summary.process += 1\n } else if (propagated.has(signal.id)) {\n summary.propagated += 1\n } else if (\n signal.role === 'AGENT' ||\n signal.role === 'CHAIN' ||\n signal.role === 'LLM' ||\n signal.role === 'TOOL'\n ) {\n summary.execution += 1\n } else {\n summary.unclassified += 1\n }\n }\n\n return summary\n}\n","import type { RunTokenUsage } from '../run-record'\nimport {\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_COST_ATTR_KEYS,\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n RUN_COST_ATTR_KEYS,\n} from './otlp-attributes'\n\nexport interface ExecutionMeasurementSpan {\n id: string\n parentId?: string\n attributes: Record<string, unknown>\n modelCall: boolean\n aggregate: boolean\n}\n\nexport interface MeasurementCoverage {\n value?: number\n reportingCalls: number\n complete: boolean\n}\n\nexport interface ExecutionMeasurements {\n tokenUsage: RunTokenUsage\n modelCallCount: number\n callSpanIds: string[]\n cost: MeasurementCoverage\n aggregate?: {\n tokenUsage: RunTokenUsage\n costUsd?: number\n }\n}\n\nconst TOKEN_MEASUREMENT_KEY_GROUPS = [\n LLM_INPUT_TOKEN_ATTR_KEYS,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n LLM_CACHED_TOKEN_ATTR_KEYS,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n] as const\n\nconst EXECUTION_MEASUREMENT_KEY_GROUPS = [\n ...TOKEN_MEASUREMENT_KEY_GROUPS,\n LLM_COST_ATTR_KEYS,\n] as const\n\ninterface RetainedCallSummary {\n callCount: number\n measurements: Array<{\n total: number\n reportingCalls: number\n }>\n}\n\n/**\n * Reconcile execution measurements across nested telemetry wrappers.\n * A measured parent is used only when a descendant call does not report the\n * same field, so aggregate wrappers neither duplicate complete child data nor\n * erase complementary parent fields.\n */\nexport function summarizeExecutionMeasurements(\n spans: ExecutionMeasurementSpan[],\n): ExecutionMeasurements {\n const byId = new Map<string, ExecutionMeasurementSpan>()\n for (const span of spans) {\n if (byId.has(span.id)) {\n throw new Error(`summarizeExecutionMeasurements: duplicate span id \"${span.id}\"`)\n }\n byId.set(span.id, span)\n }\n const tokenMeasurementKeys = TOKEN_MEASUREMENT_KEY_GROUPS.flat()\n const candidates = spans.filter(\n (span) =>\n span.modelCall ||\n (!span.aggregate && readNumber(span.attributes, tokenMeasurementKeys) !== undefined),\n )\n const candidateIds = new Set(candidates.map((span) => span.id))\n const candidateChildren = new Map<string, ExecutionMeasurementSpan[]>()\n for (const candidate of candidates) {\n const parentId = nearestCandidateParent(candidate, byId, candidateIds)\n if (!parentId) continue\n const children = candidateChildren.get(parentId) ?? []\n children.push(candidate)\n candidateChildren.set(parentId, children)\n }\n const aggregateIds = classifyAggregateSpans(candidates, candidateChildren)\n const untypedRunCostIds = new Set(\n spans\n .filter(\n (span) =>\n !span.modelCall &&\n !span.aggregate &&\n !candidateIds.has(span.id) &&\n readNumber(span.attributes, RUN_COST_ATTR_KEYS) !== undefined,\n )\n .map((span) => span.id),\n )\n const aggregateSourceIds = new Set(\n spans\n .filter(\n (span) => span.aggregate || aggregateIds.has(span.id) || untypedRunCostIds.has(span.id),\n )\n .map((span) => span.id),\n )\n\n const calls = candidates.filter((span) => !aggregateIds.has(span.id))\n const input = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_INPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const output = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_OUTPUT_TOKEN_ATTR_KEYS,\n LLM_REASONING_TOKEN_ATTR_KEYS,\n )\n const cached = reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileMeasurement(\n calls,\n byId,\n aggregateSourceIds,\n LLM_CACHE_WRITE_TOKEN_ATTR_KEYS,\n )\n const aggregate = summarizeAggregateMeasurements(\n spans,\n byId,\n new Set([...aggregateIds, ...untypedRunCostIds]),\n )\n\n return {\n tokenUsage: {\n input: input.value ?? 0,\n output: Math.max(output.value ?? 0, reasoning.value ?? 0),\n ...(reasoning.value !== undefined ? { reasoning: reasoning.value } : {}),\n ...(cached.value !== undefined ? { cached: cached.value } : {}),\n ...(cacheWrite.value !== undefined ? { cacheWrite: cacheWrite.value } : {}),\n },\n modelCallCount: calls.length,\n callSpanIds: calls.map((span) => span.id),\n cost: reconcileMeasurement(calls, byId, aggregateSourceIds, LLM_COST_ATTR_KEYS),\n ...(aggregate ? { aggregate } : {}),\n }\n}\n\nexport function recordAggregateMeasurements(\n raw: Record<string, number>,\n aggregate: ExecutionMeasurements['aggregate'],\n): void {\n if (!aggregate) return\n raw.aggregate_prompt_tokens = aggregate.tokenUsage.input\n raw.aggregate_completion_tokens = aggregate.tokenUsage.output\n if (aggregate.tokenUsage.reasoning !== undefined)\n raw.aggregate_reasoning_tokens = aggregate.tokenUsage.reasoning\n if (aggregate.tokenUsage.cached !== undefined)\n raw.aggregate_cached_tokens = aggregate.tokenUsage.cached\n if (aggregate.tokenUsage.cacheWrite !== undefined)\n raw.aggregate_cache_write_tokens = aggregate.tokenUsage.cacheWrite\n if (aggregate.costUsd !== undefined) raw.aggregate_cost_usd = aggregate.costUsd\n}\n\nfunction summarizeAggregateMeasurements(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateIds: Set<string>,\n): ExecutionMeasurements['aggregate'] {\n const aggregates = spans.filter((span) => span.aggregate || aggregateIds.has(span.id))\n const input = reconcileTopLevelMeasurement(aggregates, byId, LLM_INPUT_TOKEN_ATTR_KEYS)\n const output = reconcileTopLevelMeasurement(aggregates, byId, LLM_OUTPUT_TOKEN_ATTR_KEYS)\n const reasoning = reconcileTopLevelMeasurement(aggregates, byId, LLM_REASONING_TOKEN_ATTR_KEYS)\n const cached = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHED_TOKEN_ATTR_KEYS)\n const cacheWrite = reconcileTopLevelMeasurement(aggregates, byId, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS)\n const costUsd = reconcileTopLevelMeasurement(aggregates, byId, LLM_COST_ATTR_KEYS)\n if (\n input === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined &&\n costUsd === undefined\n )\n return undefined\n return {\n tokenUsage: {\n input: input ?? 0,\n output: output ?? reasoning ?? 0,\n ...(reasoning !== undefined ? { reasoning } : {}),\n ...(cached !== undefined ? { cached } : {}),\n ...(cacheWrite !== undefined ? { cacheWrite } : {}),\n },\n ...(costUsd !== undefined ? { costUsd } : {}),\n }\n}\n\nfunction reconcileTopLevelMeasurement(\n spans: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n keys: readonly string[],\n): number | undefined {\n const selected = new Map<string, number>()\n for (const span of spans) {\n const value = readNumber(span.attributes, keys)\n if (value !== undefined) selected.set(span.id, value)\n }\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (ancestorIds(span, byId).some((ancestorId) => selected.has(ancestorId))) {\n selected.delete(spanId)\n }\n }\n return selected.size > 0\n ? [...selected.values()].reduce((total, value) => total + value, 0)\n : undefined\n}\n\nfunction reconcileMeasurement(\n calls: ExecutionMeasurementSpan[],\n byId: Map<string, ExecutionMeasurementSpan>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n fallbackKeys?: readonly string[],\n): MeasurementCoverage {\n const selected = new Map<string, number>()\n const callIds = new Set(calls.map((call) => call.id))\n let reportingCalls = 0\n\n for (const call of calls) {\n const primary = nearestMeasurement(call, byId, callIds, aggregateSourceIds, keys)\n const fallback = fallbackKeys\n ? nearestMeasurement(call, byId, callIds, aggregateSourceIds, fallbackKeys)\n : undefined\n const source =\n primary && fallback && primary.span.id === fallback.span.id\n ? { span: primary.span, value: Math.max(primary.value, fallback.value) }\n : (primary ?? fallback)\n if (!source) continue\n reportingCalls += 1\n selected.set(source.span.id, source.value)\n }\n\n for (const spanId of [...selected.keys()]) {\n const span = byId.get(spanId)\n if (!span) continue\n if (\n ancestorIds(span, byId).some(\n (ancestorId) => selected.has(ancestorId) && !callIds.has(ancestorId),\n )\n ) {\n selected.delete(spanId)\n }\n }\n\n return {\n ...(selected.size > 0\n ? { value: [...selected.values()].reduce((total, value) => total + value, 0) }\n : {}),\n reportingCalls,\n complete: calls.length > 0 && reportingCalls === calls.length,\n }\n}\n\nfunction nearestMeasurement(\n call: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n callIds: Set<string>,\n aggregateSourceIds: Set<string>,\n keys: readonly string[],\n): { span: ExecutionMeasurementSpan; value: number } | undefined {\n let current: ExecutionMeasurementSpan | undefined = call\n const seen = new Set<string>()\n while (current && !seen.has(current.id)) {\n seen.add(current.id)\n const value = readNumber(current.attributes, keys)\n if (\n value !== undefined &&\n (current.id === call.id || (!callIds.has(current.id) && aggregateSourceIds.has(current.id)))\n ) {\n return { span: current, value }\n }\n current = current.parentId ? byId.get(current.parentId) : undefined\n }\n return undefined\n}\n\nfunction classifyAggregateSpans(\n candidates: ExecutionMeasurementSpan[],\n childrenById: Map<string, ExecutionMeasurementSpan[]>,\n): Set<string> {\n const aggregateIds = new Set<string>()\n const summaries = new Map<string, RetainedCallSummary>()\n const visiting = new Set<string>()\n\n const visit = (span: ExecutionMeasurementSpan): RetainedCallSummary => {\n const cached = summaries.get(span.id)\n if (cached) return cached\n if (visiting.has(span.id)) return emptyRetainedCallSummary()\n visiting.add(span.id)\n\n const descendants = emptyRetainedCallSummary()\n for (const child of childrenById.get(span.id) ?? []) {\n const childDescendants = visit(child)\n if (!aggregateIds.has(child.id)) addRetainedCall(descendants, child)\n addRetainedCallSummary(descendants, childDescendants)\n }\n\n if (\n descendants.callCount > 0 &&\n (!span.modelCall || hasCompatibleDescendantMeasurements(span, descendants))\n ) {\n aggregateIds.add(span.id)\n }\n\n visiting.delete(span.id)\n summaries.set(span.id, descendants)\n return descendants\n }\n\n for (const candidate of candidates) visit(candidate)\n return aggregateIds\n}\n\nfunction emptyRetainedCallSummary(): RetainedCallSummary {\n return {\n callCount: 0,\n measurements: EXECUTION_MEASUREMENT_KEY_GROUPS.map(() => ({\n total: 0,\n reportingCalls: 0,\n })),\n }\n}\n\nfunction addRetainedCall(summary: RetainedCallSummary, span: ExecutionMeasurementSpan): void {\n summary.callCount += 1\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const value = readNumber(span.attributes, EXECUTION_MEASUREMENT_KEY_GROUPS[index]!)\n if (value === undefined) continue\n const measurement = summary.measurements[index]!\n measurement.total += value\n measurement.reportingCalls += 1\n }\n}\n\nfunction addRetainedCallSummary(target: RetainedCallSummary, source: RetainedCallSummary): void {\n target.callCount += source.callCount\n for (let index = 0; index < target.measurements.length; index += 1) {\n const measurement = target.measurements[index]!\n const sourceMeasurement = source.measurements[index]!\n measurement.total += sourceMeasurement.total\n measurement.reportingCalls += sourceMeasurement.reportingCalls\n }\n}\n\nfunction hasCompatibleDescendantMeasurements(\n span: ExecutionMeasurementSpan,\n descendants: RetainedCallSummary,\n): boolean {\n let parentMeasurements = 0\n let descendantMeasurements = 0\n for (let index = 0; index < EXECUTION_MEASUREMENT_KEY_GROUPS.length; index += 1) {\n const keys = EXECUTION_MEASUREMENT_KEY_GROUPS[index]!\n const parentValue = readNumber(span.attributes, keys)\n const measurement = descendants.measurements[index]!\n if (parentValue !== undefined) parentMeasurements += 1\n if (measurement.reportingCalls > 0) descendantMeasurements += 1\n if (parentValue === undefined || measurement.reportingCalls === 0) continue\n if (measurement.reportingCalls !== descendants.callCount) continue\n if (Math.abs(parentValue - measurement.total) > 1e-12) return false\n }\n return parentMeasurements === 0 || descendantMeasurements > 0\n}\n\nfunction ancestorIds(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n): string[] {\n const ids: string[] = []\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n ids.push(parentId)\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return ids\n}\n\nfunction nearestCandidateParent(\n span: ExecutionMeasurementSpan,\n byId: Map<string, ExecutionMeasurementSpan>,\n candidateIds: Set<string>,\n): string | undefined {\n const seen = new Set<string>()\n let parentId = span.parentId\n while (parentId && !seen.has(parentId)) {\n if (candidateIds.has(parentId)) return parentId\n seen.add(parentId)\n parentId = byId.get(parentId)?.parentId\n }\n return undefined\n}\n\nfunction readNumber(\n attributes: Record<string, unknown>,\n keys: readonly string[],\n): number | undefined {\n for (const key of keys) {\n const value = attributes[key]\n const parsed =\n typeof value === 'number'\n ? value\n : typeof value === 'string' && value.length > 0\n ? Number(value)\n : Number.NaN\n if (Number.isFinite(parsed) && parsed >= 0) return parsed\n }\n return undefined\n}\n","import { ValidationError } from '../errors'\nimport { FAILURE_CLASSES, type FailureClass } from './schema'\n\nconst TASK_FAILURE_CLASS_ATTR = 'tangle.task.failure_class'\nconst TASK_FAILURE_MODE_ATTR = 'tangle.task.failure_mode'\n\ninterface AttributeCarrier {\n attributes: Record<string, unknown>\n}\n\nexport type TaskFailureLabels =\n | { failureClass?: undefined; failureMode?: undefined }\n | { failureClass: 'success'; failureMode?: undefined }\n | { failureClass: Exclude<FailureClass, 'success'>; failureMode?: string }\n\nexport function readTaskFailureLabels(\n roots: readonly AttributeCarrier[],\n context: string,\n): TaskFailureLabels {\n const failureClass = readConsistentRootString(roots, TASK_FAILURE_CLASS_ATTR, context)\n const failureMode = readConsistentRootString(roots, TASK_FAILURE_MODE_ATTR, context)\n\n if (failureClass !== undefined && !FAILURE_CLASSES.includes(failureClass as FailureClass)) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_CLASS_ATTR} must be one of ${FAILURE_CLASSES.join(', ')}`,\n )\n }\n if (failureMode !== undefined && (failureClass === undefined || failureClass === 'success')) {\n throw new ValidationError(\n `${context}: ${TASK_FAILURE_MODE_ATTR} requires a non-success ${TASK_FAILURE_CLASS_ATTR}`,\n )\n }\n\n if (failureClass === undefined) return {}\n if (failureClass === 'success') return { failureClass }\n return {\n failureClass: failureClass as Exclude<FailureClass, 'success'>,\n ...(failureMode ? { failureMode } : {}),\n }\n}\n\nfunction readConsistentRootString(\n roots: readonly AttributeCarrier[],\n key: string,\n context: string,\n): string | undefined {\n const values = new Set<string>()\n for (const root of roots) {\n if (!Object.hasOwn(root.attributes, key)) continue\n const value = root.attributes[key]\n if (typeof value !== 'string' || value.trim().length === 0) {\n throw new ValidationError(`${context}: ${key} must be a non-empty string`)\n }\n values.add(value)\n }\n\n if (values.size > 1) {\n throw new ValidationError(\n `${context}: conflicting ${key} values: ${[...values].sort().join(', ')}`,\n )\n }\n return values.values().next().value\n}\n","/**\n * Provider response and SSE usage extraction.\n *\n * Missing usage returns `null`; reported zeroes and cache-only activity remain\n * distinguishable from absent telemetry.\n */\n\nimport { SSEChunkParser } from '@tangle-network/agent-core/sse'\nimport {\n firstTokenCount,\n TOKEN_USAGE_INPUT_KEYS,\n TOKEN_USAGE_OUTPUT_KEYS,\n tokenUsageSource,\n} from '@tangle-network/agent-core/telemetry'\nimport type { RunTokenUsage } from '../run-record'\n\nexport type ExtractedUsage = RunTokenUsage\nexport type SseUsageMode = 'cumulative' | 'delta'\n\nexport interface ExtractUsageFromSseOptions {\n /** Provider usage events are cumulative snapshots unless explicitly marked as deltas. */\n mode?: SseUsageMode\n}\n\nconst INPUT_OTHER_KEYS = ['input_other'] as const\nconst EXCLUSIVE_REASONING_KEYS = ['reasoning', 'reasoning_tokens', 'reasoningTokens'] as const\nconst INCLUSIVE_REASONING_KEYS = ['reasoning_output_tokens', 'reasoningOutputTokens'] as const\nconst INPUT_DETAIL_KEYS = ['prompt_tokens_details', 'input_tokens_details'] as const\nconst OUTPUT_DETAIL_KEYS = ['completion_tokens_details', 'output_tokens_details'] as const\nconst CACHED_KEYS = [\n 'cache',\n 'cached_tokens',\n 'cached_input_tokens',\n 'cache_read_tokens',\n 'cache_read_input_tokens',\n 'cachedTokens',\n 'cachedInputTokens',\n 'cacheReadTokens',\n 'cacheReadInputTokens',\n 'input_cache_read',\n] as const\nconst CACHE_WRITE_KEYS = [\n 'cache_creation_tokens',\n 'cache_creation_input_tokens',\n 'cacheCreationTokens',\n 'cacheCreationInputTokens',\n 'input_cache_creation',\n] as const\n\nfunction nestedRecord(\n source: Record<string, unknown>,\n key: string,\n): Record<string, unknown> | undefined {\n const value = source[key]\n return value && typeof value === 'object' && !Array.isArray(value)\n ? (value as Record<string, unknown>)\n : undefined\n}\n\nfunction nestedNumber(\n source: Record<string, unknown>,\n recordKeys: readonly string[],\n valueKeys: readonly string[],\n): number | undefined {\n for (const recordKey of recordKeys) {\n const value = firstTokenCount(nestedRecord(source, recordKey), valueKeys)\n if (value !== undefined) return value\n }\n return undefined\n}\n\n/**\n * Pull `{ input, output, cached?, cacheWrite? }` from a parsed response\n * body. Accepts a top-level `usage` object (the common case) or a body that IS\n * the usage object. Returns null only when none of those categories is present.\n */\nexport function extractUsage(body: unknown): ExtractedUsage | null {\n if (!body || typeof body !== 'object') return null\n const obj = body as Record<string, unknown>\n const usage = tokenUsageSource(obj)\n const input = firstTokenCount(usage, TOKEN_USAGE_INPUT_KEYS)\n const inputOther = firstTokenCount(usage, INPUT_OTHER_KEYS)\n const output = firstTokenCount(usage, TOKEN_USAGE_OUTPUT_KEYS)\n const exclusiveReasoning = firstTokenCount(usage, EXCLUSIVE_REASONING_KEYS)\n const inclusiveReasoning =\n firstTokenCount(usage, INCLUSIVE_REASONING_KEYS) ??\n nestedNumber(usage, OUTPUT_DETAIL_KEYS, ['reasoning_tokens', 'reasoningTokens'])\n const reasoning = exclusiveReasoning ?? inclusiveReasoning\n const nestedCache = nestedRecord(usage, 'cache')\n const cached =\n firstTokenCount(usage, CACHED_KEYS) ??\n firstTokenCount(nestedCache, ['read']) ??\n nestedNumber(usage, INPUT_DETAIL_KEYS, ['cached_tokens', 'cachedTokens'])\n const cacheWrite =\n firstTokenCount(usage, CACHE_WRITE_KEYS) ?? firstTokenCount(nestedCache, ['write'])\n if (\n input === undefined &&\n inputOther === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined\n )\n return null\n const result: ExtractedUsage = {\n input: (input ?? 0) + (inputOther ?? 0),\n output:\n (output ?? (inclusiveReasoning !== undefined ? inclusiveReasoning : 0)) +\n (exclusiveReasoning ?? 0),\n }\n if (reasoning !== undefined) result.reasoning = reasoning\n if (cached !== undefined) result.cached = cached\n if (cacheWrite !== undefined) result.cacheWrite = cacheWrite\n return result\n}\n\n/**\n * Extract token usage from a complete SSE response body using the shared SSE\n * frame parser. Cumulative snapshots are the fail-safe default because summing\n * them inflates billing; callers with explicit delta events opt into `mode: 'delta'`.\n */\nexport function extractUsageFromSse(\n text: string,\n options: ExtractUsageFromSseOptions = {},\n): ExtractedUsage | null {\n const mode = options.mode ?? 'cumulative'\n let input = 0\n let output = 0\n let reasoning = 0\n let sawReasoning = false\n let cached = 0\n let sawCached = false\n let cacheWrite = 0\n let sawCacheWrite = false\n let found = false\n const parser = new SSEChunkParser<unknown>({ transform: parseSseJson })\n const events = [...parser.push(text), ...parser.flush()]\n const merge = mode === 'delta' ? (current: number, next: number) => current + next : Math.max\n\n for (const event of events) {\n const usage = extractUsage(event.data)\n if (!usage) continue\n input = merge(input, usage.input)\n output = merge(output, usage.output)\n if (usage.reasoning !== undefined) {\n reasoning = merge(reasoning, usage.reasoning)\n sawReasoning = true\n }\n if (usage.cached !== undefined) {\n cached = merge(cached, usage.cached)\n sawCached = true\n }\n if (usage.cacheWrite !== undefined) {\n cacheWrite = merge(cacheWrite, usage.cacheWrite)\n sawCacheWrite = true\n }\n found = true\n }\n if (!found) return null\n return {\n input,\n output,\n ...(sawReasoning ? { reasoning } : {}),\n ...(sawCached ? { cached } : {}),\n ...(sawCacheWrite ? { cacheWrite } : {}),\n }\n}\n\nfunction parseSseJson(raw: string): unknown | null {\n const payload = raw.trim()\n if (!payload || payload === '[DONE]') return null\n try {\n return JSON.parse(payload)\n } catch {\n return null\n }\n}\n\n/**\n * Extract usage from an HTTP `Response` without consuming the caller's body:\n * clones, reads the text, and tries the JSON parser first, then the SSE\n * accumulator. Best-effort — returns null on any read/parse miss so a usage tee\n * never takes down the underlying call.\n */\nexport async function extractUsageFromResponse(\n response: Response,\n sseOptions?: ExtractUsageFromSseOptions,\n): Promise<ExtractedUsage | null> {\n let text: string\n try {\n text = await response.clone().text()\n } catch {\n return null\n }\n let json: unknown\n try {\n json = JSON.parse(text)\n } catch {\n json = undefined\n }\n return extractUsage(json) ?? extractUsageFromSse(text, sseOptions)\n}\n"],"mappings":";;;;;;;;;;;AA0BA,SAAgB,qBAAqB,SAAyD;CAC5F,MAAM,uBAAO,IAAI,IAA8B;CAC/C,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,KAAK,IAAI,OAAO,EAAE,GACpB,MAAM,IAAI,MAAM,4CAA4C,OAAO,GAAG,EAAE;EAE1E,KAAK,IAAI,OAAO,IAAI,MAAM;CAC5B;CAEA,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,MAAM,0BAAU,IAAI,IAAY;EAChC,IAAI,WAAW,OAAO;EACtB,OAAO,YAAY,CAAC,QAAQ,IAAI,QAAQ,GAAG;GACzC,QAAQ,IAAI,QAAQ;GACpB,MAAM,SAAS,KAAK,IAAI,QAAQ;GAChC,IAAI,CAAC,QAAQ;GACb,IAAI,OAAO,OAAO,WAAW,IAAI,OAAO,EAAE;GAC1C,WAAW,OAAO;EACpB;CACF;CAEA,MAAM,UAA6B;EACjC,OAAO;EACP,WAAW;EACX,SAAS;EACT,WAAW;EACX,YAAY;EACZ,YAAY;EACZ,cAAc;CAChB;CAEA,KAAK,MAAM,UAAU,SAAS;EAC5B,IAAI,CAAC,OAAO,OAAO;EACnB,QAAQ,SAAS;EACjB,IAAI,OAAO,SAAS,aAClB,QAAQ,aAAa;OAChB,IAAI,OAAO,SAAS,aACzB,QAAQ,cAAc;OACjB,IAAI,OAAO,aAChB,QAAQ,WAAW;OACd,IAAI,WAAW,IAAI,OAAO,EAAE,GACjC,QAAQ,cAAc;OACjB,IACL,OAAO,SAAS,WAChB,OAAO,SAAS,WAChB,OAAO,SAAS,SAChB,OAAO,SAAS,QAEhB,QAAQ,aAAa;OAErB,QAAQ,gBAAgB;CAE5B;CAEA,OAAO;AACT;;;AC/CA,MAAM,+BAA+B;CACnC;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,mCAAmC,CACvC,GAAG,8BACH,kBACF;;;;;;;AAgBA,SAAgB,+BACd,OACuB;CACvB,MAAM,uBAAO,IAAI,IAAsC;CACvD,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,KAAK,IAAI,KAAK,EAAE,GAClB,MAAM,IAAI,MAAM,sDAAsD,KAAK,GAAG,EAAE;EAElF,KAAK,IAAI,KAAK,IAAI,IAAI;CACxB;CACA,MAAM,uBAAuB,6BAA6B,KAAK;CAC/D,MAAM,aAAa,MAAM,QACtB,SACC,KAAK,aACJ,CAAC,KAAK,aAAa,WAAW,KAAK,YAAY,oBAAoB,MAAM,KAAA,CAC9E;CACA,MAAM,eAAe,IAAI,IAAI,WAAW,KAAK,SAAS,KAAK,EAAE,CAAC;CAC9D,MAAM,oCAAoB,IAAI,IAAwC;CACtE,KAAK,MAAM,aAAa,YAAY;EAClC,MAAM,WAAW,uBAAuB,WAAW,MAAM,YAAY;EACrE,IAAI,CAAC,UAAU;EACf,MAAM,WAAW,kBAAkB,IAAI,QAAQ,KAAK,CAAC;EACrD,SAAS,KAAK,SAAS;EACvB,kBAAkB,IAAI,UAAU,QAAQ;CAC1C;CACA,MAAM,eAAe,uBAAuB,YAAY,iBAAiB;CACzE,MAAM,oBAAoB,IAAI,IAC5B,MACG,QACE,SACC,CAAC,KAAK,aACN,CAAC,KAAK,aACN,CAAC,aAAa,IAAI,KAAK,EAAE,KACzB,WAAW,KAAK,YAAY,kBAAkB,MAAM,KAAA,CACxD,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CACA,MAAM,qBAAqB,IAAI,IAC7B,MACG,QACE,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,KAAK,kBAAkB,IAAI,KAAK,EAAE,CACxF,CAAC,CACA,KAAK,SAAS,KAAK,EAAE,CAC1B;CAEA,MAAM,QAAQ,WAAW,QAAQ,SAAS,CAAC,aAAa,IAAI,KAAK,EAAE,CAAC;CACpE,MAAM,QAAQ,qBAAqB,OAAO,MAAM,oBAAoB,yBAAyB;CAC7F,MAAM,YAAY,qBAChB,OACA,MACA,oBACA,6BACF;CACA,MAAM,SAAS,qBACb,OACA,MACA,oBACA,4BACA,6BACF;CACA,MAAM,SAAS,qBAAqB,OAAO,MAAM,oBAAoB,0BAA0B;CAC/F,MAAM,aAAa,qBACjB,OACA,MACA,oBACA,+BACF;CACA,MAAM,YAAY,+BAChB,OACA,sBACA,IAAI,IAAI,CAAC,GAAG,cAAc,GAAG,iBAAiB,CAAC,CACjD;CAEA,OAAO;EACL,YAAY;GACV,OAAO,MAAM,SAAS;GACtB,QAAQ,KAAK,IAAI,OAAO,SAAS,GAAG,UAAU,SAAS,CAAC;GACxD,GAAI,UAAU,UAAU,KAAA,IAAY,EAAE,WAAW,UAAU,MAAM,IAAI,CAAC;GACtE,GAAI,OAAO,UAAU,KAAA,IAAY,EAAE,QAAQ,OAAO,MAAM,IAAI,CAAC;GAC7D,GAAI,WAAW,UAAU,KAAA,IAAY,EAAE,YAAY,WAAW,MAAM,IAAI,CAAC;EAC3E;EACA,gBAAgB,MAAM;EACtB,aAAa,MAAM,KAAK,SAAS,KAAK,EAAE;EACxC,MAAM,qBAAqB,OAAO,MAAM,oBAAoB,kBAAkB;EAC9E,GAAI,YAAY,EAAE,UAAU,IAAI,CAAC;CACnC;AACF;AAEA,SAAgB,4BACd,KACA,WACM;CACN,IAAI,CAAC,WAAW;CAChB,IAAI,0BAA0B,UAAU,WAAW;CACnD,IAAI,8BAA8B,UAAU,WAAW;CACvD,IAAI,UAAU,WAAW,cAAc,KAAA,GACrC,IAAI,6BAA6B,UAAU,WAAW;CACxD,IAAI,UAAU,WAAW,WAAW,KAAA,GAClC,IAAI,0BAA0B,UAAU,WAAW;CACrD,IAAI,UAAU,WAAW,eAAe,KAAA,GACtC,IAAI,+BAA+B,UAAU,WAAW;CAC1D,IAAI,UAAU,YAAY,KAAA,GAAW,IAAI,qBAAqB,UAAU;AAC1E;AAEA,SAAS,+BACP,OACA,MACA,cACoC;CACpC,MAAM,aAAa,MAAM,QAAQ,SAAS,KAAK,aAAa,aAAa,IAAI,KAAK,EAAE,CAAC;CACrF,MAAM,QAAQ,6BAA6B,YAAY,MAAM,yBAAyB;CACtF,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,YAAY,6BAA6B,YAAY,MAAM,6BAA6B;CAC9F,MAAM,SAAS,6BAA6B,YAAY,MAAM,0BAA0B;CACxF,MAAM,aAAa,6BAA6B,YAAY,MAAM,+BAA+B;CACjG,MAAM,UAAU,6BAA6B,YAAY,MAAM,kBAAkB;CACjF,IACE,UAAU,KAAA,KACV,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,KACf,YAAY,KAAA,GAEZ,OAAO,KAAA;CACT,OAAO;EACL,YAAY;GACV,OAAO,SAAS;GAChB,QAAQ,UAAU,aAAa;GAC/B,GAAI,cAAc,KAAA,IAAY,EAAE,UAAU,IAAI,CAAC;GAC/C,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;GACzC,GAAI,eAAe,KAAA,IAAY,EAAE,WAAW,IAAI,CAAC;EACnD;EACA,GAAI,YAAY,KAAA,IAAY,EAAE,QAAQ,IAAI,CAAC;CAC7C;AACF;AAEA,SAAS,6BACP,OACA,MACA,MACoB;CACpB,MAAM,2BAAW,IAAI,IAAoB;CACzC,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,QAAQ,WAAW,KAAK,YAAY,IAAI;EAC9C,IAAI,UAAU,KAAA,GAAW,SAAS,IAAI,KAAK,IAAI,KAAK;CACtD;CACA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IAAI,YAAY,MAAM,IAAI,CAAC,CAAC,MAAM,eAAe,SAAS,IAAI,UAAU,CAAC,GACvE,SAAS,OAAO,MAAM;CAE1B;CACA,OAAO,SAAS,OAAO,IACnB,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,IAChE,KAAA;AACN;AAEA,SAAS,qBACP,OACA,MACA,oBACA,MACA,cACqB;CACrB,MAAM,2BAAW,IAAI,IAAoB;CACzC,MAAM,UAAU,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,EAAE,CAAC;CACpD,IAAI,iBAAiB;CAErB,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,UAAU,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,IAAI;EAChF,MAAM,WAAW,eACb,mBAAmB,MAAM,MAAM,SAAS,oBAAoB,YAAY,IACxE,KAAA;EACJ,MAAM,SACJ,WAAW,YAAY,QAAQ,KAAK,OAAO,SAAS,KAAK,KACrD;GAAE,MAAM,QAAQ;GAAM,OAAO,KAAK,IAAI,QAAQ,OAAO,SAAS,KAAK;EAAE,IACpE,WAAW;EAClB,IAAI,CAAC,QAAQ;EACb,kBAAkB;EAClB,SAAS,IAAI,OAAO,KAAK,IAAI,OAAO,KAAK;CAC3C;CAEA,KAAK,MAAM,UAAU,CAAC,GAAG,SAAS,KAAK,CAAC,GAAG;EACzC,MAAM,OAAO,KAAK,IAAI,MAAM;EAC5B,IAAI,CAAC,MAAM;EACX,IACE,YAAY,MAAM,IAAI,CAAC,CAAC,MACrB,eAAe,SAAS,IAAI,UAAU,KAAK,CAAC,QAAQ,IAAI,UAAU,CACrE,GAEA,SAAS,OAAO,MAAM;CAE1B;CAEA,OAAO;EACL,GAAI,SAAS,OAAO,IAChB,EAAE,OAAO,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAAC,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC,EAAE,IAC3E,CAAC;EACL;EACA,UAAU,MAAM,SAAS,KAAK,mBAAmB,MAAM;CACzD;AACF;AAEA,SAAS,mBACP,MACA,MACA,SACA,oBACA,MAC+D;CAC/D,IAAI,UAAgD;CACpD,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO,WAAW,CAAC,KAAK,IAAI,QAAQ,EAAE,GAAG;EACvC,KAAK,IAAI,QAAQ,EAAE;EACnB,MAAM,QAAQ,WAAW,QAAQ,YAAY,IAAI;EACjD,IACE,UAAU,KAAA,MACT,QAAQ,OAAO,KAAK,MAAO,CAAC,QAAQ,IAAI,QAAQ,EAAE,KAAK,mBAAmB,IAAI,QAAQ,EAAE,IAEzF,OAAO;GAAE,MAAM;GAAS;EAAM;EAEhC,UAAU,QAAQ,WAAW,KAAK,IAAI,QAAQ,QAAQ,IAAI,KAAA;CAC5D;AAEF;AAEA,SAAS,uBACP,YACA,cACa;CACb,MAAM,+BAAe,IAAI,IAAY;CACrC,MAAM,4BAAY,IAAI,IAAiC;CACvD,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,SAAS,SAAwD;EACrE,MAAM,SAAS,UAAU,IAAI,KAAK,EAAE;EACpC,IAAI,QAAQ,OAAO;EACnB,IAAI,SAAS,IAAI,KAAK,EAAE,GAAG,OAAO,yBAAyB;EAC3D,SAAS,IAAI,KAAK,EAAE;EAEpB,MAAM,cAAc,yBAAyB;EAC7C,KAAK,MAAM,SAAS,aAAa,IAAI,KAAK,EAAE,KAAK,CAAC,GAAG;GACnD,MAAM,mBAAmB,MAAM,KAAK;GACpC,IAAI,CAAC,aAAa,IAAI,MAAM,EAAE,GAAG,gBAAgB,aAAa,KAAK;GACnE,uBAAuB,aAAa,gBAAgB;EACtD;EAEA,IACE,YAAY,YAAY,MACvB,CAAC,KAAK,aAAa,oCAAoC,MAAM,WAAW,IAEzE,aAAa,IAAI,KAAK,EAAE;EAG1B,SAAS,OAAO,KAAK,EAAE;EACvB,UAAU,IAAI,KAAK,IAAI,WAAW;EAClC,OAAO;CACT;CAEA,KAAK,MAAM,aAAa,YAAY,MAAM,SAAS;CACnD,OAAO;AACT;AAEA,SAAS,2BAAgD;CACvD,OAAO;EACL,WAAW;EACX,cAAc,iCAAiC,WAAW;GACxD,OAAO;GACP,gBAAgB;EAClB,EAAE;CACJ;AACF;AAEA,SAAS,gBAAgB,SAA8B,MAAsC;CAC3F,QAAQ,aAAa;CACrB,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,QAAQ,WAAW,KAAK,YAAY,iCAAiC,MAAO;EAClF,IAAI,UAAU,KAAA,GAAW;EACzB,MAAM,cAAc,QAAQ,aAAa;EACzC,YAAY,SAAS;EACrB,YAAY,kBAAkB;CAChC;AACF;AAEA,SAAS,uBAAuB,QAA6B,QAAmC;CAC9F,OAAO,aAAa,OAAO;CAC3B,KAAK,IAAI,QAAQ,GAAG,QAAQ,OAAO,aAAa,QAAQ,SAAS,GAAG;EAClE,MAAM,cAAc,OAAO,aAAa;EACxC,MAAM,oBAAoB,OAAO,aAAa;EAC9C,YAAY,SAAS,kBAAkB;EACvC,YAAY,kBAAkB,kBAAkB;CAClD;AACF;AAEA,SAAS,oCACP,MACA,aACS;CACT,IAAI,qBAAqB;CACzB,IAAI,yBAAyB;CAC7B,KAAK,IAAI,QAAQ,GAAG,QAAQ,iCAAiC,QAAQ,SAAS,GAAG;EAC/E,MAAM,OAAO,iCAAiC;EAC9C,MAAM,cAAc,WAAW,KAAK,YAAY,IAAI;EACpD,MAAM,cAAc,YAAY,aAAa;EAC7C,IAAI,gBAAgB,KAAA,GAAW,sBAAsB;EACrD,IAAI,YAAY,iBAAiB,GAAG,0BAA0B;EAC9D,IAAI,gBAAgB,KAAA,KAAa,YAAY,mBAAmB,GAAG;EACnE,IAAI,YAAY,mBAAmB,YAAY,WAAW;EAC1D,IAAI,KAAK,IAAI,cAAc,YAAY,KAAK,IAAI,OAAO,OAAO;CAChE;CACA,OAAO,uBAAuB,KAAK,yBAAyB;AAC9D;AAEA,SAAS,YACP,MACA,MACU;CACV,MAAM,MAAgB,CAAC;CACvB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,KAAK,QAAQ;EACjB,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;CACA,OAAO;AACT;AAEA,SAAS,uBACP,MACA,MACA,cACoB;CACpB,MAAM,uBAAO,IAAI,IAAY;CAC7B,IAAI,WAAW,KAAK;CACpB,OAAO,YAAY,CAAC,KAAK,IAAI,QAAQ,GAAG;EACtC,IAAI,aAAa,IAAI,QAAQ,GAAG,OAAO;EACvC,KAAK,IAAI,QAAQ;EACjB,WAAW,KAAK,IAAI,QAAQ,CAAC,EAAE;CACjC;AAEF;AAEA,SAAS,WACP,YACA,MACoB;CACpB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,QAAQ,WAAW;EACzB,MAAM,SACJ,OAAO,UAAU,WACb,QACA,OAAO,UAAU,YAAY,MAAM,SAAS,IAC1C,OAAO,KAAK,IACZ;EACR,IAAI,OAAO,SAAS,MAAM,KAAK,UAAU,GAAG,OAAO;CACrD;AAEF;;;ACpaA,MAAM,0BAA0B;AAChC,MAAM,yBAAyB;AAW/B,SAAgB,sBACd,OACA,SACmB;CACnB,MAAM,eAAe,yBAAyB,OAAO,yBAAyB,OAAO;CACrF,MAAM,cAAc,yBAAyB,OAAO,wBAAwB,OAAO;CAEnF,IAAI,iBAAiB,KAAA,KAAa,CAAC,gBAAgB,SAAS,YAA4B,GACtF,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,wBAAwB,kBAAkB,gBAAgB,KAAK,IAAI,GACpF;CAEF,IAAI,gBAAgB,KAAA,MAAc,iBAAiB,KAAA,KAAa,iBAAiB,YAC/E,MAAM,IAAI,gBACR,GAAG,QAAQ,IAAI,uBAAuB,0BAA0B,yBAClE;CAGF,IAAI,iBAAiB,KAAA,GAAW,OAAO,CAAC;CACxC,IAAI,iBAAiB,WAAW,OAAO,EAAE,aAAa;CACtD,OAAO;EACS;EACd,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;CACvC;AACF;AAEA,SAAS,yBACP,OACA,KACA,SACoB;CACpB,MAAM,yBAAS,IAAI,IAAY;CAC/B,KAAK,MAAM,QAAQ,OAAO;EACxB,IAAI,CAAC,OAAO,OAAO,KAAK,YAAY,GAAG,GAAG;EAC1C,MAAM,QAAQ,KAAK,WAAW;EAC9B,IAAI,OAAO,UAAU,YAAY,MAAM,KAAK,CAAC,CAAC,WAAW,GACvD,MAAM,IAAI,gBAAgB,GAAG,QAAQ,IAAI,IAAI,4BAA4B;EAE3E,OAAO,IAAI,KAAK;CAClB;CAEA,IAAI,OAAO,OAAO,GAChB,MAAM,IAAI,gBACR,GAAG,QAAQ,gBAAgB,IAAI,WAAW,CAAC,GAAG,MAAM,CAAC,CAAC,KAAK,CAAC,CAAC,KAAK,IAAI,GACxE;CAEF,OAAO,OAAO,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;AAChC;;;;;;;;;ACtCA,MAAM,mBAAmB,CAAC,aAAa;AACvC,MAAM,2BAA2B;CAAC;CAAa;CAAoB;AAAiB;AACpF,MAAM,2BAA2B,CAAC,2BAA2B,uBAAuB;AACpF,MAAM,oBAAoB,CAAC,yBAAyB,sBAAsB;AAC1E,MAAM,qBAAqB,CAAC,6BAA6B,uBAAuB;AAChF,MAAM,cAAc;CAClB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AACA,MAAM,mBAAmB;CACvB;CACA;CACA;CACA;CACA;AACF;AAEA,SAAS,aACP,QACA,KACqC;CACrC,MAAM,QAAQ,OAAO;CACrB,OAAO,SAAS,OAAO,UAAU,YAAY,CAAC,MAAM,QAAQ,KAAK,IAC5D,QACD,KAAA;AACN;AAEA,SAAS,aACP,QACA,YACA,WACoB;CACpB,KAAK,MAAM,aAAa,YAAY;EAClC,MAAM,QAAQ,gBAAgB,aAAa,QAAQ,SAAS,GAAG,SAAS;EACxE,IAAI,UAAU,KAAA,GAAW,OAAO;CAClC;AAEF;;;;;;AAOA,SAAgB,aAAa,MAAsC;CACjE,IAAI,CAAC,QAAQ,OAAO,SAAS,UAAU,OAAO;CAE9C,MAAM,QAAQ,iBAAiBA,IAAG;CAClC,MAAM,QAAQ,gBAAgB,OAAO,sBAAsB;CAC3D,MAAM,aAAa,gBAAgB,OAAO,gBAAgB;CAC1D,MAAM,SAAS,gBAAgB,OAAO,uBAAuB;CAC7D,MAAM,qBAAqB,gBAAgB,OAAO,wBAAwB;CAC1E,MAAM,qBACJ,gBAAgB,OAAO,wBAAwB,KAC/C,aAAa,OAAO,oBAAoB,CAAC,oBAAoB,iBAAiB,CAAC;CACjF,MAAM,YAAY,sBAAsB;CACxC,MAAM,cAAc,aAAa,OAAO,OAAO;CAC/C,MAAM,SACJ,gBAAgB,OAAO,WAAW,KAClC,gBAAgB,aAAa,CAAC,MAAM,CAAC,KACrC,aAAa,OAAO,mBAAmB,CAAC,iBAAiB,cAAc,CAAC;CAC1E,MAAM,aACJ,gBAAgB,OAAO,gBAAgB,KAAK,gBAAgB,aAAa,CAAC,OAAO,CAAC;CACpF,IACE,UAAU,KAAA,KACV,eAAe,KAAA,KACf,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,GAEf,OAAO;CACT,MAAM,SAAyB;EAC7B,QAAQ,SAAS,MAAM,cAAc;EACrC,SACG,WAAW,uBAAuB,KAAA,IAAY,qBAAqB,OACnE,sBAAsB;CAC3B;CACA,IAAI,cAAc,KAAA,GAAW,OAAO,YAAY;CAChD,IAAI,WAAW,KAAA,GAAW,OAAO,SAAS;CAC1C,IAAI,eAAe,KAAA,GAAW,OAAO,aAAa;CAClD,OAAO;AACT;;;;;;AAOA,SAAgB,oBACd,MACA,UAAsC,CAAC,GAChB;CACvB,MAAM,OAAO,QAAQ,QAAQ;CAC7B,IAAI,QAAQ;CACZ,IAAI,SAAS;CACb,IAAI,YAAY;CAChB,IAAI,eAAe;CACnB,IAAI,SAAS;CACb,IAAI,YAAY;CAChB,IAAI,aAAa;CACjB,IAAI,gBAAgB;CACpB,IAAI,QAAQ;CACZ,MAAM,SAAS,IAAI,eAAwB,EAAE,WAAW,aAAa,CAAC;CACtE,MAAM,SAAS,CAAC,GAAG,OAAO,KAAK,IAAI,GAAG,GAAG,OAAO,MAAM,CAAC;CACvD,MAAM,QAAQ,SAAS,WAAW,SAAiB,SAAiB,UAAU,OAAO,KAAK;CAE1F,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,QAAQ,aAAa,MAAM,IAAI;EACrC,IAAI,CAAC,OAAO;EACZ,QAAQ,MAAM,OAAO,MAAM,KAAK;EAChC,SAAS,MAAM,QAAQ,MAAM,MAAM;EACnC,IAAI,MAAM,cAAc,KAAA,GAAW;GACjC,YAAY,MAAM,WAAW,MAAM,SAAS;GAC5C,eAAe;EACjB;EACA,IAAI,MAAM,WAAW,KAAA,GAAW;GAC9B,SAAS,MAAM,QAAQ,MAAM,MAAM;GACnC,YAAY;EACd;EACA,IAAI,MAAM,eAAe,KAAA,GAAW;GAClC,aAAa,MAAM,YAAY,MAAM,UAAU;GAC/C,gBAAgB;EAClB;EACA,QAAQ;CACV;CACA,IAAI,CAAC,OAAO,OAAO;CACnB,OAAO;EACL;EACA;EACA,GAAI,eAAe,EAAE,UAAU,IAAI,CAAC;EACpC,GAAI,YAAY,EAAE,OAAO,IAAI,CAAC;EAC9B,GAAI,gBAAgB,EAAE,WAAW,IAAI,CAAC;CACxC;AACF;AAEA,SAAS,aAAa,KAA6B;CACjD,MAAM,UAAU,IAAI,KAAK;CACzB,IAAI,CAAC,WAAW,YAAY,UAAU,OAAO;CAC7C,IAAI;EACF,OAAO,KAAK,MAAM,OAAO;CAC3B,QAAQ;EACN,OAAO;CACT;AACF;;;;;;;AAQA,eAAsB,yBACpB,UACA,YACgC;CAChC,IAAI;CACJ,IAAI;EACF,OAAO,MAAM,SAAS,MAAM,CAAC,CAAC,KAAK;CACrC,QAAQ;EACN,OAAO;CACT;CACA,IAAI;CACJ,IAAI;EACF,OAAO,KAAK,MAAM,IAAI;CACxB,QAAQ;EACN,OAAO,KAAA;CACT;CACA,OAAO,aAAa,IAAI,KAAK,oBAAoB,MAAM,UAAU;AACnE"}
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { o as FailureClass } from "./schema-BtVldJ3T.js";
|
|
2
2
|
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
3
3
|
import { i as TraceEmitter } from "./emitter-DGQGoLyj.js";
|
|
4
|
-
import { i as AnalystFinding, l as AnalystRunResult, p as EvidenceRef } from "./types-
|
|
5
|
-
import "./exact-types-
|
|
4
|
+
import { i as AnalystFinding, l as AnalystRunResult, p as EvidenceRef } from "./types-D3jh6F98.js";
|
|
5
|
+
import "./exact-types-B0lJV3tu.js";
|
|
6
6
|
import { a as DatasetScenario, o as DatasetSplit } from "./dataset-v_Y5902-.js";
|
|
7
7
|
//#region src/control-runtime.d.ts
|
|
8
8
|
type ControlSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
@@ -425,4 +425,4 @@ declare function controlRunToFeedbackTrajectory<TState, TAction, TActionResult>(
|
|
|
425
425
|
}): FeedbackTrajectory;
|
|
426
426
|
//#endregion
|
|
427
427
|
export { ControlRunResult as $, feedbackTrajectoryToOptimizerRow as A, AnalystReviewCounts as B, analystRunToReviewRequests as C, feedbackTrajectoriesToDatasetScenarios as D, createFeedbackTrajectory as E, serializeFeedbackTrajectoriesJsonl as F, analystFindingDigest as G, AnalystReviewQuality as H, summarizePreferenceMemory as I, ControlActionOutcome as J, analystRunDigest as K, withAssignedFeedbackSplit as L, renderPreferenceMemoryMarkdown as M, replayFeedbackTrajectories as N, feedbackTrajectoriesToOptimizerRows as O, replayFeedbackTrajectory as P, ControlEvalResult as Q, AnalystFindingDigest as R, analystRunToFeedbackTrajectory as S, controlRunToFeedbackTrajectory as T, AnalystReviewSource as U, AnalystReviewDecision as V, AnalystRunDigest as W, ControlContext as X, ControlBudget as Y, ControlDecision as Z, FeedbackTrajectoryStore as _, FeedbackLabel as a, StopDecision as at, PreferenceMemoryEntry as b, FeedbackOptimizerRow as c, runAgentControlLoop as ct, FeedbackReplayResult as d, subjectiveEval as dt, ControlRuntimeConfig as et, FeedbackSeverity as f, FeedbackTrajectoryFilter as g, FeedbackTrajectory as h, FeedbackAttempt as i, ControlStopPolicies as it, parseFeedbackTrajectoriesJsonl as j, feedbackTrajectoryToDatasetScenario as k, FeedbackOutcome as l, stopOnNoProgress as lt, FeedbackTask as m, AnalystReviewRequest as n, ControlSeverity as nt, FeedbackLabelKind as o, allCriticalPassed as ot, FeedbackSplitPolicy as p, ControlActionFailureMode as q, FeedbackArtifactType as r, ControlStep as rt, FeedbackLabelSource as s, objectiveEval as st, AnalystFeedbackTrajectoryOptions as t, ControlRuntimeError as tt, FeedbackReplayAdapter as u, stopOnRepeatedAction as ut, FileSystemFeedbackTrajectoryStore as v, assignFeedbackSplit as w, ProposedSideEffect as x, InMemoryFeedbackTrajectoryStore as y, AnalystMissedIssue as z };
|
|
428
|
-
//# sourceMappingURL=feedback-trajectory-
|
|
428
|
+
//# sourceMappingURL=feedback-trajectory-BCHqzLh3.d.ts.map
|