@tangle-network/agent-eval 0.144.0 → 0.144.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -0
- package/dist/analyst/index.d.ts +88 -23
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -4
- package/dist/{analyze-runs-BScZvqMV.js → analyze-runs-DWIvOAGk.js} +2 -2
- package/dist/{analyze-runs-BScZvqMV.js.map → analyze-runs-DWIvOAGk.js.map} +1 -1
- package/dist/{benchmark-DxaZfy0w.d.ts → benchmark-CP6kWfj8.d.ts} +3 -3
- package/dist/{benchmark-DxaZfy0w.d.ts.map → benchmark-CP6kWfj8.d.ts.map} +1 -1
- package/dist/{benchmark-command-4c7N_rlw.js → benchmark-command-CQPKRUr-.js} +332 -111
- package/dist/benchmark-command-CQPKRUr-.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-05UGZ8sZ.js → benchmarks-CkG1bWFa.js} +4 -4
- package/dist/{benchmarks-05UGZ8sZ.js.map → benchmarks-CkG1bWFa.js.map} +1 -1
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-BKOtvRAB.js → campaign-DjGFyPxH.js} +74 -40
- package/dist/campaign-DjGFyPxH.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-Cgl6KasJ.d.ts → client-Bbht4xxl.d.ts} +4 -4
- package/dist/{client-Cgl6KasJ.d.ts.map → client-Bbht4xxl.d.ts.map} +1 -1
- package/dist/{completion-verifier-D15NHYSk.d.ts → completion-verifier-EJERfFwF.d.ts} +3 -3
- package/dist/{completion-verifier-D15NHYSk.d.ts.map → completion-verifier-EJERfFwF.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +8 -8
- package/dist/contract/index.js +10 -8
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{default-registry-DfHJEwYh.js → default-registry-SOyHB6qG.js} +2 -2
- package/dist/{default-registry-DfHJEwYh.js.map → default-registry-SOyHB6qG.js.map} +1 -1
- package/dist/{default-registry-D3uqKbo6.d.ts → default-registry-iXfu2trt.d.ts} +5 -5
- package/dist/{default-registry-D3uqKbo6.d.ts.map → default-registry-iXfu2trt.d.ts.map} +1 -1
- package/dist/{dspy-rlm-engine-CBFwlyaY.js → dspy-rlm-engine-IRCG8kdi.js} +61 -105
- package/dist/dspy-rlm-engine-IRCG8kdi.js.map +1 -0
- package/dist/{eval-campaign-YdkpWWoT.js → eval-campaign-lZcDIwQM.js} +3 -3
- package/dist/{eval-campaign-YdkpWWoT.js.map → eval-campaign-lZcDIwQM.js.map} +1 -1
- package/dist/{exact-types-B0lJV3tu.d.ts → exact-types-BygCBR4L.d.ts} +2 -2
- package/dist/{exact-types-B0lJV3tu.d.ts.map → exact-types-BygCBR4L.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-iK0yu4AR.d.ts → external-optimizer-contracts-CdmX2K2S.d.ts} +47 -10
- package/dist/external-optimizer-contracts-CdmX2K2S.d.ts.map +1 -0
- package/dist/{feedback-trajectory-BCHqzLh3.d.ts → feedback-trajectory-CSIkRLQX.d.ts} +3 -3
- package/dist/{feedback-trajectory-BCHqzLh3.d.ts.map → feedback-trajectory-CSIkRLQX.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-DgbFl4cv.d.ts → index-BuHs_OnD.d.ts} +17 -12
- package/dist/{index-DgbFl4cv.d.ts.map → index-BuHs_OnD.d.ts.map} +1 -1
- package/dist/{index-D2enbqA0.d.ts → index-CNOCxBLh.d.ts} +2 -2
- package/dist/{index-D2enbqA0.d.ts.map → index-CNOCxBLh.d.ts.map} +1 -1
- package/dist/{index-DFLVtPZ9.d.ts → index-DGIzNtRv.d.ts} +2 -2
- package/dist/{index-DFLVtPZ9.d.ts.map → index-DGIzNtRv.d.ts.map} +1 -1
- package/dist/{index-DtMpBKVF.d.ts → index-D_qTihaQ.d.ts} +5 -5
- package/dist/{index-DtMpBKVF.d.ts.map → index-D_qTihaQ.d.ts.map} +1 -1
- package/dist/index.d.ts +20 -20
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +17 -17
- package/dist/{insight-report-Bh_8ksel.d.ts → insight-report-D5m1z0_n.d.ts} +3 -3
- package/dist/{insight-report-Bh_8ksel.d.ts.map → insight-report-D5m1z0_n.d.ts.map} +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-Dxz0Rkwa.js → ledger-core-DXZIqu17.js} +3 -3
- package/dist/ledger-core-DXZIqu17.js.map +1 -0
- package/dist/{llm-client-B3WXSH5Y.js → llm-client-D3EoChAU.js} +7 -7
- package/dist/llm-client-D3EoChAU.js.map +1 -0
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-Ctwk079K.js → mint-DD-0oQTA.js} +2 -2
- package/dist/{mint-Ctwk079K.js.map → mint-DD-0oQTA.js.map} +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{propose-review-control-DLXz4FCX.js → propose-review-control-BciUCZoh.js} +2 -2
- package/dist/{propose-review-control-DLXz4FCX.js.map → propose-review-control-BciUCZoh.js.map} +1 -1
- package/dist/{release-report-B_bQOHM-.d.ts → release-report-BZRWdq_t.d.ts} +3 -3
- package/dist/{release-report-B_bQOHM-.d.ts.map → release-report-BZRWdq_t.d.ts.map} +1 -1
- package/dist/{release-report-B5XPBvAU.js → release-report-Sl0xfkFv.js} +2 -2
- package/dist/{release-report-B5XPBvAU.js.map → release-report-Sl0xfkFv.js.map} +1 -1
- package/dist/{replay-DjUfTrHD.js → replay-CqOsGjzU.js} +2 -9
- package/dist/replay-CqOsGjzU.js.map +1 -0
- package/dist/{replay-BuJM6kLh.d.ts → replay-DQ-55DC_.d.ts} +4 -4
- package/dist/{replay-BuJM6kLh.d.ts.map → replay-DQ-55DC_.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +1 -1
- package/dist/{researcher-C6lzl-rP.d.ts → researcher-BSiCoM1s.d.ts} +3 -3
- package/dist/{researcher-C6lzl-rP.d.ts.map → researcher-BSiCoM1s.d.ts.map} +1 -1
- package/dist/{reward-hacking-DjTi9HLb.js → reward-hacking-CyuzxKly.js} +2 -2
- package/dist/{reward-hacking-DjTi9HLb.js.map → reward-hacking-CyuzxKly.js.map} +1 -1
- package/dist/{reward-hacking-CEVVmy3h.d.ts → reward-hacking-DFgkEY4p.d.ts} +2 -2
- package/dist/{reward-hacking-CEVVmy3h.d.ts.map → reward-hacking-DFgkEY4p.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +4 -4
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-8nj3mYvx.js → rollout-C2fD1cf4.js} +2 -2
- package/dist/{rollout-8nj3mYvx.js.map → rollout-C2fD1cf4.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C2CthIfY.d.ts → rubric-predictive-validity-C4r4Y-q8.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C2CthIfY.d.ts.map → rubric-predictive-validity-C4r4Y-q8.d.ts.map} +1 -1
- package/dist/{run-evidence-8Ou28QSa.d.ts → run-evidence-H1vRpIdT.d.ts} +3 -3
- package/dist/{run-evidence-8Ou28QSa.d.ts.map → run-evidence-H1vRpIdT.d.ts.map} +1 -1
- package/dist/{run-record-vRgqWmJw.js → run-record-CWN8-VsV.js} +40 -14
- package/dist/{run-record-vRgqWmJw.js.map → run-record-CWN8-VsV.js.map} +1 -1
- package/dist/{run-record-Tb3TTtUn.d.ts → run-record-ooo9FWns.d.ts} +5 -9
- package/dist/{run-record-Tb3TTtUn.d.ts.map → run-record-ooo9FWns.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-DJQtFr95.js → semantic-concept-judge-Do5aM9wP.js} +3 -3
- package/dist/{semantic-concept-judge-DJQtFr95.js.map → semantic-concept-judge-Do5aM9wP.js.map} +1 -1
- package/dist/{server-Cu4M3NSO.js → server-Df00sdwz.js} +3 -3
- package/dist/{server-Cu4M3NSO.js.map → server-Df00sdwz.js.map} +1 -1
- package/dist/{single-run-lock-t1si1ob7.js → single-run-lock-D5iN0Xzb.js} +719 -187
- package/dist/single-run-lock-D5iN0Xzb.js.map +1 -0
- package/dist/{skill-usage-BiVEU0QY.d.ts → skill-usage-DtpLou9L.d.ts} +28 -9
- package/dist/skill-usage-DtpLou9L.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Ds8J1_K8.js → skillopt-optimization-method-C4FX42dy.js} +42 -210
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +1 -0
- package/dist/{skillopt-optimization-method-B7o01OdX.d.ts → skillopt-optimization-method-CvSJGdm3.d.ts} +12 -8
- package/dist/{skillopt-optimization-method-B7o01OdX.d.ts.map → skillopt-optimization-method-CvSJGdm3.d.ts.map} +1 -1
- package/dist/{summary-report-o3eJ3gxG.d.ts → summary-report-BOM6dfP7.d.ts} +2 -2
- package/dist/{summary-report-o3eJ3gxG.d.ts.map → summary-report-BOM6dfP7.d.ts.map} +1 -1
- package/dist/{tool-groups-Bqy4A3QB.d.ts → tool-groups-CMmsgTzj.d.ts} +7 -7
- package/dist/{tool-groups-Bqy4A3QB.d.ts.map → tool-groups-CMmsgTzj.d.ts.map} +1 -1
- package/dist/traces.d.ts +4 -4
- package/dist/traces.js +1 -1
- package/dist/types-BjMFz88h.d.ts.map +1 -1
- package/dist/{types-KEqL1pZc.d.ts → types-DcJxgsLy.d.ts} +8 -6
- package/dist/{types-KEqL1pZc.d.ts.map → types-DcJxgsLy.d.ts.map} +1 -1
- package/dist/{types-D3jh6F98.d.ts → types-y8jrxXWd.d.ts} +2 -2
- package/dist/{types-D3jh6F98.d.ts.map → types-y8jrxXWd.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/eval-surface-map.md +4 -0
- package/docs/trace-analysis.md +43 -32
- package/package.json +1 -1
- package/dist/benchmark-command-4c7N_rlw.js.map +0 -1
- package/dist/campaign-BKOtvRAB.js.map +0 -1
- package/dist/dspy-rlm-engine-CBFwlyaY.js.map +0 -1
- package/dist/external-optimizer-contracts-iK0yu4AR.d.ts.map +0 -1
- package/dist/ledger-core-Dxz0Rkwa.js.map +0 -1
- package/dist/llm-client-B3WXSH5Y.js.map +0 -1
- package/dist/replay-DjUfTrHD.js.map +0 -1
- package/dist/single-run-lock-t1si1ob7.js.map +0 -1
- package/dist/skill-usage-BiVEU0QY.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Ds8J1_K8.js.map +0 -1
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"llm-client-D3EoChAU.js","names":[],"sources":["../src/llm-client.ts"],"sourcesContent":["/**\n * LLM client with graceful degrade.\n *\n * OpenAI-compatible `/v1/chat/completions` client with:\n * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).\n * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).\n * - One retry at temperature 1 when a model explicitly requires it.\n * - Graceful json_schema → json_object degrade on 400 with schema-reject body.\n * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.\n * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI\n * directly, cli-bridge subscriptions, and any router that speaks the spec.\n *\n * Usage:\n * const { value, result } = await callLlmJson<MyType>(\n * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },\n * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },\n * )\n *\n * `createChatClient` wraps this implementation for provider-neutral package\n * entry points. Direct callers can use `callLlm` or `callLlmJson`.\n */\n\nimport {\n type CostReceiptInput,\n type CustomTokenPricing,\n costForTokenPricing,\n type MaximumCharge,\n} from './cost-ledger'\nimport { AgentEvalError, CaptureIntegrityError } from './errors'\nimport {\n defaultProviderRedactor,\n type ProviderRedactor,\n providerFromBaseUrl,\n type RawProviderEvent,\n type RawProviderSink,\n} from './trace/raw-provider-sink'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\nexport interface LlmMessage {\n role: 'system' | 'user' | 'assistant'\n /**\n * Either a plain text content string OR a multimodal content array\n * (text + image_url parts) for vision-capable models.\n */\n content:\n | string\n | Array<\n | { type: 'text'; text: string }\n | { type: 'image_url'; image_url: { url: string; detail?: 'auto' | 'low' | 'high' } }\n >\n}\n\nexport type LlmThinkingMode = 'enabled' | 'disabled'\n\nexport interface LlmCallRequest {\n model: string\n messages: LlmMessage[]\n /** Optional JSON-mode response format (response_format: json_object). */\n jsonMode?: boolean\n /** Optional structured output via JSON Schema. Falls back to json_object on 400. */\n jsonSchema?: { name: string; schema: Record<string, unknown> }\n temperature?: number\n maxTokens?: number\n /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */\n thinking?: LlmThinkingMode\n /** Per-call timeout, default 300s. */\n timeoutMs?: number\n}\n\n/** Conservative priced bound for the exact text request sent to a provider.\n * Returns undefined when output or multimodal input is not bounded, causing a\n * capped CostLedger to reject the call before execution. Pass\n * `customTokenPricing` when package pricing does not cover the model or endpoint. */\nexport function maximumChargeForLlmRequest(\n request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens' | 'thinking'>,\n options: LlmClientOptions = {},\n): MaximumCharge | undefined {\n if (request.maxTokens === undefined) return undefined\n if (!Number.isInteger(request.maxTokens) || request.maxTokens <= 0) {\n throw new RangeError(`maximumChargeForLlmRequest: maxTokens must be a positive integer`)\n }\n if (\n request.messages.some(\n (message) =>\n Array.isArray(message.content) && message.content.some((part) => part.type === 'image_url'),\n )\n ) {\n return undefined\n }\n\n const attempts = resolveMaximumAttempts(options.maximumAttempts)\n const forceJsonObject = options.jsonSchemaTransport === 'json-object'\n // A byte-level tokenizer cannot emit more input tokens than request bytes.\n // Pricing the complete body also covers role/schema framing omitted from content-only estimates.\n const requestBytes = new TextEncoder().encode(\n JSON.stringify(buildBody(request, forceJsonObject, options.thinking)),\n ).byteLength\n // A rejected response schema can trigger one JSON-mode batch with the same output limit.\n const batches = request.jsonSchema && !forceJsonObject ? 2 : 1\n const usage = {\n inputTokens: requestBytes * attempts * batches,\n outputTokens: request.maxTokens * attempts * batches,\n }\n return options.customTokenPricing\n ? { customTokenPricing: options.customTokenPricing, ...usage }\n : { model: request.model, ...usage }\n}\n\nexport interface LlmUsage {\n promptTokens: number\n completionTokens: number\n totalTokens: number\n /** False when the provider omitted or malformed prompt/completion usage. */\n captured?: boolean\n /** Reasoning-token subset of completionTokens, when reported. */\n reasoningTokens?: number\n /** Proxies populate this when prompt caching is on. */\n cachedPromptTokens?: number\n}\n\nexport interface LlmCallResult {\n /** The text content of the first choice. Empty string if none. */\n content: string\n usage: LlmUsage\n /**\n * Cost in USD. Uses the provider's reported cost when present, otherwise\n * caller-supplied token pricing. `null` when neither is available.\n */\n costUsd: number | null\n /** Model name actually used (echoed from response). */\n model: string\n /** Wall-clock duration of the HTTP call (last attempt, if retried). */\n durationMs: number\n /**\n * `finish_reason` echoed from the first choice (`stop`, `length`,\n * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.\n * Exposed so a free-form `callLlm` caller CAN detect a truncated answer\n * (`length`) instead of treating a cut-off completion as complete. Note:\n * `callLlm` does not itself reject on it — acting on this signal is the\n * caller's responsibility (in-repo free-form drivers do not yet enforce it).\n */\n finishReason?: string | null\n /**\n * True when `content.trim()` is empty. An empty completion is a silent zero\n * for free-form `callLlm` callers; this flag is the signal a caller can\n * inspect to fail loud rather than proceed on an empty string. `callLlm`\n * surfaces it but does not throw on it.\n */\n contentEmpty?: boolean\n /** Raw response body. */\n raw: Record<string, unknown>\n}\n\nexport type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>\n\n/** Convert a provider result into the canonical paid-call receipt input. */\nexport function costReceiptFromLlm(\n result: LlmCallResult,\n customTokenPricing?: CustomTokenPricing,\n): CostReceiptInput {\n const cachedTokens = result.usage.cachedPromptTokens ?? 0\n const inputTokens = Math.max(0, result.usage.promptTokens - cachedTokens)\n const providerCostUsd = providerReportedCost(result.raw)\n return {\n model: result.model,\n inputTokens,\n outputTokens: result.usage.completionTokens,\n reasoningTokens: result.usage.reasoningTokens,\n cachedTokens: cachedTokens > 0 ? cachedTokens : undefined,\n ...(providerCostUsd === undefined\n ? customTokenPricing && result.usage.captured !== false\n ? { customTokenPricing }\n : result.costUsd === null\n ? {}\n : { estimatedCostUsd: result.costUsd }\n : { actualCostUsd: providerCostUsd }),\n usageUnknown: result.usage.captured === false,\n }\n}\n\nfunction providerReportedCost(raw: Record<string, unknown>): number | undefined {\n const value = raw._response_cost ?? raw.cost_usd\n return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined\n}\n\n/** Structured-response failures retain their completed provider receipt. */\nexport function costReceiptFromLlmError(\n error: Error,\n customTokenPricing?: CustomTokenPricing,\n): CostReceiptInput | undefined {\n return error instanceof LlmResponseError\n ? costReceiptFromLlm(error.result, customTokenPricing)\n : undefined\n}\n\nexport class LlmCallError extends AgentEvalError {\n constructor(\n message: string,\n public readonly status: number,\n public readonly body: string,\n public readonly model: string,\n ) {\n super('judge', message)\n }\n}\n\n/** A provider response completed and incurred measurable usage, but its content\n * could not satisfy the caller's response contract. The response envelope is\n * retained so accounting can commit the receipt before the error propagates. */\nexport class LlmResponseError extends AgentEvalError {\n constructor(\n message: string,\n public readonly result: LlmCallResult,\n options?: { cause?: unknown },\n ) {\n super('judge', message, options)\n }\n}\n\nexport interface LlmClientOptions {\n /** Base URL (without trailing slash). Must end at the `/v1` prefix. */\n baseUrl?: string\n /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */\n apiKey?: string\n bearer?: string\n /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */\n authHeader?: { name: string; value: string }\n /** Stable provider idempotency key, reused across retries of this logical call. */\n idempotencyKey?: string\n /** Default timeout in ms. Per-call can override. */\n defaultTimeoutMs?: number\n /**\n * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to\n * each attempt's per-attempt timeout controller, so aborting it cancels\n * the in-flight fetch. A caller abort is FATAL: it is not retried even\n * though an AbortError otherwise matches the transient patterns.\n */\n signal?: AbortSignal\n /**\n * Cross-attempt wall-clock budget in ms, measured from the first attempt.\n * Before launching each attempt the loop checks the remaining budget and\n * stops retrying once it is exhausted, rather than waiting the full\n * per-attempt timeout on every retry. Bounds total time independent of\n * total attempts × `timeoutMs`.\n */\n deadlineMs?: number\n /** Total provider attempts. Default 3. */\n maximumAttempts?: number\n /** Token rates used when the provider omits cost or package pricing does not cover the model. */\n customTokenPricing?: CustomTokenPricing\n /**\n * Transport for requests that declare `jsonSchema`. `native` sends\n * `response_format: json_schema`; `json-object` sends the broadly supported\n * JSON mode and relies on the caller to include the schema in model-visible\n * instructions. Default: `native`.\n */\n jsonSchemaTransport?: 'native' | 'json-object'\n /**\n * JSON payload parsing policy. `extract` accepts fenced or prose-prefixed JSON.\n * `exact` requires the complete response content to be one JSON value.\n * Default: `extract`.\n */\n jsonPayloadMode?: 'extract' | 'exact'\n /** Default provider reasoning mode. A per-call request value takes precedence. */\n thinking?: LlmThinkingMode\n /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */\n fetch?: typeof fetch\n /**\n * Optional raw HTTP capture sink. When provided, every request, response,\n * and error (across all retry attempts) is recorded to the sink, with auth\n * headers and credential-shaped body fields redacted by default. This is\n * the layer-1 forensics primitive: structured `LlmSpan`s record intent,\n * raw events record what actually crossed the wire.\n */\n rawSink?: RawProviderSink\n /**\n * Logical provider id attached to raw events. When omitted, derived from\n * `baseUrl` via `providerFromBaseUrl`.\n */\n provider?: string\n /** Trace context attached to raw events; populated by emitter-aware callers. */\n traceContext?: { runId?: string; spanId?: string }\n /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */\n redactor?: ProviderRedactor\n}\n\n// ─── Internals ──────────────────────────────────────────────────────────\n\nconst DEFAULT_BASE_URL = 'https://router.tangle.tools/v1'\n// Flagship / reasoning models routinely take several minutes on large prompts (a\n// reflection over many failures, a long tool transcript). A tight cap aborts a\n// legitimately-slow but healthy call — and because every retry attempt re-uses\n// the same window, such a model aborts on ALL attempts and the loop throws. The\n// default is generous enough to let those complete, bounded enough that a truly\n// hung call still fails over after retries, and tunable per deployment via\n// TANGLE_LLM_TIMEOUT_MS. Per-call `req.timeoutMs` / `opts.defaultTimeoutMs`\n// still win for callers that know their model's latency.\nconst DEFAULT_TIMEOUT_MS = Number(process.env.TANGLE_LLM_TIMEOUT_MS) || 300_000\nconst DEFAULT_MAXIMUM_ATTEMPTS =\n process.env.TANGLE_LLM_MAXIMUM_ATTEMPTS === undefined\n ? 3\n : Number(process.env.TANGLE_LLM_MAXIMUM_ATTEMPTS)\n\nfunction resolveMaximumAttempts(configured: number | undefined): number {\n const attempts = configured ?? DEFAULT_MAXIMUM_ATTEMPTS\n if (!Number.isInteger(attempts) || attempts <= 0) {\n throw new RangeError('LLM maximum attempts must be a positive integer')\n }\n return attempts\n}\n\nfunction providerTokenCount(value: unknown): number | undefined {\n return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0 ? value : undefined\n}\n\nconst RETRYABLE_STATUS = new Set([429, 502, 503, 504])\n\n/**\n * Transient transport/network error signatures, matched against an error's\n * name, message, and `code`. Covers fetch/undici network failures, aborts\n * and timeouts, and — critically — HTTP/2 transport faults a keep-alive\n * connection raises mid-response: `terminated`, `NGHTTP2_INTERNAL_ERROR`,\n * `UND_ERR_*`, `other side closed`. Those last ones carry no clean HTTP\n * status; unrecognised, they escape the retry loop and surface as an\n * uncaught rejection.\n */\nconst TRANSIENT_ERROR_PATTERNS: readonly RegExp[] = [\n /AbortError/i,\n /TimeoutError/i,\n /this operation was aborted/i,\n /fetch failed/i,\n /ECONNRESET/i,\n /ETIMEDOUT/i,\n /EAI_AGAIN/i,\n /socket hang up/i,\n /stream.*ended.*unexpectedly/i,\n /terminated/i,\n /other side closed/i,\n /NGHTTP2/i,\n /UND_ERR/i,\n]\n\n/**\n * True when an error is a transient transport/network fault worth retrying,\n * as opposed to a deterministic failure (4xx schema reject, JSON parse) that\n * a retry cannot fix. Inspects `LlmCallError.status`, then the error's\n * name/message/code, then recurses into `error.cause` — undici nests the\n * real socket fault one or more levels under `.cause`.\n *\n * This is the retry classifier for the package: `callLlm` and\n * `withJudgeRetry` both route through it, so connection failures are treated\n * consistently across transports.\n */\nexport function isTransientLlmError(err: unknown): boolean {\n return classifyTransient(err, 0)\n}\n\nfunction classifyTransient(err: unknown, depth: number): boolean {\n if (err instanceof LlmCallError) return RETRYABLE_STATUS.has(err.status)\n if (!(err instanceof Error)) return false\n // Foreign transport errors can carry a numeric HTTP status without being an\n // LlmCallError. A retryable status is decisive.\n const status = (err as { status?: unknown }).status\n if (typeof status === 'number' && RETRYABLE_STATUS.has(status)) return true\n const code = (err as { code?: unknown }).code\n const haystack = `${err.name}\\n${err.message}\\n${typeof code === 'string' ? code : ''}`\n if (TRANSIENT_ERROR_PATTERNS.some((p) => p.test(haystack))) return true\n const cause = (err as { cause?: unknown }).cause\n if (depth < 4 && cause instanceof Error && cause !== err) {\n return classifyTransient(cause, depth + 1)\n }\n return false\n}\n\nfunction parseRetryAfter(headers: Headers): number | null {\n const h = headers.get('retry-after')\n if (!h) return null\n const asNumber = Number(h)\n if (Number.isFinite(asNumber) && asNumber > 0) return asNumber * 1000\n const asDate = Date.parse(h)\n if (Number.isFinite(asDate)) return Math.max(0, asDate - Date.now())\n return null\n}\n\n/** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */\nexport function backoffMs(attempt: number): number {\n return Math.min(500 * 2 ** attempt, 16_000)\n}\n\nfunction buildHeaders(opts: LlmClientOptions): Record<string, string> {\n const headers: Record<string, string> = {\n 'Content-Type': 'application/json',\n Accept: 'application/json',\n }\n if (opts.authHeader) {\n headers[opts.authHeader.name] = opts.authHeader.value\n } else if (opts.bearer || opts.apiKey) {\n headers.Authorization = `Bearer ${opts.bearer ?? opts.apiKey}`\n }\n if (opts.idempotencyKey) headers['Idempotency-Key'] = opts.idempotencyKey\n return headers\n}\n\nfunction isSchemaRejection(status: number, body: string): boolean {\n if (status !== 400) return false\n const lower = body.toLowerCase()\n return (\n lower.includes('response_format') ||\n lower.includes('json_schema') ||\n lower.includes('is unavailable') ||\n lower.includes('not supported')\n )\n}\n\nfunction isTemperatureOneRejection(status: number, body: string): boolean {\n if (status !== 400 || !/temperature/i.test(body)) return false\n return (\n /temperature[^.\\n]{0,120}\\b(?:only|must|should|required|requires?)\\b[^.\\n]{0,40}\\b1(?:\\.0+)?\\b/i.test(\n body,\n ) || /\\bonly\\s+1(?:\\.0+)?\\s+is\\s+allowed\\b[^.\\n]{0,120}\\btemperature\\b/i.test(body)\n )\n}\n\nfunction buildBody(\n req: LlmCallRequest,\n forceJsonObject: boolean,\n defaultThinking?: LlmThinkingMode,\n): Record<string, unknown> {\n const body: Record<string, unknown> = {\n model: req.model,\n messages: req.messages,\n temperature: req.temperature ?? 0,\n }\n if (req.maxTokens != null) {\n if (usesMaxCompletionTokens(req.model)) body.max_completion_tokens = req.maxTokens\n else body.max_tokens = req.maxTokens\n }\n const thinking = req.thinking ?? defaultThinking\n if (thinking !== undefined) {\n body.thinking = { type: thinking }\n }\n\n if (req.jsonSchema && !forceJsonObject) {\n body.response_format = {\n type: 'json_schema',\n json_schema: { name: req.jsonSchema.name, schema: req.jsonSchema.schema, strict: true },\n }\n } else if (req.jsonMode || req.jsonSchema) {\n body.response_format = { type: 'json_object' }\n }\n\n return body\n}\n\nfunction usesMaxCompletionTokens(model: string): boolean {\n return /^gpt-5(?:[.-]|$)/i.test(model)\n}\n\nasync function sleep(ms: number): Promise<void> {\n return new Promise((resolve) => setTimeout(resolve, ms))\n}\n\n/**\n * Combine the per-attempt timeout signal with an optional caller signal into\n * one signal the fetch listens on. Prefers the native `AbortSignal.any`; falls\n * back to manual wiring on runtimes that predate it. The caller signal is also\n * propagated to the timeout controller so aborting it cancels the in-flight\n * fetch immediately.\n */\nfunction linkSignals(timeoutController: AbortController, caller?: AbortSignal): AbortSignal {\n if (!caller) return timeoutController.signal\n if (typeof (AbortSignal as { any?: unknown }).any === 'function') {\n return AbortSignal.any([timeoutController.signal, caller])\n }\n if (caller.aborted) {\n timeoutController.abort()\n } else {\n caller.addEventListener('abort', () => timeoutController.abort(), { once: true })\n }\n return timeoutController.signal\n}\n\n/** True once the cross-attempt wall-clock budget (if any) is exhausted. */\nfunction deadlineExceeded(start: number, deadlineMs: number | undefined): boolean {\n return deadlineMs != null && Date.now() - start >= deadlineMs\n}\n\n// ─── Public API ─────────────────────────────────────────────────────────\n\n/**\n * Strip a ```json / ``` code fence if the model emitted one.\n * Idempotent for naked JSON. Some models (claude-code via router, certain\n * deepseek models) wrap output even under json_object.\n */\nexport function stripFencedJson(raw: string): string {\n const trimmed = raw.trim()\n const m = trimmed.match(/^```(?:json)?\\s*\\n?([\\s\\S]*?)\\n?```\\s*$/)\n return m ? m[1]!.trim() : trimmed\n}\n\nexport function extractJsonPayload(raw: string): string {\n const stripped = stripFencedJson(raw)\n try {\n JSON.parse(stripped)\n return stripped\n } catch {\n // A response that declares a JSON root must parse as that complete root.\n // Scanning onward could turn a truncated object into one of its valid nested\n // arrays or objects and silently change the response schema.\n if (stripped.startsWith('{') || stripped.startsWith('[')) return stripped\n }\n\n // Only prose-leading responses may contain a recoverable JSON payload.\n const starts = [...stripped.matchAll(/[[{]/g)]\n .map((match) => match.index)\n .filter((index) => index != null)\n for (const start of starts) {\n const candidate = extractBalancedJson(stripped, start)\n if (!candidate) continue\n try {\n JSON.parse(candidate)\n return candidate\n } catch {\n // Keep scanning; earlier braces may belong to prose.\n }\n }\n\n return stripped\n}\n\nfunction extractBalancedJson(input: string, start: number): string | null {\n const opener = input[start]\n const closer = opener === '{' ? '}' : opener === '[' ? ']' : null\n if (!closer) return null\n\n const stack: string[] = [closer]\n let isInString = false\n let isEscaped = false\n\n for (let i = start + 1; i < input.length; i++) {\n const char = input[i]!\n if (isEscaped) {\n isEscaped = false\n continue\n }\n if (char === '\\\\') {\n isEscaped = isInString\n continue\n }\n if (char === '\"') {\n isInString = !isInString\n continue\n }\n if (isInString) continue\n\n if (char === '{') stack.push('}')\n else if (char === '[') stack.push(']')\n else if (char === stack[stack.length - 1]) {\n stack.pop()\n if (stack.length === 0) return input.slice(start, i + 1)\n }\n }\n\n return null\n}\n\n/**\n * Low-level call. Returns raw content + usage + cost. Retries on transient\n * failures; does NOT degrade schema here — callers that want graceful\n * degrade use `callLlmJson`.\n */\nexport async function callLlm(\n req: LlmCallRequest,\n opts: LlmClientOptions = {},\n): Promise<LlmCallResult> {\n const baseUrl = (opts.baseUrl ?? DEFAULT_BASE_URL).replace(/\\/+$/, '')\n const url = `${baseUrl}/chat/completions`\n const endpoint = '/chat/completions'\n const timeoutMs = req.timeoutMs ?? opts.defaultTimeoutMs ?? DEFAULT_TIMEOUT_MS\n const maximumAttempts = resolveMaximumAttempts(opts.maximumAttempts)\n const fetchFn = opts.fetch ?? globalThis.fetch\n const headers = buildHeaders(opts)\n const provider = opts.provider ?? providerFromBaseUrl(baseUrl)\n const sink = opts.rawSink\n const redactor = opts.redactor ?? defaultProviderRedactor\n const traceContext = opts.traceContext\n const callerSignal = opts.signal\n const deadlineMs = opts.deadlineMs\n const deadlineStart = Date.now()\n if (opts.customTokenPricing) {\n costForTokenPricing(opts.customTokenPricing, { inputTokens: 0, outputTokens: 0 })\n }\n\n let lastErr: unknown\n let effectiveRequest = req\n for (let attempt = 0; attempt < maximumAttempts; attempt++) {\n // A caller cancel is fatal — never retried. Checking before each attempt\n // means an already-aborted signal short-circuits without firing fetch.\n if (callerSignal?.aborted) {\n throw new DOMException('callLlm aborted by caller signal', 'AbortError')\n }\n // Stop retrying once the cross-attempt budget is spent rather than burning\n // a full per-attempt timeout on each remaining retry.\n if (attempt > 0 && deadlineExceeded(deadlineStart, deadlineMs)) {\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n }\n const controller = new AbortController()\n const attemptSignal = linkSignals(controller, callerSignal)\n const timeoutHandle = setTimeout(() => controller.abort(), timeoutMs)\n const started = Date.now()\n const requestBody = buildBody(\n effectiveRequest,\n opts.jsonSchemaTransport === 'json-object',\n opts.thinking,\n )\n let attemptErrorRecorded = false\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'request',\n timestamp: started,\n requestHeaders: headers,\n requestBody,\n redactedFields: [],\n })\n }\n\n try {\n const res = await fetchFn(url, {\n method: 'POST',\n headers,\n body: JSON.stringify(requestBody),\n signal: attemptSignal,\n })\n clearTimeout(timeoutHandle)\n const responseHeaders = sink ? headersToObject(res.headers) : undefined\n\n if (!res.ok) {\n const body = await res.text()\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n statusCode: res.status,\n responseHeaders,\n responseBody: body,\n errorMessage: `HTTP ${res.status}`,\n redactedFields: [],\n })\n attemptErrorRecorded = true\n }\n const err = new LlmCallError(\n `LLM call failed with HTTP ${res.status}`,\n res.status,\n body,\n req.model,\n )\n if (\n isTemperatureOneRejection(res.status, body) &&\n effectiveRequest.temperature !== 1 &&\n attempt < maximumAttempts - 1 &&\n !deadlineExceeded(deadlineStart, deadlineMs)\n ) {\n lastErr = err\n effectiveRequest = { ...effectiveRequest, temperature: 1 }\n continue\n }\n if (\n RETRYABLE_STATUS.has(res.status) &&\n attempt < maximumAttempts - 1 &&\n !deadlineExceeded(deadlineStart, deadlineMs)\n ) {\n lastErr = err\n const retryAfter = parseRetryAfter(res.headers)\n await sleep(retryAfter ?? backoffMs(attempt))\n continue\n }\n throw err\n }\n\n const text = await res.text()\n let json: Record<string, unknown>\n try {\n json = JSON.parse(text) as Record<string, unknown>\n } catch (parseErr) {\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n statusCode: res.status,\n responseHeaders,\n responseBody: text,\n errorMessage: `non-JSON response: ${parseErr instanceof Error ? parseErr.message : String(parseErr)}`,\n redactedFields: [],\n })\n attemptErrorRecorded = true\n }\n throw parseErr\n }\n if (sink) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'response',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n statusCode: res.status,\n responseHeaders,\n responseBody: json,\n redactedFields: [],\n })\n }\n const choice = (\n json.choices as\n | Array<{ message?: { content?: string }; finish_reason?: string | null }>\n | undefined\n )?.[0]\n const usageRaw =\n json.usage && typeof json.usage === 'object' && !Array.isArray(json.usage)\n ? (json.usage as Record<string, unknown>)\n : undefined\n const promptTokens = providerTokenCount(usageRaw?.prompt_tokens)\n const completionTokens = providerTokenCount(usageRaw?.completion_tokens)\n const totalTokens = providerTokenCount(usageRaw?.total_tokens)\n const completionDetails =\n usageRaw?.completion_tokens_details &&\n typeof usageRaw.completion_tokens_details === 'object' &&\n !Array.isArray(usageRaw.completion_tokens_details)\n ? (usageRaw.completion_tokens_details as Record<string, unknown>)\n : undefined\n const reasoningRaw = completionDetails?.reasoning_tokens\n const reasoningTokens =\n reasoningRaw === undefined ? undefined : providerTokenCount(reasoningRaw)\n const cachedRaw =\n usageRaw?.prompt_tokens_details &&\n typeof usageRaw.prompt_tokens_details === 'object' &&\n !Array.isArray(usageRaw.prompt_tokens_details)\n ? (usageRaw.prompt_tokens_details as Record<string, unknown>).cached_tokens\n : undefined\n const cachedPromptTokens = cachedRaw === undefined ? undefined : providerTokenCount(cachedRaw)\n const usageCaptured =\n promptTokens !== undefined &&\n completionTokens !== undefined &&\n (reasoningRaw === undefined ||\n (reasoningTokens !== undefined && reasoningTokens <= completionTokens)) &&\n (cachedRaw === undefined ||\n (cachedPromptTokens !== undefined && cachedPromptTokens <= promptTokens)) &&\n (totalTokens === undefined || totalTokens === promptTokens + completionTokens)\n const costFromProxy = (json._response_cost ?? json.cost_usd) as number | undefined\n const content = choice?.message?.content ?? ''\n\n const configuredCost =\n typeof costFromProxy !== 'number' && usageCaptured && opts.customTokenPricing\n ? costForTokenPricing(opts.customTokenPricing, {\n inputTokens: promptTokens! - (cachedPromptTokens ?? 0),\n ...(cachedPromptTokens ? { cachedTokens: cachedPromptTokens } : {}),\n outputTokens: completionTokens!,\n })\n : undefined\n\n return {\n content,\n finishReason: choice?.finish_reason ?? null,\n contentEmpty: content.trim().length === 0,\n usage: {\n promptTokens: promptTokens ?? 0,\n completionTokens: completionTokens ?? 0,\n totalTokens: totalTokens ?? (promptTokens ?? 0) + (completionTokens ?? 0),\n captured: usageCaptured,\n reasoningTokens,\n cachedPromptTokens,\n },\n costUsd: typeof costFromProxy === 'number' ? costFromProxy : (configuredCost ?? null),\n model: (json.model as string) ?? req.model,\n durationMs: Date.now() - started,\n raw: json,\n }\n } catch (err) {\n clearTimeout(timeoutHandle)\n lastErr = err\n // A caller cancel is fatal even though an AbortError matches the\n // transient patterns — a cancelled call must surface immediately, not\n // be retried against the same dead intent.\n if (callerSignal?.aborted) {\n if (sink && !attemptErrorRecorded) {\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n }\n throw err\n }\n if (sink && !attemptErrorRecorded) {\n // Record only if neither the !res.ok branch nor the JSON.parse catch\n // already produced an error event for this attempt. Covers network\n // failures, timeouts, and aborts.\n await recordRaw(sink, redactor, {\n eventId: cryptoEventId(),\n runId: traceContext?.runId,\n spanId: traceContext?.spanId,\n provider,\n model: req.model,\n endpoint,\n baseUrl,\n attemptIndex: attempt,\n direction: 'error',\n timestamp: Date.now(),\n durationMs: Date.now() - started,\n errorMessage: err instanceof Error ? err.message : String(err),\n redactedFields: [],\n })\n }\n if (\n attempt < maximumAttempts - 1 &&\n isTransientLlmError(err) &&\n !deadlineExceeded(deadlineStart, deadlineMs)\n ) {\n await sleep(backoffMs(attempt))\n continue\n }\n throw err\n }\n }\n throw lastErr instanceof Error ? lastErr : new Error(String(lastErr))\n}\n\nasync function recordRaw(\n sink: RawProviderSink,\n redactor: ProviderRedactor,\n event: RawProviderEvent,\n): Promise<void> {\n // Errors from sinks must not crash the LLM call. Forensic capture is\n // best-effort; the structured trace is the system of record.\n try {\n await sink.record(redactor(event))\n } catch {\n // Intentionally swallowed.\n }\n}\n\nfunction headersToObject(h: Headers): Record<string, string> {\n const out: Record<string, string> = {}\n h.forEach((value, key) => {\n out[key] = value\n })\n return out\n}\n\nfunction cryptoEventId(): string {\n if (typeof globalThis.crypto?.randomUUID === 'function') return globalThis.crypto.randomUUID()\n return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`\n}\n\n/**\n * Structured-output call. Returns parsed JSON plus the raw result envelope.\n * Degrades `jsonSchema` → `jsonMode` on a 400 that names the schema param —\n * critical for deepseek-v3/v4, kimi-k2.6, and other models that don't accept\n * the `response_format.json_schema` shape but DO accept `json_object`.\n */\nexport async function callLlmJson<T = unknown>(\n req: LlmCallRequest,\n opts: LlmClientOptions = {},\n): Promise<{ value: T; result: LlmCallResult }> {\n const result = await callLlmStructured(req, opts)\n const value = parseJsonResult<T>(result, opts.jsonPayloadMode ?? 'extract')\n return { value, result }\n}\n\n/** Shared schema-to-JSON-mode fallback that preserves the raw result. */\nasync function callLlmStructured(\n req: LlmCallRequest,\n opts: LlmClientOptions = {},\n): Promise<LlmCallResult> {\n try {\n return await callLlm({ ...req, jsonMode: req.jsonMode ?? !req.jsonSchema }, opts)\n } catch (err) {\n if (\n opts.jsonSchemaTransport !== 'json-object' &&\n err instanceof LlmCallError &&\n isSchemaRejection(err.status, err.body) &&\n req.jsonSchema\n ) {\n const degradedReq: LlmCallRequest = { ...req, jsonMode: true, jsonSchema: undefined }\n return await callLlm(degradedReq, opts)\n }\n throw err\n }\n}\n\nfunction parseJsonResult<T>(\n result: LlmCallResult,\n jsonPayloadMode: NonNullable<LlmClientOptions['jsonPayloadMode']>,\n): T {\n try {\n if (result.finishReason === 'length') {\n throw new Error(\n `LLM returned truncated JSON content (model=${result.model}, finishReason=length)`,\n )\n }\n return parseJsonSafely<T>(result.content, result.model, jsonPayloadMode)\n } catch (error) {\n if (error instanceof LlmResponseError) throw error\n const cause = error instanceof Error ? error : new Error(String(error))\n throw new LlmResponseError(cause.message, result, { cause })\n }\n}\n\nfunction parseJsonSafely<T>(\n content: string,\n model: string,\n jsonPayloadMode: NonNullable<LlmClientOptions['jsonPayloadMode']>,\n): T {\n const payload = jsonPayloadMode === 'exact' ? content : extractJsonPayload(content)\n try {\n return JSON.parse(payload) as T\n } catch {\n throw new Error(`LLM returned non-JSON content (model=${model})`)\n }\n}\n\n// ─── Route assertion ────────────────────────────────────────────────────\n\nexport type LlmRouteAssertionReason =\n | 'no_explicit_base_url'\n | 'base_url_blocked'\n | 'base_url_not_allowed'\n | 'no_auth'\n | 'wrong_provider'\n\nexport class LlmRouteAssertionError extends CaptureIntegrityError {\n constructor(\n message: string,\n public readonly reason: LlmRouteAssertionReason,\n public readonly baseUrl: string,\n ) {\n super(message)\n }\n}\n\nexport interface LlmRouteRequirements {\n /**\n * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to\n * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using\n * the public/free-tier router is a defect — the launch reviewer needs to\n * know exactly which provider answered.\n */\n requireExplicitBaseUrl?: boolean\n /**\n * Allowlist of acceptable base URLs. Strings match by prefix\n * (case-insensitive); RegExps test against the full base URL.\n */\n allowedBaseUrls?: Array<string | RegExp>\n /** Blocklist that takes precedence over `allowedBaseUrls`. */\n blockedBaseUrls?: Array<string | RegExp>\n /** Throw if no auth header / api key is configured. */\n requireAuth?: boolean\n /**\n * Logical provider id the configured `baseUrl` is expected to match (via\n * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.\n */\n expectedProvider?: string\n}\n\n/**\n * Fail-loud assertion that the configured LLM client points at the route\n * the caller intends. Designed for the matrix-runner preflight: invoke\n * once before any LLM call to catch misconfiguration before a sweep burns\n * dollars on the wrong provider.\n *\n * Throws `LlmRouteAssertionError`. Pure — no I/O — so it's safe to call\n * from constructors and CI gates.\n */\nexport function assertLlmRoute(opts: LlmClientOptions, req: LlmRouteRequirements = {}): void {\n const baseUrlExplicit = opts.baseUrl !== undefined\n const baseUrl = (opts.baseUrl ?? DEFAULT_BASE_URL).replace(/\\/+$/, '')\n\n if (req.requireExplicitBaseUrl && !baseUrlExplicit) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: requireExplicitBaseUrl set but opts.baseUrl is undefined; would fall back to ${DEFAULT_BASE_URL}.`,\n 'no_explicit_base_url',\n baseUrl,\n )\n }\n\n if (req.blockedBaseUrls?.some((p) => matchUrl(baseUrl, p))) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: baseUrl ${baseUrl} matches a blocked pattern.`,\n 'base_url_blocked',\n baseUrl,\n )\n }\n\n if (req.allowedBaseUrls && req.allowedBaseUrls.length > 0) {\n const ok = req.allowedBaseUrls.some((p) => matchUrl(baseUrl, p))\n if (!ok) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: baseUrl ${baseUrl} is not in the allowed list (${req.allowedBaseUrls.map(describePattern).join(', ')}).`,\n 'base_url_not_allowed',\n baseUrl,\n )\n }\n }\n\n if (req.requireAuth && !opts.apiKey && !opts.bearer && !opts.authHeader) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: requireAuth set but no apiKey, bearer, or authHeader was supplied.`,\n 'no_auth',\n baseUrl,\n )\n }\n\n if (req.expectedProvider) {\n const actual = opts.provider ?? providerFromBaseUrl(baseUrl)\n if (actual !== req.expectedProvider) {\n throw new LlmRouteAssertionError(\n `assertLlmRoute: expected provider ${req.expectedProvider} but baseUrl ${baseUrl} resolves to ${actual}.`,\n 'wrong_provider',\n baseUrl,\n )\n }\n }\n}\n\nfunction matchUrl(url: string, pattern: string | RegExp): boolean {\n if (pattern instanceof RegExp) return pattern.test(url)\n return url.toLowerCase().startsWith(pattern.toLowerCase())\n}\n\nfunction describePattern(p: string | RegExp): string {\n return p instanceof RegExp ? p.source : p\n}\n\n/**\n * Probe whether a model is reachable. Returns latency + null error on\n * success; `ok=false` + error message on any failure (HTTP, timeout,\n * network, parse). Designed for sweep preflights — fail loud at the\n * boundary before burning a 30-leaf run on a misconfigured router.\n *\n * Sends a tiny `ping` message with `maxTokens=64`. Reasoning models\n * (glm-5.1, deepseek-v4) can burn the entire budget on internal reasoning\n * for short prompts, so don't tighten this further. We don't validate\n * content; HTTP 200 means reachable.\n */\nexport async function probeLlm(\n model: string,\n opts: LlmClientOptions & { timeoutMs?: number } = {},\n): Promise<{ ok: boolean; latencyMs: number; error: string | null }> {\n const start = Date.now()\n try {\n await callLlm(\n {\n model,\n messages: [{ role: 'user', content: 'ping' }],\n maxTokens: 64,\n timeoutMs: opts.timeoutMs ?? 30_000,\n },\n opts,\n )\n return { ok: true, latencyMs: Date.now() - start, error: null }\n } catch (err) {\n return {\n ok: false,\n latencyMs: Date.now() - start,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Stateful client — construct once with defaults, call many times.\n * Thin wrapper around the free functions; exists for callers that want\n * to inject a single configured instance into multiple primitives.\n */\nexport class LlmClient {\n readonly maximumAttempts: number\n private readonly opts: LlmClientOptions\n\n constructor(opts: LlmClientOptions = {}) {\n this.opts = opts\n this.maximumAttempts = resolveMaximumAttempts(opts.maximumAttempts)\n }\n\n call(req: LlmCallRequest, per?: LlmClientOptions): Promise<LlmCallResult> {\n const options = { ...this.opts, ...per }\n return req.jsonSchema ? callLlmStructured(req, options) : callLlm(req, options)\n }\n\n callJson<T = unknown>(\n req: LlmCallRequest,\n per?: LlmClientOptions,\n ): Promise<{ value: T; result: LlmCallResult }> {\n return callLlmJson<T>(req, { ...this.opts, ...per })\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA0EA,SAAgB,2BACd,SACA,UAA4B,CAAC,GACF;CAC3B,IAAI,QAAQ,cAAc,KAAA,GAAW,OAAO,KAAA;CAC5C,IAAI,CAAC,OAAO,UAAU,QAAQ,SAAS,KAAK,QAAQ,aAAa,GAC/D,MAAM,IAAI,WAAW,kEAAkE;CAEzF,IACE,QAAQ,SAAS,MACd,YACC,MAAM,QAAQ,QAAQ,OAAO,KAAK,QAAQ,QAAQ,MAAM,SAAS,KAAK,SAAS,WAAW,CAC9F,GAEA;CAGF,MAAM,WAAW,uBAAuB,QAAQ,eAAe;CAC/D,MAAM,kBAAkB,QAAQ,wBAAwB;CAGxD,MAAM,eAAe,IAAI,YAAY,CAAC,CAAC,OACrC,KAAK,UAAU,UAAU,SAAS,iBAAiB,QAAQ,QAAQ,CAAC,CACtE,CAAC,CAAC;CAEF,MAAM,UAAU,QAAQ,cAAc,CAAC,kBAAkB,IAAI;CAC7D,MAAM,QAAQ;EACZ,aAAa,eAAe,WAAW;EACvC,cAAc,QAAQ,YAAY,WAAW;CAC/C;CACA,OAAO,QAAQ,qBACX;EAAE,oBAAoB,QAAQ;EAAoB,GAAG;CAAM,IAC3D;EAAE,OAAO,QAAQ;EAAO,GAAG;CAAM;AACvC;;AAkDA,SAAgB,mBACd,QACA,oBACkB;CAClB,MAAM,eAAe,OAAO,MAAM,sBAAsB;CACxD,MAAM,cAAc,KAAK,IAAI,GAAG,OAAO,MAAM,eAAe,YAAY;CACxE,MAAM,kBAAkB,qBAAqB,OAAO,GAAG;CACvD,OAAO;EACL,OAAO,OAAO;EACd;EACA,cAAc,OAAO,MAAM;EAC3B,iBAAiB,OAAO,MAAM;EAC9B,cAAc,eAAe,IAAI,eAAe,KAAA;EAChD,GAAI,oBAAoB,KAAA,IACpB,sBAAsB,OAAO,MAAM,aAAa,QAC9C,EAAE,mBAAmB,IACrB,OAAO,YAAY,OACjB,CAAC,IACD,EAAE,kBAAkB,OAAO,QAAQ,IACvC,EAAE,eAAe,gBAAgB;EACrC,cAAc,OAAO,MAAM,aAAa;CAC1C;AACF;AAEA,SAAS,qBAAqB,KAAkD;CAC9E,MAAM,QAAQ,IAAI,kBAAkB,IAAI;CACxC,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AACrF;;AAGA,SAAgB,wBACd,OACA,oBAC8B;CAC9B,OAAO,iBAAiB,mBACpB,mBAAmB,MAAM,QAAQ,kBAAkB,IACnD,KAAA;AACN;AAEA,IAAa,eAAb,cAAkC,eAAe;CAG7B;CACA;CACA;CAJlB,YACE,SACA,QACA,MACA,OACA;EACA,MAAM,SAAS,OAAO;EAJN,KAAA,SAAA;EACA,KAAA,OAAA;EACA,KAAA,QAAA;CAGlB;AACF;;;;AAKA,IAAa,mBAAb,cAAsC,eAAe;CAGjC;CAFlB,YACE,SACA,QACA,SACA;EACA,MAAM,SAAS,SAAS,OAAO;EAHf,KAAA,SAAA;CAIlB;AACF;AAuEA,MAAM,mBAAmB;AASzB,MAAM,qBAAqB,OAAO,QAAQ,IAAI,qBAAqB,KAAK;AACxE,MAAM,2BACJ,QAAQ,IAAI,gCAAgC,KAAA,IACxC,IACA,OAAO,QAAQ,IAAI,2BAA2B;AAEpD,SAAS,uBAAuB,YAAwC;CACtE,MAAM,WAAW,cAAc;CAC/B,IAAI,CAAC,OAAO,UAAU,QAAQ,KAAK,YAAY,GAC7C,MAAM,IAAI,WAAW,iDAAiD;CAExE,OAAO;AACT;AAEA,SAAS,mBAAmB,OAAoC;CAC9D,OAAO,OAAO,UAAU,YAAY,OAAO,cAAc,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AAC1F;AAEA,MAAM,mCAAmB,IAAI,IAAI;CAAC;CAAK;CAAK;CAAK;AAAG,CAAC;;;;;;;;;;AAWrD,MAAM,2BAA8C;CAClD;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;;;;;;;;;;;;AAaA,SAAgB,oBAAoB,KAAuB;CACzD,OAAO,kBAAkB,KAAK,CAAC;AACjC;AAEA,SAAS,kBAAkB,KAAc,OAAwB;CAC/D,IAAI,eAAe,cAAc,OAAO,iBAAiB,IAAI,IAAI,MAAM;CACvE,IAAI,EAAE,eAAe,QAAQ,OAAO;CAGpC,MAAM,SAAU,IAA6B;CAC7C,IAAI,OAAO,WAAW,YAAY,iBAAiB,IAAI,MAAM,GAAG,OAAO;CACvE,MAAM,OAAQ,IAA2B;CACzC,MAAM,WAAW,GAAG,IAAI,KAAK,IAAI,IAAI,QAAQ,IAAI,OAAO,SAAS,WAAW,OAAO;CACnF,IAAI,yBAAyB,MAAM,MAAM,EAAE,KAAK,QAAQ,CAAC,GAAG,OAAO;CACnE,MAAM,QAAS,IAA4B;CAC3C,IAAI,QAAQ,KAAK,iBAAiB,SAAS,UAAU,KACnD,OAAO,kBAAkB,OAAO,QAAQ,CAAC;CAE3C,OAAO;AACT;AAEA,SAAS,gBAAgB,SAAiC;CACxD,MAAM,IAAI,QAAQ,IAAI,aAAa;CACnC,IAAI,CAAC,GAAG,OAAO;CACf,MAAM,WAAW,OAAO,CAAC;CACzB,IAAI,OAAO,SAAS,QAAQ,KAAK,WAAW,GAAG,OAAO,WAAW;CACjE,MAAM,SAAS,KAAK,MAAM,CAAC;CAC3B,IAAI,OAAO,SAAS,MAAM,GAAG,OAAO,KAAK,IAAI,GAAG,SAAS,KAAK,IAAI,CAAC;CACnE,OAAO;AACT;;AAGA,SAAgB,UAAU,SAAyB;CACjD,OAAO,KAAK,IAAI,MAAM,KAAK,SAAS,IAAM;AAC5C;AAEA,SAAS,aAAa,MAAgD;CACpE,MAAM,UAAkC;EACtC,gBAAgB;EAChB,QAAQ;CACV;CACA,IAAI,KAAK,YACP,QAAQ,KAAK,WAAW,QAAQ,KAAK,WAAW;MAC3C,IAAI,KAAK,UAAU,KAAK,QAC7B,QAAQ,gBAAgB,UAAU,KAAK,UAAU,KAAK;CAExD,IAAI,KAAK,gBAAgB,QAAQ,qBAAqB,KAAK;CAC3D,OAAO;AACT;AAEA,SAAS,kBAAkB,QAAgB,MAAuB;CAChE,IAAI,WAAW,KAAK,OAAO;CAC3B,MAAM,QAAQ,KAAK,YAAY;CAC/B,OACE,MAAM,SAAS,iBAAiB,KAChC,MAAM,SAAS,aAAa,KAC5B,MAAM,SAAS,gBAAgB,KAC/B,MAAM,SAAS,eAAe;AAElC;AAEA,SAAS,0BAA0B,QAAgB,MAAuB;CACxE,IAAI,WAAW,OAAO,CAAC,eAAe,KAAK,IAAI,GAAG,OAAO;CACzD,OACE,iGAAiG,KAC/F,IACF,KAAK,oEAAoE,KAAK,IAAI;AAEtF;AAEA,SAAS,UACP,KACA,iBACA,iBACyB;CACzB,MAAM,OAAgC;EACpC,OAAO,IAAI;EACX,UAAU,IAAI;EACd,aAAa,IAAI,eAAe;CAClC;CACA,IAAI,IAAI,aAAa,MACnB,IAAI,wBAAwB,IAAI,KAAK,GAAG,KAAK,wBAAwB,IAAI;MACpE,KAAK,aAAa,IAAI;CAE7B,MAAM,WAAW,IAAI,YAAY;CACjC,IAAI,aAAa,KAAA,GACf,KAAK,WAAW,EAAE,MAAM,SAAS;CAGnC,IAAI,IAAI,cAAc,CAAC,iBACrB,KAAK,kBAAkB;EACrB,MAAM;EACN,aAAa;GAAE,MAAM,IAAI,WAAW;GAAM,QAAQ,IAAI,WAAW;GAAQ,QAAQ;EAAK;CACxF;MACK,IAAI,IAAI,YAAY,IAAI,YAC7B,KAAK,kBAAkB,EAAE,MAAM,cAAc;CAG/C,OAAO;AACT;AAEA,SAAS,wBAAwB,OAAwB;CACvD,OAAO,oBAAoB,KAAK,KAAK;AACvC;AAEA,eAAe,MAAM,IAA2B;CAC9C,OAAO,IAAI,SAAS,YAAY,WAAW,SAAS,EAAE,CAAC;AACzD;;;;;;;;AASA,SAAS,YAAY,mBAAoC,QAAmC;CAC1F,IAAI,CAAC,QAAQ,OAAO,kBAAkB;CACtC,IAAI,OAAQ,YAAkC,QAAQ,YACpD,OAAO,YAAY,IAAI,CAAC,kBAAkB,QAAQ,MAAM,CAAC;CAE3D,IAAI,OAAO,SACT,kBAAkB,MAAM;MAExB,OAAO,iBAAiB,eAAe,kBAAkB,MAAM,GAAG,EAAE,MAAM,KAAK,CAAC;CAElF,OAAO,kBAAkB;AAC3B;;AAGA,SAAS,iBAAiB,OAAe,YAAyC;CAChF,OAAO,cAAc,QAAQ,KAAK,IAAI,IAAI,SAAS;AACrD;;;;;;AASA,SAAgB,gBAAgB,KAAqB;CACnD,MAAM,UAAU,IAAI,KAAK;CACzB,MAAM,IAAI,QAAQ,MAAM,yCAAyC;CACjE,OAAO,IAAI,EAAE,EAAE,CAAE,KAAK,IAAI;AAC5B;AAEA,SAAgB,mBAAmB,KAAqB;CACtD,MAAM,WAAW,gBAAgB,GAAG;CACpC,IAAI;EACF,KAAK,MAAM,QAAQ;EACnB,OAAO;CACT,QAAQ;EAIN,IAAI,SAAS,WAAW,GAAG,KAAK,SAAS,WAAW,GAAG,GAAG,OAAO;CACnE;CAGA,MAAM,SAAS,CAAC,GAAG,SAAS,SAAS,OAAO,CAAC,CAAC,CAC3C,KAAK,UAAU,MAAM,KAAK,CAAC,CAC3B,QAAQ,UAAU,SAAS,IAAI;CAClC,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,YAAY,oBAAoB,UAAU,KAAK;EACrD,IAAI,CAAC,WAAW;EAChB,IAAI;GACF,KAAK,MAAM,SAAS;GACpB,OAAO;EACT,QAAQ,CAER;CACF;CAEA,OAAO;AACT;AAEA,SAAS,oBAAoB,OAAe,OAA8B;CACxE,MAAM,SAAS,MAAM;CACrB,MAAM,SAAS,WAAW,MAAM,MAAM,WAAW,MAAM,MAAM;CAC7D,IAAI,CAAC,QAAQ,OAAO;CAEpB,MAAM,QAAkB,CAAC,MAAM;CAC/B,IAAI,aAAa;CACjB,IAAI,YAAY;CAEhB,KAAK,IAAI,IAAI,QAAQ,GAAG,IAAI,MAAM,QAAQ,KAAK;EAC7C,MAAM,OAAO,MAAM;EACnB,IAAI,WAAW;GACb,YAAY;GACZ;EACF;EACA,IAAI,SAAS,MAAM;GACjB,YAAY;GACZ;EACF;EACA,IAAI,SAAS,MAAK;GAChB,aAAa,CAAC;GACd;EACF;EACA,IAAI,YAAY;EAEhB,IAAI,SAAS,KAAK,MAAM,KAAK,GAAG;OAC3B,IAAI,SAAS,KAAK,MAAM,KAAK,GAAG;OAChC,IAAI,SAAS,MAAM,MAAM,SAAS,IAAI;GACzC,MAAM,IAAI;GACV,IAAI,MAAM,WAAW,GAAG,OAAO,MAAM,MAAM,OAAO,IAAI,CAAC;EACzD;CACF;CAEA,OAAO;AACT;;;;;;AAOA,eAAsB,QACpB,KACA,OAAyB,CAAC,GACF;CACxB,MAAM,WAAW,KAAK,WAAW,iBAAA,CAAkB,QAAQ,QAAQ,EAAE;CACrE,MAAM,MAAM,GAAG,QAAQ;CACvB,MAAM,WAAW;CACjB,MAAM,YAAY,IAAI,aAAa,KAAK,oBAAoB;CAC5D,MAAM,kBAAkB,uBAAuB,KAAK,eAAe;CACnE,MAAM,UAAU,KAAK,SAAS,WAAW;CACzC,MAAM,UAAU,aAAa,IAAI;CACjC,MAAM,WAAW,KAAK,YAAY,oBAAoB,OAAO;CAC7D,MAAM,OAAO,KAAK;CAClB,MAAM,WAAW,KAAK,YAAY;CAClC,MAAM,eAAe,KAAK;CAC1B,MAAM,eAAe,KAAK;CAC1B,MAAM,aAAa,KAAK;CACxB,MAAM,gBAAgB,KAAK,IAAI;CAC/B,IAAI,KAAK,oBACP,oBAAoB,KAAK,oBAAoB;EAAE,aAAa;EAAG,cAAc;CAAE,CAAC;CAGlF,IAAI;CACJ,IAAI,mBAAmB;CACvB,KAAK,IAAI,UAAU,GAAG,UAAU,iBAAiB,WAAW;EAG1D,IAAI,cAAc,SAChB,MAAM,IAAI,aAAa,oCAAoC,YAAY;EAIzE,IAAI,UAAU,KAAK,iBAAiB,eAAe,UAAU,GAC3D,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;EAEtE,MAAM,aAAa,IAAI,gBAAgB;EACvC,MAAM,gBAAgB,YAAY,YAAY,YAAY;EAC1D,MAAM,gBAAgB,iBAAiB,WAAW,MAAM,GAAG,SAAS;EACpE,MAAM,UAAU,KAAK,IAAI;EACzB,MAAM,cAAc,UAClB,kBACA,KAAK,wBAAwB,eAC7B,KAAK,QACP;EACA,IAAI,uBAAuB;EAC3B,IAAI,MACF,MAAM,UAAU,MAAM,UAAU;GAC9B,SAAS,cAAc;GACvB,OAAO,cAAc;GACrB,QAAQ,cAAc;GACtB;GACA,OAAO,IAAI;GACX;GACA;GACA,cAAc;GACd,WAAW;GACX,WAAW;GACX,gBAAgB;GAChB;GACA,gBAAgB,CAAC;EACnB,CAAC;EAGH,IAAI;GACF,MAAM,MAAM,MAAM,QAAQ,KAAK;IAC7B,QAAQ;IACR;IACA,MAAM,KAAK,UAAU,WAAW;IAChC,QAAQ;GACV,CAAC;GACD,aAAa,aAAa;GAC1B,MAAM,kBAAkB,OAAO,gBAAgB,IAAI,OAAO,IAAI,KAAA;GAE9D,IAAI,CAAC,IAAI,IAAI;IACX,MAAM,OAAO,MAAM,IAAI,KAAK;IAC5B,IAAI,MAAM;KACR,MAAM,UAAU,MAAM,UAAU;MAC9B,SAAS,cAAc;MACvB,OAAO,cAAc;MACrB,QAAQ,cAAc;MACtB;MACA,OAAO,IAAI;MACX;MACA;MACA,cAAc;MACd,WAAW;MACX,WAAW,KAAK,IAAI;MACpB,YAAY,KAAK,IAAI,IAAI;MACzB,YAAY,IAAI;MAChB;MACA,cAAc;MACd,cAAc,QAAQ,IAAI;MAC1B,gBAAgB,CAAC;KACnB,CAAC;KACD,uBAAuB;IACzB;IACA,MAAM,MAAM,IAAI,aACd,6BAA6B,IAAI,UACjC,IAAI,QACJ,MACA,IAAI,KACN;IACA,IACE,0BAA0B,IAAI,QAAQ,IAAI,KAC1C,iBAAiB,gBAAgB,KACjC,UAAU,kBAAkB,KAC5B,CAAC,iBAAiB,eAAe,UAAU,GAC3C;KACA,UAAU;KACV,mBAAmB;MAAE,GAAG;MAAkB,aAAa;KAAE;KACzD;IACF;IACA,IACE,iBAAiB,IAAI,IAAI,MAAM,KAC/B,UAAU,kBAAkB,KAC5B,CAAC,iBAAiB,eAAe,UAAU,GAC3C;KACA,UAAU;KAEV,MAAM,MADa,gBAAgB,IAAI,OAClB,KAAK,UAAU,OAAO,CAAC;KAC5C;IACF;IACA,MAAM;GACR;GAEA,MAAM,OAAO,MAAM,IAAI,KAAK;GAC5B,IAAI;GACJ,IAAI;IACF,OAAO,KAAK,MAAM,IAAI;GACxB,SAAS,UAAU;IACjB,IAAI,MAAM;KACR,MAAM,UAAU,MAAM,UAAU;MAC9B,SAAS,cAAc;MACvB,OAAO,cAAc;MACrB,QAAQ,cAAc;MACtB;MACA,OAAO,IAAI;MACX;MACA;MACA,cAAc;MACd,WAAW;MACX,WAAW,KAAK,IAAI;MACpB,YAAY,KAAK,IAAI,IAAI;MACzB,YAAY,IAAI;MAChB;MACA,cAAc;MACd,cAAc,sBAAsB,oBAAoB,QAAQ,SAAS,UAAU,OAAO,QAAQ;MAClG,gBAAgB,CAAC;KACnB,CAAC;KACD,uBAAuB;IACzB;IACA,MAAM;GACR;GACA,IAAI,MACF,MAAM,UAAU,MAAM,UAAU;IAC9B,SAAS,cAAc;IACvB,OAAO,cAAc;IACrB,QAAQ,cAAc;IACtB;IACA,OAAO,IAAI;IACX;IACA;IACA,cAAc;IACd,WAAW;IACX,WAAW,KAAK,IAAI;IACpB,YAAY,KAAK,IAAI,IAAI;IACzB,YAAY,IAAI;IAChB;IACA,cAAc;IACd,gBAAgB,CAAC;GACnB,CAAC;GAEH,MAAM,SACJ,KAAK,UAGH;GACJ,MAAM,WACJ,KAAK,SAAS,OAAO,KAAK,UAAU,YAAY,CAAC,MAAM,QAAQ,KAAK,KAAK,IACpE,KAAK,QACN,KAAA;GACN,MAAM,eAAe,mBAAmB,UAAU,aAAa;GAC/D,MAAM,mBAAmB,mBAAmB,UAAU,iBAAiB;GACvE,MAAM,cAAc,mBAAmB,UAAU,YAAY;GAO7D,MAAM,gBALJ,UAAU,6BACV,OAAO,SAAS,8BAA8B,YAC9C,CAAC,MAAM,QAAQ,SAAS,yBAAyB,IAC5C,SAAS,4BACV,KAAA,EAAA,EACkC;GACxC,MAAM,kBACJ,iBAAiB,KAAA,IAAY,KAAA,IAAY,mBAAmB,YAAY;GAC1E,MAAM,YACJ,UAAU,yBACV,OAAO,SAAS,0BAA0B,YAC1C,CAAC,MAAM,QAAQ,SAAS,qBAAqB,IACxC,SAAS,sBAAkD,gBAC5D,KAAA;GACN,MAAM,qBAAqB,cAAc,KAAA,IAAY,KAAA,IAAY,mBAAmB,SAAS;GAC7F,MAAM,gBACJ,iBAAiB,KAAA,KACjB,qBAAqB,KAAA,MACpB,iBAAiB,KAAA,KACf,oBAAoB,KAAA,KAAa,mBAAmB,sBACtD,cAAc,KAAA,KACZ,uBAAuB,KAAA,KAAa,sBAAsB,kBAC5D,gBAAgB,KAAA,KAAa,gBAAgB,eAAe;GAC/D,MAAM,gBAAiB,KAAK,kBAAkB,KAAK;GACnD,MAAM,UAAU,QAAQ,SAAS,WAAW;GAE5C,MAAM,iBACJ,OAAO,kBAAkB,YAAY,iBAAiB,KAAK,qBACvD,oBAAoB,KAAK,oBAAoB;IAC3C,aAAa,gBAAiB,sBAAsB;IACpD,GAAI,qBAAqB,EAAE,cAAc,mBAAmB,IAAI,CAAC;IACjE,cAAc;GAChB,CAAC,IACD,KAAA;GAEN,OAAO;IACL;IACA,cAAc,QAAQ,iBAAiB;IACvC,cAAc,QAAQ,KAAK,CAAC,CAAC,WAAW;IACxC,OAAO;KACL,cAAc,gBAAgB;KAC9B,kBAAkB,oBAAoB;KACtC,aAAa,gBAAgB,gBAAgB,MAAM,oBAAoB;KACvE,UAAU;KACV;KACA;IACF;IACA,SAAS,OAAO,kBAAkB,WAAW,gBAAiB,kBAAkB;IAChF,OAAQ,KAAK,SAAoB,IAAI;IACrC,YAAY,KAAK,IAAI,IAAI;IACzB,KAAK;GACP;EACF,SAAS,KAAK;GACZ,aAAa,aAAa;GAC1B,UAAU;GAIV,IAAI,cAAc,SAAS;IACzB,IAAI,QAAQ,CAAC,sBACX,MAAM,UAAU,MAAM,UAAU;KAC9B,SAAS,cAAc;KACvB,OAAO,cAAc;KACrB,QAAQ,cAAc;KACtB;KACA,OAAO,IAAI;KACX;KACA;KACA,cAAc;KACd,WAAW;KACX,WAAW,KAAK,IAAI;KACpB,YAAY,KAAK,IAAI,IAAI;KACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;KAC7D,gBAAgB,CAAC;IACnB,CAAC;IAEH,MAAM;GACR;GACA,IAAI,QAAQ,CAAC,sBAIX,MAAM,UAAU,MAAM,UAAU;IAC9B,SAAS,cAAc;IACvB,OAAO,cAAc;IACrB,QAAQ,cAAc;IACtB;IACA,OAAO,IAAI;IACX;IACA;IACA,cAAc;IACd,WAAW;IACX,WAAW,KAAK,IAAI;IACpB,YAAY,KAAK,IAAI,IAAI;IACzB,cAAc,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAC7D,gBAAgB,CAAC;GACnB,CAAC;GAEH,IACE,UAAU,kBAAkB,KAC5B,oBAAoB,GAAG,KACvB,CAAC,iBAAiB,eAAe,UAAU,GAC3C;IACA,MAAM,MAAM,UAAU,OAAO,CAAC;IAC9B;GACF;GACA,MAAM;EACR;CACF;CACA,MAAM,mBAAmB,QAAQ,UAAU,IAAI,MAAM,OAAO,OAAO,CAAC;AACtE;AAEA,eAAe,UACb,MACA,UACA,OACe;CAGf,IAAI;EACF,MAAM,KAAK,OAAO,SAAS,KAAK,CAAC;CACnC,QAAQ,CAER;AACF;AAEA,SAAS,gBAAgB,GAAoC;CAC3D,MAAM,MAA8B,CAAC;CACrC,EAAE,SAAS,OAAO,QAAQ;EACxB,IAAI,OAAO;CACb,CAAC;CACD,OAAO;AACT;AAEA,SAAS,gBAAwB;CAC/B,IAAI,OAAO,WAAW,QAAQ,eAAe,YAAY,OAAO,WAAW,OAAO,WAAW;CAC7F,OAAO,GAAG,KAAK,IAAI,CAAC,CAAC,SAAS,EAAE,EAAE,GAAG,KAAK,OAAO,CAAC,CAAC,SAAS,EAAE,CAAC,CAAC,MAAM,GAAG,EAAE;AAC7E;;;;;;;AAQA,eAAsB,YACpB,KACA,OAAyB,CAAC,GACoB;CAC9C,MAAM,SAAS,MAAM,kBAAkB,KAAK,IAAI;CAEhD,OAAO;EAAE,OADK,gBAAmB,QAAQ,KAAK,mBAAmB,SACpD;EAAG;CAAO;AACzB;;AAGA,eAAe,kBACb,KACA,OAAyB,CAAC,GACF;CACxB,IAAI;EACF,OAAO,MAAM,QAAQ;GAAE,GAAG;GAAK,UAAU,IAAI,YAAY,CAAC,IAAI;EAAW,GAAG,IAAI;CAClF,SAAS,KAAK;EACZ,IACE,KAAK,wBAAwB,iBAC7B,eAAe,gBACf,kBAAkB,IAAI,QAAQ,IAAI,IAAI,KACtC,IAAI,YAGJ,OAAO,MAAM,QAAQ;GADiB,GAAG;GAAK,UAAU;GAAM,YAAY,KAAA;EAC3C,GAAG,IAAI;EAExC,MAAM;CACR;AACF;AAEA,SAAS,gBACP,QACA,iBACG;CACH,IAAI;EACF,IAAI,OAAO,iBAAiB,UAC1B,MAAM,IAAI,MACR,8CAA8C,OAAO,MAAM,uBAC7D;EAEF,OAAO,gBAAmB,OAAO,SAAS,OAAO,OAAO,eAAe;CACzE,SAAS,OAAO;EACd,IAAI,iBAAiB,kBAAkB,MAAM;EAC7C,MAAM,QAAQ,iBAAiB,QAAQ,QAAQ,IAAI,MAAM,OAAO,KAAK,CAAC;EACtE,MAAM,IAAI,iBAAiB,MAAM,SAAS,QAAQ,EAAE,MAAM,CAAC;CAC7D;AACF;AAEA,SAAS,gBACP,SACA,OACA,iBACG;CACH,MAAM,UAAU,oBAAoB,UAAU,UAAU,mBAAmB,OAAO;CAClF,IAAI;EACF,OAAO,KAAK,MAAM,OAAO;CAC3B,QAAQ;EACN,MAAM,IAAI,MAAM,wCAAwC,MAAM,EAAE;CAClE;AACF;AAWA,IAAa,yBAAb,cAA4C,sBAAsB;CAG9C;CACA;CAHlB,YACE,SACA,QACA,SACA;EACA,MAAM,OAAO;EAHG,KAAA,SAAA;EACA,KAAA,UAAA;CAGlB;AACF;;;;;;;;;;AAmCA,SAAgB,eAAe,MAAwB,MAA4B,CAAC,GAAS;CAC3F,MAAM,kBAAkB,KAAK,YAAY,KAAA;CACzC,MAAM,WAAW,KAAK,WAAW,iBAAA,CAAkB,QAAQ,QAAQ,EAAE;CAErE,IAAI,IAAI,0BAA0B,CAAC,iBACjC,MAAM,IAAI,uBACR,gGAAgG,iBAAiB,IACjH,wBACA,OACF;CAGF,IAAI,IAAI,iBAAiB,MAAM,MAAM,SAAS,SAAS,CAAC,CAAC,GACvD,MAAM,IAAI,uBACR,2BAA2B,QAAQ,8BACnC,oBACA,OACF;CAGF,IAAI,IAAI,mBAAmB,IAAI,gBAAgB,SAAS,GAElD;MAAA,CADO,IAAI,gBAAgB,MAAM,MAAM,SAAS,SAAS,CAAC,CACxD,GACJ,MAAM,IAAI,uBACR,2BAA2B,QAAQ,+BAA+B,IAAI,gBAAgB,IAAI,eAAe,CAAC,CAAC,KAAK,IAAI,EAAE,KACtH,wBACA,OACF;CAAA;CAIJ,IAAI,IAAI,eAAe,CAAC,KAAK,UAAU,CAAC,KAAK,UAAU,CAAC,KAAK,YAC3D,MAAM,IAAI,uBACR,sFACA,WACA,OACF;CAGF,IAAI,IAAI,kBAAkB;EACxB,MAAM,SAAS,KAAK,YAAY,oBAAoB,OAAO;EAC3D,IAAI,WAAW,IAAI,kBACjB,MAAM,IAAI,uBACR,qCAAqC,IAAI,iBAAiB,eAAe,QAAQ,eAAe,OAAO,IACvG,kBACA,OACF;CAEJ;AACF;AAEA,SAAS,SAAS,KAAa,SAAmC;CAChE,IAAI,mBAAmB,QAAQ,OAAO,QAAQ,KAAK,GAAG;CACtD,OAAO,IAAI,YAAY,CAAC,CAAC,WAAW,QAAQ,YAAY,CAAC;AAC3D;AAEA,SAAS,gBAAgB,GAA4B;CACnD,OAAO,aAAa,SAAS,EAAE,SAAS;AAC1C;;;;;;;;;;;;AAaA,eAAsB,SACpB,OACA,OAAkD,CAAC,GACgB;CACnE,MAAM,QAAQ,KAAK,IAAI;CACvB,IAAI;EACF,MAAM,QACJ;GACE;GACA,UAAU,CAAC;IAAE,MAAM;IAAQ,SAAS;GAAO,CAAC;GAC5C,WAAW;GACX,WAAW,KAAK,aAAa;EAC/B,GACA,IACF;EACA,OAAO;GAAE,IAAI;GAAM,WAAW,KAAK,IAAI,IAAI;GAAO,OAAO;EAAK;CAChE,SAAS,KAAK;EACZ,OAAO;GACL,IAAI;GACJ,WAAW,KAAK,IAAI,IAAI;GACxB,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF;;;;;;AAOA,IAAa,YAAb,MAAuB;CACrB;CACA;CAEA,YAAY,OAAyB,CAAC,GAAG;EACvC,KAAK,OAAO;EACZ,KAAK,kBAAkB,uBAAuB,KAAK,eAAe;CACpE;CAEA,KAAK,KAAqB,KAAgD;EACxE,MAAM,UAAU;GAAE,GAAG,KAAK;GAAM,GAAG;EAAI;EACvC,OAAO,IAAI,aAAa,kBAAkB,KAAK,OAAO,IAAI,QAAQ,KAAK,OAAO;CAChF;CAEA,SACE,KACA,KAC8C;EAC9C,OAAO,YAAe,KAAK;GAAE,GAAG,KAAK;GAAM,GAAG;EAAI,CAAC;CACrD;AACF"}
|
|
@@ -3,7 +3,7 @@ import { s as TraceStore } from "../store-CT9YIIve.js";
|
|
|
3
3
|
import { Ct as GoldenItem, St as ContinuousCalibrationResult, a as CorpusAgreementReport, bt as ContinuousAgreement, vt as CalibrationResult, yt as CandidateScore } from "../statistics-B4u_CiFd.js";
|
|
4
4
|
import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
|
|
5
5
|
import { n as SeriesConvergenceResult, t as SeriesConvergenceOptions } from "../series-convergence-ofsqPWhs.js";
|
|
6
|
-
import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-
|
|
6
|
+
import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-C4r4Y-q8.js";
|
|
7
7
|
//#region src/meta-eval/correlation-study.d.ts
|
|
8
8
|
interface EvalMetricSpec {
|
|
9
9
|
id: string;
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { c as ValidationError } from "./errors-D-LKuDhb.js";
|
|
2
2
|
import { i as ROLLOUT_SCHEMA, s as assertMinted } from "./schema-C6DW4ZHR.js";
|
|
3
3
|
import { a as scoreOrigin, i as rolloutRewardFields } from "./reward-nw2xZGZG.js";
|
|
4
|
-
import { o as runTaskScore } from "./run-record-
|
|
4
|
+
import { o as runTaskScore } from "./run-record-CWN8-VsV.js";
|
|
5
5
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
6
6
|
//#region src/rollout/mint.ts
|
|
7
7
|
/**
|
|
@@ -315,4 +315,4 @@ async function mintRolloutRows(records, store, options = {}) {
|
|
|
315
315
|
//#endregion
|
|
316
316
|
export { unmintableReasons as n, mintRolloutRows as t };
|
|
317
317
|
|
|
318
|
-
//# sourceMappingURL=mint-
|
|
318
|
+
//# sourceMappingURL=mint-DD-0oQTA.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mint-Ctwk079K.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
|
|
1
|
+
{"version":3,"file":"mint-DD-0oQTA.js","names":[],"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst isObject = (value: unknown): value is Record<string, unknown> =>\n typeof value === 'object' && value !== null\n\ninterface MintFieldCheck {\n /** The RunRecord path, spelled the way the caller has to fix it. */\n readonly field: string\n /** True when the record carries something the line can honestly be built from. */\n readonly present: (bag: Record<string, unknown>) => boolean\n /** What the caller writes onto the record, and why that value and not another. */\n readonly remedy: string\n}\n\n/**\n * The RunRecord fields mint reads that a record can be missing even though the\n * TYPE says it cannot. There are exactly two ways that happens:\n *\n * 1. The field was OPTIONAL when the record was serialized. `costProvenance`,\n * `terminalOutcome` and `scenarioId` were optional through agent-eval\n * 0.125 and became required in 0.126, with no on-disk migration — so every\n * ledger written before 0.126 is full of records the type calls complete.\n * 2. Mint reads a level DEEPER than the record's own type is checked at:\n * `outcome.raw`, `tokenUsage.input`, `tokenUsage.output`.\n *\n * Nothing else needs a check here. Every other field mint copies is a top-level\n * scalar landing in a typed slot on the line, where an absent value arrives as\n * `undefined` and `assertMinted` refuses it by name. These are the ones where an\n * absent value instead kills the join with `TypeError: Cannot read properties of\n * undefined`, or — worse — mints a line that reads as measured.\n *\n * This is deliberately NOT `validateRunRecord`. That validator answers \"is this\n * a valid RunRecord\", which is a wider question than \"can a rollout line be\n * built from this one\": it also enforces model-snapshot discipline, the\n * `terminalFailureReason` coupling, and the `costUsd === costProvenance.usd`\n * agreement. Routing the mint door through it would refuse records mint can\n * mint honestly today (a model alias with no snapshot date, for one), which is\n * a policy change with its own blast radius and not this bug. The door asks the\n * narrower question and answers it precisely.\n */\nconst MINT_FIELD_CHECKS: readonly MintFieldCheck[] = [\n {\n field: 'costProvenance',\n present: (bag) => isObject(bag.costProvenance) && typeof bag.costProvenance.kind === 'string',\n remedy:\n \"Records written before agent-eval 0.126 predate this field and carry `costUsd: 0` as the documented uncaptured sentinel, which is NOT an observed zero. Backfill it as costProvenance: { kind: 'uncaptured', usd: null } WITH costUsd: null — an uncaptured cost whose costUsd is non-null is rejected by validateRunRecord, so provenance alone leaves the record invalid.\",\n },\n {\n field: 'tokenUsage',\n present: (bag) => isObject(bag.tokenUsage),\n remedy:\n \"The line's cost.tokens_in and cost.tokens_out are read from it. Backfill it from the provider's usage report; mint will not write 0 for tokens nobody counted.\",\n },\n {\n field: 'tokenUsage.input',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.input === 'number',\n remedy: \"The line's cost.tokens_in is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'tokenUsage.output',\n present: (bag) => !isObject(bag.tokenUsage) || typeof bag.tokenUsage.output === 'number',\n remedy: \"The line's cost.tokens_out is read from it, and a missing count is not a zero count.\",\n },\n {\n field: 'outcome',\n present: (bag) => isObject(bag.outcome),\n remedy:\n \"The line's reward, reward_source and metrics are all read from it. A record with no outcome carries no training label at all, and mint refuses an unlabeled row.\",\n },\n {\n field: 'outcome.raw',\n // Reported only when `outcome` itself is present: one absent field should\n // produce one reason per CAUSE, not one per path that dereferences it.\n present: (bag) => !isObject(bag.outcome) || isObject(bag.outcome.raw),\n remedy:\n 'It is the metric bag copied verbatim into the line\\'s outcome.metrics. `{ ...undefined }` spreads to `{}` without complaint, so an absent bag would mint as \"this run reported no metrics\" — a different claim from \"this record predates the field\". Backfill it as {} only when that is what you mean.',\n },\n {\n field: 'terminalOutcome',\n present: (bag) => typeof bag.terminalOutcome === 'string',\n remedy:\n \"It became required in agent-eval 0.126. Backfill it from root-run or process evidence, or as 'unknown' when the producer has none — mint will not decide the line's is_completed and is_truncated for you.\",\n },\n {\n field: 'scenarioId',\n present: (bag) => typeof bag.scenarioId === 'string' && bag.scenarioId.length > 0,\n remedy:\n \"It became required in agent-eval 0.126 and becomes the line's task.instance_id, which must be a non-empty string. Backfill it from the scenario the run was dealt (pre-0.126 producers often left it in outcome.raw.scenario_id).\",\n },\n]\n\n/**\n * Why a record cannot be minted, one entry per missing field, empty when it can.\n *\n * Exported so a caller can partition a whole ledger — \"which of my 2742 records\n * predate 0.126\" — without catching an exception per record, and without\n * re-deriving the field list on their side. A re-derived list is a list that\n * drifts from the door it is supposed to predict.\n *\n * Takes a `RunRecord` because that is what the caller holds and what the\n * compiler agrees they hold. The type is precisely the thing that is wrong, so\n * the checks read the record as the untyped bag it actually is on disk.\n */\nexport function unmintableReasons(record: RunRecord): string[] {\n const bag = record as unknown as Record<string, unknown>\n return MINT_FIELD_CHECKS.filter((check) => !check.present(bag)).map(\n (check) => `${check.field} is missing. ${check.remedy}`,\n )\n}\n\n/**\n * The mint door THROWS on a record it cannot build a line from. It does NOT\n * normalise an absent `costProvenance` to `{kind:'uncaptured', usd:null}`, and\n * the choice is not stylistic:\n *\n * - Normalising cannot cover the record, only part of it. `terminalOutcome`\n * feeds `is_completed` and `is_truncated`, which the rollout schema requires\n * to be BOOLEAN — there is no null to fall back to, so every possible\n * default is a claim about how the run ended. A door that quietly fixes the\n * cost and invents the ending is a door no caller can predict.\n * - Normalising the cost requires knowing what `costUsd: 0` meant, and mint\n * cannot know. A genuinely free run and an uncaptured one are the same bytes\n * in a pre-0.126 record; only the producer can tell them apart. Guessing is\n * exactly the failure this guard exists to stop — the 0.125 optional chain\n * `record.costProvenance?.kind === 'uncaptured'` already made that guess,\n * silently, and every record it touched minted `cost.usd: 0`: an unmeasured\n * cost published as a measured zero, into a training dataset.\n * - `requireTaskScore`, directly above, already refuses an unlabeled record\n * for the same reason: \"nobody graded this\" is not \"graded zero\". \"Nobody\n * billed this\" is not \"billed zero\".\n *\n * The caller who wants historical records minted backfills them at their store,\n * in one pass, where `costUsd` can be corrected alongside `costProvenance` —\n * which is the only place that decision can be made correctly. The refusal names\n * the run, names every missing field, and spells the value to write.\n */\nfunction requireMintableRecord(record: RunRecord): void {\n const reasons = unmintableReasons(record)\n if (reasons.length === 0) return\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: ${reasons.join('\\n ')}`)\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // Field presence first, and BEFORE `requireTaskScore`: that guard reads\n // `record.outcome.searchScore` on its way to the answer, so an absent\n // `outcome` would throw a bare TypeError from inside the guard whose whole\n // job is to produce a clean refusal.\n //\n // Both branches of `mintRolloutRows` — the traced line and the gap line —\n // land here, which is the point: `mintLine` is the only constructor of a\n // `MintedRolloutLine` from a RunRecord, so there is no path into the waist\n // that skips the check and no way to get this wrong from the outside.\n requireMintableRecord(record)\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEA,MAAM,UAAU,GAAY,UAAmC;CAE7D,OAAO,OADG,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC,MACpC,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;CACpE,MAAM,OAAoB;EACxB,MAAM,KAAK;EACX,MAAM,MAAM,KAAK,IAAI;EACrB,QAAQ,KAAK;EACb,YAAY,KAAK,YAAY,KAAA,IAAY,KAAK,UAAU,KAAK,YAAY,KAAA;CAC3E;CACA,IAAI,KAAK,SAAS,OAAO;EACvB,MAAM,MAAM;EACZ,MAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS;EAChD,IAAI,MAAM,KAAK,QAAQ,MAAM,KAAK,OAAO;EACzC,IAAI,IAAI,WAAW,KAAA,GAAW,KAAK,SAAS,MAAM,IAAI,MAAM;CAC9D,OAAO,IAAI,KAAK,SAAS,QAAQ;EAC/B,MAAM,OAAO;EACb,KAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;EACpC,IAAI,KAAK,WAAW,KAAA,GAAW,KAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;CACxE;CACA,OAAO;AACT;;AAGA,SAAS,kBAAkB,OAAe,OAAuC;CAC/E,MAAM,OAAO,MAAM,QAAQ,MAAoB,EAAE,SAAS,KAAK;CAC/D,MAAM,OAAO,KAAK,KAAK,SAAS;CAChC,IAAI,CAAC,MAAM,OAAO,CAAC;CACnB,MAAM,WAA0B,KAAK,SAAS,KAAK,OAAgB;EACjE,MAAM,EAAE;EACR,SAAS,MAAM,EAAE,OAAO;CAC1B,EAAE;CACF,IAAI,KAAK,WAAW,KAAA,KAAa,KAAK,WAAW,IAC/C,SAAS,KAAK;EAAE,MAAM;EAAa,SAAS,MAAM,KAAK,MAAM;CAAE,CAAC;CAElE,OAAO;AACT;AAgBA,MAAM,gBAAgE;CACpE,SAAS;CACT,QAAQ;CACR,UAAU;AACZ;;;;;;;;;AAUA,SAAS,iBAAiB,QAAyB;CACjD,IAAI,aAAa,MAAM,MAAM,KAAA,GAC3B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,wBAAwB;AAElG;AAEA,MAAM,YAAY,UAChB,OAAO,UAAU,YAAY,UAAU;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqCzC,MAAM,oBAA+C;CACnD;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,cAAc,KAAK,OAAO,IAAI,eAAe,SAAS;EACrF,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,UAAU;EACzC,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,UAAU;EAC/E,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,CAAC,SAAS,IAAI,UAAU,KAAK,OAAO,IAAI,WAAW,WAAW;EAChF,QAAQ;CACV;CACA;EACE,OAAO;EACP,UAAU,QAAQ,SAAS,IAAI,OAAO;EACtC,QACE;CACJ;CACA;EACE,OAAO;EAGP,UAAU,QAAQ,CAAC,SAAS,IAAI,OAAO,KAAK,SAAS,IAAI,QAAQ,GAAG;EACpE,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,oBAAoB;EACjD,QACE;CACJ;CACA;EACE,OAAO;EACP,UAAU,QAAQ,OAAO,IAAI,eAAe,YAAY,IAAI,WAAW,SAAS;EAChF,QACE;CACJ;AACF;;;;;;;;;;;;;AAcA,SAAgB,kBAAkB,QAA6B;CAC7D,MAAM,MAAM;CACZ,OAAO,kBAAkB,QAAQ,UAAU,CAAC,MAAM,QAAQ,GAAG,CAAC,CAAC,CAAC,KAC7D,UAAU,GAAG,MAAM,MAAM,eAAe,MAAM,QACjD;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAS,sBAAsB,QAAyB;CACtD,MAAM,UAAU,kBAAkB,MAAM;CACxC,IAAI,QAAQ,WAAW,GAAG;CAC1B,MAAM,IAAI,gBAAgB,+BAA+B,OAAO,MAAM,IAAI,QAAQ,KAAK,MAAM,GAAG;AAClG;AAEA,MAAM,iBAA8D;CAClE,QAAQ;CACR,KAAK;CACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;CAUnB,sBAAsB,MAAM;CAI5B,iBAAiB,MAAM;CAGvB,MAAM,eAAe,oBAAoB,MAAM;CAC/C,MAAM,aAAa,OAAO,eAAe,SAAS;CAClD,MAAM,kBAAkB,OAAO;CAC/B,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,cAAc,oBAAoB,eAAe,oBAAoB;CAC3E,MAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,oBAC9C;CAIN,OAAO,aACL;EACE,QAAQ;EACR,YAAY,OAAO;EACnB,mBAAmB;EACnB,QAAQ,OAAO;EACf,eAAe,OAAO;EACtB,cAAc,OAAO;EACrB,YAAY;EACZ,iBAAiB;EACjB,MAAM,QAAQ,QAAQ;EACtB,MAAM;GACJ,OAAO,QAAQ,SAAS,OAAO;GAC/B,aAAa,OAAO;GACpB,OAAO,eAAe,OAAO;GAC7B,MAAM,OAAO;GACb,KAAK;EACP;EACA,QAAQ;GACN,SAAS;GACT,iBAAiB;GACjB,OAAO,OAAO;GACd,UAAU;GACV,gBAAgB,OAAO;GACvB,aAAa,OAAO;GACpB,aAAa,OAAO;GACpB,uBAAuB,OAAO,cAAc,UAAU;GACtD,UAAU;EACZ;EACA;EACA,WAAW,CAAC;EACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;EACpC,SAAS;GACP,GAAG;GACH,eAAe,cAAc,YAAY,MAAM;GAC/C,SAAS;GAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;GACjC,cAAc;GACd,cAAc;GACd,OAAO;EACT;EACA,MAAM;GACJ,KAAK,aAAa,OAAO,OAAO;GAChC,WAAW,OAAO,WAAW;GAC7B,YAAY,OAAO,WAAW;GAC9B,kBAAkB,OAAO,WAAW,aAAa;GACjD,YAAY,OAAO,WAAW,UAAU;GACxC,aAAa,OAAO,WAAW,cAAc;GAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;EACzC;EACA,WAAW;GAAE,YAAY;GAAM,SAAS;GAAM,gBAAgB;EAAK;EACnE,YAAY;GACV,aAAa;GACb,SAAS;GACT,GAAI,QAAQ,KAAA,IAAY,EAAE,IAAI,IAAI,CAAC;EACrC;CACF,GACA,+BAA+B,OAAO,OACxC;AACF;;;;;;;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;CAC5B,MAAM,QAAQ,QAAQ,WAAW,MAAM;CACvC,MAAM,cAAc,QAAQ,MAAM,qBAAK,IAAI,KAAK,EAAA,CAAG,YAAY;CAC/D,MAAM,OAA4B,CAAC;CACnC,MAAM,gBAA0B,CAAC;CACjC,KAAK,MAAM,UAAU,SAAS;EAC5B,MAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;EAC5D,IAAI,WAAW,MAAM,WAAW,GAAG;GACjC,cAAc,KAAK,OAAO,KAAK;GAC/B,KAAK,KACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC,CACxF;GACA;EACF;EACA,IAAI,QAAQ,WAAW,MAAM,KAAK,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;EAClE,IAAI,QAAQ,aAAa,KAAA,KAAa,MAAM,SAAS,QAAQ,UAAU;GAGrE,MAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;GAC3C,MAAM,OAAO,QAAQ,WAAW;GAChC,QAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;EACvE;EACA,MAAM,eAAe,kBACnB,WAAW,MAAM,KAAK,MAAM,EAAE,IAAI,GAClC,KACF;EACA,MAAM,MACJ,aAAa,WAAW,IAAI,uDAAuD,KAAA;EACrF,KAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;CAC3E;CACA,OAAO;EAAE;EAAM;CAAc;AAC/B"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { p as CostProvenance } from "../cost-ledger-FuQvHxPm.js";
|
|
2
|
-
import { w as JudgeScore } from "../types-
|
|
2
|
+
import { w as JudgeScore } from "../types-DcJxgsLy.js";
|
|
3
3
|
import { o as MatrixResult } from "../index-DSC51roc.js";
|
|
4
4
|
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
5
5
|
//#region src/multishot/types.d.ts
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.144.
|
|
5
|
+
"version": "0.144.3",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
2
|
-
import { s as validateRunRecord } from "./run-record-
|
|
2
|
+
import { s as validateRunRecord } from "./run-record-CWN8-VsV.js";
|
|
3
3
|
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
4
4
|
import { dirname } from "node:path";
|
|
5
5
|
//#region src/action-policy.ts
|
|
@@ -1455,4 +1455,4 @@ function controlFailureClassFromVerification(verification) {
|
|
|
1455
1455
|
//#endregion
|
|
1456
1456
|
export { observeAll as _, jsonlReviewStore as a, scoreFromEvals as c, runAgentControlLoop as d, stopOnNoProgress as f, noProgressDetector as g, errorStreakDetector as h, inMemoryReviewStore as i, allCriticalPassed as l, subjectiveEval as m, runProposeReviewAsControlLoop as n, runProposeReview as o, stopOnRepeatedAction as p, createLlmReviewer as r, controlRunToRunRecord as s, controlFailureClassFromVerification as t, objectiveEval as u, repeatedActionDetector as v, evaluateActionPolicy as y };
|
|
1457
1457
|
|
|
1458
|
-
//# sourceMappingURL=propose-review-control-
|
|
1458
|
+
//# sourceMappingURL=propose-review-control-BciUCZoh.js.map
|