@tangle-network/agent-eval 0.173.3 → 0.175.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. package/CHANGELOG.md +54 -0
  2. package/README.md +1 -1
  3. package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
  4. package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
  5. package/dist/adapters/http.d.ts +2 -2
  6. package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
  7. package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
  8. package/dist/analyst/index.d.ts +7 -9
  9. package/dist/analyst/index.d.ts.map +1 -1
  10. package/dist/analyst/index.js +8 -8
  11. package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
  12. package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
  13. package/dist/{benchmark-command-BY9oscke.js → benchmark-command-D_5xG9LG.js} +13 -13
  14. package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-D_5xG9LG.js.map} +1 -1
  15. package/dist/benchmarks/index.d.ts +3 -4
  16. package/dist/benchmarks/index.d.ts.map +1 -1
  17. package/dist/benchmarks/index.js +3 -3
  18. package/dist/campaign/index.d.ts +5 -9
  19. package/dist/campaign/index.js +7 -7
  20. package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
  21. package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
  22. package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
  23. package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
  24. package/dist/cli.js +9 -2
  25. package/dist/cli.js.map +1 -1
  26. package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
  27. package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
  28. package/dist/contract/index.d.ts +9 -10
  29. package/dist/contract/index.js +8 -8
  30. package/dist/{default-registry-B0bKikCb.js → default-registry-DBqVI4pq.js} +5 -5
  31. package/dist/{default-registry-B0bKikCb.js.map → default-registry-DBqVI4pq.js.map} +1 -1
  32. package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
  33. package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
  34. package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
  35. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
  36. package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
  37. package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
  38. package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
  39. package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
  40. package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
  41. package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
  42. package/dist/experiment/index.d.ts +1 -4
  43. package/dist/experiment/index.d.ts.map +1 -1
  44. package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
  45. package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
  46. package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
  47. package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
  48. package/dist/fuzz.js +1 -1
  49. package/dist/fuzz.js.map +1 -1
  50. package/dist/hosted/index.d.ts +1 -1
  51. package/dist/{index-D0Db5X-4.d.ts → index-BAAiSF3_.d.ts} +5 -5
  52. package/dist/{index-D0Db5X-4.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
  53. package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
  54. package/dist/index-BTrx5s8m.d.ts.map +1 -0
  55. package/dist/index-DKXuBPXf.d.ts +3840 -0
  56. package/dist/index-DKXuBPXf.d.ts.map +1 -0
  57. package/dist/index.d.ts +11 -13
  58. package/dist/index.d.ts.map +1 -1
  59. package/dist/index.js +11 -11
  60. package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
  61. package/dist/integrity-DsHWCebQ.js.map +1 -0
  62. package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
  63. package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
  64. package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
  65. package/dist/llm-judge-DmNaBrXB.js.map +1 -0
  66. package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
  67. package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
  68. package/dist/multishot/golden/index.d.ts +1 -1
  69. package/dist/multishot/index.d.ts +2 -2
  70. package/dist/openapi.json +1 -1
  71. package/dist/opencode-sqlite-CNw3vubS.js +145 -0
  72. package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
  73. package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
  74. package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
  75. package/dist/report-command-DKlXfU5r.js +1528 -0
  76. package/dist/report-command-DKlXfU5r.js.map +1 -0
  77. package/dist/rl.d.ts +1 -1
  78. package/dist/rl.d.ts.map +1 -1
  79. package/dist/rl.js.map +1 -1
  80. package/dist/rollout/index.js +3 -2
  81. package/dist/{rollout-C-znbbYg.js → rollout-CGlDq1GI.js} +3 -2
  82. package/dist/{rollout-C-znbbYg.js.map → rollout-CGlDq1GI.js.map} +1 -1
  83. package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
  84. package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
  85. package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
  86. package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
  87. package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
  88. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
  89. package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
  90. package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
  91. package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
  92. package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
  93. package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
  94. package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
  95. package/dist/supervisor-run/index.d.ts +71 -6
  96. package/dist/supervisor-run/index.d.ts.map +1 -1
  97. package/dist/supervisor-run/index.js +6 -1357
  98. package/dist/supervisor-run/index.js.map +1 -1
  99. package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
  100. package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
  101. package/dist/terminal-record-Ce9_UjRz.js +539 -0
  102. package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
  103. package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
  104. package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
  105. package/dist/trace-repair/index.d.ts +1 -1
  106. package/dist/traces.d.ts +2 -2
  107. package/dist/traces.js +4 -4
  108. package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
  109. package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
  110. package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
  111. package/dist/types-DQ0e2E7y.js.map +1 -0
  112. package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
  113. package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
  114. package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
  115. package/dist/types-vUdAx2Cj.d.ts.map +1 -0
  116. package/docs/campaign-proposers.md +42 -0
  117. package/package.json +1 -1
  118. package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
  119. package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
  120. package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
  121. package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
  122. package/dist/benchmark-BjLGkfnN.d.ts +0 -236
  123. package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
  124. package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
  125. package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
  126. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
  127. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
  128. package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
  129. package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
  130. package/dist/index-BQqOjerE.d.ts.map +0 -1
  131. package/dist/index-CFDffsKz.d.ts +0 -1135
  132. package/dist/index-CFDffsKz.d.ts.map +0 -1
  133. package/dist/integrity-BWywb34E.js.map +0 -1
  134. package/dist/llm-judge-BfqMFo4h.js.map +0 -1
  135. package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
  136. package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
  137. package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
  138. package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
  139. package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
  140. package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
  141. package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
  142. package/dist/proposal-findings-bko3GGy-.js.map +0 -1
  143. package/dist/provenance-CRY67X50.d.ts +0 -1995
  144. package/dist/provenance-CRY67X50.d.ts.map +0 -1
  145. package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
  146. package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
  147. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
  148. package/dist/types-CiWITkGo.js.map +0 -1
  149. package/dist/types-CoPUTiXb.d.ts.map +0 -1
@@ -0,0 +1 @@
1
+ {"version":3,"file":"terminal-record-Ce9_UjRz.js","names":["unhandled"],"sources":["../src/supervisor-run/source-facts.ts","../src/supervisor-run/types.ts","../src/supervisor-run/terminal-record.ts"],"sourcesContent":["import type { SupervisorRunNodeRole, SupervisorRunSources } from './types'\n\n// ---------------------------------------------------------------------------\n// Journal shapes — structurally parsed. The journal is the contract, not the type.\n// ---------------------------------------------------------------------------\n\ninterface JournalEvent {\n kind?: unknown\n id?: unknown\n parent?: unknown\n label?: unknown\n role?: unknown\n profileDigest?: unknown\n runtime?: unknown\n status?: unknown\n verdict?: unknown\n reason?: unknown\n infra?: unknown\n seq?: unknown\n at?: unknown\n spend?: unknown\n spent?: unknown\n}\n\ninterface Tokens {\n input: number\n output: number\n cacheRead: number\n cacheWrite: number\n /** False when the event carried no cache counters at all — not \"zero cached\". */\n hasCache: boolean\n /**\n * False only when the record carried `cacheBreakdownKnown: false` — Runtime's mark for a\n * provider that reported a total without splitting cache reads from cache writes. The\n * counters that ARE present are then a partial split, never a measured \"zero cached\".\n */\n breakdownKnown: boolean\n}\n\nexport interface SpendLike {\n tokens: Tokens\n /**\n * False only when the record carried `tokensKnown: false` — Runtime's mark for work that\n * HAPPENED with an unreported token count. `tokens` then holds the known subtotal (often\n * zero), never the measured total. An absent flag means the record is complete.\n */\n tokensKnown: boolean\n usd: number\n /**\n * False only when the record carried `usdKnown: false` — work that HAPPENED at a price the\n * provider never reported. `usd` then holds a catalog estimate or zero, never a measured\n * price. An absent flag means the record is complete.\n */\n usdKnown: boolean\n}\n\nexport function asRecord(v: unknown): Record<string, unknown> {\n return typeof v === 'object' && v !== null ? (v as Record<string, unknown>) : {}\n}\n\nfunction num(v: unknown): number {\n return typeof v === 'number' && Number.isFinite(v) ? v : 0\n}\n\nfunction readSpend(v: unknown): SpendLike {\n const rec = asRecord(v)\n const tok = asRecord(rec.tokens)\n const cacheRead = tok.cacheRead ?? tok.cache_read\n const cacheWrite = tok.cacheWrite ?? tok.cache_write\n return {\n tokens: {\n input: num(tok.input),\n output: num(tok.output),\n cacheRead: num(cacheRead),\n cacheWrite: num(cacheWrite),\n hasCache: cacheRead !== undefined || cacheWrite !== undefined,\n breakdownKnown: tok.cacheBreakdownKnown !== false,\n },\n tokensKnown: rec.tokensKnown !== false,\n usd: num(rec.usd),\n usdKnown: rec.usdKnown !== false,\n }\n}\n\nexport function parseJsonl(text: string | null): Record<string, unknown>[] {\n return parseJsonlWithDiagnostics(text).rows\n}\n\ninterface ParsedJsonl {\n readonly rows: Record<string, unknown>[]\n readonly invalidRows: number\n}\n\nfunction parseJsonlWithDiagnostics(text: string | null): ParsedJsonl {\n if (text === null) return { rows: [], invalidRows: 0 }\n const out: Record<string, unknown>[] = []\n let invalidRows = 0\n for (const line of text.split('\\n')) {\n const trimmed = line.trim()\n if (!trimmed) continue\n try {\n const parsed: unknown = JSON.parse(trimmed)\n if (typeof parsed === 'object' && parsed !== null && !Array.isArray(parsed)) {\n out.push(parsed as Record<string, unknown>)\n } else {\n invalidRows += 1\n }\n } catch {\n invalidRows += 1\n }\n }\n return { rows: out, invalidRows }\n}\n\nexport function parseJson(text: string | null): Record<string, unknown> | null {\n if (text === null) return null\n try {\n const parsed: unknown = JSON.parse(text)\n return typeof parsed === 'object' && parsed !== null\n ? (parsed as Record<string, unknown>)\n : null\n } catch {\n return null\n }\n}\n\nfunction ms(at: unknown): number | null {\n if (typeof at !== 'string') return null\n const t = Date.parse(at)\n return Number.isFinite(t) ? t : null\n}\n\n// ---------------------------------------------------------------------------\n// The supervision tree, as parsed from the journal event stream.\n// ---------------------------------------------------------------------------\n\nexport interface SpawnRow {\n /** Zero-based row among parsed journal objects. */\n sourceRow: number\n id: string\n parent: string | null\n label: string\n /** Null when the source did not explicitly record a role. */\n role: SupervisorRunNodeRole | null\n /** Exact canonical AgentProfile digest, when the source recorded one. */\n profileDigest: string | null\n /** Exact execution runtime tag, when the source recorded one. */\n runtime: string | null\n at: number | null\n /** False when identity, parentage, label, or role was malformed. */\n valid: boolean\n invalidFields: readonly string[]\n}\n\nexport interface CloseRow {\n id: string\n kind: 'settled' | 'cancelled'\n status: string | null\n /** String verdict from legacy journals, or valid/invalid for a structured verdict. */\n verdict: string | null\n /** Structured verdict validity, when recorded. */\n valid: boolean | null\n /** Structured verdict score, preserved without boolean coercion. */\n score: number | null\n /** The verdict exactly as the journal carried it. */\n rawVerdict: unknown | null\n /** Terminal failure detail exactly as recorded. */\n reason: string | null\n /** Whether Runtime classified the failure as infrastructure-caused. */\n infra: boolean | null\n at: number | null\n spend: SpendLike\n /** False when the close event carried no spend object — not \"spent nothing\". */\n hasSpend: boolean\n}\n\nexport interface WorkerLogFacts {\n /** Position in `SupervisorRunSources.workers`, retained for exact evidence paths. */\n sourceIndex: number\n workerId: string | null\n label: string\n /** `false` means the artifact was missing; a captured empty artifact is `true`. */\n inboxCaptured: boolean\n eventsCaptured: boolean\n inboxInvalidRows: number\n eventsInvalidRows: number\n started: number | null\n /** True once a `finished` event was seen — independent of whether its `at` parsed. */\n finished: boolean\n finishedAt: number | null\n passed: boolean | null\n /** Numeric score exactly as the finished event recorded it. */\n score: number | null\n /** `patchBytes` as reported by the finished event (not the patch file's size). */\n finishedPatchBytes: number | null\n evidenceBytes: number\n /** Unique queued request ids, or null when exact accounting is unavailable. */\n steersQueued: number | null\n /** Unique matched request ids acknowledged as delivered, or null when unavailable. */\n steersDelivered: number | null\n steersQueuedUnavailable: string | null\n steersDeliveredUnavailable: string | null\n questions: number\n steerRequests: SteerRequestFact[]\n steerAcknowledgements: SteerAcknowledgementFact[]\n}\n\nexport interface SteerRequestFact {\n /** Durable inbox request id; null when the row did not retain one. */\n requestId: string | null\n row: number\n}\n\nexport interface SteerAcknowledgementFact {\n /** Worker event request id; null when the event cannot be correlated. */\n requestId: string | null\n /** `null` means the acknowledgement omitted delivery status. */\n delivered: boolean | null\n row: number\n}\n\n/**\n * The tree + timeline the report is computed from, exposed because the rollout-row\n * minter needs exactly the same parse (one parser, two consumers).\n */\nexport interface SupervisorTreeFacts {\n readonly rootId: string | null\n readonly spawns: readonly SpawnRow[]\n readonly closes: readonly CloseRow[]\n readonly workerSpawns: readonly SpawnRow[]\n readonly workerCloses: readonly CloseRow[]\n readonly brain: {\n /**\n * Summed over the root's metered rows that reported a token count. A row Runtime marked\n * `tokensKnown: false` is EXCLUDED and counted in `tokensUnknownCount` instead, because\n * its `tokens` field holds a known subtotal — folding it in would price unreported work\n * at zero tokens.\n */\n tokensIn: number\n tokensOut: number\n cacheRead: number\n cacheWrite: number\n /** False when no metered event carried cache counters — not \"nothing cached\". */\n hasCache: boolean\n /** False when any folded metered row declared `cacheBreakdownKnown: false`. */\n cacheBreakdownKnown: boolean\n /** Root metered rows that reported a token count, and rows that did not. */\n tokensKnownCount: number\n tokensUnknownCount: number\n /** Summed over the root's metered rows that reported a price, on the same rule. */\n usd: number\n /** Root metered rows that reported a price, and rows Runtime marked `usdKnown: false`. */\n usdKnownCount: number\n usdUnknownCount: number\n /** Root-attributed metered events only — every row this object folded or counted. */\n meteredCount: number\n }\n readonly workerLogs: ReadonlyMap<string, WorkerLogFacts>\n /** Every worker source row, including duplicate identities that a map cannot retain. */\n readonly workerLogRows: readonly WorkerLogFacts[]\n /** Every non-empty journal line, whatever shape it turned out to have. */\n readonly journalRows: number\n /**\n * Journal lines this parser could not interpret at all: invalid JSON, non-objects, and\n * JSON objects carrying no record shape it recognizes.\n *\n * This count is what separates \"the run was empty\" from \"the input was unreadable\".\n * Zero spawns with zero invalid rows is a positive claim that nothing happened; it must\n * only ever be made about bytes the parser actually understood.\n */\n readonly journalInvalidRows: number\n /** The `journalInvalidRows` subset that was not valid JSON, or not a JSON object. */\n readonly journalMalformedJsonRows: number\n /**\n * Records recognized but deliberately not folded into the tree, counted by kind — the\n * runtime envelope's `begin` header and the `waiting` / `woken` wait-node events. Kept\n * separate from `journalInvalidRows` because these rows were understood, not rejected.\n */\n readonly journalIgnoredRowsByKind: Readonly<Record<string, number>>\n /**\n * Every metered row read, root-attributed or not. `brain.meteredCount` covers only the\n * root's own spend; a nested driver's metered rows are understood but reach this tree\n * through that driver's settled close, so this total is what keeps row accounting exact:\n * journalRows = spawns + closes + journalMeteredRows + ignored + journalInvalidRows.\n */\n readonly journalMeteredRows: number\n /** Which dialect(s) the journal's rows were actually written in. */\n readonly journalDialect: SupervisorJournalDialect\n /** Parsed once with the journal and worker artifacts. */\n readonly state: Record<string, unknown> | null\n readonly startedAt: number | null\n readonly completedAt: number | null\n /**\n * First and last stamped tree events (spawned / settled / cancelled /\n * metered) in the journal — the bounds of observed activity. They derive a\n * lower-bound wall for a store that writes no completion stamp.\n */\n readonly firstEventAt: number | null\n readonly lastEventAt: number | null\n}\n\ninterface VerdictFacts {\n label: string | null\n valid: boolean | null\n score: number | null\n raw: unknown | null\n}\n\nfunction readVerdict(v: unknown): VerdictFacts {\n if (typeof v === 'string') return { label: v, valid: null, score: null, raw: v }\n if (typeof v !== 'object' || v === null) {\n return { label: null, valid: null, score: null, raw: null }\n }\n const rec = v as Record<string, unknown>\n const valid = typeof rec.valid === 'boolean' ? rec.valid : null\n return {\n label: valid === null ? null : valid ? 'valid' : 'invalid',\n valid,\n score: typeof rec.score === 'number' && Number.isFinite(rec.score) ? rec.score : null,\n raw: v,\n }\n}\n\nexport function workerSourceKey(\n worker: NonNullable<SupervisorRunSources['workers']>[number],\n): string {\n return worker.workerId ?? worker.label\n}\n\n// ---------------------------------------------------------------------------\n// Journal dialects — the two shapes a supervision journal is actually written in.\n// ---------------------------------------------------------------------------\n\n/**\n * How a journal's rows were shaped on disk.\n *\n * `flat` — one event object per line (`{kind:'spawned', id, ...}`). The loops\n * supervisor writes this, and `readClaudeCodeSupervisorRun` synthesizes it.\n *\n * `runtime-envelope` — the records `agent-runtime`'s `FileSpawnJournal` writes:\n * a `{kind:'begin', root, at}` header followed by `{kind:'event', root, event}`\n * envelopes (`agent-runtime/src/durable/spawn-journal.ts`, `appendRecord`). It is\n * the layout `createFileRunContext(dir)` produces at `<dir>/spawn-journal.jsonl`.\n * This dialect is READ, not tolerated: agent-runtime is the writer, so a reader\n * that could not interpret it was the defect, and `journalDialect` records which\n * shape the bytes actually had so an unexpected one is never invisible.\n *\n * `mixed` — rows of both shapes in one file. `none` — no interpretable row at all.\n */\nexport type SupervisorJournalDialect = 'none' | 'flat' | 'runtime-envelope' | 'mixed'\n\ntype JournalRowDialect = 'flat' | 'runtime-envelope'\n\n/** Event kinds this parser folds into the tree. */\nconst TREE_EVENT_KINDS = ['spawned', 'settled', 'cancelled', 'metered'] as const\ntype TreeEventKind = (typeof TREE_EVENT_KINDS)[number]\n\n/**\n * Event kinds `agent-runtime` journals that this parser recognizes but does not model as\n * tree nodes: `waiting` arms a wait node and `woken` settles it. They are counted by kind\n * rather than dropped, so \"understood, not modeled\" can never be mistaken for \"absent\".\n */\nconst UNMODELLED_EVENT_KINDS: readonly string[] = ['waiting', 'woken']\n\nfunction treeEventKind(value: unknown): TreeEventKind | null {\n return (TREE_EVENT_KINDS as readonly unknown[]).includes(value) ? (value as TreeEventKind) : null\n}\n\ntype JournalRowReading =\n | {\n outcome: 'event'\n dialect: JournalRowDialect\n kind: TreeEventKind\n event: JournalEvent\n }\n | { outcome: 'ignored'; dialect: JournalRowDialect; kind: string }\n | { outcome: 'unreadable' }\n\n/**\n * Classify one parsed journal row. Every row lands in exactly one outcome — interpreted,\n * recognized-but-not-modeled, or unreadable — because a row that falls through silently is\n * how a run with events reads as a run with none.\n */\nfunction readJournalRow(row: Record<string, unknown>): JournalRowReading {\n const kind = row.kind\n if (typeof kind !== 'string') return { outcome: 'unreadable' }\n\n // agent-runtime's FileSpawnJournal record header. It carries no event.\n if (kind === 'begin' && typeof row.root === 'string') {\n return { outcome: 'ignored', dialect: 'runtime-envelope', kind: 'begin' }\n }\n if (kind === 'event') {\n const inner = row.event\n if (typeof inner !== 'object' || inner === null || Array.isArray(inner)) {\n return { outcome: 'unreadable' }\n }\n const innerKind = (inner as Record<string, unknown>).kind\n if (typeof innerKind !== 'string') return { outcome: 'unreadable' }\n const tree = treeEventKind(innerKind)\n if (tree !== null) {\n return {\n outcome: 'event',\n dialect: 'runtime-envelope',\n kind: tree,\n event: inner as JournalEvent,\n }\n }\n // The envelope proves this IS a journaled event, so an unmodelled kind is recorded by\n // name rather than called unreadable — the caller sees exactly what went uncounted.\n return { outcome: 'ignored', dialect: 'runtime-envelope', kind: innerKind }\n }\n\n const tree = treeEventKind(kind)\n if (tree !== null) {\n return { outcome: 'event', dialect: 'flat', kind: tree, event: row as JournalEvent }\n }\n if (UNMODELLED_EVENT_KINDS.includes(kind)) {\n return { outcome: 'ignored', dialect: 'flat', kind }\n }\n // A bare object with an unknown `kind` could be any record at all — unlike the envelope\n // case there is no evidence it is a journal event, so it counts as unreadable (fail closed).\n return { outcome: 'unreadable' }\n}\n\nfunction summarizeDialect(seen: ReadonlySet<JournalRowDialect>): SupervisorJournalDialect {\n if (seen.size === 0) return 'none'\n if (seen.size > 1) return 'mixed'\n return seen.has('runtime-envelope') ? 'runtime-envelope' : 'flat'\n}\n\nexport function parseSupervisorTree(src: SupervisorRunSources): SupervisorTreeFacts {\n const parsedJournal = parseJsonlWithDiagnostics(src.journal)\n const events = parsedJournal.rows\n const state = parseJson(src.state)\n\n const spawns: SpawnRow[] = []\n const closes: CloseRow[] = []\n let brainIn = 0\n let brainOut = 0\n let brainCacheRead = 0\n let brainCacheWrite = 0\n let brainHasCache = false\n let brainCacheBreakdownKnown = true\n let brainTokensKnownCount = 0\n let brainTokensUnknownCount = 0\n let brainUsd = 0\n let brainUsdKnownCount = 0\n let brainUsdUnknownCount = 0\n let meteredCount = 0\n let rootId: string | null = null\n let unreadableRows = 0\n const ignoredByKind = new Map<string, number>()\n const dialectsSeen = new Set<JournalRowDialect>()\n const meteredRows: Array<{ id: string; spend: SpendLike; at: number | null }> = []\n for (const [sourceRow, row] of events.entries()) {\n const reading = readJournalRow(row)\n if (reading.outcome === 'unreadable') {\n unreadableRows += 1\n continue\n }\n dialectsSeen.add(reading.dialect)\n if (reading.outcome === 'ignored') {\n ignoredByKind.set(reading.kind, (ignoredByKind.get(reading.kind) ?? 0) + 1)\n continue\n }\n const { kind, event: ev } = reading\n const id = typeof ev.id === 'string' ? ev.id : ''\n if (kind === 'spawned') {\n const parent = typeof ev.parent === 'string' && ev.parent.length > 0 ? ev.parent : null\n const label = typeof ev.label === 'string' ? ev.label : ''\n const invalidFields: string[] = []\n if (id.length === 0) invalidFields.push('id')\n const parentOmittedForFirstRoot = ev.parent === undefined && rootId === null\n if (\n ev.parent !== null &&\n !parentOmittedForFirstRoot &&\n !(typeof ev.parent === 'string' && ev.parent.length > 0)\n ) {\n invalidFields.push('parent')\n }\n if (label.length === 0) invalidFields.push('label')\n if (ev.role !== undefined && ev.role !== 'supervisor' && ev.role !== 'worker') {\n invalidFields.push('role')\n }\n if (\n ev.profileDigest !== undefined &&\n !(typeof ev.profileDigest === 'string' && ev.profileDigest.length > 0)\n ) {\n invalidFields.push('profileDigest')\n }\n if (ev.runtime !== undefined && !(typeof ev.runtime === 'string' && ev.runtime.length > 0)) {\n invalidFields.push('runtime')\n }\n const role: SupervisorRunNodeRole | null =\n ev.role === 'supervisor' || ev.role === 'worker' ? ev.role : null\n const valid = invalidFields.length === 0\n if (valid && parent === null && rootId === null) rootId = id\n spawns.push({\n sourceRow,\n id,\n parent,\n label,\n role,\n profileDigest:\n typeof ev.profileDigest === 'string' && ev.profileDigest.length > 0\n ? ev.profileDigest\n : null,\n runtime: typeof ev.runtime === 'string' && ev.runtime.length > 0 ? ev.runtime : null,\n at: ms(ev.at),\n valid,\n invalidFields,\n })\n } else if (kind === 'settled') {\n const verdict = readVerdict(ev.verdict)\n closes.push({\n id,\n kind: 'settled',\n status: typeof ev.status === 'string' ? ev.status : null,\n verdict: verdict.label,\n valid: verdict.valid,\n score: verdict.score,\n rawVerdict: verdict.raw,\n reason: typeof ev.reason === 'string' ? ev.reason : null,\n infra: typeof ev.infra === 'boolean' ? ev.infra : null,\n at: ms(ev.at),\n spend: readSpend(ev.spent),\n hasSpend: asRecord(ev.spent).tokens !== undefined,\n })\n } else if (kind === 'cancelled') {\n closes.push({\n id,\n kind: 'cancelled',\n status: 'cancelled',\n verdict: typeof ev.reason === 'string' ? ev.reason : null,\n valid: null,\n score: null,\n rawVerdict: typeof ev.reason === 'string' ? ev.reason : null,\n reason: typeof ev.reason === 'string' ? ev.reason : null,\n infra: null,\n at: ms(ev.at),\n spend: {\n tokens: {\n input: 0,\n output: 0,\n cacheRead: 0,\n cacheWrite: 0,\n hasCache: false,\n breakdownKnown: true,\n },\n tokensKnown: true,\n usd: 0,\n usdKnown: true,\n },\n hasSpend: false,\n })\n } else if (kind === 'metered') {\n // Deferred: rows are attributed after the loop, once the root is known, so a metered\n // event that precedes the root spawn is never misattributed.\n meteredRows.push({ id, spend: readSpend(ev.spend), at: ms(ev.at) })\n } else {\n // `TREE_EVENT_KINDS` and these branches are one contract: adding a kind to the set\n // without a branch here would silently drop its rows, which is the defect this\n // accounting exists to prevent. The compiler refuses the omission.\n const unhandled: never = kind\n throw new Error(`unhandled journal event kind ${JSON.stringify(unhandled)}`)\n }\n }\n\n // Only root-attributed metered events are the supervisor brain's own spend. A nested\n // driver meters under its own id; its spend reaches this tree through its settled close,\n // so folding it into the brain here would count the same tokens twice.\n for (const row of meteredRows) {\n if (rootId === null || row.id !== rootId) continue\n if (row.spend.tokensKnown) {\n brainIn += row.spend.tokens.input\n brainOut += row.spend.tokens.output\n brainCacheRead += row.spend.tokens.cacheRead\n brainCacheWrite += row.spend.tokens.cacheWrite\n brainHasCache = brainHasCache || row.spend.tokens.hasCache\n brainCacheBreakdownKnown = brainCacheBreakdownKnown && row.spend.tokens.breakdownKnown\n brainTokensKnownCount += 1\n } else {\n brainTokensUnknownCount += 1\n }\n if (row.spend.usdKnown) {\n brainUsd += row.spend.usd\n brainUsdKnownCount += 1\n } else {\n brainUsdUnknownCount += 1\n }\n meteredCount += 1\n }\n\n const workerSpawns = spawns.filter((s) => s.valid && s.id !== rootId)\n const workerIds = new Set(workerSpawns.map((s) => s.id))\n const workerCloses = closes.filter((c) => workerIds.has(c.id))\n\n const workerLogs = new Map<string, WorkerLogFacts>()\n const workerLogRows: WorkerLogFacts[] = []\n for (const [sourceIndex, w] of (src.workers ?? []).entries()) {\n const facts: WorkerLogFacts = {\n sourceIndex,\n workerId: w.workerId ?? null,\n label: w.label,\n inboxCaptured: w.inbox !== null,\n eventsCaptured: w.events !== null,\n inboxInvalidRows: 0,\n eventsInvalidRows: 0,\n started: null,\n finished: false,\n finishedAt: null,\n passed: null,\n score: null,\n finishedPatchBytes: null,\n evidenceBytes: 0,\n steersQueued: null,\n steersDelivered: null,\n steersQueuedUnavailable: null,\n steersDeliveredUnavailable: null,\n questions: 0,\n steerRequests: [],\n steerAcknowledgements: [],\n }\n const parsedInbox = parseJsonlWithDiagnostics(w.inbox)\n const parsedEvents = parseJsonlWithDiagnostics(w.events)\n facts.inboxInvalidRows = parsedInbox.invalidRows\n facts.eventsInvalidRows = parsedEvents.invalidRows\n for (const [row, req] of parsedInbox.rows.entries()) {\n if (typeof req.message !== 'string' || req.message.trim().length === 0) continue\n facts.steerRequests.push({\n requestId: typeof req.id === 'string' && req.id.length > 0 ? req.id : null,\n row,\n })\n }\n for (const [row, ev] of parsedEvents.rows.entries()) {\n const kind = ev.kind\n if (kind === 'message') {\n if (ev.direction === 'up') {\n facts.questions += 1\n continue\n }\n facts.steerAcknowledgements.push({\n requestId:\n typeof ev.requestId === 'string' && ev.requestId.length > 0 ? ev.requestId : null,\n delivered: typeof ev.delivered === 'boolean' ? ev.delivered : null,\n row,\n })\n } else if (kind === 'started') {\n facts.started = ms(ev.at)\n } else if (kind === 'finished') {\n facts.finished = true\n facts.finishedAt = ms(ev.at)\n facts.passed = typeof ev.passed === 'boolean' ? ev.passed : null\n facts.score = typeof ev.score === 'number' && Number.isFinite(ev.score) ? ev.score : null\n facts.finishedPatchBytes = typeof ev.patchBytes === 'number' ? ev.patchBytes : null\n facts.evidenceBytes = typeof ev.evidence === 'string' ? ev.evidence.length : 0\n }\n }\n const requestsById = new Map<string, number>()\n for (const request of facts.steerRequests) {\n if (request.requestId === null) continue\n requestsById.set(request.requestId, (requestsById.get(request.requestId) ?? 0) + 1)\n }\n const acknowledgementsById = new Map<string, SteerAcknowledgementFact[]>()\n for (const acknowledgement of facts.steerAcknowledgements) {\n if (acknowledgement.requestId === null) continue\n const matches = acknowledgementsById.get(acknowledgement.requestId) ?? []\n matches.push(acknowledgement)\n acknowledgementsById.set(acknowledgement.requestId, matches)\n }\n const queueProblems: string[] = []\n if (!facts.inboxCaptured) queueProblems.push('inbox absent')\n if (!facts.eventsCaptured) queueProblems.push('events absent')\n if (facts.inboxInvalidRows > 0) queueProblems.push('inbox contains malformed rows')\n if (facts.eventsInvalidRows > 0) queueProblems.push('events contain malformed rows')\n if (facts.steerRequests.some((request) => request.requestId === null)) {\n queueProblems.push('queued request id missing')\n }\n if ([...requestsById.values()].some((count) => count > 1)) {\n queueProblems.push('queued request id duplicated')\n }\n if (facts.steerAcknowledgements.some((acknowledgement) => acknowledgement.requestId === null)) {\n queueProblems.push('acknowledgement request id missing')\n }\n if ([...acknowledgementsById.keys()].some((id) => !requestsById.has(id))) {\n queueProblems.push('acknowledgement has no queued request')\n }\n if (queueProblems.length === 0) {\n facts.steersQueued = requestsById.size\n } else {\n facts.steersQueuedUnavailable = [...new Set(queueProblems)].join('; ')\n }\n\n const deliveryProblems = [...queueProblems]\n if ([...acknowledgementsById.values()].some((rows) => rows.length > 1)) {\n deliveryProblems.push('acknowledgement request id duplicated')\n }\n if (facts.steerAcknowledgements.some((acknowledgement) => acknowledgement.delivered === null)) {\n deliveryProblems.push('acknowledgement delivery status missing')\n }\n if (deliveryProblems.length === 0) {\n facts.steersDelivered = [...requestsById.keys()].filter((id) => {\n const acknowledgement = acknowledgementsById.get(id)\n return acknowledgement?.length === 1 && acknowledgement[0]?.delivered === true\n }).length\n } else {\n facts.steersDeliveredUnavailable = [...new Set(deliveryProblems)].join('; ')\n }\n workerLogRows.push(facts)\n workerLogs.set(workerSourceKey(w), facts)\n }\n\n const startedAt = ms(state?.startedAt) ?? spawns[0]?.at ?? null\n const rootClose = rootId === null ? null : closes.find((close) => close.id === rootId)\n const completedAt = ms(state?.completedAt) ?? rootClose?.at ?? null\n\n let firstEventAt: number | null = null\n let lastEventAt: number | null = null\n const widenEventSpan = (at: number | null): void => {\n if (at === null) return\n if (firstEventAt === null || at < firstEventAt) firstEventAt = at\n if (lastEventAt === null || at > lastEventAt) lastEventAt = at\n }\n for (const spawn of spawns) widenEventSpan(spawn.at)\n for (const close of closes) widenEventSpan(close.at)\n for (const metered of meteredRows) widenEventSpan(metered.at)\n\n return {\n rootId,\n spawns,\n closes,\n workerSpawns,\n workerCloses,\n brain: {\n tokensIn: brainIn,\n tokensOut: brainOut,\n cacheRead: brainCacheRead,\n cacheWrite: brainCacheWrite,\n hasCache: brainHasCache,\n cacheBreakdownKnown: brainCacheBreakdownKnown,\n tokensKnownCount: brainTokensKnownCount,\n tokensUnknownCount: brainTokensUnknownCount,\n usd: brainUsd,\n usdKnownCount: brainUsdKnownCount,\n usdUnknownCount: brainUsdUnknownCount,\n meteredCount,\n },\n workerLogs,\n workerLogRows,\n journalRows: events.length + parsedJournal.invalidRows,\n journalInvalidRows: parsedJournal.invalidRows + unreadableRows,\n journalMalformedJsonRows: parsedJournal.invalidRows,\n journalIgnoredRowsByKind: Object.fromEntries(ignoredByKind),\n journalMeteredRows: meteredRows.length,\n journalDialect: summarizeDialect(dialectsSeen),\n state,\n startedAt,\n completedAt,\n firstEventAt,\n lastEventAt,\n }\n}\n","/**\n * Supervisor-run analysis — the multi-agent analogue of single-rollout trace\n * analysis. A solo rollout is one invocation with a transcript; a supervisor\n * run is a TREE of invocations (a brain that spawns, steers, and settles\n * workers) plus the event timeline that connects them. `src/trace-analyst`\n * answers \"what happened inside one session\"; this module answers \"what did\n * the tree do\" — did the brain steer anyone mid-task, how many spawn waves,\n * how concurrent, how idle, what did each role cost, what came back.\n *\n * The nodes of that tree are NOT a new shape: they are `tangle.rollout.v1`\n * rows (`src/rollout`), keyed by `parent_rollout_id`, with `role` already\n * spanning `supervisor` / `worker`. `supervisorRunRolloutLines` mints them.\n * What rollout rows deliberately do NOT carry is the inter-invocation event\n * timeline (spawn/settle/steer instants), which is what every structural\n * metric here is computed from — so the reader consumes the journal event\n * stream and emits rollout rows, rather than maintaining a parallel node type.\n *\n * ## UNAVAILABLE ≠ ZERO\n *\n * Every metric whose backing artifact can be missing is typed\n * `Measured<T> = T | { unavailable: reason }`. A supervisor that steered\n * nobody reports `steers: 0`; a supervisor whose worker logs were never\n * written reports `steers: unavailable — <reason>`. The two have driven\n * opposite conclusions about the same architecture, so they never collapse.\n */\n\nimport type { RolloutLine } from '../rollout/schema'\nimport type { SeriesDistribution } from '../statistics'\n\n// ---------------------------------------------------------------------------\n// Unavailable-aware metric type.\n// ---------------------------------------------------------------------------\n\n/** A metric that could not be computed, with the reason its artifact was missing. */\nexport interface Unavailable {\n readonly unavailable: string\n}\n\n/** A metric value, or the reason it is unknown. NEVER collapse `unavailable` to 0. */\nexport type Measured<T> = T | Unavailable\n\nexport function unavailable(reason: string): Unavailable {\n return { unavailable: reason }\n}\n\nexport function isUnavailable(v: unknown): v is Unavailable {\n return typeof v === 'object' && v !== null && typeof (v as Unavailable).unavailable === 'string'\n}\n\n/** Render a measured scalar for the markdown/headline: `0` and `unavailable` stay distinct. */\nexport function showMeasured(v: Measured<number | string | boolean | null>): string {\n if (isUnavailable(v)) return `unavailable — ${v.unavailable}`\n if (v === null) return 'null'\n return String(v)\n}\n\n// ---------------------------------------------------------------------------\n// Source contract — deliberately source-agnostic.\n// ---------------------------------------------------------------------------\n\n/** The two invocation roles a recursive supervision tree can contain. */\nexport type SupervisorRunNodeRole = 'supervisor' | 'worker'\n\n/**\n * One worker's logs, as read. `null` means the artifact did not exist; `''`\n * means the artifact was captured and contained no rows.\n */\nexport interface WorkerLogSource {\n /**\n * Stable journal node id. Readers should set this whenever their source has\n * one; `label` remains the compatibility join for older stores.\n */\n readonly workerId?: string\n /** Human-readable task label. It is not required to be unique. */\n readonly label: string\n /** Worker event stream — started / progress / finished / message events (JSONL). */\n readonly events: string | null\n /** The durable steer queue — one line per steer request (JSONL). */\n readonly inbox: string | null\n /** Worker patch byte length, or null when absent. */\n readonly patchBytes: number | null\n /** Where this worker's transcript lives, for the rollout row. Null = no such artifact. */\n readonly transcriptRef?: string | null\n /** Where this worker's delivered patch lives. Null = the store keeps no patch per worker. */\n readonly patchPath?: string | null\n /** This worker's own inference tokens, when the store records them per worker. */\n readonly tokensIn?: number | null\n readonly tokensOut?: number | null\n readonly cacheRead?: number | null\n readonly cacheWrite?: number | null\n}\n\n/**\n * Facts a SOURCE structurally cannot express, each with the reason.\n *\n * The difference between \"the artifact is missing\" and \"this store never\n * records that fact\" is the difference between a run that spent $0 and a\n * harness that does not price inference — and the second harness is where a\n * loops-shaped assumption becomes a fabricated zero. A reader declares its\n * limits once; the analyzer reports `unavailable` for everything downstream.\n *\n * `null` on a field means the source DOES carry that fact.\n */\nexport interface SourceLimits {\n /** Reason manager input/output token totals are unavailable (null = recorded). */\n readonly managerTokens: string | null\n /** Reason worker input/output token totals are unavailable (null = recorded). */\n readonly workerTokens: string | null\n /** Reason inference spend has no price in this store (null = the store prices it). */\n readonly spendUsd: string | null\n /** Reason workers carry no pass/fail verdict (null = verdicts are recorded). */\n readonly workerVerdicts: string | null\n /** Reason no delivered artifact (patch/diff) is retained per worker (null = retained). */\n readonly deliverables: string | null\n}\n\n/** A source that carries every fact the analyzer can use. */\nexport const NO_SOURCE_LIMITS: SourceLimits = {\n managerTokens: null,\n workerTokens: null,\n spendUsd: null,\n workerVerdicts: null,\n deliverables: null,\n}\n\n/**\n * Everything the pure analyzer reads — already-read bytes, never paths. Each\n * field is `null` when its artifact was absent, which is what turns the\n * dependent metrics into `unavailable` rather than 0.\n *\n * This is the whole input contract. Any store that can produce these strings\n * (an on-disk loops run, an object-store archive, a database, a test fixture)\n * is a valid source; `loopsSupervisorRunReader` is ONE implementation.\n */\nexport interface SupervisorRunSources {\n /** Stable identity of the run being analyzed (a directory, a run id, a URL). */\n readonly runRef: string\n readonly instanceId: string | null\n /** Which arm/variant of a comparison this run is, when the run belongs to one. */\n readonly arm: string | null\n /** Identity of the supervision-tree store this was read from; null = none found. */\n readonly supRunDir: string | null\n /**\n * Supervision journal — spawned / settled / cancelled / metered events (JSONL).\n * Recursive readers put `role: 'supervisor' | 'worker'` on spawned rows;\n * settled `verdict` may be a legacy string or `{ valid, score, ... }`.\n */\n readonly journal: string | null\n /**\n * Source-specific reason `journal` is null. The analyzer uses it verbatim as\n * the `unavailable` reason on every journal-dependent metric, so a non-loops\n * layout names its own journal file instead of inheriting the loops paths.\n */\n readonly journalMissingReason?: string\n /** Per-brain-call tap (JSONL): finish_reason, completion tokens, requested max tokens. */\n readonly brainLog: string | null\n /** Source-specific reason `brainLog` is absent. */\n readonly brainLogMissingReason?: string\n /** Supervisor state document (JSON). */\n readonly state: string | null\n /** Supervisor progress stream (JSONL). */\n readonly progress: string | null\n /** Per-worker logs; `null` = the worker log store itself was missing. */\n readonly workers: readonly WorkerLogSource[] | null\n /** Why `workers` is null (only set when it is). */\n readonly workersMissingReason: string | null\n /**\n * Run result document (JSON). For a Runtime run dir this is `result.json`,\n * the `SupervisedResult` that `supervise()` returned, verbatim; its `kind`\n * is the run's status. For a loops run dir it is the legacy result document.\n */\n readonly result: string | null\n /**\n * Runtime's terminal failure record (`failure.json`), written when\n * `supervise()` threw before a result landed. `null` = the store was read and\n * holds no such record; `undefined` = the store has no failure document at\n * all (the loops layout), which the analyzer reports as its own absence.\n */\n readonly failure?: string | null\n /**\n * Judge verdict document (JSON), or the matching ledger row re-encoded as\n * one. Runners that write the verdict straight to a ledger leave no judge\n * document, so the ledger row is the same fact from the same run — not a\n * substitute measurement.\n */\n readonly judge: string | null\n /** Where `judge` came from, for the report's provenance line. */\n readonly judgeSource: string | null\n /** Delivered unified-diff patch text. */\n readonly patch: string | null\n /** Outer-driver log (used for the driver's steer verbs + deadline evidence). */\n readonly driverLog: string | null\n /**\n * Worker tokens recovered from a harness session store; null = store unavailable.\n * `store` names the store in the report's provenance line (e.g. `opencode`).\n */\n readonly harnessWorkerTokens: {\n store: string\n sessions: number\n input: number\n output: number\n /** Cached prompt tokens, when the store counts them separately. */\n cacheRead?: number\n cacheWrite?: number\n } | null\n readonly harnessMissingReason: string | null\n /** What this store structurally cannot record. See `SourceLimits`. */\n readonly limits: SourceLimits\n /**\n * Where the ROOT invocation's transcript lives. Undefined lets the rollout\n * minter fall back to the loops layout (`<supRunDir>/journal.jsonl`); any\n * other store must say, or the row points at a path that never existed.\n */\n readonly rootTranscriptRef?: string | null\n /**\n * The `traces` CLI command that covers this run's harness-session layer.\n * Null falls back to the analyzer's default (an opencode worker fleet).\n */\n readonly traceCommand: string | null\n}\n\n/**\n * A source of supervisor-run bytes. Implementations own their storage layout;\n * the analyzer only ever sees `SupervisorRunSources`.\n */\nexport interface SupervisorRunReader {\n /** Stable identity of what this reader points at (for logs and report labels). */\n readonly runRef: string\n read(): Promise<SupervisorRunSources>\n}\n\n// ---------------------------------------------------------------------------\n// Report shape.\n// ---------------------------------------------------------------------------\n\nexport const SUPERVISOR_RUN_SCHEMA = 'tangle.supervisor-run@1'\nexport const SUPERVISOR_RUN_ROLLUP_SCHEMA = 'tangle.supervisor-run-rollup@1'\n\nexport interface SteerBreakdown {\n /** Stable journal node id when the reader retained one. */\n readonly workerId: string | null\n readonly worker: string\n /** Steer requests durably queued to this worker's inbox. */\n readonly queued: number\n /** Steers the worker's executor actually accepted (control event `delivered:true`). */\n readonly delivered: number\n}\n\nexport interface OrchestrationMetrics {\n readonly workersSpawned: Measured<number>\n readonly workersSettled: Measured<number>\n readonly workersCancelled: Measured<number>\n /** THE HEADLINE: mid-task steers the brain sent to live workers. 0 ≠ unavailable. */\n readonly steers: Measured<number>\n readonly steersDelivered: Measured<number>\n readonly steersByWorker: Measured<readonly SteerBreakdown[]>\n /** Outer-driver `supervisor_steer` tool calls seen in the driver log (a second steer path). */\n readonly driverSteerCalls: Measured<number>\n /**\n * Spawn waves. A wave is a maximal run of worker spawns with no settle/cancel between\n * them: wave N+1 begins at the first spawn issued after at least one worker from an\n * earlier wave has settled. Structural, not a time threshold — no tunable constant.\n */\n readonly waves: Measured<number>\n readonly waveSizes: Measured<readonly number[]>\n readonly maxConcurrency: Measured<number>\n /** Direct-child spawns issued after that parent's first direct-child settlement. */\n readonly respawns: Measured<number>\n /** Labels spawned more than once by the same parent. */\n readonly repeatedLabels: Measured<readonly string[]>\n /** Longest parent chain below the root, in worker hops. */\n readonly delegationDepth: Measured<number>\n readonly timeToFirstSpawnMs: Measured<number>\n readonly supervisorWallMs: Measured<number>\n /**\n * Which measurement `supervisorWallMs` holds — never a silent substitution.\n * `stamps`: explicit start and completion stamps. `journal-span`: the start\n * stamp (or first stamped event) to the last stamped journal event — a\n * lower bound, derived when the store wrote no completion stamp. `idleMs`,\n * `idlePct`, and `workerUtilization` cover the same span. Unavailable\n * exactly when `supervisorWallMs` is, with the same reason.\n */\n readonly supervisorWallSource: Measured<'stamps' | 'journal-span'>\n /** Wall time inside the supervisor run with ZERO live workers. */\n readonly idleMs: Measured<number>\n readonly idlePct: Measured<number>\n /** sum(worker wall) / supervisor wall. >1 means real parallelism. */\n readonly workerUtilization: Measured<number>\n}\n\nexport interface DecisionMetrics {\n readonly settledByStatus: Measured<Record<string, number>>\n readonly settledVerdicts: Measured<Record<string, number>>\n /**\n * Workers whose recorded verdict was green. A store that retains a delivered patch also\n * requires patch bytes; a store that retains none accepts the verdict alone, because the\n * verdict IS the acceptance decision that store recorded. `emptyPass` — the split that\n * needs patch bytes — is what reads unavailable there.\n */\n readonly accepted: Measured<number>\n /** Worker settled with a failing verify. */\n readonly rejected: Measured<number>\n /** Worker verified green but delivered no patch bytes — output with nothing to accept. */\n readonly emptyPass: Measured<number>\n /** Direct-child settlements a parent observed before issuing its next direct-child spawn. */\n readonly observeThenRespawn: Measured<number>\n /** Parent-local respawns with no direct-child settlement in front of them. */\n readonly respawnWithoutEvidence: Measured<number>\n /** Steer + question traffic on the live down/up legs — the only \"review while running\" signal. */\n readonly reviewActions: Measured<number>\n readonly workerEvidenceBytes: Measured<number>\n}\n\nexport interface RoleSpend {\n readonly tokensIn: Measured<number>\n readonly tokensOut: Measured<number>\n /**\n * Cached prompt tokens read/written. On a harness that caches aggressively\n * these dwarf `tokensIn`, so a report that omits them understates the context\n * each invocation actually consumed. `unavailable` = the store has no such counter.\n */\n readonly cacheRead: Measured<number>\n readonly cacheWrite: Measured<number>\n readonly usd: Measured<number>\n readonly source: string\n}\n\nexport interface PerWorkerRow {\n /** Stable journal node id when the reader retained one. */\n readonly workerId: string | null\n readonly worker: string\n /** Explicit journal role after the worker source was joined to its spawn. */\n readonly role: SupervisorRunNodeRole | null\n /** Exact execution runtime tag from the spawn event. */\n readonly runtime: string | null\n /** Exact canonical AgentProfile digest from the spawn event. */\n readonly profileDigest: string | null\n /** Terminal lifecycle status exactly as recorded. */\n readonly status: string | null\n /** Terminal failure detail exactly as recorded. */\n readonly failure: string | null\n /** Runtime infrastructure classification, when recorded. */\n readonly infra: boolean | null\n readonly wallMs: number | null\n /** `null` = this store does not attribute tokens per worker (NOT \"zero tokens\"). */\n readonly tokensIn: number | null\n readonly tokensOut: number | null\n readonly usd: number | null\n readonly patchBytes: number | null\n readonly passed: boolean | null\n /** Numeric verdict score exactly as recorded; null means no score was recorded. */\n readonly score: number | null\n}\n\n/**\n * `SeriesDistribution` (from `../statistics`) over per-worker wall\n * milliseconds. The fold itself is `summarizeNumberSeries`, exported for any\n * series — fleet wall medians, tokens-per-claim spreads — not only wall.\n */\nexport type WallDistribution = SeriesDistribution\n\n/** One spend measurement and the number of source records behind it. */\nexport interface SpendMeasurement {\n readonly usd: Measured<number>\n /** Source records folded into `usd`; 0 when the measurement is unavailable. */\n readonly records: number\n /**\n * Records the store wrote with `usdKnown: false` — work that HAPPENED at a price the\n * provider never reported. Those records are NOT folded into `usd`, so a run with any\n * of them has a `usd` that is a floor on real spend, never the measured total. Dropping\n * the whole channel instead would discard the records that DID carry a price.\n */\n readonly unknownRecords: number\n /** True exactly when `unknownRecords > 0`: `usd` covers some of the run, not all of it. */\n readonly partial: boolean\n /** Node ids behind `unknownRecords`, in journal order. Empty when none. */\n readonly unknownNodes: readonly string[]\n}\n\n/**\n * The run's total inference spend, measured two ways.\n *\n * `closeRecord` is the spend the store recorded as settled when the run\n * closed (loops `state.json` `result.spentUsd`; Runtime `result.json`\n * `spentTotal.usd`) — the billing-shaped answer. `journalDerived` is the\n * spend execution observably consumed (journal `metered` + `settled` rows) —\n * the execution-accounting answer. Neither is canonical for the other's\n * question. The two cover different records at different moments, so\n * divergence between them is itself a signal (a dropped settlement, a\n * double meter, spend after the close) — read it, never average it away.\n */\nexport interface SpendMeasurements {\n readonly journalDerived: SpendMeasurement\n readonly closeRecord: SpendMeasurement\n}\n\nexport interface EconomicsMetrics {\n /** Driver/brain inference — journal `metered` events. */\n readonly brain: RoleSpend\n /**\n * Brain completions that came back `finish_reason: \"length\"` — output TRUNCATED. Any value\n * above 0 means the supervisor planned into a wall and then acted on the half-written plan,\n * which is a defect and not a cost figure. The journal's `metered` rows carry token counts\n * but no finish reason, so this reads the per-call brain tap; a run whose supervisor\n * predates that tap reports `unavailable`, never 0.\n */\n readonly brainTruncations: Measured<number>\n /** Worker inference — journal `settled` spend plus the harness session join. */\n readonly workers: RoleSpend\n /** Both total-spend measurements, each with its own record count. */\n readonly spend: SpendMeasurements\n /**\n * One collapsed number kept for existing consumers: the close record when\n * the store wrote one, else the journal-derived sum. `totalUsdSource` names\n * the pick, and says so when the number is a partial floor because some\n * records carried `usdKnown: false`. Prefer `spend` — the collapse hides\n * which accounting question the number answers and how much of it is priced.\n */\n readonly totalUsd: Measured<number>\n /**\n * Where `totalUsd` came from. CLI-backend workers never price their own inference into\n * the journal, so on those arms the total is BRAIN-ONLY and the worker row's token\n * counts (recovered from the harness store) are the honest worker-side figure.\n */\n readonly totalUsdSource: string\n readonly costPerAcceptedPatchUsd: Measured<number>\n readonly workerWallMsDistribution: Measured<WallDistribution>\n readonly perWorker: Measured<readonly PerWorkerRow[]>\n}\n\nexport interface PatchStats {\n readonly files: number\n readonly linesAdded: number\n readonly linesRemoved: number\n readonly testFilesTouched: readonly string[]\n}\n\n/** Which record a run's terminal status was read from, so no source is a silent substitution. */\nexport type SupervisorStatusSource =\n /** Runtime's `result.json` `kind`: the `SupervisedResult` discriminant, read verbatim. */\n | 'runtime-result'\n /** Runtime's `failure.json`: `supervise()` threw before a result landed. */\n | 'runtime-failure'\n /** Control-plane-era loops `state.json` `status`. */\n | 'legacy-state'\n /** Control-plane-era loops `result.json` `sup_status`. */\n | 'legacy-result'\n\n/** The error a run directory recorded. */\nexport interface TerminalFailure {\n /**\n * Which record carried the error: Runtime's `failure.json`, or the driver\n * rejection inside a `driver-failed` no-winner `result.json`.\n */\n readonly source: 'runtime-failure' | 'runtime-result'\n /** `error.name` exactly as recorded; null when the record has none. */\n readonly name: string | null\n /** `error.message` exactly as recorded; null when the record has none. */\n readonly message: string | null\n /** ISO timestamp the record carries; null when it carries none. */\n readonly at: string | null\n /**\n * True when the directory also holds a settled `result.json`. Runtime writes\n * `failure.json` only when `supervise()` threw, never after a settle, and\n * refuses to re-enter a settled directory, so a failure beside a result is\n * the throw of an earlier attempt that a later attempt outlived. The status\n * is the settled result's; this record explains the retry, not the outcome.\n */\n readonly earlierAttempt: boolean\n}\n\nexport interface OutcomeMetrics {\n /**\n * The run's terminal status, exactly as its record spells it: Runtime's\n * `winner` / `no-winner`, `failed` for a Runtime failure record, or the\n * legacy loops status. `supStatusSource` names the record it came from.\n */\n readonly supStatus: Measured<string>\n readonly supStatusSource: Measured<SupervisorStatusSource>\n /**\n * Runtime's `reason` on a `no-winner` result (`all-children-down`,\n * `budget-exhausted`, `aborted`, `driver-failed`). `null` = the terminal\n * record carries no reason (a winner, a failure record, a legacy document).\n */\n readonly supReason: Measured<string | null>\n /**\n * The recorded error: Runtime's `failure.json`, or the driver rejection a\n * `driver-failed` no-winner carries. `null` = the run recorded a result and\n * no error. A `failure.json` beside a settled result is reported with\n * `earlierAttempt: true`. Unavailable when the store has no terminal record.\n */\n readonly failure: Measured<TerminalFailure | null>\n readonly supVerdict: Measured<string>\n readonly delivered: Measured<boolean>\n readonly judgeResolved: Measured<boolean | null>\n readonly judgeScore: Measured<number | null>\n readonly judgePassed: Measured<number | null>\n readonly judgeTotal: Measured<number | null>\n readonly verifyPass: Measured<boolean>\n readonly verifyRc: Measured<number>\n readonly patch: Measured<PatchStats>\n /** Which document the judge fields came from (a judge file, a ledger row, or nothing). */\n readonly judgeSource: string | null\n}\n\nexport interface SupervisorRunReport {\n readonly schema: typeof SUPERVISOR_RUN_SCHEMA\n /** The `runRef` of the sources this report was computed from. */\n readonly runRef: string\n readonly instanceId: string | null\n readonly arm: string | null\n readonly supervisorId: Measured<string>\n readonly supervisorProfileDigest: Measured<string>\n readonly generatedAt: string\n readonly orchestration: OrchestrationMetrics\n readonly decision: DecisionMetrics\n readonly economics: EconomicsMetrics\n readonly outcome: OutcomeMetrics\n /** Artifacts that were missing, in read order — the provenance of every `unavailable`. */\n readonly gaps: readonly string[]\n /** The `traces` CLI command that covers the harness-session layer for this run. */\n readonly traceCommand: string\n}\n\nexport interface RollupCellRow {\n readonly instanceId: string | null\n readonly arm: string | null\n readonly steers: Measured<number>\n readonly waves: Measured<number>\n readonly utilization: Measured<number>\n readonly idlePct: Measured<number>\n readonly resolved: Measured<boolean | null>\n readonly usd: Measured<number>\n}\n\nexport interface SupervisorRunRollup {\n readonly schema: typeof SUPERVISOR_RUN_ROLLUP_SCHEMA\n readonly cells: number\n readonly steersTotal: Measured<number>\n readonly cellsWithSteers: Measured<number>\n readonly cellsWithUnavailableSteers: number\n readonly wavesMean: Measured<number>\n readonly maxConcurrencyMax: Measured<number>\n readonly utilizationMean: Measured<number>\n readonly idlePctMean: Measured<number>\n readonly workersSpawnedTotal: Measured<number>\n readonly acceptedTotal: Measured<number>\n /**\n * Sum of the per-run collapsed `totalUsd`. Prefer `spendUsd`: this total\n * mixes close-record and journal-derived cells without saying which.\n */\n readonly usdTotal: Measured<number>\n /**\n * Fleet spend measured two ways. `runs` is each measurement's own\n * denominator — the cells where that measurement was available. The two\n * sums cover different run sets, so comparing the values without their\n * denominators manufactures a phantom divergence.\n */\n readonly spendUsd: {\n readonly journalDerived: { readonly value: Measured<number>; readonly runs: number }\n readonly closeRecord: { readonly value: Measured<number>; readonly runs: number }\n }\n readonly resolvedCount: Measured<number>\n readonly perCell: readonly RollupCellRow[]\n}\n\n/**\n * A supervision tree expressed in the canonical rollout row type: one\n * `RolloutLine` per invocation, joined by `parent_rollout_id`. The root row\n * carries `role: 'supervisor'`; nested supervisors retain that role and leaf\n * invocations carry `role: 'worker'`.\n */\nexport interface SupervisorRunTree {\n readonly rootId: string | null\n readonly nodes: readonly RolloutLine[]\n /** Typed reasons a rollout field could not be recovered, in read order. */\n readonly gaps: readonly SupervisorRunTreeGap[]\n}\n\n/** Stable machine-readable reasons emitted while minting a supervisor tree. */\nexport type SupervisorRunTreeGapCode =\n | 'journal-unavailable'\n | 'source-row-malformed'\n | 'root-spawn-unavailable'\n | 'root-reward-unavailable'\n | 'child-reward-unavailable'\n | 'node-role-unavailable'\n | 'node-schema-invalid'\n\nexport interface SupervisorRunTreeGap {\n readonly code: SupervisorRunTreeGapCode\n readonly message: string\n readonly nodeId?: string\n readonly count?: number\n}\n","/**\n * The run's terminal record, read from the documents the store wrote when the\n * run ended. Runtime's own record outranks every other source: `result.json` is\n * the `SupervisedResult` that `supervise()` returned, verbatim, and its `kind`\n * is the run's status. `failure.json` is the record Runtime writes when\n * `supervise()` threw before a result landed. The control-plane-era loops\n * documents (`state.json` `status`, `result.json` `sup_status`) stay readable\n * as named legacy sources, so a report always says which record it read.\n *\n * Pure: takes already-parsed documents and returns `Measured` values. The\n * analyzer and the rollout minter share it so a run never has two statuses.\n */\n\nimport { asRecord } from './source-facts'\nimport {\n type Measured,\n type SupervisorStatusSource,\n type TerminalFailure,\n unavailable,\n} from './types'\n\n/** The reason every terminal field reads `unavailable` when no store wrote one. */\nexport const NO_TERMINAL_RECORD =\n 'no terminal record: Runtime result.json kind, Runtime failure.json, or legacy state.json / result.json status'\n\n/** The status reported for a run whose directory holds Runtime's failure record and no result. */\nexport const RUNTIME_FAILED_STATUS = 'failed'\n\nexport interface TerminalRecordInput {\n /** Legacy loops `state.json`, parsed; null when absent. */\n readonly state: Record<string, unknown> | null\n /** `result.json`, parsed; Runtime's `SupervisedResult` or the legacy loops result. */\n readonly result: Record<string, unknown> | null\n /**\n * Runtime's `failure.json`, parsed. `null` when the store was read and holds\n * none; `undefined` when the store has no such document type.\n */\n readonly failure: Record<string, unknown> | null | undefined\n}\n\nexport interface TerminalRecord {\n readonly supStatus: Measured<string>\n readonly supStatusSource: Measured<SupervisorStatusSource>\n /**\n * Runtime's `reason` on a `no-winner` result. `null` when the record carries\n * none (a winner, a failure record, or a legacy document).\n */\n readonly supReason: Measured<string | null>\n /** The recorded error. `null` when the run recorded a result and no error. */\n readonly failure: Measured<TerminalFailure | null>\n /**\n * True when the run recorded a delivered result (Runtime `winner`, legacy\n * `completed`); false on every other recorded terminal state; null when no\n * terminal record exists.\n */\n readonly completed: boolean | null\n}\n\nfunction str(value: unknown): string | null {\n return typeof value === 'string' && value.length > 0 ? value : null\n}\n\nfunction errorRecord(\n doc: Record<string, unknown>,\n source: TerminalFailure['source'],\n earlierAttempt: boolean,\n): TerminalFailure | null {\n if (typeof doc.error !== 'object' || doc.error === null) return null\n const error = asRecord(doc.error)\n return {\n source,\n name: str(error.name),\n message: str(error.message),\n at: str(doc.at),\n earlierAttempt,\n }\n}\n\nexport function readTerminalRecord(input: TerminalRecordInput): TerminalRecord {\n const { state, result } = input\n const resultKind = str(result?.kind)\n const settled = resultKind !== null && result !== null\n const failureRecord =\n input.failure === null || input.failure === undefined\n ? null\n : errorRecord(input.failure, 'runtime-failure', settled)\n\n if (settled) {\n // Runtime's settle record is the status. A failure record beside it is an\n // earlier attempt's throw (Runtime never writes one after a settle), kept\n // as the recorded error with `earlierAttempt` set; a `driver-failed`\n // no-winner carries its own rejection instead.\n return {\n supStatus: resultKind,\n supStatusSource: 'runtime-result',\n supReason: str(result.reason),\n failure: failureRecord ?? errorRecord(result, 'runtime-result', false),\n completed: resultKind === 'winner',\n }\n }\n\n if (failureRecord !== null) {\n return {\n supStatus: RUNTIME_FAILED_STATUS,\n supStatusSource: 'runtime-failure',\n supReason: null,\n failure: failureRecord,\n completed: false,\n }\n }\n if (input.failure !== null && input.failure !== undefined) {\n // The document exists but carries no `error` object. The run still ended in\n // a failure; only the detail is missing, and it says so.\n return {\n supStatus: RUNTIME_FAILED_STATUS,\n supStatusSource: 'runtime-failure',\n supReason: null,\n failure: unavailable('Runtime failure.json carries no error record'),\n completed: false,\n }\n }\n\n const legacyStateStatus = str(state?.status)\n if (legacyStateStatus !== null) {\n return {\n supStatus: legacyStateStatus,\n supStatusSource: 'legacy-state',\n supReason: null,\n failure: unavailable('legacy state.json records no failure document'),\n completed: legacyStateStatus === 'completed',\n }\n }\n const legacyResultStatus = str(result?.sup_status)\n if (legacyResultStatus !== null) {\n return {\n supStatus: legacyResultStatus,\n supStatusSource: 'legacy-result',\n supReason: null,\n failure: unavailable('legacy result.json records no failure document'),\n completed: legacyResultStatus === 'completed',\n }\n }\n\n return {\n supStatus: unavailable(NO_TERMINAL_RECORD),\n supStatusSource: unavailable(NO_TERMINAL_RECORD),\n supReason: unavailable(NO_TERMINAL_RECORD),\n failure: unavailable(NO_TERMINAL_RECORD),\n completed: null,\n }\n}\n"],"mappings":";AAwDA,SAAgB,SAAS,GAAqC;CAC5D,OAAO,OAAO,MAAM,YAAY,MAAM,OAAQ,IAAgC,CAAC;AACjF;AAEA,SAAS,IAAI,GAAoB;CAC/B,OAAO,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,IAAI,IAAI;AAC3D;AAEA,SAAS,UAAU,GAAuB;CACxC,MAAM,MAAM,SAAS,CAAC;CACtB,MAAM,MAAM,SAAS,IAAI,MAAM;CAC/B,MAAM,YAAY,IAAI,aAAa,IAAI;CACvC,MAAM,aAAa,IAAI,cAAc,IAAI;CACzC,OAAO;EACL,QAAQ;GACN,OAAO,IAAI,IAAI,KAAK;GACpB,QAAQ,IAAI,IAAI,MAAM;GACtB,WAAW,IAAI,SAAS;GACxB,YAAY,IAAI,UAAU;GAC1B,UAAU,cAAc,KAAA,KAAa,eAAe,KAAA;GACpD,gBAAgB,IAAI,wBAAwB;EAC9C;EACA,aAAa,IAAI,gBAAgB;EACjC,KAAK,IAAI,IAAI,GAAG;EAChB,UAAU,IAAI,aAAa;CAC7B;AACF;AAEA,SAAgB,WAAW,MAAgD;CACzE,OAAO,0BAA0B,IAAI,CAAC,CAAC;AACzC;AAOA,SAAS,0BAA0B,MAAkC;CACnE,IAAI,SAAS,MAAM,OAAO;EAAE,MAAM,CAAC;EAAG,aAAa;CAAE;CACrD,MAAM,MAAiC,CAAC;CACxC,IAAI,cAAc;CAClB,KAAK,MAAM,QAAQ,KAAK,MAAM,IAAI,GAAG;EACnC,MAAM,UAAU,KAAK,KAAK;EAC1B,IAAI,CAAC,SAAS;EACd,IAAI;GACF,MAAM,SAAkB,KAAK,MAAM,OAAO;GAC1C,IAAI,OAAO,WAAW,YAAY,WAAW,QAAQ,CAAC,MAAM,QAAQ,MAAM,GACxE,IAAI,KAAK,MAAiC;QAE1C,eAAe;EAEnB,QAAQ;GACN,eAAe;EACjB;CACF;CACA,OAAO;EAAE,MAAM;EAAK;CAAY;AAClC;AAEA,SAAgB,UAAU,MAAqD;CAC7E,IAAI,SAAS,MAAM,OAAO;CAC1B,IAAI;EACF,MAAM,SAAkB,KAAK,MAAM,IAAI;EACvC,OAAO,OAAO,WAAW,YAAY,WAAW,OAC3C,SACD;CACN,QAAQ;EACN,OAAO;CACT;AACF;AAEA,SAAS,GAAG,IAA4B;CACtC,IAAI,OAAO,OAAO,UAAU,OAAO;CACnC,MAAM,IAAI,KAAK,MAAM,EAAE;CACvB,OAAO,OAAO,SAAS,CAAC,IAAI,IAAI;AAClC;AAkLA,SAAS,YAAY,GAA0B;CAC7C,IAAI,OAAO,MAAM,UAAU,OAAO;EAAE,OAAO;EAAG,OAAO;EAAM,OAAO;EAAM,KAAK;CAAE;CAC/E,IAAI,OAAO,MAAM,YAAY,MAAM,MACjC,OAAO;EAAE,OAAO;EAAM,OAAO;EAAM,OAAO;EAAM,KAAK;CAAK;CAE5D,MAAM,MAAM;CACZ,MAAM,QAAQ,OAAO,IAAI,UAAU,YAAY,IAAI,QAAQ;CAC3D,OAAO;EACL,OAAO,UAAU,OAAO,OAAO,QAAQ,UAAU;EACjD;EACA,OAAO,OAAO,IAAI,UAAU,YAAY,OAAO,SAAS,IAAI,KAAK,IAAI,IAAI,QAAQ;EACjF,KAAK;CACP;AACF;AAEA,SAAgB,gBACd,QACQ;CACR,OAAO,OAAO,YAAY,OAAO;AACnC;;AA2BA,MAAM,mBAAmB;CAAC;CAAW;CAAW;CAAa;AAAS;;;;;;AAQtE,MAAM,yBAA4C,CAAC,WAAW,OAAO;AAErE,SAAS,cAAc,OAAsC;CAC3D,OAAQ,iBAAwC,SAAS,KAAK,IAAK,QAA0B;AAC/F;;;;;;AAiBA,SAAS,eAAe,KAAiD;CACvE,MAAM,OAAO,IAAI;CACjB,IAAI,OAAO,SAAS,UAAU,OAAO,EAAE,SAAS,aAAa;CAG7D,IAAI,SAAS,WAAW,OAAO,IAAI,SAAS,UAC1C,OAAO;EAAE,SAAS;EAAW,SAAS;EAAoB,MAAM;CAAQ;CAE1E,IAAI,SAAS,SAAS;EACpB,MAAM,QAAQ,IAAI;EAClB,IAAI,OAAO,UAAU,YAAY,UAAU,QAAQ,MAAM,QAAQ,KAAK,GACpE,OAAO,EAAE,SAAS,aAAa;EAEjC,MAAM,YAAa,MAAkC;EACrD,IAAI,OAAO,cAAc,UAAU,OAAO,EAAE,SAAS,aAAa;EAClE,MAAM,OAAO,cAAc,SAAS;EACpC,IAAI,SAAS,MACX,OAAO;GACL,SAAS;GACT,SAAS;GACT,MAAM;GACN,OAAO;EACT;EAIF,OAAO;GAAE,SAAS;GAAW,SAAS;GAAoB,MAAM;EAAU;CAC5E;CAEA,MAAM,OAAO,cAAc,IAAI;CAC/B,IAAI,SAAS,MACX,OAAO;EAAE,SAAS;EAAS,SAAS;EAAQ,MAAM;EAAM,OAAO;CAAoB;CAErF,IAAI,uBAAuB,SAAS,IAAI,GACtC,OAAO;EAAE,SAAS;EAAW,SAAS;EAAQ;CAAK;CAIrD,OAAO,EAAE,SAAS,aAAa;AACjC;AAEA,SAAS,iBAAiB,MAAgE;CACxF,IAAI,KAAK,SAAS,GAAG,OAAO;CAC5B,IAAI,KAAK,OAAO,GAAG,OAAO;CAC1B,OAAO,KAAK,IAAI,kBAAkB,IAAI,qBAAqB;AAC7D;AAEA,SAAgB,oBAAoB,KAAgD;CAClF,MAAM,gBAAgB,0BAA0B,IAAI,OAAO;CAC3D,MAAM,SAAS,cAAc;CAC7B,MAAM,QAAQ,UAAU,IAAI,KAAK;CAEjC,MAAM,SAAqB,CAAC;CAC5B,MAAM,SAAqB,CAAC;CAC5B,IAAI,UAAU;CACd,IAAI,WAAW;CACf,IAAI,iBAAiB;CACrB,IAAI,kBAAkB;CACtB,IAAI,gBAAgB;CACpB,IAAI,2BAA2B;CAC/B,IAAI,wBAAwB;CAC5B,IAAI,0BAA0B;CAC9B,IAAI,WAAW;CACf,IAAI,qBAAqB;CACzB,IAAI,uBAAuB;CAC3B,IAAI,eAAe;CACnB,IAAI,SAAwB;CAC5B,IAAI,iBAAiB;CACrB,MAAM,gCAAgB,IAAI,IAAoB;CAC9C,MAAM,+BAAe,IAAI,IAAuB;CAChD,MAAM,cAA0E,CAAC;CACjF,KAAK,MAAM,CAAC,WAAW,QAAQ,OAAO,QAAQ,GAAG;EAC/C,MAAM,UAAU,eAAe,GAAG;EAClC,IAAI,QAAQ,YAAY,cAAc;GACpC,kBAAkB;GAClB;EACF;EACA,aAAa,IAAI,QAAQ,OAAO;EAChC,IAAI,QAAQ,YAAY,WAAW;GACjC,cAAc,IAAI,QAAQ,OAAO,cAAc,IAAI,QAAQ,IAAI,KAAK,KAAK,CAAC;GAC1E;EACF;EACA,MAAM,EAAE,MAAM,OAAO,OAAO;EAC5B,MAAM,KAAK,OAAO,GAAG,OAAO,WAAW,GAAG,KAAK;EAC/C,IAAI,SAAS,WAAW;GACtB,MAAM,SAAS,OAAO,GAAG,WAAW,YAAY,GAAG,OAAO,SAAS,IAAI,GAAG,SAAS;GACnF,MAAM,QAAQ,OAAO,GAAG,UAAU,WAAW,GAAG,QAAQ;GACxD,MAAM,gBAA0B,CAAC;GACjC,IAAI,GAAG,WAAW,GAAG,cAAc,KAAK,IAAI;GAC5C,MAAM,4BAA4B,GAAG,WAAW,KAAA,KAAa,WAAW;GACxE,IACE,GAAG,WAAW,QACd,CAAC,6BACD,EAAE,OAAO,GAAG,WAAW,YAAY,GAAG,OAAO,SAAS,IAEtD,cAAc,KAAK,QAAQ;GAE7B,IAAI,MAAM,WAAW,GAAG,cAAc,KAAK,OAAO;GAClD,IAAI,GAAG,SAAS,KAAA,KAAa,GAAG,SAAS,gBAAgB,GAAG,SAAS,UACnE,cAAc,KAAK,MAAM;GAE3B,IACE,GAAG,kBAAkB,KAAA,KACrB,EAAE,OAAO,GAAG,kBAAkB,YAAY,GAAG,cAAc,SAAS,IAEpE,cAAc,KAAK,eAAe;GAEpC,IAAI,GAAG,YAAY,KAAA,KAAa,EAAE,OAAO,GAAG,YAAY,YAAY,GAAG,QAAQ,SAAS,IACtF,cAAc,KAAK,SAAS;GAE9B,MAAM,OACJ,GAAG,SAAS,gBAAgB,GAAG,SAAS,WAAW,GAAG,OAAO;GAC/D,MAAM,QAAQ,cAAc,WAAW;GACvC,IAAI,SAAS,WAAW,QAAQ,WAAW,MAAM,SAAS;GAC1D,OAAO,KAAK;IACV;IACA;IACA;IACA;IACA;IACA,eACE,OAAO,GAAG,kBAAkB,YAAY,GAAG,cAAc,SAAS,IAC9D,GAAG,gBACH;IACN,SAAS,OAAO,GAAG,YAAY,YAAY,GAAG,QAAQ,SAAS,IAAI,GAAG,UAAU;IAChF,IAAI,GAAG,GAAG,EAAE;IACZ;IACA;GACF,CAAC;EACH,OAAO,IAAI,SAAS,WAAW;GAC7B,MAAM,UAAU,YAAY,GAAG,OAAO;GACtC,OAAO,KAAK;IACV;IACA,MAAM;IACN,QAAQ,OAAO,GAAG,WAAW,WAAW,GAAG,SAAS;IACpD,SAAS,QAAQ;IACjB,OAAO,QAAQ;IACf,OAAO,QAAQ;IACf,YAAY,QAAQ;IACpB,QAAQ,OAAO,GAAG,WAAW,WAAW,GAAG,SAAS;IACpD,OAAO,OAAO,GAAG,UAAU,YAAY,GAAG,QAAQ;IAClD,IAAI,GAAG,GAAG,EAAE;IACZ,OAAO,UAAU,GAAG,KAAK;IACzB,UAAU,SAAS,GAAG,KAAK,CAAC,CAAC,WAAW,KAAA;GAC1C,CAAC;EACH,OAAO,IAAI,SAAS,aAClB,OAAO,KAAK;GACV;GACA,MAAM;GACN,QAAQ;GACR,SAAS,OAAO,GAAG,WAAW,WAAW,GAAG,SAAS;GACrD,OAAO;GACP,OAAO;GACP,YAAY,OAAO,GAAG,WAAW,WAAW,GAAG,SAAS;GACxD,QAAQ,OAAO,GAAG,WAAW,WAAW,GAAG,SAAS;GACpD,OAAO;GACP,IAAI,GAAG,GAAG,EAAE;GACZ,OAAO;IACL,QAAQ;KACN,OAAO;KACP,QAAQ;KACR,WAAW;KACX,YAAY;KACZ,UAAU;KACV,gBAAgB;IAClB;IACA,aAAa;IACb,KAAK;IACL,UAAU;GACZ;GACA,UAAU;EACZ,CAAC;OACI,IAAI,SAAS,WAGlB,YAAY,KAAK;GAAE;GAAI,OAAO,UAAU,GAAG,KAAK;GAAG,IAAI,GAAG,GAAG,EAAE;EAAE,CAAC;OAMlE,MAAM,IAAI,MAAM,gCAAgC,KAAK,UAAUA,IAAS,GAAG;CAE/E;CAKA,KAAK,MAAM,OAAO,aAAa;EAC7B,IAAI,WAAW,QAAQ,IAAI,OAAO,QAAQ;EAC1C,IAAI,IAAI,MAAM,aAAa;GACzB,WAAW,IAAI,MAAM,OAAO;GAC5B,YAAY,IAAI,MAAM,OAAO;GAC7B,kBAAkB,IAAI,MAAM,OAAO;GACnC,mBAAmB,IAAI,MAAM,OAAO;GACpC,gBAAgB,iBAAiB,IAAI,MAAM,OAAO;GAClD,2BAA2B,4BAA4B,IAAI,MAAM,OAAO;GACxE,yBAAyB;EAC3B,OACE,2BAA2B;EAE7B,IAAI,IAAI,MAAM,UAAU;GACtB,YAAY,IAAI,MAAM;GACtB,sBAAsB;EACxB,OACE,wBAAwB;EAE1B,gBAAgB;CAClB;CAEA,MAAM,eAAe,OAAO,QAAQ,MAAM,EAAE,SAAS,EAAE,OAAO,MAAM;CACpE,MAAM,YAAY,IAAI,IAAI,aAAa,KAAK,MAAM,EAAE,EAAE,CAAC;CACvD,MAAM,eAAe,OAAO,QAAQ,MAAM,UAAU,IAAI,EAAE,EAAE,CAAC;CAE7D,MAAM,6BAAa,IAAI,IAA4B;CACnD,MAAM,gBAAkC,CAAC;CACzC,KAAK,MAAM,CAAC,aAAa,OAAO,IAAI,WAAW,CAAC,EAAA,CAAG,QAAQ,GAAG;EAC5D,MAAM,QAAwB;GAC5B;GACA,UAAU,EAAE,YAAY;GACxB,OAAO,EAAE;GACT,eAAe,EAAE,UAAU;GAC3B,gBAAgB,EAAE,WAAW;GAC7B,kBAAkB;GAClB,mBAAmB;GACnB,SAAS;GACT,UAAU;GACV,YAAY;GACZ,QAAQ;GACR,OAAO;GACP,oBAAoB;GACpB,eAAe;GACf,cAAc;GACd,iBAAiB;GACjB,yBAAyB;GACzB,4BAA4B;GAC5B,WAAW;GACX,eAAe,CAAC;GAChB,uBAAuB,CAAC;EAC1B;EACA,MAAM,cAAc,0BAA0B,EAAE,KAAK;EACrD,MAAM,eAAe,0BAA0B,EAAE,MAAM;EACvD,MAAM,mBAAmB,YAAY;EACrC,MAAM,oBAAoB,aAAa;EACvC,KAAK,MAAM,CAAC,KAAK,QAAQ,YAAY,KAAK,QAAQ,GAAG;GACnD,IAAI,OAAO,IAAI,YAAY,YAAY,IAAI,QAAQ,KAAK,CAAC,CAAC,WAAW,GAAG;GACxE,MAAM,cAAc,KAAK;IACvB,WAAW,OAAO,IAAI,OAAO,YAAY,IAAI,GAAG,SAAS,IAAI,IAAI,KAAK;IACtE;GACF,CAAC;EACH;EACA,KAAK,MAAM,CAAC,KAAK,OAAO,aAAa,KAAK,QAAQ,GAAG;GACnD,MAAM,OAAO,GAAG;GAChB,IAAI,SAAS,WAAW;IACtB,IAAI,GAAG,cAAc,MAAM;KACzB,MAAM,aAAa;KACnB;IACF;IACA,MAAM,sBAAsB,KAAK;KAC/B,WACE,OAAO,GAAG,cAAc,YAAY,GAAG,UAAU,SAAS,IAAI,GAAG,YAAY;KAC/E,WAAW,OAAO,GAAG,cAAc,YAAY,GAAG,YAAY;KAC9D;IACF,CAAC;GACH,OAAO,IAAI,SAAS,WAClB,MAAM,UAAU,GAAG,GAAG,EAAE;QACnB,IAAI,SAAS,YAAY;IAC9B,MAAM,WAAW;IACjB,MAAM,aAAa,GAAG,GAAG,EAAE;IAC3B,MAAM,SAAS,OAAO,GAAG,WAAW,YAAY,GAAG,SAAS;IAC5D,MAAM,QAAQ,OAAO,GAAG,UAAU,YAAY,OAAO,SAAS,GAAG,KAAK,IAAI,GAAG,QAAQ;IACrF,MAAM,qBAAqB,OAAO,GAAG,eAAe,WAAW,GAAG,aAAa;IAC/E,MAAM,gBAAgB,OAAO,GAAG,aAAa,WAAW,GAAG,SAAS,SAAS;GAC/E;EACF;EACA,MAAM,+BAAe,IAAI,IAAoB;EAC7C,KAAK,MAAM,WAAW,MAAM,eAAe;GACzC,IAAI,QAAQ,cAAc,MAAM;GAChC,aAAa,IAAI,QAAQ,YAAY,aAAa,IAAI,QAAQ,SAAS,KAAK,KAAK,CAAC;EACpF;EACA,MAAM,uCAAuB,IAAI,IAAwC;EACzE,KAAK,MAAM,mBAAmB,MAAM,uBAAuB;GACzD,IAAI,gBAAgB,cAAc,MAAM;GACxC,MAAM,UAAU,qBAAqB,IAAI,gBAAgB,SAAS,KAAK,CAAC;GACxE,QAAQ,KAAK,eAAe;GAC5B,qBAAqB,IAAI,gBAAgB,WAAW,OAAO;EAC7D;EACA,MAAM,gBAA0B,CAAC;EACjC,IAAI,CAAC,MAAM,eAAe,cAAc,KAAK,cAAc;EAC3D,IAAI,CAAC,MAAM,gBAAgB,cAAc,KAAK,eAAe;EAC7D,IAAI,MAAM,mBAAmB,GAAG,cAAc,KAAK,+BAA+B;EAClF,IAAI,MAAM,oBAAoB,GAAG,cAAc,KAAK,+BAA+B;EACnF,IAAI,MAAM,cAAc,MAAM,YAAY,QAAQ,cAAc,IAAI,GAClE,cAAc,KAAK,2BAA2B;EAEhD,IAAI,CAAC,GAAG,aAAa,OAAO,CAAC,CAAC,CAAC,MAAM,UAAU,QAAQ,CAAC,GACtD,cAAc,KAAK,8BAA8B;EAEnD,IAAI,MAAM,sBAAsB,MAAM,oBAAoB,gBAAgB,cAAc,IAAI,GAC1F,cAAc,KAAK,oCAAoC;EAEzD,IAAI,CAAC,GAAG,qBAAqB,KAAK,CAAC,CAAC,CAAC,MAAM,OAAO,CAAC,aAAa,IAAI,EAAE,CAAC,GACrE,cAAc,KAAK,uCAAuC;EAE5D,IAAI,cAAc,WAAW,GAC3B,MAAM,eAAe,aAAa;OAElC,MAAM,0BAA0B,CAAC,GAAG,IAAI,IAAI,aAAa,CAAC,CAAC,CAAC,KAAK,IAAI;EAGvE,MAAM,mBAAmB,CAAC,GAAG,aAAa;EAC1C,IAAI,CAAC,GAAG,qBAAqB,OAAO,CAAC,CAAC,CAAC,MAAM,SAAS,KAAK,SAAS,CAAC,GACnE,iBAAiB,KAAK,uCAAuC;EAE/D,IAAI,MAAM,sBAAsB,MAAM,oBAAoB,gBAAgB,cAAc,IAAI,GAC1F,iBAAiB,KAAK,yCAAyC;EAEjE,IAAI,iBAAiB,WAAW,GAC9B,MAAM,kBAAkB,CAAC,GAAG,aAAa,KAAK,CAAC,CAAC,CAAC,QAAQ,OAAO;GAC9D,MAAM,kBAAkB,qBAAqB,IAAI,EAAE;GACnD,OAAO,iBAAiB,WAAW,KAAK,gBAAgB,EAAE,EAAE,cAAc;EAC5E,CAAC,CAAC,CAAC;OAEH,MAAM,6BAA6B,CAAC,GAAG,IAAI,IAAI,gBAAgB,CAAC,CAAC,CAAC,KAAK,IAAI;EAE7E,cAAc,KAAK,KAAK;EACxB,WAAW,IAAI,gBAAgB,CAAC,GAAG,KAAK;CAC1C;CAEA,MAAM,YAAY,GAAG,OAAO,SAAS,KAAK,OAAO,EAAE,EAAE,MAAM;CAC3D,MAAM,YAAY,WAAW,OAAO,OAAO,OAAO,MAAM,UAAU,MAAM,OAAO,MAAM;CACrF,MAAM,cAAc,GAAG,OAAO,WAAW,KAAK,WAAW,MAAM;CAE/D,IAAI,eAA8B;CAClC,IAAI,cAA6B;CACjC,MAAM,kBAAkB,OAA4B;EAClD,IAAI,OAAO,MAAM;EACjB,IAAI,iBAAiB,QAAQ,KAAK,cAAc,eAAe;EAC/D,IAAI,gBAAgB,QAAQ,KAAK,aAAa,cAAc;CAC9D;CACA,KAAK,MAAM,SAAS,QAAQ,eAAe,MAAM,EAAE;CACnD,KAAK,MAAM,SAAS,QAAQ,eAAe,MAAM,EAAE;CACnD,KAAK,MAAM,WAAW,aAAa,eAAe,QAAQ,EAAE;CAE5D,OAAO;EACL;EACA;EACA;EACA;EACA;EACA,OAAO;GACL,UAAU;GACV,WAAW;GACX,WAAW;GACX,YAAY;GACZ,UAAU;GACV,qBAAqB;GACrB,kBAAkB;GAClB,oBAAoB;GACpB,KAAK;GACL,eAAe;GACf,iBAAiB;GACjB;EACF;EACA;EACA;EACA,aAAa,OAAO,SAAS,cAAc;EAC3C,oBAAoB,cAAc,cAAc;EAChD,0BAA0B,cAAc;EACxC,0BAA0B,OAAO,YAAY,aAAa;EAC1D,oBAAoB,YAAY;EAChC,gBAAgB,iBAAiB,YAAY;EAC7C;EACA;EACA;EACA;EACA;CACF;AACF;;;AChtBA,SAAgB,YAAY,QAA6B;CACvD,OAAO,EAAE,aAAa,OAAO;AAC/B;AAEA,SAAgB,cAAc,GAA8B;CAC1D,OAAO,OAAO,MAAM,YAAY,MAAM,QAAQ,OAAQ,EAAkB,gBAAgB;AAC1F;;AAGA,SAAgB,aAAa,GAAuD;CAClF,IAAI,cAAc,CAAC,GAAG,OAAO,iBAAiB,EAAE;CAChD,IAAI,MAAM,MAAM,OAAO;CACvB,OAAO,OAAO,CAAC;AACjB;;AA+DA,MAAa,mBAAiC;CAC5C,eAAe;CACf,cAAc;CACd,UAAU;CACV,gBAAgB;CAChB,cAAc;AAChB;AAgHA,MAAa,wBAAwB;AACrC,MAAa,+BAA+B;;;;;;;;;;;;;;;;ACtN5C,MAAa,qBACX;;AAGF,MAAa,wBAAwB;AAgCrC,SAAS,IAAI,OAA+B;CAC1C,OAAO,OAAO,UAAU,YAAY,MAAM,SAAS,IAAI,QAAQ;AACjE;AAEA,SAAS,YACP,KACA,QACA,gBACwB;CACxB,IAAI,OAAO,IAAI,UAAU,YAAY,IAAI,UAAU,MAAM,OAAO;CAChE,MAAM,QAAQ,SAAS,IAAI,KAAK;CAChC,OAAO;EACL;EACA,MAAM,IAAI,MAAM,IAAI;EACpB,SAAS,IAAI,MAAM,OAAO;EAC1B,IAAI,IAAI,IAAI,EAAE;EACd;CACF;AACF;AAEA,SAAgB,mBAAmB,OAA4C;CAC7E,MAAM,EAAE,OAAO,WAAW;CAC1B,MAAM,aAAa,IAAI,QAAQ,IAAI;CACnC,MAAM,UAAU,eAAe,QAAQ,WAAW;CAClD,MAAM,gBACJ,MAAM,YAAY,QAAQ,MAAM,YAAY,KAAA,IACxC,OACA,YAAY,MAAM,SAAS,mBAAmB,OAAO;CAE3D,IAAI,SAKF,OAAO;EACL,WAAW;EACX,iBAAiB;EACjB,WAAW,IAAI,OAAO,MAAM;EAC5B,SAAS,iBAAiB,YAAY,QAAQ,kBAAkB,KAAK;EACrE,WAAW,eAAe;CAC5B;CAGF,IAAI,kBAAkB,MACpB,OAAO;EACL,WAAW;EACX,iBAAiB;EACjB,WAAW;EACX,SAAS;EACT,WAAW;CACb;CAEF,IAAI,MAAM,YAAY,QAAQ,MAAM,YAAY,KAAA,GAG9C,OAAO;EACL,WAAW;EACX,iBAAiB;EACjB,WAAW;EACX,SAAS,YAAY,8CAA8C;EACnE,WAAW;CACb;CAGF,MAAM,oBAAoB,IAAI,OAAO,MAAM;CAC3C,IAAI,sBAAsB,MACxB,OAAO;EACL,WAAW;EACX,iBAAiB;EACjB,WAAW;EACX,SAAS,YAAY,+CAA+C;EACpE,WAAW,sBAAsB;CACnC;CAEF,MAAM,qBAAqB,IAAI,QAAQ,UAAU;CACjD,IAAI,uBAAuB,MACzB,OAAO;EACL,WAAW;EACX,iBAAiB;EACjB,WAAW;EACX,SAAS,YAAY,gDAAgD;EACrE,WAAW,uBAAuB;CACpC;CAGF,OAAO;EACL,WAAW,YAAY,kBAAkB;EACzC,iBAAiB,YAAY,kBAAkB;EAC/C,WAAW,YAAY,kBAAkB;EACzC,SAAS,YAAY,kBAAkB;EACvC,WAAW;CACb;AACF"}
@@ -1,5 +1,5 @@
1
1
  import { w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
2
- import { h as TraceAnalysisToolDescriptor } from "./engine-DhFir3Ys.js";
2
+ import { h as TraceAnalysisToolDescriptor } from "./engine-CvW_I72-.js";
3
3
  //#region src/analyst/tool-groups.d.ts
4
4
  /** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
5
5
  type TraceToolGroupName =
@@ -25,4 +25,4 @@ type TraceToolGroupName =
25
25
  declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): TraceAnalysisToolDescriptor[];
26
26
  //#endregion
27
27
  export { buildTraceToolsForGroup as n, TraceToolGroupName as t };
28
- //# sourceMappingURL=tool-groups-Cp4Xdzrp.d.ts.map
28
+ //# sourceMappingURL=tool-groups-DAe1t6zb.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tool-groups-DAe1t6zb.d.ts","names":[],"sources":["../src/analyst/tool-groups.ts"],"mappings":";;;;KAmBY;;;;;;;;;;;;;;;;;;;;iBAgDI,wBACd,OAAO,oBACP,OAAO,qBACN"}
@@ -2,7 +2,7 @@ import { r as CaptureIntegrityError } from "../errors-DEE6u6ot.js";
2
2
  import { b as CustomTokenPricing, c as CostLedgerHandle, p as CostProvenance } from "../cost-ledger-DbQdN3nO.js";
3
3
  import { u as RunTokenUsage } from "../run-record-DTv1MdjK.js";
4
4
  import { t as DefaultVerdict } from "../verdict-E4eRNf7-.js";
5
- import { a as TraceAnalystLimits, n as TraceAnalysisEngine } from "../engine-DhFir3Ys.js";
5
+ import { a as TraceAnalystLimits, n as TraceAnalysisEngine } from "../engine-CvW_I72-.js";
6
6
  import { t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
7
7
  import { a as RecordedTrajectoryStep, h as isRecordedTimeout } from "../steps-CiNVJry_.js";
8
8
  //#region src/trace-repair/mini-swe-scaffold.d.ts
package/dist/traces.d.ts CHANGED
@@ -2,10 +2,10 @@ import { c as ValidationError, o as LimitExceededError, r as CaptureIntegrityErr
2
2
  import { C as TraceEvent, E as isToolSpan, S as ToolSpan, T as isLlmSpan, _ as Span, a as FAILURE_CLASSES, b as SpanStatus, c as JudgeSpan, d as RetrievalSpan, f as Run, g as SandboxSpan, h as RunStatus, i as EventKind, l as LlmSpan, m as RunOutcome, n as BudgetLedgerEntry, o as FailureClass, p as RunLayer, r as BudgetSpec, s as GenericSpan, t as Artifact, u as Message, v as SpanBase, w as isJudgeSpan, x as TRACE_SCHEMA_VERSION, y as SpanKind } from "./schema-CR5cpjQ3.js";
3
3
  import { a as RunRecord, l as RunTerminalOutcome, s as RunSplitTag, u as RunTokenUsage } from "./run-record-DTv1MdjK.js";
4
4
  import { A as SearchSpanResult, B as ViewSpansResult, C as TRACE_ANALYSIS_LIMITS, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, S as BoundedTraceAnalysisStoreOptions, T as TraceAnalysisStoreContext, V as ViewTraceOversized, j as SearchTraceResult, k as QueryTracesPage, w as TraceAnalysisStore, z as TraceAnalystTraceSummary } from "./types-DN2WdT5S.js";
5
- import { A as REDACTION_VERSION, B as exportRunAsOtlp, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, E as AnalyzeTracesOptions, F as OtlpFlatLine, G as ExtractedUsage, H as CaptureFetchOptions, I as OTEL_AGENT_EVAL_SCOPE, J as extractUsageFromResponse, K as SseUsageMode, L as OtlpExport, M as RedactionRule, N as redactString, O as analyzeTraces, P as redactValue, R as OtlpResourceSpans, S as planTraceInsightQuestions, T as AnalyzeTracesInput, U as captureFetchToRawSink, V as CaptureFetchContext, W as ExtractUsageFromSseOptions, X as createBoundedTraceAnalysisStore, Y as extractUsageFromSse, _ as buildTraceInsightPrompt, a as ToolSpansToTraceAnalysisStoreOptions, b as domainEvidencePattern, c as TraceInsightFinding, d as TraceInsightQualityGate, f as TraceInsightQuestion, g as buildTraceInsightContext, h as TraceInsightTask, i as OtlpFileTraceStoreOptions, j as RedactionReport, k as DEFAULT_REDACTION_RULES, l as TraceInsightPanelRole, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, q as extractUsage, r as OtlpFileTraceStore, s as TraceInsightContext, t as ToolTraceMissingError, u as TraceInsightPromptInput, v as defaultTraceInsightPanel, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope, z as OtlpSpan } from "./store-tool-spans-CCZNsihA.js";
5
+ import { A as REDACTION_VERSION, B as exportRunAsOtlp, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, E as AnalyzeTracesOptions, F as OtlpFlatLine, G as ExtractedUsage, H as CaptureFetchOptions, I as OTEL_AGENT_EVAL_SCOPE, J as extractUsageFromResponse, K as SseUsageMode, L as OtlpExport, M as RedactionRule, N as redactString, O as analyzeTraces, P as redactValue, R as OtlpResourceSpans, S as planTraceInsightQuestions, T as AnalyzeTracesInput, U as captureFetchToRawSink, V as CaptureFetchContext, W as ExtractUsageFromSseOptions, X as createBoundedTraceAnalysisStore, Y as extractUsageFromSse, _ as buildTraceInsightPrompt, a as ToolSpansToTraceAnalysisStoreOptions, b as domainEvidencePattern, c as TraceInsightFinding, d as TraceInsightQualityGate, f as TraceInsightQuestion, g as buildTraceInsightContext, h as TraceInsightTask, i as OtlpFileTraceStoreOptions, j as RedactionReport, k as DEFAULT_REDACTION_RULES, l as TraceInsightPanelRole, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, q as extractUsage, r as OtlpFileTraceStore, s as TraceInsightContext, t as ToolTraceMissingError, u as TraceInsightPromptInput, v as defaultTraceInsightPanel, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope, z as OtlpSpan } from "./store-tool-spans-4o55ABER.js";
6
6
  import { G as NoopRawProviderSink, H as FileSystemRawProviderSinkOptions, J as RawProviderEvent, K as ProviderRedactor, Q as providerFromBaseUrl, U as InMemoryRawProviderSink, V as FileSystemRawProviderSink, W as InMemoryRawProviderSinkOptions, X as RawProviderSinkFilter, Y as RawProviderSink, Z as defaultProviderRedactor, q as RawProviderDirection } from "./types-gvRsyJLh.js";
7
7
  import { a as RunFilter, i as InMemoryTraceStore, n as FileSystemTraceStore, o as SpanFilter, r as FileSystemTraceStoreOptions, s as TraceStore, t as EventFilter } from "./store-BErPvYBr.js";
8
- import { _ as traceAnalystFunctionGroup, g as buildTraceAnalysisToolDescriptors, h as TraceAnalysisToolDescriptor, m as TRACE_ANALYST_TOOL_NAMESPACE, p as BuildTraceAnalysisToolsOptions } from "./engine-DhFir3Ys.js";
8
+ import { _ as traceAnalystFunctionGroup, g as buildTraceAnalysisToolDescriptors, h as TraceAnalysisToolDescriptor, m as TRACE_ANALYST_TOOL_NAMESPACE, p as BuildTraceAnalysisToolsOptions } from "./engine-CvW_I72-.js";
9
9
  import { a as TraceEmitterOptions, i as TraceEmitter, n as RunCompleteHookContext, r as SpanHandle, t as RunCompleteHook } from "./emitter-Cs0egaFd.js";
10
10
  import { C as TOOL_LATENCY_MS, D as asNumber, E as applyLlmSpanOtlpAttributes, O as contextInputTokens, S as TOOL_ARGS_CAPTURED, T as TOOL_NAME_ATTR_KEYS, _ as LlmSpanOtlpInput, a as LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, b as RUN_COST_ATTR_KEYS, c as LLM_COST_USD, d as LLM_MODEL_ATTR_KEYS, f as LLM_MODEL_NAME, g as LLM_REASONING_TOKEN_ATTR_KEYS, h as LLM_REASONING_TOKENS, i as LLM_CACHE_WRITE_TOKENS, k as firstNumberAttr, l as LLM_INPUT_TOKENS, m as LLM_OUTPUT_TOKEN_ATTR_KEYS, n as LLM_CACHED_TOKENS, o as LLM_CONTEXT_TOKENS, p as LLM_OUTPUT_TOKENS, r as LLM_CACHED_TOKEN_ATTR_KEYS, s as LLM_COST_ATTR_KEYS, t as INPUT_VALUE, u as LLM_INPUT_TOKEN_ATTR_KEYS, v as OPENINFERENCE_SPAN_KIND, w as TOOL_NAME, x as SPAN_KIND_ATTR_KEYS, y as OUTPUT_VALUE } from "./attribute-vocabulary-DLJ6303h.js";
11
11
  import { a as RunIntegrityReport, i as RunIntegrityIssueCode, n as RunIntegrityExpectations, o as assertRunCaptured, r as RunIntegrityIssue, s as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-BKTcA-HP.js";
package/dist/traces.js CHANGED
@@ -3,15 +3,15 @@ import { a as hasCapturedToolArgs, c as llmSpans, d as runsForScenario, f as too
3
3
  import { c as validateRunRecord, i as modelHasSnapshot } from "./run-record-ZIsR9Fif.js";
4
4
  import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
5
5
  import { t as TraceEmitter } from "./emitter-DeQHiDMm.js";
6
- import { C as asString, D as projectOtlpFlatLine, E as firstStringAttr, G as classifyOtlpSpanRole, K as isOtlpModelCall, O as spanEpochMillis, S as TraceNotFoundError, T as extractOtlpAttributes, W as applyToolSpanOtlpAttributes, _ as TraceAnalysisStoreContractError, a as TRACE_ANALYST_TOOL_NAMESPACE, b as TraceFileMissingError, c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, g as TraceAnalysisLimitError, h as SpanNotFoundError, k as stringField, m as TRACE_ANALYSIS_LIMITS, o as buildTraceAnalysisToolDescriptors, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, q as traceSpanKindToOpenInferenceKind, s as traceAnalystFunctionGroup, v as TraceAnalysisValidationError, w as compareSpanTime, x as TraceFileTooLargeError, y as TraceFileMalformedError } from "./kind-factory-gP6lDySe.js";
6
+ import { C as asString, D as projectOtlpFlatLine, E as firstStringAttr, G as classifyOtlpSpanRole, K as isOtlpModelCall, O as spanEpochMillis, S as TraceNotFoundError, T as extractOtlpAttributes, W as applyToolSpanOtlpAttributes, _ as TraceAnalysisStoreContractError, a as TRACE_ANALYST_TOOL_NAMESPACE, b as TraceFileMissingError, c as createBoundedTraceAnalysisStore, f as DEFAULT_TRACE_ANALYST_BUDGETS, g as TraceAnalysisLimitError, h as SpanNotFoundError, k as stringField, m as TRACE_ANALYSIS_LIMITS, o as buildTraceAnalysisToolDescriptors, p as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, q as traceSpanKindToOpenInferenceKind, s as traceAnalystFunctionGroup, v as TraceAnalysisValidationError, w as compareSpanTime, x as TraceFileTooLargeError, y as TraceFileMalformedError } from "./kind-factory-BLvL-E44.js";
7
7
  import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
8
8
  import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BrQ8mCLX.js";
9
9
  import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
10
- import { C as toOtlpAttributes, S as msToUnixNano, _ as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, a as spanStatusToOtlp, b as spanIdForWire, c as defaultTraceInsightPanel, d as inferDomainKeywords, f as planTraceInsightQuestions, g as TRACE_ANALYST_ACTOR_DESCRIPTION, h as analyzeTraces, i as epochMillisToIso, l as describeTraceInsightScope, m as tokenizeDomainWords, n as toolSpansToTraceAnalysisStore, o as buildTraceInsightContext, p as scoreTraceInsightReadiness, r as createOtlpFlatLine, s as buildTraceInsightPrompt, t as ToolTraceMissingError, u as domainEvidencePattern, v as OTEL_AGENT_EVAL_SCOPE, w as captureFetchToRawSink, x as traceIdForWire, y as exportRunAsOtlp } from "./store-tool-spans-CeNj_m2L.js";
10
+ import { C as toOtlpAttributes, S as msToUnixNano, _ as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, a as spanStatusToOtlp, b as spanIdForWire, c as defaultTraceInsightPanel, d as inferDomainKeywords, f as planTraceInsightQuestions, g as TRACE_ANALYST_ACTOR_DESCRIPTION, h as analyzeTraces, i as epochMillisToIso, l as describeTraceInsightScope, m as tokenizeDomainWords, n as toolSpansToTraceAnalysisStore, o as buildTraceInsightContext, p as scoreTraceInsightReadiness, r as createOtlpFlatLine, s as buildTraceInsightPrompt, t as ToolTraceMissingError, u as domainEvidencePattern, v as OTEL_AGENT_EVAL_SCOPE, w as captureFetchToRawSink, x as traceIdForWire, y as exportRunAsOtlp } from "./store-tool-spans-B9tjys_h.js";
11
11
  import { n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
12
12
  import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
13
- import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-Dow0pk_5.js";
14
- import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "./task-failure-attributes-CZjZeBsY.js";
13
+ import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-DV_H2HDu.js";
14
+ import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "./task-failure-attributes-CUy9mkIY.js";
15
15
  import { readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
16
16
  import { join } from "node:path";
17
17
  //#region src/trace/otel-bridge.ts
@@ -571,7 +571,7 @@ interface GenerationCandidate {
571
571
  surfaceHash: string;
572
572
  /** Mean over complete task-quality scores, or null when none were produced. */
573
573
  composite: number | null;
574
- /** Descriptive interval for `composite`, or null when no score exists. */
574
+ /** Estimated interval for `composite`, or null when uncertainty was not estimated. */
575
575
  ci95: [number, number] | null;
576
576
  /** Exact surface this candidate mutated. */
577
577
  parentSurfaceHash?: string;
@@ -670,4 +670,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
670
670
  }
671
671
  //#endregion
672
672
  export { LabeledScenarioWrite as A, ScoredSurfaceOutcome as B, JudgeDimension as C, LabeledScenarioSampleArgs as D, LabeledScenarioRecord as E, ProposeContext as F, labelTrustRank as G, SurfaceProposer as H, ProposedCandidate as I, RedactionStatus as L, OptimizerConfig as M, ParetoParent as N, LabeledScenarioSource as O, ProposalTrackContext as P, Scenario as R, JudgeConfig as S, LabelTrust as T, TraceSpan as U, SessionScript as V, isProposedCandidate as W, GateDecision as _, CampaignResult as a, GenerationRecord as b, CampaignTraceWriter as c, DispatchContext as d, DispatchFn as f, GateContribution as g, GateContext as h, CampaignCostMeter as i, MutableSurface as j, LabeledScenarioStore as k, CodeSurface as l, GateCheckStatus as m, CampaignArtifactWriter as n, CampaignScenarioIdentity as o, Gate as p, CampaignCellResult as r, CampaignTokenUsage as s, CampaignAggregates as t, ComponentSurface as u, GateResult as v, JudgeScore as w, JudgeAggregate as x, GenerationCandidate as y, ScenarioAggregate as z };
673
- //# sourceMappingURL=types-Ba5UQyVD.d.ts.map
673
+ //# sourceMappingURL=types-BJz2CPTM.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types-Ba5UQyVD.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;;UAiCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;;EAEA;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;;;;EAOV;;;;EAIA,eAAe,eAAe;IAAQ;IAAe;;;;;EAIrD;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;;;EAMA,cAAc,SAAS;;;;iBAKT,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;;;WAGjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB,gBAAgB;;;;WAIhB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;;;EAIA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;;;;;EAMA,cAAc;;UAGC;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;;;EAIA,cAAc,SAAS;;UAGR;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
1
+ {"version":3,"file":"types-BJz2CPTM.d.ts","names":[],"sources":["../src/campaign/types.ts"],"mappings":";;;;;;;;UAiCiB;EACf;EACA;EACA;;;;;EAKA;;;UAIe,iCAAiC,KAAK;EACrD;;;;;UAMe;EACf;;EAEA;EACA;EACA;EACA;EACA,QAAQ;EACR,OAAO;EACP,WAAW;EACX,MAAM;;EAEN;;EAEA;;;;;;;;EAQA;;;;KAKU,WAAW,kBAAkB,UAAU,cACjD,UAAU,WACV,KAAK,oBACF,QAAQ;;;;UAOI,cAAc,WAAW;EACxC;EACA;EACA;;EAEA;;;EAGA,sBAAsB,UAAU,WAAW,sBAAsB,UAAU,cAAc;;UAK1E;;EAEf;;EAEA;;;;;;;;;UAUe,YAAY,WAAW,kBAAkB,WAAW;EACnE;EACA,YAAY;;;;EAIZ;;;EAGA,MAAM;IACJ,UAAU;IACV,UAAU;IACV,QAAQ;;IAER,aAAa;IACb;IACA,WAAW;MACT,aAAa,QAAQ;EACzB,aAAa,UAAU;;;;;;;;;;;UAYR;EACf,YAAY;EACZ;EACA;;EAEA,UAAU;;;;;;;EAOV;;;;EAIA,eAAe,eAAe;IAAQ;IAAe;;;;;EAIrD;;;EAGA;;EAEA;;EAEA,WAAW,eAAe;;;;;;;UAUX;WACN;;;WAGA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;;WAGA;aACE;aACA;aACA;;;WAGF;;;UAIM;WACN;WACA,YAAY,SAAS;;;;;;;;;;;KAYpB,0BAA0B,mBAAmB;;;;;;;UAQxC;EACf,SAAS;;EAET;;;;EAIA;;;;;;EAMA,cAAc,SAAS;;;;iBAKT,oBACd,OAAO,iBAAiB,oBACvB,SAAS;;;;;;;;;UAkBK;EACf,SAAS;EACT;;;EAGA,YAAY;;;EAGZ;;EAEA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;;;UAMe;;;EAGf;;EAEA;EACA;EACA;EACA,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;EAC1E;IACE;IACA;;;;;UAMa,eAAe,YAAY;;;WAGjC,gBAAgB;WAChB,SAAS,cAAc;WACvB,UAAU,cAAc;;WAExB;WACA;WACA,QAAQ;;WAER,QAAQ;;;WAGR,kBAAkB;;;WAGlB,mBAAmB;;;;WAInB,gBAAgB;;;;WAIhB;;;;;;;WAOA,gBAAgB,cAAc;;WAE9B,aAAa;WACb;;;;;;;;;;;;;;UAeM,gBAAgB,YAAY;EAC3C;;;;;EAKA,QAAQ,KAAK,eAAe,aAAa,QAAQ,MAAM,iBAAiB;;;EAGxE,QAAQ;IAAQ,SAAS,cAAc;;IAAwB;IAAe;;;UAG/D;EACf;EACA;EACA,mBAAmB,qBAAqB;;UAGzB,wBAAwB;EACvC,UAAU;;;KAMA;;KAGA;UAEK;EACf;EACA,QAAQ;EACR;;UAGe,YAAY,WAAW,kBAAkB;EACxD,oBAAoB,YAAY;EAChC,oBAAoB,YAAY;;EAEhC,aAAa,YAAY,eAAe;;;;;EAKxC,sBAAsB,YAAY,eAAe;;;;;;;EAOjD,yBAAyB,YAAY,eAAe;;;EAGpD,uBAAuB,YAAY;EACnC,WAAW;EACX;IAAQ;IAAmB;;;EAE3B,aAAa;EACb;EACA,QAAQ;;UAGO;EACf,UAAU;EACV;EACA,mBAAmB;EACnB;;;UAIe,KAAK,qBAAqB,kBAAkB,WAAW;EACtE;EACA,OAAO,KAAK,YAAY,WAAW,aAAa,QAAQ;;;;UAOzC;EACf,KAAK,cAAc,aAAa,0BAA0B;EAC1D,SAAS;;UAGM;EACf,IAAI,aAAa;EACjB,aAAa,aAAa;;;;UAKX;EACf,MAAM,cAAc,kBAAkB,aAAa;EACnD,UAAU,cAAc,iBAAiB;;;;;;KAO/B,qBAAqB;;;;;UAMhB;;EAEf,YAAY,GACV,OAAO,KAAK,iBAAiB;IAC3B,UAAU;MAEX,QAAQ,eAAe;;;;;KAQhB;KAOA;;;;;;;;;;;;;;;KAgBA;;iBASI,eAAe,OAAO;;;;UAOrB,qBAAqB,kBAAkB,WAAW,UAAU;EAC3E,UAAU;EACV,UAAU;EACV,aAAa,eAAe;EAC5B,QAAQ;EACR;EACA;EACA,iBAAiB;;;;;EAKjB,aAAa;;EAEb;;UAGe,sBAAsB,kBAAkB,WAAW,UAAU,6BACpE,qBAAqB,WAAW;;EAExC;;;EAGA;;UAGe;EACf;;EAEA;;;;EAIA;EACA;IACE;IACA,SAAS,wBAAwB;IACjC;IACA;;;;;IAKA,WAAW;;;UAIE;EACf,QAAQ,OAAO,uBAAuB;EACtC,OAAO,MAAM,4BAA4B,QAAQ;EACjD,QAAQ;IACN;IACA;IACA,UAAU;;;IAGV,SAAS,OAAO;;;UAMH,mBAAmB;;;EAGlC;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,aAAa,eAAe;;;EAG5B;;EAEA,gBAAgB;;EAEhB;;;EAGA,YAAY;;;EAGZ;;;EAGA;EACA;EACA;EACA;;;;EAIA;;EAEA;;EAEA;EACA;;UAGe;EACf;EACA;EACA;EACA;;;;;;EAMA,cAAc;;UAGC;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,YAAY;EACZ;;;;;;UAOe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;;;EAIA;;;;EAIA;IACE;IACA;IACA,iBAAiB;MAAQ;MAAgB;;;;;EAI3C,YAAY;;;;;;;;;;EAUZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;EAG1E;;;;EAIA;;;;EAIA,cAAc,SAAS;;UAGR;EACf,SAAS,eAAe;EACxB,YAAY,eAAe;;EAE3B,MAAM;;EAEN;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,eAAe,qBAAqB,kBAAkB,WAAW;;EAEhF;;EAEA;EACA;;EAEA;EACA;EACA;EACA;EACA,OAAO,MAAM,mBAAmB;EAChC,YAAY;EACZ;IACE,aAAa;IACb;;EAEF,OAAO;EACP;EACA;EACA,iBAAiB;;;EAGjB,WAAW,MAAM,2BAA2B,KAAK"}
@@ -87,7 +87,7 @@ function assertNonNegativeFinite(value, field, context) {
87
87
  //#region src/analyst/types.ts
88
88
  /**
89
89
  * Analyst contract — the missing orchestration layer over agent-eval's
90
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
90
+ * existing analyzers (analyzeTraces, MultiLayerVerifier,
91
91
  * SemanticConceptJudge, JudgeFn, ...).
92
92
  *
93
93
  * Each existing primitive returns its own output shape. The Analyst
@@ -160,4 +160,4 @@ function makeProposalFinding(init) {
160
160
  //#endregion
161
161
  export { assertValidAnalystUsageReceipt as a, validateUsageSettlementTimeout as c, makeProposalFinding as i, deepFreezeCanonicalJson as l, computeFindingId as n, settleUsageReceiptFromCostLedger as o, makeFinding as r, usageReceiptFromCostLedger as s, ANALYST_SEVERITIES as t };
162
162
 
163
- //# sourceMappingURL=types-CiWITkGo.js.map
163
+ //# sourceMappingURL=types-DQ0e2E7y.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-DQ0e2E7y.js","names":[],"sources":["../src/ledger-core/deep-freeze.ts","../src/analyst/usage-receipt.ts","../src/analyst/types.ts"],"sourcesContent":["/** Freeze a detached canonical-JSON graph. Canonicalization has already ruled out cycles.\n *\n * Lives outside canonical.ts so the analyst-benchmark implementation digest,\n * which covers canonical.ts, stays bound to the published benchmark evidence. */\nexport function deepFreezeCanonicalJson<T>(value: T): T {\n if (value && typeof value === 'object' && !Object.isFrozen(value)) {\n Object.freeze(value)\n for (const nested of Object.values(value)) deepFreezeCanonicalJson(nested)\n }\n return value\n}\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n if (receipt.partialTokens) {\n const { input, output } = receipt.partialTokens\n if (receipt.tokens) {\n throw new Error(`${context}: partialTokens must be absent when tokens is complete`)\n }\n if (input === null && output === null) {\n throw new Error(`${context}: partialTokens must carry at least one reported side`)\n }\n if (input !== null) assertNonNegativeSafeInteger(input, 'partialTokens.input', context)\n if (output !== null) assertNonNegativeSafeInteger(output, 'partialTokens.output', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n","/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\n/** Finding severity. `AnalystSeverity` derives from this array, so the type\n * and the schema that validates a finding cannot name different levels. */\nexport const ANALYST_SEVERITIES = ['critical', 'high', 'medium', 'low', 'info'] as const\n\nexport type AnalystSeverity = (typeof ANALYST_SEVERITIES)[number]\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n /**\n * Optional live-execution port. A runtime that owns a sandbox or checkout\n * fills it so an analyst can execute a bounded probe against the run's\n * produced state instead of reasoning about it from the trace alone. This\n * package defines only the port: no field here reaches for an agent loop,\n * and an absent probe means the analyst works from recorded evidence.\n */\n probe?: ExecutionProbe\n}\n\n// ── Live-execution port ─────────────────────────────────────────────\n\n/** One bounded command an analyst asks the probe to run. */\nexport interface ExecutionProbeRequest {\n command: string\n /** Working directory inside the probed environment. */\n cwd?: string\n /** Hard wall-clock deadline for this one execution. */\n timeoutMs: number\n /** Bytes of combined output retained; the prober truncates beyond it. */\n maxOutputBytes?: number\n signal?: AbortSignal\n}\n\n/**\n * Typed outcome of one probe execution. `succeeded: false` is a PROBE failure\n * (the environment could not run the command); a command that ran and exited\n * non-zero is a successful observation with a non-zero `exitCode`.\n */\nexport type ExecutionProbeOutcome =\n | {\n succeeded: true\n exitCode: number\n stdout: string\n stderr: string\n durationMs: number\n /** True when output was cut at `maxOutputBytes`. */\n truncated: boolean\n }\n | { succeeded: false; error: { class: string; message: string } }\n\n/**\n * The seam a runtime fills to let analysts observe produced state live.\n * Implementations own sandboxing, credentials, and cleanup; analysts only\n * submit bounded requests and read typed outcomes.\n */\nexport interface ExecutionProbe {\n /** One plain sentence naming what is being probed (e.g. a sandbox id). */\n readonly description: string\n execute(request: ExecutionProbeRequest): Promise<ExecutionProbeOutcome>\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n /**\n * Token counts the provider reported on only one side. Present exactly when\n * `tokens` is null and at least one side WAS reported: `RunTokenUsage` has no\n * nullable side, so a one-sided count cannot live in `tokens` without writing\n * a zero nobody measured. Read it as a lower bound, never as a total — the\n * field exists so a null `tokens` cannot hide a real count.\n */\n partialTokens?: { input: number | null; output: number | null }\n /**\n * True when the token counts were DERIVED by the transport (from character\n * lengths, say) rather than measured by the model provider. `cost.kind` is\n * `estimated` both for a rate estimate over exact tokens and for one over\n * derived tokens; this is the field that separates them.\n */\n tokensEstimated?: boolean\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (a recursive engine returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n"],"mappings":";;;;;;AAIA,SAAgB,wBAA2B,OAAa;CACtD,IAAI,SAAS,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;EACjE,OAAO,OAAO,KAAK;EACnB,KAAK,MAAM,UAAU,OAAO,OAAO,KAAK,GAAG,wBAAwB,MAAM;CAC3E;CACA,OAAO;AACT;;ACJA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;CAEvE,IAAI,QAAQ,eAAe;EACzB,MAAM,EAAE,OAAO,WAAW,QAAQ;EAClC,IAAI,QAAQ,QACV,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;EAEpF,IAAI,UAAU,QAAQ,WAAW,MAC/B,MAAM,IAAI,MAAM,GAAG,QAAQ,sDAAsD;EAEnF,IAAI,UAAU,MAAM,6BAA6B,OAAO,uBAAuB,OAAO;EACtF,IAAI,WAAW,MAAM,6BAA6B,QAAQ,wBAAwB,OAAO;CAC3F;AACF;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E;;;;;;;;;;;;;;;;;;;;;AC3DA,MAAa,qBAAqB;CAAC;CAAY;CAAQ;CAAU;CAAO;AAAM;;;;;;;;;AAiO9E,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD"}
@@ -1,5 +1,5 @@
1
1
  import { s as RunSplitTag } from "./run-record-DTv1MdjK.js";
2
- import { R as Scenario, d as DispatchContext } from "./types-Ba5UQyVD.js";
2
+ import { R as Scenario, d as DispatchContext } from "./types-BJz2CPTM.js";
3
3
  //#region src/benchmarks/types.d.ts
4
4
  type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
5
5
  type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
@@ -90,4 +90,4 @@ declare const BENCHMARK_SPLIT_SEED = "agent-eval-v1";
90
90
  declare function deterministicSplit(itemId: string, seed?: string): RunSplitTag;
91
91
  //#endregion
92
92
  export { BenchmarkFamily as a, BenchmarkSource as c, BenchmarkEvaluation as i, BenchmarkTaskKind as l, BenchmarkAdapter as n, BenchmarkResponder as o, BenchmarkDatasetItem as r, BenchmarkScenario as s, BENCHMARK_SPLIT_SEED as t, deterministicSplit as u };
93
- //# sourceMappingURL=types-BDV4PiMR.d.ts.map
93
+ //# sourceMappingURL=types-Dd1ejaeI.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types-BDV4PiMR.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
1
+ {"version":3,"file":"types-Dd1ejaeI.d.ts","names":[],"sources":["../src/benchmarks/types.ts"],"mappings":";;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC"}
@@ -107,8 +107,19 @@ interface SupervisorRunSources {
107
107
  readonly workers: readonly WorkerLogSource[] | null;
108
108
  /** Why `workers` is null (only set when it is). */
109
109
  readonly workersMissingReason: string | null;
110
- /** Run result document (JSON). */
110
+ /**
111
+ * Run result document (JSON). For a Runtime run dir this is `result.json`,
112
+ * the `SupervisedResult` that `supervise()` returned, verbatim; its `kind`
113
+ * is the run's status. For a loops run dir it is the legacy result document.
114
+ */
111
115
  readonly result: string | null;
116
+ /**
117
+ * Runtime's terminal failure record (`failure.json`), written when
118
+ * `supervise()` threw before a result landed. `null` = the store was read and
119
+ * holds no such record; `undefined` = the store has no failure document at
120
+ * all (the loops layout), which the analyzer reports as its own absence.
121
+ */
122
+ readonly failure?: string | null;
112
123
  /**
113
124
  * Judge verdict document (JSON), or the matching ledger row re-encoded as
114
125
  * one. Runners that write the verdict straight to a ledger leave no judge
@@ -350,8 +361,59 @@ interface PatchStats {
350
361
  readonly linesRemoved: number;
351
362
  readonly testFilesTouched: readonly string[];
352
363
  }
364
+ /** Which record a run's terminal status was read from, so no source is a silent substitution. */
365
+ type SupervisorStatusSource =
366
+ /** Runtime's `result.json` `kind`: the `SupervisedResult` discriminant, read verbatim. */
367
+ 'runtime-result' |
368
+ /** Runtime's `failure.json`: `supervise()` threw before a result landed. */
369
+ 'runtime-failure' |
370
+ /** Control-plane-era loops `state.json` `status`. */
371
+ 'legacy-state' |
372
+ /** Control-plane-era loops `result.json` `sup_status`. */
373
+ 'legacy-result';
374
+ /** The error a run directory recorded. */
375
+ interface TerminalFailure {
376
+ /**
377
+ * Which record carried the error: Runtime's `failure.json`, or the driver
378
+ * rejection inside a `driver-failed` no-winner `result.json`.
379
+ */
380
+ readonly source: 'runtime-failure' | 'runtime-result';
381
+ /** `error.name` exactly as recorded; null when the record has none. */
382
+ readonly name: string | null;
383
+ /** `error.message` exactly as recorded; null when the record has none. */
384
+ readonly message: string | null;
385
+ /** ISO timestamp the record carries; null when it carries none. */
386
+ readonly at: string | null;
387
+ /**
388
+ * True when the directory also holds a settled `result.json`. Runtime writes
389
+ * `failure.json` only when `supervise()` threw, never after a settle, and
390
+ * refuses to re-enter a settled directory, so a failure beside a result is
391
+ * the throw of an earlier attempt that a later attempt outlived. The status
392
+ * is the settled result's; this record explains the retry, not the outcome.
393
+ */
394
+ readonly earlierAttempt: boolean;
395
+ }
353
396
  interface OutcomeMetrics {
397
+ /**
398
+ * The run's terminal status, exactly as its record spells it: Runtime's
399
+ * `winner` / `no-winner`, `failed` for a Runtime failure record, or the
400
+ * legacy loops status. `supStatusSource` names the record it came from.
401
+ */
354
402
  readonly supStatus: Measured<string>;
403
+ readonly supStatusSource: Measured<SupervisorStatusSource>;
404
+ /**
405
+ * Runtime's `reason` on a `no-winner` result (`all-children-down`,
406
+ * `budget-exhausted`, `aborted`, `driver-failed`). `null` = the terminal
407
+ * record carries no reason (a winner, a failure record, a legacy document).
408
+ */
409
+ readonly supReason: Measured<string | null>;
410
+ /**
411
+ * The recorded error: Runtime's `failure.json`, or the driver rejection a
412
+ * `driver-failed` no-winner carries. `null` = the run recorded a result and
413
+ * no error. A `failure.json` beside a settled result is reported with
414
+ * `earlierAttempt: true`. Unavailable when the store has no terminal record.
415
+ */
416
+ readonly failure: Measured<TerminalFailure | null>;
355
417
  readonly supVerdict: Measured<string>;
356
418
  readonly delivered: Measured<boolean>;
357
419
  readonly judgeResolved: Measured<boolean | null>;
@@ -449,5 +511,5 @@ interface SupervisorRunTreeGap {
449
511
  readonly count?: number;
450
512
  }
451
513
  //#endregion
452
- export { unavailable as A, SupervisorRunTreeGap as C, WorkerLogSource as D, WallDistribution as E, isUnavailable as O, SupervisorRunTree as S, Unavailable as T, SupervisorRunNodeRole as _, OrchestrationMetrics as a, SupervisorRunRollup as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SteerBreakdown as g, SpendMeasurements as h, NO_SOURCE_LIMITS as i, showMeasured as k, RoleSpend as l, SpendMeasurement as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunReader as v, SupervisorRunTreeGapCode as w, SupervisorRunSources as x, SupervisorRunReport as y };
453
- //# sourceMappingURL=types-CoPUTiXb.d.ts.map
514
+ export { isUnavailable as A, SupervisorRunTreeGap as C, Unavailable as D, TerminalFailure as E, unavailable as M, WallDistribution as O, SupervisorRunTree as S, SupervisorStatusSource as T, SupervisorRunNodeRole as _, OrchestrationMetrics as a, SupervisorRunRollup as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SteerBreakdown as g, SpendMeasurements as h, NO_SOURCE_LIMITS as i, showMeasured as j, WorkerLogSource as k, RoleSpend as l, SpendMeasurement as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunReader as v, SupervisorRunTreeGapCode as w, SupervisorRunSources as x, SupervisorRunReport as y };
515
+ //# sourceMappingURL=types-vUdAx2Cj.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-vUdAx2Cj.d.ts","names":[],"sources":["../src/supervisor-run/types.ts"],"mappings":";;;;UAkCiB;WACN;;;KAIC,SAAS,KAAK,IAAI;iBAEd,YAAY,iBAAiB;iBAI7B,cAAc,aAAa,KAAK;;iBAKhC,aAAa,GAAG;;KAWpB;;;;;UAMK;;;;;WAKN;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;WACA;WACA;WACA;;;;;;;;;;;;;UAcM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;;cAIE,kBAAkB;;;;;;;;;;UAiBd;;WAEN;WACA;;WAEA;;WAEA;;;;;;WAMA;;;;;;WAMA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA,kBAAkB;;WAElB;;;;;;WAMA;;;;;;;WAOA;;;;;;;WAOA;;WAEA;;WAEA;;WAEA;;;;;WAKA;IACP;IACA;IACA;IACA;;IAEA;IACA;;WAEO;;WAEA,QAAQ;;;;;;WAMR;;;;;WAKA;;;;;;UAOM;;WAEN;EACT,QAAQ,QAAQ;;cAOL;cACA;UAEI;;WAEN;WACA;;WAEA;;WAEA;;UAGM;WACN,gBAAgB;WAChB,gBAAgB;WAChB,kBAAkB;;WAElB,QAAQ;WACR,iBAAiB;WACjB,gBAAgB,kBAAkB;;WAElC,kBAAkB;;;;;;WAMlB,OAAO;WACP,WAAW;WACX,gBAAgB;;WAEhB,UAAU;;WAEV,gBAAgB;;WAEhB,iBAAiB;WACjB,oBAAoB;WACpB,kBAAkB;;;;;;;;;WASlB,sBAAsB;;WAEtB,QAAQ;WACR,SAAS;;WAET,mBAAmB;;UAGb;WACN,iBAAiB,SAAS;WAC1B,iBAAiB,SAAS;;;;;;;WAO1B,UAAU;;WAEV,UAAU;;WAEV,WAAW;;WAEX,oBAAoB;;WAEpB,wBAAwB;;WAExB,eAAe;WACf,qBAAqB;;UAGf;WACN,UAAU;WACV,WAAW;;;;;;WAMX,WAAW;WACX,YAAY;WACZ,KAAK;WACL;;UAGM;;WAEN;WACA;;WAEA,MAAM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;WACA;;WAEA;WACA;WACA;WACA;WACA;;WAEA;;;;;;;KAQC,mBAAmB;;UAGd;WACN,KAAK;;WAEL;;;;;;;WAOA;;WAEA;;WAEA;;;;;;;;;;;;;;UAeM;WACN,gBAAgB;WAChB,aAAa;;UAGP;;WAEN,OAAO;;;;;;;;WAQP,kBAAkB;;WAElB,SAAS;;WAET,OAAO;;;;;;;;WAQP,UAAU;;;;;;WAMV;WACA,yBAAyB;WACzB,0BAA0B,SAAS;WACnC,WAAW,kBAAkB;;UAGvB;WACN;WACA;WACA;WACA;;;KAIC;;;;;;;;;;UAWK;;;;;WAKN;;WAEA;;WAEA;;WAEA;;;;;;;;WAQA;;UAGM;;;;;;WAMN,WAAW;WACX,iBAAiB,SAAS;;;;;;WAM1B,WAAW;;;;;;;WAOX,SAAS,SAAS;WAClB,YAAY;WACZ,WAAW;WACX,eAAe;WACf,YAAY;WACZ,aAAa;WACb,YAAY;WACZ,YAAY;WACZ,UAAU;WACV,OAAO,SAAS;;WAEhB;;UAGM;WACN,eAAe;;WAEf;WACA;WACA;WACA,cAAc;WACd,yBAAyB;WACzB;WACA,eAAe;WACf,UAAU;WACV,WAAW;WACX,SAAS;;WAET;;WAEA;;UAGM;WACN;WACA;WACA,QAAQ;WACR,OAAO;WACP,aAAa;WACb,SAAS;WACT,UAAU;WACV,KAAK;;UAGC;WACN,eAAe;WACf;WACA,aAAa;WACb,iBAAiB;WACjB;WACA,WAAW;WACX,mBAAmB;WACnB,iBAAiB;WACjB,aAAa;WACb,qBAAqB;WACrB,eAAe;;;;;WAKf,UAAU;;;;;;;WAOV;aACE;eAA2B,OAAO;eAA2B;;aAC7D;eAAwB,OAAO;eAA2B;;;WAE5D,eAAe;WACf,kBAAkB;;;;;;;;UASZ;WACN;WACA,gBAAgB;;WAEhB,eAAe;;;KAId;UASK;WACN,MAAM;WACN;WACA;WACA"}
@@ -20,6 +20,48 @@ Use it when one surface must get better.
20
20
  Use it when two or more methods must be compared at equal budget.
21
21
  Runnable versions: [`examples/self-improve-optimizer`](../examples/self-improve-optimizer/) and [`examples/compare-optimization-methods`](../examples/compare-optimization-methods/).
22
22
 
23
+ ## Read An Improvement Result
24
+
25
+ `selfImprove({ method })` executes the complete method once and measures its selected surface on final cases.
26
+ The method may select the unchanged baseline; that result returns `gateDecision: 'hold'` and an empty diff.
27
+ Agent Eval does not score train and selection cases again or choose a different surface after the method finishes.
28
+
29
+ The result type has two modes:
30
+
31
+ | Mode | Result | Search evidence | Cost |
32
+ |---|---|---|---|
33
+ | `proposer` | `SelfImproveProposerResult` | Native `raw.generations`, `generationsExplored`, and optional `searchHistory` | Shared `cost` ledger summary |
34
+ | `method` | `SelfImproveMethodResult` | Actual `raw.method` and its optional `searchHistory` | Combined method and final `cost`; receipt breakdown in `ledgerCost` |
35
+
36
+ Both types are exported from the package root and `/contract`.
37
+ `SelfImproveResult` is their union; branch on `result.mode` before reading mode-specific fields.
38
+ Calls with a concrete `method` or `proposer` infer the corresponding result type.
39
+ Method mode has no native generation count or fabricated native search measurements.
40
+ Its durable `method-provenance.json` uses schema `tangle.method-improvement` and records partition, measurement, and cost-receipt digests.
41
+ Proposer mode retains `LoopProvenanceRecord`.
42
+
43
+ When method holdout is deferred, `baseline` and `winner.compositeMean` are `null`, `lift` is absent, and the decision is `hold`.
44
+ The selected surface remains available in `winner.surface`.
45
+ Method cost preserves the larger of reported search spend and newly recorded search receipts, then adds final measurements without counting receipts twice.
46
+ Underreported spending and incomplete receipts remain explicit; `raw.method.cost` retains the original report.
47
+ Inspect `cost.accountingComplete` and `cost.incompleteReasons` before treating the known subtotal as complete spending.
48
+ The shared dollar limit controls calls admitted through the cost ledger; arbitrary off-ledger callbacks must enforce their own spending limits.
49
+
50
+ Native generation records report `ci95: null` because search does not estimate candidate uncertainty.
51
+ Final comparisons retain their independently computed statistics.
52
+ Every final case and replica must have complete execution and judge results before comparison.
53
+
54
+ ## Bind Cached Measurements To Their Evaluator
55
+
56
+ Candidate surface content is part of native search and final measurement identity.
57
+ Pass a stable `dispatchRef` for execution behavior outside that surface, such as the worker revision and tool configuration.
58
+ Change it when that behavior changes; function names cannot identify captured state.
59
+ Set `judgeVersion` when a judge's scoring behavior changes.
60
+
61
+ To reuse `premeasuredBaseline` in proposer mode, measure the same train cases, seed, replicas, execution revision, and judges.
62
+ The standalone campaign must use `dispatchRef: surfaceDispatchRef(baselineSurface, dispatchRef)` from `/campaign`.
63
+ Agent Eval refuses a prior baseline whose evaluator manifest differs.
64
+
23
65
  ## Adapt A Third-Party Text Optimizer
24
66
 
25
67
  `externalTextOptimizationMethod()` is the general adapter for a package that already owns text or component search.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.173.3",
3
+ "version": "0.175.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {