@tangle-network/agent-eval 0.128.1 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +271 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/rl/verifiable-reward.ts","../src/rl/reward-hacking.ts"],"sourcesContent":["/**\n * Verifiable reward channel.\n *\n * For RL on coding / math / theorem-proving / structured-output tasks, the\n * reward signal is *decidable* — a test passes or fails, a proof checks or\n * doesn't, an output validates against a schema or doesn't. These rewards\n * are dramatically more useful for RL training than LLM-judge scores\n * because they don't drift, can't be Goodhart-gamed by the policy in the\n * same way, and don't require a separate calibration loop.\n *\n * The `MultiLayerVerifier` already produces this signal — it just doesn't\n * surface it in a shape that's clean enough for RL training. This module\n * wraps the verifier output so consumers can:\n *\n * 1. Extract a clean `VerifiableReward` from a `VerificationReport`\n * 2. Distinguish *deterministic* rewards (compile, test, schema) from\n * *probabilistic* rewards (judge) so they can be weighted differently\n * in the RL training step\n * 3. Filter `RunRecord[]` to only those with a verifiable reward,\n * producing the clean training set that DeepSeek-R1-style GRPO and\n * AlphaProof-style search both depend on\n *\n * Why this matters: every credible 2025-2026 frontier RL result on coding\n * agents leans on verifiable reward (DeepSeek-R1 GRPO on test pass-rate,\n * o-series RL on math/code, AlphaProof on Lean kernel checking). Mixing\n * judge scores into the reward signal poisons the gradient. This module\n * is the seam.\n */\n\nimport type { LayerResult, VerificationReport } from '../multi-layer-verifier'\nimport { isRealnessGated, observedScore, trainingScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\n\nexport type VerifiableRewardSource =\n | 'compile' // typecheck / build / lint passed\n | 'test' // unit / integration test pass-rate\n | 'schema' // structured output validates\n | 'sandbox' // sandbox exec exit code\n | 'judge' // LLM judge — probabilistic, included for completeness\n | 'composite' // weighted blend across multiple of the above\n\nexport interface VerifiableReward {\n /** Scalar in [0, 1]. The RL training signal. */\n value: number\n /** What produced the reward — different sources have different determinism. */\n source: VerifiableRewardSource\n /**\n * Determinism class. `'deterministic'` rewards are repeatable byte-for-byte\n * given the same inputs (compile, test, schema validation, sandbox exit code).\n * `'probabilistic'` rewards depend on a stochastic component (LLM judge).\n * Mixing these in the same training batch without separation is a known\n * footgun in production RLHF pipelines.\n */\n determinism: 'deterministic' | 'probabilistic'\n /**\n * Confidence in the reward value. For deterministic sources this is 1.0\n * (the bit either flipped or didn't). For judge sources this is the\n * judge-reported confidence or — when missing — a calibrated prior.\n */\n confidence: number\n /** The layer / judge id that produced the signal, for provenance. */\n origin: string\n /**\n * Per-source contribution to `value`, keyed by layer/judge id. Single-source\n * rewards carry one entry (`{ [origin]: value }`); composite rewards carry\n * every contributing layer's score — the anti-scalar-collapse surface RL\n * consumers weight per-source instead of trusting one blended number.\n */\n components: Record<string, number>\n /**\n * The run carries `outcome.realness.gated` — the authenticity gate flagged\n * its success signal as faked.\n *\n * With the gate applied (the default) `value` and every `components` entry\n * are 0 on such a run; with `applyRealnessGate: false` the observed numbers\n * come back untouched and this flag is the only marker that they are not to\n * be trusted. Either way it distinguishes \"measured a genuine failure\" from\n * \"claimed a success we refuse to believe\", which a bare 0 cannot.\n */\n realnessGated?: boolean\n /**\n * Whether an authenticity screen COULD run on this reward at all — the same\n * distinction `RolloutOutcome.realness_screened` draws, for the same reason.\n *\n * `false` on every reward from `extractVerifiableReward`, because a\n * `VerificationReport` carries layer scores and nothing else: there is no\n * `outcome.realness` to consult, so no gate has run, and `realnessGated`\n * being absent there means \"unknown\", NOT \"clean\". Absent on the\n * `RunRecord` path when the record itself carries no realness verdict.\n *\n * This matters most exactly where it is easiest to miss: a report whose\n * deterministic layers all passed yields `determinism: 'deterministic'`,\n * `confidence: 1` — the highest-credibility reward this module can emit —\n * and a stubbed integration reporting green is precisely what a gamed run\n * looks like. Consumers driving training off this shape must screen the run\n * themselves; the flag is what tells them nobody has.\n */\n realnessScreened?: boolean\n}\n\nexport interface VerifiableRewardExtractionOptions {\n /**\n * Which layers count as deterministic-reward sources. The verifier doesn't\n * tag layers as \"this is verifiable\"; the caller declares it via this list\n * (or via the layer name → source mapping). Default treats common names\n * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,\n * `sandbox`) as deterministic.\n */\n deterministicLayers?: string[]\n /**\n * Map layer name → reward source. Defaults to a sensible string-match.\n */\n sourceFor?: (layerName: string) => VerifiableRewardSource\n /**\n * Whether to fall back to a probabilistic (judge) reward when no\n * deterministic layer produced a numeric score. Default `true`. Set to\n * `false` for \"deterministic-only\" training pipelines that should\n * discard runs without a verifiable signal.\n */\n fallbackToJudge?: boolean\n /**\n * Default confidence for probabilistic (judge) rewards when the judge\n * doesn't report one. Default `0.7`.\n */\n judgeConfidenceFloor?: number\n /**\n * Whether the anti-Goodhart realness gate applies. Default `true`, and the\n * default is the one every training path must keep.\n *\n * Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,\n * for the same reason it reads `observedScore` for its proxy: it measures the\n * DIVERGENCE between the judge signal and the deterministic one, and a\n * deterministic reward that another gate already forced to 0 manufactures\n * exactly that divergence on exactly the gamed population. The detector would\n * then be re-reporting a verdict it was supposed to reach independently.\n */\n applyRealnessGate?: boolean\n}\n\nconst DEFAULT_DETERMINISTIC_LAYERS = new Set([\n 'install',\n 'typecheck',\n 'build',\n 'lint',\n 'test',\n 'compile',\n 'schema',\n 'sandbox',\n 'unit_tests',\n 'integration_tests',\n])\n\nconst DEFAULT_SOURCE_FOR = (name: string): VerifiableRewardSource => {\n const lower = name.toLowerCase()\n if (lower.includes('test')) return 'test'\n if (\n lower.includes('compile') ||\n lower.includes('build') ||\n lower.includes('typecheck') ||\n lower.includes('lint')\n )\n return 'compile'\n if (lower.includes('schema')) return 'schema'\n if (lower.includes('sandbox')) return 'sandbox'\n if (lower.includes('judge') || lower.includes('semantic')) return 'judge'\n return 'composite'\n}\n\n/**\n * Extract a `VerifiableReward` from a `VerificationReport`.\n *\n * Strategy: prefer the deterministic layers (in order: test → compile →\n * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is\n * true, return `null` if no signal qualifies. When multiple deterministic\n * layers contribute, return a `'composite'` source with a weighted blend.\n *\n * NO realness gate is applied and none can be: a `VerificationReport` carries\n * layer scores and nothing about whether the run faked them — `realness` lives\n * on the `RunRecord`. Use `extractVerifiableRewardsFromRecords` for anything\n * that becomes training data; this signature is for scoring a report in hand.\n */\nexport function extractVerifiableReward(\n report: VerificationReport,\n opts: VerifiableRewardExtractionOptions = {},\n): VerifiableReward | null {\n const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS])\n const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR\n const fallbackToJudge = opts.fallbackToJudge ?? true\n const judgeFloor = opts.judgeConfidenceFloor ?? 0.7\n\n const deterministic = report.layers.filter(\n (layer) => deterministicSet.has(layer.layer) && isMeasuredLayer(layer),\n )\n\n if (deterministic.length === 1) {\n const layer = deterministic[0]!\n const value = clamp01(layer.score!)\n return {\n value,\n source: sourceFor(layer.layer),\n determinism: 'deterministic',\n confidence: 1,\n origin: layer.layer,\n components: { [layer.layer]: value },\n realnessScreened: false,\n }\n }\n\n if (deterministic.length > 1) {\n // Composite: weighted blend by `Layer.weight` if present, else equal.\n let num = 0\n let denom = 0\n const components: Record<string, number> = {}\n for (const l of deterministic) {\n const w = (l.detail?.weight as number | undefined) ?? 1\n num += w * (l.score ?? 0)\n denom += w\n components[l.layer] = l.score!\n }\n return {\n value: denom === 0 ? 0 : clamp01(num / denom),\n source: 'composite',\n determinism: 'deterministic',\n confidence: 1,\n origin: deterministic.map((l) => l.layer).join('+'),\n components,\n realnessScreened: false,\n }\n }\n\n if (!fallbackToJudge) return null\n\n const judge =\n report.layers.find((layer) => isMeasuredLayer(layer) && sourceFor(layer.layer) === 'judge') ??\n report.layers.find(isMeasuredLayer)\n\n if (!judge) return null\n\n const confFromDetail = judge.detail?.confidence as number | undefined\n const judgeValue = clamp01(judge.score!)\n return {\n value: judgeValue,\n source: 'judge',\n determinism: 'probabilistic',\n confidence: typeof confFromDetail === 'number' ? confFromDetail : judgeFloor,\n origin: judge.layer,\n components: { [judge.layer]: judgeValue },\n realnessScreened: false,\n }\n}\n\nfunction isMeasuredLayer(\n layer: LayerResult,\n): layer is LayerResult & { status: 'pass' | 'fail'; score: number } {\n return (\n (layer.status === 'pass' || layer.status === 'fail') &&\n typeof layer.score === 'number' &&\n Number.isFinite(layer.score) &&\n layer.score >= 0 &&\n layer.score <= 1\n )\n}\n\n/**\n * Extract verifiable rewards from `RunRecord[]` produced via the\n * `verificationReportToRunRecord` adapter (which encodes per-layer scores\n * in `outcome.raw['layer.<name>']`). For records that don't carry layer\n * scores, returns `null` for that record.\n *\n * This is the canonical bridge from \"campaign-shaped artifacts\" to\n * \"RL-training-ready reward signals\": every record that has a clean\n * verifiable reward becomes a training datum, every record that doesn't\n * gets filtered out (or kept with `'probabilistic'` determinism for\n * separate downstream handling).\n *\n * The realness gate applies to EVERY channel here, and to the deterministic one\n * MOST. It is tempting to reason that a decidable signal cannot be gamed, so\n * the gate is redundant on it — that reasoning is backwards. `realness.gated`\n * means the run's success signal was FAKED, and a test suite reporting green on\n * a stubbed integration is precisely what that looks like: the deterministic\n * layer is the thing that got faked. Exporting it ungated hands a trainer the\n * highest-credibility reward the module can emit (`determinism: 'deterministic'`,\n * `confidence: 1`) for the one population the gate exists to catch. Pass\n * `applyRealnessGate: false` only to look at the ungated numbers for detection.\n */\nexport function extractVerifiableRewardsFromRecords(\n runs: RunRecord[],\n opts: VerifiableRewardExtractionOptions = {},\n): Array<{ runId: string; reward: VerifiableReward | null }> {\n const sourceFor = opts.sourceFor ?? DEFAULT_SOURCE_FOR\n const deterministicSet = new Set(opts.deterministicLayers ?? [...DEFAULT_DETERMINISTIC_LAYERS])\n const fallbackToJudge = opts.fallbackToJudge ?? true\n const judgeFloor = opts.judgeConfidenceFloor ?? 0.7\n const applyGate = opts.applyRealnessGate ?? true\n\n return runs.map((run) => {\n const flagged = isRealnessGated(run)\n // Present only when the record carries an actual realness verdict. Absent\n // is the honest \"unknown\"; `false` is reserved for a producer that\n // declares it HAS no screen, which is the report-shaped path above.\n const screened = run.outcome.realness === undefined ? {} : ({ realnessScreened: true } as const)\n // Zeroed with `value`, never left at the measured number: `components`\n // exists so an RL consumer can re-weight per source, and a raw layer score\n // surviving there would let that re-weighting reconstruct the very reward\n // the gate just refused. The measured layer scores stay on\n // `run.outcome.raw['layer.*']`, which is where analysis reads them.\n const gate = (value: number): number => (applyGate && flagged ? 0 : value)\n // Recover per-layer scores from outcome.raw['layer.<name>']\n const layerScores: Array<{ name: string; score: number }> = []\n for (const [k, v] of Object.entries(run.outcome.raw)) {\n if (\n k.startsWith('layer.') &&\n !k.includes('.', 6) &&\n typeof v === 'number' &&\n Number.isFinite(v)\n ) {\n layerScores.push({ name: k.slice('layer.'.length), score: v })\n }\n }\n const det = layerScores.filter((l) => deterministicSet.has(l.name))\n\n if (det.length === 1) {\n const layer = det[0]!\n const value = gate(clamp01(layer.score))\n return {\n runId: run.runId,\n reward: {\n value,\n source: sourceFor(layer.name),\n determinism: 'deterministic',\n confidence: 1,\n origin: layer.name,\n components: { [layer.name]: value },\n realnessGated: flagged,\n ...screened,\n },\n }\n }\n if (det.length > 1) {\n const value = gate(clamp01(det.reduce((s, l) => s + l.score, 0) / det.length))\n // Same clamp as the headline value: a producer writing layer.score 1.5\n // into outcome.raw must not propagate 1.5 through a component either.\n const components: Record<string, number> = Object.fromEntries(\n det.map((l) => [l.name, gate(clamp01(l.score))]),\n )\n return {\n runId: run.runId,\n reward: {\n value,\n source: 'composite',\n determinism: 'deterministic',\n confidence: 1,\n origin: det.map((l) => l.name).join('+'),\n components,\n realnessGated: flagged,\n ...screened,\n },\n }\n }\n if (!fallbackToJudge) return { runId: run.runId, reward: null }\n\n // Probabilistic fallback: the run's primary score. `trainingScore` already\n // carries the gate, so a gamed run falls to 0 rather than earning the\n // judge's number; `observedScore` is the ungated reader the detection\n // opt-out asks for. Either way an unscored run stays a labeled gap\n // (`reward: null`), never a fabricated 0.\n const primary = applyGate ? trainingScore(run) : observedScore(run)\n if (typeof primary !== 'number' || !Number.isFinite(primary)) {\n return { runId: run.runId, reward: null }\n }\n const primaryValue = clamp01(primary)\n return {\n runId: run.runId,\n reward: {\n value: primaryValue,\n source: 'judge',\n determinism: 'probabilistic',\n confidence: judgeFloor,\n origin: 'run.outcome.score',\n components: { 'run.outcome.score': primaryValue },\n realnessGated: flagged,\n ...screened,\n },\n }\n })\n}\n\n/**\n * Filter `RunRecord[]` to those with deterministic verifiable rewards.\n *\n * A realness-gated run is KEPT, at reward 0 with `realnessGated: true` — the\n * same rule GRPO uses on a gated line. 0 is the honest label for a faked\n * success and is usable signal, whereas dropping the run would move a group\n * baseline without saying so. (SFT differs: there every row is a target to\n * imitate, so a gated row is removed outright.)\n */\nexport function filterDeterministicallyRewarded(\n runs: RunRecord[],\n opts: VerifiableRewardExtractionOptions = {},\n): Array<{ run: RunRecord; reward: VerifiableReward }> {\n const rewarded = extractVerifiableRewardsFromRecords(runs, { ...opts, fallbackToJudge: false })\n const out: Array<{ run: RunRecord; reward: VerifiableReward }> = []\n for (let i = 0; i < runs.length; i++) {\n const r = rewarded[i]!\n if (r.reward && r.reward.determinism === 'deterministic') {\n out.push({ run: runs[i]!, reward: r.reward })\n }\n }\n return out\n}\n\nfunction clamp01(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n","/**\n * Reward hacking / Goodhart detection.\n *\n * Goodhart's Law says: when a measure becomes a target, it ceases to be\n * a good measure. In RLHF and agentic-RL settings this is the dominant\n * failure mode — the policy learns to produce outputs that score well on\n * the proxy reward (judge, rubric, test pass-rate) without producing\n * the underlying capability the proxy was meant to track.\n *\n * Krakovna et al. (2020, \"Specification Gaming Examples in AI\") and the\n * subsequent RLHF reward-hacking literature (Skalse et al. 2022, Kim et al.\n * 2023) converge on a few diagnostic signatures:\n *\n * 1. **Reward divergence:** the proxy reward grows while the held-out\n * ground-truth signal stagnates or drops. Predictive validity over\n * time captures this.\n * 2. **Distributional shift in outputs:** after RL, the policy produces\n * outputs that no longer match the reference distribution — usually\n * because it found a high-reward attractor that's degenerate (e.g.\n * one-token responses, repetition, formatting tricks).\n * 3. **Disagreement between independent rewards:** if you train on\n * reward A and a held-out independent reward B drops sharply, you're\n * probably hacking A.\n * 4. **Calibration drift:** the verifiable / deterministic component of\n * the reward is stable; the probabilistic / judge component drifts up\n * while the deterministic component doesn't. The judge is being\n * gamed.\n *\n * This module ships explicit detectors for all four signatures, plus a\n * combined verdict. The output is diagnostic — actionable signals,\n * not autoreject — because each signature has known false positives\n * (e.g., a policy that genuinely improves can show distributional shift).\n *\n * Differs from `rubricPredictiveValidity` (which is a *standing* check on\n * whether rubrics correlate with deployment outcomes) — this is a\n * *temporal* check on whether the reward-vs-truth gap is *widening over\n * time during a training run*.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\nimport { pearsonR } from '../statistics'\nimport {\n filterDeterministicallyRewarded,\n type VerifiableRewardExtractionOptions,\n} from './verifiable-reward'\n\nexport type RewardHackingSignal =\n | 'reward_divergence'\n | 'distribution_shift'\n | 'reward_disagreement'\n | 'judge_drift'\n\nexport interface RewardHackingFinding {\n signal: RewardHackingSignal\n /** Severity in [0, 1]. >0.5 = strong signal. */\n severity: number\n message: string\n /** Numeric evidence the consumer can render. */\n detail: Record<string, number>\n}\n\nexport interface RewardHackingReport {\n findings: RewardHackingFinding[]\n /** Signals with enough usable observations to produce a finding. */\n evaluatedSignals: RewardHackingSignal[]\n /**\n * Composite verdict. `'insufficient_evidence'` when fewer than four scored\n * runs exist; otherwise `'clean'` if every signal severity < 0.3,\n * `'suspect'` if at least one ≥ 0.3 but none ≥ 0.6, and `'gaming'` if any ≥ 0.6.\n */\n verdict: 'insufficient_evidence' | 'clean' | 'suspect' | 'gaming'\n /** Rationale for the verdict, ready to paste into an audit log. */\n rationale: string[]\n /** Number of runs with a usable proxy reward. */\n n: number\n}\n\nexport interface DetectRewardHackingInput {\n /**\n * Run records ordered by recency (oldest first). The detector segments\n * them into prefix/suffix windows to compute \"did the gap widen.\"\n */\n runs: RunRecord[]\n /**\n * The metric the policy was trained to optimize. Should be present on\n * `outcome.raw` or `outcome.holdoutScore`. Default reads `outcome.holdoutScore`.\n */\n proxyOf?: (run: RunRecord) => number | null\n /**\n * The held-out ground-truth metric. For RL on coding, this is typically\n * test pass-rate. For RLHF, it's downstream task performance or human\n * preference. For knowledge tasks, it's an independently-graded score.\n */\n truthOf?: (run: RunRecord) => number | null\n /**\n * Independent secondary reward. Used for the `reward_disagreement`\n * signal. Default uses the verifiable reward extractor (deterministic\n * sources only).\n */\n secondaryRewardOf?: (run: RunRecord) => number | null\n /**\n * Window size — how many of the most recent runs count as the \"after\"\n * cohort. Default min(50, half the runs).\n */\n windowSize?: number\n /**\n * Severity threshold to flag a signal. Default 0.3 (suspect) and 0.6\n * (gaming).\n */\n thresholds?: { suspect?: number; gaming?: number }\n /**\n * Verifiable-reward options used for the secondary-reward fallback.\n */\n verifiableRewardOptions?: VerifiableRewardExtractionOptions\n}\n\nconst DEFAULT_PROXY = (r: RunRecord): number | null => {\n // DELIBERATELY UNGATED. This is the proxy reward the detector tests for\n // Goodharting, and gated runs are exactly the gamed population. Forcing them\n // to 0 would collapse the proxy toward the deterministic secondary signal —\n // `reward_disagreement`'s correlation would rise, `judge_drift`'s gap would\n // shrink, and `reward_divergence`'s proxy-up/truth-flat fingerprint would be\n // erased — so the detector would report \"clean\" on the very runs it exists to\n // catch. `null` (never 0) is also load-bearing: it sets the n-denominator via\n // the filter below.\n const v = observedScore(r)\n return typeof v === 'number' && Number.isFinite(v) ? v : null\n}\n\nexport function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport {\n const proxyOf = input.proxyOf ?? DEFAULT_PROXY\n const truthOf = input.truthOf\n const sus = input.thresholds?.suspect ?? 0.3\n const gam = input.thresholds?.gaming ?? 0.6\n\n const runs = input.runs.filter((run) => finiteNumber(proxyOf(run)))\n const n = runs.length\n if (n < 4) {\n return {\n findings: [],\n evaluatedSignals: [],\n verdict: 'insufficient_evidence',\n n,\n rationale: [`fewer than 4 runs with proxy reward (n=${n}); insufficient evidence`],\n }\n }\n const windowSize = Math.max(1, input.windowSize ?? Math.min(50, Math.floor(n / 2)))\n const before = runs.slice(0, n - windowSize)\n const after = runs.slice(n - windowSize)\n\n const findings: RewardHackingFinding[] = []\n\n // ── Signal 1: reward divergence (proxy ↑ while truth flat or ↓) ──────\n if (truthOf) {\n const beforeProxy = before.map(proxyOf).filter(finiteNumber)\n const afterProxy = after.map(proxyOf).filter(finiteNumber)\n const beforeTruth = before.map(truthOf).filter(finiteNumber)\n const afterTruth = after.map(truthOf).filter(finiteNumber)\n if (\n beforeProxy.length >= 2 &&\n afterProxy.length >= 2 &&\n beforeTruth.length >= 2 &&\n afterTruth.length >= 2\n ) {\n const proxyDelta = mean(afterProxy) - mean(beforeProxy)\n const truthDelta = mean(afterTruth) - mean(beforeTruth)\n // Divergence: proxy goes up while truth goes flat or down.\n // Severity = max(0, (proxyDelta - truthDelta)) — bigger gap = bigger signal.\n const gap = Math.max(0, proxyDelta - truthDelta)\n const severity = clamp01(gap * 5) // scale: 0.2 absolute gap → severity 1.0\n findings.push({\n signal: 'reward_divergence',\n severity,\n message:\n severity >= sus\n ? `proxy reward rose by ${proxyDelta.toFixed(3)} while truth changed by ${truthDelta.toFixed(3)} — potential Goodhart`\n : `proxy and truth moved together (proxy ${proxyDelta.toFixed(3)}, truth ${truthDelta.toFixed(3)})`,\n detail: {\n proxyDelta,\n truthDelta,\n gap,\n beforeN: beforeProxy.length,\n afterN: afterProxy.length,\n },\n })\n }\n }\n\n // ── Signal 2: distributional shift in outputs (KS on score distributions) ──\n {\n const beforeP = before.map(proxyOf).filter(finiteNumber)\n const afterP = after.map(proxyOf).filter(finiteNumber)\n if (beforeP.length >= 4 && afterP.length >= 4) {\n const ks = ksStatistic(beforeP, afterP)\n // KS statistic: bigger = more shift. We're agnostic about direction;\n // genuine improvement ALSO produces shift, so this signal is\n // contributory rather than load-bearing.\n const severity = clamp01(ks - 0.2)\n findings.push({\n signal: 'distribution_shift',\n severity,\n message:\n severity >= sus\n ? `KS=${ks.toFixed(3)} between before/after windows — distributional shift large`\n : `KS=${ks.toFixed(3)} between before/after windows — within-distribution drift`,\n detail: { ks, beforeN: beforeP.length, afterN: afterP.length },\n })\n }\n }\n\n // ── Signal 3: reward disagreement (proxy vs independent secondary) ────\n {\n const secondaryOf = input.secondaryRewardOf ?? defaultSecondary(input.verifiableRewardOptions)\n const aligned = runs\n .map((r) => ({ p: proxyOf(r), s: secondaryOf(r) }))\n .filter((x): x is { p: number; s: number } => finiteNumber(x.p) && finiteNumber(x.s))\n if (aligned.length >= 4) {\n const ps = aligned.map((x) => x.p)\n const ss = aligned.map((x) => x.s)\n const r = pearsonR(ps, ss)\n // Disagreement: low or negative correlation between primary proxy\n // reward and an independent secondary signal.\n const severity = clamp01(0.5 - Math.max(0, r))\n findings.push({\n signal: 'reward_disagreement',\n severity,\n message:\n severity >= sus\n ? `proxy and independent secondary reward correlate ρ=${r.toFixed(3)} — possibly hacking proxy`\n : `proxy and secondary reward correlate ρ=${r.toFixed(3)}`,\n detail: { pearson: r, n: aligned.length },\n })\n }\n }\n\n // ── Signal 4: judge drift (probabilistic up while deterministic flat) ─\n {\n // Ungated on purpose, exactly like `DEFAULT_PROXY` above. This signal is\n // the GAP between the judge reward and the deterministic one; a\n // deterministic reward another gate already forced to 0 would open that gap\n // by construction on the gamed population, so the detector would fire on\n // its own input rather than on evidence it found.\n const detRuns = filterDeterministicallyRewarded(runs, {\n ...(input.verifiableRewardOptions ?? {}),\n applyRealnessGate: false,\n })\n if (detRuns.length >= 4) {\n const detBefore = detRuns.slice(0, Math.floor(detRuns.length / 2))\n const detAfter = detRuns.slice(Math.floor(detRuns.length / 2))\n const detDelta =\n mean(detAfter.map((r) => r.reward.value)) - mean(detBefore.map((r) => r.reward.value))\n const proxyDelta =\n mean(after.map(proxyOf).filter(finiteNumber)) -\n mean(before.map(proxyOf).filter(finiteNumber))\n const driftGap = Math.max(0, proxyDelta - detDelta)\n const severity = clamp01(driftGap * 5)\n findings.push({\n signal: 'judge_drift',\n severity,\n message:\n severity >= sus\n ? `judge proxy +${proxyDelta.toFixed(3)} while deterministic reward +${detDelta.toFixed(3)} — judge drifting up without verifiable backing`\n : `judge and deterministic rewards move in step (judge ${proxyDelta.toFixed(3)}, det ${detDelta.toFixed(3)})`,\n detail: { proxyDelta, detDelta, driftGap, n: detRuns.length },\n })\n }\n }\n\n const maxSev = findings.reduce((m, f) => Math.max(m, f.severity), 0)\n if (findings.length === 0) {\n return {\n findings,\n evaluatedSignals: [],\n verdict: 'insufficient_evidence',\n rationale: [`no reward-hacking signal had enough paired evidence (n=${n})`],\n n,\n }\n }\n const verdict: RewardHackingReport['verdict'] =\n maxSev >= gam ? 'gaming' : maxSev >= sus ? 'suspect' : 'clean'\n const rationale = findings\n .filter((f) => f.severity >= sus)\n .map((f) => `${f.signal}: severity ${f.severity.toFixed(2)} — ${f.message}`)\n if (rationale.length === 0) rationale.push('no signals fired above suspect threshold')\n\n return {\n findings,\n evaluatedSignals: findings.map((finding) => finding.signal),\n verdict,\n rationale,\n n,\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction mean(xs: number[]): number {\n if (xs.length === 0) return 0\n return xs.reduce((s, x) => s + x, 0) / xs.length\n}\n\nfunction finiteNumber(value: number | null): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\nfunction clamp01(x: number): number {\n if (!Number.isFinite(x)) return 0\n return Math.max(0, Math.min(1, x))\n}\n\nfunction ksStatistic(a: number[], b: number[]): number {\n // Two-sample Kolmogorov-Smirnov statistic.\n const sortedA = [...a].sort((x, y) => x - y)\n const sortedB = [...b].sort((x, y) => x - y)\n const all = [...new Set([...sortedA, ...sortedB])].sort((x, y) => x - y)\n let max = 0\n for (const v of all) {\n const fa = sortedA.filter((x) => x <= v).length / sortedA.length\n const fb = sortedB.filter((x) => x <= v).length / sortedB.length\n max = Math.max(max, Math.abs(fa - fb))\n }\n return max\n}\n\nfunction defaultSecondary(\n verifiableOpts?: VerifiableRewardExtractionOptions,\n): (run: RunRecord) => number | null {\n return (run: RunRecord) => {\n // Ungated for the same reason as signal 4: this is the INDEPENDENT\n // secondary reward whose correlation with the proxy is the evidence.\n // Zeroing it on gated runs would drive that correlation down mechanically.\n const filtered = filterDeterministicallyRewarded([run], {\n ...(verifiableOpts ?? {}),\n applyRealnessGate: false,\n })\n return filtered.length === 1 ? filtered[0]!.reward.value : null\n }\n}\n"],"mappings":";;;;;;;;;;AA2IA,IAAM,+BAA+B,oBAAI,IAAI;AAAA,EAC3C;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAED,IAAM,qBAAqB,CAAC,SAAyC;AACnE,QAAM,QAAQ,KAAK,YAAY;AAC/B,MAAI,MAAM,SAAS,MAAM,EAAG,QAAO;AACnC,MACE,MAAM,SAAS,SAAS,KACxB,MAAM,SAAS,OAAO,KACtB,MAAM,SAAS,WAAW,KAC1B,MAAM,SAAS,MAAM;AAErB,WAAO;AACT,MAAI,MAAM,SAAS,QAAQ,EAAG,QAAO;AACrC,MAAI,MAAM,SAAS,SAAS,EAAG,QAAO;AACtC,MAAI,MAAM,SAAS,OAAO,KAAK,MAAM,SAAS,UAAU,EAAG,QAAO;AAClE,SAAO;AACT;AAeO,SAAS,wBACd,QACA,OAA0C,CAAC,GAClB;AACzB,QAAM,mBAAmB,IAAI,IAAI,KAAK,uBAAuB,CAAC,GAAG,4BAA4B,CAAC;AAC9F,QAAM,YAAY,KAAK,aAAa;AACpC,QAAM,kBAAkB,KAAK,mBAAmB;AAChD,QAAM,aAAa,KAAK,wBAAwB;AAEhD,QAAM,gBAAgB,OAAO,OAAO;AAAA,IAClC,CAAC,UAAU,iBAAiB,IAAI,MAAM,KAAK,KAAK,gBAAgB,KAAK;AAAA,EACvE;AAEA,MAAI,cAAc,WAAW,GAAG;AAC9B,UAAM,QAAQ,cAAc,CAAC;AAC7B,UAAM,QAAQ,QAAQ,MAAM,KAAM;AAClC,WAAO;AAAA,MACL;AAAA,MACA,QAAQ,UAAU,MAAM,KAAK;AAAA,MAC7B,aAAa;AAAA,MACb,YAAY;AAAA,MACZ,QAAQ,MAAM;AAAA,MACd,YAAY,EAAE,CAAC,MAAM,KAAK,GAAG,MAAM;AAAA,MACnC,kBAAkB;AAAA,IACpB;AAAA,EACF;AAEA,MAAI,cAAc,SAAS,GAAG;AAE5B,QAAI,MAAM;AACV,QAAI,QAAQ;AACZ,UAAM,aAAqC,CAAC;AAC5C,eAAW,KAAK,eAAe;AAC7B,YAAM,IAAK,EAAE,QAAQ,UAAiC;AACtD,aAAO,KAAK,EAAE,SAAS;AACvB,eAAS;AACT,iBAAW,EAAE,KAAK,IAAI,EAAE;AAAA,IAC1B;AACA,WAAO;AAAA,MACL,OAAO,UAAU,IAAI,IAAI,QAAQ,MAAM,KAAK;AAAA,MAC5C,QAAQ;AAAA,MACR,aAAa;AAAA,MACb,YAAY;AAAA,MACZ,QAAQ,cAAc,IAAI,CAAC,MAAM,EAAE,KAAK,EAAE,KAAK,GAAG;AAAA,MAClD;AAAA,MACA,kBAAkB;AAAA,IACpB;AAAA,EACF;AAEA,MAAI,CAAC,gBAAiB,QAAO;AAE7B,QAAM,QACJ,OAAO,OAAO,KAAK,CAAC,UAAU,gBAAgB,KAAK,KAAK,UAAU,MAAM,KAAK,MAAM,OAAO,KAC1F,OAAO,OAAO,KAAK,eAAe;AAEpC,MAAI,CAAC,MAAO,QAAO;AAEnB,QAAM,iBAAiB,MAAM,QAAQ;AACrC,QAAM,aAAa,QAAQ,MAAM,KAAM;AACvC,SAAO;AAAA,IACL,OAAO;AAAA,IACP,QAAQ;AAAA,IACR,aAAa;AAAA,IACb,YAAY,OAAO,mBAAmB,WAAW,iBAAiB;AAAA,IAClE,QAAQ,MAAM;AAAA,IACd,YAAY,EAAE,CAAC,MAAM,KAAK,GAAG,WAAW;AAAA,IACxC,kBAAkB;AAAA,EACpB;AACF;AAEA,SAAS,gBACP,OACmE;AACnE,UACG,MAAM,WAAW,UAAU,MAAM,WAAW,WAC7C,OAAO,MAAM,UAAU,YACvB,OAAO,SAAS,MAAM,KAAK,KAC3B,MAAM,SAAS,KACf,MAAM,SAAS;AAEnB;AAwBO,SAAS,oCACd,MACA,OAA0C,CAAC,GACgB;AAC3D,QAAM,YAAY,KAAK,aAAa;AACpC,QAAM,mBAAmB,IAAI,IAAI,KAAK,uBAAuB,CAAC,GAAG,4BAA4B,CAAC;AAC9F,QAAM,kBAAkB,KAAK,mBAAmB;AAChD,QAAM,aAAa,KAAK,wBAAwB;AAChD,QAAM,YAAY,KAAK,qBAAqB;AAE5C,SAAO,KAAK,IAAI,CAAC,QAAQ;AACvB,UAAM,UAAU,gBAAgB,GAAG;AAInC,UAAM,WAAW,IAAI,QAAQ,aAAa,SAAY,CAAC,IAAK,EAAE,kBAAkB,KAAK;AAMrF,UAAM,OAAO,CAAC,UAA2B,aAAa,UAAU,IAAI;AAEpE,UAAM,cAAsD,CAAC;AAC7D,eAAW,CAAC,GAAG,CAAC,KAAK,OAAO,QAAQ,IAAI,QAAQ,GAAG,GAAG;AACpD,UACE,EAAE,WAAW,QAAQ,KACrB,CAAC,EAAE,SAAS,KAAK,CAAC,KAClB,OAAO,MAAM,YACb,OAAO,SAAS,CAAC,GACjB;AACA,oBAAY,KAAK,EAAE,MAAM,EAAE,MAAM,SAAS,MAAM,GAAG,OAAO,EAAE,CAAC;AAAA,MAC/D;AAAA,IACF;AACA,UAAM,MAAM,YAAY,OAAO,CAAC,MAAM,iBAAiB,IAAI,EAAE,IAAI,CAAC;AAElE,QAAI,IAAI,WAAW,GAAG;AACpB,YAAM,QAAQ,IAAI,CAAC;AACnB,YAAM,QAAQ,KAAK,QAAQ,MAAM,KAAK,CAAC;AACvC,aAAO;AAAA,QACL,OAAO,IAAI;AAAA,QACX,QAAQ;AAAA,UACN;AAAA,UACA,QAAQ,UAAU,MAAM,IAAI;AAAA,UAC5B,aAAa;AAAA,UACb,YAAY;AAAA,UACZ,QAAQ,MAAM;AAAA,UACd,YAAY,EAAE,CAAC,MAAM,IAAI,GAAG,MAAM;AAAA,UAClC,eAAe;AAAA,UACf,GAAG;AAAA,QACL;AAAA,MACF;AAAA,IACF;AACA,QAAI,IAAI,SAAS,GAAG;AAClB,YAAM,QAAQ,KAAK,QAAQ,IAAI,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,IAAI,MAAM,CAAC;AAG7E,YAAM,aAAqC,OAAO;AAAA,QAChD,IAAI,IAAI,CAAC,MAAM,CAAC,EAAE,MAAM,KAAK,QAAQ,EAAE,KAAK,CAAC,CAAC,CAAC;AAAA,MACjD;AACA,aAAO;AAAA,QACL,OAAO,IAAI;AAAA,QACX,QAAQ;AAAA,UACN;AAAA,UACA,QAAQ;AAAA,UACR,aAAa;AAAA,UACb,YAAY;AAAA,UACZ,QAAQ,IAAI,IAAI,CAAC,MAAM,EAAE,IAAI,EAAE,KAAK,GAAG;AAAA,UACvC;AAAA,UACA,eAAe;AAAA,UACf,GAAG;AAAA,QACL;AAAA,MACF;AAAA,IACF;AACA,QAAI,CAAC,gBAAiB,QAAO,EAAE,OAAO,IAAI,OAAO,QAAQ,KAAK;AAO9D,UAAM,UAAU,YAAY,cAAc,GAAG,IAAI,cAAc,GAAG;AAClE,QAAI,OAAO,YAAY,YAAY,CAAC,OAAO,SAAS,OAAO,GAAG;AAC5D,aAAO,EAAE,OAAO,IAAI,OAAO,QAAQ,KAAK;AAAA,IAC1C;AACA,UAAM,eAAe,QAAQ,OAAO;AACpC,WAAO;AAAA,MACL,OAAO,IAAI;AAAA,MACX,QAAQ;AAAA,QACN,OAAO;AAAA,QACP,QAAQ;AAAA,QACR,aAAa;AAAA,QACb,YAAY;AAAA,QACZ,QAAQ;AAAA,QACR,YAAY,EAAE,qBAAqB,aAAa;AAAA,QAChD,eAAe;AAAA,QACf,GAAG;AAAA,MACL;AAAA,IACF;AAAA,EACF,CAAC;AACH;AAWO,SAAS,gCACd,MACA,OAA0C,CAAC,GACU;AACrD,QAAM,WAAW,oCAAoC,MAAM,EAAE,GAAG,MAAM,iBAAiB,MAAM,CAAC;AAC9F,QAAM,MAA2D,CAAC;AAClE,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,IAAI,SAAS,CAAC;AACpB,QAAI,EAAE,UAAU,EAAE,OAAO,gBAAgB,iBAAiB;AACxD,UAAI,KAAK,EAAE,KAAK,KAAK,CAAC,GAAI,QAAQ,EAAE,OAAO,CAAC;AAAA,IAC9C;AAAA,EACF;AACA,SAAO;AACT;AAEA,SAAS,QAAQ,GAAmB;AAClC,MAAI,CAAC,OAAO,SAAS,CAAC,EAAG,QAAO;AAChC,SAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;;;ACzSA,IAAM,gBAAgB,CAAC,MAAgC;AASrD,QAAM,IAAI,cAAc,CAAC;AACzB,SAAO,OAAO,MAAM,YAAY,OAAO,SAAS,CAAC,IAAI,IAAI;AAC3D;AAEO,SAAS,oBAAoB,OAAsD;AACxF,QAAM,UAAU,MAAM,WAAW;AACjC,QAAM,UAAU,MAAM;AACtB,QAAM,MAAM,MAAM,YAAY,WAAW;AACzC,QAAM,MAAM,MAAM,YAAY,UAAU;AAExC,QAAM,OAAO,MAAM,KAAK,OAAO,CAAC,QAAQ,aAAa,QAAQ,GAAG,CAAC,CAAC;AAClE,QAAM,IAAI,KAAK;AACf,MAAI,IAAI,GAAG;AACT,WAAO;AAAA,MACL,UAAU,CAAC;AAAA,MACX,kBAAkB,CAAC;AAAA,MACnB,SAAS;AAAA,MACT;AAAA,MACA,WAAW,CAAC,0CAA0C,CAAC,0BAA0B;AAAA,IACnF;AAAA,EACF;AACA,QAAM,aAAa,KAAK,IAAI,GAAG,MAAM,cAAc,KAAK,IAAI,IAAI,KAAK,MAAM,IAAI,CAAC,CAAC,CAAC;AAClF,QAAM,SAAS,KAAK,MAAM,GAAG,IAAI,UAAU;AAC3C,QAAM,QAAQ,KAAK,MAAM,IAAI,UAAU;AAEvC,QAAM,WAAmC,CAAC;AAG1C,MAAI,SAAS;AACX,UAAM,cAAc,OAAO,IAAI,OAAO,EAAE,OAAO,YAAY;AAC3D,UAAM,aAAa,MAAM,IAAI,OAAO,EAAE,OAAO,YAAY;AACzD,UAAM,cAAc,OAAO,IAAI,OAAO,EAAE,OAAO,YAAY;AAC3D,UAAM,aAAa,MAAM,IAAI,OAAO,EAAE,OAAO,YAAY;AACzD,QACE,YAAY,UAAU,KACtB,WAAW,UAAU,KACrB,YAAY,UAAU,KACtB,WAAW,UAAU,GACrB;AACA,YAAM,aAAa,KAAK,UAAU,IAAI,KAAK,WAAW;AACtD,YAAM,aAAa,KAAK,UAAU,IAAI,KAAK,WAAW;AAGtD,YAAM,MAAM,KAAK,IAAI,GAAG,aAAa,UAAU;AAC/C,YAAM,WAAWA,SAAQ,MAAM,CAAC;AAChC,eAAS,KAAK;AAAA,QACZ,QAAQ;AAAA,QACR;AAAA,QACA,SACE,YAAY,MACR,wBAAwB,WAAW,QAAQ,CAAC,CAAC,2BAA2B,WAAW,QAAQ,CAAC,CAAC,+BAC7F,yCAAyC,WAAW,QAAQ,CAAC,CAAC,WAAW,WAAW,QAAQ,CAAC,CAAC;AAAA,QACpG,QAAQ;AAAA,UACN;AAAA,UACA;AAAA,UACA;AAAA,UACA,SAAS,YAAY;AAAA,UACrB,QAAQ,WAAW;AAAA,QACrB;AAAA,MACF,CAAC;AAAA,IACH;AAAA,EACF;AAGA;AACE,UAAM,UAAU,OAAO,IAAI,OAAO,EAAE,OAAO,YAAY;AACvD,UAAM,SAAS,MAAM,IAAI,OAAO,EAAE,OAAO,YAAY;AACrD,QAAI,QAAQ,UAAU,KAAK,OAAO,UAAU,GAAG;AAC7C,YAAM,KAAK,YAAY,SAAS,MAAM;AAItC,YAAM,WAAWA,SAAQ,KAAK,GAAG;AACjC,eAAS,KAAK;AAAA,QACZ,QAAQ;AAAA,QACR;AAAA,QACA,SACE,YAAY,MACR,MAAM,GAAG,QAAQ,CAAC,CAAC,oEACnB,MAAM,GAAG,QAAQ,CAAC,CAAC;AAAA,QACzB,QAAQ,EAAE,IAAI,SAAS,QAAQ,QAAQ,QAAQ,OAAO,OAAO;AAAA,MAC/D,CAAC;AAAA,IACH;AAAA,EACF;AAGA;AACE,UAAM,cAAc,MAAM,qBAAqB,iBAAiB,MAAM,uBAAuB;AAC7F,UAAM,UAAU,KACb,IAAI,CAAC,OAAO,EAAE,GAAG,QAAQ,CAAC,GAAG,GAAG,YAAY,CAAC,EAAE,EAAE,EACjD,OAAO,CAAC,MAAqC,aAAa,EAAE,CAAC,KAAK,aAAa,EAAE,CAAC,CAAC;AACtF,QAAI,QAAQ,UAAU,GAAG;AACvB,YAAM,KAAK,QAAQ,IAAI,CAAC,MAAM,EAAE,CAAC;AACjC,YAAM,KAAK,QAAQ,IAAI,CAAC,MAAM,EAAE,CAAC;AACjC,YAAM,IAAI,SAAS,IAAI,EAAE;AAGzB,YAAM,WAAWA,SAAQ,MAAM,KAAK,IAAI,GAAG,CAAC,CAAC;AAC7C,eAAS,KAAK;AAAA,QACZ,QAAQ;AAAA,QACR;AAAA,QACA,SACE,YAAY,MACR,2DAAsD,EAAE,QAAQ,CAAC,CAAC,mCAClE,+CAA0C,EAAE,QAAQ,CAAC,CAAC;AAAA,QAC5D,QAAQ,EAAE,SAAS,GAAG,GAAG,QAAQ,OAAO;AAAA,MAC1C,CAAC;AAAA,IACH;AAAA,EACF;AAGA;AAME,UAAM,UAAU,gCAAgC,MAAM;AAAA,MACpD,GAAI,MAAM,2BAA2B,CAAC;AAAA,MACtC,mBAAmB;AAAA,IACrB,CAAC;AACD,QAAI,QAAQ,UAAU,GAAG;AACvB,YAAM,YAAY,QAAQ,MAAM,GAAG,KAAK,MAAM,QAAQ,SAAS,CAAC,CAAC;AACjE,YAAM,WAAW,QAAQ,MAAM,KAAK,MAAM,QAAQ,SAAS,CAAC,CAAC;AAC7D,YAAM,WACJ,KAAK,SAAS,IAAI,CAAC,MAAM,EAAE,OAAO,KAAK,CAAC,IAAI,KAAK,UAAU,IAAI,CAAC,MAAM,EAAE,OAAO,KAAK,CAAC;AACvF,YAAM,aACJ,KAAK,MAAM,IAAI,OAAO,EAAE,OAAO,YAAY,CAAC,IAC5C,KAAK,OAAO,IAAI,OAAO,EAAE,OAAO,YAAY,CAAC;AAC/C,YAAM,WAAW,KAAK,IAAI,GAAG,aAAa,QAAQ;AAClD,YAAM,WAAWA,SAAQ,WAAW,CAAC;AACrC,eAAS,KAAK;AAAA,QACZ,QAAQ;AAAA,QACR;AAAA,QACA,SACE,YAAY,MACR,gBAAgB,WAAW,QAAQ,CAAC,CAAC,gCAAgC,SAAS,QAAQ,CAAC,CAAC,yDACxF,uDAAuD,WAAW,QAAQ,CAAC,CAAC,SAAS,SAAS,QAAQ,CAAC,CAAC;AAAA,QAC9G,QAAQ,EAAE,YAAY,UAAU,UAAU,GAAG,QAAQ,OAAO;AAAA,MAC9D,CAAC;AAAA,IACH;AAAA,EACF;AAEA,QAAM,SAAS,SAAS,OAAO,CAAC,GAAG,MAAM,KAAK,IAAI,GAAG,EAAE,QAAQ,GAAG,CAAC;AACnE,MAAI,SAAS,WAAW,GAAG;AACzB,WAAO;AAAA,MACL;AAAA,MACA,kBAAkB,CAAC;AAAA,MACnB,SAAS;AAAA,MACT,WAAW,CAAC,0DAA0D,CAAC,GAAG;AAAA,MAC1E;AAAA,IACF;AAAA,EACF;AACA,QAAM,UACJ,UAAU,MAAM,WAAW,UAAU,MAAM,YAAY;AACzD,QAAM,YAAY,SACf,OAAO,CAAC,MAAM,EAAE,YAAY,GAAG,EAC/B,IAAI,CAAC,MAAM,GAAG,EAAE,MAAM,cAAc,EAAE,SAAS,QAAQ,CAAC,CAAC,WAAM,EAAE,OAAO,EAAE;AAC7E,MAAI,UAAU,WAAW,EAAG,WAAU,KAAK,0CAA0C;AAErF,SAAO;AAAA,IACL;AAAA,IACA,kBAAkB,SAAS,IAAI,CAAC,YAAY,QAAQ,MAAM;AAAA,IAC1D;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACF;AAIA,SAAS,KAAK,IAAsB;AAClC,MAAI,GAAG,WAAW,EAAG,QAAO;AAC5B,SAAO,GAAG,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;AAEA,SAAS,aAAa,OAAuC;AAC3D,SAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;AAEA,SAASA,SAAQ,GAAmB;AAClC,MAAI,CAAC,OAAO,SAAS,CAAC,EAAG,QAAO;AAChC,SAAO,KAAK,IAAI,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;AACnC;AAEA,SAAS,YAAY,GAAa,GAAqB;AAErD,QAAM,UAAU,CAAC,GAAG,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,IAAI,CAAC;AAC3C,QAAM,UAAU,CAAC,GAAG,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,IAAI,CAAC;AAC3C,QAAM,MAAM,CAAC,GAAG,oBAAI,IAAI,CAAC,GAAG,SAAS,GAAG,OAAO,CAAC,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,IAAI,CAAC;AACvE,MAAI,MAAM;AACV,aAAW,KAAK,KAAK;AACnB,UAAM,KAAK,QAAQ,OAAO,CAAC,MAAM,KAAK,CAAC,EAAE,SAAS,QAAQ;AAC1D,UAAM,KAAK,QAAQ,OAAO,CAAC,MAAM,KAAK,CAAC,EAAE,SAAS,QAAQ;AAC1D,UAAM,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,EAAE,CAAC;AAAA,EACvC;AACA,SAAO;AACT;AAEA,SAAS,iBACP,gBACmC;AACnC,SAAO,CAAC,QAAmB;AAIzB,UAAM,WAAW,gCAAgC,CAAC,GAAG,GAAG;AAAA,MACtD,GAAI,kBAAkB,CAAC;AAAA,MACvB,mBAAmB;AAAA,IACrB,CAAC;AACD,WAAO,SAAS,WAAW,IAAI,SAAS,CAAC,EAAG,OAAO,QAAQ;AAAA,EAC7D;AACF;","names":["clamp01"]}
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  cohensD
3
- } from "./chunk-MHELPNRP.js";
3
+ } from "./chunk-ZHTZ4EYI.js";
4
4
  import {
5
5
  argHash,
6
6
  groupBy,
@@ -573,4 +573,4 @@ export {
573
573
  iqr,
574
574
  welchsTTest
575
575
  };
576
- //# sourceMappingURL=chunk-P5W7RQKK.js.map
576
+ //# sourceMappingURL=chunk-FXTVJPYD.js.map
@@ -1,3 +1,7 @@
1
+ import {
2
+ observedScore
3
+ } from "./chunk-OIUOT4QD.js";
4
+
1
5
  // src/rl/active-curriculum.ts
2
6
  function varianceBasedCurriculum(observations, candidateCells, opts) {
3
7
  const variancePrior = opts.variancePrior ?? 0.05;
@@ -93,7 +97,7 @@ function observationsFromRunRecords(runs, opts = {}) {
93
97
  const out = [];
94
98
  for (const r of runs) {
95
99
  if (!r.scenarioId) continue;
96
- const score = useHoldout ? r.outcome.holdoutScore ?? r.outcome.searchScore : r.outcome.searchScore ?? r.outcome.holdoutScore;
100
+ const score = observedScore(r, useHoldout ? "holdout" : "search");
97
101
  if (typeof score !== "number" || !Number.isFinite(score)) continue;
98
102
  out.push({
99
103
  variantId: r.candidateId,
@@ -146,4 +150,4 @@ export {
146
150
  thompsonCurriculum,
147
151
  observationsFromRunRecords
148
152
  };
149
- //# sourceMappingURL=chunk-VZSRQ272.js.map
153
+ //# sourceMappingURL=chunk-G7MGMCZD.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/rl/active-curriculum.ts"],"sourcesContent":["/**\n * Adaptive curriculum / active scenario selection.\n *\n * Fixed scenario sets waste sample budget on cells the policy already\n * passes (no information left) and cells the policy never passes (no\n * gradient available either). Active learning over scenarios fixes this\n * by allocating the next sample budget to cells where the policy's\n * outcome is *uncertain* — those carry the most decision-relevant signal.\n *\n * This module ships two complementary strategies:\n *\n * 1. **Variance-based** — score each (variant, scenario) cell by the\n * empirical variance of past observations. Allocate next-round budget\n * proportional to variance. Standard active-learning-by-uncertainty\n * heuristic; works well when the policy is non-deterministic and\n * cells differ in observation noise.\n *\n * 2. **Bandit-based (Thompson sampling)** — model each (variant,\n * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick\n * cells whose posterior mean is closest to the per-scenario decision\n * threshold. The right primitive when scenarios are\n * \"pass/fail\" rather than continuous, and when promotion gates fire\n * at a known threshold (e.g., 0.5).\n *\n * The output is a *next-round budget allocation* — a list of (variant,\n * scenario, count) triples. The consumer's matrix runner consumes the\n * allocation, runs those cells, feeds the new observations back. Loop.\n *\n * Out of scope (deliberate): scenario *generation* — that's the\n * adversarial primitive's job. This module allocates over an existing\n * scenario pool.\n */\n\nimport { observedScore } from '../rollout/reward'\nimport type { RunRecord } from '../run-record'\n\nexport interface CellObservation {\n variantId: string\n scenarioId: string\n /** Observed score in [0, 1]. */\n score: number\n /** For Bernoulli arms — derive from the score with a threshold if needed. */\n pass?: boolean\n}\n\nexport interface CurriculumAllocation {\n variantId: string\n scenarioId: string\n /** How many additional reps to run on this cell. */\n count: number\n /** Strategy-specific reason for the allocation. */\n reason: string\n}\n\nexport interface VarianceCurriculumOptions {\n /** Total reps to allocate across all cells. */\n budget: number\n /**\n * Smoothing prior on variance — keeps the allocator from concentrating\n * on a cell with one observation just because its 1-sample variance is\n * 0. Default 0.05.\n */\n variancePrior?: number\n /**\n * Minimum reps per cell — even when the variance estimate is low, give\n * every cell at least this many. Default 1.\n */\n floorPerCell?: number\n}\n\n/**\n * Variance-proportional allocation. For each cell, estimate variance from\n * past observations + a prior, then allocate the budget proportional to\n * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule\n * (Neyman 1934) that balances \"explore noisy cells\" with \"explore\n * under-sampled cells.\"\n */\nexport function varianceBasedCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: VarianceCurriculumOptions,\n): CurriculumAllocation[] {\n const variancePrior = opts.variancePrior ?? 0.05\n const floor = opts.floorPerCell ?? 1\n const budget = opts.budget\n\n const grouped = new Map<string, number[]>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const arr = grouped.get(k) ?? []\n arr.push(o.score)\n grouped.set(k, arr)\n }\n\n const cellStats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const samples = grouped.get(k) ?? []\n const n = samples.length\n const mean = n === 0 ? 0.5 : samples.reduce((s, v) => s + v, 0) / n\n const variance =\n n < 2\n ? variancePrior\n : samples.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1) + variancePrior\n // Neyman optimal allocation: weight ∝ √variance; add √(1/n) to break\n // ties toward under-sampled cells.\n const weight = Math.sqrt(variance) + 1 / Math.sqrt(Math.max(1, n))\n return { variantId: c.variantId, scenarioId: c.scenarioId, n, mean, variance, weight }\n })\n\n // Reserve floor*N for the floor; allocate the rest proportional to weight.\n const floorTotal = floor * cellStats.length\n if (floorTotal >= budget) {\n const each = Math.max(1, Math.floor(budget / Math.max(1, cellStats.length)))\n return cellStats.map((c) => ({\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: each,\n reason: `floor allocation (budget tight; n=${c.n})`,\n }))\n }\n const remaining = budget - floorTotal\n const totalWeight = cellStats.reduce((s, c) => s + c.weight, 0)\n return cellStats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * remaining)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: floor + proportional,\n reason: `variance ${c.variance.toFixed(3)} (n=${c.n}, mean=${c.mean.toFixed(3)})`,\n }\n })\n}\n\nexport interface ThompsonCurriculumOptions {\n budget: number\n /**\n * The per-scenario decision threshold. Cells whose posterior mean is\n * closest to this get the most budget — that's where the next observation\n * has the highest information value for the gate decision. Default 0.5.\n */\n decisionThreshold?: number\n /** Beta prior parameters. Default α=β=1 (uniform). */\n priorAlpha?: number\n priorBeta?: number\n /** Seed the Thompson sampler. Default unset (Math.random). */\n seed?: number\n}\n\n/**\n * Thompson-sampling-style allocation for pass/fail cells. For each cell:\n *\n * - Maintain Beta(α + passes, β + failures) posterior on pass-rate\n * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):\n * cells whose sampled posterior straddles the decision boundary get\n * the most weight; cells already clearly above or below get less.\n *\n * This is the right primitive when promotion gates fire at a known\n * threshold and you want to sharpen the posterior near the boundary.\n */\nexport function thompsonCurriculum(\n observations: CellObservation[],\n candidateCells: Array<{ variantId: string; scenarioId: string }>,\n opts: ThompsonCurriculumOptions,\n): CurriculumAllocation[] {\n const threshold = opts.decisionThreshold ?? 0.5\n const alpha0 = opts.priorAlpha ?? 1\n const beta0 = opts.priorBeta ?? 1\n const rng = makeRng(opts.seed)\n\n const grouped = new Map<string, { passes: number; failures: number }>()\n for (const o of observations) {\n const k = `${o.variantId}::${o.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const pass = o.pass ?? o.score >= threshold\n if (pass) cur.passes += 1\n else cur.failures += 1\n grouped.set(k, cur)\n }\n\n const stats = candidateCells.map((c) => {\n const k = `${c.variantId}::${c.scenarioId}`\n const cur = grouped.get(k) ?? { passes: 0, failures: 0 }\n const a = alpha0 + cur.passes\n const b = beta0 + cur.failures\n // Sample a single Beta draw — the Thompson signal.\n const sampled = sampleBeta(a, b, rng)\n const distance = Math.abs(sampled - threshold)\n // Information-near-threshold weight: closer = higher.\n // Use Gaussian-shaped kernel with σ tuned to posterior std.\n const variance = (a * b) / ((a + b) ** 2 * (a + b + 1))\n const sigma = Math.max(0.05, Math.sqrt(variance))\n const weight = Math.exp(-((distance / sigma) ** 2))\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n n: cur.passes + cur.failures,\n sampled,\n sigma,\n weight,\n a,\n b,\n }\n })\n\n const totalWeight = stats.reduce((s, c) => s + c.weight, 0)\n return stats.map((c) => {\n const proportional = totalWeight === 0 ? 0 : Math.round((c.weight / totalWeight) * opts.budget)\n return {\n variantId: c.variantId,\n scenarioId: c.scenarioId,\n count: Math.max(0, proportional),\n reason: `Beta(${c.a.toFixed(1)},${c.b.toFixed(1)}) sample=${c.sampled.toFixed(3)} (target ${threshold})`,\n }\n })\n}\n\n/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */\nexport function observationsFromRunRecords(\n runs: RunRecord[],\n opts: { passThreshold?: number; useHoldout?: boolean } = {},\n): CellObservation[] {\n const threshold = opts.passThreshold ?? 0.5\n const useHoldout = opts.useHoldout ?? true\n const out: CellObservation[] = []\n for (const r of runs) {\n if (!r.scenarioId) continue\n // Ungated on purpose, and the precedence is caller policy, not a default:\n // `useHoldout: false` means \"score this curriculum on the search split when\n // both exist\". This feeds sampling COUNTS, not an exported reward. Known\n // risk: a gamed run's high score inflates the cell's Beta posterior, so the\n // curriculum stops sampling a cell it wrongly believes is solved. The fix\n // for that is an upstream filter on gated records — zeroing the score here\n // would push the posterior the opposite way and be equally wrong.\n const score = observedScore(r, useHoldout ? 'holdout' : 'search')\n if (typeof score !== 'number' || !Number.isFinite(score)) continue\n out.push({\n variantId: r.candidateId,\n scenarioId: r.scenarioId,\n score,\n pass: score >= threshold,\n })\n }\n return out\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction makeRng(seed?: number): () => number {\n if (seed === undefined) return Math.random\n let s = seed >>> 0\n return () => {\n s = (s + 0x6d2b79f5) >>> 0\n let t = s\n t = Math.imul(t ^ (t >>> 15), t | 1)\n t ^= t + Math.imul(t ^ (t >>> 7), t | 61)\n return ((t ^ (t >>> 14)) >>> 0) / 4294967296\n }\n}\n\n/**\n * Sample from Beta(α, β) via the Marsaglia–Tsang method using two Gamma\n * variates. Accuracy is good for α, β > 1; we floor the parameters at 1\n * to avoid degenerate cases.\n */\nfunction sampleBeta(alpha: number, beta: number, rng: () => number): number {\n const a = Math.max(1, alpha)\n const b = Math.max(1, beta)\n const x = sampleGamma(a, rng)\n const y = sampleGamma(b, rng)\n return x / (x + y)\n}\n\nfunction sampleGamma(shape: number, rng: () => number): number {\n // Marsaglia–Tsang for shape ≥ 1.\n const d = shape - 1 / 3\n const c = 1 / Math.sqrt(9 * d)\n while (true) {\n let x: number\n let v: number\n do {\n const u1 = rng() || 1e-12\n const u2 = rng() || 1e-12\n // Box-Muller for a normal sample.\n x = Math.sqrt(-2 * Math.log(u1)) * Math.cos(2 * Math.PI * u2)\n v = 1 + c * x\n } while (v <= 0)\n v = v * v * v\n const u = rng()\n if (u < 1 - 0.0331 * x ** 4) return d * v\n if (Math.log(u) < 0.5 * x * x + d * (1 - v + Math.log(v))) return d * v\n }\n}\n"],"mappings":";;;;;AA6EO,SAAS,wBACd,cACA,gBACA,MACwB;AACxB,QAAM,gBAAgB,KAAK,iBAAiB;AAC5C,QAAM,QAAQ,KAAK,gBAAgB;AACnC,QAAM,SAAS,KAAK;AAEpB,QAAM,UAAU,oBAAI,IAAsB;AAC1C,aAAW,KAAK,cAAc;AAC5B,UAAM,IAAI,GAAG,EAAE,SAAS,KAAK,EAAE,UAAU;AACzC,UAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,CAAC;AAC/B,QAAI,KAAK,EAAE,KAAK;AAChB,YAAQ,IAAI,GAAG,GAAG;AAAA,EACpB;AAEA,QAAM,YAAY,eAAe,IAAI,CAAC,MAAM;AAC1C,UAAM,IAAI,GAAG,EAAE,SAAS,KAAK,EAAE,UAAU;AACzC,UAAM,UAAU,QAAQ,IAAI,CAAC,KAAK,CAAC;AACnC,UAAM,IAAI,QAAQ;AAClB,UAAM,OAAO,MAAM,IAAI,MAAM,QAAQ,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;AAClE,UAAM,WACJ,IAAI,IACA,gBACA,QAAQ,OAAO,CAAC,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,KAAK,IAAI,KAAK;AAGnE,UAAM,SAAS,KAAK,KAAK,QAAQ,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,GAAG,CAAC,CAAC;AACjE,WAAO,EAAE,WAAW,EAAE,WAAW,YAAY,EAAE,YAAY,GAAG,MAAM,UAAU,OAAO;AAAA,EACvF,CAAC;AAGD,QAAM,aAAa,QAAQ,UAAU;AACrC,MAAI,cAAc,QAAQ;AACxB,UAAM,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,SAAS,KAAK,IAAI,GAAG,UAAU,MAAM,CAAC,CAAC;AAC3E,WAAO,UAAU,IAAI,CAAC,OAAO;AAAA,MAC3B,WAAW,EAAE;AAAA,MACb,YAAY,EAAE;AAAA,MACd,OAAO;AAAA,MACP,QAAQ,qCAAqC,EAAE,CAAC;AAAA,IAClD,EAAE;AAAA,EACJ;AACA,QAAM,YAAY,SAAS;AAC3B,QAAM,cAAc,UAAU,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;AAC9D,SAAO,UAAU,IAAI,CAAC,MAAM;AAC1B,UAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,SAAS;AAC5F,WAAO;AAAA,MACL,WAAW,EAAE;AAAA,MACb,YAAY,EAAE;AAAA,MACd,OAAO,QAAQ;AAAA,MACf,QAAQ,YAAY,EAAE,SAAS,QAAQ,CAAC,CAAC,OAAO,EAAE,CAAC,UAAU,EAAE,KAAK,QAAQ,CAAC,CAAC;AAAA,IAChF;AAAA,EACF,CAAC;AACH;AA4BO,SAAS,mBACd,cACA,gBACA,MACwB;AACxB,QAAM,YAAY,KAAK,qBAAqB;AAC5C,QAAM,SAAS,KAAK,cAAc;AAClC,QAAM,QAAQ,KAAK,aAAa;AAChC,QAAM,MAAM,QAAQ,KAAK,IAAI;AAE7B,QAAM,UAAU,oBAAI,IAAkD;AACtE,aAAW,KAAK,cAAc;AAC5B,UAAM,IAAI,GAAG,EAAE,SAAS,KAAK,EAAE,UAAU;AACzC,UAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,EAAE,QAAQ,GAAG,UAAU,EAAE;AACvD,UAAM,OAAO,EAAE,QAAQ,EAAE,SAAS;AAClC,QAAI,KAAM,KAAI,UAAU;AAAA,QACnB,KAAI,YAAY;AACrB,YAAQ,IAAI,GAAG,GAAG;AAAA,EACpB;AAEA,QAAM,QAAQ,eAAe,IAAI,CAAC,MAAM;AACtC,UAAM,IAAI,GAAG,EAAE,SAAS,KAAK,EAAE,UAAU;AACzC,UAAM,MAAM,QAAQ,IAAI,CAAC,KAAK,EAAE,QAAQ,GAAG,UAAU,EAAE;AACvD,UAAM,IAAI,SAAS,IAAI;AACvB,UAAM,IAAI,QAAQ,IAAI;AAEtB,UAAM,UAAU,WAAW,GAAG,GAAG,GAAG;AACpC,UAAM,WAAW,KAAK,IAAI,UAAU,SAAS;AAG7C,UAAM,WAAY,IAAI,MAAO,IAAI,MAAM,KAAK,IAAI,IAAI;AACpD,UAAM,QAAQ,KAAK,IAAI,MAAM,KAAK,KAAK,QAAQ,CAAC;AAChD,UAAM,SAAS,KAAK,IAAI,GAAG,WAAW,UAAU,EAAE;AAClD,WAAO;AAAA,MACL,WAAW,EAAE;AAAA,MACb,YAAY,EAAE;AAAA,MACd,GAAG,IAAI,SAAS,IAAI;AAAA,MACpB;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,IACF;AAAA,EACF,CAAC;AAED,QAAM,cAAc,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;AAC1D,SAAO,MAAM,IAAI,CAAC,MAAM;AACtB,UAAM,eAAe,gBAAgB,IAAI,IAAI,KAAK,MAAO,EAAE,SAAS,cAAe,KAAK,MAAM;AAC9F,WAAO;AAAA,MACL,WAAW,EAAE;AAAA,MACb,YAAY,EAAE;AAAA,MACd,OAAO,KAAK,IAAI,GAAG,YAAY;AAAA,MAC/B,QAAQ,QAAQ,EAAE,EAAE,QAAQ,CAAC,CAAC,IAAI,EAAE,EAAE,QAAQ,CAAC,CAAC,YAAY,EAAE,QAAQ,QAAQ,CAAC,CAAC,YAAY,SAAS;AAAA,IACvG;AAAA,EACF,CAAC;AACH;AAGO,SAAS,2BACd,MACA,OAAyD,CAAC,GACvC;AACnB,QAAM,YAAY,KAAK,iBAAiB;AACxC,QAAM,aAAa,KAAK,cAAc;AACtC,QAAM,MAAyB,CAAC;AAChC,aAAW,KAAK,MAAM;AACpB,QAAI,CAAC,EAAE,WAAY;AAQnB,UAAM,QAAQ,cAAc,GAAG,aAAa,YAAY,QAAQ;AAChE,QAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,EAAG;AAC1D,QAAI,KAAK;AAAA,MACP,WAAW,EAAE;AAAA,MACb,YAAY,EAAE;AAAA,MACd;AAAA,MACA,MAAM,SAAS;AAAA,IACjB,CAAC;AAAA,EACH;AACA,SAAO;AACT;AAIA,SAAS,QAAQ,MAA6B;AAC5C,MAAI,SAAS,OAAW,QAAO,KAAK;AACpC,MAAI,IAAI,SAAS;AACjB,SAAO,MAAM;AACX,QAAK,IAAI,eAAgB;AACzB,QAAI,IAAI;AACR,QAAI,KAAK,KAAK,IAAK,MAAM,IAAK,IAAI,CAAC;AACnC,SAAK,IAAI,KAAK,KAAK,IAAK,MAAM,GAAI,IAAI,EAAE;AACxC,aAAS,IAAK,MAAM,QAAS,KAAK;AAAA,EACpC;AACF;AAOA,SAAS,WAAW,OAAe,MAAc,KAA2B;AAC1E,QAAM,IAAI,KAAK,IAAI,GAAG,KAAK;AAC3B,QAAM,IAAI,KAAK,IAAI,GAAG,IAAI;AAC1B,QAAM,IAAI,YAAY,GAAG,GAAG;AAC5B,QAAM,IAAI,YAAY,GAAG,GAAG;AAC5B,SAAO,KAAK,IAAI;AAClB;AAEA,SAAS,YAAY,OAAe,KAA2B;AAE7D,QAAM,IAAI,QAAQ,IAAI;AACtB,QAAM,IAAI,IAAI,KAAK,KAAK,IAAI,CAAC;AAC7B,SAAO,MAAM;AACX,QAAI;AACJ,QAAI;AACJ,OAAG;AACD,YAAM,KAAK,IAAI,KAAK;AACpB,YAAM,KAAK,IAAI,KAAK;AAEpB,UAAI,KAAK,KAAK,KAAK,KAAK,IAAI,EAAE,CAAC,IAAI,KAAK,IAAI,IAAI,KAAK,KAAK,EAAE;AAC5D,UAAI,IAAI,IAAI;AAAA,IACd,SAAS,KAAK;AACd,QAAI,IAAI,IAAI;AACZ,UAAM,IAAI,IAAI;AACd,QAAI,IAAI,IAAI,SAAS,KAAK,EAAG,QAAO,IAAI;AACxC,QAAI,KAAK,IAAI,CAAC,IAAI,MAAM,IAAI,IAAI,KAAK,IAAI,IAAI,KAAK,IAAI,CAAC,GAAI,QAAO,IAAI;AAAA,EACxE;AACF;","names":[]}
@@ -1,12 +1,17 @@
1
1
  import {
2
- ROLLOUT_SCHEMA
3
- } from "./chunk-UWZZKKU7.js";
2
+ ROLLOUT_SCHEMA,
3
+ assertMinted
4
+ } from "./chunk-PC5DOSM7.js";
4
5
  import {
5
6
  buildTrajectory
6
7
  } from "./chunk-RZTMDUO7.js";
7
8
  import {
8
9
  runTaskScore
9
- } from "./chunk-2JX3CFMB.js";
10
+ } from "./chunk-56TAVBOK.js";
11
+ import {
12
+ rolloutRewardFields,
13
+ scoreOrigin
14
+ } from "./chunk-OIUOT4QD.js";
10
15
  import {
11
16
  ValidationError
12
17
  } from "./chunk-ONWEPEDO.js";
@@ -48,21 +53,15 @@ function finalConversation(spans, scrub) {
48
53
  }
49
54
  return messages;
50
55
  }
51
- function scoredReward(record) {
52
- const score = runTaskScore(record);
53
- if (score === void 0) {
56
+ var REWARD_SOURCE = {
57
+ holdout: "run-record/holdout-score",
58
+ search: "run-record/search-score",
59
+ unscored: "run-record/unscored"
60
+ };
61
+ function requireTaskScore(record) {
62
+ if (runTaskScore(record) === void 0) {
54
63
  throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`);
55
64
  }
56
- const gated = record.outcome.realness?.gated === true;
57
- return {
58
- reward: gated ? 0 : score,
59
- gated,
60
- source: record.outcome.holdoutScore !== void 0 ? "run-record/holdout-score" : "run-record/search-score"
61
- };
62
- }
63
- function rolloutReward(record) {
64
- const { reward, gated } = scoredReward(record);
65
- return { reward, gated };
66
65
  }
67
66
  var SPLIT_FROM_TAG = {
68
67
  search: "search",
@@ -70,69 +69,81 @@ var SPLIT_FROM_TAG = {
70
69
  holdout: "holdout"
71
70
  };
72
71
  function mintLine(record, steps, messages, options, capturedAt, gap) {
73
- const { reward, gated, source } = scoredReward(record);
72
+ requireTaskScore(record);
73
+ const rewardFields = rolloutRewardFields(record);
74
74
  const uncaptured = record.costProvenance.kind === "uncaptured";
75
75
  const terminalOutcome = record.terminalOutcome;
76
76
  const isCompleted = terminalOutcome === "succeeded" || terminalOutcome === "failed";
77
77
  const isTruncated = terminalOutcome === "cancelled" || terminalOutcome === "incomplete";
78
78
  const terminalError = terminalOutcome === "failed" || terminalOutcome === "cancelled" || terminalOutcome === "incomplete" ? record.terminalFailureReason ?? `run ended ${terminalOutcome}` : null;
79
- return {
80
- schema: ROLLOUT_SCHEMA,
81
- rollout_id: record.runId,
82
- parent_rollout_id: null,
83
- run_id: record.runId,
84
- experiment_id: record.experimentId,
85
- candidate_id: record.candidateId,
86
- generation: null,
87
- candidate_index: null,
88
- role: options.role ?? "agent",
89
- task: {
90
- suite: options.suite ?? record.experimentId,
91
- instance_id: record.scenarioId,
92
- split: SPLIT_FROM_TAG[record.splitTag],
93
- seed: record.seed,
94
- rep: 0
95
- },
96
- policy: {
97
- harness: null,
98
- harness_version: null,
99
- model: record.model,
100
- provider: null,
101
- profile_commit: record.commitSha,
102
- prompt_hash: record.promptHash,
103
- config_hash: record.configHash,
104
- agent_profile_cell_id: record.agentProfile?.cellId ?? null,
105
- sampling: null
79
+ return assertMinted(
80
+ {
81
+ schema: ROLLOUT_SCHEMA,
82
+ rollout_id: record.runId,
83
+ parent_rollout_id: null,
84
+ run_id: record.runId,
85
+ experiment_id: record.experimentId,
86
+ candidate_id: record.candidateId,
87
+ generation: null,
88
+ candidate_index: null,
89
+ role: options.role ?? "agent",
90
+ task: {
91
+ suite: options.suite ?? record.experimentId,
92
+ instance_id: record.scenarioId,
93
+ split: SPLIT_FROM_TAG[record.splitTag],
94
+ seed: record.seed,
95
+ rep: 0
96
+ },
97
+ policy: {
98
+ harness: null,
99
+ harness_version: null,
100
+ model: record.model,
101
+ provider: null,
102
+ profile_commit: record.commitSha,
103
+ prompt_hash: record.promptHash,
104
+ config_hash: record.configHash,
105
+ agent_profile_cell_id: record.agentProfile?.cellId ?? null,
106
+ sampling: null
107
+ },
108
+ messages,
109
+ tool_defs: [],
110
+ ...steps.length > 0 ? { steps } : {},
111
+ outcome: {
112
+ ...rewardFields,
113
+ reward_source: REWARD_SOURCE[scoreOrigin(record)],
114
+ verdict: null,
115
+ // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`
116
+ // holds the per-layer verifier scores (`layer.*`) that the reward was
117
+ // derived from, so on a gated run this dict is the reward signal in
118
+ // component form — but filtering it at this call site is the pattern
119
+ // that has now leaked twice, because the next producer to write a
120
+ // reward-bearing field forgets. The gate is applied to the whole
121
+ // outcome once, in `assertMinted` below (`gateGamedOutcome`), which
122
+ // moves the block to `provenance.gated_evidence` when the run is gated
123
+ // and leaves it here untouched when it is not.
124
+ metrics: { ...record.outcome.raw },
125
+ is_completed: isCompleted,
126
+ is_truncated: isTruncated,
127
+ error: terminalError
128
+ },
129
+ cost: {
130
+ usd: uncaptured ? null : record.costUsd,
131
+ tokens_in: record.tokenUsage.input,
132
+ tokens_out: record.tokenUsage.output,
133
+ tokens_reasoning: record.tokenUsage.reasoning ?? null,
134
+ cache_read: record.tokenUsage.cached ?? null,
135
+ cache_write: record.tokenUsage.cacheWrite ?? null,
136
+ wall_s: Math.round(record.wallMs / 1e3)
137
+ },
138
+ artifacts: { patch_path: null, run_dir: null, transcript_ref: null },
139
+ provenance: {
140
+ captured_at: capturedAt,
141
+ capture: "mint",
142
+ ...gap !== void 0 ? { gap } : {}
143
+ }
106
144
  },
107
- messages,
108
- tool_defs: [],
109
- ...steps.length > 0 ? { steps } : {},
110
- outcome: {
111
- reward,
112
- reward_source: source,
113
- verdict: null,
114
- metrics: { ...record.outcome.raw },
115
- is_completed: isCompleted,
116
- is_truncated: isTruncated,
117
- error: terminalError,
118
- realness_gated: gated
119
- },
120
- cost: {
121
- usd: uncaptured ? null : record.costUsd,
122
- tokens_in: record.tokenUsage.input,
123
- tokens_out: record.tokenUsage.output,
124
- tokens_reasoning: record.tokenUsage.reasoning ?? null,
125
- cache_read: record.tokenUsage.cached ?? null,
126
- cache_write: record.tokenUsage.cacheWrite ?? null,
127
- wall_s: Math.round(record.wallMs / 1e3)
128
- },
129
- artifacts: { patch_path: null, run_dir: null, transcript_ref: null },
130
- provenance: {
131
- captured_at: capturedAt,
132
- capture: "mint",
133
- ...gap !== void 0 ? { gap } : {}
134
- }
135
- };
145
+ `minted rollout line for run ${record.runId}`
146
+ );
136
147
  }
137
148
  async function mintRolloutRows(records, store, options = {}) {
138
149
  const scrub = options.scrub ?? ((t) => t);
@@ -165,7 +176,6 @@ async function mintRolloutRows(records, store, options = {}) {
165
176
  }
166
177
 
167
178
  export {
168
- rolloutReward,
169
179
  mintRolloutRows
170
180
  };
171
- //# sourceMappingURL=chunk-IHQDPH7D.js.map
181
+ //# sourceMappingURL=chunk-H23X7XKK.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/rollout/mint.ts"],"sourcesContent":["/**\n * Rollout minting — `tangle.rollout.v1` lines joined from the records the\n * substrate ALREADY keeps. There is no separate rollout store: a rollout\n * is the JOIN of a RunRecord (identity, provenance, cost, outcome) with\n * its trace (spans share `runId`), projected into the canonical line.\n *\n * Composition, not duplication:\n * - identity/provenance → `RunRecord` (candidateId, splitTag, agentProfile, hashes)\n * - step structure → `buildTrajectory` over the shared TraceStore\n * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)\n * - PRM / reward-model → `reward-model-export.ts`\n *\n * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is\n * never exported with a positive reward OR with any of the numbers that reward\n * was computed from. The gate travels into the training data (`reward` forced\n * to 0, `realness_gated: true`) and the whole outcome is transformed by\n * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and\n * `verdict` to `provenance.gated_evidence`. Mint returns\n * `MintedRolloutLine[]`: the brand the training exporters require, which only\n * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.\n *\n * A record carrying NEITHER split score is REJECTED (`ValidationError`), never\n * minted at 0 — \"nobody graded this\" is not the same claim as \"graded a total\n * failure\", and a trainer reading 0 learns the second. Lines that already\n * carry `reward: null` (interchange imports, existing ledgers) remain valid on\n * the wire; only the RunRecord→line door refuses.\n *\n * Records without spans become labeled GAP LINES (messages: [],\n * provenance.gap) — present in the output AND surfaced in\n * `missingTraces`; a capture gap is a finding, never a silent omission.\n */\n\nimport { ValidationError } from '../errors'\nimport { type RunRecord, runTaskScore } from '../run-record'\nimport type { LlmSpan, Message, Span, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory } from '../trajectory'\nimport { rolloutRewardFields, scoreOrigin } from './reward'\nimport {\n assertMinted,\n type ChatMessage,\n type MintedRolloutLine,\n ROLLOUT_SCHEMA,\n type RolloutRole,\n type RolloutSplit,\n type RolloutStep,\n} from './schema'\n\n/** Redactor applied to every exported string (secrets, PII). Identity by default. */\nexport type RolloutScrubber = (text: string) => string\n\nexport interface MintRolloutOptions {\n scrub?: RolloutScrubber\n /** Cap steps per line (longest runs first drop middle steps). Default: no cap. */\n maxSteps?: number\n /** Role recorded on every minted line. Default 'agent' (a solo eval run). */\n role?: RolloutRole\n /** Task suite label. Default: the record's `experimentId`. */\n suite?: string\n /** Injected clock for deterministic output. */\n now?: () => Date\n}\n\nexport interface MintRolloutResult {\n rows: MintedRolloutLine[]\n /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */\n missingTraces: string[]\n}\n\nconst asText = (v: unknown, scrub: RolloutScrubber): string => {\n const s = typeof v === 'string' ? v : JSON.stringify(v)\n return scrub(s ?? '')\n}\n\nfunction projectStep(span: Span, scrub: RolloutScrubber): RolloutStep {\n const base: RolloutStep = {\n kind: span.kind,\n name: scrub(span.name),\n status: span.status,\n durationMs: span.endedAt !== undefined ? span.endedAt - span.startedAt : undefined,\n }\n if (span.kind === 'llm') {\n const llm = span as LlmSpan\n const last = llm.messages[llm.messages.length - 1]\n if (last) base.input = scrub(last.content)\n if (llm.output !== undefined) base.output = scrub(llm.output)\n } else if (span.kind === 'tool') {\n const tool = span as ToolSpan\n base.input = asText(tool.args, scrub)\n if (tool.result !== undefined) base.output = asText(tool.result, scrub)\n }\n return base\n}\n\n/** The final llm span's history + output is the completed conversation. */\nfunction finalConversation(spans: Span[], scrub: RolloutScrubber): ChatMessage[] {\n const llms = spans.filter((s): s is LlmSpan => s.kind === 'llm')\n const last = llms[llms.length - 1]\n if (!last) return []\n const messages: ChatMessage[] = last.messages.map((m: Message) => ({\n role: m.role,\n content: scrub(m.content),\n }))\n if (last.output !== undefined && last.output !== '') {\n messages.push({ role: 'assistant', content: scrub(last.output) })\n }\n return messages\n}\n\n// The reward derivations live in the leaf module `./reward` so gate and\n// reporting code can import them without dragging in the trace store; they are\n// re-exported here because the derivations shipped from this path.\nexport {\n isRealnessGated,\n observedScore,\n observedSplitScore,\n type ScoreOrigin,\n type ScorePreference,\n scoreOrigin,\n trainingReward,\n trainingScore,\n} from './reward'\n\nconst REWARD_SOURCE: Record<ReturnType<typeof scoreOrigin>, string> = {\n holdout: 'run-record/holdout-score',\n search: 'run-record/search-score',\n unscored: 'run-record/unscored',\n}\n\n/**\n * The mint door refuses an execution-only record: a missing training label is\n * not a zero reward, and not a mintable line either. Lines that already carry\n * `reward: null` — interchange imports, existing ledgers — stay valid on the\n * wire and keep their labeled gap; this guard is only about the\n * RunRecord→line door, where the producer can still be told to go score the\n * run instead of shipping an unlabeled row.\n */\nfunction requireTaskScore(record: RunRecord): void {\n if (runTaskScore(record) === undefined) {\n throw new ValidationError(`Cannot mint rollout for run ${record.runId}: task score is missing`)\n }\n}\n\nconst SPLIT_FROM_TAG: Record<RunRecord['splitTag'], RolloutSplit> = {\n search: 'search',\n dev: 'dev',\n holdout: 'holdout',\n}\n\nfunction mintLine(\n record: RunRecord,\n steps: RolloutStep[],\n messages: ChatMessage[],\n options: MintRolloutOptions,\n capturedAt: string,\n gap?: string,\n): MintedRolloutLine {\n // A missing task score is refused before anything is built: an\n // execution-only record has no training label, and a missing label is\n // neither a zero reward nor a mintable row.\n requireTaskScore(record)\n // `reward` and `realness_gated` come out of one call, so neither door into\n // the waist can write one and forget the other.\n const rewardFields = rolloutRewardFields(record)\n const uncaptured = record.costProvenance.kind === 'uncaptured'\n const terminalOutcome = record.terminalOutcome\n const isCompleted = terminalOutcome === 'succeeded' || terminalOutcome === 'failed'\n const isTruncated = terminalOutcome === 'cancelled' || terminalOutcome === 'incomplete'\n const terminalError =\n terminalOutcome === 'failed' ||\n terminalOutcome === 'cancelled' ||\n terminalOutcome === 'incomplete'\n ? (record.terminalFailureReason ?? `run ended ${terminalOutcome}`)\n : null\n // `assertMinted` rather than a cast: mint is the producer the whole gate\n // rests on, so it proves the line it just built is valid instead of asserting\n // it by fiat. The brand is unforgeable precisely because nobody casts to it.\n return assertMinted(\n {\n schema: ROLLOUT_SCHEMA,\n rollout_id: record.runId,\n parent_rollout_id: null,\n run_id: record.runId,\n experiment_id: record.experimentId,\n candidate_id: record.candidateId,\n generation: null,\n candidate_index: null,\n role: options.role ?? 'agent',\n task: {\n suite: options.suite ?? record.experimentId,\n instance_id: record.scenarioId,\n split: SPLIT_FROM_TAG[record.splitTag],\n seed: record.seed,\n rep: 0,\n },\n policy: {\n harness: null,\n harness_version: null,\n model: record.model,\n provider: null,\n profile_commit: record.commitSha,\n prompt_hash: record.promptHash,\n config_hash: record.configHash,\n agent_profile_cell_id: record.agentProfile?.cellId ?? null,\n sampling: null,\n },\n messages,\n tool_defs: [],\n ...(steps.length > 0 ? { steps } : {}),\n outcome: {\n ...rewardFields,\n reward_source: REWARD_SOURCE[scoreOrigin(record)],\n verdict: null,\n // A verbatim bulk copy, deliberately UNFILTERED here. `outcome.raw`\n // holds the per-layer verifier scores (`layer.*`) that the reward was\n // derived from, so on a gated run this dict is the reward signal in\n // component form — but filtering it at this call site is the pattern\n // that has now leaked twice, because the next producer to write a\n // reward-bearing field forgets. The gate is applied to the whole\n // outcome once, in `assertMinted` below (`gateGamedOutcome`), which\n // moves the block to `provenance.gated_evidence` when the run is gated\n // and leaves it here untouched when it is not.\n metrics: { ...record.outcome.raw },\n is_completed: isCompleted,\n is_truncated: isTruncated,\n error: terminalError,\n },\n cost: {\n usd: uncaptured ? null : record.costUsd,\n tokens_in: record.tokenUsage.input,\n tokens_out: record.tokenUsage.output,\n tokens_reasoning: record.tokenUsage.reasoning ?? null,\n cache_read: record.tokenUsage.cached ?? null,\n cache_write: record.tokenUsage.cacheWrite ?? null,\n wall_s: Math.round(record.wallMs / 1000),\n },\n artifacts: { patch_path: null, run_dir: null, transcript_ref: null },\n provenance: {\n captured_at: capturedAt,\n capture: 'mint',\n ...(gap !== undefined ? { gap } : {}),\n },\n },\n `minted rollout line for run ${record.runId}`,\n )\n}\n\n/**\n * Join RunRecords with their traces into canonical rollout lines. Records\n * without spans are emitted as labeled gap lines and reported in\n * `missingTraces`. Execution-only records without a task score are rejected\n * because a missing training label is not a zero reward.\n */\nexport async function mintRolloutRows(\n records: RunRecord[],\n store: TraceStore,\n options: MintRolloutOptions = {},\n): Promise<MintRolloutResult> {\n const scrub = options.scrub ?? ((t) => t)\n const capturedAt = (options.now?.() ?? new Date()).toISOString()\n const rows: MintedRolloutLine[] = []\n const missingTraces: string[] = []\n for (const record of records) {\n const trajectory = await buildTrajectory(store, record.runId)\n if (trajectory.steps.length === 0) {\n missingTraces.push(record.runId)\n rows.push(\n mintLine(record, [], [], options, capturedAt, 'no trace spans recorded for this runId'),\n )\n continue\n }\n let steps = trajectory.steps.map((s) => projectStep(s.span, scrub))\n if (options.maxSteps !== undefined && steps.length > options.maxSteps) {\n // Keep the head and tail — the middle of a long run is the least\n // informative for outcome attribution.\n const head = Math.ceil(options.maxSteps / 2)\n const tail = options.maxSteps - head\n steps = [...steps.slice(0, head), ...steps.slice(steps.length - tail)]\n }\n const conversation = finalConversation(\n trajectory.steps.map((s) => s.span),\n scrub,\n )\n const gap =\n conversation.length === 0 ? 'trace has no llm spans — no conversation to inline' : undefined\n rows.push(mintLine(record, steps, conversation, options, capturedAt, gap))\n }\n return { rows, missingTraces }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAqEA,IAAM,SAAS,CAAC,GAAY,UAAmC;AAC7D,QAAM,IAAI,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC;AACtD,SAAO,MAAM,KAAK,EAAE;AACtB;AAEA,SAAS,YAAY,MAAY,OAAqC;AACpE,QAAM,OAAoB;AAAA,IACxB,MAAM,KAAK;AAAA,IACX,MAAM,MAAM,KAAK,IAAI;AAAA,IACrB,QAAQ,KAAK;AAAA,IACb,YAAY,KAAK,YAAY,SAAY,KAAK,UAAU,KAAK,YAAY;AAAA,EAC3E;AACA,MAAI,KAAK,SAAS,OAAO;AACvB,UAAM,MAAM;AACZ,UAAM,OAAO,IAAI,SAAS,IAAI,SAAS,SAAS,CAAC;AACjD,QAAI,KAAM,MAAK,QAAQ,MAAM,KAAK,OAAO;AACzC,QAAI,IAAI,WAAW,OAAW,MAAK,SAAS,MAAM,IAAI,MAAM;AAAA,EAC9D,WAAW,KAAK,SAAS,QAAQ;AAC/B,UAAM,OAAO;AACb,SAAK,QAAQ,OAAO,KAAK,MAAM,KAAK;AACpC,QAAI,KAAK,WAAW,OAAW,MAAK,SAAS,OAAO,KAAK,QAAQ,KAAK;AAAA,EACxE;AACA,SAAO;AACT;AAGA,SAAS,kBAAkB,OAAe,OAAuC;AAC/E,QAAM,OAAO,MAAM,OAAO,CAAC,MAAoB,EAAE,SAAS,KAAK;AAC/D,QAAM,OAAO,KAAK,KAAK,SAAS,CAAC;AACjC,MAAI,CAAC,KAAM,QAAO,CAAC;AACnB,QAAM,WAA0B,KAAK,SAAS,IAAI,CAAC,OAAgB;AAAA,IACjE,MAAM,EAAE;AAAA,IACR,SAAS,MAAM,EAAE,OAAO;AAAA,EAC1B,EAAE;AACF,MAAI,KAAK,WAAW,UAAa,KAAK,WAAW,IAAI;AACnD,aAAS,KAAK,EAAE,MAAM,aAAa,SAAS,MAAM,KAAK,MAAM,EAAE,CAAC;AAAA,EAClE;AACA,SAAO;AACT;AAgBA,IAAM,gBAAgE;AAAA,EACpE,SAAS;AAAA,EACT,QAAQ;AAAA,EACR,UAAU;AACZ;AAUA,SAAS,iBAAiB,QAAyB;AACjD,MAAI,aAAa,MAAM,MAAM,QAAW;AACtC,UAAM,IAAI,gBAAgB,+BAA+B,OAAO,KAAK,yBAAyB;AAAA,EAChG;AACF;AAEA,IAAM,iBAA8D;AAAA,EAClE,QAAQ;AAAA,EACR,KAAK;AAAA,EACL,SAAS;AACX;AAEA,SAAS,SACP,QACA,OACA,UACA,SACA,YACA,KACmB;AAInB,mBAAiB,MAAM;AAGvB,QAAM,eAAe,oBAAoB,MAAM;AAC/C,QAAM,aAAa,OAAO,eAAe,SAAS;AAClD,QAAM,kBAAkB,OAAO;AAC/B,QAAM,cAAc,oBAAoB,eAAe,oBAAoB;AAC3E,QAAM,cAAc,oBAAoB,eAAe,oBAAoB;AAC3E,QAAM,gBACJ,oBAAoB,YACpB,oBAAoB,eACpB,oBAAoB,eACf,OAAO,yBAAyB,aAAa,eAAe,KAC7D;AAIN,SAAO;AAAA,IACL;AAAA,MACE,QAAQ;AAAA,MACR,YAAY,OAAO;AAAA,MACnB,mBAAmB;AAAA,MACnB,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,cAAc,OAAO;AAAA,MACrB,YAAY;AAAA,MACZ,iBAAiB;AAAA,MACjB,MAAM,QAAQ,QAAQ;AAAA,MACtB,MAAM;AAAA,QACJ,OAAO,QAAQ,SAAS,OAAO;AAAA,QAC/B,aAAa,OAAO;AAAA,QACpB,OAAO,eAAe,OAAO,QAAQ;AAAA,QACrC,MAAM,OAAO;AAAA,QACb,KAAK;AAAA,MACP;AAAA,MACA,QAAQ;AAAA,QACN,SAAS;AAAA,QACT,iBAAiB;AAAA,QACjB,OAAO,OAAO;AAAA,QACd,UAAU;AAAA,QACV,gBAAgB,OAAO;AAAA,QACvB,aAAa,OAAO;AAAA,QACpB,aAAa,OAAO;AAAA,QACpB,uBAAuB,OAAO,cAAc,UAAU;AAAA,QACtD,UAAU;AAAA,MACZ;AAAA,MACA;AAAA,MACA,WAAW,CAAC;AAAA,MACZ,GAAI,MAAM,SAAS,IAAI,EAAE,MAAM,IAAI,CAAC;AAAA,MACpC,SAAS;AAAA,QACP,GAAG;AAAA,QACH,eAAe,cAAc,YAAY,MAAM,CAAC;AAAA,QAChD,SAAS;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,QAUT,SAAS,EAAE,GAAG,OAAO,QAAQ,IAAI;AAAA,QACjC,cAAc;AAAA,QACd,cAAc;AAAA,QACd,OAAO;AAAA,MACT;AAAA,MACA,MAAM;AAAA,QACJ,KAAK,aAAa,OAAO,OAAO;AAAA,QAChC,WAAW,OAAO,WAAW;AAAA,QAC7B,YAAY,OAAO,WAAW;AAAA,QAC9B,kBAAkB,OAAO,WAAW,aAAa;AAAA,QACjD,YAAY,OAAO,WAAW,UAAU;AAAA,QACxC,aAAa,OAAO,WAAW,cAAc;AAAA,QAC7C,QAAQ,KAAK,MAAM,OAAO,SAAS,GAAI;AAAA,MACzC;AAAA,MACA,WAAW,EAAE,YAAY,MAAM,SAAS,MAAM,gBAAgB,KAAK;AAAA,MACnE,YAAY;AAAA,QACV,aAAa;AAAA,QACb,SAAS;AAAA,QACT,GAAI,QAAQ,SAAY,EAAE,IAAI,IAAI,CAAC;AAAA,MACrC;AAAA,IACF;AAAA,IACA,+BAA+B,OAAO,KAAK;AAAA,EAC7C;AACF;AAQA,eAAsB,gBACpB,SACA,OACA,UAA8B,CAAC,GACH;AAC5B,QAAM,QAAQ,QAAQ,UAAU,CAAC,MAAM;AACvC,QAAM,cAAc,QAAQ,MAAM,KAAK,oBAAI,KAAK,GAAG,YAAY;AAC/D,QAAM,OAA4B,CAAC;AACnC,QAAM,gBAA0B,CAAC;AACjC,aAAW,UAAU,SAAS;AAC5B,UAAM,aAAa,MAAM,gBAAgB,OAAO,OAAO,KAAK;AAC5D,QAAI,WAAW,MAAM,WAAW,GAAG;AACjC,oBAAc,KAAK,OAAO,KAAK;AAC/B,WAAK;AAAA,QACH,SAAS,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,YAAY,wCAAwC;AAAA,MACxF;AACA;AAAA,IACF;AACA,QAAI,QAAQ,WAAW,MAAM,IAAI,CAAC,MAAM,YAAY,EAAE,MAAM,KAAK,CAAC;AAClE,QAAI,QAAQ,aAAa,UAAa,MAAM,SAAS,QAAQ,UAAU;AAGrE,YAAM,OAAO,KAAK,KAAK,QAAQ,WAAW,CAAC;AAC3C,YAAM,OAAO,QAAQ,WAAW;AAChC,cAAQ,CAAC,GAAG,MAAM,MAAM,GAAG,IAAI,GAAG,GAAG,MAAM,MAAM,MAAM,SAAS,IAAI,CAAC;AAAA,IACvE;AACA,UAAM,eAAe;AAAA,MACnB,WAAW,MAAM,IAAI,CAAC,MAAM,EAAE,IAAI;AAAA,MAClC;AAAA,IACF;AACA,UAAM,MACJ,aAAa,WAAW,IAAI,4DAAuD;AACrF,SAAK,KAAK,SAAS,QAAQ,OAAO,cAAc,SAAS,YAAY,GAAG,CAAC;AAAA,EAC3E;AACA,SAAO,EAAE,MAAM,cAAc;AAC/B;","names":[]}
@@ -146,16 +146,14 @@ ${reasoning}`;
146
146
  }
147
147
 
148
148
  // src/rollout/readers/opencode-sqlite.ts
149
+ import { createRequire } from "module";
149
150
  import { homedir as homedir2 } from "os";
150
151
  import { join as join2 } from "path";
151
152
  var DEFAULT_OPENCODE_DB = join2(homedir2(), ".local", "share", "opencode", "opencode.db");
152
- var NODE_SQLITE_SPECIFIER = ["node", "sqlite"].join(":");
153
+ var nodeRequire = createRequire(import.meta.url);
153
154
  async function openOpencodeDb(path = DEFAULT_OPENCODE_DB) {
154
155
  try {
155
- const { DatabaseSync } = await import(
156
- /* @vite-ignore */
157
- NODE_SQLITE_SPECIFIER
158
- );
156
+ const { DatabaseSync } = nodeRequire("node:sqlite");
159
157
  const db = new DatabaseSync(path, { readOnly: true });
160
158
  db.prepare("SELECT id FROM session LIMIT 1").get();
161
159
  return db;
@@ -288,4 +286,4 @@ export {
288
286
  findOpencodeSessionById,
289
287
  readOpencodeSessionMessages
290
288
  };
291
- //# sourceMappingURL=chunk-VBQ3CRKH.js.map
289
+ //# sourceMappingURL=chunk-HPWUNB47.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["../src/rollout/readers/claude-jsonl.ts","../src/rollout/readers/opencode-sqlite.ts"],"sourcesContent":["/**\n * Backfill reader over Claude Code project transcripts\n * (~/.claude/projects/<cwd-slug>/<sessionId>.jsonl) → canonical\n * chat-with-tools messages plus per-session token usage.\n *\n * Transcript lines consumed: type:\"user\" (string content or content blocks —\n * text + tool_result) and type:\"assistant\" (content blocks — thinking, text,\n * tool_use; message.usage carries tokens). Sidechain lines (isSidechain=true,\n * subagent threads) are separate invocations and are excluded from the main\n * transcript. Everything else (queue-operation, attachment, last-prompt…) is\n * transport metadata, not conversation.\n */\n\nimport { readdir, readFile } from 'node:fs/promises'\nimport { homedir } from 'node:os'\nimport { join } from 'node:path'\nimport type { ChatMessage, ChatToolCall } from '../schema'\n\nexport const DEFAULT_CLAUDE_PROJECTS_DIR = join(homedir(), '.claude', 'projects')\n\n/** Claude Code's project-directory slug for a working directory. */\nexport function claudeProjectSlug(cwd: string): string {\n return cwd.replace(/[^a-zA-Z0-9-]/g, '-')\n}\n\nexport interface ClaudeTranscriptRef {\n sessionId: string\n path: string\n}\n\n/** Transcript files recorded for sessions launched from `cwd`. */\nexport async function findClaudeTranscripts(\n cwd: string,\n projectsDir: string = DEFAULT_CLAUDE_PROJECTS_DIR,\n): Promise<ClaudeTranscriptRef[]> {\n const dir = join(projectsDir, claudeProjectSlug(cwd))\n const names = await readdir(dir).catch(() => [])\n return names\n .filter((n) => n.endsWith('.jsonl'))\n .sort()\n .map((n) => ({ sessionId: n.replace(/\\.jsonl$/, ''), path: join(dir, n) }))\n}\n\nexport interface ClaudeUsageTotals {\n tokensIn: number\n tokensOut: number\n cacheRead: number\n cacheWrite: number\n}\n\nexport interface ClaudeTranscript {\n messages: ChatMessage[]\n usage: ClaudeUsageTotals\n /** Timestamp of the first conversation line; null = empty transcript. */\n startedAt: string | null\n endedAt: string | null\n model: string | null\n}\n\nconst isRecord = (v: unknown): v is Record<string, unknown> =>\n typeof v === 'object' && v !== null && !Array.isArray(v)\n\n/**\n * One conversation line of a transcript, still in Claude Code's own shape.\n *\n * This is the single line-level parse of the format. `readClaudeTranscript`\n * projects it to canonical messages + usage; the supervision-tree reader\n * (`src/supervisor-run/claude-code-reader.ts`) projects the SAME entries to\n * spawn/settle/steer instants. Two projections, one parser — a second\n * transcript parser is how the two views silently disagree.\n */\nexport interface ClaudeEntry {\n readonly type: 'user' | 'assistant'\n /** ISO instant of the line; null when the line carried none. */\n readonly timestamp: string | null\n /** The Anthropic message body (`role`, `content`, `model`, `usage`). */\n readonly message: Record<string, unknown>\n /** Claude Code's structured tool result, when the line carries one. */\n readonly toolUseResult: unknown\n /** True on subagent threads — a separate invocation, not this transcript's turn. */\n readonly isSidechain: boolean\n /** Subagent id Claude Code stamps on sidechain lines; null on main-thread lines. */\n readonly agentId: string | null\n}\n\n/** Parse transcript jsonl text into conversation lines. Non-conversation lines are dropped. */\nexport function parseClaudeEntries(raw: string): ClaudeEntry[] {\n const out: ClaudeEntry[] = []\n for (const line of raw.split('\\n')) {\n if (!line.trim()) continue\n let entry: Record<string, unknown>\n try {\n const parsed: unknown = JSON.parse(line)\n if (!isRecord(parsed)) continue\n entry = parsed\n } catch {\n continue\n }\n if (entry.type !== 'user' && entry.type !== 'assistant') continue\n const message = entry.message\n if (!isRecord(message)) continue\n out.push({\n type: entry.type,\n timestamp: typeof entry.timestamp === 'string' ? entry.timestamp : null,\n message,\n toolUseResult: entry.toolUseResult,\n isSidechain: entry.isSidechain === true,\n agentId: typeof entry.agentId === 'string' ? entry.agentId : null,\n })\n }\n return out\n}\n\nexport interface ReadClaudeTranscriptOptions {\n /**\n * Read the sidechain (subagent) thread instead of skipping it. Subagent\n * transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain\n * lines end to end, so their usage is invisible without this.\n */\n readonly includeSidechain?: boolean\n}\n\nfunction blockText(content: unknown): string {\n if (typeof content === 'string') return content\n if (!Array.isArray(content)) return ''\n return content\n .filter(\n (b): b is Record<string, unknown> =>\n isRecord(b) && b.type === 'text' && typeof b.text === 'string',\n )\n .map((b) => b.text as string)\n .join('\\n')\n}\n\n/** Parse one transcript jsonl into canonical messages + usage totals. */\nexport async function readClaudeTranscript(\n path: string,\n options: ReadClaudeTranscriptOptions = {},\n): Promise<ClaudeTranscript> {\n return transcriptFromEntries(parseClaudeEntries(await readFile(path, 'utf8')), options)\n}\n\n/** The messages+usage projection of already-parsed entries. */\nexport function transcriptFromEntries(\n entries: readonly ClaudeEntry[],\n options: ReadClaudeTranscriptOptions = {},\n): ClaudeTranscript {\n const wantSidechain = options.includeSidechain === true\n const messages: ChatMessage[] = []\n const usage: ClaudeUsageTotals = { tokensIn: 0, tokensOut: 0, cacheRead: 0, cacheWrite: 0 }\n let startedAt: string | null = null\n let endedAt: string | null = null\n let model: string | null = null\n // Claude Code writes one jsonl line PER CONTENT BLOCK of an API message,\n // repeating message.id and usage on each — merge blocks into one canonical\n // assistant turn and count usage once per API message id.\n let lastAssistantApiId: string | null = null\n let lastAssistantIndex = -1\n\n for (const entry of entries) {\n if (entry.isSidechain !== wantSidechain) continue\n const message = entry.message\n if (entry.timestamp !== null) {\n if (startedAt === null) startedAt = entry.timestamp\n endedAt = entry.timestamp\n }\n\n if (entry.type === 'user') {\n lastAssistantApiId = null\n lastAssistantIndex = -1\n const content = message.content\n if (typeof content === 'string') {\n messages.push({ role: 'user', content })\n continue\n }\n if (!Array.isArray(content)) continue\n // A user line may interleave tool_result blocks (answers to the prior\n // assistant tool_use) with plain text; preserve order.\n let userText = ''\n for (const block of content) {\n if (!isRecord(block)) continue\n if (block.type === 'tool_result' && typeof block.tool_use_id === 'string') {\n messages.push({\n role: 'tool',\n tool_call_id: block.tool_use_id,\n content:\n blockText(block.content) || (typeof block.content === 'string' ? block.content : ''),\n })\n } else if (block.type === 'text' && typeof block.text === 'string') {\n userText += (userText.length > 0 ? '\\n' : '') + block.text\n }\n }\n if (userText.length > 0) messages.push({ role: 'user', content: userText })\n continue\n }\n\n // assistant\n if (typeof message.model === 'string') model = message.model\n const apiId = typeof message.id === 'string' ? message.id : null\n const continuesTurn = apiId !== null && apiId === lastAssistantApiId && lastAssistantIndex >= 0\n const msgUsage = message.usage\n if (isRecord(msgUsage) && !continuesTurn) {\n usage.tokensIn += typeof msgUsage.input_tokens === 'number' ? msgUsage.input_tokens : 0\n usage.tokensOut += typeof msgUsage.output_tokens === 'number' ? msgUsage.output_tokens : 0\n usage.cacheRead +=\n typeof msgUsage.cache_read_input_tokens === 'number' ? msgUsage.cache_read_input_tokens : 0\n usage.cacheWrite +=\n typeof msgUsage.cache_creation_input_tokens === 'number'\n ? msgUsage.cache_creation_input_tokens\n : 0\n }\n const content = message.content\n if (!Array.isArray(content)) continue\n let reasoning = ''\n let text = ''\n const toolCalls: ChatToolCall[] = []\n for (const block of content) {\n if (!isRecord(block)) continue\n if (\n block.type === 'thinking' &&\n typeof block.thinking === 'string' &&\n block.thinking.length > 0\n ) {\n reasoning += (reasoning.length > 0 ? '\\n' : '') + block.thinking\n } else if (block.type === 'text' && typeof block.text === 'string') {\n text += (text.length > 0 ? '\\n' : '') + block.text\n } else if (block.type === 'tool_use' && typeof block.id === 'string') {\n toolCalls.push({\n id: block.id,\n type: 'function',\n function: {\n name: typeof block.name === 'string' ? block.name : 'unknown',\n arguments: JSON.stringify(block.input ?? {}),\n },\n })\n }\n }\n if (reasoning.length === 0 && text.length === 0 && toolCalls.length === 0) continue\n if (continuesTurn) {\n const prev = messages[lastAssistantIndex]!\n if (text.length > 0) prev.content = prev.content === null ? text : `${prev.content}\\n${text}`\n if (reasoning.length > 0) {\n prev.reasoning_content =\n prev.reasoning_content === undefined\n ? reasoning\n : `${prev.reasoning_content}\\n${reasoning}`\n }\n if (toolCalls.length > 0) prev.tool_calls = [...(prev.tool_calls ?? []), ...toolCalls]\n continue\n }\n messages.push({\n role: 'assistant',\n content: text.length > 0 ? text : null,\n ...(reasoning.length > 0 ? { reasoning_content: reasoning } : {}),\n ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),\n })\n lastAssistantApiId = apiId\n lastAssistantIndex = messages.length - 1\n }\n\n return { messages, usage, startedAt, endedAt, model }\n}\n","/**\n * Read-only backfill reader over the opencode sqlite store\n * (~/.local/share/opencode/opencode.db) → canonical chat-with-tools messages.\n *\n * Schema consumed (observed, 2026-07): `session` rows carry directory /\n * parent_id / agent / model / cost / tokens_*; `message` rows carry a JSON\n * `data` blob ({role, modelID, providerID, tokens, cost, finish}); `part`\n * rows carry the actual content ({type: text|reasoning|tool|step-start|\n * step-finish|snapshot…}). Tool parts hold {callID, state:{input, output,\n * status}} — both the call and its result, which we split into an assistant\n * tool_call plus a role:\"tool\" result message.\n *\n * The store is mutable and can be corrupt (a `.corrupt-bak` sibling ships\n * next to it in the wild), so `openOpencodeDb` returns null instead of\n * throwing — callers record a gap line, never crash the backfill.\n */\n\nimport { createRequire } from 'node:module'\nimport { homedir } from 'node:os'\nimport { join } from 'node:path'\nimport type { DatabaseSync } from 'node:sqlite'\nimport type { ChatMessage, ChatToolCall } from '../schema'\n\nexport const DEFAULT_OPENCODE_DB = join(homedir(), '.local', 'share', 'opencode', 'opencode.db')\n\nexport interface OpencodeSessionRow {\n id: string\n parentId: string | null\n directory: string\n agent: string | null\n /** Raw session.model JSON: {id, providerID, variant} where present. */\n model: { id?: string; providerID?: string } | null\n costUsd: number\n tokensInput: number\n tokensOutput: number\n tokensReasoning: number\n tokensCacheRead: number\n tokensCacheWrite: number\n timeCreated: number\n timeUpdated: number\n}\n\n// `node:sqlite` is loaded through CommonJS `require`, not `import()`. esbuild\n// (bundling) and Vite (tests) both rewrite a dynamic import and strip the\n// `node:` prefix under an es20xx target, turning this builtin into a bogus\n// \"sqlite\" package lookup; composing the specifier at runtime does not reliably\n// defeat that (it still resolved through Vite's transform in some workers, so\n// the failure moved around as test files were added). A require obtained from\n// `createRequire` is not an analyzable module reference in either tool, so\n// neither can rewrite it.\nconst nodeRequire = createRequire(import.meta.url)\n\n/** Open the store read-only; null = unavailable/corrupt (caller records a gap). */\nexport async function openOpencodeDb(\n path: string = DEFAULT_OPENCODE_DB,\n): Promise<DatabaseSync | null> {\n try {\n const { DatabaseSync } = nodeRequire('node:sqlite') as typeof import('node:sqlite')\n const db = new DatabaseSync(path, { readOnly: true })\n // Probe: a corrupt store can open() fine and fail on first page read.\n db.prepare('SELECT id FROM session LIMIT 1').get()\n return db\n } catch {\n return null\n }\n}\n\nconst isRecord = (v: unknown): v is Record<string, unknown> =>\n typeof v === 'object' && v !== null && !Array.isArray(v)\n\nfunction parseSessionRow(row: Record<string, unknown>): OpencodeSessionRow {\n let model: OpencodeSessionRow['model'] = null\n if (typeof row.model === 'string' && row.model.length > 0) {\n try {\n const parsed: unknown = JSON.parse(row.model)\n if (isRecord(parsed)) model = parsed as { id?: string; providerID?: string }\n } catch {\n model = null\n }\n }\n return {\n id: String(row.id),\n parentId: row.parent_id === null || row.parent_id === undefined ? null : String(row.parent_id),\n directory: String(row.directory),\n agent: row.agent === null || row.agent === undefined ? null : String(row.agent),\n model,\n costUsd: Number(row.cost ?? 0),\n tokensInput: Number(row.tokens_input ?? 0),\n tokensOutput: Number(row.tokens_output ?? 0),\n tokensReasoning: Number(row.tokens_reasoning ?? 0),\n tokensCacheRead: Number(row.tokens_cache_read ?? 0),\n tokensCacheWrite: Number(row.tokens_cache_write ?? 0),\n timeCreated: Number(row.time_created ?? 0),\n timeUpdated: Number(row.time_updated ?? 0),\n }\n}\n\nconst SESSION_COLUMNS =\n 'id, parent_id, directory, agent, model, cost, tokens_input, tokens_output, tokens_reasoning, tokens_cache_read, tokens_cache_write, time_created, time_updated'\n\n/** Sessions whose cwd is `directory` (the worker-clone join key). */\nexport function findOpencodeSessionsByDirectory(\n db: DatabaseSync,\n directory: string,\n): OpencodeSessionRow[] {\n const rows = db\n .prepare(`SELECT ${SESSION_COLUMNS} FROM session WHERE directory = ? ORDER BY time_created`)\n .all(directory) as Array<Record<string, unknown>>\n return rows.map(parseSessionRow)\n}\n\nexport function findOpencodeSessionById(\n db: DatabaseSync,\n sessionId: string,\n): OpencodeSessionRow | null {\n const row = db.prepare(`SELECT ${SESSION_COLUMNS} FROM session WHERE id = ?`).get(sessionId) as\n | Record<string, unknown>\n | undefined\n return row === undefined ? null : parseSessionRow(row)\n}\n\ninterface OpencodePart {\n type?: string\n text?: string\n tool?: string\n callID?: string\n state?: { status?: string; input?: unknown; output?: unknown }\n}\n\nfunction toolResultContent(output: unknown): string {\n if (typeof output === 'string') return output\n if (output === null || output === undefined) return ''\n return JSON.stringify(output)\n}\n\n/**\n * Convert one session's message+part rows into canonical messages.\n * An opencode assistant message row spans several model steps; each step's\n * parts (reasoning → text → tool …) become one assistant message followed by\n * the role:\"tool\" results of its calls, preserving order.\n */\nexport function readOpencodeSessionMessages(db: DatabaseSync, sessionId: string): ChatMessage[] {\n const messageRows = db\n .prepare('SELECT id, data FROM message WHERE session_id = ? ORDER BY time_created, id')\n .all(sessionId) as Array<{ id: string; data: string }>\n const partsStmt = db.prepare('SELECT data FROM part WHERE message_id = ? ORDER BY id')\n\n const messages: ChatMessage[] = []\n for (const messageRow of messageRows) {\n let data: Record<string, unknown>\n try {\n const parsed: unknown = JSON.parse(messageRow.data)\n if (!isRecord(parsed)) continue\n data = parsed\n } catch {\n continue\n }\n const parts: OpencodePart[] = []\n for (const row of partsStmt.all(messageRow.id) as Array<{ data: string }>) {\n try {\n const parsed: unknown = JSON.parse(row.data)\n if (isRecord(parsed)) parts.push(parsed as OpencodePart)\n } catch {\n // Malformed part payload: skip the part, keep the message.\n }\n }\n\n if (data.role === 'user') {\n const text = parts\n .filter((p) => p.type === 'text' && typeof p.text === 'string')\n .map((p) => p.text as string)\n .join('\\n')\n messages.push({ role: 'user', content: text })\n continue\n }\n if (data.role !== 'assistant') continue\n\n // Split the row into steps at step-start boundaries; parts before the\n // first step-start (none observed, but tolerated) form an implicit step.\n const steps: OpencodePart[][] = []\n let current: OpencodePart[] = []\n for (const part of parts) {\n if (part.type === 'step-start') {\n if (current.length > 0) steps.push(current)\n current = []\n continue\n }\n if (part.type === 'step-finish' || part.type === 'snapshot' || part.type === 'patch') continue\n current.push(part)\n }\n if (current.length > 0) steps.push(current)\n\n for (const step of steps) {\n const reasoning = step\n .filter((p) => p.type === 'reasoning' && typeof p.text === 'string' && p.text.length > 0)\n .map((p) => p.text as string)\n .join('\\n')\n const text = step\n .filter((p) => p.type === 'text' && typeof p.text === 'string')\n .map((p) => p.text as string)\n .join('\\n')\n const toolParts = step.filter((p) => p.type === 'tool' && typeof p.callID === 'string')\n const toolCalls: ChatToolCall[] = toolParts.map((p) => ({\n id: p.callID as string,\n type: 'function',\n function: {\n name: p.tool ?? 'unknown',\n arguments: JSON.stringify(p.state?.input ?? {}),\n },\n }))\n if (reasoning.length === 0 && text.length === 0 && toolCalls.length === 0) continue\n messages.push({\n role: 'assistant',\n content: text.length > 0 ? text : null,\n ...(reasoning.length > 0 ? { reasoning_content: reasoning } : {}),\n ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),\n })\n for (const p of toolParts) {\n messages.push({\n role: 'tool',\n tool_call_id: p.callID as string,\n name: p.tool ?? 'unknown',\n content: toolResultContent(p.state?.output),\n })\n }\n }\n }\n return messages\n}\n"],"mappings":";AAaA,SAAS,SAAS,gBAAgB;AAClC,SAAS,eAAe;AACxB,SAAS,YAAY;AAGd,IAAM,8BAA8B,KAAK,QAAQ,GAAG,WAAW,UAAU;AAGzE,SAAS,kBAAkB,KAAqB;AACrD,SAAO,IAAI,QAAQ,kBAAkB,GAAG;AAC1C;AAQA,eAAsB,sBACpB,KACA,cAAsB,6BACU;AAChC,QAAM,MAAM,KAAK,aAAa,kBAAkB,GAAG,CAAC;AACpD,QAAM,QAAQ,MAAM,QAAQ,GAAG,EAAE,MAAM,MAAM,CAAC,CAAC;AAC/C,SAAO,MACJ,OAAO,CAAC,MAAM,EAAE,SAAS,QAAQ,CAAC,EAClC,KAAK,EACL,IAAI,CAAC,OAAO,EAAE,WAAW,EAAE,QAAQ,YAAY,EAAE,GAAG,MAAM,KAAK,KAAK,CAAC,EAAE,EAAE;AAC9E;AAkBA,IAAM,WAAW,CAAC,MAChB,OAAO,MAAM,YAAY,MAAM,QAAQ,CAAC,MAAM,QAAQ,CAAC;AA0BlD,SAAS,mBAAmB,KAA4B;AAC7D,QAAM,MAAqB,CAAC;AAC5B,aAAW,QAAQ,IAAI,MAAM,IAAI,GAAG;AAClC,QAAI,CAAC,KAAK,KAAK,EAAG;AAClB,QAAI;AACJ,QAAI;AACF,YAAM,SAAkB,KAAK,MAAM,IAAI;AACvC,UAAI,CAAC,SAAS,MAAM,EAAG;AACvB,cAAQ;AAAA,IACV,QAAQ;AACN;AAAA,IACF;AACA,QAAI,MAAM,SAAS,UAAU,MAAM,SAAS,YAAa;AACzD,UAAM,UAAU,MAAM;AACtB,QAAI,CAAC,SAAS,OAAO,EAAG;AACxB,QAAI,KAAK;AAAA,MACP,MAAM,MAAM;AAAA,MACZ,WAAW,OAAO,MAAM,cAAc,WAAW,MAAM,YAAY;AAAA,MACnE;AAAA,MACA,eAAe,MAAM;AAAA,MACrB,aAAa,MAAM,gBAAgB;AAAA,MACnC,SAAS,OAAO,MAAM,YAAY,WAAW,MAAM,UAAU;AAAA,IAC/D,CAAC;AAAA,EACH;AACA,SAAO;AACT;AAWA,SAAS,UAAU,SAA0B;AAC3C,MAAI,OAAO,YAAY,SAAU,QAAO;AACxC,MAAI,CAAC,MAAM,QAAQ,OAAO,EAAG,QAAO;AACpC,SAAO,QACJ;AAAA,IACC,CAAC,MACC,SAAS,CAAC,KAAK,EAAE,SAAS,UAAU,OAAO,EAAE,SAAS;AAAA,EAC1D,EACC,IAAI,CAAC,MAAM,EAAE,IAAc,EAC3B,KAAK,IAAI;AACd;AAGA,eAAsB,qBACpB,MACA,UAAuC,CAAC,GACb;AAC3B,SAAO,sBAAsB,mBAAmB,MAAM,SAAS,MAAM,MAAM,CAAC,GAAG,OAAO;AACxF;AAGO,SAAS,sBACd,SACA,UAAuC,CAAC,GACtB;AAClB,QAAM,gBAAgB,QAAQ,qBAAqB;AACnD,QAAM,WAA0B,CAAC;AACjC,QAAM,QAA2B,EAAE,UAAU,GAAG,WAAW,GAAG,WAAW,GAAG,YAAY,EAAE;AAC1F,MAAI,YAA2B;AAC/B,MAAI,UAAyB;AAC7B,MAAI,QAAuB;AAI3B,MAAI,qBAAoC;AACxC,MAAI,qBAAqB;AAEzB,aAAW,SAAS,SAAS;AAC3B,QAAI,MAAM,gBAAgB,cAAe;AACzC,UAAM,UAAU,MAAM;AACtB,QAAI,MAAM,cAAc,MAAM;AAC5B,UAAI,cAAc,KAAM,aAAY,MAAM;AAC1C,gBAAU,MAAM;AAAA,IAClB;AAEA,QAAI,MAAM,SAAS,QAAQ;AACzB,2BAAqB;AACrB,2BAAqB;AACrB,YAAMA,WAAU,QAAQ;AACxB,UAAI,OAAOA,aAAY,UAAU;AAC/B,iBAAS,KAAK,EAAE,MAAM,QAAQ,SAAAA,SAAQ,CAAC;AACvC;AAAA,MACF;AACA,UAAI,CAAC,MAAM,QAAQA,QAAO,EAAG;AAG7B,UAAI,WAAW;AACf,iBAAW,SAASA,UAAS;AAC3B,YAAI,CAAC,SAAS,KAAK,EAAG;AACtB,YAAI,MAAM,SAAS,iBAAiB,OAAO,MAAM,gBAAgB,UAAU;AACzE,mBAAS,KAAK;AAAA,YACZ,MAAM;AAAA,YACN,cAAc,MAAM;AAAA,YACpB,SACE,UAAU,MAAM,OAAO,MAAM,OAAO,MAAM,YAAY,WAAW,MAAM,UAAU;AAAA,UACrF,CAAC;AAAA,QACH,WAAW,MAAM,SAAS,UAAU,OAAO,MAAM,SAAS,UAAU;AAClE,uBAAa,SAAS,SAAS,IAAI,OAAO,MAAM,MAAM;AAAA,QACxD;AAAA,MACF;AACA,UAAI,SAAS,SAAS,EAAG,UAAS,KAAK,EAAE,MAAM,QAAQ,SAAS,SAAS,CAAC;AAC1E;AAAA,IACF;AAGA,QAAI,OAAO,QAAQ,UAAU,SAAU,SAAQ,QAAQ;AACvD,UAAM,QAAQ,OAAO,QAAQ,OAAO,WAAW,QAAQ,KAAK;AAC5D,UAAM,gBAAgB,UAAU,QAAQ,UAAU,sBAAsB,sBAAsB;AAC9F,UAAM,WAAW,QAAQ;AACzB,QAAI,SAAS,QAAQ,KAAK,CAAC,eAAe;AACxC,YAAM,YAAY,OAAO,SAAS,iBAAiB,WAAW,SAAS,eAAe;AACtF,YAAM,aAAa,OAAO,SAAS,kBAAkB,WAAW,SAAS,gBAAgB;AACzF,YAAM,aACJ,OAAO,SAAS,4BAA4B,WAAW,SAAS,0BAA0B;AAC5F,YAAM,cACJ,OAAO,SAAS,gCAAgC,WAC5C,SAAS,8BACT;AAAA,IACR;AACA,UAAM,UAAU,QAAQ;AACxB,QAAI,CAAC,MAAM,QAAQ,OAAO,EAAG;AAC7B,QAAI,YAAY;AAChB,QAAI,OAAO;AACX,UAAM,YAA4B,CAAC;AACnC,eAAW,SAAS,SAAS;AAC3B,UAAI,CAAC,SAAS,KAAK,EAAG;AACtB,UACE,MAAM,SAAS,cACf,OAAO,MAAM,aAAa,YAC1B,MAAM,SAAS,SAAS,GACxB;AACA,sBAAc,UAAU,SAAS,IAAI,OAAO,MAAM,MAAM;AAAA,MAC1D,WAAW,MAAM,SAAS,UAAU,OAAO,MAAM,SAAS,UAAU;AAClE,iBAAS,KAAK,SAAS,IAAI,OAAO,MAAM,MAAM;AAAA,MAChD,WAAW,MAAM,SAAS,cAAc,OAAO,MAAM,OAAO,UAAU;AACpE,kBAAU,KAAK;AAAA,UACb,IAAI,MAAM;AAAA,UACV,MAAM;AAAA,UACN,UAAU;AAAA,YACR,MAAM,OAAO,MAAM,SAAS,WAAW,MAAM,OAAO;AAAA,YACpD,WAAW,KAAK,UAAU,MAAM,SAAS,CAAC,CAAC;AAAA,UAC7C;AAAA,QACF,CAAC;AAAA,MACH;AAAA,IACF;AACA,QAAI,UAAU,WAAW,KAAK,KAAK,WAAW,KAAK,UAAU,WAAW,EAAG;AAC3E,QAAI,eAAe;AACjB,YAAM,OAAO,SAAS,kBAAkB;AACxC,UAAI,KAAK,SAAS,EAAG,MAAK,UAAU,KAAK,YAAY,OAAO,OAAO,GAAG,KAAK,OAAO;AAAA,EAAK,IAAI;AAC3F,UAAI,UAAU,SAAS,GAAG;AACxB,aAAK,oBACH,KAAK,sBAAsB,SACvB,YACA,GAAG,KAAK,iBAAiB;AAAA,EAAK,SAAS;AAAA,MAC/C;AACA,UAAI,UAAU,SAAS,EAAG,MAAK,aAAa,CAAC,GAAI,KAAK,cAAc,CAAC,GAAI,GAAG,SAAS;AACrF;AAAA,IACF;AACA,aAAS,KAAK;AAAA,MACZ,MAAM;AAAA,MACN,SAAS,KAAK,SAAS,IAAI,OAAO;AAAA,MAClC,GAAI,UAAU,SAAS,IAAI,EAAE,mBAAmB,UAAU,IAAI,CAAC;AAAA,MAC/D,GAAI,UAAU,SAAS,IAAI,EAAE,YAAY,UAAU,IAAI,CAAC;AAAA,IAC1D,CAAC;AACD,yBAAqB;AACrB,yBAAqB,SAAS,SAAS;AAAA,EACzC;AAEA,SAAO,EAAE,UAAU,OAAO,WAAW,SAAS,MAAM;AACtD;;;ACpPA,SAAS,qBAAqB;AAC9B,SAAS,WAAAC,gBAAe;AACxB,SAAS,QAAAC,aAAY;AAId,IAAM,sBAAsBA,MAAKD,SAAQ,GAAG,UAAU,SAAS,YAAY,aAAa;AA2B/F,IAAM,cAAc,cAAc,YAAY,GAAG;AAGjD,eAAsB,eACpB,OAAe,qBACe;AAC9B,MAAI;AACF,UAAM,EAAE,aAAa,IAAI,YAAY,aAAa;AAClD,UAAM,KAAK,IAAI,aAAa,MAAM,EAAE,UAAU,KAAK,CAAC;AAEpD,OAAG,QAAQ,gCAAgC,EAAE,IAAI;AACjD,WAAO;AAAA,EACT,QAAQ;AACN,WAAO;AAAA,EACT;AACF;AAEA,IAAME,YAAW,CAAC,MAChB,OAAO,MAAM,YAAY,MAAM,QAAQ,CAAC,MAAM,QAAQ,CAAC;AAEzD,SAAS,gBAAgB,KAAkD;AACzE,MAAI,QAAqC;AACzC,MAAI,OAAO,IAAI,UAAU,YAAY,IAAI,MAAM,SAAS,GAAG;AACzD,QAAI;AACF,YAAM,SAAkB,KAAK,MAAM,IAAI,KAAK;AAC5C,UAAIA,UAAS,MAAM,EAAG,SAAQ;AAAA,IAChC,QAAQ;AACN,cAAQ;AAAA,IACV;AAAA,EACF;AACA,SAAO;AAAA,IACL,IAAI,OAAO,IAAI,EAAE;AAAA,IACjB,UAAU,IAAI,cAAc,QAAQ,IAAI,cAAc,SAAY,OAAO,OAAO,IAAI,SAAS;AAAA,IAC7F,WAAW,OAAO,IAAI,SAAS;AAAA,IAC/B,OAAO,IAAI,UAAU,QAAQ,IAAI,UAAU,SAAY,OAAO,OAAO,IAAI,KAAK;AAAA,IAC9E;AAAA,IACA,SAAS,OAAO,IAAI,QAAQ,CAAC;AAAA,IAC7B,aAAa,OAAO,IAAI,gBAAgB,CAAC;AAAA,IACzC,cAAc,OAAO,IAAI,iBAAiB,CAAC;AAAA,IAC3C,iBAAiB,OAAO,IAAI,oBAAoB,CAAC;AAAA,IACjD,iBAAiB,OAAO,IAAI,qBAAqB,CAAC;AAAA,IAClD,kBAAkB,OAAO,IAAI,sBAAsB,CAAC;AAAA,IACpD,aAAa,OAAO,IAAI,gBAAgB,CAAC;AAAA,IACzC,aAAa,OAAO,IAAI,gBAAgB,CAAC;AAAA,EAC3C;AACF;AAEA,IAAM,kBACJ;AAGK,SAAS,gCACd,IACA,WACsB;AACtB,QAAM,OAAO,GACV,QAAQ,UAAU,eAAe,yDAAyD,EAC1F,IAAI,SAAS;AAChB,SAAO,KAAK,IAAI,eAAe;AACjC;AAEO,SAAS,wBACd,IACA,WAC2B;AAC3B,QAAM,MAAM,GAAG,QAAQ,UAAU,eAAe,4BAA4B,EAAE,IAAI,SAAS;AAG3F,SAAO,QAAQ,SAAY,OAAO,gBAAgB,GAAG;AACvD;AAUA,SAAS,kBAAkB,QAAyB;AAClD,MAAI,OAAO,WAAW,SAAU,QAAO;AACvC,MAAI,WAAW,QAAQ,WAAW,OAAW,QAAO;AACpD,SAAO,KAAK,UAAU,MAAM;AAC9B;AAQO,SAAS,4BAA4B,IAAkB,WAAkC;AAC9F,QAAM,cAAc,GACjB,QAAQ,6EAA6E,EACrF,IAAI,SAAS;AAChB,QAAM,YAAY,GAAG,QAAQ,wDAAwD;AAErF,QAAM,WAA0B,CAAC;AACjC,aAAW,cAAc,aAAa;AACpC,QAAI;AACJ,QAAI;AACF,YAAM,SAAkB,KAAK,MAAM,WAAW,IAAI;AAClD,UAAI,CAACA,UAAS,MAAM,EAAG;AACvB,aAAO;AAAA,IACT,QAAQ;AACN;AAAA,IACF;AACA,UAAM,QAAwB,CAAC;AAC/B,eAAW,OAAO,UAAU,IAAI,WAAW,EAAE,GAA8B;AACzE,UAAI;AACF,cAAM,SAAkB,KAAK,MAAM,IAAI,IAAI;AAC3C,YAAIA,UAAS,MAAM,EAAG,OAAM,KAAK,MAAsB;AAAA,MACzD,QAAQ;AAAA,MAER;AAAA,IACF;AAEA,QAAI,KAAK,SAAS,QAAQ;AACxB,YAAM,OAAO,MACV,OAAO,CAAC,MAAM,EAAE,SAAS,UAAU,OAAO,EAAE,SAAS,QAAQ,EAC7D,IAAI,CAAC,MAAM,EAAE,IAAc,EAC3B,KAAK,IAAI;AACZ,eAAS,KAAK,EAAE,MAAM,QAAQ,SAAS,KAAK,CAAC;AAC7C;AAAA,IACF;AACA,QAAI,KAAK,SAAS,YAAa;AAI/B,UAAM,QAA0B,CAAC;AACjC,QAAI,UAA0B,CAAC;AAC/B,eAAW,QAAQ,OAAO;AACxB,UAAI,KAAK,SAAS,cAAc;AAC9B,YAAI,QAAQ,SAAS,EAAG,OAAM,KAAK,OAAO;AAC1C,kBAAU,CAAC;AACX;AAAA,MACF;AACA,UAAI,KAAK,SAAS,iBAAiB,KAAK,SAAS,cAAc,KAAK,SAAS,QAAS;AACtF,cAAQ,KAAK,IAAI;AAAA,IACnB;AACA,QAAI,QAAQ,SAAS,EAAG,OAAM,KAAK,OAAO;AAE1C,eAAW,QAAQ,OAAO;AACxB,YAAM,YAAY,KACf,OAAO,CAAC,MAAM,EAAE,SAAS,eAAe,OAAO,EAAE,SAAS,YAAY,EAAE,KAAK,SAAS,CAAC,EACvF,IAAI,CAAC,MAAM,EAAE,IAAc,EAC3B,KAAK,IAAI;AACZ,YAAM,OAAO,KACV,OAAO,CAAC,MAAM,EAAE,SAAS,UAAU,OAAO,EAAE,SAAS,QAAQ,EAC7D,IAAI,CAAC,MAAM,EAAE,IAAc,EAC3B,KAAK,IAAI;AACZ,YAAM,YAAY,KAAK,OAAO,CAAC,MAAM,EAAE,SAAS,UAAU,OAAO,EAAE,WAAW,QAAQ;AACtF,YAAM,YAA4B,UAAU,IAAI,CAAC,OAAO;AAAA,QACtD,IAAI,EAAE;AAAA,QACN,MAAM;AAAA,QACN,UAAU;AAAA,UACR,MAAM,EAAE,QAAQ;AAAA,UAChB,WAAW,KAAK,UAAU,EAAE,OAAO,SAAS,CAAC,CAAC;AAAA,QAChD;AAAA,MACF,EAAE;AACF,UAAI,UAAU,WAAW,KAAK,KAAK,WAAW,KAAK,UAAU,WAAW,EAAG;AAC3E,eAAS,KAAK;AAAA,QACZ,MAAM;AAAA,QACN,SAAS,KAAK,SAAS,IAAI,OAAO;AAAA,QAClC,GAAI,UAAU,SAAS,IAAI,EAAE,mBAAmB,UAAU,IAAI,CAAC;AAAA,QAC/D,GAAI,UAAU,SAAS,IAAI,EAAE,YAAY,UAAU,IAAI,CAAC;AAAA,MAC1D,CAAC;AACD,iBAAW,KAAK,WAAW;AACzB,iBAAS,KAAK;AAAA,UACZ,MAAM;AAAA,UACN,cAAc,EAAE;AAAA,UAChB,MAAM,EAAE,QAAQ;AAAA,UAChB,SAAS,kBAAkB,EAAE,OAAO,MAAM;AAAA,QAC5C,CAAC;AAAA,MACH;AAAA,IACF;AAAA,EACF;AACA,SAAO;AACT;","names":["content","homedir","join","isRecord"]}
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  fsCampaignStorage,
3
3
  runCampaign
4
- } from "./chunk-EZJEIH2R.js";
4
+ } from "./chunk-C6LXANRU.js";
5
5
  import {
6
6
  __export
7
7
  } from "./chunk-PZ5AY32C.js";
@@ -390,7 +390,7 @@ function summarizeBenchmarkCampaign(input) {
390
390
  totalCells: input.campaign.cells.length,
391
391
  cellsFailed: input.campaign.aggregates.cellsFailed,
392
392
  cellsCached: input.campaign.aggregates.cellsCached,
393
- totalCostUsd: input.campaign.aggregates.totalCostUsd,
393
+ totalCostUsd: input.campaign.aggregates.cost.totalCostUsd,
394
394
  splits: summarizeSlices(successful, (row) => row.scenario?.splitTag ?? "unknown", [
395
395
  "search",
396
396
  "dev",
@@ -763,4 +763,4 @@ export {
763
763
  retrievalMetricsAtCutoff,
764
764
  benchmarks_exports
765
765
  };
766
- //# sourceMappingURL=chunk-XPRT64IE.js.map
766
+ //# sourceMappingURL=chunk-IYCLP2N2.js.map