@tangle-network/agent-eval 0.93.0 → 0.94.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. package/dist/adapters/otel.d.ts +4 -4
  2. package/dist/analyst/index.d.ts +10 -10
  3. package/dist/analyst/index.js +6 -6
  4. package/dist/{analyze-runs-DwCEkpO_.d.ts → analyze-runs-B6Ljo_dI.d.ts} +2 -2
  5. package/dist/{baseline-DE36-Np7.d.ts → baseline-Bbid3WoO.d.ts} +1 -1
  6. package/dist/belief-state/index.d.ts +2 -2
  7. package/dist/belief-state/index.js +1 -1
  8. package/dist/builder-eval/index.d.ts +3 -3
  9. package/dist/builder-eval/index.js +2 -2
  10. package/dist/{calibration-Cpr3WaX3.d.ts → calibration-BPmzuVPk.d.ts} +1 -1
  11. package/dist/campaign/index.d.ts +13 -13
  12. package/dist/campaign/index.js +8 -8
  13. package/dist/{chunk-TWS7AZEY.js → chunk-2K6UUZ7P.js} +2 -2
  14. package/dist/{chunk-Y47J2LJ3.js → chunk-2KNZHH3P.js} +5 -5
  15. package/dist/{chunk-LMZQ2Z4U.js → chunk-56GC6VYO.js} +126 -1
  16. package/dist/chunk-56GC6VYO.js.map +1 -0
  17. package/dist/{chunk-QG2OVF2D.js → chunk-CTBHKLEU.js} +99 -30
  18. package/dist/chunk-CTBHKLEU.js.map +1 -0
  19. package/dist/{chunk-CVVHBFGN.js → chunk-CWNP4DV4.js} +53 -5
  20. package/dist/chunk-CWNP4DV4.js.map +1 -0
  21. package/dist/{chunk-QS3RBQPI.js → chunk-D5KBVULY.js} +2 -2
  22. package/dist/{chunk-ZFIBGEOL.js → chunk-EGPMSBEZ.js} +4 -4
  23. package/dist/{chunk-GSH6QNNS.js → chunk-KW53MSA5.js} +101 -78
  24. package/dist/chunk-KW53MSA5.js.map +1 -0
  25. package/dist/{chunk-UMMZHCPB.js → chunk-MIFZUPEK.js} +7 -5
  26. package/dist/chunk-MIFZUPEK.js.map +1 -0
  27. package/dist/{chunk-XY4DDNEG.js → chunk-MPQWFX6Y.js} +2 -2
  28. package/dist/{chunk-L3JOU6XM.js → chunk-NACM5RS7.js} +3 -3
  29. package/dist/{chunk-D3V5B42D.js → chunk-Q5LIB7BC.js} +7 -4
  30. package/dist/{chunk-D3V5B42D.js.map → chunk-Q5LIB7BC.js.map} +1 -1
  31. package/dist/{chunk-QAY5UIJO.js → chunk-QMUEXQJS.js} +99 -57
  32. package/dist/chunk-QMUEXQJS.js.map +1 -0
  33. package/dist/{chunk-4FBZZIYD.js → chunk-QTMYW64F.js} +2 -2
  34. package/dist/{chunk-6SOJM3VR.js → chunk-SD2YFWQQ.js} +4 -4
  35. package/dist/{chunk-UHMJT4T7.js → chunk-TBDR6PAI.js} +2 -2
  36. package/dist/{chunk-47X6LRCE.js → chunk-UFSG7ACU.js} +10 -7
  37. package/dist/chunk-UFSG7ACU.js.map +1 -0
  38. package/dist/{chunk-CY6U5S3X.js → chunk-URWVKRCS.js} +2 -2
  39. package/dist/{chunk-T375SUOZ.js → chunk-ZOPLCJSY.js} +86 -31
  40. package/dist/chunk-ZOPLCJSY.js.map +1 -0
  41. package/dist/cli.js +2 -2
  42. package/dist/contract/index.d.ts +17 -17
  43. package/dist/contract/index.js +11 -11
  44. package/dist/{control-_Qb7skHX.d.ts → control-D6qwHXIR.d.ts} +4 -4
  45. package/dist/{control-runtime-DuFBYg7A.d.ts → control-runtime-Acf9CGhw.d.ts} +2 -2
  46. package/dist/control.d.ts +5 -5
  47. package/dist/{corpus-BoR-041R.d.ts → corpus-B8A4BDR3.d.ts} +1 -1
  48. package/dist/{counterfactual-Dwibr5IW.d.ts → counterfactual-DlOz8PBx.d.ts} +3 -3
  49. package/dist/{default-registry-zoGHUQEH.d.ts → default-registry-6dhErQbs.d.ts} +2 -2
  50. package/dist/diagnose.d.ts +9 -9
  51. package/dist/diagnose.js +1 -1
  52. package/dist/{emitter-DEZwY14K.d.ts → emitter-C2rqGH_l.d.ts} +1 -1
  53. package/dist/{failure-cluster-CL7IVgkJ.d.ts → failure-cluster-DH9Flgcf.d.ts} +1 -1
  54. package/dist/{feedback-trajectory-D9OVLrg9.d.ts → feedback-trajectory-BxY0cKfs.d.ts} +1 -1
  55. package/dist/governance/index.d.ts +2 -2
  56. package/dist/{harness-optimizer-EnEnQPsr.d.ts → harness-optimizer-mOl9XX_O.d.ts} +1 -1
  57. package/dist/hosted/index.d.ts +4 -4
  58. package/dist/index.d.ts +68 -46
  59. package/dist/index.js +261 -87
  60. package/dist/index.js.map +1 -1
  61. package/dist/{insight-report-BBwvOh6x.d.ts → insight-report-DWl3z9tl.d.ts} +1 -1
  62. package/dist/{integrity-VJ9A7aST.d.ts → integrity-D2t12mMw.d.ts} +1 -1
  63. package/dist/{kind-factory-5b7xXXOr.d.ts → kind-factory-0BhLSI27.d.ts} +1 -1
  64. package/dist/knowledge/index.d.ts +3 -3
  65. package/dist/{llm-client-BeEcAokY.d.ts → llm-client-Bj7g0rqu.d.ts} +31 -0
  66. package/dist/meta-eval/index.d.ts +4 -4
  67. package/dist/meta-eval/index.js +1 -1
  68. package/dist/openapi.json +1 -1
  69. package/dist/pipelines/index.d.ts +6 -6
  70. package/dist/pipelines/index.js +40 -15
  71. package/dist/pipelines/index.js.map +1 -1
  72. package/dist/prm/index.d.ts +11 -7
  73. package/dist/prm/index.js +55 -12
  74. package/dist/prm/index.js.map +1 -1
  75. package/dist/{provenance-LnqRT0sS.d.ts → provenance-P-bCL2Fo.d.ts} +3 -3
  76. package/dist/{query-CqTxMwDw.d.ts → query-B7GGjRox.d.ts} +2 -1
  77. package/dist/{red-team-BXHil6c8.d.ts → red-team-BWdoyleI.d.ts} +1 -1
  78. package/dist/{release-report-euXIV_Sk.d.ts → release-report-BEbWmVYj.d.ts} +1 -1
  79. package/dist/reporting.d.ts +6 -6
  80. package/dist/reporting.js +3 -3
  81. package/dist/{researcher-DE6Gpnb4.d.ts → researcher-B0C2_fVO.d.ts} +5 -5
  82. package/dist/rl.d.ts +10 -10
  83. package/dist/rl.js +4 -4
  84. package/dist/{rubric-BOfxn4ja.d.ts → rubric-Cc6UHvUb.d.ts} +2 -2
  85. package/dist/{run-campaign-RDGAM5KJ.js → run-campaign-7WNXMDSN.js} +3 -3
  86. package/dist/{run-critic-BAIjX99r.d.ts → run-critic-CmMf05uV.d.ts} +1 -1
  87. package/dist/{run-improvement-loop-5z_l5zDz.d.ts → run-improvement-loop-DBahB8Ax.d.ts} +1 -1
  88. package/dist/{semantic-concept-judge-Dn8Z6KEG.d.ts → semantic-concept-judge-B9MgmBnM.d.ts} +3 -3
  89. package/dist/{statistics-C7PozGrZ.d.ts → statistics-CCJpTGOS.d.ts} +117 -2
  90. package/dist/{store-CKUAgsJz.d.ts → store-BcFXE6LG.d.ts} +16 -1
  91. package/dist/{summary-report-DGmUucwQ.d.ts → summary-report-BDOFevaT.d.ts} +1 -1
  92. package/dist/{test-graded-scenario-BdVaPyHT.d.ts → test-graded-scenario-DeODGLra.d.ts} +21 -2
  93. package/dist/traces.d.ts +19 -6
  94. package/dist/traces.js +4 -4
  95. package/dist/{trajectory-GEdXJCL5.d.ts → trajectory-2TkpSEVh.d.ts} +1 -1
  96. package/dist/{types-Croy5h7V.d.ts → types-C7DGg5ex.d.ts} +2 -0
  97. package/dist/{types-2VVIL04s.d.ts → types-Ce17tDlG.d.ts} +2 -2
  98. package/dist/wire/index.d.ts +4 -4
  99. package/dist/wire/index.js +2 -2
  100. package/dist/workflow/index.d.ts +13 -13
  101. package/dist/workflow/index.js +1 -1
  102. package/package.json +1 -1
  103. package/dist/chunk-47X6LRCE.js.map +0 -1
  104. package/dist/chunk-CVVHBFGN.js.map +0 -1
  105. package/dist/chunk-GSH6QNNS.js.map +0 -1
  106. package/dist/chunk-LMZQ2Z4U.js.map +0 -1
  107. package/dist/chunk-QAY5UIJO.js.map +0 -1
  108. package/dist/chunk-QG2OVF2D.js.map +0 -1
  109. package/dist/chunk-T375SUOZ.js.map +0 -1
  110. package/dist/chunk-UMMZHCPB.js.map +0 -1
  111. /package/dist/{chunk-TWS7AZEY.js.map → chunk-2K6UUZ7P.js.map} +0 -0
  112. /package/dist/{chunk-Y47J2LJ3.js.map → chunk-2KNZHH3P.js.map} +0 -0
  113. /package/dist/{chunk-QS3RBQPI.js.map → chunk-D5KBVULY.js.map} +0 -0
  114. /package/dist/{chunk-ZFIBGEOL.js.map → chunk-EGPMSBEZ.js.map} +0 -0
  115. /package/dist/{chunk-XY4DDNEG.js.map → chunk-MPQWFX6Y.js.map} +0 -0
  116. /package/dist/{chunk-L3JOU6XM.js.map → chunk-NACM5RS7.js.map} +0 -0
  117. /package/dist/{chunk-4FBZZIYD.js.map → chunk-QTMYW64F.js.map} +0 -0
  118. /package/dist/{chunk-6SOJM3VR.js.map → chunk-SD2YFWQQ.js.map} +0 -0
  119. /package/dist/{chunk-UHMJT4T7.js.map → chunk-TBDR6PAI.js.map} +0 -0
  120. /package/dist/{chunk-CY6U5S3X.js.map → chunk-URWVKRCS.js.map} +0 -0
  121. /package/dist/{run-campaign-RDGAM5KJ.js.map → run-campaign-7WNXMDSN.js.map} +0 -0
@@ -1 +1 @@
1
- {"version":3,"sources":["../../src/prm/builtin-rubrics.ts","../../src/prm/inference.ts","../../src/prm/rubric.ts"],"sourcesContent":["/**\n * Built-in reference rubrics. Consumers combine these with domain\n * rubrics. All are deterministic, rule-based — cheap to run + easy\n * to unit-test. LLM-based rubrics are trivially authored by\n * following the StepRubric contract.\n */\n\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { StepRubric } from './rubric'\n\n/** Penalize very short or very long assistant outputs. */\nexport function outputLengthRubric(\n args: { minChars?: number; maxChars?: number; weight?: number } = {},\n): StepRubric {\n const min = args.minChars ?? 20\n const max = args.maxChars ?? 8000\n return {\n id: 'output-length',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const len = (llm.output ?? '').length\n if (len === 0) return { score: 0, rationale: 'empty output' }\n if (len < min)\n return { score: Math.max(0, len / min), rationale: `below min (${len} < ${min})` }\n if (len > max)\n return {\n score: Math.max(0, 1 - (len - max) / max),\n rationale: `above max (${len} > ${max})`,\n }\n return { score: 1, rationale: `${len} chars in bounds` }\n },\n }\n}\n\n/** Reward tool calls that succeeded (status='ok') with an informative result. */\nexport function toolSuccessRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-success',\n kinds: ['tool'],\n weight: args.weight ?? 1,\n async grade({ step }) {\n const tool = step.span as ToolSpan\n if (tool.status === 'error')\n return { score: 0, rationale: `error: ${tool.error ?? 'unknown'}` }\n const r = tool.result\n if (r === null || r === undefined) return { score: 0.3, rationale: 'empty result' }\n const asText = typeof r === 'string' ? r : JSON.stringify(r)\n if (asText.length < 4) return { score: 0.5, rationale: 'tiny result' }\n return { score: 1, rationale: `${tool.toolName} ok` }\n },\n }\n}\n\n/** Penalize tool calls that duplicate a prior call with identical args. */\nexport function toolNonRedundantRubric(args: { weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 0.5\n return {\n id: 'tool-non-redundant',\n kinds: ['tool'],\n weight,\n async grade({ step, prior }) {\n const tool = step.span as ToolSpan\n const priorMatches = prior.filter((p) => {\n if (p.span.kind !== 'tool') return false\n const pt = p.span as ToolSpan\n return (\n pt.toolName === tool.toolName && stableStringify(pt.args) === stableStringify(tool.args)\n )\n })\n if (priorMatches.length === 0) return { score: 1, rationale: 'novel call' }\n return {\n score: Math.max(0, 1 - priorMatches.length * 0.5),\n rationale: `${priorMatches.length} duplicate(s)`,\n }\n },\n }\n}\n\n/** Penalize LLM outputs that contain common refusal markers when a refusal\n * is NOT expected (caller inverts weight for scenarios where refusal IS expected). */\nexport function nonRefusalRubric(args: { markers?: RegExp[]; weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 1\n const markers = args.markers ?? [\n /\\bi\\s+(?:can(?:not|'t)|won't|will\\s+not)\\b/i,\n /\\b(?:as\\s+an?\\s+)?ai\\b.*?\\b(?:can't|cannot)\\b/i,\n ]\n return {\n id: 'non-refusal',\n kinds: ['llm'],\n weight,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const out = llm.output ?? ''\n const refused = markers.some((re) => re.test(out))\n return refused\n ? { score: 0, rationale: 'refusal marker present' }\n : { score: 1, rationale: 'no refusal' }\n },\n }\n}\n\n/** Reward outputs that invoke the next-step tool the trajectory actually uses\n * (i.e. the LLM span announced \"I will call X\" and the following tool span IS X). */\nexport function toolIntentAlignmentRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-intent-alignment',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step, next }) {\n const llm = step.span as LlmSpan\n const nextTool = next.find((s) => s.span.kind === 'tool')\n if (!nextTool) return null\n const toolName = (nextTool.span as ToolSpan).toolName\n const out = (llm.output ?? '').toLowerCase()\n const mentioned = out.includes(toolName.toLowerCase())\n return mentioned\n ? { score: 1, rationale: `mentioned \"${toolName}\" before calling it` }\n : { score: 0.5, rationale: `called \"${toolName}\" without announcing it` }\n },\n }\n}\n\nfunction stableStringify(value: unknown): string {\n if (value === null || typeof value !== 'object') return JSON.stringify(value)\n if (Array.isArray(value)) return `[${value.map(stableStringify).join(',')}]`\n const keys = Object.keys(value as Record<string, unknown>).sort()\n return `{${keys.map((k) => `${JSON.stringify(k)}:${stableStringify((value as Record<string, unknown>)[k])}`).join(',')}}`\n}\n","/**\n * Inference-time PRM scoring — pick the best of N candidate trajectories\n * using a trained reward model (or a rule-based PRM as a proxy).\n *\n * The canonical Best-of-N pattern: generate N completions, score each\n * with a PRM, pick the winner. Here the scoring loop is framework-agnostic\n * — supply a TraceStore + PrmGrader + N run IDs → get ranking + winner.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport type { PrmGradedTrace, PrmGrader } from './rubric'\n\nexport interface BestOfNResult {\n winner: PrmGradedTrace\n ranked: PrmGradedTrace[]\n /** Standard deviation of aggregate scores — small = candidates were homogenous. */\n stdDev: number\n}\n\nexport async function prmBestOfN(\n store: TraceStore,\n grader: PrmGrader,\n runIds: string[],\n): Promise<BestOfNResult> {\n if (runIds.length === 0) throw new Error('prmBestOfN: at least 1 candidate required')\n const graded = await Promise.all(runIds.map((id) => grader.grade(store, id)))\n const ranked = [...graded].sort((a, b) => b.aggregateScore - a.aggregateScore)\n const mean = graded.reduce((a, g) => a + g.aggregateScore, 0) / graded.length\n const variance = graded.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / graded.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n\n/**\n * Weighted vote across multiple graders — use when you want a PRM ensemble\n * (e.g. rule-based + LLM-based + trained model). Each grader produces its\n * own ranking; we aggregate via rank-sum (Borda count) so no single grader\n * dominates via a different score scale.\n */\nexport async function prmEnsembleBestOfN(\n store: TraceStore,\n graders: PrmGrader[],\n runIds: string[],\n): Promise<BestOfNResult> {\n if (graders.length === 0) throw new Error('prmEnsembleBestOfN: at least 1 grader')\n const perGrader = await Promise.all(\n graders.map(async (g) => {\n const graded = await Promise.all(runIds.map((id) => g.grade(store, id)))\n return graded.sort((a, b) => b.aggregateScore - a.aggregateScore)\n }),\n )\n // Borda: rank-sum across graders.\n const bordaScores = new Map<string, number>()\n for (const ranking of perGrader) {\n ranking.forEach((g, rank) => {\n bordaScores.set(g.runId, (bordaScores.get(g.runId) ?? 0) + (ranking.length - rank))\n })\n }\n // Return a synthesized ranking using the first grader's graded traces\n // ordered by Borda score. aggregateScore field kept for UX.\n const canonical = perGrader[0]!\n const byRun = new Map(canonical.map((g) => [g.runId, g]))\n const ranked = [...byRun.values()].sort(\n (a, b) => (bordaScores.get(b.runId) ?? 0) - (bordaScores.get(a.runId) ?? 0),\n )\n const mean = ranked.reduce((a, g) => a + g.aggregateScore, 0) / ranked.length\n const variance = ranked.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / ranked.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n","/**\n * Process Reward Modeling — per-step rubric grading.\n *\n * A StepRubric inspects one span and returns a score + rationale.\n * PrmGrader applies an array of rubrics to every LLM span in a\n * trajectory (consumers can broaden to tool/retrieval spans via the\n * `kind` filter on each rubric).\n *\n * Why this matters: outcome-only eval (did the final artifact work?)\n * gives sparse reward — most agent turns are unattributable. PRMs\n * densify the signal so optimizers and RL fine-tuning can assign\n * credit per turn.\n */\n\nimport { TraceEmitter } from '../trace/emitter'\nimport type { JudgeSpan, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface StepContext {\n trajectory: Trajectory\n step: TrajectoryStep\n /** Steps preceding `step` in trajectory order. */\n prior: TrajectoryStep[]\n /** Steps following `step`. */\n next: TrajectoryStep[]\n}\n\nexport interface StepRubric {\n id: string\n /** Only grade spans of these kinds (default: all). */\n kinds?: Array<Span['kind']>\n /** Weight in the aggregate score. Default 1. */\n weight?: number\n /** Returns score in 0..1 + optional rationale/evidence. Return `null` to\n * skip grading (rubric doesn't apply to this step). */\n grade: (\n ctx: StepContext,\n ) => Promise<{ score: number; rationale?: string; evidence?: string } | null>\n}\n\nexport interface GradedStep {\n spanId: string\n rubricId: string\n score: number\n weight: number\n rationale?: string\n evidence?: string\n}\n\nexport interface PrmGradedTrace {\n runId: string\n steps: GradedStep[]\n /** Weighted mean of all graded steps; 0..1. */\n aggregateScore: number\n /** Number of spans graded — useful for sanity-checking coverage. */\n gradedCount: number\n /** Number of spans in the trajectory that no rubric matched. */\n ungradedCount: number\n}\n\nexport class PrmGrader {\n constructor(private rubrics: StepRubric[]) {\n if (rubrics.length === 0) throw new Error('PrmGrader: at least 1 rubric required')\n }\n\n /**\n * Grade every eligible span in a run. Emits a JudgeVerdict span for each\n * (rubric × span) verdict so the result is visible to downstream pipelines\n * (judgeAgreementView, etc.) — PRM is just \"a judge that runs per span.\"\n */\n async grade(store: TraceStore, runId: string): Promise<PrmGradedTrace> {\n const trajectory = await buildTrajectory(store, runId)\n const emitter = new TraceEmitter(store, { runId })\n const steps: GradedStep[] = []\n let ungraded = 0\n for (let i = 0; i < trajectory.steps.length; i++) {\n const step = trajectory.steps[i]!\n const ctx: StepContext = {\n trajectory,\n step,\n prior: trajectory.steps.slice(0, i),\n next: trajectory.steps.slice(i + 1),\n }\n let gradedThis = false\n for (const rubric of this.rubrics) {\n if (rubric.kinds && !rubric.kinds.includes(step.span.kind)) continue\n const verdict = await rubric.grade(ctx)\n if (verdict === null) continue\n const weight = rubric.weight ?? 1\n steps.push({\n spanId: step.span.spanId,\n rubricId: rubric.id,\n score: verdict.score,\n weight,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n })\n gradedThis = true\n // Persist the verdict as a JudgeSpan so the query pipelines see it\n await emitter.recordJudge({\n judgeId: `prm:${rubric.id}`,\n targetSpanId: step.span.spanId,\n dimension: 'step_quality',\n score: verdict.score,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n name: `prm:${rubric.id}`,\n })\n }\n if (!gradedThis) ungraded++\n }\n\n const totalWeight = steps.reduce((a, s) => a + s.weight, 0)\n const aggregateScore =\n totalWeight === 0 ? 0 : steps.reduce((a, s) => a + s.score * s.weight, 0) / totalWeight\n\n return { runId, steps, aggregateScore, gradedCount: steps.length, ungradedCount: ungraded }\n }\n}\n\n/** Helper: reads JudgeVerdict spans that PRM emitted so downstream pipelines\n * can distinguish PRM verdicts from human or top-level LLM judges. */\nexport function isPrmVerdict(verdict: JudgeSpan): boolean {\n return verdict.judgeId.startsWith('prm:')\n}\n"],"mappings":";;;;;;;;;;;;;;AAWO,SAAS,mBACd,OAAkE,CAAC,GACvD;AACZ,QAAM,MAAM,KAAK,YAAY;AAC7B,QAAM,MAAM,KAAK,YAAY;AAC7B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,OAAO,IAAI,UAAU,IAAI;AAC/B,UAAI,QAAQ,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,eAAe;AAC5D,UAAI,MAAM;AACR,eAAO,EAAE,OAAO,KAAK,IAAI,GAAG,MAAM,GAAG,GAAG,WAAW,cAAc,GAAG,MAAM,GAAG,IAAI;AACnF,UAAI,MAAM;AACR,eAAO;AAAA,UACL,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,GAAG;AAAA,UACxC,WAAW,cAAc,GAAG,MAAM,GAAG;AAAA,QACvC;AACF,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,GAAG,mBAAmB;AAAA,IACzD;AAAA,EACF;AACF;AAGO,SAAS,kBAAkB,OAA4B,CAAC,GAAe;AAC5E,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,OAAO,KAAK;AAClB,UAAI,KAAK,WAAW;AAClB,eAAO,EAAE,OAAO,GAAG,WAAW,UAAU,KAAK,SAAS,SAAS,GAAG;AACpE,YAAM,IAAI,KAAK;AACf,UAAI,MAAM,QAAQ,MAAM,OAAW,QAAO,EAAE,OAAO,KAAK,WAAW,eAAe;AAClF,YAAM,SAAS,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC;AAC3D,UAAI,OAAO,SAAS,EAAG,QAAO,EAAE,OAAO,KAAK,WAAW,cAAc;AACrE,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,KAAK,QAAQ,MAAM;AAAA,IACtD;AAAA,EACF;AACF;AAGO,SAAS,uBAAuB,OAA4B,CAAC,GAAe;AACjF,QAAM,SAAS,KAAK,UAAU;AAC9B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd;AAAA,IACA,MAAM,MAAM,EAAE,MAAM,MAAM,GAAG;AAC3B,YAAM,OAAO,KAAK;AAClB,YAAM,eAAe,MAAM,OAAO,CAAC,MAAM;AACvC,YAAI,EAAE,KAAK,SAAS,OAAQ,QAAO;AACnC,cAAM,KAAK,EAAE;AACb,eACE,GAAG,aAAa,KAAK,YAAY,gBAAgB,GAAG,IAAI,MAAM,gBAAgB,KAAK,IAAI;AAAA,MAE3F,CAAC;AACD,UAAI,aAAa,WAAW,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,aAAa;AAC1E,aAAO;AAAA,QACL,OAAO,KAAK,IAAI,GAAG,IAAI,aAAa,SAAS,GAAG;AAAA,QAChD,WAAW,GAAG,aAAa,MAAM;AAAA,MACnC;AAAA,IACF;AAAA,EACF;AACF;AAIO,SAAS,iBAAiB,OAAgD,CAAC,GAAe;AAC/F,QAAM,SAAS,KAAK,UAAU;AAC9B,QAAM,UAAU,KAAK,WAAW;AAAA,IAC9B;AAAA,IACA;AAAA,EACF;AACA,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb;AAAA,IACA,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,MAAM,IAAI,UAAU;AAC1B,YAAM,UAAU,QAAQ,KAAK,CAAC,OAAO,GAAG,KAAK,GAAG,CAAC;AACjD,aAAO,UACH,EAAE,OAAO,GAAG,WAAW,yBAAyB,IAChD,EAAE,OAAO,GAAG,WAAW,aAAa;AAAA,IAC1C;AAAA,EACF;AACF;AAIO,SAAS,0BAA0B,OAA4B,CAAC,GAAe;AACpF,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,MAAM,KAAK,GAAG;AAC1B,YAAM,MAAM,KAAK;AACjB,YAAM,WAAW,KAAK,KAAK,CAAC,MAAM,EAAE,KAAK,SAAS,MAAM;AACxD,UAAI,CAAC,SAAU,QAAO;AACtB,YAAM,WAAY,SAAS,KAAkB;AAC7C,YAAM,OAAO,IAAI,UAAU,IAAI,YAAY;AAC3C,YAAM,YAAY,IAAI,SAAS,SAAS,YAAY,CAAC;AACrD,aAAO,YACH,EAAE,OAAO,GAAG,WAAW,cAAc,QAAQ,sBAAsB,IACnE,EAAE,OAAO,KAAK,WAAW,WAAW,QAAQ,0BAA0B;AAAA,IAC5E;AAAA,EACF;AACF;AAEA,SAAS,gBAAgB,OAAwB;AAC/C,MAAI,UAAU,QAAQ,OAAO,UAAU,SAAU,QAAO,KAAK,UAAU,KAAK;AAC5E,MAAI,MAAM,QAAQ,KAAK,EAAG,QAAO,IAAI,MAAM,IAAI,eAAe,EAAE,KAAK,GAAG,CAAC;AACzE,QAAM,OAAO,OAAO,KAAK,KAAgC,EAAE,KAAK;AAChE,SAAO,IAAI,KAAK,IAAI,CAAC,MAAM,GAAG,KAAK,UAAU,CAAC,CAAC,IAAI,gBAAiB,MAAkC,CAAC,CAAC,CAAC,EAAE,EAAE,KAAK,GAAG,CAAC;AACxH;;;AC9GA,eAAsB,WACpB,OACA,QACA,QACwB;AACxB,MAAI,OAAO,WAAW,EAAG,OAAM,IAAI,MAAM,2CAA2C;AACpF,QAAM,SAAS,MAAM,QAAQ,IAAI,OAAO,IAAI,CAAC,OAAO,OAAO,MAAM,OAAO,EAAE,CAAC,CAAC;AAC5E,QAAM,SAAS,CAAC,GAAG,MAAM,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAC7E,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;AAQA,eAAsB,mBACpB,OACA,SACA,QACwB;AACxB,MAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AACjF,QAAM,YAAY,MAAM,QAAQ;AAAA,IAC9B,QAAQ,IAAI,OAAO,MAAM;AACvB,YAAM,SAAS,MAAM,QAAQ,IAAI,OAAO,IAAI,CAAC,OAAO,EAAE,MAAM,OAAO,EAAE,CAAC,CAAC;AACvE,aAAO,OAAO,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAAA,IAClE,CAAC;AAAA,EACH;AAEA,QAAM,cAAc,oBAAI,IAAoB;AAC5C,aAAW,WAAW,WAAW;AAC/B,YAAQ,QAAQ,CAAC,GAAG,SAAS;AAC3B,kBAAY,IAAI,EAAE,QAAQ,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,QAAQ,SAAS,KAAK;AAAA,IACpF,CAAC;AAAA,EACH;AAGA,QAAM,YAAY,UAAU,CAAC;AAC7B,QAAM,QAAQ,IAAI,IAAI,UAAU,IAAI,CAAC,MAAM,CAAC,EAAE,OAAO,CAAC,CAAC,CAAC;AACxD,QAAM,SAAS,CAAC,GAAG,MAAM,OAAO,CAAC,EAAE;AAAA,IACjC,CAAC,GAAG,OAAO,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,YAAY,IAAI,EAAE,KAAK,KAAK;AAAA,EAC3E;AACA,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;;;ACNO,IAAM,YAAN,MAAgB;AAAA,EACrB,YAAoB,SAAuB;AAAvB;AAClB,QAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AAAA,EACnF;AAAA,EAFoB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASpB,MAAM,MAAM,OAAmB,OAAwC;AACrE,UAAM,aAAa,MAAM,gBAAgB,OAAO,KAAK;AACrD,UAAM,UAAU,IAAI,aAAa,OAAO,EAAE,MAAM,CAAC;AACjD,UAAM,QAAsB,CAAC;AAC7B,QAAI,WAAW;AACf,aAAS,IAAI,GAAG,IAAI,WAAW,MAAM,QAAQ,KAAK;AAChD,YAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,YAAM,MAAmB;AAAA,QACvB;AAAA,QACA;AAAA,QACA,OAAO,WAAW,MAAM,MAAM,GAAG,CAAC;AAAA,QAClC,MAAM,WAAW,MAAM,MAAM,IAAI,CAAC;AAAA,MACpC;AACA,UAAI,aAAa;AACjB,iBAAW,UAAU,KAAK,SAAS;AACjC,YAAI,OAAO,SAAS,CAAC,OAAO,MAAM,SAAS,KAAK,KAAK,IAAI,EAAG;AAC5D,cAAM,UAAU,MAAM,OAAO,MAAM,GAAG;AACtC,YAAI,YAAY,KAAM;AACtB,cAAM,SAAS,OAAO,UAAU;AAChC,cAAM,KAAK;AAAA,UACT,QAAQ,KAAK,KAAK;AAAA,UAClB,UAAU,OAAO;AAAA,UACjB,OAAO,QAAQ;AAAA,UACf;AAAA,UACA,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,QACpB,CAAC;AACD,qBAAa;AAEb,cAAM,QAAQ,YAAY;AAAA,UACxB,SAAS,OAAO,OAAO,EAAE;AAAA,UACzB,cAAc,KAAK,KAAK;AAAA,UACxB,WAAW;AAAA,UACX,OAAO,QAAQ;AAAA,UACf,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,UAClB,MAAM,OAAO,OAAO,EAAE;AAAA,QACxB,CAAC;AAAA,MACH;AACA,UAAI,CAAC,WAAY;AAAA,IACnB;AAEA,UAAM,cAAc,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;AAC1D,UAAM,iBACJ,gBAAgB,IAAI,IAAI,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,EAAE,QAAQ,CAAC,IAAI;AAE9E,WAAO,EAAE,OAAO,OAAO,gBAAgB,aAAa,MAAM,QAAQ,eAAe,SAAS;AAAA,EAC5F;AACF;AAIO,SAAS,aAAa,SAA6B;AACxD,SAAO,QAAQ,QAAQ,WAAW,MAAM;AAC1C;","names":[]}
1
+ {"version":3,"sources":["../../src/prm/builtin-rubrics.ts","../../src/prm/inference.ts","../../src/prm/rubric.ts"],"sourcesContent":["/**\n * Built-in reference rubrics. Consumers combine these with domain\n * rubrics. All are deterministic, rule-based — cheap to run + easy\n * to unit-test. LLM-based rubrics are trivially authored by\n * following the StepRubric contract.\n */\n\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { StepRubric } from './rubric'\n\n/** Penalize very short or very long assistant outputs. */\nexport function outputLengthRubric(\n args: { minChars?: number; maxChars?: number; weight?: number } = {},\n): StepRubric {\n const min = args.minChars ?? 20\n const max = args.maxChars ?? 8000\n return {\n id: 'output-length',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const len = (llm.output ?? '').length\n if (len === 0) return { score: 0, rationale: 'empty output' }\n if (len < min)\n return { score: Math.max(0, len / min), rationale: `below min (${len} < ${min})` }\n if (len > max)\n return {\n score: Math.max(0, 1 - (len - max) / max),\n rationale: `above max (${len} > ${max})`,\n }\n return { score: 1, rationale: `${len} chars in bounds` }\n },\n }\n}\n\n/** Reward tool calls that succeeded (status='ok') with an informative result. */\nexport function toolSuccessRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-success',\n kinds: ['tool'],\n weight: args.weight ?? 1,\n async grade({ step }) {\n const tool = step.span as ToolSpan\n if (tool.status === 'error')\n return { score: 0, rationale: `error: ${tool.error ?? 'unknown'}` }\n const r = tool.result\n if (r === null || r === undefined) return { score: 0.3, rationale: 'empty result' }\n const asText = typeof r === 'string' ? r : JSON.stringify(r)\n if (asText.length < 4) return { score: 0.5, rationale: 'tiny result' }\n return { score: 1, rationale: `${tool.toolName} ok` }\n },\n }\n}\n\n/** Penalize tool calls that duplicate a prior call with identical args. */\nexport function toolNonRedundantRubric(args: { weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 0.5\n return {\n id: 'tool-non-redundant',\n kinds: ['tool'],\n weight,\n async grade({ step, prior }) {\n const tool = step.span as ToolSpan\n const priorMatches = prior.filter((p) => {\n if (p.span.kind !== 'tool') return false\n const pt = p.span as ToolSpan\n return (\n pt.toolName === tool.toolName && stableStringify(pt.args) === stableStringify(tool.args)\n )\n })\n if (priorMatches.length === 0) return { score: 1, rationale: 'novel call' }\n return {\n score: Math.max(0, 1 - priorMatches.length * 0.5),\n rationale: `${priorMatches.length} duplicate(s)`,\n }\n },\n }\n}\n\n/** Penalize LLM outputs that contain common refusal markers when a refusal\n * is NOT expected (caller inverts weight for scenarios where refusal IS expected). */\nexport function nonRefusalRubric(args: { markers?: RegExp[]; weight?: number } = {}): StepRubric {\n const weight = args.weight ?? 1\n const markers = args.markers ?? [\n /\\bi\\s+(?:can(?:not|'t)|won't|will\\s+not)\\b/i,\n /\\b(?:as\\s+an?\\s+)?ai\\b.*?\\b(?:can't|cannot)\\b/i,\n ]\n return {\n id: 'non-refusal',\n kinds: ['llm'],\n weight,\n async grade({ step }) {\n const llm = step.span as LlmSpan\n const out = llm.output ?? ''\n const refused = markers.some((re) => re.test(out))\n return refused\n ? { score: 0, rationale: 'refusal marker present' }\n : { score: 1, rationale: 'no refusal' }\n },\n }\n}\n\n/** Reward outputs that invoke the next-step tool the trajectory actually uses\n * (i.e. the LLM span announced \"I will call X\" and the following tool span IS X). */\nexport function toolIntentAlignmentRubric(args: { weight?: number } = {}): StepRubric {\n return {\n id: 'tool-intent-alignment',\n kinds: ['llm'],\n weight: args.weight ?? 0.5,\n async grade({ step, next }) {\n const llm = step.span as LlmSpan\n const nextTool = next.find((s) => s.span.kind === 'tool')\n if (!nextTool) return null\n const toolName = (nextTool.span as ToolSpan).toolName\n const out = (llm.output ?? '').toLowerCase()\n const mentioned = out.includes(toolName.toLowerCase())\n return mentioned\n ? { score: 1, rationale: `mentioned \"${toolName}\" before calling it` }\n : { score: 0.5, rationale: `called \"${toolName}\" without announcing it` }\n },\n }\n}\n\nfunction stableStringify(value: unknown): string {\n if (value === null || typeof value !== 'object') return JSON.stringify(value)\n if (Array.isArray(value)) return `[${value.map(stableStringify).join(',')}]`\n const keys = Object.keys(value as Record<string, unknown>).sort()\n return `{${keys.map((k) => `${JSON.stringify(k)}:${stableStringify((value as Record<string, unknown>)[k])}`).join(',')}}`\n}\n","/**\n * Inference-time PRM scoring — pick the best of N candidate trajectories\n * using a trained reward model (or a rule-based PRM as a proxy).\n *\n * The canonical Best-of-N pattern: generate N completions, score each\n * with a PRM, pick the winner. Here the scoring loop is framework-agnostic\n * — supply a TraceStore + PrmGrader + N run IDs → get ranking + winner.\n */\n\nimport type { TraceStore } from '../trace/store'\nimport type { PrmGradedTrace, PrmGrader } from './rubric'\n\nexport interface BestOfNResult {\n winner: PrmGradedTrace\n ranked: PrmGradedTrace[]\n /** Standard deviation of aggregate scores — small = candidates were homogenous. */\n stdDev: number\n}\n\n/** Default max concurrent grader calls. Bounds LLM fan-out so a wide ensemble\n * doesn't trigger a provider rate-limit storm. */\nconst DEFAULT_PRM_CONCURRENCY = 4\n\nexport interface PrmBestOfNOptions {\n /** Max concurrent `grader.grade` calls. Default 4. */\n concurrency?: number\n}\n\nexport async function prmBestOfN(\n store: TraceStore,\n grader: PrmGrader,\n runIds: string[],\n options: PrmBestOfNOptions = {},\n): Promise<BestOfNResult> {\n if (runIds.length === 0) throw new Error('prmBestOfN: at least 1 candidate required')\n const concurrency = Math.max(1, options.concurrency ?? DEFAULT_PRM_CONCURRENCY)\n const graded: PrmGradedTrace[] = new Array(runIds.length)\n let cursor = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = cursor++\n if (i >= runIds.length) return\n graded[i] = await grader.grade(store, runIds[i]!)\n }\n }\n await Promise.all(Array.from({ length: Math.min(concurrency, runIds.length) }, () => worker()))\n const ranked = [...graded].sort((a, b) => b.aggregateScore - a.aggregateScore)\n const mean = graded.reduce((a, g) => a + g.aggregateScore, 0) / graded.length\n const variance = graded.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / graded.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n\n/**\n * Weighted vote across multiple graders — use when you want a PRM ensemble\n * (e.g. rule-based + LLM-based + trained model). Each grader produces its\n * own ranking; we aggregate via rank-sum (Borda count) so no single grader\n * dominates via a different score scale.\n */\nexport async function prmEnsembleBestOfN(\n store: TraceStore,\n graders: PrmGrader[],\n runIds: string[],\n options: PrmBestOfNOptions = {},\n): Promise<BestOfNResult> {\n if (graders.length === 0) throw new Error('prmEnsembleBestOfN: at least 1 grader')\n if (runIds.length === 0) throw new Error('prmEnsembleBestOfN: at least 1 candidate required')\n const concurrency = Math.max(1, options.concurrency ?? DEFAULT_PRM_CONCURRENCY)\n\n // Flatten the (grader, runId) product and run it through a single bounded\n // pool. Nested unbounded fan-out (graders × runIds) would launch every LLM\n // call at once. allSettled isolates failures: one grader (or one runId)\n // failing doesn't void the whole ensemble.\n type Job = { graderIdx: number; runId: string }\n const jobs: Job[] = []\n for (let gi = 0; gi < graders.length; gi++) {\n for (const runId of runIds) jobs.push({ graderIdx: gi, runId })\n }\n const settled: PromiseSettledResult<PrmGradedTrace>[] = new Array(jobs.length)\n let cursor = 0\n async function worker(): Promise<void> {\n while (true) {\n const i = cursor++\n if (i >= jobs.length) return\n const job = jobs[i]!\n settled[i] = await graders[job.graderIdx]!.grade(store, job.runId)\n .then((value): PromiseSettledResult<PrmGradedTrace> => ({ status: 'fulfilled', value }))\n .catch((reason): PromiseSettledResult<PrmGradedTrace> => ({ status: 'rejected', reason }))\n }\n }\n await Promise.all(Array.from({ length: Math.min(concurrency, jobs.length) }, () => worker()))\n\n // Regroup fulfilled results into per-grader rankings. A grader contributes\n // only the candidates it successfully graded; a grader that graded nothing\n // is dropped from the vote rather than skewing it with phantom zeros.\n const perGrader: PrmGradedTrace[][] = graders.map(() => [])\n const failures: { graderIdx: number; runId: string; error: string }[] = []\n for (let i = 0; i < jobs.length; i++) {\n const job = jobs[i]!\n const result = settled[i]!\n if (result.status === 'fulfilled') perGrader[job.graderIdx]!.push(result.value)\n else {\n const error = result.reason instanceof Error ? result.reason.message : String(result.reason)\n failures.push({ graderIdx: job.graderIdx, runId: job.runId, error })\n }\n }\n const survivingGraders = perGrader.filter((ranking) => ranking.length > 0)\n if (survivingGraders.length === 0) {\n throw new Error(\n `prmEnsembleBestOfN: every grader failed on every candidate (${failures.length} call(s)). First error: ${failures[0]?.error ?? 'unknown'}`,\n )\n }\n for (const ranking of survivingGraders)\n ranking.sort((a, b) => b.aggregateScore - a.aggregateScore)\n\n // Borda: rank-sum across surviving graders.\n const bordaScores = new Map<string, number>()\n for (const ranking of survivingGraders) {\n ranking.forEach((g, rank) => {\n bordaScores.set(g.runId, (bordaScores.get(g.runId) ?? 0) + (ranking.length - rank))\n })\n }\n // Synthesize a ranking from the union of every successfully-graded trace,\n // ordered by Borda score. aggregateScore field kept for UX. Using the union\n // (not just the first grader) keeps a candidate that one grader dropped but\n // another graded.\n const byRun = new Map<string, PrmGradedTrace>()\n for (const ranking of survivingGraders) {\n for (const g of ranking) if (!byRun.has(g.runId)) byRun.set(g.runId, g)\n }\n const ranked = [...byRun.values()].sort(\n (a, b) => (bordaScores.get(b.runId) ?? 0) - (bordaScores.get(a.runId) ?? 0),\n )\n const mean = ranked.reduce((a, g) => a + g.aggregateScore, 0) / ranked.length\n const variance = ranked.reduce((a, g) => a + (g.aggregateScore - mean) ** 2, 0) / ranked.length\n return { winner: ranked[0]!, ranked, stdDev: Math.sqrt(variance) }\n}\n","/**\n * Process Reward Modeling — per-step rubric grading.\n *\n * A StepRubric inspects one span and returns a score + rationale.\n * PrmGrader applies an array of rubrics to every LLM span in a\n * trajectory (consumers can broaden to tool/retrieval spans via the\n * `kind` filter on each rubric).\n *\n * Why this matters: outcome-only eval (did the final artifact work?)\n * gives sparse reward — most agent turns are unattributable. PRMs\n * densify the signal so optimizers and RL fine-tuning can assign\n * credit per turn.\n */\n\nimport { TraceEmitter } from '../trace/emitter'\nimport type { JudgeSpan, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\nexport interface StepContext {\n trajectory: Trajectory\n step: TrajectoryStep\n /** Steps preceding `step` in trajectory order. */\n prior: TrajectoryStep[]\n /** Steps following `step`. */\n next: TrajectoryStep[]\n}\n\nexport interface StepRubric {\n id: string\n /** Only grade spans of these kinds (default: all). */\n kinds?: Array<Span['kind']>\n /** Weight in the aggregate score. Default 1. */\n weight?: number\n /** Returns score in 0..1 + optional rationale/evidence. Return `null` to\n * skip grading (rubric doesn't apply to this step). */\n grade: (\n ctx: StepContext,\n ) => Promise<{ score: number; rationale?: string; evidence?: string } | null>\n}\n\nexport interface GradedStep {\n spanId: string\n rubricId: string\n score: number\n weight: number\n rationale?: string\n evidence?: string\n}\n\nexport interface PrmGradedTrace {\n runId: string\n steps: GradedStep[]\n /** Weighted mean of all graded steps; 0..1. */\n aggregateScore: number\n /** Number of spans graded — useful for sanity-checking coverage. */\n gradedCount: number\n /** Number of spans in the trajectory that no rubric matched. */\n ungradedCount: number\n}\n\nexport class PrmGrader {\n constructor(private rubrics: StepRubric[]) {\n if (rubrics.length === 0) throw new Error('PrmGrader: at least 1 rubric required')\n }\n\n /**\n * Grade every eligible span in a run. Emits a JudgeVerdict span for each\n * (rubric × span) verdict so the result is visible to downstream pipelines\n * (judgeAgreementView, etc.) — PRM is just \"a judge that runs per span.\"\n */\n async grade(store: TraceStore, runId: string): Promise<PrmGradedTrace> {\n const trajectory = await buildTrajectory(store, runId)\n const emitter = new TraceEmitter(store, { runId })\n const steps: GradedStep[] = []\n let ungraded = 0\n for (let i = 0; i < trajectory.steps.length; i++) {\n const step = trajectory.steps[i]!\n const ctx: StepContext = {\n trajectory,\n step,\n prior: trajectory.steps.slice(0, i),\n next: trajectory.steps.slice(i + 1),\n }\n let gradedThis = false\n for (const rubric of this.rubrics) {\n if (rubric.kinds && !rubric.kinds.includes(step.span.kind)) continue\n const verdict = await rubric.grade(ctx)\n if (verdict === null) continue\n const weight = rubric.weight ?? 1\n steps.push({\n spanId: step.span.spanId,\n rubricId: rubric.id,\n score: verdict.score,\n weight,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n })\n gradedThis = true\n // Persist the verdict as a JudgeSpan so the query pipelines see it\n await emitter.recordJudge({\n judgeId: `prm:${rubric.id}`,\n targetSpanId: step.span.spanId,\n dimension: 'step_quality',\n score: verdict.score,\n rationale: verdict.rationale,\n evidence: verdict.evidence,\n name: `prm:${rubric.id}`,\n })\n }\n if (!gradedThis) ungraded++\n }\n\n const totalWeight = steps.reduce((a, s) => a + s.weight, 0)\n const aggregateScore =\n totalWeight === 0 ? 0 : steps.reduce((a, s) => a + s.score * s.weight, 0) / totalWeight\n\n return { runId, steps, aggregateScore, gradedCount: steps.length, ungradedCount: ungraded }\n }\n}\n\n/** Helper: reads JudgeVerdict spans that PRM emitted so downstream pipelines\n * can distinguish PRM verdicts from human or top-level LLM judges. */\nexport function isPrmVerdict(verdict: JudgeSpan): boolean {\n return verdict.judgeId.startsWith('prm:')\n}\n"],"mappings":";;;;;;;;;;;;;;AAWO,SAAS,mBACd,OAAkE,CAAC,GACvD;AACZ,QAAM,MAAM,KAAK,YAAY;AAC7B,QAAM,MAAM,KAAK,YAAY;AAC7B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,OAAO,IAAI,UAAU,IAAI;AAC/B,UAAI,QAAQ,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,eAAe;AAC5D,UAAI,MAAM;AACR,eAAO,EAAE,OAAO,KAAK,IAAI,GAAG,MAAM,GAAG,GAAG,WAAW,cAAc,GAAG,MAAM,GAAG,IAAI;AACnF,UAAI,MAAM;AACR,eAAO;AAAA,UACL,OAAO,KAAK,IAAI,GAAG,KAAK,MAAM,OAAO,GAAG;AAAA,UACxC,WAAW,cAAc,GAAG,MAAM,GAAG;AAAA,QACvC;AACF,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,GAAG,mBAAmB;AAAA,IACzD;AAAA,EACF;AACF;AAGO,SAAS,kBAAkB,OAA4B,CAAC,GAAe;AAC5E,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,OAAO,KAAK;AAClB,UAAI,KAAK,WAAW;AAClB,eAAO,EAAE,OAAO,GAAG,WAAW,UAAU,KAAK,SAAS,SAAS,GAAG;AACpE,YAAM,IAAI,KAAK;AACf,UAAI,MAAM,QAAQ,MAAM,OAAW,QAAO,EAAE,OAAO,KAAK,WAAW,eAAe;AAClF,YAAM,SAAS,OAAO,MAAM,WAAW,IAAI,KAAK,UAAU,CAAC;AAC3D,UAAI,OAAO,SAAS,EAAG,QAAO,EAAE,OAAO,KAAK,WAAW,cAAc;AACrE,aAAO,EAAE,OAAO,GAAG,WAAW,GAAG,KAAK,QAAQ,MAAM;AAAA,IACtD;AAAA,EACF;AACF;AAGO,SAAS,uBAAuB,OAA4B,CAAC,GAAe;AACjF,QAAM,SAAS,KAAK,UAAU;AAC9B,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,MAAM;AAAA,IACd;AAAA,IACA,MAAM,MAAM,EAAE,MAAM,MAAM,GAAG;AAC3B,YAAM,OAAO,KAAK;AAClB,YAAM,eAAe,MAAM,OAAO,CAAC,MAAM;AACvC,YAAI,EAAE,KAAK,SAAS,OAAQ,QAAO;AACnC,cAAM,KAAK,EAAE;AACb,eACE,GAAG,aAAa,KAAK,YAAY,gBAAgB,GAAG,IAAI,MAAM,gBAAgB,KAAK,IAAI;AAAA,MAE3F,CAAC;AACD,UAAI,aAAa,WAAW,EAAG,QAAO,EAAE,OAAO,GAAG,WAAW,aAAa;AAC1E,aAAO;AAAA,QACL,OAAO,KAAK,IAAI,GAAG,IAAI,aAAa,SAAS,GAAG;AAAA,QAChD,WAAW,GAAG,aAAa,MAAM;AAAA,MACnC;AAAA,IACF;AAAA,EACF;AACF;AAIO,SAAS,iBAAiB,OAAgD,CAAC,GAAe;AAC/F,QAAM,SAAS,KAAK,UAAU;AAC9B,QAAM,UAAU,KAAK,WAAW;AAAA,IAC9B;AAAA,IACA;AAAA,EACF;AACA,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb;AAAA,IACA,MAAM,MAAM,EAAE,KAAK,GAAG;AACpB,YAAM,MAAM,KAAK;AACjB,YAAM,MAAM,IAAI,UAAU;AAC1B,YAAM,UAAU,QAAQ,KAAK,CAAC,OAAO,GAAG,KAAK,GAAG,CAAC;AACjD,aAAO,UACH,EAAE,OAAO,GAAG,WAAW,yBAAyB,IAChD,EAAE,OAAO,GAAG,WAAW,aAAa;AAAA,IAC1C;AAAA,EACF;AACF;AAIO,SAAS,0BAA0B,OAA4B,CAAC,GAAe;AACpF,SAAO;AAAA,IACL,IAAI;AAAA,IACJ,OAAO,CAAC,KAAK;AAAA,IACb,QAAQ,KAAK,UAAU;AAAA,IACvB,MAAM,MAAM,EAAE,MAAM,KAAK,GAAG;AAC1B,YAAM,MAAM,KAAK;AACjB,YAAM,WAAW,KAAK,KAAK,CAAC,MAAM,EAAE,KAAK,SAAS,MAAM;AACxD,UAAI,CAAC,SAAU,QAAO;AACtB,YAAM,WAAY,SAAS,KAAkB;AAC7C,YAAM,OAAO,IAAI,UAAU,IAAI,YAAY;AAC3C,YAAM,YAAY,IAAI,SAAS,SAAS,YAAY,CAAC;AACrD,aAAO,YACH,EAAE,OAAO,GAAG,WAAW,cAAc,QAAQ,sBAAsB,IACnE,EAAE,OAAO,KAAK,WAAW,WAAW,QAAQ,0BAA0B;AAAA,IAC5E;AAAA,EACF;AACF;AAEA,SAAS,gBAAgB,OAAwB;AAC/C,MAAI,UAAU,QAAQ,OAAO,UAAU,SAAU,QAAO,KAAK,UAAU,KAAK;AAC5E,MAAI,MAAM,QAAQ,KAAK,EAAG,QAAO,IAAI,MAAM,IAAI,eAAe,EAAE,KAAK,GAAG,CAAC;AACzE,QAAM,OAAO,OAAO,KAAK,KAAgC,EAAE,KAAK;AAChE,SAAO,IAAI,KAAK,IAAI,CAAC,MAAM,GAAG,KAAK,UAAU,CAAC,CAAC,IAAI,gBAAiB,MAAkC,CAAC,CAAC,CAAC,EAAE,EAAE,KAAK,GAAG,CAAC;AACxH;;;AC5GA,IAAM,0BAA0B;AAOhC,eAAsB,WACpB,OACA,QACA,QACA,UAA6B,CAAC,GACN;AACxB,MAAI,OAAO,WAAW,EAAG,OAAM,IAAI,MAAM,2CAA2C;AACpF,QAAM,cAAc,KAAK,IAAI,GAAG,QAAQ,eAAe,uBAAuB;AAC9E,QAAM,SAA2B,IAAI,MAAM,OAAO,MAAM;AACxD,MAAI,SAAS;AACb,iBAAe,SAAwB;AACrC,WAAO,MAAM;AACX,YAAM,IAAI;AACV,UAAI,KAAK,OAAO,OAAQ;AACxB,aAAO,CAAC,IAAI,MAAM,OAAO,MAAM,OAAO,OAAO,CAAC,CAAE;AAAA,IAClD;AAAA,EACF;AACA,QAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,OAAO,MAAM,EAAE,GAAG,MAAM,OAAO,CAAC,CAAC;AAC9F,QAAM,SAAS,CAAC,GAAG,MAAM,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAC7E,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;AAQA,eAAsB,mBACpB,OACA,SACA,QACA,UAA6B,CAAC,GACN;AACxB,MAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AACjF,MAAI,OAAO,WAAW,EAAG,OAAM,IAAI,MAAM,mDAAmD;AAC5F,QAAM,cAAc,KAAK,IAAI,GAAG,QAAQ,eAAe,uBAAuB;AAO9E,QAAM,OAAc,CAAC;AACrB,WAAS,KAAK,GAAG,KAAK,QAAQ,QAAQ,MAAM;AAC1C,eAAW,SAAS,OAAQ,MAAK,KAAK,EAAE,WAAW,IAAI,MAAM,CAAC;AAAA,EAChE;AACA,QAAM,UAAkD,IAAI,MAAM,KAAK,MAAM;AAC7E,MAAI,SAAS;AACb,iBAAe,SAAwB;AACrC,WAAO,MAAM;AACX,YAAM,IAAI;AACV,UAAI,KAAK,KAAK,OAAQ;AACtB,YAAM,MAAM,KAAK,CAAC;AAClB,cAAQ,CAAC,IAAI,MAAM,QAAQ,IAAI,SAAS,EAAG,MAAM,OAAO,IAAI,KAAK,EAC9D,KAAK,CAAC,WAAiD,EAAE,QAAQ,aAAa,MAAM,EAAE,EACtF,MAAM,CAAC,YAAkD,EAAE,QAAQ,YAAY,OAAO,EAAE;AAAA,IAC7F;AAAA,EACF;AACA,QAAM,QAAQ,IAAI,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,KAAK,MAAM,EAAE,GAAG,MAAM,OAAO,CAAC,CAAC;AAK5F,QAAM,YAAgC,QAAQ,IAAI,MAAM,CAAC,CAAC;AAC1D,QAAM,WAAkE,CAAC;AACzE,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,MAAM,KAAK,CAAC;AAClB,UAAM,SAAS,QAAQ,CAAC;AACxB,QAAI,OAAO,WAAW,YAAa,WAAU,IAAI,SAAS,EAAG,KAAK,OAAO,KAAK;AAAA,SACzE;AACH,YAAM,QAAQ,OAAO,kBAAkB,QAAQ,OAAO,OAAO,UAAU,OAAO,OAAO,MAAM;AAC3F,eAAS,KAAK,EAAE,WAAW,IAAI,WAAW,OAAO,IAAI,OAAO,MAAM,CAAC;AAAA,IACrE;AAAA,EACF;AACA,QAAM,mBAAmB,UAAU,OAAO,CAAC,YAAY,QAAQ,SAAS,CAAC;AACzE,MAAI,iBAAiB,WAAW,GAAG;AACjC,UAAM,IAAI;AAAA,MACR,+DAA+D,SAAS,MAAM,2BAA2B,SAAS,CAAC,GAAG,SAAS,SAAS;AAAA,IAC1I;AAAA,EACF;AACA,aAAW,WAAW;AACpB,YAAQ,KAAK,CAAC,GAAG,MAAM,EAAE,iBAAiB,EAAE,cAAc;AAG5D,QAAM,cAAc,oBAAI,IAAoB;AAC5C,aAAW,WAAW,kBAAkB;AACtC,YAAQ,QAAQ,CAAC,GAAG,SAAS;AAC3B,kBAAY,IAAI,EAAE,QAAQ,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,QAAQ,SAAS,KAAK;AAAA,IACpF,CAAC;AAAA,EACH;AAKA,QAAM,QAAQ,oBAAI,IAA4B;AAC9C,aAAW,WAAW,kBAAkB;AACtC,eAAW,KAAK,QAAS,KAAI,CAAC,MAAM,IAAI,EAAE,KAAK,EAAG,OAAM,IAAI,EAAE,OAAO,CAAC;AAAA,EACxE;AACA,QAAM,SAAS,CAAC,GAAG,MAAM,OAAO,CAAC,EAAE;AAAA,IACjC,CAAC,GAAG,OAAO,YAAY,IAAI,EAAE,KAAK,KAAK,MAAM,YAAY,IAAI,EAAE,KAAK,KAAK;AAAA,EAC3E;AACA,QAAM,OAAO,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,gBAAgB,CAAC,IAAI,OAAO;AACvE,QAAM,WAAW,OAAO,OAAO,CAAC,GAAG,MAAM,KAAK,EAAE,iBAAiB,SAAS,GAAG,CAAC,IAAI,OAAO;AACzF,SAAO,EAAE,QAAQ,OAAO,CAAC,GAAI,QAAQ,QAAQ,KAAK,KAAK,QAAQ,EAAE;AACnE;;;AC1EO,IAAM,YAAN,MAAgB;AAAA,EACrB,YAAoB,SAAuB;AAAvB;AAClB,QAAI,QAAQ,WAAW,EAAG,OAAM,IAAI,MAAM,uCAAuC;AAAA,EACnF;AAAA,EAFoB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASpB,MAAM,MAAM,OAAmB,OAAwC;AACrE,UAAM,aAAa,MAAM,gBAAgB,OAAO,KAAK;AACrD,UAAM,UAAU,IAAI,aAAa,OAAO,EAAE,MAAM,CAAC;AACjD,UAAM,QAAsB,CAAC;AAC7B,QAAI,WAAW;AACf,aAAS,IAAI,GAAG,IAAI,WAAW,MAAM,QAAQ,KAAK;AAChD,YAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,YAAM,MAAmB;AAAA,QACvB;AAAA,QACA;AAAA,QACA,OAAO,WAAW,MAAM,MAAM,GAAG,CAAC;AAAA,QAClC,MAAM,WAAW,MAAM,MAAM,IAAI,CAAC;AAAA,MACpC;AACA,UAAI,aAAa;AACjB,iBAAW,UAAU,KAAK,SAAS;AACjC,YAAI,OAAO,SAAS,CAAC,OAAO,MAAM,SAAS,KAAK,KAAK,IAAI,EAAG;AAC5D,cAAM,UAAU,MAAM,OAAO,MAAM,GAAG;AACtC,YAAI,YAAY,KAAM;AACtB,cAAM,SAAS,OAAO,UAAU;AAChC,cAAM,KAAK;AAAA,UACT,QAAQ,KAAK,KAAK;AAAA,UAClB,UAAU,OAAO;AAAA,UACjB,OAAO,QAAQ;AAAA,UACf;AAAA,UACA,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,QACpB,CAAC;AACD,qBAAa;AAEb,cAAM,QAAQ,YAAY;AAAA,UACxB,SAAS,OAAO,OAAO,EAAE;AAAA,UACzB,cAAc,KAAK,KAAK;AAAA,UACxB,WAAW;AAAA,UACX,OAAO,QAAQ;AAAA,UACf,WAAW,QAAQ;AAAA,UACnB,UAAU,QAAQ;AAAA,UAClB,MAAM,OAAO,OAAO,EAAE;AAAA,QACxB,CAAC;AAAA,MACH;AACA,UAAI,CAAC,WAAY;AAAA,IACnB;AAEA,UAAM,cAAc,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,CAAC;AAC1D,UAAM,iBACJ,gBAAgB,IAAI,IAAI,MAAM,OAAO,CAAC,GAAG,MAAM,IAAI,EAAE,QAAQ,EAAE,QAAQ,CAAC,IAAI;AAE9E,WAAO,EAAE,OAAO,OAAO,gBAAgB,aAAa,MAAM,QAAQ,eAAe,SAAS;AAAA,EAC5F;AACF;AAIO,SAAS,aAAa,SAA6B;AACxD,SAAO,QAAQ,QAAQ,WAAW,MAAM;AAC1C;","names":[]}
@@ -1,9 +1,9 @@
1
1
  import { o as Mutator, I as ImprovementDriver, S as Scenario, G as Gate, k as GateResult, j as GateContext, g as CampaignResult, M as MutableSurface, c as GateDecision } from './types-BU-7W85F.js';
2
- import { R as RedTeamCase } from './red-team-BXHil6c8.js';
2
+ import { R as RedTeamCase } from './red-team-BWdoyleI.js';
3
3
  import { R as RunRecord } from './run-record-e7vj1uZQ.js';
4
4
  import { D as Direction } from './pareto-E-pembql.js';
5
- import { a as PairedBootstrapResult } from './statistics-C7PozGrZ.js';
6
- import { b as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-5z_l5zDz.js';
5
+ import { a as PairedBootstrapResult } from './statistics-CCJpTGOS.js';
6
+ import { b as RunCampaignOptions, C as CampaignStorage } from './run-improvement-loop-DBahB8Ax.js';
7
7
  import { HostedClient, TraceSpanEvent } from './hosted/index.js';
8
8
 
9
9
  /**
@@ -1,5 +1,5 @@
1
1
  import { L as LlmSpan, J as JudgeSpan, R as Run, F as FailureClass, T as ToolSpan } from './schema-m0gsnbt3.js';
2
- import { T as TraceStore } from './store-CKUAgsJz.js';
2
+ import { T as TraceStore } from './store-BcFXE6LG.js';
3
3
 
4
4
  /**
5
5
  * Typed query helpers over TraceStore.
@@ -23,6 +23,7 @@ declare function aggregateLlm(spans: LlmSpan[]): {
23
23
  inputTokens: number;
24
24
  outputTokens: number;
25
25
  cachedTokens: number;
26
+ reasoningTokens: number;
26
27
  costUsd: number;
27
28
  };
28
29
  /** Pick the outcome's failure class when present, else derive 'success' from run status. */
@@ -1,5 +1,5 @@
1
1
  import { a as DatasetScenario, b as Dataset } from './dataset-BbGkaN2I.js';
2
- import { T as TraceStore } from './store-CKUAgsJz.js';
2
+ import { T as TraceStore } from './store-BcFXE6LG.js';
3
3
 
4
4
  /**
5
5
  * Red-team battery — adversarial scenario corpus with per-category
@@ -1,5 +1,5 @@
1
1
  import { D as DatasetSplit, c as DatasetManifest, a as DatasetScenario } from './dataset-BbGkaN2I.js';
2
- import { m as GateDecision } from './summary-report-DGmUucwQ.js';
2
+ import { m as GateDecision } from './summary-report-BDOFevaT.js';
3
3
  import { R as RunRecord, a as RunSplitTag } from './run-record-e7vj1uZQ.js';
4
4
 
5
5
  /**
@@ -1,15 +1,15 @@
1
1
  export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-Cy_W-hWZ.js';
2
- export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-euXIV_Sk.js';
2
+ export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-BEbWmVYj.js';
3
3
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
4
- export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-C7PozGrZ.js';
5
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-DGmUucwQ.js';
4
+ export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-CCJpTGOS.js';
5
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BDOFevaT.js';
6
6
  import './run-record-e7vj1uZQ.js';
7
7
  import './errors-CzMUYo7b.js';
8
8
  import './schema-m0gsnbt3.js';
9
9
  import './outcome-store-rnXLEqSn.js';
10
10
  import './dataset-BbGkaN2I.js';
11
11
  import './judge-calibration-0p2QcWNE.js';
12
- import './types-Croy5h7V.js';
12
+ import './types-C7DGg5ex.js';
13
13
  import '@tangle-network/tcloud';
14
- import './failure-cluster-CL7IVgkJ.js';
15
- import './store-CKUAgsJz.js';
14
+ import './failure-cluster-DH9Flgcf.js';
15
+ import './store-BcFXE6LG.js';
package/dist/reporting.js CHANGED
@@ -4,7 +4,7 @@ import {
4
4
  evaluateReleaseConfidence,
5
5
  judgeReplayGate,
6
6
  renderReleaseReport
7
- } from "./chunk-4FBZZIYD.js";
7
+ } from "./chunk-QTMYW64F.js";
8
8
  import {
9
9
  rubricPredictiveValidity
10
10
  } from "./chunk-YRZ4M5GS.js";
@@ -18,12 +18,12 @@ import {
18
18
  paretoChart,
19
19
  researchReport,
20
20
  summaryTable
21
- } from "./chunk-CY6U5S3X.js";
21
+ } from "./chunk-URWVKRCS.js";
22
22
  import {
23
23
  benjaminiHochberg,
24
24
  pairedBootstrap,
25
25
  wilcoxonSignedRank
26
- } from "./chunk-LMZQ2Z4U.js";
26
+ } from "./chunk-56GC6VYO.js";
27
27
  import "./chunk-VSMTAMNK.js";
28
28
  import "./chunk-3BFEG2F6.js";
29
29
  import "./chunk-PZ5AY32C.js";
@@ -1,11 +1,11 @@
1
1
  import { a as RunSplitTag, b as RunTokenUsage, c as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, d as AgentProfileCellInput, R as RunRecord } from './run-record-e7vj1uZQ.js';
2
- import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-BeEcAokY.js';
3
- import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-DGmUucwQ.js';
4
- import { T as TraceEmitter, R as RunCompleteHook } from './emitter-DEZwY14K.js';
5
- import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-VJ9A7aST.js';
2
+ import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-Bj7g0rqu.js';
3
+ import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-BDOFevaT.js';
4
+ import { T as TraceEmitter, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
5
+ import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-D2t12mMw.js';
6
6
  import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
7
7
  import { F as FailureClass } from './schema-m0gsnbt3.js';
8
- import { T as TraceStore } from './store-CKUAgsJz.js';
8
+ import { T as TraceStore } from './store-BcFXE6LG.js';
9
9
 
10
10
  /**
11
11
  * EvalCampaign — opinionated matrix runner that wires the four
package/dist/rl.d.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
2
- import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-BoR-041R.js';
3
- export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-BoR-041R.js';
2
+ import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-B8A4BDR3.js';
3
+ export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-B8A4BDR3.js';
4
4
  import { g as CampaignResult } from './types-BU-7W85F.js';
5
5
  import { a as RunSplitTag, R as RunRecord } from './run-record-e7vj1uZQ.js';
6
6
  import { a as VerificationReport } from './multi-layer-verifier-DUZXrPDA.js';
@@ -8,19 +8,19 @@ export { A as AdversarialMutation, a as AdversarialScenario, b as AdversarialSea
8
8
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
9
9
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
10
10
  import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-Cy_W-hWZ.js';
11
- import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-DE6Gpnb4.js';
12
- export { r as runEvalCampaign } from './researcher-DE6Gpnb4.js';
11
+ import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-B0C2_fVO.js';
12
+ export { r as runEvalCampaign } from './researcher-B0C2_fVO.js';
13
13
  import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
14
14
  import './schema-m0gsnbt3.js';
15
- import './store-CKUAgsJz.js';
15
+ import './store-BcFXE6LG.js';
16
16
  import './errors-CzMUYo7b.js';
17
17
  import './verdict-C9MlYujm.js';
18
- import './llm-client-BeEcAokY.js';
18
+ import './llm-client-Bj7g0rqu.js';
19
19
  import './raw-provider-sink-C46HDghv.js';
20
- import './summary-report-DGmUucwQ.js';
21
- import './failure-cluster-CL7IVgkJ.js';
22
- import './emitter-DEZwY14K.js';
23
- import './integrity-VJ9A7aST.js';
20
+ import './summary-report-BDOFevaT.js';
21
+ import './failure-cluster-DH9Flgcf.js';
22
+ import './emitter-C2rqGH_l.js';
23
+ import './integrity-D2t12mMw.js';
24
24
 
25
25
  /**
26
26
  * Test-time compute scaling curves.
package/dist/rl.js CHANGED
@@ -16,19 +16,19 @@ import {
16
16
  } from "./chunk-3RF76KTD.js";
17
17
  import {
18
18
  runEvalCampaign
19
- } from "./chunk-GSH6QNNS.js";
20
- import "./chunk-CVVHBFGN.js";
19
+ } from "./chunk-KW53MSA5.js";
20
+ import "./chunk-CWNP4DV4.js";
21
21
  import {
22
22
  rubricPredictiveValidity
23
23
  } from "./chunk-YRZ4M5GS.js";
24
24
  import {
25
25
  evaluateInterimReleaseConfidence
26
26
  } from "./chunk-MAZ26DC7.js";
27
- import "./chunk-CY6U5S3X.js";
27
+ import "./chunk-URWVKRCS.js";
28
28
  import {
29
29
  benjaminiHochberg,
30
30
  wilcoxonSignedRank
31
- } from "./chunk-LMZQ2Z4U.js";
31
+ } from "./chunk-56GC6VYO.js";
32
32
  import {
33
33
  observationsFromRunRecords,
34
34
  thompsonCurriculum,
@@ -1,6 +1,6 @@
1
1
  import { S as Span, J as JudgeSpan } from './schema-m0gsnbt3.js';
2
- import { T as TraceStore } from './store-CKUAgsJz.js';
3
- import { T as Trajectory, a as TrajectoryStep } from './trajectory-GEdXJCL5.js';
2
+ import { T as TraceStore } from './store-BcFXE6LG.js';
3
+ import { T as Trajectory, a as TrajectoryStep } from './trajectory-2TkpSEVh.js';
4
4
 
5
5
  /**
6
6
  * Process Reward Modeling — per-step rubric grading.
@@ -1,10 +1,10 @@
1
1
  import {
2
2
  runCampaign
3
- } from "./chunk-TWS7AZEY.js";
4
- import "./chunk-LMZQ2Z4U.js";
3
+ } from "./chunk-2K6UUZ7P.js";
4
+ import "./chunk-56GC6VYO.js";
5
5
  import "./chunk-3BFEG2F6.js";
6
6
  import "./chunk-PZ5AY32C.js";
7
7
  export {
8
8
  runCampaign
9
9
  };
10
- //# sourceMappingURL=run-campaign-RDGAM5KJ.js.map
10
+ //# sourceMappingURL=run-campaign-7WNXMDSN.js.map
@@ -1,5 +1,5 @@
1
1
  import { R as Run, S as Span, a as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-m0gsnbt3.js';
2
- import { T as TraceStore } from './store-CKUAgsJz.js';
2
+ import { T as TraceStore } from './store-BcFXE6LG.js';
3
3
 
4
4
  interface RunScore {
5
5
  success: number;
@@ -1,4 +1,4 @@
1
- import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
1
+ import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
2
2
  import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-BU-7W85F.js';
3
3
 
4
4
  /**
@@ -1,9 +1,9 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
2
  import { z } from 'zod';
3
- import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-2VVIL04s.js';
4
- import { T as TraceAnalystKindSpec } from './kind-factory-5b7xXXOr.js';
3
+ import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-Ce17tDlG.js';
4
+ import { T as TraceAnalystKindSpec } from './kind-factory-0BhLSI27.js';
5
5
  import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
6
- import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
6
+ import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
7
7
  import { S as Severity } from './multi-layer-verifier-DUZXrPDA.js';
8
8
 
9
9
  interface CreateAnalystAiConfig {
@@ -1,5 +1,5 @@
1
1
  import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-0p2QcWNE.js';
2
- import { J as JudgeScore } from './types-Croy5h7V.js';
2
+ import { J as JudgeScore } from './types-C7DGg5ex.js';
3
3
 
4
4
  /** Identity: dimensions already follow "higher = better" by prompt convention
5
5
  * (inverted dims like hallucination are scored 10 = best at the source). */
@@ -199,6 +199,39 @@ declare function pairedMde(opts: {
199
199
  power?: number;
200
200
  twoSided?: boolean;
201
201
  }): number;
202
+ /**
203
+ * Number of paired observations needed for a McNemar test to reach a target
204
+ * power — the pre-registration companion to {@link mcnemar}. Parametrised by the
205
+ * expected discordant-cell probabilities `p10` (P[treatment wins on a pair]) and
206
+ * `p01` (P[control wins]); concordant pairs carry no information, so the count
207
+ * is driven entirely by the discordant rate. Lachin's (1992) asymptotic normal
208
+ * approximation: with discordant rate `pDisc = p10 + p01` and marginal effect
209
+ * `δ = p10 − p01`,
210
+ * n = ( z_{1-α/2}·√pDisc + z_{1-β}·√(pDisc − δ²) )² / δ².
211
+ * Returns Infinity when there is no effect (p10 === p01). Asymptotic — at the
212
+ * tiny discordant counts where the exact {@link mcnemar} differs from the normal
213
+ * approximation, treat the result as a lower bound and prefer the discordant-pair
214
+ * floor.
215
+ */
216
+ declare function mcnemarRequiredN(opts: {
217
+ p10: number;
218
+ p01: number;
219
+ alpha?: number;
220
+ power?: number;
221
+ twoSided?: boolean;
222
+ }): number;
223
+ /**
224
+ * Power of a McNemar test at a given number of paired observations, the inverse
225
+ * of {@link mcnemarRequiredN} (same Lachin asymptotic model, same parameters).
226
+ * Returns a value in [0, 1]; equals `alpha` when there is no effect.
227
+ */
228
+ declare function mcnemarPower(opts: {
229
+ p10: number;
230
+ p01: number;
231
+ nPairs: number;
232
+ alpha?: number;
233
+ twoSided?: boolean;
234
+ }): number;
202
235
  /** Bonferroni adjustment: multiply every p-value by the test count, clamp at 1. */
203
236
  declare function bonferroni(pValues: number[], alpha?: number): {
204
237
  adjusted: number[];
@@ -245,6 +278,88 @@ interface PairedBootstrapOptions {
245
278
  * gain is real at the confidence level. Throws on unequal sample sizes.
246
279
  */
247
280
  declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
281
+ /** A binomial proportion estimate with a confidence interval. */
282
+ interface ProportionInterval {
283
+ /** Point estimate successes / n (0 when n = 0). */
284
+ estimate: number;
285
+ /** Lower bound, clamped to [0, 1]. */
286
+ lower: number;
287
+ /** Upper bound, clamped to [0, 1]. */
288
+ upper: number;
289
+ }
290
+ /**
291
+ * Wilson score interval for a binomial proportion. Correct at small n and near
292
+ * 0/1, where the normal (Wald) approximation produces bounds outside [0, 1] and
293
+ * understates coverage. Use this for any pass-rate / hit-rate / realness-rate
294
+ * CI — the continuous `confidenceInterval` assumes the wrong distribution for a
295
+ * proportion. `n = 0 ⇒ {0, 0, 0}`.
296
+ */
297
+ declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
298
+ /** Result of a McNemar paired-binary significance test. */
299
+ interface McNemarResult {
300
+ /** Total paired observations. */
301
+ n: number;
302
+ /** Discordant pairs (b + c) — the only ones that carry signal. */
303
+ nDiscordant: number;
304
+ /** Pairs where treatment succeeded and control failed ("newly correct"). */
305
+ b: number;
306
+ /** Pairs where control succeeded and treatment failed ("newly wrong"). */
307
+ c: number;
308
+ /** Continuity-corrected chi-square statistic (reference; exact p drives the call). */
309
+ statistic: number;
310
+ /** Two-sided p-value. Exact (binomial sign test on discordant pairs). */
311
+ pValue: number;
312
+ }
313
+ /**
314
+ * McNemar's test for paired binary outcomes — the correct significance test for
315
+ * "does treatment change the success rate vs control on the SAME items". Only
316
+ * discordant pairs (one arm right, the other wrong) carry information; concordant
317
+ * pairs are uninformative, so a paired t-test / two-proportion z-test on the raw
318
+ * rates is wrong here. The p-value is exact: under H0 the b "treatment-wins" are
319
+ * Binomial(b + c, 0.5), so the two-sided p is the doubled binomial tail — correct
320
+ * at the small discordant counts typical of eval runs (no continuity-corrected
321
+ * chi-square approximation needed, though it is returned as `statistic` for
322
+ * reference). Inputs are paired 0/1 (or boolean) arrays, control first to match
323
+ * the module's (before, after) convention. Throws on unequal lengths.
324
+ */
325
+ declare function mcnemar(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>): McNemarResult;
326
+ /** A paired binary effect size (treatment rate − control rate) with a CI. */
327
+ interface RiskDifferenceResult {
328
+ /** Total paired observations. */
329
+ n: number;
330
+ /** Discordant pairs: treatment-win count. */
331
+ b: number;
332
+ /** Discordant pairs: control-win count. */
333
+ c: number;
334
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
335
+ riskDifference: number;
336
+ /** Lower bound of the CI, clamped to [-1, 1]. */
337
+ lower: number;
338
+ /** Upper bound of the CI, clamped to [-1, 1]. */
339
+ upper: number;
340
+ /** Confidence level used. */
341
+ confidence: number;
342
+ }
343
+ /**
344
+ * Paired risk difference (the effect-size companion to {@link mcnemar}): the
345
+ * change in success rate p(treatment) − p(control) on matched items, which for
346
+ * paired binary data equals (b − c) / n. The CI uses the paired variance from
347
+ * the discordant counts, not the independent-samples formula (which overstates
348
+ * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
349
+ * arrays, control first. Throws on unequal lengths.
350
+ */
351
+ declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
352
+ /**
353
+ * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
354
+ * Language Models Trained on Code"). Given `n` independent samples for one
355
+ * problem of which `c` pass, the probability that at least one of a random k of
356
+ * them passes is 1 − C(n−c, k) / C(n, k). Estimating pass@k as "did any of the
357
+ * first k pass" is biased high at small n; this is the variance-reduced estimator
358
+ * averaged implicitly over all k-subsets. Average the per-problem values across
359
+ * the suite for the corpus pass@k. Computed in the numerically stable product
360
+ * form. Requires 1 ≤ k ≤ n and 0 ≤ c ≤ n.
361
+ */
362
+ declare function passAtK(n: number, c: number, k: number): number;
248
363
  interface EProcessOptions {
249
364
  /** Type-I error budget. The process decides when wealth ≥ 1/alpha
250
365
  * (Ville's inequality). Default 0.05. */
@@ -315,4 +430,4 @@ declare function eProcess(opts?: EProcessOptions): EProcess;
315
430
  * gate verdicts is non-reproducible by construction). */
316
431
  declare function mulberry32(seed: number): () => number;
317
432
 
318
- export { partialCredit as A, requiredSampleSize as B, type CorpusAgreementReport as C, weightedComposite as D, type EProcessState as E, weightedMean as F, type PairedBootstrapOptions as P, type WeightedCompositeInput as W, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type WeightedCompositeResult as j, bonferroni as k, cliffsDelta as l, cohensD as m, confidenceInterval as n, corpusInterRaterAgreement as o, pairedBootstrap as p, corpusInterRaterAgreementFromJudgeScores as q, eProcess as r, interRaterReliability as s, interpretCliffs as t, mannWhitneyU as u, mulberry32 as v, wilcoxonSignedRank as w, normalizeScores as x, pairedMde as y, pairedTTest as z };
433
+ export { mulberry32 as A, normalizeScores as B, type CorpusAgreementReport as C, pairedMde as D, type EProcessState as E, pairedRiskDifference as F, pairedTTest as G, partialCredit as H, passAtK as I, requiredSampleSize as J, weightedComposite as K, weightedMean as L, type McNemarResult as M, wilson as N, type PairedBootstrapOptions as P, type RiskDifferenceResult as R, type WeightedCompositeInput as W, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type ProportionInterval as j, type WeightedCompositeResult as k, bonferroni as l, cliffsDelta as m, cohensD as n, confidenceInterval as o, pairedBootstrap as p, corpusInterRaterAgreement as q, corpusInterRaterAgreementFromJudgeScores as r, eProcess as s, interRaterReliability as t, interpretCliffs as u, mannWhitneyU as v, wilcoxonSignedRank as w, mcnemar as x, mcnemarPower as y, mcnemarRequiredN as z };
@@ -77,12 +77,27 @@ declare class FileSystemTraceStore implements TraceStore {
77
77
  private maxBytes;
78
78
  /** Lazy in-memory index for queries — populated on first read. */
79
79
  private index?;
80
- private loaded;
80
+ /** Memoized index build — concurrent first reads share one build, and an
81
+ * append racing an in-flight load awaits this so its row isn't lost. */
82
+ private indexPromise?;
83
+ /** Strictly-increasing rollover stamp. Date.now() alone collides when two
84
+ * rollovers land in the same millisecond, overwriting a rolled file. */
85
+ private lastRolloverStamp;
86
+ /**
87
+ * Per-file append serialization. stat → conditional-rename → appendFile is a
88
+ * read-modify-write on the active file; without a lock two concurrent appends
89
+ * to the same `name` can both pass the size check (exceeding maxBytes) or one
90
+ * can append to a file the other just renamed away. Each entry chains the
91
+ * next append behind the prior one for that file.
92
+ */
93
+ private appendLocks;
81
94
  constructor(options: FileSystemTraceStoreOptions);
82
95
  private ensureDir;
83
96
  private append;
97
+ private appendLocked;
84
98
  private insertInto;
85
99
  private load;
100
+ private buildIndex;
86
101
  appendRun(run: Run): Promise<void>;
87
102
  updateRun(runId: string, patch: Partial<Run>): Promise<void>;
88
103
  appendSpan(span: Span): Promise<void>;
@@ -1,5 +1,5 @@
1
1
  import { R as RunRecord } from './run-record-e7vj1uZQ.js';
2
- import { F as FailureClusterReport } from './failure-cluster-CL7IVgkJ.js';
2
+ import { F as FailureClusterReport } from './failure-cluster-DH9Flgcf.js';
3
3
 
4
4
  /**
5
5
  * HeldOutGate — first-class held-out paired-delta promotion gate.
@@ -1,6 +1,6 @@
1
- import { T as TraceEmitter } from './emitter-DEZwY14K.js';
1
+ import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
2
2
  import { R as Run, F as FailureClass } from './schema-m0gsnbt3.js';
3
- import { T as TraceStore } from './store-CKUAgsJz.js';
3
+ import { T as TraceStore } from './store-BcFXE6LG.js';
4
4
 
5
5
  /**
6
6
  * SandboxHarness — executes a scenario in an isolated environment and
@@ -27,6 +27,12 @@ interface HarnessConfig {
27
27
  cwd?: string;
28
28
  /** Max wall-clock per phase in ms. Default 10 minutes. */
29
29
  timeoutMs?: number;
30
+ /**
31
+ * Cap on captured stdout+stderr bytes per phase. A runaway process can
32
+ * otherwise grow the in-memory buffer without bound. Once hit, further
33
+ * output is dropped and `outputTruncated` is set. Default 16 MiB.
34
+ */
35
+ maxOutputBytes?: number;
30
36
  /** Image for the docker driver. */
31
37
  image?: string;
32
38
  /** Extra env vars (validated; shell-escaped). */
@@ -49,6 +55,19 @@ interface SandboxResult {
49
55
  wallMs: number;
50
56
  testsTotal?: number;
51
57
  testsPassed?: number;
58
+ /**
59
+ * True when the process was killed because it exceeded `timeoutMs`. A
60
+ * SIGKILLed child can still close with exit code 0; callers MUST treat
61
+ * a timed-out phase as a hard failure regardless of `exitCode`, never
62
+ * as a pass. `undefined`/`false` means the process completed on its own.
63
+ */
64
+ killedByTimeout?: boolean;
65
+ /**
66
+ * True when captured stdout/stderr hit `maxOutputBytes` and further
67
+ * output was dropped. The result is still returned (the process was
68
+ * not killed for this), but downstream parsers see truncated text.
69
+ */
70
+ outputTruncated?: boolean;
52
71
  }
53
72
  interface SandboxDriver {
54
73
  id: string;
package/dist/traces.d.ts CHANGED
@@ -1,12 +1,12 @@
1
1
  import { N as NotFoundError, R as ReplayError } from './errors-CzMUYo7b.js';
2
2
  import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
3
3
  export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
4
- import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-DEZwY14K.js';
5
- export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-DEZwY14K.js';
6
- export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-VJ9A7aST.js';
7
- import { T as TraceStore } from './store-CKUAgsJz.js';
8
- export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-CKUAgsJz.js';
9
- export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CqTxMwDw.js';
4
+ import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
5
+ export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-C2rqGH_l.js';
6
+ export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-D2t12mMw.js';
7
+ import { T as TraceStore } from './store-BcFXE6LG.js';
8
+ export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BcFXE6LG.js';
9
+ export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-B7GGjRox.js';
10
10
  export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
11
11
  import { R as Run } from './schema-m0gsnbt3.js';
12
12
  export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
@@ -474,6 +474,14 @@ interface OtlpFileTraceStoreOptions {
474
474
  perCallByteCeiling?: number;
475
475
  /** Override the per-match text budget. */
476
476
  perMatchTextBudget?: number;
477
+ /**
478
+ * Hard ceiling on the trace file size in bytes. The store reads the
479
+ * whole file into one Buffer and indexes it in memory, so an
480
+ * unbounded file OOMs the process. Above this size the store fails
481
+ * loud with `TraceFileTooLargeError` instead of degrading silently.
482
+ * Default 256 MiB.
483
+ */
484
+ maxFileBytes?: number;
477
485
  }
478
486
  declare class OtlpFileTraceStore implements TraceAnalysisStore {
479
487
  private readonly path;
@@ -481,6 +489,7 @@ declare class OtlpFileTraceStore implements TraceAnalysisStore {
481
489
  private readonly perAttributeSpanBudget;
482
490
  private readonly perCallByteCeiling;
483
491
  private readonly perMatchTextBudget;
492
+ private readonly maxFileBytes;
484
493
  private indexPromise?;
485
494
  /** Cached UTF-8 buffer of the file. We pin it once because every
486
495
  * read needs slice access and re-reading on each call balloons the
@@ -518,6 +527,10 @@ declare class OtlpFileTraceStore implements TraceAnalysisStore {
518
527
  * before the first agent call. */
519
528
  ensureIndexed(): Promise<void>;
520
529
  private buffer;
530
+ /** Stat-then-read so an oversized file fails loud BEFORE we allocate a
531
+ * multi-hundred-MB Buffer and OOM the process. A missing file surfaces
532
+ * as TraceFileMissingError; any other stat/read error propagates. */
533
+ private readGuarded;
521
534
  private index;
522
535
  private buildIndex;
523
536
  private matchedTraces;