@tangle-network/agent-eval 0.100.0 → 0.100.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/dist/adapters/http.d.ts +2 -2
  2. package/dist/adapters/langchain.d.ts +2 -2
  3. package/dist/adapters/otel.d.ts +4 -4
  4. package/dist/analyst/index.d.ts +8 -7
  5. package/dist/analyst/index.js +35 -27
  6. package/dist/analyst/index.js.map +1 -1
  7. package/dist/{analyze-runs-DtT6F_6T.d.ts → analyze-runs-BlJRBniC.d.ts} +3 -3
  8. package/dist/belief-state/index.d.ts +3 -3
  9. package/dist/benchmarks/index.d.ts +2 -2
  10. package/dist/campaign/index.d.ts +37 -13
  11. package/dist/campaign/index.js +76 -7
  12. package/dist/campaign/index.js.map +1 -1
  13. package/dist/{chunk-WMBLMTUE.js → chunk-2KTBHICD.js} +139 -16
  14. package/dist/chunk-2KTBHICD.js.map +1 -0
  15. package/dist/{chunk-IZCEK2HR.js → chunk-2MLIEQSN.js} +3 -2
  16. package/dist/{chunk-IZCEK2HR.js.map → chunk-2MLIEQSN.js.map} +1 -1
  17. package/dist/{chunk-OKQ2LAT7.js → chunk-4LWD6GC7.js} +7 -5
  18. package/dist/{chunk-OKQ2LAT7.js.map → chunk-4LWD6GC7.js.map} +1 -1
  19. package/dist/{chunk-LO6IOIJ2.js → chunk-ABOIVNXL.js} +2 -240
  20. package/dist/chunk-ABOIVNXL.js.map +1 -0
  21. package/dist/{chunk-NZEQVRH5.js → chunk-BOETF6BU.js} +2 -2
  22. package/dist/{chunk-S4SYLDFX.js → chunk-FRI6RG3P.js} +3 -2
  23. package/dist/chunk-FRI6RG3P.js.map +1 -0
  24. package/dist/chunk-G6S73VA7.js +248 -0
  25. package/dist/chunk-G6S73VA7.js.map +1 -0
  26. package/dist/{chunk-GMGRBNVT.js → chunk-G7IB3GJ5.js} +2 -2
  27. package/dist/chunk-IN3SHQML.js +664 -0
  28. package/dist/chunk-IN3SHQML.js.map +1 -0
  29. package/dist/{chunk-SJISCGWD.js → chunk-JU6ZX3CX.js} +2 -2
  30. package/dist/{chunk-3NHEO6ZC.js → chunk-L5TVEZFT.js} +2 -2
  31. package/dist/{chunk-77T4STFI.js → chunk-PMF5WIBX.js} +3 -3
  32. package/dist/{chunk-OYU4D7FY.js → chunk-VWQ6PO5O.js} +2 -2
  33. package/dist/{code-agent-session-CPHRCb4-.d.ts → code-agent-session-B6ZcDwyA.d.ts} +1 -1
  34. package/dist/contract/index.d.ts +16 -16
  35. package/dist/contract/index.js +6 -5
  36. package/dist/contract/index.js.map +1 -1
  37. package/dist/{control-Doncu-B_.d.ts → control-DC8TELh0.d.ts} +1 -1
  38. package/dist/control.d.ts +2 -2
  39. package/dist/control.js +3 -2
  40. package/dist/{corpus-D4YW9UoJ.d.ts → corpus-ONOzGFmG.d.ts} +1 -1
  41. package/dist/{default-registry-GyE8X5SP.d.ts → default-registry-Dhrc__SE.d.ts} +2 -2
  42. package/dist/diagnose.d.ts +3 -3
  43. package/dist/diagnose.js +2 -1
  44. package/dist/diagnose.js.map +1 -1
  45. package/dist/{gepa-H6mlM0KN.d.ts → gepa-BRgNnmGZ.d.ts} +1 -1
  46. package/dist/hosted/index.d.ts +4 -4
  47. package/dist/{index-_Y4oNOOb.d.ts → index-W96macmS.d.ts} +1 -1
  48. package/dist/index.d.ts +68 -22
  49. package/dist/index.js +113 -43
  50. package/dist/index.js.map +1 -1
  51. package/dist/{insight-report-BnRjTibG.d.ts → insight-report-C02J3q4T.d.ts} +1 -1
  52. package/dist/{kind-factory-X3eDYbKn.d.ts → kind-factory-OgqQSvLi.d.ts} +1 -1
  53. package/dist/meta-eval/index.d.ts +2 -2
  54. package/dist/multishot/index.d.ts +2 -2
  55. package/dist/openapi.json +1 -1
  56. package/dist/policy-edit-Dccm9tyA.d.ts +103 -0
  57. package/dist/{pre-registration-CMm8cvrh.d.ts → pre-registration-DB8oDqZJ.d.ts} +3 -3
  58. package/dist/{provenance-Bg_RttR8.d.ts → provenance-B0SZw1z2.d.ts} +3 -3
  59. package/dist/{release-report-pidWUMZ2.d.ts → release-report-B1tA6pKu.d.ts} +2 -2
  60. package/dist/reporting.d.ts +4 -4
  61. package/dist/{researcher-Jr8ME1dZ.d.ts → researcher-Ba2y1Foi.d.ts} +2 -2
  62. package/dist/rl.d.ts +8 -8
  63. package/dist/rl.js +3 -2
  64. package/dist/rl.js.map +1 -1
  65. package/dist/{rubric-predictive-validity-C2hDKM8Z.d.ts → rubric-predictive-validity-w7tun-q3.d.ts} +1 -1
  66. package/dist/{run-record-CP2ObebC.d.ts → run-record-DEwidcqn.d.ts} +1 -1
  67. package/dist/{runtime-trajectory-BOUUjI0y.d.ts → runtime-trajectory-OJDaTYHN.d.ts} +1 -1
  68. package/dist/{semantic-concept-judge-DSBB2Cfp.d.ts → semantic-concept-judge-J8xvjdc3.d.ts} +2 -2
  69. package/dist/{summary-report-CInXwsza.d.ts → summary-report-C4uzRWh8.d.ts} +1 -1
  70. package/dist/traces.d.ts +1 -1
  71. package/dist/traces.js +4 -3
  72. package/dist/{types-B5x54y6n.d.ts → types-BEzCBMQD.d.ts} +2 -2
  73. package/dist/{types-BTI16iFl.d.ts → types-Cv1bo4_a.d.ts} +1 -1
  74. package/dist/workflow/index.d.ts +4 -4
  75. package/dist/workflow/index.js +2 -1
  76. package/dist/workflow/index.js.map +1 -1
  77. package/package.json +1 -1
  78. package/dist/chunk-BUTW4RGG.js +0 -32
  79. package/dist/chunk-BUTW4RGG.js.map +0 -1
  80. package/dist/chunk-LO6IOIJ2.js.map +0 -1
  81. package/dist/chunk-S4SYLDFX.js.map +0 -1
  82. package/dist/chunk-WMBLMTUE.js.map +0 -1
  83. /package/dist/{chunk-NZEQVRH5.js.map → chunk-BOETF6BU.js.map} +0 -0
  84. /package/dist/{chunk-GMGRBNVT.js.map → chunk-G7IB3GJ5.js.map} +0 -0
  85. /package/dist/{chunk-SJISCGWD.js.map → chunk-JU6ZX3CX.js.map} +0 -0
  86. /package/dist/{chunk-3NHEO6ZC.js.map → chunk-L5TVEZFT.js.map} +0 -0
  87. /package/dist/{chunk-77T4STFI.js.map → chunk-PMF5WIBX.js.map} +0 -0
  88. /package/dist/{chunk-OYU4D7FY.js.map → chunk-VWQ6PO5O.js.map} +0 -0
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/diagnose/causal-sweep.ts","../src/diagnose/remediation.ts","../src/diagnose/repair.ts"],"sourcesContent":["/**\n * Causal sweep — WHY did this run fail?\n *\n * Orchestrates the dormant counterfactual primitives into a responsibility\n * report: for each candidate step, run `reps` counterfactual replays per\n * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`\n * is the execution seam) and reduce the per-rep score deltas into a mean\n * effect + bootstrap confidence interval (via `confidenceInterval`).\n *\n * Why `reps` is REQUIRED: a single intervention delta is one stochastic\n * draw — LLM re-execution from a prefix is sampled, so one replay cannot\n * distinguish \"this step caused the failure\" from sampling noise. The\n * signal is the distribution of deltas across reps; the CI over that\n * distribution is what lets a caller say \"this step's effect excludes\n * zero\" instead of eyeballing a point estimate.\n *\n * Budget discipline: the sweep never silently drops cells. When the\n * remaining budget cannot fund a full `reps`-sized cell, the sweep halts\n * and every step not fully probed is named in `uncovered`.\n */\n\nimport {\n attributeCounterfactuals,\n type CounterfactualMutation,\n type CounterfactualResult,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport { confidenceInterval } from '../statistics'\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\n/** Stable reference to a trajectory step — carried through reports,\n * findings, and corpus records so evidence stays addressable. */\nexport interface StepRef {\n index: number\n spanId: string\n kind: Span['kind']\n name: string\n}\n\nexport function stepRefOf(step: TrajectoryStep): StepRef {\n return {\n index: step.index,\n spanId: step.span.spanId,\n kind: step.span.kind,\n name: step.span.name,\n }\n}\n\nexport interface CausalSweepOptions {\n store: TraceStore\n /** The failed run to diagnose. Its `outcome.score` is the baseline every\n * counterfactual delta is measured against. */\n runId: string\n /** Execution seam — identical contract to `runCounterfactual`: re-runs the\n * agent from the mutation point and MUST `endRun` with a numeric score. */\n runner: CounterfactualRunner\n /** Trajectory indices to probe. Default: every llm + tool span — the kinds\n * the existing `CounterfactualMutation` set targets. */\n candidateSteps?: number[]\n /**\n * Mutations to probe a given step with. Returned mutations MUST target\n * `step.index`. Default probes are the payload-free existing kinds:\n * - tool span → `swap-tool-result` with `newResult: null` (knockout:\n * how much did the run depend on this tool's information?)\n * - llm span → `truncate-after` (re-roll: how much did the realized\n * turn deviate from the policy's typical continuation?)\n * `swap-model` / `inject-system-message` need consumer payloads, so they\n * are opt-in via this callback.\n */\n mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[]\n /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */\n reps: number\n /** Hard cap on total counterfactual replays across the whole sweep. */\n budget: number\n /** Seed for the bootstrap CI resampler. Deterministic default so two\n * sweeps over the same deltas report identical intervals. */\n ciSeed?: number\n /** Bootstrap CI confidence level. Default 0.95. */\n ciConfidence?: number\n}\n\nexport interface StepResponsibility {\n stepRef: StepRef\n mutationKind: CounterfactualMutation['kind']\n /** Mean of per-rep score deltas (counterfactual − original). */\n meanEffect: number\n /** Bootstrap CI over the per-rep deltas. */\n ci: { mean: number; lower: number; upper: number }\n /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */\n ciExcludesZero: boolean\n reps: number\n /** Raw per-rep deltas — downstream evidence, never re-derived. */\n deltas: number[]\n /** Replay run ids (layer='meta', parentRunId=original) for audit. */\n counterfactualRunIds: string[]\n}\n\nexport interface CausalResponsibilityReport {\n runId: string\n originalScore: number\n /** Ranked by |meanEffect| descending — the blame ordering. */\n steps: StepResponsibility[]\n /** Kind-level aggregate from the existing `attributeCounterfactuals`. */\n byMutationKind: ReturnType<typeof attributeCounterfactuals>\n replaysUsed: number\n budget: number\n /** Steps planned but not fully probed before the budget ran out.\n * Named, never silent: an absent step is \"no effect found\"; an\n * uncovered step is \"not measured\". */\n uncovered: StepRef[]\n}\n\nconst DEFAULT_CI_SEED = 0x5eed\n\nfunction defaultMutations(step: TrajectoryStep): CounterfactualMutation[] {\n if (step.span.kind === 'tool') {\n return [{ kind: 'swap-tool-result', at: step.index, newResult: null }]\n }\n if (step.span.kind === 'llm') {\n return [{ kind: 'truncate-after', at: step.index }]\n }\n return []\n}\n\nexport async function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport> {\n if (!Number.isInteger(opts.reps) || opts.reps < 2) {\n throw new ValidationError(\n `causalSweep: reps must be an integer >= 2 (got ${opts.reps}) — a single-intervention delta is one stochastic draw, not a measurement`,\n )\n }\n if (!Number.isInteger(opts.budget) || opts.budget < 1) {\n throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`)\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `causalSweep: run ${opts.runId} has no numeric outcome.score — deltas have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n const candidates = resolveCandidates(trajectory, opts.candidateSteps)\n const mutationsFor = opts.mutationsPerStep ?? defaultMutations\n\n interface Cell {\n step: TrajectoryStep\n mutation: CounterfactualMutation\n }\n const cells: Cell[] = []\n for (const step of candidates) {\n const mutations = mutationsFor(step)\n for (const m of mutations) {\n if (m.at !== step.index) {\n throw new ValidationError(\n `causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} — mutations must target the step they were asked for`,\n )\n }\n cells.push({ step, mutation: m })\n }\n }\n\n const responsibilities: StepResponsibility[] = []\n const allResults: CounterfactualResult[] = []\n const uncoveredIndices = new Set<number>()\n let replaysUsed = 0\n let halted = false\n\n for (const cell of cells) {\n if (halted || replaysUsed + opts.reps > opts.budget) {\n // A partial cell would report a CI over fewer reps than requested —\n // weaker evidence masquerading as the real thing. Halt and name it.\n halted = true\n uncoveredIndices.add(cell.step.index)\n continue\n }\n const deltas: number[] = []\n const cfRunIds: string[] = []\n for (let rep = 0; rep < opts.reps; rep++) {\n const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner)\n replaysUsed++\n const d = result.delta.deltaScore\n if (typeof d !== 'number' || !Number.isFinite(d)) {\n throw new ValidationError(\n `causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`,\n )\n }\n deltas.push(d)\n cfRunIds.push(result.counterfactualRunId)\n allResults.push(result)\n }\n const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {\n seed: opts.ciSeed ?? DEFAULT_CI_SEED,\n })\n responsibilities.push({\n stepRef: stepRefOf(cell.step),\n mutationKind: cell.mutation.kind,\n meanEffect: ci.mean,\n ci,\n ciExcludesZero: ci.lower > 0 || ci.upper < 0,\n reps: opts.reps,\n deltas,\n counterfactualRunIds: cfRunIds,\n })\n }\n\n responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect))\n\n // A step probed under one mutation but cut off under another appears in\n // BOTH steps and uncovered — partial coverage is named, not blended.\n const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf)\n\n return {\n runId: opts.runId,\n originalScore,\n steps: responsibilities,\n byMutationKind: attributeCounterfactuals(allResults),\n replaysUsed,\n budget: opts.budget,\n uncovered,\n }\n}\n\nfunction resolveCandidates(trajectory: Trajectory, indices?: number[]): TrajectoryStep[] {\n if (indices === undefined) {\n return trajectory.steps.filter((s) => s.span.kind === 'llm' || s.span.kind === 'tool')\n }\n return indices.map((i) => {\n const step = trajectory.steps[i]\n if (!step) {\n throw new ValidationError(\n `causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`,\n )\n }\n return step\n })\n}\n","/**\n * Remediation adapters — HOW DO WE MAKE IT HAPPEN?\n *\n * The diagnose chain ends by feeding existing improvement machinery,\n * not by building new machinery:\n *\n * - `toAnalystFindings` → the analyst contract (`makeFinding`), so\n * responsibility evidence flows into the same registry / steering /\n * diff pipeline every other analyst feeds.\n * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the\n * diagnosed failure + validated repair as a permanent scenario.\n * - `suggestInvariant` → a plain-data hint in the shape the\n * trace-contracts machinery consumes (`never` / `without` clauses).\n */\n\nimport type { AnalystFinding, AnalystSeverity, EvidenceRef } from '../analyst/types'\nimport { makeFinding } from '../analyst/types'\nimport type { CounterfactualMutation } from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { CorpusRecord } from '../rl/corpus'\nimport type { RunRecord } from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CausalResponsibilityReport, StepResponsibility } from './causal-sweep'\nimport type { RepairReport, ValidatedRepair } from './repair'\n\nexport const DIAGNOSE_ANALYST_ID = 'diagnose-causal-sweep'\n\n/** Severity from causal effect size. Effects whose CI includes zero are\n * 'info' regardless of magnitude — an indistinguishable-from-noise effect\n * must not steer remediation priority. */\nexport function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity {\n if (!responsibility.ciExcludesZero) return 'info'\n const magnitude = Math.abs(responsibility.meanEffect)\n if (magnitude >= 0.5) return 'critical'\n if (magnitude >= 0.25) return 'high'\n if (magnitude >= 0.1) return 'medium'\n return 'low'\n}\n\n/** Deterministic human-readable rendering of a mutation — used in\n * recommended actions, corpus completions, and invariant hints. */\nexport function describeMutation(mutation: CounterfactualMutation): string {\n switch (mutation.kind) {\n case 'swap-model':\n return `use model '${mutation.newModel}' at step ${mutation.at}`\n case 'swap-tool-result':\n return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`\n case 'truncate-after':\n return `stop the run after step ${mutation.at}`\n case 'inject-system-message':\n return `inject system message at step ${mutation.at}: ${mutation.content}`\n case 'custom':\n return `${mutation.describe} (step ${mutation.at})`\n }\n}\n\n/**\n * Lift a responsibility report (and optionally its validated repairs) into\n * `AnalystFinding`s via the real `makeFinding` factory. One finding per\n * probed step; a validated repair for that step upgrades the finding with\n * a `recommended_action` + the replay-validation evidence.\n *\n * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,\n * so `derived_from_judge` stays unset and they may steer.\n */\nexport function toAnalystFindings(\n report: CausalResponsibilityReport,\n repairs?: RepairReport,\n): AnalystFinding[] {\n const repairByStep = new Map<string, ValidatedRepair>()\n for (const r of repairs?.repairs ?? []) {\n if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r)\n }\n\n return report.steps.map((resp) => {\n const repair = repairByStep.get(resp.stepRef.spanId)\n const evidence: EvidenceRef[] = [\n {\n kind: 'span',\n uri: `span://${resp.stepRef.spanId}`,\n excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`,\n },\n {\n kind: 'metric',\n uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,\n excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(', ')}]`,\n },\n ...resp.counterfactualRunIds.map((id): EvidenceRef => ({ kind: 'span', uri: `run://${id}` })),\n ]\n return makeFinding({\n analyst_id: DIAGNOSE_ANALYST_ID,\n severity: severityFromEffect(resp),\n area: 'causal-attribution',\n claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,\n rationale: resp.ciExcludesZero\n ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero`\n : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero — not distinguishable from noise`,\n evidence_refs: evidence,\n recommended_action: repair ? describeMutation(repair.mutation) : undefined,\n validation_plan: repair\n ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs!.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})`\n : undefined,\n confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,\n subject: resp.stepRef.spanId,\n metadata: {\n stepRef: resp.stepRef,\n mutationKind: resp.mutationKind,\n meanEffect: resp.meanEffect,\n ci: resp.ci,\n deltas: resp.deltas,\n counterfactualRunIds: resp.counterfactualRunIds,\n ...(repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}),\n },\n })\n })\n}\n\n/**\n * Pin the diagnosed failure as a permanent corpus scenario. Takes the\n * original run's `RunRecord` projection plus a validated repair and emits\n * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw\n * failure and the diagnosed entry).\n *\n * `completion` defaults to the validated mutation's rendering — \"what\n * should have happened\" in machine-derived form. Supply `prompt` (and\n * optionally a richer `completion`) when the trajectory text is available\n * so the record is harvestable by `buildDatasetFromCorpus`.\n */\nexport function toCorpusRecord(\n run: RunRecord,\n repair: ValidatedRepair,\n opts: { prompt?: string; completion?: string } = {},\n): CorpusRecord {\n const record: CorpusRecord = {\n ...run,\n runId: `${run.runId}#repair:${repair.stepRef.spanId}`,\n outcome: {\n ...run.outcome,\n raw: {\n ...run.outcome.raw,\n diagnose_blamed_step_index: repair.stepRef.index,\n diagnose_repair_mean_score: repair.meanScore,\n diagnose_repair_delta_score: repair.deltaScore,\n diagnose_repair_reps: repair.reps,\n },\n },\n prompt: opts.prompt,\n completion: opts.completion ?? describeMutation(repair.mutation),\n }\n // Boundary check — a corpus record that fails RunRecord validation would\n // poison every downstream harvest.\n validateRunRecord(record)\n return record\n}\n\n/** Plain-data invariant hint. The trace-contracts machinery consumes this\n * shape: `never` is a pattern that must not appear in a passing trace;\n * `without` is a guard whose absence makes the failure reachable. */\nexport interface InvariantHint {\n description: string\n never?: string\n without?: string\n}\n\n/**\n * Derive an invariant hint from a validated repair. Deterministic per\n * mutation kind — the hint names the contract a trace must satisfy so\n * the diagnosed failure cannot silently recur.\n */\nexport function suggestInvariant(repair: ValidatedRepair): InvariantHint {\n const { stepRef, mutation } = repair\n const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`\n switch (mutation.kind) {\n case 'swap-tool-result':\n return {\n description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `unvalidated result from tool '${stepRef.name}' flows downstream`,\n without: `result guard on tool '${stepRef.name}'`,\n }\n case 'swap-model':\n return {\n description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`,\n }\n case 'inject-system-message':\n return {\n description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n without: `system message present at '${stepRef.name}': ${mutation.content}`,\n }\n case 'truncate-after':\n return {\n description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) — continuation past this step caused the failure`,\n never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`,\n }\n case 'custom':\n return {\n description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n }\n default: {\n const exhausted: never = mutation\n throw new ValidationError(\n `suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`,\n )\n }\n }\n}\n","/**\n * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?\n *\n * Takes the blamed steps from a `CausalResponsibilityReport`, asks a\n * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate\n * mutations, and machine-verifies each candidate by replaying the run\n * WITH the mutation applied (through the same `runCounterfactual` seam\n * the sweep uses).\n *\n * A repair is \"what should have happened\" ONLY when every validation\n * replay crosses `flipThreshold` — a prescription is never speculated,\n * it is demonstrated. Candidates that don't flip, or whose replay\n * errors, land in `rejected` with a typed reason; nothing is dropped\n * silently.\n */\n\nimport {\n type CounterfactualMutation,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\nimport type { StepRef, StepResponsibility } from './causal-sweep'\n\n/** Context handed to `proposeFix` so an LLM-backed proposer can see the\n * full trajectory plus the responsibility evidence for the blamed step. */\nexport interface RepairContext {\n runId: string\n trajectory: Trajectory\n originalScore: number\n responsibility: StepResponsibility\n}\n\nexport interface PrescribeRepairOptions {\n store: TraceStore\n /** The failed run the sweep diagnosed. */\n runId: string\n /** Execution seam — same `CounterfactualRunner` contract as the sweep. */\n runner: CounterfactualRunner\n /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */\n blamed: StepResponsibility[]\n /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.\n * Returned mutations MUST target the blamed step's index. */\n proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>\n /** Score every validation replay must reach for the repair to count. Default 0.5. */\n flipThreshold?: number\n /** Validation replays per candidate mutation. Default 3. */\n repsToValidate?: number\n /** Max candidate mutations tried per step. Default: all proposed. */\n maxAttemptsPerStep?: number\n}\n\nexport interface ValidatedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n /** Always true — presence in `repairs` IS the machine-verified claim. */\n validated: true\n /** Mean counterfactual score across the validation reps. */\n meanScore: number\n /** meanScore − originalScore. */\n deltaScore: number\n reps: number\n /** Replay run ids backing the validation — audit trail. */\n counterfactualRunIds: string[]\n}\n\nexport interface RejectedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n reason: 'did-not-flip' | 'error'\n /** Present for 'did-not-flip': mean delta over the reps that ran. */\n deltaScore?: number\n /** Present for 'error': the message, preserved for diagnosis. */\n error?: string\n}\n\nexport interface RepairReport {\n runId: string\n originalScore: number\n flipThreshold: number\n repairs: ValidatedRepair[]\n rejected: RejectedRepair[]\n replaysUsed: number\n}\n\nexport async function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport> {\n const flipThreshold = opts.flipThreshold ?? 0.5\n const repsToValidate = opts.repsToValidate ?? 3\n if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {\n throw new ValidationError(\n `prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`,\n )\n }\n const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY\n if (maxAttempts < 1) {\n throw new ValidationError(\n `prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`,\n )\n }\n if (opts.blamed.length === 0) {\n throw new ValidationError('prescribeRepair: blamed is empty — nothing to repair')\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `prescribeRepair: run ${opts.runId} has no numeric outcome.score — flips have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n\n const repairs: ValidatedRepair[] = []\n const rejected: RejectedRepair[] = []\n let replaysUsed = 0\n\n for (const responsibility of opts.blamed) {\n const step = trajectory.steps[responsibility.stepRef.index]\n if (!step || step.span.spanId !== responsibility.stepRef.spanId) {\n throw new ValidationError(\n `prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} — stale report?`,\n )\n }\n\n const candidates = await opts.proposeFix(step, {\n runId: opts.runId,\n trajectory,\n originalScore,\n responsibility,\n })\n const toTry = candidates.slice(0, maxAttempts)\n\n for (const mutation of toTry) {\n if (mutation.at !== step.index) {\n throw new ValidationError(\n `prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`,\n )\n }\n const scores: number[] = []\n const cfRunIds: string[] = []\n let failure: string | undefined\n for (let rep = 0; rep < repsToValidate; rep++) {\n try {\n const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner)\n replaysUsed++\n const score = result.delta.counterfactualOutcomeScore\n if (typeof score !== 'number' || !Number.isFinite(score)) {\n failure = `validation rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`\n break\n }\n scores.push(score)\n cfRunIds.push(result.counterfactualRunId)\n } catch (err) {\n replaysUsed++\n failure = err instanceof Error ? err.message : String(err)\n break\n }\n }\n\n if (failure !== undefined) {\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'error',\n error: failure,\n })\n continue\n }\n\n const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length\n const everyRepFlipped = scores.every((s) => s >= flipThreshold)\n if (everyRepFlipped) {\n repairs.push({\n stepRef: responsibility.stepRef,\n mutation,\n validated: true,\n meanScore,\n deltaScore: meanScore - originalScore,\n reps: repsToValidate,\n counterfactualRunIds: cfRunIds,\n })\n // First validated repair per step IS the prescription; remaining\n // candidates are untried, not rejected — we don't fabricate verdicts.\n break\n }\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'did-not-flip',\n deltaScore: meanScore - originalScore,\n })\n }\n }\n\n return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;AA2CO,SAAS,UAAU,MAA+B;AACvD,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK,KAAK;AAAA,IAClB,MAAM,KAAK,KAAK;AAAA,IAChB,MAAM,KAAK,KAAK;AAAA,EAClB;AACF;AAkEA,IAAM,kBAAkB;AAExB,SAAS,iBAAiB,MAAgD;AACxE,MAAI,KAAK,KAAK,SAAS,QAAQ;AAC7B,WAAO,CAAC,EAAE,MAAM,oBAAoB,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC;AAAA,EACvE;AACA,MAAI,KAAK,KAAK,SAAS,OAAO;AAC5B,WAAO,CAAC,EAAE,MAAM,kBAAkB,IAAI,KAAK,MAAM,CAAC;AAAA,EACpD;AACA,SAAO,CAAC;AACV;AAEA,eAAsB,YAAY,MAA+D;AAC/F,MAAI,CAAC,OAAO,UAAU,KAAK,IAAI,KAAK,KAAK,OAAO,GAAG;AACjD,UAAM,IAAI;AAAA,MACR,kDAAkD,KAAK,IAAI;AAAA,IAC7D;AAAA,EACF;AACA,MAAI,CAAC,OAAO,UAAU,KAAK,MAAM,KAAK,KAAK,SAAS,GAAG;AACrD,UAAM,IAAI,gBAAgB,oDAAoD,KAAK,MAAM,GAAG;AAAA,EAC9F;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,oBAAoB,KAAK,KAAK,YAAY;AACtF,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,oBAAoB,KAAK,KAAK;AAAA,IAChC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAC/D,QAAM,aAAa,kBAAkB,YAAY,KAAK,cAAc;AACpE,QAAM,eAAe,KAAK,oBAAoB;AAM9C,QAAM,QAAgB,CAAC;AACvB,aAAW,QAAQ,YAAY;AAC7B,UAAM,YAAY,aAAa,IAAI;AACnC,eAAW,KAAK,WAAW;AACzB,UAAI,EAAE,OAAO,KAAK,OAAO;AACvB,cAAM,IAAI;AAAA,UACR,kEAAkE,EAAE,EAAE,mBAAmB,KAAK,KAAK;AAAA,QACrG;AAAA,MACF;AACA,YAAM,KAAK,EAAE,MAAM,UAAU,EAAE,CAAC;AAAA,IAClC;AAAA,EACF;AAEA,QAAM,mBAAyC,CAAC;AAChD,QAAM,aAAqC,CAAC;AAC5C,QAAM,mBAAmB,oBAAI,IAAY;AACzC,MAAI,cAAc;AAClB,MAAI,SAAS;AAEb,aAAW,QAAQ,OAAO;AACxB,QAAI,UAAU,cAAc,KAAK,OAAO,KAAK,QAAQ;AAGnD,eAAS;AACT,uBAAiB,IAAI,KAAK,KAAK,KAAK;AACpC;AAAA,IACF;AACA,UAAM,SAAmB,CAAC;AAC1B,UAAM,WAAqB,CAAC;AAC5B,aAAS,MAAM,GAAG,MAAM,KAAK,MAAM,OAAO;AACxC,YAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,KAAK,UAAU,KAAK,MAAM;AACzF;AACA,YAAM,IAAI,OAAO,MAAM;AACvB,UAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;AAChD,cAAM,IAAI;AAAA,UACR,+CAA+C,KAAK,KAAK,KAAK,KAAK,KAAK,SAAS,IAAI,SAAS,GAAG;AAAA,QACnG;AAAA,MACF;AACA,aAAO,KAAK,CAAC;AACb,eAAS,KAAK,OAAO,mBAAmB;AACxC,iBAAW,KAAK,MAAM;AAAA,IACxB;AACA,UAAM,KAAK,mBAAmB,QAAQ,KAAK,gBAAgB,MAAM;AAAA,MAC/D,MAAM,KAAK,UAAU;AAAA,IACvB,CAAC;AACD,qBAAiB,KAAK;AAAA,MACpB,SAAS,UAAU,KAAK,IAAI;AAAA,MAC5B,cAAc,KAAK,SAAS;AAAA,MAC5B,YAAY,GAAG;AAAA,MACf;AAAA,MACA,gBAAgB,GAAG,QAAQ,KAAK,GAAG,QAAQ;AAAA,MAC3C,MAAM,KAAK;AAAA,MACX;AAAA,MACA,sBAAsB;AAAA,IACxB,CAAC;AAAA,EACH;AAEA,mBAAiB,KAAK,CAAC,GAAG,MAAM,KAAK,IAAI,EAAE,UAAU,IAAI,KAAK,IAAI,EAAE,UAAU,CAAC;AAI/E,QAAM,YAAY,WAAW,OAAO,CAAC,MAAM,iBAAiB,IAAI,EAAE,KAAK,CAAC,EAAE,IAAI,SAAS;AAEvF,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ;AAAA,IACA,OAAO;AAAA,IACP,gBAAgB,yBAAyB,UAAU;AAAA,IACnD;AAAA,IACA,QAAQ,KAAK;AAAA,IACb;AAAA,EACF;AACF;AAEA,SAAS,kBAAkB,YAAwB,SAAsC;AACvF,MAAI,YAAY,QAAW;AACzB,WAAO,WAAW,MAAM,OAAO,CAAC,MAAM,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,MAAM;AAAA,EACvF;AACA,SAAO,QAAQ,IAAI,CAAC,MAAM;AACxB,UAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,QAAI,CAAC,MAAM;AACT,YAAM,IAAI;AAAA,QACR,qCAAqC,CAAC,qBAAqB,WAAW,MAAM,MAAM;AAAA,MACpF;AAAA,IACF;AACA,WAAO;AAAA,EACT,CAAC;AACH;;;ACzNO,IAAM,sBAAsB;AAK5B,SAAS,mBAAmB,gBAAqD;AACtF,MAAI,CAAC,eAAe,eAAgB,QAAO;AAC3C,QAAM,YAAY,KAAK,IAAI,eAAe,UAAU;AACpD,MAAI,aAAa,IAAK,QAAO;AAC7B,MAAI,aAAa,KAAM,QAAO;AAC9B,MAAI,aAAa,IAAK,QAAO;AAC7B,SAAO;AACT;AAIO,SAAS,iBAAiB,UAA0C;AACzE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO,cAAc,SAAS,QAAQ,aAAa,SAAS,EAAE;AAAA,IAChE,KAAK;AACH,aAAO,mCAAmC,SAAS,EAAE,SAAS,KAAK,UAAU,SAAS,SAAS,CAAC;AAAA,IAClG,KAAK;AACH,aAAO,2BAA2B,SAAS,EAAE;AAAA,IAC/C,KAAK;AACH,aAAO,iCAAiC,SAAS,EAAE,KAAK,SAAS,OAAO;AAAA,IAC1E,KAAK;AACH,aAAO,GAAG,SAAS,QAAQ,UAAU,SAAS,EAAE;AAAA,EACpD;AACF;AAWO,SAAS,kBACd,QACA,SACkB;AAClB,QAAM,eAAe,oBAAI,IAA6B;AACtD,aAAW,KAAK,SAAS,WAAW,CAAC,GAAG;AACtC,QAAI,CAAC,aAAa,IAAI,EAAE,QAAQ,MAAM,EAAG,cAAa,IAAI,EAAE,QAAQ,QAAQ,CAAC;AAAA,EAC/E;AAEA,SAAO,OAAO,MAAM,IAAI,CAAC,SAAS;AAChC,UAAM,SAAS,aAAa,IAAI,KAAK,QAAQ,MAAM;AACnD,UAAM,WAA0B;AAAA,MAC9B;AAAA,QACE,MAAM;AAAA,QACN,KAAK,UAAU,KAAK,QAAQ,MAAM;AAAA,QAClC,SAAS,QAAQ,KAAK,QAAQ,KAAK,KAAK,KAAK,QAAQ,IAAI,KAAK,KAAK,QAAQ,IAAI,iBAAiB,KAAK,WAAW,QAAQ,CAAC,CAAC,QAAQ,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,UAAU,KAAK,IAAI;AAAA,MAC5M;AAAA,MACA;AAAA,QACE,MAAM;AAAA,QACN,KAAK,qBAAqB,OAAO,KAAK,SAAS,KAAK,QAAQ,KAAK,IAAI,KAAK,YAAY;AAAA,QACtF,SAAS,WAAW,KAAK,OAAO,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC,EAAE,KAAK,IAAI,CAAC;AAAA,MACrE;AAAA,MACA,GAAG,KAAK,qBAAqB,IAAI,CAAC,QAAqB,EAAE,MAAM,QAAQ,KAAK,SAAS,EAAE,GAAG,EAAE;AAAA,IAC9F;AACA,WAAO,YAAY;AAAA,MACjB,YAAY;AAAA,MACZ,UAAU,mBAAmB,IAAI;AAAA,MACjC,MAAM;AAAA,MACN,OAAO,SAAS,KAAK,QAAQ,IAAI,MAAM,KAAK,QAAQ,IAAI,uDAAuD,KAAK,YAAY;AAAA,MAChI,WAAW,KAAK,iBACZ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,oBAChJ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC;AAAA,MACpJ,eAAe;AAAA,MACf,oBAAoB,SAAS,iBAAiB,OAAO,QAAQ,IAAI;AAAA,MACjE,iBAAiB,SACb,qBAAqB,OAAO,IAAI,IAAI,OAAO,IAAI,mBAAmB,QAAS,aAAa,UAAU,OAAO,UAAU,QAAQ,CAAC,CAAC,WAAW,OAAO,WAAW,QAAQ,CAAC,CAAC,MACpK;AAAA,MACJ,YAAY,SAAS,OAAO,KAAK,iBAAiB,OAAO;AAAA,MACzD,SAAS,KAAK,QAAQ;AAAA,MACtB,UAAU;AAAA,QACR,SAAS,KAAK;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,YAAY,KAAK;AAAA,QACjB,IAAI,KAAK;AAAA,QACT,QAAQ,KAAK;AAAA,QACb,sBAAsB,KAAK;AAAA,QAC3B,GAAI,SAAS,EAAE,QAAQ,EAAE,UAAU,OAAO,UAAU,WAAW,OAAO,UAAU,EAAE,IAAI,CAAC;AAAA,MACzF;AAAA,IACF,CAAC;AAAA,EACH,CAAC;AACH;AAaO,SAAS,eACd,KACA,QACA,OAAiD,CAAC,GACpC;AACd,QAAM,SAAuB;AAAA,IAC3B,GAAG;AAAA,IACH,OAAO,GAAG,IAAI,KAAK,WAAW,OAAO,QAAQ,MAAM;AAAA,IACnD,SAAS;AAAA,MACP,GAAG,IAAI;AAAA,MACP,KAAK;AAAA,QACH,GAAG,IAAI,QAAQ;AAAA,QACf,4BAA4B,OAAO,QAAQ;AAAA,QAC3C,4BAA4B,OAAO;AAAA,QACnC,6BAA6B,OAAO;AAAA,QACpC,sBAAsB,OAAO;AAAA,MAC/B;AAAA,IACF;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,YAAY,KAAK,cAAc,iBAAiB,OAAO,QAAQ;AAAA,EACjE;AAGA,oBAAkB,MAAM;AACxB,SAAO;AACT;AAgBO,SAAS,iBAAiB,QAAwC;AACvE,QAAM,EAAE,SAAS,SAAS,IAAI;AAC9B,QAAM,KAAK,QAAQ,QAAQ,KAAK,KAAK,QAAQ,IAAI,KAAK,QAAQ,IAAI;AAClE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO;AAAA,QACL,aAAa,uBAAuB,QAAQ,IAAI,4FAA4F,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QACxK,OAAO,iCAAiC,QAAQ,IAAI;AAAA,QACpD,SAAS,yBAAyB,QAAQ,IAAI;AAAA,MAChD;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,yBAAyB,EAAE,QAAQ,SAAS,QAAQ,gCAAgC,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC7H,OAAO,aAAa,QAAQ,IAAI,iCAAiC,SAAS,QAAQ;AAAA,MACpF;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,iCAAiC,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC3G,SAAS,8BAA8B,QAAQ,IAAI,MAAM,SAAS,OAAO;AAAA,MAC3E;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,kBAAkB,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC5F,OAAO,wBAAwB,QAAQ,IAAI,YAAY,QAAQ,KAAK;AAAA,MACtE;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,GAAG,SAAS,QAAQ,OAAO,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,MACvG;AAAA,IACF,SAAS;AACP,YAAM,YAAmB;AACzB,YAAM,IAAI;AAAA,QACR,2CAA2C,KAAK,UAAU,SAAS,CAAC;AAAA,MACtE;AAAA,IACF;AAAA,EACF;AACF;;;ACtHA,eAAsB,gBAAgB,MAAqD;AACzF,QAAM,gBAAgB,KAAK,iBAAiB;AAC5C,QAAM,iBAAiB,KAAK,kBAAkB;AAC9C,MAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GAAG;AAC3D,UAAM,IAAI;AAAA,MACR,gEAAgE,cAAc;AAAA,IAChF;AAAA,EACF;AACA,QAAM,cAAc,KAAK,sBAAsB,OAAO;AACtD,MAAI,cAAc,GAAG;AACnB,UAAM,IAAI;AAAA,MACR,yDAAyD,KAAK,kBAAkB;AAAA,IAClF;AAAA,EACF;AACA,MAAI,KAAK,OAAO,WAAW,GAAG;AAC5B,UAAM,IAAI,gBAAgB,2DAAsD;AAAA,EAClF;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,wBAAwB,KAAK,KAAK,YAAY;AAC1F,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,wBAAwB,KAAK,KAAK;AAAA,IACpC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAE/D,QAAM,UAA6B,CAAC;AACpC,QAAM,WAA6B,CAAC;AACpC,MAAI,cAAc;AAElB,aAAW,kBAAkB,KAAK,QAAQ;AACxC,UAAM,OAAO,WAAW,MAAM,eAAe,QAAQ,KAAK;AAC1D,QAAI,CAAC,QAAQ,KAAK,KAAK,WAAW,eAAe,QAAQ,QAAQ;AAC/D,YAAM,IAAI;AAAA,QACR,sCAAsC,eAAe,QAAQ,KAAK,WAAW,eAAe,QAAQ,MAAM,uBAAuB,KAAK,KAAK;AAAA,MAC7I;AAAA,IACF;AAEA,UAAM,aAAa,MAAM,KAAK,WAAW,MAAM;AAAA,MAC7C,OAAO,KAAK;AAAA,MACZ;AAAA,MACA;AAAA,MACA;AAAA,IACF,CAAC;AACD,UAAM,QAAQ,WAAW,MAAM,GAAG,WAAW;AAE7C,eAAW,YAAY,OAAO;AAC5B,UAAI,SAAS,OAAO,KAAK,OAAO;AAC9B,cAAM,IAAI;AAAA,UACR,gEAAgE,SAAS,EAAE,0BAA0B,KAAK,KAAK;AAAA,QACjH;AAAA,MACF;AACA,YAAM,SAAmB,CAAC;AAC1B,YAAM,WAAqB,CAAC;AAC5B,UAAI;AACJ,eAAS,MAAM,GAAG,MAAM,gBAAgB,OAAO;AAC7C,YAAI;AACF,gBAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,UAAU,KAAK,MAAM;AACpF;AACA,gBAAM,QAAQ,OAAO,MAAM;AAC3B,cAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;AACxD,sBAAU,kBAAkB,GAAG;AAC/B;AAAA,UACF;AACA,iBAAO,KAAK,KAAK;AACjB,mBAAS,KAAK,OAAO,mBAAmB;AAAA,QAC1C,SAAS,KAAK;AACZ;AACA,oBAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AACzD;AAAA,QACF;AAAA,MACF;AAEA,UAAI,YAAY,QAAW;AACzB,iBAAS,KAAK;AAAA,UACZ,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,QAAQ;AAAA,UACR,OAAO;AAAA,QACT,CAAC;AACD;AAAA,MACF;AAEA,YAAM,YAAY,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;AAC7D,YAAM,kBAAkB,OAAO,MAAM,CAAC,MAAM,KAAK,aAAa;AAC9D,UAAI,iBAAiB;AACnB,gBAAQ,KAAK;AAAA,UACX,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,WAAW;AAAA,UACX;AAAA,UACA,YAAY,YAAY;AAAA,UACxB,MAAM;AAAA,UACN,sBAAsB;AAAA,QACxB,CAAC;AAGD;AAAA,MACF;AACA,eAAS,KAAK;AAAA,QACZ,SAAS,eAAe;AAAA,QACxB;AAAA,QACA,QAAQ;AAAA,QACR,YAAY,YAAY;AAAA,MAC1B,CAAC;AAAA,IACH;AAAA,EACF;AAEA,SAAO,EAAE,OAAO,KAAK,OAAO,eAAe,eAAe,SAAS,UAAU,YAAY;AAC3F;","names":[]}
1
+ {"version":3,"sources":["../src/diagnose/causal-sweep.ts","../src/diagnose/remediation.ts","../src/diagnose/repair.ts"],"sourcesContent":["/**\n * Causal sweep — WHY did this run fail?\n *\n * Orchestrates the dormant counterfactual primitives into a responsibility\n * report: for each candidate step, run `reps` counterfactual replays per\n * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`\n * is the execution seam) and reduce the per-rep score deltas into a mean\n * effect + bootstrap confidence interval (via `confidenceInterval`).\n *\n * Why `reps` is REQUIRED: a single intervention delta is one stochastic\n * draw — LLM re-execution from a prefix is sampled, so one replay cannot\n * distinguish \"this step caused the failure\" from sampling noise. The\n * signal is the distribution of deltas across reps; the CI over that\n * distribution is what lets a caller say \"this step's effect excludes\n * zero\" instead of eyeballing a point estimate.\n *\n * Budget discipline: the sweep never silently drops cells. When the\n * remaining budget cannot fund a full `reps`-sized cell, the sweep halts\n * and every step not fully probed is named in `uncovered`.\n */\n\nimport {\n attributeCounterfactuals,\n type CounterfactualMutation,\n type CounterfactualResult,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport { confidenceInterval } from '../statistics'\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\n/** Stable reference to a trajectory step — carried through reports,\n * findings, and corpus records so evidence stays addressable. */\nexport interface StepRef {\n index: number\n spanId: string\n kind: Span['kind']\n name: string\n}\n\nexport function stepRefOf(step: TrajectoryStep): StepRef {\n return {\n index: step.index,\n spanId: step.span.spanId,\n kind: step.span.kind,\n name: step.span.name,\n }\n}\n\nexport interface CausalSweepOptions {\n store: TraceStore\n /** The failed run to diagnose. Its `outcome.score` is the baseline every\n * counterfactual delta is measured against. */\n runId: string\n /** Execution seam — identical contract to `runCounterfactual`: re-runs the\n * agent from the mutation point and MUST `endRun` with a numeric score. */\n runner: CounterfactualRunner\n /** Trajectory indices to probe. Default: every llm + tool span — the kinds\n * the existing `CounterfactualMutation` set targets. */\n candidateSteps?: number[]\n /**\n * Mutations to probe a given step with. Returned mutations MUST target\n * `step.index`. Default probes are the payload-free existing kinds:\n * - tool span → `swap-tool-result` with `newResult: null` (knockout:\n * how much did the run depend on this tool's information?)\n * - llm span → `truncate-after` (re-roll: how much did the realized\n * turn deviate from the policy's typical continuation?)\n * `swap-model` / `inject-system-message` need consumer payloads, so they\n * are opt-in via this callback.\n */\n mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[]\n /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */\n reps: number\n /** Hard cap on total counterfactual replays across the whole sweep. */\n budget: number\n /** Seed for the bootstrap CI resampler. Deterministic default so two\n * sweeps over the same deltas report identical intervals. */\n ciSeed?: number\n /** Bootstrap CI confidence level. Default 0.95. */\n ciConfidence?: number\n}\n\nexport interface StepResponsibility {\n stepRef: StepRef\n mutationKind: CounterfactualMutation['kind']\n /** Mean of per-rep score deltas (counterfactual − original). */\n meanEffect: number\n /** Bootstrap CI over the per-rep deltas. */\n ci: { mean: number; lower: number; upper: number }\n /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */\n ciExcludesZero: boolean\n reps: number\n /** Raw per-rep deltas — downstream evidence, never re-derived. */\n deltas: number[]\n /** Replay run ids (layer='meta', parentRunId=original) for audit. */\n counterfactualRunIds: string[]\n}\n\nexport interface CausalResponsibilityReport {\n runId: string\n originalScore: number\n /** Ranked by |meanEffect| descending — the blame ordering. */\n steps: StepResponsibility[]\n /** Kind-level aggregate from the existing `attributeCounterfactuals`. */\n byMutationKind: ReturnType<typeof attributeCounterfactuals>\n replaysUsed: number\n budget: number\n /** Steps planned but not fully probed before the budget ran out.\n * Named, never silent: an absent step is \"no effect found\"; an\n * uncovered step is \"not measured\". */\n uncovered: StepRef[]\n}\n\nconst DEFAULT_CI_SEED = 0x5eed\n\nfunction defaultMutations(step: TrajectoryStep): CounterfactualMutation[] {\n if (step.span.kind === 'tool') {\n return [{ kind: 'swap-tool-result', at: step.index, newResult: null }]\n }\n if (step.span.kind === 'llm') {\n return [{ kind: 'truncate-after', at: step.index }]\n }\n return []\n}\n\nexport async function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport> {\n if (!Number.isInteger(opts.reps) || opts.reps < 2) {\n throw new ValidationError(\n `causalSweep: reps must be an integer >= 2 (got ${opts.reps}) — a single-intervention delta is one stochastic draw, not a measurement`,\n )\n }\n if (!Number.isInteger(opts.budget) || opts.budget < 1) {\n throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`)\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `causalSweep: run ${opts.runId} has no numeric outcome.score — deltas have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n const candidates = resolveCandidates(trajectory, opts.candidateSteps)\n const mutationsFor = opts.mutationsPerStep ?? defaultMutations\n\n interface Cell {\n step: TrajectoryStep\n mutation: CounterfactualMutation\n }\n const cells: Cell[] = []\n for (const step of candidates) {\n const mutations = mutationsFor(step)\n for (const m of mutations) {\n if (m.at !== step.index) {\n throw new ValidationError(\n `causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} — mutations must target the step they were asked for`,\n )\n }\n cells.push({ step, mutation: m })\n }\n }\n\n const responsibilities: StepResponsibility[] = []\n const allResults: CounterfactualResult[] = []\n const uncoveredIndices = new Set<number>()\n let replaysUsed = 0\n let halted = false\n\n for (const cell of cells) {\n if (halted || replaysUsed + opts.reps > opts.budget) {\n // A partial cell would report a CI over fewer reps than requested —\n // weaker evidence masquerading as the real thing. Halt and name it.\n halted = true\n uncoveredIndices.add(cell.step.index)\n continue\n }\n const deltas: number[] = []\n const cfRunIds: string[] = []\n for (let rep = 0; rep < opts.reps; rep++) {\n const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner)\n replaysUsed++\n const d = result.delta.deltaScore\n if (typeof d !== 'number' || !Number.isFinite(d)) {\n throw new ValidationError(\n `causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`,\n )\n }\n deltas.push(d)\n cfRunIds.push(result.counterfactualRunId)\n allResults.push(result)\n }\n const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {\n seed: opts.ciSeed ?? DEFAULT_CI_SEED,\n })\n responsibilities.push({\n stepRef: stepRefOf(cell.step),\n mutationKind: cell.mutation.kind,\n meanEffect: ci.mean,\n ci,\n ciExcludesZero: ci.lower > 0 || ci.upper < 0,\n reps: opts.reps,\n deltas,\n counterfactualRunIds: cfRunIds,\n })\n }\n\n responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect))\n\n // A step probed under one mutation but cut off under another appears in\n // BOTH steps and uncovered — partial coverage is named, not blended.\n const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf)\n\n return {\n runId: opts.runId,\n originalScore,\n steps: responsibilities,\n byMutationKind: attributeCounterfactuals(allResults),\n replaysUsed,\n budget: opts.budget,\n uncovered,\n }\n}\n\nfunction resolveCandidates(trajectory: Trajectory, indices?: number[]): TrajectoryStep[] {\n if (indices === undefined) {\n return trajectory.steps.filter((s) => s.span.kind === 'llm' || s.span.kind === 'tool')\n }\n return indices.map((i) => {\n const step = trajectory.steps[i]\n if (!step) {\n throw new ValidationError(\n `causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`,\n )\n }\n return step\n })\n}\n","/**\n * Remediation adapters — HOW DO WE MAKE IT HAPPEN?\n *\n * The diagnose chain ends by feeding existing improvement machinery,\n * not by building new machinery:\n *\n * - `toAnalystFindings` → the analyst contract (`makeFinding`), so\n * responsibility evidence flows into the same registry / steering /\n * diff pipeline every other analyst feeds.\n * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the\n * diagnosed failure + validated repair as a permanent scenario.\n * - `suggestInvariant` → a plain-data hint in the shape the\n * trace-contracts machinery consumes (`never` / `without` clauses).\n */\n\nimport type { AnalystFinding, AnalystSeverity, EvidenceRef } from '../analyst/types'\nimport { makeFinding } from '../analyst/types'\nimport type { CounterfactualMutation } from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { CorpusRecord } from '../rl/corpus'\nimport type { RunRecord } from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CausalResponsibilityReport, StepResponsibility } from './causal-sweep'\nimport type { RepairReport, ValidatedRepair } from './repair'\n\nexport const DIAGNOSE_ANALYST_ID = 'diagnose-causal-sweep'\n\n/** Severity from causal effect size. Effects whose CI includes zero are\n * 'info' regardless of magnitude — an indistinguishable-from-noise effect\n * must not steer remediation priority. */\nexport function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity {\n if (!responsibility.ciExcludesZero) return 'info'\n const magnitude = Math.abs(responsibility.meanEffect)\n if (magnitude >= 0.5) return 'critical'\n if (magnitude >= 0.25) return 'high'\n if (magnitude >= 0.1) return 'medium'\n return 'low'\n}\n\n/** Deterministic human-readable rendering of a mutation — used in\n * recommended actions, corpus completions, and invariant hints. */\nexport function describeMutation(mutation: CounterfactualMutation): string {\n switch (mutation.kind) {\n case 'swap-model':\n return `use model '${mutation.newModel}' at step ${mutation.at}`\n case 'swap-tool-result':\n return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`\n case 'truncate-after':\n return `stop the run after step ${mutation.at}`\n case 'inject-system-message':\n return `inject system message at step ${mutation.at}: ${mutation.content}`\n case 'custom':\n return `${mutation.describe} (step ${mutation.at})`\n }\n}\n\n/**\n * Lift a responsibility report (and optionally its validated repairs) into\n * `AnalystFinding`s via the real `makeFinding` factory. One finding per\n * probed step; a validated repair for that step upgrades the finding with\n * a `recommended_action` + the replay-validation evidence.\n *\n * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,\n * so `derived_from_judge` stays unset and they may steer.\n */\nexport function toAnalystFindings(\n report: CausalResponsibilityReport,\n repairs?: RepairReport,\n): AnalystFinding[] {\n const repairByStep = new Map<string, ValidatedRepair>()\n for (const r of repairs?.repairs ?? []) {\n if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r)\n }\n\n return report.steps.map((resp) => {\n const repair = repairByStep.get(resp.stepRef.spanId)\n const evidence: EvidenceRef[] = [\n {\n kind: 'span',\n uri: `span://${resp.stepRef.spanId}`,\n excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`,\n },\n {\n kind: 'metric',\n uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,\n excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(', ')}]`,\n },\n ...resp.counterfactualRunIds.map((id): EvidenceRef => ({ kind: 'span', uri: `run://${id}` })),\n ]\n return makeFinding({\n analyst_id: DIAGNOSE_ANALYST_ID,\n severity: severityFromEffect(resp),\n area: 'causal-attribution',\n claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,\n rationale: resp.ciExcludesZero\n ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero`\n : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero — not distinguishable from noise`,\n evidence_refs: evidence,\n recommended_action: repair ? describeMutation(repair.mutation) : undefined,\n validation_plan: repair\n ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs!.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})`\n : undefined,\n confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,\n subject: resp.stepRef.spanId,\n metadata: {\n stepRef: resp.stepRef,\n mutationKind: resp.mutationKind,\n meanEffect: resp.meanEffect,\n ci: resp.ci,\n deltas: resp.deltas,\n counterfactualRunIds: resp.counterfactualRunIds,\n ...(repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}),\n },\n })\n })\n}\n\n/**\n * Pin the diagnosed failure as a permanent corpus scenario. Takes the\n * original run's `RunRecord` projection plus a validated repair and emits\n * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw\n * failure and the diagnosed entry).\n *\n * `completion` defaults to the validated mutation's rendering — \"what\n * should have happened\" in machine-derived form. Supply `prompt` (and\n * optionally a richer `completion`) when the trajectory text is available\n * so the record is harvestable by `buildDatasetFromCorpus`.\n */\nexport function toCorpusRecord(\n run: RunRecord,\n repair: ValidatedRepair,\n opts: { prompt?: string; completion?: string } = {},\n): CorpusRecord {\n const record: CorpusRecord = {\n ...run,\n runId: `${run.runId}#repair:${repair.stepRef.spanId}`,\n outcome: {\n ...run.outcome,\n raw: {\n ...run.outcome.raw,\n diagnose_blamed_step_index: repair.stepRef.index,\n diagnose_repair_mean_score: repair.meanScore,\n diagnose_repair_delta_score: repair.deltaScore,\n diagnose_repair_reps: repair.reps,\n },\n },\n prompt: opts.prompt,\n completion: opts.completion ?? describeMutation(repair.mutation),\n }\n // Boundary check — a corpus record that fails RunRecord validation would\n // poison every downstream harvest.\n validateRunRecord(record)\n return record\n}\n\n/** Plain-data invariant hint. The trace-contracts machinery consumes this\n * shape: `never` is a pattern that must not appear in a passing trace;\n * `without` is a guard whose absence makes the failure reachable. */\nexport interface InvariantHint {\n description: string\n never?: string\n without?: string\n}\n\n/**\n * Derive an invariant hint from a validated repair. Deterministic per\n * mutation kind — the hint names the contract a trace must satisfy so\n * the diagnosed failure cannot silently recur.\n */\nexport function suggestInvariant(repair: ValidatedRepair): InvariantHint {\n const { stepRef, mutation } = repair\n const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`\n switch (mutation.kind) {\n case 'swap-tool-result':\n return {\n description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `unvalidated result from tool '${stepRef.name}' flows downstream`,\n without: `result guard on tool '${stepRef.name}'`,\n }\n case 'swap-model':\n return {\n description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`,\n }\n case 'inject-system-message':\n return {\n description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n without: `system message present at '${stepRef.name}': ${mutation.content}`,\n }\n case 'truncate-after':\n return {\n description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) — continuation past this step caused the failure`,\n never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`,\n }\n case 'custom':\n return {\n description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n }\n default: {\n const exhausted: never = mutation\n throw new ValidationError(\n `suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`,\n )\n }\n }\n}\n","/**\n * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?\n *\n * Takes the blamed steps from a `CausalResponsibilityReport`, asks a\n * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate\n * mutations, and machine-verifies each candidate by replaying the run\n * WITH the mutation applied (through the same `runCounterfactual` seam\n * the sweep uses).\n *\n * A repair is \"what should have happened\" ONLY when every validation\n * replay crosses `flipThreshold` — a prescription is never speculated,\n * it is demonstrated. Candidates that don't flip, or whose replay\n * errors, land in `rejected` with a typed reason; nothing is dropped\n * silently.\n */\n\nimport {\n type CounterfactualMutation,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\nimport type { StepRef, StepResponsibility } from './causal-sweep'\n\n/** Context handed to `proposeFix` so an LLM-backed proposer can see the\n * full trajectory plus the responsibility evidence for the blamed step. */\nexport interface RepairContext {\n runId: string\n trajectory: Trajectory\n originalScore: number\n responsibility: StepResponsibility\n}\n\nexport interface PrescribeRepairOptions {\n store: TraceStore\n /** The failed run the sweep diagnosed. */\n runId: string\n /** Execution seam — same `CounterfactualRunner` contract as the sweep. */\n runner: CounterfactualRunner\n /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */\n blamed: StepResponsibility[]\n /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.\n * Returned mutations MUST target the blamed step's index. */\n proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>\n /** Score every validation replay must reach for the repair to count. Default 0.5. */\n flipThreshold?: number\n /** Validation replays per candidate mutation. Default 3. */\n repsToValidate?: number\n /** Max candidate mutations tried per step. Default: all proposed. */\n maxAttemptsPerStep?: number\n}\n\nexport interface ValidatedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n /** Always true — presence in `repairs` IS the machine-verified claim. */\n validated: true\n /** Mean counterfactual score across the validation reps. */\n meanScore: number\n /** meanScore − originalScore. */\n deltaScore: number\n reps: number\n /** Replay run ids backing the validation — audit trail. */\n counterfactualRunIds: string[]\n}\n\nexport interface RejectedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n reason: 'did-not-flip' | 'error'\n /** Present for 'did-not-flip': mean delta over the reps that ran. */\n deltaScore?: number\n /** Present for 'error': the message, preserved for diagnosis. */\n error?: string\n}\n\nexport interface RepairReport {\n runId: string\n originalScore: number\n flipThreshold: number\n repairs: ValidatedRepair[]\n rejected: RejectedRepair[]\n replaysUsed: number\n}\n\nexport async function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport> {\n const flipThreshold = opts.flipThreshold ?? 0.5\n const repsToValidate = opts.repsToValidate ?? 3\n if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {\n throw new ValidationError(\n `prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`,\n )\n }\n const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY\n if (maxAttempts < 1) {\n throw new ValidationError(\n `prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`,\n )\n }\n if (opts.blamed.length === 0) {\n throw new ValidationError('prescribeRepair: blamed is empty — nothing to repair')\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `prescribeRepair: run ${opts.runId} has no numeric outcome.score — flips have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n\n const repairs: ValidatedRepair[] = []\n const rejected: RejectedRepair[] = []\n let replaysUsed = 0\n\n for (const responsibility of opts.blamed) {\n const step = trajectory.steps[responsibility.stepRef.index]\n if (!step || step.span.spanId !== responsibility.stepRef.spanId) {\n throw new ValidationError(\n `prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} — stale report?`,\n )\n }\n\n const candidates = await opts.proposeFix(step, {\n runId: opts.runId,\n trajectory,\n originalScore,\n responsibility,\n })\n const toTry = candidates.slice(0, maxAttempts)\n\n for (const mutation of toTry) {\n if (mutation.at !== step.index) {\n throw new ValidationError(\n `prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`,\n )\n }\n const scores: number[] = []\n const cfRunIds: string[] = []\n let failure: string | undefined\n for (let rep = 0; rep < repsToValidate; rep++) {\n try {\n const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner)\n replaysUsed++\n const score = result.delta.counterfactualOutcomeScore\n if (typeof score !== 'number' || !Number.isFinite(score)) {\n failure = `validation rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`\n break\n }\n scores.push(score)\n cfRunIds.push(result.counterfactualRunId)\n } catch (err) {\n replaysUsed++\n failure = err instanceof Error ? err.message : String(err)\n break\n }\n }\n\n if (failure !== undefined) {\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'error',\n error: failure,\n })\n continue\n }\n\n const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length\n const everyRepFlipped = scores.every((s) => s >= flipThreshold)\n if (everyRepFlipped) {\n repairs.push({\n stepRef: responsibility.stepRef,\n mutation,\n validated: true,\n meanScore,\n deltaScore: meanScore - originalScore,\n reps: repsToValidate,\n counterfactualRunIds: cfRunIds,\n })\n // First validated repair per step IS the prescription; remaining\n // candidates are untried, not rejected — we don't fabricate verdicts.\n break\n }\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'did-not-flip',\n deltaScore: meanScore - originalScore,\n })\n }\n }\n\n return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;AA2CO,SAAS,UAAU,MAA+B;AACvD,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK,KAAK;AAAA,IAClB,MAAM,KAAK,KAAK;AAAA,IAChB,MAAM,KAAK,KAAK;AAAA,EAClB;AACF;AAkEA,IAAM,kBAAkB;AAExB,SAAS,iBAAiB,MAAgD;AACxE,MAAI,KAAK,KAAK,SAAS,QAAQ;AAC7B,WAAO,CAAC,EAAE,MAAM,oBAAoB,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC;AAAA,EACvE;AACA,MAAI,KAAK,KAAK,SAAS,OAAO;AAC5B,WAAO,CAAC,EAAE,MAAM,kBAAkB,IAAI,KAAK,MAAM,CAAC;AAAA,EACpD;AACA,SAAO,CAAC;AACV;AAEA,eAAsB,YAAY,MAA+D;AAC/F,MAAI,CAAC,OAAO,UAAU,KAAK,IAAI,KAAK,KAAK,OAAO,GAAG;AACjD,UAAM,IAAI;AAAA,MACR,kDAAkD,KAAK,IAAI;AAAA,IAC7D;AAAA,EACF;AACA,MAAI,CAAC,OAAO,UAAU,KAAK,MAAM,KAAK,KAAK,SAAS,GAAG;AACrD,UAAM,IAAI,gBAAgB,oDAAoD,KAAK,MAAM,GAAG;AAAA,EAC9F;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,oBAAoB,KAAK,KAAK,YAAY;AACtF,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,oBAAoB,KAAK,KAAK;AAAA,IAChC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAC/D,QAAM,aAAa,kBAAkB,YAAY,KAAK,cAAc;AACpE,QAAM,eAAe,KAAK,oBAAoB;AAM9C,QAAM,QAAgB,CAAC;AACvB,aAAW,QAAQ,YAAY;AAC7B,UAAM,YAAY,aAAa,IAAI;AACnC,eAAW,KAAK,WAAW;AACzB,UAAI,EAAE,OAAO,KAAK,OAAO;AACvB,cAAM,IAAI;AAAA,UACR,kEAAkE,EAAE,EAAE,mBAAmB,KAAK,KAAK;AAAA,QACrG;AAAA,MACF;AACA,YAAM,KAAK,EAAE,MAAM,UAAU,EAAE,CAAC;AAAA,IAClC;AAAA,EACF;AAEA,QAAM,mBAAyC,CAAC;AAChD,QAAM,aAAqC,CAAC;AAC5C,QAAM,mBAAmB,oBAAI,IAAY;AACzC,MAAI,cAAc;AAClB,MAAI,SAAS;AAEb,aAAW,QAAQ,OAAO;AACxB,QAAI,UAAU,cAAc,KAAK,OAAO,KAAK,QAAQ;AAGnD,eAAS;AACT,uBAAiB,IAAI,KAAK,KAAK,KAAK;AACpC;AAAA,IACF;AACA,UAAM,SAAmB,CAAC;AAC1B,UAAM,WAAqB,CAAC;AAC5B,aAAS,MAAM,GAAG,MAAM,KAAK,MAAM,OAAO;AACxC,YAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,KAAK,UAAU,KAAK,MAAM;AACzF;AACA,YAAM,IAAI,OAAO,MAAM;AACvB,UAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;AAChD,cAAM,IAAI;AAAA,UACR,+CAA+C,KAAK,KAAK,KAAK,KAAK,KAAK,SAAS,IAAI,SAAS,GAAG;AAAA,QACnG;AAAA,MACF;AACA,aAAO,KAAK,CAAC;AACb,eAAS,KAAK,OAAO,mBAAmB;AACxC,iBAAW,KAAK,MAAM;AAAA,IACxB;AACA,UAAM,KAAK,mBAAmB,QAAQ,KAAK,gBAAgB,MAAM;AAAA,MAC/D,MAAM,KAAK,UAAU;AAAA,IACvB,CAAC;AACD,qBAAiB,KAAK;AAAA,MACpB,SAAS,UAAU,KAAK,IAAI;AAAA,MAC5B,cAAc,KAAK,SAAS;AAAA,MAC5B,YAAY,GAAG;AAAA,MACf;AAAA,MACA,gBAAgB,GAAG,QAAQ,KAAK,GAAG,QAAQ;AAAA,MAC3C,MAAM,KAAK;AAAA,MACX;AAAA,MACA,sBAAsB;AAAA,IACxB,CAAC;AAAA,EACH;AAEA,mBAAiB,KAAK,CAAC,GAAG,MAAM,KAAK,IAAI,EAAE,UAAU,IAAI,KAAK,IAAI,EAAE,UAAU,CAAC;AAI/E,QAAM,YAAY,WAAW,OAAO,CAAC,MAAM,iBAAiB,IAAI,EAAE,KAAK,CAAC,EAAE,IAAI,SAAS;AAEvF,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ;AAAA,IACA,OAAO;AAAA,IACP,gBAAgB,yBAAyB,UAAU;AAAA,IACnD;AAAA,IACA,QAAQ,KAAK;AAAA,IACb;AAAA,EACF;AACF;AAEA,SAAS,kBAAkB,YAAwB,SAAsC;AACvF,MAAI,YAAY,QAAW;AACzB,WAAO,WAAW,MAAM,OAAO,CAAC,MAAM,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,MAAM;AAAA,EACvF;AACA,SAAO,QAAQ,IAAI,CAAC,MAAM;AACxB,UAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,QAAI,CAAC,MAAM;AACT,YAAM,IAAI;AAAA,QACR,qCAAqC,CAAC,qBAAqB,WAAW,MAAM,MAAM;AAAA,MACpF;AAAA,IACF;AACA,WAAO;AAAA,EACT,CAAC;AACH;;;ACzNO,IAAM,sBAAsB;AAK5B,SAAS,mBAAmB,gBAAqD;AACtF,MAAI,CAAC,eAAe,eAAgB,QAAO;AAC3C,QAAM,YAAY,KAAK,IAAI,eAAe,UAAU;AACpD,MAAI,aAAa,IAAK,QAAO;AAC7B,MAAI,aAAa,KAAM,QAAO;AAC9B,MAAI,aAAa,IAAK,QAAO;AAC7B,SAAO;AACT;AAIO,SAAS,iBAAiB,UAA0C;AACzE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO,cAAc,SAAS,QAAQ,aAAa,SAAS,EAAE;AAAA,IAChE,KAAK;AACH,aAAO,mCAAmC,SAAS,EAAE,SAAS,KAAK,UAAU,SAAS,SAAS,CAAC;AAAA,IAClG,KAAK;AACH,aAAO,2BAA2B,SAAS,EAAE;AAAA,IAC/C,KAAK;AACH,aAAO,iCAAiC,SAAS,EAAE,KAAK,SAAS,OAAO;AAAA,IAC1E,KAAK;AACH,aAAO,GAAG,SAAS,QAAQ,UAAU,SAAS,EAAE;AAAA,EACpD;AACF;AAWO,SAAS,kBACd,QACA,SACkB;AAClB,QAAM,eAAe,oBAAI,IAA6B;AACtD,aAAW,KAAK,SAAS,WAAW,CAAC,GAAG;AACtC,QAAI,CAAC,aAAa,IAAI,EAAE,QAAQ,MAAM,EAAG,cAAa,IAAI,EAAE,QAAQ,QAAQ,CAAC;AAAA,EAC/E;AAEA,SAAO,OAAO,MAAM,IAAI,CAAC,SAAS;AAChC,UAAM,SAAS,aAAa,IAAI,KAAK,QAAQ,MAAM;AACnD,UAAM,WAA0B;AAAA,MAC9B;AAAA,QACE,MAAM;AAAA,QACN,KAAK,UAAU,KAAK,QAAQ,MAAM;AAAA,QAClC,SAAS,QAAQ,KAAK,QAAQ,KAAK,KAAK,KAAK,QAAQ,IAAI,KAAK,KAAK,QAAQ,IAAI,iBAAiB,KAAK,WAAW,QAAQ,CAAC,CAAC,QAAQ,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,UAAU,KAAK,IAAI;AAAA,MAC5M;AAAA,MACA;AAAA,QACE,MAAM;AAAA,QACN,KAAK,qBAAqB,OAAO,KAAK,SAAS,KAAK,QAAQ,KAAK,IAAI,KAAK,YAAY;AAAA,QACtF,SAAS,WAAW,KAAK,OAAO,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC,EAAE,KAAK,IAAI,CAAC;AAAA,MACrE;AAAA,MACA,GAAG,KAAK,qBAAqB,IAAI,CAAC,QAAqB,EAAE,MAAM,QAAQ,KAAK,SAAS,EAAE,GAAG,EAAE;AAAA,IAC9F;AACA,WAAO,YAAY;AAAA,MACjB,YAAY;AAAA,MACZ,UAAU,mBAAmB,IAAI;AAAA,MACjC,MAAM;AAAA,MACN,OAAO,SAAS,KAAK,QAAQ,IAAI,MAAM,KAAK,QAAQ,IAAI,uDAAuD,KAAK,YAAY;AAAA,MAChI,WAAW,KAAK,iBACZ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,oBAChJ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC;AAAA,MACpJ,eAAe;AAAA,MACf,oBAAoB,SAAS,iBAAiB,OAAO,QAAQ,IAAI;AAAA,MACjE,iBAAiB,SACb,qBAAqB,OAAO,IAAI,IAAI,OAAO,IAAI,mBAAmB,QAAS,aAAa,UAAU,OAAO,UAAU,QAAQ,CAAC,CAAC,WAAW,OAAO,WAAW,QAAQ,CAAC,CAAC,MACpK;AAAA,MACJ,YAAY,SAAS,OAAO,KAAK,iBAAiB,OAAO;AAAA,MACzD,SAAS,KAAK,QAAQ;AAAA,MACtB,UAAU;AAAA,QACR,SAAS,KAAK;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,YAAY,KAAK;AAAA,QACjB,IAAI,KAAK;AAAA,QACT,QAAQ,KAAK;AAAA,QACb,sBAAsB,KAAK;AAAA,QAC3B,GAAI,SAAS,EAAE,QAAQ,EAAE,UAAU,OAAO,UAAU,WAAW,OAAO,UAAU,EAAE,IAAI,CAAC;AAAA,MACzF;AAAA,IACF,CAAC;AAAA,EACH,CAAC;AACH;AAaO,SAAS,eACd,KACA,QACA,OAAiD,CAAC,GACpC;AACd,QAAM,SAAuB;AAAA,IAC3B,GAAG;AAAA,IACH,OAAO,GAAG,IAAI,KAAK,WAAW,OAAO,QAAQ,MAAM;AAAA,IACnD,SAAS;AAAA,MACP,GAAG,IAAI;AAAA,MACP,KAAK;AAAA,QACH,GAAG,IAAI,QAAQ;AAAA,QACf,4BAA4B,OAAO,QAAQ;AAAA,QAC3C,4BAA4B,OAAO;AAAA,QACnC,6BAA6B,OAAO;AAAA,QACpC,sBAAsB,OAAO;AAAA,MAC/B;AAAA,IACF;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,YAAY,KAAK,cAAc,iBAAiB,OAAO,QAAQ;AAAA,EACjE;AAGA,oBAAkB,MAAM;AACxB,SAAO;AACT;AAgBO,SAAS,iBAAiB,QAAwC;AACvE,QAAM,EAAE,SAAS,SAAS,IAAI;AAC9B,QAAM,KAAK,QAAQ,QAAQ,KAAK,KAAK,QAAQ,IAAI,KAAK,QAAQ,IAAI;AAClE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO;AAAA,QACL,aAAa,uBAAuB,QAAQ,IAAI,4FAA4F,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QACxK,OAAO,iCAAiC,QAAQ,IAAI;AAAA,QACpD,SAAS,yBAAyB,QAAQ,IAAI;AAAA,MAChD;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,yBAAyB,EAAE,QAAQ,SAAS,QAAQ,gCAAgC,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC7H,OAAO,aAAa,QAAQ,IAAI,iCAAiC,SAAS,QAAQ;AAAA,MACpF;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,iCAAiC,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC3G,SAAS,8BAA8B,QAAQ,IAAI,MAAM,SAAS,OAAO;AAAA,MAC3E;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,kBAAkB,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC5F,OAAO,wBAAwB,QAAQ,IAAI,YAAY,QAAQ,KAAK;AAAA,MACtE;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,GAAG,SAAS,QAAQ,OAAO,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,MACvG;AAAA,IACF,SAAS;AACP,YAAM,YAAmB;AACzB,YAAM,IAAI;AAAA,QACR,2CAA2C,KAAK,UAAU,SAAS,CAAC;AAAA,MACtE;AAAA,IACF;AAAA,EACF;AACF;;;ACtHA,eAAsB,gBAAgB,MAAqD;AACzF,QAAM,gBAAgB,KAAK,iBAAiB;AAC5C,QAAM,iBAAiB,KAAK,kBAAkB;AAC9C,MAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GAAG;AAC3D,UAAM,IAAI;AAAA,MACR,gEAAgE,cAAc;AAAA,IAChF;AAAA,EACF;AACA,QAAM,cAAc,KAAK,sBAAsB,OAAO;AACtD,MAAI,cAAc,GAAG;AACnB,UAAM,IAAI;AAAA,MACR,yDAAyD,KAAK,kBAAkB;AAAA,IAClF;AAAA,EACF;AACA,MAAI,KAAK,OAAO,WAAW,GAAG;AAC5B,UAAM,IAAI,gBAAgB,2DAAsD;AAAA,EAClF;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,wBAAwB,KAAK,KAAK,YAAY;AAC1F,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,wBAAwB,KAAK,KAAK;AAAA,IACpC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAE/D,QAAM,UAA6B,CAAC;AACpC,QAAM,WAA6B,CAAC;AACpC,MAAI,cAAc;AAElB,aAAW,kBAAkB,KAAK,QAAQ;AACxC,UAAM,OAAO,WAAW,MAAM,eAAe,QAAQ,KAAK;AAC1D,QAAI,CAAC,QAAQ,KAAK,KAAK,WAAW,eAAe,QAAQ,QAAQ;AAC/D,YAAM,IAAI;AAAA,QACR,sCAAsC,eAAe,QAAQ,KAAK,WAAW,eAAe,QAAQ,MAAM,uBAAuB,KAAK,KAAK;AAAA,MAC7I;AAAA,IACF;AAEA,UAAM,aAAa,MAAM,KAAK,WAAW,MAAM;AAAA,MAC7C,OAAO,KAAK;AAAA,MACZ;AAAA,MACA;AAAA,MACA;AAAA,IACF,CAAC;AACD,UAAM,QAAQ,WAAW,MAAM,GAAG,WAAW;AAE7C,eAAW,YAAY,OAAO;AAC5B,UAAI,SAAS,OAAO,KAAK,OAAO;AAC9B,cAAM,IAAI;AAAA,UACR,gEAAgE,SAAS,EAAE,0BAA0B,KAAK,KAAK;AAAA,QACjH;AAAA,MACF;AACA,YAAM,SAAmB,CAAC;AAC1B,YAAM,WAAqB,CAAC;AAC5B,UAAI;AACJ,eAAS,MAAM,GAAG,MAAM,gBAAgB,OAAO;AAC7C,YAAI;AACF,gBAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,UAAU,KAAK,MAAM;AACpF;AACA,gBAAM,QAAQ,OAAO,MAAM;AAC3B,cAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;AACxD,sBAAU,kBAAkB,GAAG;AAC/B;AAAA,UACF;AACA,iBAAO,KAAK,KAAK;AACjB,mBAAS,KAAK,OAAO,mBAAmB;AAAA,QAC1C,SAAS,KAAK;AACZ;AACA,oBAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AACzD;AAAA,QACF;AAAA,MACF;AAEA,UAAI,YAAY,QAAW;AACzB,iBAAS,KAAK;AAAA,UACZ,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,QAAQ;AAAA,UACR,OAAO;AAAA,QACT,CAAC;AACD;AAAA,MACF;AAEA,YAAM,YAAY,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;AAC7D,YAAM,kBAAkB,OAAO,MAAM,CAAC,MAAM,KAAK,aAAa;AAC9D,UAAI,iBAAiB;AACnB,gBAAQ,KAAK;AAAA,UACX,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,WAAW;AAAA,UACX;AAAA,UACA,YAAY,YAAY;AAAA,UACxB,MAAM;AAAA,UACN,sBAAsB;AAAA,QACxB,CAAC;AAGD;AAAA,MACF;AACA,eAAS,KAAK;AAAA,QACZ,SAAS,eAAe;AAAA,QACxB;AAAA,QACA,QAAQ;AAAA,QACR,YAAY,YAAY;AAAA,MAC1B,CAAC;AAAA,IACH;AAAA,EACF;AAEA,SAAO,EAAE,OAAO,KAAK,OAAO,eAAe,eAAe,SAAS,UAAU,YAAY;AAC3F;","names":[]}
@@ -1,4 +1,4 @@
1
- import { S as Scenario, C as CampaignResult, G as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-BTI16iFl.js';
1
+ import { S as Scenario, C as CampaignResult, G as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-Cv1bo4_a.js';
2
2
  import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
3
3
 
4
4
  /**
@@ -1,10 +1,10 @@
1
- import { M as MutableSurface, j as GateDecision } from '../types-BTI16iFl.js';
2
- import { I as InsightReport } from '../insight-report-BnRjTibG.js';
3
- import '../run-record-CP2ObebC.js';
1
+ import { M as MutableSurface, j as GateDecision } from '../types-Cv1bo4_a.js';
2
+ import { I as InsightReport } from '../insight-report-C02J3q4T.js';
3
+ import '../run-record-DEwidcqn.js';
4
4
  import '@tangle-network/agent-interface';
5
5
  import '../errors-CzMUYo7b.js';
6
6
  import '../schema-m0gsnbt3.js';
7
- import '../summary-report-CInXwsza.js';
7
+ import '../summary-report-C4uzRWh8.js';
8
8
  import '../failure-cluster-DH9Flgcf.js';
9
9
  import '../store-BcFXE6LG.js';
10
10
  import '../judge-calibration-0p2QcWNE.js';
@@ -1,4 +1,4 @@
1
- import { a as RunSplitTag } from './run-record-CP2ObebC.js';
1
+ import { b as RunSplitTag } from './run-record-DEwidcqn.js';
2
2
 
3
3
  /**
4
4
  * Shared types for the reference benchmark wrappers under
package/dist/index.d.ts CHANGED
@@ -1,12 +1,13 @@
1
- export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-Doncu-B_.js';
2
- import { R as RunRecord, a as RunSplitTag } from './run-record-CP2ObebC.js';
3
- export { e as AGENT_PROFILE_KINDS, f as AgentInterfaceProfileLike, A as AgentProfileCell, d as AgentProfileCellInput, g as AgentProfileCellSchemaVersion, h as AgentProfileCellValidationError, i as AgentProfileDimensionValue, j as AgentProfileHarness, k as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, b as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as parseRunRecordSafe, y as requireAgentProfileCell, z as roundTripRunRecord, B as toAgentProfileJson, C as validateAgentProfileCell, D as validateRunRecord, E as verifyAgentProfileCell } from './run-record-CP2ObebC.js';
4
- import { B as BehavioralMetrics } from './semantic-concept-judge-DSBB2Cfp.js';
5
- export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-DSBB2Cfp.js';
6
- import { l as ChatRequest, p as CreateChatClientOpts } from './types-B5x54y6n.js';
7
- export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-B5x54y6n.js';
8
- export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-GyE8X5SP.js';
9
- export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-X3eDYbKn.js';
1
+ export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-DC8TELh0.js';
2
+ import { R as RunRecord, b as RunSplitTag } from './run-record-DEwidcqn.js';
3
+ export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as parseRunRecordSafe, y as requireAgentProfileCell, z as roundTripRunRecord, B as toAgentProfileJson, C as validateAgentProfileCell, D as validateRunRecord, E as verifyAgentProfileCell } from './run-record-DEwidcqn.js';
4
+ import { B as BehavioralMetrics } from './semantic-concept-judge-J8xvjdc3.js';
5
+ export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-J8xvjdc3.js';
6
+ import { l as ChatRequest, p as CreateChatClientOpts } from './types-BEzCBMQD.js';
7
+ export { a as Analyst, b as AnalystContext, g as AnalystCost, A as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-BEzCBMQD.js';
8
+ export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-Dhrc__SE.js';
9
+ export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-OgqQSvLi.js';
10
+ export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-Dccm9tyA.js';
10
11
  import { TCloud } from '@tangle-network/tcloud';
11
12
  import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
12
13
  export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
@@ -17,11 +18,11 @@ import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors
17
18
  export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-CzMUYo7b.js';
18
19
  import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-BxY0cKfs.js';
19
20
  export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-BxY0cKfs.js';
20
- import { b as CorrectnessChecker } from './pre-registration-CMm8cvrh.js';
21
- export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-CMm8cvrh.js';
21
+ import { b as CorrectnessChecker } from './pre-registration-DB8oDqZJ.js';
22
+ export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-DB8oDqZJ.js';
22
23
  export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
23
- import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-pidWUMZ2.js';
24
- export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-pidWUMZ2.js';
24
+ import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-B1tA6pKu.js';
25
+ export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-B1tA6pKu.js';
25
26
  export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as ProportionInterval, R as RiskDifferenceResult, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as pearsonR, K as ranks, L as requiredSampleSize, N as spearmanR, O as weightedComposite, Q as weightedMean, w as wilcoxonSignedRank, S as wilson } from './statistics-xP-cWc5k.js';
26
27
  import { OtelExporter, OtelExportConfig } from './traces.js';
27
28
  export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
@@ -29,8 +30,8 @@ import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesR
29
30
  export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
30
31
  import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
31
32
  export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
32
- import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-BTI16iFl.js';
33
- import { A as AnalyzeRunsOptions } from './analyze-runs-DtT6F_6T.js';
33
+ import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-Cv1bo4_a.js';
34
+ import { A as AnalyzeRunsOptions } from './analyze-runs-BlJRBniC.js';
34
35
  import { S as SteeringBundle } from './harness-optimizer-mOl9XX_O.js';
35
36
  export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-mOl9XX_O.js';
36
37
  import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-DeODGLra.js';
@@ -46,7 +47,7 @@ export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionRep
46
47
  import { T as TraceStore, R as RunFilter } from './store-BcFXE6LG.js';
47
48
  export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BcFXE6LG.js';
48
49
  export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-DH9Flgcf.js';
49
- export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-BOUUjI0y.js';
50
+ export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-OJDaTYHN.js';
50
51
  import { a as BaselineReport } from './baseline-Bbid3WoO.js';
51
52
  export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-Bbid3WoO.js';
52
53
  import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
@@ -69,16 +70,16 @@ import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from
69
70
  export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DUZXrPDA.js';
70
71
  import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
71
72
  export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-Bj7g0rqu.js';
72
- export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-_Y4oNOOb.js';
73
- export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Jr8ME1dZ.js';
74
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-CInXwsza.js';
73
+ export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-W96macmS.js';
74
+ export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Ba2y1Foi.js';
75
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-C4uzRWh8.js';
75
76
  export { L as LockedJsonlAppender } from './testing-C21CHsq2.js';
76
77
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
77
- import { e as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-H6mlM0KN.js';
78
+ import { e as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-BRgNnmGZ.js';
78
79
  export { IntegrityResult, IntegrityViolation, JourneySpec, PerfBaseline, PerfGateResult, PerfRegression, PerfScenario, PerfStat, ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords } from './perf/index.js';
79
80
  import '@ax-llm/ax';
80
81
  import 'zod';
81
- import './insight-report-BnRjTibG.js';
82
+ import './insight-report-C02J3q4T.js';
82
83
  import './outcome-store-rnXLEqSn.js';
83
84
 
84
85
  /**
@@ -2354,6 +2355,51 @@ declare class ExperimentTracker {
2354
2355
  verdictFor(experimentId: string): Promise<ImprovementVerdictResult>;
2355
2356
  }
2356
2357
 
2358
+ /**
2359
+ * Cross-profile leaderboard — rank agent profiles (or any grouping) over a corpus
2360
+ * of RunRecords on pass-rate with a Wilson CI, joined with the cost / token /
2361
+ * latency rollup every benchmark wants next to the score.
2362
+ *
2363
+ * This is substrate, not benchmark-specific: any consumer of agent-eval gets a
2364
+ * cross-harness × model leaderboard without re-implementing grouping, binomial
2365
+ * CIs, or the observability join. The benchmark supplies only what "passed"
2366
+ * means; everything else composes existing primitives (`wilson`, the RunRecord
2367
+ * cost/token/wall fields, the AgentProfileCell grouping key).
2368
+ */
2369
+
2370
+ interface LeaderboardRow {
2371
+ /** Group key — the agent-profile cellId by default, else a caller dimension. */
2372
+ key: string;
2373
+ /** Display label: `harness · model` when the profile carries them, else the key. */
2374
+ label: string;
2375
+ harness?: string;
2376
+ model?: string;
2377
+ rank: number;
2378
+ /** Measured runs in this group. */
2379
+ n: number;
2380
+ passRate: number;
2381
+ passRateCi95: [number, number];
2382
+ /** Observability rollup — means over the group; null when no run reported it. */
2383
+ meanCostUsd: number | null;
2384
+ meanTokensIn: number | null;
2385
+ meanTokensOut: number | null;
2386
+ meanWallMs: number | null;
2387
+ }
2388
+ interface LeaderboardOptions {
2389
+ /** Per-run pass predicate — the benchmark defines what "passed" means (real-build
2390
+ * hit-rate, test-pass, etc.). The leaderboard owns only the grouping + stats. */
2391
+ passed: (record: RunRecord) => boolean;
2392
+ /** Group key. Default: the agent-profile cellId (harness × model × dimensions),
2393
+ * so persona/harness/model sweeps separate without parsing labels. */
2394
+ groupBy?: (record: RunRecord) => string;
2395
+ }
2396
+ /**
2397
+ * Group `records`, compute pass-rate + Wilson CI + cost/token/latency means per
2398
+ * group, and rank by pass-rate (cost as the tiebreaker — cheaper wins a tie).
2399
+ * Pure projection: no I/O, deterministic, safe to call on any RunRecord[].
2400
+ */
2401
+ declare function leaderboard(records: RunRecord[], opts: LeaderboardOptions): LeaderboardRow[];
2402
+
2357
2403
  /**
2358
2404
  * muffled-gate-scanner — test helper that greps consumer source for
2359
2405
  * gate + measurement anti-patterns and fails with file:line locations.
@@ -5973,4 +6019,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeCo
5973
6019
  */
5974
6020
  declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
5975
6021
 
5976
- export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BehavioralMetrics, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeScoreInput, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, type RoutedField, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolMatcher, ToolSpan, TraceAnalystSpan, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCrossFamily, assertModelsServed, assertNoHiddenLeak, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, blendHeldout, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, classifyTreatment, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, hashContent, hashToUnit, hiddenGrade, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isHiddenDestination, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, routeFields, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
6022
+ export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BehavioralMetrics, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeScoreInput, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, type RoutedField, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolMatcher, ToolSpan, TraceAnalystSpan, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCrossFamily, assertModelsServed, assertNoHiddenLeak, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, blendHeldout, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, classifyTreatment, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, hashContent, hashToUnit, hiddenGrade, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isHiddenDestination, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, leaderboard, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, routeFields, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };