@tangle-network/agent-eval 0.100.0 → 0.100.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/analyst/index.d.ts +8 -7
- package/dist/analyst/index.js +35 -27
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DtT6F_6T.d.ts → analyze-runs-BlJRBniC.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/campaign/index.d.ts +37 -13
- package/dist/campaign/index.js +76 -7
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-WMBLMTUE.js → chunk-2KTBHICD.js} +139 -16
- package/dist/chunk-2KTBHICD.js.map +1 -0
- package/dist/{chunk-IZCEK2HR.js → chunk-2MLIEQSN.js} +3 -2
- package/dist/{chunk-IZCEK2HR.js.map → chunk-2MLIEQSN.js.map} +1 -1
- package/dist/{chunk-OKQ2LAT7.js → chunk-4LWD6GC7.js} +7 -5
- package/dist/{chunk-OKQ2LAT7.js.map → chunk-4LWD6GC7.js.map} +1 -1
- package/dist/{chunk-LO6IOIJ2.js → chunk-ABOIVNXL.js} +2 -240
- package/dist/chunk-ABOIVNXL.js.map +1 -0
- package/dist/{chunk-NZEQVRH5.js → chunk-BOETF6BU.js} +2 -2
- package/dist/{chunk-S4SYLDFX.js → chunk-FRI6RG3P.js} +3 -2
- package/dist/chunk-FRI6RG3P.js.map +1 -0
- package/dist/chunk-G6S73VA7.js +248 -0
- package/dist/chunk-G6S73VA7.js.map +1 -0
- package/dist/{chunk-GMGRBNVT.js → chunk-G7IB3GJ5.js} +2 -2
- package/dist/chunk-IN3SHQML.js +664 -0
- package/dist/chunk-IN3SHQML.js.map +1 -0
- package/dist/{chunk-SJISCGWD.js → chunk-JU6ZX3CX.js} +2 -2
- package/dist/{chunk-3NHEO6ZC.js → chunk-L5TVEZFT.js} +2 -2
- package/dist/{chunk-77T4STFI.js → chunk-PMF5WIBX.js} +3 -3
- package/dist/{chunk-OYU4D7FY.js → chunk-VWQ6PO5O.js} +2 -2
- package/dist/{code-agent-session-CPHRCb4-.d.ts → code-agent-session-B6ZcDwyA.d.ts} +1 -1
- package/dist/contract/index.d.ts +16 -16
- package/dist/contract/index.js +6 -5
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-Doncu-B_.d.ts → control-DC8TELh0.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +3 -2
- package/dist/{corpus-D4YW9UoJ.d.ts → corpus-ONOzGFmG.d.ts} +1 -1
- package/dist/{default-registry-GyE8X5SP.d.ts → default-registry-Dhrc__SE.d.ts} +2 -2
- package/dist/diagnose.d.ts +3 -3
- package/dist/diagnose.js +2 -1
- package/dist/diagnose.js.map +1 -1
- package/dist/{gepa-H6mlM0KN.d.ts → gepa-BRgNnmGZ.d.ts} +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-_Y4oNOOb.d.ts → index-W96macmS.d.ts} +1 -1
- package/dist/index.d.ts +22 -21
- package/dist/index.js +47 -21
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BnRjTibG.d.ts → insight-report-C02J3q4T.d.ts} +1 -1
- package/dist/{kind-factory-X3eDYbKn.d.ts → kind-factory-OgqQSvLi.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/policy-edit-Dccm9tyA.d.ts +103 -0
- package/dist/{pre-registration-CMm8cvrh.d.ts → pre-registration-DB8oDqZJ.d.ts} +3 -3
- package/dist/{provenance-Bg_RttR8.d.ts → provenance-B0SZw1z2.d.ts} +3 -3
- package/dist/{release-report-pidWUMZ2.d.ts → release-report-B1tA6pKu.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Jr8ME1dZ.d.ts → researcher-Ba2y1Foi.d.ts} +2 -2
- package/dist/rl.d.ts +8 -8
- package/dist/rl.js +3 -2
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-C2hDKM8Z.d.ts → rubric-predictive-validity-w7tun-q3.d.ts} +1 -1
- package/dist/{run-record-CP2ObebC.d.ts → run-record-DEwidcqn.d.ts} +1 -1
- package/dist/{runtime-trajectory-BOUUjI0y.d.ts → runtime-trajectory-OJDaTYHN.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DSBB2Cfp.d.ts → semantic-concept-judge-J8xvjdc3.d.ts} +2 -2
- package/dist/{summary-report-CInXwsza.d.ts → summary-report-C4uzRWh8.d.ts} +1 -1
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +4 -3
- package/dist/{types-B5x54y6n.d.ts → types-BEzCBMQD.d.ts} +2 -2
- package/dist/{types-BTI16iFl.d.ts → types-Cv1bo4_a.d.ts} +1 -1
- package/dist/workflow/index.d.ts +4 -4
- package/dist/workflow/index.js +2 -1
- package/dist/workflow/index.js.map +1 -1
- package/package.json +1 -1
- package/dist/chunk-BUTW4RGG.js +0 -32
- package/dist/chunk-BUTW4RGG.js.map +0 -1
- package/dist/chunk-LO6IOIJ2.js.map +0 -1
- package/dist/chunk-S4SYLDFX.js.map +0 -1
- package/dist/chunk-WMBLMTUE.js.map +0 -1
- /package/dist/{chunk-NZEQVRH5.js.map → chunk-BOETF6BU.js.map} +0 -0
- /package/dist/{chunk-GMGRBNVT.js.map → chunk-G7IB3GJ5.js.map} +0 -0
- /package/dist/{chunk-SJISCGWD.js.map → chunk-JU6ZX3CX.js.map} +0 -0
- /package/dist/{chunk-3NHEO6ZC.js.map → chunk-L5TVEZFT.js.map} +0 -0
- /package/dist/{chunk-77T4STFI.js.map → chunk-PMF5WIBX.js.map} +0 -0
- /package/dist/{chunk-OYU4D7FY.js.map → chunk-VWQ6PO5O.js.map} +0 -0
package/dist/diagnose.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/diagnose/causal-sweep.ts","../src/diagnose/remediation.ts","../src/diagnose/repair.ts"],"sourcesContent":["/**\n * Causal sweep — WHY did this run fail?\n *\n * Orchestrates the dormant counterfactual primitives into a responsibility\n * report: for each candidate step, run `reps` counterfactual replays per\n * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`\n * is the execution seam) and reduce the per-rep score deltas into a mean\n * effect + bootstrap confidence interval (via `confidenceInterval`).\n *\n * Why `reps` is REQUIRED: a single intervention delta is one stochastic\n * draw — LLM re-execution from a prefix is sampled, so one replay cannot\n * distinguish \"this step caused the failure\" from sampling noise. The\n * signal is the distribution of deltas across reps; the CI over that\n * distribution is what lets a caller say \"this step's effect excludes\n * zero\" instead of eyeballing a point estimate.\n *\n * Budget discipline: the sweep never silently drops cells. When the\n * remaining budget cannot fund a full `reps`-sized cell, the sweep halts\n * and every step not fully probed is named in `uncovered`.\n */\n\nimport {\n attributeCounterfactuals,\n type CounterfactualMutation,\n type CounterfactualResult,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport { confidenceInterval } from '../statistics'\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\n/** Stable reference to a trajectory step — carried through reports,\n * findings, and corpus records so evidence stays addressable. */\nexport interface StepRef {\n index: number\n spanId: string\n kind: Span['kind']\n name: string\n}\n\nexport function stepRefOf(step: TrajectoryStep): StepRef {\n return {\n index: step.index,\n spanId: step.span.spanId,\n kind: step.span.kind,\n name: step.span.name,\n }\n}\n\nexport interface CausalSweepOptions {\n store: TraceStore\n /** The failed run to diagnose. Its `outcome.score` is the baseline every\n * counterfactual delta is measured against. */\n runId: string\n /** Execution seam — identical contract to `runCounterfactual`: re-runs the\n * agent from the mutation point and MUST `endRun` with a numeric score. */\n runner: CounterfactualRunner\n /** Trajectory indices to probe. Default: every llm + tool span — the kinds\n * the existing `CounterfactualMutation` set targets. */\n candidateSteps?: number[]\n /**\n * Mutations to probe a given step with. Returned mutations MUST target\n * `step.index`. Default probes are the payload-free existing kinds:\n * - tool span → `swap-tool-result` with `newResult: null` (knockout:\n * how much did the run depend on this tool's information?)\n * - llm span → `truncate-after` (re-roll: how much did the realized\n * turn deviate from the policy's typical continuation?)\n * `swap-model` / `inject-system-message` need consumer payloads, so they\n * are opt-in via this callback.\n */\n mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[]\n /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */\n reps: number\n /** Hard cap on total counterfactual replays across the whole sweep. */\n budget: number\n /** Seed for the bootstrap CI resampler. Deterministic default so two\n * sweeps over the same deltas report identical intervals. */\n ciSeed?: number\n /** Bootstrap CI confidence level. Default 0.95. */\n ciConfidence?: number\n}\n\nexport interface StepResponsibility {\n stepRef: StepRef\n mutationKind: CounterfactualMutation['kind']\n /** Mean of per-rep score deltas (counterfactual − original). */\n meanEffect: number\n /** Bootstrap CI over the per-rep deltas. */\n ci: { mean: number; lower: number; upper: number }\n /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */\n ciExcludesZero: boolean\n reps: number\n /** Raw per-rep deltas — downstream evidence, never re-derived. */\n deltas: number[]\n /** Replay run ids (layer='meta', parentRunId=original) for audit. */\n counterfactualRunIds: string[]\n}\n\nexport interface CausalResponsibilityReport {\n runId: string\n originalScore: number\n /** Ranked by |meanEffect| descending — the blame ordering. */\n steps: StepResponsibility[]\n /** Kind-level aggregate from the existing `attributeCounterfactuals`. */\n byMutationKind: ReturnType<typeof attributeCounterfactuals>\n replaysUsed: number\n budget: number\n /** Steps planned but not fully probed before the budget ran out.\n * Named, never silent: an absent step is \"no effect found\"; an\n * uncovered step is \"not measured\". */\n uncovered: StepRef[]\n}\n\nconst DEFAULT_CI_SEED = 0x5eed\n\nfunction defaultMutations(step: TrajectoryStep): CounterfactualMutation[] {\n if (step.span.kind === 'tool') {\n return [{ kind: 'swap-tool-result', at: step.index, newResult: null }]\n }\n if (step.span.kind === 'llm') {\n return [{ kind: 'truncate-after', at: step.index }]\n }\n return []\n}\n\nexport async function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport> {\n if (!Number.isInteger(opts.reps) || opts.reps < 2) {\n throw new ValidationError(\n `causalSweep: reps must be an integer >= 2 (got ${opts.reps}) — a single-intervention delta is one stochastic draw, not a measurement`,\n )\n }\n if (!Number.isInteger(opts.budget) || opts.budget < 1) {\n throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`)\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `causalSweep: run ${opts.runId} has no numeric outcome.score — deltas have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n const candidates = resolveCandidates(trajectory, opts.candidateSteps)\n const mutationsFor = opts.mutationsPerStep ?? defaultMutations\n\n interface Cell {\n step: TrajectoryStep\n mutation: CounterfactualMutation\n }\n const cells: Cell[] = []\n for (const step of candidates) {\n const mutations = mutationsFor(step)\n for (const m of mutations) {\n if (m.at !== step.index) {\n throw new ValidationError(\n `causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} — mutations must target the step they were asked for`,\n )\n }\n cells.push({ step, mutation: m })\n }\n }\n\n const responsibilities: StepResponsibility[] = []\n const allResults: CounterfactualResult[] = []\n const uncoveredIndices = new Set<number>()\n let replaysUsed = 0\n let halted = false\n\n for (const cell of cells) {\n if (halted || replaysUsed + opts.reps > opts.budget) {\n // A partial cell would report a CI over fewer reps than requested —\n // weaker evidence masquerading as the real thing. Halt and name it.\n halted = true\n uncoveredIndices.add(cell.step.index)\n continue\n }\n const deltas: number[] = []\n const cfRunIds: string[] = []\n for (let rep = 0; rep < opts.reps; rep++) {\n const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner)\n replaysUsed++\n const d = result.delta.deltaScore\n if (typeof d !== 'number' || !Number.isFinite(d)) {\n throw new ValidationError(\n `causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`,\n )\n }\n deltas.push(d)\n cfRunIds.push(result.counterfactualRunId)\n allResults.push(result)\n }\n const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {\n seed: opts.ciSeed ?? DEFAULT_CI_SEED,\n })\n responsibilities.push({\n stepRef: stepRefOf(cell.step),\n mutationKind: cell.mutation.kind,\n meanEffect: ci.mean,\n ci,\n ciExcludesZero: ci.lower > 0 || ci.upper < 0,\n reps: opts.reps,\n deltas,\n counterfactualRunIds: cfRunIds,\n })\n }\n\n responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect))\n\n // A step probed under one mutation but cut off under another appears in\n // BOTH steps and uncovered — partial coverage is named, not blended.\n const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf)\n\n return {\n runId: opts.runId,\n originalScore,\n steps: responsibilities,\n byMutationKind: attributeCounterfactuals(allResults),\n replaysUsed,\n budget: opts.budget,\n uncovered,\n }\n}\n\nfunction resolveCandidates(trajectory: Trajectory, indices?: number[]): TrajectoryStep[] {\n if (indices === undefined) {\n return trajectory.steps.filter((s) => s.span.kind === 'llm' || s.span.kind === 'tool')\n }\n return indices.map((i) => {\n const step = trajectory.steps[i]\n if (!step) {\n throw new ValidationError(\n `causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`,\n )\n }\n return step\n })\n}\n","/**\n * Remediation adapters — HOW DO WE MAKE IT HAPPEN?\n *\n * The diagnose chain ends by feeding existing improvement machinery,\n * not by building new machinery:\n *\n * - `toAnalystFindings` → the analyst contract (`makeFinding`), so\n * responsibility evidence flows into the same registry / steering /\n * diff pipeline every other analyst feeds.\n * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the\n * diagnosed failure + validated repair as a permanent scenario.\n * - `suggestInvariant` → a plain-data hint in the shape the\n * trace-contracts machinery consumes (`never` / `without` clauses).\n */\n\nimport type { AnalystFinding, AnalystSeverity, EvidenceRef } from '../analyst/types'\nimport { makeFinding } from '../analyst/types'\nimport type { CounterfactualMutation } from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { CorpusRecord } from '../rl/corpus'\nimport type { RunRecord } from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CausalResponsibilityReport, StepResponsibility } from './causal-sweep'\nimport type { RepairReport, ValidatedRepair } from './repair'\n\nexport const DIAGNOSE_ANALYST_ID = 'diagnose-causal-sweep'\n\n/** Severity from causal effect size. Effects whose CI includes zero are\n * 'info' regardless of magnitude — an indistinguishable-from-noise effect\n * must not steer remediation priority. */\nexport function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity {\n if (!responsibility.ciExcludesZero) return 'info'\n const magnitude = Math.abs(responsibility.meanEffect)\n if (magnitude >= 0.5) return 'critical'\n if (magnitude >= 0.25) return 'high'\n if (magnitude >= 0.1) return 'medium'\n return 'low'\n}\n\n/** Deterministic human-readable rendering of a mutation — used in\n * recommended actions, corpus completions, and invariant hints. */\nexport function describeMutation(mutation: CounterfactualMutation): string {\n switch (mutation.kind) {\n case 'swap-model':\n return `use model '${mutation.newModel}' at step ${mutation.at}`\n case 'swap-tool-result':\n return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`\n case 'truncate-after':\n return `stop the run after step ${mutation.at}`\n case 'inject-system-message':\n return `inject system message at step ${mutation.at}: ${mutation.content}`\n case 'custom':\n return `${mutation.describe} (step ${mutation.at})`\n }\n}\n\n/**\n * Lift a responsibility report (and optionally its validated repairs) into\n * `AnalystFinding`s via the real `makeFinding` factory. One finding per\n * probed step; a validated repair for that step upgrades the finding with\n * a `recommended_action` + the replay-validation evidence.\n *\n * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,\n * so `derived_from_judge` stays unset and they may steer.\n */\nexport function toAnalystFindings(\n report: CausalResponsibilityReport,\n repairs?: RepairReport,\n): AnalystFinding[] {\n const repairByStep = new Map<string, ValidatedRepair>()\n for (const r of repairs?.repairs ?? []) {\n if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r)\n }\n\n return report.steps.map((resp) => {\n const repair = repairByStep.get(resp.stepRef.spanId)\n const evidence: EvidenceRef[] = [\n {\n kind: 'span',\n uri: `span://${resp.stepRef.spanId}`,\n excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`,\n },\n {\n kind: 'metric',\n uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,\n excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(', ')}]`,\n },\n ...resp.counterfactualRunIds.map((id): EvidenceRef => ({ kind: 'span', uri: `run://${id}` })),\n ]\n return makeFinding({\n analyst_id: DIAGNOSE_ANALYST_ID,\n severity: severityFromEffect(resp),\n area: 'causal-attribution',\n claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,\n rationale: resp.ciExcludesZero\n ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero`\n : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero — not distinguishable from noise`,\n evidence_refs: evidence,\n recommended_action: repair ? describeMutation(repair.mutation) : undefined,\n validation_plan: repair\n ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs!.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})`\n : undefined,\n confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,\n subject: resp.stepRef.spanId,\n metadata: {\n stepRef: resp.stepRef,\n mutationKind: resp.mutationKind,\n meanEffect: resp.meanEffect,\n ci: resp.ci,\n deltas: resp.deltas,\n counterfactualRunIds: resp.counterfactualRunIds,\n ...(repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}),\n },\n })\n })\n}\n\n/**\n * Pin the diagnosed failure as a permanent corpus scenario. Takes the\n * original run's `RunRecord` projection plus a validated repair and emits\n * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw\n * failure and the diagnosed entry).\n *\n * `completion` defaults to the validated mutation's rendering — \"what\n * should have happened\" in machine-derived form. Supply `prompt` (and\n * optionally a richer `completion`) when the trajectory text is available\n * so the record is harvestable by `buildDatasetFromCorpus`.\n */\nexport function toCorpusRecord(\n run: RunRecord,\n repair: ValidatedRepair,\n opts: { prompt?: string; completion?: string } = {},\n): CorpusRecord {\n const record: CorpusRecord = {\n ...run,\n runId: `${run.runId}#repair:${repair.stepRef.spanId}`,\n outcome: {\n ...run.outcome,\n raw: {\n ...run.outcome.raw,\n diagnose_blamed_step_index: repair.stepRef.index,\n diagnose_repair_mean_score: repair.meanScore,\n diagnose_repair_delta_score: repair.deltaScore,\n diagnose_repair_reps: repair.reps,\n },\n },\n prompt: opts.prompt,\n completion: opts.completion ?? describeMutation(repair.mutation),\n }\n // Boundary check — a corpus record that fails RunRecord validation would\n // poison every downstream harvest.\n validateRunRecord(record)\n return record\n}\n\n/** Plain-data invariant hint. The trace-contracts machinery consumes this\n * shape: `never` is a pattern that must not appear in a passing trace;\n * `without` is a guard whose absence makes the failure reachable. */\nexport interface InvariantHint {\n description: string\n never?: string\n without?: string\n}\n\n/**\n * Derive an invariant hint from a validated repair. Deterministic per\n * mutation kind — the hint names the contract a trace must satisfy so\n * the diagnosed failure cannot silently recur.\n */\nexport function suggestInvariant(repair: ValidatedRepair): InvariantHint {\n const { stepRef, mutation } = repair\n const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`\n switch (mutation.kind) {\n case 'swap-tool-result':\n return {\n description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `unvalidated result from tool '${stepRef.name}' flows downstream`,\n without: `result guard on tool '${stepRef.name}'`,\n }\n case 'swap-model':\n return {\n description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`,\n }\n case 'inject-system-message':\n return {\n description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n without: `system message present at '${stepRef.name}': ${mutation.content}`,\n }\n case 'truncate-after':\n return {\n description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) — continuation past this step caused the failure`,\n never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`,\n }\n case 'custom':\n return {\n description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n }\n default: {\n const exhausted: never = mutation\n throw new ValidationError(\n `suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`,\n )\n }\n }\n}\n","/**\n * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?\n *\n * Takes the blamed steps from a `CausalResponsibilityReport`, asks a\n * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate\n * mutations, and machine-verifies each candidate by replaying the run\n * WITH the mutation applied (through the same `runCounterfactual` seam\n * the sweep uses).\n *\n * A repair is \"what should have happened\" ONLY when every validation\n * replay crosses `flipThreshold` — a prescription is never speculated,\n * it is demonstrated. Candidates that don't flip, or whose replay\n * errors, land in `rejected` with a typed reason; nothing is dropped\n * silently.\n */\n\nimport {\n type CounterfactualMutation,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\nimport type { StepRef, StepResponsibility } from './causal-sweep'\n\n/** Context handed to `proposeFix` so an LLM-backed proposer can see the\n * full trajectory plus the responsibility evidence for the blamed step. */\nexport interface RepairContext {\n runId: string\n trajectory: Trajectory\n originalScore: number\n responsibility: StepResponsibility\n}\n\nexport interface PrescribeRepairOptions {\n store: TraceStore\n /** The failed run the sweep diagnosed. */\n runId: string\n /** Execution seam — same `CounterfactualRunner` contract as the sweep. */\n runner: CounterfactualRunner\n /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */\n blamed: StepResponsibility[]\n /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.\n * Returned mutations MUST target the blamed step's index. */\n proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>\n /** Score every validation replay must reach for the repair to count. Default 0.5. */\n flipThreshold?: number\n /** Validation replays per candidate mutation. Default 3. */\n repsToValidate?: number\n /** Max candidate mutations tried per step. Default: all proposed. */\n maxAttemptsPerStep?: number\n}\n\nexport interface ValidatedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n /** Always true — presence in `repairs` IS the machine-verified claim. */\n validated: true\n /** Mean counterfactual score across the validation reps. */\n meanScore: number\n /** meanScore − originalScore. */\n deltaScore: number\n reps: number\n /** Replay run ids backing the validation — audit trail. */\n counterfactualRunIds: string[]\n}\n\nexport interface RejectedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n reason: 'did-not-flip' | 'error'\n /** Present for 'did-not-flip': mean delta over the reps that ran. */\n deltaScore?: number\n /** Present for 'error': the message, preserved for diagnosis. */\n error?: string\n}\n\nexport interface RepairReport {\n runId: string\n originalScore: number\n flipThreshold: number\n repairs: ValidatedRepair[]\n rejected: RejectedRepair[]\n replaysUsed: number\n}\n\nexport async function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport> {\n const flipThreshold = opts.flipThreshold ?? 0.5\n const repsToValidate = opts.repsToValidate ?? 3\n if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {\n throw new ValidationError(\n `prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`,\n )\n }\n const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY\n if (maxAttempts < 1) {\n throw new ValidationError(\n `prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`,\n )\n }\n if (opts.blamed.length === 0) {\n throw new ValidationError('prescribeRepair: blamed is empty — nothing to repair')\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `prescribeRepair: run ${opts.runId} has no numeric outcome.score — flips have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n\n const repairs: ValidatedRepair[] = []\n const rejected: RejectedRepair[] = []\n let replaysUsed = 0\n\n for (const responsibility of opts.blamed) {\n const step = trajectory.steps[responsibility.stepRef.index]\n if (!step || step.span.spanId !== responsibility.stepRef.spanId) {\n throw new ValidationError(\n `prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} — stale report?`,\n )\n }\n\n const candidates = await opts.proposeFix(step, {\n runId: opts.runId,\n trajectory,\n originalScore,\n responsibility,\n })\n const toTry = candidates.slice(0, maxAttempts)\n\n for (const mutation of toTry) {\n if (mutation.at !== step.index) {\n throw new ValidationError(\n `prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`,\n )\n }\n const scores: number[] = []\n const cfRunIds: string[] = []\n let failure: string | undefined\n for (let rep = 0; rep < repsToValidate; rep++) {\n try {\n const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner)\n replaysUsed++\n const score = result.delta.counterfactualOutcomeScore\n if (typeof score !== 'number' || !Number.isFinite(score)) {\n failure = `validation rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`\n break\n }\n scores.push(score)\n cfRunIds.push(result.counterfactualRunId)\n } catch (err) {\n replaysUsed++\n failure = err instanceof Error ? err.message : String(err)\n break\n }\n }\n\n if (failure !== undefined) {\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'error',\n error: failure,\n })\n continue\n }\n\n const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length\n const everyRepFlipped = scores.every((s) => s >= flipThreshold)\n if (everyRepFlipped) {\n repairs.push({\n stepRef: responsibility.stepRef,\n mutation,\n validated: true,\n meanScore,\n deltaScore: meanScore - originalScore,\n reps: repsToValidate,\n counterfactualRunIds: cfRunIds,\n })\n // First validated repair per step IS the prescription; remaining\n // candidates are untried, not rejected — we don't fabricate verdicts.\n break\n }\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'did-not-flip',\n deltaScore: meanScore - originalScore,\n })\n }\n }\n\n return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;AA2CO,SAAS,UAAU,MAA+B;AACvD,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK,KAAK;AAAA,IAClB,MAAM,KAAK,KAAK;AAAA,IAChB,MAAM,KAAK,KAAK;AAAA,EAClB;AACF;AAkEA,IAAM,kBAAkB;AAExB,SAAS,iBAAiB,MAAgD;AACxE,MAAI,KAAK,KAAK,SAAS,QAAQ;AAC7B,WAAO,CAAC,EAAE,MAAM,oBAAoB,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC;AAAA,EACvE;AACA,MAAI,KAAK,KAAK,SAAS,OAAO;AAC5B,WAAO,CAAC,EAAE,MAAM,kBAAkB,IAAI,KAAK,MAAM,CAAC;AAAA,EACpD;AACA,SAAO,CAAC;AACV;AAEA,eAAsB,YAAY,MAA+D;AAC/F,MAAI,CAAC,OAAO,UAAU,KAAK,IAAI,KAAK,KAAK,OAAO,GAAG;AACjD,UAAM,IAAI;AAAA,MACR,kDAAkD,KAAK,IAAI;AAAA,IAC7D;AAAA,EACF;AACA,MAAI,CAAC,OAAO,UAAU,KAAK,MAAM,KAAK,KAAK,SAAS,GAAG;AACrD,UAAM,IAAI,gBAAgB,oDAAoD,KAAK,MAAM,GAAG;AAAA,EAC9F;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,oBAAoB,KAAK,KAAK,YAAY;AACtF,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,oBAAoB,KAAK,KAAK;AAAA,IAChC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAC/D,QAAM,aAAa,kBAAkB,YAAY,KAAK,cAAc;AACpE,QAAM,eAAe,KAAK,oBAAoB;AAM9C,QAAM,QAAgB,CAAC;AACvB,aAAW,QAAQ,YAAY;AAC7B,UAAM,YAAY,aAAa,IAAI;AACnC,eAAW,KAAK,WAAW;AACzB,UAAI,EAAE,OAAO,KAAK,OAAO;AACvB,cAAM,IAAI;AAAA,UACR,kEAAkE,EAAE,EAAE,mBAAmB,KAAK,KAAK;AAAA,QACrG;AAAA,MACF;AACA,YAAM,KAAK,EAAE,MAAM,UAAU,EAAE,CAAC;AAAA,IAClC;AAAA,EACF;AAEA,QAAM,mBAAyC,CAAC;AAChD,QAAM,aAAqC,CAAC;AAC5C,QAAM,mBAAmB,oBAAI,IAAY;AACzC,MAAI,cAAc;AAClB,MAAI,SAAS;AAEb,aAAW,QAAQ,OAAO;AACxB,QAAI,UAAU,cAAc,KAAK,OAAO,KAAK,QAAQ;AAGnD,eAAS;AACT,uBAAiB,IAAI,KAAK,KAAK,KAAK;AACpC;AAAA,IACF;AACA,UAAM,SAAmB,CAAC;AAC1B,UAAM,WAAqB,CAAC;AAC5B,aAAS,MAAM,GAAG,MAAM,KAAK,MAAM,OAAO;AACxC,YAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,KAAK,UAAU,KAAK,MAAM;AACzF;AACA,YAAM,IAAI,OAAO,MAAM;AACvB,UAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;AAChD,cAAM,IAAI;AAAA,UACR,+CAA+C,KAAK,KAAK,KAAK,KAAK,KAAK,SAAS,IAAI,SAAS,GAAG;AAAA,QACnG;AAAA,MACF;AACA,aAAO,KAAK,CAAC;AACb,eAAS,KAAK,OAAO,mBAAmB;AACxC,iBAAW,KAAK,MAAM;AAAA,IACxB;AACA,UAAM,KAAK,mBAAmB,QAAQ,KAAK,gBAAgB,MAAM;AAAA,MAC/D,MAAM,KAAK,UAAU;AAAA,IACvB,CAAC;AACD,qBAAiB,KAAK;AAAA,MACpB,SAAS,UAAU,KAAK,IAAI;AAAA,MAC5B,cAAc,KAAK,SAAS;AAAA,MAC5B,YAAY,GAAG;AAAA,MACf;AAAA,MACA,gBAAgB,GAAG,QAAQ,KAAK,GAAG,QAAQ;AAAA,MAC3C,MAAM,KAAK;AAAA,MACX;AAAA,MACA,sBAAsB;AAAA,IACxB,CAAC;AAAA,EACH;AAEA,mBAAiB,KAAK,CAAC,GAAG,MAAM,KAAK,IAAI,EAAE,UAAU,IAAI,KAAK,IAAI,EAAE,UAAU,CAAC;AAI/E,QAAM,YAAY,WAAW,OAAO,CAAC,MAAM,iBAAiB,IAAI,EAAE,KAAK,CAAC,EAAE,IAAI,SAAS;AAEvF,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ;AAAA,IACA,OAAO;AAAA,IACP,gBAAgB,yBAAyB,UAAU;AAAA,IACnD;AAAA,IACA,QAAQ,KAAK;AAAA,IACb;AAAA,EACF;AACF;AAEA,SAAS,kBAAkB,YAAwB,SAAsC;AACvF,MAAI,YAAY,QAAW;AACzB,WAAO,WAAW,MAAM,OAAO,CAAC,MAAM,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,MAAM;AAAA,EACvF;AACA,SAAO,QAAQ,IAAI,CAAC,MAAM;AACxB,UAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,QAAI,CAAC,MAAM;AACT,YAAM,IAAI;AAAA,QACR,qCAAqC,CAAC,qBAAqB,WAAW,MAAM,MAAM;AAAA,MACpF;AAAA,IACF;AACA,WAAO;AAAA,EACT,CAAC;AACH;;;ACzNO,IAAM,sBAAsB;AAK5B,SAAS,mBAAmB,gBAAqD;AACtF,MAAI,CAAC,eAAe,eAAgB,QAAO;AAC3C,QAAM,YAAY,KAAK,IAAI,eAAe,UAAU;AACpD,MAAI,aAAa,IAAK,QAAO;AAC7B,MAAI,aAAa,KAAM,QAAO;AAC9B,MAAI,aAAa,IAAK,QAAO;AAC7B,SAAO;AACT;AAIO,SAAS,iBAAiB,UAA0C;AACzE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO,cAAc,SAAS,QAAQ,aAAa,SAAS,EAAE;AAAA,IAChE,KAAK;AACH,aAAO,mCAAmC,SAAS,EAAE,SAAS,KAAK,UAAU,SAAS,SAAS,CAAC;AAAA,IAClG,KAAK;AACH,aAAO,2BAA2B,SAAS,EAAE;AAAA,IAC/C,KAAK;AACH,aAAO,iCAAiC,SAAS,EAAE,KAAK,SAAS,OAAO;AAAA,IAC1E,KAAK;AACH,aAAO,GAAG,SAAS,QAAQ,UAAU,SAAS,EAAE;AAAA,EACpD;AACF;AAWO,SAAS,kBACd,QACA,SACkB;AAClB,QAAM,eAAe,oBAAI,IAA6B;AACtD,aAAW,KAAK,SAAS,WAAW,CAAC,GAAG;AACtC,QAAI,CAAC,aAAa,IAAI,EAAE,QAAQ,MAAM,EAAG,cAAa,IAAI,EAAE,QAAQ,QAAQ,CAAC;AAAA,EAC/E;AAEA,SAAO,OAAO,MAAM,IAAI,CAAC,SAAS;AAChC,UAAM,SAAS,aAAa,IAAI,KAAK,QAAQ,MAAM;AACnD,UAAM,WAA0B;AAAA,MAC9B;AAAA,QACE,MAAM;AAAA,QACN,KAAK,UAAU,KAAK,QAAQ,MAAM;AAAA,QAClC,SAAS,QAAQ,KAAK,QAAQ,KAAK,KAAK,KAAK,QAAQ,IAAI,KAAK,KAAK,QAAQ,IAAI,iBAAiB,KAAK,WAAW,QAAQ,CAAC,CAAC,QAAQ,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,UAAU,KAAK,IAAI;AAAA,MAC5M;AAAA,MACA;AAAA,QACE,MAAM;AAAA,QACN,KAAK,qBAAqB,OAAO,KAAK,SAAS,KAAK,QAAQ,KAAK,IAAI,KAAK,YAAY;AAAA,QACtF,SAAS,WAAW,KAAK,OAAO,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC,EAAE,KAAK,IAAI,CAAC;AAAA,MACrE;AAAA,MACA,GAAG,KAAK,qBAAqB,IAAI,CAAC,QAAqB,EAAE,MAAM,QAAQ,KAAK,SAAS,EAAE,GAAG,EAAE;AAAA,IAC9F;AACA,WAAO,YAAY;AAAA,MACjB,YAAY;AAAA,MACZ,UAAU,mBAAmB,IAAI;AAAA,MACjC,MAAM;AAAA,MACN,OAAO,SAAS,KAAK,QAAQ,IAAI,MAAM,KAAK,QAAQ,IAAI,uDAAuD,KAAK,YAAY;AAAA,MAChI,WAAW,KAAK,iBACZ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,oBAChJ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC;AAAA,MACpJ,eAAe;AAAA,MACf,oBAAoB,SAAS,iBAAiB,OAAO,QAAQ,IAAI;AAAA,MACjE,iBAAiB,SACb,qBAAqB,OAAO,IAAI,IAAI,OAAO,IAAI,mBAAmB,QAAS,aAAa,UAAU,OAAO,UAAU,QAAQ,CAAC,CAAC,WAAW,OAAO,WAAW,QAAQ,CAAC,CAAC,MACpK;AAAA,MACJ,YAAY,SAAS,OAAO,KAAK,iBAAiB,OAAO;AAAA,MACzD,SAAS,KAAK,QAAQ;AAAA,MACtB,UAAU;AAAA,QACR,SAAS,KAAK;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,YAAY,KAAK;AAAA,QACjB,IAAI,KAAK;AAAA,QACT,QAAQ,KAAK;AAAA,QACb,sBAAsB,KAAK;AAAA,QAC3B,GAAI,SAAS,EAAE,QAAQ,EAAE,UAAU,OAAO,UAAU,WAAW,OAAO,UAAU,EAAE,IAAI,CAAC;AAAA,MACzF;AAAA,IACF,CAAC;AAAA,EACH,CAAC;AACH;AAaO,SAAS,eACd,KACA,QACA,OAAiD,CAAC,GACpC;AACd,QAAM,SAAuB;AAAA,IAC3B,GAAG;AAAA,IACH,OAAO,GAAG,IAAI,KAAK,WAAW,OAAO,QAAQ,MAAM;AAAA,IACnD,SAAS;AAAA,MACP,GAAG,IAAI;AAAA,MACP,KAAK;AAAA,QACH,GAAG,IAAI,QAAQ;AAAA,QACf,4BAA4B,OAAO,QAAQ;AAAA,QAC3C,4BAA4B,OAAO;AAAA,QACnC,6BAA6B,OAAO;AAAA,QACpC,sBAAsB,OAAO;AAAA,MAC/B;AAAA,IACF;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,YAAY,KAAK,cAAc,iBAAiB,OAAO,QAAQ;AAAA,EACjE;AAGA,oBAAkB,MAAM;AACxB,SAAO;AACT;AAgBO,SAAS,iBAAiB,QAAwC;AACvE,QAAM,EAAE,SAAS,SAAS,IAAI;AAC9B,QAAM,KAAK,QAAQ,QAAQ,KAAK,KAAK,QAAQ,IAAI,KAAK,QAAQ,IAAI;AAClE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO;AAAA,QACL,aAAa,uBAAuB,QAAQ,IAAI,4FAA4F,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QACxK,OAAO,iCAAiC,QAAQ,IAAI;AAAA,QACpD,SAAS,yBAAyB,QAAQ,IAAI;AAAA,MAChD;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,yBAAyB,EAAE,QAAQ,SAAS,QAAQ,gCAAgC,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC7H,OAAO,aAAa,QAAQ,IAAI,iCAAiC,SAAS,QAAQ;AAAA,MACpF;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,iCAAiC,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC3G,SAAS,8BAA8B,QAAQ,IAAI,MAAM,SAAS,OAAO;AAAA,MAC3E;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,kBAAkB,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC5F,OAAO,wBAAwB,QAAQ,IAAI,YAAY,QAAQ,KAAK;AAAA,MACtE;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,GAAG,SAAS,QAAQ,OAAO,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,MACvG;AAAA,IACF,SAAS;AACP,YAAM,YAAmB;AACzB,YAAM,IAAI;AAAA,QACR,2CAA2C,KAAK,UAAU,SAAS,CAAC;AAAA,MACtE;AAAA,IACF;AAAA,EACF;AACF;;;ACtHA,eAAsB,gBAAgB,MAAqD;AACzF,QAAM,gBAAgB,KAAK,iBAAiB;AAC5C,QAAM,iBAAiB,KAAK,kBAAkB;AAC9C,MAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GAAG;AAC3D,UAAM,IAAI;AAAA,MACR,gEAAgE,cAAc;AAAA,IAChF;AAAA,EACF;AACA,QAAM,cAAc,KAAK,sBAAsB,OAAO;AACtD,MAAI,cAAc,GAAG;AACnB,UAAM,IAAI;AAAA,MACR,yDAAyD,KAAK,kBAAkB;AAAA,IAClF;AAAA,EACF;AACA,MAAI,KAAK,OAAO,WAAW,GAAG;AAC5B,UAAM,IAAI,gBAAgB,2DAAsD;AAAA,EAClF;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,wBAAwB,KAAK,KAAK,YAAY;AAC1F,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,wBAAwB,KAAK,KAAK;AAAA,IACpC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAE/D,QAAM,UAA6B,CAAC;AACpC,QAAM,WAA6B,CAAC;AACpC,MAAI,cAAc;AAElB,aAAW,kBAAkB,KAAK,QAAQ;AACxC,UAAM,OAAO,WAAW,MAAM,eAAe,QAAQ,KAAK;AAC1D,QAAI,CAAC,QAAQ,KAAK,KAAK,WAAW,eAAe,QAAQ,QAAQ;AAC/D,YAAM,IAAI;AAAA,QACR,sCAAsC,eAAe,QAAQ,KAAK,WAAW,eAAe,QAAQ,MAAM,uBAAuB,KAAK,KAAK;AAAA,MAC7I;AAAA,IACF;AAEA,UAAM,aAAa,MAAM,KAAK,WAAW,MAAM;AAAA,MAC7C,OAAO,KAAK;AAAA,MACZ;AAAA,MACA;AAAA,MACA;AAAA,IACF,CAAC;AACD,UAAM,QAAQ,WAAW,MAAM,GAAG,WAAW;AAE7C,eAAW,YAAY,OAAO;AAC5B,UAAI,SAAS,OAAO,KAAK,OAAO;AAC9B,cAAM,IAAI;AAAA,UACR,gEAAgE,SAAS,EAAE,0BAA0B,KAAK,KAAK;AAAA,QACjH;AAAA,MACF;AACA,YAAM,SAAmB,CAAC;AAC1B,YAAM,WAAqB,CAAC;AAC5B,UAAI;AACJ,eAAS,MAAM,GAAG,MAAM,gBAAgB,OAAO;AAC7C,YAAI;AACF,gBAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,UAAU,KAAK,MAAM;AACpF;AACA,gBAAM,QAAQ,OAAO,MAAM;AAC3B,cAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;AACxD,sBAAU,kBAAkB,GAAG;AAC/B;AAAA,UACF;AACA,iBAAO,KAAK,KAAK;AACjB,mBAAS,KAAK,OAAO,mBAAmB;AAAA,QAC1C,SAAS,KAAK;AACZ;AACA,oBAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AACzD;AAAA,QACF;AAAA,MACF;AAEA,UAAI,YAAY,QAAW;AACzB,iBAAS,KAAK;AAAA,UACZ,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,QAAQ;AAAA,UACR,OAAO;AAAA,QACT,CAAC;AACD;AAAA,MACF;AAEA,YAAM,YAAY,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;AAC7D,YAAM,kBAAkB,OAAO,MAAM,CAAC,MAAM,KAAK,aAAa;AAC9D,UAAI,iBAAiB;AACnB,gBAAQ,KAAK;AAAA,UACX,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,WAAW;AAAA,UACX;AAAA,UACA,YAAY,YAAY;AAAA,UACxB,MAAM;AAAA,UACN,sBAAsB;AAAA,QACxB,CAAC;AAGD;AAAA,MACF;AACA,eAAS,KAAK;AAAA,QACZ,SAAS,eAAe;AAAA,QACxB;AAAA,QACA,QAAQ;AAAA,QACR,YAAY,YAAY;AAAA,MAC1B,CAAC;AAAA,IACH;AAAA,EACF;AAEA,SAAO,EAAE,OAAO,KAAK,OAAO,eAAe,eAAe,SAAS,UAAU,YAAY;AAC3F;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../src/diagnose/causal-sweep.ts","../src/diagnose/remediation.ts","../src/diagnose/repair.ts"],"sourcesContent":["/**\n * Causal sweep — WHY did this run fail?\n *\n * Orchestrates the dormant counterfactual primitives into a responsibility\n * report: for each candidate step, run `reps` counterfactual replays per\n * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`\n * is the execution seam) and reduce the per-rep score deltas into a mean\n * effect + bootstrap confidence interval (via `confidenceInterval`).\n *\n * Why `reps` is REQUIRED: a single intervention delta is one stochastic\n * draw — LLM re-execution from a prefix is sampled, so one replay cannot\n * distinguish \"this step caused the failure\" from sampling noise. The\n * signal is the distribution of deltas across reps; the CI over that\n * distribution is what lets a caller say \"this step's effect excludes\n * zero\" instead of eyeballing a point estimate.\n *\n * Budget discipline: the sweep never silently drops cells. When the\n * remaining budget cannot fund a full `reps`-sized cell, the sweep halts\n * and every step not fully probed is named in `uncovered`.\n */\n\nimport {\n attributeCounterfactuals,\n type CounterfactualMutation,\n type CounterfactualResult,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport { confidenceInterval } from '../statistics'\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\n/** Stable reference to a trajectory step — carried through reports,\n * findings, and corpus records so evidence stays addressable. */\nexport interface StepRef {\n index: number\n spanId: string\n kind: Span['kind']\n name: string\n}\n\nexport function stepRefOf(step: TrajectoryStep): StepRef {\n return {\n index: step.index,\n spanId: step.span.spanId,\n kind: step.span.kind,\n name: step.span.name,\n }\n}\n\nexport interface CausalSweepOptions {\n store: TraceStore\n /** The failed run to diagnose. Its `outcome.score` is the baseline every\n * counterfactual delta is measured against. */\n runId: string\n /** Execution seam — identical contract to `runCounterfactual`: re-runs the\n * agent from the mutation point and MUST `endRun` with a numeric score. */\n runner: CounterfactualRunner\n /** Trajectory indices to probe. Default: every llm + tool span — the kinds\n * the existing `CounterfactualMutation` set targets. */\n candidateSteps?: number[]\n /**\n * Mutations to probe a given step with. Returned mutations MUST target\n * `step.index`. Default probes are the payload-free existing kinds:\n * - tool span → `swap-tool-result` with `newResult: null` (knockout:\n * how much did the run depend on this tool's information?)\n * - llm span → `truncate-after` (re-roll: how much did the realized\n * turn deviate from the policy's typical continuation?)\n * `swap-model` / `inject-system-message` need consumer payloads, so they\n * are opt-in via this callback.\n */\n mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[]\n /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */\n reps: number\n /** Hard cap on total counterfactual replays across the whole sweep. */\n budget: number\n /** Seed for the bootstrap CI resampler. Deterministic default so two\n * sweeps over the same deltas report identical intervals. */\n ciSeed?: number\n /** Bootstrap CI confidence level. Default 0.95. */\n ciConfidence?: number\n}\n\nexport interface StepResponsibility {\n stepRef: StepRef\n mutationKind: CounterfactualMutation['kind']\n /** Mean of per-rep score deltas (counterfactual − original). */\n meanEffect: number\n /** Bootstrap CI over the per-rep deltas. */\n ci: { mean: number; lower: number; upper: number }\n /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */\n ciExcludesZero: boolean\n reps: number\n /** Raw per-rep deltas — downstream evidence, never re-derived. */\n deltas: number[]\n /** Replay run ids (layer='meta', parentRunId=original) for audit. */\n counterfactualRunIds: string[]\n}\n\nexport interface CausalResponsibilityReport {\n runId: string\n originalScore: number\n /** Ranked by |meanEffect| descending — the blame ordering. */\n steps: StepResponsibility[]\n /** Kind-level aggregate from the existing `attributeCounterfactuals`. */\n byMutationKind: ReturnType<typeof attributeCounterfactuals>\n replaysUsed: number\n budget: number\n /** Steps planned but not fully probed before the budget ran out.\n * Named, never silent: an absent step is \"no effect found\"; an\n * uncovered step is \"not measured\". */\n uncovered: StepRef[]\n}\n\nconst DEFAULT_CI_SEED = 0x5eed\n\nfunction defaultMutations(step: TrajectoryStep): CounterfactualMutation[] {\n if (step.span.kind === 'tool') {\n return [{ kind: 'swap-tool-result', at: step.index, newResult: null }]\n }\n if (step.span.kind === 'llm') {\n return [{ kind: 'truncate-after', at: step.index }]\n }\n return []\n}\n\nexport async function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport> {\n if (!Number.isInteger(opts.reps) || opts.reps < 2) {\n throw new ValidationError(\n `causalSweep: reps must be an integer >= 2 (got ${opts.reps}) — a single-intervention delta is one stochastic draw, not a measurement`,\n )\n }\n if (!Number.isInteger(opts.budget) || opts.budget < 1) {\n throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`)\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `causalSweep: run ${opts.runId} has no numeric outcome.score — deltas have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n const candidates = resolveCandidates(trajectory, opts.candidateSteps)\n const mutationsFor = opts.mutationsPerStep ?? defaultMutations\n\n interface Cell {\n step: TrajectoryStep\n mutation: CounterfactualMutation\n }\n const cells: Cell[] = []\n for (const step of candidates) {\n const mutations = mutationsFor(step)\n for (const m of mutations) {\n if (m.at !== step.index) {\n throw new ValidationError(\n `causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} — mutations must target the step they were asked for`,\n )\n }\n cells.push({ step, mutation: m })\n }\n }\n\n const responsibilities: StepResponsibility[] = []\n const allResults: CounterfactualResult[] = []\n const uncoveredIndices = new Set<number>()\n let replaysUsed = 0\n let halted = false\n\n for (const cell of cells) {\n if (halted || replaysUsed + opts.reps > opts.budget) {\n // A partial cell would report a CI over fewer reps than requested —\n // weaker evidence masquerading as the real thing. Halt and name it.\n halted = true\n uncoveredIndices.add(cell.step.index)\n continue\n }\n const deltas: number[] = []\n const cfRunIds: string[] = []\n for (let rep = 0; rep < opts.reps; rep++) {\n const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner)\n replaysUsed++\n const d = result.delta.deltaScore\n if (typeof d !== 'number' || !Number.isFinite(d)) {\n throw new ValidationError(\n `causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`,\n )\n }\n deltas.push(d)\n cfRunIds.push(result.counterfactualRunId)\n allResults.push(result)\n }\n const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {\n seed: opts.ciSeed ?? DEFAULT_CI_SEED,\n })\n responsibilities.push({\n stepRef: stepRefOf(cell.step),\n mutationKind: cell.mutation.kind,\n meanEffect: ci.mean,\n ci,\n ciExcludesZero: ci.lower > 0 || ci.upper < 0,\n reps: opts.reps,\n deltas,\n counterfactualRunIds: cfRunIds,\n })\n }\n\n responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect))\n\n // A step probed under one mutation but cut off under another appears in\n // BOTH steps and uncovered — partial coverage is named, not blended.\n const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf)\n\n return {\n runId: opts.runId,\n originalScore,\n steps: responsibilities,\n byMutationKind: attributeCounterfactuals(allResults),\n replaysUsed,\n budget: opts.budget,\n uncovered,\n }\n}\n\nfunction resolveCandidates(trajectory: Trajectory, indices?: number[]): TrajectoryStep[] {\n if (indices === undefined) {\n return trajectory.steps.filter((s) => s.span.kind === 'llm' || s.span.kind === 'tool')\n }\n return indices.map((i) => {\n const step = trajectory.steps[i]\n if (!step) {\n throw new ValidationError(\n `causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`,\n )\n }\n return step\n })\n}\n","/**\n * Remediation adapters — HOW DO WE MAKE IT HAPPEN?\n *\n * The diagnose chain ends by feeding existing improvement machinery,\n * not by building new machinery:\n *\n * - `toAnalystFindings` → the analyst contract (`makeFinding`), so\n * responsibility evidence flows into the same registry / steering /\n * diff pipeline every other analyst feeds.\n * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the\n * diagnosed failure + validated repair as a permanent scenario.\n * - `suggestInvariant` → a plain-data hint in the shape the\n * trace-contracts machinery consumes (`never` / `without` clauses).\n */\n\nimport type { AnalystFinding, AnalystSeverity, EvidenceRef } from '../analyst/types'\nimport { makeFinding } from '../analyst/types'\nimport type { CounterfactualMutation } from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { CorpusRecord } from '../rl/corpus'\nimport type { RunRecord } from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CausalResponsibilityReport, StepResponsibility } from './causal-sweep'\nimport type { RepairReport, ValidatedRepair } from './repair'\n\nexport const DIAGNOSE_ANALYST_ID = 'diagnose-causal-sweep'\n\n/** Severity from causal effect size. Effects whose CI includes zero are\n * 'info' regardless of magnitude — an indistinguishable-from-noise effect\n * must not steer remediation priority. */\nexport function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity {\n if (!responsibility.ciExcludesZero) return 'info'\n const magnitude = Math.abs(responsibility.meanEffect)\n if (magnitude >= 0.5) return 'critical'\n if (magnitude >= 0.25) return 'high'\n if (magnitude >= 0.1) return 'medium'\n return 'low'\n}\n\n/** Deterministic human-readable rendering of a mutation — used in\n * recommended actions, corpus completions, and invariant hints. */\nexport function describeMutation(mutation: CounterfactualMutation): string {\n switch (mutation.kind) {\n case 'swap-model':\n return `use model '${mutation.newModel}' at step ${mutation.at}`\n case 'swap-tool-result':\n return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`\n case 'truncate-after':\n return `stop the run after step ${mutation.at}`\n case 'inject-system-message':\n return `inject system message at step ${mutation.at}: ${mutation.content}`\n case 'custom':\n return `${mutation.describe} (step ${mutation.at})`\n }\n}\n\n/**\n * Lift a responsibility report (and optionally its validated repairs) into\n * `AnalystFinding`s via the real `makeFinding` factory. One finding per\n * probed step; a validated repair for that step upgrades the finding with\n * a `recommended_action` + the replay-validation evidence.\n *\n * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,\n * so `derived_from_judge` stays unset and they may steer.\n */\nexport function toAnalystFindings(\n report: CausalResponsibilityReport,\n repairs?: RepairReport,\n): AnalystFinding[] {\n const repairByStep = new Map<string, ValidatedRepair>()\n for (const r of repairs?.repairs ?? []) {\n if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r)\n }\n\n return report.steps.map((resp) => {\n const repair = repairByStep.get(resp.stepRef.spanId)\n const evidence: EvidenceRef[] = [\n {\n kind: 'span',\n uri: `span://${resp.stepRef.spanId}`,\n excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`,\n },\n {\n kind: 'metric',\n uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,\n excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(', ')}]`,\n },\n ...resp.counterfactualRunIds.map((id): EvidenceRef => ({ kind: 'span', uri: `run://${id}` })),\n ]\n return makeFinding({\n analyst_id: DIAGNOSE_ANALYST_ID,\n severity: severityFromEffect(resp),\n area: 'causal-attribution',\n claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,\n rationale: resp.ciExcludesZero\n ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero`\n : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero — not distinguishable from noise`,\n evidence_refs: evidence,\n recommended_action: repair ? describeMutation(repair.mutation) : undefined,\n validation_plan: repair\n ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs!.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})`\n : undefined,\n confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,\n subject: resp.stepRef.spanId,\n metadata: {\n stepRef: resp.stepRef,\n mutationKind: resp.mutationKind,\n meanEffect: resp.meanEffect,\n ci: resp.ci,\n deltas: resp.deltas,\n counterfactualRunIds: resp.counterfactualRunIds,\n ...(repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}),\n },\n })\n })\n}\n\n/**\n * Pin the diagnosed failure as a permanent corpus scenario. Takes the\n * original run's `RunRecord` projection plus a validated repair and emits\n * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw\n * failure and the diagnosed entry).\n *\n * `completion` defaults to the validated mutation's rendering — \"what\n * should have happened\" in machine-derived form. Supply `prompt` (and\n * optionally a richer `completion`) when the trajectory text is available\n * so the record is harvestable by `buildDatasetFromCorpus`.\n */\nexport function toCorpusRecord(\n run: RunRecord,\n repair: ValidatedRepair,\n opts: { prompt?: string; completion?: string } = {},\n): CorpusRecord {\n const record: CorpusRecord = {\n ...run,\n runId: `${run.runId}#repair:${repair.stepRef.spanId}`,\n outcome: {\n ...run.outcome,\n raw: {\n ...run.outcome.raw,\n diagnose_blamed_step_index: repair.stepRef.index,\n diagnose_repair_mean_score: repair.meanScore,\n diagnose_repair_delta_score: repair.deltaScore,\n diagnose_repair_reps: repair.reps,\n },\n },\n prompt: opts.prompt,\n completion: opts.completion ?? describeMutation(repair.mutation),\n }\n // Boundary check — a corpus record that fails RunRecord validation would\n // poison every downstream harvest.\n validateRunRecord(record)\n return record\n}\n\n/** Plain-data invariant hint. The trace-contracts machinery consumes this\n * shape: `never` is a pattern that must not appear in a passing trace;\n * `without` is a guard whose absence makes the failure reachable. */\nexport interface InvariantHint {\n description: string\n never?: string\n without?: string\n}\n\n/**\n * Derive an invariant hint from a validated repair. Deterministic per\n * mutation kind — the hint names the contract a trace must satisfy so\n * the diagnosed failure cannot silently recur.\n */\nexport function suggestInvariant(repair: ValidatedRepair): InvariantHint {\n const { stepRef, mutation } = repair\n const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`\n switch (mutation.kind) {\n case 'swap-tool-result':\n return {\n description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `unvalidated result from tool '${stepRef.name}' flows downstream`,\n without: `result guard on tool '${stepRef.name}'`,\n }\n case 'swap-model':\n return {\n description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`,\n }\n case 'inject-system-message':\n return {\n description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n without: `system message present at '${stepRef.name}': ${mutation.content}`,\n }\n case 'truncate-after':\n return {\n description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) — continuation past this step caused the failure`,\n never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`,\n }\n case 'custom':\n return {\n description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n }\n default: {\n const exhausted: never = mutation\n throw new ValidationError(\n `suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`,\n )\n }\n }\n}\n","/**\n * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?\n *\n * Takes the blamed steps from a `CausalResponsibilityReport`, asks a\n * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate\n * mutations, and machine-verifies each candidate by replaying the run\n * WITH the mutation applied (through the same `runCounterfactual` seam\n * the sweep uses).\n *\n * A repair is \"what should have happened\" ONLY when every validation\n * replay crosses `flipThreshold` — a prescription is never speculated,\n * it is demonstrated. Candidates that don't flip, or whose replay\n * errors, land in `rejected` with a typed reason; nothing is dropped\n * silently.\n */\n\nimport {\n type CounterfactualMutation,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\nimport type { StepRef, StepResponsibility } from './causal-sweep'\n\n/** Context handed to `proposeFix` so an LLM-backed proposer can see the\n * full trajectory plus the responsibility evidence for the blamed step. */\nexport interface RepairContext {\n runId: string\n trajectory: Trajectory\n originalScore: number\n responsibility: StepResponsibility\n}\n\nexport interface PrescribeRepairOptions {\n store: TraceStore\n /** The failed run the sweep diagnosed. */\n runId: string\n /** Execution seam — same `CounterfactualRunner` contract as the sweep. */\n runner: CounterfactualRunner\n /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */\n blamed: StepResponsibility[]\n /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.\n * Returned mutations MUST target the blamed step's index. */\n proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>\n /** Score every validation replay must reach for the repair to count. Default 0.5. */\n flipThreshold?: number\n /** Validation replays per candidate mutation. Default 3. */\n repsToValidate?: number\n /** Max candidate mutations tried per step. Default: all proposed. */\n maxAttemptsPerStep?: number\n}\n\nexport interface ValidatedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n /** Always true — presence in `repairs` IS the machine-verified claim. */\n validated: true\n /** Mean counterfactual score across the validation reps. */\n meanScore: number\n /** meanScore − originalScore. */\n deltaScore: number\n reps: number\n /** Replay run ids backing the validation — audit trail. */\n counterfactualRunIds: string[]\n}\n\nexport interface RejectedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n reason: 'did-not-flip' | 'error'\n /** Present for 'did-not-flip': mean delta over the reps that ran. */\n deltaScore?: number\n /** Present for 'error': the message, preserved for diagnosis. */\n error?: string\n}\n\nexport interface RepairReport {\n runId: string\n originalScore: number\n flipThreshold: number\n repairs: ValidatedRepair[]\n rejected: RejectedRepair[]\n replaysUsed: number\n}\n\nexport async function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport> {\n const flipThreshold = opts.flipThreshold ?? 0.5\n const repsToValidate = opts.repsToValidate ?? 3\n if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {\n throw new ValidationError(\n `prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`,\n )\n }\n const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY\n if (maxAttempts < 1) {\n throw new ValidationError(\n `prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`,\n )\n }\n if (opts.blamed.length === 0) {\n throw new ValidationError('prescribeRepair: blamed is empty — nothing to repair')\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `prescribeRepair: run ${opts.runId} has no numeric outcome.score — flips have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n\n const repairs: ValidatedRepair[] = []\n const rejected: RejectedRepair[] = []\n let replaysUsed = 0\n\n for (const responsibility of opts.blamed) {\n const step = trajectory.steps[responsibility.stepRef.index]\n if (!step || step.span.spanId !== responsibility.stepRef.spanId) {\n throw new ValidationError(\n `prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} — stale report?`,\n )\n }\n\n const candidates = await opts.proposeFix(step, {\n runId: opts.runId,\n trajectory,\n originalScore,\n responsibility,\n })\n const toTry = candidates.slice(0, maxAttempts)\n\n for (const mutation of toTry) {\n if (mutation.at !== step.index) {\n throw new ValidationError(\n `prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`,\n )\n }\n const scores: number[] = []\n const cfRunIds: string[] = []\n let failure: string | undefined\n for (let rep = 0; rep < repsToValidate; rep++) {\n try {\n const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner)\n replaysUsed++\n const score = result.delta.counterfactualOutcomeScore\n if (typeof score !== 'number' || !Number.isFinite(score)) {\n failure = `validation rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`\n break\n }\n scores.push(score)\n cfRunIds.push(result.counterfactualRunId)\n } catch (err) {\n replaysUsed++\n failure = err instanceof Error ? err.message : String(err)\n break\n }\n }\n\n if (failure !== undefined) {\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'error',\n error: failure,\n })\n continue\n }\n\n const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length\n const everyRepFlipped = scores.every((s) => s >= flipThreshold)\n if (everyRepFlipped) {\n repairs.push({\n stepRef: responsibility.stepRef,\n mutation,\n validated: true,\n meanScore,\n deltaScore: meanScore - originalScore,\n reps: repsToValidate,\n counterfactualRunIds: cfRunIds,\n })\n // First validated repair per step IS the prescription; remaining\n // candidates are untried, not rejected — we don't fabricate verdicts.\n break\n }\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'did-not-flip',\n deltaScore: meanScore - originalScore,\n })\n }\n }\n\n return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;AA2CO,SAAS,UAAU,MAA+B;AACvD,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK,KAAK;AAAA,IAClB,MAAM,KAAK,KAAK;AAAA,IAChB,MAAM,KAAK,KAAK;AAAA,EAClB;AACF;AAkEA,IAAM,kBAAkB;AAExB,SAAS,iBAAiB,MAAgD;AACxE,MAAI,KAAK,KAAK,SAAS,QAAQ;AAC7B,WAAO,CAAC,EAAE,MAAM,oBAAoB,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC;AAAA,EACvE;AACA,MAAI,KAAK,KAAK,SAAS,OAAO;AAC5B,WAAO,CAAC,EAAE,MAAM,kBAAkB,IAAI,KAAK,MAAM,CAAC;AAAA,EACpD;AACA,SAAO,CAAC;AACV;AAEA,eAAsB,YAAY,MAA+D;AAC/F,MAAI,CAAC,OAAO,UAAU,KAAK,IAAI,KAAK,KAAK,OAAO,GAAG;AACjD,UAAM,IAAI;AAAA,MACR,kDAAkD,KAAK,IAAI;AAAA,IAC7D;AAAA,EACF;AACA,MAAI,CAAC,OAAO,UAAU,KAAK,MAAM,KAAK,KAAK,SAAS,GAAG;AACrD,UAAM,IAAI,gBAAgB,oDAAoD,KAAK,MAAM,GAAG;AAAA,EAC9F;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,oBAAoB,KAAK,KAAK,YAAY;AACtF,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,oBAAoB,KAAK,KAAK;AAAA,IAChC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAC/D,QAAM,aAAa,kBAAkB,YAAY,KAAK,cAAc;AACpE,QAAM,eAAe,KAAK,oBAAoB;AAM9C,QAAM,QAAgB,CAAC;AACvB,aAAW,QAAQ,YAAY;AAC7B,UAAM,YAAY,aAAa,IAAI;AACnC,eAAW,KAAK,WAAW;AACzB,UAAI,EAAE,OAAO,KAAK,OAAO;AACvB,cAAM,IAAI;AAAA,UACR,kEAAkE,EAAE,EAAE,mBAAmB,KAAK,KAAK;AAAA,QACrG;AAAA,MACF;AACA,YAAM,KAAK,EAAE,MAAM,UAAU,EAAE,CAAC;AAAA,IAClC;AAAA,EACF;AAEA,QAAM,mBAAyC,CAAC;AAChD,QAAM,aAAqC,CAAC;AAC5C,QAAM,mBAAmB,oBAAI,IAAY;AACzC,MAAI,cAAc;AAClB,MAAI,SAAS;AAEb,aAAW,QAAQ,OAAO;AACxB,QAAI,UAAU,cAAc,KAAK,OAAO,KAAK,QAAQ;AAGnD,eAAS;AACT,uBAAiB,IAAI,KAAK,KAAK,KAAK;AACpC;AAAA,IACF;AACA,UAAM,SAAmB,CAAC;AAC1B,UAAM,WAAqB,CAAC;AAC5B,aAAS,MAAM,GAAG,MAAM,KAAK,MAAM,OAAO;AACxC,YAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,KAAK,UAAU,KAAK,MAAM;AACzF;AACA,YAAM,IAAI,OAAO,MAAM;AACvB,UAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;AAChD,cAAM,IAAI;AAAA,UACR,+CAA+C,KAAK,KAAK,KAAK,KAAK,KAAK,SAAS,IAAI,SAAS,GAAG;AAAA,QACnG;AAAA,MACF;AACA,aAAO,KAAK,CAAC;AACb,eAAS,KAAK,OAAO,mBAAmB;AACxC,iBAAW,KAAK,MAAM;AAAA,IACxB;AACA,UAAM,KAAK,mBAAmB,QAAQ,KAAK,gBAAgB,MAAM;AAAA,MAC/D,MAAM,KAAK,UAAU;AAAA,IACvB,CAAC;AACD,qBAAiB,KAAK;AAAA,MACpB,SAAS,UAAU,KAAK,IAAI;AAAA,MAC5B,cAAc,KAAK,SAAS;AAAA,MAC5B,YAAY,GAAG;AAAA,MACf;AAAA,MACA,gBAAgB,GAAG,QAAQ,KAAK,GAAG,QAAQ;AAAA,MAC3C,MAAM,KAAK;AAAA,MACX;AAAA,MACA,sBAAsB;AAAA,IACxB,CAAC;AAAA,EACH;AAEA,mBAAiB,KAAK,CAAC,GAAG,MAAM,KAAK,IAAI,EAAE,UAAU,IAAI,KAAK,IAAI,EAAE,UAAU,CAAC;AAI/E,QAAM,YAAY,WAAW,OAAO,CAAC,MAAM,iBAAiB,IAAI,EAAE,KAAK,CAAC,EAAE,IAAI,SAAS;AAEvF,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ;AAAA,IACA,OAAO;AAAA,IACP,gBAAgB,yBAAyB,UAAU;AAAA,IACnD;AAAA,IACA,QAAQ,KAAK;AAAA,IACb;AAAA,EACF;AACF;AAEA,SAAS,kBAAkB,YAAwB,SAAsC;AACvF,MAAI,YAAY,QAAW;AACzB,WAAO,WAAW,MAAM,OAAO,CAAC,MAAM,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,MAAM;AAAA,EACvF;AACA,SAAO,QAAQ,IAAI,CAAC,MAAM;AACxB,UAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,QAAI,CAAC,MAAM;AACT,YAAM,IAAI;AAAA,QACR,qCAAqC,CAAC,qBAAqB,WAAW,MAAM,MAAM;AAAA,MACpF;AAAA,IACF;AACA,WAAO;AAAA,EACT,CAAC;AACH;;;ACzNO,IAAM,sBAAsB;AAK5B,SAAS,mBAAmB,gBAAqD;AACtF,MAAI,CAAC,eAAe,eAAgB,QAAO;AAC3C,QAAM,YAAY,KAAK,IAAI,eAAe,UAAU;AACpD,MAAI,aAAa,IAAK,QAAO;AAC7B,MAAI,aAAa,KAAM,QAAO;AAC9B,MAAI,aAAa,IAAK,QAAO;AAC7B,SAAO;AACT;AAIO,SAAS,iBAAiB,UAA0C;AACzE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO,cAAc,SAAS,QAAQ,aAAa,SAAS,EAAE;AAAA,IAChE,KAAK;AACH,aAAO,mCAAmC,SAAS,EAAE,SAAS,KAAK,UAAU,SAAS,SAAS,CAAC;AAAA,IAClG,KAAK;AACH,aAAO,2BAA2B,SAAS,EAAE;AAAA,IAC/C,KAAK;AACH,aAAO,iCAAiC,SAAS,EAAE,KAAK,SAAS,OAAO;AAAA,IAC1E,KAAK;AACH,aAAO,GAAG,SAAS,QAAQ,UAAU,SAAS,EAAE;AAAA,EACpD;AACF;AAWO,SAAS,kBACd,QACA,SACkB;AAClB,QAAM,eAAe,oBAAI,IAA6B;AACtD,aAAW,KAAK,SAAS,WAAW,CAAC,GAAG;AACtC,QAAI,CAAC,aAAa,IAAI,EAAE,QAAQ,MAAM,EAAG,cAAa,IAAI,EAAE,QAAQ,QAAQ,CAAC;AAAA,EAC/E;AAEA,SAAO,OAAO,MAAM,IAAI,CAAC,SAAS;AAChC,UAAM,SAAS,aAAa,IAAI,KAAK,QAAQ,MAAM;AACnD,UAAM,WAA0B;AAAA,MAC9B;AAAA,QACE,MAAM;AAAA,QACN,KAAK,UAAU,KAAK,QAAQ,MAAM;AAAA,QAClC,SAAS,QAAQ,KAAK,QAAQ,KAAK,KAAK,KAAK,QAAQ,IAAI,KAAK,KAAK,QAAQ,IAAI,iBAAiB,KAAK,WAAW,QAAQ,CAAC,CAAC,QAAQ,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,UAAU,KAAK,IAAI;AAAA,MAC5M;AAAA,MACA;AAAA,QACE,MAAM;AAAA,QACN,KAAK,qBAAqB,OAAO,KAAK,SAAS,KAAK,QAAQ,KAAK,IAAI,KAAK,YAAY;AAAA,QACtF,SAAS,WAAW,KAAK,OAAO,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC,EAAE,KAAK,IAAI,CAAC;AAAA,MACrE;AAAA,MACA,GAAG,KAAK,qBAAqB,IAAI,CAAC,QAAqB,EAAE,MAAM,QAAQ,KAAK,SAAS,EAAE,GAAG,EAAE;AAAA,IAC9F;AACA,WAAO,YAAY;AAAA,MACjB,YAAY;AAAA,MACZ,UAAU,mBAAmB,IAAI;AAAA,MACjC,MAAM;AAAA,MACN,OAAO,SAAS,KAAK,QAAQ,IAAI,MAAM,KAAK,QAAQ,IAAI,uDAAuD,KAAK,YAAY;AAAA,MAChI,WAAW,KAAK,iBACZ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,oBAChJ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC;AAAA,MACpJ,eAAe;AAAA,MACf,oBAAoB,SAAS,iBAAiB,OAAO,QAAQ,IAAI;AAAA,MACjE,iBAAiB,SACb,qBAAqB,OAAO,IAAI,IAAI,OAAO,IAAI,mBAAmB,QAAS,aAAa,UAAU,OAAO,UAAU,QAAQ,CAAC,CAAC,WAAW,OAAO,WAAW,QAAQ,CAAC,CAAC,MACpK;AAAA,MACJ,YAAY,SAAS,OAAO,KAAK,iBAAiB,OAAO;AAAA,MACzD,SAAS,KAAK,QAAQ;AAAA,MACtB,UAAU;AAAA,QACR,SAAS,KAAK;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,YAAY,KAAK;AAAA,QACjB,IAAI,KAAK;AAAA,QACT,QAAQ,KAAK;AAAA,QACb,sBAAsB,KAAK;AAAA,QAC3B,GAAI,SAAS,EAAE,QAAQ,EAAE,UAAU,OAAO,UAAU,WAAW,OAAO,UAAU,EAAE,IAAI,CAAC;AAAA,MACzF;AAAA,IACF,CAAC;AAAA,EACH,CAAC;AACH;AAaO,SAAS,eACd,KACA,QACA,OAAiD,CAAC,GACpC;AACd,QAAM,SAAuB;AAAA,IAC3B,GAAG;AAAA,IACH,OAAO,GAAG,IAAI,KAAK,WAAW,OAAO,QAAQ,MAAM;AAAA,IACnD,SAAS;AAAA,MACP,GAAG,IAAI;AAAA,MACP,KAAK;AAAA,QACH,GAAG,IAAI,QAAQ;AAAA,QACf,4BAA4B,OAAO,QAAQ;AAAA,QAC3C,4BAA4B,OAAO;AAAA,QACnC,6BAA6B,OAAO;AAAA,QACpC,sBAAsB,OAAO;AAAA,MAC/B;AAAA,IACF;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,YAAY,KAAK,cAAc,iBAAiB,OAAO,QAAQ;AAAA,EACjE;AAGA,oBAAkB,MAAM;AACxB,SAAO;AACT;AAgBO,SAAS,iBAAiB,QAAwC;AACvE,QAAM,EAAE,SAAS,SAAS,IAAI;AAC9B,QAAM,KAAK,QAAQ,QAAQ,KAAK,KAAK,QAAQ,IAAI,KAAK,QAAQ,IAAI;AAClE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO;AAAA,QACL,aAAa,uBAAuB,QAAQ,IAAI,4FAA4F,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QACxK,OAAO,iCAAiC,QAAQ,IAAI;AAAA,QACpD,SAAS,yBAAyB,QAAQ,IAAI;AAAA,MAChD;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,yBAAyB,EAAE,QAAQ,SAAS,QAAQ,gCAAgC,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC7H,OAAO,aAAa,QAAQ,IAAI,iCAAiC,SAAS,QAAQ;AAAA,MACpF;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,iCAAiC,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC3G,SAAS,8BAA8B,QAAQ,IAAI,MAAM,SAAS,OAAO;AAAA,MAC3E;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,kBAAkB,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC5F,OAAO,wBAAwB,QAAQ,IAAI,YAAY,QAAQ,KAAK;AAAA,MACtE;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,GAAG,SAAS,QAAQ,OAAO,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,MACvG;AAAA,IACF,SAAS;AACP,YAAM,YAAmB;AACzB,YAAM,IAAI;AAAA,QACR,2CAA2C,KAAK,UAAU,SAAS,CAAC;AAAA,MACtE;AAAA,IACF;AAAA,EACF;AACF;;;ACtHA,eAAsB,gBAAgB,MAAqD;AACzF,QAAM,gBAAgB,KAAK,iBAAiB;AAC5C,QAAM,iBAAiB,KAAK,kBAAkB;AAC9C,MAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GAAG;AAC3D,UAAM,IAAI;AAAA,MACR,gEAAgE,cAAc;AAAA,IAChF;AAAA,EACF;AACA,QAAM,cAAc,KAAK,sBAAsB,OAAO;AACtD,MAAI,cAAc,GAAG;AACnB,UAAM,IAAI;AAAA,MACR,yDAAyD,KAAK,kBAAkB;AAAA,IAClF;AAAA,EACF;AACA,MAAI,KAAK,OAAO,WAAW,GAAG;AAC5B,UAAM,IAAI,gBAAgB,2DAAsD;AAAA,EAClF;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,wBAAwB,KAAK,KAAK,YAAY;AAC1F,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,wBAAwB,KAAK,KAAK;AAAA,IACpC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAE/D,QAAM,UAA6B,CAAC;AACpC,QAAM,WAA6B,CAAC;AACpC,MAAI,cAAc;AAElB,aAAW,kBAAkB,KAAK,QAAQ;AACxC,UAAM,OAAO,WAAW,MAAM,eAAe,QAAQ,KAAK;AAC1D,QAAI,CAAC,QAAQ,KAAK,KAAK,WAAW,eAAe,QAAQ,QAAQ;AAC/D,YAAM,IAAI;AAAA,QACR,sCAAsC,eAAe,QAAQ,KAAK,WAAW,eAAe,QAAQ,MAAM,uBAAuB,KAAK,KAAK;AAAA,MAC7I;AAAA,IACF;AAEA,UAAM,aAAa,MAAM,KAAK,WAAW,MAAM;AAAA,MAC7C,OAAO,KAAK;AAAA,MACZ;AAAA,MACA;AAAA,MACA;AAAA,IACF,CAAC;AACD,UAAM,QAAQ,WAAW,MAAM,GAAG,WAAW;AAE7C,eAAW,YAAY,OAAO;AAC5B,UAAI,SAAS,OAAO,KAAK,OAAO;AAC9B,cAAM,IAAI;AAAA,UACR,gEAAgE,SAAS,EAAE,0BAA0B,KAAK,KAAK;AAAA,QACjH;AAAA,MACF;AACA,YAAM,SAAmB,CAAC;AAC1B,YAAM,WAAqB,CAAC;AAC5B,UAAI;AACJ,eAAS,MAAM,GAAG,MAAM,gBAAgB,OAAO;AAC7C,YAAI;AACF,gBAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,UAAU,KAAK,MAAM;AACpF;AACA,gBAAM,QAAQ,OAAO,MAAM;AAC3B,cAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;AACxD,sBAAU,kBAAkB,GAAG;AAC/B;AAAA,UACF;AACA,iBAAO,KAAK,KAAK;AACjB,mBAAS,KAAK,OAAO,mBAAmB;AAAA,QAC1C,SAAS,KAAK;AACZ;AACA,oBAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AACzD;AAAA,QACF;AAAA,MACF;AAEA,UAAI,YAAY,QAAW;AACzB,iBAAS,KAAK;AAAA,UACZ,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,QAAQ;AAAA,UACR,OAAO;AAAA,QACT,CAAC;AACD;AAAA,MACF;AAEA,YAAM,YAAY,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;AAC7D,YAAM,kBAAkB,OAAO,MAAM,CAAC,MAAM,KAAK,aAAa;AAC9D,UAAI,iBAAiB;AACnB,gBAAQ,KAAK;AAAA,UACX,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,WAAW;AAAA,UACX;AAAA,UACA,YAAY,YAAY;AAAA,UACxB,MAAM;AAAA,UACN,sBAAsB;AAAA,QACxB,CAAC;AAGD;AAAA,MACF;AACA,eAAS,KAAK;AAAA,QACZ,SAAS,eAAe;AAAA,QACxB;AAAA,QACA,QAAQ;AAAA,QACR,YAAY,YAAY;AAAA,MAC1B,CAAC;AAAA,IACH;AAAA,EACF;AAEA,SAAO,EAAE,OAAO,KAAK,OAAO,eAAe,eAAe,SAAS,UAAU,YAAY;AAC3F;","names":[]}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { S as Scenario, C as CampaignResult, G as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-
|
|
1
|
+
import { S as Scenario, C as CampaignResult, G as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-Cv1bo4_a.js';
|
|
2
2
|
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
3
3
|
|
|
4
4
|
/**
|
package/dist/hosted/index.d.ts
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { M as MutableSurface, j as GateDecision } from '../types-
|
|
2
|
-
import { I as InsightReport } from '../insight-report-
|
|
3
|
-
import '../run-record-
|
|
1
|
+
import { M as MutableSurface, j as GateDecision } from '../types-Cv1bo4_a.js';
|
|
2
|
+
import { I as InsightReport } from '../insight-report-C02J3q4T.js';
|
|
3
|
+
import '../run-record-DEwidcqn.js';
|
|
4
4
|
import '@tangle-network/agent-interface';
|
|
5
5
|
import '../errors-CzMUYo7b.js';
|
|
6
6
|
import '../schema-m0gsnbt3.js';
|
|
7
|
-
import '../summary-report-
|
|
7
|
+
import '../summary-report-C4uzRWh8.js';
|
|
8
8
|
import '../failure-cluster-DH9Flgcf.js';
|
|
9
9
|
import '../store-BcFXE6LG.js';
|
|
10
10
|
import '../judge-calibration-0p2QcWNE.js';
|
package/dist/index.d.ts
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord,
|
|
3
|
-
export {
|
|
4
|
-
import { B as BehavioralMetrics } from './semantic-concept-judge-
|
|
5
|
-
export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-
|
|
6
|
-
import { l as ChatRequest, p as CreateChatClientOpts } from './types-
|
|
7
|
-
export {
|
|
8
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-
|
|
9
|
-
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-DC8TELh0.js';
|
|
2
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-DEwidcqn.js';
|
|
3
|
+
export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as parseRunRecordSafe, y as requireAgentProfileCell, z as roundTripRunRecord, B as toAgentProfileJson, C as validateAgentProfileCell, D as validateRunRecord, E as verifyAgentProfileCell } from './run-record-DEwidcqn.js';
|
|
4
|
+
import { B as BehavioralMetrics } from './semantic-concept-judge-J8xvjdc3.js';
|
|
5
|
+
export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-J8xvjdc3.js';
|
|
6
|
+
import { l as ChatRequest, p as CreateChatClientOpts } from './types-BEzCBMQD.js';
|
|
7
|
+
export { a as Analyst, b as AnalystContext, g as AnalystCost, A as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-BEzCBMQD.js';
|
|
8
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-Dhrc__SE.js';
|
|
9
|
+
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-OgqQSvLi.js';
|
|
10
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-Dccm9tyA.js';
|
|
10
11
|
import { TCloud } from '@tangle-network/tcloud';
|
|
11
12
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
|
|
12
13
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
|
|
@@ -17,11 +18,11 @@ import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors
|
|
|
17
18
|
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-CzMUYo7b.js';
|
|
18
19
|
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-BxY0cKfs.js';
|
|
19
20
|
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-BxY0cKfs.js';
|
|
20
|
-
import { b as CorrectnessChecker } from './pre-registration-
|
|
21
|
-
export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-
|
|
21
|
+
import { b as CorrectnessChecker } from './pre-registration-DB8oDqZJ.js';
|
|
22
|
+
export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-DB8oDqZJ.js';
|
|
22
23
|
export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
|
|
23
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
24
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
24
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-B1tA6pKu.js';
|
|
25
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-B1tA6pKu.js';
|
|
25
26
|
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as ProportionInterval, R as RiskDifferenceResult, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as pearsonR, K as ranks, L as requiredSampleSize, N as spearmanR, O as weightedComposite, Q as weightedMean, w as wilcoxonSignedRank, S as wilson } from './statistics-xP-cWc5k.js';
|
|
26
27
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
27
28
|
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
@@ -29,8 +30,8 @@ import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesR
|
|
|
29
30
|
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
30
31
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
31
32
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
32
|
-
import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-
|
|
33
|
-
import { A as AnalyzeRunsOptions } from './analyze-runs-
|
|
33
|
+
import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-Cv1bo4_a.js';
|
|
34
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-BlJRBniC.js';
|
|
34
35
|
import { S as SteeringBundle } from './harness-optimizer-mOl9XX_O.js';
|
|
35
36
|
export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-mOl9XX_O.js';
|
|
36
37
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-DeODGLra.js';
|
|
@@ -46,7 +47,7 @@ export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionRep
|
|
|
46
47
|
import { T as TraceStore, R as RunFilter } from './store-BcFXE6LG.js';
|
|
47
48
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BcFXE6LG.js';
|
|
48
49
|
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-DH9Flgcf.js';
|
|
49
|
-
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-
|
|
50
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-OJDaTYHN.js';
|
|
50
51
|
import { a as BaselineReport } from './baseline-Bbid3WoO.js';
|
|
51
52
|
export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-Bbid3WoO.js';
|
|
52
53
|
import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
|
|
@@ -69,16 +70,16 @@ import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from
|
|
|
69
70
|
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DUZXrPDA.js';
|
|
70
71
|
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
71
72
|
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-Bj7g0rqu.js';
|
|
72
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-
|
|
73
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
74
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
73
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-W96macmS.js';
|
|
74
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Ba2y1Foi.js';
|
|
75
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-C4uzRWh8.js';
|
|
75
76
|
export { L as LockedJsonlAppender } from './testing-C21CHsq2.js';
|
|
76
77
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
77
|
-
import { e as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-
|
|
78
|
+
import { e as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-BRgNnmGZ.js';
|
|
78
79
|
export { IntegrityResult, IntegrityViolation, JourneySpec, PerfBaseline, PerfGateResult, PerfRegression, PerfScenario, PerfStat, ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords } from './perf/index.js';
|
|
79
80
|
import '@ax-llm/ax';
|
|
80
81
|
import 'zod';
|
|
81
|
-
import './insight-report-
|
|
82
|
+
import './insight-report-C02J3q4T.js';
|
|
82
83
|
import './outcome-store-rnXLEqSn.js';
|
|
83
84
|
|
|
84
85
|
/**
|
package/dist/index.js
CHANGED
|
@@ -51,7 +51,7 @@ import {
|
|
|
51
51
|
llmJudge,
|
|
52
52
|
parseCorrectnessResponse,
|
|
53
53
|
verifyCompletion
|
|
54
|
-
} from "./chunk-
|
|
54
|
+
} from "./chunk-VWQ6PO5O.js";
|
|
55
55
|
import {
|
|
56
56
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
57
57
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -75,7 +75,7 @@ import {
|
|
|
75
75
|
scoreRedTeamOutput,
|
|
76
76
|
surfaceContentHash,
|
|
77
77
|
toolNamesForRun
|
|
78
|
-
} from "./chunk-
|
|
78
|
+
} from "./chunk-2KTBHICD.js";
|
|
79
79
|
import {
|
|
80
80
|
BackendIntegrityError,
|
|
81
81
|
assertRealBackend,
|
|
@@ -133,7 +133,7 @@ import {
|
|
|
133
133
|
defaultIsMaterial,
|
|
134
134
|
diffFindings,
|
|
135
135
|
runSemanticConceptJudge
|
|
136
|
-
} from "./chunk-
|
|
136
|
+
} from "./chunk-JU6ZX3CX.js";
|
|
137
137
|
import {
|
|
138
138
|
LockedJsonlAppender,
|
|
139
139
|
Mutex
|
|
@@ -141,12 +141,24 @@ import {
|
|
|
141
141
|
import {
|
|
142
142
|
buildDefaultAnalystRegistry,
|
|
143
143
|
computeTraceMetrics
|
|
144
|
-
} from "./chunk-
|
|
144
|
+
} from "./chunk-L5TVEZFT.js";
|
|
145
145
|
import {
|
|
146
146
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
147
|
+
POLICY_EDIT_AXES,
|
|
148
|
+
POLICY_EDIT_TARGET_SURFACES,
|
|
149
|
+
PolicyEditValidationError,
|
|
150
|
+
admitPolicyEdit,
|
|
147
151
|
aggregateRunScore,
|
|
148
|
-
|
|
149
|
-
|
|
152
|
+
applyPolicyEditToSurface,
|
|
153
|
+
clamp01,
|
|
154
|
+
computePolicyEditId,
|
|
155
|
+
isPolicyEdit,
|
|
156
|
+
makePolicyEdit,
|
|
157
|
+
policyEditFromFinding,
|
|
158
|
+
policyEditsFromFindings,
|
|
159
|
+
scorePolicyEditReadiness,
|
|
160
|
+
validatePolicyEdit
|
|
161
|
+
} from "./chunk-IN3SHQML.js";
|
|
150
162
|
import {
|
|
151
163
|
AnalystRegistry,
|
|
152
164
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -156,7 +168,7 @@ import {
|
|
|
156
168
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
157
169
|
createTraceAnalystKind,
|
|
158
170
|
renderPriorFindings
|
|
159
|
-
} from "./chunk-
|
|
171
|
+
} from "./chunk-FRI6RG3P.js";
|
|
160
172
|
import {
|
|
161
173
|
controlFailureClassFromVerification,
|
|
162
174
|
controlRunToRunRecord,
|
|
@@ -167,7 +179,7 @@ import {
|
|
|
167
179
|
runProposeReview,
|
|
168
180
|
runProposeReviewAsControlLoop,
|
|
169
181
|
scoreFromEvals
|
|
170
|
-
} from "./chunk-
|
|
182
|
+
} from "./chunk-BOETF6BU.js";
|
|
171
183
|
import {
|
|
172
184
|
allCriticalPassed,
|
|
173
185
|
errorStreakDetector,
|
|
@@ -189,7 +201,7 @@ import {
|
|
|
189
201
|
} from "./chunk-VDGPPGE3.js";
|
|
190
202
|
import {
|
|
191
203
|
runEvalCampaign
|
|
192
|
-
} from "./chunk-
|
|
204
|
+
} from "./chunk-4LWD6GC7.js";
|
|
193
205
|
import {
|
|
194
206
|
LlmCallError,
|
|
195
207
|
LlmClient,
|
|
@@ -292,13 +304,13 @@ import {
|
|
|
292
304
|
scoreTraceInsightReadiness,
|
|
293
305
|
tokenizeDomainWords,
|
|
294
306
|
traceAnalystOnRunComplete
|
|
295
|
-
} from "./chunk-
|
|
307
|
+
} from "./chunk-PMF5WIBX.js";
|
|
296
308
|
import {
|
|
297
309
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
298
310
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
299
311
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
300
312
|
analyzeTraces
|
|
301
|
-
} from "./chunk-
|
|
313
|
+
} from "./chunk-2MLIEQSN.js";
|
|
302
314
|
import {
|
|
303
315
|
DEFAULT_REDACTION_RULES,
|
|
304
316
|
REDACTION_VERSION,
|
|
@@ -370,29 +382,31 @@ import {
|
|
|
370
382
|
defaultProviderRedactor,
|
|
371
383
|
providerFromBaseUrl
|
|
372
384
|
} from "./chunk-PC4UYEBM.js";
|
|
385
|
+
import {
|
|
386
|
+
RunRecordValidationError,
|
|
387
|
+
isRunRecord,
|
|
388
|
+
parseRunRecordSafe,
|
|
389
|
+
roundTripRunRecord,
|
|
390
|
+
validateRunRecord
|
|
391
|
+
} from "./chunk-G6S73VA7.js";
|
|
392
|
+
import {
|
|
393
|
+
TraceEmitter,
|
|
394
|
+
llmSpanFromProvider
|
|
395
|
+
} from "./chunk-TVVP3ZZQ.js";
|
|
373
396
|
import {
|
|
374
397
|
AGENT_PROFILE_KINDS,
|
|
375
398
|
AgentProfileCellValidationError,
|
|
376
|
-
RunRecordValidationError,
|
|
377
399
|
agentProfileCellHashMaterial,
|
|
378
400
|
agentProfileCellKey,
|
|
379
401
|
assertRunAgentProfileCell,
|
|
380
402
|
buildAgentInterfaceProfileCell,
|
|
381
403
|
buildAgentProfileCell,
|
|
382
404
|
groupRunsByAgentProfileCell,
|
|
383
|
-
isRunRecord,
|
|
384
|
-
parseRunRecordSafe,
|
|
385
405
|
requireAgentProfileCell,
|
|
386
|
-
roundTripRunRecord,
|
|
387
406
|
toAgentProfileJson,
|
|
388
407
|
validateAgentProfileCell,
|
|
389
|
-
validateRunRecord,
|
|
390
408
|
verifyAgentProfileCell
|
|
391
|
-
} from "./chunk-
|
|
392
|
-
import {
|
|
393
|
-
TraceEmitter,
|
|
394
|
-
llmSpanFromProvider
|
|
395
|
-
} from "./chunk-TVVP3ZZQ.js";
|
|
409
|
+
} from "./chunk-ABOIVNXL.js";
|
|
396
410
|
import {
|
|
397
411
|
canonicalize,
|
|
398
412
|
evaluateHypothesis,
|
|
@@ -10126,7 +10140,10 @@ export {
|
|
|
10126
10140
|
OPENINFERENCE_SPAN_KIND,
|
|
10127
10141
|
OTEL_AGENT_EVAL_SCOPE,
|
|
10128
10142
|
OtlpFileTraceStore,
|
|
10143
|
+
POLICY_EDIT_AXES,
|
|
10144
|
+
POLICY_EDIT_TARGET_SURFACES,
|
|
10129
10145
|
PairwiseSteeringOptimizer,
|
|
10146
|
+
PolicyEditValidationError,
|
|
10130
10147
|
ProductClient,
|
|
10131
10148
|
PromptRegistry,
|
|
10132
10149
|
REDACTION_VERSION,
|
|
@@ -10165,6 +10182,7 @@ export {
|
|
|
10165
10182
|
ValidationError,
|
|
10166
10183
|
VerificationError,
|
|
10167
10184
|
acquisitionPlansForKnowledgeGaps,
|
|
10185
|
+
admitPolicyEdit,
|
|
10168
10186
|
adversarialJudge,
|
|
10169
10187
|
agentProfileCellHashMaterial,
|
|
10170
10188
|
agentProfileCellKey,
|
|
@@ -10181,6 +10199,7 @@ export {
|
|
|
10181
10199
|
analyzeSeries,
|
|
10182
10200
|
analyzeTraces,
|
|
10183
10201
|
appendScorecard,
|
|
10202
|
+
applyPolicyEditToSurface,
|
|
10184
10203
|
argHash,
|
|
10185
10204
|
asNumber,
|
|
10186
10205
|
asString,
|
|
@@ -10255,6 +10274,7 @@ export {
|
|
|
10255
10274
|
composeValidators,
|
|
10256
10275
|
computeExperimentStats,
|
|
10257
10276
|
computeFindingId,
|
|
10277
|
+
computePolicyEditId,
|
|
10258
10278
|
computeToolUseMetrics,
|
|
10259
10279
|
computeTraceMetrics,
|
|
10260
10280
|
confidenceInterval,
|
|
@@ -10393,6 +10413,7 @@ export {
|
|
|
10393
10413
|
isLlmSpan,
|
|
10394
10414
|
isModelPriced,
|
|
10395
10415
|
isOtelConfigured,
|
|
10416
|
+
isPolicyEdit,
|
|
10396
10417
|
isRetrievalSpan,
|
|
10397
10418
|
isRunRecord,
|
|
10398
10419
|
isSandboxSpan,
|
|
@@ -10421,6 +10442,7 @@ export {
|
|
|
10421
10442
|
lowercaseMutator,
|
|
10422
10443
|
makeEvalTools,
|
|
10423
10444
|
makeFinding,
|
|
10445
|
+
makePolicyEdit,
|
|
10424
10446
|
mannWhitneyU,
|
|
10425
10447
|
matchGoldens,
|
|
10426
10448
|
matchSpan,
|
|
@@ -10465,6 +10487,8 @@ export {
|
|
|
10465
10487
|
pearsonR,
|
|
10466
10488
|
pixelDeltaRatio,
|
|
10467
10489
|
planTraceInsightQuestions,
|
|
10490
|
+
policyEditFromFinding,
|
|
10491
|
+
policyEditsFromFindings,
|
|
10468
10492
|
politenessPrefixMutator,
|
|
10469
10493
|
positionalBias,
|
|
10470
10494
|
preflightModels,
|
|
@@ -10540,6 +10564,7 @@ export {
|
|
|
10540
10564
|
scoreContinuity,
|
|
10541
10565
|
scoreFromEvals,
|
|
10542
10566
|
scoreKnowledgeReadiness,
|
|
10567
|
+
scorePolicyEditReadiness,
|
|
10543
10568
|
scorePrReviewComments,
|
|
10544
10569
|
scorePrReviewSource,
|
|
10545
10570
|
scoreRedTeamOutput,
|
|
@@ -10589,6 +10614,7 @@ export {
|
|
|
10589
10614
|
urlContains,
|
|
10590
10615
|
userQuestionsForKnowledgeGaps,
|
|
10591
10616
|
validateAgentProfileCell,
|
|
10617
|
+
validatePolicyEdit,
|
|
10592
10618
|
validateRunRecord,
|
|
10593
10619
|
verbosityBias,
|
|
10594
10620
|
verifyAgentProfileCell,
|