@tangle-network/agent-eval 0.109.1 → 0.110.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +16 -19
- package/dist/campaign/index.js +7 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-GTERJI6Q.js} +4 -4
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-R6D7NEYJ.js → chunk-RNB2NICW.js} +11 -11
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +20 -23
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-BUNP3606.d.ts} +143 -2
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-DMvsfknv.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-OVPVM4JC.js.map → chunk-GTERJI6Q.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-R6D7NEYJ.js.map → chunk-RNB2NICW.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/diagnose.js.map
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/diagnose/causal-sweep.ts","../src/diagnose/remediation.ts","../src/diagnose/repair.ts"],"sourcesContent":["/**\n * Causal sweep — WHY did this run fail?\n *\n * Orchestrates the dormant counterfactual primitives into a responsibility\n * report: for each candidate step, run `reps` counterfactual replays per\n * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`\n * is the execution seam) and reduce the per-rep score deltas into a mean\n * effect + bootstrap confidence interval (via `confidenceInterval`).\n *\n * Why `reps` is REQUIRED: a single intervention delta is one stochastic\n * draw — LLM re-execution from a prefix is sampled, so one replay cannot\n * distinguish \"this step caused the failure\" from sampling noise. The\n * signal is the distribution of deltas across reps; the CI over that\n * distribution is what lets a caller say \"this step's effect excludes\n * zero\" instead of eyeballing a point estimate.\n *\n * Budget discipline: the sweep never silently drops cells. When the\n * remaining budget cannot fund a full `reps`-sized cell, the sweep halts\n * and every step not fully probed is named in `uncovered`.\n */\n\nimport {\n attributeCounterfactuals,\n type CounterfactualMutation,\n type CounterfactualResult,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport { confidenceInterval } from '../statistics'\nimport type { Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\n\n/** Stable reference to a trajectory step — carried through reports,\n * findings, and corpus records so evidence stays addressable. */\nexport interface StepRef {\n index: number\n spanId: string\n kind: Span['kind']\n name: string\n}\n\nexport function stepRefOf(step: TrajectoryStep): StepRef {\n return {\n index: step.index,\n spanId: step.span.spanId,\n kind: step.span.kind,\n name: step.span.name,\n }\n}\n\nexport interface CausalSweepOptions {\n store: TraceStore\n /** The failed run to diagnose. Its `outcome.score` is the baseline every\n * counterfactual delta is measured against. */\n runId: string\n /** Execution seam — identical contract to `runCounterfactual`: re-runs the\n * agent from the mutation point and MUST `endRun` with a numeric score. */\n runner: CounterfactualRunner\n /** Trajectory indices to probe. Default: every llm + tool span — the kinds\n * the existing `CounterfactualMutation` set targets. */\n candidateSteps?: number[]\n /**\n * Mutations to probe a given step with. Returned mutations MUST target\n * `step.index`. Default probes are the payload-free existing kinds:\n * - tool span → `swap-tool-result` with `newResult: null` (knockout:\n * how much did the run depend on this tool's information?)\n * - llm span → `truncate-after` (re-roll: how much did the realized\n * turn deviate from the policy's typical continuation?)\n * `swap-model` / `inject-system-message` need consumer payloads, so they\n * are opt-in via this callback.\n */\n mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[]\n /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */\n reps: number\n /** Hard cap on total counterfactual replays across the whole sweep. */\n budget: number\n /** Seed for the bootstrap CI resampler. Deterministic default so two\n * sweeps over the same deltas report identical intervals. */\n ciSeed?: number\n /** Bootstrap CI confidence level. Default 0.95. */\n ciConfidence?: number\n}\n\nexport interface StepResponsibility {\n stepRef: StepRef\n mutationKind: CounterfactualMutation['kind']\n /** Mean of per-rep score deltas (counterfactual − original). */\n meanEffect: number\n /** Bootstrap CI over the per-rep deltas. */\n ci: { mean: number; lower: number; upper: number }\n /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */\n ciExcludesZero: boolean\n reps: number\n /** Raw per-rep deltas — downstream evidence, never re-derived. */\n deltas: number[]\n /** Replay run ids (layer='meta', parentRunId=original) for audit. */\n counterfactualRunIds: string[]\n}\n\nexport interface CausalResponsibilityReport {\n runId: string\n originalScore: number\n /** Ranked by |meanEffect| descending — the blame ordering. */\n steps: StepResponsibility[]\n /** Kind-level aggregate from the existing `attributeCounterfactuals`. */\n byMutationKind: ReturnType<typeof attributeCounterfactuals>\n replaysUsed: number\n budget: number\n /** Steps planned but not fully probed before the budget ran out.\n * Named, never silent: an absent step is \"no effect found\"; an\n * uncovered step is \"not measured\". */\n uncovered: StepRef[]\n}\n\nconst DEFAULT_CI_SEED = 0x5eed\n\nfunction defaultMutations(step: TrajectoryStep): CounterfactualMutation[] {\n if (step.span.kind === 'tool') {\n return [{ kind: 'swap-tool-result', at: step.index, newResult: null }]\n }\n if (step.span.kind === 'llm') {\n return [{ kind: 'truncate-after', at: step.index }]\n }\n return []\n}\n\nexport async function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport> {\n if (!Number.isInteger(opts.reps) || opts.reps < 2) {\n throw new ValidationError(\n `causalSweep: reps must be an integer >= 2 (got ${opts.reps}) — a single-intervention delta is one stochastic draw, not a measurement`,\n )\n }\n if (!Number.isInteger(opts.budget) || opts.budget < 1) {\n throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`)\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `causalSweep: run ${opts.runId} has no numeric outcome.score — deltas have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n const candidates = resolveCandidates(trajectory, opts.candidateSteps)\n const mutationsFor = opts.mutationsPerStep ?? defaultMutations\n\n interface Cell {\n step: TrajectoryStep\n mutation: CounterfactualMutation\n }\n const cells: Cell[] = []\n for (const step of candidates) {\n const mutations = mutationsFor(step)\n for (const m of mutations) {\n if (m.at !== step.index) {\n throw new ValidationError(\n `causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} — mutations must target the step they were asked for`,\n )\n }\n cells.push({ step, mutation: m })\n }\n }\n\n const responsibilities: StepResponsibility[] = []\n const allResults: CounterfactualResult[] = []\n const uncoveredIndices = new Set<number>()\n let replaysUsed = 0\n let halted = false\n\n for (const cell of cells) {\n if (halted || replaysUsed + opts.reps > opts.budget) {\n // A partial cell would report a CI over fewer reps than requested —\n // weaker evidence masquerading as the real thing. Halt and name it.\n halted = true\n uncoveredIndices.add(cell.step.index)\n continue\n }\n const deltas: number[] = []\n const cfRunIds: string[] = []\n for (let rep = 0; rep < opts.reps; rep++) {\n const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner)\n replaysUsed++\n const d = result.delta.deltaScore\n if (typeof d !== 'number' || !Number.isFinite(d)) {\n throw new ValidationError(\n `causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`,\n )\n }\n deltas.push(d)\n cfRunIds.push(result.counterfactualRunId)\n allResults.push(result)\n }\n const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {\n seed: opts.ciSeed ?? DEFAULT_CI_SEED,\n })\n responsibilities.push({\n stepRef: stepRefOf(cell.step),\n mutationKind: cell.mutation.kind,\n meanEffect: ci.mean,\n ci,\n ciExcludesZero: ci.lower > 0 || ci.upper < 0,\n reps: opts.reps,\n deltas,\n counterfactualRunIds: cfRunIds,\n })\n }\n\n responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect))\n\n // A step probed under one mutation but cut off under another appears in\n // BOTH steps and uncovered — partial coverage is named, not blended.\n const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf)\n\n return {\n runId: opts.runId,\n originalScore,\n steps: responsibilities,\n byMutationKind: attributeCounterfactuals(allResults),\n replaysUsed,\n budget: opts.budget,\n uncovered,\n }\n}\n\nfunction resolveCandidates(trajectory: Trajectory, indices?: number[]): TrajectoryStep[] {\n if (indices === undefined) {\n return trajectory.steps.filter((s) => s.span.kind === 'llm' || s.span.kind === 'tool')\n }\n return indices.map((i) => {\n const step = trajectory.steps[i]\n if (!step) {\n throw new ValidationError(\n `causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`,\n )\n }\n return step\n })\n}\n","/**\n * Remediation adapters — HOW DO WE MAKE IT HAPPEN?\n *\n * The diagnose chain ends by feeding existing improvement machinery,\n * not by building new machinery:\n *\n * - `toAnalystFindings` → the analyst contract (`makeFinding`), so\n * responsibility evidence flows into the same registry / steering /\n * diff pipeline every other analyst feeds.\n * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the\n * diagnosed failure + validated repair as a permanent scenario.\n * - `suggestInvariant` → a plain-data hint in the shape the\n * trace-contracts machinery consumes (`never` / `without` clauses).\n */\n\nimport type { AnalystFinding, AnalystSeverity, EvidenceRef } from '../analyst/types'\nimport { makeFinding } from '../analyst/types'\nimport type { CounterfactualMutation } from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { CorpusRecord } from '../rl/corpus'\nimport type { RunRecord } from '../run-record'\nimport { validateRunRecord } from '../run-record'\nimport type { CausalResponsibilityReport, StepResponsibility } from './causal-sweep'\nimport type { RepairReport, ValidatedRepair } from './repair'\n\nexport const DIAGNOSE_ANALYST_ID = 'diagnose-causal-sweep'\n\n/** Severity from causal effect size. Effects whose CI includes zero are\n * 'info' regardless of magnitude — an indistinguishable-from-noise effect\n * must not steer remediation priority. */\nexport function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity {\n if (!responsibility.ciExcludesZero) return 'info'\n const magnitude = Math.abs(responsibility.meanEffect)\n if (magnitude >= 0.5) return 'critical'\n if (magnitude >= 0.25) return 'high'\n if (magnitude >= 0.1) return 'medium'\n return 'low'\n}\n\n/** Deterministic human-readable rendering of a mutation — used in\n * recommended actions, corpus completions, and invariant hints. */\nexport function describeMutation(mutation: CounterfactualMutation): string {\n switch (mutation.kind) {\n case 'swap-model':\n return `use model '${mutation.newModel}' at step ${mutation.at}`\n case 'swap-tool-result':\n return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`\n case 'truncate-after':\n return `stop the run after step ${mutation.at}`\n case 'inject-system-message':\n return `inject system message at step ${mutation.at}: ${mutation.content}`\n case 'custom':\n return `${mutation.describe} (step ${mutation.at})`\n }\n}\n\n/**\n * Lift a responsibility report (and optionally its validated repairs) into\n * `AnalystFinding`s via the real `makeFinding` factory. One finding per\n * probed step; a validated repair for that step upgrades the finding with\n * a `recommended_action` + the replay-validation evidence.\n *\n * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,\n * so `derived_from_judge` stays unset and they may steer.\n */\nexport function toAnalystFindings(\n report: CausalResponsibilityReport,\n repairs?: RepairReport,\n): AnalystFinding[] {\n const repairByStep = new Map<string, ValidatedRepair>()\n for (const r of repairs?.repairs ?? []) {\n if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r)\n }\n\n return report.steps.map((resp) => {\n const repair = repairByStep.get(resp.stepRef.spanId)\n const evidence: EvidenceRef[] = [\n {\n kind: 'span',\n uri: `span://${resp.stepRef.spanId}`,\n excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`,\n },\n {\n kind: 'metric',\n uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,\n excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(', ')}]`,\n },\n ...resp.counterfactualRunIds.map((id): EvidenceRef => ({ kind: 'span', uri: `run://${id}` })),\n ]\n return makeFinding({\n analyst_id: DIAGNOSE_ANALYST_ID,\n severity: severityFromEffect(resp),\n area: 'causal-attribution',\n claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,\n rationale: resp.ciExcludesZero\n ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero`\n : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero — not distinguishable from noise`,\n evidence_refs: evidence,\n recommended_action: repair ? describeMutation(repair.mutation) : undefined,\n validation_plan: repair\n ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs!.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})`\n : undefined,\n confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,\n subject: resp.stepRef.spanId,\n metadata: {\n stepRef: resp.stepRef,\n mutationKind: resp.mutationKind,\n meanEffect: resp.meanEffect,\n ci: resp.ci,\n deltas: resp.deltas,\n counterfactualRunIds: resp.counterfactualRunIds,\n ...(repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}),\n },\n })\n })\n}\n\n/**\n * Pin the diagnosed failure as a permanent corpus scenario. Takes the\n * original run's `RunRecord` projection plus a validated repair and emits\n * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw\n * failure and the diagnosed entry).\n *\n * `completion` defaults to the validated mutation's rendering — \"what\n * should have happened\" in machine-derived form. Supply `prompt` (and\n * optionally a richer `completion`) when the trajectory text is available\n * so the record is harvestable by `buildDatasetFromCorpus`.\n */\nexport function toCorpusRecord(\n run: RunRecord,\n repair: ValidatedRepair,\n opts: { prompt?: string; completion?: string } = {},\n): CorpusRecord {\n const record: CorpusRecord = {\n ...run,\n runId: `${run.runId}#repair:${repair.stepRef.spanId}`,\n outcome: {\n ...run.outcome,\n raw: {\n ...run.outcome.raw,\n diagnose_blamed_step_index: repair.stepRef.index,\n diagnose_repair_mean_score: repair.meanScore,\n diagnose_repair_delta_score: repair.deltaScore,\n diagnose_repair_reps: repair.reps,\n },\n },\n prompt: opts.prompt,\n completion: opts.completion ?? describeMutation(repair.mutation),\n }\n // Boundary check — a corpus record that fails RunRecord validation would\n // poison every downstream harvest.\n validateRunRecord(record)\n return record\n}\n\n/** Plain-data invariant hint. The trace-contracts machinery consumes this\n * shape: `never` is a pattern that must not appear in a passing trace;\n * `without` is a guard whose absence makes the failure reachable. */\nexport interface InvariantHint {\n description: string\n never?: string\n without?: string\n}\n\n/**\n * Derive an invariant hint from a validated repair. Deterministic per\n * mutation kind — the hint names the contract a trace must satisfy so\n * the diagnosed failure cannot silently recur.\n */\nexport function suggestInvariant(repair: ValidatedRepair): InvariantHint {\n const { stepRef, mutation } = repair\n const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`\n switch (mutation.kind) {\n case 'swap-tool-result':\n return {\n description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `unvalidated result from tool '${stepRef.name}' flows downstream`,\n without: `result guard on tool '${stepRef.name}'`,\n }\n case 'swap-model':\n return {\n description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`,\n }\n case 'inject-system-message':\n return {\n description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n without: `system message present at '${stepRef.name}': ${mutation.content}`,\n }\n case 'truncate-after':\n return {\n description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) — continuation past this step caused the failure`,\n never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`,\n }\n case 'custom':\n return {\n description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,\n }\n default: {\n const exhausted: never = mutation\n throw new ValidationError(\n `suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`,\n )\n }\n }\n}\n","/**\n * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?\n *\n * Takes the blamed steps from a `CausalResponsibilityReport`, asks a\n * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate\n * mutations, and machine-verifies each candidate by replaying the run\n * WITH the mutation applied (through the same `runCounterfactual` seam\n * the sweep uses).\n *\n * A repair is \"what should have happened\" ONLY when every validation\n * replay crosses `flipThreshold` — a prescription is never speculated,\n * it is demonstrated. Candidates that don't flip, or whose replay\n * errors, land in `rejected` with a typed reason; nothing is dropped\n * silently.\n */\n\nimport {\n type CounterfactualMutation,\n type CounterfactualRunner,\n runCounterfactual,\n} from '../counterfactual'\nimport { ValidationError } from '../errors'\nimport type { TraceStore } from '../trace/store'\nimport { buildTrajectory, type Trajectory, type TrajectoryStep } from '../trajectory'\nimport type { StepRef, StepResponsibility } from './causal-sweep'\n\n/** Context handed to `proposeFix` so an LLM-backed proposer can see the\n * full trajectory plus the responsibility evidence for the blamed step. */\nexport interface RepairContext {\n runId: string\n trajectory: Trajectory\n originalScore: number\n responsibility: StepResponsibility\n}\n\nexport interface PrescribeRepairOptions {\n store: TraceStore\n /** The failed run the sweep diagnosed. */\n runId: string\n /** Execution seam — same `CounterfactualRunner` contract as the sweep. */\n runner: CounterfactualRunner\n /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */\n blamed: StepResponsibility[]\n /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.\n * Returned mutations MUST target the blamed step's index. */\n proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>\n /** Score every validation replay must reach for the repair to count. Default 0.5. */\n flipThreshold?: number\n /** Validation replays per candidate mutation. Default 3. */\n repsToValidate?: number\n /** Max candidate mutations tried per step. Default: all proposed. */\n maxAttemptsPerStep?: number\n}\n\nexport interface ValidatedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n /** Always true — presence in `repairs` IS the machine-verified claim. */\n validated: true\n /** Mean counterfactual score across the validation reps. */\n meanScore: number\n /** meanScore − originalScore. */\n deltaScore: number\n reps: number\n /** Replay run ids backing the validation — audit trail. */\n counterfactualRunIds: string[]\n}\n\nexport interface RejectedRepair {\n stepRef: StepRef\n mutation: CounterfactualMutation\n reason: 'did-not-flip' | 'error'\n /** Present for 'did-not-flip': mean delta over the reps that ran. */\n deltaScore?: number\n /** Present for 'error': the message, preserved for diagnosis. */\n error?: string\n}\n\nexport interface RepairReport {\n runId: string\n originalScore: number\n flipThreshold: number\n repairs: ValidatedRepair[]\n rejected: RejectedRepair[]\n replaysUsed: number\n}\n\nexport async function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport> {\n const flipThreshold = opts.flipThreshold ?? 0.5\n const repsToValidate = opts.repsToValidate ?? 3\n if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {\n throw new ValidationError(\n `prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`,\n )\n }\n const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY\n if (maxAttempts < 1) {\n throw new ValidationError(\n `prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`,\n )\n }\n if (opts.blamed.length === 0) {\n throw new ValidationError('prescribeRepair: blamed is empty — nothing to repair')\n }\n\n const originalRun = await opts.store.getRun(opts.runId)\n if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`)\n const originalScore = originalRun.outcome?.score\n if (typeof originalScore !== 'number' || !Number.isFinite(originalScore)) {\n throw new ValidationError(\n `prescribeRepair: run ${opts.runId} has no numeric outcome.score — flips have no baseline`,\n )\n }\n\n const trajectory = await buildTrajectory(opts.store, opts.runId)\n\n const repairs: ValidatedRepair[] = []\n const rejected: RejectedRepair[] = []\n let replaysUsed = 0\n\n for (const responsibility of opts.blamed) {\n const step = trajectory.steps[responsibility.stepRef.index]\n if (!step || step.span.spanId !== responsibility.stepRef.spanId) {\n throw new ValidationError(\n `prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} — stale report?`,\n )\n }\n\n const candidates = await opts.proposeFix(step, {\n runId: opts.runId,\n trajectory,\n originalScore,\n responsibility,\n })\n const toTry = candidates.slice(0, maxAttempts)\n\n for (const mutation of toTry) {\n if (mutation.at !== step.index) {\n throw new ValidationError(\n `prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`,\n )\n }\n const scores: number[] = []\n const cfRunIds: string[] = []\n let failure: string | undefined\n for (let rep = 0; rep < repsToValidate; rep++) {\n try {\n const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner)\n replaysUsed++\n const score = result.delta.counterfactualOutcomeScore\n if (typeof score !== 'number' || !Number.isFinite(score)) {\n failure = `validation rep ${rep} produced no numeric score — the runner must endRun with a numeric outcome.score`\n break\n }\n scores.push(score)\n cfRunIds.push(result.counterfactualRunId)\n } catch (err) {\n replaysUsed++\n failure = err instanceof Error ? err.message : String(err)\n break\n }\n }\n\n if (failure !== undefined) {\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'error',\n error: failure,\n })\n continue\n }\n\n const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length\n const everyRepFlipped = scores.every((s) => s >= flipThreshold)\n if (everyRepFlipped) {\n repairs.push({\n stepRef: responsibility.stepRef,\n mutation,\n validated: true,\n meanScore,\n deltaScore: meanScore - originalScore,\n reps: repsToValidate,\n counterfactualRunIds: cfRunIds,\n })\n // First validated repair per step IS the prescription; remaining\n // candidates are untried, not rejected — we don't fabricate verdicts.\n break\n }\n rejected.push({\n stepRef: responsibility.stepRef,\n mutation,\n reason: 'did-not-flip',\n deltaScore: meanScore - originalScore,\n })\n }\n }\n\n return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;AA2CO,SAAS,UAAU,MAA+B;AACvD,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK,KAAK;AAAA,IAClB,MAAM,KAAK,KAAK;AAAA,IAChB,MAAM,KAAK,KAAK;AAAA,EAClB;AACF;AAkEA,IAAM,kBAAkB;AAExB,SAAS,iBAAiB,MAAgD;AACxE,MAAI,KAAK,KAAK,SAAS,QAAQ;AAC7B,WAAO,CAAC,EAAE,MAAM,oBAAoB,IAAI,KAAK,OAAO,WAAW,KAAK,CAAC;AAAA,EACvE;AACA,MAAI,KAAK,KAAK,SAAS,OAAO;AAC5B,WAAO,CAAC,EAAE,MAAM,kBAAkB,IAAI,KAAK,MAAM,CAAC;AAAA,EACpD;AACA,SAAO,CAAC;AACV;AAEA,eAAsB,YAAY,MAA+D;AAC/F,MAAI,CAAC,OAAO,UAAU,KAAK,IAAI,KAAK,KAAK,OAAO,GAAG;AACjD,UAAM,IAAI;AAAA,MACR,kDAAkD,KAAK,IAAI;AAAA,IAC7D;AAAA,EACF;AACA,MAAI,CAAC,OAAO,UAAU,KAAK,MAAM,KAAK,KAAK,SAAS,GAAG;AACrD,UAAM,IAAI,gBAAgB,oDAAoD,KAAK,MAAM,GAAG;AAAA,EAC9F;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,oBAAoB,KAAK,KAAK,YAAY;AACtF,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,oBAAoB,KAAK,KAAK;AAAA,IAChC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAC/D,QAAM,aAAa,kBAAkB,YAAY,KAAK,cAAc;AACpE,QAAM,eAAe,KAAK,oBAAoB;AAM9C,QAAM,QAAgB,CAAC;AACvB,aAAW,QAAQ,YAAY;AAC7B,UAAM,YAAY,aAAa,IAAI;AACnC,eAAW,KAAK,WAAW;AACzB,UAAI,EAAE,OAAO,KAAK,OAAO;AACvB,cAAM,IAAI;AAAA,UACR,kEAAkE,EAAE,EAAE,mBAAmB,KAAK,KAAK;AAAA,QACrG;AAAA,MACF;AACA,YAAM,KAAK,EAAE,MAAM,UAAU,EAAE,CAAC;AAAA,IAClC;AAAA,EACF;AAEA,QAAM,mBAAyC,CAAC;AAChD,QAAM,aAAqC,CAAC;AAC5C,QAAM,mBAAmB,oBAAI,IAAY;AACzC,MAAI,cAAc;AAClB,MAAI,SAAS;AAEb,aAAW,QAAQ,OAAO;AACxB,QAAI,UAAU,cAAc,KAAK,OAAO,KAAK,QAAQ;AAGnD,eAAS;AACT,uBAAiB,IAAI,KAAK,KAAK,KAAK;AACpC;AAAA,IACF;AACA,UAAM,SAAmB,CAAC;AAC1B,UAAM,WAAqB,CAAC;AAC5B,aAAS,MAAM,GAAG,MAAM,KAAK,MAAM,OAAO;AACxC,YAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,KAAK,UAAU,KAAK,MAAM;AACzF;AACA,YAAM,IAAI,OAAO,MAAM;AACvB,UAAI,OAAO,MAAM,YAAY,CAAC,OAAO,SAAS,CAAC,GAAG;AAChD,cAAM,IAAI;AAAA,UACR,+CAA+C,KAAK,KAAK,KAAK,KAAK,KAAK,SAAS,IAAI,SAAS,GAAG;AAAA,QACnG;AAAA,MACF;AACA,aAAO,KAAK,CAAC;AACb,eAAS,KAAK,OAAO,mBAAmB;AACxC,iBAAW,KAAK,MAAM;AAAA,IACxB;AACA,UAAM,KAAK,mBAAmB,QAAQ,KAAK,gBAAgB,MAAM;AAAA,MAC/D,MAAM,KAAK,UAAU;AAAA,IACvB,CAAC;AACD,qBAAiB,KAAK;AAAA,MACpB,SAAS,UAAU,KAAK,IAAI;AAAA,MAC5B,cAAc,KAAK,SAAS;AAAA,MAC5B,YAAY,GAAG;AAAA,MACf;AAAA,MACA,gBAAgB,GAAG,QAAQ,KAAK,GAAG,QAAQ;AAAA,MAC3C,MAAM,KAAK;AAAA,MACX;AAAA,MACA,sBAAsB;AAAA,IACxB,CAAC;AAAA,EACH;AAEA,mBAAiB,KAAK,CAAC,GAAG,MAAM,KAAK,IAAI,EAAE,UAAU,IAAI,KAAK,IAAI,EAAE,UAAU,CAAC;AAI/E,QAAM,YAAY,WAAW,OAAO,CAAC,MAAM,iBAAiB,IAAI,EAAE,KAAK,CAAC,EAAE,IAAI,SAAS;AAEvF,SAAO;AAAA,IACL,OAAO,KAAK;AAAA,IACZ;AAAA,IACA,OAAO;AAAA,IACP,gBAAgB,yBAAyB,UAAU;AAAA,IACnD;AAAA,IACA,QAAQ,KAAK;AAAA,IACb;AAAA,EACF;AACF;AAEA,SAAS,kBAAkB,YAAwB,SAAsC;AACvF,MAAI,YAAY,QAAW;AACzB,WAAO,WAAW,MAAM,OAAO,CAAC,MAAM,EAAE,KAAK,SAAS,SAAS,EAAE,KAAK,SAAS,MAAM;AAAA,EACvF;AACA,SAAO,QAAQ,IAAI,CAAC,MAAM;AACxB,UAAM,OAAO,WAAW,MAAM,CAAC;AAC/B,QAAI,CAAC,MAAM;AACT,YAAM,IAAI;AAAA,QACR,qCAAqC,CAAC,qBAAqB,WAAW,MAAM,MAAM;AAAA,MACpF;AAAA,IACF;AACA,WAAO;AAAA,EACT,CAAC;AACH;;;ACzNO,IAAM,sBAAsB;AAK5B,SAAS,mBAAmB,gBAAqD;AACtF,MAAI,CAAC,eAAe,eAAgB,QAAO;AAC3C,QAAM,YAAY,KAAK,IAAI,eAAe,UAAU;AACpD,MAAI,aAAa,IAAK,QAAO;AAC7B,MAAI,aAAa,KAAM,QAAO;AAC9B,MAAI,aAAa,IAAK,QAAO;AAC7B,SAAO;AACT;AAIO,SAAS,iBAAiB,UAA0C;AACzE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO,cAAc,SAAS,QAAQ,aAAa,SAAS,EAAE;AAAA,IAChE,KAAK;AACH,aAAO,mCAAmC,SAAS,EAAE,SAAS,KAAK,UAAU,SAAS,SAAS,CAAC;AAAA,IAClG,KAAK;AACH,aAAO,2BAA2B,SAAS,EAAE;AAAA,IAC/C,KAAK;AACH,aAAO,iCAAiC,SAAS,EAAE,KAAK,SAAS,OAAO;AAAA,IAC1E,KAAK;AACH,aAAO,GAAG,SAAS,QAAQ,UAAU,SAAS,EAAE;AAAA,EACpD;AACF;AAWO,SAAS,kBACd,QACA,SACkB;AAClB,QAAM,eAAe,oBAAI,IAA6B;AACtD,aAAW,KAAK,SAAS,WAAW,CAAC,GAAG;AACtC,QAAI,CAAC,aAAa,IAAI,EAAE,QAAQ,MAAM,EAAG,cAAa,IAAI,EAAE,QAAQ,QAAQ,CAAC;AAAA,EAC/E;AAEA,SAAO,OAAO,MAAM,IAAI,CAAC,SAAS;AAChC,UAAM,SAAS,aAAa,IAAI,KAAK,QAAQ,MAAM;AACnD,UAAM,WAA0B;AAAA,MAC9B;AAAA,QACE,MAAM;AAAA,QACN,KAAK,UAAU,KAAK,QAAQ,MAAM;AAAA,QAClC,SAAS,QAAQ,KAAK,QAAQ,KAAK,KAAK,KAAK,QAAQ,IAAI,KAAK,KAAK,QAAQ,IAAI,iBAAiB,KAAK,WAAW,QAAQ,CAAC,CAAC,QAAQ,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,UAAU,KAAK,IAAI;AAAA,MAC5M;AAAA,MACA;AAAA,QACE,MAAM;AAAA,QACN,KAAK,qBAAqB,OAAO,KAAK,SAAS,KAAK,QAAQ,KAAK,IAAI,KAAK,YAAY;AAAA,QACtF,SAAS,WAAW,KAAK,OAAO,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC,EAAE,KAAK,IAAI,CAAC;AAAA,MACrE;AAAA,MACA,GAAG,KAAK,qBAAqB,IAAI,CAAC,QAAqB,EAAE,MAAM,QAAQ,KAAK,SAAS,EAAE,GAAG,EAAE;AAAA,IAC9F;AACA,WAAO,YAAY;AAAA,MACjB,YAAY;AAAA,MACZ,UAAU,mBAAmB,IAAI;AAAA,MACjC,MAAM;AAAA,MACN,OAAO,SAAS,KAAK,QAAQ,IAAI,MAAM,KAAK,QAAQ,IAAI,uDAAuD,KAAK,YAAY;AAAA,MAChI,WAAW,KAAK,iBACZ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,oBAChJ,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,SAAS,KAAK,IAAI,gCAAgC,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG,MAAM,QAAQ,CAAC,CAAC;AAAA,MACpJ,eAAe;AAAA,MACf,oBAAoB,SAAS,iBAAiB,OAAO,QAAQ,IAAI;AAAA,MACjE,iBAAiB,SACb,qBAAqB,OAAO,IAAI,IAAI,OAAO,IAAI,mBAAmB,QAAS,aAAa,UAAU,OAAO,UAAU,QAAQ,CAAC,CAAC,WAAW,OAAO,WAAW,QAAQ,CAAC,CAAC,MACpK;AAAA,MACJ,YAAY,SAAS,OAAO,KAAK,iBAAiB,OAAO;AAAA,MACzD,SAAS,KAAK,QAAQ;AAAA,MACtB,UAAU;AAAA,QACR,SAAS,KAAK;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,YAAY,KAAK;AAAA,QACjB,IAAI,KAAK;AAAA,QACT,QAAQ,KAAK;AAAA,QACb,sBAAsB,KAAK;AAAA,QAC3B,GAAI,SAAS,EAAE,QAAQ,EAAE,UAAU,OAAO,UAAU,WAAW,OAAO,UAAU,EAAE,IAAI,CAAC;AAAA,MACzF;AAAA,IACF,CAAC;AAAA,EACH,CAAC;AACH;AAaO,SAAS,eACd,KACA,QACA,OAAiD,CAAC,GACpC;AACd,QAAM,SAAuB;AAAA,IAC3B,GAAG;AAAA,IACH,OAAO,GAAG,IAAI,KAAK,WAAW,OAAO,QAAQ,MAAM;AAAA,IACnD,SAAS;AAAA,MACP,GAAG,IAAI;AAAA,MACP,KAAK;AAAA,QACH,GAAG,IAAI,QAAQ;AAAA,QACf,4BAA4B,OAAO,QAAQ;AAAA,QAC3C,4BAA4B,OAAO;AAAA,QACnC,6BAA6B,OAAO;AAAA,QACpC,sBAAsB,OAAO;AAAA,MAC/B;AAAA,IACF;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,YAAY,KAAK,cAAc,iBAAiB,OAAO,QAAQ;AAAA,EACjE;AAGA,oBAAkB,MAAM;AACxB,SAAO;AACT;AAgBO,SAAS,iBAAiB,QAAwC;AACvE,QAAM,EAAE,SAAS,SAAS,IAAI;AAC9B,QAAM,KAAK,QAAQ,QAAQ,KAAK,KAAK,QAAQ,IAAI,KAAK,QAAQ,IAAI;AAClE,UAAQ,SAAS,MAAM;AAAA,IACrB,KAAK;AACH,aAAO;AAAA,QACL,aAAa,uBAAuB,QAAQ,IAAI,4FAA4F,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QACxK,OAAO,iCAAiC,QAAQ,IAAI;AAAA,QACpD,SAAS,yBAAyB,QAAQ,IAAI;AAAA,MAChD;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,yBAAyB,EAAE,QAAQ,SAAS,QAAQ,gCAAgC,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC7H,OAAO,aAAa,QAAQ,IAAI,iCAAiC,SAAS,QAAQ;AAAA,MACpF;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,iCAAiC,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC3G,SAAS,8BAA8B,QAAQ,IAAI,MAAM,SAAS,OAAO;AAAA,MAC3E;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,kBAAkB,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,QAC5F,OAAO,wBAAwB,QAAQ,IAAI,YAAY,QAAQ,KAAK;AAAA,MACtE;AAAA,IACF,KAAK;AACH,aAAO;AAAA,QACL,aAAa,GAAG,SAAS,QAAQ,OAAO,EAAE,+BAA+B,OAAO,WAAW,QAAQ,CAAC,CAAC;AAAA,MACvG;AAAA,IACF,SAAS;AACP,YAAM,YAAmB;AACzB,YAAM,IAAI;AAAA,QACR,2CAA2C,KAAK,UAAU,SAAS,CAAC;AAAA,MACtE;AAAA,IACF;AAAA,EACF;AACF;;;ACtHA,eAAsB,gBAAgB,MAAqD;AACzF,QAAM,gBAAgB,KAAK,iBAAiB;AAC5C,QAAM,iBAAiB,KAAK,kBAAkB;AAC9C,MAAI,CAAC,OAAO,UAAU,cAAc,KAAK,iBAAiB,GAAG;AAC3D,UAAM,IAAI;AAAA,MACR,gEAAgE,cAAc;AAAA,IAChF;AAAA,EACF;AACA,QAAM,cAAc,KAAK,sBAAsB,OAAO;AACtD,MAAI,cAAc,GAAG;AACnB,UAAM,IAAI;AAAA,MACR,yDAAyD,KAAK,kBAAkB;AAAA,IAClF;AAAA,EACF;AACA,MAAI,KAAK,OAAO,WAAW,GAAG;AAC5B,UAAM,IAAI,gBAAgB,2DAAsD;AAAA,EAClF;AAEA,QAAM,cAAc,MAAM,KAAK,MAAM,OAAO,KAAK,KAAK;AACtD,MAAI,CAAC,YAAa,OAAM,IAAI,gBAAgB,wBAAwB,KAAK,KAAK,YAAY;AAC1F,QAAM,gBAAgB,YAAY,SAAS;AAC3C,MAAI,OAAO,kBAAkB,YAAY,CAAC,OAAO,SAAS,aAAa,GAAG;AACxE,UAAM,IAAI;AAAA,MACR,wBAAwB,KAAK,KAAK;AAAA,IACpC;AAAA,EACF;AAEA,QAAM,aAAa,MAAM,gBAAgB,KAAK,OAAO,KAAK,KAAK;AAE/D,QAAM,UAA6B,CAAC;AACpC,QAAM,WAA6B,CAAC;AACpC,MAAI,cAAc;AAElB,aAAW,kBAAkB,KAAK,QAAQ;AACxC,UAAM,OAAO,WAAW,MAAM,eAAe,QAAQ,KAAK;AAC1D,QAAI,CAAC,QAAQ,KAAK,KAAK,WAAW,eAAe,QAAQ,QAAQ;AAC/D,YAAM,IAAI;AAAA,QACR,sCAAsC,eAAe,QAAQ,KAAK,WAAW,eAAe,QAAQ,MAAM,uBAAuB,KAAK,KAAK;AAAA,MAC7I;AAAA,IACF;AAEA,UAAM,aAAa,MAAM,KAAK,WAAW,MAAM;AAAA,MAC7C,OAAO,KAAK;AAAA,MACZ;AAAA,MACA;AAAA,MACA;AAAA,IACF,CAAC;AACD,UAAM,QAAQ,WAAW,MAAM,GAAG,WAAW;AAE7C,eAAW,YAAY,OAAO;AAC5B,UAAI,SAAS,OAAO,KAAK,OAAO;AAC9B,cAAM,IAAI;AAAA,UACR,gEAAgE,SAAS,EAAE,0BAA0B,KAAK,KAAK;AAAA,QACjH;AAAA,MACF;AACA,YAAM,SAAmB,CAAC;AAC1B,YAAM,WAAqB,CAAC;AAC5B,UAAI;AACJ,eAAS,MAAM,GAAG,MAAM,gBAAgB,OAAO;AAC7C,YAAI;AACF,gBAAM,SAAS,MAAM,kBAAkB,KAAK,OAAO,KAAK,OAAO,UAAU,KAAK,MAAM;AACpF;AACA,gBAAM,QAAQ,OAAO,MAAM;AAC3B,cAAI,OAAO,UAAU,YAAY,CAAC,OAAO,SAAS,KAAK,GAAG;AACxD,sBAAU,kBAAkB,GAAG;AAC/B;AAAA,UACF;AACA,iBAAO,KAAK,KAAK;AACjB,mBAAS,KAAK,OAAO,mBAAmB;AAAA,QAC1C,SAAS,KAAK;AACZ;AACA,oBAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AACzD;AAAA,QACF;AAAA,MACF;AAEA,UAAI,YAAY,QAAW;AACzB,iBAAS,KAAK;AAAA,UACZ,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,QAAQ;AAAA,UACR,OAAO;AAAA,QACT,CAAC;AACD;AAAA,MACF;AAEA,YAAM,YAAY,OAAO,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;AAC7D,YAAM,kBAAkB,OAAO,MAAM,CAAC,MAAM,KAAK,aAAa;AAC9D,UAAI,iBAAiB;AACnB,gBAAQ,KAAK;AAAA,UACX,SAAS,eAAe;AAAA,UACxB;AAAA,UACA,WAAW;AAAA,UACX;AAAA,UACA,YAAY,YAAY;AAAA,UACxB,MAAM;AAAA,UACN,sBAAsB;AAAA,QACxB,CAAC;AAGD;AAAA,MACF;AACA,eAAS,KAAK;AAAA,QACZ,SAAS,eAAe;AAAA,QACxB;AAAA,QACA,QAAQ;AAAA,QACR,YAAY,YAAY;AAAA,MAC1B,CAAC;AAAA,IACH;AAAA,EACF;AAEA,SAAO,EAAE,OAAO,KAAK,OAAO,eAAe,eAAe,SAAS,UAAU,YAAY;AAC3F;","names":[]}
|
|
@@ -1,169 +0,0 @@
|
|
|
1
|
-
import { C as ControlEvalResult, a as ControlRunResult, b as ControlStep } from './control-runtime-Acf9CGhw.js';
|
|
2
|
-
import { D as DatasetSplit, a as DatasetScenario } from './dataset-DS7ytHZU.js';
|
|
3
|
-
|
|
4
|
-
type FeedbackArtifactType = 'text' | 'code' | 'plan' | 'research' | 'action' | 'ui' | 'decision' | 'data' | 'other';
|
|
5
|
-
type FeedbackLabelSource = 'user' | 'judge' | 'environment' | 'metric' | 'policy' | 'system';
|
|
6
|
-
type FeedbackLabelKind = 'approve' | 'reject' | 'select' | 'edit' | 'rank' | 'rate' | 'comment' | 'metric_outcome' | 'policy_block' | 'revision_request';
|
|
7
|
-
type FeedbackSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
8
|
-
interface FeedbackTask {
|
|
9
|
-
intent: string;
|
|
10
|
-
context?: unknown;
|
|
11
|
-
}
|
|
12
|
-
interface ProposedSideEffect {
|
|
13
|
-
type: string;
|
|
14
|
-
risk?: 'low' | 'medium' | 'high';
|
|
15
|
-
costUsd?: number;
|
|
16
|
-
externalSideEffect?: boolean;
|
|
17
|
-
requiresApproval?: boolean;
|
|
18
|
-
metadata?: Record<string, unknown>;
|
|
19
|
-
}
|
|
20
|
-
interface FeedbackLabel {
|
|
21
|
-
id?: string;
|
|
22
|
-
source: FeedbackLabelSource;
|
|
23
|
-
kind: FeedbackLabelKind;
|
|
24
|
-
value: unknown;
|
|
25
|
-
reason?: string;
|
|
26
|
-
severity?: FeedbackSeverity;
|
|
27
|
-
createdAt: string;
|
|
28
|
-
metadata?: Record<string, unknown>;
|
|
29
|
-
}
|
|
30
|
-
interface FeedbackAttempt {
|
|
31
|
-
id: string;
|
|
32
|
-
stepIndex: number;
|
|
33
|
-
artifactType: FeedbackArtifactType;
|
|
34
|
-
artifact: unknown;
|
|
35
|
-
options?: unknown[];
|
|
36
|
-
proposedAction?: ProposedSideEffect;
|
|
37
|
-
evals?: ControlEvalResult[];
|
|
38
|
-
feedback?: FeedbackLabel[];
|
|
39
|
-
createdAt: string;
|
|
40
|
-
metadata?: Record<string, unknown>;
|
|
41
|
-
}
|
|
42
|
-
interface FeedbackOutcome {
|
|
43
|
-
success?: boolean;
|
|
44
|
-
score?: number;
|
|
45
|
-
metrics?: Record<string, number>;
|
|
46
|
-
costUsd?: number;
|
|
47
|
-
detail?: string;
|
|
48
|
-
observedAt?: string;
|
|
49
|
-
metadata?: Record<string, unknown>;
|
|
50
|
-
}
|
|
51
|
-
interface FeedbackTrajectory {
|
|
52
|
-
id: string;
|
|
53
|
-
projectId?: string;
|
|
54
|
-
scenarioId?: string;
|
|
55
|
-
task: FeedbackTask;
|
|
56
|
-
attempts: FeedbackAttempt[];
|
|
57
|
-
labels: FeedbackLabel[];
|
|
58
|
-
outcome?: FeedbackOutcome;
|
|
59
|
-
split?: DatasetSplit;
|
|
60
|
-
tags?: Record<string, string>;
|
|
61
|
-
createdAt: string;
|
|
62
|
-
updatedAt?: string;
|
|
63
|
-
metadata?: Record<string, unknown>;
|
|
64
|
-
}
|
|
65
|
-
interface FeedbackTrajectoryStore {
|
|
66
|
-
save(trajectory: FeedbackTrajectory): Promise<void>;
|
|
67
|
-
get(id: string): Promise<FeedbackTrajectory | null>;
|
|
68
|
-
list(filter?: FeedbackTrajectoryFilter): Promise<FeedbackTrajectory[]>;
|
|
69
|
-
appendAttempt(id: string, attempt: FeedbackAttempt): Promise<FeedbackTrajectory>;
|
|
70
|
-
appendLabel(id: string, label: FeedbackLabel, attemptId?: string): Promise<FeedbackTrajectory>;
|
|
71
|
-
}
|
|
72
|
-
interface FeedbackTrajectoryFilter {
|
|
73
|
-
projectId?: string;
|
|
74
|
-
scenarioId?: string;
|
|
75
|
-
split?: DatasetSplit;
|
|
76
|
-
tag?: [string, string];
|
|
77
|
-
}
|
|
78
|
-
interface FeedbackSplitPolicy {
|
|
79
|
-
trainPct?: number;
|
|
80
|
-
devPct?: number;
|
|
81
|
-
testPct?: number;
|
|
82
|
-
holdoutPct?: number;
|
|
83
|
-
}
|
|
84
|
-
interface PreferenceMemoryEntry {
|
|
85
|
-
instruction: string;
|
|
86
|
-
rationale: string;
|
|
87
|
-
weight: number;
|
|
88
|
-
sourceTrajectoryId: string;
|
|
89
|
-
sourceLabelId?: string;
|
|
90
|
-
category?: string;
|
|
91
|
-
}
|
|
92
|
-
interface FeedbackOptimizerRow {
|
|
93
|
-
scenarioId: string;
|
|
94
|
-
trajectoryId: string;
|
|
95
|
-
labelKinds: FeedbackLabelKind[];
|
|
96
|
-
score?: number;
|
|
97
|
-
metadata?: Record<string, unknown>;
|
|
98
|
-
}
|
|
99
|
-
interface FeedbackReplayResult {
|
|
100
|
-
trajectoryId: string;
|
|
101
|
-
pass: boolean;
|
|
102
|
-
score?: number;
|
|
103
|
-
labels: FeedbackLabel[];
|
|
104
|
-
outcome?: FeedbackOutcome;
|
|
105
|
-
metadata?: Record<string, unknown>;
|
|
106
|
-
}
|
|
107
|
-
interface FeedbackReplayAdapter {
|
|
108
|
-
replay(trajectory: FeedbackTrajectory): Promise<Omit<FeedbackReplayResult, 'trajectoryId'>> | Omit<FeedbackReplayResult, 'trajectoryId'>;
|
|
109
|
-
}
|
|
110
|
-
declare class InMemoryFeedbackTrajectoryStore implements FeedbackTrajectoryStore {
|
|
111
|
-
private readonly trajectories;
|
|
112
|
-
save(trajectory: FeedbackTrajectory): Promise<void>;
|
|
113
|
-
get(id: string): Promise<FeedbackTrajectory | null>;
|
|
114
|
-
list(filter?: FeedbackTrajectoryFilter): Promise<FeedbackTrajectory[]>;
|
|
115
|
-
appendAttempt(id: string, attempt: FeedbackAttempt): Promise<FeedbackTrajectory>;
|
|
116
|
-
appendLabel(id: string, label: FeedbackLabel, attemptId?: string): Promise<FeedbackTrajectory>;
|
|
117
|
-
}
|
|
118
|
-
declare class FileSystemFeedbackTrajectoryStore implements FeedbackTrajectoryStore {
|
|
119
|
-
private readonly dir;
|
|
120
|
-
private readonly memory;
|
|
121
|
-
private loaded;
|
|
122
|
-
constructor(options: {
|
|
123
|
-
dir: string;
|
|
124
|
-
});
|
|
125
|
-
save(trajectory: FeedbackTrajectory): Promise<void>;
|
|
126
|
-
get(id: string): Promise<FeedbackTrajectory | null>;
|
|
127
|
-
list(filter?: FeedbackTrajectoryFilter): Promise<FeedbackTrajectory[]>;
|
|
128
|
-
appendAttempt(id: string, attempt: FeedbackAttempt): Promise<FeedbackTrajectory>;
|
|
129
|
-
appendLabel(id: string, label: FeedbackLabel, attemptId?: string): Promise<FeedbackTrajectory>;
|
|
130
|
-
private append;
|
|
131
|
-
private load;
|
|
132
|
-
}
|
|
133
|
-
declare function createFeedbackTrajectory(input: {
|
|
134
|
-
id?: string;
|
|
135
|
-
projectId?: string;
|
|
136
|
-
scenarioId?: string;
|
|
137
|
-
task: FeedbackTask;
|
|
138
|
-
attempts?: FeedbackAttempt[];
|
|
139
|
-
labels?: FeedbackLabel[];
|
|
140
|
-
outcome?: FeedbackOutcome;
|
|
141
|
-
split?: DatasetSplit;
|
|
142
|
-
tags?: Record<string, string>;
|
|
143
|
-
createdAt?: string;
|
|
144
|
-
metadata?: Record<string, unknown>;
|
|
145
|
-
}): FeedbackTrajectory;
|
|
146
|
-
declare function assignFeedbackSplit(trajectory: Pick<FeedbackTrajectory, 'id' | 'projectId' | 'scenarioId' | 'task'>, policy?: FeedbackSplitPolicy): DatasetSplit;
|
|
147
|
-
declare function withAssignedFeedbackSplit(trajectory: FeedbackTrajectory, policy?: FeedbackSplitPolicy): FeedbackTrajectory;
|
|
148
|
-
declare function feedbackTrajectoryToDatasetScenario(trajectory: FeedbackTrajectory): DatasetScenario;
|
|
149
|
-
declare function feedbackTrajectoriesToDatasetScenarios(trajectories: FeedbackTrajectory[]): DatasetScenario[];
|
|
150
|
-
declare function feedbackTrajectoryToOptimizerRow(trajectory: FeedbackTrajectory): FeedbackOptimizerRow;
|
|
151
|
-
declare function feedbackTrajectoriesToOptimizerRows(trajectories: FeedbackTrajectory[]): FeedbackOptimizerRow[];
|
|
152
|
-
declare function replayFeedbackTrajectory(trajectory: FeedbackTrajectory, adapter: FeedbackReplayAdapter): Promise<FeedbackReplayResult>;
|
|
153
|
-
declare function replayFeedbackTrajectories(trajectories: FeedbackTrajectory[], adapter: FeedbackReplayAdapter): Promise<FeedbackReplayResult[]>;
|
|
154
|
-
declare function summarizePreferenceMemory(trajectories: FeedbackTrajectory[], options?: {
|
|
155
|
-
maxEntries?: number;
|
|
156
|
-
}): PreferenceMemoryEntry[];
|
|
157
|
-
declare function renderPreferenceMemoryMarkdown(entries: PreferenceMemoryEntry[]): string;
|
|
158
|
-
declare function serializeFeedbackTrajectoriesJsonl(trajectories: FeedbackTrajectory[]): string;
|
|
159
|
-
declare function parseFeedbackTrajectoriesJsonl(jsonl: string): FeedbackTrajectory[];
|
|
160
|
-
declare function controlRunToFeedbackTrajectory<TState, TAction, TActionResult>(run: ControlRunResult<TState, TAction, TActionResult>, options?: {
|
|
161
|
-
projectId?: string;
|
|
162
|
-
scenarioId?: string;
|
|
163
|
-
artifactType?: FeedbackArtifactType;
|
|
164
|
-
artifactFromStep?: (step: ControlStep<TState, TAction, TActionResult>) => unknown;
|
|
165
|
-
proposedActionFromStep?: (step: ControlStep<TState, TAction, TActionResult>) => ProposedSideEffect | undefined;
|
|
166
|
-
createdAt?: string;
|
|
167
|
-
}): FeedbackTrajectory;
|
|
168
|
-
|
|
169
|
-
export { replayFeedbackTrajectory as A, serializeFeedbackTrajectoriesJsonl as B, summarizePreferenceMemory as C, withAssignedFeedbackSplit as D, type FeedbackTrajectoryStore as F, InMemoryFeedbackTrajectoryStore as I, type PreferenceMemoryEntry as P, type FeedbackTrajectory as a, type FeedbackLabel as b, type FeedbackArtifactType as c, type FeedbackAttempt as d, type FeedbackLabelKind as e, type FeedbackLabelSource as f, type FeedbackOptimizerRow as g, type FeedbackOutcome as h, type FeedbackReplayAdapter as i, type FeedbackReplayResult as j, type FeedbackSeverity as k, type FeedbackSplitPolicy as l, type FeedbackTask as m, type FeedbackTrajectoryFilter as n, FileSystemFeedbackTrajectoryStore as o, type ProposedSideEffect as p, assignFeedbackSplit as q, controlRunToFeedbackTrajectory as r, createFeedbackTrajectory as s, feedbackTrajectoriesToDatasetScenarios as t, feedbackTrajectoriesToOptimizerRows as u, feedbackTrajectoryToDatasetScenario as v, feedbackTrajectoryToOptimizerRow as w, parseFeedbackTrajectoriesJsonl as x, renderPreferenceMemoryMarkdown as y, replayFeedbackTrajectories as z };
|
|
@@ -1,135 +0,0 @@
|
|
|
1
|
-
import { c as DatasetManifest } from '../dataset-DS7ytHZU.js';
|
|
2
|
-
import { a as CalibrationResult } from '../judge-calibration-7C-IDmKr.js';
|
|
3
|
-
import { b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
4
|
-
import { d as RedTeamReport } from '../red-team-KmmiqBlY.js';
|
|
5
|
-
import { T as TraceStore } from '../store-BcFXE6LG.js';
|
|
6
|
-
import '../errors-oeQrLqXC.js';
|
|
7
|
-
import '../schema-m0gsnbt3.js';
|
|
8
|
-
|
|
9
|
-
/**
|
|
10
|
-
* Governance reporting — shared types.
|
|
11
|
-
*
|
|
12
|
-
* The framework collects a `GovernanceContext` (traces + outcomes +
|
|
13
|
-
* dataset manifests + red-team results + judge calibration) and each
|
|
14
|
-
* specific template (NIST AI RMF, SOC2, EU AI Act) renders a
|
|
15
|
-
* structured report from it.
|
|
16
|
-
*
|
|
17
|
-
* Reports are machine-readable JSON first; human-readable Markdown is a
|
|
18
|
-
* pure transform on top. External auditors consume the Markdown; CI
|
|
19
|
-
* consumes the JSON.
|
|
20
|
-
*/
|
|
21
|
-
|
|
22
|
-
interface GovernanceContext {
|
|
23
|
-
/** Legal / org identity for the report. */
|
|
24
|
-
organization: string;
|
|
25
|
-
/** System / agent identifier. */
|
|
26
|
-
systemName: string;
|
|
27
|
-
/** ISO8601 period the report covers. */
|
|
28
|
-
periodStart: string;
|
|
29
|
-
periodEnd: string;
|
|
30
|
-
/** Versioned dataset manifests used during the period. */
|
|
31
|
-
datasets: DatasetManifest[];
|
|
32
|
-
traceStore: TraceStore;
|
|
33
|
-
outcomeStore?: OutcomeStore;
|
|
34
|
-
/** Cached red-team results for the period, if available. */
|
|
35
|
-
redTeam?: RedTeamReport;
|
|
36
|
-
/** Judge-vs-human calibration results, if measured. */
|
|
37
|
-
judgeCalibration?: CalibrationResult[];
|
|
38
|
-
/** Responsible owner for the system — role + name + email. */
|
|
39
|
-
owner: {
|
|
40
|
-
role: string;
|
|
41
|
-
name: string;
|
|
42
|
-
email: string;
|
|
43
|
-
};
|
|
44
|
-
}
|
|
45
|
-
interface GovernanceFinding {
|
|
46
|
-
id: string;
|
|
47
|
-
severity: 'info' | 'low' | 'medium' | 'high' | 'critical';
|
|
48
|
-
/** Control reference the finding maps to (e.g. "NIST-AI-RMF:MEASURE-2.1"). */
|
|
49
|
-
control: string;
|
|
50
|
-
summary: string;
|
|
51
|
-
evidence?: string;
|
|
52
|
-
remediation?: string;
|
|
53
|
-
}
|
|
54
|
-
interface GovernanceReport {
|
|
55
|
-
framework: 'NIST-AI-RMF' | 'SOC2' | 'EU-AI-ACT';
|
|
56
|
-
version: string;
|
|
57
|
-
context: Pick<GovernanceContext, 'organization' | 'systemName' | 'periodStart' | 'periodEnd' | 'owner'>;
|
|
58
|
-
summary: {
|
|
59
|
-
findings: number;
|
|
60
|
-
byeverity: Record<GovernanceFinding['severity'], number>;
|
|
61
|
-
overall: 'compliant' | 'compliant-with-findings' | 'non-compliant';
|
|
62
|
-
};
|
|
63
|
-
findings: GovernanceFinding[];
|
|
64
|
-
/** Framework-specific structured payload (mapped controls, risk class, etc.). */
|
|
65
|
-
payload: Record<string, unknown>;
|
|
66
|
-
generatedAt: string;
|
|
67
|
-
}
|
|
68
|
-
declare function renderMarkdown(report: GovernanceReport): string;
|
|
69
|
-
declare function summarize(findings: GovernanceFinding[]): GovernanceReport['summary'];
|
|
70
|
-
|
|
71
|
-
/**
|
|
72
|
-
* EU AI Act — risk-class classification + compliance checklist.
|
|
73
|
-
*
|
|
74
|
-
* Classification is declarative: caller supplies the domain/use-case
|
|
75
|
-
* signals (biometric? critical infrastructure? education? employment?
|
|
76
|
-
* access to services?) and we map to the Act's risk tiers:
|
|
77
|
-
* - "unacceptable" (prohibited)
|
|
78
|
-
* - "high" (Annex III — strict obligations)
|
|
79
|
-
* - "limited" (transparency obligations)
|
|
80
|
-
* - "minimal" (voluntary codes of conduct)
|
|
81
|
-
*
|
|
82
|
-
* Then the compliance checklist enumerates Article 9 (risk mgmt),
|
|
83
|
-
* 10 (data + data governance), 11 (technical documentation), 13
|
|
84
|
-
* (transparency), 14 (human oversight), 15 (accuracy + robustness)
|
|
85
|
-
* requirements and flags gaps.
|
|
86
|
-
*/
|
|
87
|
-
|
|
88
|
-
type EuRiskClass = 'unacceptable' | 'high' | 'limited' | 'minimal';
|
|
89
|
-
interface UseCaseSignals {
|
|
90
|
-
/** Used for biometric identification in public spaces? (Art. 5 — unacceptable). */
|
|
91
|
-
biometricPublic?: boolean;
|
|
92
|
-
/** Social scoring by public authorities? (Art. 5). */
|
|
93
|
-
socialScoring?: boolean;
|
|
94
|
-
/** Subliminal manipulation? (Art. 5). */
|
|
95
|
-
subliminal?: boolean;
|
|
96
|
-
/** Annex III sector: critical infrastructure / education / employment /
|
|
97
|
-
* access to essential services / law enforcement / migration /
|
|
98
|
-
* administration of justice / democratic processes? */
|
|
99
|
-
annexIII?: boolean;
|
|
100
|
-
/** Interacts directly with natural persons (chatbot, agent)? — limited risk. */
|
|
101
|
-
chatbot?: boolean;
|
|
102
|
-
/** Generates synthetic media (image/audio/video/text deepfakes)? — limited risk. */
|
|
103
|
-
generatesSyntheticMedia?: boolean;
|
|
104
|
-
}
|
|
105
|
-
declare function classifyEuAiRisk(signals: UseCaseSignals): EuRiskClass;
|
|
106
|
-
declare function euAiActReport(ctx: GovernanceContext, signals: UseCaseSignals): Promise<GovernanceReport>;
|
|
107
|
-
|
|
108
|
-
/**
|
|
109
|
-
* NIST AI RMF 1.0 — Govern / Map / Measure / Manage mapping.
|
|
110
|
-
*
|
|
111
|
-
* Each subcategory derives its status from concrete framework state:
|
|
112
|
-
* MEASURE 2.x: do we have a calibration regime? contamination controls?
|
|
113
|
-
* MEASURE 2.7: are red-team results available?
|
|
114
|
-
* MANAGE 1.x: are outcome metrics captured? correlation measured?
|
|
115
|
-
* GOVERN 1.x: dataset + prompt provenance recorded?
|
|
116
|
-
*
|
|
117
|
-
* We ship the mapping and the derivation rules; consumers supply the
|
|
118
|
-
* GovernanceContext.
|
|
119
|
-
*/
|
|
120
|
-
|
|
121
|
-
declare function nistAiRmfReport(ctx: GovernanceContext): Promise<GovernanceReport>;
|
|
122
|
-
|
|
123
|
-
/**
|
|
124
|
-
* SOC 2 — Common Criteria 7 (system operations + change management)
|
|
125
|
-
* audit trail derived from the trace corpus.
|
|
126
|
-
*
|
|
127
|
-
* This is NOT a formal SOC2 report — that requires an external
|
|
128
|
-
* auditor. What we ship is the machine-readable *evidence* package
|
|
129
|
-
* that an auditor consumes: run counts, deploy events, access log
|
|
130
|
-
* summary, anomaly tracking, response-time SLOs.
|
|
131
|
-
*/
|
|
132
|
-
|
|
133
|
-
declare function soc2Report(ctx: GovernanceContext): Promise<GovernanceReport>;
|
|
134
|
-
|
|
135
|
-
export { type EuRiskClass, type GovernanceContext, type GovernanceFinding, type GovernanceReport, type UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize };
|
package/dist/governance/index.js
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
classifyEuAiRisk,
|
|
3
|
-
euAiActReport,
|
|
4
|
-
nistAiRmfReport,
|
|
5
|
-
renderMarkdown,
|
|
6
|
-
soc2Report,
|
|
7
|
-
summarize
|
|
8
|
-
} from "../chunk-KKHDIONI.js";
|
|
9
|
-
import "../chunk-PZ5AY32C.js";
|
|
10
|
-
export {
|
|
11
|
-
classifyEuAiRisk,
|
|
12
|
-
euAiActReport,
|
|
13
|
-
nistAiRmfReport,
|
|
14
|
-
renderMarkdown,
|
|
15
|
-
soc2Report,
|
|
16
|
-
summarize
|
|
17
|
-
};
|
|
18
|
-
//# sourceMappingURL=index.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
|
@@ -1,112 +0,0 @@
|
|
|
1
|
-
import { S as Span } from '../schema-m0gsnbt3.js';
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* Groundedness — "did the retrieval PROVIDER surface what the task needed?"
|
|
5
|
-
*
|
|
6
|
-
* A search/research provider returns text; the task needed certain facts or
|
|
7
|
-
* symbols to be solvable (the CURRENT API, a version number, a function name).
|
|
8
|
-
* This module scores how much of that required knowledge the provider's results
|
|
9
|
-
* actually surfaced — isolating PROVIDER quality (was the right thing
|
|
10
|
-
* retrievable / returned) from AGENT skill (did the agent then use it). A high
|
|
11
|
-
* groundedness score with a failed run blames the agent; a low score blames the
|
|
12
|
-
* provider. That separation is the whole point — pass/fail alone cannot make it.
|
|
13
|
-
*
|
|
14
|
-
* Structural sibling of `../authenticity`:
|
|
15
|
-
* - authenticity scores the agent's PRODUCED files for realness.
|
|
16
|
-
* - groundedness scores the provider's RETRIEVED text for coverage.
|
|
17
|
-
* Both are pure deterministic scorers whose DOMAIN config is supplied by the
|
|
18
|
-
* consumer (authenticity: `AuthenticitySignals`; groundedness:
|
|
19
|
-
* `requiredKnowledge: string[]`) — neither bakes in a benchmark's vocabulary.
|
|
20
|
-
*
|
|
21
|
-
* Relationship to `keyword-coverage-judge`: that judge scores the agent's
|
|
22
|
-
* SERVED OUTPUT (HTML + assets) for expected concepts — a different input
|
|
23
|
-
* (produced deliverable) answering a different question (deliverable quality).
|
|
24
|
-
* Groundedness reads the RETRIEVAL side (provider results). They are
|
|
25
|
-
* complementary coverage scorers over different stages of the run, not
|
|
26
|
-
* duplicates; do not collapse one into the other.
|
|
27
|
-
*
|
|
28
|
-
* Two seams, neither forked:
|
|
29
|
-
* - PURE SCORER `scoreGroundedness(resultText, requiredKnowledge)` — case-
|
|
30
|
-
* insensitive substring containment over a deduped key set. Fail-open: with
|
|
31
|
-
* no required knowledge there is nothing to ground, so `score = 1`.
|
|
32
|
-
* - TRACE EXTRACTOR `extractRetrievedText(spans, opts?)` — pulls the provider's
|
|
33
|
-
* returned text out of the canonical `TraceSchema` spans (`RetrievalSpan.hits`
|
|
34
|
-
* + provider `ToolSpan.result`) instead of re-parsing bespoke run files. This
|
|
35
|
-
* is the retrieval-side analog of `extractProducedState` (events → produced
|
|
36
|
-
* files): structural span input, no IO, no disk walking.
|
|
37
|
-
*/
|
|
38
|
-
|
|
39
|
-
interface GroundednessResult {
|
|
40
|
-
/** 0..1 share of required knowledge surfaced by the provider's results.
|
|
41
|
-
* 1 when there is nothing to ground (`requiredKnowledge` empty) — fail-open. */
|
|
42
|
-
score: number;
|
|
43
|
-
/** The required-knowledge keys the result text surfaced (deduped, original casing). */
|
|
44
|
-
found: string[];
|
|
45
|
-
/** The required-knowledge keys the result text did NOT surface. */
|
|
46
|
-
missing: string[];
|
|
47
|
-
/** Distinct required-knowledge keys after dedup — the denominator of `score`. */
|
|
48
|
-
total: number;
|
|
49
|
-
/** Did the provider return any result text at all? Distinguishes "provider
|
|
50
|
-
* surfaced nothing" (`!hadResults`) from "returned text but missed the facts"
|
|
51
|
-
* (`hadResults && score < 1`) — the same provider-vs-agent split as the score. */
|
|
52
|
-
hadResults: boolean;
|
|
53
|
-
}
|
|
54
|
-
/**
|
|
55
|
-
* Score how much of `requiredKnowledge` the retrieval provider's `resultText`
|
|
56
|
-
* surfaced. Pure — same inputs, same output. No IO, no LLM.
|
|
57
|
-
*
|
|
58
|
-
* Matching is case-insensitive substring containment: each required key is
|
|
59
|
-
* checked against the lower-cased result text. This is intentionally the same
|
|
60
|
-
* cheap, deterministic containment the authenticity scorer uses for its
|
|
61
|
-
* structural signals — a key is "surfaced" if the provider's returned text
|
|
62
|
-
* mentions it. Semantic / paraphrase coverage is a separate (LLM) layer a
|
|
63
|
-
* consumer can stack on top, exactly as authenticity stacks its nuance judge.
|
|
64
|
-
*
|
|
65
|
-
* Fail-open at `total === 0`: a task with no required knowledge has nothing for
|
|
66
|
-
* the provider to ground, so it cannot be penalized (`score = 1`). The benchmark
|
|
67
|
-
* caller decides what `requiredKnowledge` is — the substrate never derives it.
|
|
68
|
-
*/
|
|
69
|
-
declare function scoreGroundedness(resultText: string, requiredKnowledge: readonly string[]): GroundednessResult;
|
|
70
|
-
/**
|
|
71
|
-
* Predicate selecting which `ToolSpan`s are retrieval-PROVIDER calls (whose
|
|
72
|
-
* `result` carries returned text), by tool name. A parameter — never a baked-in
|
|
73
|
-
* literal — so the substrate stays free of any one benchmark's tool vocabulary,
|
|
74
|
-
* exactly as `AuthenticitySignals` keeps all domain regexes consumer-supplied.
|
|
75
|
-
*/
|
|
76
|
-
type ProviderToolMatcher = (toolName: string) => boolean;
|
|
77
|
-
/**
|
|
78
|
-
* Default provider matcher: tool names that look like search/research but not a
|
|
79
|
-
* plain fetch/read. A sensible starting point for the common "search arm" shape;
|
|
80
|
-
* any consumer with different tool names passes its own matcher. `RetrievalSpan`s
|
|
81
|
-
* are ALWAYS included regardless of this matcher (they are retrieval by kind);
|
|
82
|
-
* the matcher only selects which generic `ToolSpan`s also count as provider calls.
|
|
83
|
-
*/
|
|
84
|
-
declare const defaultProviderToolMatcher: ProviderToolMatcher;
|
|
85
|
-
interface ExtractRetrievedTextOptions {
|
|
86
|
-
/** Which `ToolSpan`s count as provider calls. Default: {@link defaultProviderToolMatcher}. */
|
|
87
|
-
isProviderTool?: ProviderToolMatcher;
|
|
88
|
-
}
|
|
89
|
-
/**
|
|
90
|
-
* Extract the retrieval PROVIDER's returned text from a span stream — the
|
|
91
|
-
* retrieval-side analog of `extractProducedState`. Reads the canonical
|
|
92
|
-
* `TraceSchema` carriers, NOT bespoke run files:
|
|
93
|
-
* - every `RetrievalSpan`'s `hits[].content` (kind 'retrieval' — the
|
|
94
|
-
* substrate's first-class search/research result carrier; the same `.hits`
|
|
95
|
-
* the `bad_retrieval` failure detector already reads), and
|
|
96
|
-
* - `ToolSpan.result` for tool spans whose `toolName` the provider matcher
|
|
97
|
-
* accepts (kind 'tool').
|
|
98
|
-
*
|
|
99
|
-
* Pure and total: spans of other kinds, and provider tools with no result, are
|
|
100
|
-
* skipped. Returns one text blob ready for `scoreGroundedness`.
|
|
101
|
-
*/
|
|
102
|
-
declare function extractRetrievedText(spans: readonly Span[], opts?: ExtractRetrievedTextOptions): string;
|
|
103
|
-
/**
|
|
104
|
-
* Extract the provider's retrieved text from a run's spans and score it against
|
|
105
|
-
* `requiredKnowledge` in one call — the analog of authenticity's file-in
|
|
106
|
-
* convenience. The primary contract is the standalone `scoreGroundedness`; this
|
|
107
|
-
* is the ergonomic path for a consumer holding a persisted run's `Span[]`
|
|
108
|
-
* (e.g. from `TraceStore.spans(...)`).
|
|
109
|
-
*/
|
|
110
|
-
declare function scoreGroundednessForRun(spans: readonly Span[], requiredKnowledge: readonly string[], opts?: ExtractRetrievedTextOptions): GroundednessResult;
|
|
111
|
-
|
|
112
|
-
export { type ExtractRetrievedTextOptions, type GroundednessResult, type ProviderToolMatcher, defaultProviderToolMatcher, extractRetrievedText, scoreGroundedness, scoreGroundednessForRun };
|
|
@@ -1,77 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
isRetrievalSpan,
|
|
3
|
-
isToolSpan
|
|
4
|
-
} from "../chunk-5BKGXME7.js";
|
|
5
|
-
import "../chunk-PZ5AY32C.js";
|
|
6
|
-
|
|
7
|
-
// src/groundedness/index.ts
|
|
8
|
-
function dedupeKeys(keys) {
|
|
9
|
-
const seen = /* @__PURE__ */ new Set();
|
|
10
|
-
const out = [];
|
|
11
|
-
for (const raw of keys) {
|
|
12
|
-
const k = raw.trim();
|
|
13
|
-
if (!k) continue;
|
|
14
|
-
const lower = k.toLowerCase();
|
|
15
|
-
if (seen.has(lower)) continue;
|
|
16
|
-
seen.add(lower);
|
|
17
|
-
out.push(k);
|
|
18
|
-
}
|
|
19
|
-
return out;
|
|
20
|
-
}
|
|
21
|
-
function scoreGroundedness(resultText, requiredKnowledge) {
|
|
22
|
-
const keys = dedupeKeys(requiredKnowledge);
|
|
23
|
-
const total = keys.length;
|
|
24
|
-
const text = resultText ?? "";
|
|
25
|
-
const hadResults = text.trim().length > 0;
|
|
26
|
-
const haystack = text.toLowerCase();
|
|
27
|
-
if (total === 0) {
|
|
28
|
-
return { score: 1, found: [], missing: [], total: 0, hadResults };
|
|
29
|
-
}
|
|
30
|
-
const found = [];
|
|
31
|
-
const missing = [];
|
|
32
|
-
for (const key of keys) {
|
|
33
|
-
if (haystack.includes(key.toLowerCase())) found.push(key);
|
|
34
|
-
else missing.push(key);
|
|
35
|
-
}
|
|
36
|
-
return { score: found.length / total, found, missing, total, hadResults };
|
|
37
|
-
}
|
|
38
|
-
var defaultProviderToolMatcher = (name) => /search|research/i.test(name) && !/fetch/i.test(name);
|
|
39
|
-
function resultToText(result) {
|
|
40
|
-
if (result == null) return "";
|
|
41
|
-
if (typeof result === "string") return result;
|
|
42
|
-
try {
|
|
43
|
-
return JSON.stringify(result);
|
|
44
|
-
} catch {
|
|
45
|
-
return String(result);
|
|
46
|
-
}
|
|
47
|
-
}
|
|
48
|
-
function retrievalSpanText(span) {
|
|
49
|
-
return span.hits.map((h) => h.content ?? "").filter((c) => c.length > 0).join("\n");
|
|
50
|
-
}
|
|
51
|
-
function extractRetrievedText(spans, opts = {}) {
|
|
52
|
-
const isProviderTool = opts.isProviderTool ?? defaultProviderToolMatcher;
|
|
53
|
-
const parts = [];
|
|
54
|
-
for (const span of spans) {
|
|
55
|
-
if (isRetrievalSpan(span)) {
|
|
56
|
-
const t = retrievalSpanText(span);
|
|
57
|
-
if (t) parts.push(t);
|
|
58
|
-
} else if (isToolSpan(span)) {
|
|
59
|
-
const ts = span;
|
|
60
|
-
if (isProviderTool(ts.toolName)) {
|
|
61
|
-
const t = resultToText(ts.result);
|
|
62
|
-
if (t) parts.push(t);
|
|
63
|
-
}
|
|
64
|
-
}
|
|
65
|
-
}
|
|
66
|
-
return parts.join("\n");
|
|
67
|
-
}
|
|
68
|
-
function scoreGroundednessForRun(spans, requiredKnowledge, opts = {}) {
|
|
69
|
-
return scoreGroundedness(extractRetrievedText(spans, opts), requiredKnowledge);
|
|
70
|
-
}
|
|
71
|
-
export {
|
|
72
|
-
defaultProviderToolMatcher,
|
|
73
|
-
extractRetrievedText,
|
|
74
|
-
scoreGroundedness,
|
|
75
|
-
scoreGroundednessForRun
|
|
76
|
-
};
|
|
77
|
-
//# sourceMappingURL=index.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["../../src/groundedness/index.ts"],"sourcesContent":["/**\n * Groundedness — \"did the retrieval PROVIDER surface what the task needed?\"\n *\n * A search/research provider returns text; the task needed certain facts or\n * symbols to be solvable (the CURRENT API, a version number, a function name).\n * This module scores how much of that required knowledge the provider's results\n * actually surfaced — isolating PROVIDER quality (was the right thing\n * retrievable / returned) from AGENT skill (did the agent then use it). A high\n * groundedness score with a failed run blames the agent; a low score blames the\n * provider. That separation is the whole point — pass/fail alone cannot make it.\n *\n * Structural sibling of `../authenticity`:\n * - authenticity scores the agent's PRODUCED files for realness.\n * - groundedness scores the provider's RETRIEVED text for coverage.\n * Both are pure deterministic scorers whose DOMAIN config is supplied by the\n * consumer (authenticity: `AuthenticitySignals`; groundedness:\n * `requiredKnowledge: string[]`) — neither bakes in a benchmark's vocabulary.\n *\n * Relationship to `keyword-coverage-judge`: that judge scores the agent's\n * SERVED OUTPUT (HTML + assets) for expected concepts — a different input\n * (produced deliverable) answering a different question (deliverable quality).\n * Groundedness reads the RETRIEVAL side (provider results). They are\n * complementary coverage scorers over different stages of the run, not\n * duplicates; do not collapse one into the other.\n *\n * Two seams, neither forked:\n * - PURE SCORER `scoreGroundedness(resultText, requiredKnowledge)` — case-\n * insensitive substring containment over a deduped key set. Fail-open: with\n * no required knowledge there is nothing to ground, so `score = 1`.\n * - TRACE EXTRACTOR `extractRetrievedText(spans, opts?)` — pulls the provider's\n * returned text out of the canonical `TraceSchema` spans (`RetrievalSpan.hits`\n * + provider `ToolSpan.result`) instead of re-parsing bespoke run files. This\n * is the retrieval-side analog of `extractProducedState` (events → produced\n * files): structural span input, no IO, no disk walking.\n */\n\nimport type { RetrievalSpan, Span, ToolSpan } from '../trace/schema'\nimport { isRetrievalSpan, isToolSpan } from '../trace/schema'\n\n// ── Pure scorer ──────────────────────────────────────────────────────────────\n\nexport interface GroundednessResult {\n /** 0..1 share of required knowledge surfaced by the provider's results.\n * 1 when there is nothing to ground (`requiredKnowledge` empty) — fail-open. */\n score: number\n /** The required-knowledge keys the result text surfaced (deduped, original casing). */\n found: string[]\n /** The required-knowledge keys the result text did NOT surface. */\n missing: string[]\n /** Distinct required-knowledge keys after dedup — the denominator of `score`. */\n total: number\n /** Did the provider return any result text at all? Distinguishes \"provider\n * surfaced nothing\" (`!hadResults`) from \"returned text but missed the facts\"\n * (`hadResults && score < 1`) — the same provider-vs-agent split as the score. */\n hadResults: boolean\n}\n\n/**\n * Dedup a knowledge-key list, case-insensitively, keeping first-seen casing and\n * dropping blanks. The score denominator is distinct keys, so a config that\n * lists the same symbol twice (or with different casing) can't inflate `total`.\n */\nfunction dedupeKeys(keys: readonly string[]): string[] {\n const seen = new Set<string>()\n const out: string[] = []\n for (const raw of keys) {\n const k = raw.trim()\n if (!k) continue\n const lower = k.toLowerCase()\n if (seen.has(lower)) continue\n seen.add(lower)\n out.push(k)\n }\n return out\n}\n\n/**\n * Score how much of `requiredKnowledge` the retrieval provider's `resultText`\n * surfaced. Pure — same inputs, same output. No IO, no LLM.\n *\n * Matching is case-insensitive substring containment: each required key is\n * checked against the lower-cased result text. This is intentionally the same\n * cheap, deterministic containment the authenticity scorer uses for its\n * structural signals — a key is \"surfaced\" if the provider's returned text\n * mentions it. Semantic / paraphrase coverage is a separate (LLM) layer a\n * consumer can stack on top, exactly as authenticity stacks its nuance judge.\n *\n * Fail-open at `total === 0`: a task with no required knowledge has nothing for\n * the provider to ground, so it cannot be penalized (`score = 1`). The benchmark\n * caller decides what `requiredKnowledge` is — the substrate never derives it.\n */\nexport function scoreGroundedness(\n resultText: string,\n requiredKnowledge: readonly string[],\n): GroundednessResult {\n const keys = dedupeKeys(requiredKnowledge)\n const total = keys.length\n const text = resultText ?? ''\n const hadResults = text.trim().length > 0\n const haystack = text.toLowerCase()\n\n if (total === 0) {\n return { score: 1, found: [], missing: [], total: 0, hadResults }\n }\n\n const found: string[] = []\n const missing: string[] = []\n for (const key of keys) {\n if (haystack.includes(key.toLowerCase())) found.push(key)\n else missing.push(key)\n }\n\n return { score: found.length / total, found, missing, total, hadResults }\n}\n\n// ── Trace extractor ────────────────────────────────────────────────────────\n\n/**\n * Predicate selecting which `ToolSpan`s are retrieval-PROVIDER calls (whose\n * `result` carries returned text), by tool name. A parameter — never a baked-in\n * literal — so the substrate stays free of any one benchmark's tool vocabulary,\n * exactly as `AuthenticitySignals` keeps all domain regexes consumer-supplied.\n */\nexport type ProviderToolMatcher = (toolName: string) => boolean\n\n/**\n * Default provider matcher: tool names that look like search/research but not a\n * plain fetch/read. A sensible starting point for the common \"search arm\" shape;\n * any consumer with different tool names passes its own matcher. `RetrievalSpan`s\n * are ALWAYS included regardless of this matcher (they are retrieval by kind);\n * the matcher only selects which generic `ToolSpan`s also count as provider calls.\n */\nexport const defaultProviderToolMatcher: ProviderToolMatcher = (name) =>\n /search|research/i.test(name) && !/fetch/i.test(name)\n\nexport interface ExtractRetrievedTextOptions {\n /** Which `ToolSpan`s count as provider calls. Default: {@link defaultProviderToolMatcher}. */\n isProviderTool?: ProviderToolMatcher\n}\n\n/** Stringify a `ToolSpan.result` of unknown shape into searchable text. */\nfunction resultToText(result: unknown): string {\n if (result == null) return ''\n if (typeof result === 'string') return result\n try {\n return JSON.stringify(result)\n } catch {\n return String(result)\n }\n}\n\n/** Pull the retrieved text out of a `RetrievalSpan`: every hit's `content`. */\nfunction retrievalSpanText(span: RetrievalSpan): string {\n return span.hits\n .map((h) => h.content ?? '')\n .filter((c) => c.length > 0)\n .join('\\n')\n}\n\n/**\n * Extract the retrieval PROVIDER's returned text from a span stream — the\n * retrieval-side analog of `extractProducedState`. Reads the canonical\n * `TraceSchema` carriers, NOT bespoke run files:\n * - every `RetrievalSpan`'s `hits[].content` (kind 'retrieval' — the\n * substrate's first-class search/research result carrier; the same `.hits`\n * the `bad_retrieval` failure detector already reads), and\n * - `ToolSpan.result` for tool spans whose `toolName` the provider matcher\n * accepts (kind 'tool').\n *\n * Pure and total: spans of other kinds, and provider tools with no result, are\n * skipped. Returns one text blob ready for `scoreGroundedness`.\n */\nexport function extractRetrievedText(\n spans: readonly Span[],\n opts: ExtractRetrievedTextOptions = {},\n): string {\n const isProviderTool = opts.isProviderTool ?? defaultProviderToolMatcher\n const parts: string[] = []\n for (const span of spans) {\n if (isRetrievalSpan(span)) {\n const t = retrievalSpanText(span)\n if (t) parts.push(t)\n } else if (isToolSpan(span)) {\n const ts = span as ToolSpan\n if (isProviderTool(ts.toolName)) {\n const t = resultToText(ts.result)\n if (t) parts.push(t)\n }\n }\n }\n return parts.join('\\n')\n}\n\n// ── Convenience: extract-then-score ───────────────────────────────────────────\n\n/**\n * Extract the provider's retrieved text from a run's spans and score it against\n * `requiredKnowledge` in one call — the analog of authenticity's file-in\n * convenience. The primary contract is the standalone `scoreGroundedness`; this\n * is the ergonomic path for a consumer holding a persisted run's `Span[]`\n * (e.g. from `TraceStore.spans(...)`).\n */\nexport function scoreGroundednessForRun(\n spans: readonly Span[],\n requiredKnowledge: readonly string[],\n opts: ExtractRetrievedTextOptions = {},\n): GroundednessResult {\n return scoreGroundedness(extractRetrievedText(spans, opts), requiredKnowledge)\n}\n"],"mappings":";;;;;;;AA8DA,SAAS,WAAW,MAAmC;AACrD,QAAM,OAAO,oBAAI,IAAY;AAC7B,QAAM,MAAgB,CAAC;AACvB,aAAW,OAAO,MAAM;AACtB,UAAM,IAAI,IAAI,KAAK;AACnB,QAAI,CAAC,EAAG;AACR,UAAM,QAAQ,EAAE,YAAY;AAC5B,QAAI,KAAK,IAAI,KAAK,EAAG;AACrB,SAAK,IAAI,KAAK;AACd,QAAI,KAAK,CAAC;AAAA,EACZ;AACA,SAAO;AACT;AAiBO,SAAS,kBACd,YACA,mBACoB;AACpB,QAAM,OAAO,WAAW,iBAAiB;AACzC,QAAM,QAAQ,KAAK;AACnB,QAAM,OAAO,cAAc;AAC3B,QAAM,aAAa,KAAK,KAAK,EAAE,SAAS;AACxC,QAAM,WAAW,KAAK,YAAY;AAElC,MAAI,UAAU,GAAG;AACf,WAAO,EAAE,OAAO,GAAG,OAAO,CAAC,GAAG,SAAS,CAAC,GAAG,OAAO,GAAG,WAAW;AAAA,EAClE;AAEA,QAAM,QAAkB,CAAC;AACzB,QAAM,UAAoB,CAAC;AAC3B,aAAW,OAAO,MAAM;AACtB,QAAI,SAAS,SAAS,IAAI,YAAY,CAAC,EAAG,OAAM,KAAK,GAAG;AAAA,QACnD,SAAQ,KAAK,GAAG;AAAA,EACvB;AAEA,SAAO,EAAE,OAAO,MAAM,SAAS,OAAO,OAAO,SAAS,OAAO,WAAW;AAC1E;AAmBO,IAAM,6BAAkD,CAAC,SAC9D,mBAAmB,KAAK,IAAI,KAAK,CAAC,SAAS,KAAK,IAAI;AAQtD,SAAS,aAAa,QAAyB;AAC7C,MAAI,UAAU,KAAM,QAAO;AAC3B,MAAI,OAAO,WAAW,SAAU,QAAO;AACvC,MAAI;AACF,WAAO,KAAK,UAAU,MAAM;AAAA,EAC9B,QAAQ;AACN,WAAO,OAAO,MAAM;AAAA,EACtB;AACF;AAGA,SAAS,kBAAkB,MAA6B;AACtD,SAAO,KAAK,KACT,IAAI,CAAC,MAAM,EAAE,WAAW,EAAE,EAC1B,OAAO,CAAC,MAAM,EAAE,SAAS,CAAC,EAC1B,KAAK,IAAI;AACd;AAeO,SAAS,qBACd,OACA,OAAoC,CAAC,GAC7B;AACR,QAAM,iBAAiB,KAAK,kBAAkB;AAC9C,QAAM,QAAkB,CAAC;AACzB,aAAW,QAAQ,OAAO;AACxB,QAAI,gBAAgB,IAAI,GAAG;AACzB,YAAM,IAAI,kBAAkB,IAAI;AAChC,UAAI,EAAG,OAAM,KAAK,CAAC;AAAA,IACrB,WAAW,WAAW,IAAI,GAAG;AAC3B,YAAM,KAAK;AACX,UAAI,eAAe,GAAG,QAAQ,GAAG;AAC/B,cAAM,IAAI,aAAa,GAAG,MAAM;AAChC,YAAI,EAAG,OAAM,KAAK,CAAC;AAAA,MACrB;AAAA,IACF;AAAA,EACF;AACA,SAAO,MAAM,KAAK,IAAI;AACxB;AAWO,SAAS,wBACd,OACA,mBACA,OAAoC,CAAC,GACjB;AACpB,SAAO,kBAAkB,qBAAqB,OAAO,IAAI,GAAG,iBAAiB;AAC/E;","names":[]}
|