@tangle-network/agent-eval 0.109.1 → 0.110.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +62 -20
- package/dist/campaign/index.js +9 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-R6D7NEYJ.js → chunk-GBI5J5DB.js} +81 -11
- package/dist/chunk-GBI5J5DB.js.map +1 -0
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-J4AKLZEV.js} +15 -4
- package/dist/{chunk-OVPVM4JC.js.map → chunk-J4AKLZEV.js.map} +1 -1
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +26 -28
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-T8T215nw.d.ts} +149 -6
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-CyxkvEi9.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/docs/improvement-glossary.md +14 -13
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-R6D7NEYJ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/diagnose.d.ts
DELETED
|
@@ -1,252 +0,0 @@
|
|
|
1
|
-
import { C as CounterfactualMutation, a as attributeCounterfactuals, b as CounterfactualRunner } from './counterfactual-DlOz8PBx.js';
|
|
2
|
-
export { c as CounterfactualContext, d as CounterfactualResult } from './counterfactual-DlOz8PBx.js';
|
|
3
|
-
import { S as Span } from './schema-m0gsnbt3.js';
|
|
4
|
-
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
5
|
-
import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
|
|
6
|
-
import { h as AnalystSeverity, A as AnalystFinding } from './types-D1ytG0Yg.js';
|
|
7
|
-
import { C as CorpusRecord } from './corpus-eBVwhCp1.js';
|
|
8
|
-
import { R as RunRecord } from './run-record-I-Z3JNvO.js';
|
|
9
|
-
import './emitter-C2rqGH_l.js';
|
|
10
|
-
import './store-C1YxJDEK.js';
|
|
11
|
-
import './types-C7DGg5ex.js';
|
|
12
|
-
import '@tangle-network/tcloud';
|
|
13
|
-
import './llm-client-DyqEH4jH.js';
|
|
14
|
-
import './errors-oeQrLqXC.js';
|
|
15
|
-
import './raw-provider-sink-C46HDghv.js';
|
|
16
|
-
import '@tangle-network/agent-interface';
|
|
17
|
-
|
|
18
|
-
/**
|
|
19
|
-
* Causal sweep — WHY did this run fail?
|
|
20
|
-
*
|
|
21
|
-
* Orchestrates the dormant counterfactual primitives into a responsibility
|
|
22
|
-
* report: for each candidate step, run `reps` counterfactual replays per
|
|
23
|
-
* mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`
|
|
24
|
-
* is the execution seam) and reduce the per-rep score deltas into a mean
|
|
25
|
-
* effect + bootstrap confidence interval (via `confidenceInterval`).
|
|
26
|
-
*
|
|
27
|
-
* Why `reps` is REQUIRED: a single intervention delta is one stochastic
|
|
28
|
-
* draw — LLM re-execution from a prefix is sampled, so one replay cannot
|
|
29
|
-
* distinguish "this step caused the failure" from sampling noise. The
|
|
30
|
-
* signal is the distribution of deltas across reps; the CI over that
|
|
31
|
-
* distribution is what lets a caller say "this step's effect excludes
|
|
32
|
-
* zero" instead of eyeballing a point estimate.
|
|
33
|
-
*
|
|
34
|
-
* Budget discipline: the sweep never silently drops cells. When the
|
|
35
|
-
* remaining budget cannot fund a full `reps`-sized cell, the sweep halts
|
|
36
|
-
* and every step not fully probed is named in `uncovered`.
|
|
37
|
-
*/
|
|
38
|
-
|
|
39
|
-
/** Stable reference to a trajectory step — carried through reports,
|
|
40
|
-
* findings, and corpus records so evidence stays addressable. */
|
|
41
|
-
interface StepRef {
|
|
42
|
-
index: number;
|
|
43
|
-
spanId: string;
|
|
44
|
-
kind: Span['kind'];
|
|
45
|
-
name: string;
|
|
46
|
-
}
|
|
47
|
-
declare function stepRefOf(step: TrajectoryStep): StepRef;
|
|
48
|
-
interface CausalSweepOptions {
|
|
49
|
-
store: TraceStore;
|
|
50
|
-
/** The failed run to diagnose. Its `outcome.score` is the baseline every
|
|
51
|
-
* counterfactual delta is measured against. */
|
|
52
|
-
runId: string;
|
|
53
|
-
/** Execution seam — identical contract to `runCounterfactual`: re-runs the
|
|
54
|
-
* agent from the mutation point and MUST `endRun` with a numeric score. */
|
|
55
|
-
runner: CounterfactualRunner;
|
|
56
|
-
/** Trajectory indices to probe. Default: every llm + tool span — the kinds
|
|
57
|
-
* the existing `CounterfactualMutation` set targets. */
|
|
58
|
-
candidateSteps?: number[];
|
|
59
|
-
/**
|
|
60
|
-
* Mutations to probe a given step with. Returned mutations MUST target
|
|
61
|
-
* `step.index`. Default probes are the payload-free existing kinds:
|
|
62
|
-
* - tool span → `swap-tool-result` with `newResult: null` (knockout:
|
|
63
|
-
* how much did the run depend on this tool's information?)
|
|
64
|
-
* - llm span → `truncate-after` (re-roll: how much did the realized
|
|
65
|
-
* turn deviate from the policy's typical continuation?)
|
|
66
|
-
* `swap-model` / `inject-system-message` need consumer payloads, so they
|
|
67
|
-
* are opt-in via this callback.
|
|
68
|
-
*/
|
|
69
|
-
mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[];
|
|
70
|
-
/** Replays per (step, mutation) cell. Minimum 2 — see module doc. */
|
|
71
|
-
reps: number;
|
|
72
|
-
/** Hard cap on total counterfactual replays across the whole sweep. */
|
|
73
|
-
budget: number;
|
|
74
|
-
/** Seed for the bootstrap CI resampler. Deterministic default so two
|
|
75
|
-
* sweeps over the same deltas report identical intervals. */
|
|
76
|
-
ciSeed?: number;
|
|
77
|
-
/** Bootstrap CI confidence level. Default 0.95. */
|
|
78
|
-
ciConfidence?: number;
|
|
79
|
-
}
|
|
80
|
-
interface StepResponsibility {
|
|
81
|
-
stepRef: StepRef;
|
|
82
|
-
mutationKind: CounterfactualMutation['kind'];
|
|
83
|
-
/** Mean of per-rep score deltas (counterfactual − original). */
|
|
84
|
-
meanEffect: number;
|
|
85
|
-
/** Bootstrap CI over the per-rep deltas. */
|
|
86
|
-
ci: {
|
|
87
|
-
mean: number;
|
|
88
|
-
lower: number;
|
|
89
|
-
upper: number;
|
|
90
|
-
};
|
|
91
|
-
/** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */
|
|
92
|
-
ciExcludesZero: boolean;
|
|
93
|
-
reps: number;
|
|
94
|
-
/** Raw per-rep deltas — downstream evidence, never re-derived. */
|
|
95
|
-
deltas: number[];
|
|
96
|
-
/** Replay run ids (layer='meta', parentRunId=original) for audit. */
|
|
97
|
-
counterfactualRunIds: string[];
|
|
98
|
-
}
|
|
99
|
-
interface CausalResponsibilityReport {
|
|
100
|
-
runId: string;
|
|
101
|
-
originalScore: number;
|
|
102
|
-
/** Ranked by |meanEffect| descending — the blame ordering. */
|
|
103
|
-
steps: StepResponsibility[];
|
|
104
|
-
/** Kind-level aggregate from the existing `attributeCounterfactuals`. */
|
|
105
|
-
byMutationKind: ReturnType<typeof attributeCounterfactuals>;
|
|
106
|
-
replaysUsed: number;
|
|
107
|
-
budget: number;
|
|
108
|
-
/** Steps planned but not fully probed before the budget ran out.
|
|
109
|
-
* Named, never silent: an absent step is "no effect found"; an
|
|
110
|
-
* uncovered step is "not measured". */
|
|
111
|
-
uncovered: StepRef[];
|
|
112
|
-
}
|
|
113
|
-
declare function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport>;
|
|
114
|
-
|
|
115
|
-
/**
|
|
116
|
-
* Replay-validated repair — WHAT SHOULD HAVE HAPPENED?
|
|
117
|
-
*
|
|
118
|
-
* Takes the blamed steps from a `CausalResponsibilityReport`, asks a
|
|
119
|
-
* consumer-supplied `proposeFix` (LLM-backed in live use) for candidate
|
|
120
|
-
* mutations, and machine-verifies each candidate by replaying the run
|
|
121
|
-
* WITH the mutation applied (through the same `runCounterfactual` seam
|
|
122
|
-
* the sweep uses).
|
|
123
|
-
*
|
|
124
|
-
* A repair is "what should have happened" ONLY when every validation
|
|
125
|
-
* replay crosses `flipThreshold` — a prescription is never speculated,
|
|
126
|
-
* it is demonstrated. Candidates that don't flip, or whose replay
|
|
127
|
-
* errors, land in `rejected` with a typed reason; nothing is dropped
|
|
128
|
-
* silently.
|
|
129
|
-
*/
|
|
130
|
-
|
|
131
|
-
/** Context handed to `proposeFix` so an LLM-backed proposer can see the
|
|
132
|
-
* full trajectory plus the responsibility evidence for the blamed step. */
|
|
133
|
-
interface RepairContext {
|
|
134
|
-
runId: string;
|
|
135
|
-
trajectory: Trajectory;
|
|
136
|
-
originalScore: number;
|
|
137
|
-
responsibility: StepResponsibility;
|
|
138
|
-
}
|
|
139
|
-
interface PrescribeRepairOptions {
|
|
140
|
-
store: TraceStore;
|
|
141
|
-
/** The failed run the sweep diagnosed. */
|
|
142
|
-
runId: string;
|
|
143
|
-
/** Execution seam — same `CounterfactualRunner` contract as the sweep. */
|
|
144
|
-
runner: CounterfactualRunner;
|
|
145
|
-
/** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */
|
|
146
|
-
blamed: StepResponsibility[];
|
|
147
|
-
/** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.
|
|
148
|
-
* Returned mutations MUST target the blamed step's index. */
|
|
149
|
-
proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>;
|
|
150
|
-
/** Score every validation replay must reach for the repair to count. Default 0.5. */
|
|
151
|
-
flipThreshold?: number;
|
|
152
|
-
/** Validation replays per candidate mutation. Default 3. */
|
|
153
|
-
repsToValidate?: number;
|
|
154
|
-
/** Max candidate mutations tried per step. Default: all proposed. */
|
|
155
|
-
maxAttemptsPerStep?: number;
|
|
156
|
-
}
|
|
157
|
-
interface ValidatedRepair {
|
|
158
|
-
stepRef: StepRef;
|
|
159
|
-
mutation: CounterfactualMutation;
|
|
160
|
-
/** Always true — presence in `repairs` IS the machine-verified claim. */
|
|
161
|
-
validated: true;
|
|
162
|
-
/** Mean counterfactual score across the validation reps. */
|
|
163
|
-
meanScore: number;
|
|
164
|
-
/** meanScore − originalScore. */
|
|
165
|
-
deltaScore: number;
|
|
166
|
-
reps: number;
|
|
167
|
-
/** Replay run ids backing the validation — audit trail. */
|
|
168
|
-
counterfactualRunIds: string[];
|
|
169
|
-
}
|
|
170
|
-
interface RejectedRepair {
|
|
171
|
-
stepRef: StepRef;
|
|
172
|
-
mutation: CounterfactualMutation;
|
|
173
|
-
reason: 'did-not-flip' | 'error';
|
|
174
|
-
/** Present for 'did-not-flip': mean delta over the reps that ran. */
|
|
175
|
-
deltaScore?: number;
|
|
176
|
-
/** Present for 'error': the message, preserved for diagnosis. */
|
|
177
|
-
error?: string;
|
|
178
|
-
}
|
|
179
|
-
interface RepairReport {
|
|
180
|
-
runId: string;
|
|
181
|
-
originalScore: number;
|
|
182
|
-
flipThreshold: number;
|
|
183
|
-
repairs: ValidatedRepair[];
|
|
184
|
-
rejected: RejectedRepair[];
|
|
185
|
-
replaysUsed: number;
|
|
186
|
-
}
|
|
187
|
-
declare function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport>;
|
|
188
|
-
|
|
189
|
-
/**
|
|
190
|
-
* Remediation adapters — HOW DO WE MAKE IT HAPPEN?
|
|
191
|
-
*
|
|
192
|
-
* The diagnose chain ends by feeding existing improvement machinery,
|
|
193
|
-
* not by building new machinery:
|
|
194
|
-
*
|
|
195
|
-
* - `toAnalystFindings` → the analyst contract (`makeFinding`), so
|
|
196
|
-
* responsibility evidence flows into the same registry / steering /
|
|
197
|
-
* diff pipeline every other analyst feeds.
|
|
198
|
-
* - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the
|
|
199
|
-
* diagnosed failure + validated repair as a permanent scenario.
|
|
200
|
-
* - `suggestInvariant` → a plain-data hint in the shape the
|
|
201
|
-
* trace-contracts machinery consumes (`never` / `without` clauses).
|
|
202
|
-
*/
|
|
203
|
-
|
|
204
|
-
declare const DIAGNOSE_ANALYST_ID = "diagnose-causal-sweep";
|
|
205
|
-
/** Severity from causal effect size. Effects whose CI includes zero are
|
|
206
|
-
* 'info' regardless of magnitude — an indistinguishable-from-noise effect
|
|
207
|
-
* must not steer remediation priority. */
|
|
208
|
-
declare function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity;
|
|
209
|
-
/** Deterministic human-readable rendering of a mutation — used in
|
|
210
|
-
* recommended actions, corpus completions, and invariant hints. */
|
|
211
|
-
declare function describeMutation(mutation: CounterfactualMutation): string;
|
|
212
|
-
/**
|
|
213
|
-
* Lift a responsibility report (and optionally its validated repairs) into
|
|
214
|
-
* `AnalystFinding`s via the real `makeFinding` factory. One finding per
|
|
215
|
-
* probed step; a validated repair for that step upgrades the finding with
|
|
216
|
-
* a `recommended_action` + the replay-validation evidence.
|
|
217
|
-
*
|
|
218
|
-
* Findings are OBSERVED causal probes (replay deltas), not judge verdicts,
|
|
219
|
-
* so `derived_from_judge` stays unset and they may steer.
|
|
220
|
-
*/
|
|
221
|
-
declare function toAnalystFindings(report: CausalResponsibilityReport, repairs?: RepairReport): AnalystFinding[];
|
|
222
|
-
/**
|
|
223
|
-
* Pin the diagnosed failure as a permanent corpus scenario. Takes the
|
|
224
|
-
* original run's `RunRecord` projection plus a validated repair and emits
|
|
225
|
-
* a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw
|
|
226
|
-
* failure and the diagnosed entry).
|
|
227
|
-
*
|
|
228
|
-
* `completion` defaults to the validated mutation's rendering — "what
|
|
229
|
-
* should have happened" in machine-derived form. Supply `prompt` (and
|
|
230
|
-
* optionally a richer `completion`) when the trajectory text is available
|
|
231
|
-
* so the record is harvestable by `buildDatasetFromCorpus`.
|
|
232
|
-
*/
|
|
233
|
-
declare function toCorpusRecord(run: RunRecord, repair: ValidatedRepair, opts?: {
|
|
234
|
-
prompt?: string;
|
|
235
|
-
completion?: string;
|
|
236
|
-
}): CorpusRecord;
|
|
237
|
-
/** Plain-data invariant hint. The trace-contracts machinery consumes this
|
|
238
|
-
* shape: `never` is a pattern that must not appear in a passing trace;
|
|
239
|
-
* `without` is a guard whose absence makes the failure reachable. */
|
|
240
|
-
interface InvariantHint {
|
|
241
|
-
description: string;
|
|
242
|
-
never?: string;
|
|
243
|
-
without?: string;
|
|
244
|
-
}
|
|
245
|
-
/**
|
|
246
|
-
* Derive an invariant hint from a validated repair. Deterministic per
|
|
247
|
-
* mutation kind — the hint names the contract a trace must satisfy so
|
|
248
|
-
* the diagnosed failure cannot silently recur.
|
|
249
|
-
*/
|
|
250
|
-
declare function suggestInvariant(repair: ValidatedRepair): InvariantHint;
|
|
251
|
-
|
|
252
|
-
export { type CausalResponsibilityReport, type CausalSweepOptions, CounterfactualMutation, CounterfactualRunner, DIAGNOSE_ANALYST_ID, type InvariantHint, type PrescribeRepairOptions, type RejectedRepair, type RepairContext, type RepairReport, type StepRef, type StepResponsibility, type ValidatedRepair, causalSweep, describeMutation, prescribeRepair, severityFromEffect, stepRefOf, suggestInvariant, toAnalystFindings, toCorpusRecord };
|
package/dist/diagnose.js
DELETED
|
@@ -1,382 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
attributeCounterfactuals,
|
|
3
|
-
runCounterfactual
|
|
4
|
-
} from "./chunk-6SK5VFYK.js";
|
|
5
|
-
import {
|
|
6
|
-
buildTrajectory
|
|
7
|
-
} from "./chunk-RZTMDUO7.js";
|
|
8
|
-
import {
|
|
9
|
-
makeFinding
|
|
10
|
-
} from "./chunk-45EEMHTC.js";
|
|
11
|
-
import {
|
|
12
|
-
confidenceInterval
|
|
13
|
-
} from "./chunk-XMSYF4A7.js";
|
|
14
|
-
import {
|
|
15
|
-
validateRunRecord
|
|
16
|
-
} from "./chunk-VK6HBGAE.js";
|
|
17
|
-
import "./chunk-TVVP3ZZQ.js";
|
|
18
|
-
import "./chunk-XJYR7XFV.js";
|
|
19
|
-
import "./chunk-VSMTAMNK.js";
|
|
20
|
-
import {
|
|
21
|
-
ValidationError
|
|
22
|
-
} from "./chunk-ONWEPEDO.js";
|
|
23
|
-
import "./chunk-PZ5AY32C.js";
|
|
24
|
-
|
|
25
|
-
// src/diagnose/causal-sweep.ts
|
|
26
|
-
function stepRefOf(step) {
|
|
27
|
-
return {
|
|
28
|
-
index: step.index,
|
|
29
|
-
spanId: step.span.spanId,
|
|
30
|
-
kind: step.span.kind,
|
|
31
|
-
name: step.span.name
|
|
32
|
-
};
|
|
33
|
-
}
|
|
34
|
-
var DEFAULT_CI_SEED = 24301;
|
|
35
|
-
function defaultMutations(step) {
|
|
36
|
-
if (step.span.kind === "tool") {
|
|
37
|
-
return [{ kind: "swap-tool-result", at: step.index, newResult: null }];
|
|
38
|
-
}
|
|
39
|
-
if (step.span.kind === "llm") {
|
|
40
|
-
return [{ kind: "truncate-after", at: step.index }];
|
|
41
|
-
}
|
|
42
|
-
return [];
|
|
43
|
-
}
|
|
44
|
-
async function causalSweep(opts) {
|
|
45
|
-
if (!Number.isInteger(opts.reps) || opts.reps < 2) {
|
|
46
|
-
throw new ValidationError(
|
|
47
|
-
`causalSweep: reps must be an integer >= 2 (got ${opts.reps}) \u2014 a single-intervention delta is one stochastic draw, not a measurement`
|
|
48
|
-
);
|
|
49
|
-
}
|
|
50
|
-
if (!Number.isInteger(opts.budget) || opts.budget < 1) {
|
|
51
|
-
throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`);
|
|
52
|
-
}
|
|
53
|
-
const originalRun = await opts.store.getRun(opts.runId);
|
|
54
|
-
if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`);
|
|
55
|
-
const originalScore = originalRun.outcome?.score;
|
|
56
|
-
if (typeof originalScore !== "number" || !Number.isFinite(originalScore)) {
|
|
57
|
-
throw new ValidationError(
|
|
58
|
-
`causalSweep: run ${opts.runId} has no numeric outcome.score \u2014 deltas have no baseline`
|
|
59
|
-
);
|
|
60
|
-
}
|
|
61
|
-
const trajectory = await buildTrajectory(opts.store, opts.runId);
|
|
62
|
-
const candidates = resolveCandidates(trajectory, opts.candidateSteps);
|
|
63
|
-
const mutationsFor = opts.mutationsPerStep ?? defaultMutations;
|
|
64
|
-
const cells = [];
|
|
65
|
-
for (const step of candidates) {
|
|
66
|
-
const mutations = mutationsFor(step);
|
|
67
|
-
for (const m of mutations) {
|
|
68
|
-
if (m.at !== step.index) {
|
|
69
|
-
throw new ValidationError(
|
|
70
|
-
`causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} \u2014 mutations must target the step they were asked for`
|
|
71
|
-
);
|
|
72
|
-
}
|
|
73
|
-
cells.push({ step, mutation: m });
|
|
74
|
-
}
|
|
75
|
-
}
|
|
76
|
-
const responsibilities = [];
|
|
77
|
-
const allResults = [];
|
|
78
|
-
const uncoveredIndices = /* @__PURE__ */ new Set();
|
|
79
|
-
let replaysUsed = 0;
|
|
80
|
-
let halted = false;
|
|
81
|
-
for (const cell of cells) {
|
|
82
|
-
if (halted || replaysUsed + opts.reps > opts.budget) {
|
|
83
|
-
halted = true;
|
|
84
|
-
uncoveredIndices.add(cell.step.index);
|
|
85
|
-
continue;
|
|
86
|
-
}
|
|
87
|
-
const deltas = [];
|
|
88
|
-
const cfRunIds = [];
|
|
89
|
-
for (let rep = 0; rep < opts.reps; rep++) {
|
|
90
|
-
const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner);
|
|
91
|
-
replaysUsed++;
|
|
92
|
-
const d = result.delta.deltaScore;
|
|
93
|
-
if (typeof d !== "number" || !Number.isFinite(d)) {
|
|
94
|
-
throw new ValidationError(
|
|
95
|
-
`causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score \u2014 the runner must endRun with a numeric outcome.score`
|
|
96
|
-
);
|
|
97
|
-
}
|
|
98
|
-
deltas.push(d);
|
|
99
|
-
cfRunIds.push(result.counterfactualRunId);
|
|
100
|
-
allResults.push(result);
|
|
101
|
-
}
|
|
102
|
-
const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {
|
|
103
|
-
seed: opts.ciSeed ?? DEFAULT_CI_SEED
|
|
104
|
-
});
|
|
105
|
-
responsibilities.push({
|
|
106
|
-
stepRef: stepRefOf(cell.step),
|
|
107
|
-
mutationKind: cell.mutation.kind,
|
|
108
|
-
meanEffect: ci.mean,
|
|
109
|
-
ci,
|
|
110
|
-
ciExcludesZero: ci.lower > 0 || ci.upper < 0,
|
|
111
|
-
reps: opts.reps,
|
|
112
|
-
deltas,
|
|
113
|
-
counterfactualRunIds: cfRunIds
|
|
114
|
-
});
|
|
115
|
-
}
|
|
116
|
-
responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect));
|
|
117
|
-
const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf);
|
|
118
|
-
return {
|
|
119
|
-
runId: opts.runId,
|
|
120
|
-
originalScore,
|
|
121
|
-
steps: responsibilities,
|
|
122
|
-
byMutationKind: attributeCounterfactuals(allResults),
|
|
123
|
-
replaysUsed,
|
|
124
|
-
budget: opts.budget,
|
|
125
|
-
uncovered
|
|
126
|
-
};
|
|
127
|
-
}
|
|
128
|
-
function resolveCandidates(trajectory, indices) {
|
|
129
|
-
if (indices === void 0) {
|
|
130
|
-
return trajectory.steps.filter((s) => s.span.kind === "llm" || s.span.kind === "tool");
|
|
131
|
-
}
|
|
132
|
-
return indices.map((i) => {
|
|
133
|
-
const step = trajectory.steps[i];
|
|
134
|
-
if (!step) {
|
|
135
|
-
throw new ValidationError(
|
|
136
|
-
`causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`
|
|
137
|
-
);
|
|
138
|
-
}
|
|
139
|
-
return step;
|
|
140
|
-
});
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
// src/diagnose/remediation.ts
|
|
144
|
-
var DIAGNOSE_ANALYST_ID = "diagnose-causal-sweep";
|
|
145
|
-
function severityFromEffect(responsibility) {
|
|
146
|
-
if (!responsibility.ciExcludesZero) return "info";
|
|
147
|
-
const magnitude = Math.abs(responsibility.meanEffect);
|
|
148
|
-
if (magnitude >= 0.5) return "critical";
|
|
149
|
-
if (magnitude >= 0.25) return "high";
|
|
150
|
-
if (magnitude >= 0.1) return "medium";
|
|
151
|
-
return "low";
|
|
152
|
-
}
|
|
153
|
-
function describeMutation(mutation) {
|
|
154
|
-
switch (mutation.kind) {
|
|
155
|
-
case "swap-model":
|
|
156
|
-
return `use model '${mutation.newModel}' at step ${mutation.at}`;
|
|
157
|
-
case "swap-tool-result":
|
|
158
|
-
return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`;
|
|
159
|
-
case "truncate-after":
|
|
160
|
-
return `stop the run after step ${mutation.at}`;
|
|
161
|
-
case "inject-system-message":
|
|
162
|
-
return `inject system message at step ${mutation.at}: ${mutation.content}`;
|
|
163
|
-
case "custom":
|
|
164
|
-
return `${mutation.describe} (step ${mutation.at})`;
|
|
165
|
-
}
|
|
166
|
-
}
|
|
167
|
-
function toAnalystFindings(report, repairs) {
|
|
168
|
-
const repairByStep = /* @__PURE__ */ new Map();
|
|
169
|
-
for (const r of repairs?.repairs ?? []) {
|
|
170
|
-
if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r);
|
|
171
|
-
}
|
|
172
|
-
return report.steps.map((resp) => {
|
|
173
|
-
const repair = repairByStep.get(resp.stepRef.spanId);
|
|
174
|
-
const evidence = [
|
|
175
|
-
{
|
|
176
|
-
kind: "span",
|
|
177
|
-
uri: `span://${resp.stepRef.spanId}`,
|
|
178
|
-
excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`
|
|
179
|
-
},
|
|
180
|
-
{
|
|
181
|
-
kind: "metric",
|
|
182
|
-
uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,
|
|
183
|
-
excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(", ")}]`
|
|
184
|
-
},
|
|
185
|
-
...resp.counterfactualRunIds.map((id) => ({ kind: "span", uri: `run://${id}` }))
|
|
186
|
-
];
|
|
187
|
-
return makeFinding({
|
|
188
|
-
analyst_id: DIAGNOSE_ANALYST_ID,
|
|
189
|
-
severity: severityFromEffect(resp),
|
|
190
|
-
area: "causal-attribution",
|
|
191
|
-
claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,
|
|
192
|
-
rationale: resp.ciExcludesZero ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero` : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero \u2014 not distinguishable from noise`,
|
|
193
|
-
evidence_refs: evidence,
|
|
194
|
-
recommended_action: repair ? describeMutation(repair.mutation) : void 0,
|
|
195
|
-
validation_plan: repair ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})` : void 0,
|
|
196
|
-
confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,
|
|
197
|
-
subject: resp.stepRef.spanId,
|
|
198
|
-
metadata: {
|
|
199
|
-
stepRef: resp.stepRef,
|
|
200
|
-
mutationKind: resp.mutationKind,
|
|
201
|
-
meanEffect: resp.meanEffect,
|
|
202
|
-
ci: resp.ci,
|
|
203
|
-
deltas: resp.deltas,
|
|
204
|
-
counterfactualRunIds: resp.counterfactualRunIds,
|
|
205
|
-
...repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}
|
|
206
|
-
}
|
|
207
|
-
});
|
|
208
|
-
});
|
|
209
|
-
}
|
|
210
|
-
function toCorpusRecord(run, repair, opts = {}) {
|
|
211
|
-
const record = {
|
|
212
|
-
...run,
|
|
213
|
-
runId: `${run.runId}#repair:${repair.stepRef.spanId}`,
|
|
214
|
-
outcome: {
|
|
215
|
-
...run.outcome,
|
|
216
|
-
raw: {
|
|
217
|
-
...run.outcome.raw,
|
|
218
|
-
diagnose_blamed_step_index: repair.stepRef.index,
|
|
219
|
-
diagnose_repair_mean_score: repair.meanScore,
|
|
220
|
-
diagnose_repair_delta_score: repair.deltaScore,
|
|
221
|
-
diagnose_repair_reps: repair.reps
|
|
222
|
-
}
|
|
223
|
-
},
|
|
224
|
-
prompt: opts.prompt,
|
|
225
|
-
completion: opts.completion ?? describeMutation(repair.mutation)
|
|
226
|
-
};
|
|
227
|
-
validateRunRecord(record);
|
|
228
|
-
return record;
|
|
229
|
-
}
|
|
230
|
-
function suggestInvariant(repair) {
|
|
231
|
-
const { stepRef, mutation } = repair;
|
|
232
|
-
const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`;
|
|
233
|
-
switch (mutation.kind) {
|
|
234
|
-
case "swap-tool-result":
|
|
235
|
-
return {
|
|
236
|
-
description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,
|
|
237
|
-
never: `unvalidated result from tool '${stepRef.name}' flows downstream`,
|
|
238
|
-
without: `result guard on tool '${stepRef.name}'`
|
|
239
|
-
};
|
|
240
|
-
case "swap-model":
|
|
241
|
-
return {
|
|
242
|
-
description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,
|
|
243
|
-
never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`
|
|
244
|
-
};
|
|
245
|
-
case "inject-system-message":
|
|
246
|
-
return {
|
|
247
|
-
description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,
|
|
248
|
-
without: `system message present at '${stepRef.name}': ${mutation.content}`
|
|
249
|
-
};
|
|
250
|
-
case "truncate-after":
|
|
251
|
-
return {
|
|
252
|
-
description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) \u2014 continuation past this step caused the failure`,
|
|
253
|
-
never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`
|
|
254
|
-
};
|
|
255
|
-
case "custom":
|
|
256
|
-
return {
|
|
257
|
-
description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`
|
|
258
|
-
};
|
|
259
|
-
default: {
|
|
260
|
-
const exhausted = mutation;
|
|
261
|
-
throw new ValidationError(
|
|
262
|
-
`suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`
|
|
263
|
-
);
|
|
264
|
-
}
|
|
265
|
-
}
|
|
266
|
-
}
|
|
267
|
-
|
|
268
|
-
// src/diagnose/repair.ts
|
|
269
|
-
async function prescribeRepair(opts) {
|
|
270
|
-
const flipThreshold = opts.flipThreshold ?? 0.5;
|
|
271
|
-
const repsToValidate = opts.repsToValidate ?? 3;
|
|
272
|
-
if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {
|
|
273
|
-
throw new ValidationError(
|
|
274
|
-
`prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`
|
|
275
|
-
);
|
|
276
|
-
}
|
|
277
|
-
const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY;
|
|
278
|
-
if (maxAttempts < 1) {
|
|
279
|
-
throw new ValidationError(
|
|
280
|
-
`prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`
|
|
281
|
-
);
|
|
282
|
-
}
|
|
283
|
-
if (opts.blamed.length === 0) {
|
|
284
|
-
throw new ValidationError("prescribeRepair: blamed is empty \u2014 nothing to repair");
|
|
285
|
-
}
|
|
286
|
-
const originalRun = await opts.store.getRun(opts.runId);
|
|
287
|
-
if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`);
|
|
288
|
-
const originalScore = originalRun.outcome?.score;
|
|
289
|
-
if (typeof originalScore !== "number" || !Number.isFinite(originalScore)) {
|
|
290
|
-
throw new ValidationError(
|
|
291
|
-
`prescribeRepair: run ${opts.runId} has no numeric outcome.score \u2014 flips have no baseline`
|
|
292
|
-
);
|
|
293
|
-
}
|
|
294
|
-
const trajectory = await buildTrajectory(opts.store, opts.runId);
|
|
295
|
-
const repairs = [];
|
|
296
|
-
const rejected = [];
|
|
297
|
-
let replaysUsed = 0;
|
|
298
|
-
for (const responsibility of opts.blamed) {
|
|
299
|
-
const step = trajectory.steps[responsibility.stepRef.index];
|
|
300
|
-
if (!step || step.span.spanId !== responsibility.stepRef.spanId) {
|
|
301
|
-
throw new ValidationError(
|
|
302
|
-
`prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} \u2014 stale report?`
|
|
303
|
-
);
|
|
304
|
-
}
|
|
305
|
-
const candidates = await opts.proposeFix(step, {
|
|
306
|
-
runId: opts.runId,
|
|
307
|
-
trajectory,
|
|
308
|
-
originalScore,
|
|
309
|
-
responsibility
|
|
310
|
-
});
|
|
311
|
-
const toTry = candidates.slice(0, maxAttempts);
|
|
312
|
-
for (const mutation of toTry) {
|
|
313
|
-
if (mutation.at !== step.index) {
|
|
314
|
-
throw new ValidationError(
|
|
315
|
-
`prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`
|
|
316
|
-
);
|
|
317
|
-
}
|
|
318
|
-
const scores = [];
|
|
319
|
-
const cfRunIds = [];
|
|
320
|
-
let failure;
|
|
321
|
-
for (let rep = 0; rep < repsToValidate; rep++) {
|
|
322
|
-
try {
|
|
323
|
-
const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner);
|
|
324
|
-
replaysUsed++;
|
|
325
|
-
const score = result.delta.counterfactualOutcomeScore;
|
|
326
|
-
if (typeof score !== "number" || !Number.isFinite(score)) {
|
|
327
|
-
failure = `validation rep ${rep} produced no numeric score \u2014 the runner must endRun with a numeric outcome.score`;
|
|
328
|
-
break;
|
|
329
|
-
}
|
|
330
|
-
scores.push(score);
|
|
331
|
-
cfRunIds.push(result.counterfactualRunId);
|
|
332
|
-
} catch (err) {
|
|
333
|
-
replaysUsed++;
|
|
334
|
-
failure = err instanceof Error ? err.message : String(err);
|
|
335
|
-
break;
|
|
336
|
-
}
|
|
337
|
-
}
|
|
338
|
-
if (failure !== void 0) {
|
|
339
|
-
rejected.push({
|
|
340
|
-
stepRef: responsibility.stepRef,
|
|
341
|
-
mutation,
|
|
342
|
-
reason: "error",
|
|
343
|
-
error: failure
|
|
344
|
-
});
|
|
345
|
-
continue;
|
|
346
|
-
}
|
|
347
|
-
const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length;
|
|
348
|
-
const everyRepFlipped = scores.every((s) => s >= flipThreshold);
|
|
349
|
-
if (everyRepFlipped) {
|
|
350
|
-
repairs.push({
|
|
351
|
-
stepRef: responsibility.stepRef,
|
|
352
|
-
mutation,
|
|
353
|
-
validated: true,
|
|
354
|
-
meanScore,
|
|
355
|
-
deltaScore: meanScore - originalScore,
|
|
356
|
-
reps: repsToValidate,
|
|
357
|
-
counterfactualRunIds: cfRunIds
|
|
358
|
-
});
|
|
359
|
-
break;
|
|
360
|
-
}
|
|
361
|
-
rejected.push({
|
|
362
|
-
stepRef: responsibility.stepRef,
|
|
363
|
-
mutation,
|
|
364
|
-
reason: "did-not-flip",
|
|
365
|
-
deltaScore: meanScore - originalScore
|
|
366
|
-
});
|
|
367
|
-
}
|
|
368
|
-
}
|
|
369
|
-
return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed };
|
|
370
|
-
}
|
|
371
|
-
export {
|
|
372
|
-
DIAGNOSE_ANALYST_ID,
|
|
373
|
-
causalSweep,
|
|
374
|
-
describeMutation,
|
|
375
|
-
prescribeRepair,
|
|
376
|
-
severityFromEffect,
|
|
377
|
-
stepRefOf,
|
|
378
|
-
suggestInvariant,
|
|
379
|
-
toAnalystFindings,
|
|
380
|
-
toCorpusRecord
|
|
381
|
-
};
|
|
382
|
-
//# sourceMappingURL=diagnose.js.map
|