@tangle-network/agent-eval 0.109.1 → 0.110.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/CHANGELOG.md +9 -0
  2. package/dist/analyst/index.d.ts +10 -12
  3. package/dist/analyst/index.js +8 -11
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
  6. package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/benchmarks/index.d.ts +4 -4
  9. package/dist/benchmarks/index.js +7 -8
  10. package/dist/builder-eval/index.d.ts +4 -4
  11. package/dist/builder-eval/index.js +1 -2
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +62 -20
  15. package/dist/campaign/index.js +9 -8
  16. package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
  17. package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
  18. package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
  19. package/dist/chunk-7NX6ZSBG.js.map +1 -0
  20. package/dist/{chunk-R6D7NEYJ.js → chunk-GBI5J5DB.js} +81 -11
  21. package/dist/chunk-GBI5J5DB.js.map +1 -0
  22. package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
  23. package/dist/chunk-IMWDSFUM.js.map +1 -0
  24. package/dist/{chunk-OVPVM4JC.js → chunk-J4AKLZEV.js} +15 -4
  25. package/dist/{chunk-OVPVM4JC.js.map → chunk-J4AKLZEV.js.map} +1 -1
  26. package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
  27. package/dist/chunk-MHNQWM4I.js.map +1 -0
  28. package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
  29. package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
  30. package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
  31. package/dist/chunk-PLOMR3HP.js.map +1 -0
  32. package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
  33. package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
  34. package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
  35. package/dist/chunk-XRGOKCMO.js.map +1 -0
  36. package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
  37. package/dist/contract/index.d.ts +26 -28
  38. package/dist/contract/index.js +11 -13
  39. package/dist/contract/index.js.map +1 -1
  40. package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
  41. package/dist/control.d.ts +8 -9
  42. package/dist/control.js +6 -8
  43. package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
  44. package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
  45. package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
  46. package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
  47. package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
  48. package/dist/{gepa-B3x5Ulcv.d.ts → gepa-T8T215nw.d.ts} +149 -6
  49. package/dist/hosted/index.d.ts +7 -7
  50. package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
  51. package/dist/index.d.ts +645 -61
  52. package/dist/index.js +1282 -190
  53. package/dist/index.js.map +1 -1
  54. package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
  55. package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
  56. package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
  57. package/dist/meta-eval/index.d.ts +5 -5
  58. package/dist/meta-eval/index.js +1 -2
  59. package/dist/meta-eval/index.js.map +1 -1
  60. package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
  61. package/dist/multishot/index.d.ts +3 -3
  62. package/dist/openapi.json +1 -1
  63. package/dist/pipelines/index.d.ts +6 -7
  64. package/dist/pipelines/index.js +3 -6
  65. package/dist/pipelines/index.js.map +1 -1
  66. package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
  67. package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
  68. package/dist/{provenance-DdDhf6cg.d.ts → provenance-CyxkvEi9.d.ts} +3 -5
  69. package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
  70. package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
  71. package/dist/reporting.d.ts +8 -8
  72. package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
  73. package/dist/rl.d.ts +568 -15
  74. package/dist/rl.js +4 -4
  75. package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
  76. package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
  77. package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
  78. package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
  80. package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
  81. package/dist/storyboard/index.d.ts +1 -1
  82. package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
  83. package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
  84. package/dist/traces.d.ts +54 -11
  85. package/dist/traces.js +25 -27
  86. package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
  87. package/dist/wire/index.d.ts +5 -6
  88. package/docs/improvement-glossary.md +14 -13
  89. package/package.json +1 -71
  90. package/dist/adapters/http.d.ts +0 -142
  91. package/dist/adapters/http.js +0 -203
  92. package/dist/adapters/http.js.map +0 -1
  93. package/dist/adapters/langchain.d.ts +0 -95
  94. package/dist/adapters/langchain.js +0 -34
  95. package/dist/adapters/langchain.js.map +0 -1
  96. package/dist/adapters/otel.d.ts +0 -112
  97. package/dist/adapters/otel.js +0 -110
  98. package/dist/adapters/otel.js.map +0 -1
  99. package/dist/chunk-2OGPXHOB.js.map +0 -1
  100. package/dist/chunk-45EEMHTC.js +0 -35
  101. package/dist/chunk-45EEMHTC.js.map +0 -1
  102. package/dist/chunk-5BKGXME7.js +0 -65
  103. package/dist/chunk-5BKGXME7.js.map +0 -1
  104. package/dist/chunk-5PK3626Q.js.map +0 -1
  105. package/dist/chunk-6SK5VFYK.js +0 -100
  106. package/dist/chunk-6SK5VFYK.js.map +0 -1
  107. package/dist/chunk-DBDRR6GF.js.map +0 -1
  108. package/dist/chunk-DJWX3GVS.js +0 -81
  109. package/dist/chunk-DJWX3GVS.js.map +0 -1
  110. package/dist/chunk-FOUG2VVS.js +0 -855
  111. package/dist/chunk-FOUG2VVS.js.map +0 -1
  112. package/dist/chunk-JZXGWLK5.js.map +0 -1
  113. package/dist/chunk-K7QEIHHJ.js +0 -613
  114. package/dist/chunk-K7QEIHHJ.js.map +0 -1
  115. package/dist/chunk-KKHDIONI.js +0 -414
  116. package/dist/chunk-KKHDIONI.js.map +0 -1
  117. package/dist/chunk-KMPRBJK4.js +0 -74
  118. package/dist/chunk-KMPRBJK4.js.map +0 -1
  119. package/dist/chunk-Q2JRAWRI.js +0 -196
  120. package/dist/chunk-Q2JRAWRI.js.map +0 -1
  121. package/dist/chunk-R6D7NEYJ.js.map +0 -1
  122. package/dist/chunk-RZTMDUO7.js +0 -49
  123. package/dist/chunk-RZTMDUO7.js.map +0 -1
  124. package/dist/chunk-STGVSCDH.js +0 -202
  125. package/dist/chunk-STGVSCDH.js.map +0 -1
  126. package/dist/chunk-YEHAEDUD.js.map +0 -1
  127. package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
  128. package/dist/corpus-eBVwhCp1.d.ts +0 -560
  129. package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
  130. package/dist/diagnose.d.ts +0 -252
  131. package/dist/diagnose.js +0 -382
  132. package/dist/diagnose.js.map +0 -1
  133. package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
  134. package/dist/governance/index.d.ts +0 -135
  135. package/dist/governance/index.js +0 -18
  136. package/dist/governance/index.js.map +0 -1
  137. package/dist/groundedness/index.d.ts +0 -112
  138. package/dist/groundedness/index.js +0 -77
  139. package/dist/groundedness/index.js.map +0 -1
  140. package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
  141. package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
  142. package/dist/knowledge/index.d.ts +0 -103
  143. package/dist/knowledge/index.js +0 -18
  144. package/dist/knowledge/index.js.map +0 -1
  145. package/dist/pareto-E-pembql.d.ts +0 -81
  146. package/dist/perf/index.d.ts +0 -123
  147. package/dist/perf/index.js +0 -18
  148. package/dist/perf/index.js.map +0 -1
  149. package/dist/prm/index.d.ts +0 -104
  150. package/dist/prm/index.js +0 -265
  151. package/dist/prm/index.js.map +0 -1
  152. package/dist/product-benchmark/index.d.ts +0 -247
  153. package/dist/product-benchmark/index.js +0 -37
  154. package/dist/product-benchmark/index.js.map +0 -1
  155. package/dist/red-team-KmmiqBlY.d.ts +0 -63
  156. package/dist/redact-B40YG2M_.d.ts +0 -45
  157. package/dist/rubric-Cc6UHvUb.d.ts +0 -73
  158. package/dist/run-critic-CmMf05uV.d.ts +0 -56
  159. package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
  160. package/dist/telemetry/file.d.ts +0 -19
  161. package/dist/telemetry/file.js +0 -45
  162. package/dist/telemetry/file.js.map +0 -1
  163. package/dist/telemetry/index.d.ts +0 -38
  164. package/dist/telemetry/index.js +0 -130
  165. package/dist/telemetry/index.js.map +0 -1
  166. package/dist/testing-C21CHsq2.d.ts +0 -20
  167. package/dist/testing.d.ts +0 -1
  168. package/dist/testing.js +0 -8
  169. package/dist/testing.js.map +0 -1
  170. package/dist/trajectory-2TkpSEVh.d.ts +0 -33
  171. package/dist/workflow/index.d.ts +0 -496
  172. package/dist/workflow/index.js +0 -2178
  173. package/dist/workflow/index.js.map +0 -1
  174. /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
  175. /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
  176. /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
  177. /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
@@ -1,252 +0,0 @@
1
- import { C as CounterfactualMutation, a as attributeCounterfactuals, b as CounterfactualRunner } from './counterfactual-DlOz8PBx.js';
2
- export { c as CounterfactualContext, d as CounterfactualResult } from './counterfactual-DlOz8PBx.js';
3
- import { S as Span } from './schema-m0gsnbt3.js';
4
- import { T as TraceStore } from './store-BcFXE6LG.js';
5
- import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
6
- import { h as AnalystSeverity, A as AnalystFinding } from './types-D1ytG0Yg.js';
7
- import { C as CorpusRecord } from './corpus-eBVwhCp1.js';
8
- import { R as RunRecord } from './run-record-I-Z3JNvO.js';
9
- import './emitter-C2rqGH_l.js';
10
- import './store-C1YxJDEK.js';
11
- import './types-C7DGg5ex.js';
12
- import '@tangle-network/tcloud';
13
- import './llm-client-DyqEH4jH.js';
14
- import './errors-oeQrLqXC.js';
15
- import './raw-provider-sink-C46HDghv.js';
16
- import '@tangle-network/agent-interface';
17
-
18
- /**
19
- * Causal sweep — WHY did this run fail?
20
- *
21
- * Orchestrates the dormant counterfactual primitives into a responsibility
22
- * report: for each candidate step, run `reps` counterfactual replays per
23
- * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`
24
- * is the execution seam) and reduce the per-rep score deltas into a mean
25
- * effect + bootstrap confidence interval (via `confidenceInterval`).
26
- *
27
- * Why `reps` is REQUIRED: a single intervention delta is one stochastic
28
- * draw — LLM re-execution from a prefix is sampled, so one replay cannot
29
- * distinguish "this step caused the failure" from sampling noise. The
30
- * signal is the distribution of deltas across reps; the CI over that
31
- * distribution is what lets a caller say "this step's effect excludes
32
- * zero" instead of eyeballing a point estimate.
33
- *
34
- * Budget discipline: the sweep never silently drops cells. When the
35
- * remaining budget cannot fund a full `reps`-sized cell, the sweep halts
36
- * and every step not fully probed is named in `uncovered`.
37
- */
38
-
39
- /** Stable reference to a trajectory step — carried through reports,
40
- * findings, and corpus records so evidence stays addressable. */
41
- interface StepRef {
42
- index: number;
43
- spanId: string;
44
- kind: Span['kind'];
45
- name: string;
46
- }
47
- declare function stepRefOf(step: TrajectoryStep): StepRef;
48
- interface CausalSweepOptions {
49
- store: TraceStore;
50
- /** The failed run to diagnose. Its `outcome.score` is the baseline every
51
- * counterfactual delta is measured against. */
52
- runId: string;
53
- /** Execution seam — identical contract to `runCounterfactual`: re-runs the
54
- * agent from the mutation point and MUST `endRun` with a numeric score. */
55
- runner: CounterfactualRunner;
56
- /** Trajectory indices to probe. Default: every llm + tool span — the kinds
57
- * the existing `CounterfactualMutation` set targets. */
58
- candidateSteps?: number[];
59
- /**
60
- * Mutations to probe a given step with. Returned mutations MUST target
61
- * `step.index`. Default probes are the payload-free existing kinds:
62
- * - tool span → `swap-tool-result` with `newResult: null` (knockout:
63
- * how much did the run depend on this tool's information?)
64
- * - llm span → `truncate-after` (re-roll: how much did the realized
65
- * turn deviate from the policy's typical continuation?)
66
- * `swap-model` / `inject-system-message` need consumer payloads, so they
67
- * are opt-in via this callback.
68
- */
69
- mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[];
70
- /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */
71
- reps: number;
72
- /** Hard cap on total counterfactual replays across the whole sweep. */
73
- budget: number;
74
- /** Seed for the bootstrap CI resampler. Deterministic default so two
75
- * sweeps over the same deltas report identical intervals. */
76
- ciSeed?: number;
77
- /** Bootstrap CI confidence level. Default 0.95. */
78
- ciConfidence?: number;
79
- }
80
- interface StepResponsibility {
81
- stepRef: StepRef;
82
- mutationKind: CounterfactualMutation['kind'];
83
- /** Mean of per-rep score deltas (counterfactual − original). */
84
- meanEffect: number;
85
- /** Bootstrap CI over the per-rep deltas. */
86
- ci: {
87
- mean: number;
88
- lower: number;
89
- upper: number;
90
- };
91
- /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */
92
- ciExcludesZero: boolean;
93
- reps: number;
94
- /** Raw per-rep deltas — downstream evidence, never re-derived. */
95
- deltas: number[];
96
- /** Replay run ids (layer='meta', parentRunId=original) for audit. */
97
- counterfactualRunIds: string[];
98
- }
99
- interface CausalResponsibilityReport {
100
- runId: string;
101
- originalScore: number;
102
- /** Ranked by |meanEffect| descending — the blame ordering. */
103
- steps: StepResponsibility[];
104
- /** Kind-level aggregate from the existing `attributeCounterfactuals`. */
105
- byMutationKind: ReturnType<typeof attributeCounterfactuals>;
106
- replaysUsed: number;
107
- budget: number;
108
- /** Steps planned but not fully probed before the budget ran out.
109
- * Named, never silent: an absent step is "no effect found"; an
110
- * uncovered step is "not measured". */
111
- uncovered: StepRef[];
112
- }
113
- declare function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport>;
114
-
115
- /**
116
- * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?
117
- *
118
- * Takes the blamed steps from a `CausalResponsibilityReport`, asks a
119
- * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate
120
- * mutations, and machine-verifies each candidate by replaying the run
121
- * WITH the mutation applied (through the same `runCounterfactual` seam
122
- * the sweep uses).
123
- *
124
- * A repair is "what should have happened" ONLY when every validation
125
- * replay crosses `flipThreshold` — a prescription is never speculated,
126
- * it is demonstrated. Candidates that don't flip, or whose replay
127
- * errors, land in `rejected` with a typed reason; nothing is dropped
128
- * silently.
129
- */
130
-
131
- /** Context handed to `proposeFix` so an LLM-backed proposer can see the
132
- * full trajectory plus the responsibility evidence for the blamed step. */
133
- interface RepairContext {
134
- runId: string;
135
- trajectory: Trajectory;
136
- originalScore: number;
137
- responsibility: StepResponsibility;
138
- }
139
- interface PrescribeRepairOptions {
140
- store: TraceStore;
141
- /** The failed run the sweep diagnosed. */
142
- runId: string;
143
- /** Execution seam — same `CounterfactualRunner` contract as the sweep. */
144
- runner: CounterfactualRunner;
145
- /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */
146
- blamed: StepResponsibility[];
147
- /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.
148
- * Returned mutations MUST target the blamed step's index. */
149
- proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>;
150
- /** Score every validation replay must reach for the repair to count. Default 0.5. */
151
- flipThreshold?: number;
152
- /** Validation replays per candidate mutation. Default 3. */
153
- repsToValidate?: number;
154
- /** Max candidate mutations tried per step. Default: all proposed. */
155
- maxAttemptsPerStep?: number;
156
- }
157
- interface ValidatedRepair {
158
- stepRef: StepRef;
159
- mutation: CounterfactualMutation;
160
- /** Always true — presence in `repairs` IS the machine-verified claim. */
161
- validated: true;
162
- /** Mean counterfactual score across the validation reps. */
163
- meanScore: number;
164
- /** meanScore − originalScore. */
165
- deltaScore: number;
166
- reps: number;
167
- /** Replay run ids backing the validation — audit trail. */
168
- counterfactualRunIds: string[];
169
- }
170
- interface RejectedRepair {
171
- stepRef: StepRef;
172
- mutation: CounterfactualMutation;
173
- reason: 'did-not-flip' | 'error';
174
- /** Present for 'did-not-flip': mean delta over the reps that ran. */
175
- deltaScore?: number;
176
- /** Present for 'error': the message, preserved for diagnosis. */
177
- error?: string;
178
- }
179
- interface RepairReport {
180
- runId: string;
181
- originalScore: number;
182
- flipThreshold: number;
183
- repairs: ValidatedRepair[];
184
- rejected: RejectedRepair[];
185
- replaysUsed: number;
186
- }
187
- declare function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport>;
188
-
189
- /**
190
- * Remediation adapters — HOW DO WE MAKE IT HAPPEN?
191
- *
192
- * The diagnose chain ends by feeding existing improvement machinery,
193
- * not by building new machinery:
194
- *
195
- * - `toAnalystFindings` → the analyst contract (`makeFinding`), so
196
- * responsibility evidence flows into the same registry / steering /
197
- * diff pipeline every other analyst feeds.
198
- * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the
199
- * diagnosed failure + validated repair as a permanent scenario.
200
- * - `suggestInvariant` → a plain-data hint in the shape the
201
- * trace-contracts machinery consumes (`never` / `without` clauses).
202
- */
203
-
204
- declare const DIAGNOSE_ANALYST_ID = "diagnose-causal-sweep";
205
- /** Severity from causal effect size. Effects whose CI includes zero are
206
- * 'info' regardless of magnitude — an indistinguishable-from-noise effect
207
- * must not steer remediation priority. */
208
- declare function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity;
209
- /** Deterministic human-readable rendering of a mutation — used in
210
- * recommended actions, corpus completions, and invariant hints. */
211
- declare function describeMutation(mutation: CounterfactualMutation): string;
212
- /**
213
- * Lift a responsibility report (and optionally its validated repairs) into
214
- * `AnalystFinding`s via the real `makeFinding` factory. One finding per
215
- * probed step; a validated repair for that step upgrades the finding with
216
- * a `recommended_action` + the replay-validation evidence.
217
- *
218
- * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,
219
- * so `derived_from_judge` stays unset and they may steer.
220
- */
221
- declare function toAnalystFindings(report: CausalResponsibilityReport, repairs?: RepairReport): AnalystFinding[];
222
- /**
223
- * Pin the diagnosed failure as a permanent corpus scenario. Takes the
224
- * original run's `RunRecord` projection plus a validated repair and emits
225
- * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw
226
- * failure and the diagnosed entry).
227
- *
228
- * `completion` defaults to the validated mutation's rendering — "what
229
- * should have happened" in machine-derived form. Supply `prompt` (and
230
- * optionally a richer `completion`) when the trajectory text is available
231
- * so the record is harvestable by `buildDatasetFromCorpus`.
232
- */
233
- declare function toCorpusRecord(run: RunRecord, repair: ValidatedRepair, opts?: {
234
- prompt?: string;
235
- completion?: string;
236
- }): CorpusRecord;
237
- /** Plain-data invariant hint. The trace-contracts machinery consumes this
238
- * shape: `never` is a pattern that must not appear in a passing trace;
239
- * `without` is a guard whose absence makes the failure reachable. */
240
- interface InvariantHint {
241
- description: string;
242
- never?: string;
243
- without?: string;
244
- }
245
- /**
246
- * Derive an invariant hint from a validated repair. Deterministic per
247
- * mutation kind — the hint names the contract a trace must satisfy so
248
- * the diagnosed failure cannot silently recur.
249
- */
250
- declare function suggestInvariant(repair: ValidatedRepair): InvariantHint;
251
-
252
- export { type CausalResponsibilityReport, type CausalSweepOptions, CounterfactualMutation, CounterfactualRunner, DIAGNOSE_ANALYST_ID, type InvariantHint, type PrescribeRepairOptions, type RejectedRepair, type RepairContext, type RepairReport, type StepRef, type StepResponsibility, type ValidatedRepair, causalSweep, describeMutation, prescribeRepair, severityFromEffect, stepRefOf, suggestInvariant, toAnalystFindings, toCorpusRecord };
package/dist/diagnose.js DELETED
@@ -1,382 +0,0 @@
1
- import {
2
- attributeCounterfactuals,
3
- runCounterfactual
4
- } from "./chunk-6SK5VFYK.js";
5
- import {
6
- buildTrajectory
7
- } from "./chunk-RZTMDUO7.js";
8
- import {
9
- makeFinding
10
- } from "./chunk-45EEMHTC.js";
11
- import {
12
- confidenceInterval
13
- } from "./chunk-XMSYF4A7.js";
14
- import {
15
- validateRunRecord
16
- } from "./chunk-VK6HBGAE.js";
17
- import "./chunk-TVVP3ZZQ.js";
18
- import "./chunk-XJYR7XFV.js";
19
- import "./chunk-VSMTAMNK.js";
20
- import {
21
- ValidationError
22
- } from "./chunk-ONWEPEDO.js";
23
- import "./chunk-PZ5AY32C.js";
24
-
25
- // src/diagnose/causal-sweep.ts
26
- function stepRefOf(step) {
27
- return {
28
- index: step.index,
29
- spanId: step.span.spanId,
30
- kind: step.span.kind,
31
- name: step.span.name
32
- };
33
- }
34
- var DEFAULT_CI_SEED = 24301;
35
- function defaultMutations(step) {
36
- if (step.span.kind === "tool") {
37
- return [{ kind: "swap-tool-result", at: step.index, newResult: null }];
38
- }
39
- if (step.span.kind === "llm") {
40
- return [{ kind: "truncate-after", at: step.index }];
41
- }
42
- return [];
43
- }
44
- async function causalSweep(opts) {
45
- if (!Number.isInteger(opts.reps) || opts.reps < 2) {
46
- throw new ValidationError(
47
- `causalSweep: reps must be an integer >= 2 (got ${opts.reps}) \u2014 a single-intervention delta is one stochastic draw, not a measurement`
48
- );
49
- }
50
- if (!Number.isInteger(opts.budget) || opts.budget < 1) {
51
- throw new ValidationError(`causalSweep: budget must be an integer >= 1 (got ${opts.budget})`);
52
- }
53
- const originalRun = await opts.store.getRun(opts.runId);
54
- if (!originalRun) throw new ValidationError(`causalSweep: run ${opts.runId} not found`);
55
- const originalScore = originalRun.outcome?.score;
56
- if (typeof originalScore !== "number" || !Number.isFinite(originalScore)) {
57
- throw new ValidationError(
58
- `causalSweep: run ${opts.runId} has no numeric outcome.score \u2014 deltas have no baseline`
59
- );
60
- }
61
- const trajectory = await buildTrajectory(opts.store, opts.runId);
62
- const candidates = resolveCandidates(trajectory, opts.candidateSteps);
63
- const mutationsFor = opts.mutationsPerStep ?? defaultMutations;
64
- const cells = [];
65
- for (const step of candidates) {
66
- const mutations = mutationsFor(step);
67
- for (const m of mutations) {
68
- if (m.at !== step.index) {
69
- throw new ValidationError(
70
- `causalSweep: mutationsPerStep returned a mutation targeting at=${m.at} for step index=${step.index} \u2014 mutations must target the step they were asked for`
71
- );
72
- }
73
- cells.push({ step, mutation: m });
74
- }
75
- }
76
- const responsibilities = [];
77
- const allResults = [];
78
- const uncoveredIndices = /* @__PURE__ */ new Set();
79
- let replaysUsed = 0;
80
- let halted = false;
81
- for (const cell of cells) {
82
- if (halted || replaysUsed + opts.reps > opts.budget) {
83
- halted = true;
84
- uncoveredIndices.add(cell.step.index);
85
- continue;
86
- }
87
- const deltas = [];
88
- const cfRunIds = [];
89
- for (let rep = 0; rep < opts.reps; rep++) {
90
- const result = await runCounterfactual(opts.store, opts.runId, cell.mutation, opts.runner);
91
- replaysUsed++;
92
- const d = result.delta.deltaScore;
93
- if (typeof d !== "number" || !Number.isFinite(d)) {
94
- throw new ValidationError(
95
- `causalSweep: counterfactual replay for step ${cell.step.index} (${cell.mutation.kind}) rep ${rep} produced no numeric score \u2014 the runner must endRun with a numeric outcome.score`
96
- );
97
- }
98
- deltas.push(d);
99
- cfRunIds.push(result.counterfactualRunId);
100
- allResults.push(result);
101
- }
102
- const ci = confidenceInterval(deltas, opts.ciConfidence ?? 0.95, {
103
- seed: opts.ciSeed ?? DEFAULT_CI_SEED
104
- });
105
- responsibilities.push({
106
- stepRef: stepRefOf(cell.step),
107
- mutationKind: cell.mutation.kind,
108
- meanEffect: ci.mean,
109
- ci,
110
- ciExcludesZero: ci.lower > 0 || ci.upper < 0,
111
- reps: opts.reps,
112
- deltas,
113
- counterfactualRunIds: cfRunIds
114
- });
115
- }
116
- responsibilities.sort((a, b) => Math.abs(b.meanEffect) - Math.abs(a.meanEffect));
117
- const uncovered = candidates.filter((s) => uncoveredIndices.has(s.index)).map(stepRefOf);
118
- return {
119
- runId: opts.runId,
120
- originalScore,
121
- steps: responsibilities,
122
- byMutationKind: attributeCounterfactuals(allResults),
123
- replaysUsed,
124
- budget: opts.budget,
125
- uncovered
126
- };
127
- }
128
- function resolveCandidates(trajectory, indices) {
129
- if (indices === void 0) {
130
- return trajectory.steps.filter((s) => s.span.kind === "llm" || s.span.kind === "tool");
131
- }
132
- return indices.map((i) => {
133
- const step = trajectory.steps[i];
134
- if (!step) {
135
- throw new ValidationError(
136
- `causalSweep: candidateSteps index ${i} out of range [0, ${trajectory.steps.length})`
137
- );
138
- }
139
- return step;
140
- });
141
- }
142
-
143
- // src/diagnose/remediation.ts
144
- var DIAGNOSE_ANALYST_ID = "diagnose-causal-sweep";
145
- function severityFromEffect(responsibility) {
146
- if (!responsibility.ciExcludesZero) return "info";
147
- const magnitude = Math.abs(responsibility.meanEffect);
148
- if (magnitude >= 0.5) return "critical";
149
- if (magnitude >= 0.25) return "high";
150
- if (magnitude >= 0.1) return "medium";
151
- return "low";
152
- }
153
- function describeMutation(mutation) {
154
- switch (mutation.kind) {
155
- case "swap-model":
156
- return `use model '${mutation.newModel}' at step ${mutation.at}`;
157
- case "swap-tool-result":
158
- return `replace the tool result at step ${mutation.at} with ${JSON.stringify(mutation.newResult)}`;
159
- case "truncate-after":
160
- return `stop the run after step ${mutation.at}`;
161
- case "inject-system-message":
162
- return `inject system message at step ${mutation.at}: ${mutation.content}`;
163
- case "custom":
164
- return `${mutation.describe} (step ${mutation.at})`;
165
- }
166
- }
167
- function toAnalystFindings(report, repairs) {
168
- const repairByStep = /* @__PURE__ */ new Map();
169
- for (const r of repairs?.repairs ?? []) {
170
- if (!repairByStep.has(r.stepRef.spanId)) repairByStep.set(r.stepRef.spanId, r);
171
- }
172
- return report.steps.map((resp) => {
173
- const repair = repairByStep.get(resp.stepRef.spanId);
174
- const evidence = [
175
- {
176
- kind: "span",
177
- uri: `span://${resp.stepRef.spanId}`,
178
- excerpt: `step ${resp.stepRef.index} (${resp.stepRef.kind} '${resp.stepRef.name}') meanEffect=${resp.meanEffect.toFixed(4)} ci=[${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] reps=${resp.reps}`
179
- },
180
- {
181
- kind: "metric",
182
- uri: `metric://diagnose/${report.runId}/step/${resp.stepRef.index}/${resp.mutationKind}`,
183
- excerpt: `deltas=[${resp.deltas.map((d) => d.toFixed(4)).join(", ")}]`
184
- },
185
- ...resp.counterfactualRunIds.map((id) => ({ kind: "span", uri: `run://${id}` }))
186
- ];
187
- return makeFinding({
188
- analyst_id: DIAGNOSE_ANALYST_ID,
189
- severity: severityFromEffect(resp),
190
- area: "causal-attribution",
191
- claim: `step '${resp.stepRef.name}' (${resp.stepRef.kind}) is causally responsible for the run outcome under ${resp.mutationKind}`,
192
- rationale: resp.ciExcludesZero ? `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] excludes zero` : `mean effect ${resp.meanEffect.toFixed(4)} over ${resp.reps} counterfactual replays; CI [${resp.ci.lower.toFixed(4)}, ${resp.ci.upper.toFixed(4)}] includes zero \u2014 not distinguishable from noise`,
193
- evidence_refs: evidence,
194
- recommended_action: repair ? describeMutation(repair.mutation) : void 0,
195
- validation_plan: repair ? `replay-validated: ${repair.reps}/${repair.reps} reps scored >= ${repairs.flipThreshold} (mean ${repair.meanScore.toFixed(4)}, delta ${repair.deltaScore.toFixed(4)})` : void 0,
196
- confidence: repair ? 0.95 : resp.ciExcludesZero ? 0.85 : 0.3,
197
- subject: resp.stepRef.spanId,
198
- metadata: {
199
- stepRef: resp.stepRef,
200
- mutationKind: resp.mutationKind,
201
- meanEffect: resp.meanEffect,
202
- ci: resp.ci,
203
- deltas: resp.deltas,
204
- counterfactualRunIds: resp.counterfactualRunIds,
205
- ...repair ? { repair: { mutation: repair.mutation, meanScore: repair.meanScore } } : {}
206
- }
207
- });
208
- });
209
- }
210
- function toCorpusRecord(run, repair, opts = {}) {
211
- const record = {
212
- ...run,
213
- runId: `${run.runId}#repair:${repair.stepRef.spanId}`,
214
- outcome: {
215
- ...run.outcome,
216
- raw: {
217
- ...run.outcome.raw,
218
- diagnose_blamed_step_index: repair.stepRef.index,
219
- diagnose_repair_mean_score: repair.meanScore,
220
- diagnose_repair_delta_score: repair.deltaScore,
221
- diagnose_repair_reps: repair.reps
222
- }
223
- },
224
- prompt: opts.prompt,
225
- completion: opts.completion ?? describeMutation(repair.mutation)
226
- };
227
- validateRunRecord(record);
228
- return record;
229
- }
230
- function suggestInvariant(repair) {
231
- const { stepRef, mutation } = repair;
232
- const at = `step ${stepRef.index} (${stepRef.kind} '${stepRef.name}')`;
233
- switch (mutation.kind) {
234
- case "swap-tool-result":
235
- return {
236
- description: `the result of tool '${stepRef.name}' was causally responsible for the failure; a replaced result flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,
237
- never: `unvalidated result from tool '${stepRef.name}' flows downstream`,
238
- without: `result guard on tool '${stepRef.name}'`
239
- };
240
- case "swap-model":
241
- return {
242
- description: `swapping the model at ${at} to '${mutation.newModel}' flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,
243
- never: `llm span '${stepRef.name}' runs on a model other than '${mutation.newModel}'`
244
- };
245
- case "inject-system-message":
246
- return {
247
- description: `injecting a system message at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`,
248
- without: `system message present at '${stepRef.name}': ${mutation.content}`
249
- };
250
- case "truncate-after":
251
- return {
252
- description: `stopping after ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)}) \u2014 continuation past this step caused the failure`,
253
- never: `spans execute after '${stepRef.name}' (index ${stepRef.index})`
254
- };
255
- case "custom":
256
- return {
257
- description: `${mutation.describe} at ${at} flipped the outcome (delta ${repair.deltaScore.toFixed(4)})`
258
- };
259
- default: {
260
- const exhausted = mutation;
261
- throw new ValidationError(
262
- `suggestInvariant: unknown mutation kind ${JSON.stringify(exhausted)}`
263
- );
264
- }
265
- }
266
- }
267
-
268
- // src/diagnose/repair.ts
269
- async function prescribeRepair(opts) {
270
- const flipThreshold = opts.flipThreshold ?? 0.5;
271
- const repsToValidate = opts.repsToValidate ?? 3;
272
- if (!Number.isInteger(repsToValidate) || repsToValidate < 1) {
273
- throw new ValidationError(
274
- `prescribeRepair: repsToValidate must be an integer >= 1 (got ${repsToValidate})`
275
- );
276
- }
277
- const maxAttempts = opts.maxAttemptsPerStep ?? Number.POSITIVE_INFINITY;
278
- if (maxAttempts < 1) {
279
- throw new ValidationError(
280
- `prescribeRepair: maxAttemptsPerStep must be >= 1 (got ${opts.maxAttemptsPerStep})`
281
- );
282
- }
283
- if (opts.blamed.length === 0) {
284
- throw new ValidationError("prescribeRepair: blamed is empty \u2014 nothing to repair");
285
- }
286
- const originalRun = await opts.store.getRun(opts.runId);
287
- if (!originalRun) throw new ValidationError(`prescribeRepair: run ${opts.runId} not found`);
288
- const originalScore = originalRun.outcome?.score;
289
- if (typeof originalScore !== "number" || !Number.isFinite(originalScore)) {
290
- throw new ValidationError(
291
- `prescribeRepair: run ${opts.runId} has no numeric outcome.score \u2014 flips have no baseline`
292
- );
293
- }
294
- const trajectory = await buildTrajectory(opts.store, opts.runId);
295
- const repairs = [];
296
- const rejected = [];
297
- let replaysUsed = 0;
298
- for (const responsibility of opts.blamed) {
299
- const step = trajectory.steps[responsibility.stepRef.index];
300
- if (!step || step.span.spanId !== responsibility.stepRef.spanId) {
301
- throw new ValidationError(
302
- `prescribeRepair: blamed step index=${responsibility.stepRef.index} spanId=${responsibility.stepRef.spanId} does not match run ${opts.runId} \u2014 stale report?`
303
- );
304
- }
305
- const candidates = await opts.proposeFix(step, {
306
- runId: opts.runId,
307
- trajectory,
308
- originalScore,
309
- responsibility
310
- });
311
- const toTry = candidates.slice(0, maxAttempts);
312
- for (const mutation of toTry) {
313
- if (mutation.at !== step.index) {
314
- throw new ValidationError(
315
- `prescribeRepair: proposeFix returned a mutation targeting at=${mutation.at} for blamed step index=${step.index}`
316
- );
317
- }
318
- const scores = [];
319
- const cfRunIds = [];
320
- let failure;
321
- for (let rep = 0; rep < repsToValidate; rep++) {
322
- try {
323
- const result = await runCounterfactual(opts.store, opts.runId, mutation, opts.runner);
324
- replaysUsed++;
325
- const score = result.delta.counterfactualOutcomeScore;
326
- if (typeof score !== "number" || !Number.isFinite(score)) {
327
- failure = `validation rep ${rep} produced no numeric score \u2014 the runner must endRun with a numeric outcome.score`;
328
- break;
329
- }
330
- scores.push(score);
331
- cfRunIds.push(result.counterfactualRunId);
332
- } catch (err) {
333
- replaysUsed++;
334
- failure = err instanceof Error ? err.message : String(err);
335
- break;
336
- }
337
- }
338
- if (failure !== void 0) {
339
- rejected.push({
340
- stepRef: responsibility.stepRef,
341
- mutation,
342
- reason: "error",
343
- error: failure
344
- });
345
- continue;
346
- }
347
- const meanScore = scores.reduce((a, b) => a + b, 0) / scores.length;
348
- const everyRepFlipped = scores.every((s) => s >= flipThreshold);
349
- if (everyRepFlipped) {
350
- repairs.push({
351
- stepRef: responsibility.stepRef,
352
- mutation,
353
- validated: true,
354
- meanScore,
355
- deltaScore: meanScore - originalScore,
356
- reps: repsToValidate,
357
- counterfactualRunIds: cfRunIds
358
- });
359
- break;
360
- }
361
- rejected.push({
362
- stepRef: responsibility.stepRef,
363
- mutation,
364
- reason: "did-not-flip",
365
- deltaScore: meanScore - originalScore
366
- });
367
- }
368
- }
369
- return { runId: opts.runId, originalScore, flipThreshold, repairs, rejected, replaysUsed };
370
- }
371
- export {
372
- DIAGNOSE_ANALYST_ID,
373
- causalSweep,
374
- describeMutation,
375
- prescribeRepair,
376
- severityFromEffect,
377
- stepRefOf,
378
- suggestInvariant,
379
- toAnalystFindings,
380
- toCorpusRecord
381
- };
382
- //# sourceMappingURL=diagnose.js.map