@tangle-network/agent-eval 0.86.0 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/dist/adapters/http.d.ts +3 -3
  2. package/dist/adapters/langchain.d.ts +3 -3
  3. package/dist/adapters/otel.d.ts +6 -6
  4. package/dist/adversarial-DIVcDoI_.d.ts +88 -0
  5. package/dist/analyst/index.d.ts +11 -10
  6. package/dist/analyst/index.js +13 -8
  7. package/dist/analyst/index.js.map +1 -1
  8. package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
  9. package/dist/belief-state/index.d.ts +4 -4
  10. package/dist/belief-state/index.js +1 -1
  11. package/dist/benchmarks/index.d.ts +3 -3
  12. package/dist/campaign/index.d.ts +165 -18
  13. package/dist/campaign/index.js +289 -14
  14. package/dist/campaign/index.js.map +1 -1
  15. package/dist/chunk-45EEMHTC.js +35 -0
  16. package/dist/chunk-45EEMHTC.js.map +1 -0
  17. package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
  18. package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
  19. package/dist/chunk-5HRORJQY.js.map +1 -0
  20. package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
  21. package/dist/chunk-BOD4O7OF.js +40 -0
  22. package/dist/chunk-BOD4O7OF.js.map +1 -0
  23. package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
  24. package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
  25. package/dist/chunk-D3V5B42D.js.map +1 -0
  26. package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
  27. package/dist/chunk-FIUKOSWI.js.map +1 -0
  28. package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
  29. package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
  30. package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
  31. package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
  32. package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
  33. package/dist/chunk-QAY5UIJO.js.map +1 -0
  34. package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
  35. package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
  36. package/dist/chunk-REVYNR6C.js +100 -0
  37. package/dist/chunk-REVYNR6C.js.map +1 -0
  38. package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
  39. package/dist/chunk-UHMJT4T7.js +200 -0
  40. package/dist/chunk-UHMJT4T7.js.map +1 -0
  41. package/dist/chunk-UMMZHCPB.js +190 -0
  42. package/dist/chunk-UMMZHCPB.js.map +1 -0
  43. package/dist/chunk-VZSRQ272.js +149 -0
  44. package/dist/chunk-VZSRQ272.js.map +1 -0
  45. package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
  46. package/dist/chunk-XY4DDNEG.js.map +1 -0
  47. package/dist/chunk-Y47J2LJ3.js +859 -0
  48. package/dist/chunk-Y47J2LJ3.js.map +1 -0
  49. package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
  50. package/dist/chunk-ZFIBGEOL.js.map +1 -0
  51. package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
  52. package/dist/contract/index.d.ts +24 -95
  53. package/dist/contract/index.js +16 -755
  54. package/dist/contract/index.js.map +1 -1
  55. package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
  56. package/dist/control.d.ts +5 -5
  57. package/dist/corpus-BoR-041R.d.ts +560 -0
  58. package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
  59. package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
  60. package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
  61. package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
  62. package/dist/diagnose.d.ts +251 -0
  63. package/dist/diagnose.js +381 -0
  64. package/dist/diagnose.js.map +1 -0
  65. package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
  66. package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
  67. package/dist/fuzz.d.ts +484 -0
  68. package/dist/fuzz.js +613 -0
  69. package/dist/fuzz.js.map +1 -0
  70. package/dist/governance/index.d.ts +4 -4
  71. package/dist/hosted/index.d.ts +6 -6
  72. package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
  73. package/dist/index.d.ts +717 -455
  74. package/dist/index.js +1590 -793
  75. package/dist/index.js.map +1 -1
  76. package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
  77. package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
  78. package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
  79. package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
  80. package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
  81. package/dist/matrix/index.d.ts +2 -2
  82. package/dist/meta-eval/index.d.ts +177 -3
  83. package/dist/meta-eval/index.js +260 -1
  84. package/dist/meta-eval/index.js.map +1 -1
  85. package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
  86. package/dist/multishot/index.d.ts +25 -11
  87. package/dist/multishot/index.js +36 -7
  88. package/dist/multishot/index.js.map +1 -1
  89. package/dist/openapi.json +1 -1
  90. package/dist/pipelines/index.js +2 -2
  91. package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
  92. package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
  93. package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
  94. package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
  95. package/dist/reporting.d.ts +8 -8
  96. package/dist/reporting.js +3 -3
  97. package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
  98. package/dist/rl.d.ts +194 -656
  99. package/dist/rl.js +236 -154
  100. package/dist/rl.js.map +1 -1
  101. package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
  102. package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
  103. package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
  104. package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
  105. package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
  106. package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
  107. package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
  108. package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
  109. package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
  110. package/dist/traces.d.ts +3 -3
  111. package/dist/traces.js +8 -6
  112. package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
  113. package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
  114. package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
  115. package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
  116. package/dist/wire/index.d.ts +3 -3
  117. package/dist/workflow/index.d.ts +12 -11
  118. package/dist/workflow/index.js +1 -1
  119. package/package.json +11 -1
  120. package/dist/chunk-BABOZOSN.js.map +0 -1
  121. package/dist/chunk-L5G7OUKD.js.map +0 -1
  122. package/dist/chunk-SHTXZ4O2.js +0 -113
  123. package/dist/chunk-SHTXZ4O2.js.map +0 -1
  124. package/dist/chunk-VIDQF3F5.js.map +0 -1
  125. package/dist/chunk-VUINJM5M.js.map +0 -1
  126. package/dist/chunk-YGYXHNAQ.js.map +0 -1
  127. package/dist/chunk-YV7J7X5N.js.map +0 -1
  128. /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
  129. /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
  130. /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
  131. /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
  132. /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
  133. /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
  134. /package/dist/{run-campaign-4Y5V5CN3.js.map → run-campaign-RDGAM5KJ.js.map} +0 -0
@@ -0,0 +1,85 @@
1
+ import { T as TraceEmitter } from './emitter-DEZwY14K.js';
2
+ import { T as TraceStore } from './store-CKUAgsJz.js';
3
+ import { T as Trajectory, a as TrajectoryStep } from './trajectory-GEdXJCL5.js';
4
+
5
+ /**
6
+ * Counterfactual replay — "what would have happened if we'd changed
7
+ * exactly one thing at turn N?"
8
+ *
9
+ * The framework does NOT drive the agent — it sets up the replay
10
+ * context (prior spans, prior state, mutation spec) and records the
11
+ * resulting divergence. Consumers supply an `executeFrom(ctx)` callback
12
+ * that runs their agent starting from turn N with the mutation applied.
13
+ *
14
+ * Counterfactual runs are recorded as a new Run with `layer='meta'` and
15
+ * `parentRunId = originalRunId`, so downstream diff + correlation
16
+ * pipelines see them natively.
17
+ */
18
+
19
+ type CounterfactualMutation = {
20
+ kind: 'swap-model';
21
+ at: number;
22
+ newModel: string;
23
+ } | {
24
+ kind: 'swap-tool-result';
25
+ at: number;
26
+ newResult: unknown;
27
+ } | {
28
+ kind: 'truncate-after';
29
+ at: number;
30
+ } | {
31
+ kind: 'inject-system-message';
32
+ at: number;
33
+ content: string;
34
+ } | {
35
+ kind: 'custom';
36
+ at: number;
37
+ describe: string;
38
+ apply: (step: TrajectoryStep) => TrajectoryStep;
39
+ };
40
+ interface CounterfactualContext {
41
+ originalRunId: string;
42
+ originalTrajectory: Trajectory;
43
+ /** Steps up to (but not including) the mutation point — the prefix the
44
+ * replayed agent inherits as its prior conversation/tool history. */
45
+ prefix: TrajectoryStep[];
46
+ mutation: CounterfactualMutation;
47
+ /** Pre-applied mutation on the step at `mutation.at`. Consumers use this
48
+ * as the FIRST step the replayed agent emits (they decide whether to
49
+ * re-emit it or continue from there). */
50
+ mutatedStep: TrajectoryStep;
51
+ }
52
+ interface CounterfactualResult {
53
+ counterfactualRunId: string;
54
+ originalRunId: string;
55
+ mutation: CounterfactualMutation;
56
+ /** Structured delta summary — caller can extend via scoring. */
57
+ delta: {
58
+ originalOutcomeScore: number | null;
59
+ counterfactualOutcomeScore: number | null;
60
+ deltaScore: number | null;
61
+ };
62
+ }
63
+ interface CounterfactualRunner {
64
+ /**
65
+ * Execute the agent from `ctx.prefix` with the mutation applied.
66
+ * MUST emit spans into the provided emitter so they become part of
67
+ * the counterfactual run. MUST call emitter.endRun() with a verdict.
68
+ */
69
+ executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>;
70
+ }
71
+ declare function runCounterfactual(store: TraceStore, originalRunId: string, mutation: CounterfactualMutation, runner: CounterfactualRunner): Promise<CounterfactualResult>;
72
+ /**
73
+ * Aggregate a batch of counterfactuals into a simple attribution table:
74
+ * which mutation kinds move outcomes most? (Useful when you run a grid
75
+ * over the same trajectory — swap-model at every llm span, swap-tool
76
+ * at every tool span — and want a ranked summary.)
77
+ */
78
+ declare function attributeCounterfactuals(results: CounterfactualResult[]): Array<{
79
+ mutationKind: CounterfactualMutation['kind'];
80
+ n: number;
81
+ meanAbsDelta: number;
82
+ meanSignedDelta: number;
83
+ }>;
84
+
85
+ export { type CounterfactualMutation as C, attributeCounterfactuals as a, type CounterfactualRunner as b, type CounterfactualContext as c, type CounterfactualResult as d, runCounterfactual as r };
@@ -1,4 +1,4 @@
1
- import { V as ValidationError } from './errors-Dwqw-T_m.js';
1
+ import { V as ValidationError } from './errors-CzMUYo7b.js';
2
2
 
3
3
  /**
4
4
  * Dataset — versioned, sliceable, content-hashed scenario collection.
@@ -1,4 +1,6 @@
1
- import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-Cu3u_x59.js';
1
+ import { AxAIService } from '@ax-llm/ax';
2
+ import { T as TraceAnalystKindSpec } from './kind-factory-5b7xXXOr.js';
3
+ import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-2VVIL04s.js';
2
4
 
3
5
  /**
4
6
  * AnalystRegistry — orchestrate N analysts against one run.
@@ -125,4 +127,29 @@ declare class AnalystRegistry {
125
127
  private routeInput;
126
128
  }
127
129
 
128
- export { type AnalystHooks as A, type BudgetPolicy as B, type RegistryRunOpts as R, AnalystRegistry as a, type AnalystRegistryOptions as b };
130
+ /**
131
+ * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
132
+ * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
133
+ *
134
+ * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
135
+ * model and is model-agnostic by construction). The agentic RLM kinds are
136
+ * registered only when an `ai` service is supplied — so a caller with no LLM
137
+ * still gets the full behavioral/efficiency diagnosis, and the substrate's
138
+ * "any model (including no model)" guarantee holds at the suite level.
139
+ */
140
+
141
+ interface DefaultAnalystRegistryOptions {
142
+ /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
143
+ ai?: AxAIService;
144
+ /** Model for the agentic kinds (falls back to the ai service default). */
145
+ model?: string;
146
+ /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
147
+ kinds?: readonly TraceAnalystKindSpec[];
148
+ /** Set false to omit the deterministic behavioral analyst (default: include). */
149
+ includeBehavioral?: boolean;
150
+ /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
151
+ registry?: AnalystRegistryOptions;
152
+ }
153
+ declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
154
+
155
+ export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a, buildDefaultAnalystRegistry as b, type AnalystRegistryOptions as c };
@@ -0,0 +1,251 @@
1
+ import { C as CounterfactualMutation, a as attributeCounterfactuals, b as CounterfactualRunner } from './counterfactual-Dwibr5IW.js';
2
+ export { c as CounterfactualContext, d as CounterfactualResult } from './counterfactual-Dwibr5IW.js';
3
+ import { S as Span } from './schema-m0gsnbt3.js';
4
+ import { T as TraceStore } from './store-CKUAgsJz.js';
5
+ import { a as TrajectoryStep, T as Trajectory } from './trajectory-GEdXJCL5.js';
6
+ import { h as AnalystSeverity, c as AnalystFinding } from './types-2VVIL04s.js';
7
+ import { C as CorpusRecord } from './corpus-BoR-041R.js';
8
+ import { R as RunRecord } from './run-record-e7vj1uZQ.js';
9
+ import './emitter-DEZwY14K.js';
10
+ import './store-C1YxJDEK.js';
11
+ import './types-Croy5h7V.js';
12
+ import '@tangle-network/tcloud';
13
+ import './llm-client-BeEcAokY.js';
14
+ import './errors-CzMUYo7b.js';
15
+ import './raw-provider-sink-C46HDghv.js';
16
+
17
+ /**
18
+ * Causal sweep — WHY did this run fail?
19
+ *
20
+ * Orchestrates the dormant counterfactual primitives into a responsibility
21
+ * report: for each candidate step, run `reps` counterfactual replays per
22
+ * mutation (via `runCounterfactual` — the consumer's `CounterfactualRunner`
23
+ * is the execution seam) and reduce the per-rep score deltas into a mean
24
+ * effect + bootstrap confidence interval (via `confidenceInterval`).
25
+ *
26
+ * Why `reps` is REQUIRED: a single intervention delta is one stochastic
27
+ * draw — LLM re-execution from a prefix is sampled, so one replay cannot
28
+ * distinguish "this step caused the failure" from sampling noise. The
29
+ * signal is the distribution of deltas across reps; the CI over that
30
+ * distribution is what lets a caller say "this step's effect excludes
31
+ * zero" instead of eyeballing a point estimate.
32
+ *
33
+ * Budget discipline: the sweep never silently drops cells. When the
34
+ * remaining budget cannot fund a full `reps`-sized cell, the sweep halts
35
+ * and every step not fully probed is named in `uncovered`.
36
+ */
37
+
38
+ /** Stable reference to a trajectory step — carried through reports,
39
+ * findings, and corpus records so evidence stays addressable. */
40
+ interface StepRef {
41
+ index: number;
42
+ spanId: string;
43
+ kind: Span['kind'];
44
+ name: string;
45
+ }
46
+ declare function stepRefOf(step: TrajectoryStep): StepRef;
47
+ interface CausalSweepOptions {
48
+ store: TraceStore;
49
+ /** The failed run to diagnose. Its `outcome.score` is the baseline every
50
+ * counterfactual delta is measured against. */
51
+ runId: string;
52
+ /** Execution seam — identical contract to `runCounterfactual`: re-runs the
53
+ * agent from the mutation point and MUST `endRun` with a numeric score. */
54
+ runner: CounterfactualRunner;
55
+ /** Trajectory indices to probe. Default: every llm + tool span — the kinds
56
+ * the existing `CounterfactualMutation` set targets. */
57
+ candidateSteps?: number[];
58
+ /**
59
+ * Mutations to probe a given step with. Returned mutations MUST target
60
+ * `step.index`. Default probes are the payload-free existing kinds:
61
+ * - tool span → `swap-tool-result` with `newResult: null` (knockout:
62
+ * how much did the run depend on this tool's information?)
63
+ * - llm span → `truncate-after` (re-roll: how much did the realized
64
+ * turn deviate from the policy's typical continuation?)
65
+ * `swap-model` / `inject-system-message` need consumer payloads, so they
66
+ * are opt-in via this callback.
67
+ */
68
+ mutationsPerStep?: (step: TrajectoryStep) => CounterfactualMutation[];
69
+ /** Replays per (step, mutation) cell. Minimum 2 — see module doc. */
70
+ reps: number;
71
+ /** Hard cap on total counterfactual replays across the whole sweep. */
72
+ budget: number;
73
+ /** Seed for the bootstrap CI resampler. Deterministic default so two
74
+ * sweeps over the same deltas report identical intervals. */
75
+ ciSeed?: number;
76
+ /** Bootstrap CI confidence level. Default 0.95. */
77
+ ciConfidence?: number;
78
+ }
79
+ interface StepResponsibility {
80
+ stepRef: StepRef;
81
+ mutationKind: CounterfactualMutation['kind'];
82
+ /** Mean of per-rep score deltas (counterfactual − original). */
83
+ meanEffect: number;
84
+ /** Bootstrap CI over the per-rep deltas. */
85
+ ci: {
86
+ mean: number;
87
+ lower: number;
88
+ upper: number;
89
+ };
90
+ /** `ci.lower > 0 || ci.upper < 0` — the effect is distinguishable from noise. */
91
+ ciExcludesZero: boolean;
92
+ reps: number;
93
+ /** Raw per-rep deltas — downstream evidence, never re-derived. */
94
+ deltas: number[];
95
+ /** Replay run ids (layer='meta', parentRunId=original) for audit. */
96
+ counterfactualRunIds: string[];
97
+ }
98
+ interface CausalResponsibilityReport {
99
+ runId: string;
100
+ originalScore: number;
101
+ /** Ranked by |meanEffect| descending — the blame ordering. */
102
+ steps: StepResponsibility[];
103
+ /** Kind-level aggregate from the existing `attributeCounterfactuals`. */
104
+ byMutationKind: ReturnType<typeof attributeCounterfactuals>;
105
+ replaysUsed: number;
106
+ budget: number;
107
+ /** Steps planned but not fully probed before the budget ran out.
108
+ * Named, never silent: an absent step is "no effect found"; an
109
+ * uncovered step is "not measured". */
110
+ uncovered: StepRef[];
111
+ }
112
+ declare function causalSweep(opts: CausalSweepOptions): Promise<CausalResponsibilityReport>;
113
+
114
+ /**
115
+ * Replay-validated repair — WHAT SHOULD HAVE HAPPENED?
116
+ *
117
+ * Takes the blamed steps from a `CausalResponsibilityReport`, asks a
118
+ * consumer-supplied `proposeFix` (LLM-backed in live use) for candidate
119
+ * mutations, and machine-verifies each candidate by replaying the run
120
+ * WITH the mutation applied (through the same `runCounterfactual` seam
121
+ * the sweep uses).
122
+ *
123
+ * A repair is "what should have happened" ONLY when every validation
124
+ * replay crosses `flipThreshold` — a prescription is never speculated,
125
+ * it is demonstrated. Candidates that don't flip, or whose replay
126
+ * errors, land in `rejected` with a typed reason; nothing is dropped
127
+ * silently.
128
+ */
129
+
130
+ /** Context handed to `proposeFix` so an LLM-backed proposer can see the
131
+ * full trajectory plus the responsibility evidence for the blamed step. */
132
+ interface RepairContext {
133
+ runId: string;
134
+ trajectory: Trajectory;
135
+ originalScore: number;
136
+ responsibility: StepResponsibility;
137
+ }
138
+ interface PrescribeRepairOptions {
139
+ store: TraceStore;
140
+ /** The failed run the sweep diagnosed. */
141
+ runId: string;
142
+ /** Execution seam — same `CounterfactualRunner` contract as the sweep. */
143
+ runner: CounterfactualRunner;
144
+ /** Blamed steps from `causalSweep` — typically `report.steps.slice(0, k)`. */
145
+ blamed: StepResponsibility[];
146
+ /** Candidate-fix generator. Consumer-supplied; LLM-backed in live use.
147
+ * Returned mutations MUST target the blamed step's index. */
148
+ proposeFix: (step: TrajectoryStep, context: RepairContext) => Promise<CounterfactualMutation[]>;
149
+ /** Score every validation replay must reach for the repair to count. Default 0.5. */
150
+ flipThreshold?: number;
151
+ /** Validation replays per candidate mutation. Default 3. */
152
+ repsToValidate?: number;
153
+ /** Max candidate mutations tried per step. Default: all proposed. */
154
+ maxAttemptsPerStep?: number;
155
+ }
156
+ interface ValidatedRepair {
157
+ stepRef: StepRef;
158
+ mutation: CounterfactualMutation;
159
+ /** Always true — presence in `repairs` IS the machine-verified claim. */
160
+ validated: true;
161
+ /** Mean counterfactual score across the validation reps. */
162
+ meanScore: number;
163
+ /** meanScore − originalScore. */
164
+ deltaScore: number;
165
+ reps: number;
166
+ /** Replay run ids backing the validation — audit trail. */
167
+ counterfactualRunIds: string[];
168
+ }
169
+ interface RejectedRepair {
170
+ stepRef: StepRef;
171
+ mutation: CounterfactualMutation;
172
+ reason: 'did-not-flip' | 'error';
173
+ /** Present for 'did-not-flip': mean delta over the reps that ran. */
174
+ deltaScore?: number;
175
+ /** Present for 'error': the message, preserved for diagnosis. */
176
+ error?: string;
177
+ }
178
+ interface RepairReport {
179
+ runId: string;
180
+ originalScore: number;
181
+ flipThreshold: number;
182
+ repairs: ValidatedRepair[];
183
+ rejected: RejectedRepair[];
184
+ replaysUsed: number;
185
+ }
186
+ declare function prescribeRepair(opts: PrescribeRepairOptions): Promise<RepairReport>;
187
+
188
+ /**
189
+ * Remediation adapters — HOW DO WE MAKE IT HAPPEN?
190
+ *
191
+ * The diagnose chain ends by feeding existing improvement machinery,
192
+ * not by building new machinery:
193
+ *
194
+ * - `toAnalystFindings` → the analyst contract (`makeFinding`), so
195
+ * responsibility evidence flows into the same registry / steering /
196
+ * diff pipeline every other analyst feeds.
197
+ * - `toCorpusRecord` → the RL corpus (`CorpusRecord`), pinning the
198
+ * diagnosed failure + validated repair as a permanent scenario.
199
+ * - `suggestInvariant` → a plain-data hint in the shape the
200
+ * trace-contracts machinery consumes (`never` / `without` clauses).
201
+ */
202
+
203
+ declare const DIAGNOSE_ANALYST_ID = "diagnose-causal-sweep";
204
+ /** Severity from causal effect size. Effects whose CI includes zero are
205
+ * 'info' regardless of magnitude — an indistinguishable-from-noise effect
206
+ * must not steer remediation priority. */
207
+ declare function severityFromEffect(responsibility: StepResponsibility): AnalystSeverity;
208
+ /** Deterministic human-readable rendering of a mutation — used in
209
+ * recommended actions, corpus completions, and invariant hints. */
210
+ declare function describeMutation(mutation: CounterfactualMutation): string;
211
+ /**
212
+ * Lift a responsibility report (and optionally its validated repairs) into
213
+ * `AnalystFinding`s via the real `makeFinding` factory. One finding per
214
+ * probed step; a validated repair for that step upgrades the finding with
215
+ * a `recommended_action` + the replay-validation evidence.
216
+ *
217
+ * Findings are OBSERVED causal probes (replay deltas), not judge verdicts,
218
+ * so `derived_from_judge` stays unset and they may steer.
219
+ */
220
+ declare function toAnalystFindings(report: CausalResponsibilityReport, repairs?: RepairReport): AnalystFinding[];
221
+ /**
222
+ * Pin the diagnosed failure as a permanent corpus scenario. Takes the
223
+ * original run's `RunRecord` projection plus a validated repair and emits
224
+ * a fresh `CorpusRecord` (new runId, so corpus dedup keeps both the raw
225
+ * failure and the diagnosed entry).
226
+ *
227
+ * `completion` defaults to the validated mutation's rendering — "what
228
+ * should have happened" in machine-derived form. Supply `prompt` (and
229
+ * optionally a richer `completion`) when the trajectory text is available
230
+ * so the record is harvestable by `buildDatasetFromCorpus`.
231
+ */
232
+ declare function toCorpusRecord(run: RunRecord, repair: ValidatedRepair, opts?: {
233
+ prompt?: string;
234
+ completion?: string;
235
+ }): CorpusRecord;
236
+ /** Plain-data invariant hint. The trace-contracts machinery consumes this
237
+ * shape: `never` is a pattern that must not appear in a passing trace;
238
+ * `without` is a guard whose absence makes the failure reachable. */
239
+ interface InvariantHint {
240
+ description: string;
241
+ never?: string;
242
+ without?: string;
243
+ }
244
+ /**
245
+ * Derive an invariant hint from a validated repair. Deterministic per
246
+ * mutation kind — the hint names the contract a trace must satisfy so
247
+ * the diagnosed failure cannot silently recur.
248
+ */
249
+ declare function suggestInvariant(repair: ValidatedRepair): InvariantHint;
250
+
251
+ export { type CausalResponsibilityReport, type CausalSweepOptions, CounterfactualMutation, CounterfactualRunner, DIAGNOSE_ANALYST_ID, type InvariantHint, type PrescribeRepairOptions, type RejectedRepair, type RepairContext, type RepairReport, type StepRef, type StepResponsibility, type ValidatedRepair, causalSweep, describeMutation, prescribeRepair, severityFromEffect, stepRefOf, suggestInvariant, toAnalystFindings, toCorpusRecord };