@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,541 +0,0 @@
1
- import { S as Scenario, G as Gate, m as GateResult, l as GateContext, C as CampaignResult, p as Mutator, c as SurfaceProposer, M as MutableSurface, n as GenerationCandidate, d as GateDecision } from './types-BSw1rOUB.js';
2
- import { n as RedTeamCase, D as Direction, h as RunCampaignOptions } from './gepa-eESocoDi.js';
3
- import { R as RunRecord } from './run-record-BDH49H2E.js';
4
- import { a as PairedBootstrapResult } from './statistics-KUnG73jH.js';
5
- import { P as PolicyEditCandidateRecord } from './policy-edit-wG9uFEFm.js';
6
- import { HostedClient, TraceSpanEvent } from './hosted/index.js';
7
- import { C as CampaignStorage } from './storage-DrX3v_5B.js';
8
-
9
- /**
10
- * Compose multiple `Gate` implementations — every gate must pass for the
11
- * composite to ship. Closes the alignment reviewer's "default-only
12
- * heldOutGate + costGate would happily promote a reward-hacked prompt"
13
- * concern by making safety gates first-class composable defaults.
14
- */
15
-
16
- /** Compose gates — all must `ship` for the composite to `ship`. First
17
- * non-ship verdict short-circuits the composite verdict, but ALL gates run
18
- * (so the result records every gate's reason — useful for diagnostics). */
19
- declare function composeGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(...gates: Array<Gate<TArtifact, TScenario>>): Gate<TArtifact, TScenario>;
20
-
21
- /**
22
- * `defaultProductionGate` — composes the substrate's existing safety
23
- * primitives (red-team / reward-hacking / canary / heldout) into a single
24
- * Gate.decide shape. Closes the alignment + Anthropic-SI reviewers' "safety
25
- * primitives are off the critical path" blocker.
26
- *
27
- * The composition is opinionated — when consumers wire `runImprovementLoop`,
28
- * THIS gate is the default. Consumers can still pass a custom gate to
29
- * override; the recommended pattern is to compose THIS gate with whatever
30
- * extra domain-specific gates they need (`composeGate(defaultProductionGate(...), customGate)`).
31
- */
32
-
33
- interface DefaultProductionGateOptions {
34
- /** Required: scenarios held out from training; substrate compares
35
- * candidate-on-holdout vs baseline-on-holdout. */
36
- holdoutScenarios: Scenario[];
37
- /** Minimum held-out lift the **paired-bootstrap CI lower bound** must clear
38
- * to ship — NOT a point estimate. Default 0 ⇒ "confidently positive at the
39
- * confidence level". Interpreted in the judge's native composite scale (set
40
- * e.g. 2 for a 0-100 rubric to require a ≥2-point significant gain). */
41
- deltaThreshold?: number;
42
- /** Confidence level for the held-out + dimension bootstraps. Default 0.95. */
43
- confidence?: number;
44
- /** Bootstrap resamples. Default 2000. */
45
- bootstrapResamples?: number;
46
- /** Fixed bootstrap seed for a deterministic verdict. Default 1337. */
47
- bootstrapSeed?: number;
48
- /** Minimum paired holdout observations (scenarios × reps) before a
49
- * significance claim is allowed; below it the gate HOLDS with `few_runs`
50
- * rather than reading a degenerate CI. Default 3. */
51
- minProductiveRuns?: number;
52
- /** Ship statistic for the held-out significance test. Default `'mean'`
53
- * (tie-robust — see `heldoutSignificance`). Pass `'median'` for
54
- * outlier-robustness at the cost of tie-blindness. */
55
- heldoutStatistic?: 'mean' | 'median';
56
- /** Critical judge dimensions that must NOT significantly regress even when
57
- * the net composite rises (anti-Goodhart). The gate HOLDS if any listed
58
- * dimension's paired-delta CI lower bound < −`regressionTolerance`. E.g.
59
- * `['hallucination_free']` for a legal agent. */
60
- criticalDimensions?: string[];
61
- /** Tolerance for the per-dimension regression guard, in the dimension's
62
- * native scale. When omitted it auto-scales off observed magnitudes:
63
- * 0.05 on [0,1], 5 on 0-100. */
64
- regressionTolerance?: number;
65
- /** Total $ budget for ALL cells in this campaign — including baseline + candidate.
66
- * Composite verdict refuses to ship when spend exceeded budget. */
67
- budgetUsd?: number;
68
- /** Red-team cases to probe candidate outputs against. When omitted the
69
- * substrate uses `DEFAULT_RED_TEAM_CORPUS`. Provide a domain-specific
70
- * battery for tighter coverage. */
71
- redTeamBattery?: RedTeamCase[];
72
- /** Run records (oldest-first) needed for the reward-hacking detector.
73
- * Substrate populates from prior production-loop generations. */
74
- recentRuns?: RunRecord[];
75
- /** When true, the gate refuses to ship if the reward-hacking detector
76
- * fires at the `gaming` severity. Default true. */
77
- blockOnRewardHackingGaming?: boolean;
78
- }
79
- /**
80
- * Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
81
- */
82
- declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
83
-
84
- /**
85
- * @module
86
- * Composable held-out promotion gate backed by paired bootstrap confidence.
87
- *
88
- * Pair by full `scenario:rep` cellId, bootstrap the paired candidate-minus-
89
- * baseline delta, and ship only when CI.low strictly clears the threshold with
90
- * at least `minProductiveRuns` paired observations.
91
- *
92
- * Use when you want held-out significance as ONE of N composed gates instead
93
- * of the full `defaultProductionGate` stack (which adds critical-dimension
94
- * regression + reward-hacking guards on top).
95
- */
96
-
97
- interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
98
- scenarios: TScenario[];
99
- /** Effect-size threshold the CI lower bound must clear, in the judge's native
100
- * scale. Default 0.5. Equality holds; CI.low must be greater than this value. */
101
- deltaThreshold?: number;
102
- /** Bootstrap CI confidence. Default 0.95. */
103
- confidence?: number;
104
- /** Minimum paired holdout observations to claim significance. Default 3. */
105
- minProductiveRuns?: number;
106
- /** Bootstrap resamples. Default 2000. */
107
- resamples?: number;
108
- /** Fixed bootstrap seed for deterministic verdicts. Default 1337. */
109
- bootstrapSeed?: number;
110
- }
111
- /**
112
- * Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
113
- * of the candidate-minus-baseline composite delta clears `deltaThreshold`.
114
- */
115
- declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
116
-
117
- /**
118
- * Power preflight — "can this budget detect the effect you are hunting?"
119
- *
120
- * The failure it prevents (measured, twice): a live prompt-improvement campaign ran
121
- * 333 sandbox cells over 5.6 hours and produced a +0.08 holdout lift the ship gate
122
- * (paired bootstrap, CI.low > 0.05) could not distinguish from zero — because at
123
- * that holdout size and worker variance the MINIMUM DETECTABLE lift was larger than
124
- * any effect a prompt change plausibly produces. The budget was spent learning what
125
- * a 30-second calculation on the baseline cells already knew. No eval framework we
126
- * know of surfaces this; every underpowered improvement run everywhere ends in an
127
- * uninformative "hold".
128
- *
129
- * Model: the ship rule is `CI.low(paired Δ) > deltaThreshold`. Approximating the
130
- * bootstrap CI as normal, `CI.low ≈ effect − z·sd_Δ/√n`, so the smallest shippable
131
- * true effect is `MDE = deltaThreshold + z·sd_Δ/√n`. The paired-delta SD is unknown
132
- * before the candidate exists; we bound it by the zero-correlation case
133
- * `sd_Δ ≤ √2·sd_baseline` — a CONSERVATIVE (upper) MDE, which is the correct
134
- * direction for a warning. Pairing is per cell (`scenario:rep`), so reps multiply n.
135
- *
136
- * Standalone by design: feed it any baseline composites (a `gate:'none'` run, a
137
- * live-proof table) BEFORE budgeting the real search; `selfImprove` also attaches
138
- * it to every result and warns when the run was structurally unable to ship.
139
- */
140
- interface PowerPreflightOptions {
141
- /** Per-cell baseline composites on the HOLDOUT scenarios (one per scenario:rep cell). */
142
- baselineComposites: number[];
143
- /** Paired observations the budgeted comparison will produce
144
- * (holdout scenarios × reps). Defaults to `baselineComposites.length`. */
145
- pairedN?: number;
146
- /** The ship gate's effect-size threshold. Default 0.05 (defaultProductionGate). */
147
- deltaThreshold?: number;
148
- /** CI confidence the gate uses. Default 0.95. */
149
- confidence?: number;
150
- /** True when the holdout is scored by the SAME judge/scorer family as the gate
151
- * (selfImprove's default composition — one judge scores everything). Under a
152
- * shared channel, raising paired n reduces only the IDIOSYNCRATIC noise share;
153
- * systematic judge bias is untouched, so the MDE here is a lower bound and the
154
- * only full debiaser is an independent second scoring channel
155
- * (recursive-self-improvement S1c, closed form in EXP-023 P0). Default false. */
156
- sharedScorerChannel?: boolean;
157
- }
158
- interface PowerPreflight {
159
- /** Paired observations the comparison will have. */
160
- n: number;
161
- /** Baseline per-cell composite standard deviation (the variance the effect must beat). */
162
- sd: number;
163
- /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
164
- mde: number;
165
- /** Baseline holdout composite mean. */
166
- baselineMean: number;
167
- /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
168
- headroom: number;
169
- /** True when even the largest achievable effect (headroom) is below the MDE —
170
- * the run is structurally unable to ship regardless of proposal quality.
171
- * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
172
- underpowered: boolean;
173
- /** True when composites look [0,1]-scaled; headroom/underpowered are only
174
- * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
175
- scaleAssumed: boolean;
176
- deltaThreshold: number;
177
- confidence: number;
178
- /** Set when the holdout shares the gate's scoring channel: more cells cannot
179
- * buy back systematic judge bias — treat the MDE as a lower bound. */
180
- sharedChannelCaveat?: string;
181
- /** One actionable sentence for humans and logs. */
182
- recommendation: string;
183
- }
184
- /** Estimate the minimum detectable lift a paired-holdout improvement run can
185
- * ship at a given budget, from the baseline holdout composites — call it BEFORE
186
- * spending a search to learn whether the effect you are hunting is even
187
- * observable at this holdout size and worker variance. */
188
- declare function powerPreflight(opts: PowerPreflightOptions): PowerPreflight;
189
-
190
- /**
191
- * Promotion policy over the evidence VECTOR — the substrate's answer to "never
192
- * collapse the multi-objective promotion decision into one scalar." A
193
- * `defaultProductionGate` is one opinionated composition; this module factors
194
- * the decision into two reusable pieces so MANY policies can compete over the
195
- * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
196
- *
197
- * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
198
- * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
199
- * paretoPolicy(ev) // the default strategy
200
- * paretoSignificanceGate(options): Gate // bus + policy as a Gate
201
- *
202
- * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
203
- * potential gain source AND a safety floor (unlike `defaultProductionGate`,
204
- * where only `composite` can win and `criticalDimensions` are pure floors). A
205
- * candidate ships iff it weakly DOMINATES the baseline at the confidence level —
206
- * no objective credibly worse (CI floor breach) AND at least one objective
207
- * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
208
- * (NOT folded into hold: "gather more reps" and "reject" are different actions).
209
- *
210
- * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
211
- * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
212
- * constraints (compose with a budget gate via `composeGate`), not faked CIs.
213
- */
214
-
215
- /** Where an objective's per-cell scalar comes from. `composite` reads the
216
- * judge's composite; `dimension` reads a named per-dimension score. */
217
- type ObjectiveSource = {
218
- kind: 'composite';
219
- } | {
220
- kind: 'dimension';
221
- dimension: string;
222
- };
223
- interface PromotionObjective {
224
- /** Stable label used in reports + `contributingGates`. */
225
- name: string;
226
- source: ObjectiveSource;
227
- /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
228
- * the paired delta so a positive bootstrap always means "candidate better". */
229
- direction: Direction;
230
- /** The good-direction paired-delta CI lower bound must EXCEED this to count
231
- * as a significant gain on this axis. Interpreted in the judge's native
232
- * scale. Default 0 (⇒ "confidently better"). */
233
- gainThreshold?: number;
234
- /** A floor breach (regression) is declared when the good-direction CI lower
235
- * bound is below −floorTolerance. When omitted it auto-scales off observed
236
- * magnitudes (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
237
- floorTolerance?: number;
238
- }
239
- /** Per-axis verdict from the good-direction paired bootstrap. */
240
- type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
241
- interface AxisEvidence {
242
- name: string;
243
- source: ObjectiveSource;
244
- direction: Direction;
245
- /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
246
- * a positive value means the candidate is better on this axis. */
247
- bootstrap: PairedBootstrapResult;
248
- /** Paired observations contributing to this axis. */
249
- n: number;
250
- gainThreshold: number;
251
- floorTolerance: number;
252
- verdict: AxisVerdict;
253
- }
254
- interface EvidenceVector {
255
- /** One entry per objective — NOTHING averaged across axes. */
256
- axes: AxisEvidence[];
257
- /** Smallest paired n across axes that produced observations — the binding
258
- * evidence-sufficiency constraint. 0 when no axis produced observations. */
259
- minN: number;
260
- /** Aggregate per-side cost from the gate context (a constraint input, not a
261
- * CI axis — see the module header). */
262
- cost: {
263
- candidate: number;
264
- baseline: number;
265
- };
266
- }
267
- /** A promotion strategy: a pure function from the evidence vector to a verdict.
268
- * Many policies can run over the same `EvidenceVector` and disagree — that's
269
- * the point (competing strategies, shared evidence). */
270
- type PromotionPolicy = (ev: EvidenceVector) => GateResult;
271
- interface BuildEvidenceVectorOptions {
272
- /** Minimum paired observations before an axis can claim significance; below
273
- * it the axis is `few_runs`. Default 3. */
274
- minProductiveRuns?: number;
275
- /** Confidence level for every axis bootstrap. Default 0.95. */
276
- confidence?: number;
277
- /** Bootstrap resamples. Default 2000. */
278
- resamples?: number;
279
- /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
280
- seed?: number;
281
- }
282
- /**
283
- * The Evidence Bus. For each objective, pair candidate vs baseline by full
284
- * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
285
- * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
286
- * a single source of truth governs pairing granularity + scale handling.
287
- */
288
- declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
289
- /**
290
- * The default strategy: symmetric multi-objective Pareto significance. Ship iff
291
- * the candidate weakly dominates the baseline at the confidence level — no axis
292
- * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
293
- * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
294
- * need_more_work. Statistically equivalent → hold (never ship noise).
295
- */
296
- declare const paretoPolicy: PromotionPolicy;
297
- interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
298
- /** The objective vector. Every axis is both a gain source and a safety floor. */
299
- objectives: PromotionObjective[];
300
- /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
301
- * to run a stricter/looser strategy over the SAME bus (competing policies). */
302
- policy?: PromotionPolicy;
303
- /** Override the gate name in reports. */
304
- name?: string;
305
- }
306
- /**
307
- * Wrap the bus + a policy as a `Gate`. Plugs into the existing
308
- * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
309
- * loop behavior is unchanged because consumers opt in by passing this gate.
310
- */
311
- declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
312
-
313
- /**
314
- * `runEval` — the simplest preset over `runCampaign`. No optimizer, no
315
- * gate, no auto-PR. Just: run scenarios through dispatch, score with
316
- * judges, return CampaignResult.
317
- *
318
- * The 80% case for consumers who want a scorecard, not an improvement loop.
319
- */
320
-
321
- interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
322
- runDir: string;
323
- }
324
- /**
325
- * Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
326
- */
327
- declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
328
-
329
- /**
330
- * `evolutionaryProposer` — adapts a stateless `Mutator` (population mutation:
331
- * GEPA / AxGEPA / reflective-mutation) into a `SurfaceProposer`. This is
332
- * the evolutionary strategy: each generation, mutate the current best surface
333
- * into N candidates, measure, select. No generation memory beyond the current
334
- * surface; the loop body handles ranking + promotion.
335
- *
336
- * The reflective alternative is agent-runtime's runtime proposer with a
337
- * `reflectiveGenerator` / `agenticGenerator`: it reasons over the report +
338
- * trace findings to propose targeted edits rather than blind mutations. Both
339
- * conform to `SurfaceProposer`; the improvement loop is identical either way.
340
- */
341
-
342
- interface EvolutionaryProposerOptions<TFindings = unknown> {
343
- mutator: Mutator<TFindings>;
344
- /** External findings fed to the mutator each generation. Default: []. */
345
- findings?: TFindings[];
346
- }
347
- /**
348
- * Wrap a stateless `Mutator` (GEPA, AxGEPA, reflective-mutation) as a `SurfaceProposer` that mutates the current best surface into N candidates each generation.
349
- */
350
- declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryProposerOptions<TFindings>): SurfaceProposer<TFindings>;
351
-
352
- /**
353
- * Loop provenance — the durable, queryable record of WHAT a self-improvement
354
- * loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
355
- * an eval-run to the underlying candidate→cell→gate→promote chain.
356
- *
357
- * Two artifacts, one source of truth:
358
- *
359
- * 1. `LoopProvenanceRecord` — a structured JSON record capturing every
360
- * candidate (surfaceHash + label + rationale + structured cause), its measured composite,
361
- * the gate decision + reasons + delta, the held-out lift, the explicit
362
- * baseline→candidate diff, and BACKEND PROVENANCE (the
363
- * `assertRealBackend` verdict + worker call count + model). This is the
364
- * ingestable audit artifact: the +lift recomputes from it, the "because
365
- * Z" rationale survives in it, and a stub backend is detectable from it.
366
- *
367
- * 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
368
- * `TraceSpanEvent`s, pivoted on the substrate's standard
369
- * `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
370
- * `tangle.generation` attributes (the same pivots `/adapters/otel`
371
- * reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
372
- * not just the `cost.*` spans `runCampaign` already emits per cell.
373
- *
374
- * The record is built from the substrate's own loop result + the per-call
375
- * `RunRecord`s the worker emitted — no new measurement, no recomputation that
376
- * could drift from what the gate actually saw.
377
- */
378
-
379
- interface LoopProvenanceCandidate {
380
- /** Generation index this candidate was proposed in. */
381
- generation: number;
382
- /** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
383
- surfaceHash: string;
384
- /** Full sha256 content hash — byte-identical-verifiable. */
385
- contentHash: string;
386
- /** Proposer label, when the proposer returned a `ProposedCandidate`. */
387
- label?: string;
388
- /** Proposer rationale — the "because Z". When the proposer returned a bare
389
- * surface (blind mutator) this is absent. */
390
- rationale?: string;
391
- /** Exact validated cause when the proposer emitted a structured record. */
392
- candidateRecord?: PolicyEditCandidateRecord;
393
- /** Exact complete incumbent this candidate mutated. */
394
- parentSurfaceHash: string;
395
- /** Search-split composite of the exact parent. */
396
- parentComposite: number;
397
- /** Search-split composite change relative to the exact parent. */
398
- observedDeltaFromParent?: number;
399
- /** Whether the candidate completed every designed cell and could be selected. */
400
- eligibleForPromotion: boolean;
401
- /** Designed-denominator receipt retained even for incomplete candidates. */
402
- coverage: NonNullable<GenerationCandidate['coverage']>;
403
- /** Mean composite this candidate scored on the search split. */
404
- composite: number;
405
- /** Whether this candidate was promoted out of its generation. */
406
- promoted: boolean;
407
- }
408
- interface LoopProvenanceBackend {
409
- /** `assertRealBackend`-grade verdict over the worker call records. */
410
- verdict: 'real' | 'mixed' | 'stub';
411
- /** Number of worker LLM calls captured (the audit's "worker call count"). */
412
- workerCallCount: number;
413
- /** Distinct model ids observed across worker calls. */
414
- models: string[];
415
- totalInputTokens: number;
416
- totalOutputTokens: number;
417
- totalCostUsd: number;
418
- }
419
- /**
420
- * The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
421
- * ADDS the rationale + the explicit baseline→candidate diff (both omitted from
422
- * the bare hosted event) + backend provenance.
423
- */
424
- interface LoopProvenanceRecord {
425
- schema: 'tangle.loop-provenance.v3';
426
- runId: string;
427
- runDir: string;
428
- timestamp: string;
429
- /** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
430
- baselineContentHash: string;
431
- winnerContentHash: string;
432
- /** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
433
- winnerLabel?: string;
434
- winnerRationale?: string;
435
- /** The explicit baseline→winner unified diff the gate decided on. */
436
- diff: string;
437
- /** Every candidate across every generation, with its rationale and structured cause. */
438
- candidates: LoopProvenanceCandidate[];
439
- /** Baseline composite on the search split that generated the candidates. */
440
- baselineSearchComposite: number;
441
- /** The gate verdict — decision + reasons + contributing gates + delta. */
442
- gate: {
443
- decision: GateDecision;
444
- reasons: string[];
445
- delta?: number;
446
- contributingGates: Array<{
447
- name: string;
448
- passed: boolean;
449
- }>;
450
- };
451
- /** baseline-on-holdout composite mean. */
452
- baselineHoldoutComposite: number;
453
- /** winner-on-holdout composite mean. */
454
- winnerHoldoutComposite: number;
455
- /** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. */
456
- heldOutLift: number;
457
- /** Backend provenance: stub-vs-real verdict + worker call count + models. */
458
- backend: LoopProvenanceBackend;
459
- totalCostUsd: number;
460
- totalDurationMs: number;
461
- }
462
- interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
463
- runId: string;
464
- runDir: string;
465
- timestamp: string;
466
- baselineSurface: MutableSurface;
467
- winnerSurface: MutableSurface;
468
- winnerLabel?: string;
469
- winnerRationale?: string;
470
- diff: string;
471
- /** Baseline composite on the search split, distinct from holdout scoring. */
472
- baselineSearchComposite: number;
473
- /** Per-generation candidate records straight off the loop result. */
474
- generations: Array<{
475
- generationIndex: number;
476
- candidates: GenerationCandidate[];
477
- promoted: string[];
478
- /** Surfaces measured this generation, keyed by surface hash so the content
479
- * hash can be computed and the loop identity rechecked from real bytes. */
480
- surfaces: Array<{
481
- surfaceHash: string;
482
- surface: MutableSurface;
483
- }>;
484
- }>;
485
- gate: GateResult;
486
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
487
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
488
- /** Worker call records — the source for backend provenance. */
489
- workerRecords: ReadonlyArray<RunRecord>;
490
- totalCostUsd: number;
491
- totalDurationMs: number;
492
- }
493
- /** Build the durable provenance record from a completed loop result. */
494
- declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario>(args: BuildLoopProvenanceArgs<TArtifact, TScenario>): LoopProvenanceRecord;
495
- /**
496
- * Build the loop's OTLP-ingestable spans from a provenance record. One root
497
- * span per loop (`tangle.runId`), one span per generation, one span per
498
- * candidate (carrying its surfaceHash + label), and one span for the gate
499
- * decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
500
- * the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
501
- * reads, so the hosted collector reconstructs the full tree.
502
- *
503
- * Times are synthesized monotonically off a single base so the span tree is
504
- * orderable; the substrate does not retain per-candidate wall-clock starts.
505
- */
506
- declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
507
- baseTimeMs?: number;
508
- }): TraceSpanEvent[];
509
- /** Canonical durable paths under the run dir. */
510
- declare function provenanceRecordPath(runDir: string): string;
511
- /**
512
- * Canonical path for the durable OTLP spans JSONL file under a loop run directory.
513
- */
514
- declare function provenanceSpansPath(runDir: string): string;
515
- interface EmitLoopProvenanceResult {
516
- record: LoopProvenanceRecord;
517
- spans: TraceSpanEvent[];
518
- /** Absolute paths the record + spans were written to, when storage persists. */
519
- recordPath: string;
520
- spansPath: string;
521
- }
522
- interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends BuildLoopProvenanceArgs<TArtifact, TScenario> {
523
- /** Storage the record + spans are written through. */
524
- storage: CampaignStorage;
525
- /** When set, the spans are also shipped to the hosted `/v1/ingest/traces`
526
- * endpoint so the collector receives the full loop, not just `cost.*`. */
527
- hostedClient?: HostedClient;
528
- }
529
- /**
530
- * Build the provenance record + OTel spans and persist them durably under the
531
- * run dir (and ship spans to a hosted collector when one is wired). Returns
532
- * both artifacts so the caller can assert on / re-derive from them.
533
- *
534
- * Fail-loud: the durable write throws on storage failure (a swallowed write is
535
- * exactly the "emitted but lost" failure this closes). The hosted span ship is
536
- * the one best-effort leg — its failure is logged, not thrown, so an offline
537
- * collector never fails the loop (the durable artifact is the source of truth).
538
- */
539
- declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
540
-
541
- export { type AxisEvidence as A, type BuildEvidenceVectorOptions as B, type DefaultProductionGateOptions as D, type EvidenceVector as E, type HeldOutGateOptions as H, type LoopProvenanceRecord as L, type ObjectiveSource as O, type PowerPreflight as P, type RunEvalOptions as R, type AxisVerdict as a, type EvolutionaryProposerOptions as b, type ParetoSignificanceGateOptions as c, type PromotionObjective as d, type PromotionPolicy as e, buildEvidenceVector as f, composeGate as g, defaultProductionGate as h, evolutionaryProposer as i, heldOutGate as j, paretoSignificanceGate as k, type BuildLoopProvenanceArgs as l, type EmitLoopProvenanceArgs as m, type EmitLoopProvenanceResult as n, type LoopProvenanceBackend as o, paretoPolicy as p, type LoopProvenanceCandidate as q, runEval as r, type PowerPreflightOptions as s, buildLoopProvenanceRecord as t, emitLoopProvenance as u, loopProvenanceSpans as v, powerPreflight as w, provenanceRecordPath as x, provenanceSpansPath as y };
@@ -1,35 +0,0 @@
1
- import { L as LlmSpan, T as ToolSpan, J as JudgeSpan, R as Run, F as FailureClass } from './schema-B3Q3l9Z_.js';
2
- import { T as TraceStore } from './store-DGqD0Pyo.js';
3
-
4
- /**
5
- * Typed query helpers over TraceStore.
6
- *
7
- * Not a full SQL engine — a minimal, composable set of operators that
8
- * cover the canned-pipeline use cases. For ad-hoc analytics, persist to
9
- * NDJSON and point DuckDB at it; the schema is stable so external SQL
10
- * tooling works out of the box.
11
- */
12
-
13
- declare function runsForScenario(store: TraceStore, scenarioId: string): Promise<Run[]>;
14
- declare function llmSpans(store: TraceStore, runId?: string): Promise<LlmSpan[]>;
15
- declare function toolSpans(store: TraceStore, runId?: string, toolName?: string): Promise<ToolSpan[]>;
16
- /** Query judge-kind spans from the trace store, optionally scoped to a single run. */
17
- declare function judgeSpans(store: TraceStore, runId?: string): Promise<JudgeSpan[]>;
18
- /** Group spans by any key selector. */
19
- declare function groupBy<T, K extends string | number>(items: T[], key: (t: T) => K): Map<K, T[]>;
20
- /** Hash tool arguments to an orderless-key-stable string for de-duplication. */
21
- declare function argHash(args: unknown): string;
22
- /** Whether argument-based comparisons are valid for this tool call. */
23
- declare function hasCapturedToolArgs(span: ToolSpan): boolean;
24
- /** Sum an LLM-span array into aggregate token + cost. */
25
- declare function aggregateLlm(spans: LlmSpan[]): {
26
- inputTokens: number;
27
- outputTokens: number;
28
- cachedTokens: number;
29
- reasoningTokens: number;
30
- costUsd: number;
31
- };
32
- /** Pick the outcome's failure class when present, else derive 'success' from run status. */
33
- declare function runFailureClass(run: Run): FailureClass;
34
-
35
- export { aggregateLlm as a, argHash as b, runsForScenario as c, groupBy as g, hasCapturedToolArgs as h, judgeSpans as j, llmSpans as l, runFailureClass as r, toolSpans as t };