@tangle-network/agent-eval 0.94.0 → 0.95.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README.md +44 -30
  3. package/dist/adapters/http.d.ts +8 -7
  4. package/dist/adapters/http.js.map +1 -1
  5. package/dist/adapters/langchain.d.ts +3 -2
  6. package/dist/adapters/otel.d.ts +5 -4
  7. package/dist/analyst/index.d.ts +11 -31
  8. package/dist/analyst/index.js +5 -65
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
  11. package/dist/belief-state/index.d.ts +4 -3
  12. package/dist/benchmarks/index.d.ts +3 -2
  13. package/dist/campaign/index.d.ts +727 -616
  14. package/dist/campaign/index.js +1863 -1316
  15. package/dist/campaign/index.js.map +1 -1
  16. package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
  17. package/dist/chunk-2T4EZACH.js.map +1 -0
  18. package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
  19. package/dist/chunk-77T4STFI.js.map +1 -0
  20. package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
  21. package/dist/chunk-7QTQKIDD.js.map +1 -0
  22. package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
  23. package/dist/chunk-AQ5WQAIV.js.map +1 -0
  24. package/dist/chunk-DJWX3GVS.js +81 -0
  25. package/dist/chunk-DJWX3GVS.js.map +1 -0
  26. package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
  27. package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
  28. package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
  29. package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
  30. package/dist/chunk-KKWJD5E6.js.map +1 -0
  31. package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
  32. package/dist/chunk-LO6IOIJ2.js.map +1 -0
  33. package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
  34. package/dist/chunk-NZEQVRH5.js.map +1 -0
  35. package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
  36. package/dist/chunk-PSWWQXHF.js.map +1 -0
  37. package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
  38. package/dist/chunk-S4SYLDFX.js.map +1 -0
  39. package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
  40. package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
  41. package/dist/chunk-YBIGNSCZ.js.map +1 -0
  42. package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
  43. package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
  44. package/dist/contract/index.d.ts +91 -43
  45. package/dist/contract/index.js +127 -17
  46. package/dist/contract/index.js.map +1 -1
  47. package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
  48. package/dist/control.d.ts +3 -2
  49. package/dist/control.js +2 -2
  50. package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
  51. package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
  52. package/dist/diagnose.d.ts +4 -3
  53. package/dist/diagnose.js +1 -1
  54. package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
  55. package/dist/hosted/index.d.ts +5 -4
  56. package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
  57. package/dist/index.d.ts +76 -81
  58. package/dist/index.js +66 -31
  59. package/dist/index.js.map +1 -1
  60. package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
  61. package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
  62. package/dist/matrix/index.d.ts +1 -1
  63. package/dist/meta-eval/index.d.ts +3 -2
  64. package/dist/multishot/index.d.ts +4 -4
  65. package/dist/multishot/index.js.map +1 -1
  66. package/dist/openapi.json +1 -1
  67. package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
  68. package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
  69. package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
  70. package/dist/reporting.d.ts +5 -4
  71. package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
  72. package/dist/rl.d.ts +516 -515
  73. package/dist/rl.js +612 -612
  74. package/dist/rl.js.map +1 -1
  75. package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
  76. package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
  77. package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
  78. package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
  80. package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
  81. package/dist/testing-C21CHsq2.d.ts +20 -0
  82. package/dist/testing.d.ts +1 -0
  83. package/dist/testing.js +8 -0
  84. package/dist/testing.js.map +1 -0
  85. package/dist/traces.d.ts +26 -10
  86. package/dist/traces.js +41 -11
  87. package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
  88. package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
  89. package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
  90. package/dist/workflow/index.d.ts +5 -4
  91. package/dist/workflow/index.js +1 -1
  92. package/docs/campaign-proposers.md +170 -0
  93. package/docs/concepts.md +8 -4
  94. package/docs/customer-journeys.md +15 -13
  95. package/docs/design/loop-taxonomy.md +34 -66
  96. package/docs/distributed-driver.md +14 -14
  97. package/docs/feature-guide.md +1 -1
  98. package/docs/hosted-ingest-spec.md +2 -3
  99. package/docs/multi-shot-optimization.md +8 -8
  100. package/docs/product-eval-adoption.md +1 -1
  101. package/docs/self-improvement-map.md +33 -29
  102. package/package.json +8 -14
  103. package/dist/chunk-2K6UUZ7P.js.map +0 -1
  104. package/dist/chunk-CTBHKLEU.js.map +0 -1
  105. package/dist/chunk-E4GH6USR.js.map +0 -1
  106. package/dist/chunk-EGPMSBEZ.js.map +0 -1
  107. package/dist/chunk-KWRRMR3J.js.map +0 -1
  108. package/dist/chunk-MIFZUPEK.js.map +0 -1
  109. package/dist/chunk-MPQWFX6Y.js.map +0 -1
  110. package/dist/chunk-Q5LIB7BC.js.map +0 -1
  111. package/dist/chunk-QMUEXQJS.js.map +0 -1
  112. package/dist/chunk-SD2YFWQQ.js.map +0 -1
  113. package/docs/design/external-agent-wedge.md +0 -89
  114. package/docs/design/phase-d-rfc.md +0 -125
  115. package/docs/design/phase4-consumer-migration.md +0 -70
  116. package/docs/design/primitives-integration-spec.md +0 -393
  117. package/docs/design/product-self-improvement-loop.md +0 -146
  118. package/docs/design/self-improvement-engine.md +0 -140
  119. package/docs/design/self-improvement-protocol.md +0 -223
  120. package/docs/design/self-improvement-roadmap.md +0 -106
  121. package/docs/design/substrate-gaps.md +0 -118
  122. package/docs/phase-b-pairing-kit.md +0 -188
  123. package/docs/phase-b-runbook.md +0 -176
  124. package/docs/pilot/README.md +0 -62
  125. package/docs/pilot/customer-checklist.md +0 -90
  126. package/docs/pilot/integration-foreign-stack.md +0 -296
  127. package/docs/pilot/integration-tangle-stack.md +0 -248
  128. package/docs/pilot/one-pager.md +0 -161
  129. package/docs/pilot/sample-insight-report.json +0 -172
  130. package/docs/quickstart-external.md +0 -229
  131. package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
  132. package/docs/research/research-roadmap.md +0 -205
  133. package/docs/specs/driver-honest-spec.md +0 -251
  134. package/docs/specs/hermes-self-improvement-audit.md +0 -93
  135. package/docs/specs/profile-versioning.md +0 -291
  136. package/docs/three-package-architecture.md +0 -168
  137. /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
  138. /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
  139. /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
  140. /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
package/dist/rl.d.ts CHANGED
@@ -1,26 +1,246 @@
1
- export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
2
- import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-B8A4BDR3.js';
3
- export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-B8A4BDR3.js';
4
- import { g as CampaignResult } from './types-BU-7W85F.js';
5
- import { a as RunSplitTag, R as RunRecord } from './run-record-e7vj1uZQ.js';
6
- import { a as VerificationReport } from './multi-layer-verifier-DUZXrPDA.js';
1
+ import { R as RunRecord, a as RunSplitTag } from './run-record-CP2ObebC.js';
7
2
  export { A as AdversarialMutation, a as AdversarialScenario, b as AdversarialSearchOptions, c as AdversarialSearchReport, d as adversarialScenarioSearch } from './adversarial-DIVcDoI_.js';
3
+ import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-D4YW9UoJ.js';
4
+ export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-D4YW9UoJ.js';
5
+ export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
8
6
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
9
7
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
10
- import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-Cy_W-hWZ.js';
11
- import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-B0C2_fVO.js';
12
- export { r as runEvalCampaign } from './researcher-B0C2_fVO.js';
8
+ import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-C2hDKM8Z.js';
9
+ import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-Jr8ME1dZ.js';
10
+ export { r as runEvalCampaign } from './researcher-Jr8ME1dZ.js';
11
+ import { a as VerificationReport } from './multi-layer-verifier-DUZXrPDA.js';
13
12
  import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
13
+ import { C as CampaignResult } from './types-DQRY8ZT-.js';
14
+ import '@tangle-network/agent-interface';
15
+ import './errors-CzMUYo7b.js';
14
16
  import './schema-m0gsnbt3.js';
15
17
  import './store-BcFXE6LG.js';
16
- import './errors-CzMUYo7b.js';
17
- import './verdict-C9MlYujm.js';
18
18
  import './llm-client-Bj7g0rqu.js';
19
19
  import './raw-provider-sink-C46HDghv.js';
20
- import './summary-report-BDOFevaT.js';
20
+ import './summary-report-CInXwsza.js';
21
21
  import './failure-cluster-DH9Flgcf.js';
22
22
  import './emitter-C2rqGH_l.js';
23
23
  import './integrity-D2t12mMw.js';
24
+ import './verdict-C9MlYujm.js';
25
+
26
+ /**
27
+ * Adaptive curriculum / active scenario selection.
28
+ *
29
+ * Fixed scenario sets waste sample budget on cells the policy already
30
+ * passes (no information left) and cells the policy never passes (no
31
+ * gradient available either). Active learning over scenarios fixes this
32
+ * by allocating the next sample budget to cells where the policy's
33
+ * outcome is *uncertain* — those carry the most decision-relevant signal.
34
+ *
35
+ * This module ships two complementary strategies:
36
+ *
37
+ * 1. **Variance-based** — score each (variant, scenario) cell by the
38
+ * empirical variance of past observations. Allocate next-round budget
39
+ * proportional to variance. Standard active-learning-by-uncertainty
40
+ * heuristic; works well when the policy is non-deterministic and
41
+ * cells differ in observation noise.
42
+ *
43
+ * 2. **Bandit-based (Thompson sampling)** — model each (variant,
44
+ * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
45
+ * cells whose posterior mean is closest to the per-scenario decision
46
+ * threshold. The right primitive when scenarios are
47
+ * "pass/fail" rather than continuous, and when promotion gates fire
48
+ * at a known threshold (e.g., 0.5).
49
+ *
50
+ * The output is a *next-round budget allocation* — a list of (variant,
51
+ * scenario, count) triples. The consumer's matrix runner consumes the
52
+ * allocation, runs those cells, feeds the new observations back. Loop.
53
+ *
54
+ * Out of scope (deliberate): scenario *generation* — that's the
55
+ * adversarial primitive's job. This module allocates over an existing
56
+ * scenario pool.
57
+ */
58
+
59
+ interface CellObservation {
60
+ variantId: string;
61
+ scenarioId: string;
62
+ /** Observed score in [0, 1]. */
63
+ score: number;
64
+ /** For Bernoulli arms — derive from the score with a threshold if needed. */
65
+ pass?: boolean;
66
+ }
67
+ interface CurriculumAllocation {
68
+ variantId: string;
69
+ scenarioId: string;
70
+ /** How many additional reps to run on this cell. */
71
+ count: number;
72
+ /** Strategy-specific reason for the allocation. */
73
+ reason: string;
74
+ }
75
+ interface VarianceCurriculumOptions {
76
+ /** Total reps to allocate across all cells. */
77
+ budget: number;
78
+ /**
79
+ * Smoothing prior on variance — keeps the allocator from concentrating
80
+ * on a cell with one observation just because its 1-sample variance is
81
+ * 0. Default 0.05.
82
+ */
83
+ variancePrior?: number;
84
+ /**
85
+ * Minimum reps per cell — even when the variance estimate is low, give
86
+ * every cell at least this many. Default 1.
87
+ */
88
+ floorPerCell?: number;
89
+ }
90
+ /**
91
+ * Variance-proportional allocation. For each cell, estimate variance from
92
+ * past observations + a prior, then allocate the budget proportional to
93
+ * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule
94
+ * (Neyman 1934) that balances "explore noisy cells" with "explore
95
+ * under-sampled cells."
96
+ */
97
+ declare function varianceBasedCurriculum(observations: CellObservation[], candidateCells: Array<{
98
+ variantId: string;
99
+ scenarioId: string;
100
+ }>, opts: VarianceCurriculumOptions): CurriculumAllocation[];
101
+ interface ThompsonCurriculumOptions {
102
+ budget: number;
103
+ /**
104
+ * The per-scenario decision threshold. Cells whose posterior mean is
105
+ * closest to this get the most budget — that's where the next observation
106
+ * has the highest information value for the gate decision. Default 0.5.
107
+ */
108
+ decisionThreshold?: number;
109
+ /** Beta prior parameters. Default α=β=1 (uniform). */
110
+ priorAlpha?: number;
111
+ priorBeta?: number;
112
+ /** Seed the Thompson sampler. Default unset (Math.random). */
113
+ seed?: number;
114
+ }
115
+ /**
116
+ * Thompson-sampling-style allocation for pass/fail cells. For each cell:
117
+ *
118
+ * - Maintain Beta(α + passes, β + failures) posterior on pass-rate
119
+ * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):
120
+ * cells whose sampled posterior straddles the decision boundary get
121
+ * the most weight; cells already clearly above or below get less.
122
+ *
123
+ * This is the right primitive when promotion gates fire at a known
124
+ * threshold and you want to sharpen the posterior near the boundary.
125
+ */
126
+ declare function thompsonCurriculum(observations: CellObservation[], candidateCells: Array<{
127
+ variantId: string;
128
+ scenarioId: string;
129
+ }>, opts: ThompsonCurriculumOptions): CurriculumAllocation[];
130
+ /** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
131
+ declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
132
+ passThreshold?: number;
133
+ useHoldout?: boolean;
134
+ }): CellObservation[];
135
+
136
+ /**
137
+ * Sample-efficient adaptation evaluation.
138
+ *
139
+ * For foundation-model-based agents, the load-bearing capability isn't
140
+ * raw end-state performance — it's *how fast the agent reaches that
141
+ * performance from cold start*. The same model with a worse prompt that
142
+ * adapts in 5 demonstrations beats the same model with a better prompt
143
+ * that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
144
+ * reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
145
+ * in-context examples or fine-tune steps.
146
+ *
147
+ * This module ships:
148
+ *
149
+ * 1. `runAdaptationCurve` — given a runner that takes k demonstrations
150
+ * and returns a score, produce the (k, score) curve.
151
+ * 2. `compareAdaptationCurves` — paired comparison across two policies.
152
+ * Returns per-k delta with bootstrap CIs and an "area-under-curve"
153
+ * summary statistic.
154
+ * 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
155
+ * the policy reliably passes (≥ pass-rate threshold over reps).
156
+ *
157
+ * Use cases:
158
+ * - Compare two prompt designs that have similar end-state performance
159
+ * but different in-context efficiency.
160
+ * - Decide between fine-tuning and prompting based on adaptation cost.
161
+ * - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
162
+ */
163
+ interface AdaptationRunner<S> {
164
+ /**
165
+ * Runs the policy on `scenario` with `k` demonstrations. Returns a
166
+ * scalar score in [0, 1]. The runner is responsible for any caching;
167
+ * the harness calls it once per (scenario, k, rep) cell.
168
+ */
169
+ run(args: {
170
+ scenario: S;
171
+ k: number;
172
+ rep: number;
173
+ }): Promise<number>;
174
+ }
175
+ interface RunAdaptationCurveOptions<S> {
176
+ scenarios: S[];
177
+ /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
178
+ ks?: number[];
179
+ /** Reps per (scenario, k) cell. Default 3. */
180
+ reps?: number;
181
+ runner: AdaptationRunner<S>;
182
+ /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
183
+ passThreshold?: number;
184
+ }
185
+ interface AdaptationPoint {
186
+ k: number;
187
+ meanScore: number;
188
+ passRate: number;
189
+ std: number;
190
+ n: number;
191
+ /** Per-scenario means at this k. */
192
+ perScenario: Array<{
193
+ scenarioId: string;
194
+ meanScore: number;
195
+ passes: number;
196
+ total: number;
197
+ }>;
198
+ }
199
+ interface AdaptationCurve {
200
+ points: AdaptationPoint[];
201
+ /**
202
+ * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
203
+ * tested reaches it.
204
+ */
205
+ firstPassK: number | null;
206
+ /**
207
+ * Area under the (k, meanScore) curve, normalized by max-k. A
208
+ * single-number summary of "how well does this policy adapt from
209
+ * cold-start to fully-conditioned." Higher = better adapter.
210
+ */
211
+ adaptationArea: number;
212
+ }
213
+ declare function runAdaptationCurve<S extends {
214
+ scenarioId?: string;
215
+ }>(opts: RunAdaptationCurveOptions<S>): Promise<AdaptationCurve>;
216
+ interface CompareCurvesResult {
217
+ perK: Array<{
218
+ k: number;
219
+ deltaMean: number;
220
+ aLow: number;
221
+ aHigh: number;
222
+ bLow: number;
223
+ bHigh: number;
224
+ }>;
225
+ areaDelta: number;
226
+ firstPassKDelta: number | null;
227
+ /** Verdict: 'a_better' | 'b_better' | 'similar'. */
228
+ verdict: 'a_better' | 'b_better' | 'similar';
229
+ /** Rationale, ready to render. */
230
+ rationale: string;
231
+ }
232
+ /**
233
+ * Paired comparison of two adaptation curves. Per-k deltas with 95%
234
+ * bootstrap CIs (constructed from each curve's `perScenario` per-k means
235
+ * — the bootstrap unit is the scenario, not the rep).
236
+ */
237
+ declare function compareAdaptationCurves(a: AdaptationCurve, b: AdaptationCurve, opts?: {
238
+ confidence?: number;
239
+ bootstrapResamples?: number;
240
+ seed?: number;
241
+ }): CompareCurvesResult;
242
+ /** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
243
+ declare function firstPassK(curve: AdaptationCurve, threshold?: number): number | null;
24
244
 
25
245
  /**
26
246
  * Test-time compute scaling curves.
@@ -267,173 +487,70 @@ declare function injectIrrelevantClause<S extends {
267
487
  }>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
268
488
 
269
489
  /**
270
- * Adapters: convert measurement outputs into the canonical `RunRecord[]`
271
- * artifact that `replayCache`, `pairedEvalueSequence`, and
272
- * `rubricPredictiveValidity` consume. Two sources:
273
- * - `campaignToRunRecords` — the campaign substrate's per-cell results
274
- * (the modern path: `runCampaign` / `runImprovementLoop` → records).
275
- * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
490
+ * `PredictiveValidityResearcher` concrete `Researcher` implementation
491
+ * that drives selection from outcome-anchored predictive validity.
276
492
  *
277
- * Adapters are thin and explicit — every mandatory `RunRecord` field comes
278
- * from a caller-supplied context (`commitSha`, `model`, `promptHash`,
279
- * `configHash`) plus the cell's runtime data. The validator still rejects
280
- * bare-alias model strings the caller snapshot-pins.
493
+ * Each method:
494
+ *
495
+ * - `inspectFailures(runs)` synthesizes failure modes from the
496
+ * bottom-quartile of `RunRecord`s on the configured proxy reward.
497
+ * - `proposeChange(failures)` — proposes steering changes that target
498
+ * the rubrics with the lowest predictive validity (decorative ones).
499
+ * Either reduce their weight in the composite, or recalibrate them.
500
+ * - `applyChange(changes, baseline)` — merges the proposed steering
501
+ * into the experiment plan.
502
+ * - `evaluateChange(plan)` — re-runs the predictive-validity check on
503
+ * the post-change runs and reports the delta.
504
+ *
505
+ * The result is a closed loop: the rubric weights drift toward the ones
506
+ * that actually predict deployment outcomes, automatically. Pair with
507
+ * `runRLCampaign` for the full auto-research story.
281
508
  */
282
509
 
283
- interface AdapterContext {
284
- /** Logical experiment id — typically the campaign or sweep identifier. */
285
- experimentId: string;
286
- /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
287
- model: string;
288
- /** Git SHA the harness was run from. */
289
- commitSha: string;
290
- /** Hash of the effective prompt sent to the model. */
291
- promptHash: string;
292
- /** Hash of the effective config (model, temperature, tools, judges, splits). */
293
- configHash: string;
294
- /** Default split tag. Default `'search'`. */
295
- splitTag?: RunSplitTag;
296
- /** Default cost in USD when the source doesn't record one. Default `0`. */
297
- defaultCostUsd?: number;
298
- }
299
- /**
300
- * Convert a `CampaignResult` into canonical `RunRecord[]` — one record per
301
- * scored cell. The cell's mean judge composite becomes the split score; every
302
- * judge dimension is carried through to `outcome.raw`. A cell that errored
303
- * becomes a record with `failureMode: 'cell_error'` (kept, not dropped — an
304
- * unscored cell is signal). `candidateId` identifies the measured surface
305
- * (defaults to the campaign manifest hash).
306
- */
307
- declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterContext & {
308
- candidateId?: string;
309
- }): RunRecord[];
310
- /**
311
- * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
312
- * `outcome.searchScore` (or `holdoutScore`) is `report.blendedScore`;
313
- * `outcome.raw` carries every layer's score + a pass indicator; `failureMode`
314
- * is the first failing layer's reason.
315
- */
316
- declare function verificationReportToRunRecord(report: VerificationReport, ctx: AdapterContext & {
317
- candidateId: string;
318
- scenarioId?: string;
319
- }, opts?: {
320
- runId?: string;
321
- }): RunRecord;
322
-
323
- /**
324
- * Bradley-Terry / Elo tournament evaluation.
325
- *
326
- * For multi-candidate sweeps, comparing every candidate's score against
327
- * a fixed comparator wastes information — the comparator becomes a high-
328
- * variance reference and rank flips between near-tied middle-rank
329
- * candidates are dominated by noise. Pairwise tournaments fix this:
330
- * every (i, j) pair contributes a comparison to a Bradley-Terry MLE that
331
- * estimates each candidate's strength on a unified scale.
332
- *
333
- * For online updating (rolling campaigns where new candidates arrive
334
- * over time), we also ship classical Elo with configurable K-factor.
335
- *
336
- * References:
337
- * - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete
338
- * block designs. Biometrika, 39(3/4), 324–345.
339
- * - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry
340
- * models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm
341
- * used here.)
342
- * - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.
343
- *
344
- * This is a useful primitive because most LLM-eval communities (Chatbot
345
- * Arena, AlpacaEval, ELO-style ablation) have converged on pairwise
346
- * tournament eval as the most sample-efficient and most rank-stable
347
- * method when you have many candidates.
348
- */
349
- interface PairwiseOutcome {
350
- /** Winner candidate id. */
351
- winner: string;
352
- /** Loser candidate id. */
353
- loser: string;
354
- /**
355
- * Optional draw flag. When true, both candidates get half-credit
356
- * (Bradley-Terry handles draws as half-wins for each side).
357
- */
358
- draw?: boolean;
510
+ interface PredictiveValidityResearcherOptions {
511
+ outcomes: OutcomeStore;
512
+ outcomeMetrics: string[];
513
+ /** Score threshold below which a run counts as a "failure." Default 0.5. */
514
+ failureThreshold?: number;
515
+ /** Spearman bucket below which a rubric is "decorative." Default 0.4. */
516
+ decorativeThreshold?: number;
517
+ /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
518
+ steeringNamespace?: string;
519
+ /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
520
+ rubrics?: string[];
359
521
  /**
360
- * Optional weightuseful if some pairwise comparisons are stronger
361
- * signals than others (e.g. a paired test with a wider score gap is
362
- * a more confident comparison). Default 1.
522
+ * Snapshot stash hook called with the most recent predictive-validity
523
+ * report. Useful when a downstream system wants to log rubric drift over
524
+ * time. Default no-op.
363
525
  */
364
- weight?: number;
365
- }
366
- interface BradleyTerryRating {
367
- candidateId: string;
368
- /** Latent strength θ ≥ 0 from the BT MLE. */
369
- strength: number;
370
- /** Log-strength = log(θ) — interpretable on a linear scale. */
371
- logStrength: number;
372
- /** Number of pairwise comparisons this candidate appears in. */
373
- n: number;
374
- /** Win count (+ 0.5 per draw). */
375
- wins: number;
376
- }
377
- interface BradleyTerryFit {
378
- ratings: BradleyTerryRating[];
379
- /** Iterations of the MM algorithm before convergence. */
380
- iterations: number;
381
- /** Final maximum |θ_new - θ_old| / θ_old. */
382
- finalDelta: number;
383
- converged: boolean;
384
- }
385
- /**
386
- * Bradley-Terry MLE via Hunter's MM algorithm.
387
- *
388
- * Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
389
- * where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
390
- *
391
- * Returns log-strengths normalized so the smallest is 0 (any constant
392
- * offset is unobservable in BT — only differences are identified).
393
- */
394
- declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
395
- tolerance?: number;
396
- maxIterations?: number;
397
- smoothing?: number;
398
- }): BradleyTerryFit;
399
- /**
400
- * Online Elo updates. Use when comparisons arrive over time and you want
401
- * a running rating without re-fitting the full BT MLE on every update.
402
- *
403
- * Initialize ratings to `defaultRating` (1500 by default). Each call to
404
- * `applyEloUpdate` mutates the map in place and returns the deltas so
405
- * the caller can log per-comparison rating changes.
406
- */
407
- interface EloOptions {
408
- /** Default rating for unseen candidates. Default 1500. */
409
- defaultRating?: number;
410
- /** K-factor controls the step size. Default 32 (FIDE-ish). */
411
- kFactor?: number;
526
+ onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>;
412
527
  }
413
- declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseOutcome, opts?: EloOptions): {
414
- winnerDelta: number;
415
- loserDelta: number;
416
- };
417
528
  /**
418
- * Build pairwise outcomes from the campaign artifact: for every scenario
419
- * shared by two candidates, the higher-scoring run wins. Useful when you
420
- * want a tournament view of an existing campaign without an additional
421
- * pairwise judge call.
529
+ * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
530
+ * rubrics that don't predict deployment outcomes don't earn weight.
422
531
  */
423
- interface BuildPairwiseFromCampaignInput {
424
- runs: Array<{
425
- candidateId: string;
426
- /** Stable identifier for the matching unit (typically scenarioId). */
427
- matchKey: string;
428
- score: number;
429
- }>;
532
+ declare class PredictiveValidityResearcher implements Researcher {
533
+ private opts;
534
+ private lastReport;
535
+ constructor(opts: PredictiveValidityResearcherOptions);
536
+ inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
537
+ proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
538
+ applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
539
+ evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
430
540
  /**
431
- * Tied-score margin. Below this, the comparison is a draw. Default 0
432
- * (no ties).
541
+ * Run the predictive-validity check explicitly against a fresh RunRecord
542
+ * set. Updates the researcher's cached report so subsequent
543
+ * `proposeChange` calls have evidence to draw from.
433
544
  */
434
- drawMargin?: number;
545
+ runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport>;
546
+ /**
547
+ * Force-feed a predictive-validity report into the researcher state —
548
+ * useful when the consumer ran the report out-of-band and wants the
549
+ * researcher's later proposals informed by it.
550
+ */
551
+ setReport(report: RubricPredictiveValidityReport): void;
552
+ getLastReport(): RubricPredictiveValidityReport | null;
435
553
  }
436
- declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
437
554
 
438
555
  /**
439
556
  * Verifiable reward channel.
@@ -487,361 +604,76 @@ interface VerifiableReward {
487
604
  /** The layer / judge id that produced the signal, for provenance. */
488
605
  origin: string;
489
606
  /**
490
- * Per-source contribution to `value`, keyed by layer/judge id. Single-source
491
- * rewards carry one entry (`{ [origin]: value }`); composite rewards carry
492
- * every contributing layer's score — the anti-scalar-collapse surface RL
493
- * consumers weight per-source instead of trusting one blended number.
494
- */
495
- components: Record<string, number>;
496
- /**
497
- * @deprecated Read `components` for per-source reward values. Kept for
498
- * published-API compatibility: single-source rewards carry the layer's
499
- * diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
500
- * the same per-layer scores `components` now holds.
501
- */
502
- breakdown?: Record<string, number>;
503
- }
504
- interface VerifiableRewardExtractionOptions {
505
- /**
506
- * Which layers count as deterministic-reward sources. The verifier doesn't
507
- * tag layers as "this is verifiable"; the caller declares it via this list
508
- * (or via the layer name → source mapping). Default treats common names
509
- * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
510
- * `sandbox`) as deterministic.
511
- */
512
- deterministicLayers?: string[];
513
- /**
514
- * Map layer name → reward source. Defaults to a sensible string-match.
515
- */
516
- sourceFor?: (layerName: string) => VerifiableRewardSource;
517
- /**
518
- * Whether to fall back to a probabilistic (judge) reward when no
519
- * deterministic layer produced a numeric score. Default `true`. Set to
520
- * `false` for "deterministic-only" training pipelines that should
521
- * discard runs without a verifiable signal.
522
- */
523
- fallbackToJudge?: boolean;
524
- /**
525
- * Default confidence for probabilistic (judge) rewards when the judge
526
- * doesn't report one. Default `0.7`.
527
- */
528
- judgeConfidenceFloor?: number;
529
- }
530
- /**
531
- * Extract a `VerifiableReward` from a `VerificationReport`.
532
- *
533
- * Strategy: prefer the deterministic layers (in order: test → compile →
534
- * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
535
- * true, return `null` if no signal qualifies. When multiple deterministic
536
- * layers contribute, return a `'composite'` source with a weighted blend.
537
- */
538
- declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
539
- /**
540
- * Extract verifiable rewards from `RunRecord[]` produced via the
541
- * `verificationReportToRunRecord` adapter (which encodes per-layer scores
542
- * in `outcome.raw['layer.<name>']`). For records that don't carry layer
543
- * scores, returns `null` for that record.
544
- *
545
- * This is the canonical bridge from "campaign-shaped artifacts" to
546
- * "RL-training-ready reward signals": every record that has a clean
547
- * verifiable reward becomes a training datum, every record that doesn't
548
- * gets filtered out (or kept with `'probabilistic'` determinism for
549
- * separate downstream handling).
550
- */
551
- declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
552
- runId: string;
553
- reward: VerifiableReward | null;
554
- }>;
555
- /** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
556
- declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
557
- run: RunRecord;
558
- reward: VerifiableReward;
559
- }>;
560
-
561
- /**
562
- * Adaptive curriculum / active scenario selection.
563
- *
564
- * Fixed scenario sets waste sample budget on cells the policy already
565
- * passes (no information left) and cells the policy never passes (no
566
- * gradient available either). Active learning over scenarios fixes this
567
- * by allocating the next sample budget to cells where the policy's
568
- * outcome is *uncertain* — those carry the most decision-relevant signal.
569
- *
570
- * This module ships two complementary strategies:
571
- *
572
- * 1. **Variance-based** — score each (variant, scenario) cell by the
573
- * empirical variance of past observations. Allocate next-round budget
574
- * proportional to variance. Standard active-learning-by-uncertainty
575
- * heuristic; works well when the policy is non-deterministic and
576
- * cells differ in observation noise.
577
- *
578
- * 2. **Bandit-based (Thompson sampling)** — model each (variant,
579
- * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
580
- * cells whose posterior mean is closest to the per-scenario decision
581
- * threshold. The right primitive when scenarios are
582
- * "pass/fail" rather than continuous, and when promotion gates fire
583
- * at a known threshold (e.g., 0.5).
584
- *
585
- * The output is a *next-round budget allocation* — a list of (variant,
586
- * scenario, count) triples. The consumer's matrix runner consumes the
587
- * allocation, runs those cells, feeds the new observations back. Loop.
588
- *
589
- * Out of scope (deliberate): scenario *generation* — that's the
590
- * adversarial primitive's job. This module allocates over an existing
591
- * scenario pool.
592
- */
593
-
594
- interface CellObservation {
595
- variantId: string;
596
- scenarioId: string;
597
- /** Observed score in [0, 1]. */
598
- score: number;
599
- /** For Bernoulli arms — derive from the score with a threshold if needed. */
600
- pass?: boolean;
601
- }
602
- interface CurriculumAllocation {
603
- variantId: string;
604
- scenarioId: string;
605
- /** How many additional reps to run on this cell. */
606
- count: number;
607
- /** Strategy-specific reason for the allocation. */
608
- reason: string;
609
- }
610
- interface VarianceCurriculumOptions {
611
- /** Total reps to allocate across all cells. */
612
- budget: number;
613
- /**
614
- * Smoothing prior on variance — keeps the allocator from concentrating
615
- * on a cell with one observation just because its 1-sample variance is
616
- * 0. Default 0.05.
617
- */
618
- variancePrior?: number;
619
- /**
620
- * Minimum reps per cell — even when the variance estimate is low, give
621
- * every cell at least this many. Default 1.
622
- */
623
- floorPerCell?: number;
624
- }
625
- /**
626
- * Variance-proportional allocation. For each cell, estimate variance from
627
- * past observations + a prior, then allocate the budget proportional to
628
- * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule
629
- * (Neyman 1934) that balances "explore noisy cells" with "explore
630
- * under-sampled cells."
631
- */
632
- declare function varianceBasedCurriculum(observations: CellObservation[], candidateCells: Array<{
633
- variantId: string;
634
- scenarioId: string;
635
- }>, opts: VarianceCurriculumOptions): CurriculumAllocation[];
636
- interface ThompsonCurriculumOptions {
637
- budget: number;
638
- /**
639
- * The per-scenario decision threshold. Cells whose posterior mean is
640
- * closest to this get the most budget — that's where the next observation
641
- * has the highest information value for the gate decision. Default 0.5.
642
- */
643
- decisionThreshold?: number;
644
- /** Beta prior parameters. Default α=β=1 (uniform). */
645
- priorAlpha?: number;
646
- priorBeta?: number;
647
- /** Seed the Thompson sampler. Default unset (Math.random). */
648
- seed?: number;
649
- }
650
- /**
651
- * Thompson-sampling-style allocation for pass/fail cells. For each cell:
652
- *
653
- * - Maintain Beta(α + passes, β + failures) posterior on pass-rate
654
- * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):
655
- * cells whose sampled posterior straddles the decision boundary get
656
- * the most weight; cells already clearly above or below get less.
657
- *
658
- * This is the right primitive when promotion gates fire at a known
659
- * threshold and you want to sharpen the posterior near the boundary.
660
- */
661
- declare function thompsonCurriculum(observations: CellObservation[], candidateCells: Array<{
662
- variantId: string;
663
- scenarioId: string;
664
- }>, opts: ThompsonCurriculumOptions): CurriculumAllocation[];
665
- /** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
666
- declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
667
- passThreshold?: number;
668
- useHoldout?: boolean;
669
- }): CellObservation[];
670
-
671
- /**
672
- * Sample-efficient adaptation evaluation.
673
- *
674
- * For foundation-model-based agents, the load-bearing capability isn't
675
- * raw end-state performance — it's *how fast the agent reaches that
676
- * performance from cold start*. The same model with a worse prompt that
677
- * adapts in 5 demonstrations beats the same model with a better prompt
678
- * that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
679
- * reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
680
- * in-context examples or fine-tune steps.
681
- *
682
- * This module ships:
683
- *
684
- * 1. `runAdaptationCurve` — given a runner that takes k demonstrations
685
- * and returns a score, produce the (k, score) curve.
686
- * 2. `compareAdaptationCurves` — paired comparison across two policies.
687
- * Returns per-k delta with bootstrap CIs and an "area-under-curve"
688
- * summary statistic.
689
- * 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
690
- * the policy reliably passes (≥ pass-rate threshold over reps).
691
- *
692
- * Use cases:
693
- * - Compare two prompt designs that have similar end-state performance
694
- * but different in-context efficiency.
695
- * - Decide between fine-tuning and prompting based on adaptation cost.
696
- * - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
697
- */
698
- interface AdaptationRunner<S> {
699
- /**
700
- * Runs the policy on `scenario` with `k` demonstrations. Returns a
701
- * scalar score in [0, 1]. The runner is responsible for any caching;
702
- * the harness calls it once per (scenario, k, rep) cell.
703
- */
704
- run(args: {
705
- scenario: S;
706
- k: number;
707
- rep: number;
708
- }): Promise<number>;
709
- }
710
- interface RunAdaptationCurveOptions<S> {
711
- scenarios: S[];
712
- /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
713
- ks?: number[];
714
- /** Reps per (scenario, k) cell. Default 3. */
715
- reps?: number;
716
- runner: AdaptationRunner<S>;
717
- /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
718
- passThreshold?: number;
719
- }
720
- interface AdaptationPoint {
721
- k: number;
722
- meanScore: number;
723
- passRate: number;
724
- std: number;
725
- n: number;
726
- /** Per-scenario means at this k. */
727
- perScenario: Array<{
728
- scenarioId: string;
729
- meanScore: number;
730
- passes: number;
731
- total: number;
732
- }>;
733
- }
734
- interface AdaptationCurve {
735
- points: AdaptationPoint[];
736
- /**
737
- * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
738
- * tested reaches it.
739
- */
740
- firstPassK: number | null;
741
- /**
742
- * Area under the (k, meanScore) curve, normalized by max-k. A
743
- * single-number summary of "how well does this policy adapt from
744
- * cold-start to fully-conditioned." Higher = better adapter.
745
- */
746
- adaptationArea: number;
747
- }
748
- declare function runAdaptationCurve<S extends {
749
- scenarioId?: string;
750
- }>(opts: RunAdaptationCurveOptions<S>): Promise<AdaptationCurve>;
751
- interface CompareCurvesResult {
752
- perK: Array<{
753
- k: number;
754
- deltaMean: number;
755
- aLow: number;
756
- aHigh: number;
757
- bLow: number;
758
- bHigh: number;
759
- }>;
760
- areaDelta: number;
761
- firstPassKDelta: number | null;
762
- /** Verdict: 'a_better' | 'b_better' | 'similar'. */
763
- verdict: 'a_better' | 'b_better' | 'similar';
764
- /** Rationale, ready to render. */
765
- rationale: string;
766
- }
767
- /**
768
- * Paired comparison of two adaptation curves. Per-k deltas with 95%
769
- * bootstrap CIs (constructed from each curve's `perScenario` per-k means
770
- * — the bootstrap unit is the scenario, not the rep).
771
- */
772
- declare function compareAdaptationCurves(a: AdaptationCurve, b: AdaptationCurve, opts?: {
773
- confidence?: number;
774
- bootstrapResamples?: number;
775
- seed?: number;
776
- }): CompareCurvesResult;
777
- /** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
778
- declare function firstPassK(curve: AdaptationCurve, threshold?: number): number | null;
779
-
780
- /**
781
- * `PredictiveValidityResearcher` — concrete `Researcher` implementation
782
- * that drives selection from outcome-anchored predictive validity.
783
- *
784
- * Each method:
785
- *
786
- * - `inspectFailures(runs)` — synthesizes failure modes from the
787
- * bottom-quartile of `RunRecord`s on the configured proxy reward.
788
- * - `proposeChange(failures)` — proposes steering changes that target
789
- * the rubrics with the lowest predictive validity (decorative ones).
790
- * Either reduce their weight in the composite, or recalibrate them.
791
- * - `applyChange(changes, baseline)` — merges the proposed steering
792
- * into the experiment plan.
793
- * - `evaluateChange(plan)` — re-runs the predictive-validity check on
794
- * the post-change runs and reports the delta.
795
- *
796
- * The result is a closed loop: the rubric weights drift toward the ones
797
- * that actually predict deployment outcomes, automatically. Pair with
798
- * `runRLCampaign` for the full auto-research story.
799
- */
800
-
801
- interface PredictiveValidityResearcherOptions {
802
- outcomes: OutcomeStore;
803
- outcomeMetrics: string[];
804
- /** Score threshold below which a run counts as a "failure." Default 0.5. */
805
- failureThreshold?: number;
806
- /** Spearman bucket below which a rubric is "decorative." Default 0.4. */
807
- decorativeThreshold?: number;
808
- /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
809
- steeringNamespace?: string;
810
- /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
811
- rubrics?: string[];
607
+ * Per-source contribution to `value`, keyed by layer/judge id. Single-source
608
+ * rewards carry one entry (`{ [origin]: value }`); composite rewards carry
609
+ * every contributing layer's score — the anti-scalar-collapse surface RL
610
+ * consumers weight per-source instead of trusting one blended number.
611
+ */
612
+ components: Record<string, number>;
812
613
  /**
813
- * Snapshot stash hook called with the most recent predictive-validity
814
- * report. Useful when a downstream system wants to log rubric drift over
815
- * time. Default no-op.
614
+ * @deprecated Read `components` for per-source reward values. Kept for
615
+ * published-API compatibility: single-source rewards carry the layer's
616
+ * diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
617
+ * the same per-layer scores `components` now holds.
816
618
  */
817
- onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>;
619
+ breakdown?: Record<string, number>;
818
620
  }
819
- /**
820
- * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
821
- * rubrics that don't predict deployment outcomes don't earn weight.
822
- */
823
- declare class PredictiveValidityResearcher implements Researcher {
824
- private opts;
825
- private lastReport;
826
- constructor(opts: PredictiveValidityResearcherOptions);
827
- inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
828
- proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
829
- applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
830
- evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
621
+ interface VerifiableRewardExtractionOptions {
831
622
  /**
832
- * Run the predictive-validity check explicitly against a fresh RunRecord
833
- * set. Updates the researcher's cached report so subsequent
834
- * `proposeChange` calls have evidence to draw from.
623
+ * Which layers count as deterministic-reward sources. The verifier doesn't
624
+ * tag layers as "this is verifiable"; the caller declares it via this list
625
+ * (or via the layer name source mapping). Default treats common names
626
+ * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
627
+ * `sandbox`) as deterministic.
835
628
  */
836
- runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport>;
629
+ deterministicLayers?: string[];
837
630
  /**
838
- * Force-feed a predictive-validity report into the researcher state
839
- * useful when the consumer ran the report out-of-band and wants the
840
- * researcher's later proposals informed by it.
631
+ * Map layer name reward source. Defaults to a sensible string-match.
841
632
  */
842
- setReport(report: RubricPredictiveValidityReport): void;
843
- getLastReport(): RubricPredictiveValidityReport | null;
633
+ sourceFor?: (layerName: string) => VerifiableRewardSource;
634
+ /**
635
+ * Whether to fall back to a probabilistic (judge) reward when no
636
+ * deterministic layer produced a numeric score. Default `true`. Set to
637
+ * `false` for "deterministic-only" training pipelines that should
638
+ * discard runs without a verifiable signal.
639
+ */
640
+ fallbackToJudge?: boolean;
641
+ /**
642
+ * Default confidence for probabilistic (judge) rewards when the judge
643
+ * doesn't report one. Default `0.7`.
644
+ */
645
+ judgeConfidenceFloor?: number;
844
646
  }
647
+ /**
648
+ * Extract a `VerifiableReward` from a `VerificationReport`.
649
+ *
650
+ * Strategy: prefer the deterministic layers (in order: test → compile →
651
+ * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
652
+ * true, return `null` if no signal qualifies. When multiple deterministic
653
+ * layers contribute, return a `'composite'` source with a weighted blend.
654
+ */
655
+ declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
656
+ /**
657
+ * Extract verifiable rewards from `RunRecord[]` produced via the
658
+ * `verificationReportToRunRecord` adapter (which encodes per-layer scores
659
+ * in `outcome.raw['layer.<name>']`). For records that don't carry layer
660
+ * scores, returns `null` for that record.
661
+ *
662
+ * This is the canonical bridge from "campaign-shaped artifacts" to
663
+ * "RL-training-ready reward signals": every record that has a clean
664
+ * verifiable reward becomes a training datum, every record that doesn't
665
+ * gets filtered out (or kept with `'probabilistic'` determinism for
666
+ * separate downstream handling).
667
+ */
668
+ declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
669
+ runId: string;
670
+ reward: VerifiableReward | null;
671
+ }>;
672
+ /** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
673
+ declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
674
+ run: RunRecord;
675
+ reward: VerifiableReward;
676
+ }>;
845
677
 
846
678
  /**
847
679
  * Reward hacking / Goodhart detection.
@@ -1023,6 +855,60 @@ interface RLCampaignResult<V> {
1023
855
  }
1024
856
  declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult<V>>;
1025
857
 
858
+ /**
859
+ * Adapters: convert measurement outputs into the canonical `RunRecord[]`
860
+ * artifact that `replayCache`, `pairedEvalueSequence`, and
861
+ * `rubricPredictiveValidity` consume. Two sources:
862
+ * - `campaignToRunRecords` — the campaign substrate's per-cell results
863
+ * (the modern path: `runCampaign` / `runImprovementLoop` → records).
864
+ * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
865
+ *
866
+ * Adapters are thin and explicit — every mandatory `RunRecord` field comes
867
+ * from a caller-supplied context (`commitSha`, `model`, `promptHash`,
868
+ * `configHash`) plus the cell's runtime data. The validator still rejects
869
+ * bare-alias model strings — the caller snapshot-pins.
870
+ */
871
+
872
+ interface AdapterContext {
873
+ /** Logical experiment id — typically the campaign or sweep identifier. */
874
+ experimentId: string;
875
+ /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
876
+ model: string;
877
+ /** Git SHA the harness was run from. */
878
+ commitSha: string;
879
+ /** Hash of the effective prompt sent to the model. */
880
+ promptHash: string;
881
+ /** Hash of the effective config (model, temperature, tools, judges, splits). */
882
+ configHash: string;
883
+ /** Default split tag. Default `'search'`. */
884
+ splitTag?: RunSplitTag;
885
+ /** Default cost in USD when the source doesn't record one. Default `0`. */
886
+ defaultCostUsd?: number;
887
+ }
888
+ /**
889
+ * Convert a `CampaignResult` into canonical `RunRecord[]` — one record per
890
+ * scored cell. The cell's mean judge composite becomes the split score; every
891
+ * judge dimension is carried through to `outcome.raw`. A cell that errored
892
+ * becomes a record with `failureMode: 'cell_error'` (kept, not dropped — an
893
+ * unscored cell is signal). `candidateId` identifies the measured surface
894
+ * (defaults to the campaign manifest hash).
895
+ */
896
+ declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterContext & {
897
+ candidateId?: string;
898
+ }): RunRecord[];
899
+ /**
900
+ * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
901
+ * `outcome.searchScore` (or `holdoutScore`) is `report.blendedScore`;
902
+ * `outcome.raw` carries every layer's score + a pass indicator; `failureMode`
903
+ * is the first failing layer's reason.
904
+ */
905
+ declare function verificationReportToRunRecord(report: VerificationReport, ctx: AdapterContext & {
906
+ candidateId: string;
907
+ scenarioId?: string;
908
+ }, opts?: {
909
+ runId?: string;
910
+ }): RunRecord;
911
+
1026
912
  /**
1027
913
  * Simulator fidelity — score a user SIMULATOR's realism against real-user
1028
914
  * trace distributions.
@@ -1189,4 +1075,119 @@ interface EasyModeReport {
1189
1075
  */
1190
1076
  declare function easyModeCheck(simulated: RunRecord[], production: RunRecord[], opts?: EasyModeOptions): EasyModeReport;
1191
1077
 
1078
+ /**
1079
+ * Bradley-Terry / Elo tournament evaluation.
1080
+ *
1081
+ * For multi-candidate sweeps, comparing every candidate's score against
1082
+ * a fixed comparator wastes information — the comparator becomes a high-
1083
+ * variance reference and rank flips between near-tied middle-rank
1084
+ * candidates are dominated by noise. Pairwise tournaments fix this:
1085
+ * every (i, j) pair contributes a comparison to a Bradley-Terry MLE that
1086
+ * estimates each candidate's strength on a unified scale.
1087
+ *
1088
+ * For online updating (rolling campaigns where new candidates arrive
1089
+ * over time), we also ship classical Elo with configurable K-factor.
1090
+ *
1091
+ * References:
1092
+ * - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete
1093
+ * block designs. Biometrika, 39(3/4), 324–345.
1094
+ * - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry
1095
+ * models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm
1096
+ * used here.)
1097
+ * - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.
1098
+ *
1099
+ * This is a useful primitive because most LLM-eval communities (Chatbot
1100
+ * Arena, AlpacaEval, ELO-style ablation) have converged on pairwise
1101
+ * tournament eval as the most sample-efficient and most rank-stable
1102
+ * method when you have many candidates.
1103
+ */
1104
+ interface PairwiseOutcome {
1105
+ /** Winner candidate id. */
1106
+ winner: string;
1107
+ /** Loser candidate id. */
1108
+ loser: string;
1109
+ /**
1110
+ * Optional draw flag. When true, both candidates get half-credit
1111
+ * (Bradley-Terry handles draws as half-wins for each side).
1112
+ */
1113
+ draw?: boolean;
1114
+ /**
1115
+ * Optional weight — useful if some pairwise comparisons are stronger
1116
+ * signals than others (e.g. a paired test with a wider score gap is
1117
+ * a more confident comparison). Default 1.
1118
+ */
1119
+ weight?: number;
1120
+ }
1121
+ interface BradleyTerryRating {
1122
+ candidateId: string;
1123
+ /** Latent strength θ ≥ 0 from the BT MLE. */
1124
+ strength: number;
1125
+ /** Log-strength = log(θ) — interpretable on a linear scale. */
1126
+ logStrength: number;
1127
+ /** Number of pairwise comparisons this candidate appears in. */
1128
+ n: number;
1129
+ /** Win count (+ 0.5 per draw). */
1130
+ wins: number;
1131
+ }
1132
+ interface BradleyTerryFit {
1133
+ ratings: BradleyTerryRating[];
1134
+ /** Iterations of the MM algorithm before convergence. */
1135
+ iterations: number;
1136
+ /** Final maximum |θ_new - θ_old| / θ_old. */
1137
+ finalDelta: number;
1138
+ converged: boolean;
1139
+ }
1140
+ /**
1141
+ * Bradley-Terry MLE via Hunter's MM algorithm.
1142
+ *
1143
+ * Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
1144
+ * where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
1145
+ *
1146
+ * Returns log-strengths normalized so the smallest is 0 (any constant
1147
+ * offset is unobservable in BT — only differences are identified).
1148
+ */
1149
+ declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
1150
+ tolerance?: number;
1151
+ maxIterations?: number;
1152
+ smoothing?: number;
1153
+ }): BradleyTerryFit;
1154
+ /**
1155
+ * Online Elo updates. Use when comparisons arrive over time and you want
1156
+ * a running rating without re-fitting the full BT MLE on every update.
1157
+ *
1158
+ * Initialize ratings to `defaultRating` (1500 by default). Each call to
1159
+ * `applyEloUpdate` mutates the map in place and returns the deltas so
1160
+ * the caller can log per-comparison rating changes.
1161
+ */
1162
+ interface EloOptions {
1163
+ /** Default rating for unseen candidates. Default 1500. */
1164
+ defaultRating?: number;
1165
+ /** K-factor controls the step size. Default 32 (FIDE-ish). */
1166
+ kFactor?: number;
1167
+ }
1168
+ declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseOutcome, opts?: EloOptions): {
1169
+ winnerDelta: number;
1170
+ loserDelta: number;
1171
+ };
1172
+ /**
1173
+ * Build pairwise outcomes from the campaign artifact: for every scenario
1174
+ * shared by two candidates, the higher-scoring run wins. Useful when you
1175
+ * want a tournament view of an existing campaign without an additional
1176
+ * pairwise judge call.
1177
+ */
1178
+ interface BuildPairwiseFromCampaignInput {
1179
+ runs: Array<{
1180
+ candidateId: string;
1181
+ /** Stable identifier for the matching unit (typically scenarioId). */
1182
+ matchKey: string;
1183
+ score: number;
1184
+ }>;
1185
+ /**
1186
+ * Tied-score margin. Below this, the comparison is a draw. Default 0
1187
+ * (no ties).
1188
+ */
1189
+ drawMargin?: number;
1190
+ }
1191
+ declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
1192
+
1192
1193
  export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DetectRewardHackingInput, DpoExportRow, DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, ExtractPreferencesOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, GrpoExportRow, GrpoLookups, OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, PreferenceExtractionReport, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, SftExportRow, SftLookups, type SimFidelityOptions, type ThompsonCurriculumOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, applyEloUpdate, bestOfN, bucketLabel, buildPairwiseFromCampaign, campaignToRunRecords, compareAdaptationCurves, defaultBehaviorFeatures, detectRewardHacking, easyModeCheck, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, jsDivergence, observationsFromRunRecords, paretoFrontier, quantileEdges, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runRLCampaign, selfConsistency, shuffleOrder, simFidelityReport, thompsonCurriculum, varianceBasedCurriculum, verificationReportToRunRecord };