@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
package/dist/rl.d.ts
CHANGED
|
@@ -1,26 +1,246 @@
|
|
|
1
|
-
|
|
2
|
-
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-B8A4BDR3.js';
|
|
3
|
-
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-B8A4BDR3.js';
|
|
4
|
-
import { g as CampaignResult } from './types-BU-7W85F.js';
|
|
5
|
-
import { a as RunSplitTag, R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
6
|
-
import { a as VerificationReport } from './multi-layer-verifier-DUZXrPDA.js';
|
|
1
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-CP2ObebC.js';
|
|
7
2
|
export { A as AdversarialMutation, a as AdversarialScenario, b as AdversarialSearchOptions, c as AdversarialSearchReport, d as adversarialScenarioSearch } from './adversarial-DIVcDoI_.js';
|
|
3
|
+
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-D4YW9UoJ.js';
|
|
4
|
+
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-D4YW9UoJ.js';
|
|
5
|
+
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
8
6
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
9
7
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
10
|
-
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-
|
|
11
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
12
|
-
export { r as runEvalCampaign } from './researcher-
|
|
8
|
+
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-C2hDKM8Z.js';
|
|
9
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-Jr8ME1dZ.js';
|
|
10
|
+
export { r as runEvalCampaign } from './researcher-Jr8ME1dZ.js';
|
|
11
|
+
import { a as VerificationReport } from './multi-layer-verifier-DUZXrPDA.js';
|
|
13
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
|
+
import { C as CampaignResult } from './types-DQRY8ZT-.js';
|
|
14
|
+
import '@tangle-network/agent-interface';
|
|
15
|
+
import './errors-CzMUYo7b.js';
|
|
14
16
|
import './schema-m0gsnbt3.js';
|
|
15
17
|
import './store-BcFXE6LG.js';
|
|
16
|
-
import './errors-CzMUYo7b.js';
|
|
17
|
-
import './verdict-C9MlYujm.js';
|
|
18
18
|
import './llm-client-Bj7g0rqu.js';
|
|
19
19
|
import './raw-provider-sink-C46HDghv.js';
|
|
20
|
-
import './summary-report-
|
|
20
|
+
import './summary-report-CInXwsza.js';
|
|
21
21
|
import './failure-cluster-DH9Flgcf.js';
|
|
22
22
|
import './emitter-C2rqGH_l.js';
|
|
23
23
|
import './integrity-D2t12mMw.js';
|
|
24
|
+
import './verdict-C9MlYujm.js';
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Adaptive curriculum / active scenario selection.
|
|
28
|
+
*
|
|
29
|
+
* Fixed scenario sets waste sample budget on cells the policy already
|
|
30
|
+
* passes (no information left) and cells the policy never passes (no
|
|
31
|
+
* gradient available either). Active learning over scenarios fixes this
|
|
32
|
+
* by allocating the next sample budget to cells where the policy's
|
|
33
|
+
* outcome is *uncertain* — those carry the most decision-relevant signal.
|
|
34
|
+
*
|
|
35
|
+
* This module ships two complementary strategies:
|
|
36
|
+
*
|
|
37
|
+
* 1. **Variance-based** — score each (variant, scenario) cell by the
|
|
38
|
+
* empirical variance of past observations. Allocate next-round budget
|
|
39
|
+
* proportional to variance. Standard active-learning-by-uncertainty
|
|
40
|
+
* heuristic; works well when the policy is non-deterministic and
|
|
41
|
+
* cells differ in observation noise.
|
|
42
|
+
*
|
|
43
|
+
* 2. **Bandit-based (Thompson sampling)** — model each (variant,
|
|
44
|
+
* scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
|
|
45
|
+
* cells whose posterior mean is closest to the per-scenario decision
|
|
46
|
+
* threshold. The right primitive when scenarios are
|
|
47
|
+
* "pass/fail" rather than continuous, and when promotion gates fire
|
|
48
|
+
* at a known threshold (e.g., 0.5).
|
|
49
|
+
*
|
|
50
|
+
* The output is a *next-round budget allocation* — a list of (variant,
|
|
51
|
+
* scenario, count) triples. The consumer's matrix runner consumes the
|
|
52
|
+
* allocation, runs those cells, feeds the new observations back. Loop.
|
|
53
|
+
*
|
|
54
|
+
* Out of scope (deliberate): scenario *generation* — that's the
|
|
55
|
+
* adversarial primitive's job. This module allocates over an existing
|
|
56
|
+
* scenario pool.
|
|
57
|
+
*/
|
|
58
|
+
|
|
59
|
+
interface CellObservation {
|
|
60
|
+
variantId: string;
|
|
61
|
+
scenarioId: string;
|
|
62
|
+
/** Observed score in [0, 1]. */
|
|
63
|
+
score: number;
|
|
64
|
+
/** For Bernoulli arms — derive from the score with a threshold if needed. */
|
|
65
|
+
pass?: boolean;
|
|
66
|
+
}
|
|
67
|
+
interface CurriculumAllocation {
|
|
68
|
+
variantId: string;
|
|
69
|
+
scenarioId: string;
|
|
70
|
+
/** How many additional reps to run on this cell. */
|
|
71
|
+
count: number;
|
|
72
|
+
/** Strategy-specific reason for the allocation. */
|
|
73
|
+
reason: string;
|
|
74
|
+
}
|
|
75
|
+
interface VarianceCurriculumOptions {
|
|
76
|
+
/** Total reps to allocate across all cells. */
|
|
77
|
+
budget: number;
|
|
78
|
+
/**
|
|
79
|
+
* Smoothing prior on variance — keeps the allocator from concentrating
|
|
80
|
+
* on a cell with one observation just because its 1-sample variance is
|
|
81
|
+
* 0. Default 0.05.
|
|
82
|
+
*/
|
|
83
|
+
variancePrior?: number;
|
|
84
|
+
/**
|
|
85
|
+
* Minimum reps per cell — even when the variance estimate is low, give
|
|
86
|
+
* every cell at least this many. Default 1.
|
|
87
|
+
*/
|
|
88
|
+
floorPerCell?: number;
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* Variance-proportional allocation. For each cell, estimate variance from
|
|
92
|
+
* past observations + a prior, then allocate the budget proportional to
|
|
93
|
+
* (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule
|
|
94
|
+
* (Neyman 1934) that balances "explore noisy cells" with "explore
|
|
95
|
+
* under-sampled cells."
|
|
96
|
+
*/
|
|
97
|
+
declare function varianceBasedCurriculum(observations: CellObservation[], candidateCells: Array<{
|
|
98
|
+
variantId: string;
|
|
99
|
+
scenarioId: string;
|
|
100
|
+
}>, opts: VarianceCurriculumOptions): CurriculumAllocation[];
|
|
101
|
+
interface ThompsonCurriculumOptions {
|
|
102
|
+
budget: number;
|
|
103
|
+
/**
|
|
104
|
+
* The per-scenario decision threshold. Cells whose posterior mean is
|
|
105
|
+
* closest to this get the most budget — that's where the next observation
|
|
106
|
+
* has the highest information value for the gate decision. Default 0.5.
|
|
107
|
+
*/
|
|
108
|
+
decisionThreshold?: number;
|
|
109
|
+
/** Beta prior parameters. Default α=β=1 (uniform). */
|
|
110
|
+
priorAlpha?: number;
|
|
111
|
+
priorBeta?: number;
|
|
112
|
+
/** Seed the Thompson sampler. Default unset (Math.random). */
|
|
113
|
+
seed?: number;
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Thompson-sampling-style allocation for pass/fail cells. For each cell:
|
|
117
|
+
*
|
|
118
|
+
* - Maintain Beta(α + passes, β + failures) posterior on pass-rate
|
|
119
|
+
* - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):
|
|
120
|
+
* cells whose sampled posterior straddles the decision boundary get
|
|
121
|
+
* the most weight; cells already clearly above or below get less.
|
|
122
|
+
*
|
|
123
|
+
* This is the right primitive when promotion gates fire at a known
|
|
124
|
+
* threshold and you want to sharpen the posterior near the boundary.
|
|
125
|
+
*/
|
|
126
|
+
declare function thompsonCurriculum(observations: CellObservation[], candidateCells: Array<{
|
|
127
|
+
variantId: string;
|
|
128
|
+
scenarioId: string;
|
|
129
|
+
}>, opts: ThompsonCurriculumOptions): CurriculumAllocation[];
|
|
130
|
+
/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
|
|
131
|
+
declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
|
|
132
|
+
passThreshold?: number;
|
|
133
|
+
useHoldout?: boolean;
|
|
134
|
+
}): CellObservation[];
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* Sample-efficient adaptation evaluation.
|
|
138
|
+
*
|
|
139
|
+
* For foundation-model-based agents, the load-bearing capability isn't
|
|
140
|
+
* raw end-state performance — it's *how fast the agent reaches that
|
|
141
|
+
* performance from cold start*. The same model with a worse prompt that
|
|
142
|
+
* adapts in 5 demonstrations beats the same model with a better prompt
|
|
143
|
+
* that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
|
|
144
|
+
* reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
|
|
145
|
+
* in-context examples or fine-tune steps.
|
|
146
|
+
*
|
|
147
|
+
* This module ships:
|
|
148
|
+
*
|
|
149
|
+
* 1. `runAdaptationCurve` — given a runner that takes k demonstrations
|
|
150
|
+
* and returns a score, produce the (k, score) curve.
|
|
151
|
+
* 2. `compareAdaptationCurves` — paired comparison across two policies.
|
|
152
|
+
* Returns per-k delta with bootstrap CIs and an "area-under-curve"
|
|
153
|
+
* summary statistic.
|
|
154
|
+
* 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
|
|
155
|
+
* the policy reliably passes (≥ pass-rate threshold over reps).
|
|
156
|
+
*
|
|
157
|
+
* Use cases:
|
|
158
|
+
* - Compare two prompt designs that have similar end-state performance
|
|
159
|
+
* but different in-context efficiency.
|
|
160
|
+
* - Decide between fine-tuning and prompting based on adaptation cost.
|
|
161
|
+
* - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
|
|
162
|
+
*/
|
|
163
|
+
interface AdaptationRunner<S> {
|
|
164
|
+
/**
|
|
165
|
+
* Runs the policy on `scenario` with `k` demonstrations. Returns a
|
|
166
|
+
* scalar score in [0, 1]. The runner is responsible for any caching;
|
|
167
|
+
* the harness calls it once per (scenario, k, rep) cell.
|
|
168
|
+
*/
|
|
169
|
+
run(args: {
|
|
170
|
+
scenario: S;
|
|
171
|
+
k: number;
|
|
172
|
+
rep: number;
|
|
173
|
+
}): Promise<number>;
|
|
174
|
+
}
|
|
175
|
+
interface RunAdaptationCurveOptions<S> {
|
|
176
|
+
scenarios: S[];
|
|
177
|
+
/** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
|
|
178
|
+
ks?: number[];
|
|
179
|
+
/** Reps per (scenario, k) cell. Default 3. */
|
|
180
|
+
reps?: number;
|
|
181
|
+
runner: AdaptationRunner<S>;
|
|
182
|
+
/** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
|
|
183
|
+
passThreshold?: number;
|
|
184
|
+
}
|
|
185
|
+
interface AdaptationPoint {
|
|
186
|
+
k: number;
|
|
187
|
+
meanScore: number;
|
|
188
|
+
passRate: number;
|
|
189
|
+
std: number;
|
|
190
|
+
n: number;
|
|
191
|
+
/** Per-scenario means at this k. */
|
|
192
|
+
perScenario: Array<{
|
|
193
|
+
scenarioId: string;
|
|
194
|
+
meanScore: number;
|
|
195
|
+
passes: number;
|
|
196
|
+
total: number;
|
|
197
|
+
}>;
|
|
198
|
+
}
|
|
199
|
+
interface AdaptationCurve {
|
|
200
|
+
points: AdaptationPoint[];
|
|
201
|
+
/**
|
|
202
|
+
* Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
|
|
203
|
+
* tested reaches it.
|
|
204
|
+
*/
|
|
205
|
+
firstPassK: number | null;
|
|
206
|
+
/**
|
|
207
|
+
* Area under the (k, meanScore) curve, normalized by max-k. A
|
|
208
|
+
* single-number summary of "how well does this policy adapt from
|
|
209
|
+
* cold-start to fully-conditioned." Higher = better adapter.
|
|
210
|
+
*/
|
|
211
|
+
adaptationArea: number;
|
|
212
|
+
}
|
|
213
|
+
declare function runAdaptationCurve<S extends {
|
|
214
|
+
scenarioId?: string;
|
|
215
|
+
}>(opts: RunAdaptationCurveOptions<S>): Promise<AdaptationCurve>;
|
|
216
|
+
interface CompareCurvesResult {
|
|
217
|
+
perK: Array<{
|
|
218
|
+
k: number;
|
|
219
|
+
deltaMean: number;
|
|
220
|
+
aLow: number;
|
|
221
|
+
aHigh: number;
|
|
222
|
+
bLow: number;
|
|
223
|
+
bHigh: number;
|
|
224
|
+
}>;
|
|
225
|
+
areaDelta: number;
|
|
226
|
+
firstPassKDelta: number | null;
|
|
227
|
+
/** Verdict: 'a_better' | 'b_better' | 'similar'. */
|
|
228
|
+
verdict: 'a_better' | 'b_better' | 'similar';
|
|
229
|
+
/** Rationale, ready to render. */
|
|
230
|
+
rationale: string;
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* Paired comparison of two adaptation curves. Per-k deltas with 95%
|
|
234
|
+
* bootstrap CIs (constructed from each curve's `perScenario` per-k means
|
|
235
|
+
* — the bootstrap unit is the scenario, not the rep).
|
|
236
|
+
*/
|
|
237
|
+
declare function compareAdaptationCurves(a: AdaptationCurve, b: AdaptationCurve, opts?: {
|
|
238
|
+
confidence?: number;
|
|
239
|
+
bootstrapResamples?: number;
|
|
240
|
+
seed?: number;
|
|
241
|
+
}): CompareCurvesResult;
|
|
242
|
+
/** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
|
|
243
|
+
declare function firstPassK(curve: AdaptationCurve, threshold?: number): number | null;
|
|
24
244
|
|
|
25
245
|
/**
|
|
26
246
|
* Test-time compute scaling curves.
|
|
@@ -267,173 +487,70 @@ declare function injectIrrelevantClause<S extends {
|
|
|
267
487
|
}>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
|
|
268
488
|
|
|
269
489
|
/**
|
|
270
|
-
*
|
|
271
|
-
*
|
|
272
|
-
* `rubricPredictiveValidity` consume. Two sources:
|
|
273
|
-
* - `campaignToRunRecords` — the campaign substrate's per-cell results
|
|
274
|
-
* (the modern path: `runCampaign` / `runImprovementLoop` → records).
|
|
275
|
-
* - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
|
|
490
|
+
* `PredictiveValidityResearcher` — concrete `Researcher` implementation
|
|
491
|
+
* that drives selection from outcome-anchored predictive validity.
|
|
276
492
|
*
|
|
277
|
-
*
|
|
278
|
-
*
|
|
279
|
-
* `
|
|
280
|
-
*
|
|
493
|
+
* Each method:
|
|
494
|
+
*
|
|
495
|
+
* - `inspectFailures(runs)` — synthesizes failure modes from the
|
|
496
|
+
* bottom-quartile of `RunRecord`s on the configured proxy reward.
|
|
497
|
+
* - `proposeChange(failures)` — proposes steering changes that target
|
|
498
|
+
* the rubrics with the lowest predictive validity (decorative ones).
|
|
499
|
+
* Either reduce their weight in the composite, or recalibrate them.
|
|
500
|
+
* - `applyChange(changes, baseline)` — merges the proposed steering
|
|
501
|
+
* into the experiment plan.
|
|
502
|
+
* - `evaluateChange(plan)` — re-runs the predictive-validity check on
|
|
503
|
+
* the post-change runs and reports the delta.
|
|
504
|
+
*
|
|
505
|
+
* The result is a closed loop: the rubric weights drift toward the ones
|
|
506
|
+
* that actually predict deployment outcomes, automatically. Pair with
|
|
507
|
+
* `runRLCampaign` for the full auto-research story.
|
|
281
508
|
*/
|
|
282
509
|
|
|
283
|
-
interface
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
/**
|
|
287
|
-
|
|
288
|
-
/**
|
|
289
|
-
|
|
290
|
-
/**
|
|
291
|
-
|
|
292
|
-
/**
|
|
293
|
-
|
|
294
|
-
/** Default split tag. Default `'search'`. */
|
|
295
|
-
splitTag?: RunSplitTag;
|
|
296
|
-
/** Default cost in USD when the source doesn't record one. Default `0`. */
|
|
297
|
-
defaultCostUsd?: number;
|
|
298
|
-
}
|
|
299
|
-
/**
|
|
300
|
-
* Convert a `CampaignResult` into canonical `RunRecord[]` — one record per
|
|
301
|
-
* scored cell. The cell's mean judge composite becomes the split score; every
|
|
302
|
-
* judge dimension is carried through to `outcome.raw`. A cell that errored
|
|
303
|
-
* becomes a record with `failureMode: 'cell_error'` (kept, not dropped — an
|
|
304
|
-
* unscored cell is signal). `candidateId` identifies the measured surface
|
|
305
|
-
* (defaults to the campaign manifest hash).
|
|
306
|
-
*/
|
|
307
|
-
declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterContext & {
|
|
308
|
-
candidateId?: string;
|
|
309
|
-
}): RunRecord[];
|
|
310
|
-
/**
|
|
311
|
-
* Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
|
|
312
|
-
* `outcome.searchScore` (or `holdoutScore`) is `report.blendedScore`;
|
|
313
|
-
* `outcome.raw` carries every layer's score + a pass indicator; `failureMode`
|
|
314
|
-
* is the first failing layer's reason.
|
|
315
|
-
*/
|
|
316
|
-
declare function verificationReportToRunRecord(report: VerificationReport, ctx: AdapterContext & {
|
|
317
|
-
candidateId: string;
|
|
318
|
-
scenarioId?: string;
|
|
319
|
-
}, opts?: {
|
|
320
|
-
runId?: string;
|
|
321
|
-
}): RunRecord;
|
|
322
|
-
|
|
323
|
-
/**
|
|
324
|
-
* Bradley-Terry / Elo tournament evaluation.
|
|
325
|
-
*
|
|
326
|
-
* For multi-candidate sweeps, comparing every candidate's score against
|
|
327
|
-
* a fixed comparator wastes information — the comparator becomes a high-
|
|
328
|
-
* variance reference and rank flips between near-tied middle-rank
|
|
329
|
-
* candidates are dominated by noise. Pairwise tournaments fix this:
|
|
330
|
-
* every (i, j) pair contributes a comparison to a Bradley-Terry MLE that
|
|
331
|
-
* estimates each candidate's strength on a unified scale.
|
|
332
|
-
*
|
|
333
|
-
* For online updating (rolling campaigns where new candidates arrive
|
|
334
|
-
* over time), we also ship classical Elo with configurable K-factor.
|
|
335
|
-
*
|
|
336
|
-
* References:
|
|
337
|
-
* - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete
|
|
338
|
-
* block designs. Biometrika, 39(3/4), 324–345.
|
|
339
|
-
* - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry
|
|
340
|
-
* models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm
|
|
341
|
-
* used here.)
|
|
342
|
-
* - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.
|
|
343
|
-
*
|
|
344
|
-
* This is a useful primitive because most LLM-eval communities (Chatbot
|
|
345
|
-
* Arena, AlpacaEval, ELO-style ablation) have converged on pairwise
|
|
346
|
-
* tournament eval as the most sample-efficient and most rank-stable
|
|
347
|
-
* method when you have many candidates.
|
|
348
|
-
*/
|
|
349
|
-
interface PairwiseOutcome {
|
|
350
|
-
/** Winner candidate id. */
|
|
351
|
-
winner: string;
|
|
352
|
-
/** Loser candidate id. */
|
|
353
|
-
loser: string;
|
|
354
|
-
/**
|
|
355
|
-
* Optional draw flag. When true, both candidates get half-credit
|
|
356
|
-
* (Bradley-Terry handles draws as half-wins for each side).
|
|
357
|
-
*/
|
|
358
|
-
draw?: boolean;
|
|
510
|
+
interface PredictiveValidityResearcherOptions {
|
|
511
|
+
outcomes: OutcomeStore;
|
|
512
|
+
outcomeMetrics: string[];
|
|
513
|
+
/** Score threshold below which a run counts as a "failure." Default 0.5. */
|
|
514
|
+
failureThreshold?: number;
|
|
515
|
+
/** Spearman bucket below which a rubric is "decorative." Default 0.4. */
|
|
516
|
+
decorativeThreshold?: number;
|
|
517
|
+
/** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
|
|
518
|
+
steeringNamespace?: string;
|
|
519
|
+
/** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
|
|
520
|
+
rubrics?: string[];
|
|
359
521
|
/**
|
|
360
|
-
*
|
|
361
|
-
*
|
|
362
|
-
*
|
|
522
|
+
* Snapshot stash hook — called with the most recent predictive-validity
|
|
523
|
+
* report. Useful when a downstream system wants to log rubric drift over
|
|
524
|
+
* time. Default no-op.
|
|
363
525
|
*/
|
|
364
|
-
|
|
365
|
-
}
|
|
366
|
-
interface BradleyTerryRating {
|
|
367
|
-
candidateId: string;
|
|
368
|
-
/** Latent strength θ ≥ 0 from the BT MLE. */
|
|
369
|
-
strength: number;
|
|
370
|
-
/** Log-strength = log(θ) — interpretable on a linear scale. */
|
|
371
|
-
logStrength: number;
|
|
372
|
-
/** Number of pairwise comparisons this candidate appears in. */
|
|
373
|
-
n: number;
|
|
374
|
-
/** Win count (+ 0.5 per draw). */
|
|
375
|
-
wins: number;
|
|
376
|
-
}
|
|
377
|
-
interface BradleyTerryFit {
|
|
378
|
-
ratings: BradleyTerryRating[];
|
|
379
|
-
/** Iterations of the MM algorithm before convergence. */
|
|
380
|
-
iterations: number;
|
|
381
|
-
/** Final maximum |θ_new - θ_old| / θ_old. */
|
|
382
|
-
finalDelta: number;
|
|
383
|
-
converged: boolean;
|
|
384
|
-
}
|
|
385
|
-
/**
|
|
386
|
-
* Bradley-Terry MLE via Hunter's MM algorithm.
|
|
387
|
-
*
|
|
388
|
-
* Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
|
|
389
|
-
* where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
|
|
390
|
-
*
|
|
391
|
-
* Returns log-strengths normalized so the smallest is 0 (any constant
|
|
392
|
-
* offset is unobservable in BT — only differences are identified).
|
|
393
|
-
*/
|
|
394
|
-
declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
|
|
395
|
-
tolerance?: number;
|
|
396
|
-
maxIterations?: number;
|
|
397
|
-
smoothing?: number;
|
|
398
|
-
}): BradleyTerryFit;
|
|
399
|
-
/**
|
|
400
|
-
* Online Elo updates. Use when comparisons arrive over time and you want
|
|
401
|
-
* a running rating without re-fitting the full BT MLE on every update.
|
|
402
|
-
*
|
|
403
|
-
* Initialize ratings to `defaultRating` (1500 by default). Each call to
|
|
404
|
-
* `applyEloUpdate` mutates the map in place and returns the deltas so
|
|
405
|
-
* the caller can log per-comparison rating changes.
|
|
406
|
-
*/
|
|
407
|
-
interface EloOptions {
|
|
408
|
-
/** Default rating for unseen candidates. Default 1500. */
|
|
409
|
-
defaultRating?: number;
|
|
410
|
-
/** K-factor controls the step size. Default 32 (FIDE-ish). */
|
|
411
|
-
kFactor?: number;
|
|
526
|
+
onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>;
|
|
412
527
|
}
|
|
413
|
-
declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseOutcome, opts?: EloOptions): {
|
|
414
|
-
winnerDelta: number;
|
|
415
|
-
loserDelta: number;
|
|
416
|
-
};
|
|
417
528
|
/**
|
|
418
|
-
*
|
|
419
|
-
*
|
|
420
|
-
* want a tournament view of an existing campaign without an additional
|
|
421
|
-
* pairwise judge call.
|
|
529
|
+
* Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
|
|
530
|
+
* rubrics that don't predict deployment outcomes don't earn weight.
|
|
422
531
|
*/
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
532
|
+
declare class PredictiveValidityResearcher implements Researcher {
|
|
533
|
+
private opts;
|
|
534
|
+
private lastReport;
|
|
535
|
+
constructor(opts: PredictiveValidityResearcherOptions);
|
|
536
|
+
inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
|
|
537
|
+
proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
|
|
538
|
+
applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
|
|
539
|
+
evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
|
|
430
540
|
/**
|
|
431
|
-
*
|
|
432
|
-
*
|
|
541
|
+
* Run the predictive-validity check explicitly against a fresh RunRecord
|
|
542
|
+
* set. Updates the researcher's cached report so subsequent
|
|
543
|
+
* `proposeChange` calls have evidence to draw from.
|
|
433
544
|
*/
|
|
434
|
-
|
|
545
|
+
runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport>;
|
|
546
|
+
/**
|
|
547
|
+
* Force-feed a predictive-validity report into the researcher state —
|
|
548
|
+
* useful when the consumer ran the report out-of-band and wants the
|
|
549
|
+
* researcher's later proposals informed by it.
|
|
550
|
+
*/
|
|
551
|
+
setReport(report: RubricPredictiveValidityReport): void;
|
|
552
|
+
getLastReport(): RubricPredictiveValidityReport | null;
|
|
435
553
|
}
|
|
436
|
-
declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
|
|
437
554
|
|
|
438
555
|
/**
|
|
439
556
|
* Verifiable reward channel.
|
|
@@ -487,361 +604,76 @@ interface VerifiableReward {
|
|
|
487
604
|
/** The layer / judge id that produced the signal, for provenance. */
|
|
488
605
|
origin: string;
|
|
489
606
|
/**
|
|
490
|
-
* Per-source contribution to `value`, keyed by layer/judge id. Single-source
|
|
491
|
-
* rewards carry one entry (`{ [origin]: value }`); composite rewards carry
|
|
492
|
-
* every contributing layer's score — the anti-scalar-collapse surface RL
|
|
493
|
-
* consumers weight per-source instead of trusting one blended number.
|
|
494
|
-
*/
|
|
495
|
-
components: Record<string, number>;
|
|
496
|
-
/**
|
|
497
|
-
* @deprecated Read `components` for per-source reward values. Kept for
|
|
498
|
-
* published-API compatibility: single-source rewards carry the layer's
|
|
499
|
-
* diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
|
|
500
|
-
* the same per-layer scores `components` now holds.
|
|
501
|
-
*/
|
|
502
|
-
breakdown?: Record<string, number>;
|
|
503
|
-
}
|
|
504
|
-
interface VerifiableRewardExtractionOptions {
|
|
505
|
-
/**
|
|
506
|
-
* Which layers count as deterministic-reward sources. The verifier doesn't
|
|
507
|
-
* tag layers as "this is verifiable"; the caller declares it via this list
|
|
508
|
-
* (or via the layer name → source mapping). Default treats common names
|
|
509
|
-
* (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
|
|
510
|
-
* `sandbox`) as deterministic.
|
|
511
|
-
*/
|
|
512
|
-
deterministicLayers?: string[];
|
|
513
|
-
/**
|
|
514
|
-
* Map layer name → reward source. Defaults to a sensible string-match.
|
|
515
|
-
*/
|
|
516
|
-
sourceFor?: (layerName: string) => VerifiableRewardSource;
|
|
517
|
-
/**
|
|
518
|
-
* Whether to fall back to a probabilistic (judge) reward when no
|
|
519
|
-
* deterministic layer produced a numeric score. Default `true`. Set to
|
|
520
|
-
* `false` for "deterministic-only" training pipelines that should
|
|
521
|
-
* discard runs without a verifiable signal.
|
|
522
|
-
*/
|
|
523
|
-
fallbackToJudge?: boolean;
|
|
524
|
-
/**
|
|
525
|
-
* Default confidence for probabilistic (judge) rewards when the judge
|
|
526
|
-
* doesn't report one. Default `0.7`.
|
|
527
|
-
*/
|
|
528
|
-
judgeConfidenceFloor?: number;
|
|
529
|
-
}
|
|
530
|
-
/**
|
|
531
|
-
* Extract a `VerifiableReward` from a `VerificationReport`.
|
|
532
|
-
*
|
|
533
|
-
* Strategy: prefer the deterministic layers (in order: test → compile →
|
|
534
|
-
* schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
|
|
535
|
-
* true, return `null` if no signal qualifies. When multiple deterministic
|
|
536
|
-
* layers contribute, return a `'composite'` source with a weighted blend.
|
|
537
|
-
*/
|
|
538
|
-
declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
|
|
539
|
-
/**
|
|
540
|
-
* Extract verifiable rewards from `RunRecord[]` produced via the
|
|
541
|
-
* `verificationReportToRunRecord` adapter (which encodes per-layer scores
|
|
542
|
-
* in `outcome.raw['layer.<name>']`). For records that don't carry layer
|
|
543
|
-
* scores, returns `null` for that record.
|
|
544
|
-
*
|
|
545
|
-
* This is the canonical bridge from "campaign-shaped artifacts" to
|
|
546
|
-
* "RL-training-ready reward signals": every record that has a clean
|
|
547
|
-
* verifiable reward becomes a training datum, every record that doesn't
|
|
548
|
-
* gets filtered out (or kept with `'probabilistic'` determinism for
|
|
549
|
-
* separate downstream handling).
|
|
550
|
-
*/
|
|
551
|
-
declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
|
|
552
|
-
runId: string;
|
|
553
|
-
reward: VerifiableReward | null;
|
|
554
|
-
}>;
|
|
555
|
-
/** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
|
|
556
|
-
declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
|
|
557
|
-
run: RunRecord;
|
|
558
|
-
reward: VerifiableReward;
|
|
559
|
-
}>;
|
|
560
|
-
|
|
561
|
-
/**
|
|
562
|
-
* Adaptive curriculum / active scenario selection.
|
|
563
|
-
*
|
|
564
|
-
* Fixed scenario sets waste sample budget on cells the policy already
|
|
565
|
-
* passes (no information left) and cells the policy never passes (no
|
|
566
|
-
* gradient available either). Active learning over scenarios fixes this
|
|
567
|
-
* by allocating the next sample budget to cells where the policy's
|
|
568
|
-
* outcome is *uncertain* — those carry the most decision-relevant signal.
|
|
569
|
-
*
|
|
570
|
-
* This module ships two complementary strategies:
|
|
571
|
-
*
|
|
572
|
-
* 1. **Variance-based** — score each (variant, scenario) cell by the
|
|
573
|
-
* empirical variance of past observations. Allocate next-round budget
|
|
574
|
-
* proportional to variance. Standard active-learning-by-uncertainty
|
|
575
|
-
* heuristic; works well when the policy is non-deterministic and
|
|
576
|
-
* cells differ in observation noise.
|
|
577
|
-
*
|
|
578
|
-
* 2. **Bandit-based (Thompson sampling)** — model each (variant,
|
|
579
|
-
* scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
|
|
580
|
-
* cells whose posterior mean is closest to the per-scenario decision
|
|
581
|
-
* threshold. The right primitive when scenarios are
|
|
582
|
-
* "pass/fail" rather than continuous, and when promotion gates fire
|
|
583
|
-
* at a known threshold (e.g., 0.5).
|
|
584
|
-
*
|
|
585
|
-
* The output is a *next-round budget allocation* — a list of (variant,
|
|
586
|
-
* scenario, count) triples. The consumer's matrix runner consumes the
|
|
587
|
-
* allocation, runs those cells, feeds the new observations back. Loop.
|
|
588
|
-
*
|
|
589
|
-
* Out of scope (deliberate): scenario *generation* — that's the
|
|
590
|
-
* adversarial primitive's job. This module allocates over an existing
|
|
591
|
-
* scenario pool.
|
|
592
|
-
*/
|
|
593
|
-
|
|
594
|
-
interface CellObservation {
|
|
595
|
-
variantId: string;
|
|
596
|
-
scenarioId: string;
|
|
597
|
-
/** Observed score in [0, 1]. */
|
|
598
|
-
score: number;
|
|
599
|
-
/** For Bernoulli arms — derive from the score with a threshold if needed. */
|
|
600
|
-
pass?: boolean;
|
|
601
|
-
}
|
|
602
|
-
interface CurriculumAllocation {
|
|
603
|
-
variantId: string;
|
|
604
|
-
scenarioId: string;
|
|
605
|
-
/** How many additional reps to run on this cell. */
|
|
606
|
-
count: number;
|
|
607
|
-
/** Strategy-specific reason for the allocation. */
|
|
608
|
-
reason: string;
|
|
609
|
-
}
|
|
610
|
-
interface VarianceCurriculumOptions {
|
|
611
|
-
/** Total reps to allocate across all cells. */
|
|
612
|
-
budget: number;
|
|
613
|
-
/**
|
|
614
|
-
* Smoothing prior on variance — keeps the allocator from concentrating
|
|
615
|
-
* on a cell with one observation just because its 1-sample variance is
|
|
616
|
-
* 0. Default 0.05.
|
|
617
|
-
*/
|
|
618
|
-
variancePrior?: number;
|
|
619
|
-
/**
|
|
620
|
-
* Minimum reps per cell — even when the variance estimate is low, give
|
|
621
|
-
* every cell at least this many. Default 1.
|
|
622
|
-
*/
|
|
623
|
-
floorPerCell?: number;
|
|
624
|
-
}
|
|
625
|
-
/**
|
|
626
|
-
* Variance-proportional allocation. For each cell, estimate variance from
|
|
627
|
-
* past observations + a prior, then allocate the budget proportional to
|
|
628
|
-
* (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule
|
|
629
|
-
* (Neyman 1934) that balances "explore noisy cells" with "explore
|
|
630
|
-
* under-sampled cells."
|
|
631
|
-
*/
|
|
632
|
-
declare function varianceBasedCurriculum(observations: CellObservation[], candidateCells: Array<{
|
|
633
|
-
variantId: string;
|
|
634
|
-
scenarioId: string;
|
|
635
|
-
}>, opts: VarianceCurriculumOptions): CurriculumAllocation[];
|
|
636
|
-
interface ThompsonCurriculumOptions {
|
|
637
|
-
budget: number;
|
|
638
|
-
/**
|
|
639
|
-
* The per-scenario decision threshold. Cells whose posterior mean is
|
|
640
|
-
* closest to this get the most budget — that's where the next observation
|
|
641
|
-
* has the highest information value for the gate decision. Default 0.5.
|
|
642
|
-
*/
|
|
643
|
-
decisionThreshold?: number;
|
|
644
|
-
/** Beta prior parameters. Default α=β=1 (uniform). */
|
|
645
|
-
priorAlpha?: number;
|
|
646
|
-
priorBeta?: number;
|
|
647
|
-
/** Seed the Thompson sampler. Default unset (Math.random). */
|
|
648
|
-
seed?: number;
|
|
649
|
-
}
|
|
650
|
-
/**
|
|
651
|
-
* Thompson-sampling-style allocation for pass/fail cells. For each cell:
|
|
652
|
-
*
|
|
653
|
-
* - Maintain Beta(α + passes, β + failures) posterior on pass-rate
|
|
654
|
-
* - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):
|
|
655
|
-
* cells whose sampled posterior straddles the decision boundary get
|
|
656
|
-
* the most weight; cells already clearly above or below get less.
|
|
657
|
-
*
|
|
658
|
-
* This is the right primitive when promotion gates fire at a known
|
|
659
|
-
* threshold and you want to sharpen the posterior near the boundary.
|
|
660
|
-
*/
|
|
661
|
-
declare function thompsonCurriculum(observations: CellObservation[], candidateCells: Array<{
|
|
662
|
-
variantId: string;
|
|
663
|
-
scenarioId: string;
|
|
664
|
-
}>, opts: ThompsonCurriculumOptions): CurriculumAllocation[];
|
|
665
|
-
/** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
|
|
666
|
-
declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
|
|
667
|
-
passThreshold?: number;
|
|
668
|
-
useHoldout?: boolean;
|
|
669
|
-
}): CellObservation[];
|
|
670
|
-
|
|
671
|
-
/**
|
|
672
|
-
* Sample-efficient adaptation evaluation.
|
|
673
|
-
*
|
|
674
|
-
* For foundation-model-based agents, the load-bearing capability isn't
|
|
675
|
-
* raw end-state performance — it's *how fast the agent reaches that
|
|
676
|
-
* performance from cold start*. The same model with a worse prompt that
|
|
677
|
-
* adapts in 5 demonstrations beats the same model with a better prompt
|
|
678
|
-
* that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
|
|
679
|
-
* reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
|
|
680
|
-
* in-context examples or fine-tune steps.
|
|
681
|
-
*
|
|
682
|
-
* This module ships:
|
|
683
|
-
*
|
|
684
|
-
* 1. `runAdaptationCurve` — given a runner that takes k demonstrations
|
|
685
|
-
* and returns a score, produce the (k, score) curve.
|
|
686
|
-
* 2. `compareAdaptationCurves` — paired comparison across two policies.
|
|
687
|
-
* Returns per-k delta with bootstrap CIs and an "area-under-curve"
|
|
688
|
-
* summary statistic.
|
|
689
|
-
* 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
|
|
690
|
-
* the policy reliably passes (≥ pass-rate threshold over reps).
|
|
691
|
-
*
|
|
692
|
-
* Use cases:
|
|
693
|
-
* - Compare two prompt designs that have similar end-state performance
|
|
694
|
-
* but different in-context efficiency.
|
|
695
|
-
* - Decide between fine-tuning and prompting based on adaptation cost.
|
|
696
|
-
* - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
|
|
697
|
-
*/
|
|
698
|
-
interface AdaptationRunner<S> {
|
|
699
|
-
/**
|
|
700
|
-
* Runs the policy on `scenario` with `k` demonstrations. Returns a
|
|
701
|
-
* scalar score in [0, 1]. The runner is responsible for any caching;
|
|
702
|
-
* the harness calls it once per (scenario, k, rep) cell.
|
|
703
|
-
*/
|
|
704
|
-
run(args: {
|
|
705
|
-
scenario: S;
|
|
706
|
-
k: number;
|
|
707
|
-
rep: number;
|
|
708
|
-
}): Promise<number>;
|
|
709
|
-
}
|
|
710
|
-
interface RunAdaptationCurveOptions<S> {
|
|
711
|
-
scenarios: S[];
|
|
712
|
-
/** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
|
|
713
|
-
ks?: number[];
|
|
714
|
-
/** Reps per (scenario, k) cell. Default 3. */
|
|
715
|
-
reps?: number;
|
|
716
|
-
runner: AdaptationRunner<S>;
|
|
717
|
-
/** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
|
|
718
|
-
passThreshold?: number;
|
|
719
|
-
}
|
|
720
|
-
interface AdaptationPoint {
|
|
721
|
-
k: number;
|
|
722
|
-
meanScore: number;
|
|
723
|
-
passRate: number;
|
|
724
|
-
std: number;
|
|
725
|
-
n: number;
|
|
726
|
-
/** Per-scenario means at this k. */
|
|
727
|
-
perScenario: Array<{
|
|
728
|
-
scenarioId: string;
|
|
729
|
-
meanScore: number;
|
|
730
|
-
passes: number;
|
|
731
|
-
total: number;
|
|
732
|
-
}>;
|
|
733
|
-
}
|
|
734
|
-
interface AdaptationCurve {
|
|
735
|
-
points: AdaptationPoint[];
|
|
736
|
-
/**
|
|
737
|
-
* Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
|
|
738
|
-
* tested reaches it.
|
|
739
|
-
*/
|
|
740
|
-
firstPassK: number | null;
|
|
741
|
-
/**
|
|
742
|
-
* Area under the (k, meanScore) curve, normalized by max-k. A
|
|
743
|
-
* single-number summary of "how well does this policy adapt from
|
|
744
|
-
* cold-start to fully-conditioned." Higher = better adapter.
|
|
745
|
-
*/
|
|
746
|
-
adaptationArea: number;
|
|
747
|
-
}
|
|
748
|
-
declare function runAdaptationCurve<S extends {
|
|
749
|
-
scenarioId?: string;
|
|
750
|
-
}>(opts: RunAdaptationCurveOptions<S>): Promise<AdaptationCurve>;
|
|
751
|
-
interface CompareCurvesResult {
|
|
752
|
-
perK: Array<{
|
|
753
|
-
k: number;
|
|
754
|
-
deltaMean: number;
|
|
755
|
-
aLow: number;
|
|
756
|
-
aHigh: number;
|
|
757
|
-
bLow: number;
|
|
758
|
-
bHigh: number;
|
|
759
|
-
}>;
|
|
760
|
-
areaDelta: number;
|
|
761
|
-
firstPassKDelta: number | null;
|
|
762
|
-
/** Verdict: 'a_better' | 'b_better' | 'similar'. */
|
|
763
|
-
verdict: 'a_better' | 'b_better' | 'similar';
|
|
764
|
-
/** Rationale, ready to render. */
|
|
765
|
-
rationale: string;
|
|
766
|
-
}
|
|
767
|
-
/**
|
|
768
|
-
* Paired comparison of two adaptation curves. Per-k deltas with 95%
|
|
769
|
-
* bootstrap CIs (constructed from each curve's `perScenario` per-k means
|
|
770
|
-
* — the bootstrap unit is the scenario, not the rep).
|
|
771
|
-
*/
|
|
772
|
-
declare function compareAdaptationCurves(a: AdaptationCurve, b: AdaptationCurve, opts?: {
|
|
773
|
-
confidence?: number;
|
|
774
|
-
bootstrapResamples?: number;
|
|
775
|
-
seed?: number;
|
|
776
|
-
}): CompareCurvesResult;
|
|
777
|
-
/** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
|
|
778
|
-
declare function firstPassK(curve: AdaptationCurve, threshold?: number): number | null;
|
|
779
|
-
|
|
780
|
-
/**
|
|
781
|
-
* `PredictiveValidityResearcher` — concrete `Researcher` implementation
|
|
782
|
-
* that drives selection from outcome-anchored predictive validity.
|
|
783
|
-
*
|
|
784
|
-
* Each method:
|
|
785
|
-
*
|
|
786
|
-
* - `inspectFailures(runs)` — synthesizes failure modes from the
|
|
787
|
-
* bottom-quartile of `RunRecord`s on the configured proxy reward.
|
|
788
|
-
* - `proposeChange(failures)` — proposes steering changes that target
|
|
789
|
-
* the rubrics with the lowest predictive validity (decorative ones).
|
|
790
|
-
* Either reduce their weight in the composite, or recalibrate them.
|
|
791
|
-
* - `applyChange(changes, baseline)` — merges the proposed steering
|
|
792
|
-
* into the experiment plan.
|
|
793
|
-
* - `evaluateChange(plan)` — re-runs the predictive-validity check on
|
|
794
|
-
* the post-change runs and reports the delta.
|
|
795
|
-
*
|
|
796
|
-
* The result is a closed loop: the rubric weights drift toward the ones
|
|
797
|
-
* that actually predict deployment outcomes, automatically. Pair with
|
|
798
|
-
* `runRLCampaign` for the full auto-research story.
|
|
799
|
-
*/
|
|
800
|
-
|
|
801
|
-
interface PredictiveValidityResearcherOptions {
|
|
802
|
-
outcomes: OutcomeStore;
|
|
803
|
-
outcomeMetrics: string[];
|
|
804
|
-
/** Score threshold below which a run counts as a "failure." Default 0.5. */
|
|
805
|
-
failureThreshold?: number;
|
|
806
|
-
/** Spearman bucket below which a rubric is "decorative." Default 0.4. */
|
|
807
|
-
decorativeThreshold?: number;
|
|
808
|
-
/** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
|
|
809
|
-
steeringNamespace?: string;
|
|
810
|
-
/** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
|
|
811
|
-
rubrics?: string[];
|
|
607
|
+
* Per-source contribution to `value`, keyed by layer/judge id. Single-source
|
|
608
|
+
* rewards carry one entry (`{ [origin]: value }`); composite rewards carry
|
|
609
|
+
* every contributing layer's score — the anti-scalar-collapse surface RL
|
|
610
|
+
* consumers weight per-source instead of trusting one blended number.
|
|
611
|
+
*/
|
|
612
|
+
components: Record<string, number>;
|
|
812
613
|
/**
|
|
813
|
-
*
|
|
814
|
-
*
|
|
815
|
-
*
|
|
614
|
+
* @deprecated Read `components` for per-source reward values. Kept for
|
|
615
|
+
* published-API compatibility: single-source rewards carry the layer's
|
|
616
|
+
* diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
|
|
617
|
+
* the same per-layer scores `components` now holds.
|
|
816
618
|
*/
|
|
817
|
-
|
|
619
|
+
breakdown?: Record<string, number>;
|
|
818
620
|
}
|
|
819
|
-
|
|
820
|
-
* Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
|
|
821
|
-
* rubrics that don't predict deployment outcomes don't earn weight.
|
|
822
|
-
*/
|
|
823
|
-
declare class PredictiveValidityResearcher implements Researcher {
|
|
824
|
-
private opts;
|
|
825
|
-
private lastReport;
|
|
826
|
-
constructor(opts: PredictiveValidityResearcherOptions);
|
|
827
|
-
inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
|
|
828
|
-
proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
|
|
829
|
-
applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
|
|
830
|
-
evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
|
|
621
|
+
interface VerifiableRewardExtractionOptions {
|
|
831
622
|
/**
|
|
832
|
-
*
|
|
833
|
-
*
|
|
834
|
-
*
|
|
623
|
+
* Which layers count as deterministic-reward sources. The verifier doesn't
|
|
624
|
+
* tag layers as "this is verifiable"; the caller declares it via this list
|
|
625
|
+
* (or via the layer name → source mapping). Default treats common names
|
|
626
|
+
* (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
|
|
627
|
+
* `sandbox`) as deterministic.
|
|
835
628
|
*/
|
|
836
|
-
|
|
629
|
+
deterministicLayers?: string[];
|
|
837
630
|
/**
|
|
838
|
-
*
|
|
839
|
-
* useful when the consumer ran the report out-of-band and wants the
|
|
840
|
-
* researcher's later proposals informed by it.
|
|
631
|
+
* Map layer name → reward source. Defaults to a sensible string-match.
|
|
841
632
|
*/
|
|
842
|
-
|
|
843
|
-
|
|
633
|
+
sourceFor?: (layerName: string) => VerifiableRewardSource;
|
|
634
|
+
/**
|
|
635
|
+
* Whether to fall back to a probabilistic (judge) reward when no
|
|
636
|
+
* deterministic layer produced a numeric score. Default `true`. Set to
|
|
637
|
+
* `false` for "deterministic-only" training pipelines that should
|
|
638
|
+
* discard runs without a verifiable signal.
|
|
639
|
+
*/
|
|
640
|
+
fallbackToJudge?: boolean;
|
|
641
|
+
/**
|
|
642
|
+
* Default confidence for probabilistic (judge) rewards when the judge
|
|
643
|
+
* doesn't report one. Default `0.7`.
|
|
644
|
+
*/
|
|
645
|
+
judgeConfidenceFloor?: number;
|
|
844
646
|
}
|
|
647
|
+
/**
|
|
648
|
+
* Extract a `VerifiableReward` from a `VerificationReport`.
|
|
649
|
+
*
|
|
650
|
+
* Strategy: prefer the deterministic layers (in order: test → compile →
|
|
651
|
+
* schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
|
|
652
|
+
* true, return `null` if no signal qualifies. When multiple deterministic
|
|
653
|
+
* layers contribute, return a `'composite'` source with a weighted blend.
|
|
654
|
+
*/
|
|
655
|
+
declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
|
|
656
|
+
/**
|
|
657
|
+
* Extract verifiable rewards from `RunRecord[]` produced via the
|
|
658
|
+
* `verificationReportToRunRecord` adapter (which encodes per-layer scores
|
|
659
|
+
* in `outcome.raw['layer.<name>']`). For records that don't carry layer
|
|
660
|
+
* scores, returns `null` for that record.
|
|
661
|
+
*
|
|
662
|
+
* This is the canonical bridge from "campaign-shaped artifacts" to
|
|
663
|
+
* "RL-training-ready reward signals": every record that has a clean
|
|
664
|
+
* verifiable reward becomes a training datum, every record that doesn't
|
|
665
|
+
* gets filtered out (or kept with `'probabilistic'` determinism for
|
|
666
|
+
* separate downstream handling).
|
|
667
|
+
*/
|
|
668
|
+
declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
|
|
669
|
+
runId: string;
|
|
670
|
+
reward: VerifiableReward | null;
|
|
671
|
+
}>;
|
|
672
|
+
/** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
|
|
673
|
+
declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
|
|
674
|
+
run: RunRecord;
|
|
675
|
+
reward: VerifiableReward;
|
|
676
|
+
}>;
|
|
845
677
|
|
|
846
678
|
/**
|
|
847
679
|
* Reward hacking / Goodhart detection.
|
|
@@ -1023,6 +855,60 @@ interface RLCampaignResult<V> {
|
|
|
1023
855
|
}
|
|
1024
856
|
declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult<V>>;
|
|
1025
857
|
|
|
858
|
+
/**
|
|
859
|
+
* Adapters: convert measurement outputs into the canonical `RunRecord[]`
|
|
860
|
+
* artifact that `replayCache`, `pairedEvalueSequence`, and
|
|
861
|
+
* `rubricPredictiveValidity` consume. Two sources:
|
|
862
|
+
* - `campaignToRunRecords` — the campaign substrate's per-cell results
|
|
863
|
+
* (the modern path: `runCampaign` / `runImprovementLoop` → records).
|
|
864
|
+
* - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
|
|
865
|
+
*
|
|
866
|
+
* Adapters are thin and explicit — every mandatory `RunRecord` field comes
|
|
867
|
+
* from a caller-supplied context (`commitSha`, `model`, `promptHash`,
|
|
868
|
+
* `configHash`) plus the cell's runtime data. The validator still rejects
|
|
869
|
+
* bare-alias model strings — the caller snapshot-pins.
|
|
870
|
+
*/
|
|
871
|
+
|
|
872
|
+
interface AdapterContext {
|
|
873
|
+
/** Logical experiment id — typically the campaign or sweep identifier. */
|
|
874
|
+
experimentId: string;
|
|
875
|
+
/** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
|
|
876
|
+
model: string;
|
|
877
|
+
/** Git SHA the harness was run from. */
|
|
878
|
+
commitSha: string;
|
|
879
|
+
/** Hash of the effective prompt sent to the model. */
|
|
880
|
+
promptHash: string;
|
|
881
|
+
/** Hash of the effective config (model, temperature, tools, judges, splits). */
|
|
882
|
+
configHash: string;
|
|
883
|
+
/** Default split tag. Default `'search'`. */
|
|
884
|
+
splitTag?: RunSplitTag;
|
|
885
|
+
/** Default cost in USD when the source doesn't record one. Default `0`. */
|
|
886
|
+
defaultCostUsd?: number;
|
|
887
|
+
}
|
|
888
|
+
/**
|
|
889
|
+
* Convert a `CampaignResult` into canonical `RunRecord[]` — one record per
|
|
890
|
+
* scored cell. The cell's mean judge composite becomes the split score; every
|
|
891
|
+
* judge dimension is carried through to `outcome.raw`. A cell that errored
|
|
892
|
+
* becomes a record with `failureMode: 'cell_error'` (kept, not dropped — an
|
|
893
|
+
* unscored cell is signal). `candidateId` identifies the measured surface
|
|
894
|
+
* (defaults to the campaign manifest hash).
|
|
895
|
+
*/
|
|
896
|
+
declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterContext & {
|
|
897
|
+
candidateId?: string;
|
|
898
|
+
}): RunRecord[];
|
|
899
|
+
/**
|
|
900
|
+
* Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
|
|
901
|
+
* `outcome.searchScore` (or `holdoutScore`) is `report.blendedScore`;
|
|
902
|
+
* `outcome.raw` carries every layer's score + a pass indicator; `failureMode`
|
|
903
|
+
* is the first failing layer's reason.
|
|
904
|
+
*/
|
|
905
|
+
declare function verificationReportToRunRecord(report: VerificationReport, ctx: AdapterContext & {
|
|
906
|
+
candidateId: string;
|
|
907
|
+
scenarioId?: string;
|
|
908
|
+
}, opts?: {
|
|
909
|
+
runId?: string;
|
|
910
|
+
}): RunRecord;
|
|
911
|
+
|
|
1026
912
|
/**
|
|
1027
913
|
* Simulator fidelity — score a user SIMULATOR's realism against real-user
|
|
1028
914
|
* trace distributions.
|
|
@@ -1189,4 +1075,119 @@ interface EasyModeReport {
|
|
|
1189
1075
|
*/
|
|
1190
1076
|
declare function easyModeCheck(simulated: RunRecord[], production: RunRecord[], opts?: EasyModeOptions): EasyModeReport;
|
|
1191
1077
|
|
|
1078
|
+
/**
|
|
1079
|
+
* Bradley-Terry / Elo tournament evaluation.
|
|
1080
|
+
*
|
|
1081
|
+
* For multi-candidate sweeps, comparing every candidate's score against
|
|
1082
|
+
* a fixed comparator wastes information — the comparator becomes a high-
|
|
1083
|
+
* variance reference and rank flips between near-tied middle-rank
|
|
1084
|
+
* candidates are dominated by noise. Pairwise tournaments fix this:
|
|
1085
|
+
* every (i, j) pair contributes a comparison to a Bradley-Terry MLE that
|
|
1086
|
+
* estimates each candidate's strength on a unified scale.
|
|
1087
|
+
*
|
|
1088
|
+
* For online updating (rolling campaigns where new candidates arrive
|
|
1089
|
+
* over time), we also ship classical Elo with configurable K-factor.
|
|
1090
|
+
*
|
|
1091
|
+
* References:
|
|
1092
|
+
* - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete
|
|
1093
|
+
* block designs. Biometrika, 39(3/4), 324–345.
|
|
1094
|
+
* - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry
|
|
1095
|
+
* models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm
|
|
1096
|
+
* used here.)
|
|
1097
|
+
* - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.
|
|
1098
|
+
*
|
|
1099
|
+
* This is a useful primitive because most LLM-eval communities (Chatbot
|
|
1100
|
+
* Arena, AlpacaEval, ELO-style ablation) have converged on pairwise
|
|
1101
|
+
* tournament eval as the most sample-efficient and most rank-stable
|
|
1102
|
+
* method when you have many candidates.
|
|
1103
|
+
*/
|
|
1104
|
+
interface PairwiseOutcome {
|
|
1105
|
+
/** Winner candidate id. */
|
|
1106
|
+
winner: string;
|
|
1107
|
+
/** Loser candidate id. */
|
|
1108
|
+
loser: string;
|
|
1109
|
+
/**
|
|
1110
|
+
* Optional draw flag. When true, both candidates get half-credit
|
|
1111
|
+
* (Bradley-Terry handles draws as half-wins for each side).
|
|
1112
|
+
*/
|
|
1113
|
+
draw?: boolean;
|
|
1114
|
+
/**
|
|
1115
|
+
* Optional weight — useful if some pairwise comparisons are stronger
|
|
1116
|
+
* signals than others (e.g. a paired test with a wider score gap is
|
|
1117
|
+
* a more confident comparison). Default 1.
|
|
1118
|
+
*/
|
|
1119
|
+
weight?: number;
|
|
1120
|
+
}
|
|
1121
|
+
interface BradleyTerryRating {
|
|
1122
|
+
candidateId: string;
|
|
1123
|
+
/** Latent strength θ ≥ 0 from the BT MLE. */
|
|
1124
|
+
strength: number;
|
|
1125
|
+
/** Log-strength = log(θ) — interpretable on a linear scale. */
|
|
1126
|
+
logStrength: number;
|
|
1127
|
+
/** Number of pairwise comparisons this candidate appears in. */
|
|
1128
|
+
n: number;
|
|
1129
|
+
/** Win count (+ 0.5 per draw). */
|
|
1130
|
+
wins: number;
|
|
1131
|
+
}
|
|
1132
|
+
interface BradleyTerryFit {
|
|
1133
|
+
ratings: BradleyTerryRating[];
|
|
1134
|
+
/** Iterations of the MM algorithm before convergence. */
|
|
1135
|
+
iterations: number;
|
|
1136
|
+
/** Final maximum |θ_new - θ_old| / θ_old. */
|
|
1137
|
+
finalDelta: number;
|
|
1138
|
+
converged: boolean;
|
|
1139
|
+
}
|
|
1140
|
+
/**
|
|
1141
|
+
* Bradley-Terry MLE via Hunter's MM algorithm.
|
|
1142
|
+
*
|
|
1143
|
+
* Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
|
|
1144
|
+
* where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
|
|
1145
|
+
*
|
|
1146
|
+
* Returns log-strengths normalized so the smallest is 0 (any constant
|
|
1147
|
+
* offset is unobservable in BT — only differences are identified).
|
|
1148
|
+
*/
|
|
1149
|
+
declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
|
|
1150
|
+
tolerance?: number;
|
|
1151
|
+
maxIterations?: number;
|
|
1152
|
+
smoothing?: number;
|
|
1153
|
+
}): BradleyTerryFit;
|
|
1154
|
+
/**
|
|
1155
|
+
* Online Elo updates. Use when comparisons arrive over time and you want
|
|
1156
|
+
* a running rating without re-fitting the full BT MLE on every update.
|
|
1157
|
+
*
|
|
1158
|
+
* Initialize ratings to `defaultRating` (1500 by default). Each call to
|
|
1159
|
+
* `applyEloUpdate` mutates the map in place and returns the deltas so
|
|
1160
|
+
* the caller can log per-comparison rating changes.
|
|
1161
|
+
*/
|
|
1162
|
+
interface EloOptions {
|
|
1163
|
+
/** Default rating for unseen candidates. Default 1500. */
|
|
1164
|
+
defaultRating?: number;
|
|
1165
|
+
/** K-factor controls the step size. Default 32 (FIDE-ish). */
|
|
1166
|
+
kFactor?: number;
|
|
1167
|
+
}
|
|
1168
|
+
declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseOutcome, opts?: EloOptions): {
|
|
1169
|
+
winnerDelta: number;
|
|
1170
|
+
loserDelta: number;
|
|
1171
|
+
};
|
|
1172
|
+
/**
|
|
1173
|
+
* Build pairwise outcomes from the campaign artifact: for every scenario
|
|
1174
|
+
* shared by two candidates, the higher-scoring run wins. Useful when you
|
|
1175
|
+
* want a tournament view of an existing campaign without an additional
|
|
1176
|
+
* pairwise judge call.
|
|
1177
|
+
*/
|
|
1178
|
+
interface BuildPairwiseFromCampaignInput {
|
|
1179
|
+
runs: Array<{
|
|
1180
|
+
candidateId: string;
|
|
1181
|
+
/** Stable identifier for the matching unit (typically scenarioId). */
|
|
1182
|
+
matchKey: string;
|
|
1183
|
+
score: number;
|
|
1184
|
+
}>;
|
|
1185
|
+
/**
|
|
1186
|
+
* Tied-score margin. Below this, the comparison is a draw. Default 0
|
|
1187
|
+
* (no ties).
|
|
1188
|
+
*/
|
|
1189
|
+
drawMargin?: number;
|
|
1190
|
+
}
|
|
1191
|
+
declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
|
|
1192
|
+
|
|
1192
1193
|
export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DetectRewardHackingInput, DpoExportRow, DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, ExtractPreferencesOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, GrpoExportRow, GrpoLookups, OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, PreferenceExtractionReport, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, SftExportRow, SftLookups, type SimFidelityOptions, type ThompsonCurriculumOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, applyEloUpdate, bestOfN, bucketLabel, buildPairwiseFromCampaign, campaignToRunRecords, compareAdaptationCurves, defaultBehaviorFeatures, detectRewardHacking, easyModeCheck, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, jsDivergence, observationsFromRunRecords, paretoFrontier, quantileEdges, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runRLCampaign, selfConsistency, shuffleOrder, simFidelityReport, thompsonCurriculum, varianceBasedCurriculum, verificationReportToRunRecord };
|