@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
package/dist/rl.d.ts DELETED
@@ -1,4092 +0,0 @@
1
- type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
2
- type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
3
- [key: string]: AgentProfileJson;
4
- };
5
- type AgentProfileDimensionValue = string | number | boolean | null;
6
- interface AgentProfileSource {
7
- /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
8
- kind: string;
9
- /** sha256 over the canonical source profile object. */
10
- hash: string;
11
- }
12
- interface AgentProfileSourceInput {
13
- kind: string;
14
- /** Precomputed sha256 for callers that already sign their profile artifact. */
15
- hash?: string;
16
- /** Full canonical runtime profile; hashed and then discarded from the cell. */
17
- profile?: AgentProfileJson;
18
- }
19
- interface AgentProfileHarness {
20
- id: string;
21
- version?: string;
22
- hash?: string;
23
- }
24
- interface AgentProfileCellInput {
25
- profileId: string;
26
- sourceProfile: AgentProfileSourceInput;
27
- harness?: AgentProfileHarness;
28
- model?: string;
29
- promptHash?: string;
30
- dimensions?: Record<string, AgentProfileDimensionValue>;
31
- }
32
- interface AgentProfileCell {
33
- schemaVersion: AgentProfileCellSchemaVersion;
34
- cellId: string;
35
- profileId: string;
36
- sourceProfile: AgentProfileSource;
37
- harness?: AgentProfileHarness;
38
- model?: string;
39
- promptHash?: string;
40
- dimensions?: Record<string, AgentProfileDimensionValue>;
41
- }
42
-
43
- type RunStatus = 'running' | 'completed' | 'failed' | 'aborted';
44
- interface BudgetSpec {
45
- tokens?: number;
46
- wallMs?: number;
47
- calls?: number;
48
- usd?: number;
49
- }
50
- interface RunOutcome$1 {
51
- score?: number;
52
- pass?: boolean;
53
- failureClass?: FailureClass;
54
- notes?: string;
55
- }
56
- /**
57
- * Layer — optional classification in a nested build workflow.
58
- * `builder`: the meta-agent editing a project (e.g. agent-builder Forge chat).
59
- * `app-build`: sandbox harness that compiled + tested the generated scaffold.
60
- * `app-runtime`: a run of the generated agent against a domain scenario.
61
- * `meta`: any meta-eval (judge replay, correlation analysis).
62
- */
63
- type RunLayer = 'builder' | 'app-build' | 'app-runtime' | 'meta' | 'custom';
64
- interface Run {
65
- runId: string;
66
- /**
67
- * Stable identifier of the scenario being executed.
68
- *
69
- * Always populated on the persisted Run — but `TraceEmitter.startRun` accepts
70
- * input WITHOUT this field, substituting a sensible default
71
- * (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) when the caller has no
72
- * curated scenario to anchor to (runtime / operator / meta-eval runs). This
73
- * keeps the persisted shape unambiguous for downstream filters + aggregations
74
- * while removing the boilerplate of inventing placeholder ids at the call site.
75
- */
76
- scenarioId: string;
77
- variantId?: string;
78
- datasetVersion?: string;
79
- /** Git SHA of agent code at run time. */
80
- codeSha?: string;
81
- /** Hash of the prompt template + any system prompt. */
82
- promptSha?: string;
83
- /** Model id + date + system-prompt hash, concatenated. */
84
- modelFingerprint?: string;
85
- seed?: number;
86
- /** Arbitrary environment markers (shell, docker version, tz). */
87
- envFingerprint?: Record<string, string>;
88
- /** Version of the redaction rules applied to this run. */
89
- redactionVersion?: string;
90
- /** Parent run in a nested build workflow. A builder run's children are
91
- * app-build runs; those children are app-runtime runs. */
92
- parentRunId?: string;
93
- /** Stable project identifier — groups runs across chats + sessions. */
94
- projectId?: string;
95
- /** Chat/conversation identifier within a project. */
96
- chatId?: string;
97
- /** Layer classification — hint for aggregation; not enforced. */
98
- layer?: RunLayer;
99
- startedAt: number;
100
- endedAt?: number;
101
- status: RunStatus;
102
- outcome?: RunOutcome$1;
103
- budget?: BudgetSpec;
104
- /** Free-form labels for downstream grouping. */
105
- tags?: Record<string, string>;
106
- }
107
- type SpanKind = 'agent' | 'llm' | 'tool' | 'retrieval' | 'judge' | 'sandbox' | 'custom';
108
- type SpanStatus = 'ok' | 'error';
109
- interface SpanBase {
110
- spanId: string;
111
- parentSpanId?: string;
112
- runId: string;
113
- kind: SpanKind;
114
- name: string;
115
- startedAt: number;
116
- endedAt?: number;
117
- status?: SpanStatus;
118
- error?: string;
119
- /** Anything not covered by typed fields. Kept deliberately free-form. */
120
- attributes?: Record<string, unknown>;
121
- }
122
- interface Message {
123
- role: 'system' | 'user' | 'assistant' | 'tool';
124
- content: string;
125
- tokens?: number;
126
- /** Multi-modal content descriptors; blobs themselves live in Artifacts. */
127
- images?: Array<{
128
- artifactId?: string;
129
- url?: string;
130
- mime?: string;
131
- }>;
132
- }
133
- interface LlmSpan extends SpanBase {
134
- kind: 'llm';
135
- model: string;
136
- messages: Message[];
137
- output?: string;
138
- inputTokens?: number;
139
- /** All generated tokens, including the reasoning subset when present. */
140
- outputTokens?: number;
141
- cachedTokens?: number;
142
- cacheWriteTokens?: number;
143
- /** Reasoning-token subset of `outputTokens`. */
144
- reasoningTokens?: number;
145
- costUsd?: number;
146
- finishReason?: string;
147
- }
148
- interface ToolSpan extends SpanBase {
149
- kind: 'tool';
150
- toolName: string;
151
- args: unknown;
152
- /** False when the source observed the call but did not capture its arguments. */
153
- argsCaptured?: boolean;
154
- result?: unknown;
155
- latencyMs?: number;
156
- }
157
- interface RetrievalSpan extends SpanBase {
158
- kind: 'retrieval';
159
- query: string;
160
- hits: Array<{
161
- docId: string;
162
- score: number;
163
- content?: string;
164
- }>;
165
- }
166
- interface JudgeSpan extends SpanBase {
167
- kind: 'judge';
168
- judgeId: string;
169
- /** Span this judgment applies to. */
170
- targetSpanId: string;
171
- dimension: string;
172
- /** Numeric score (free-range; interpretation up to the judge). */
173
- score: number;
174
- rationale?: string;
175
- evidence?: string;
176
- }
177
- interface SandboxSpan extends SpanBase {
178
- kind: 'sandbox';
179
- image?: string;
180
- command?: string;
181
- exitCode?: number;
182
- testsTotal?: number;
183
- testsPassed?: number;
184
- stdoutHash?: string;
185
- stderrHash?: string;
186
- /** Duration in ms; the harness fills this explicitly (endedAt - startedAt may miss setup). */
187
- wallMs?: number;
188
- }
189
- interface GenericSpan extends SpanBase {
190
- kind: 'agent' | 'custom';
191
- }
192
- type Span = LlmSpan | ToolSpan | RetrievalSpan | JudgeSpan | SandboxSpan | GenericSpan;
193
- type EventKind = 'log' | 'error' | 'budget_decrement' | 'budget_breach' | 'state_mutation' | 'policy_violation' | 'redaction_applied' | 'custom';
194
- interface TraceEvent {
195
- eventId: string;
196
- runId: string;
197
- spanId?: string;
198
- kind: EventKind;
199
- timestamp: number;
200
- payload: Record<string, unknown>;
201
- }
202
- interface BudgetLedgerEntry {
203
- runId: string;
204
- dimension: keyof BudgetSpec;
205
- limit: number;
206
- consumed: number;
207
- remaining: number;
208
- timestamp: number;
209
- breached: boolean;
210
- /** Span that triggered this entry, if any. */
211
- spanId?: string;
212
- }
213
- interface Artifact {
214
- artifactId: string;
215
- runId: string;
216
- spanId?: string;
217
- contentType: string;
218
- sizeBytes: number;
219
- /** sha256 in hex. */
220
- hash: string;
221
- /** External storage URL (R2, S3, filesystem path). */
222
- storageUrl?: string;
223
- /** Inline content for small blobs — keep under ~64KB. */
224
- inlineContent?: string;
225
- }
226
- type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
227
-
228
- /**
229
- * Paper-grade RunRecord schema + runtime validator.
230
- *
231
- * Every run that participates in a promotion gate, paper table, or
232
- * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
233
- * fields are exactly those the paper "Two Loops, Three Roles" requires
234
- * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
235
- * holdout split tag and either a `searchScore` or a `holdoutScore`.
236
- *
237
- * This is intentionally NOT a replacement for the rich `Run` /
238
- * `ProposeReviewReport` / `ScenarioResult` types already in the
239
- * package. Those are runtime structures with full provenance. A
240
- * `RunRecord` is the analysis-time projection — the JSON-friendly
241
- * row you'd put in a parquet file or paste into a notebook.
242
- *
243
- * Validate at the boundary:
244
- *
245
- * const rec = validateRunRecord(rawJson) // throws on missing
246
- * const ok = isRunRecord(rawJson) // boolean check
247
- * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
248
- *
249
- * The validator runs in pure TS — zod is intentionally NOT a
250
- * dependency. Round-trip tested in `tests/run-record.test.ts`.
251
- */
252
-
253
- /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
254
- * combined train+test pool that the optimizer is allowed to read. */
255
- type RunSplitTag = 'search' | 'dev' | 'holdout';
256
- interface RunTokenUsage {
257
- input: number;
258
- /** All generated tokens charged as output, including reasoning tokens. */
259
- output: number;
260
- /** Reasoning-token subset of `output`, when the provider reports it. */
261
- reasoning?: number;
262
- /** Prompt tokens served from a provider cache. */
263
- cached?: number;
264
- /** Prompt tokens written into a provider cache. */
265
- cacheWrite?: number;
266
- }
267
- /**
268
- * How a run's USD amount was obtained.
269
- *
270
- * `costUsd` remains mandatory for wire compatibility. New producers should
271
- * always populate this discriminated union so a missing bill is never
272
- * mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
273
- * the legacy `0` sentinel while this field carries the truthful null.
274
- */
275
- type RunCostProvenance = {
276
- kind: 'observed';
277
- usd: number;
278
- } | {
279
- kind: 'estimated';
280
- usd: number;
281
- } | {
282
- kind: 'uncaptured';
283
- usd: null;
284
- };
285
- interface RunJudgeMetadata {
286
- model: string;
287
- promptVersion: string;
288
- /** [0,1] confidence the judge declared. Constant judge confidence
289
- * across many runs is a fallback signal (see `canary.ts`). */
290
- confidence: number;
291
- /** True if the judge degraded to a fallback path (rules-only,
292
- * prior-call cache, etc.). The canary uses this to alert. */
293
- fallback: boolean;
294
- }
295
- /**
296
- * Per-judge / per-dimension breakdown for runs scored by an ensemble of
297
- * judges over a multi-dimensional rubric.
298
- *
299
- * The collapsed `outcome.searchScore` / `holdoutScore` carries the
300
- * composite the gate uses. The full breakdown belongs here so consumers
301
- * can answer "which judge disagreed?", "which dimension dragged the
302
- * composite down?", and "did half the panel fail?" without re-running.
303
- *
304
- * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
305
- * `composite` are convenience projections — derivable but precomputed so
306
- * downstream IRR primitives (`interRaterReliability`,
307
- * `corpusInterRaterAgreement`) and reporters don't pay the same
308
- * aggregation twice.
309
- *
310
- * Fail-loud discipline: judges that errored out land in `failedJudges`
311
- * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
312
- * run); the explicit list makes a partial-failure recorded as such.
313
- */
314
- interface JudgeScoresRecord {
315
- /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
316
- perJudge: Record<string, Record<string, number>>;
317
- /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
318
- perDimMean: Record<string, number>;
319
- /** Composite mean across all dims and judges. Mirrors the score
320
- * the gate sees on `outcome.searchScore` / `holdoutScore`. */
321
- composite: number;
322
- /** Judges that errored or returned an unparseable verdict. Recorded
323
- * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
324
- * not inferred from missing keys in `perJudge`. */
325
- failedJudges?: string[];
326
- /** Free-form notes the judges emitted (joined across judges or
327
- * first-judge only — consumer's choice). */
328
- notes?: string;
329
- }
330
- interface RunOutcome {
331
- /** Score on the search/optimization split. Optional because a
332
- * holdout-only evaluation only fills `holdoutScore`. */
333
- searchScore?: number;
334
- /** Score on the held-out split. Optional because a search-only run
335
- * only fills `searchScore`. At least one must be present. */
336
- holdoutScore?: number;
337
- /** Bag of any other metric the run produced — judge dimensions,
338
- * pass/fail counters, latency stats, etc. Numeric only — keeps
339
- * reporters honest. */
340
- raw: Record<string, number>;
341
- /** Per-judge / per-dim breakdown. Consumers writing ensemble
342
- * judgements populate this; substrate primitives like
343
- * `interRaterReliability` and `corpusInterRaterAgreement` accept
344
- * these records as input. Optional — single-judge or scalar-only
345
- * runs leave it unset. */
346
- judgeScores?: JudgeScoresRecord;
347
- /** Authenticity / realness verdict — did the run build the REAL thing on the
348
- * intended infra, or fake it (see `./authenticity`)? Optional: only domains
349
- * with an authenticity config populate it. Carried in the corpus so the
350
- * flywheel / off-policy learning can optimize for real completion, not gamed
351
- * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
352
- * must not count as a real success regardless of `score`. */
353
- realness?: {
354
- score: number;
355
- gated: boolean;
356
- reason?: string;
357
- };
358
- }
359
- /**
360
- * Mandatory paper-grade fields for a single evaluation run. Optional
361
- * fields are extension points; mandatory fields throw if missing.
362
- *
363
- * Hash discipline:
364
- * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
365
- * model (after any steering bundle merge).
366
- * - `configHash` is the sha256 of the effective run config (model,
367
- * temperature, tools, judges, splits). The pair (promptHash,
368
- * configHash) uniquely identifies an experiment cell.
369
- *
370
- * Model snapshot discipline:
371
- * - `model` MUST encode a snapshot version. Bare aliases like
372
- * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
373
- * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
374
- */
375
- interface RunRecord {
376
- /** UUID for the run. */
377
- runId: string;
378
- /** Logical experiment grouping (a treatment vs a baseline within
379
- * the same sweep should share `experimentId`). */
380
- experimentId: string;
381
- /** Stable identifier for the candidate (variant) being run. The
382
- * promotion gate compares two `candidateId`s on matched items. */
383
- candidateId: string;
384
- /** RNG seed for the run. Always recorded — silent re-seeding is
385
- * the most common cause of non-reproducible numbers. */
386
- seed: number;
387
- /** Model identifier WITH snapshot version. */
388
- model: string;
389
- /** sha256 of the effective prompt (post-steering). */
390
- promptHash: string;
391
- /** sha256 of the effective config. */
392
- configHash: string;
393
- /** Git SHA the harness was run from. */
394
- commitSha: string;
395
- /** End-to-end wall-clock duration in milliseconds. */
396
- wallMs: number;
397
- /** Time spent queued before execution started, if known. */
398
- queueMs?: number;
399
- /** Total USD cost. Mandatory — runs without a cost number are
400
- * unbounded by definition and must not be admitted into the gate.
401
- * `0` is retained as the compatibility sentinel for an uncaptured amount;
402
- * inspect `costProvenance` before treating it as observed. */
403
- costUsd: number;
404
- /** Observed, model-priced estimate, or genuinely uncaptured USD amount.
405
- * Optional only so existing serialized RunRecords remain valid. */
406
- costProvenance?: RunCostProvenance;
407
- /** Token usage breakdown. */
408
- tokenUsage: RunTokenUsage;
409
- /** Judge-side metadata, if a judge was used. */
410
- judgeMetadata?: RunJudgeMetadata;
411
- /** Per-split scores + raw bag. */
412
- outcome: RunOutcome;
413
- /** Canonical, cross-agent failure class drawn from the shared
414
- * `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
415
- * "which failure dominates across the whole fleet" answerable in ONE
416
- * vocabulary — every agent classifies against the same enum. Producers
417
- * set it via the substrate classifier; leave unset only when the failure
418
- * genuinely can't be classified. */
419
- failureClass?: FailureClass;
420
- /** Free-form domain-specific failure detail, scoped UNDER `failureClass`
421
- * (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
422
- * The within-agent drill-down; `failureClass` is the cross-agent key. */
423
- failureMode?: string;
424
- /** Which split this run was drawn from. */
425
- splitTag: RunSplitTag;
426
- /**
427
- * Stable scenario identifier the run was scored against. Optional for
428
- * backwards compatibility, but **strongly recommended**: every primitive
429
- * that pairs runs by scenario (preferences, paired stats, BT tournament)
430
- * keys on this. The campaign artifact populates it canonically; legacy
431
- * runs without it fall back to inference from `outcome.raw.scenario_id`
432
- * or `experimentId`.
433
- */
434
- scenarioId?: string;
435
- /**
436
- * Canonical identity for the agent profile cell that produced this row:
437
- * profile artifact hash plus optional harness/model/prompt/reporting
438
- * dimensions. Use `agentProfile.cellId` to group persona sweeps and
439
- * longitudinal reports by the complete source profile, not by a loose
440
- * candidate label or opaque config hash.
441
- */
442
- agentProfile?: AgentProfileCell;
443
- }
444
-
445
- /**
446
- * Adaptive curriculum / active scenario selection.
447
- *
448
- * Fixed scenario sets waste sample budget on cells the policy already
449
- * passes (no information left) and cells the policy never passes (no
450
- * gradient available either). Active learning over scenarios fixes this
451
- * by allocating the next sample budget to cells where the policy's
452
- * outcome is *uncertain* — those carry the most decision-relevant signal.
453
- *
454
- * This module ships two complementary strategies:
455
- *
456
- * 1. **Variance-based** — score each (variant, scenario) cell by the
457
- * empirical variance of past observations. Allocate next-round budget
458
- * proportional to variance. Standard active-learning-by-uncertainty
459
- * heuristic; works well when the policy is non-deterministic and
460
- * cells differ in observation noise.
461
- *
462
- * 2. **Bandit-based (Thompson sampling)** — model each (variant,
463
- * scenario) cell as a Beta-Bernoulli arm; sample a posterior; pick
464
- * cells whose posterior mean is closest to the per-scenario decision
465
- * threshold. The right primitive when scenarios are
466
- * "pass/fail" rather than continuous, and when promotion gates fire
467
- * at a known threshold (e.g., 0.5).
468
- *
469
- * The output is a *next-round budget allocation* — a list of (variant,
470
- * scenario, count) triples. The consumer's matrix runner consumes the
471
- * allocation, runs those cells, feeds the new observations back. Loop.
472
- *
473
- * Out of scope (deliberate): scenario *generation* — that's the
474
- * adversarial primitive's job. This module allocates over an existing
475
- * scenario pool.
476
- */
477
-
478
- interface CellObservation {
479
- variantId: string;
480
- scenarioId: string;
481
- /** Observed score in [0, 1]. */
482
- score: number;
483
- /** For Bernoulli arms — derive from the score with a threshold if needed. */
484
- pass?: boolean;
485
- }
486
- interface CurriculumAllocation {
487
- variantId: string;
488
- scenarioId: string;
489
- /** How many additional reps to run on this cell. */
490
- count: number;
491
- /** Strategy-specific reason for the allocation. */
492
- reason: string;
493
- }
494
- interface VarianceCurriculumOptions {
495
- /** Total reps to allocate across all cells. */
496
- budget: number;
497
- /**
498
- * Smoothing prior on variance — keeps the allocator from concentrating
499
- * on a cell with one observation just because its 1-sample variance is
500
- * 0. Default 0.05.
501
- */
502
- variancePrior?: number;
503
- /**
504
- * Minimum reps per cell — even when the variance estimate is low, give
505
- * every cell at least this many. Default 1.
506
- */
507
- floorPerCell?: number;
508
- }
509
- /**
510
- * Variance-proportional allocation. For each cell, estimate variance from
511
- * past observations + a prior, then allocate the budget proportional to
512
- * (sqrt(variance) + 1/sqrt(n)) — a classical optimal-allocation rule
513
- * (Neyman 1934) that balances "explore noisy cells" with "explore
514
- * under-sampled cells."
515
- */
516
- declare function varianceBasedCurriculum(observations: CellObservation[], candidateCells: Array<{
517
- variantId: string;
518
- scenarioId: string;
519
- }>, opts: VarianceCurriculumOptions): CurriculumAllocation[];
520
- interface ThompsonCurriculumOptions {
521
- budget: number;
522
- /**
523
- * The per-scenario decision threshold. Cells whose posterior mean is
524
- * closest to this get the most budget — that's where the next observation
525
- * has the highest information value for the gate decision. Default 0.5.
526
- */
527
- decisionThreshold?: number;
528
- /** Beta prior parameters. Default α=β=1 (uniform). */
529
- priorAlpha?: number;
530
- priorBeta?: number;
531
- /** Seed the Thompson sampler. Default unset (Math.random). */
532
- seed?: number;
533
- }
534
- /**
535
- * Thompson-sampling-style allocation for pass/fail cells. For each cell:
536
- *
537
- * - Maintain Beta(α + passes, β + failures) posterior on pass-rate
538
- * - Allocation weight ∝ exp(-((sampledMean - threshold) / σ)^2):
539
- * cells whose sampled posterior straddles the decision boundary get
540
- * the most weight; cells already clearly above or below get less.
541
- *
542
- * This is the right primitive when promotion gates fire at a known
543
- * threshold and you want to sharpen the posterior near the boundary.
544
- */
545
- declare function thompsonCurriculum(observations: CellObservation[], candidateCells: Array<{
546
- variantId: string;
547
- scenarioId: string;
548
- }>, opts: ThompsonCurriculumOptions): CurriculumAllocation[];
549
- /** Convenience: extract `CellObservation[]` directly from `RunRecord[]`. */
550
- declare function observationsFromRunRecords(runs: RunRecord[], opts?: {
551
- passThreshold?: number;
552
- useHoldout?: boolean;
553
- }): CellObservation[];
554
-
555
- /**
556
- * Sample-efficient adaptation evaluation.
557
- *
558
- * For foundation-model-based agents, the load-bearing capability isn't
559
- * raw end-state performance — it's *how fast the agent reaches that
560
- * performance from cold start*. The same model with a worse prompt that
561
- * adapts in 5 demonstrations beats the same model with a better prompt
562
- * that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
563
- * reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
564
- * in-context examples or fine-tune steps.
565
- *
566
- * This module ships:
567
- *
568
- * 1. `runAdaptationCurve` — given a runner that takes k demonstrations
569
- * and returns a score, produce the (k, score) curve.
570
- * 2. `compareAdaptationCurves` — paired comparison across two policies.
571
- * Returns per-k delta with bootstrap CIs and an "area-under-curve"
572
- * summary statistic.
573
- * 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
574
- * the policy reliably passes (≥ pass-rate threshold over reps).
575
- *
576
- * Use cases:
577
- * - Compare two prompt designs that have similar end-state performance
578
- * but different in-context efficiency.
579
- * - Decide between fine-tuning and prompting based on adaptation cost.
580
- * - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
581
- */
582
- interface AdaptationRunner<S> {
583
- /**
584
- * Runs the policy on `scenario` with `k` demonstrations. Returns a
585
- * scalar score in [0, 1]. The runner is responsible for any caching;
586
- * the harness calls it once per (scenario, k, rep) cell.
587
- */
588
- run(args: {
589
- scenario: S;
590
- k: number;
591
- rep: number;
592
- }): Promise<number>;
593
- }
594
- interface RunAdaptationCurveOptions<S> {
595
- scenarios: S[];
596
- /** Number-of-shots to evaluate at. Default `[0, 1, 2, 4, 8, 16]`. */
597
- ks?: number[];
598
- /** Reps per (scenario, k) cell. Default 3. */
599
- reps?: number;
600
- runner: AdaptationRunner<S>;
601
- /** Pass-rate threshold for `firstPassK` reporting. Default 0.5. */
602
- passThreshold?: number;
603
- }
604
- interface AdaptationPoint {
605
- k: number;
606
- meanScore: number;
607
- passRate: number;
608
- std: number;
609
- n: number;
610
- /** Per-scenario means at this k. */
611
- perScenario: Array<{
612
- scenarioId: string;
613
- meanScore: number;
614
- passes: number;
615
- total: number;
616
- }>;
617
- }
618
- interface AdaptationCurve {
619
- points: AdaptationPoint[];
620
- /**
621
- * Smallest `k` at which `passRate ≥ passThreshold`. `null` if no `k`
622
- * tested reaches it.
623
- */
624
- firstPassK: number | null;
625
- /**
626
- * Area under the (k, meanScore) curve, normalized by max-k. A
627
- * single-number summary of "how well does this policy adapt from
628
- * cold-start to fully-conditioned." Higher = better adapter.
629
- */
630
- adaptationArea: number;
631
- }
632
- declare function runAdaptationCurve<S extends {
633
- scenarioId?: string;
634
- }>(opts: RunAdaptationCurveOptions<S>): Promise<AdaptationCurve>;
635
- interface CompareCurvesResult {
636
- perK: Array<{
637
- k: number;
638
- deltaMean: number;
639
- aLow: number;
640
- aHigh: number;
641
- bLow: number;
642
- bHigh: number;
643
- }>;
644
- areaDelta: number;
645
- firstPassKDelta: number | null;
646
- /** Verdict: 'a_better' | 'b_better' | 'similar'. */
647
- verdict: 'a_better' | 'b_better' | 'similar';
648
- /** Rationale, ready to render. */
649
- rationale: string;
650
- }
651
- /**
652
- * Paired comparison of two adaptation curves. Per-k deltas with 95%
653
- * bootstrap CIs (constructed from each curve's `perScenario` per-k means
654
- * — the bootstrap unit is the scenario, not the rep).
655
- */
656
- declare function compareAdaptationCurves(a: AdaptationCurve, b: AdaptationCurve, opts?: {
657
- confidence?: number;
658
- bootstrapResamples?: number;
659
- seed?: number;
660
- }): CompareCurvesResult;
661
- /** First k at which the curve's per-scenario pass rate reliably hits the threshold. */
662
- declare function firstPassK(curve: AdaptationCurve, threshold?: number): number | null;
663
-
664
- /**
665
- * Adversarial mutation contract.
666
- *
667
- * `AdversarialMutation<S>` is the scenario-mutation strategy the fuzz harness
668
- * (`fuzzAgent`, src/fuzz) drives: paraphrase, edge-case substitution, or
669
- * compositional combination of a scenario the policy currently passes, looking
670
- * for the tail inputs that break it. The harness supplies the loop; consumers
671
- * supply the mutations and the failure detector.
672
- */
673
- interface AdversarialMutation<S> {
674
- id: string;
675
- /**
676
- * Mutate one scenario. Return null to skip; return one or more new
677
- * scenarios. The harness deduplicates by `mutateScenarioId(scenario)`.
678
- */
679
- mutate(parent: S, rng: () => number): Promise<S[]> | S[];
680
- }
681
-
682
- /**
683
- * Test-time compute scaling curves.
684
- *
685
- * The test-time-compute frontier paper (Snell et al. 2024) and the
686
- * subsequent o1-style scaling work both show that LLM-agent capability
687
- * is a function of the compute budget at inference, not just of the
688
- * training run. The right way to characterize a candidate is therefore
689
- * a *curve* — score at compute budgets {1×, 4×, 16×, …} — not a single
690
- * point.
691
- *
692
- * This module ships:
693
- *
694
- * 1. The compute-curve harness — `runComputeCurve(runner, budgets)` —
695
- * that evaluates one candidate at a sequence of compute budgets
696
- * and returns the (compute, score) curve.
697
- * 2. A best-of-N evaluator — `bestOfN(runner, n, scoreFn)` — the
698
- * simplest test-time-compute scaling primitive: sample N
699
- * independent rollouts, return the best.
700
- * 3. A self-consistency evaluator — `selfConsistency(runner, n)` —
701
- * the majority-vote variant of best-of-N for tasks with a small
702
- * categorical answer space.
703
- * 4. Pareto-frontier extraction over multiple candidates — given
704
- * (candidate, compute, score) tuples, return the set of
705
- * candidate-compute combinations that aren't dominated.
706
- *
707
- * Caveat: "compute" here is the caller's notion of a compute unit. For
708
- * agent eval that's typically wall-time × parallelism, or token budget,
709
- * or LLM-call count. We accept whatever the caller provides; the curve
710
- * is on whatever axis they pick.
711
- */
712
- interface ComputeCurveBudget {
713
- /** Identifier — for the report. Common: '1x', '4x', '16x'. */
714
- id: string;
715
- /** Numeric value on the chosen axis (tokens, calls, USD, ms — caller picks). */
716
- cost: number;
717
- /** Free-form metadata (the caller can carry per-budget config). */
718
- meta?: Record<string, unknown>;
719
- }
720
- interface ComputeCurvePoint {
721
- budgetId: string;
722
- cost: number;
723
- score: number;
724
- /** Number of underlying samples used at this budget. */
725
- samples: number;
726
- /** Optional spread / variance information. */
727
- std?: number;
728
- /** Any extra metrics the runner returned. */
729
- metrics?: Record<string, number>;
730
- }
731
- interface ComputeCurve {
732
- candidateId: string;
733
- points: ComputeCurvePoint[];
734
- /** Rough exponent fit: score ≈ a + b * log(cost). Useful for "how steep is the curve?" */
735
- logSlope: number | null;
736
- /** Best (highest-score) point on the curve. */
737
- best: ComputeCurvePoint;
738
- }
739
- interface RunComputeCurveOptions {
740
- candidateId: string;
741
- budgets: ComputeCurveBudget[];
742
- /**
743
- * Run the candidate at one budget. Returns the realized score plus
744
- * optional spread + extra metrics.
745
- */
746
- runAtBudget: (budget: ComputeCurveBudget) => Promise<{
747
- score: number;
748
- samples: number;
749
- std?: number;
750
- metrics?: Record<string, number>;
751
- }>;
752
- }
753
- declare function runComputeCurve(opts: RunComputeCurveOptions): Promise<ComputeCurve>;
754
- interface ComputeBestOfNOptions<O> {
755
- /** Number of independent samples to draw. */
756
- n: number;
757
- /** Sampler — produces one rollout. */
758
- sample: (sampleIdx: number) => Promise<O>;
759
- /** Score one rollout. */
760
- scoreFn: (rollout: O) => Promise<number> | number;
761
- }
762
- interface ComputeBestOfNResult<O> {
763
- best: O;
764
- bestScore: number;
765
- scores: number[];
766
- meanScore: number;
767
- /** Index of the best rollout, for diagnostics. */
768
- bestIndex: number;
769
- }
770
- /** The simplest test-time scaling primitive. */
771
- declare function bestOfN<O>(opts: ComputeBestOfNOptions<O>): Promise<ComputeBestOfNResult<O>>;
772
- interface SelfConsistencyOptions<O> {
773
- n: number;
774
- sample: (sampleIdx: number) => Promise<O>;
775
- /** Extract the canonical answer key (string) from a rollout. */
776
- answerKey: (rollout: O) => string;
777
- }
778
- interface SelfConsistencyResult<O> {
779
- /** Modal answer (the majority vote). */
780
- answer: string;
781
- /** Fraction of samples voting for the modal answer in [0, 1]. */
782
- agreement: number;
783
- /** Histogram of all answers. */
784
- histogram: Record<string, number>;
785
- /** A representative rollout that voted for the modal answer. */
786
- representative: O;
787
- /** All rollouts. */
788
- rollouts: O[];
789
- }
790
- /**
791
- * Self-consistency / majority-vote test-time scaling. For tasks with a
792
- * small categorical answer space (math problems, multiple choice).
793
- */
794
- declare function selfConsistency<O>(opts: SelfConsistencyOptions<O>): Promise<SelfConsistencyResult<O>>;
795
- /**
796
- * Pareto frontier over (candidate, compute, score) tuples. A point is on
797
- * the frontier iff no other point dominates it in both score (higher
798
- * better) and cost (lower better). Returns the frontier sorted ascending
799
- * by cost.
800
- */
801
- interface ParetoPointInput {
802
- candidateId: string;
803
- budgetId: string;
804
- cost: number;
805
- score: number;
806
- }
807
- declare function paretoFrontier(points: ParetoPointInput[]): ParetoPointInput[];
808
-
809
- /**
810
- * Contamination probe — held-out perturbation tests.
811
- *
812
- * The bug class: once a benchmark scenario set is published, models train
813
- * on it, and your scores become invalid. SWE-Bench-Verified, GPQA, and
814
- * MMLU-Pro all exist because their predecessors got contaminated within
815
- * months. The right defense is to keep a held-out *perturbed* version of
816
- * every scenario — same task, slightly different surface — and check
817
- * whether scores diverge significantly. Genuine capability transfers; rote
818
- * memorization doesn't.
819
- *
820
- * This module ships the probe contract:
821
- *
822
- * 1. A `ScenarioPerturbation` strategy type — function that produces a
823
- * perturbed scenario from an original.
824
- * 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs
825
- * both halves and reports per-scenario score divergence + a global
826
- * contamination verdict via paired Wilcoxon.
827
- * 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,
828
- * `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the
829
- * task's structural difficulty while breaking surface memorization.
830
- *
831
- * The verdict is conservative: if the perturbed-vs-original score
832
- * difference is statistically significant (BH-adjusted p < 0.05) AND
833
- * the median drop is > 5 percentage points, we flag *contamination
834
- * suspected*. False positives are possible (the perturbation might
835
- * actually be harder); the default is to flag for review, not to
836
- * autoreject.
837
- */
838
- type ScenarioPerturbationKind = 'rename_variables' | 'shuffle_order' | 'paraphrase' | 'inject_irrelevant_clause' | 'custom';
839
- interface ScenarioPerturbation<S> {
840
- kind: ScenarioPerturbationKind;
841
- /** Apply to one scenario, return its perturbed sibling. */
842
- apply: (scenario: S) => Promise<S> | S;
843
- /** Optional id — for the report. */
844
- id?: string;
845
- }
846
- interface ContaminationProbeInput<S> {
847
- /** Identity of every scenario. The probe's `runFingerprint` keys on these. */
848
- scenarioId: (s: S) => string;
849
- /** Original scenarios. */
850
- originals: S[];
851
- /**
852
- * Either pre-computed perturbations (one per original, same order) OR a
853
- * `perturbation` strategy that synthesizes them on the fly.
854
- */
855
- perturbed?: S[];
856
- perturbation?: ScenarioPerturbation<S>;
857
- /**
858
- * Run the policy/agent against one scenario and return a scalar score
859
- * in [0, 1]. The probe doesn't care what the policy is — that's the
860
- * caller's contract.
861
- */
862
- scoreFn: (s: S) => Promise<number>;
863
- }
864
- interface ContaminationProbeOptions {
865
- /** Drop scores below this from the probe; treats partial failures separately. Default 0. */
866
- scoreFloor?: number;
867
- /**
868
- * BH-FDR threshold for declaring contamination on each per-scenario
869
- * delta. Default 0.05.
870
- */
871
- fdr?: number;
872
- /**
873
- * Minimum median per-scenario drop to flag global contamination. Default
874
- * 0.05 (5 percentage points). Smaller drops may be noise.
875
- */
876
- minMedianDrop?: number;
877
- }
878
- interface ContaminationProbeReport {
879
- perScenario: Array<{
880
- scenarioId: string;
881
- originalScore: number;
882
- perturbedScore: number;
883
- delta: number;
884
- /** Per-scenario q-value (single-test BH for a single scenario). Mainly for display. */
885
- qValue: number;
886
- }>;
887
- /** Wilcoxon paired-test on the deltas. */
888
- pairedTest: {
889
- w: number;
890
- p: number;
891
- };
892
- medianDelta: number;
893
- meanDelta: number;
894
- contaminationSuspected: boolean;
895
- reason: string;
896
- /** Number of scenarios processed. */
897
- n: number;
898
- }
899
- declare function runContaminationProbe<S>(input: ContaminationProbeInput<S>, opts?: ContaminationProbeOptions): Promise<ContaminationProbeReport>;
900
- /**
901
- * Identifier-rename perturbation for code/text scenarios. Replaces every
902
- * occurrence of the listed identifiers with synthesized aliases. Use when
903
- * the scenario's structural difficulty is independent of variable names
904
- * (e.g. SWE-Bench-style coding tasks).
905
- */
906
- declare function renameVariables<S extends {
907
- prompt: string;
908
- }>(identifiers: string[], rename?: (name: string, idx: number) => string): ScenarioPerturbation<S>;
909
- /**
910
- * Order-shuffle perturbation. Reshuffles a list-shaped section of the
911
- * prompt (for QA scenarios that present options A/B/C/D — answer depends
912
- * on the option labels, not order). Caller provides the section extractor.
913
- */
914
- declare function shuffleOrder<S extends {
915
- prompt: string;
916
- }>(shuffleSection: (prompt: string, rng: () => number) => string, seed: number): ScenarioPerturbation<S>;
917
- /**
918
- * Inject-irrelevant-clause perturbation. Adds a benign sentence that
919
- * shouldn't change the answer. Tests for "did the model just memorize
920
- * the input string."
921
- */
922
- declare function injectIrrelevantClause<S extends {
923
- prompt: string;
924
- }>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
925
-
926
- /**
927
- * Preference dataset extraction — bridge from `RunRecord[]` to RL training.
928
- *
929
- * Production RLHF / DPO / KTO / SimPO pipelines need preference triples:
930
- * `(prompt, chosen, rejected)`. The campaign artifact already contains the
931
- * ingredients — every (variantId, scenarioId, seed) cell is a candidate
932
- * that ran the same prompt against the same scenario, scored by the same
933
- * judge — but turning that into a clean preference dataset requires
934
- * deciding *what counts as a preference*.
935
- *
936
- * This module ships three preference-extraction strategies with explicit
937
- * tradeoffs, plus a unified output type compatible with HuggingFace TRL,
938
- * Anthropic finetuning JSONL, and OpenAI fine-tuning APIs. The strategies
939
- * are deliberately not auto-magical — picking the wrong one corrupts the
940
- * gradient.
941
- *
942
- * Strategies:
943
- *
944
- * 1. **`paired-by-scenario-and-seed`** — exact-match comparisons. For
945
- * each scenario × seed pair, compare every (variantA, variantB) on
946
- * that exact (scenario, seed). Matches scenarios so the comparison
947
- * isolates variant effects. Highest signal-to-noise; smallest
948
- * dataset (only matched pairs count).
949
- *
950
- * 2. **`paired-by-scenario`** — looser matching. For each scenario,
951
- * compare every (variantA, variantB) where both have ≥ 1 run on the
952
- * same scenario. Aggregates across seeds to compute mean scores per
953
- * (variant, scenario), then forms preferences from the means. More
954
- * data, lower per-pair signal.
955
- *
956
- * 3. **`top-vs-bottom`** — coarsest. Within each scenario, the highest-
957
- * scoring run is `chosen`, the lowest is `rejected`. Smallest dataset
958
- * per scenario but biggest score gap per pair. Useful for early
959
- * bootstrapping when you have few variants.
960
- *
961
- * The output `PreferenceTriple` is *agent-eval-canonical* but trivially
962
- * mappable to TRL's `DPODataset` shape (`prompt`, `chosen`, `rejected`)
963
- * via the `toTRLFormat` helper.
964
- */
965
-
966
- type PreferenceStrategy = 'paired-by-scenario-and-seed' | 'paired-by-scenario' | 'top-vs-bottom';
967
- interface PreferenceTriple {
968
- /** The scenario (input) the variants were run against. */
969
- scenarioId: string;
970
- /** RunRecord ids on each side, for traceability. */
971
- chosenRunId: string;
972
- rejectedRunId: string;
973
- /** Variant ids — load-bearing for the RL update. */
974
- chosenVariantId: string;
975
- rejectedVariantId: string;
976
- /** The score gap between chosen and rejected. Larger = stronger signal. */
977
- marginScore: number;
978
- /**
979
- * Optional `(chosen_score, rejected_score)` pair for soft-margin DPO
980
- * variants. Omitted for `top-vs-bottom` runs that don't carry meaningful
981
- * scalar gaps.
982
- */
983
- scores?: {
984
- chosen: number;
985
- rejected: number;
986
- };
987
- /** Tie-breaker — when multiple seeds match this scenario, the one used. */
988
- seed?: number;
989
- /**
990
- * Free-form metadata propagated from the run records — e.g. original
991
- * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.
992
- */
993
- meta: {
994
- chosenPromptHash: string;
995
- rejectedPromptHash: string;
996
- chosenConfigHash: string;
997
- rejectedConfigHash: string;
998
- chosenModel: string;
999
- rejectedModel: string;
1000
- };
1001
- }
1002
- interface ExtractPreferencesOptions {
1003
- strategy?: PreferenceStrategy;
1004
- /**
1005
- * Minimum score gap required to admit a pair. Pairs below this are
1006
- * dropped — they're noise, not signal. Default 0.05 (5% of [0,1]).
1007
- */
1008
- minMargin?: number;
1009
- /**
1010
- * Optional split tag filter — restrict to runs from one split. Default
1011
- * `'holdout'` (the canonical "real" signal).
1012
- */
1013
- splitTag?: RunRecord['splitTag'];
1014
- /**
1015
- * Optional reward extractor that overrides `outcome.holdoutScore` /
1016
- * `outcome.searchScore`. Use to drive preferences off a verifiable
1017
- * reward instead of the headline score.
1018
- */
1019
- rewardOf?: (run: RunRecord) => number | null;
1020
- }
1021
- interface PreferenceExtractionReport {
1022
- pairs: PreferenceTriple[];
1023
- /** Number of (scenario, seed) cells inspected. */
1024
- cellsInspected: number;
1025
- /** Number of pairs filtered by `minMargin`. */
1026
- pairsBelowMargin: number;
1027
- /** Number of cells with only one variant (no comparison possible). */
1028
- cellsSingleton: number;
1029
- /** Strategy used. */
1030
- strategy: PreferenceStrategy;
1031
- }
1032
- /**
1033
- * Convert `RunRecord[]` to preference triples for RL training.
1034
- *
1035
- * Returns a structured report so callers can see how much data was
1036
- * dropped and why (low-margin pairs, singleton cells). For production
1037
- * pipelines, you usually want to:
1038
- *
1039
- * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
1040
- * 2. Call this with `strategy: 'paired-by-scenario-and-seed'` and a
1041
- * verifiable-reward extractor as `rewardOf`
1042
- * 3. Pass `report.pairs` to `toTRLFormat` and pipe to your DPO trainer
1043
- */
1044
- declare function extractPreferences(runs: RunRecord[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
1045
- /**
1046
- * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
1047
- * but the prompt isn't stored on the RunRecord — only its hash. The caller
1048
- * passes a `promptOf(promptHash)` lookup that the TRL trainer can use.
1049
- */
1050
- declare function toTRLFormat(triples: PreferenceTriple[], promptOf: (hash: string) => string): Array<{
1051
- prompt: string;
1052
- chosen: string;
1053
- rejected: string;
1054
- }>;
1055
- /**
1056
- * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
1057
- * shape. Same caveat as TRL: prompt + outputs are content the caller has
1058
- * to map back from the run record / raw event log.
1059
- */
1060
- declare function toAnthropicFormat(triples: PreferenceTriple[]): Array<{
1061
- scenarioId: string;
1062
- chosenRunId: string;
1063
- rejectedRunId: string;
1064
- margin: number;
1065
- }>;
1066
-
1067
- interface RunFilter {
1068
- scenarioId?: string;
1069
- variantId?: string;
1070
- status?: RunStatus;
1071
- since?: number;
1072
- until?: number;
1073
- tag?: {
1074
- key: string;
1075
- value: string;
1076
- };
1077
- parentRunId?: string;
1078
- projectId?: string;
1079
- chatId?: string;
1080
- layer?: RunLayer;
1081
- }
1082
- interface SpanFilter {
1083
- runId?: string;
1084
- parentSpanId?: string;
1085
- kind?: SpanKind;
1086
- name?: string;
1087
- toolName?: string;
1088
- judgeId?: string;
1089
- since?: number;
1090
- until?: number;
1091
- }
1092
- interface EventFilter {
1093
- runId?: string;
1094
- spanId?: string;
1095
- kind?: EventKind;
1096
- since?: number;
1097
- until?: number;
1098
- }
1099
- interface TraceStore {
1100
- appendRun(run: Run): Promise<void>;
1101
- updateRun(runId: string, patch: Partial<Run>): Promise<void>;
1102
- appendSpan(span: Span): Promise<void>;
1103
- updateSpan(spanId: string, patch: Partial<Span>): Promise<void>;
1104
- appendEvent(event: TraceEvent): Promise<void>;
1105
- appendArtifact(artifact: Artifact): Promise<void>;
1106
- appendBudgetEntry(entry: BudgetLedgerEntry): Promise<void>;
1107
- getRun(runId: string): Promise<Run | undefined>;
1108
- listRuns(filter?: RunFilter): Promise<Run[]>;
1109
- spans(filter?: SpanFilter): Promise<Span[]>;
1110
- events(filter?: EventFilter): Promise<TraceEvent[]>;
1111
- budget(runId: string): Promise<BudgetLedgerEntry[]>;
1112
- artifacts(runId: string): Promise<Artifact[]>;
1113
- }
1114
-
1115
- /**
1116
- * Process reward extraction — step-level credit assignment from trace spans.
1117
- *
1118
- * RL on long-horizon agents needs *step-level* rewards, not run-level
1119
- * ones. The classic credit-assignment problem (Sutton & Barto) requires
1120
- * knowing which sub-decisions in a trajectory contributed to the
1121
- * outcome. Modern systems (DeepSeek-R1, OpenAI o-series, Lightman et al.
1122
- * "Let's Verify Step by Step" 2023) train *process reward models* (PRMs)
1123
- * that score every step, then do RL with the PRM as the reward signal.
1124
- *
1125
- * This module extracts `StepReward[]` from trace spans — one per
1126
- * meaningful step — and ships:
1127
- *
1128
- * 1. `extractStepRewards(store, runId, opts)` — span → step-reward
1129
- * conversion using configurable per-span scorers (LLM judge over the
1130
- * span output, deterministic checkers, or a learned PRM).
1131
- * 2. `runwiseStepRewardSummary(stepRewards)` — aggregate the per-step
1132
- * signal into a credit-assignment-aware run-level score.
1133
- * 3. `prmTrainingPairs(stepRewards, options)` — produce the
1134
- * `(prefix, suffix_chosen, suffix_rejected)` triples that PRM
1135
- * training pipelines consume.
1136
- *
1137
- * What we ship: the *extraction* and *aggregation* infrastructure plus
1138
- * the data shape PRM training expects. We do NOT ship the actual PRM
1139
- * training (gradient descent over a transformer is out of scope for a
1140
- * TS package). The interface is the contract; downstream consumers wire
1141
- * their preferred trainer.
1142
- *
1143
- * Caveat the panel will land: this is descriptive credit assignment
1144
- * (which steps correlate with outcome), not causal credit assignment
1145
- * (which steps caused outcome). For causal claims you need
1146
- * counterfactual rollouts or a learned dynamics model. Future work; the
1147
- * descriptive version is what production PRM training actually uses.
1148
- */
1149
-
1150
- interface StepReward {
1151
- /** Trace span this reward attaches to. */
1152
- spanId: string;
1153
- runId: string;
1154
- /** Index in the trajectory (0-based, in started-at order). */
1155
- stepIndex: number;
1156
- /** Span kind (typically 'tool', 'llm', 'judge'). */
1157
- kind: Span['kind'];
1158
- /** Span name — for the consumer's downstream filtering. */
1159
- name: string;
1160
- /** Step-level reward in [0, 1]. */
1161
- reward: number;
1162
- /**
1163
- * Determinism class. Mirrors the verifiable-reward distinction:
1164
- * deterministic = test/compile/schema check; probabilistic = LLM judge.
1165
- */
1166
- determinism: 'deterministic' | 'probabilistic';
1167
- /** Optional rationale / evidence — the trainer typically discards. */
1168
- rationale?: string;
1169
- /** Optional weight — how much this step contributes to credit assignment. */
1170
- weight?: number;
1171
- }
1172
- interface StepScorer {
1173
- /** Span kinds this scorer applies to. */
1174
- appliesTo: Span['kind'][];
1175
- /** Returns null to skip the span; returns a `StepReward` shape (without index/runId/spanId, which are filled in). */
1176
- score(span: Span): Promise<Omit<StepReward, 'spanId' | 'runId' | 'stepIndex'>> | null | undefined;
1177
- }
1178
- interface ExtractStepRewardsOptions {
1179
- /**
1180
- * Ordered list of scorers. Each span runs through scorers in order;
1181
- * the first non-null result wins. If no scorer applies, the span is
1182
- * skipped (not all spans are training-worthy).
1183
- */
1184
- scorers: StepScorer[];
1185
- /** Optional filter — return null to drop the span entirely before scoring. */
1186
- preFilter?: (span: Span) => boolean;
1187
- }
1188
- declare function extractStepRewards(store: TraceStore, runId: string, opts: ExtractStepRewardsOptions): Promise<StepReward[]>;
1189
- interface RunwiseStepSummary {
1190
- runId: string;
1191
- totalSteps: number;
1192
- meanReward: number;
1193
- /** Sum-of-rewards (weighted by `weight ?? 1`). Use as the run-level proxy. */
1194
- sumWeightedReward: number;
1195
- /** Fraction of steps where reward < 0.5 — proxy for "where the policy was wrong." */
1196
- failureFraction: number;
1197
- /** Maximum drop in reward between consecutive steps — diagnoses a step where things went sideways. */
1198
- worstStepDelta: number;
1199
- worstStepIndex: number | null;
1200
- }
1201
- declare function runwiseStepRewardSummary(stepRewards: StepReward[]): RunwiseStepSummary;
1202
- interface PrmTrainingTriple {
1203
- /** Prefix run-id (or composite key) — the trajectory up to step k-1. */
1204
- prefixRunId: string;
1205
- prefixStepIndex: number;
1206
- /** The step that came next on a high-reward trajectory. */
1207
- chosenSpanId: string;
1208
- chosenReward: number;
1209
- /** A step from a divergent low-reward trajectory at the same prefix length. */
1210
- rejectedSpanId: string;
1211
- rejectedReward: number;
1212
- /** The prefix run came from this run; the rejected step came from `rejectedRunId`. */
1213
- rejectedRunId: string;
1214
- marginScore: number;
1215
- }
1216
- /**
1217
- * Build PRM training triples. The shape: pair runs that share an early
1218
- * prefix (same scenario, same first N steps) and diverge later — at the
1219
- * point of divergence, the high-reward run's next step is `chosen`, the
1220
- * low-reward run's next step is `rejected`. This is the canonical PRM
1221
- * training data shape from Lightman et al. and DeepSeek-R1 process
1222
- * supervision.
1223
- *
1224
- * Implementation note: we don't have a way to detect "same prefix" in
1225
- * the general agent setting (token-level prefixes require hashing model
1226
- * outputs). The current heuristic groups by `(scenarioId, prefixSpanName
1227
- * sequence)` — runs are paired when their first K span names match. For
1228
- * production use this should be replaced with a proper trajectory-prefix
1229
- * hash; the heuristic is good enough for early-stage scaffolding.
1230
- */
1231
- declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, opts?: {
1232
- minMargin?: number;
1233
- minPrefixLength?: number;
1234
- }): PrmTrainingTriple[];
1235
-
1236
- /**
1237
- * Trainer-format exporters.
1238
- *
1239
- * agent-eval produces canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`,
1240
- * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
1241
- * different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
1242
- * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
1243
- * JSONL conventions. Rather than ship N adapters, this module ships the
1244
- * canonical formats most production pipelines accept and ergonomic helpers
1245
- * for the rest.
1246
- *
1247
- * Shapes:
1248
- * - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed
1249
- * by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.
1250
- * - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.
1251
- * Consumed by prime-rl GRPO, verl, OpenRLHF.
1252
- * - **SFT** — `{messages[]}` JSONL with chosen completion as the final
1253
- * assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,
1254
- * Anthropic finetuning.
1255
- * - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.
1256
- * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
1257
- *
1258
- * Why ship this in agent-eval rather than a separate adapter package: the
1259
- * canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`, etc.) are
1260
- * agent-eval's contract; without first-party exporters consumers reverse-
1261
- * engineer the mapping every release. The exporters codify it.
1262
- *
1263
- * The exporters take callbacks for any field that isn't on the canonical
1264
- * artifact (specifically: prompt + completion text, since the package
1265
- * stores only their hashes by design — full text is the consumer's
1266
- * trace store / raw event log).
1267
- */
1268
-
1269
- interface DpoLookups {
1270
- /** Resolve the prompt text for a run (typically from a trace store / raw event sink). */
1271
- promptOf: (runId: string) => string | Promise<string>;
1272
- /** Resolve the assistant completion text for a run. */
1273
- completionOf: (runId: string) => string | Promise<string>;
1274
- }
1275
- interface DpoExportRow {
1276
- prompt: string;
1277
- chosen: string;
1278
- rejected: string;
1279
- /** Carried-through margin. Some KTO / IPO variants use this. */
1280
- margin?: number;
1281
- /** Free-form metadata for downstream filtering / sharding. */
1282
- meta?: Record<string, unknown>;
1283
- }
1284
- /**
1285
- * Convert preference triples to TRL-compatible DPO rows. The shape
1286
- * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
1287
- * entry; every major DPO trainer accepts it.
1288
- */
1289
- declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups): Promise<DpoExportRow[]>;
1290
- /** Serialize DPO rows as JSONL. One line per row. */
1291
- declare function toDpoJsonl(rows: DpoExportRow[]): string;
1292
- interface GrpoLookups {
1293
- promptOf: (runId: string) => string | Promise<string>;
1294
- completionOf: (runId: string) => string | Promise<string>;
1295
- /** Optional: derive a custom reward from the run. Defaults to score. */
1296
- rewardOf?: (run: RunRecord) => number | null;
1297
- }
1298
- interface GrpoExportRow {
1299
- prompt: string;
1300
- completions: string[];
1301
- rewards: number[];
1302
- /** runIds in the same order as `completions[]` for traceability. */
1303
- runIds: string[];
1304
- meta?: Record<string, unknown>;
1305
- }
1306
- /**
1307
- * Convert RunRecord[] grouped by `(scenarioId)` into GRPO offline rows —
1308
- * one row per scenario, with one completion per run on that scenario.
1309
- *
1310
- * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
1311
- * within a group of completions for the same prompt; this is the
1312
- * canonical input format.
1313
- */
1314
- declare function toGrpoRows(runs: RunRecord[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
1315
- declare function toGrpoJsonl(rows: GrpoExportRow[]): string;
1316
- interface SftLookups {
1317
- promptOf: (runId: string) => string | Promise<string>;
1318
- completionOf: (runId: string) => string | Promise<string>;
1319
- /** Optional system message. Default omits. */
1320
- systemOf?: (run: RunRecord) => string | null | undefined;
1321
- /** Filter — return false to skip the run (e.g., low score, failed cases). */
1322
- include?: (run: RunRecord) => boolean;
1323
- }
1324
- interface SftExportRow {
1325
- messages: Array<{
1326
- role: 'system' | 'user' | 'assistant';
1327
- content: string;
1328
- }>;
1329
- meta?: Record<string, unknown>;
1330
- }
1331
- /**
1332
- * Convert RunRecord[] into Hugging Face / OpenAI / Anthropic-style
1333
- * conversational SFT rows. By default every record becomes one row;
1334
- * pass `include` to filter (e.g., keep only `score >= 0.8` for
1335
- * rejection-sampling SFT).
1336
- */
1337
- declare function toSftRows(runs: RunRecord[], lookups: SftLookups): Promise<SftExportRow[]>;
1338
- declare function toSftJsonl(rows: SftExportRow[]): string;
1339
- interface PrmLookups {
1340
- /** Resolve the prompt text for a run. */
1341
- promptOf: (runId: string) => string | Promise<string>;
1342
- /** Resolve the trajectory step text for a (runId, spanId) pair. */
1343
- stepTextOf: (runId: string, spanId: string) => string | Promise<string>;
1344
- /** Optional: sequence of prefix span ids leading up to the divergence. */
1345
- prefixOf?: (runId: string, prefixStepIndex: number) => string[] | Promise<string[]>;
1346
- }
1347
- interface PrmExportRow {
1348
- prompt: string;
1349
- /** Span ids for the steps before divergence — caller resolves text via `stepTextOf`. */
1350
- prefixSpanIds: string[];
1351
- prefixStepText: string[];
1352
- chosenStep: string;
1353
- rejectedStep: string;
1354
- chosenReward: number;
1355
- rejectedReward: number;
1356
- marginScore: number;
1357
- meta?: Record<string, unknown>;
1358
- }
1359
- /**
1360
- * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
1361
- * callback resolves span text from the consumer's trace store.
1362
- */
1363
- declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups): Promise<PrmExportRow[]>;
1364
- declare function toPrmJsonl(rows: PrmExportRow[]): string;
1365
- interface StepRewardJsonlRow {
1366
- runId: string;
1367
- spanId: string;
1368
- stepIndex: number;
1369
- reward: number;
1370
- determinism: 'deterministic' | 'probabilistic';
1371
- weight: number;
1372
- }
1373
- declare function stepRewardsToJsonl(stepRewards: StepReward[]): string;
1374
-
1375
- /**
1376
- * RL dataset packaging + datasheet — the publishable, sellable bundle.
1377
- *
1378
- * The format exporters (`toGrpoRows` / `toSftRows` / `toDpoRows`) already
1379
- * produce trainer-ready shapes (prime-rl GRPO, TRL DPO, conversational SFT).
1380
- * What turns that into a dataset someone can PUBLISH or BUY is the provenance
1381
- * + a datasheet: which models produced it, which prompt/agent versions, how the
1382
- * reward was derived (deterministic verifiable vs probabilistic judge — the
1383
- * credibility axis a buyer checks first), the split discipline, the reward
1384
- * distribution, the quality gates, the license, and the intended/out-of-scope
1385
- * uses. This module computes those facts from the `RunRecord[]` and renders a
1386
- * "Datasheet for Datasets" (Gebru et al. 2018) card alongside the format files.
1387
- *
1388
- * It composes the existing `rl/exporters` — it does not reimplement any trainer
1389
- * format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization
1390
- * with per-token loss masks) is a downstream Python stage that consumes the
1391
- * `messages`/`completions` this bundle emits.
1392
- */
1393
-
1394
- type RewardKind = 'deterministic' | 'probabilistic' | 'mixed';
1395
- type DatasetFormat = 'grpo' | 'sft' | 'dpo';
1396
- /** Caller-declared context — the qualitative half of the datasheet that can't
1397
- * be computed from records. */
1398
- interface RlDatasetConfig {
1399
- name: string;
1400
- version: string;
1401
- /** Product/task domain, e.g. 'legal-m&a', 'tax-1040'. */
1402
- domain: string;
1403
- /** SPDX id or a named commercial license. Required — an unlicensed dataset
1404
- * cannot be published or sold. */
1405
- license: string;
1406
- /** How the reward was produced. `kind: 'deterministic'` (a test/schema/XPath
1407
- * decided it) is the credibility signal; 'probabilistic' = LLM-judge. */
1408
- reward: {
1409
- kind: RewardKind;
1410
- source: string;
1411
- description: string;
1412
- };
1413
- intendedUse: string;
1414
- outOfScope: string;
1415
- limitations: string;
1416
- /** ISO timestamp — passed in (the substrate forbids Date.now()). */
1417
- createdAtIso: string;
1418
- /** Default: ['grpo', 'sft']. */
1419
- formats?: DatasetFormat[];
1420
- /** Quality gates already run, recorded on the card for the buyer. */
1421
- qualityGates?: {
1422
- contaminationProbe?: 'passed' | 'failed' | 'not-run';
1423
- dedup?: boolean;
1424
- verifiableRewardFilter?: boolean;
1425
- };
1426
- }
1427
- interface RewardStats {
1428
- n: number;
1429
- mean: number;
1430
- median: number;
1431
- min: number;
1432
- max: number;
1433
- std: number;
1434
- }
1435
- interface RlDatasetStats {
1436
- records: number;
1437
- /** Record count per split — a publishable dataset must declare its holdout. */
1438
- splits: Record<RunSplitTag, number>;
1439
- reward: RewardStats;
1440
- /** Distinct snapshot-pinned models that produced the trajectories. */
1441
- models: string[];
1442
- /** Distinct effective-prompt hashes (the agent profile/prompt versions). */
1443
- promptHashes: string[];
1444
- commitShas: string[];
1445
- totalTokens: {
1446
- input: number;
1447
- output: number;
1448
- };
1449
- totalCostUsd: number;
1450
- }
1451
- interface RlDatasetManifest extends RlDatasetConfig {
1452
- formats: DatasetFormat[];
1453
- rowCounts: Partial<Record<DatasetFormat, number>>;
1454
- stats: RlDatasetStats;
1455
- }
1456
- interface RlDatasetBundle {
1457
- manifest: RlDatasetManifest;
1458
- /** Relative filename -> contents. Write these to a directory to publish. */
1459
- files: Record<string, string>;
1460
- }
1461
- /**
1462
- * Package graded `RunRecord[]` into a publishable RL dataset bundle: the
1463
- * trainer-format JSONL files + a manifest + a datasheet. DPO requires
1464
- * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
1465
- * the records directly via the supplied lookups. Throws on an empty corpus —
1466
- * an empty dataset must never be published.
1467
- */
1468
- declare function buildRlDataset(records: RunRecord[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
1469
- triples: PreferenceTriple[];
1470
- lookups: DpoLookups;
1471
- }): Promise<RlDatasetBundle>;
1472
- /** Render the "Datasheet for Datasets" card — the artifact a buyer reads. */
1473
- declare function datasheetToMarkdown(m: RlDatasetManifest): string;
1474
-
1475
- /**
1476
- * RL corpus — the durable, append-only accumulation of graded RunRecords that
1477
- * every eval run deposits BY DEFAULT.
1478
- *
1479
- * The dataset is the free exhaust of the normal eval process: we run evals
1480
- * constantly to get an agent production-ready, and those runs already produce
1481
- * graded trajectories. Instead of writing them to an ephemeral run dir and
1482
- * throwing them away, `appendToCorpus` accumulates them into a durable corpus;
1483
- * `buildDatasetFromCorpus` later harvests the whole corpus into a publishable
1484
- * bundle. No separate data-collection campaign — the data accrues from work we
1485
- * do anyway. This is the "best things for free by our process" layer.
1486
- *
1487
- * Trajectory text rides on the record as top-level `prompt` / `completion`
1488
- * (what the eval harnesses capture; the RunRecord validator ignores the extra
1489
- * keys). The harvest reads them directly — no trace store round-trip needed.
1490
- */
1491
-
1492
- /** A corpus record is a RunRecord carrying the trajectory text the harness
1493
- * captured. `prompt`/`completion` are top-level (the validator ignores extras). */
1494
- type CorpusRecord = RunRecord & {
1495
- prompt?: string;
1496
- completion?: string;
1497
- };
1498
- interface CorpusAppendResult {
1499
- appended: number;
1500
- /** Skipped because a record with the same runId was already in the corpus
1501
- * (idempotent appends — NOT re-run collapsing; re-runs get fresh runIds). */
1502
- skipped: number;
1503
- total: number;
1504
- }
1505
- /**
1506
- * Append graded records to the corpus (append-only JSONL). Deduplicates by
1507
- * `runId` against what's already on disk so re-running the same harness is
1508
- * idempotent. Creates the file and parent dir. This is the call every eval
1509
- * harness makes by default after producing its records.
1510
- */
1511
- declare function appendToCorpus(records: CorpusRecord[], corpusPath: string): CorpusAppendResult;
1512
- /** Read the full corpus. Returns [] if the corpus does not exist yet. */
1513
- declare function readCorpus(corpusPath: string): CorpusRecord[];
1514
- interface HarvestOptions {
1515
- /** Keep only records scoring >= this (rejection-sampling for SFT). */
1516
- minScore?: number;
1517
- /** Keep only these splits (e.g. ['holdout'] for an eval-only dataset). */
1518
- splits?: RunRecord['splitTag'][];
1519
- }
1520
- /**
1521
- * Harvest the accumulated corpus into a publishable RL dataset bundle. Reads
1522
- * trajectory text from each record's top-level `prompt`/`completion`; records
1523
- * missing either are excluded (a graded score with no trajectory can't train).
1524
- * Optionally filters by score / split. Throws (via buildRlDataset) if nothing
1525
- * survives — an empty dataset must never be published.
1526
- */
1527
- declare function buildDatasetFromCorpus(corpusPath: string, config: RlDatasetConfig, opts?: HarvestOptions): Promise<RlDatasetBundle>;
1528
-
1529
- /**
1530
- * Off-policy evaluation primitives.
1531
- *
1532
- * Standard inverse-probability-weighted (IPS), self-normalized
1533
- * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
1534
- * value of a *target* policy given trajectories collected under a
1535
- * *behavior* policy. This is the canonical RL eval task: "we have last
1536
- * week's runs, we changed the policy — how would the new one do without
1537
- * re-running?"
1538
- *
1539
- * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
1540
- * & Joachims 2015 for SNIPS) but the *application* to LLM-agent
1541
- * evaluation needs care:
1542
- *
1543
- * - The "policy" is the (prompt, tool config, model snapshot) triple.
1544
- * Two policies have the same probability over an action *iff* their
1545
- * LLM call would emit the same token with the same probability —
1546
- * which is generally unknowable without the model log-probs.
1547
- * - For LLM agents, propensity scores must be supplied by the caller
1548
- * (logged in the trace, recovered from token log-probs, or estimated
1549
- * via a learned propensity model). We do NOT estimate propensity here.
1550
- * - Doubly-robust requires a Q-function (model-based reward predictor).
1551
- * We accept any callable; consumers pass either a tabular average,
1552
- * a regression fit, or a learned reward model.
1553
- *
1554
- * Bias / variance tradeoffs:
1555
- * - IPS: unbiased; high variance for small overlap, infinite variance
1556
- * when target has support outside behavior.
1557
- * - SNIPS: lower variance, slight bias; usually preferred in practice.
1558
- * - DR: doubly-robust — unbiased if either propensity OR Q-function is
1559
- * correct. Lowest practical variance when Q is decent. Use this.
1560
- *
1561
- * Caveat the panel will land: on the LLM-agent setting, propensity scores
1562
- * recovered from token log-probs are noisy, the action space is enormous,
1563
- * and overlap is often poor. These estimators are useful but not magic;
1564
- * complement with `replayCampaign` (exact replay where the request hashes
1565
- * match) for high-confidence answers and OPE for the gap.
1566
- */
1567
- interface OffPolicyTrajectory {
1568
- /** Stable id, for traceability through the dataset. */
1569
- runId: string;
1570
- /** Reward observed under the behavior policy (the realized outcome). */
1571
- reward: number;
1572
- /**
1573
- * Behavior-policy probability of the action that was taken. For LLM
1574
- * agents this is typically `exp(sum(token_log_probs))` over the chosen
1575
- * trajectory. Must be in (0, 1].
1576
- */
1577
- behaviorProb: number;
1578
- /**
1579
- * Target-policy probability of the same action. For replay-style
1580
- * counterfactual evaluation this is what the *new* policy would have
1581
- * assigned to the *old* trajectory. Must be in [0, 1].
1582
- */
1583
- targetProb: number;
1584
- /**
1585
- * Optional model-based reward prediction at the same context. Used by
1586
- * `doublyRobust`. Set to `null` for IPS-only evaluation.
1587
- */
1588
- qHat?: number | null;
1589
- }
1590
- interface OffPolicyEstimate {
1591
- /** Estimated value of the target policy. */
1592
- value: number;
1593
- /** Standard error of the estimate. */
1594
- standardError: number;
1595
- /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */
1596
- effectiveSampleSize: number;
1597
- /** Number of trajectories used. */
1598
- n: number;
1599
- /**
1600
- * Diagnostic: maximum importance weight observed. Large values (>>10x
1601
- * mean) are a red flag — variance is dominated by a few outliers.
1602
- */
1603
- maxImportanceWeight: number;
1604
- }
1605
- interface OffPolicyOptions {
1606
- /**
1607
- * Cap importance weights at this value (Ionides 2008 truncated IS) to
1608
- * trade unbiasedness for variance reduction. Default `Infinity` (no cap).
1609
- * Set e.g. `10` for stable estimates when the policies are close.
1610
- */
1611
- weightCap?: number;
1612
- /** Reward clipping range. Default `[0, 1]`. */
1613
- rewardClip?: {
1614
- low: number;
1615
- high: number;
1616
- };
1617
- }
1618
- /**
1619
- * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator
1620
- * of E[reward under target policy]. Variance scales with the spread of
1621
- * target/behavior ratios.
1622
- */
1623
- declare function inverseProbabilityWeighting(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
1624
- /**
1625
- * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at
1626
- * the cost of small bias (vanishing as N grows). The right default for
1627
- * LLM-agent evaluation where overlap is often poor.
1628
- */
1629
- declare function selfNormalizedImportanceWeighting(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
1630
- /**
1631
- * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).
1632
- *
1633
- * V_DR = (1/N) * sum_i [ q_hat_i + (target_prob_i / behavior_prob_i) * (r_i - q_hat_i) ]
1634
- *
1635
- * Unbiased if EITHER:
1636
- * - the importance ratios are correct (IPS-style validity), OR
1637
- * - the Q-hat function is correct (model-based validity).
1638
- *
1639
- * In practice both are imperfect, but the residual bias is the *product*
1640
- * of both errors — much smaller than either alone. This is why DR is the
1641
- * default in production OPE pipelines.
1642
- *
1643
- * Requires `qHat` on every trajectory. If any are `null`, the estimator
1644
- * falls back to SNIPS for those entries (loud-fallback behavior; the
1645
- * report's `n` reflects the full set but `effectiveSampleSize` accounts
1646
- * for the lost variance reduction).
1647
- */
1648
- declare function doublyRobust(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): OffPolicyEstimate;
1649
- /**
1650
- * Convenience: run all three estimators and return them side-by-side.
1651
- * The recommended diagnostic — agreement across estimators is a much
1652
- * stronger signal than any single one.
1653
- */
1654
- declare function offPolicyEstimateAll(trajectories: OffPolicyTrajectory[], opts?: OffPolicyOptions): {
1655
- ips: OffPolicyEstimate;
1656
- snips: OffPolicyEstimate;
1657
- dr: OffPolicyEstimate;
1658
- };
1659
-
1660
- /**
1661
- * OutcomeStore — deployment outcomes attached to Run IDs.
1662
- *
1663
- * Outcomes arrive asynchronously from production telemetry after the
1664
- * eval run completed: user ratings, retention flags, conversion events,
1665
- * revenue, support-ticket rate, anything a product team can measure.
1666
- * The store is a peer to TraceStore — separate lifecycle, same runId
1667
- * foreign key.
1668
- *
1669
- * The whole point of this module is to make the meta-eval correlation
1670
- * question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.
1671
- */
1672
- interface DeploymentOutcome {
1673
- runId: string;
1674
- capturedAt: number;
1675
- /** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */
1676
- metrics: Record<string, number>;
1677
- /** Dimensions for stratified analysis — cohort, region, user_segment. */
1678
- labels?: Record<string, string>;
1679
- /** Free-form provenance (source system, pipeline version). */
1680
- source?: string;
1681
- }
1682
- interface OutcomeFilter {
1683
- runIds?: string[];
1684
- since?: number;
1685
- until?: number;
1686
- label?: {
1687
- key: string;
1688
- value: string;
1689
- };
1690
- source?: string;
1691
- }
1692
- interface OutcomeStore {
1693
- append(outcome: DeploymentOutcome): Promise<void>;
1694
- /** All outcomes attached to this run (a single run can have many — multiple
1695
- * capture windows over deployment time). */
1696
- forRun(runId: string): Promise<DeploymentOutcome[]>;
1697
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
1698
- }
1699
- declare class InMemoryOutcomeStore implements OutcomeStore {
1700
- private items;
1701
- append(outcome: DeploymentOutcome): Promise<void>;
1702
- forRun(runId: string): Promise<DeploymentOutcome[]>;
1703
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
1704
- }
1705
- interface FileSystemOutcomeStoreOptions {
1706
- dir: string;
1707
- maxBytes?: number;
1708
- }
1709
- declare class FileSystemOutcomeStore implements OutcomeStore {
1710
- private dir;
1711
- private maxBytes;
1712
- private memo?;
1713
- private loaded;
1714
- constructor(options: FileSystemOutcomeStoreOptions);
1715
- private ensureDir;
1716
- append(outcome: DeploymentOutcome): Promise<void>;
1717
- private load;
1718
- forRun(runId: string): Promise<DeploymentOutcome[]>;
1719
- list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
1720
- }
1721
-
1722
- /**
1723
- * Rubric predictive validity — does our eval rubric predict deployment
1724
- * outcomes?
1725
- *
1726
- * `correlationStudy` (already in this package) joins a `TraceStore` to an
1727
- * `OutcomeStore` and computes Pearson + Spearman + bootstrap CI for each
1728
- * (eval-metric, outcome-metric) pair. That answers "does X correlate with
1729
- * Y at all." `rubricPredictiveValidity` is the campaign-shaped wrapper
1730
- * around it: take a sequence of `RunRecord`s (the canonical campaign
1731
- * artifact) and a `DeploymentOutcomeStore`, join on `runId`, return a
1732
- * ranked verdict on every rubric whose dimension scores were captured in
1733
- * `outcome.raw`.
1734
- *
1735
- * The point — quoting the methodology doc — is that **without this loop
1736
- * every rubric is faith-based**. Once it's wired, you know which rubrics
1737
- * have earned their promotion power and which ones are decoration.
1738
- *
1739
- * const validity = await rubricPredictiveValidity({
1740
- * runs: lastQuarter,
1741
- * outcomes: shipFlagOutcomeStore,
1742
- * outcomeMetrics: ['revenue_lift', 'retention_30d', 'csat'],
1743
- * rubrics: ['anti_slop', 'semantic_concept', 'tool_recovery'],
1744
- * })
1745
- * for (const r of validity.ranked) {
1746
- * console.log(`${r.rubric} → ${r.bestOutcome}: ρ=${r.spearman.toFixed(2)}`)
1747
- * }
1748
- *
1749
- * The function is intentionally read-only. Use the verdict to deprecate
1750
- * decorative rubrics, re-weight composite scores, or trigger a
1751
- * recalibration sweep when predictive validity drops below a threshold.
1752
- */
1753
-
1754
- interface RubricOutcomePair {
1755
- rubric: string;
1756
- outcome: string;
1757
- n: number;
1758
- pearson: number;
1759
- spearman: number;
1760
- ci95: {
1761
- low: number;
1762
- high: number;
1763
- };
1764
- /**
1765
- * Verdict bucket. `load_bearing` ≥ 0.7, `informative` ≥ 0.4,
1766
- * `decorative` < 0.4 in absolute correlation. A negative correlation
1767
- * with a desired outcome is also `decorative` — actively misleading
1768
- * is worse than uninformative.
1769
- */
1770
- verdict: 'load_bearing' | 'informative' | 'decorative';
1771
- }
1772
- interface RubricRanking {
1773
- rubric: string;
1774
- /** Outcome metric this rubric correlated best with. */
1775
- bestOutcome: string;
1776
- spearman: number;
1777
- pearson: number;
1778
- n: number;
1779
- verdict: RubricOutcomePair['verdict'];
1780
- }
1781
- interface RubricPredictiveValidityReport {
1782
- pairs: RubricOutcomePair[];
1783
- /** Per-rubric best pair, sorted descending by |spearman|. */
1784
- ranked: RubricRanking[];
1785
- joinedSamples: number;
1786
- skippedRuns: number;
1787
- /** Rubrics that were declared but never produced a usable score. */
1788
- rubricsWithoutData: string[];
1789
- }
1790
-
1791
- /**
1792
- * HeldOutGate — first-class held-out paired-delta promotion gate.
1793
- *
1794
- * Encodes the "honesty override" pattern that lived inline in
1795
- * `~/webb/redteam/scripts/agent-eval-autoresearch.ts:138–171`.
1796
- * The optimizer's best-guess is one thing; what we should actually
1797
- * ship is another. The gate is the line between them.
1798
- *
1799
- * A candidate is promoted iff ALL three pass:
1800
- *
1801
- * 1. **Productive runs**: the candidate has at least
1802
- * `minProductiveRuns` paired observations on items where BOTH
1803
- * candidate and baseline produced a real (non-silent) score.
1804
- * 2. **Paired delta**: the lower bound of the bootstrap CI on the
1805
- * median per-item delta (candidate − baseline) on the HOLDOUT
1806
- * split is strictly greater than `pairedDeltaThreshold`.
1807
- * 3. **Overfit gap**: the candidate's gap between search-split
1808
- * score and holdout-split score is no worse (more positive)
1809
- * than the baseline's gap by more than `overfitGapThreshold`.
1810
- * "Better on search, worse on holdout" is the canonical
1811
- * overfit pattern; this catches it.
1812
- *
1813
- * The decision carries a machine-readable `rejectionCode` plus an
1814
- * `evidence` block with every number the gate looked at, so the
1815
- * downstream researcher / paper / dashboard can re-derive the
1816
- * verdict without re-running.
1817
- *
1818
- * See also:
1819
- * - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank`
1820
- * - `src/run-record.ts` for the input row schema
1821
- * - `src/reference-replay.ts` for the older, reference-replay-
1822
- * specific promotion path (still useful for replay-style evals).
1823
- */
1824
-
1825
- type HeldOutGateRejectionCode = 'few_runs' | 'negative_delta' | 'overfit_gap' | 'cost_ceiling';
1826
- interface GateEvidence {
1827
- /** Number of paired (candidate, baseline) holdout observations used. */
1828
- productiveRuns: number;
1829
- /** Median of (candidate − baseline) paired holdout deltas. */
1830
- medianPairedDelta: number;
1831
- /** Bootstrap CI on the median paired holdout delta. */
1832
- pairedCI: {
1833
- low: number;
1834
- high: number;
1835
- };
1836
- /** Wilcoxon signed-rank p-value on the paired holdout deltas. */
1837
- pairedPValue: number;
1838
- /** Mean candidate score on the search split (NaN if none). */
1839
- searchScore: number;
1840
- /** Mean candidate score on the holdout split (NaN if none). */
1841
- holdoutScore: number;
1842
- /** Candidate (search − holdout) gap. */
1843
- overfitGap: number;
1844
- /** Baseline (search − holdout) gap. */
1845
- baselineOverfitGap: number;
1846
- /** Median per-task USD cost across the candidate's runs. Recorded
1847
- * even when no `costPerTaskCeiling` is configured so downstream
1848
- * dashboards (intelligence.tangle.tools) can render \$/task per
1849
- * generation regardless of gating policy. */
1850
- medianCandidateCost: number;
1851
- /** Median per-task USD cost across the baseline runs, for
1852
- * symmetric reporting. */
1853
- medianBaselineCost: number;
1854
- }
1855
- interface GateDecision$1 {
1856
- /** Final promote/no-promote verdict. */
1857
- promote: boolean;
1858
- /** The candidate that was evaluated. */
1859
- candidateId: string;
1860
- /** The baseline it was compared against. */
1861
- baselineId: string;
1862
- /** Every number the gate looked at, for audit + paper export. */
1863
- evidence: GateEvidence;
1864
- /** Human-readable reason. */
1865
- reason: string;
1866
- /** Machine-readable rejection code, or null on promote. */
1867
- rejectionCode: HeldOutGateRejectionCode | null;
1868
- }
1869
-
1870
- /**
1871
- * Researcher interface — stable hook for an external autonomous-research
1872
- * agent to drive the meta-loop.
1873
- *
1874
- * Implementations live downstream (typically in a private repo that
1875
- * runs the actual LLM). This package ships only the contract + a
1876
- * `NoopResearcher` so consumers can wire the surface without being
1877
- * forced to implement every method up front.
1878
- *
1879
- * The four methods mirror the four stages of the paper "Two Loops,
1880
- * Three Roles":
1881
- *
1882
- * inspectFailures — given the observed runs, what failure modes
1883
- * are present? (data → diagnosis)
1884
- * proposeChange — given diagnosed failure modes, what
1885
- * structural changes should we try?
1886
- * (diagnosis → plan delta)
1887
- * applyChange — fold the proposed deltas into a concrete
1888
- * experiment plan against an existing baseline.
1889
- * (plan delta → executable plan)
1890
- * evaluateChange — run the plan, return runs + the gate verdict.
1891
- * (executable plan → verdict)
1892
- *
1893
- * Composition is the discipline: a Researcher implementation MUST
1894
- * keep these four steps separate and inspectable. Conflating
1895
- * "diagnose + propose + run" into a single LLM call defeats the
1896
- * point of the framework — you can't audit which step lied.
1897
- *
1898
- * THIS INTERFACE IS STABLE. Breaking changes require a new module
1899
- * (e.g. `Researcher2`) so existing implementations keep working.
1900
- */
1901
-
1902
- /** A diagnosed failure mode with the run-IDs that exhibit it. */
1903
- interface FailureMode {
1904
- /** Short machine-readable code. Must be stable across runs of the
1905
- * same researcher to enable longitudinal tracking. */
1906
- code: string;
1907
- /** Human-readable description for the paper / dashboard. */
1908
- description: string;
1909
- evidence: {
1910
- /** Run IDs (from `RunRecord.runId`) where this failure mode was
1911
- * observed. */
1912
- runIds: string[];
1913
- /** Number of run samples that informed the diagnosis. */
1914
- samples: number;
1915
- };
1916
- }
1917
- /** A single steering change the researcher wants to try. */
1918
- interface SteeringChange {
1919
- kind: 'reviewer_prompt' | 'skill_add' | 'skill_remove' | 'threshold' | 'budget';
1920
- /** Implementation-specific payload. Researcher implementations
1921
- * define the schema — keep this `unknown` here to avoid coupling
1922
- * the public interface to any one researcher's internal model. */
1923
- payload: unknown;
1924
- /** Why the researcher proposed this change. Goes into the audit
1925
- * trail next to the failure-mode evidence. */
1926
- rationale: string;
1927
- /** Optional self-reported expected delta on the headline metric. */
1928
- expectedDelta?: number;
1929
- }
1930
- /** A single experiment plan, mapped onto the search/holdout splits. */
1931
- interface ExperimentPlan {
1932
- baselineCandidateId: string;
1933
- proposedCandidateId: string;
1934
- changes: SteeringChange[];
1935
- /** USD ceiling for the entire experiment. The runner must stop
1936
- * before exceeding this and report a partial result. */
1937
- evaluationBudgetUsd: number;
1938
- /** Item IDs (your dataset keys) for the search vs holdout splits. */
1939
- splits: {
1940
- search: string[];
1941
- holdout: string[];
1942
- };
1943
- }
1944
- /** Result of running a plan: every run, plus the gate verdict. */
1945
- interface ExperimentResult {
1946
- plan: ExperimentPlan;
1947
- runs: RunRecord[];
1948
- gateDecision: GateDecision$1;
1949
- }
1950
- /**
1951
- * The researcher loop. Stable, four-step, inspectable.
1952
- *
1953
- * ┌──────────┐ inspectFailures ┌──────────┐ proposeChange ┌──────────┐
1954
- * │ runs │ ─────────────────▶│ failures │ ──────────────▶│ changes │
1955
- * └──────────┘ └──────────┘ └────┬─────┘
1956
- * │
1957
- * ▼
1958
- * ┌────────────────┐ applyChange ┌────────┐
1959
- * │ ExperimentPlan │ ◀────────────│ base │
1960
- * └────────┬───────┘ └────────┘
1961
- * │
1962
- * evaluateChange ▼
1963
- * ┌────────────────┐
1964
- * │ ExperimentResult│
1965
- * └────────────────┘
1966
- */
1967
- interface Researcher {
1968
- inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
1969
- proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
1970
- applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
1971
- evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
1972
- }
1973
-
1974
- /**
1975
- * `PredictiveValidityResearcher` — concrete `Researcher` implementation
1976
- * that drives selection from outcome-anchored predictive validity.
1977
- *
1978
- * Each method:
1979
- *
1980
- * - `inspectFailures(runs)` — synthesizes failure modes from the
1981
- * bottom-quartile of `RunRecord`s on the configured proxy reward.
1982
- * - `proposeChange(failures)` — proposes steering changes that target
1983
- * the rubrics with the lowest predictive validity (decorative ones).
1984
- * Either reduce their weight in the composite, or recalibrate them.
1985
- * - `applyChange(changes, baseline)` — merges the proposed steering
1986
- * into the experiment plan.
1987
- * - `evaluateChange(plan)` — re-runs the predictive-validity check on
1988
- * the post-change runs and reports the delta.
1989
- *
1990
- * The result is a closed loop: the rubric weights drift toward the ones
1991
- * that actually predict deployment outcomes, automatically. Pair with
1992
- * `runRLCampaign` for the full auto-research story.
1993
- */
1994
-
1995
- interface PredictiveValidityResearcherOptions {
1996
- outcomes: OutcomeStore;
1997
- outcomeMetrics: string[];
1998
- /** Score threshold below which a run counts as a "failure." Default 0.5. */
1999
- failureThreshold?: number;
2000
- /** Spearman bucket below which a rubric is "decorative." Default 0.4. */
2001
- decorativeThreshold?: number;
2002
- /** Optional steering-namespace prefix for proposed changes. Default `'rubric_weight'`. */
2003
- steeringNamespace?: string;
2004
- /** Override the rubric set the researcher inspects. Default: every numeric `outcome.raw` key seen. */
2005
- rubrics?: string[];
2006
- /**
2007
- * Snapshot stash hook — called with the most recent predictive-validity
2008
- * report. Useful when a downstream system wants to log rubric drift over
2009
- * time. Default no-op.
2010
- */
2011
- onReport?: (report: RubricPredictiveValidityReport) => void | Promise<void>;
2012
- }
2013
- /**
2014
- * Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
2015
- * rubrics that don't predict deployment outcomes don't earn weight.
2016
- */
2017
- declare class PredictiveValidityResearcher implements Researcher {
2018
- private opts;
2019
- private lastReport;
2020
- constructor(opts: PredictiveValidityResearcherOptions);
2021
- inspectFailures(runs: RunRecord[]): Promise<FailureMode[]>;
2022
- proposeChange(failures: FailureMode[]): Promise<SteeringChange[]>;
2023
- applyChange(changes: SteeringChange[], baseline: ExperimentPlan): Promise<ExperimentPlan>;
2024
- evaluateChange(plan: ExperimentPlan): Promise<ExperimentResult>;
2025
- /**
2026
- * Run the predictive-validity check explicitly against a fresh RunRecord
2027
- * set. Updates the researcher's cached report so subsequent
2028
- * `proposeChange` calls have evidence to draw from.
2029
- */
2030
- runValidityCheck(runs: RunRecord[]): Promise<RubricPredictiveValidityReport>;
2031
- /**
2032
- * Force-feed a predictive-validity report into the researcher state —
2033
- * useful when the consumer ran the report out-of-band and wants the
2034
- * researcher's later proposals informed by it.
2035
- */
2036
- setReport(report: RubricPredictiveValidityReport): void;
2037
- getLastReport(): RubricPredictiveValidityReport | null;
2038
- }
2039
-
2040
- /**
2041
- * Validator-output verdict — substrate primitive for "did this output pass,
2042
- * and how well?"
2043
- *
2044
- * Used by:
2045
- * - `@tangle-network/agent-eval/matrix` — verdict per cell in the cartesian.
2046
- * - `@tangle-network/agent-runtime` — Validator<Output, Verdict = DefaultVerdict>.
2047
- * Runtime keeps `Validator` because it's coupled to runtime-shaped
2048
- * `ValidationCtx` (iteration, signal, traceEmitter); the verdict TYPE
2049
- * itself is a substrate concept and lives here.
2050
- *
2051
- * Repo layering: agent-eval is the substrate (no upward deps). Both
2052
- * agent-runtime and agent-knowledge consume this type FROM agent-eval —
2053
- * never the other way around. See CLAUDE.md "Repo layering" for the rule.
2054
- */
2055
- /**
2056
- * Minimal verdict shape — `valid` + `score` are required; `scores` +
2057
- * `notes` are optional surface. Validators that need richer shapes
2058
- * parameterise `Validator<Output, MyVerdict>` with their own type.
2059
- *
2060
- * Need structured extras? Extend DefaultVerdict with typed fields — never
2061
- * serialize extras into `notes`.
2062
- */
2063
- interface DefaultVerdict {
2064
- /** Whether the output meets the validator's pass criteria. */
2065
- valid: boolean;
2066
- /** Aggregate score in [0, 1]. Drivers use this for winner selection. */
2067
- score: number;
2068
- /** Per-dimension scores. Free-form; weighted into `score` by the validator. */
2069
- scores?: Record<string, number>;
2070
- /** Human-readable rationale; surfaces in trace + final-result `winner.verdict`. */
2071
- notes?: string;
2072
- }
2073
-
2074
- /**
2075
- * Multi-layer verifier — ordered pipeline of verification layers.
2076
- *
2077
- * Different contract from {@link JudgeRunner} (which runs parallel
2078
- * specs against a sandbox). MultiLayerVerifier is a DAG of layers
2079
- * (install → typecheck → build → lint → serve → semantic → …) with
2080
- * dependency-based skip, per-layer findings, soft-fail semantics, and
2081
- * an aggregated `blendedScore` across all passed layers.
2082
- *
2083
- * Use when you want:
2084
- * - ordered stages where a failing upstream stage skips downstream ones
2085
- * - each stage produces rich `findings` (severity + message + evidence)
2086
- * - a single composite score across stages with per-stage weights
2087
- * - soft-fail stages whose failure doesn't abort the pipeline
2088
- *
2089
- * Use {@link JudgeRunner} when you want:
2090
- * - N independent judges running in parallel against the same artifact
2091
- * - no inter-judge dependencies
2092
- * - boolean `passed` per judge + overall
2093
- *
2094
- * Both primitives compose — JudgeRunner can be invoked as a single
2095
- * layer inside a MultiLayerVerifier if that suits the caller.
2096
- */
2097
-
2098
- type LayerStatus = 'pass' | 'fail' | 'skipped' | 'error' | 'timeout';
2099
- type Severity = 'critical' | 'major' | 'minor' | 'info';
2100
- interface Finding {
2101
- severity: Severity;
2102
- message: string;
2103
- evidence?: string;
2104
- /** Optional layer name the finding belongs to (set by the verifier if omitted). */
2105
- layer?: string;
2106
- /**
2107
- * Free-form structured payload — used by `multiToolchainLayer` to attach
2108
- * `{ adapter: 'pnpm' }`, by judges to attach evidence pointers, etc.
2109
- * Renderers MAY interrogate; agent-eval primitives never assume shape.
2110
- */
2111
- detail?: Record<string, unknown>;
2112
- }
2113
- interface LayerResult {
2114
- layer: string;
2115
- status: LayerStatus;
2116
- /** 0..1 score, optional — layers that don't produce a numeric score omit. */
2117
- score?: number;
2118
- durationMs: number;
2119
- findings: Finding[];
2120
- /** Short human-readable summary (one line). */
2121
- reason?: string;
2122
- /**
2123
- * Numeric layer-level diagnostics: error counts, warning counts,
2124
- * cyclomatic complexity, total adapter wall-time, etc. Keyed by
2125
- * diagnostic name; null = "diagnostic not applicable / not measured."
2126
- * Renderers that know the keys can display them; ones that don't,
2127
- * ignore. Free-form on purpose — consumers type the value shape in
2128
- * their own namespace.
2129
- */
2130
- diagnostics?: Record<string, number | null>;
2131
- /** Any rich per-layer detail — rendered as-is by consumers that know the layer. */
2132
- detail?: Record<string, unknown>;
2133
- }
2134
- /** Extends the substrate verdict spine: `valid` = `allPass` and `score` =
2135
- * `blendedScore` — derived where the report is aggregated, so spine
2136
- * consumers (drivers, gates) read this report without an adapter. */
2137
- interface VerificationReport extends DefaultVerdict {
2138
- layers: LayerResult[];
2139
- passCount: number;
2140
- failCount: number;
2141
- skippedCount: number;
2142
- errorCount: number;
2143
- /** True iff at least one scored layer ran AND every scored layer passed. */
2144
- allPass: boolean;
2145
- /**
2146
- * Weighted mean of `score` across contributing layers. 0 when no layers
2147
- * contributed. See {@link Layer.failContributesToScore} for fail semantics.
2148
- */
2149
- blendedScore: number;
2150
- durationMs: number;
2151
- startedAt: string;
2152
- finishedAt: string;
2153
- }
2154
-
2155
- /**
2156
- * Verifiable reward channel.
2157
- *
2158
- * For RL on coding / math / theorem-proving / structured-output tasks, the
2159
- * reward signal is *decidable* — a test passes or fails, a proof checks or
2160
- * doesn't, an output validates against a schema or doesn't. These rewards
2161
- * are dramatically more useful for RL training than LLM-judge scores
2162
- * because they don't drift, can't be Goodhart-gamed by the policy in the
2163
- * same way, and don't require a separate calibration loop.
2164
- *
2165
- * The `MultiLayerVerifier` already produces this signal — it just doesn't
2166
- * surface it in a shape that's clean enough for RL training. This module
2167
- * wraps the verifier output so consumers can:
2168
- *
2169
- * 1. Extract a clean `VerifiableReward` from a `VerificationReport`
2170
- * 2. Distinguish *deterministic* rewards (compile, test, schema) from
2171
- * *probabilistic* rewards (judge) so they can be weighted differently
2172
- * in the RL training step
2173
- * 3. Filter `RunRecord[]` to only those with a verifiable reward,
2174
- * producing the clean training set that DeepSeek-R1-style GRPO and
2175
- * AlphaProof-style search both depend on
2176
- *
2177
- * Why this matters: every credible 2025-2026 frontier RL result on coding
2178
- * agents leans on verifiable reward (DeepSeek-R1 GRPO on test pass-rate,
2179
- * o-series RL on math/code, AlphaProof on Lean kernel checking). Mixing
2180
- * judge scores into the reward signal poisons the gradient. This module
2181
- * is the seam.
2182
- */
2183
-
2184
- type VerifiableRewardSource = 'compile' | 'test' | 'schema' | 'sandbox' | 'judge' | 'composite';
2185
- interface VerifiableReward {
2186
- /** Scalar in [0, 1]. The RL training signal. */
2187
- value: number;
2188
- /** What produced the reward — different sources have different determinism. */
2189
- source: VerifiableRewardSource;
2190
- /**
2191
- * Determinism class. `'deterministic'` rewards are repeatable byte-for-byte
2192
- * given the same inputs (compile, test, schema validation, sandbox exit code).
2193
- * `'probabilistic'` rewards depend on a stochastic component (LLM judge).
2194
- * Mixing these in the same training batch without separation is a known
2195
- * footgun in production RLHF pipelines.
2196
- */
2197
- determinism: 'deterministic' | 'probabilistic';
2198
- /**
2199
- * Confidence in the reward value. For deterministic sources this is 1.0
2200
- * (the bit either flipped or didn't). For judge sources this is the
2201
- * judge-reported confidence or — when missing — a calibrated prior.
2202
- */
2203
- confidence: number;
2204
- /** The layer / judge id that produced the signal, for provenance. */
2205
- origin: string;
2206
- /**
2207
- * Per-source contribution to `value`, keyed by layer/judge id. Single-source
2208
- * rewards carry one entry (`{ [origin]: value }`); composite rewards carry
2209
- * every contributing layer's score — the anti-scalar-collapse surface RL
2210
- * consumers weight per-source instead of trusting one blended number.
2211
- */
2212
- components: Record<string, number>;
2213
- /**
2214
- * @deprecated Read `components` for per-source reward values. Kept for
2215
- * published-API compatibility: single-source rewards carry the layer's
2216
- * diagnostics here (e.g. `{ tests_passed: 7 }`); composite rewards carry
2217
- * the same per-layer scores `components` now holds.
2218
- */
2219
- breakdown?: Record<string, number>;
2220
- }
2221
- interface VerifiableRewardExtractionOptions {
2222
- /**
2223
- * Which layers count as deterministic-reward sources. The verifier doesn't
2224
- * tag layers as "this is verifiable"; the caller declares it via this list
2225
- * (or via the layer name → source mapping). Default treats common names
2226
- * (`install`, `typecheck`, `build`, `lint`, `test`, `compile`, `schema`,
2227
- * `sandbox`) as deterministic.
2228
- */
2229
- deterministicLayers?: string[];
2230
- /**
2231
- * Map layer name → reward source. Defaults to a sensible string-match.
2232
- */
2233
- sourceFor?: (layerName: string) => VerifiableRewardSource;
2234
- /**
2235
- * Whether to fall back to a probabilistic (judge) reward when no
2236
- * deterministic layer produced a numeric score. Default `true`. Set to
2237
- * `false` for "deterministic-only" training pipelines that should
2238
- * discard runs without a verifiable signal.
2239
- */
2240
- fallbackToJudge?: boolean;
2241
- /**
2242
- * Default confidence for probabilistic (judge) rewards when the judge
2243
- * doesn't report one. Default `0.7`.
2244
- */
2245
- judgeConfidenceFloor?: number;
2246
- }
2247
- /**
2248
- * Extract a `VerifiableReward` from a `VerificationReport`.
2249
- *
2250
- * Strategy: prefer the deterministic layers (in order: test → compile →
2251
- * schema → sandbox), fall back to the judge layer if `fallbackToJudge` is
2252
- * true, return `null` if no signal qualifies. When multiple deterministic
2253
- * layers contribute, return a `'composite'` source with a weighted blend.
2254
- */
2255
- declare function extractVerifiableReward(report: VerificationReport, opts?: VerifiableRewardExtractionOptions): VerifiableReward | null;
2256
- /**
2257
- * Extract verifiable rewards from `RunRecord[]` produced via the
2258
- * `verificationReportToRunRecord` adapter (which encodes per-layer scores
2259
- * in `outcome.raw['layer.<name>']`). For records that don't carry layer
2260
- * scores, returns `null` for that record.
2261
- *
2262
- * This is the canonical bridge from "campaign-shaped artifacts" to
2263
- * "RL-training-ready reward signals": every record that has a clean
2264
- * verifiable reward becomes a training datum, every record that doesn't
2265
- * gets filtered out (or kept with `'probabilistic'` determinism for
2266
- * separate downstream handling).
2267
- */
2268
- declare function extractVerifiableRewardsFromRecords(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
2269
- runId: string;
2270
- reward: VerifiableReward | null;
2271
- }>;
2272
- /** Filter `RunRecord[]` to those with deterministic verifiable rewards. */
2273
- declare function filterDeterministicallyRewarded(runs: RunRecord[], opts?: VerifiableRewardExtractionOptions): Array<{
2274
- run: RunRecord;
2275
- reward: VerifiableReward;
2276
- }>;
2277
-
2278
- /**
2279
- * Reward hacking / Goodhart detection.
2280
- *
2281
- * Goodhart's Law says: when a measure becomes a target, it ceases to be
2282
- * a good measure. In RLHF and agentic-RL settings this is the dominant
2283
- * failure mode — the policy learns to produce outputs that score well on
2284
- * the proxy reward (judge, rubric, test pass-rate) without producing
2285
- * the underlying capability the proxy was meant to track.
2286
- *
2287
- * Krakovna et al. (2020, "Specification Gaming Examples in AI") and the
2288
- * subsequent RLHF reward-hacking literature (Skalse et al. 2022, Kim et al.
2289
- * 2023) converge on a few diagnostic signatures:
2290
- *
2291
- * 1. **Reward divergence:** the proxy reward grows while the held-out
2292
- * ground-truth signal stagnates or drops. Predictive validity over
2293
- * time captures this.
2294
- * 2. **Distributional shift in outputs:** after RL, the policy produces
2295
- * outputs that no longer match the reference distribution — usually
2296
- * because it found a high-reward attractor that's degenerate (e.g.
2297
- * one-token responses, repetition, formatting tricks).
2298
- * 3. **Disagreement between independent rewards:** if you train on
2299
- * reward A and a held-out independent reward B drops sharply, you're
2300
- * probably hacking A.
2301
- * 4. **Calibration drift:** the verifiable / deterministic component of
2302
- * the reward is stable; the probabilistic / judge component drifts up
2303
- * while the deterministic component doesn't. The judge is being
2304
- * gamed.
2305
- *
2306
- * This module ships explicit detectors for all four signatures, plus a
2307
- * combined verdict. The output is diagnostic — actionable signals,
2308
- * not autoreject — because each signature has known false positives
2309
- * (e.g., a policy that genuinely improves can show distributional shift).
2310
- *
2311
- * Differs from `rubricPredictiveValidity` (which is a *standing* check on
2312
- * whether rubrics correlate with deployment outcomes) — this is a
2313
- * *temporal* check on whether the reward-vs-truth gap is *widening over
2314
- * time during a training run*.
2315
- */
2316
-
2317
- type RewardHackingSignal = 'reward_divergence' | 'distribution_shift' | 'reward_disagreement' | 'judge_drift';
2318
- interface RewardHackingFinding {
2319
- signal: RewardHackingSignal;
2320
- /** Severity in [0, 1]. >0.5 = strong signal. */
2321
- severity: number;
2322
- message: string;
2323
- /** Numeric evidence the consumer can render. */
2324
- detail: Record<string, number>;
2325
- }
2326
- interface RewardHackingReport {
2327
- findings: RewardHackingFinding[];
2328
- /**
2329
- * Composite verdict. `'clean'` if every signal severity < 0.3;
2330
- * `'suspect'` if at least one ≥ 0.3 but none ≥ 0.6; `'gaming'` if any ≥ 0.6.
2331
- */
2332
- verdict: 'clean' | 'suspect' | 'gaming';
2333
- /** Rationale for the verdict, ready to paste into an audit log. */
2334
- rationale: string[];
2335
- /** Number of paired (proxy, truth) data points the report saw. */
2336
- n: number;
2337
- }
2338
- interface DetectRewardHackingInput {
2339
- /**
2340
- * Run records ordered by recency (oldest first). The detector segments
2341
- * them into prefix/suffix windows to compute "did the gap widen."
2342
- */
2343
- runs: RunRecord[];
2344
- /**
2345
- * The metric the policy was trained to optimize. Should be present on
2346
- * `outcome.raw` or `outcome.holdoutScore`. Default reads `outcome.holdoutScore`.
2347
- */
2348
- proxyOf?: (run: RunRecord) => number | null;
2349
- /**
2350
- * The held-out ground-truth metric. For RL on coding, this is typically
2351
- * test pass-rate. For RLHF, it's downstream task performance or human
2352
- * preference. For knowledge tasks, it's an independently-graded score.
2353
- */
2354
- truthOf?: (run: RunRecord) => number | null;
2355
- /**
2356
- * Independent secondary reward. Used for the `reward_disagreement`
2357
- * signal. Default uses the verifiable reward extractor (deterministic
2358
- * sources only).
2359
- */
2360
- secondaryRewardOf?: (run: RunRecord) => number | null;
2361
- /**
2362
- * Window size — how many of the most recent runs count as the "after"
2363
- * cohort. Default min(50, half the runs).
2364
- */
2365
- windowSize?: number;
2366
- /**
2367
- * Severity threshold to flag a signal. Default 0.3 (suspect) and 0.6
2368
- * (gaming).
2369
- */
2370
- thresholds?: {
2371
- suspect?: number;
2372
- gaming?: number;
2373
- };
2374
- /**
2375
- * Verifiable-reward options used for the secondary-reward fallback.
2376
- */
2377
- verifiableRewardOptions?: VerifiableRewardExtractionOptions;
2378
- }
2379
- declare function detectRewardHacking(input: DetectRewardHackingInput): RewardHackingReport;
2380
-
2381
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
2382
- interface ChannelRollup {
2383
- channel: CostChannel;
2384
- calls: number;
2385
- inputTokens: number;
2386
- outputTokens: number;
2387
- reasoningTokens?: number;
2388
- cachedTokens: number;
2389
- cacheWriteTokens?: number;
2390
- costUsd: number;
2391
- unpricedCalls: number;
2392
- unknownUsageCalls: number;
2393
- }
2394
- interface CostLedgerSummary {
2395
- totalCalls: number;
2396
- pendingCalls: number;
2397
- unresolvedCalls: number;
2398
- reservedCostUsd: number;
2399
- inputTokens: number;
2400
- outputTokens: number;
2401
- reasoningTokens?: number;
2402
- cachedTokens: number;
2403
- cacheWriteTokens?: number;
2404
- totalCostUsd: number;
2405
- byChannel: ChannelRollup[];
2406
- unpricedModels: string[];
2407
- fullyPriced: boolean;
2408
- usageComplete: boolean;
2409
- accountingComplete: boolean;
2410
- incompleteReasons: string[];
2411
- }
2412
-
2413
- /**
2414
- * RawProviderSink — first-class persistence for the actual HTTP-level
2415
- * request/response bodies of every LLM provider call.
2416
- *
2417
- * Why this is a separate sink from the structured `LlmSpan`:
2418
- *
2419
- * - `LlmSpan` records the *intent* — model name, messages, output text,
2420
- * usage. It's what dashboards read; it's NOT enough for forensics.
2421
- * - When a downstream consumer reports "the verifier used the wrong route"
2422
- * or "tokens look right but reasoning was missing," the only way to
2423
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
2424
- * a different `model` value than what actually answered); the raw
2425
- * response is ground truth.
2426
- *
2427
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
2428
- * matrix runner / BuilderSession sets it up automatically) and every
2429
- * request, response, and error is recorded — including retries, with the
2430
- * attempt index attached so a flaky call's full event chain is recoverable.
2431
- *
2432
- * Redaction is enforced at sink time. The default redactor strips
2433
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
2434
- * payload field whose key matches `apiKey | api_key | bearer | password |
2435
- * secret | token` (case-insensitive). Override via the sink constructor or
2436
- * the per-call `redactor`. The `redactedFields` array on the persisted
2437
- * event lets a reviewer see what was stripped without exposing the values.
2438
- */
2439
- type RawProviderDirection = 'request' | 'response' | 'error';
2440
- interface RawProviderEvent {
2441
- /** Stable id. Generated by the sink if omitted. */
2442
- eventId: string;
2443
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
2444
- runId?: string;
2445
- spanId?: string;
2446
- /**
2447
- * Logical provider name. Free-form so callers can use whatever id matches
2448
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
2449
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
2450
- */
2451
- provider: string;
2452
- model: string;
2453
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
2454
- endpoint: string;
2455
- /** Base URL used for the call (already-normalised — no trailing slash). */
2456
- baseUrl: string;
2457
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
2458
- attemptIndex: number;
2459
- direction: RawProviderDirection;
2460
- /** Unix ms. */
2461
- timestamp: number;
2462
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
2463
- durationMs?: number;
2464
- statusCode?: number;
2465
- requestHeaders?: Record<string, string>;
2466
- requestBody?: unknown;
2467
- responseHeaders?: Record<string, string>;
2468
- responseBody?: unknown;
2469
- /** Set on `direction: 'error'` events. */
2470
- errorMessage?: string;
2471
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
2472
- redactedFields: string[];
2473
- }
2474
- interface RawProviderSinkFilter {
2475
- runId?: string;
2476
- spanId?: string;
2477
- direction?: RawProviderDirection;
2478
- attemptIndex?: number;
2479
- }
2480
- interface RawProviderSink {
2481
- record(event: RawProviderEvent): Promise<void>;
2482
- /** Optional listing — implementations that durably persist (file, db) should support this. */
2483
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
2484
- /** Optional teardown for backed implementations. */
2485
- close?(): Promise<void>;
2486
- }
2487
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
2488
-
2489
- /**
2490
- * LLM client with graceful degrade.
2491
- *
2492
- * OpenAI-compatible `/v1/chat/completions` client with:
2493
- * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
2494
- * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
2495
- * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
2496
- * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
2497
- * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
2498
- * directly, cli-bridge subscriptions, and any router that speaks the spec.
2499
- *
2500
- * Usage:
2501
- * const { value, result } = await callLlmJson<MyType>(
2502
- * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
2503
- * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
2504
- * )
2505
- *
2506
- * This is THE llm-calling seam for agent-eval primitives that need structured
2507
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
2508
- * that need free-form text use `callLlm` and parse output themselves.
2509
- */
2510
-
2511
- interface LlmUsage {
2512
- promptTokens: number;
2513
- completionTokens: number;
2514
- totalTokens: number;
2515
- /** False when the provider omitted or malformed prompt/completion usage. */
2516
- captured?: boolean;
2517
- /** Proxies populate this when prompt caching is on. */
2518
- cachedPromptTokens?: number;
2519
- }
2520
- interface LlmCallResult {
2521
- /** The text content of the first choice. Empty string if none. */
2522
- content: string;
2523
- usage: LlmUsage;
2524
- /**
2525
- * Cost in USD. Pulled from proxy's `_response_cost` field when present;
2526
- * `null` when neither the proxy nor the caller can derive it.
2527
- */
2528
- costUsd: number | null;
2529
- /** Model name actually used (echoed from response). */
2530
- model: string;
2531
- /** Wall-clock duration of the HTTP call (last attempt, if retried). */
2532
- durationMs: number;
2533
- /**
2534
- * `finish_reason` echoed from the first choice (`stop`, `length`,
2535
- * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
2536
- * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
2537
- * (`length`) instead of treating a cut-off completion as complete. Note:
2538
- * `callLlm` does not itself reject on it — acting on this signal is the
2539
- * caller's responsibility (in-repo free-form drivers do not yet enforce it).
2540
- */
2541
- finishReason?: string | null;
2542
- /**
2543
- * True when `content.trim()` is empty. An empty completion is a silent zero
2544
- * for free-form `callLlm` callers; this flag is the signal a caller can
2545
- * inspect to fail loud rather than proceed on an empty string. `callLlm`
2546
- * surfaces it but does not throw on it.
2547
- */
2548
- contentEmpty?: boolean;
2549
- /** Raw response body. */
2550
- raw: Record<string, unknown>;
2551
- }
2552
- type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
2553
- interface LlmClientOptions {
2554
- /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
2555
- baseUrl?: string;
2556
- /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
2557
- apiKey?: string;
2558
- bearer?: string;
2559
- /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
2560
- authHeader?: {
2561
- name: string;
2562
- value: string;
2563
- };
2564
- /** Stable provider idempotency key, reused across retries of this logical call. */
2565
- idempotencyKey?: string;
2566
- /** Default timeout in ms. Per-call can override. */
2567
- defaultTimeoutMs?: number;
2568
- /**
2569
- * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
2570
- * each attempt's per-attempt timeout controller, so aborting it cancels
2571
- * the in-flight fetch. A caller abort is FATAL: it is not retried even
2572
- * though an AbortError otherwise matches the transient patterns.
2573
- */
2574
- signal?: AbortSignal;
2575
- /**
2576
- * Cross-attempt wall-clock budget in ms, measured from the first attempt.
2577
- * Before launching each attempt the loop checks the remaining budget and
2578
- * stops retrying once it is exhausted, rather than waiting the full
2579
- * per-attempt timeout on every retry. Bounds total time independent of
2580
- * total attempts × `timeoutMs`.
2581
- */
2582
- deadlineMs?: number;
2583
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
2584
- maxRetries?: number;
2585
- /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
2586
- fetch?: typeof fetch;
2587
- /**
2588
- * Optional raw HTTP capture sink. When provided, every request, response,
2589
- * and error (across all retry attempts) is recorded to the sink, with auth
2590
- * headers and credential-shaped body fields redacted by default. This is
2591
- * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
2592
- * raw events record what actually crossed the wire.
2593
- */
2594
- rawSink?: RawProviderSink;
2595
- /**
2596
- * Logical provider id attached to raw events. When omitted, derived from
2597
- * `baseUrl` via `providerFromBaseUrl`.
2598
- */
2599
- provider?: string;
2600
- /** Trace context attached to raw events; populated by emitter-aware callers. */
2601
- traceContext?: {
2602
- runId?: string;
2603
- spanId?: string;
2604
- };
2605
- /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
2606
- redactor?: ProviderRedactor;
2607
- }
2608
- interface LlmRouteRequirements {
2609
- /**
2610
- * Throw if `opts.baseUrl` is undefined, i.e. the call would fall back to
2611
- * `DEFAULT_BASE_URL`. Set this for evaluation runs where silently using
2612
- * the public/free-tier router is a defect — the launch reviewer needs to
2613
- * know exactly which provider answered.
2614
- */
2615
- requireExplicitBaseUrl?: boolean;
2616
- /**
2617
- * Allowlist of acceptable base URLs. Strings match by prefix
2618
- * (case-insensitive); RegExps test against the full base URL.
2619
- */
2620
- allowedBaseUrls?: Array<string | RegExp>;
2621
- /** Blocklist that takes precedence over `allowedBaseUrls`. */
2622
- blockedBaseUrls?: Array<string | RegExp>;
2623
- /** Throw if no auth header / api key is configured. */
2624
- requireAuth?: boolean;
2625
- /**
2626
- * Logical provider id the configured `baseUrl` is expected to match (via
2627
- * `providerFromBaseUrl`). Mainly useful when paired with `requireExplicitBaseUrl`.
2628
- */
2629
- expectedProvider?: string;
2630
- }
2631
-
2632
- /**
2633
- * FailureClusterView — groups failed runs by (failureClass, triggerTool,
2634
- * argHash-prefix) so weekly reviews can prioritize the top-N clusters.
2635
- *
2636
- * Each cluster includes: N runs, scenarios affected, representative
2637
- * error message, a proposed mitigation hint (rule → action table).
2638
- */
2639
-
2640
- interface FailureCluster {
2641
- failureClass: FailureClass;
2642
- /** Tool name when the trigger was a tool span, else undefined. */
2643
- toolName?: string;
2644
- /** First 16 chars of argHash — clusters similar args. */
2645
- argPrefix?: string;
2646
- /**
2647
- * Source dimension when the trigger was a judge span (e.g. `'format'`,
2648
- * `'safety'`, `'correctness'`). Lets cross-template aggregators
2649
- * group failures by the dimension that fired without overloading
2650
- * `argPrefix`. Optional — clusters without this field deserialize cleanly.
2651
- */
2652
- dimension?: string;
2653
- runCount: number;
2654
- scenarioIds: string[];
2655
- exampleError?: string;
2656
- exampleRunId: string;
2657
- }
2658
- interface FailureClusterReport {
2659
- clusters: FailureCluster[];
2660
- totalFailures: number;
2661
- totalRuns: number;
2662
- }
2663
-
2664
- /**
2665
- * Reporting helpers — production summaries and paper-quality figures — sit alongside `reporter.ts` rather
2666
- * than replacing it.
2667
- *
2668
- * Three artefacts:
2669
- *
2670
- * - `summaryTable` Markdown table of per-candidate means,
2671
- * 95% bootstrap CIs, BH-adjusted Wilcoxon
2672
- * p-values, and Cohen's d versus a
2673
- * comparator candidate.
2674
- * - `paretoChart` Abstract spec for a cost vs quality
2675
- * scatter, with gate decisions overlaid.
2676
- * Returns numbers + labels — caller
2677
- * chooses the plotting library.
2678
- * - `gainHistogram`
2679
- * Per-item paired holdout deltas as a
2680
- * histogram spec (bins + counts + median +
2681
- * CI). Same "data, not images" contract.
2682
- *
2683
- * The figure types are PlotSpecs — JSON-friendly, library-agnostic.
2684
- * They aren't React components and they aren't PNGs; they are
2685
- * what you'd hand to vega-lite, plotly, matplotlib, or your own
2686
- * Canvas renderer to draw the actual figure.
2687
- */
2688
-
2689
- interface SummaryTableRow {
2690
- candidateId: string;
2691
- n: number;
2692
- mean: number;
2693
- ciLow: number;
2694
- ciHigh: number;
2695
- /** BH-adjusted q-value vs comparator. NaN if no comparator. */
2696
- qValue: number;
2697
- /** Cohen's d vs comparator. NaN if no comparator. */
2698
- cohensD: number;
2699
- }
2700
- interface SummaryTable {
2701
- rows: SummaryTableRow[];
2702
- comparator: string | null;
2703
- split: 'search' | 'holdout';
2704
- /** Pre-rendered markdown — drop into a paper or PR. */
2705
- markdown: string;
2706
- }
2707
- interface ParetoPoint {
2708
- candidateId: string;
2709
- /** Mean USD cost per run on the chosen split. */
2710
- cost: number;
2711
- /** Mean score on the chosen split. */
2712
- quality: number;
2713
- /** Number of runs that informed this point. */
2714
- n: number;
2715
- /** Whether this candidate is on the Pareto frontier — high
2716
- * quality, low cost, no dominator. */
2717
- onFrontier: boolean;
2718
- /** Optional gate verdict for this candidate, if a `GateDecision`
2719
- * for it was passed in. */
2720
- gate?: 'promote' | 'reject_few_runs' | 'reject_negative_delta' | 'reject_overfit_gap' | null;
2721
- }
2722
- interface ParetoFigureSpec {
2723
- kind: 'pareto-cost-quality';
2724
- split: 'search' | 'holdout';
2725
- points: ParetoPoint[];
2726
- axes: {
2727
- x: 'costUsd';
2728
- y: 'score';
2729
- };
2730
- }
2731
- interface GainDistributionBin {
2732
- /** Inclusive lower edge. */
2733
- lo: number;
2734
- /** Exclusive upper edge (or inclusive if it's the last bin). */
2735
- hi: number;
2736
- /** Number of pairs whose delta lands in this bin. */
2737
- count: number;
2738
- }
2739
- interface GainDistributionFigureSpec {
2740
- kind: 'gain-distribution';
2741
- candidateId: string;
2742
- comparator: string;
2743
- split: 'search' | 'holdout';
2744
- /** Number of pairs used. */
2745
- n: number;
2746
- bins: GainDistributionBin[];
2747
- median: number;
2748
- ci: {
2749
- low: number;
2750
- high: number;
2751
- };
2752
- }
2753
- type ResearchReportDecision = 'promote' | 'hold' | 'reject' | 'equivalent' | 'needs_more_data';
2754
- interface ResearchReportOptions {
2755
- /** Human-readable report title. */
2756
- title?: string;
2757
- /** Comparator candidate id. Required for statistical decision guidance. */
2758
- comparator?: string;
2759
- /** Which split to use for the primary decision. Default 'holdout'. */
2760
- split?: 'search' | 'holdout';
2761
- /** Confidence level used by lower-level report helpers. Default 0.95. */
2762
- confidence?: number;
2763
- /** FDR threshold for q-values. Default 0.05. */
2764
- fdr?: number;
2765
- /**
2766
- * Soft floor on paired observations before issuing a directional
2767
- * promote / reject. Below this we report `needs_more_data` and surface the
2768
- * minimum detectable effect at the current N. Default 20 — chosen so the
2769
- * Wilcoxon signed-rank approximation is reasonable and so the paired
2770
- * bootstrap CI has non-degenerate coverage. Hard floor is enforced at
2771
- * `RESEARCH_REPORT_HARD_PAIR_FLOOR` (6) regardless of this value.
2772
- */
2773
- minPairs?: number;
2774
- /**
2775
- * Region of Practical Equivalence on the paired delta. When a candidate's
2776
- * paired-delta CI is fully contained in `[low, high]`, the decision is
2777
- * `equivalent` rather than `hold`. Sourced from the domain owner — there is
2778
- * no statistically-defensible default.
2779
- */
2780
- rope?: {
2781
- low: number;
2782
- high: number;
2783
- };
2784
- /**
2785
- * Power for the minimum detectable effect (MDE) reported on each candidate.
2786
- * Default 0.8.
2787
- */
2788
- mdePower?: number;
2789
- /**
2790
- * Two-sided alpha for the MDE. Default matches `fdr` so the reported MDE
2791
- * lines up with the test the report actually runs.
2792
- */
2793
- mdeAlpha?: number;
2794
- /** Optional held-out gate decisions keyed by candidate id. */
2795
- gateDecisions?: Record<string, GateDecision$1>;
2796
- /** Optional failure clusters from failureClusterView. */
2797
- failureClusters?: FailureClusterReport;
2798
- /** Build gain histograms for these candidates. Defaults to all non-comparator candidates. */
2799
- candidateIds?: string[];
2800
- /** Deterministic bootstrap seed passed to gainHistogram and the posterior helper. */
2801
- seed?: number;
2802
- /** Report timestamp. Defaults to current time. */
2803
- generatedAt?: string;
2804
- /**
2805
- * Hash of a preregistered protocol (e.g. `signManifest({...}).contentHash`).
2806
- * Embedded verbatim in the report so the analysis can be cited as the
2807
- * preregistered one rather than a post-hoc fishing expedition.
2808
- */
2809
- preregistrationHash?: string;
2810
- }
2811
- interface ResearchReportRecommendation {
2812
- decision: ResearchReportDecision;
2813
- candidateId: string | null;
2814
- rationale: string[];
2815
- risks: string[];
2816
- nextActions: string[];
2817
- }
2818
- interface ResearchReportCandidate {
2819
- candidateId: string;
2820
- n: number;
2821
- mean: number;
2822
- ciLow: number;
2823
- ciHigh: number;
2824
- qValue: number;
2825
- cohensD: number;
2826
- meanDeltaVsComparator: number | null;
2827
- pairedN: number;
2828
- medianGain: number | null;
2829
- meanGain: number | null;
2830
- gainCi: {
2831
- low: number;
2832
- high: number;
2833
- } | null;
2834
- /**
2835
- * Bayesian-bootstrap-style posterior summaries on the paired delta. Computed
2836
- * from the same resamples that produce the gain CI; interpretable as
2837
- * "fraction of resamples in which the candidate beats the comparator on
2838
- * matched pairs."
2839
- */
2840
- prGreaterThanZero: number | null;
2841
- prInRope: number | null;
2842
- /**
2843
- * Minimum detectable effect (in score units) at the candidate's paired N,
2844
- * the configured power, and the configured alpha. Standardised by the
2845
- * observed paired-delta SD and inverted via `requiredSampleSize`. Reported
2846
- * for every candidate so a `needs_more_data` verdict is actionable.
2847
- */
2848
- mde: number | null;
2849
- onParetoFrontier: boolean;
2850
- gate?: ParetoPoint['gate'];
2851
- decision: ResearchReportDecision;
2852
- decisionReason: string;
2853
- }
2854
- interface ResearchReportMethodology {
2855
- /**
2856
- * Plain-language assumptions the report depends on. Read these first when
2857
- * deciding whether the verdict is load-bearing for a launch decision.
2858
- */
2859
- assumptions: string[];
2860
- /** Tests and estimators the verdict was computed from. */
2861
- methods: string[];
2862
- /** Alternatives the author considered and why this report didn't take them. */
2863
- alternatives: string[];
2864
- /** Failure modes — when this report should NOT drive a decision. */
2865
- whenNotToApply: string[];
2866
- /** Citations for the methodological choices above. */
2867
- citations: string[];
2868
- }
2869
- interface ResearchReport {
2870
- kind: 'agent-eval-research-report';
2871
- title: string;
2872
- generatedAt: string;
2873
- split: 'search' | 'holdout';
2874
- comparator: string | null;
2875
- /**
2876
- * SHA-256 over the canonicalised set of `(runId, candidateId, split)` triples
2877
- * the report was computed from, plus the comparator and split. Stable across
2878
- * key insertion order; recomputable by the reader to verify provenance.
2879
- */
2880
- runFingerprint: string;
2881
- preregistrationHash: string | null;
2882
- rope: {
2883
- low: number;
2884
- high: number;
2885
- } | null;
2886
- executiveSummary: string[];
2887
- recommendation: ResearchReportRecommendation;
2888
- candidates: ResearchReportCandidate[];
2889
- summary: SummaryTable;
2890
- charts: {
2891
- pareto: ParetoFigureSpec;
2892
- gains: GainDistributionFigureSpec[];
2893
- };
2894
- methodology: ResearchReportMethodology;
2895
- failureClusters?: FailureClusterReport;
2896
- markdown: string;
2897
- html: string;
2898
- }
2899
-
2900
- /**
2901
- * TraceEmitter — hierarchical span builder that auto-parents using an
2902
- * internal stack. One emitter per Run; emitters do NOT share state.
2903
- *
2904
- * Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)
2905
- * return a `SpanHandle` with `.end()` / `.fail()` so callers don't
2906
- * have to thread spanIds manually. For async workflows that can't use
2907
- * the stack (e.g. fan-out parallel calls), pass `parentSpanId`
2908
- * explicitly.
2909
- */
2910
-
2911
- interface SpanHandle<S extends Span = Span> {
2912
- span: S;
2913
- end(patch?: Partial<S>): Promise<void>;
2914
- fail(error: string | Error, patch?: Partial<S>): Promise<void>;
2915
- }
2916
- interface RunCompleteHookContext {
2917
- runId: string;
2918
- emitter: TraceEmitter;
2919
- store: TraceStore;
2920
- /** Outcome the caller passed to `endRun` (undefined for `abortRun`). */
2921
- outcome?: RunOutcome$1;
2922
- /** Final run status. */
2923
- status: 'completed' | 'failed' | 'aborted';
2924
- }
2925
- type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void;
2926
- interface TraceEmitterOptions {
2927
- runId?: string;
2928
- /** Inject a clock for deterministic tests. */
2929
- now?: () => number;
2930
- /** Inject an id generator for deterministic tests. */
2931
- id?: () => string;
2932
- /**
2933
- * Hooks fired after `endRun` / `abortRun` writes the final run state.
2934
- * Designed for trace-analyst auto-execution, integrity assertions, and
2935
- * outbound notifications. Hooks run sequentially in the order supplied.
2936
- *
2937
- * By default a hook that throws is swallowed and logged as a `note` event
2938
- * on the run — auto-orchestration must not crash the underlying flow.
2939
- * Set `hookErrors: 'throw'` to propagate.
2940
- */
2941
- onRunComplete?: RunCompleteHook[];
2942
- /** `'swallow'` (default) | `'throw'`. */
2943
- hookErrors?: 'swallow' | 'throw';
2944
- }
2945
- declare class TraceEmitter {
2946
- private store;
2947
- private stack;
2948
- private _runId;
2949
- private now;
2950
- private id;
2951
- private hooks;
2952
- private hookErrors;
2953
- constructor(store: TraceStore, options?: TraceEmitterOptions);
2954
- get runId(): string;
2955
- get traceStore(): TraceStore;
2956
- /** Append a hook after construction (e.g. attach the trace analyst). */
2957
- addRunCompleteHook(hook: RunCompleteHook): void;
2958
- /**
2959
- * Begin a Run.
2960
- *
2961
- * `scenarioId` is required on the persisted Run shape — every Run downstream
2962
- * gets a non-empty scenarioId so filters and aggregations stay simple — but
2963
- * the INPUT here accepts it as optional. When omitted, startRun substitutes
2964
- * a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so
2965
- * runtime / operator / meta-eval runs that have no curated-scenario corpus
2966
- * to anchor to don't have to invent placeholder strings at the call site.
2967
- */
2968
- startRun(run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & {
2969
- scenarioId?: string;
2970
- }): Promise<Run>;
2971
- endRun(outcome?: RunOutcome$1): Promise<void>;
2972
- abortRun(reason: string): Promise<void>;
2973
- private runHooks;
2974
- span<S extends Span = Span>(init: {
2975
- kind: SpanKind;
2976
- name: string;
2977
- parentSpanId?: string;
2978
- attributes?: Record<string, unknown>;
2979
- } & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>): Promise<SpanHandle<S>>;
2980
- private handle;
2981
- private pop;
2982
- llm(init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<LlmSpan>>;
2983
- tool(init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<ToolSpan>>;
2984
- retrieval(init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<RetrievalSpan>>;
2985
- recordJudge(verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>): Promise<JudgeSpan>;
2986
- sandbox(init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<SandboxSpan>>;
2987
- emit(event: {
2988
- kind: EventKind;
2989
- spanId?: string;
2990
- payload?: Record<string, unknown>;
2991
- }): Promise<TraceEvent>;
2992
- recordBudget(entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & {
2993
- timestamp?: number;
2994
- }): Promise<BudgetLedgerEntry>;
2995
- recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact>;
2996
- /**
2997
- * Runs `fn` inside a span; auto-ends on success, auto-fails on throw.
2998
- * Returns the fn's return value. Use this for the 95% case.
2999
- */
3000
- within<T>(init: Parameters<TraceEmitter['span']>[0], fn: (handle: SpanHandle) => Promise<T>): Promise<T>;
3001
- }
3002
-
3003
- /**
3004
- * Run-completion integrity check — at end of run, verify the expected event
3005
- * types were actually captured. The point is the launch-review failure mode:
3006
- * a run *appears* successful but the raw provider events were never written,
3007
- * so a downstream reviewer can't reconstruct what happened.
3008
- *
3009
- * Pattern:
3010
- *
3011
- * const report = await assertRunCaptured(store, runId, {
3012
- * llmSpansMin: 1,
3013
- * judgeSpansMin: 1,
3014
- * rawSink: providerSink, // must have ≥ 1 event for this run
3015
- * requireRawCoverageOfLlmSpans: true, // every llm span has matching raw events
3016
- * })
3017
- * if (!report.ok) throwIfRunIncomplete(report) // or mark run failed and continue
3018
- *
3019
- * The function is read-only on the store and returns a structured report;
3020
- * the caller chooses the failure mode (throw, mark run failed, log warning).
3021
- * `throwIfRunIncomplete` is the convenient strict mode.
3022
- */
3023
-
3024
- interface RunIntegrityExpectations {
3025
- /** Minimum LLM span count. Default 0 (no requirement). */
3026
- llmSpansMin?: number;
3027
- /** Minimum judge span count. Default 0. */
3028
- judgeSpansMin?: number;
3029
- /** Minimum tool span count. Default 0. */
3030
- toolSpansMin?: number;
3031
- /**
3032
- * Raw provider sink to consult for capture verification. When present,
3033
- * the check requires at least one raw event for the run.
3034
- */
3035
- rawSink?: RawProviderSink;
3036
- /** Minimum raw provider event count. Default 0; ignored when `rawSink` absent. */
3037
- rawProviderEventsMin?: number;
3038
- /**
3039
- * Every LLM span must have at least one matching raw `request` event
3040
- * (matched by spanId). Catches the common bug where the structured span
3041
- * was emitted but the raw HTTP capture was wired to a different sink.
3042
- */
3043
- requireRawCoverageOfLlmSpans?: boolean;
3044
- /** Run outcome must be set (not null/undefined). Default false. */
3045
- requireOutcome?: boolean;
3046
- }
3047
- type RunIntegrityIssueCode = 'no_run' | 'missing_llm_spans' | 'missing_judge_spans' | 'missing_tool_spans' | 'missing_raw_events' | 'no_raw_sink' | 'orphan_llm_span' | 'missing_outcome';
3048
- interface RunIntegrityIssue {
3049
- code: RunIntegrityIssueCode;
3050
- message: string;
3051
- detail?: Record<string, unknown>;
3052
- }
3053
- interface RunIntegrityReport {
3054
- ok: boolean;
3055
- runId: string;
3056
- llmSpanCount: number;
3057
- judgeSpanCount: number;
3058
- toolSpanCount: number;
3059
- rawProviderEventCount: number;
3060
- /**
3061
- * Coverage of LLM spans by raw provider events keyed on spanId.
3062
- * `total` is the number of LLM spans; `covered` is the count with at
3063
- * least one matching `request` raw event.
3064
- */
3065
- rawSpanCoverage: {
3066
- covered: number;
3067
- total: number;
3068
- };
3069
- issues: RunIntegrityIssue[];
3070
- }
3071
-
3072
- /**
3073
- * EvalCampaign — opinionated matrix runner that wires the four
3074
- * capture-integrity directives by construction.
3075
- *
3076
- * The canonical benchmark shape — matrix runner → for each
3077
- * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the
3078
- * run → analyze — has a bug class at the integration boundary: raw
3079
- * events not captured, route silently wrong, integrity not asserted,
3080
- * analyst never run. The directives in `SKILL.md § Capture integrity`
3081
- * are the mitigations.
3082
- *
3083
- * `EvalCampaign` is the structural fix — consumers don't wire the
3084
- * integrity surface themselves; the campaign owns it. Specifically:
3085
- *
3086
- * - calls `assertLlmRoute` once at preflight before any work runs
3087
- * - constructs a per-run `TraceStore` and `RawProviderSink` via factories
3088
- * - constructs the `TraceEmitter` with `onRunComplete: [analyst hook]`
3089
- * - hands the runner an `LlmClientOptions` pre-wired with the sink and
3090
- * trace context — the runner can't accidentally call an LLM without
3091
- * capturing the raw HTTP envelope
3092
- * - calls `assertRunCaptured` after every `endRun` and routes failures
3093
- * through a configurable policy (`throw` / `mark_failed` / `log`)
3094
- * - assembles per-run `RunRecord`s and runs `researchReport` at the end
3095
- * so the campaign artifact is launch-decision-grade by default
3096
- * - embeds the campaign fingerprint (a SHA-256 over the canonicalised
3097
- * run set) and optional `preregistrationHash` in the report
3098
- *
3099
- * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`
3100
- * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped
3101
- * lives in the campaign. This is the inversion-of-control point — consumers
3102
- * stop writing matrix runners and start writing scenario-runners.
3103
- *
3104
- * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):
3105
- *
3106
- * - Distributed/cluster execution (concurrency is local async)
3107
- * - Adaptive sampling / sequential interim looks
3108
- * - Resume from partial state across crashes
3109
- * - LLM-call retry beyond what `LlmClient` already does
3110
- */
3111
-
3112
- interface CampaignVariant<V> {
3113
- id: string;
3114
- payload: V;
3115
- }
3116
- interface CampaignScenario {
3117
- scenarioId: string;
3118
- /** Free-form metadata propagated to runs and reports. */
3119
- tags?: Record<string, string>;
3120
- }
3121
- interface CampaignRunContext<V> {
3122
- /** Stable run id. The campaign generates this; the runner does not. */
3123
- runId: string;
3124
- /** Logical experiment id (campaignId by default; overridable per-run via opts). */
3125
- experimentId: string;
3126
- variant: V;
3127
- variantId: string;
3128
- scenarioId: string;
3129
- scenarioTags: Record<string, string>;
3130
- seed: number;
3131
- splitTag: RunSplitTag;
3132
- /**
3133
- * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired
3134
- * (analyst auto-execution if configured, plus integrity check). The
3135
- * runner MUST call `emitter.startRun` before doing any work and either
3136
- * `emitter.endRun` or `emitter.abortRun` before returning.
3137
- */
3138
- emitter: TraceEmitter;
3139
- store: TraceStore;
3140
- rawSink: RawProviderSink;
3141
- /**
3142
- * Pre-wired LLM client options — `rawSink` and `traceContext` are populated
3143
- * so any `callLlm(req, ctx.llmOpts)` automatically captures raw HTTP. The
3144
- * runner can spread additional fields if needed.
3145
- */
3146
- llmOpts: LlmClientOptions;
3147
- }
3148
- interface CampaignRunOutcome {
3149
- /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */
3150
- pass: boolean;
3151
- /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */
3152
- score: number;
3153
- /** Mandatory cost in USD. Use 0 + raw.cost_unknown=1 only if truly unknown. */
3154
- costUsd: number;
3155
- tokenUsage: RunTokenUsage;
3156
- /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
3157
- model: string;
3158
- /** sha256 of the effective prompt sent to the model. */
3159
- promptHash: string;
3160
- /** sha256 of the effective config (model, temperature, tools, judges, splits). */
3161
- configHash: string;
3162
- /** Optional extra numeric metrics to land in `outcome.raw`. */
3163
- raw?: Record<string, number>;
3164
- /** Canonical cross-agent failure class from the shared `FAILURE_CLASSES`
3165
- * taxonomy. Propagated to `RunRecord.failureClass` so campaign runs
3166
- * aggregate failures in the same vocabulary as every other producer. */
3167
- failureClass?: FailureClass;
3168
- /** Optional free-form failure detail, scoped under `failureClass`. */
3169
- failureMode?: string;
3170
- /** Optional judge metadata when a judge was used. */
3171
- judgeMetadata?: RunJudgeMetadata;
3172
- /**
3173
- * Optional per-judge / per-dim breakdown for ensemble-judged runs.
3174
- * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.
3175
- * Single-judge or scalar-only runs leave this unset.
3176
- */
3177
- judgeScores?: JudgeScoresRecord;
3178
- /**
3179
- * Agent profile cell observed by the runner. When supplied, it overrides
3180
- * `EvalCampaignOptions.agentProfile` for this run and must match the
3181
- * outcome's `model` and `promptHash`.
3182
- */
3183
- agentProfile?: AgentProfileCell | AgentProfileCellInput;
3184
- }
3185
- type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>;
3186
- type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log';
3187
- interface EvalCampaignOptions<V> {
3188
- /**
3189
- * Stable id for the campaign. Used as the default `experimentId` on
3190
- * every run, and folded into the campaign fingerprint.
3191
- */
3192
- campaignId: string;
3193
- variants: CampaignVariant<V>[];
3194
- scenarios: CampaignScenario[];
3195
- /** Default `[0, 1, 2]`. */
3196
- seeds?: number[];
3197
- /** Default `'holdout'` — the split that anchors a launch decision. */
3198
- splitTag?: RunSplitTag;
3199
- /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */
3200
- commitSha: string;
3201
- /**
3202
- * LLM client config. Augmented per-run with `rawSink` and `traceContext`
3203
- * before being passed to the runner. The campaign asserts this config
3204
- * matches `routeRequirements` once at preflight.
3205
- */
3206
- llmOpts: LlmClientOptions;
3207
- /**
3208
- * Default `{ requireExplicitBaseUrl: true, requireAuth: true }` — fail
3209
- * loud if the campaign would silently fall back to the public router or
3210
- * run unauthenticated. Override with an empty object to disable.
3211
- */
3212
- routeRequirements?: LlmRouteRequirements;
3213
- /**
3214
- * Per-run TraceStore factory. Common shape: a fresh store per run keyed
3215
- * on `runId`. Implementations that share a store across the campaign
3216
- * are valid — the campaign only writes through `emitter`.
3217
- */
3218
- storeFactory: (params: CampaignFactoryParams) => TraceStore;
3219
- /**
3220
- * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`
3221
- * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;
3222
- * otherwise required. Forensic capture is non-negotiable in a campaign
3223
- * run — pass `NoopRawProviderSink` explicitly if you want to opt out.
3224
- */
3225
- rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink;
3226
- /**
3227
- * Filesystem root for default `rawSinkFactory`. Ignored if
3228
- * `rawSinkFactory` is supplied.
3229
- */
3230
- workDir?: string;
3231
- /**
3232
- * Extra `onRunComplete` hooks the campaign appends (after its own
3233
- * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.
3234
- */
3235
- onRunComplete?: RunCompleteHook[];
3236
- /**
3237
- * Per-run integrity expectations. Defaults to:
3238
- * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.
3239
- * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.
3240
- */
3241
- integrity?: RunIntegrityExpectations;
3242
- /** Behaviour when integrity fails. Default `'mark_failed'`. */
3243
- onIntegrityFailure?: CampaignIntegrityPolicy;
3244
- /**
3245
- * Per-run runner. Receives a fully-wired context; produces an outcome
3246
- * the campaign converts into a `RunRecord`.
3247
- */
3248
- runner: CampaignRunner<V>;
3249
- /**
3250
- * If set, the campaign computes `researchReport` at the end. `comparator`
3251
- * is a `variantId`. Other fields are forwarded verbatim.
3252
- */
3253
- report?: {
3254
- comparator?: string;
3255
- } & Omit<ResearchReportOptions, 'comparator' | 'preregistrationHash' | 'generatedAt'>;
3256
- /**
3257
- * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).
3258
- * Embedded in the campaign fingerprint and the research report.
3259
- */
3260
- preregistrationHash?: string;
3261
- /** Local concurrency. Default `1` (sequential). */
3262
- concurrency?: number;
3263
- /**
3264
- * Override the time source. Tests pass a mock to make wallMs deterministic.
3265
- */
3266
- now?: () => number;
3267
- /** Override the runId generator. Tests pin this. */
3268
- runId?: (params: CampaignFactoryParams) => string;
3269
- /**
3270
- * Agent profile cell for campaign runs. Static profiles can pass an object;
3271
- * routers or variant-specific harnesses can pass a factory. The campaign
3272
- * stamps the built cell onto every `RunRecord` and rejects profile/model or
3273
- * profile/prompt contradictions.
3274
- */
3275
- agentProfile?: AgentProfileCell | AgentProfileCellInput | ((params: CampaignFactoryParams & {
3276
- variant: V;
3277
- scenarioTags: Record<string, string>;
3278
- }) => AgentProfileCell | AgentProfileCellInput | Promise<AgentProfileCell | AgentProfileCellInput>);
3279
- }
3280
- interface CampaignFactoryParams {
3281
- campaignId: string;
3282
- runId: string;
3283
- variantId: string;
3284
- scenarioId: string;
3285
- seed: number;
3286
- }
3287
- interface FailedRun {
3288
- runId: string;
3289
- variantId: string;
3290
- scenarioId: string;
3291
- seed: number;
3292
- reason: string;
3293
- error?: string;
3294
- }
3295
- interface EvalCampaignResult {
3296
- campaignId: string;
3297
- /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, baseUrl, provider, preregistrationHash)`. */
3298
- campaignFingerprint: string;
3299
- preregistrationHash: string | null;
3300
- /** Successful runs only. Failed runs land in `failedRuns`. */
3301
- runs: RunRecord[];
3302
- /** Integrity reports for every successful run. */
3303
- integrityReports: RunIntegrityReport[];
3304
- failedRuns: FailedRun[];
3305
- /** Computed when `report` is set on options. */
3306
- report?: ResearchReport;
3307
- startedAt: string;
3308
- endedAt: string;
3309
- }
3310
- declare function runEvalCampaign<V>(opts: EvalCampaignOptions<V>): Promise<EvalCampaignResult>;
3311
-
3312
- /**
3313
- * Always-valid sequential evaluation.
3314
- *
3315
- * `researchReport` assumes a single pre-specified analysis. Real
3316
- * consumers run campaigns weekly / nightly / per-PR; each new run silently
3317
- * inflates the false-discovery rate, because the BH-FDR guarantee is for
3318
- * the *first* look, not the 47th. Without time-uniform inference,
3319
- * launch-decision teams either (a) don't peek, which forfeits the cost
3320
- * advantage of stop-when-decisive, or (b) peek and pretend they didn't,
3321
- * which forfeits scientific validity.
3322
- *
3323
- * This module ships **e-value-based confidence sequences** for paired
3324
- * bounded outcomes. The methodology is the predictable plug-in betting
3325
- * martingale of Waudby-Smith & Ramdas (2024) — provably valid at *any*
3326
- * stopping time. Concretely:
3327
- *
3328
- * For paired deltas D_1, D_2, … ∈ [-c, c] with the null H_0: E[D] ≤ 0,
3329
- * a betting fraction λ_i is chosen using only D_{1..i-1} (predictable
3330
- * plug-in), and the running e-value is
3331
- *
3332
- * E_t = ∏_{i=1}^{t} (1 + λ_i · D_i)
3333
- *
3334
- * E_t is a non-negative martingale under H_0 with E[E_t] ≤ 1, so by
3335
- * Ville's inequality, P(∃ t : E_t ≥ 1/α) ≤ α — we can reject the null
3336
- * at any time without inflating the type-I error.
3337
- *
3338
- * Combined with `runEvalCampaign`, every consumer running rolling
3339
- * campaigns gains the ability to ship the moment evidence is decisive,
3340
- * stop-early on dead-on-arrival variants, and accumulate evidence across
3341
- * partial runs without spending the FDR budget. No new sweep is wasted.
3342
- *
3343
- * References:
3344
- * - Howard, S. R., Ramdas, A., McAuliffe, J., Sekhon, J. (2021).
3345
- * Time-uniform, nonparametric, nonasymptotic confidence sequences.
3346
- * Annals of Statistics, 49(2), 1055–1080.
3347
- * - Waudby-Smith, I., Ramdas, A. (2024). Estimating means of bounded
3348
- * random variables by betting. JRSS B, 86(1), 1–27.
3349
- */
3350
- type SequentialDecision = 'promote_now' | 'continue' | 'reject_now' | 'equivalent';
3351
- interface InterimReleaseConfidence {
3352
- candidates: Array<{
3353
- candidateId: string;
3354
- decision: SequentialDecision;
3355
- decisionFiredAt: number | null;
3356
- finalEvalue: number;
3357
- finalPValue: number;
3358
- pairs: number;
3359
- csLow: number;
3360
- csHigh: number;
3361
- }>;
3362
- /**
3363
- * Campaign-level recommendation: pick the strongest 'promote_now', else
3364
- * 'continue' if any candidate is still live, else 'reject_now' if every
3365
- * candidate is dead, else 'equivalent'.
3366
- */
3367
- recommendation: {
3368
- decision: SequentialDecision;
3369
- candidateId: string | null;
3370
- };
3371
- }
3372
-
3373
- /**
3374
- * `runRLCampaign` — top-level orchestrator that runs the matrix and
3375
- * produces every RL-ready artifact in one call.
3376
- *
3377
- * Wires:
3378
- * 1. `runEvalCampaign` for the matrix run (capture, integrity, hooks)
3379
- * 2. `extractVerifiableReward` over each run, separating deterministic
3380
- * from probabilistic reward sources for the trainer
3381
- * 3. `extractPreferences` to produce DPO/PPO/KTO triples
3382
- * 4. `evaluateInterimReleaseConfidence` over paired deltas (anytime-valid)
3383
- * 5. `rubricPredictiveValidity` against an outcome store, when provided
3384
- * 6. `detectRewardHacking` as a standing hygiene check
3385
- * 7. Trainer-format export rows ready for prime-rl / TRL / verl
3386
- *
3387
- * The output `RLCampaignResult` is a single, audit-ready artifact: every
3388
- * stage's output is in there. The consumer's downstream fits in a single
3389
- * line: pass `result.preferences` to their DPO trainer, `result.grpoRows`
3390
- * to GRPO, `result.runs` plus `result.rewardSignals` to a custom RL loop.
3391
- */
3392
-
3393
- interface RunRLCampaignOptions<V> extends EvalCampaignOptions<V> {
3394
- /** Preference-extraction options. Default uses paired-by-scenario-and-seed with min-margin 0.05. */
3395
- preferences?: ExtractPreferencesOptions;
3396
- /** Verifiable-reward extraction options. */
3397
- verifiableReward?: VerifiableRewardExtractionOptions;
3398
- /** Outcome store + metric names — when supplied, runs `rubricPredictiveValidity` post-campaign. */
3399
- outcomeStore?: OutcomeStore;
3400
- outcomeMetrics?: string[];
3401
- /** Anytime-valid sequential evaluation options. */
3402
- sequential?: {
3403
- alpha?: number;
3404
- bound?: number;
3405
- rope?: {
3406
- low: number;
3407
- high: number;
3408
- };
3409
- };
3410
- /** Trainer-format export lookups. When provided, the orchestrator builds the corresponding rows. */
3411
- trainerExport?: {
3412
- dpo?: DpoLookups;
3413
- grpo?: GrpoLookups;
3414
- sft?: SftLookups;
3415
- };
3416
- }
3417
- interface RLCampaignResult<V> {
3418
- campaign: EvalCampaignResult;
3419
- /** Per-run verifiable reward (deterministic when available, probabilistic fallback otherwise). */
3420
- rewardSignals: Array<{
3421
- runId: string;
3422
- reward: VerifiableReward | null;
3423
- }>;
3424
- /** Preference extraction report. */
3425
- preferences: PreferenceExtractionReport;
3426
- /** Anytime-valid interim verdict over the paired deltas (vs comparator). */
3427
- interimConfidence: InterimReleaseConfidence | null;
3428
- /** Standing reward-hacking hygiene check. */
3429
- rewardHacking: RewardHackingReport;
3430
- /** Predictive validity, when an outcome store was supplied. */
3431
- predictiveValidity: RubricPredictiveValidityReport | null;
3432
- /** Trainer-export rows, populated only for the formats the caller requested via `trainerExport`. */
3433
- trainerRows: {
3434
- dpo?: DpoExportRow[];
3435
- grpo?: GrpoExportRow[];
3436
- sft?: SftExportRow[];
3437
- };
3438
- /**
3439
- * One-line top-level summary the consumer can log.
3440
- */
3441
- summary: string;
3442
- /**
3443
- * Convenience type-tag — consumers can branch on `result.kind`.
3444
- */
3445
- kind: 'agent-eval-rl-campaign';
3446
- unusedVariant?: V;
3447
- }
3448
- declare function runRLCampaign<V>(opts: RunRLCampaignOptions<V>): Promise<RLCampaignResult<V>>;
3449
-
3450
- /**
3451
- * Analyst contract — the missing orchestration layer over agent-eval's
3452
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
3453
- * SemanticConceptJudge, JudgeFn, ...).
3454
- *
3455
- * Each existing primitive returns its own output shape. The Analyst
3456
- * contract is the single envelope every primitive lifts into, so a
3457
- * registry can run N analysts against a run and a single renderer can
3458
- * compose findings without knowing which analyzer produced them.
3459
- *
3460
- * The contract is intentionally domain-agnostic: nothing here knows
3461
- * about code, voice, RAG, or any particular agent stack. Analysts
3462
- * declare what INPUT KIND they need (a trace store, an artifact dir,
3463
- * a RunRecord, a JudgeInput, or `custom`), and the registry routes
3464
- * the matching input from `AnalystRunInputs`.
3465
- */
3466
-
3467
- interface EvidenceRef {
3468
- /**
3469
- * Where the evidence lives. `span` and `event` refer to OTLP trace
3470
- * elements; `artifact` to a file inside the run's artifact tree;
3471
- * `finding` to another AnalystFinding (cross-analyst chaining);
3472
- * `metric` to a named scalar reading the renderer knows how to read.
3473
- */
3474
- kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
3475
- uri: string;
3476
- excerpt?: string;
3477
- }
3478
-
3479
- type PolicyEditSchemaVersion = 'policy-edit/v1';
3480
- declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
3481
- type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
3482
- declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
3483
- type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
3484
- type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
3485
- type PolicyEditGainDirection = 'increase' | 'decrease';
3486
- type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
3487
- interface PolicyEditTarget {
3488
- surface: PolicyEditTargetSurface;
3489
- /** Stable path inside the target surface, for example `system-prompt:tools`
3490
- * or `budget.maxTurns`. */
3491
- path?: string;
3492
- /** Optional canonical deployment identity. Store the existing cell, not a
3493
- * local profile shape. */
3494
- agentProfileCell?: AgentProfileCell;
3495
- /** Human label when the path is not enough for a readable audit trail. */
3496
- label?: string;
3497
- }
3498
- type PolicyEditChange = {
3499
- kind: 'text';
3500
- mode: 'append' | 'prepend' | 'replace';
3501
- value: string;
3502
- /** Required when `mode === 'replace'`; exact match only. */
3503
- find?: string;
3504
- } | {
3505
- kind: 'json';
3506
- mode: 'set' | 'merge' | 'remove';
3507
- path: string;
3508
- value?: AgentProfileJson;
3509
- };
3510
- interface PolicyEditExpectedGain {
3511
- /** Metric this edit is expected to move, e.g. `holdout.composite`. */
3512
- metric: string;
3513
- direction: PolicyEditGainDirection;
3514
- /** Positive magnitude in the metric's native units. */
3515
- amount: number;
3516
- unit?: PolicyEditGainUnit;
3517
- rationale?: string;
3518
- }
3519
- interface PolicyEditSource {
3520
- findingIds: string[];
3521
- analystIds: string[];
3522
- evidenceRefs: EvidenceRef[];
3523
- /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
3524
- derivedFromJudge?: boolean;
3525
- }
3526
- interface PolicyEdit {
3527
- schemaVersion: PolicyEditSchemaVersion;
3528
- editId: string;
3529
- axis: PolicyEditAxis;
3530
- target: PolicyEditTarget;
3531
- change: PolicyEditChange;
3532
- claim: string;
3533
- expectedGain: PolicyEditExpectedGain;
3534
- confidence: number;
3535
- risk: PolicyEditRisk;
3536
- source: PolicyEditSource;
3537
- rationale?: string;
3538
- validationPlan?: string;
3539
- metadata?: Record<string, unknown>;
3540
- }
3541
- declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
3542
- /** JSON-safe attribution carried with a measured candidate and its scores. */
3543
- interface PolicyEditCandidateRecord {
3544
- schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
3545
- policyEdit: PolicyEdit;
3546
- }
3547
-
3548
- /**
3549
- * Pass A substrate types — `runCampaign` is the one primitive every
3550
- * eval flow composes from. Three contracts in this file:
3551
- *
3552
- * - `Scenario` input set
3553
- * - `DispatchFn` how to run one scenario → artifact
3554
- * - `CampaignResult` defined output schema (the contract downstream tools depend on)
3555
- *
3556
- * Three more lifted from earlier substrate work (re-exported):
3557
- *
3558
- * - `JudgeConfig` pluggable dimensional scorer (0.38)
3559
- * - `Mutator` optimization-loop surface mutator
3560
- * - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
3561
- *
3562
- * No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
3563
- * can build dashboards / CI gates / regression diffs against a stable schema.
3564
- */
3565
-
3566
- /** Stable identifier + kind tag for any scenario. Consumers
3567
- * extend with their per-domain payload (persona, task, requirement, ...). */
3568
- interface Scenario {
3569
- id: string;
3570
- kind: string;
3571
- tags?: string[];
3572
- }
3573
- /** Redacted identity of a complete scenario payload retained in campaign results. */
3574
- interface CampaignScenarioIdentity extends Pick<Scenario, 'id' | 'kind'> {
3575
- scenarioDigest: `sha256:${string}`;
3576
- }
3577
- /** The canonical judge verdict shape — one declaration, shared by campaign
3578
- * judges and the multishot judge runner (which re-exports this type).
3579
- *
3580
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
3581
- * multishot runner emits 0-10. Cross-scale comparison must go through
3582
- * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
3583
- * promotion-policy) — never renormalize a producer's values in place, as
3584
- * downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
3585
- * `>= 7` gates) key on the producer's native scale. */
3586
- interface JudgeScore {
3587
- dimensions: Record<string, number>;
3588
- composite: number;
3589
- notes: string;
3590
- /** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
3591
- llmCall?: LlmCallMetadata;
3592
- /** Set when the judge itself failed (call error, unparseable output).
3593
- * `composite`/`dimensions` carry no signal — aggregators MUST exclude
3594
- * failed scores from means instead of folding them into zeros. */
3595
- failed?: true;
3596
- /** Ensemble extras (populated by `ensembleJudge`): max per-dimension
3597
- * spread across surviving judges — the inter-rater signal. */
3598
- maxDisagreement?: number;
3599
- /** Ensemble extras: judge identities whose verdict failed. */
3600
- failedJudges?: string[];
3601
- /** Ensemble extras: each surviving judge's per-dimension scores. */
3602
- perJudge?: Record<string, Record<string, number>>;
3603
- }
3604
- /** Five-valued verdict taxonomy (MOSS-paper alignment). */
3605
- type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
3606
- interface GateResult {
3607
- decision: GateDecision;
3608
- reasons: string[];
3609
- contributingGates: Array<{
3610
- name: string;
3611
- passed: boolean;
3612
- detail: unknown;
3613
- }>;
3614
- delta?: number;
3615
- }
3616
- /** Token usage accumulated for a cell. Aliased to the canonical `RunTokenUsage`
3617
- * (run-record.ts, same package) so a cell maps onto a `RunRecord` for the
3618
- * backend-integrity guard with ONE source of truth — a field added to
3619
- * `RunTokenUsage` is a compile error here, not a silent drift. */
3620
- type CampaignTokenUsage = RunTokenUsage;
3621
- interface CampaignCellResult<TArtifact> {
3622
- /** Manifest that produced this cell. Resumability refuses to reuse a cell
3623
- * whose manifest differs from the current run. */
3624
- manifestHash?: string;
3625
- cellId: string;
3626
- scenarioId: string;
3627
- rep: number;
3628
- generation?: number;
3629
- artifact: TArtifact;
3630
- judgeScores: Record<string, JudgeScore>;
3631
- costUsd: number;
3632
- /** True when at least one priced receipt used the model table instead of a provider bill. */
3633
- costEstimated?: boolean;
3634
- /** Exact durable receipts required to reuse this cached result. */
3635
- costCallIds?: string[];
3636
- /** Agent-call token usage committed by `ctx.cost.runPaidCall`.
3637
- * `{ input: 0, output: 0 }` when no paid agent call was recorded. */
3638
- tokenUsage: CampaignTokenUsage;
3639
- /** Concrete model from the latest committed agent receipt. Consumed by
3640
- * `buildRunRecord` to pin the model when the declared profile uses a
3641
- * runtime-resolved sentinel. */
3642
- resolvedModel?: string;
3643
- durationMs: number;
3644
- seed: number;
3645
- cached: boolean;
3646
- error?: string;
3647
- }
3648
- interface JudgeAggregate {
3649
- mean: number;
3650
- stdev: number;
3651
- ci95: [number, number];
3652
- n: number;
3653
- }
3654
- interface ScenarioAggregate {
3655
- meanComposite: number;
3656
- ci95: [number, number];
3657
- n: number;
3658
- }
3659
- interface GenerationRecord {
3660
- generationIndex: number;
3661
- candidates: GenerationCandidate[];
3662
- promoted: string[];
3663
- }
3664
- /** One scored candidate surface in a generation. `dimensions` + `scenarios`
3665
- * let a reflective proposer ground its next proposal on WHICH
3666
- * dimensions the candidate is weakest on and WHICH scenarios it best/worst
3667
- * handled — the evidence a blind `Mutator` cannot see. */
3668
- interface GenerationCandidate {
3669
- surfaceHash: string;
3670
- composite: number;
3671
- ci95: [number, number];
3672
- /** Exact surface this candidate mutated. */
3673
- parentSurfaceHash?: string;
3674
- /** Measured search-split composite of the exact parent surface. */
3675
- parentComposite?: number;
3676
- /** Candidate composite minus its parent's composite. Present only when the
3677
- * candidate completed the designed denominator. */
3678
- observedDeltaFromParent?: number;
3679
- /** Whether this candidate had a scorable result for every designed campaign
3680
- * cell and was therefore eligible for ranking, promotion, and Pareto
3681
- * selection. Older externally-authored records may omit this field; loop
3682
- * records always populate it. */
3683
- eligibleForPromotion?: boolean;
3684
- /** Exact denominator receipt for selection eligibility. Scores stay
3685
- * descriptive: an incomplete candidate is retained with its observed score
3686
- * and errors instead of receiving an invented penalty. */
3687
- coverage?: {
3688
- expectedCells: number;
3689
- scorableCells: number;
3690
- unscorableCells: Array<{
3691
- cellId: string;
3692
- reason: string;
3693
- }>;
3694
- };
3695
- /** Mean score per judge dimension across all cells (scenarios × reps ×
3696
- * judges that reported the dimension). */
3697
- dimensions: Record<string, number>;
3698
- /** Per-scenario composite (mean over reps + judges), plus the judge's
3699
- * free-form `notes` for that scenario — the "why it scored low" evidence a
3700
- * reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
3701
- * (which checks/lines/dimensions failed and how), NOT case-specific ground
3702
- * truth: leaking expected answers into the prompt is memorization, and the
3703
- * held-out gate would reject it anyway. */
3704
- scenarios: Array<{
3705
- scenarioId: string;
3706
- composite: number;
3707
- notes?: string;
3708
- }>;
3709
- /** Proposer-supplied short label for the change. Present when the proposer
3710
- * returned a `ProposedCandidate`; absent for bare-surface mutators. */
3711
- label?: string;
3712
- /** Proposer-supplied rationale — WHY this candidate was proposed. The
3713
- * "because rationale Z" the audit requires to survive to the result.
3714
- * Present when the proposer returned a `ProposedCandidate`. */
3715
- rationale?: string;
3716
- /** Exact structured cause threaded from the proposer, when available. */
3717
- candidateRecord?: PolicyEditCandidateRecord;
3718
- }
3719
- interface CampaignAggregates {
3720
- byJudge: Record<string, JudgeAggregate>;
3721
- byScenario: Record<string, ScenarioAggregate>;
3722
- /** Canonical campaign accounting, including worker and judge calls. */
3723
- cost: CostLedgerSummary;
3724
- /** Compatibility alias of `cost.totalCostUsd`. */
3725
- totalCostUsd: number;
3726
- cellsExecuted: number;
3727
- cellsSkipped: number;
3728
- cellsCached: number;
3729
- cellsFailed: number;
3730
- }
3731
- interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
3732
- /** sha256(scenarios, judges, dispatch source ref, optimizer config, seed). Stable identity for reruns. */
3733
- manifestHash: string;
3734
- /** Canonical identity of the exact scenario payloads and replicate count. */
3735
- splitDigest: `sha256:${string}`;
3736
- seed: number;
3737
- /** Replicates designed for every scenario in this campaign. */
3738
- reps: number;
3739
- startedAt: string;
3740
- endedAt: string;
3741
- durationMs: number;
3742
- cells: Array<CampaignCellResult<TArtifact>>;
3743
- aggregates: CampaignAggregates;
3744
- optimization?: {
3745
- generations: GenerationRecord[];
3746
- winnerSurfaceHash?: string;
3747
- };
3748
- gate?: GateResult;
3749
- prUrl?: string;
3750
- runDir: string;
3751
- artifactsByPath: Record<string, string>;
3752
- /** Redacted identities that let consumers verify the exact scenario payloads
3753
- * without retaining customer task content in the result. */
3754
- scenarios: Array<CampaignScenarioIdentity & Pick<TScenario, 'id' | 'kind'>>;
3755
- }
3756
-
3757
- /**
3758
- * Adapters: convert measurement outputs into the canonical `RunRecord[]`
3759
- * artifact that `replayCache`, `pairedEvalueSequence`, and
3760
- * `rubricPredictiveValidity` consume. Two sources:
3761
- * - `campaignToRunRecords` — the campaign substrate's per-cell results
3762
- * (the modern path: `runCampaign` / `runImprovementLoop` → records).
3763
- * - `verificationReportToRunRecord` — a `MultiLayerVerifier` report.
3764
- *
3765
- * Adapters are thin and explicit — every mandatory `RunRecord` field comes
3766
- * from a caller-supplied context (`commitSha`, `model`, `promptHash`,
3767
- * `configHash`) plus the cell's runtime data. The validator still rejects
3768
- * bare-alias model strings — the caller snapshot-pins.
3769
- */
3770
-
3771
- interface AdapterContext {
3772
- /** Logical experiment id — typically the campaign or sweep identifier. */
3773
- experimentId: string;
3774
- /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */
3775
- model: string;
3776
- /** Git SHA the harness was run from. */
3777
- commitSha: string;
3778
- /** Hash of the effective prompt sent to the model. */
3779
- promptHash: string;
3780
- /** Hash of the effective config (model, temperature, tools, judges, splits). */
3781
- configHash: string;
3782
- /** Default split tag. Default `'search'`. */
3783
- splitTag?: RunSplitTag;
3784
- /** Default cost in USD when the source doesn't record one. Default `0`. */
3785
- defaultCostUsd?: number;
3786
- }
3787
- /**
3788
- * Convert a `CampaignResult` into canonical `RunRecord[]` — one record per
3789
- * scored cell. The cell's mean judge composite becomes the split score; every
3790
- * judge dimension is carried through to `outcome.raw`. A cell that errored
3791
- * becomes a record with `failureMode: 'cell_error'` (kept, not dropped — an
3792
- * unscored cell is signal). `candidateId` identifies the measured surface
3793
- * (defaults to the campaign manifest hash).
3794
- */
3795
- declare function campaignToRunRecords(campaign: CampaignResult, ctx: AdapterContext & {
3796
- candidateId?: string;
3797
- }): RunRecord[];
3798
- /**
3799
- * Convert a `MultiLayerVerifier` `VerificationReport` into a `RunRecord`.
3800
- * `outcome.searchScore` (or `holdoutScore`) is `report.blendedScore`;
3801
- * `outcome.raw` carries every layer's score + a pass indicator; `failureMode`
3802
- * is the first failing layer's reason.
3803
- */
3804
- declare function verificationReportToRunRecord(report: VerificationReport, ctx: AdapterContext & {
3805
- candidateId: string;
3806
- scenarioId?: string;
3807
- }, opts?: {
3808
- runId?: string;
3809
- }): RunRecord;
3810
-
3811
- /**
3812
- * Simulator fidelity — score a user SIMULATOR's realism against real-user
3813
- * trace distributions.
3814
- *
3815
- * Synthetic-persona evals (`PersonaConfig`-driven canonical evals, fuzz
3816
- * user-simulator objectives) stand in for real users in most of the numbers
3817
- * we publish. The standing threat is the Sim2Real gap: a simulator that is
3818
- * distributionally unlike production creates "easy mode" and silently
3819
- * inflates every score built on it. This module measures that gap from the
3820
- * SAME artifact both sides already produce — `RunRecord`s — so no new
3821
- * capture pipeline is needed:
3822
- *
3823
- * - `simFidelityReport` — per-feature Jensen-Shannon divergence between
3824
- * simulated and production record distributions, collapsed into a
3825
- * fidelity coefficient in [0,1].
3826
- * - `easyModeCheck` — the headline academic failure mode (sim inflates
3827
- * pass-rate over production) as its own named artifact.
3828
- *
3829
- * Every synthetic-persona eval result should publish its fidelity
3830
- * coefficient alongside the score — a number from an unrepresentative
3831
- * simulator is an unlabeled estimate. Wire-in points:
3832
- *
3833
- * - canonical persona evals: pass the campaign's `RunRecord`s as
3834
- * `simulated` and intake-adapter output (`contract/intake`: OTel spans,
3835
- * feedback tables, coding-agent sessions) as `production`
3836
- * - the fuzz user-sim objective: use `1 - report.fidelity` as a realism
3837
- * penalty when searching over generated personas
3838
- * - the durable corpus (`./corpus`): both sides read straight from
3839
- * `readCorpus` — tag sim vs production by `experimentId`
3840
- */
3841
-
3842
- /** Extracts a flat behavioral feature map from one record. `string` values
3843
- * are categorical, `number` values are quantile-bucketed over the union of
3844
- * both sides, `null` means the feature is absent on this record and is
3845
- * counted explicitly as its own category (never silently dropped). */
3846
- type BehaviorFeatures = (record: RunRecord) => Record<string, string | number | null>;
3847
- /** Reserved histogram category for `null` feature values. A capture-rate
3848
- * difference (one side instruments a signal, the other does not) registers
3849
- * as divergence by design: a simulator that produces no tool traces is not
3850
- * representative of production that does. */
3851
- declare const ABSENT_CATEGORY = "(absent)";
3852
- /** Minimum non-null observations PER SIDE for a feature to enter the
3853
- * fidelity mean. Below this the JSD estimate is sampling noise. */
3854
- declare const DEFAULT_MIN_N_PER_FEATURE = 20;
3855
- /** Quantile buckets used to discretize numeric features. Quartiles balance
3856
- * resolution against per-bucket sample size at the default minN. */
3857
- declare const DEFAULT_QUANTILE_BUCKETS = 4;
3858
- /** Fidelity at or above this → 'representative'; below → 'skewed'.
3859
- * 1 − 0.8 = mean JSD 0.2 ≈ distributions that mostly overlap with one
3860
- * clearly shifted mode — the point where per-feature shifts start changing
3861
- * which failure classes an eval can even observe. */
3862
- declare const REPRESENTATIVE_MIN_FIDELITY = 0.8;
3863
- /**
3864
- * Default feature set — ONLY fields verified present on both simulated and
3865
- * production records:
3866
- *
3867
- * - `score`, `wall_ms`, `output_tokens` — mandatory per the `RunRecord`
3868
- * validator (non-finite values read as absent rather than poisoning a
3869
- * bucket).
3870
- * - `failure_class` — optional taxonomy field; absent counted explicitly.
3871
- * - `turn_count`, `tool_errors`, `tool_error_recovery` — derived from the
3872
- * `outcome.raw` counters the intake adapters and eval harnesses write
3873
- * (`turns_completed`, `assistant_messages`, `tool_errors`,
3874
- * `turns_aborted`); absent on records whose producer did not capture
3875
- * them, counted explicitly.
3876
- * - `completion_length` — from the optional `CorpusRecord` trajectory
3877
- * text; the message-length proxy when records come from the corpus.
3878
- *
3879
- * `RunRecord` carries event COUNTS, not event ordering, so
3880
- * `tool_error_recovery` is a counts-only derivation: errors occurred and the
3881
- * run still completed cleanly ('recovered') vs aborted or classified as a
3882
- * failure ('unrecovered') — not a literal error→retry sequence check.
3883
- */
3884
- declare const defaultBehaviorFeatures: BehaviorFeatures;
3885
- /**
3886
- * Jensen-Shannon divergence between two categorical histograms (raw counts;
3887
- * normalized internally). Log base 2 → bounded [0,1]: 0 = identical
3888
- * distributions, 1 = disjoint support. Symmetric, defined even where the
3889
- * supports differ — exactly the regime sim-vs-production comparison lives in.
3890
- * Throws on zero-mass or negative/non-finite counts: an empty histogram has
3891
- * no distribution and a silent 0 would read as "perfectly representative".
3892
- */
3893
- declare function jsDivergence(p: Record<string, number>, q: Record<string, number>): number;
3894
- /**
3895
- * Deterministic quantile edges over a value set (the UNION of both sides, so
3896
- * sim and production land in the same buckets). Linear interpolation between
3897
- * order statistics; duplicate edges from heavy ties collapse into fewer,
3898
- * wider buckets. Returns `bucketCount - 1` edges before deduplication.
3899
- */
3900
- declare function quantileEdges(values: number[], bucketCount?: number): number[];
3901
- /** Stable half-open bucket label for a value against quantile edges:
3902
- * `[-inf,e0)`, `[e0,e1)`, …, `[eLast,+inf)`. */
3903
- declare function bucketLabel(value: number, edges: number[]): string;
3904
- interface FeatureShift {
3905
- /** Category label (a string value, a numeric bucket, or `ABSENT_CATEGORY`). */
3906
- value: string;
3907
- /** Probability of this category among ALL simulated records (nulls included
3908
- * via `ABSENT_CATEGORY`, so each side's shifts sum to 1). */
3909
- pSim: number;
3910
- /** Probability among ALL production records. */
3911
- pProd: number;
3912
- }
3913
- interface FeatureDivergence {
3914
- feature: string;
3915
- /** Jensen-Shannon divergence in [0,1] for this feature. */
3916
- divergence: number;
3917
- /** Largest |pSim − pProd| categories, descending — where the sim deviates. */
3918
- topShifts: FeatureShift[];
3919
- /** Non-null observations on the simulated side. */
3920
- nSim: number;
3921
- /** Non-null observations on the production side. */
3922
- nProd: number;
3923
- }
3924
- type FidelityVerdict = 'representative' | 'skewed' | 'insufficient-data';
3925
- interface FidelityReport {
3926
- perDimension: FeatureDivergence[];
3927
- /** 1 − mean divergence over features with sufficient data. NaN when the
3928
- * verdict is 'insufficient-data' — a 0 would read as "maximally skewed"
3929
- * and silently poison downstream aggregation; check `verdict` first. */
3930
- fidelity: number;
3931
- /** Features excluded because either side had fewer than `minNPerFeature`
3932
- * non-null observations. Named, never silently dropped. */
3933
- insufficientData: string[];
3934
- /** 'representative' when fidelity >= REPRESENTATIVE_MIN_FIDELITY (0.8),
3935
- * 'skewed' below, 'insufficient-data' when no feature met minN. */
3936
- verdict: FidelityVerdict;
3937
- }
3938
- interface SimFidelityOptions {
3939
- /** Feature extractor. Defaults to `defaultBehaviorFeatures`. */
3940
- features?: BehaviorFeatures;
3941
- /** Minimum non-null observations per side per feature. Default 20. */
3942
- minNPerFeature?: number;
3943
- }
3944
- /**
3945
- * Compare a simulator's RunRecords against production RunRecords, feature by
3946
- * feature. Numeric features are bucketed by deterministic quantiles of the
3947
- * union; nulls count as an explicit `ABSENT_CATEGORY`. Throws on empty
3948
- * inputs — "no records" is a wiring error, not a distribution.
3949
- */
3950
- declare function simFidelityReport(simulated: RunRecord[], production: RunRecord[], opts?: SimFidelityOptions): FidelityReport;
3951
- interface EasyModeOptions {
3952
- /** A run passes when its score (holdout, else search) >= this. Default 0.5
3953
- * — matches the pass-threshold convention across the rl/ primitives. */
3954
- passThreshold?: number;
3955
- /** Pass-rate gap above which the sim is flagged inflated. Default 0.1 —
3956
- * a 10-point inflation is enough to flip most promotion gates. */
3957
- inflationTolerance?: number;
3958
- }
3959
- interface EasyModeReport {
3960
- simPassRate: number;
3961
- prodPassRate: number;
3962
- /** simPassRate − prodPassRate. Positive = the simulator is easier than reality. */
3963
- gap: number;
3964
- /** True when gap > inflationTolerance: numbers measured against this
3965
- * simulator overstate production performance. */
3966
- inflated: boolean;
3967
- }
3968
- /**
3969
- * The headline simulator failure mode as its own named artifact: a simulator
3970
- * that creates "easy mode" inflates pass-rate relative to production, and
3971
- * every score measured against it overstates reality. Throws on empty inputs
3972
- * and on records carrying neither score — a silently-skipped record would
3973
- * bias the very rate this check exists to keep honest.
3974
- */
3975
- declare function easyModeCheck(simulated: RunRecord[], production: RunRecord[], opts?: EasyModeOptions): EasyModeReport;
3976
-
3977
- /**
3978
- * Bradley-Terry / Elo tournament evaluation.
3979
- *
3980
- * For multi-candidate sweeps, comparing every candidate's score against
3981
- * a fixed comparator wastes information — the comparator becomes a high-
3982
- * variance reference and rank flips between near-tied middle-rank
3983
- * candidates are dominated by noise. Pairwise tournaments fix this:
3984
- * every (i, j) pair contributes a comparison to a Bradley-Terry MLE that
3985
- * estimates each candidate's strength on a unified scale.
3986
- *
3987
- * For online updating (rolling campaigns where new candidates arrive
3988
- * over time), we also ship classical Elo with configurable K-factor.
3989
- *
3990
- * References:
3991
- * - Bradley, R. A., Terry, M. E. (1952). Rank analysis of incomplete
3992
- * block designs. Biometrika, 39(3/4), 324–345.
3993
- * - Hunter, D. R. (2004). MM algorithms for generalized Bradley-Terry
3994
- * models. Annals of Statistics, 32(1), 384–406. (The MLE algorithm
3995
- * used here.)
3996
- * - Elo, A. E. (1978). The Rating of Chess Players, Past and Present.
3997
- *
3998
- * This is a useful primitive because most LLM-eval communities (Chatbot
3999
- * Arena, AlpacaEval, ELO-style ablation) have converged on pairwise
4000
- * tournament eval as the most sample-efficient and most rank-stable
4001
- * method when you have many candidates.
4002
- */
4003
- interface PairwiseOutcome {
4004
- /** Winner candidate id. */
4005
- winner: string;
4006
- /** Loser candidate id. */
4007
- loser: string;
4008
- /**
4009
- * Optional draw flag. When true, both candidates get half-credit
4010
- * (Bradley-Terry handles draws as half-wins for each side).
4011
- */
4012
- draw?: boolean;
4013
- /**
4014
- * Optional weight — useful if some pairwise comparisons are stronger
4015
- * signals than others (e.g. a paired test with a wider score gap is
4016
- * a more confident comparison). Default 1.
4017
- */
4018
- weight?: number;
4019
- }
4020
- interface BradleyTerryRating {
4021
- candidateId: string;
4022
- /** Latent strength θ ≥ 0 from the BT MLE. */
4023
- strength: number;
4024
- /** Log-strength = log(θ) — interpretable on a linear scale. */
4025
- logStrength: number;
4026
- /** Number of pairwise comparisons this candidate appears in. */
4027
- n: number;
4028
- /** Win count (+ 0.5 per draw). */
4029
- wins: number;
4030
- }
4031
- interface BradleyTerryFit {
4032
- ratings: BradleyTerryRating[];
4033
- /** Iterations of the MM algorithm before convergence. */
4034
- iterations: number;
4035
- /** Final maximum |θ_new - θ_old| / θ_old. */
4036
- finalDelta: number;
4037
- converged: boolean;
4038
- }
4039
- /**
4040
- * Bradley-Terry MLE via Hunter's MM algorithm.
4041
- *
4042
- * Iteration: θ_i^new = W_i / Σ_{j ≠ i} N_ij / (θ_i + θ_j)
4043
- * where W_i = wins by i (+ 0.5 per draw), N_ij = total comparisons.
4044
- *
4045
- * Returns log-strengths normalized so the smallest is 0 (any constant
4046
- * offset is unobservable in BT — only differences are identified).
4047
- */
4048
- declare function fitBradleyTerry(outcomes: PairwiseOutcome[], opts?: {
4049
- tolerance?: number;
4050
- maxIterations?: number;
4051
- smoothing?: number;
4052
- }): BradleyTerryFit;
4053
- /**
4054
- * Online Elo updates. Use when comparisons arrive over time and you want
4055
- * a running rating without re-fitting the full BT MLE on every update.
4056
- *
4057
- * Initialize ratings to `defaultRating` (1500 by default). Each call to
4058
- * `applyEloUpdate` mutates the map in place and returns the deltas so
4059
- * the caller can log per-comparison rating changes.
4060
- */
4061
- interface EloOptions {
4062
- /** Default rating for unseen candidates. Default 1500. */
4063
- defaultRating?: number;
4064
- /** K-factor controls the step size. Default 32 (FIDE-ish). */
4065
- kFactor?: number;
4066
- }
4067
- declare function applyEloUpdate(ratings: Map<string, number>, outcome: PairwiseOutcome, opts?: EloOptions): {
4068
- winnerDelta: number;
4069
- loserDelta: number;
4070
- };
4071
- /**
4072
- * Build pairwise outcomes from the campaign artifact: for every scenario
4073
- * shared by two candidates, the higher-scoring run wins. Useful when you
4074
- * want a tournament view of an existing campaign without an additional
4075
- * pairwise judge call.
4076
- */
4077
- interface BuildPairwiseFromCampaignInput {
4078
- runs: Array<{
4079
- candidateId: string;
4080
- /** Stable identifier for the matching unit (typically scenarioId). */
4081
- matchKey: string;
4082
- score: number;
4083
- }>;
4084
- /**
4085
- * Tied-score margin. Below this, the comparison is a draw. Default 0
4086
- * (no ties).
4087
- */
4088
- drawMargin?: number;
4089
- }
4090
- declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
4091
-
4092
- export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type AdversarialMutation, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DatasetFormat, type DeploymentOutcome, type DetectRewardHackingInput, type DpoExportRow, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type GrpoExportRow, type GrpoLookups, type HarvestOptions, InMemoryOutcomeStore, type OffPolicyEstimate, type OffPolicyOptions, type OffPolicyTrajectory, type OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, appendToCorpus, applyEloUpdate, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, doublyRobust, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, inverseProbabilityWeighting, jsDivergence, observationsFromRunRecords, offPolicyEstimateAll, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runEvalCampaign, runRLCampaign, runwiseStepRewardSummary, selfConsistency, selfNormalizedImportanceWeighting, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, varianceBasedCurriculum, verificationReportToRunRecord };