@tangle-network/agent-eval 0.109.1 → 0.110.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/CHANGELOG.md +9 -0
  2. package/dist/analyst/index.d.ts +10 -12
  3. package/dist/analyst/index.js +8 -11
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
  6. package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/benchmarks/index.d.ts +4 -4
  9. package/dist/benchmarks/index.js +7 -8
  10. package/dist/builder-eval/index.d.ts +4 -4
  11. package/dist/builder-eval/index.js +1 -2
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +62 -20
  15. package/dist/campaign/index.js +9 -8
  16. package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
  17. package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
  18. package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
  19. package/dist/chunk-7NX6ZSBG.js.map +1 -0
  20. package/dist/{chunk-R6D7NEYJ.js → chunk-GBI5J5DB.js} +81 -11
  21. package/dist/chunk-GBI5J5DB.js.map +1 -0
  22. package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
  23. package/dist/chunk-IMWDSFUM.js.map +1 -0
  24. package/dist/{chunk-OVPVM4JC.js → chunk-J4AKLZEV.js} +15 -4
  25. package/dist/{chunk-OVPVM4JC.js.map → chunk-J4AKLZEV.js.map} +1 -1
  26. package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
  27. package/dist/chunk-MHNQWM4I.js.map +1 -0
  28. package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
  29. package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
  30. package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
  31. package/dist/chunk-PLOMR3HP.js.map +1 -0
  32. package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
  33. package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
  34. package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
  35. package/dist/chunk-XRGOKCMO.js.map +1 -0
  36. package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
  37. package/dist/contract/index.d.ts +26 -28
  38. package/dist/contract/index.js +11 -13
  39. package/dist/contract/index.js.map +1 -1
  40. package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
  41. package/dist/control.d.ts +8 -9
  42. package/dist/control.js +6 -8
  43. package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
  44. package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
  45. package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
  46. package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
  47. package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
  48. package/dist/{gepa-B3x5Ulcv.d.ts → gepa-T8T215nw.d.ts} +149 -6
  49. package/dist/hosted/index.d.ts +7 -7
  50. package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
  51. package/dist/index.d.ts +645 -61
  52. package/dist/index.js +1282 -190
  53. package/dist/index.js.map +1 -1
  54. package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
  55. package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
  56. package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
  57. package/dist/meta-eval/index.d.ts +5 -5
  58. package/dist/meta-eval/index.js +1 -2
  59. package/dist/meta-eval/index.js.map +1 -1
  60. package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
  61. package/dist/multishot/index.d.ts +3 -3
  62. package/dist/openapi.json +1 -1
  63. package/dist/pipelines/index.d.ts +6 -7
  64. package/dist/pipelines/index.js +3 -6
  65. package/dist/pipelines/index.js.map +1 -1
  66. package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
  67. package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
  68. package/dist/{provenance-DdDhf6cg.d.ts → provenance-CyxkvEi9.d.ts} +3 -5
  69. package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
  70. package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
  71. package/dist/reporting.d.ts +8 -8
  72. package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
  73. package/dist/rl.d.ts +568 -15
  74. package/dist/rl.js +4 -4
  75. package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
  76. package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
  77. package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
  78. package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
  80. package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
  81. package/dist/storyboard/index.d.ts +1 -1
  82. package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
  83. package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
  84. package/dist/traces.d.ts +54 -11
  85. package/dist/traces.js +25 -27
  86. package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
  87. package/dist/wire/index.d.ts +5 -6
  88. package/docs/improvement-glossary.md +14 -13
  89. package/package.json +1 -71
  90. package/dist/adapters/http.d.ts +0 -142
  91. package/dist/adapters/http.js +0 -203
  92. package/dist/adapters/http.js.map +0 -1
  93. package/dist/adapters/langchain.d.ts +0 -95
  94. package/dist/adapters/langchain.js +0 -34
  95. package/dist/adapters/langchain.js.map +0 -1
  96. package/dist/adapters/otel.d.ts +0 -112
  97. package/dist/adapters/otel.js +0 -110
  98. package/dist/adapters/otel.js.map +0 -1
  99. package/dist/chunk-2OGPXHOB.js.map +0 -1
  100. package/dist/chunk-45EEMHTC.js +0 -35
  101. package/dist/chunk-45EEMHTC.js.map +0 -1
  102. package/dist/chunk-5BKGXME7.js +0 -65
  103. package/dist/chunk-5BKGXME7.js.map +0 -1
  104. package/dist/chunk-5PK3626Q.js.map +0 -1
  105. package/dist/chunk-6SK5VFYK.js +0 -100
  106. package/dist/chunk-6SK5VFYK.js.map +0 -1
  107. package/dist/chunk-DBDRR6GF.js.map +0 -1
  108. package/dist/chunk-DJWX3GVS.js +0 -81
  109. package/dist/chunk-DJWX3GVS.js.map +0 -1
  110. package/dist/chunk-FOUG2VVS.js +0 -855
  111. package/dist/chunk-FOUG2VVS.js.map +0 -1
  112. package/dist/chunk-JZXGWLK5.js.map +0 -1
  113. package/dist/chunk-K7QEIHHJ.js +0 -613
  114. package/dist/chunk-K7QEIHHJ.js.map +0 -1
  115. package/dist/chunk-KKHDIONI.js +0 -414
  116. package/dist/chunk-KKHDIONI.js.map +0 -1
  117. package/dist/chunk-KMPRBJK4.js +0 -74
  118. package/dist/chunk-KMPRBJK4.js.map +0 -1
  119. package/dist/chunk-Q2JRAWRI.js +0 -196
  120. package/dist/chunk-Q2JRAWRI.js.map +0 -1
  121. package/dist/chunk-R6D7NEYJ.js.map +0 -1
  122. package/dist/chunk-RZTMDUO7.js +0 -49
  123. package/dist/chunk-RZTMDUO7.js.map +0 -1
  124. package/dist/chunk-STGVSCDH.js +0 -202
  125. package/dist/chunk-STGVSCDH.js.map +0 -1
  126. package/dist/chunk-YEHAEDUD.js.map +0 -1
  127. package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
  128. package/dist/corpus-eBVwhCp1.d.ts +0 -560
  129. package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
  130. package/dist/diagnose.d.ts +0 -252
  131. package/dist/diagnose.js +0 -382
  132. package/dist/diagnose.js.map +0 -1
  133. package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
  134. package/dist/governance/index.d.ts +0 -135
  135. package/dist/governance/index.js +0 -18
  136. package/dist/governance/index.js.map +0 -1
  137. package/dist/groundedness/index.d.ts +0 -112
  138. package/dist/groundedness/index.js +0 -77
  139. package/dist/groundedness/index.js.map +0 -1
  140. package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
  141. package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
  142. package/dist/knowledge/index.d.ts +0 -103
  143. package/dist/knowledge/index.js +0 -18
  144. package/dist/knowledge/index.js.map +0 -1
  145. package/dist/pareto-E-pembql.d.ts +0 -81
  146. package/dist/perf/index.d.ts +0 -123
  147. package/dist/perf/index.js +0 -18
  148. package/dist/perf/index.js.map +0 -1
  149. package/dist/prm/index.d.ts +0 -104
  150. package/dist/prm/index.js +0 -265
  151. package/dist/prm/index.js.map +0 -1
  152. package/dist/product-benchmark/index.d.ts +0 -247
  153. package/dist/product-benchmark/index.js +0 -37
  154. package/dist/product-benchmark/index.js.map +0 -1
  155. package/dist/red-team-KmmiqBlY.d.ts +0 -63
  156. package/dist/redact-B40YG2M_.d.ts +0 -45
  157. package/dist/rubric-Cc6UHvUb.d.ts +0 -73
  158. package/dist/run-critic-CmMf05uV.d.ts +0 -56
  159. package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
  160. package/dist/telemetry/file.d.ts +0 -19
  161. package/dist/telemetry/file.js +0 -45
  162. package/dist/telemetry/file.js.map +0 -1
  163. package/dist/telemetry/index.d.ts +0 -38
  164. package/dist/telemetry/index.js +0 -130
  165. package/dist/telemetry/index.js.map +0 -1
  166. package/dist/testing-C21CHsq2.d.ts +0 -20
  167. package/dist/testing.d.ts +0 -1
  168. package/dist/testing.js +0 -8
  169. package/dist/testing.js.map +0 -1
  170. package/dist/trajectory-2TkpSEVh.d.ts +0 -33
  171. package/dist/workflow/index.d.ts +0 -496
  172. package/dist/workflow/index.js +0 -2178
  173. package/dist/workflow/index.js.map +0 -1
  174. /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
  175. /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
  176. /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
  177. /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/rl.d.ts CHANGED
@@ -1,26 +1,24 @@
1
- import { R as RunRecord, b as RunSplitTag } from './run-record-I-Z3JNvO.js';
1
+ import { R as RunRecord, b as RunSplitTag } from './run-record-DksGsfgv.js';
2
2
  export { A as AdversarialMutation } from './adversarial-B7loGVVX.js';
3
- import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-eBVwhCp1.js';
4
- export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-eBVwhCp1.js';
3
+ import { S as Span } from './schema-SGWcK9wa.js';
4
+ import { T as TraceStore } from './store-BsVi7ncX.js';
5
5
  export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
6
6
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
7
7
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
8
- import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-DPnyG-CE.js';
9
- import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-Wc7dx6GM.js';
10
- export { r as runEvalCampaign } from './researcher-Wc7dx6GM.js';
11
- import { a as VerificationReport } from './multi-layer-verifier-CI4jdX-q.js';
8
+ import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-C-fMteAW.js';
9
+ import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-CaH0CwFC.js';
10
+ export { r as runEvalCampaign } from './researcher-CaH0CwFC.js';
11
+ import { a as VerificationReport } from './multi-layer-verifier-BsqKuLyN.js';
12
12
  import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
13
- import { C as CampaignResult } from './types-BdIv5dvA.js';
13
+ import { C as CampaignResult } from './types-v--ctu-b.js';
14
14
  import '@tangle-network/agent-interface';
15
15
  import './errors-oeQrLqXC.js';
16
- import './schema-m0gsnbt3.js';
17
- import './store-BcFXE6LG.js';
18
16
  import './llm-client-DyqEH4jH.js';
19
17
  import './raw-provider-sink-C46HDghv.js';
20
- import './summary-report-QMZVe3P-.js';
21
- import './failure-cluster-DH9Flgcf.js';
22
- import './emitter-C2rqGH_l.js';
23
- import './integrity-DqGZg3st.js';
18
+ import './summary-report-Bz-0-t8v.js';
19
+ import './failure-cluster-C48PiReX.js';
20
+ import './emitter-BRchAAAx.js';
21
+ import './integrity-qemeBAyx.js';
24
22
  import './verdict-C9MlYujm.js';
25
23
 
26
24
  /**
@@ -486,6 +484,561 @@ declare function injectIrrelevantClause<S extends {
486
484
  prompt: string;
487
485
  }>(clause: string, position?: 'prefix' | 'suffix'): ScenarioPerturbation<S>;
488
486
 
487
+ /**
488
+ * Preference dataset extraction — bridge from `RunRecord[]` to RL training.
489
+ *
490
+ * Production RLHF / DPO / KTO / SimPO pipelines need preference triples:
491
+ * `(prompt, chosen, rejected)`. The campaign artifact already contains the
492
+ * ingredients — every (variantId, scenarioId, seed) cell is a candidate
493
+ * that ran the same prompt against the same scenario, scored by the same
494
+ * judge — but turning that into a clean preference dataset requires
495
+ * deciding *what counts as a preference*.
496
+ *
497
+ * This module ships three preference-extraction strategies with explicit
498
+ * tradeoffs, plus a unified output type compatible with HuggingFace TRL,
499
+ * Anthropic finetuning JSONL, and OpenAI fine-tuning APIs. The strategies
500
+ * are deliberately not auto-magical — picking the wrong one corrupts the
501
+ * gradient.
502
+ *
503
+ * Strategies:
504
+ *
505
+ * 1. **`paired-by-scenario-and-seed`** — exact-match comparisons. For
506
+ * each scenario × seed pair, compare every (variantA, variantB) on
507
+ * that exact (scenario, seed). Matches scenarios so the comparison
508
+ * isolates variant effects. Highest signal-to-noise; smallest
509
+ * dataset (only matched pairs count).
510
+ *
511
+ * 2. **`paired-by-scenario`** — looser matching. For each scenario,
512
+ * compare every (variantA, variantB) where both have ≥ 1 run on the
513
+ * same scenario. Aggregates across seeds to compute mean scores per
514
+ * (variant, scenario), then forms preferences from the means. More
515
+ * data, lower per-pair signal.
516
+ *
517
+ * 3. **`top-vs-bottom`** — coarsest. Within each scenario, the highest-
518
+ * scoring run is `chosen`, the lowest is `rejected`. Smallest dataset
519
+ * per scenario but biggest score gap per pair. Useful for early
520
+ * bootstrapping when you have few variants.
521
+ *
522
+ * The output `PreferenceTriple` is *agent-eval-canonical* but trivially
523
+ * mappable to TRL's `DPODataset` shape (`prompt`, `chosen`, `rejected`)
524
+ * via the `toTRLFormat` helper.
525
+ */
526
+
527
+ type PreferenceStrategy = 'paired-by-scenario-and-seed' | 'paired-by-scenario' | 'top-vs-bottom';
528
+ interface PreferenceTriple {
529
+ /** The scenario (input) the variants were run against. */
530
+ scenarioId: string;
531
+ /** RunRecord ids on each side, for traceability. */
532
+ chosenRunId: string;
533
+ rejectedRunId: string;
534
+ /** Variant ids — load-bearing for the RL update. */
535
+ chosenVariantId: string;
536
+ rejectedVariantId: string;
537
+ /** The score gap between chosen and rejected. Larger = stronger signal. */
538
+ marginScore: number;
539
+ /**
540
+ * Optional `(chosen_score, rejected_score)` pair for soft-margin DPO
541
+ * variants. Omitted for `top-vs-bottom` runs that don't carry meaningful
542
+ * scalar gaps.
543
+ */
544
+ scores?: {
545
+ chosen: number;
546
+ rejected: number;
547
+ };
548
+ /** Tie-breaker — when multiple seeds match this scenario, the one used. */
549
+ seed?: number;
550
+ /**
551
+ * Free-form metadata propagated from the run records — e.g. original
552
+ * prompt-hash, model, etc. Lets the RL trainer reconstruct the prompt.
553
+ */
554
+ meta: {
555
+ chosenPromptHash: string;
556
+ rejectedPromptHash: string;
557
+ chosenConfigHash: string;
558
+ rejectedConfigHash: string;
559
+ chosenModel: string;
560
+ rejectedModel: string;
561
+ };
562
+ }
563
+ interface ExtractPreferencesOptions {
564
+ strategy?: PreferenceStrategy;
565
+ /**
566
+ * Minimum score gap required to admit a pair. Pairs below this are
567
+ * dropped — they're noise, not signal. Default 0.05 (5% of [0,1]).
568
+ */
569
+ minMargin?: number;
570
+ /**
571
+ * Optional split tag filter — restrict to runs from one split. Default
572
+ * `'holdout'` (the canonical "real" signal).
573
+ */
574
+ splitTag?: RunRecord['splitTag'];
575
+ /**
576
+ * Optional reward extractor that overrides `outcome.holdoutScore` /
577
+ * `outcome.searchScore`. Use to drive preferences off a verifiable
578
+ * reward instead of the headline score.
579
+ */
580
+ rewardOf?: (run: RunRecord) => number | null;
581
+ }
582
+ interface PreferenceExtractionReport {
583
+ pairs: PreferenceTriple[];
584
+ /** Number of (scenario, seed) cells inspected. */
585
+ cellsInspected: number;
586
+ /** Number of pairs filtered by `minMargin`. */
587
+ pairsBelowMargin: number;
588
+ /** Number of cells with only one variant (no comparison possible). */
589
+ cellsSingleton: number;
590
+ /** Strategy used. */
591
+ strategy: PreferenceStrategy;
592
+ }
593
+ /**
594
+ * Convert `RunRecord[]` to preference triples for RL training.
595
+ *
596
+ * Returns a structured report so callers can see how much data was
597
+ * dropped and why (low-margin pairs, singleton cells). For production
598
+ * pipelines, you usually want to:
599
+ *
600
+ * 1. Run a campaign producing 5–10 variants × 50–200 scenarios × 3 seeds
601
+ * 2. Call this with `strategy: 'paired-by-scenario-and-seed'` and a
602
+ * verifiable-reward extractor as `rewardOf`
603
+ * 3. Pass `report.pairs` to `toTRLFormat` and pipe to your DPO trainer
604
+ */
605
+ declare function extractPreferences(runs: RunRecord[], opts?: ExtractPreferencesOptions): PreferenceExtractionReport;
606
+ /**
607
+ * TRL-compatible export. TRL's `DPODataset` is `{ prompt, chosen, rejected }`
608
+ * but the prompt isn't stored on the RunRecord — only its hash. The caller
609
+ * passes a `promptOf(promptHash)` lookup that the TRL trainer can use.
610
+ */
611
+ declare function toTRLFormat(triples: PreferenceTriple[], promptOf: (hash: string) => string): Array<{
612
+ prompt: string;
613
+ chosen: string;
614
+ rejected: string;
615
+ }>;
616
+ /**
617
+ * Anthropic finetuning JSONL export — `{ system, user, assistant_chosen, assistant_rejected }`
618
+ * shape. Same caveat as TRL: prompt + outputs are content the caller has
619
+ * to map back from the run record / raw event log.
620
+ */
621
+ declare function toAnthropicFormat(triples: PreferenceTriple[]): Array<{
622
+ scenarioId: string;
623
+ chosenRunId: string;
624
+ rejectedRunId: string;
625
+ margin: number;
626
+ }>;
627
+
628
+ /**
629
+ * Process reward extraction — step-level credit assignment from trace spans.
630
+ *
631
+ * RL on long-horizon agents needs *step-level* rewards, not run-level
632
+ * ones. The classic credit-assignment problem (Sutton & Barto) requires
633
+ * knowing which sub-decisions in a trajectory contributed to the
634
+ * outcome. Modern systems (DeepSeek-R1, OpenAI o-series, Lightman et al.
635
+ * "Let's Verify Step by Step" 2023) train *process reward models* (PRMs)
636
+ * that score every step, then do RL with the PRM as the reward signal.
637
+ *
638
+ * This module extracts `StepReward[]` from trace spans — one per
639
+ * meaningful step — and ships:
640
+ *
641
+ * 1. `extractStepRewards(store, runId, opts)` — span → step-reward
642
+ * conversion using configurable per-span scorers (LLM judge over the
643
+ * span output, deterministic checkers, or a learned PRM).
644
+ * 2. `runwiseStepRewardSummary(stepRewards)` — aggregate the per-step
645
+ * signal into a credit-assignment-aware run-level score.
646
+ * 3. `prmTrainingPairs(stepRewards, options)` — produce the
647
+ * `(prefix, suffix_chosen, suffix_rejected)` triples that PRM
648
+ * training pipelines consume.
649
+ *
650
+ * What we ship: the *extraction* and *aggregation* infrastructure plus
651
+ * the data shape PRM training expects. We do NOT ship the actual PRM
652
+ * training (gradient descent over a transformer is out of scope for a
653
+ * TS package). The interface is the contract; downstream consumers wire
654
+ * their preferred trainer.
655
+ *
656
+ * Caveat the panel will land: this is descriptive credit assignment
657
+ * (which steps correlate with outcome), not causal credit assignment
658
+ * (which steps caused outcome). For causal claims you need
659
+ * counterfactual rollouts or a learned dynamics model. Future work; the
660
+ * descriptive version is what production PRM training actually uses.
661
+ */
662
+
663
+ interface StepReward {
664
+ /** Trace span this reward attaches to. */
665
+ spanId: string;
666
+ runId: string;
667
+ /** Index in the trajectory (0-based, in started-at order). */
668
+ stepIndex: number;
669
+ /** Span kind (typically 'tool', 'llm', 'judge'). */
670
+ kind: Span['kind'];
671
+ /** Span name — for the consumer's downstream filtering. */
672
+ name: string;
673
+ /** Step-level reward in [0, 1]. */
674
+ reward: number;
675
+ /**
676
+ * Determinism class. Mirrors the verifiable-reward distinction:
677
+ * deterministic = test/compile/schema check; probabilistic = LLM judge.
678
+ */
679
+ determinism: 'deterministic' | 'probabilistic';
680
+ /** Optional rationale / evidence — the trainer typically discards. */
681
+ rationale?: string;
682
+ /** Optional weight — how much this step contributes to credit assignment. */
683
+ weight?: number;
684
+ }
685
+ interface StepScorer {
686
+ /** Span kinds this scorer applies to. */
687
+ appliesTo: Span['kind'][];
688
+ /** Returns null to skip the span; returns a `StepReward` shape (without index/runId/spanId, which are filled in). */
689
+ score(span: Span): Promise<Omit<StepReward, 'spanId' | 'runId' | 'stepIndex'>> | null | undefined;
690
+ }
691
+ interface ExtractStepRewardsOptions {
692
+ /**
693
+ * Ordered list of scorers. Each span runs through scorers in order;
694
+ * the first non-null result wins. If no scorer applies, the span is
695
+ * skipped (not all spans are training-worthy).
696
+ */
697
+ scorers: StepScorer[];
698
+ /** Optional filter — return null to drop the span entirely before scoring. */
699
+ preFilter?: (span: Span) => boolean;
700
+ }
701
+ declare function extractStepRewards(store: TraceStore, runId: string, opts: ExtractStepRewardsOptions): Promise<StepReward[]>;
702
+ interface RunwiseStepSummary {
703
+ runId: string;
704
+ totalSteps: number;
705
+ meanReward: number;
706
+ /** Sum-of-rewards (weighted by `weight ?? 1`). Use as the run-level proxy. */
707
+ sumWeightedReward: number;
708
+ /** Fraction of steps where reward < 0.5 — proxy for "where the policy was wrong." */
709
+ failureFraction: number;
710
+ /** Maximum drop in reward between consecutive steps — diagnoses a step where things went sideways. */
711
+ worstStepDelta: number;
712
+ worstStepIndex: number | null;
713
+ }
714
+ declare function runwiseStepRewardSummary(stepRewards: StepReward[]): RunwiseStepSummary;
715
+ interface PrmTrainingTriple {
716
+ /** Prefix run-id (or composite key) — the trajectory up to step k-1. */
717
+ prefixRunId: string;
718
+ prefixStepIndex: number;
719
+ /** The step that came next on a high-reward trajectory. */
720
+ chosenSpanId: string;
721
+ chosenReward: number;
722
+ /** A step from a divergent low-reward trajectory at the same prefix length. */
723
+ rejectedSpanId: string;
724
+ rejectedReward: number;
725
+ /** The prefix run came from this run; the rejected step came from `rejectedRunId`. */
726
+ rejectedRunId: string;
727
+ marginScore: number;
728
+ }
729
+ /**
730
+ * Build PRM training triples. The shape: pair runs that share an early
731
+ * prefix (same scenario, same first N steps) and diverge later — at the
732
+ * point of divergence, the high-reward run's next step is `chosen`, the
733
+ * low-reward run's next step is `rejected`. This is the canonical PRM
734
+ * training data shape from Lightman et al. and DeepSeek-R1 process
735
+ * supervision.
736
+ *
737
+ * Implementation note: we don't have a way to detect "same prefix" in
738
+ * the general agent setting (token-level prefixes require hashing model
739
+ * outputs). The current heuristic groups by `(scenarioId, prefixSpanName
740
+ * sequence)` — runs are paired when their first K span names match. For
741
+ * production use this should be replaced with a proper trajectory-prefix
742
+ * hash; the heuristic is good enough for early-stage scaffolding.
743
+ */
744
+ declare function prmTrainingPairs(stepRewardsByRun: Map<string, StepReward[]>, opts?: {
745
+ minMargin?: number;
746
+ minPrefixLength?: number;
747
+ }): PrmTrainingTriple[];
748
+
749
+ /**
750
+ * Trainer-format exporters.
751
+ *
752
+ * agent-eval produces canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`,
753
+ * `StepReward[]`, `PrmTrainingTriple[]`). RL training pipelines consume
754
+ * different shapes — Hugging Face TRL, Prime Intellect's prime-rl, OpenAI
755
+ * fine-tuning, Anthropic finetuning, OpenRLHF, verl. Each has its own
756
+ * JSONL conventions. Rather than ship N adapters, this module ships the
757
+ * canonical formats most production pipelines accept and ergonomic helpers
758
+ * for the rest.
759
+ *
760
+ * Shapes:
761
+ * - **DPO / IPO / KTO** — `{prompt, chosen, rejected}` JSONL. Consumed
762
+ * by HuggingFace TRL, prime-rl's offline DPO, OpenRLHF.
763
+ * - **GRPO offline** — `{prompt, completions[], rewards[]}` JSONL.
764
+ * Consumed by prime-rl GRPO, verl, OpenRLHF.
765
+ * - **SFT** — `{messages[]}` JSONL with chosen completion as the final
766
+ * assistant turn. Consumed by HF SFT trainers, OpenAI fine-tuning,
767
+ * Anthropic finetuning.
768
+ * - **PRM** — `{prompt, prefix_steps[], chosen_step, rejected_step}` JSONL.
769
+ * Consumed by Lightman-style PRM trainers and prime-rl's PRM mode.
770
+ *
771
+ * Why ship this in agent-eval rather than a separate adapter package: the
772
+ * canonical artifacts (`RunRecord[]`, `PreferenceTriple[]`, etc.) are
773
+ * agent-eval's contract; without first-party exporters consumers reverse-
774
+ * engineer the mapping every release. The exporters codify it.
775
+ *
776
+ * The exporters take callbacks for any field that isn't on the canonical
777
+ * artifact (specifically: prompt + completion text, since the package
778
+ * stores only their hashes by design — full text is the consumer's
779
+ * trace store / raw event log).
780
+ */
781
+
782
+ interface DpoLookups {
783
+ /** Resolve the prompt text for a run (typically from a trace store / raw event sink). */
784
+ promptOf: (runId: string) => string | Promise<string>;
785
+ /** Resolve the assistant completion text for a run. */
786
+ completionOf: (runId: string) => string | Promise<string>;
787
+ }
788
+ interface DpoExportRow {
789
+ prompt: string;
790
+ chosen: string;
791
+ rejected: string;
792
+ /** Carried-through margin. Some KTO / IPO variants use this. */
793
+ margin?: number;
794
+ /** Free-form metadata for downstream filtering / sharding. */
795
+ meta?: Record<string, unknown>;
796
+ }
797
+ /**
798
+ * Convert preference triples to TRL-compatible DPO rows. The shape
799
+ * `{prompt, chosen, rejected}` is the canonical HuggingFace DPODataset
800
+ * entry; every major DPO trainer accepts it.
801
+ */
802
+ declare function toDpoRows(triples: PreferenceTriple[], lookups: DpoLookups): Promise<DpoExportRow[]>;
803
+ /** Serialize DPO rows as JSONL. One line per row. */
804
+ declare function toDpoJsonl(rows: DpoExportRow[]): string;
805
+ interface GrpoLookups {
806
+ promptOf: (runId: string) => string | Promise<string>;
807
+ completionOf: (runId: string) => string | Promise<string>;
808
+ /** Optional: derive a custom reward from the run. Defaults to score. */
809
+ rewardOf?: (run: RunRecord) => number | null;
810
+ }
811
+ interface GrpoExportRow {
812
+ prompt: string;
813
+ completions: string[];
814
+ rewards: number[];
815
+ /** runIds in the same order as `completions[]` for traceability. */
816
+ runIds: string[];
817
+ meta?: Record<string, unknown>;
818
+ }
819
+ /**
820
+ * Convert RunRecord[] grouped by `(scenarioId)` into GRPO offline rows —
821
+ * one row per scenario, with one completion per run on that scenario.
822
+ *
823
+ * GRPO (Shao et al. 2024 / DeepSeek-R1) trains on relative advantages
824
+ * within a group of completions for the same prompt; this is the
825
+ * canonical input format.
826
+ */
827
+ declare function toGrpoRows(runs: RunRecord[], lookups: GrpoLookups): Promise<GrpoExportRow[]>;
828
+ declare function toGrpoJsonl(rows: GrpoExportRow[]): string;
829
+ interface SftLookups {
830
+ promptOf: (runId: string) => string | Promise<string>;
831
+ completionOf: (runId: string) => string | Promise<string>;
832
+ /** Optional system message. Default omits. */
833
+ systemOf?: (run: RunRecord) => string | null | undefined;
834
+ /** Filter — return false to skip the run (e.g., low score, failed cases). */
835
+ include?: (run: RunRecord) => boolean;
836
+ }
837
+ interface SftExportRow {
838
+ messages: Array<{
839
+ role: 'system' | 'user' | 'assistant';
840
+ content: string;
841
+ }>;
842
+ meta?: Record<string, unknown>;
843
+ }
844
+ /**
845
+ * Convert RunRecord[] into Hugging Face / OpenAI / Anthropic-style
846
+ * conversational SFT rows. By default every record becomes one row;
847
+ * pass `include` to filter (e.g., keep only `score >= 0.8` for
848
+ * rejection-sampling SFT).
849
+ */
850
+ declare function toSftRows(runs: RunRecord[], lookups: SftLookups): Promise<SftExportRow[]>;
851
+ declare function toSftJsonl(rows: SftExportRow[]): string;
852
+ interface PrmLookups {
853
+ /** Resolve the prompt text for a run. */
854
+ promptOf: (runId: string) => string | Promise<string>;
855
+ /** Resolve the trajectory step text for a (runId, spanId) pair. */
856
+ stepTextOf: (runId: string, spanId: string) => string | Promise<string>;
857
+ /** Optional: sequence of prefix span ids leading up to the divergence. */
858
+ prefixOf?: (runId: string, prefixStepIndex: number) => string[] | Promise<string[]>;
859
+ }
860
+ interface PrmExportRow {
861
+ prompt: string;
862
+ /** Span ids for the steps before divergence — caller resolves text via `stepTextOf`. */
863
+ prefixSpanIds: string[];
864
+ prefixStepText: string[];
865
+ chosenStep: string;
866
+ rejectedStep: string;
867
+ chosenReward: number;
868
+ rejectedReward: number;
869
+ marginScore: number;
870
+ meta?: Record<string, unknown>;
871
+ }
872
+ /**
873
+ * Convert PRM training triples to JSONL rows. Caller's `stepTextOf`
874
+ * callback resolves span text from the consumer's trace store.
875
+ */
876
+ declare function toPrmRows(triples: PrmTrainingTriple[], lookups: PrmLookups): Promise<PrmExportRow[]>;
877
+ declare function toPrmJsonl(rows: PrmExportRow[]): string;
878
+ interface StepRewardJsonlRow {
879
+ runId: string;
880
+ spanId: string;
881
+ stepIndex: number;
882
+ reward: number;
883
+ determinism: 'deterministic' | 'probabilistic';
884
+ weight: number;
885
+ }
886
+ declare function stepRewardsToJsonl(stepRewards: StepReward[]): string;
887
+
888
+ /**
889
+ * RL dataset packaging + datasheet — the publishable, sellable bundle.
890
+ *
891
+ * The format exporters (`toGrpoRows` / `toSftRows` / `toDpoRows`) already
892
+ * produce trainer-ready shapes (prime-rl GRPO, TRL DPO, conversational SFT).
893
+ * What turns that into a dataset someone can PUBLISH or BUY is the provenance
894
+ * + a datasheet: which models produced it, which prompt/agent versions, how the
895
+ * reward was derived (deterministic verifiable vs probabilistic judge — the
896
+ * credibility axis a buyer checks first), the split discipline, the reward
897
+ * distribution, the quality gates, the license, and the intended/out-of-scope
898
+ * uses. This module computes those facts from the `RunRecord[]` and renders a
899
+ * "Datasheet for Datasets" (Gebru et al. 2018) card alongside the format files.
900
+ *
901
+ * It composes the existing `rl/exporters` — it does not reimplement any trainer
902
+ * format. The renderers token-identity step (DeepSeek/Kimi/Qwen tokenization
903
+ * with per-token loss masks) is a downstream Python stage that consumes the
904
+ * `messages`/`completions` this bundle emits.
905
+ */
906
+
907
+ type RewardKind = 'deterministic' | 'probabilistic' | 'mixed';
908
+ type DatasetFormat = 'grpo' | 'sft' | 'dpo';
909
+ /** Caller-declared context — the qualitative half of the datasheet that can't
910
+ * be computed from records. */
911
+ interface RlDatasetConfig {
912
+ name: string;
913
+ version: string;
914
+ /** Product/task domain, e.g. 'legal-m&a', 'tax-1040'. */
915
+ domain: string;
916
+ /** SPDX id or a named commercial license. Required — an unlicensed dataset
917
+ * cannot be published or sold. */
918
+ license: string;
919
+ /** How the reward was produced. `kind: 'deterministic'` (a test/schema/XPath
920
+ * decided it) is the credibility signal; 'probabilistic' = LLM-judge. */
921
+ reward: {
922
+ kind: RewardKind;
923
+ source: string;
924
+ description: string;
925
+ };
926
+ intendedUse: string;
927
+ outOfScope: string;
928
+ limitations: string;
929
+ /** ISO timestamp — passed in (the substrate forbids Date.now()). */
930
+ createdAtIso: string;
931
+ /** Default: ['grpo', 'sft']. */
932
+ formats?: DatasetFormat[];
933
+ /** Quality gates already run, recorded on the card for the buyer. */
934
+ qualityGates?: {
935
+ contaminationProbe?: 'passed' | 'failed' | 'not-run';
936
+ dedup?: boolean;
937
+ verifiableRewardFilter?: boolean;
938
+ };
939
+ }
940
+ interface RewardStats {
941
+ n: number;
942
+ mean: number;
943
+ median: number;
944
+ min: number;
945
+ max: number;
946
+ std: number;
947
+ }
948
+ interface RlDatasetStats {
949
+ records: number;
950
+ /** Record count per split — a publishable dataset must declare its holdout. */
951
+ splits: Record<RunSplitTag, number>;
952
+ reward: RewardStats;
953
+ /** Distinct snapshot-pinned models that produced the trajectories. */
954
+ models: string[];
955
+ /** Distinct effective-prompt hashes (the agent profile/prompt versions). */
956
+ promptHashes: string[];
957
+ commitShas: string[];
958
+ totalTokens: {
959
+ input: number;
960
+ output: number;
961
+ };
962
+ totalCostUsd: number;
963
+ }
964
+ interface RlDatasetManifest extends RlDatasetConfig {
965
+ formats: DatasetFormat[];
966
+ rowCounts: Partial<Record<DatasetFormat, number>>;
967
+ stats: RlDatasetStats;
968
+ }
969
+ interface RlDatasetBundle {
970
+ manifest: RlDatasetManifest;
971
+ /** Relative filename -> contents. Write these to a directory to publish. */
972
+ files: Record<string, string>;
973
+ }
974
+ /**
975
+ * Package graded `RunRecord[]` into a publishable RL dataset bundle: the
976
+ * trainer-format JSONL files + a manifest + a datasheet. DPO requires
977
+ * pre-extracted preference triples (pass `preferences`); GRPO/SFT derive from
978
+ * the records directly via the supplied lookups. Throws on an empty corpus —
979
+ * an empty dataset must never be published.
980
+ */
981
+ declare function buildRlDataset(records: RunRecord[], lookups: GrpoLookups & SftLookups, config: RlDatasetConfig, preferences?: {
982
+ triples: PreferenceTriple[];
983
+ lookups: DpoLookups;
984
+ }): Promise<RlDatasetBundle>;
985
+ /** Render the "Datasheet for Datasets" card — the artifact a buyer reads. */
986
+ declare function datasheetToMarkdown(m: RlDatasetManifest): string;
987
+
988
+ /**
989
+ * RL corpus — the durable, append-only accumulation of graded RunRecords that
990
+ * every eval run deposits BY DEFAULT.
991
+ *
992
+ * The dataset is the free exhaust of the normal eval process: we run evals
993
+ * constantly to get an agent production-ready, and those runs already produce
994
+ * graded trajectories. Instead of writing them to an ephemeral run dir and
995
+ * throwing them away, `appendToCorpus` accumulates them into a durable corpus;
996
+ * `buildDatasetFromCorpus` later harvests the whole corpus into a publishable
997
+ * bundle. No separate data-collection campaign — the data accrues from work we
998
+ * do anyway. This is the "best things for free by our process" layer.
999
+ *
1000
+ * Trajectory text rides on the record as top-level `prompt` / `completion`
1001
+ * (what the eval harnesses capture; the RunRecord validator ignores the extra
1002
+ * keys). The harvest reads them directly — no trace store round-trip needed.
1003
+ */
1004
+
1005
+ /** A corpus record is a RunRecord carrying the trajectory text the harness
1006
+ * captured. `prompt`/`completion` are top-level (the validator ignores extras). */
1007
+ type CorpusRecord = RunRecord & {
1008
+ prompt?: string;
1009
+ completion?: string;
1010
+ };
1011
+ interface CorpusAppendResult {
1012
+ appended: number;
1013
+ /** Skipped because a record with the same runId was already in the corpus
1014
+ * (idempotent appends — NOT re-run collapsing; re-runs get fresh runIds). */
1015
+ skipped: number;
1016
+ total: number;
1017
+ }
1018
+ /**
1019
+ * Append graded records to the corpus (append-only JSONL). Deduplicates by
1020
+ * `runId` against what's already on disk so re-running the same harness is
1021
+ * idempotent. Creates the file and parent dir. This is the call every eval
1022
+ * harness makes by default after producing its records.
1023
+ */
1024
+ declare function appendToCorpus(records: CorpusRecord[], corpusPath: string): CorpusAppendResult;
1025
+ /** Read the full corpus. Returns [] if the corpus does not exist yet. */
1026
+ declare function readCorpus(corpusPath: string): CorpusRecord[];
1027
+ interface HarvestOptions {
1028
+ /** Keep only records scoring >= this (rejection-sampling for SFT). */
1029
+ minScore?: number;
1030
+ /** Keep only these splits (e.g. ['holdout'] for an eval-only dataset). */
1031
+ splits?: RunRecord['splitTag'][];
1032
+ }
1033
+ /**
1034
+ * Harvest the accumulated corpus into a publishable RL dataset bundle. Reads
1035
+ * trajectory text from each record's top-level `prompt`/`completion`; records
1036
+ * missing either are excluded (a graded score with no trajectory can't train).
1037
+ * Optionally filters by score / split. Throws (via buildRlDataset) if nothing
1038
+ * survives — an empty dataset must never be published.
1039
+ */
1040
+ declare function buildDatasetFromCorpus(corpusPath: string, config: RlDatasetConfig, opts?: HarvestOptions): Promise<RlDatasetBundle>;
1041
+
489
1042
  /**
490
1043
  * `PredictiveValidityResearcher` — concrete `Researcher` implementation
491
1044
  * that drives selection from outcome-anchored predictive validity.
@@ -1190,4 +1743,4 @@ interface BuildPairwiseFromCampaignInput {
1190
1743
  }
1191
1744
  declare function buildPairwiseFromCampaign(input: BuildPairwiseFromCampaignInput): PairwiseOutcome[];
1192
1745
 
1193
- export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DetectRewardHackingInput, DpoExportRow, DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, ExtractPreferencesOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, GrpoExportRow, GrpoLookups, OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, PreferenceExtractionReport, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, SftExportRow, SftLookups, type SimFidelityOptions, type ThompsonCurriculumOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, applyEloUpdate, bestOfN, bucketLabel, buildPairwiseFromCampaign, campaignToRunRecords, compareAdaptationCurves, defaultBehaviorFeatures, detectRewardHacking, easyModeCheck, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, jsDivergence, observationsFromRunRecords, paretoFrontier, quantileEdges, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runRLCampaign, selfConsistency, shuffleOrder, simFidelityReport, thompsonCurriculum, varianceBasedCurriculum, verificationReportToRunRecord };
1746
+ export { ABSENT_CATEGORY, type AdaptationCurve, type AdaptationPoint, type AdaptationRunner, type AdapterContext, type BehaviorFeatures, type BradleyTerryFit, type BradleyTerryRating, type BuildPairwiseFromCampaignInput, type CellObservation, type CompareCurvesResult, type ComputeBestOfNOptions, type ComputeBestOfNResult, type ComputeCurve, type ComputeCurveBudget, type ComputeCurvePoint, type ContaminationProbeInput, type ContaminationProbeOptions, type ContaminationProbeReport, type CorpusAppendResult, type CorpusRecord, type CurriculumAllocation, DEFAULT_MIN_N_PER_FEATURE, DEFAULT_QUANTILE_BUCKETS, type DatasetFormat, type DetectRewardHackingInput, type DpoExportRow, type DpoLookups, type EasyModeOptions, type EasyModeReport, type EloOptions, type ExtractPreferencesOptions, type ExtractStepRewardsOptions, type FeatureDivergence, type FeatureShift, type FidelityReport, type FidelityVerdict, type GrpoExportRow, type GrpoLookups, type HarvestOptions, OutcomeStore, type PairwiseOutcome, type ParetoPointInput, PredictiveValidityResearcher, type PredictiveValidityResearcherOptions, type PreferenceExtractionReport, type PreferenceStrategy, type PreferenceTriple, type PrmExportRow, type PrmLookups, type PrmTrainingTriple, REPRESENTATIVE_MIN_FIDELITY, type RLCampaignResult, type RewardHackingFinding, type RewardHackingReport, type RewardHackingSignal, type RewardKind, type RewardStats, type RlDatasetBundle, type RlDatasetConfig, type RlDatasetManifest, type RlDatasetStats, type RunAdaptationCurveOptions, type RunComputeCurveOptions, type RunRLCampaignOptions, type RunwiseStepSummary, type ScenarioPerturbation, type ScenarioPerturbationKind, type SelfConsistencyOptions, type SelfConsistencyResult, type SftExportRow, type SftLookups, type SimFidelityOptions, type StepReward, type StepRewardJsonlRow, type StepScorer, type ThompsonCurriculumOptions, type VarianceCurriculumOptions, type VerifiableReward, type VerifiableRewardExtractionOptions, type VerifiableRewardSource, appendToCorpus, applyEloUpdate, bestOfN, bucketLabel, buildDatasetFromCorpus, buildPairwiseFromCampaign, buildRlDataset, campaignToRunRecords, compareAdaptationCurves, datasheetToMarkdown, defaultBehaviorFeatures, detectRewardHacking, easyModeCheck, extractPreferences, extractStepRewards, extractVerifiableReward, extractVerifiableRewardsFromRecords, filterDeterministicallyRewarded, firstPassK, fitBradleyTerry, injectIrrelevantClause, jsDivergence, observationsFromRunRecords, paretoFrontier, prmTrainingPairs, quantileEdges, readCorpus, renameVariables, runAdaptationCurve, runComputeCurve, runContaminationProbe, runRLCampaign, runwiseStepRewardSummary, selfConsistency, shuffleOrder, simFidelityReport, stepRewardsToJsonl, thompsonCurriculum, toAnthropicFormat, toDpoJsonl, toDpoRows, toGrpoJsonl, toGrpoRows, toPrmJsonl, toPrmRows, toSftJsonl, toSftRows, toTRLFormat, varianceBasedCurriculum, verificationReportToRunRecord };
package/dist/rl.js CHANGED
@@ -10,14 +10,13 @@ import {
10
10
  } from "./chunk-3RF76KTD.js";
11
11
  import {
12
12
  runEvalCampaign
13
- } from "./chunk-LIEJUH2I.js";
13
+ } from "./chunk-6PL5MGDL.js";
14
14
  import {
15
15
  detectRewardHacking,
16
16
  extractVerifiableReward,
17
17
  extractVerifiableRewardsFromRecords,
18
18
  filterDeterministicallyRewarded
19
19
  } from "./chunk-N22ZO7FV.js";
20
- import "./chunk-FUCQVFMU.js";
21
20
  import {
22
21
  rubricPredictiveValidity
23
22
  } from "./chunk-X6NIXVOD.js";
@@ -35,11 +34,12 @@ import {
35
34
  varianceBasedCurriculum
36
35
  } from "./chunk-VZSRQ272.js";
37
36
  import "./chunk-TT4KNT67.js";
38
- import "./chunk-PC4UYEBM.js";
39
- import "./chunk-VK6HBGAE.js";
40
37
  import "./chunk-TVVP3ZZQ.js";
38
+ import "./chunk-VK6HBGAE.js";
41
39
  import "./chunk-XJYR7XFV.js";
42
40
  import "./chunk-VSMTAMNK.js";
41
+ import "./chunk-FUCQVFMU.js";
42
+ import "./chunk-PC4UYEBM.js";
43
43
  import {
44
44
  ValidationError
45
45
  } from "./chunk-ONWEPEDO.js";
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-I-Z3JNvO.js';
1
+ import { R as RunRecord } from './run-record-DksGsfgv.js';
2
2
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
3
3
 
4
4
  /**
@@ -1,6 +1,6 @@
1
1
  import { AgentProfile } from '@tangle-network/agent-interface';
2
2
  import { V as ValidationError } from './errors-oeQrLqXC.js';
3
- import { F as FailureClass } from './schema-m0gsnbt3.js';
3
+ import { F as FailureClass } from './schema-SGWcK9wa.js';
4
4
 
5
5
  type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
6
6
  type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
@@ -1,4 +1,4 @@
1
- import { b as RunSplitTag } from './run-record-I-Z3JNvO.js';
1
+ import { b as RunSplitTag } from './run-record-DksGsfgv.js';
2
2
 
3
3
  interface RuntimeTrajectoryHookEvent {
4
4
  id: string;