@tangle-network/agent-eval 0.128.1 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +271 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/rl.js CHANGED
@@ -3,50 +3,66 @@ import {
3
3
  inverseProbabilityWeighting,
4
4
  offPolicyEstimateAll,
5
5
  selfNormalizedImportanceWeighting
6
- } from "./chunk-VGRCHJON.js";
6
+ } from "./chunk-T6RLYGAD.js";
7
7
  import {
8
8
  FileSystemOutcomeStore,
9
9
  InMemoryOutcomeStore
10
10
  } from "./chunk-3RF76KTD.js";
11
11
  import {
12
12
  runEvalCampaign
13
- } from "./chunk-TBL77AUT.js";
13
+ } from "./chunk-YQN4ICPP.js";
14
+ import {
15
+ mintRolloutRows
16
+ } from "./chunk-H23X7XKK.js";
17
+ import {
18
+ isSplitEligible
19
+ } from "./chunk-OWN5NPMC.js";
20
+ import {
21
+ assertRewardGate
22
+ } from "./chunk-PC5DOSM7.js";
23
+ import "./chunk-RZTMDUO7.js";
14
24
  import {
15
25
  detectRewardHacking,
16
26
  extractVerifiableReward,
17
27
  extractVerifiableRewardsFromRecords,
18
28
  filterDeterministicallyRewarded
19
- } from "./chunk-NYLOYM6N.js";
29
+ } from "./chunk-EG66UGL4.js";
20
30
  import {
21
31
  campaignCellToRunRecord
22
- } from "./chunk-2MKQIFS4.js";
23
- import "./chunk-PBE2LOSS.js";
32
+ } from "./chunk-E7QXT7SX.js";
33
+ import "./chunk-SFLLL76A.js";
24
34
  import {
25
35
  rubricPredictiveValidity
26
- } from "./chunk-S5YLIBFX.js";
36
+ } from "./chunk-TQ7LNKZ3.js";
27
37
  import {
28
38
  evaluateInterimReleaseConfidence
29
39
  } from "./chunk-MAZ26DC7.js";
30
- import "./chunk-VLOATJQ2.js";
31
- import "./chunk-DPUHNQLN.js";
40
+ import "./chunk-TJVT4QFF.js";
41
+ import "./chunk-7FO3TNPI.js";
32
42
  import {
33
43
  benjaminiHochberg,
34
44
  wilcoxonSignedRank
35
- } from "./chunk-MHELPNRP.js";
45
+ } from "./chunk-ZHTZ4EYI.js";
36
46
  import {
37
47
  observationsFromRunRecords,
38
48
  thompsonCurriculum,
39
49
  varianceBasedCurriculum
40
- } from "./chunk-VZSRQ272.js";
41
- import "./chunk-WS3NZZQQ.js";
50
+ } from "./chunk-G7MGMCZD.js";
51
+ import "./chunk-VCZ5FQYW.js";
42
52
  import "./chunk-VI2UW6B6.js";
43
- import "./chunk-TT4KNT67.js";
53
+ import {
54
+ InMemoryTraceStore
55
+ } from "./chunk-U4PHLT2N.js";
44
56
  import "./chunk-PC4UYEBM.js";
45
57
  import "./chunk-VQMK5FMP.js";
46
58
  import {
47
59
  runTaskScore
48
- } from "./chunk-2JX3CFMB.js";
60
+ } from "./chunk-56TAVBOK.js";
49
61
  import "./chunk-MA6HLL3S.js";
62
+ import {
63
+ observedSplitScore,
64
+ trainingScore
65
+ } from "./chunk-OIUOT4QD.js";
50
66
  import {
51
67
  ValidationError
52
68
  } from "./chunk-ONWEPEDO.js";
@@ -367,10 +383,119 @@ function escapeRegex(s) {
367
383
  import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
368
384
  import { dirname } from "path";
369
385
 
386
+ // src/rl/rollout-input.ts
387
+ function trainableLineReward(line) {
388
+ assertRewardGate(line, "trainable reward");
389
+ const { reward } = line.outcome;
390
+ if (reward === null || !Number.isFinite(reward)) return null;
391
+ return reward;
392
+ }
393
+ function isLineRealnessGated(line) {
394
+ return line.outcome.realness_gated === true;
395
+ }
396
+ function push(index, key, line) {
397
+ const existing = index.get(key);
398
+ if (existing === void 0) index.set(key, [line]);
399
+ else existing.push(line);
400
+ }
401
+ function invocationIndex(lines) {
402
+ const byRollout = /* @__PURE__ */ new Map();
403
+ const byRun = /* @__PURE__ */ new Map();
404
+ for (const line of lines) {
405
+ push(byRollout, line.rollout_id, line);
406
+ push(byRun, line.run_id, line);
407
+ }
408
+ return { byRollout, byRun };
409
+ }
410
+ function resolveInvocation(index, id) {
411
+ const rollouts = index.byRollout.get(id) ?? [];
412
+ const runs = index.byRun.get(id) ?? [];
413
+ if (rollouts.length > 1) return { kind: "ambiguous", count: rollouts.length };
414
+ const exact = rollouts[0];
415
+ if (exact !== void 0) {
416
+ if (runs.some((line) => line !== exact)) {
417
+ return { kind: "ambiguous", count: 1 + runs.filter((line) => line !== exact).length };
418
+ }
419
+ return { kind: "resolved", line: exact };
420
+ }
421
+ if (runs.length > 1) return { kind: "ambiguous", count: runs.length };
422
+ const only = runs[0];
423
+ return only === void 0 ? { kind: "missing" } : { kind: "resolved", line: only };
424
+ }
425
+ function auditInvocationAdmission(items, idsOf, context, requirement, inspect) {
426
+ if (context === void 0 || context === null) {
427
+ throw new Error(
428
+ `${requirement.exporter}: a ${requirement.contextType} is required \u2014 ${requirement.because} Pass \`{ lines: (await mintRolloutRows(...)).rows }\`.`
429
+ );
430
+ }
431
+ const index = invocationIndex(context.lines);
432
+ const audit = {
433
+ admitted: [],
434
+ gatedDrops: 0,
435
+ ambiguousDrops: 0,
436
+ ambiguous: []
437
+ };
438
+ for (const item of items) {
439
+ const lines = [];
440
+ let ambiguous = false;
441
+ for (const id of idsOf(item)) {
442
+ const resolution = resolveInvocation(index, id);
443
+ if (resolution.kind === "missing") {
444
+ throw new Error(
445
+ `${requirement.exporter}: no rollout line supplied for run ${id} \u2014 its realness gate and capture quality are unknown`
446
+ );
447
+ }
448
+ if (resolution.kind === "ambiguous") {
449
+ ambiguous = true;
450
+ if (!audit.ambiguous.some((entry) => entry.id === id)) {
451
+ audit.ambiguous.push({ id, invocations: resolution.count });
452
+ }
453
+ continue;
454
+ }
455
+ lines.push(resolution.line);
456
+ }
457
+ for (const line of lines) {
458
+ assertRewardGate(line, requirement.exporter);
459
+ inspect?.(line);
460
+ }
461
+ if (ambiguous) {
462
+ audit.ambiguousDrops++;
463
+ continue;
464
+ }
465
+ if (lines.some(isLineRealnessGated)) {
466
+ audit.gatedDrops++;
467
+ continue;
468
+ }
469
+ audit.admitted.push(item);
470
+ }
471
+ return audit;
472
+ }
473
+ function admitUngatedByInvocation(items, idsOf, context, requirement, inspect) {
474
+ const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect);
475
+ if (audit.ambiguousDrops > 0) {
476
+ const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(", ");
477
+ console.warn(
478
+ `[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more than one invocation in the supplied lines, so the realness gate cannot be read for the invocation the artifact meant. Reference the \`rollout_id\` instead of the \`run_id\`, or supply a context holding one invocation per run.`
479
+ );
480
+ }
481
+ return audit.admitted;
482
+ }
483
+
370
484
  // src/rl/exporters.ts
371
- async function toDpoRows(triples, lookups) {
485
+ var DPO_CONTEXT_REQUIREMENT = {
486
+ exporter: "DPO export",
487
+ contextType: "DpoLineContext",
488
+ because: "a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair \u2014 which is DPO trained to PREFER the gaming trajectory."
489
+ };
490
+ async function toDpoRows(triples, lookups, context) {
491
+ const admitted = admitUngatedByInvocation(
492
+ triples,
493
+ (t) => [t.chosenRunId, t.rejectedRunId],
494
+ context,
495
+ DPO_CONTEXT_REQUIREMENT
496
+ );
372
497
  const out = [];
373
- for (const t of triples) {
498
+ for (const t of admitted) {
374
499
  const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
375
500
  Promise.resolve(lookups.promptOf(t.chosenRunId)),
376
501
  Promise.resolve(lookups.promptOf(t.rejectedRunId)),
@@ -403,46 +528,41 @@ async function toDpoRows(triples, lookups) {
403
528
  function toDpoJsonl(rows) {
404
529
  return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
405
530
  }
406
- async function toGrpoRows(runs, lookups) {
407
- const rewardOf2 = lookups.rewardOf ?? defaultReward;
408
- const promptHashByScenario = /* @__PURE__ */ new Map();
531
+ async function toGrpoRows(lines, lookups) {
532
+ return grpoRowsFromLines(lines, lookups);
533
+ }
534
+ async function grpoRowsFromLines(lines, lookups) {
409
535
  const grouped = /* @__PURE__ */ new Map();
410
- for (const r of runs) {
411
- const reward2 = rewardOf2(r);
412
- if (!isTrainingRunEligible(r, reward2, lookups)) continue;
413
- const existingPromptHash = promptHashByScenario.get(r.scenarioId);
414
- if (existingPromptHash !== void 0 && existingPromptHash !== r.promptHash) {
415
- throw new Error(
416
- `toGrpoRows: scenario "${r.scenarioId}" contains mixed prompt identities "${existingPromptHash}" and "${r.promptHash}"`
417
- );
418
- }
419
- promptHashByScenario.set(r.scenarioId, r.promptHash);
420
- const key = `${r.scenarioId}\0${r.promptHash}`;
421
- const group = grouped.get(key) ?? {
422
- scenarioId: r.scenarioId,
423
- promptHash: r.promptHash,
424
- scored: []
425
- };
426
- group.scored.push({ run: r, reward: reward2 });
427
- grouped.set(key, group);
536
+ for (const line of lines) {
537
+ if (!isSelectedSplit(line, lookups)) continue;
538
+ const arr = grouped.get(line.task.instance_id) ?? [];
539
+ arr.push(line);
540
+ grouped.set(line.task.instance_id, arr);
428
541
  }
429
542
  const rows = [];
430
- for (const { scenarioId, promptHash, scored } of grouped.values()) {
543
+ for (const [scenarioId, group] of grouped.entries()) {
544
+ if (group.length === 0) continue;
545
+ const scored = [];
546
+ for (const line of group) {
547
+ const reward = trainableLineReward(line);
548
+ if (reward === null) continue;
549
+ scored.push({ line, reward });
550
+ }
431
551
  if (scored.length < 2) continue;
432
552
  const prompts = await Promise.all(
433
- scored.map(({ run }) => Promise.resolve(lookups.promptOf(run.runId)))
553
+ scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id)))
434
554
  );
435
555
  const prompt = prompts[0];
436
556
  if (prompts.some((value) => value !== prompt)) {
437
557
  throw new Error(
438
- `toGrpoRows: prompt identity "${promptHash}" resolves to different text within scenario "${scenarioId}"`
558
+ `toGrpoRows: scenario "${scenarioId}" resolves to different prompt text within one group`
439
559
  );
440
560
  }
441
561
  const completions = await Promise.all(
442
- scored.map(({ run }) => Promise.resolve(lookups.completionOf(run.runId)))
562
+ scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id)))
443
563
  );
444
- const rewards = scored.map(({ reward: reward2 }) => reward2);
445
- const runIds = scored.map(({ run }) => run.runId);
564
+ const rewards = scored.map(({ reward }) => reward);
565
+ const runIds = scored.map(({ line }) => line.run_id);
446
566
  rows.push({
447
567
  prompt,
448
568
  completions,
@@ -450,7 +570,6 @@ async function toGrpoRows(runs, lookups) {
450
570
  runIds,
451
571
  meta: {
452
572
  scenarioId,
453
- promptHash,
454
573
  n: completions.length,
455
574
  meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
456
575
  }
@@ -461,17 +580,30 @@ async function toGrpoRows(runs, lookups) {
461
580
  function toGrpoJsonl(rows) {
462
581
  return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
463
582
  }
464
- async function toSftRows(runs, lookups) {
583
+ async function toSftRows(lines, lookups) {
584
+ return sftRowsFromLines(lines, lookups);
585
+ }
586
+ async function sftRowsFromLines(lines, lookups) {
465
587
  const include = lookups.include ?? (() => true);
588
+ const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0;
589
+ if (!Number.isFinite(minimumQualityExclusive)) {
590
+ throw new Error("minimumQualityExclusive must be finite");
591
+ }
466
592
  const rows = [];
467
- for (const r of runs) {
468
- const score = runTaskScore(r);
469
- if (!isTrainingRunEligible(r, score, lookups)) continue;
470
- if (!include(r)) continue;
471
- const system = lookups.systemOf?.(r);
593
+ for (const line of lines) {
594
+ assertRewardGate(line, "SFT export");
595
+ if (isLineRealnessGated(line)) continue;
596
+ if (!isSelectedSplit(line, lookups)) continue;
597
+ const score = trainableLineReward(line);
598
+ if (score === null || score <= minimumQualityExclusive) continue;
599
+ if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {
600
+ continue;
601
+ }
602
+ if (!include(line)) continue;
603
+ const system = lookups.systemOf?.(line);
472
604
  const [prompt, completion] = await Promise.all([
473
- Promise.resolve(lookups.promptOf(r.runId)),
474
- Promise.resolve(lookups.completionOf(r.runId))
605
+ Promise.resolve(lookups.promptOf(line.run_id)),
606
+ Promise.resolve(lookups.completionOf(line.run_id))
475
607
  ]);
476
608
  const messages = [];
477
609
  if (system) messages.push({ role: "system", content: system });
@@ -480,11 +612,11 @@ async function toSftRows(runs, lookups) {
480
612
  rows.push({
481
613
  messages,
482
614
  meta: {
483
- runId: r.runId,
484
- candidateId: r.candidateId,
485
- scenarioId: r.scenarioId,
615
+ runId: line.run_id,
616
+ candidateId: line.candidate_id ?? null,
617
+ scenarioId: line.task.instance_id,
486
618
  score,
487
- model: r.model
619
+ model: line.policy.model
488
620
  }
489
621
  });
490
622
  }
@@ -493,9 +625,10 @@ async function toSftRows(runs, lookups) {
493
625
  function toSftJsonl(rows) {
494
626
  return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
495
627
  }
496
- async function toPrmRows(triples, lookups) {
628
+ async function toPrmRows(triples, lookups, context) {
629
+ const admitted = admitPrmTriples(triples, context);
497
630
  const rows = [];
498
- for (const t of triples) {
631
+ for (const t of admitted) {
499
632
  const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
500
633
  const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
501
634
  const prefixStepText = [];
@@ -524,11 +657,59 @@ async function toPrmRows(triples, lookups) {
524
657
  }
525
658
  return rows;
526
659
  }
660
+ function assertPrmTrainableLine(line, mintedWithMaxSteps) {
661
+ const id = line.rollout_id;
662
+ if (line.provenance.gap !== void 0) {
663
+ throw new Error(
664
+ `PRM export: rollout ${id} is a gap line (${line.provenance.gap}) \u2014 refusing to build a process-reward row from a trajectory that was never captured`
665
+ );
666
+ }
667
+ if (line.steps === void 0 || line.steps.length === 0) {
668
+ throw new Error(
669
+ `PRM export: rollout ${id} carries no steps \u2014 refusing to build a process-reward row with no trajectory`
670
+ );
671
+ }
672
+ if (line.outcome.is_truncated) {
673
+ throw new Error(
674
+ `PRM export: rollout ${id} is marked truncated \u2014 refusing to assign step-level credit over a partial trajectory`
675
+ );
676
+ }
677
+ if (mintedWithMaxSteps !== void 0 && line.steps.length >= mintedWithMaxSteps) {
678
+ throw new Error(
679
+ `PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} \u2014 its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`
680
+ );
681
+ }
682
+ }
683
+ var PRM_CONTEXT_REQUIREMENT = {
684
+ exporter: "PRM export",
685
+ contextType: "PrmLineContext",
686
+ because: "without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one."
687
+ };
688
+ function admitPrmTriples(triples, context) {
689
+ return admitUngatedByInvocation(
690
+ triples,
691
+ (t) => [t.prefixRunId, t.rejectedRunId],
692
+ context,
693
+ PRM_CONTEXT_REQUIREMENT,
694
+ (line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps)
695
+ );
696
+ }
527
697
  function toPrmJsonl(rows) {
528
698
  return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
529
699
  }
530
- function stepRewardsToJsonl(stepRewards) {
531
- const rows = stepRewards.map((s) => ({
700
+ var STEP_REWARD_CONTEXT_REQUIREMENT = {
701
+ exporter: "step-reward export",
702
+ contextType: "RolloutLineContext",
703
+ because: "a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success \u2014 so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere."
704
+ };
705
+ function stepRewardsToJsonl(stepRewards, context) {
706
+ const admitted = admitUngatedByInvocation(
707
+ stepRewards,
708
+ (s) => [s.runId],
709
+ context,
710
+ STEP_REWARD_CONTEXT_REQUIREMENT
711
+ );
712
+ const rows = admitted.map((s) => ({
532
713
  runId: s.runId,
533
714
  spanId: s.spanId,
534
715
  stepIndex: s.stepIndex,
@@ -538,24 +719,9 @@ function stepRewardsToJsonl(stepRewards) {
538
719
  }));
539
720
  return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
540
721
  }
541
- function defaultReward(run) {
542
- return runTaskScore(run) ?? null;
543
- }
544
- function isTrainingRunEligible(run, quality, options = {}) {
545
- const minimumQualityExclusive = options.minimumQualityExclusive ?? 0;
546
- if (!Number.isFinite(minimumQualityExclusive)) {
547
- throw new Error("minimumQualityExclusive must be finite");
548
- }
549
- if (quality === null || quality === void 0) return false;
550
- if (!Number.isFinite(quality)) {
551
- throw new Error(`training quality for run "${run.runId}" must be finite`);
552
- }
553
- if (quality <= minimumQualityExclusive) return false;
554
- if (run.terminalOutcome !== "succeeded") return false;
555
- if (run.failureClass !== void 0 || run.terminalFailureReason !== void 0) return false;
556
- if (run.outcome.realness?.gated === true) return false;
557
- if (run.splitTag === "search") return true;
558
- return run.splitTag === "holdout" && options.allowHeldOutTrainingData === true;
722
+ function isSelectedSplit(line, options) {
723
+ if (options.splitFilter !== void 0) return options.splitFilter.includes(line.task.split);
724
+ return isSplitEligible(line, options);
559
725
  }
560
726
 
561
727
  // src/rl/dataset.ts
@@ -584,15 +750,8 @@ function validateDatasetFormats(value) {
584
750
  }
585
751
  return formats;
586
752
  }
587
- function reward(r, rewardOf2) {
588
- const value = rewardOf2 ? rewardOf2(r) : runTaskScore(r) ?? null;
589
- if (value !== null && !Number.isFinite(value)) {
590
- throw new Error(`buildRlDataset: reward for run "${r.runId}" must be finite`);
591
- }
592
- return value;
593
- }
594
753
  function distinct(xs) {
595
- return [...new Set(xs)].sort();
754
+ return [...new Set(xs.filter((x) => typeof x === "string" && x.length > 0))].sort();
596
755
  }
597
756
  function computeRewardStats(values) {
598
757
  if (values.length === 0) {
@@ -606,47 +765,50 @@ function computeRewardStats(values) {
606
765
  const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
607
766
  return { n, mean, median, min: sorted[0], max: sorted[n - 1], std: Math.sqrt(variance) };
608
767
  }
609
- function computeStats(records, rewardOf2) {
610
- const splits = { search: 0, dev: 0, holdout: 0 };
768
+ function computeStatsFromLines(lines) {
769
+ const splits = { search: 0, dev: 0, holdout: 0, canary: 0 };
611
770
  let inTok = 0;
612
771
  let outTok = 0;
613
772
  let cost = 0;
773
+ let rolloutsWithoutCost = 0;
614
774
  const rewards = [];
615
- for (const r of records) {
616
- splits[r.splitTag] = (splits[r.splitTag] ?? 0) + 1;
617
- inTok += r.tokenUsage.input;
618
- outTok += r.tokenUsage.output;
619
- cost += r.costUsd ?? 0;
620
- const rw = reward(r, rewardOf2);
775
+ for (const line of lines) {
776
+ splits[line.task.split] += 1;
777
+ inTok += line.cost.tokens_in ?? 0;
778
+ outTok += line.cost.tokens_out ?? 0;
779
+ if (line.cost.usd === null) rolloutsWithoutCost++;
780
+ else cost += line.cost.usd;
781
+ const rw = trainableLineReward(line);
621
782
  if (rw !== null) rewards.push(rw);
622
783
  }
623
784
  return {
624
- records: records.length,
785
+ records: lines.length,
625
786
  scoredRecords: rewards.length,
626
787
  splits,
627
788
  reward: computeRewardStats(rewards),
628
- models: distinct(records.map((r) => r.model)),
629
- promptHashes: distinct(records.map((r) => r.promptHash)),
630
- commitShas: distinct(records.map((r) => r.commitSha)),
789
+ models: distinct(lines.map((l) => l.policy.model)),
790
+ promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),
791
+ commitShas: distinct(lines.map((l) => l.policy.profile_commit)),
631
792
  totalTokens: { input: inTok, output: outTok },
632
- totalCostUsd: cost
793
+ totalCostUsd: cost,
794
+ rolloutsWithoutCost
633
795
  };
634
796
  }
635
- async function buildRlDataset(records, lookups, config, preferences) {
636
- if (records.length === 0) {
637
- throw new Error("buildRlDataset: no records \u2014 refusing to package an empty dataset");
797
+ async function buildRlDataset(lines, lookups, config, preferences) {
798
+ if (lines.length === 0) {
799
+ throw new Error("buildRlDataset: no rollout lines \u2014 refusing to package an empty dataset");
638
800
  }
639
801
  const formats = validateDatasetFormats(config.formats === void 0 ? ["sft"] : config.formats);
640
802
  const files = {};
641
803
  const rowCounts = {};
642
804
  if (formats.includes("grpo")) {
643
- const rows = await toGrpoRows(records, lookups);
805
+ const rows = await toGrpoRows(lines, lookups);
644
806
  requireRows("grpo", rows.length);
645
807
  files["train.grpo.jsonl"] = toGrpoJsonl(rows);
646
808
  rowCounts.grpo = rows.length;
647
809
  }
648
810
  if (formats.includes("sft")) {
649
- const rows = await toSftRows(records, lookups);
811
+ const rows = await toSftRows(lines, lookups);
650
812
  requireRows("sft", rows.length);
651
813
  files["train.sft.jsonl"] = toSftJsonl(rows);
652
814
  rowCounts.sft = rows.length;
@@ -655,7 +817,7 @@ async function buildRlDataset(records, lookups, config, preferences) {
655
817
  if (!preferences) {
656
818
  throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
657
819
  }
658
- const rows = await toDpoRows(preferences.triples, preferences.lookups);
820
+ const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines });
659
821
  requireRows("dpo", rows.length);
660
822
  files["train.dpo.jsonl"] = toDpoJsonl(rows);
661
823
  rowCounts.dpo = rows.length;
@@ -667,7 +829,7 @@ async function buildRlDataset(records, lookups, config, preferences) {
667
829
  ...config,
668
830
  formats,
669
831
  rowCounts,
670
- stats: computeStats(records, lookups.rewardOf)
832
+ stats: computeStatsFromLines(lines)
671
833
  };
672
834
  files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}
673
835
  `;
@@ -688,7 +850,8 @@ function stat(value) {
688
850
  function datasheetToMarkdown(m) {
689
851
  const s = m.stats;
690
852
  const total = s.records || 1;
691
- const splitLines = ["search", "dev", "holdout"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
853
+ const splitLines = ["search", "dev", "holdout", "canary"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
854
+ const costNote = s.rolloutsWithoutCost > 0 ? ` (floor \u2014 ${s.rolloutsWithoutCost} rollout(s) never captured a cost)` : "";
692
855
  const deterministic = m.reward.kind === "deterministic";
693
856
  return [
694
857
  `# Dataset: ${m.name} \`v${m.version}\``,
@@ -714,7 +877,7 @@ function datasheetToMarkdown(m) {
714
877
  `- **Models:** ${s.models.join(", ")}`,
715
878
  `- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
716
879
  `- **Commits:** ${s.commitShas.join(", ")}`,
717
- `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}`,
880
+ `- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,
718
881
  "",
719
882
  "## Quality gates",
720
883
  `- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
@@ -765,7 +928,8 @@ function readCorpus(corpusPath) {
765
928
  return out;
766
929
  }
767
930
  function rewardOf(r) {
768
- return runTaskScore(r) ?? null;
931
+ const v = trainingScore(r);
932
+ return typeof v === "number" && Number.isFinite(v) ? v : null;
769
933
  }
770
934
  async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
771
935
  let records = readCorpus(corpusPath).filter(
@@ -775,8 +939,8 @@ async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
775
939
  records = records.filter((r) => rewardOf(r) !== null);
776
940
  if (opts.minScore != null) {
777
941
  records = records.filter((r) => {
778
- const reward2 = rewardOf(r);
779
- return reward2 !== null && reward2 >= opts.minScore;
942
+ const reward = rewardOf(r);
943
+ return reward !== null && reward >= opts.minScore;
780
944
  });
781
945
  }
782
946
  const text = new Map(
@@ -787,7 +951,8 @@ async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
787
951
  completionOf: (id) => text.get(id)?.completion ?? "",
788
952
  allowHeldOutTrainingData: opts.allowHeldOutTrainingData
789
953
  };
790
- return buildRlDataset(records, lookups, config);
954
+ const { rows } = await mintRolloutRows(records, new InMemoryTraceStore());
955
+ return buildRlDataset(rows, lookups, config);
791
956
  }
792
957
 
793
958
  // src/rl/predictive-validity-researcher.ts
@@ -897,7 +1062,8 @@ var PredictiveValidityResearcher = class {
897
1062
  overfitGap: null,
898
1063
  baselineOverfitGap: null,
899
1064
  medianCandidateCost: null,
900
- medianBaselineCost: null
1065
+ medianBaselineCost: null,
1066
+ realnessGatedRuns: 0
901
1067
  },
902
1068
  reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
903
1069
  rejectionCode: "few_runs"
@@ -938,29 +1104,56 @@ var PredictiveValidityResearcher = class {
938
1104
  };
939
1105
 
940
1106
  // src/rl/preferences.ts
941
- var DEFAULT_REWARD = (run) => {
942
- return runTaskScore(run) ?? null;
943
- };
944
- function extractPreferences(runs, opts = {}) {
1107
+ var SPLIT_DEFAULT = "search";
1108
+ function extractPreferences(lines, opts = {}) {
945
1109
  const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
946
1110
  const minMargin = opts.minMargin ?? 0.05;
947
- const splitTag = opts.splitTag;
948
- const rewardOf2 = opts.rewardOf ?? DEFAULT_REWARD;
949
- if (splitTag === "holdout" && opts.allowHeldOutTrainingData !== true) {
1111
+ const requestedSplit = opts.split;
1112
+ if (requestedSplit === "holdout" && opts.allowHeldOutTrainingData !== true) {
1113
+ throw new Error('extractPreferences: split "holdout" requires allowHeldOutTrainingData: true');
1114
+ }
1115
+ if (requestedSplit === "dev" || requestedSplit === "canary") {
950
1116
  throw new Error(
951
- 'extractPreferences: splitTag "holdout" requires allowHeldOutTrainingData: true'
1117
+ `extractPreferences: split "${requestedSplit}" is evaluation-only; train from "search"`
952
1118
  );
953
1119
  }
954
- if (splitTag === "dev") {
955
- throw new Error('extractPreferences: splitTag "dev" is evaluation-only; train from "search"');
956
- }
957
- const scoredEntries = [];
958
- for (const run of runs) {
959
- if (splitTag !== void 0 && run.splitTag !== splitTag) continue;
960
- const s = rewardOf2(run);
961
- if (!isTrainingRunEligible(run, s, opts)) continue;
962
- scoredEntries.push({ run, score: s });
1120
+ const candidates = candidatesFromLines(lines, opts);
1121
+ const report = pairCandidates(candidates.rows, strategy, minMargin);
1122
+ return { ...report, linesWithoutCandidateId: candidates.withoutCandidateId };
1123
+ }
1124
+ function candidatesFromLines(lines, opts) {
1125
+ const split = opts.split ?? SPLIT_DEFAULT;
1126
+ const rows = [];
1127
+ let withoutCandidateId = 0;
1128
+ for (const line of lines) {
1129
+ if (line.task.split !== split) continue;
1130
+ if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {
1131
+ continue;
1132
+ }
1133
+ const score = trainableLineReward(line);
1134
+ if (score === null) continue;
1135
+ const candidateId = line.candidate_id;
1136
+ if (candidateId === null || candidateId === void 0 || candidateId.length === 0) {
1137
+ withoutCandidateId++;
1138
+ continue;
1139
+ }
1140
+ rows.push({
1141
+ scenarioId: line.task.instance_id,
1142
+ runId: line.run_id,
1143
+ candidateId,
1144
+ seed: line.task.seed,
1145
+ score,
1146
+ // `policy.*` is nullable on the wire; a minted line always carries these
1147
+ // (RunRecord makes them mandatory). Empty string marks "not recorded" so
1148
+ // `toTRLFormat`'s hash lookup fails visibly instead of silently matching.
1149
+ promptHash: line.policy.prompt_hash ?? "",
1150
+ configHash: line.policy.config_hash ?? "",
1151
+ model: line.policy.model ?? ""
1152
+ });
963
1153
  }
1154
+ return { rows, withoutCandidateId };
1155
+ }
1156
+ function pairCandidates(scoredEntries, strategy, minMargin) {
964
1157
  const pairs = [];
965
1158
  let pairsBelowMargin = 0;
966
1159
  let cellsSingleton = 0;
@@ -968,13 +1161,12 @@ function extractPreferences(runs, opts = {}) {
968
1161
  if (strategy === "paired-by-scenario-and-seed") {
969
1162
  const groups = /* @__PURE__ */ new Map();
970
1163
  for (const e of scoredEntries) {
971
- const sid = scenarioOf(e.run);
972
- const key = `${sid}::${e.run.seed}`;
1164
+ const key = `${e.scenarioId}::${e.seed}`;
973
1165
  const arr = groups.get(key) ?? [];
974
1166
  arr.push(e);
975
1167
  groups.set(key, arr);
976
1168
  }
977
- for (const [key, members] of groups.entries()) {
1169
+ for (const members of groups.values()) {
978
1170
  cellsInspected++;
979
1171
  if (members.length < 2) {
980
1172
  cellsSingleton++;
@@ -984,8 +1176,8 @@ function extractPreferences(runs, opts = {}) {
984
1176
  for (let j = i + 1; j < members.length; j++) {
985
1177
  const a = members[i];
986
1178
  const b = members[j];
987
- if (a.run.candidateId === b.run.candidateId) continue;
988
- const result = makePair(a, b, key.split("::")[0], minMargin);
1179
+ if (a.candidateId === b.candidateId) continue;
1180
+ const result = makePair(a, b, a.scenarioId, minMargin);
989
1181
  if (result.kind === "admit") pairs.push(result.pair);
990
1182
  else pairsBelowMargin++;
991
1183
  }
@@ -994,24 +1186,22 @@ function extractPreferences(runs, opts = {}) {
994
1186
  } else if (strategy === "paired-by-scenario") {
995
1187
  const byScenarioVariant = /* @__PURE__ */ new Map();
996
1188
  for (const e of scoredEntries) {
997
- const sid = scenarioOf(e.run);
998
- let perScenario = byScenarioVariant.get(sid);
1189
+ let perScenario = byScenarioVariant.get(e.scenarioId);
999
1190
  if (!perScenario) {
1000
1191
  perScenario = /* @__PURE__ */ new Map();
1001
- byScenarioVariant.set(sid, perScenario);
1192
+ byScenarioVariant.set(e.scenarioId, perScenario);
1002
1193
  }
1003
- const cur = perScenario.get(e.run.candidateId);
1194
+ const cur = perScenario.get(e.candidateId);
1004
1195
  if (cur) {
1005
1196
  cur.sum += e.score;
1006
1197
  cur.n++;
1007
- } else perScenario.set(e.run.candidateId, { run: e.run, sum: e.score, n: 1 });
1198
+ } else perScenario.set(e.candidateId, { entry: e, sum: e.score, n: 1 });
1008
1199
  }
1009
1200
  for (const [sid, perVariant] of byScenarioVariant.entries()) {
1010
1201
  cellsInspected++;
1011
- const arr = [...perVariant.entries()].map(([vid, agg]) => ({
1012
- run: agg.run,
1013
- score: agg.sum / agg.n,
1014
- variantId: vid
1202
+ const arr = [...perVariant.values()].map((agg) => ({
1203
+ ...agg.entry,
1204
+ score: agg.sum / agg.n
1015
1205
  }));
1016
1206
  if (arr.length < 2) {
1017
1207
  cellsSingleton++;
@@ -1028,10 +1218,9 @@ function extractPreferences(runs, opts = {}) {
1028
1218
  } else {
1029
1219
  const byScenario = /* @__PURE__ */ new Map();
1030
1220
  for (const e of scoredEntries) {
1031
- const sid = scenarioOf(e.run);
1032
- const arr = byScenario.get(sid) ?? [];
1221
+ const arr = byScenario.get(e.scenarioId) ?? [];
1033
1222
  arr.push(e);
1034
- byScenario.set(sid, arr);
1223
+ byScenario.set(e.scenarioId, arr);
1035
1224
  }
1036
1225
  for (const [sid, arr] of byScenario.entries()) {
1037
1226
  cellsInspected++;
@@ -1042,7 +1231,7 @@ function extractPreferences(runs, opts = {}) {
1042
1231
  const sorted = [...arr].sort((a, b) => a.score - b.score);
1043
1232
  const top = sorted[sorted.length - 1];
1044
1233
  const bot = sorted[0];
1045
- if (top.run.candidateId === bot.run.candidateId) {
1234
+ if (top.candidateId === bot.candidateId) {
1046
1235
  cellsSingleton++;
1047
1236
  continue;
1048
1237
  }
@@ -1053,8 +1242,51 @@ function extractPreferences(runs, opts = {}) {
1053
1242
  }
1054
1243
  return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
1055
1244
  }
1056
- function toAnthropicFormat(triples) {
1057
- return triples.map((t) => ({
1245
+ var PREFERENCE_RUN_IDS = (t) => [
1246
+ t.chosenRunId,
1247
+ t.rejectedRunId
1248
+ ];
1249
+ var TRL_CONTEXT_REQUIREMENT = {
1250
+ exporter: "TRL preference export",
1251
+ contextType: "RolloutLineContext",
1252
+ because: "a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair."
1253
+ };
1254
+ var ANTHROPIC_CONTEXT_REQUIREMENT = {
1255
+ exporter: "Anthropic preference export",
1256
+ contextType: "RolloutLineContext",
1257
+ because: "a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one."
1258
+ };
1259
+ async function toTRLFormat(triples, lookups, context) {
1260
+ const admitted = admitUngatedByInvocation(
1261
+ triples,
1262
+ PREFERENCE_RUN_IDS,
1263
+ context,
1264
+ TRL_CONTEXT_REQUIREMENT
1265
+ );
1266
+ const out = [];
1267
+ for (const t of admitted) {
1268
+ const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
1269
+ Promise.resolve(lookups.promptOf(t.chosenRunId)),
1270
+ Promise.resolve(lookups.promptOf(t.rejectedRunId)),
1271
+ Promise.resolve(lookups.completionOf(t.chosenRunId)),
1272
+ Promise.resolve(lookups.completionOf(t.rejectedRunId))
1273
+ ]);
1274
+ if (chosenPrompt !== rejectedPrompt) {
1275
+ throw new Error(
1276
+ `toTRLFormat: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`
1277
+ );
1278
+ }
1279
+ out.push({ prompt: chosenPrompt, chosen, rejected });
1280
+ }
1281
+ return out;
1282
+ }
1283
+ function toAnthropicFormat(triples, context) {
1284
+ return admitUngatedByInvocation(
1285
+ triples,
1286
+ PREFERENCE_RUN_IDS,
1287
+ context,
1288
+ ANTHROPIC_CONTEXT_REQUIREMENT
1289
+ ).map((t) => ({
1058
1290
  scenarioId: t.scenarioId,
1059
1291
  chosenRunId: t.chosenRunId,
1060
1292
  rejectedRunId: t.rejectedRunId,
@@ -1065,31 +1297,29 @@ function makePair(a, b, scenarioId, minMargin) {
1065
1297
  const margin = Math.abs(a.score - b.score);
1066
1298
  if (margin < minMargin) return { kind: "reject" };
1067
1299
  const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
1300
+ const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : void 0;
1068
1301
  return {
1069
1302
  kind: "admit",
1070
1303
  pair: {
1071
1304
  scenarioId,
1072
- chosenRunId: chosen.run.runId,
1073
- rejectedRunId: rejected.run.runId,
1074
- chosenVariantId: chosen.run.candidateId,
1075
- rejectedVariantId: rejected.run.candidateId,
1305
+ chosenRunId: chosen.runId,
1306
+ rejectedRunId: rejected.runId,
1307
+ chosenVariantId: chosen.candidateId,
1308
+ rejectedVariantId: rejected.candidateId,
1076
1309
  marginScore: chosen.score - rejected.score,
1077
1310
  scores: { chosen: chosen.score, rejected: rejected.score },
1078
- seed: chosen.run.seed === rejected.run.seed ? chosen.run.seed : void 0,
1311
+ seed,
1079
1312
  meta: {
1080
- chosenPromptHash: chosen.run.promptHash,
1081
- rejectedPromptHash: rejected.run.promptHash,
1082
- chosenConfigHash: chosen.run.configHash,
1083
- rejectedConfigHash: rejected.run.configHash,
1084
- chosenModel: chosen.run.model,
1085
- rejectedModel: rejected.run.model
1313
+ chosenPromptHash: chosen.promptHash,
1314
+ rejectedPromptHash: rejected.promptHash,
1315
+ chosenConfigHash: chosen.configHash,
1316
+ rejectedConfigHash: rejected.configHash,
1317
+ chosenModel: chosen.model,
1318
+ rejectedModel: rejected.model
1086
1319
  }
1087
1320
  }
1088
1321
  };
1089
1322
  }
1090
- function scenarioOf(run) {
1091
- return run.scenarioId;
1092
- }
1093
1323
 
1094
1324
  // src/rl/process-reward.ts
1095
1325
  async function extractStepRewards(store, runId, opts) {
@@ -1224,11 +1454,13 @@ async function runRLCampaign(opts) {
1224
1454
  campaign.runs,
1225
1455
  opts.verifiableReward ?? {}
1226
1456
  );
1227
- const preferences = extractPreferences(campaign.runs, {
1457
+ const scoredRuns = campaign.runs.filter((run) => runTaskScore(run) !== void 0);
1458
+ const { rows: rolloutLines } = await mintRolloutRows(scoredRuns, new InMemoryTraceStore());
1459
+ const preferences = extractPreferences(rolloutLines, {
1228
1460
  ...opts.preferences,
1229
1461
  strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
1230
1462
  minMargin: opts.preferences?.minMargin ?? 0.05,
1231
- splitTag: opts.preferences?.splitTag ?? splitTag
1463
+ split: opts.preferences?.split ?? splitTag
1232
1464
  });
1233
1465
  let interimConfidence = null;
1234
1466
  if (opts.report?.comparator) {
@@ -1257,13 +1489,15 @@ async function runRLCampaign(opts) {
1257
1489
  }
1258
1490
  const trainerRows = {};
1259
1491
  if (opts.trainerExport?.dpo) {
1260
- trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo);
1492
+ trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, {
1493
+ lines: rolloutLines
1494
+ });
1261
1495
  }
1262
1496
  if (opts.trainerExport?.grpo) {
1263
- trainerRows.grpo = await toGrpoRows(campaign.runs, opts.trainerExport.grpo);
1497
+ trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo);
1264
1498
  }
1265
1499
  if (opts.trainerExport?.sft) {
1266
- trainerRows.sft = await toSftRows(campaign.runs, opts.trainerExport.sft);
1500
+ trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft);
1267
1501
  }
1268
1502
  const summary = buildSummary({
1269
1503
  campaign,
@@ -1438,7 +1672,13 @@ var defaultBehaviorFeatures = (record) => {
1438
1672
  const turnsAborted = finiteOrNull(raw.turns_aborted);
1439
1673
  const completion = record.completion;
1440
1674
  return {
1441
- score: finiteOrNull(record.outcome?.holdoutScore) ?? finiteOrNull(record.outcome?.searchScore),
1675
+ // RAW (`observedSplitScore`), deliberately: this feature vector is one
1676
+ // half of a sim-vs-production divergence measurement. Gating a gamed run to
1677
+ // 0 would move the simulated distribution toward production and report the
1678
+ // simulator as MORE faithful precisely where it is being gamed. Each split
1679
+ // is read separately rather than through `observedScore` so a non-finite
1680
+ // holdout score falls back to search instead of poisoning the bucket.
1681
+ score: finiteOrNull(observedSplitScore(record, "holdout")) ?? finiteOrNull(observedSplitScore(record, "search")),
1442
1682
  failure_class: record.failureClass ?? null,
1443
1683
  wall_ms: finiteOrNull(record.wallMs),
1444
1684
  output_tokens: finiteOrNull(record.tokenUsage?.output),
@@ -1620,7 +1860,7 @@ function easyModeCheck(simulated, production, opts = {}) {
1620
1860
  const passRate = (records, side) => {
1621
1861
  let passes = 0;
1622
1862
  for (const r of records) {
1623
- const score = finiteOrNull(r.outcome?.holdoutScore) ?? finiteOrNull(r.outcome?.searchScore);
1863
+ const score = finiteOrNull(observedSplitScore(r, "holdout")) ?? finiteOrNull(observedSplitScore(r, "search"));
1624
1864
  if (score === null) {
1625
1865
  throw new ValidationError(
1626
1866
  `easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`
@@ -1769,12 +2009,16 @@ export {
1769
2009
  ABSENT_CATEGORY,
1770
2010
  DEFAULT_MIN_N_PER_FEATURE,
1771
2011
  DEFAULT_QUANTILE_BUCKETS,
2012
+ DPO_CONTEXT_REQUIREMENT,
1772
2013
  FileSystemOutcomeStore,
1773
2014
  InMemoryOutcomeStore,
2015
+ PRM_CONTEXT_REQUIREMENT,
1774
2016
  PredictiveValidityResearcher,
1775
2017
  REPRESENTATIVE_MIN_FIDELITY,
2018
+ STEP_REWARD_CONTEXT_REQUIREMENT,
1776
2019
  appendToCorpus,
1777
2020
  applyEloUpdate,
2021
+ assertPrmTrainableLine,
1778
2022
  bestOfN,
1779
2023
  bucketLabel,
1780
2024
  buildDatasetFromCorpus,
@@ -1796,7 +2040,6 @@ export {
1796
2040
  fitBradleyTerry,
1797
2041
  injectIrrelevantClause,
1798
2042
  inverseProbabilityWeighting,
1799
- isTrainingRunEligible,
1800
2043
  jsDivergence,
1801
2044
  observationsFromRunRecords,
1802
2045
  offPolicyEstimateAll,
@@ -1826,6 +2069,7 @@ export {
1826
2069
  toPrmRows,
1827
2070
  toSftJsonl,
1828
2071
  toSftRows,
2072
+ toTRLFormat,
1829
2073
  validateDatasetFormats,
1830
2074
  varianceBasedCurriculum,
1831
2075
  verificationReportToRunRecord