@tangle-network/agent-eval 0.128.1 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +271 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/rl.js
CHANGED
|
@@ -3,50 +3,66 @@ import {
|
|
|
3
3
|
inverseProbabilityWeighting,
|
|
4
4
|
offPolicyEstimateAll,
|
|
5
5
|
selfNormalizedImportanceWeighting
|
|
6
|
-
} from "./chunk-
|
|
6
|
+
} from "./chunk-T6RLYGAD.js";
|
|
7
7
|
import {
|
|
8
8
|
FileSystemOutcomeStore,
|
|
9
9
|
InMemoryOutcomeStore
|
|
10
10
|
} from "./chunk-3RF76KTD.js";
|
|
11
11
|
import {
|
|
12
12
|
runEvalCampaign
|
|
13
|
-
} from "./chunk-
|
|
13
|
+
} from "./chunk-YQN4ICPP.js";
|
|
14
|
+
import {
|
|
15
|
+
mintRolloutRows
|
|
16
|
+
} from "./chunk-H23X7XKK.js";
|
|
17
|
+
import {
|
|
18
|
+
isSplitEligible
|
|
19
|
+
} from "./chunk-OWN5NPMC.js";
|
|
20
|
+
import {
|
|
21
|
+
assertRewardGate
|
|
22
|
+
} from "./chunk-PC5DOSM7.js";
|
|
23
|
+
import "./chunk-RZTMDUO7.js";
|
|
14
24
|
import {
|
|
15
25
|
detectRewardHacking,
|
|
16
26
|
extractVerifiableReward,
|
|
17
27
|
extractVerifiableRewardsFromRecords,
|
|
18
28
|
filterDeterministicallyRewarded
|
|
19
|
-
} from "./chunk-
|
|
29
|
+
} from "./chunk-EG66UGL4.js";
|
|
20
30
|
import {
|
|
21
31
|
campaignCellToRunRecord
|
|
22
|
-
} from "./chunk-
|
|
23
|
-
import "./chunk-
|
|
32
|
+
} from "./chunk-E7QXT7SX.js";
|
|
33
|
+
import "./chunk-SFLLL76A.js";
|
|
24
34
|
import {
|
|
25
35
|
rubricPredictiveValidity
|
|
26
|
-
} from "./chunk-
|
|
36
|
+
} from "./chunk-TQ7LNKZ3.js";
|
|
27
37
|
import {
|
|
28
38
|
evaluateInterimReleaseConfidence
|
|
29
39
|
} from "./chunk-MAZ26DC7.js";
|
|
30
|
-
import "./chunk-
|
|
31
|
-
import "./chunk-
|
|
40
|
+
import "./chunk-TJVT4QFF.js";
|
|
41
|
+
import "./chunk-7FO3TNPI.js";
|
|
32
42
|
import {
|
|
33
43
|
benjaminiHochberg,
|
|
34
44
|
wilcoxonSignedRank
|
|
35
|
-
} from "./chunk-
|
|
45
|
+
} from "./chunk-ZHTZ4EYI.js";
|
|
36
46
|
import {
|
|
37
47
|
observationsFromRunRecords,
|
|
38
48
|
thompsonCurriculum,
|
|
39
49
|
varianceBasedCurriculum
|
|
40
|
-
} from "./chunk-
|
|
41
|
-
import "./chunk-
|
|
50
|
+
} from "./chunk-G7MGMCZD.js";
|
|
51
|
+
import "./chunk-VCZ5FQYW.js";
|
|
42
52
|
import "./chunk-VI2UW6B6.js";
|
|
43
|
-
import
|
|
53
|
+
import {
|
|
54
|
+
InMemoryTraceStore
|
|
55
|
+
} from "./chunk-U4PHLT2N.js";
|
|
44
56
|
import "./chunk-PC4UYEBM.js";
|
|
45
57
|
import "./chunk-VQMK5FMP.js";
|
|
46
58
|
import {
|
|
47
59
|
runTaskScore
|
|
48
|
-
} from "./chunk-
|
|
60
|
+
} from "./chunk-56TAVBOK.js";
|
|
49
61
|
import "./chunk-MA6HLL3S.js";
|
|
62
|
+
import {
|
|
63
|
+
observedSplitScore,
|
|
64
|
+
trainingScore
|
|
65
|
+
} from "./chunk-OIUOT4QD.js";
|
|
50
66
|
import {
|
|
51
67
|
ValidationError
|
|
52
68
|
} from "./chunk-ONWEPEDO.js";
|
|
@@ -367,10 +383,119 @@ function escapeRegex(s) {
|
|
|
367
383
|
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "fs";
|
|
368
384
|
import { dirname } from "path";
|
|
369
385
|
|
|
386
|
+
// src/rl/rollout-input.ts
|
|
387
|
+
function trainableLineReward(line) {
|
|
388
|
+
assertRewardGate(line, "trainable reward");
|
|
389
|
+
const { reward } = line.outcome;
|
|
390
|
+
if (reward === null || !Number.isFinite(reward)) return null;
|
|
391
|
+
return reward;
|
|
392
|
+
}
|
|
393
|
+
function isLineRealnessGated(line) {
|
|
394
|
+
return line.outcome.realness_gated === true;
|
|
395
|
+
}
|
|
396
|
+
function push(index, key, line) {
|
|
397
|
+
const existing = index.get(key);
|
|
398
|
+
if (existing === void 0) index.set(key, [line]);
|
|
399
|
+
else existing.push(line);
|
|
400
|
+
}
|
|
401
|
+
function invocationIndex(lines) {
|
|
402
|
+
const byRollout = /* @__PURE__ */ new Map();
|
|
403
|
+
const byRun = /* @__PURE__ */ new Map();
|
|
404
|
+
for (const line of lines) {
|
|
405
|
+
push(byRollout, line.rollout_id, line);
|
|
406
|
+
push(byRun, line.run_id, line);
|
|
407
|
+
}
|
|
408
|
+
return { byRollout, byRun };
|
|
409
|
+
}
|
|
410
|
+
function resolveInvocation(index, id) {
|
|
411
|
+
const rollouts = index.byRollout.get(id) ?? [];
|
|
412
|
+
const runs = index.byRun.get(id) ?? [];
|
|
413
|
+
if (rollouts.length > 1) return { kind: "ambiguous", count: rollouts.length };
|
|
414
|
+
const exact = rollouts[0];
|
|
415
|
+
if (exact !== void 0) {
|
|
416
|
+
if (runs.some((line) => line !== exact)) {
|
|
417
|
+
return { kind: "ambiguous", count: 1 + runs.filter((line) => line !== exact).length };
|
|
418
|
+
}
|
|
419
|
+
return { kind: "resolved", line: exact };
|
|
420
|
+
}
|
|
421
|
+
if (runs.length > 1) return { kind: "ambiguous", count: runs.length };
|
|
422
|
+
const only = runs[0];
|
|
423
|
+
return only === void 0 ? { kind: "missing" } : { kind: "resolved", line: only };
|
|
424
|
+
}
|
|
425
|
+
function auditInvocationAdmission(items, idsOf, context, requirement, inspect) {
|
|
426
|
+
if (context === void 0 || context === null) {
|
|
427
|
+
throw new Error(
|
|
428
|
+
`${requirement.exporter}: a ${requirement.contextType} is required \u2014 ${requirement.because} Pass \`{ lines: (await mintRolloutRows(...)).rows }\`.`
|
|
429
|
+
);
|
|
430
|
+
}
|
|
431
|
+
const index = invocationIndex(context.lines);
|
|
432
|
+
const audit = {
|
|
433
|
+
admitted: [],
|
|
434
|
+
gatedDrops: 0,
|
|
435
|
+
ambiguousDrops: 0,
|
|
436
|
+
ambiguous: []
|
|
437
|
+
};
|
|
438
|
+
for (const item of items) {
|
|
439
|
+
const lines = [];
|
|
440
|
+
let ambiguous = false;
|
|
441
|
+
for (const id of idsOf(item)) {
|
|
442
|
+
const resolution = resolveInvocation(index, id);
|
|
443
|
+
if (resolution.kind === "missing") {
|
|
444
|
+
throw new Error(
|
|
445
|
+
`${requirement.exporter}: no rollout line supplied for run ${id} \u2014 its realness gate and capture quality are unknown`
|
|
446
|
+
);
|
|
447
|
+
}
|
|
448
|
+
if (resolution.kind === "ambiguous") {
|
|
449
|
+
ambiguous = true;
|
|
450
|
+
if (!audit.ambiguous.some((entry) => entry.id === id)) {
|
|
451
|
+
audit.ambiguous.push({ id, invocations: resolution.count });
|
|
452
|
+
}
|
|
453
|
+
continue;
|
|
454
|
+
}
|
|
455
|
+
lines.push(resolution.line);
|
|
456
|
+
}
|
|
457
|
+
for (const line of lines) {
|
|
458
|
+
assertRewardGate(line, requirement.exporter);
|
|
459
|
+
inspect?.(line);
|
|
460
|
+
}
|
|
461
|
+
if (ambiguous) {
|
|
462
|
+
audit.ambiguousDrops++;
|
|
463
|
+
continue;
|
|
464
|
+
}
|
|
465
|
+
if (lines.some(isLineRealnessGated)) {
|
|
466
|
+
audit.gatedDrops++;
|
|
467
|
+
continue;
|
|
468
|
+
}
|
|
469
|
+
audit.admitted.push(item);
|
|
470
|
+
}
|
|
471
|
+
return audit;
|
|
472
|
+
}
|
|
473
|
+
function admitUngatedByInvocation(items, idsOf, context, requirement, inspect) {
|
|
474
|
+
const audit = auditInvocationAdmission(items, idsOf, context, requirement, inspect);
|
|
475
|
+
if (audit.ambiguousDrops > 0) {
|
|
476
|
+
const named = audit.ambiguous.map((e) => `${e.id} (${e.invocations} invocations)`).join(", ");
|
|
477
|
+
console.warn(
|
|
478
|
+
`[${requirement.exporter}] dropped ${audit.ambiguousDrops} item(s): ${named} name more than one invocation in the supplied lines, so the realness gate cannot be read for the invocation the artifact meant. Reference the \`rollout_id\` instead of the \`run_id\`, or supply a context holding one invocation per run.`
|
|
479
|
+
);
|
|
480
|
+
}
|
|
481
|
+
return audit.admitted;
|
|
482
|
+
}
|
|
483
|
+
|
|
370
484
|
// src/rl/exporters.ts
|
|
371
|
-
|
|
485
|
+
var DPO_CONTEXT_REQUIREMENT = {
|
|
486
|
+
exporter: "DPO export",
|
|
487
|
+
contextType: "DpoLineContext",
|
|
488
|
+
because: "a PreferenceTriple carries only run ids and a bare margin number, so without the minted rollout lines this exporter cannot see the realness gate and will write a run that faked its success onto the CHOSEN side of the pair \u2014 which is DPO trained to PREFER the gaming trajectory."
|
|
489
|
+
};
|
|
490
|
+
async function toDpoRows(triples, lookups, context) {
|
|
491
|
+
const admitted = admitUngatedByInvocation(
|
|
492
|
+
triples,
|
|
493
|
+
(t) => [t.chosenRunId, t.rejectedRunId],
|
|
494
|
+
context,
|
|
495
|
+
DPO_CONTEXT_REQUIREMENT
|
|
496
|
+
);
|
|
372
497
|
const out = [];
|
|
373
|
-
for (const t of
|
|
498
|
+
for (const t of admitted) {
|
|
374
499
|
const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
|
|
375
500
|
Promise.resolve(lookups.promptOf(t.chosenRunId)),
|
|
376
501
|
Promise.resolve(lookups.promptOf(t.rejectedRunId)),
|
|
@@ -403,46 +528,41 @@ async function toDpoRows(triples, lookups) {
|
|
|
403
528
|
function toDpoJsonl(rows) {
|
|
404
529
|
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
405
530
|
}
|
|
406
|
-
async function toGrpoRows(
|
|
407
|
-
|
|
408
|
-
|
|
531
|
+
async function toGrpoRows(lines, lookups) {
|
|
532
|
+
return grpoRowsFromLines(lines, lookups);
|
|
533
|
+
}
|
|
534
|
+
async function grpoRowsFromLines(lines, lookups) {
|
|
409
535
|
const grouped = /* @__PURE__ */ new Map();
|
|
410
|
-
for (const
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
throw new Error(
|
|
416
|
-
`toGrpoRows: scenario "${r.scenarioId}" contains mixed prompt identities "${existingPromptHash}" and "${r.promptHash}"`
|
|
417
|
-
);
|
|
418
|
-
}
|
|
419
|
-
promptHashByScenario.set(r.scenarioId, r.promptHash);
|
|
420
|
-
const key = `${r.scenarioId}\0${r.promptHash}`;
|
|
421
|
-
const group = grouped.get(key) ?? {
|
|
422
|
-
scenarioId: r.scenarioId,
|
|
423
|
-
promptHash: r.promptHash,
|
|
424
|
-
scored: []
|
|
425
|
-
};
|
|
426
|
-
group.scored.push({ run: r, reward: reward2 });
|
|
427
|
-
grouped.set(key, group);
|
|
536
|
+
for (const line of lines) {
|
|
537
|
+
if (!isSelectedSplit(line, lookups)) continue;
|
|
538
|
+
const arr = grouped.get(line.task.instance_id) ?? [];
|
|
539
|
+
arr.push(line);
|
|
540
|
+
grouped.set(line.task.instance_id, arr);
|
|
428
541
|
}
|
|
429
542
|
const rows = [];
|
|
430
|
-
for (const
|
|
543
|
+
for (const [scenarioId, group] of grouped.entries()) {
|
|
544
|
+
if (group.length === 0) continue;
|
|
545
|
+
const scored = [];
|
|
546
|
+
for (const line of group) {
|
|
547
|
+
const reward = trainableLineReward(line);
|
|
548
|
+
if (reward === null) continue;
|
|
549
|
+
scored.push({ line, reward });
|
|
550
|
+
}
|
|
431
551
|
if (scored.length < 2) continue;
|
|
432
552
|
const prompts = await Promise.all(
|
|
433
|
-
scored.map(({
|
|
553
|
+
scored.map(({ line }) => Promise.resolve(lookups.promptOf(line.run_id)))
|
|
434
554
|
);
|
|
435
555
|
const prompt = prompts[0];
|
|
436
556
|
if (prompts.some((value) => value !== prompt)) {
|
|
437
557
|
throw new Error(
|
|
438
|
-
`toGrpoRows:
|
|
558
|
+
`toGrpoRows: scenario "${scenarioId}" resolves to different prompt text within one group`
|
|
439
559
|
);
|
|
440
560
|
}
|
|
441
561
|
const completions = await Promise.all(
|
|
442
|
-
scored.map(({
|
|
562
|
+
scored.map(({ line }) => Promise.resolve(lookups.completionOf(line.run_id)))
|
|
443
563
|
);
|
|
444
|
-
const rewards = scored.map(({ reward
|
|
445
|
-
const runIds = scored.map(({
|
|
564
|
+
const rewards = scored.map(({ reward }) => reward);
|
|
565
|
+
const runIds = scored.map(({ line }) => line.run_id);
|
|
446
566
|
rows.push({
|
|
447
567
|
prompt,
|
|
448
568
|
completions,
|
|
@@ -450,7 +570,6 @@ async function toGrpoRows(runs, lookups) {
|
|
|
450
570
|
runIds,
|
|
451
571
|
meta: {
|
|
452
572
|
scenarioId,
|
|
453
|
-
promptHash,
|
|
454
573
|
n: completions.length,
|
|
455
574
|
meanReward: rewards.reduce((s, x) => s + x, 0) / rewards.length
|
|
456
575
|
}
|
|
@@ -461,17 +580,30 @@ async function toGrpoRows(runs, lookups) {
|
|
|
461
580
|
function toGrpoJsonl(rows) {
|
|
462
581
|
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
463
582
|
}
|
|
464
|
-
async function toSftRows(
|
|
583
|
+
async function toSftRows(lines, lookups) {
|
|
584
|
+
return sftRowsFromLines(lines, lookups);
|
|
585
|
+
}
|
|
586
|
+
async function sftRowsFromLines(lines, lookups) {
|
|
465
587
|
const include = lookups.include ?? (() => true);
|
|
588
|
+
const minimumQualityExclusive = lookups.minimumQualityExclusive ?? 0;
|
|
589
|
+
if (!Number.isFinite(minimumQualityExclusive)) {
|
|
590
|
+
throw new Error("minimumQualityExclusive must be finite");
|
|
591
|
+
}
|
|
466
592
|
const rows = [];
|
|
467
|
-
for (const
|
|
468
|
-
|
|
469
|
-
if (
|
|
470
|
-
if (!
|
|
471
|
-
const
|
|
593
|
+
for (const line of lines) {
|
|
594
|
+
assertRewardGate(line, "SFT export");
|
|
595
|
+
if (isLineRealnessGated(line)) continue;
|
|
596
|
+
if (!isSelectedSplit(line, lookups)) continue;
|
|
597
|
+
const score = trainableLineReward(line);
|
|
598
|
+
if (score === null || score <= minimumQualityExclusive) continue;
|
|
599
|
+
if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {
|
|
600
|
+
continue;
|
|
601
|
+
}
|
|
602
|
+
if (!include(line)) continue;
|
|
603
|
+
const system = lookups.systemOf?.(line);
|
|
472
604
|
const [prompt, completion] = await Promise.all([
|
|
473
|
-
Promise.resolve(lookups.promptOf(
|
|
474
|
-
Promise.resolve(lookups.completionOf(
|
|
605
|
+
Promise.resolve(lookups.promptOf(line.run_id)),
|
|
606
|
+
Promise.resolve(lookups.completionOf(line.run_id))
|
|
475
607
|
]);
|
|
476
608
|
const messages = [];
|
|
477
609
|
if (system) messages.push({ role: "system", content: system });
|
|
@@ -480,11 +612,11 @@ async function toSftRows(runs, lookups) {
|
|
|
480
612
|
rows.push({
|
|
481
613
|
messages,
|
|
482
614
|
meta: {
|
|
483
|
-
runId:
|
|
484
|
-
candidateId:
|
|
485
|
-
scenarioId:
|
|
615
|
+
runId: line.run_id,
|
|
616
|
+
candidateId: line.candidate_id ?? null,
|
|
617
|
+
scenarioId: line.task.instance_id,
|
|
486
618
|
score,
|
|
487
|
-
model:
|
|
619
|
+
model: line.policy.model
|
|
488
620
|
}
|
|
489
621
|
});
|
|
490
622
|
}
|
|
@@ -493,9 +625,10 @@ async function toSftRows(runs, lookups) {
|
|
|
493
625
|
function toSftJsonl(rows) {
|
|
494
626
|
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
495
627
|
}
|
|
496
|
-
async function toPrmRows(triples, lookups) {
|
|
628
|
+
async function toPrmRows(triples, lookups, context) {
|
|
629
|
+
const admitted = admitPrmTriples(triples, context);
|
|
497
630
|
const rows = [];
|
|
498
|
-
for (const t of
|
|
631
|
+
for (const t of admitted) {
|
|
499
632
|
const prompt = await Promise.resolve(lookups.promptOf(t.prefixRunId));
|
|
500
633
|
const prefixSpanIds = lookups.prefixOf ? await Promise.resolve(lookups.prefixOf(t.prefixRunId, t.prefixStepIndex)) : [];
|
|
501
634
|
const prefixStepText = [];
|
|
@@ -524,11 +657,59 @@ async function toPrmRows(triples, lookups) {
|
|
|
524
657
|
}
|
|
525
658
|
return rows;
|
|
526
659
|
}
|
|
660
|
+
function assertPrmTrainableLine(line, mintedWithMaxSteps) {
|
|
661
|
+
const id = line.rollout_id;
|
|
662
|
+
if (line.provenance.gap !== void 0) {
|
|
663
|
+
throw new Error(
|
|
664
|
+
`PRM export: rollout ${id} is a gap line (${line.provenance.gap}) \u2014 refusing to build a process-reward row from a trajectory that was never captured`
|
|
665
|
+
);
|
|
666
|
+
}
|
|
667
|
+
if (line.steps === void 0 || line.steps.length === 0) {
|
|
668
|
+
throw new Error(
|
|
669
|
+
`PRM export: rollout ${id} carries no steps \u2014 refusing to build a process-reward row with no trajectory`
|
|
670
|
+
);
|
|
671
|
+
}
|
|
672
|
+
if (line.outcome.is_truncated) {
|
|
673
|
+
throw new Error(
|
|
674
|
+
`PRM export: rollout ${id} is marked truncated \u2014 refusing to assign step-level credit over a partial trajectory`
|
|
675
|
+
);
|
|
676
|
+
}
|
|
677
|
+
if (mintedWithMaxSteps !== void 0 && line.steps.length >= mintedWithMaxSteps) {
|
|
678
|
+
throw new Error(
|
|
679
|
+
`PRM export: rollout ${id} has ${line.steps.length} steps at the mint cap of ${mintedWithMaxSteps} \u2014 its middle steps may have been dropped, and a capped trajectory carries no marker to prove otherwise`
|
|
680
|
+
);
|
|
681
|
+
}
|
|
682
|
+
}
|
|
683
|
+
var PRM_CONTEXT_REQUIREMENT = {
|
|
684
|
+
exporter: "PRM export",
|
|
685
|
+
contextType: "PrmLineContext",
|
|
686
|
+
because: "without the minted rollout lines this exporter cannot see the realness gate (a triple carries only a bare reward number) and cannot tell a fully-captured trajectory from a capped or empty one."
|
|
687
|
+
};
|
|
688
|
+
function admitPrmTriples(triples, context) {
|
|
689
|
+
return admitUngatedByInvocation(
|
|
690
|
+
triples,
|
|
691
|
+
(t) => [t.prefixRunId, t.rejectedRunId],
|
|
692
|
+
context,
|
|
693
|
+
PRM_CONTEXT_REQUIREMENT,
|
|
694
|
+
(line) => assertPrmTrainableLine(line, context.mintedWithMaxSteps)
|
|
695
|
+
);
|
|
696
|
+
}
|
|
527
697
|
function toPrmJsonl(rows) {
|
|
528
698
|
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
529
699
|
}
|
|
530
|
-
|
|
531
|
-
|
|
700
|
+
var STEP_REWARD_CONTEXT_REQUIREMENT = {
|
|
701
|
+
exporter: "step-reward export",
|
|
702
|
+
contextType: "RolloutLineContext",
|
|
703
|
+
because: "a StepReward carries a runId and a bare per-step reward, and nothing that says whether that run faked its success \u2014 so without the minted rollout lines this exporter ships the step-level components of a gamed run at full value while the run-level scalar sits at 0 elsewhere."
|
|
704
|
+
};
|
|
705
|
+
function stepRewardsToJsonl(stepRewards, context) {
|
|
706
|
+
const admitted = admitUngatedByInvocation(
|
|
707
|
+
stepRewards,
|
|
708
|
+
(s) => [s.runId],
|
|
709
|
+
context,
|
|
710
|
+
STEP_REWARD_CONTEXT_REQUIREMENT
|
|
711
|
+
);
|
|
712
|
+
const rows = admitted.map((s) => ({
|
|
532
713
|
runId: s.runId,
|
|
533
714
|
spanId: s.spanId,
|
|
534
715
|
stepIndex: s.stepIndex,
|
|
@@ -538,24 +719,9 @@ function stepRewardsToJsonl(stepRewards) {
|
|
|
538
719
|
}));
|
|
539
720
|
return rows.map((r) => JSON.stringify(r)).join("\n") + (rows.length > 0 ? "\n" : "");
|
|
540
721
|
}
|
|
541
|
-
function
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
function isTrainingRunEligible(run, quality, options = {}) {
|
|
545
|
-
const minimumQualityExclusive = options.minimumQualityExclusive ?? 0;
|
|
546
|
-
if (!Number.isFinite(minimumQualityExclusive)) {
|
|
547
|
-
throw new Error("minimumQualityExclusive must be finite");
|
|
548
|
-
}
|
|
549
|
-
if (quality === null || quality === void 0) return false;
|
|
550
|
-
if (!Number.isFinite(quality)) {
|
|
551
|
-
throw new Error(`training quality for run "${run.runId}" must be finite`);
|
|
552
|
-
}
|
|
553
|
-
if (quality <= minimumQualityExclusive) return false;
|
|
554
|
-
if (run.terminalOutcome !== "succeeded") return false;
|
|
555
|
-
if (run.failureClass !== void 0 || run.terminalFailureReason !== void 0) return false;
|
|
556
|
-
if (run.outcome.realness?.gated === true) return false;
|
|
557
|
-
if (run.splitTag === "search") return true;
|
|
558
|
-
return run.splitTag === "holdout" && options.allowHeldOutTrainingData === true;
|
|
722
|
+
function isSelectedSplit(line, options) {
|
|
723
|
+
if (options.splitFilter !== void 0) return options.splitFilter.includes(line.task.split);
|
|
724
|
+
return isSplitEligible(line, options);
|
|
559
725
|
}
|
|
560
726
|
|
|
561
727
|
// src/rl/dataset.ts
|
|
@@ -584,15 +750,8 @@ function validateDatasetFormats(value) {
|
|
|
584
750
|
}
|
|
585
751
|
return formats;
|
|
586
752
|
}
|
|
587
|
-
function reward(r, rewardOf2) {
|
|
588
|
-
const value = rewardOf2 ? rewardOf2(r) : runTaskScore(r) ?? null;
|
|
589
|
-
if (value !== null && !Number.isFinite(value)) {
|
|
590
|
-
throw new Error(`buildRlDataset: reward for run "${r.runId}" must be finite`);
|
|
591
|
-
}
|
|
592
|
-
return value;
|
|
593
|
-
}
|
|
594
753
|
function distinct(xs) {
|
|
595
|
-
return [...new Set(xs)].sort();
|
|
754
|
+
return [...new Set(xs.filter((x) => typeof x === "string" && x.length > 0))].sort();
|
|
596
755
|
}
|
|
597
756
|
function computeRewardStats(values) {
|
|
598
757
|
if (values.length === 0) {
|
|
@@ -606,47 +765,50 @@ function computeRewardStats(values) {
|
|
|
606
765
|
const variance = sorted.reduce((s, x) => s + (x - mean) ** 2, 0) / n;
|
|
607
766
|
return { n, mean, median, min: sorted[0], max: sorted[n - 1], std: Math.sqrt(variance) };
|
|
608
767
|
}
|
|
609
|
-
function
|
|
610
|
-
const splits = { search: 0, dev: 0, holdout: 0 };
|
|
768
|
+
function computeStatsFromLines(lines) {
|
|
769
|
+
const splits = { search: 0, dev: 0, holdout: 0, canary: 0 };
|
|
611
770
|
let inTok = 0;
|
|
612
771
|
let outTok = 0;
|
|
613
772
|
let cost = 0;
|
|
773
|
+
let rolloutsWithoutCost = 0;
|
|
614
774
|
const rewards = [];
|
|
615
|
-
for (const
|
|
616
|
-
splits[
|
|
617
|
-
inTok +=
|
|
618
|
-
outTok +=
|
|
619
|
-
cost
|
|
620
|
-
|
|
775
|
+
for (const line of lines) {
|
|
776
|
+
splits[line.task.split] += 1;
|
|
777
|
+
inTok += line.cost.tokens_in ?? 0;
|
|
778
|
+
outTok += line.cost.tokens_out ?? 0;
|
|
779
|
+
if (line.cost.usd === null) rolloutsWithoutCost++;
|
|
780
|
+
else cost += line.cost.usd;
|
|
781
|
+
const rw = trainableLineReward(line);
|
|
621
782
|
if (rw !== null) rewards.push(rw);
|
|
622
783
|
}
|
|
623
784
|
return {
|
|
624
|
-
records:
|
|
785
|
+
records: lines.length,
|
|
625
786
|
scoredRecords: rewards.length,
|
|
626
787
|
splits,
|
|
627
788
|
reward: computeRewardStats(rewards),
|
|
628
|
-
models: distinct(
|
|
629
|
-
promptHashes: distinct(
|
|
630
|
-
commitShas: distinct(
|
|
789
|
+
models: distinct(lines.map((l) => l.policy.model)),
|
|
790
|
+
promptHashes: distinct(lines.map((l) => l.policy.prompt_hash)),
|
|
791
|
+
commitShas: distinct(lines.map((l) => l.policy.profile_commit)),
|
|
631
792
|
totalTokens: { input: inTok, output: outTok },
|
|
632
|
-
totalCostUsd: cost
|
|
793
|
+
totalCostUsd: cost,
|
|
794
|
+
rolloutsWithoutCost
|
|
633
795
|
};
|
|
634
796
|
}
|
|
635
|
-
async function buildRlDataset(
|
|
636
|
-
if (
|
|
637
|
-
throw new Error("buildRlDataset: no
|
|
797
|
+
async function buildRlDataset(lines, lookups, config, preferences) {
|
|
798
|
+
if (lines.length === 0) {
|
|
799
|
+
throw new Error("buildRlDataset: no rollout lines \u2014 refusing to package an empty dataset");
|
|
638
800
|
}
|
|
639
801
|
const formats = validateDatasetFormats(config.formats === void 0 ? ["sft"] : config.formats);
|
|
640
802
|
const files = {};
|
|
641
803
|
const rowCounts = {};
|
|
642
804
|
if (formats.includes("grpo")) {
|
|
643
|
-
const rows = await toGrpoRows(
|
|
805
|
+
const rows = await toGrpoRows(lines, lookups);
|
|
644
806
|
requireRows("grpo", rows.length);
|
|
645
807
|
files["train.grpo.jsonl"] = toGrpoJsonl(rows);
|
|
646
808
|
rowCounts.grpo = rows.length;
|
|
647
809
|
}
|
|
648
810
|
if (formats.includes("sft")) {
|
|
649
|
-
const rows = await toSftRows(
|
|
811
|
+
const rows = await toSftRows(lines, lookups);
|
|
650
812
|
requireRows("sft", rows.length);
|
|
651
813
|
files["train.sft.jsonl"] = toSftJsonl(rows);
|
|
652
814
|
rowCounts.sft = rows.length;
|
|
@@ -655,7 +817,7 @@ async function buildRlDataset(records, lookups, config, preferences) {
|
|
|
655
817
|
if (!preferences) {
|
|
656
818
|
throw new Error("buildRlDataset: format 'dpo' requires `preferences` (triples + lookups)");
|
|
657
819
|
}
|
|
658
|
-
const rows = await toDpoRows(preferences.triples, preferences.lookups);
|
|
820
|
+
const rows = await toDpoRows(preferences.triples, preferences.lookups, { lines });
|
|
659
821
|
requireRows("dpo", rows.length);
|
|
660
822
|
files["train.dpo.jsonl"] = toDpoJsonl(rows);
|
|
661
823
|
rowCounts.dpo = rows.length;
|
|
@@ -667,7 +829,7 @@ async function buildRlDataset(records, lookups, config, preferences) {
|
|
|
667
829
|
...config,
|
|
668
830
|
formats,
|
|
669
831
|
rowCounts,
|
|
670
|
-
stats:
|
|
832
|
+
stats: computeStatsFromLines(lines)
|
|
671
833
|
};
|
|
672
834
|
files["manifest.json"] = `${JSON.stringify(manifest, null, 2)}
|
|
673
835
|
`;
|
|
@@ -688,7 +850,8 @@ function stat(value) {
|
|
|
688
850
|
function datasheetToMarkdown(m) {
|
|
689
851
|
const s = m.stats;
|
|
690
852
|
const total = s.records || 1;
|
|
691
|
-
const splitLines = ["search", "dev", "holdout"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
|
|
853
|
+
const splitLines = ["search", "dev", "holdout", "canary"].map((k) => ` - \`${k}\`: ${s.splits[k]} (${pct(s.splits[k] / total)})`).join("\n");
|
|
854
|
+
const costNote = s.rolloutsWithoutCost > 0 ? ` (floor \u2014 ${s.rolloutsWithoutCost} rollout(s) never captured a cost)` : "";
|
|
692
855
|
const deterministic = m.reward.kind === "deterministic";
|
|
693
856
|
return [
|
|
694
857
|
`# Dataset: ${m.name} \`v${m.version}\``,
|
|
@@ -714,7 +877,7 @@ function datasheetToMarkdown(m) {
|
|
|
714
877
|
`- **Models:** ${s.models.join(", ")}`,
|
|
715
878
|
`- **Prompt/agent versions (sha256):** ${s.promptHashes.length} distinct`,
|
|
716
879
|
`- **Commits:** ${s.commitShas.join(", ")}`,
|
|
717
|
-
`- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}`,
|
|
880
|
+
`- **Tokens:** ${s.totalTokens.input} in / ${s.totalTokens.output} out | **Cost:** $${s.totalCostUsd.toFixed(2)}${costNote}`,
|
|
718
881
|
"",
|
|
719
882
|
"## Quality gates",
|
|
720
883
|
`- Contamination probe: ${m.qualityGates?.contaminationProbe ?? "not-run"}`,
|
|
@@ -765,7 +928,8 @@ function readCorpus(corpusPath) {
|
|
|
765
928
|
return out;
|
|
766
929
|
}
|
|
767
930
|
function rewardOf(r) {
|
|
768
|
-
|
|
931
|
+
const v = trainingScore(r);
|
|
932
|
+
return typeof v === "number" && Number.isFinite(v) ? v : null;
|
|
769
933
|
}
|
|
770
934
|
async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
|
|
771
935
|
let records = readCorpus(corpusPath).filter(
|
|
@@ -775,8 +939,8 @@ async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
|
|
|
775
939
|
records = records.filter((r) => rewardOf(r) !== null);
|
|
776
940
|
if (opts.minScore != null) {
|
|
777
941
|
records = records.filter((r) => {
|
|
778
|
-
const
|
|
779
|
-
return
|
|
942
|
+
const reward = rewardOf(r);
|
|
943
|
+
return reward !== null && reward >= opts.minScore;
|
|
780
944
|
});
|
|
781
945
|
}
|
|
782
946
|
const text = new Map(
|
|
@@ -787,7 +951,8 @@ async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
|
|
|
787
951
|
completionOf: (id) => text.get(id)?.completion ?? "",
|
|
788
952
|
allowHeldOutTrainingData: opts.allowHeldOutTrainingData
|
|
789
953
|
};
|
|
790
|
-
|
|
954
|
+
const { rows } = await mintRolloutRows(records, new InMemoryTraceStore());
|
|
955
|
+
return buildRlDataset(rows, lookups, config);
|
|
791
956
|
}
|
|
792
957
|
|
|
793
958
|
// src/rl/predictive-validity-researcher.ts
|
|
@@ -897,7 +1062,8 @@ var PredictiveValidityResearcher = class {
|
|
|
897
1062
|
overfitGap: null,
|
|
898
1063
|
baselineOverfitGap: null,
|
|
899
1064
|
medianCandidateCost: null,
|
|
900
|
-
medianBaselineCost: null
|
|
1065
|
+
medianBaselineCost: null,
|
|
1066
|
+
realnessGatedRuns: 0
|
|
901
1067
|
},
|
|
902
1068
|
reason: "predictive-validity researcher does not execute plans; the caller is expected to run the sweep and call rubricPredictiveValidity directly with the resulting RunRecord[].",
|
|
903
1069
|
rejectionCode: "few_runs"
|
|
@@ -938,29 +1104,56 @@ var PredictiveValidityResearcher = class {
|
|
|
938
1104
|
};
|
|
939
1105
|
|
|
940
1106
|
// src/rl/preferences.ts
|
|
941
|
-
var
|
|
942
|
-
|
|
943
|
-
};
|
|
944
|
-
function extractPreferences(runs, opts = {}) {
|
|
1107
|
+
var SPLIT_DEFAULT = "search";
|
|
1108
|
+
function extractPreferences(lines, opts = {}) {
|
|
945
1109
|
const strategy = opts.strategy ?? "paired-by-scenario-and-seed";
|
|
946
1110
|
const minMargin = opts.minMargin ?? 0.05;
|
|
947
|
-
const
|
|
948
|
-
|
|
949
|
-
|
|
1111
|
+
const requestedSplit = opts.split;
|
|
1112
|
+
if (requestedSplit === "holdout" && opts.allowHeldOutTrainingData !== true) {
|
|
1113
|
+
throw new Error('extractPreferences: split "holdout" requires allowHeldOutTrainingData: true');
|
|
1114
|
+
}
|
|
1115
|
+
if (requestedSplit === "dev" || requestedSplit === "canary") {
|
|
950
1116
|
throw new Error(
|
|
951
|
-
|
|
1117
|
+
`extractPreferences: split "${requestedSplit}" is evaluation-only; train from "search"`
|
|
952
1118
|
);
|
|
953
1119
|
}
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
}
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
1120
|
+
const candidates = candidatesFromLines(lines, opts);
|
|
1121
|
+
const report = pairCandidates(candidates.rows, strategy, minMargin);
|
|
1122
|
+
return { ...report, linesWithoutCandidateId: candidates.withoutCandidateId };
|
|
1123
|
+
}
|
|
1124
|
+
function candidatesFromLines(lines, opts) {
|
|
1125
|
+
const split = opts.split ?? SPLIT_DEFAULT;
|
|
1126
|
+
const rows = [];
|
|
1127
|
+
let withoutCandidateId = 0;
|
|
1128
|
+
for (const line of lines) {
|
|
1129
|
+
if (line.task.split !== split) continue;
|
|
1130
|
+
if (!line.outcome.is_completed || line.outcome.is_truncated || line.outcome.error !== null) {
|
|
1131
|
+
continue;
|
|
1132
|
+
}
|
|
1133
|
+
const score = trainableLineReward(line);
|
|
1134
|
+
if (score === null) continue;
|
|
1135
|
+
const candidateId = line.candidate_id;
|
|
1136
|
+
if (candidateId === null || candidateId === void 0 || candidateId.length === 0) {
|
|
1137
|
+
withoutCandidateId++;
|
|
1138
|
+
continue;
|
|
1139
|
+
}
|
|
1140
|
+
rows.push({
|
|
1141
|
+
scenarioId: line.task.instance_id,
|
|
1142
|
+
runId: line.run_id,
|
|
1143
|
+
candidateId,
|
|
1144
|
+
seed: line.task.seed,
|
|
1145
|
+
score,
|
|
1146
|
+
// `policy.*` is nullable on the wire; a minted line always carries these
|
|
1147
|
+
// (RunRecord makes them mandatory). Empty string marks "not recorded" so
|
|
1148
|
+
// `toTRLFormat`'s hash lookup fails visibly instead of silently matching.
|
|
1149
|
+
promptHash: line.policy.prompt_hash ?? "",
|
|
1150
|
+
configHash: line.policy.config_hash ?? "",
|
|
1151
|
+
model: line.policy.model ?? ""
|
|
1152
|
+
});
|
|
963
1153
|
}
|
|
1154
|
+
return { rows, withoutCandidateId };
|
|
1155
|
+
}
|
|
1156
|
+
function pairCandidates(scoredEntries, strategy, minMargin) {
|
|
964
1157
|
const pairs = [];
|
|
965
1158
|
let pairsBelowMargin = 0;
|
|
966
1159
|
let cellsSingleton = 0;
|
|
@@ -968,13 +1161,12 @@ function extractPreferences(runs, opts = {}) {
|
|
|
968
1161
|
if (strategy === "paired-by-scenario-and-seed") {
|
|
969
1162
|
const groups = /* @__PURE__ */ new Map();
|
|
970
1163
|
for (const e of scoredEntries) {
|
|
971
|
-
const
|
|
972
|
-
const key = `${sid}::${e.run.seed}`;
|
|
1164
|
+
const key = `${e.scenarioId}::${e.seed}`;
|
|
973
1165
|
const arr = groups.get(key) ?? [];
|
|
974
1166
|
arr.push(e);
|
|
975
1167
|
groups.set(key, arr);
|
|
976
1168
|
}
|
|
977
|
-
for (const
|
|
1169
|
+
for (const members of groups.values()) {
|
|
978
1170
|
cellsInspected++;
|
|
979
1171
|
if (members.length < 2) {
|
|
980
1172
|
cellsSingleton++;
|
|
@@ -984,8 +1176,8 @@ function extractPreferences(runs, opts = {}) {
|
|
|
984
1176
|
for (let j = i + 1; j < members.length; j++) {
|
|
985
1177
|
const a = members[i];
|
|
986
1178
|
const b = members[j];
|
|
987
|
-
if (a.
|
|
988
|
-
const result = makePair(a, b,
|
|
1179
|
+
if (a.candidateId === b.candidateId) continue;
|
|
1180
|
+
const result = makePair(a, b, a.scenarioId, minMargin);
|
|
989
1181
|
if (result.kind === "admit") pairs.push(result.pair);
|
|
990
1182
|
else pairsBelowMargin++;
|
|
991
1183
|
}
|
|
@@ -994,24 +1186,22 @@ function extractPreferences(runs, opts = {}) {
|
|
|
994
1186
|
} else if (strategy === "paired-by-scenario") {
|
|
995
1187
|
const byScenarioVariant = /* @__PURE__ */ new Map();
|
|
996
1188
|
for (const e of scoredEntries) {
|
|
997
|
-
|
|
998
|
-
let perScenario = byScenarioVariant.get(sid);
|
|
1189
|
+
let perScenario = byScenarioVariant.get(e.scenarioId);
|
|
999
1190
|
if (!perScenario) {
|
|
1000
1191
|
perScenario = /* @__PURE__ */ new Map();
|
|
1001
|
-
byScenarioVariant.set(
|
|
1192
|
+
byScenarioVariant.set(e.scenarioId, perScenario);
|
|
1002
1193
|
}
|
|
1003
|
-
const cur = perScenario.get(e.
|
|
1194
|
+
const cur = perScenario.get(e.candidateId);
|
|
1004
1195
|
if (cur) {
|
|
1005
1196
|
cur.sum += e.score;
|
|
1006
1197
|
cur.n++;
|
|
1007
|
-
} else perScenario.set(e.
|
|
1198
|
+
} else perScenario.set(e.candidateId, { entry: e, sum: e.score, n: 1 });
|
|
1008
1199
|
}
|
|
1009
1200
|
for (const [sid, perVariant] of byScenarioVariant.entries()) {
|
|
1010
1201
|
cellsInspected++;
|
|
1011
|
-
const arr = [...perVariant.
|
|
1012
|
-
|
|
1013
|
-
score: agg.sum / agg.n
|
|
1014
|
-
variantId: vid
|
|
1202
|
+
const arr = [...perVariant.values()].map((agg) => ({
|
|
1203
|
+
...agg.entry,
|
|
1204
|
+
score: agg.sum / agg.n
|
|
1015
1205
|
}));
|
|
1016
1206
|
if (arr.length < 2) {
|
|
1017
1207
|
cellsSingleton++;
|
|
@@ -1028,10 +1218,9 @@ function extractPreferences(runs, opts = {}) {
|
|
|
1028
1218
|
} else {
|
|
1029
1219
|
const byScenario = /* @__PURE__ */ new Map();
|
|
1030
1220
|
for (const e of scoredEntries) {
|
|
1031
|
-
const
|
|
1032
|
-
const arr = byScenario.get(sid) ?? [];
|
|
1221
|
+
const arr = byScenario.get(e.scenarioId) ?? [];
|
|
1033
1222
|
arr.push(e);
|
|
1034
|
-
byScenario.set(
|
|
1223
|
+
byScenario.set(e.scenarioId, arr);
|
|
1035
1224
|
}
|
|
1036
1225
|
for (const [sid, arr] of byScenario.entries()) {
|
|
1037
1226
|
cellsInspected++;
|
|
@@ -1042,7 +1231,7 @@ function extractPreferences(runs, opts = {}) {
|
|
|
1042
1231
|
const sorted = [...arr].sort((a, b) => a.score - b.score);
|
|
1043
1232
|
const top = sorted[sorted.length - 1];
|
|
1044
1233
|
const bot = sorted[0];
|
|
1045
|
-
if (top.
|
|
1234
|
+
if (top.candidateId === bot.candidateId) {
|
|
1046
1235
|
cellsSingleton++;
|
|
1047
1236
|
continue;
|
|
1048
1237
|
}
|
|
@@ -1053,8 +1242,51 @@ function extractPreferences(runs, opts = {}) {
|
|
|
1053
1242
|
}
|
|
1054
1243
|
return { pairs, cellsInspected, pairsBelowMargin, cellsSingleton, strategy };
|
|
1055
1244
|
}
|
|
1056
|
-
|
|
1057
|
-
|
|
1245
|
+
var PREFERENCE_RUN_IDS = (t) => [
|
|
1246
|
+
t.chosenRunId,
|
|
1247
|
+
t.rejectedRunId
|
|
1248
|
+
];
|
|
1249
|
+
var TRL_CONTEXT_REQUIREMENT = {
|
|
1250
|
+
exporter: "TRL preference export",
|
|
1251
|
+
contextType: "RolloutLineContext",
|
|
1252
|
+
because: "a PreferenceTriple carries only run ids and hashes, so without the minted rollout lines this exporter cannot see the realness gate and will put a run that faked its success on the CHOSEN side of a DPO pair."
|
|
1253
|
+
};
|
|
1254
|
+
var ANTHROPIC_CONTEXT_REQUIREMENT = {
|
|
1255
|
+
exporter: "Anthropic preference export",
|
|
1256
|
+
contextType: "RolloutLineContext",
|
|
1257
|
+
because: "a PreferenceTriple carries only run ids and a bare margin, so without the minted rollout lines this exporter cannot see the realness gate and will name a run that faked its success as the preferred one."
|
|
1258
|
+
};
|
|
1259
|
+
async function toTRLFormat(triples, lookups, context) {
|
|
1260
|
+
const admitted = admitUngatedByInvocation(
|
|
1261
|
+
triples,
|
|
1262
|
+
PREFERENCE_RUN_IDS,
|
|
1263
|
+
context,
|
|
1264
|
+
TRL_CONTEXT_REQUIREMENT
|
|
1265
|
+
);
|
|
1266
|
+
const out = [];
|
|
1267
|
+
for (const t of admitted) {
|
|
1268
|
+
const [chosenPrompt, rejectedPrompt, chosen, rejected] = await Promise.all([
|
|
1269
|
+
Promise.resolve(lookups.promptOf(t.chosenRunId)),
|
|
1270
|
+
Promise.resolve(lookups.promptOf(t.rejectedRunId)),
|
|
1271
|
+
Promise.resolve(lookups.completionOf(t.chosenRunId)),
|
|
1272
|
+
Promise.resolve(lookups.completionOf(t.rejectedRunId))
|
|
1273
|
+
]);
|
|
1274
|
+
if (chosenPrompt !== rejectedPrompt) {
|
|
1275
|
+
throw new Error(
|
|
1276
|
+
`toTRLFormat: preference "${t.chosenRunId}"/"${t.rejectedRunId}" resolves to different prompts`
|
|
1277
|
+
);
|
|
1278
|
+
}
|
|
1279
|
+
out.push({ prompt: chosenPrompt, chosen, rejected });
|
|
1280
|
+
}
|
|
1281
|
+
return out;
|
|
1282
|
+
}
|
|
1283
|
+
function toAnthropicFormat(triples, context) {
|
|
1284
|
+
return admitUngatedByInvocation(
|
|
1285
|
+
triples,
|
|
1286
|
+
PREFERENCE_RUN_IDS,
|
|
1287
|
+
context,
|
|
1288
|
+
ANTHROPIC_CONTEXT_REQUIREMENT
|
|
1289
|
+
).map((t) => ({
|
|
1058
1290
|
scenarioId: t.scenarioId,
|
|
1059
1291
|
chosenRunId: t.chosenRunId,
|
|
1060
1292
|
rejectedRunId: t.rejectedRunId,
|
|
@@ -1065,31 +1297,29 @@ function makePair(a, b, scenarioId, minMargin) {
|
|
|
1065
1297
|
const margin = Math.abs(a.score - b.score);
|
|
1066
1298
|
if (margin < minMargin) return { kind: "reject" };
|
|
1067
1299
|
const [chosen, rejected] = a.score > b.score ? [a, b] : [b, a];
|
|
1300
|
+
const seed = chosen.seed !== null && chosen.seed === rejected.seed ? chosen.seed : void 0;
|
|
1068
1301
|
return {
|
|
1069
1302
|
kind: "admit",
|
|
1070
1303
|
pair: {
|
|
1071
1304
|
scenarioId,
|
|
1072
|
-
chosenRunId: chosen.
|
|
1073
|
-
rejectedRunId: rejected.
|
|
1074
|
-
chosenVariantId: chosen.
|
|
1075
|
-
rejectedVariantId: rejected.
|
|
1305
|
+
chosenRunId: chosen.runId,
|
|
1306
|
+
rejectedRunId: rejected.runId,
|
|
1307
|
+
chosenVariantId: chosen.candidateId,
|
|
1308
|
+
rejectedVariantId: rejected.candidateId,
|
|
1076
1309
|
marginScore: chosen.score - rejected.score,
|
|
1077
1310
|
scores: { chosen: chosen.score, rejected: rejected.score },
|
|
1078
|
-
seed
|
|
1311
|
+
seed,
|
|
1079
1312
|
meta: {
|
|
1080
|
-
chosenPromptHash: chosen.
|
|
1081
|
-
rejectedPromptHash: rejected.
|
|
1082
|
-
chosenConfigHash: chosen.
|
|
1083
|
-
rejectedConfigHash: rejected.
|
|
1084
|
-
chosenModel: chosen.
|
|
1085
|
-
rejectedModel: rejected.
|
|
1313
|
+
chosenPromptHash: chosen.promptHash,
|
|
1314
|
+
rejectedPromptHash: rejected.promptHash,
|
|
1315
|
+
chosenConfigHash: chosen.configHash,
|
|
1316
|
+
rejectedConfigHash: rejected.configHash,
|
|
1317
|
+
chosenModel: chosen.model,
|
|
1318
|
+
rejectedModel: rejected.model
|
|
1086
1319
|
}
|
|
1087
1320
|
}
|
|
1088
1321
|
};
|
|
1089
1322
|
}
|
|
1090
|
-
function scenarioOf(run) {
|
|
1091
|
-
return run.scenarioId;
|
|
1092
|
-
}
|
|
1093
1323
|
|
|
1094
1324
|
// src/rl/process-reward.ts
|
|
1095
1325
|
async function extractStepRewards(store, runId, opts) {
|
|
@@ -1224,11 +1454,13 @@ async function runRLCampaign(opts) {
|
|
|
1224
1454
|
campaign.runs,
|
|
1225
1455
|
opts.verifiableReward ?? {}
|
|
1226
1456
|
);
|
|
1227
|
-
const
|
|
1457
|
+
const scoredRuns = campaign.runs.filter((run) => runTaskScore(run) !== void 0);
|
|
1458
|
+
const { rows: rolloutLines } = await mintRolloutRows(scoredRuns, new InMemoryTraceStore());
|
|
1459
|
+
const preferences = extractPreferences(rolloutLines, {
|
|
1228
1460
|
...opts.preferences,
|
|
1229
1461
|
strategy: opts.preferences?.strategy ?? "paired-by-scenario-and-seed",
|
|
1230
1462
|
minMargin: opts.preferences?.minMargin ?? 0.05,
|
|
1231
|
-
|
|
1463
|
+
split: opts.preferences?.split ?? splitTag
|
|
1232
1464
|
});
|
|
1233
1465
|
let interimConfidence = null;
|
|
1234
1466
|
if (opts.report?.comparator) {
|
|
@@ -1257,13 +1489,15 @@ async function runRLCampaign(opts) {
|
|
|
1257
1489
|
}
|
|
1258
1490
|
const trainerRows = {};
|
|
1259
1491
|
if (opts.trainerExport?.dpo) {
|
|
1260
|
-
trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo
|
|
1492
|
+
trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, {
|
|
1493
|
+
lines: rolloutLines
|
|
1494
|
+
});
|
|
1261
1495
|
}
|
|
1262
1496
|
if (opts.trainerExport?.grpo) {
|
|
1263
|
-
trainerRows.grpo = await toGrpoRows(
|
|
1497
|
+
trainerRows.grpo = await toGrpoRows(rolloutLines, opts.trainerExport.grpo);
|
|
1264
1498
|
}
|
|
1265
1499
|
if (opts.trainerExport?.sft) {
|
|
1266
|
-
trainerRows.sft = await toSftRows(
|
|
1500
|
+
trainerRows.sft = await toSftRows(rolloutLines, opts.trainerExport.sft);
|
|
1267
1501
|
}
|
|
1268
1502
|
const summary = buildSummary({
|
|
1269
1503
|
campaign,
|
|
@@ -1438,7 +1672,13 @@ var defaultBehaviorFeatures = (record) => {
|
|
|
1438
1672
|
const turnsAborted = finiteOrNull(raw.turns_aborted);
|
|
1439
1673
|
const completion = record.completion;
|
|
1440
1674
|
return {
|
|
1441
|
-
|
|
1675
|
+
// RAW (`observedSplitScore`), deliberately: this feature vector is one
|
|
1676
|
+
// half of a sim-vs-production divergence measurement. Gating a gamed run to
|
|
1677
|
+
// 0 would move the simulated distribution toward production and report the
|
|
1678
|
+
// simulator as MORE faithful precisely where it is being gamed. Each split
|
|
1679
|
+
// is read separately rather than through `observedScore` so a non-finite
|
|
1680
|
+
// holdout score falls back to search instead of poisoning the bucket.
|
|
1681
|
+
score: finiteOrNull(observedSplitScore(record, "holdout")) ?? finiteOrNull(observedSplitScore(record, "search")),
|
|
1442
1682
|
failure_class: record.failureClass ?? null,
|
|
1443
1683
|
wall_ms: finiteOrNull(record.wallMs),
|
|
1444
1684
|
output_tokens: finiteOrNull(record.tokenUsage?.output),
|
|
@@ -1620,7 +1860,7 @@ function easyModeCheck(simulated, production, opts = {}) {
|
|
|
1620
1860
|
const passRate = (records, side) => {
|
|
1621
1861
|
let passes = 0;
|
|
1622
1862
|
for (const r of records) {
|
|
1623
|
-
const score = finiteOrNull(r
|
|
1863
|
+
const score = finiteOrNull(observedSplitScore(r, "holdout")) ?? finiteOrNull(observedSplitScore(r, "search"));
|
|
1624
1864
|
if (score === null) {
|
|
1625
1865
|
throw new ValidationError(
|
|
1626
1866
|
`easyModeCheck: ${side} run "${r.runId}" carries neither holdoutScore nor searchScore`
|
|
@@ -1769,12 +2009,16 @@ export {
|
|
|
1769
2009
|
ABSENT_CATEGORY,
|
|
1770
2010
|
DEFAULT_MIN_N_PER_FEATURE,
|
|
1771
2011
|
DEFAULT_QUANTILE_BUCKETS,
|
|
2012
|
+
DPO_CONTEXT_REQUIREMENT,
|
|
1772
2013
|
FileSystemOutcomeStore,
|
|
1773
2014
|
InMemoryOutcomeStore,
|
|
2015
|
+
PRM_CONTEXT_REQUIREMENT,
|
|
1774
2016
|
PredictiveValidityResearcher,
|
|
1775
2017
|
REPRESENTATIVE_MIN_FIDELITY,
|
|
2018
|
+
STEP_REWARD_CONTEXT_REQUIREMENT,
|
|
1776
2019
|
appendToCorpus,
|
|
1777
2020
|
applyEloUpdate,
|
|
2021
|
+
assertPrmTrainableLine,
|
|
1778
2022
|
bestOfN,
|
|
1779
2023
|
bucketLabel,
|
|
1780
2024
|
buildDatasetFromCorpus,
|
|
@@ -1796,7 +2040,6 @@ export {
|
|
|
1796
2040
|
fitBradleyTerry,
|
|
1797
2041
|
injectIrrelevantClause,
|
|
1798
2042
|
inverseProbabilityWeighting,
|
|
1799
|
-
isTrainingRunEligible,
|
|
1800
2043
|
jsDivergence,
|
|
1801
2044
|
observationsFromRunRecords,
|
|
1802
2045
|
offPolicyEstimateAll,
|
|
@@ -1826,6 +2069,7 @@ export {
|
|
|
1826
2069
|
toPrmRows,
|
|
1827
2070
|
toSftJsonl,
|
|
1828
2071
|
toSftRows,
|
|
2072
|
+
toTRLFormat,
|
|
1829
2073
|
validateDatasetFormats,
|
|
1830
2074
|
varianceBasedCurriculum,
|
|
1831
2075
|
verificationReportToRunRecord
|