@tangle-network/agent-bench 0.3.5 → 0.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/README.md +13 -1
- package/dist/adapters.d.ts +15 -0
- package/dist/adapters.js +43 -0
- package/dist/adapters.js.map +1 -0
- package/dist/benchmarks/_harness.d.ts +125 -0
- package/dist/benchmarks/_harness.js +33 -0
- package/dist/benchmarks/_harness.js.map +1 -0
- package/dist/benchmarks/aec-bench.d.ts +27 -0
- package/dist/benchmarks/aec-bench.js +8 -0
- package/dist/benchmarks/aec-bench.js.map +1 -0
- package/dist/benchmarks/agentbench.d.ts +16 -0
- package/dist/benchmarks/agentbench.js +10 -0
- package/dist/benchmarks/agentbench.js.map +1 -0
- package/dist/benchmarks/appworld.d.ts +37 -0
- package/dist/benchmarks/appworld.js +14 -0
- package/dist/benchmarks/appworld.js.map +1 -0
- package/dist/benchmarks/bfcl.d.ts +18 -0
- package/dist/benchmarks/bfcl.js +10 -0
- package/dist/benchmarks/bfcl.js.map +1 -0
- package/dist/benchmarks/cad-design.d.ts +45 -0
- package/dist/benchmarks/cad-design.js +7 -0
- package/dist/benchmarks/cad-design.js.map +1 -0
- package/dist/benchmarks/cadbench.d.ts +19 -0
- package/dist/benchmarks/cadbench.js +8 -0
- package/dist/benchmarks/cadbench.js.map +1 -0
- package/dist/benchmarks/cadgenbench.d.ts +22 -0
- package/dist/benchmarks/cadgenbench.js +8 -0
- package/dist/benchmarks/cadgenbench.js.map +1 -0
- package/dist/benchmarks/commit0.d.ts +31 -0
- package/dist/benchmarks/commit0.js +10 -0
- package/dist/benchmarks/commit0.js.map +1 -0
- package/dist/benchmarks/crag.d.ts +14 -0
- package/dist/benchmarks/crag.js +9 -0
- package/dist/benchmarks/crag.js.map +1 -0
- package/dist/benchmarks/dabstep.d.ts +18 -0
- package/dist/benchmarks/dabstep.js +10 -0
- package/dist/benchmarks/dabstep.js.map +1 -0
- package/dist/benchmarks/enterpriseops-gym.d.ts +38 -0
- package/dist/benchmarks/enterpriseops-gym.js +10 -0
- package/dist/benchmarks/enterpriseops-gym.js.map +1 -0
- package/dist/benchmarks/finresearchbench.d.ts +15 -0
- package/dist/benchmarks/finresearchbench.js +8 -0
- package/dist/benchmarks/finresearchbench.js.map +1 -0
- package/dist/benchmarks/finsearchcomp.d.ts +49 -0
- package/dist/benchmarks/finsearchcomp.js +7 -0
- package/dist/benchmarks/finsearchcomp.js.map +1 -0
- package/dist/benchmarks/frames.d.ts +59 -0
- package/dist/benchmarks/frames.js +13 -0
- package/dist/benchmarks/frames.js.map +1 -0
- package/dist/benchmarks/hotpotqa.d.ts +48 -0
- package/dist/benchmarks/hotpotqa.js +15 -0
- package/dist/benchmarks/hotpotqa.js.map +1 -0
- package/dist/benchmarks/humaneval.d.ts +62 -0
- package/dist/benchmarks/humaneval.js +17 -0
- package/dist/benchmarks/humaneval.js.map +1 -0
- package/dist/benchmarks/mind2web.d.ts +41 -0
- package/dist/benchmarks/mind2web.js +9 -0
- package/dist/benchmarks/mind2web.js.map +1 -0
- package/dist/benchmarks/nomiracl.d.ts +15 -0
- package/dist/benchmarks/nomiracl.js +9 -0
- package/dist/benchmarks/nomiracl.js.map +1 -0
- package/dist/benchmarks/open-rag-bench.d.ts +14 -0
- package/dist/benchmarks/open-rag-bench.js +9 -0
- package/dist/benchmarks/open-rag-bench.js.map +1 -0
- package/dist/benchmarks/programbench.d.ts +38 -0
- package/dist/benchmarks/programbench.js +10 -0
- package/dist/benchmarks/programbench.js.map +1 -0
- package/dist/benchmarks/rag-shared.d.ts +42 -0
- package/dist/benchmarks/rag-shared.js +39 -0
- package/dist/benchmarks/rag-shared.js.map +1 -0
- package/dist/benchmarks/ragbench.d.ts +16 -0
- package/dist/benchmarks/ragbench.js +9 -0
- package/dist/benchmarks/ragbench.js.map +1 -0
- package/dist/benchmarks/simpleqa.d.ts +64 -0
- package/dist/benchmarks/simpleqa.js +11 -0
- package/dist/benchmarks/simpleqa.js.map +1 -0
- package/dist/benchmarks/swe-bench.d.ts +56 -0
- package/dist/benchmarks/swe-bench.js +14 -0
- package/dist/benchmarks/swe-bench.js.map +1 -0
- package/dist/benchmarks/t2-ragbench.d.ts +14 -0
- package/dist/benchmarks/t2-ragbench.js +9 -0
- package/dist/benchmarks/t2-ragbench.js.map +1 -0
- package/dist/benchmarks/tau-bench-shared.d.ts +26 -0
- package/dist/benchmarks/tau-bench-shared.js +10 -0
- package/dist/benchmarks/tau-bench-shared.js.map +1 -0
- package/dist/benchmarks/tau2-bench.d.ts +7 -0
- package/dist/benchmarks/tau2-bench.js +11 -0
- package/dist/benchmarks/tau2-bench.js.map +1 -0
- package/dist/benchmarks/tau3-banking.d.ts +15 -0
- package/dist/benchmarks/tau3-banking.js +9 -0
- package/dist/benchmarks/tau3-banking.js.map +1 -0
- package/dist/benchmarks/terminal-bench.d.ts +24 -0
- package/dist/benchmarks/terminal-bench.js +8 -0
- package/dist/benchmarks/terminal-bench.js.map +1 -0
- package/dist/benchmarks/toollm.d.ts +16 -0
- package/dist/benchmarks/toollm.js +10 -0
- package/dist/benchmarks/toollm.js.map +1 -0
- package/dist/benchmarks/trata-hedge.d.ts +32 -0
- package/dist/benchmarks/trata-hedge.js +7 -0
- package/dist/benchmarks/trata-hedge.js.map +1 -0
- package/dist/benchmarks/types.d.ts +107 -0
- package/dist/benchmarks/types.js +1 -0
- package/dist/benchmarks/types.js.map +1 -0
- package/dist/benchmarks/webarena-verified.d.ts +16 -0
- package/dist/benchmarks/webarena-verified.js +10 -0
- package/dist/benchmarks/webarena-verified.js.map +1 -0
- package/dist/chunk-2PVVP7GN.js +197 -0
- package/dist/chunk-2PVVP7GN.js.map +1 -0
- package/dist/chunk-2XU6OGEN.js +170 -0
- package/dist/chunk-2XU6OGEN.js.map +1 -0
- package/dist/chunk-53UPUNBZ.js +325 -0
- package/dist/chunk-53UPUNBZ.js.map +1 -0
- package/dist/chunk-5H5XV76F.js +240 -0
- package/dist/chunk-5H5XV76F.js.map +1 -0
- package/dist/chunk-7WSD27QQ.js +118 -0
- package/dist/chunk-7WSD27QQ.js.map +1 -0
- package/dist/chunk-C7T7WEK2.js +103 -0
- package/dist/chunk-C7T7WEK2.js.map +1 -0
- package/dist/chunk-CKUVRZ2T.js +251 -0
- package/dist/chunk-CKUVRZ2T.js.map +1 -0
- package/dist/chunk-HBSWHQNJ.js +30 -0
- package/dist/chunk-HBSWHQNJ.js.map +1 -0
- package/dist/chunk-HHXFIHXC.js +116 -0
- package/dist/chunk-HHXFIHXC.js.map +1 -0
- package/dist/chunk-IFAV6KEM.js +276 -0
- package/dist/chunk-IFAV6KEM.js.map +1 -0
- package/dist/chunk-INNOYXCP.js +387 -0
- package/dist/chunk-INNOYXCP.js.map +1 -0
- package/dist/chunk-J3KDJNX2.js +182 -0
- package/dist/chunk-J3KDJNX2.js.map +1 -0
- package/dist/chunk-JRWWGMK7.js +148 -0
- package/dist/chunk-JRWWGMK7.js.map +1 -0
- package/dist/chunk-JTHWEDEW.js +32 -0
- package/dist/chunk-JTHWEDEW.js.map +1 -0
- package/dist/chunk-KDIKRJGB.js +120 -0
- package/dist/chunk-KDIKRJGB.js.map +1 -0
- package/dist/chunk-LRRD7NAG.js +301 -0
- package/dist/chunk-LRRD7NAG.js.map +1 -0
- package/dist/chunk-ODT47UAY.js +221 -0
- package/dist/chunk-ODT47UAY.js.map +1 -0
- package/dist/chunk-PA2ZKHJC.js +230 -0
- package/dist/chunk-PA2ZKHJC.js.map +1 -0
- package/dist/chunk-PUIRNYI7.js +189 -0
- package/dist/chunk-PUIRNYI7.js.map +1 -0
- package/dist/chunk-PWQVGAJB.js +144 -0
- package/dist/chunk-PWQVGAJB.js.map +1 -0
- package/dist/chunk-R36V2VP7.js +169 -0
- package/dist/chunk-R36V2VP7.js.map +1 -0
- package/dist/chunk-R67DFVLO.js +142 -0
- package/dist/chunk-R67DFVLO.js.map +1 -0
- package/dist/chunk-SEVJPLZC.js +260 -0
- package/dist/chunk-SEVJPLZC.js.map +1 -0
- package/dist/chunk-SYDW647C.js +318 -0
- package/dist/chunk-SYDW647C.js.map +1 -0
- package/dist/chunk-TBKU5XQI.js +228 -0
- package/dist/chunk-TBKU5XQI.js.map +1 -0
- package/dist/chunk-TSWPNOYM.js +147 -0
- package/dist/chunk-TSWPNOYM.js.map +1 -0
- package/dist/chunk-UAIOHCUK.js +27 -0
- package/dist/chunk-UAIOHCUK.js.map +1 -0
- package/dist/chunk-UPAMRDX4.js +233 -0
- package/dist/chunk-UPAMRDX4.js.map +1 -0
- package/dist/chunk-VQRS7VUC.js +342 -0
- package/dist/chunk-VQRS7VUC.js.map +1 -0
- package/dist/chunk-X3BTXCJ4.js +262 -0
- package/dist/chunk-X3BTXCJ4.js.map +1 -0
- package/dist/chunk-X5YKXC6V.js +211 -0
- package/dist/chunk-X5YKXC6V.js.map +1 -0
- package/dist/chunk-Y6O2OCUO.js +130 -0
- package/dist/chunk-Y6O2OCUO.js.map +1 -0
- package/dist/chunk-YCGY7UIZ.js +208 -0
- package/dist/chunk-YCGY7UIZ.js.map +1 -0
- package/dist/chunk-Z7ML6L77.js +162 -0
- package/dist/chunk-Z7ML6L77.js.map +1 -0
- package/dist/chunk-ZEWMTR5M.js +136 -0
- package/dist/chunk-ZEWMTR5M.js.map +1 -0
- package/dist/index.d.ts +355 -0
- package/dist/index.js +1908 -0
- package/dist/index.js.map +1 -0
- package/package.json +26 -9
- package/scripts/run-package-tests.mjs +30 -8
- package/scripts/verify-packed-consumer.mjs +12 -1
- package/scripts/verify-pier-agent.mts +1 -0
- package/src/benchmarks/humaneval.test.mts +122 -0
- package/src/benchmarks/humaneval.ts +100 -27
- package/src/david-attribution.mts +78 -0
- package/src/david-goliath.mts +149 -0
- package/src/hev-improve.mts +25 -6
- package/src/humaneval-object-ablation.mts +201 -0
- package/src/live-improve-campaign-mbpp.mts +641 -0
- package/src/live-improve-campaign.mts +500 -0
- package/src/mbpp-structural.mts +12 -7
- package/src/stream-observe.py +45 -0
- package/src/stream-observe.tpl.html +247 -0
- package/src/supervisor-arena.mts +816 -0
- package/src/swe-arena/analyze.ts +211 -0
- package/src/swe-arena/arms.ts +788 -0
- package/src/swe-arena/bootstrap-meta.mts +188 -0
- package/src/swe-arena/bootstrap-meta.test.mts +51 -0
- package/src/swe-arena/calibrate.ts +116 -0
- package/src/swe-arena/capabilities.mts +76 -0
- package/src/swe-arena/capabilities.test.mts +57 -0
- package/src/swe-arena/capacity.ts +194 -0
- package/src/swe-arena/cell-evidence.mts +405 -0
- package/src/swe-arena/cell-evidence.test.mts +248 -0
- package/src/swe-arena/diagnosis-ensemble.test.mts +210 -0
- package/src/swe-arena/diagnosis-ensemble.ts +520 -0
- package/src/swe-arena/execution.test.mts +1170 -0
- package/src/swe-arena/fixtures/analyze.py +80 -0
- package/src/swe-arena/fixtures/excludes.txt +8 -0
- package/src/swe-arena/fixtures/gen1-salvage/README.md +45 -0
- package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +116 -0
- package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +293 -0
- package/src/swe-arena/fixtures/holdout-preregister.log +12 -0
- package/src/swe-arena/fixtures/holdout.json +44 -0
- package/src/swe-arena/fixtures/instances.json +146 -0
- package/src/swe-arena/fixtures/ledger.jsonl +12 -0
- package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +36 -0
- package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +33 -0
- package/src/swe-arena/fixtures/rejudge.jsonl +15 -0
- package/src/swe-arena/fixtures/rematch.jsonl +3 -0
- package/src/swe-arena/fixtures/rematch2.jsonl +3 -0
- package/src/swe-arena/fixtures/rematch3.jsonl +3 -0
- package/src/swe-arena/fixtures/sup-journal-true.json +19 -0
- package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +48 -0
- package/src/swe-arena/fixtures/verify/django__django-11532.sh +50 -0
- package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +76 -0
- package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +44 -0
- package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +32 -0
- package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +51 -0
- package/src/swe-arena/fixtures/worker-tokens.json +42 -0
- package/src/swe-arena/fixtures.ts +104 -0
- package/src/swe-arena/holdout-certify.mts +408 -0
- package/src/swe-arena/holdout-certify.test.mts +160 -0
- package/src/swe-arena/judge-child.mts +37 -0
- package/src/swe-arena/manifest.mts +293 -0
- package/src/swe-arena/manifest.test.mts +169 -0
- package/src/swe-arena/materialize.ts +142 -0
- package/src/swe-arena/outer-loop.mts +2145 -0
- package/src/swe-arena/outer-loop.test.mts +696 -0
- package/src/swe-arena/parity.test.mts +87 -0
- package/src/swe-arena/proc.test.mts +174 -0
- package/src/swe-arena/proc.ts +260 -0
- package/src/swe-arena/profiles/default-author.profile.json +4 -0
- package/src/swe-arena/proposer-fanout.mts +489 -0
- package/src/swe-arena/proposer-fanout.test.mts +372 -0
- package/src/swe-arena/reconcile.ts +0 -0
- package/src/swe-arena/replay.mts +183 -0
- package/src/swe-arena/replay.test.mts +300 -0
- package/src/swe-arena/run-experiment.mts +361 -0
- package/src/swe-arena/run-supervisor.mjs +297 -0
- package/src/swe-arena/run-supervisor.test.mts +498 -0
- package/src/swe-arena/serialized-judge.ts +414 -0
- package/src/swe-arena/types.ts +166 -0
- package/src/swe-code-improve.mts +328 -0
- package/src/swe-emit-patch.mts +104 -0
- package/src/swe-improve.mts +232 -0
- package/src/swe-jail.ts +2 -2
- package/src/swe-local-proof.mts +169 -0
- package/src/swe-repro-calibrate.mts +446 -0
- package/src/swe-stream.mts +1497 -0
|
@@ -0,0 +1,405 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cell-derived scoring evidence — the round's ground truth read from the
|
|
3
|
+
* LIB's campaign cells, never from in-process dispatch-order bookkeeping.
|
|
4
|
+
*
|
|
5
|
+
* Root cause this kills (r4-mroh3rkt): `improve()` resumes its campaign from
|
|
6
|
+
* runDir and replays cached cells WITHOUT dispatching them, so any recorder
|
|
7
|
+
* keyed on "what this process dispatched" mislabels arms — the resumed run
|
|
8
|
+
* published candidate b08d31c910's cells as "baseline 0/3" while the measured
|
|
9
|
+
* baseline was 1/3. Campaign cells carry their own attribution instead:
|
|
10
|
+
*
|
|
11
|
+
* - the campaign DIRECTORY names the arm (`baseline/` vs
|
|
12
|
+
* `gen-<g>/candidate-<i>/` under the improve runDir — run-campaign.ts
|
|
13
|
+
* writes one `<cellId>/cached-result.json` per conclusive cell), and
|
|
14
|
+
* - each cell's artifact names its loops commit (`R4Artifact.commit`).
|
|
15
|
+
*
|
|
16
|
+
* Everything here is pure over cells (plus the two disk readers), so the
|
|
17
|
+
* aggregation is unit-testable against a synthetic cell set reproducing the
|
|
18
|
+
* resume-replay shape with zero dispatch.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { readdir, readFile } from 'node:fs/promises'
|
|
22
|
+
import { join } from 'node:path'
|
|
23
|
+
|
|
24
|
+
// ---------------------------------------------------------------------------
|
|
25
|
+
// The evaluated artifact. One cell = one (surface × scenario × rep); every
|
|
26
|
+
// cell carries the official-judge outcome + recovered spend. (The `kind`
|
|
27
|
+
// discriminant stays: cached cells on disk carry it, and it keeps replayed
|
|
28
|
+
// artifacts distinguishable from a null/errored cell.)
|
|
29
|
+
// ---------------------------------------------------------------------------
|
|
30
|
+
|
|
31
|
+
export interface R4Artifact {
|
|
32
|
+
kind: 'swe-arm'
|
|
33
|
+
iid: string
|
|
34
|
+
commit: string
|
|
35
|
+
resolved: boolean
|
|
36
|
+
verifyPass: boolean
|
|
37
|
+
patchLines: number
|
|
38
|
+
wallS: number
|
|
39
|
+
/** Runtime spend-tree total (state.json `result.spentTokens`, winner AND
|
|
40
|
+
* no-winner arms). `null` = state.json unreadable, a telemetry gap. */
|
|
41
|
+
spentTokens: number | null
|
|
42
|
+
spentUsd: number | null
|
|
43
|
+
/** spentTokens + opencode-sqlite worker-session tokens. */
|
|
44
|
+
recoveredTokens: number | null
|
|
45
|
+
/** Worker-session token split from the opencode sqlite join — the
|
|
46
|
+
* usage the campaign CostLedger receipt reports. */
|
|
47
|
+
workerTokIn: number | null
|
|
48
|
+
workerTokOut: number | null
|
|
49
|
+
judgeAttempts: number | null
|
|
50
|
+
judgeWallS: number | null
|
|
51
|
+
runDir: string
|
|
52
|
+
patchPath: string
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** The minimal slice of a lib `CampaignCellResult<R4Artifact>` the scoring
|
|
56
|
+
* reads. Structural so both in-memory campaign results and parsed
|
|
57
|
+
* `cached-result.json` files satisfy it. */
|
|
58
|
+
export interface EvidenceCell {
|
|
59
|
+
scenarioId: string
|
|
60
|
+
rep: number
|
|
61
|
+
/** `null` on an errored cell (the lib records failed cells with a null
|
|
62
|
+
* artifact; it never caches them). */
|
|
63
|
+
artifact: R4Artifact | null
|
|
64
|
+
error?: string
|
|
65
|
+
costUsd?: number
|
|
66
|
+
tokenUsage?: { input: number; output: number }
|
|
67
|
+
cached?: boolean
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Adapt a lib campaign's cells (in-memory result) to `EvidenceCell`s. */
|
|
71
|
+
export function cellsFromCampaign(campaign: {
|
|
72
|
+
cells: Array<{
|
|
73
|
+
scenarioId: string
|
|
74
|
+
rep: number
|
|
75
|
+
artifact: unknown
|
|
76
|
+
error?: string
|
|
77
|
+
costUsd: number
|
|
78
|
+
tokenUsage: { input: number; output: number }
|
|
79
|
+
cached: boolean
|
|
80
|
+
}>
|
|
81
|
+
}): EvidenceCell[] {
|
|
82
|
+
return campaign.cells.map((cell) => ({
|
|
83
|
+
scenarioId: cell.scenarioId,
|
|
84
|
+
rep: cell.rep,
|
|
85
|
+
artifact: (cell.artifact ?? null) as R4Artifact | null,
|
|
86
|
+
...(cell.error ? { error: cell.error } : {}),
|
|
87
|
+
costUsd: cell.costUsd,
|
|
88
|
+
tokenUsage: { input: cell.tokenUsage.input, output: cell.tokenUsage.output },
|
|
89
|
+
cached: cell.cached,
|
|
90
|
+
}))
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// ---------------------------------------------------------------------------
|
|
94
|
+
// Replicate semantics — repsPerInstance. Single-rep scoring provably flips
|
|
95
|
+
// instance outcomes run-to-run (judge flake + capacity noise both observed),
|
|
96
|
+
// so an instance counts RESOLVED only when EVERY replicate cell resolved (AND
|
|
97
|
+
// — fail-closed for keep-if-better), and coverage requires every replicate of
|
|
98
|
+
// every instance to hold a real boolean verdict.
|
|
99
|
+
// ---------------------------------------------------------------------------
|
|
100
|
+
|
|
101
|
+
export interface ReplicateRun {
|
|
102
|
+
iid: string
|
|
103
|
+
resolved: boolean | null
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** Instances where ALL `reps` replicates resolved (missing replicates never count). */
|
|
107
|
+
export function resolvedInstanceCount(runs: ReplicateRun[], iids: string[], reps: number): number {
|
|
108
|
+
let count = 0
|
|
109
|
+
for (const iid of iids) {
|
|
110
|
+
const mine = runs.filter((r) => r.iid === iid)
|
|
111
|
+
if (mine.length === reps && mine.every((r) => r.resolved === true)) count += 1
|
|
112
|
+
}
|
|
113
|
+
return count
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** Every instance has exactly `reps` replicates, each with a conclusive verdict. */
|
|
117
|
+
export function replicateCoverageComplete(runs: ReplicateRun[], iids: string[], reps: number): boolean {
|
|
118
|
+
return iids.every((iid) => {
|
|
119
|
+
const mine = runs.filter((r) => r.iid === iid)
|
|
120
|
+
return mine.length === reps && mine.every((r) => r.resolved !== null)
|
|
121
|
+
})
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** One `ReplicateRun` per swe cell. An errored/artifact-less cell is an
|
|
125
|
+
* inconclusive replicate (`resolved: null`) — never a fabricated boolean. */
|
|
126
|
+
export function replicateRunsFromCells(cells: EvidenceCell[]): ReplicateRun[] {
|
|
127
|
+
return cells
|
|
128
|
+
.filter((c) => c.artifact === null || c.artifact.kind === 'swe-arm')
|
|
129
|
+
.map((c) => ({
|
|
130
|
+
iid: c.scenarioId,
|
|
131
|
+
resolved: c.artifact !== null && c.artifact.kind === 'swe-arm' && !c.error ? c.artifact.resolved : null,
|
|
132
|
+
}))
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** Σ wall seconds across the swe cells (errored cells contribute 0). */
|
|
136
|
+
export function sumWallSFromCells(cells: EvidenceCell[]): number {
|
|
137
|
+
return cells.reduce(
|
|
138
|
+
(s, c) => s + (c.artifact !== null && c.artifact.kind === 'swe-arm' ? c.artifact.wallS : 0),
|
|
139
|
+
0,
|
|
140
|
+
)
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** Per-replicate staircase row — one per swe cell, straight off the artifact. */
|
|
144
|
+
export interface StaircasePerInstance {
|
|
145
|
+
iid: string
|
|
146
|
+
/** Replicate index (0-based) — repsPerInstance cells per instance. */
|
|
147
|
+
rep: number
|
|
148
|
+
resolved: boolean | null
|
|
149
|
+
verify_pass: boolean | null
|
|
150
|
+
patch_lines: number | null
|
|
151
|
+
wall_s: number | null
|
|
152
|
+
spentTokens: number | null
|
|
153
|
+
recoveredTokens: number | null
|
|
154
|
+
judgeAttempts: number | null
|
|
155
|
+
/** Campaign-cell CostLedger spend for this replicate (worker receipt). */
|
|
156
|
+
costUsd: number | null
|
|
157
|
+
error?: string
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export function perInstanceFromCells(cells: EvidenceCell[]): StaircasePerInstance[] {
|
|
161
|
+
const rows: StaircasePerInstance[] = []
|
|
162
|
+
for (const cell of cells) {
|
|
163
|
+
const a = cell.artifact !== null && cell.artifact.kind === 'swe-arm' && !cell.error ? cell.artifact : null
|
|
164
|
+
rows.push({
|
|
165
|
+
iid: cell.scenarioId,
|
|
166
|
+
rep: cell.rep,
|
|
167
|
+
resolved: a ? a.resolved : null,
|
|
168
|
+
verify_pass: a ? a.verifyPass : null,
|
|
169
|
+
patch_lines: a ? a.patchLines : null,
|
|
170
|
+
wall_s: a ? a.wallS : null,
|
|
171
|
+
spentTokens: a ? a.spentTokens : null,
|
|
172
|
+
recoveredTokens: a ? a.recoveredTokens : null,
|
|
173
|
+
judgeAttempts: a ? a.judgeAttempts : null,
|
|
174
|
+
costUsd: cell.costUsd ?? null,
|
|
175
|
+
...(cell.error ? { error: cell.error } : {}),
|
|
176
|
+
})
|
|
177
|
+
}
|
|
178
|
+
return rows
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// ---------------------------------------------------------------------------
|
|
182
|
+
// Premeasured-baseline drift. The gate's denominator is the stored
|
|
183
|
+
// premeasured baseline artifact ({surfaceHash, campaign}) that the LIB
|
|
184
|
+
// validates before skipping the baseline campaign — surface hash, seed, reps,
|
|
185
|
+
// and split digest all fail loud on mismatch. A resumed runDir can still hold
|
|
186
|
+
// baseline cells cached by an OLDER run of the same surface; when those
|
|
187
|
+
// contradict the validated artifact, the contradiction is logged loud and the
|
|
188
|
+
// artifact still rules.
|
|
189
|
+
// ---------------------------------------------------------------------------
|
|
190
|
+
|
|
191
|
+
/** AND-verdict per instance from campaign cells. Only instances with full,
|
|
192
|
+
* conclusive replicate coverage produce a verdict — a partial record has no
|
|
193
|
+
* AND-verdict to compare. */
|
|
194
|
+
export function instanceVerdictsFromCells(
|
|
195
|
+
cells: EvidenceCell[],
|
|
196
|
+
iids: string[],
|
|
197
|
+
reps: number,
|
|
198
|
+
): Record<string, boolean> {
|
|
199
|
+
const runs = replicateRunsFromCells(cells)
|
|
200
|
+
const verdicts: Record<string, boolean> = {}
|
|
201
|
+
for (const iid of iids) {
|
|
202
|
+
const mine = runs.filter((r) => r.iid === iid)
|
|
203
|
+
if (mine.length !== reps || mine.some((r) => r.resolved === null)) continue
|
|
204
|
+
verdicts[iid] = mine.every((r) => r.resolved === true)
|
|
205
|
+
}
|
|
206
|
+
return verdicts
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/** Per-instance contradictions between the validated premeasured artifact's
|
|
210
|
+
* verdicts and locally cached baseline cells. Only instances with full-reps,
|
|
211
|
+
* conclusive coverage on BOTH sides are compared — a partial record has no
|
|
212
|
+
* AND-verdict to contradict with. The caller logs these loud and the
|
|
213
|
+
* premeasured artifact STILL rules. */
|
|
214
|
+
export function baselineDriftWarnings(
|
|
215
|
+
expected: Record<string, boolean>,
|
|
216
|
+
runs: ReplicateRun[],
|
|
217
|
+
iids: string[],
|
|
218
|
+
reps: number,
|
|
219
|
+
): string[] {
|
|
220
|
+
const warnings: string[] = []
|
|
221
|
+
for (const iid of iids) {
|
|
222
|
+
const want = expected[iid]
|
|
223
|
+
if (typeof want !== 'boolean') continue
|
|
224
|
+
const mine = runs.filter((r) => r.iid === iid)
|
|
225
|
+
if (mine.length !== reps || mine.some((r) => r.resolved === null)) continue
|
|
226
|
+
const measured = mine.every((r) => r.resolved === true)
|
|
227
|
+
if (measured !== want) {
|
|
228
|
+
warnings.push(
|
|
229
|
+
`${iid}: premeasured=${want} but cached baseline cells measured ${measured} ` +
|
|
230
|
+
`(reps: ${mine.map((r) => String(r.resolved)).join('/')}) — the validated premeasured artifact rules`,
|
|
231
|
+
)
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
return warnings
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// ---------------------------------------------------------------------------
|
|
238
|
+
// protocol_v2 keep-if-better.
|
|
239
|
+
// ---------------------------------------------------------------------------
|
|
240
|
+
|
|
241
|
+
export type StaircaseVerdict =
|
|
242
|
+
| 'accepted'
|
|
243
|
+
| 'rejected-no-gain'
|
|
244
|
+
| 'rejected-cost'
|
|
245
|
+
| 'rejected-out-of-space'
|
|
246
|
+
| 'rejected-incomplete'
|
|
247
|
+
/** Killed by the gen-3 pre-filter (change-space/tsc/smoke) BEFORE any full
|
|
248
|
+
* evaluation — the candidate never became a measured surface. Emitted by
|
|
249
|
+
* the outer loop's kill-row writer, never by `decideVerdict`. */
|
|
250
|
+
| 'rejected-prefilter'
|
|
251
|
+
|
|
252
|
+
/** protocol_v2 keep-if-better: improvement-set resolved-count must RISE and
|
|
253
|
+
* cost must stay within the guard. Fail-closed on unprovable cost. */
|
|
254
|
+
export function decideVerdict(input: {
|
|
255
|
+
violations: string[]
|
|
256
|
+
coverageComplete: boolean
|
|
257
|
+
resolvedCount: number
|
|
258
|
+
parentResolvedCount: number
|
|
259
|
+
costRatio: number | null
|
|
260
|
+
costGuardRatio: number
|
|
261
|
+
}): StaircaseVerdict {
|
|
262
|
+
if (input.violations.length > 0) return 'rejected-out-of-space'
|
|
263
|
+
if (!input.coverageComplete) return 'rejected-incomplete'
|
|
264
|
+
if (input.resolvedCount <= input.parentResolvedCount) return 'rejected-no-gain'
|
|
265
|
+
if (input.costRatio === null || input.costRatio > input.costGuardRatio) return 'rejected-cost'
|
|
266
|
+
return 'accepted'
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
// ---------------------------------------------------------------------------
|
|
270
|
+
// Disk readers — the lib's per-cell caches. run-campaign.ts writes
|
|
271
|
+
// `<campaignDir>/<sanitized cellId>/cached-result.json` for every conclusive
|
|
272
|
+
// cell (errored cells are never cached — a missing replicate reads as
|
|
273
|
+
// coverage-incomplete downstream, fail-closed).
|
|
274
|
+
// ---------------------------------------------------------------------------
|
|
275
|
+
|
|
276
|
+
/** Parse every `<cellDir>/cached-result.json` under one campaign dir. Missing dir = []. */
|
|
277
|
+
export async function loadCampaignCells(campaignDir: string): Promise<EvidenceCell[]> {
|
|
278
|
+
const entries = await readdir(campaignDir, { withFileTypes: true }).catch(() => [])
|
|
279
|
+
const cells: EvidenceCell[] = []
|
|
280
|
+
for (const entry of entries) {
|
|
281
|
+
if (!entry.isDirectory()) continue
|
|
282
|
+
const path = join(campaignDir, entry.name, 'cached-result.json')
|
|
283
|
+
const raw = await readFile(path, 'utf8').catch(() => null)
|
|
284
|
+
if (raw === null) continue
|
|
285
|
+
let parsed: Record<string, unknown>
|
|
286
|
+
try {
|
|
287
|
+
parsed = JSON.parse(raw) as Record<string, unknown>
|
|
288
|
+
} catch {
|
|
289
|
+
throw new Error(`loadCampaignCells: corrupt cell cache ${path}`)
|
|
290
|
+
}
|
|
291
|
+
if (typeof parsed.scenarioId !== 'string' || typeof parsed.rep !== 'number') {
|
|
292
|
+
throw new Error(`loadCampaignCells: ${path} is not a campaign cell (scenarioId/rep missing)`)
|
|
293
|
+
}
|
|
294
|
+
cells.push({
|
|
295
|
+
scenarioId: parsed.scenarioId,
|
|
296
|
+
rep: parsed.rep,
|
|
297
|
+
artifact: (parsed.artifact ?? null) as R4Artifact | null,
|
|
298
|
+
...(typeof parsed.error === 'string' ? { error: parsed.error } : {}),
|
|
299
|
+
...(typeof parsed.costUsd === 'number' ? { costUsd: parsed.costUsd } : {}),
|
|
300
|
+
...(parsed.tokenUsage && typeof parsed.tokenUsage === 'object'
|
|
301
|
+
? { tokenUsage: parsed.tokenUsage as { input: number; output: number } }
|
|
302
|
+
: {}),
|
|
303
|
+
cached: true,
|
|
304
|
+
})
|
|
305
|
+
}
|
|
306
|
+
return cells
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
export interface CandidateCellGroup {
|
|
310
|
+
generation: number
|
|
311
|
+
candidateIndex: number
|
|
312
|
+
dir: string
|
|
313
|
+
cells: EvidenceCell[]
|
|
314
|
+
/** The loops commit the cells' artifacts name (null when no artifact
|
|
315
|
+
* carries one — e.g. an all-errored, never-cached candidate). */
|
|
316
|
+
commit: string | null
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/** Scan `gen-<g>/candidate-<i>/` campaign dirs under the improve runDir.
|
|
320
|
+
* Attribution is directory + artifact-commit — dispatch order plays no part. */
|
|
321
|
+
export async function loadCandidateCellGroups(improveRunDir: string): Promise<CandidateCellGroup[]> {
|
|
322
|
+
const groups: CandidateCellGroup[] = []
|
|
323
|
+
const top = await readdir(improveRunDir, { withFileTypes: true }).catch(() => [])
|
|
324
|
+
for (const genEntry of top) {
|
|
325
|
+
const genMatch = /^gen-(\d+)$/.exec(genEntry.name)
|
|
326
|
+
if (!genEntry.isDirectory() || !genMatch) continue
|
|
327
|
+
const genDir = join(improveRunDir, genEntry.name)
|
|
328
|
+
for (const candEntry of await readdir(genDir, { withFileTypes: true }).catch(() => [])) {
|
|
329
|
+
const candMatch = /^candidate-(\d+)$/.exec(candEntry.name)
|
|
330
|
+
if (!candEntry.isDirectory() || !candMatch) continue
|
|
331
|
+
const dir = join(genDir, candEntry.name)
|
|
332
|
+
const cells = await loadCampaignCells(dir)
|
|
333
|
+
const commits = new Set(
|
|
334
|
+
cells.map((c) => c.artifact?.commit).filter((c): c is string => typeof c === 'string'),
|
|
335
|
+
)
|
|
336
|
+
if (commits.size > 1) {
|
|
337
|
+
throw new Error(
|
|
338
|
+
`loadCandidateCellGroups: ${dir} mixes commits [${[...commits].join(', ')}] — one candidate dir must hold one surface`,
|
|
339
|
+
)
|
|
340
|
+
}
|
|
341
|
+
groups.push({
|
|
342
|
+
generation: Number(genMatch[1]),
|
|
343
|
+
candidateIndex: Number(candMatch[1]),
|
|
344
|
+
dir,
|
|
345
|
+
cells,
|
|
346
|
+
commit: [...commits][0] ?? null,
|
|
347
|
+
})
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
return groups.sort((a, b) => a.generation - b.generation || a.candidateIndex - b.candidateIndex)
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
// ---------------------------------------------------------------------------
|
|
354
|
+
// Gate evidence — the would-be-keep operator brief, derived from cells. The
|
|
355
|
+
// lib's deferred-holdout gate always holds; this evidence tells the operator
|
|
356
|
+
// whether the pre-registered holdout run is worth approving.
|
|
357
|
+
// ---------------------------------------------------------------------------
|
|
358
|
+
|
|
359
|
+
export interface GateEvidence {
|
|
360
|
+
candResolved: number
|
|
361
|
+
baseResolved: number
|
|
362
|
+
candWallS: number
|
|
363
|
+
baseWallS: number
|
|
364
|
+
costRatio: number | null
|
|
365
|
+
coverageComplete: boolean
|
|
366
|
+
verdict: StaircaseVerdict
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
/** Score the winner-vs-baseline comparison for the operator brief. Both sides
|
|
370
|
+
* come from campaign cells — the baseline side is the lib-validated
|
|
371
|
+
* premeasured campaign (or the bootstrap run's freshly measured one). */
|
|
372
|
+
export function gateEvidenceFromCells(input: {
|
|
373
|
+
winnerCells: EvidenceCell[]
|
|
374
|
+
baselineCells: EvidenceCell[]
|
|
375
|
+
/** Dispatch-time change-space violations of the winner's diff. */
|
|
376
|
+
violations: string[]
|
|
377
|
+
iids: string[]
|
|
378
|
+
reps: number
|
|
379
|
+
costGuardRatio: number
|
|
380
|
+
}): GateEvidence {
|
|
381
|
+
const winnerRuns = replicateRunsFromCells(input.winnerCells)
|
|
382
|
+
const baselineRuns = replicateRunsFromCells(input.baselineCells)
|
|
383
|
+
const candResolved = resolvedInstanceCount(winnerRuns, input.iids, input.reps)
|
|
384
|
+
const baseResolved = resolvedInstanceCount(baselineRuns, input.iids, input.reps)
|
|
385
|
+
const candWallS = sumWallSFromCells(input.winnerCells)
|
|
386
|
+
const baseWallS = sumWallSFromCells(input.baselineCells)
|
|
387
|
+
const costRatio = baseWallS > 0 ? candWallS / baseWallS : null
|
|
388
|
+
const coverageComplete = replicateCoverageComplete(winnerRuns, input.iids, input.reps)
|
|
389
|
+
return {
|
|
390
|
+
candResolved,
|
|
391
|
+
baseResolved,
|
|
392
|
+
candWallS,
|
|
393
|
+
baseWallS,
|
|
394
|
+
costRatio,
|
|
395
|
+
coverageComplete,
|
|
396
|
+
verdict: decideVerdict({
|
|
397
|
+
violations: input.violations,
|
|
398
|
+
coverageComplete,
|
|
399
|
+
resolvedCount: candResolved,
|
|
400
|
+
parentResolvedCount: baseResolved,
|
|
401
|
+
costRatio,
|
|
402
|
+
costGuardRatio: input.costGuardRatio,
|
|
403
|
+
}),
|
|
404
|
+
}
|
|
405
|
+
}
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cell-derived scoring: reps-AND aggregation as a pure function over lib
|
|
3
|
+
* campaign cells, and the disk readers over the lib's cached-result.json
|
|
4
|
+
* caches — including a reproduction of the r4-mroh3rkt resume shape (baseline
|
|
5
|
+
* cells replayed from cache, only a candidate dispatched in-process) proving
|
|
6
|
+
* attribution comes from the campaign directory + artifact commit, never from
|
|
7
|
+
* dispatch order. No arms, no docker, no tokens.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
|
|
11
|
+
import { tmpdir } from 'node:os'
|
|
12
|
+
import { join } from 'node:path'
|
|
13
|
+
import { afterAll, describe, expect, it } from 'vitest'
|
|
14
|
+
import {
|
|
15
|
+
baselineDriftWarnings,
|
|
16
|
+
cellsFromCampaign,
|
|
17
|
+
gateEvidenceFromCells,
|
|
18
|
+
instanceVerdictsFromCells,
|
|
19
|
+
loadCampaignCells,
|
|
20
|
+
loadCandidateCellGroups,
|
|
21
|
+
perInstanceFromCells,
|
|
22
|
+
replicateRunsFromCells,
|
|
23
|
+
resolvedInstanceCount,
|
|
24
|
+
sumWallSFromCells,
|
|
25
|
+
type EvidenceCell,
|
|
26
|
+
type R4Artifact,
|
|
27
|
+
} from './cell-evidence.mts'
|
|
28
|
+
|
|
29
|
+
const IIDS = ['astropy__astropy-13033', 'django__django-11532', 'matplotlib__matplotlib-20826']
|
|
30
|
+
|
|
31
|
+
function sweArtifact(iid: string, commit: string, resolved: boolean, wallS = 100): R4Artifact {
|
|
32
|
+
return {
|
|
33
|
+
kind: 'swe-arm',
|
|
34
|
+
iid,
|
|
35
|
+
commit,
|
|
36
|
+
resolved,
|
|
37
|
+
verifyPass: resolved,
|
|
38
|
+
patchLines: 10,
|
|
39
|
+
wallS,
|
|
40
|
+
spentTokens: 1000,
|
|
41
|
+
spentUsd: 0.01,
|
|
42
|
+
recoveredTokens: 1500,
|
|
43
|
+
workerTokIn: 400,
|
|
44
|
+
workerTokOut: 100,
|
|
45
|
+
judgeAttempts: 1,
|
|
46
|
+
judgeWallS: 30,
|
|
47
|
+
runDir: `/tmp/none/${iid}`,
|
|
48
|
+
patchPath: `/tmp/none/${iid}.patch`,
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function cell(iid: string, rep: number, artifact: R4Artifact | null, error?: string): EvidenceCell {
|
|
53
|
+
return {
|
|
54
|
+
scenarioId: iid,
|
|
55
|
+
rep,
|
|
56
|
+
artifact,
|
|
57
|
+
...(error ? { error } : {}),
|
|
58
|
+
costUsd: 0.01,
|
|
59
|
+
tokenUsage: { input: 400, output: 100 },
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
describe('cell adapters', () => {
|
|
64
|
+
it('cellsFromCampaign keeps errored cells with a null artifact', () => {
|
|
65
|
+
const cells = cellsFromCampaign({
|
|
66
|
+
cells: [
|
|
67
|
+
{ scenarioId: 'a', rep: 0, artifact: sweArtifact('a', 'c1', true), costUsd: 0.5, tokenUsage: { input: 1, output: 2 }, cached: false },
|
|
68
|
+
{ scenarioId: 'a', rep: 1, artifact: null, error: 'dispatch timeout', costUsd: 0, tokenUsage: { input: 0, output: 0 }, cached: false },
|
|
69
|
+
],
|
|
70
|
+
})
|
|
71
|
+
expect(cells).toHaveLength(2)
|
|
72
|
+
expect(cells[0]!.artifact?.kind).toBe('swe-arm')
|
|
73
|
+
expect(cells[1]!.artifact).toBeNull()
|
|
74
|
+
expect(cells[1]!.error).toMatch(/timeout/)
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
it('replicateRunsFromCells: error/artifact-less cells are inconclusive, never a boolean', () => {
|
|
78
|
+
const runs = replicateRunsFromCells([
|
|
79
|
+
cell('a', 0, sweArtifact('a', 'c1', true)),
|
|
80
|
+
cell('a', 1, sweArtifact('a', 'c1', false), 'judge inconclusive'),
|
|
81
|
+
cell('b', 0, null, 'change-space violation'),
|
|
82
|
+
])
|
|
83
|
+
expect(runs).toEqual([
|
|
84
|
+
{ iid: 'a', resolved: true },
|
|
85
|
+
{ iid: 'a', resolved: null },
|
|
86
|
+
{ iid: 'b', resolved: null },
|
|
87
|
+
])
|
|
88
|
+
})
|
|
89
|
+
|
|
90
|
+
it('perInstanceFromCells maps artifact fields and carries cell cost', () => {
|
|
91
|
+
const rows = perInstanceFromCells([
|
|
92
|
+
cell('a', 0, sweArtifact('a', 'c1', true, 250)),
|
|
93
|
+
cell('b', 1, null, 'boom'),
|
|
94
|
+
])
|
|
95
|
+
expect(rows).toHaveLength(2)
|
|
96
|
+
expect(rows[0]).toMatchObject({ iid: 'a', rep: 0, resolved: true, wall_s: 250, costUsd: 0.01, judgeAttempts: 1 })
|
|
97
|
+
expect(rows[1]).toMatchObject({ iid: 'b', rep: 1, resolved: null, wall_s: null, error: 'boom' })
|
|
98
|
+
})
|
|
99
|
+
|
|
100
|
+
it('sumWallSFromCells sums swe wall only', () => {
|
|
101
|
+
expect(
|
|
102
|
+
sumWallSFromCells([
|
|
103
|
+
cell('a', 0, sweArtifact('a', 'c1', true, 100)),
|
|
104
|
+
cell('a', 1, sweArtifact('a', 'c1', true, 40)),
|
|
105
|
+
cell('b', 0, null, 'err'),
|
|
106
|
+
]),
|
|
107
|
+
).toBe(140)
|
|
108
|
+
})
|
|
109
|
+
})
|
|
110
|
+
|
|
111
|
+
describe('disk readers + the r4-mroh3rkt resume shape', () => {
|
|
112
|
+
const roots: string[] = []
|
|
113
|
+
afterAll(async () => {
|
|
114
|
+
for (const root of roots) await rm(root, { recursive: true, force: true })
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
async function writeCachedCell(campaignDir: string, c: EvidenceCell): Promise<void> {
|
|
118
|
+
const cellDir = join(campaignDir, `cell-${c.scenarioId.replace(/[^a-zA-Z0-9_-]/g, '_')}-r${c.rep}`)
|
|
119
|
+
await mkdir(cellDir, { recursive: true })
|
|
120
|
+
await writeFile(
|
|
121
|
+
join(cellDir, 'cached-result.json'),
|
|
122
|
+
JSON.stringify({
|
|
123
|
+
cellId: `cell-${c.scenarioId}-r${c.rep}`,
|
|
124
|
+
scenarioId: c.scenarioId,
|
|
125
|
+
rep: c.rep,
|
|
126
|
+
artifact: c.artifact,
|
|
127
|
+
...(c.error ? { error: c.error } : {}),
|
|
128
|
+
costUsd: c.costUsd ?? 0,
|
|
129
|
+
tokenUsage: c.tokenUsage ?? { input: 0, output: 0 },
|
|
130
|
+
judgeScores: {},
|
|
131
|
+
durationMs: 1,
|
|
132
|
+
seed: 42,
|
|
133
|
+
cached: false,
|
|
134
|
+
}),
|
|
135
|
+
)
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** The exact resume shape behind r4-mroh3rkt: the BASELINE campaign exists
|
|
139
|
+
* only as cached cells on disk (measured 1/3 — matplotlib both reps), while
|
|
140
|
+
* the process only ever dispatched candidate b08d31c910 (0/3). */
|
|
141
|
+
async function makeResumedRunDir(): Promise<string> {
|
|
142
|
+
const root = await mkdtemp(join(tmpdir(), 'r4-cells-'))
|
|
143
|
+
roots.push(root)
|
|
144
|
+
const improveRun = join(root, 'improve-run')
|
|
145
|
+
const baselineDir = join(improveRun, 'baseline')
|
|
146
|
+
for (const iid of IIDS) {
|
|
147
|
+
const resolved = iid.startsWith('matplotlib')
|
|
148
|
+
for (const rep of [0, 1]) {
|
|
149
|
+
await writeCachedCell(baselineDir, cell(iid, rep, sweArtifact(iid, 'basecommit0', resolved, 100)))
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
const candDir = join(improveRun, 'gen-0', 'candidate-0')
|
|
153
|
+
for (const iid of IIDS) {
|
|
154
|
+
for (const rep of [0, 1]) {
|
|
155
|
+
await writeCachedCell(candDir, cell(iid, rep, sweArtifact(iid, 'b08d31c910', false, 110)))
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
return improveRun
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
it('loadCampaignCells reads every cached cell; a missing dir is empty', async () => {
|
|
162
|
+
const improveRun = await makeResumedRunDir()
|
|
163
|
+
const cells = await loadCampaignCells(join(improveRun, 'baseline'))
|
|
164
|
+
expect(cells).toHaveLength(6)
|
|
165
|
+
expect(cells.every((c) => c.cached)).toBe(true)
|
|
166
|
+
expect(await loadCampaignCells(join(improveRun, 'no-such-campaign'))).toEqual([])
|
|
167
|
+
})
|
|
168
|
+
|
|
169
|
+
it('candidate groups attribute by directory + commit — baseline cells never leak in', async () => {
|
|
170
|
+
const improveRun = await makeResumedRunDir()
|
|
171
|
+
const groups = await loadCandidateCellGroups(improveRun)
|
|
172
|
+
expect(groups).toHaveLength(1)
|
|
173
|
+
expect(groups[0]).toMatchObject({ generation: 0, candidateIndex: 0, commit: 'b08d31c910' })
|
|
174
|
+
expect(groups[0]!.cells).toHaveLength(6)
|
|
175
|
+
expect(groups[0]!.cells.every((c) => c.artifact?.commit === 'b08d31c910')).toBe(true)
|
|
176
|
+
})
|
|
177
|
+
|
|
178
|
+
it('r4-mroh3rkt regression: the resumed run grades winner 0/3 vs baseline 1/3, not 0/3 vs 0/3', async () => {
|
|
179
|
+
const improveRun = await makeResumedRunDir()
|
|
180
|
+
const groups = await loadCandidateCellGroups(improveRun)
|
|
181
|
+
const winner = groups.find((g) => g.commit === 'b08d31c910')!
|
|
182
|
+
const baselineCells = await loadCampaignCells(join(improveRun, 'baseline'))
|
|
183
|
+
|
|
184
|
+
// Attribution is campaign directory + artifact commit: the candidate's
|
|
185
|
+
// cells can never masquerade as the baseline, dispatched or replayed.
|
|
186
|
+
const ev = gateEvidenceFromCells({
|
|
187
|
+
winnerCells: winner.cells,
|
|
188
|
+
baselineCells,
|
|
189
|
+
violations: [],
|
|
190
|
+
iids: IIDS,
|
|
191
|
+
reps: 2,
|
|
192
|
+
costGuardRatio: 1.2,
|
|
193
|
+
})
|
|
194
|
+
expect(ev.candResolved).toBe(0)
|
|
195
|
+
expect(ev.baseResolved).toBe(1) // the bug published 0/3 here
|
|
196
|
+
expect(ev.verdict).toBe('rejected-no-gain')
|
|
197
|
+
expect(ev.coverageComplete).toBe(true)
|
|
198
|
+
expect(ev.costRatio).toBeCloseTo(660 / 600, 5)
|
|
199
|
+
})
|
|
200
|
+
|
|
201
|
+
it('a contradicting cached baseline raises drift warnings and the premeasured artifact rules', async () => {
|
|
202
|
+
const improveRun = await makeResumedRunDir()
|
|
203
|
+
const cachedBaseline = await loadCampaignCells(join(improveRun, 'baseline'))
|
|
204
|
+
// Premeasured artifact verdicts contradicting the cached cells on astropy.
|
|
205
|
+
const expected = {
|
|
206
|
+
'astropy__astropy-13033': true, // cached cells measured F/F
|
|
207
|
+
'django__django-11532': false,
|
|
208
|
+
'matplotlib__matplotlib-20826': true,
|
|
209
|
+
}
|
|
210
|
+
const warnings = baselineDriftWarnings(expected, replicateRunsFromCells(cachedBaseline), IIDS, 2)
|
|
211
|
+
expect(warnings).toHaveLength(1)
|
|
212
|
+
expect(warnings[0]).toContain('astropy__astropy-13033: premeasured=true')
|
|
213
|
+
expect(warnings[0]).toContain('premeasured artifact rules')
|
|
214
|
+
// The AND-verdicts of the cached campaign (the drift comparator source).
|
|
215
|
+
expect(instanceVerdictsFromCells(cachedBaseline, IIDS, 2)).toEqual({
|
|
216
|
+
'astropy__astropy-13033': false,
|
|
217
|
+
'django__django-11532': false,
|
|
218
|
+
'matplotlib__matplotlib-20826': true,
|
|
219
|
+
})
|
|
220
|
+
})
|
|
221
|
+
|
|
222
|
+
it('a missing replicate keeps the candidate coverage-incomplete (errored cells are never cached)', async () => {
|
|
223
|
+
const improveRun = await makeResumedRunDir()
|
|
224
|
+
const partialDir = join(improveRun, 'gen-0', 'candidate-1')
|
|
225
|
+
await writeCachedCell(partialDir, cell(IIDS[0]!, 0, sweArtifact(IIDS[0]!, 'deadbeef01', true)))
|
|
226
|
+
const groups = await loadCandidateCellGroups(improveRun)
|
|
227
|
+
const partial = groups.find((g) => g.commit === 'deadbeef01')!
|
|
228
|
+
const ev = gateEvidenceFromCells({
|
|
229
|
+
winnerCells: partial.cells,
|
|
230
|
+
baselineCells: await loadCampaignCells(join(improveRun, 'baseline')),
|
|
231
|
+
violations: [],
|
|
232
|
+
iids: IIDS,
|
|
233
|
+
reps: 2,
|
|
234
|
+
costGuardRatio: 1.2,
|
|
235
|
+
})
|
|
236
|
+
expect(ev.coverageComplete).toBe(false)
|
|
237
|
+
expect(ev.verdict).toBe('rejected-incomplete')
|
|
238
|
+
expect(resolvedInstanceCount(replicateRunsFromCells(partial.cells), IIDS, 2)).toBe(0)
|
|
239
|
+
})
|
|
240
|
+
|
|
241
|
+
it('a candidate dir mixing two commits fails loud', async () => {
|
|
242
|
+
const improveRun = await makeResumedRunDir()
|
|
243
|
+
const dir = join(improveRun, 'gen-1', 'candidate-0')
|
|
244
|
+
await writeCachedCell(dir, cell(IIDS[0]!, 0, sweArtifact(IIDS[0]!, 'commitaaaa1', true)))
|
|
245
|
+
await writeCachedCell(dir, cell(IIDS[1]!, 0, sweArtifact(IIDS[1]!, 'commitbbbb2', true)))
|
|
246
|
+
await expect(loadCandidateCellGroups(improveRun)).rejects.toThrow(/mixes commits/)
|
|
247
|
+
})
|
|
248
|
+
})
|