@tangle-network/agent-bench 0.3.6 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/CHANGELOG.md +7 -0
  2. package/dist/adapters.js +2 -2
  3. package/dist/benchmarks/humaneval.d.ts +10 -1
  4. package/dist/benchmarks/humaneval.js +5 -3
  5. package/dist/{chunk-PPYSEKFM.js → chunk-5H5XV76F.js} +73 -15
  6. package/dist/chunk-5H5XV76F.js.map +1 -0
  7. package/dist/{chunk-5SBJCB6W.js → chunk-PWQVGAJB.js} +2 -2
  8. package/dist/index.js +2 -2
  9. package/package.json +5 -4
  10. package/scripts/run-package-tests.mjs +30 -8
  11. package/scripts/verify-pier-agent.mts +1 -0
  12. package/src/benchmarks/humaneval.test.mts +122 -0
  13. package/src/benchmarks/humaneval.ts +100 -27
  14. package/src/david-attribution.mts +78 -0
  15. package/src/david-goliath.mts +149 -0
  16. package/src/hev-improve.mts +25 -6
  17. package/src/humaneval-object-ablation.mts +201 -0
  18. package/src/live-improve-campaign-mbpp.mts +641 -0
  19. package/src/live-improve-campaign.mts +500 -0
  20. package/src/mbpp-structural.mts +12 -7
  21. package/src/stream-observe.py +45 -0
  22. package/src/stream-observe.tpl.html +247 -0
  23. package/src/supervisor-arena.mts +816 -0
  24. package/src/swe-arena/analyze.ts +211 -0
  25. package/src/swe-arena/arms.ts +788 -0
  26. package/src/swe-arena/bootstrap-meta.mts +188 -0
  27. package/src/swe-arena/bootstrap-meta.test.mts +51 -0
  28. package/src/swe-arena/calibrate.ts +116 -0
  29. package/src/swe-arena/capabilities.mts +76 -0
  30. package/src/swe-arena/capabilities.test.mts +57 -0
  31. package/src/swe-arena/capacity.ts +194 -0
  32. package/src/swe-arena/cell-evidence.mts +405 -0
  33. package/src/swe-arena/cell-evidence.test.mts +248 -0
  34. package/src/swe-arena/diagnosis-ensemble.test.mts +210 -0
  35. package/src/swe-arena/diagnosis-ensemble.ts +520 -0
  36. package/src/swe-arena/execution.test.mts +1170 -0
  37. package/src/swe-arena/fixtures/analyze.py +80 -0
  38. package/src/swe-arena/fixtures/excludes.txt +8 -0
  39. package/src/swe-arena/fixtures/gen1-salvage/README.md +45 -0
  40. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +116 -0
  41. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +293 -0
  42. package/src/swe-arena/fixtures/holdout-preregister.log +12 -0
  43. package/src/swe-arena/fixtures/holdout.json +44 -0
  44. package/src/swe-arena/fixtures/instances.json +146 -0
  45. package/src/swe-arena/fixtures/ledger.jsonl +12 -0
  46. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +36 -0
  47. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +33 -0
  48. package/src/swe-arena/fixtures/rejudge.jsonl +15 -0
  49. package/src/swe-arena/fixtures/rematch.jsonl +3 -0
  50. package/src/swe-arena/fixtures/rematch2.jsonl +3 -0
  51. package/src/swe-arena/fixtures/rematch3.jsonl +3 -0
  52. package/src/swe-arena/fixtures/sup-journal-true.json +19 -0
  53. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +48 -0
  54. package/src/swe-arena/fixtures/verify/django__django-11532.sh +50 -0
  55. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +76 -0
  56. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +44 -0
  57. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +32 -0
  58. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +51 -0
  59. package/src/swe-arena/fixtures/worker-tokens.json +42 -0
  60. package/src/swe-arena/fixtures.ts +104 -0
  61. package/src/swe-arena/holdout-certify.mts +408 -0
  62. package/src/swe-arena/holdout-certify.test.mts +160 -0
  63. package/src/swe-arena/judge-child.mts +37 -0
  64. package/src/swe-arena/manifest.mts +293 -0
  65. package/src/swe-arena/manifest.test.mts +169 -0
  66. package/src/swe-arena/materialize.ts +142 -0
  67. package/src/swe-arena/outer-loop.mts +2145 -0
  68. package/src/swe-arena/outer-loop.test.mts +696 -0
  69. package/src/swe-arena/parity.test.mts +87 -0
  70. package/src/swe-arena/proc.test.mts +174 -0
  71. package/src/swe-arena/proc.ts +260 -0
  72. package/src/swe-arena/profiles/default-author.profile.json +4 -0
  73. package/src/swe-arena/proposer-fanout.mts +489 -0
  74. package/src/swe-arena/proposer-fanout.test.mts +372 -0
  75. package/src/swe-arena/reconcile.ts +0 -0
  76. package/src/swe-arena/replay.mts +183 -0
  77. package/src/swe-arena/replay.test.mts +300 -0
  78. package/src/swe-arena/run-experiment.mts +361 -0
  79. package/src/swe-arena/run-supervisor.mjs +297 -0
  80. package/src/swe-arena/run-supervisor.test.mts +498 -0
  81. package/src/swe-arena/serialized-judge.ts +414 -0
  82. package/src/swe-arena/types.ts +166 -0
  83. package/src/swe-code-improve.mts +328 -0
  84. package/src/swe-emit-patch.mts +104 -0
  85. package/src/swe-improve.mts +232 -0
  86. package/src/swe-jail.ts +2 -2
  87. package/src/swe-local-proof.mts +169 -0
  88. package/src/swe-repro-calibrate.mts +446 -0
  89. package/src/swe-stream.mts +1497 -0
  90. package/dist/chunk-PPYSEKFM.js.map +0 -1
  91. /package/dist/{chunk-5SBJCB6W.js.map → chunk-PWQVGAJB.js.map} +0 -0
@@ -0,0 +1,405 @@
1
+ /**
2
+ * Cell-derived scoring evidence — the round's ground truth read from the
3
+ * LIB's campaign cells, never from in-process dispatch-order bookkeeping.
4
+ *
5
+ * Root cause this kills (r4-mroh3rkt): `improve()` resumes its campaign from
6
+ * runDir and replays cached cells WITHOUT dispatching them, so any recorder
7
+ * keyed on "what this process dispatched" mislabels arms — the resumed run
8
+ * published candidate b08d31c910's cells as "baseline 0/3" while the measured
9
+ * baseline was 1/3. Campaign cells carry their own attribution instead:
10
+ *
11
+ * - the campaign DIRECTORY names the arm (`baseline/` vs
12
+ * `gen-<g>/candidate-<i>/` under the improve runDir — run-campaign.ts
13
+ * writes one `<cellId>/cached-result.json` per conclusive cell), and
14
+ * - each cell's artifact names its loops commit (`R4Artifact.commit`).
15
+ *
16
+ * Everything here is pure over cells (plus the two disk readers), so the
17
+ * aggregation is unit-testable against a synthetic cell set reproducing the
18
+ * resume-replay shape with zero dispatch.
19
+ */
20
+
21
+ import { readdir, readFile } from 'node:fs/promises'
22
+ import { join } from 'node:path'
23
+
24
+ // ---------------------------------------------------------------------------
25
+ // The evaluated artifact. One cell = one (surface × scenario × rep); every
26
+ // cell carries the official-judge outcome + recovered spend. (The `kind`
27
+ // discriminant stays: cached cells on disk carry it, and it keeps replayed
28
+ // artifacts distinguishable from a null/errored cell.)
29
+ // ---------------------------------------------------------------------------
30
+
31
+ export interface R4Artifact {
32
+ kind: 'swe-arm'
33
+ iid: string
34
+ commit: string
35
+ resolved: boolean
36
+ verifyPass: boolean
37
+ patchLines: number
38
+ wallS: number
39
+ /** Runtime spend-tree total (state.json `result.spentTokens`, winner AND
40
+ * no-winner arms). `null` = state.json unreadable, a telemetry gap. */
41
+ spentTokens: number | null
42
+ spentUsd: number | null
43
+ /** spentTokens + opencode-sqlite worker-session tokens. */
44
+ recoveredTokens: number | null
45
+ /** Worker-session token split from the opencode sqlite join — the
46
+ * usage the campaign CostLedger receipt reports. */
47
+ workerTokIn: number | null
48
+ workerTokOut: number | null
49
+ judgeAttempts: number | null
50
+ judgeWallS: number | null
51
+ runDir: string
52
+ patchPath: string
53
+ }
54
+
55
+ /** The minimal slice of a lib `CampaignCellResult<R4Artifact>` the scoring
56
+ * reads. Structural so both in-memory campaign results and parsed
57
+ * `cached-result.json` files satisfy it. */
58
+ export interface EvidenceCell {
59
+ scenarioId: string
60
+ rep: number
61
+ /** `null` on an errored cell (the lib records failed cells with a null
62
+ * artifact; it never caches them). */
63
+ artifact: R4Artifact | null
64
+ error?: string
65
+ costUsd?: number
66
+ tokenUsage?: { input: number; output: number }
67
+ cached?: boolean
68
+ }
69
+
70
+ /** Adapt a lib campaign's cells (in-memory result) to `EvidenceCell`s. */
71
+ export function cellsFromCampaign(campaign: {
72
+ cells: Array<{
73
+ scenarioId: string
74
+ rep: number
75
+ artifact: unknown
76
+ error?: string
77
+ costUsd: number
78
+ tokenUsage: { input: number; output: number }
79
+ cached: boolean
80
+ }>
81
+ }): EvidenceCell[] {
82
+ return campaign.cells.map((cell) => ({
83
+ scenarioId: cell.scenarioId,
84
+ rep: cell.rep,
85
+ artifact: (cell.artifact ?? null) as R4Artifact | null,
86
+ ...(cell.error ? { error: cell.error } : {}),
87
+ costUsd: cell.costUsd,
88
+ tokenUsage: { input: cell.tokenUsage.input, output: cell.tokenUsage.output },
89
+ cached: cell.cached,
90
+ }))
91
+ }
92
+
93
+ // ---------------------------------------------------------------------------
94
+ // Replicate semantics — repsPerInstance. Single-rep scoring provably flips
95
+ // instance outcomes run-to-run (judge flake + capacity noise both observed),
96
+ // so an instance counts RESOLVED only when EVERY replicate cell resolved (AND
97
+ // — fail-closed for keep-if-better), and coverage requires every replicate of
98
+ // every instance to hold a real boolean verdict.
99
+ // ---------------------------------------------------------------------------
100
+
101
+ export interface ReplicateRun {
102
+ iid: string
103
+ resolved: boolean | null
104
+ }
105
+
106
+ /** Instances where ALL `reps` replicates resolved (missing replicates never count). */
107
+ export function resolvedInstanceCount(runs: ReplicateRun[], iids: string[], reps: number): number {
108
+ let count = 0
109
+ for (const iid of iids) {
110
+ const mine = runs.filter((r) => r.iid === iid)
111
+ if (mine.length === reps && mine.every((r) => r.resolved === true)) count += 1
112
+ }
113
+ return count
114
+ }
115
+
116
+ /** Every instance has exactly `reps` replicates, each with a conclusive verdict. */
117
+ export function replicateCoverageComplete(runs: ReplicateRun[], iids: string[], reps: number): boolean {
118
+ return iids.every((iid) => {
119
+ const mine = runs.filter((r) => r.iid === iid)
120
+ return mine.length === reps && mine.every((r) => r.resolved !== null)
121
+ })
122
+ }
123
+
124
+ /** One `ReplicateRun` per swe cell. An errored/artifact-less cell is an
125
+ * inconclusive replicate (`resolved: null`) — never a fabricated boolean. */
126
+ export function replicateRunsFromCells(cells: EvidenceCell[]): ReplicateRun[] {
127
+ return cells
128
+ .filter((c) => c.artifact === null || c.artifact.kind === 'swe-arm')
129
+ .map((c) => ({
130
+ iid: c.scenarioId,
131
+ resolved: c.artifact !== null && c.artifact.kind === 'swe-arm' && !c.error ? c.artifact.resolved : null,
132
+ }))
133
+ }
134
+
135
+ /** Σ wall seconds across the swe cells (errored cells contribute 0). */
136
+ export function sumWallSFromCells(cells: EvidenceCell[]): number {
137
+ return cells.reduce(
138
+ (s, c) => s + (c.artifact !== null && c.artifact.kind === 'swe-arm' ? c.artifact.wallS : 0),
139
+ 0,
140
+ )
141
+ }
142
+
143
+ /** Per-replicate staircase row — one per swe cell, straight off the artifact. */
144
+ export interface StaircasePerInstance {
145
+ iid: string
146
+ /** Replicate index (0-based) — repsPerInstance cells per instance. */
147
+ rep: number
148
+ resolved: boolean | null
149
+ verify_pass: boolean | null
150
+ patch_lines: number | null
151
+ wall_s: number | null
152
+ spentTokens: number | null
153
+ recoveredTokens: number | null
154
+ judgeAttempts: number | null
155
+ /** Campaign-cell CostLedger spend for this replicate (worker receipt). */
156
+ costUsd: number | null
157
+ error?: string
158
+ }
159
+
160
+ export function perInstanceFromCells(cells: EvidenceCell[]): StaircasePerInstance[] {
161
+ const rows: StaircasePerInstance[] = []
162
+ for (const cell of cells) {
163
+ const a = cell.artifact !== null && cell.artifact.kind === 'swe-arm' && !cell.error ? cell.artifact : null
164
+ rows.push({
165
+ iid: cell.scenarioId,
166
+ rep: cell.rep,
167
+ resolved: a ? a.resolved : null,
168
+ verify_pass: a ? a.verifyPass : null,
169
+ patch_lines: a ? a.patchLines : null,
170
+ wall_s: a ? a.wallS : null,
171
+ spentTokens: a ? a.spentTokens : null,
172
+ recoveredTokens: a ? a.recoveredTokens : null,
173
+ judgeAttempts: a ? a.judgeAttempts : null,
174
+ costUsd: cell.costUsd ?? null,
175
+ ...(cell.error ? { error: cell.error } : {}),
176
+ })
177
+ }
178
+ return rows
179
+ }
180
+
181
+ // ---------------------------------------------------------------------------
182
+ // Premeasured-baseline drift. The gate's denominator is the stored
183
+ // premeasured baseline artifact ({surfaceHash, campaign}) that the LIB
184
+ // validates before skipping the baseline campaign — surface hash, seed, reps,
185
+ // and split digest all fail loud on mismatch. A resumed runDir can still hold
186
+ // baseline cells cached by an OLDER run of the same surface; when those
187
+ // contradict the validated artifact, the contradiction is logged loud and the
188
+ // artifact still rules.
189
+ // ---------------------------------------------------------------------------
190
+
191
+ /** AND-verdict per instance from campaign cells. Only instances with full,
192
+ * conclusive replicate coverage produce a verdict — a partial record has no
193
+ * AND-verdict to compare. */
194
+ export function instanceVerdictsFromCells(
195
+ cells: EvidenceCell[],
196
+ iids: string[],
197
+ reps: number,
198
+ ): Record<string, boolean> {
199
+ const runs = replicateRunsFromCells(cells)
200
+ const verdicts: Record<string, boolean> = {}
201
+ for (const iid of iids) {
202
+ const mine = runs.filter((r) => r.iid === iid)
203
+ if (mine.length !== reps || mine.some((r) => r.resolved === null)) continue
204
+ verdicts[iid] = mine.every((r) => r.resolved === true)
205
+ }
206
+ return verdicts
207
+ }
208
+
209
+ /** Per-instance contradictions between the validated premeasured artifact's
210
+ * verdicts and locally cached baseline cells. Only instances with full-reps,
211
+ * conclusive coverage on BOTH sides are compared — a partial record has no
212
+ * AND-verdict to contradict with. The caller logs these loud and the
213
+ * premeasured artifact STILL rules. */
214
+ export function baselineDriftWarnings(
215
+ expected: Record<string, boolean>,
216
+ runs: ReplicateRun[],
217
+ iids: string[],
218
+ reps: number,
219
+ ): string[] {
220
+ const warnings: string[] = []
221
+ for (const iid of iids) {
222
+ const want = expected[iid]
223
+ if (typeof want !== 'boolean') continue
224
+ const mine = runs.filter((r) => r.iid === iid)
225
+ if (mine.length !== reps || mine.some((r) => r.resolved === null)) continue
226
+ const measured = mine.every((r) => r.resolved === true)
227
+ if (measured !== want) {
228
+ warnings.push(
229
+ `${iid}: premeasured=${want} but cached baseline cells measured ${measured} ` +
230
+ `(reps: ${mine.map((r) => String(r.resolved)).join('/')}) — the validated premeasured artifact rules`,
231
+ )
232
+ }
233
+ }
234
+ return warnings
235
+ }
236
+
237
+ // ---------------------------------------------------------------------------
238
+ // protocol_v2 keep-if-better.
239
+ // ---------------------------------------------------------------------------
240
+
241
+ export type StaircaseVerdict =
242
+ | 'accepted'
243
+ | 'rejected-no-gain'
244
+ | 'rejected-cost'
245
+ | 'rejected-out-of-space'
246
+ | 'rejected-incomplete'
247
+ /** Killed by the gen-3 pre-filter (change-space/tsc/smoke) BEFORE any full
248
+ * evaluation — the candidate never became a measured surface. Emitted by
249
+ * the outer loop's kill-row writer, never by `decideVerdict`. */
250
+ | 'rejected-prefilter'
251
+
252
+ /** protocol_v2 keep-if-better: improvement-set resolved-count must RISE and
253
+ * cost must stay within the guard. Fail-closed on unprovable cost. */
254
+ export function decideVerdict(input: {
255
+ violations: string[]
256
+ coverageComplete: boolean
257
+ resolvedCount: number
258
+ parentResolvedCount: number
259
+ costRatio: number | null
260
+ costGuardRatio: number
261
+ }): StaircaseVerdict {
262
+ if (input.violations.length > 0) return 'rejected-out-of-space'
263
+ if (!input.coverageComplete) return 'rejected-incomplete'
264
+ if (input.resolvedCount <= input.parentResolvedCount) return 'rejected-no-gain'
265
+ if (input.costRatio === null || input.costRatio > input.costGuardRatio) return 'rejected-cost'
266
+ return 'accepted'
267
+ }
268
+
269
+ // ---------------------------------------------------------------------------
270
+ // Disk readers — the lib's per-cell caches. run-campaign.ts writes
271
+ // `<campaignDir>/<sanitized cellId>/cached-result.json` for every conclusive
272
+ // cell (errored cells are never cached — a missing replicate reads as
273
+ // coverage-incomplete downstream, fail-closed).
274
+ // ---------------------------------------------------------------------------
275
+
276
+ /** Parse every `<cellDir>/cached-result.json` under one campaign dir. Missing dir = []. */
277
+ export async function loadCampaignCells(campaignDir: string): Promise<EvidenceCell[]> {
278
+ const entries = await readdir(campaignDir, { withFileTypes: true }).catch(() => [])
279
+ const cells: EvidenceCell[] = []
280
+ for (const entry of entries) {
281
+ if (!entry.isDirectory()) continue
282
+ const path = join(campaignDir, entry.name, 'cached-result.json')
283
+ const raw = await readFile(path, 'utf8').catch(() => null)
284
+ if (raw === null) continue
285
+ let parsed: Record<string, unknown>
286
+ try {
287
+ parsed = JSON.parse(raw) as Record<string, unknown>
288
+ } catch {
289
+ throw new Error(`loadCampaignCells: corrupt cell cache ${path}`)
290
+ }
291
+ if (typeof parsed.scenarioId !== 'string' || typeof parsed.rep !== 'number') {
292
+ throw new Error(`loadCampaignCells: ${path} is not a campaign cell (scenarioId/rep missing)`)
293
+ }
294
+ cells.push({
295
+ scenarioId: parsed.scenarioId,
296
+ rep: parsed.rep,
297
+ artifact: (parsed.artifact ?? null) as R4Artifact | null,
298
+ ...(typeof parsed.error === 'string' ? { error: parsed.error } : {}),
299
+ ...(typeof parsed.costUsd === 'number' ? { costUsd: parsed.costUsd } : {}),
300
+ ...(parsed.tokenUsage && typeof parsed.tokenUsage === 'object'
301
+ ? { tokenUsage: parsed.tokenUsage as { input: number; output: number } }
302
+ : {}),
303
+ cached: true,
304
+ })
305
+ }
306
+ return cells
307
+ }
308
+
309
+ export interface CandidateCellGroup {
310
+ generation: number
311
+ candidateIndex: number
312
+ dir: string
313
+ cells: EvidenceCell[]
314
+ /** The loops commit the cells' artifacts name (null when no artifact
315
+ * carries one — e.g. an all-errored, never-cached candidate). */
316
+ commit: string | null
317
+ }
318
+
319
+ /** Scan `gen-<g>/candidate-<i>/` campaign dirs under the improve runDir.
320
+ * Attribution is directory + artifact-commit — dispatch order plays no part. */
321
+ export async function loadCandidateCellGroups(improveRunDir: string): Promise<CandidateCellGroup[]> {
322
+ const groups: CandidateCellGroup[] = []
323
+ const top = await readdir(improveRunDir, { withFileTypes: true }).catch(() => [])
324
+ for (const genEntry of top) {
325
+ const genMatch = /^gen-(\d+)$/.exec(genEntry.name)
326
+ if (!genEntry.isDirectory() || !genMatch) continue
327
+ const genDir = join(improveRunDir, genEntry.name)
328
+ for (const candEntry of await readdir(genDir, { withFileTypes: true }).catch(() => [])) {
329
+ const candMatch = /^candidate-(\d+)$/.exec(candEntry.name)
330
+ if (!candEntry.isDirectory() || !candMatch) continue
331
+ const dir = join(genDir, candEntry.name)
332
+ const cells = await loadCampaignCells(dir)
333
+ const commits = new Set(
334
+ cells.map((c) => c.artifact?.commit).filter((c): c is string => typeof c === 'string'),
335
+ )
336
+ if (commits.size > 1) {
337
+ throw new Error(
338
+ `loadCandidateCellGroups: ${dir} mixes commits [${[...commits].join(', ')}] — one candidate dir must hold one surface`,
339
+ )
340
+ }
341
+ groups.push({
342
+ generation: Number(genMatch[1]),
343
+ candidateIndex: Number(candMatch[1]),
344
+ dir,
345
+ cells,
346
+ commit: [...commits][0] ?? null,
347
+ })
348
+ }
349
+ }
350
+ return groups.sort((a, b) => a.generation - b.generation || a.candidateIndex - b.candidateIndex)
351
+ }
352
+
353
+ // ---------------------------------------------------------------------------
354
+ // Gate evidence — the would-be-keep operator brief, derived from cells. The
355
+ // lib's deferred-holdout gate always holds; this evidence tells the operator
356
+ // whether the pre-registered holdout run is worth approving.
357
+ // ---------------------------------------------------------------------------
358
+
359
+ export interface GateEvidence {
360
+ candResolved: number
361
+ baseResolved: number
362
+ candWallS: number
363
+ baseWallS: number
364
+ costRatio: number | null
365
+ coverageComplete: boolean
366
+ verdict: StaircaseVerdict
367
+ }
368
+
369
+ /** Score the winner-vs-baseline comparison for the operator brief. Both sides
370
+ * come from campaign cells — the baseline side is the lib-validated
371
+ * premeasured campaign (or the bootstrap run's freshly measured one). */
372
+ export function gateEvidenceFromCells(input: {
373
+ winnerCells: EvidenceCell[]
374
+ baselineCells: EvidenceCell[]
375
+ /** Dispatch-time change-space violations of the winner's diff. */
376
+ violations: string[]
377
+ iids: string[]
378
+ reps: number
379
+ costGuardRatio: number
380
+ }): GateEvidence {
381
+ const winnerRuns = replicateRunsFromCells(input.winnerCells)
382
+ const baselineRuns = replicateRunsFromCells(input.baselineCells)
383
+ const candResolved = resolvedInstanceCount(winnerRuns, input.iids, input.reps)
384
+ const baseResolved = resolvedInstanceCount(baselineRuns, input.iids, input.reps)
385
+ const candWallS = sumWallSFromCells(input.winnerCells)
386
+ const baseWallS = sumWallSFromCells(input.baselineCells)
387
+ const costRatio = baseWallS > 0 ? candWallS / baseWallS : null
388
+ const coverageComplete = replicateCoverageComplete(winnerRuns, input.iids, input.reps)
389
+ return {
390
+ candResolved,
391
+ baseResolved,
392
+ candWallS,
393
+ baseWallS,
394
+ costRatio,
395
+ coverageComplete,
396
+ verdict: decideVerdict({
397
+ violations: input.violations,
398
+ coverageComplete,
399
+ resolvedCount: candResolved,
400
+ parentResolvedCount: baseResolved,
401
+ costRatio,
402
+ costGuardRatio: input.costGuardRatio,
403
+ }),
404
+ }
405
+ }
@@ -0,0 +1,248 @@
1
+ /**
2
+ * Cell-derived scoring: reps-AND aggregation as a pure function over lib
3
+ * campaign cells, and the disk readers over the lib's cached-result.json
4
+ * caches — including a reproduction of the r4-mroh3rkt resume shape (baseline
5
+ * cells replayed from cache, only a candidate dispatched in-process) proving
6
+ * attribution comes from the campaign directory + artifact commit, never from
7
+ * dispatch order. No arms, no docker, no tokens.
8
+ */
9
+
10
+ import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
11
+ import { tmpdir } from 'node:os'
12
+ import { join } from 'node:path'
13
+ import { afterAll, describe, expect, it } from 'vitest'
14
+ import {
15
+ baselineDriftWarnings,
16
+ cellsFromCampaign,
17
+ gateEvidenceFromCells,
18
+ instanceVerdictsFromCells,
19
+ loadCampaignCells,
20
+ loadCandidateCellGroups,
21
+ perInstanceFromCells,
22
+ replicateRunsFromCells,
23
+ resolvedInstanceCount,
24
+ sumWallSFromCells,
25
+ type EvidenceCell,
26
+ type R4Artifact,
27
+ } from './cell-evidence.mts'
28
+
29
+ const IIDS = ['astropy__astropy-13033', 'django__django-11532', 'matplotlib__matplotlib-20826']
30
+
31
+ function sweArtifact(iid: string, commit: string, resolved: boolean, wallS = 100): R4Artifact {
32
+ return {
33
+ kind: 'swe-arm',
34
+ iid,
35
+ commit,
36
+ resolved,
37
+ verifyPass: resolved,
38
+ patchLines: 10,
39
+ wallS,
40
+ spentTokens: 1000,
41
+ spentUsd: 0.01,
42
+ recoveredTokens: 1500,
43
+ workerTokIn: 400,
44
+ workerTokOut: 100,
45
+ judgeAttempts: 1,
46
+ judgeWallS: 30,
47
+ runDir: `/tmp/none/${iid}`,
48
+ patchPath: `/tmp/none/${iid}.patch`,
49
+ }
50
+ }
51
+
52
+ function cell(iid: string, rep: number, artifact: R4Artifact | null, error?: string): EvidenceCell {
53
+ return {
54
+ scenarioId: iid,
55
+ rep,
56
+ artifact,
57
+ ...(error ? { error } : {}),
58
+ costUsd: 0.01,
59
+ tokenUsage: { input: 400, output: 100 },
60
+ }
61
+ }
62
+
63
+ describe('cell adapters', () => {
64
+ it('cellsFromCampaign keeps errored cells with a null artifact', () => {
65
+ const cells = cellsFromCampaign({
66
+ cells: [
67
+ { scenarioId: 'a', rep: 0, artifact: sweArtifact('a', 'c1', true), costUsd: 0.5, tokenUsage: { input: 1, output: 2 }, cached: false },
68
+ { scenarioId: 'a', rep: 1, artifact: null, error: 'dispatch timeout', costUsd: 0, tokenUsage: { input: 0, output: 0 }, cached: false },
69
+ ],
70
+ })
71
+ expect(cells).toHaveLength(2)
72
+ expect(cells[0]!.artifact?.kind).toBe('swe-arm')
73
+ expect(cells[1]!.artifact).toBeNull()
74
+ expect(cells[1]!.error).toMatch(/timeout/)
75
+ })
76
+
77
+ it('replicateRunsFromCells: error/artifact-less cells are inconclusive, never a boolean', () => {
78
+ const runs = replicateRunsFromCells([
79
+ cell('a', 0, sweArtifact('a', 'c1', true)),
80
+ cell('a', 1, sweArtifact('a', 'c1', false), 'judge inconclusive'),
81
+ cell('b', 0, null, 'change-space violation'),
82
+ ])
83
+ expect(runs).toEqual([
84
+ { iid: 'a', resolved: true },
85
+ { iid: 'a', resolved: null },
86
+ { iid: 'b', resolved: null },
87
+ ])
88
+ })
89
+
90
+ it('perInstanceFromCells maps artifact fields and carries cell cost', () => {
91
+ const rows = perInstanceFromCells([
92
+ cell('a', 0, sweArtifact('a', 'c1', true, 250)),
93
+ cell('b', 1, null, 'boom'),
94
+ ])
95
+ expect(rows).toHaveLength(2)
96
+ expect(rows[0]).toMatchObject({ iid: 'a', rep: 0, resolved: true, wall_s: 250, costUsd: 0.01, judgeAttempts: 1 })
97
+ expect(rows[1]).toMatchObject({ iid: 'b', rep: 1, resolved: null, wall_s: null, error: 'boom' })
98
+ })
99
+
100
+ it('sumWallSFromCells sums swe wall only', () => {
101
+ expect(
102
+ sumWallSFromCells([
103
+ cell('a', 0, sweArtifact('a', 'c1', true, 100)),
104
+ cell('a', 1, sweArtifact('a', 'c1', true, 40)),
105
+ cell('b', 0, null, 'err'),
106
+ ]),
107
+ ).toBe(140)
108
+ })
109
+ })
110
+
111
+ describe('disk readers + the r4-mroh3rkt resume shape', () => {
112
+ const roots: string[] = []
113
+ afterAll(async () => {
114
+ for (const root of roots) await rm(root, { recursive: true, force: true })
115
+ })
116
+
117
+ async function writeCachedCell(campaignDir: string, c: EvidenceCell): Promise<void> {
118
+ const cellDir = join(campaignDir, `cell-${c.scenarioId.replace(/[^a-zA-Z0-9_-]/g, '_')}-r${c.rep}`)
119
+ await mkdir(cellDir, { recursive: true })
120
+ await writeFile(
121
+ join(cellDir, 'cached-result.json'),
122
+ JSON.stringify({
123
+ cellId: `cell-${c.scenarioId}-r${c.rep}`,
124
+ scenarioId: c.scenarioId,
125
+ rep: c.rep,
126
+ artifact: c.artifact,
127
+ ...(c.error ? { error: c.error } : {}),
128
+ costUsd: c.costUsd ?? 0,
129
+ tokenUsage: c.tokenUsage ?? { input: 0, output: 0 },
130
+ judgeScores: {},
131
+ durationMs: 1,
132
+ seed: 42,
133
+ cached: false,
134
+ }),
135
+ )
136
+ }
137
+
138
+ /** The exact resume shape behind r4-mroh3rkt: the BASELINE campaign exists
139
+ * only as cached cells on disk (measured 1/3 — matplotlib both reps), while
140
+ * the process only ever dispatched candidate b08d31c910 (0/3). */
141
+ async function makeResumedRunDir(): Promise<string> {
142
+ const root = await mkdtemp(join(tmpdir(), 'r4-cells-'))
143
+ roots.push(root)
144
+ const improveRun = join(root, 'improve-run')
145
+ const baselineDir = join(improveRun, 'baseline')
146
+ for (const iid of IIDS) {
147
+ const resolved = iid.startsWith('matplotlib')
148
+ for (const rep of [0, 1]) {
149
+ await writeCachedCell(baselineDir, cell(iid, rep, sweArtifact(iid, 'basecommit0', resolved, 100)))
150
+ }
151
+ }
152
+ const candDir = join(improveRun, 'gen-0', 'candidate-0')
153
+ for (const iid of IIDS) {
154
+ for (const rep of [0, 1]) {
155
+ await writeCachedCell(candDir, cell(iid, rep, sweArtifact(iid, 'b08d31c910', false, 110)))
156
+ }
157
+ }
158
+ return improveRun
159
+ }
160
+
161
+ it('loadCampaignCells reads every cached cell; a missing dir is empty', async () => {
162
+ const improveRun = await makeResumedRunDir()
163
+ const cells = await loadCampaignCells(join(improveRun, 'baseline'))
164
+ expect(cells).toHaveLength(6)
165
+ expect(cells.every((c) => c.cached)).toBe(true)
166
+ expect(await loadCampaignCells(join(improveRun, 'no-such-campaign'))).toEqual([])
167
+ })
168
+
169
+ it('candidate groups attribute by directory + commit — baseline cells never leak in', async () => {
170
+ const improveRun = await makeResumedRunDir()
171
+ const groups = await loadCandidateCellGroups(improveRun)
172
+ expect(groups).toHaveLength(1)
173
+ expect(groups[0]).toMatchObject({ generation: 0, candidateIndex: 0, commit: 'b08d31c910' })
174
+ expect(groups[0]!.cells).toHaveLength(6)
175
+ expect(groups[0]!.cells.every((c) => c.artifact?.commit === 'b08d31c910')).toBe(true)
176
+ })
177
+
178
+ it('r4-mroh3rkt regression: the resumed run grades winner 0/3 vs baseline 1/3, not 0/3 vs 0/3', async () => {
179
+ const improveRun = await makeResumedRunDir()
180
+ const groups = await loadCandidateCellGroups(improveRun)
181
+ const winner = groups.find((g) => g.commit === 'b08d31c910')!
182
+ const baselineCells = await loadCampaignCells(join(improveRun, 'baseline'))
183
+
184
+ // Attribution is campaign directory + artifact commit: the candidate's
185
+ // cells can never masquerade as the baseline, dispatched or replayed.
186
+ const ev = gateEvidenceFromCells({
187
+ winnerCells: winner.cells,
188
+ baselineCells,
189
+ violations: [],
190
+ iids: IIDS,
191
+ reps: 2,
192
+ costGuardRatio: 1.2,
193
+ })
194
+ expect(ev.candResolved).toBe(0)
195
+ expect(ev.baseResolved).toBe(1) // the bug published 0/3 here
196
+ expect(ev.verdict).toBe('rejected-no-gain')
197
+ expect(ev.coverageComplete).toBe(true)
198
+ expect(ev.costRatio).toBeCloseTo(660 / 600, 5)
199
+ })
200
+
201
+ it('a contradicting cached baseline raises drift warnings and the premeasured artifact rules', async () => {
202
+ const improveRun = await makeResumedRunDir()
203
+ const cachedBaseline = await loadCampaignCells(join(improveRun, 'baseline'))
204
+ // Premeasured artifact verdicts contradicting the cached cells on astropy.
205
+ const expected = {
206
+ 'astropy__astropy-13033': true, // cached cells measured F/F
207
+ 'django__django-11532': false,
208
+ 'matplotlib__matplotlib-20826': true,
209
+ }
210
+ const warnings = baselineDriftWarnings(expected, replicateRunsFromCells(cachedBaseline), IIDS, 2)
211
+ expect(warnings).toHaveLength(1)
212
+ expect(warnings[0]).toContain('astropy__astropy-13033: premeasured=true')
213
+ expect(warnings[0]).toContain('premeasured artifact rules')
214
+ // The AND-verdicts of the cached campaign (the drift comparator source).
215
+ expect(instanceVerdictsFromCells(cachedBaseline, IIDS, 2)).toEqual({
216
+ 'astropy__astropy-13033': false,
217
+ 'django__django-11532': false,
218
+ 'matplotlib__matplotlib-20826': true,
219
+ })
220
+ })
221
+
222
+ it('a missing replicate keeps the candidate coverage-incomplete (errored cells are never cached)', async () => {
223
+ const improveRun = await makeResumedRunDir()
224
+ const partialDir = join(improveRun, 'gen-0', 'candidate-1')
225
+ await writeCachedCell(partialDir, cell(IIDS[0]!, 0, sweArtifact(IIDS[0]!, 'deadbeef01', true)))
226
+ const groups = await loadCandidateCellGroups(improveRun)
227
+ const partial = groups.find((g) => g.commit === 'deadbeef01')!
228
+ const ev = gateEvidenceFromCells({
229
+ winnerCells: partial.cells,
230
+ baselineCells: await loadCampaignCells(join(improveRun, 'baseline')),
231
+ violations: [],
232
+ iids: IIDS,
233
+ reps: 2,
234
+ costGuardRatio: 1.2,
235
+ })
236
+ expect(ev.coverageComplete).toBe(false)
237
+ expect(ev.verdict).toBe('rejected-incomplete')
238
+ expect(resolvedInstanceCount(replicateRunsFromCells(partial.cells), IIDS, 2)).toBe(0)
239
+ })
240
+
241
+ it('a candidate dir mixing two commits fails loud', async () => {
242
+ const improveRun = await makeResumedRunDir()
243
+ const dir = join(improveRun, 'gen-1', 'candidate-0')
244
+ await writeCachedCell(dir, cell(IIDS[0]!, 0, sweArtifact(IIDS[0]!, 'commitaaaa1', true)))
245
+ await writeCachedCell(dir, cell(IIDS[1]!, 0, sweArtifact(IIDS[1]!, 'commitbbbb2', true)))
246
+ await expect(loadCandidateCellGroups(improveRun)).rejects.toThrow(/mixes commits/)
247
+ })
248
+ })