@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,87 +0,0 @@
1
- /**
2
- * PARITY GATE for M2: the typed execution path (materialize → patch apply →
3
- * extractPatch → serialized-judge → official swebench verdict) must reproduce
4
- * the pinned M1 fixture verdicts for two committed arm patches — one resolved
5
- * (pallets__flask-5014 SOLO), one unresolved (pydata__xarray-4687 SUP) —
6
- * WITHOUT executing any arm. Zero model tokens; docker time only (it pulls the
7
- * two instance images if absent and runs the official judge, ~5-25 min each).
8
- *
9
- * Opt-in because of that docker cost:
10
- *
11
- * SWE_ARENA_PARITY=1 ../node_modules/.bin/vitest run src/swe-arena/parity.test.mts
12
- *
13
- * Expected verdicts are DERIVED from the pinned fixtures (ledger + rejudge
14
- * reconciliation), not hardcoded — parity means agreeing with the record.
15
- *
16
- * TODO(operator approval): the full 12-instance ARM parity re-run (typed path
17
- * vs the bash harness on the same 12 ids, both arms) costs ~$1 model spend +
18
- * ~4h wall. Do not run without an explicit operator go — this 2-patch gate is
19
- * the milestone check.
20
- */
21
-
22
- import { mkdtemp, rm } from 'node:fs/promises'
23
- import { tmpdir } from 'node:os'
24
- import { join } from 'node:path'
25
- import { afterAll, beforeAll, describe, expect, it } from 'vitest'
26
- import { loadLedger, loadRejudge } from './fixtures.ts'
27
- import { reconcile } from './reconcile.ts'
28
- import { createSerializedJudge } from './serialized-judge.ts'
29
- import { PARITY_CASES, replayPatchParity, type ParityCaseResult } from './run-experiment.mts'
30
-
31
- const RUN_PARITY = process.env.SWE_ARENA_PARITY === '1'
32
- // materialize + image pull + official judge, twice; generous but finite.
33
- const PARITY_BUDGET_MS = 5_400_000
34
-
35
- describe.runIf(RUN_PARITY)('dry-run parity vs pinned M1 verdicts (docker, no tokens)', () => {
36
- let results: ParityCaseResult[] = []
37
- let workDir = ''
38
-
39
- beforeAll(async () => {
40
- workDir = await mkdtemp(join(tmpdir(), 'swe-arena-parity-'))
41
- const judge = createSerializedJudge() // real judge child, 1800s ceiling, serialized
42
- results = await replayPatchParity(PARITY_CASES, { workDir, judge })
43
- for (const r of results) {
44
- console.log(`PARITY ${r.iid} [${r.arm}] verdict=${JSON.stringify(r.verdict)}`)
45
- }
46
- }, PARITY_BUDGET_MS)
47
-
48
- afterAll(async () => {
49
- if (workDir) await rm(workDir, { recursive: true, force: true })
50
- })
51
-
52
- it('re-extraction preserves each committed patch’s changed-file set', () => {
53
- expect(results).toHaveLength(2)
54
- for (const r of results) {
55
- expect(r.applyRc).toBe(0)
56
- expect(r.extractedFiles, `${r.iid} [${r.arm}]`).toEqual(r.fixtureFiles)
57
- expect(r.extractedPatchLines).toBeGreaterThan(0)
58
- }
59
- })
60
-
61
- it('pallets__flask-5014 SOLO: typed judge verdict == pinned ledger verdict (resolved)', () => {
62
- const pinned = loadLedger().find((r) => r.iid === 'pallets__flask-5014')
63
- expect(pinned?.solo_resolved).toBe(true) // sanity: the fixture side of the parity claim
64
- const r = results.find((x) => x.iid === 'pallets__flask-5014')
65
- expect(r?.verdict.attempts).toBeGreaterThan(0)
66
- expect(r?.verdict.resolved).toBe(pinned?.solo_resolved)
67
- })
68
-
69
- it('pydata__xarray-4687 SUP: typed judge verdict == reconciled verdict (unresolved)', () => {
70
- // The authoritative record for this arm is the personal re-judge
71
- // (sup-final), which M1's reconciliation applies over the ledger.
72
- const reconciled = reconcile(loadLedger(), loadRejudge()).valid.find(
73
- (r) => r.iid === 'pydata__xarray-4687',
74
- )
75
- expect(reconciled?.sup.source).toBe('sup-final')
76
- expect(reconciled?.sup.resolved).toBe(false)
77
- const r = results.find((x) => x.iid === 'pydata__xarray-4687')
78
- expect(r?.verdict.attempts).toBeGreaterThan(0)
79
- expect(r?.verdict.resolved).toBe(reconciled?.sup.resolved)
80
- })
81
- })
82
-
83
- describe.runIf(!RUN_PARITY)('dry-run parity (skipped)', () => {
84
- it('is opt-in: set SWE_ARENA_PARITY=1 to run the docker-backed parity gate', () => {
85
- expect(RUN_PARITY).toBe(false)
86
- })
87
- })
@@ -1,296 +0,0 @@
1
- /**
2
- * Premeasured-baseline artifact builder — reconstruct the lib's
3
- * `PremeasuredOptimizationBaseline` ({surfaceHash, campaign}) from a PRIOR
4
- * run's on-disk baseline campaign cells.
5
- *
6
- * Why this exists: the gen-3 run (r4-mrwc0awe) measured the full 6-instance ×
7
- * 2-rep baseline but never wrote the artifact — its bootstrap writer keyed on
8
- * the Pareto frontier's generation −1 entry, and the frontier the lib returned
9
- * held no such entry, so the measurement survives ONLY as per-cell
10
- * `cached-result.json` caches under `<gen3>/improve-run/baseline/`. Those
11
- * caches cannot be replayed into a NEW runDir: the lib's cache-hit path
12
- * requires every `costCallIds` receipt in the CURRENT run's ledger (tagged
13
- * with the current runDir), which a fresh outDir cannot satisfy. The
14
- * premeasured-artifact path has no such coupling — `validatedPremeasuredBaseline`
15
- * checks surface hash, seed, reps, and split digest, then uses the campaign
16
- * as-is — so rebuilding the artifact is the designed way to pin a prior run's
17
- * measured baseline.
18
- *
19
- * Everything identity-bearing is REAL, never fabricated:
20
- * - cells: verbatim `cached-result.json` records (verdicts, spend, receipts);
21
- * - surfaceHash: recomputed from the loops repo's base-ref tip (identical
22
- * incumbent surface shape the lib builds: baseCommit == candidateCommit,
23
- * empty patch) — verified equal to gen-3's recorded baseline hash;
24
- * - splitDigest/scenarios: the lib's own `campaignSplitDigest` /
25
- * `campaignScenarioIdentity` over the exact scenario payloads;
26
- * - seed/manifestHash: carried from the cells, uniformity asserted;
27
- * - run window: reconstructed from cache-file mtimes (endedAt = last cell
28
- * write; startedAt = earliest write minus that cell's duration).
29
- *
30
- * tsx src/swe-arena/premeasured-from-cells.mts <config.json> --from <priorBaselineDir>
31
- */
32
-
33
- import { createHash } from 'node:crypto'
34
- import { readdir, readFile, stat, writeFile } from 'node:fs/promises'
35
- import { join } from 'node:path'
36
- import process from 'node:process'
37
- import { pathToFileURL } from 'node:url'
38
- import {
39
- assertCampaignSplitIdentity,
40
- campaignScenarioIdentity,
41
- campaignSplitDigest,
42
- surfaceHash,
43
- type CampaignCellResult,
44
- type CampaignResult,
45
- type CodeSurface,
46
- type PremeasuredOptimizationBaseline,
47
- type Scenario,
48
- } from '@tangle-network/agent-eval/campaign'
49
- import type { R4Artifact } from './cell-evidence.mts'
50
- import type { OuterLoopConfig } from './outer-loop.mts'
51
- import { runOk } from './proc.ts'
52
-
53
- export type FullCell = CampaignCellResult<R4Artifact> & { mtimeMs?: number }
54
-
55
- /** Read every `<dir>/<cell>/cached-result.json` as a FULL lib cell record. */
56
- export async function loadFullCampaignCells(campaignDir: string): Promise<FullCell[]> {
57
- const entries = await readdir(campaignDir, { withFileTypes: true })
58
- const cells: FullCell[] = []
59
- for (const entry of entries) {
60
- if (!entry.isDirectory()) continue
61
- const path = join(campaignDir, entry.name, 'cached-result.json')
62
- const raw = await readFile(path, 'utf8').catch(() => null)
63
- if (raw === null) continue
64
- const cell = JSON.parse(raw) as FullCell
65
- if (typeof cell.scenarioId !== 'string' || typeof cell.rep !== 'number') {
66
- throw new Error(`premeasured-from-cells: ${path} is not a campaign cell`)
67
- }
68
- cell.mtimeMs = (await stat(path)).mtimeMs
69
- cells.push(cell)
70
- }
71
- return cells
72
- }
73
-
74
- /** The incumbent surface hash the lib will compute for `baseRef`'s tip: an
75
- * unchanged code surface (candidate == base, empty patch). Identity material
76
- * is {baseCommit, baseTree, candidateTree, patch} — worktree/ref excluded. */
77
- export async function incumbentSurfaceHash(loopsRepo: string, baseRef: string): Promise<string> {
78
- const baseCommit = (await runOk('git', ['-C', loopsRepo, 'rev-parse', `${baseRef}^{commit}`])).stdout.trim()
79
- const tree = (await runOk('git', ['-C', loopsRepo, 'rev-parse', `${baseRef}^{tree}`])).stdout.trim()
80
- const surface: CodeSurface = {
81
- kind: 'code',
82
- worktreeRef: loopsRepo,
83
- baseRef,
84
- baseCommit,
85
- baseTree: tree,
86
- candidateCommit: baseCommit,
87
- candidateTree: tree,
88
- patch: {
89
- format: 'git-diff-binary',
90
- sha256: `sha256:${createHash('sha256').update(Buffer.alloc(0)).digest('hex')}`,
91
- byteLength: 0,
92
- },
93
- }
94
- return surfaceHash(surface)
95
- }
96
-
97
- function meanStdCi(values: number[]): { mean: number; stdev: number; ci95: [number, number]; n: number } {
98
- const n = values.length
99
- const mean = n === 0 ? 0 : values.reduce((s, v) => s + v, 0) / n
100
- const variance = n <= 1 ? 0 : values.reduce((s, v) => s + (v - mean) ** 2, 0) / (n - 1)
101
- const stdev = Math.sqrt(variance)
102
- const half = n === 0 ? 0 : (1.96 * stdev) / Math.sqrt(n)
103
- return { mean, stdev, ci95: [mean - half, mean + half], n }
104
- }
105
-
106
- /** Assemble the artifact from full cells. Pure over inputs; every check fails
107
- * loud so a wrong artifact can never validate downstream by accident. */
108
- export function buildPremeasuredFromCells(input: {
109
- cells: FullCell[]
110
- instances: string[]
111
- reps: number
112
- surfaceHash: string
113
- /** Provenance label recorded as the campaign's runDir (the SOURCE dir). */
114
- sourceDir: string
115
- }): PremeasuredOptimizationBaseline<R4Artifact, Scenario> {
116
- const { cells, instances, reps } = input
117
- if (cells.length === 0) throw new Error('premeasured-from-cells: no cells')
118
- for (const iid of instances) {
119
- for (let rep = 0; rep < reps; rep++) {
120
- const mine = cells.filter((c) => c.scenarioId === iid && c.rep === rep)
121
- if (mine.length !== 1) {
122
- throw new Error(`premeasured-from-cells: expected exactly 1 cell for ${iid} rep ${rep}, got ${mine.length}`)
123
- }
124
- }
125
- }
126
- const extras = cells.filter((c) => !instances.includes(c.scenarioId) || c.rep >= reps)
127
- if (extras.length > 0) {
128
- throw new Error(
129
- `premeasured-from-cells: ${extras.length} cell(s) outside the ${instances.length}x${reps} design ` +
130
- `(e.g. ${extras[0]!.scenarioId}#r${extras[0]!.rep})`,
131
- )
132
- }
133
- // Per-cell seeds derive from the campaign base seed (run-campaign:
134
- // `cellSeed = seed + groupIndex * reps + rep`), so a complete design carries
135
- // the contiguous range [base, base + cells). The campaign-level seed the lib
136
- // validates is the base.
137
- const seeds = cells.map((c) => c.seed).sort((a, b) => a - b)
138
- const baseSeed = seeds[0]!
139
- for (let i = 0; i < seeds.length; i++) {
140
- if (seeds[i] !== baseSeed + i) {
141
- throw new Error(
142
- `premeasured-from-cells: cell seeds are not the contiguous derived range from base ${baseSeed} ` +
143
- `(got ${seeds.join(', ')})`,
144
- )
145
- }
146
- }
147
- const manifests = new Set(cells.map((c) => c.manifestHash ?? 'absent'))
148
- if (manifests.size !== 1) {
149
- throw new Error(`premeasured-from-cells: non-uniform manifestHash across cells (${[...manifests].join(', ')})`)
150
- }
151
- for (const cell of cells) {
152
- if (cell.error !== undefined) throw new Error(`premeasured-from-cells: cell ${cell.cellId} carries an error`)
153
- if (cell.artifact === null || cell.artifact.kind !== 'swe-arm') {
154
- throw new Error(`premeasured-from-cells: cell ${cell.cellId} has no swe-arm artifact`)
155
- }
156
- if (cell.costProvenance.kind === 'uncaptured') {
157
- throw new Error(`premeasured-from-cells: cell ${cell.cellId} has uncaptured cost`)
158
- }
159
- if (
160
- cell.costProvenance.usd !== cell.costUsd
161
- ) {
162
- throw new Error(
163
- `premeasured-from-cells: cell ${cell.cellId} cost provenance does not match costUsd`,
164
- )
165
- }
166
- }
167
-
168
- const scenarios: Scenario[] = instances.map((iid) => ({ id: iid, kind: 'swe-instance' }))
169
- const splitDigest = campaignSplitDigest(scenarios, reps)
170
- const identities = scenarios.map((s) => campaignScenarioIdentity(s))
171
- assertCampaignSplitIdentity(identities, reps, splitDigest)
172
-
173
- const sorted = [...cells].sort(
174
- (a, b) => instances.indexOf(a.scenarioId) - instances.indexOf(b.scenarioId) || a.rep - b.rep,
175
- )
176
- const composites = (cell: FullCell): number[] => Object.values(cell.judgeScores).map((s) => s.composite)
177
- const judgeNames = [...new Set(sorted.flatMap((c) => Object.keys(c.judgeScores)))]
178
- const byJudge = Object.fromEntries(
179
- judgeNames.map((name) => {
180
- const { mean, stdev, ci95, n } = meanStdCi(
181
- sorted.filter((c) => c.judgeScores[name]).map((c) => c.judgeScores[name]!.composite),
182
- )
183
- return [name, { mean, stdev, ci95, n }]
184
- }),
185
- )
186
- const byScenario = Object.fromEntries(
187
- instances.map((iid) => {
188
- const { mean, ci95, n } = meanStdCi(sorted.filter((c) => c.scenarioId === iid).flatMap(composites))
189
- return [iid, { meanComposite: mean, ci95, n }]
190
- }),
191
- )
192
- const totalCostUsd = sorted.reduce((s, c) => s + c.costUsd, 0)
193
- const costProvenance: CampaignCellResult<R4Artifact>['costProvenance'] = sorted.some(
194
- (cell) => cell.costProvenance.kind === 'estimated',
195
- )
196
- ? { kind: 'estimated', usd: totalCostUsd }
197
- : { kind: 'observed', usd: totalCostUsd }
198
- const inputTokens = sorted.reduce((s, c) => s + c.tokenUsage.input, 0)
199
- const outputTokens = sorted.reduce((s, c) => s + c.tokenUsage.output, 0)
200
- const totalCalls = sorted.reduce((s, c) => s + (c.costCallIds?.length ?? 0), 0)
201
- const models = [...new Set(sorted.map((c) => c.resolvedModel).filter((m): m is string => typeof m === 'string'))]
202
-
203
- const mtimes = sorted.filter((c) => typeof c.mtimeMs === 'number')
204
- const endedMs = mtimes.length > 0 ? Math.max(...mtimes.map((c) => c.mtimeMs!)) : 0
205
- const startedMs = mtimes.length > 0 ? Math.min(...mtimes.map((c) => c.mtimeMs! - c.durationMs)) : 0
206
-
207
- const campaign: CampaignResult<R4Artifact, Scenario> = {
208
- manifestHash: sorted[0]!.manifestHash ?? '',
209
- splitDigest,
210
- seed: baseSeed,
211
- reps,
212
- startedAt: new Date(startedMs).toISOString(),
213
- endedAt: new Date(endedMs).toISOString(),
214
- durationMs: Math.max(0, endedMs - startedMs),
215
- // Verbatim prior-run cells (marked cached — they were measured, and this
216
- // artifact replays them without dispatch). The transient mtime rider is
217
- // dropped from the persisted record.
218
- cells: sorted.map(({ mtimeMs: _mtimeMs, ...cell }) => ({ ...cell, cached: true })),
219
- aggregates: {
220
- byJudge,
221
- byScenario,
222
- cost: {
223
- totalCalls,
224
- pendingCalls: 0,
225
- unresolvedCalls: 0,
226
- reservedCostUsd: 0,
227
- inputTokens,
228
- outputTokens,
229
- cachedTokens: 0,
230
- totalCostUsd,
231
- costProvenance,
232
- byChannel: [
233
- {
234
- channel: 'agent',
235
- calls: totalCalls,
236
- inputTokens,
237
- outputTokens,
238
- cachedTokens: 0,
239
- costUsd: totalCostUsd,
240
- unpricedCalls: 0,
241
- unknownUsageCalls: 0,
242
- },
243
- ],
244
- unpricedModels: [],
245
- fullyPriced: true,
246
- usageComplete: true,
247
- accountingComplete: true,
248
- incompleteReasons: [],
249
- },
250
- cellsExecuted: sorted.length,
251
- cellsSkipped: 0,
252
- cellsCached: sorted.length,
253
- cellsFailed: 0,
254
- },
255
- runDir: input.sourceDir,
256
- artifactsByPath: {},
257
- scenarios: identities,
258
- }
259
- void models
260
- return { surfaceHash: input.surfaceHash, campaign }
261
- }
262
-
263
- // ---------------------------------------------------------------------------
264
- // CLI.
265
- // ---------------------------------------------------------------------------
266
-
267
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
268
-
269
- if (isMain) {
270
- const argv = process.argv.slice(2)
271
- const configPath = argv[0]
272
- const fromIdx = argv.indexOf('--from')
273
- const fromDir = fromIdx !== -1 ? argv[fromIdx + 1] : undefined
274
- if (!configPath || configPath.startsWith('--') || !fromDir) {
275
- console.error('usage: tsx src/swe-arena/premeasured-from-cells.mts <config.json> --from <priorBaselineCampaignDir>')
276
- process.exit(2)
277
- }
278
- const config = JSON.parse(await readFile(configPath, 'utf8')) as OuterLoopConfig
279
- const reps = config.repsPerInstance ?? 1
280
- const cells = await loadFullCampaignCells(fromDir)
281
- const hash = await incumbentSurfaceHash(config.loopsRepo, config.loopsBaseRef)
282
- const artifact = buildPremeasuredFromCells({
283
- cells,
284
- instances: config.instances,
285
- reps,
286
- surfaceHash: hash,
287
- sourceDir: fromDir,
288
- })
289
- await writeFile(config.premeasuredBaselinePath, JSON.stringify(artifact, null, 1))
290
- const resolved = artifact.campaign.cells.map(
291
- (c) => `${c.scenarioId}#r${c.rep}=${(c.artifact as R4Artifact).resolved}`,
292
- )
293
- console.log(`premeasured baseline artifact → ${config.premeasuredBaselinePath}`)
294
- console.log(`surfaceHash=${hash} splitDigest=${artifact.campaign.splitDigest} seed=${artifact.campaign.seed} reps=${reps}`)
295
- console.log(resolved.join('\n'))
296
- }
@@ -1,201 +0,0 @@
1
- import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
2
- import { tmpdir } from 'node:os'
3
- import { join } from 'node:path'
4
- import { campaignSplitDigest, assertCampaignSplitIdentity, type Scenario } from '@tangle-network/agent-eval/campaign'
5
- import { afterEach, beforeEach, describe, expect, it } from 'vitest'
6
- import { cellsFromCampaign, replicateRunsFromCells, resolvedInstanceCount } from './cell-evidence.mts'
7
- import type { R4Artifact } from './cell-evidence.mts'
8
- import {
9
- buildPremeasuredFromCells,
10
- incumbentSurfaceHash,
11
- loadFullCampaignCells,
12
- type FullCell,
13
- } from './premeasured-from-cells.mts'
14
- import { runOk } from './proc.ts'
15
-
16
- const artifact = (iid: string, resolved: boolean): R4Artifact => ({
17
- kind: 'swe-arm',
18
- iid,
19
- commit: 'c0ffee',
20
- resolved,
21
- verifyPass: resolved,
22
- patchLines: 3,
23
- wallS: 100,
24
- spentTokens: 10,
25
- spentUsd: 0.01,
26
- recoveredTokens: 20,
27
- workerTokIn: 5,
28
- workerTokOut: 5,
29
- judgeAttempts: 1,
30
- judgeWallS: 9,
31
- runDir: '/runs/x',
32
- patchPath: '/patches/x.patch',
33
- })
34
-
35
- // Per-cell seeds mirror run-campaign's derivation: base 42 + groupIndex*reps + rep.
36
- const derivedSeed = (iid: string, rep: number): number => 42 + (iid === 'inst-a' ? 0 : 2) + rep
37
-
38
- const cell = (iid: string, rep: number, resolved: boolean, over: Partial<FullCell> = {}): FullCell => ({
39
- manifestHash: 'm1',
40
- cellId: `${iid}:${rep}`,
41
- scenarioId: iid,
42
- rep,
43
- artifact: artifact(iid, resolved),
44
- judgeScores: {
45
- 'swe-arena-official-judge': {
46
- composite: resolved ? 1 : 0,
47
- dimensions: { resolved: resolved ? 1 : 0 },
48
- notes: `official judge: ${iid} resolved=${resolved}`,
49
- },
50
- },
51
- costUsd: 0.05,
52
- costProvenance: { kind: 'observed', usd: 0.05 },
53
- costCallIds: [`call-${iid}-${rep}`],
54
- tokenUsage: { input: 100, output: 50 },
55
- resolvedModel: 'zai-coding-plan/glm-5.2',
56
- durationMs: 60_000,
57
- seed: derivedSeed(iid, rep),
58
- cached: false,
59
- mtimeMs: 1_000_000 + rep * 1000,
60
- ...over,
61
- })
62
-
63
- const INSTANCES = ['inst-a', 'inst-b']
64
-
65
- describe('buildPremeasuredFromCells', () => {
66
- const cells = [cell('inst-a', 0, false), cell('inst-a', 1, true), cell('inst-b', 0, true), cell('inst-b', 1, true)]
67
-
68
- it('assembles a campaign whose split identity passes the lib validation checks', () => {
69
- const out = buildPremeasuredFromCells({
70
- cells,
71
- instances: INSTANCES,
72
- reps: 2,
73
- surfaceHash: 'abc123',
74
- sourceDir: '/prior/baseline',
75
- })
76
- expect(out.surfaceHash).toBe('abc123')
77
- const scenarios: Scenario[] = INSTANCES.map((id) => ({ id, kind: 'swe-instance' }))
78
- // The exact checks validatedPremeasuredBaseline runs (minus surface):
79
- expect(out.campaign.reps).toBe(2)
80
- expect(out.campaign.seed).toBe(42)
81
- expect(out.campaign.splitDigest).toBe(campaignSplitDigest(scenarios, 2))
82
- expect(() => assertCampaignSplitIdentity(out.campaign.scenarios, 2, out.campaign.splitDigest)).not.toThrow()
83
- // Cells replay verbatim into the harness scoring (AND rule: a=F, b=T → 1/2).
84
- const evidence = cellsFromCampaign(out.campaign)
85
- expect(resolvedInstanceCount(replicateRunsFromCells(evidence), INSTANCES, 2)).toBe(1)
86
- expect(out.campaign.cells.every((c) => c.cached)).toBe(true)
87
- expect(out.campaign.cells.every((c) => !('mtimeMs' in c))).toBe(true)
88
- // Honest spend rollup from the real cells.
89
- expect(out.campaign.aggregates.cost.totalCostUsd).toBeCloseTo(0.2)
90
- expect(out.campaign.aggregates.cost.costProvenance).toEqual({
91
- kind: 'observed',
92
- usd: 0.2,
93
- })
94
- expect(out.campaign.aggregates.cost.inputTokens).toBe(400)
95
- expect(out.campaign.runDir).toBe('/prior/baseline')
96
- // Window reconstructed from mtimes: ends at the last cell write.
97
- expect(out.campaign.endedAt).toBe(new Date(1_001_000).toISOString())
98
- })
99
-
100
- it('fails loud on a missing replicate, an extra cell, or a design mismatch', () => {
101
- expect(() =>
102
- buildPremeasuredFromCells({
103
- cells: cells.slice(0, 3),
104
- instances: INSTANCES,
105
- reps: 2,
106
- surfaceHash: 'x',
107
- sourceDir: '/p',
108
- }),
109
- ).toThrow(/expected exactly 1 cell for inst-b rep 1/)
110
- expect(() =>
111
- buildPremeasuredFromCells({
112
- cells: [...cells, cell('inst-c', 0, true)],
113
- instances: INSTANCES,
114
- reps: 2,
115
- surfaceHash: 'x',
116
- sourceDir: '/p',
117
- }),
118
- ).toThrow(/outside the 2x2 design/)
119
- })
120
-
121
- it('fails loud on broken seed derivation or manifest drift, an errored cell, or a missing artifact', () => {
122
- const seedDrift = [cell('inst-a', 0, false, { seed: 7 }), cell('inst-a', 1, true), cell('inst-b', 0, true), cell('inst-b', 1, true)]
123
- expect(() =>
124
- buildPremeasuredFromCells({ cells: seedDrift, instances: INSTANCES, reps: 2, surfaceHash: 'x', sourceDir: '/p' }),
125
- ).toThrow(/contiguous derived range/)
126
- const manifestDrift = [cell('inst-a', 0, false, { manifestHash: 'OTHER' }), cell('inst-a', 1, true), cell('inst-b', 0, true), cell('inst-b', 1, true)]
127
- expect(() =>
128
- buildPremeasuredFromCells({ cells: manifestDrift, instances: INSTANCES, reps: 2, surfaceHash: 'x', sourceDir: '/p' }),
129
- ).toThrow(/non-uniform manifestHash/)
130
- const errored = [cell('inst-a', 0, false, { error: 'boom' }), cell('inst-a', 1, true), cell('inst-b', 0, true), cell('inst-b', 1, true)]
131
- expect(() =>
132
- buildPremeasuredFromCells({ cells: errored, instances: INSTANCES, reps: 2, surfaceHash: 'x', sourceDir: '/p' }),
133
- ).toThrow(/carries an error/)
134
- const uncapturedCost = [
135
- cell('inst-a', 0, false, { costProvenance: { kind: 'uncaptured', usd: null } }),
136
- cell('inst-a', 1, true),
137
- cell('inst-b', 0, true),
138
- cell('inst-b', 1, true),
139
- ]
140
- expect(() =>
141
- buildPremeasuredFromCells({
142
- cells: uncapturedCost,
143
- instances: INSTANCES,
144
- reps: 2,
145
- surfaceHash: 'x',
146
- sourceDir: '/p',
147
- }),
148
- ).toThrow(/has uncaptured cost/)
149
- })
150
- })
151
-
152
- describe('loadFullCampaignCells + incumbentSurfaceHash (real fs/git)', () => {
153
- let dir: string
154
- let repo: string
155
-
156
- beforeEach(async () => {
157
- dir = await mkdtemp(join(tmpdir(), 'premeasured-'))
158
- repo = await mkdtemp(join(tmpdir(), 'premeasured-repo-'))
159
- })
160
-
161
- afterEach(async () => {
162
- await rm(dir, { recursive: true, force: true })
163
- await rm(repo, { recursive: true, force: true })
164
- })
165
-
166
- it('round-trips full cells (costCallIds and manifest preserved) from disk', async () => {
167
- const c = cell('inst-a', 0, true)
168
- await mkdir(join(dir, 'inst-a_0'), { recursive: true })
169
- const { mtimeMs: _m, ...persisted } = c
170
- await writeFile(join(dir, 'inst-a_0', 'cached-result.json'), JSON.stringify(persisted))
171
- const loaded = await loadFullCampaignCells(dir)
172
- expect(loaded).toHaveLength(1)
173
- expect(loaded[0]).toMatchObject({
174
- scenarioId: 'inst-a',
175
- rep: 0,
176
- manifestHash: 'm1',
177
- costCallIds: ['call-inst-a-0'],
178
- seed: 42,
179
- })
180
- expect(typeof loaded[0]!.mtimeMs).toBe('number')
181
- })
182
-
183
- it('computes the same incumbent hash for the unchanged tip (baseCommit == candidateCommit, empty patch)', async () => {
184
- await runOk('git', ['init', '-q', '-b', 'main', repo])
185
- await runOk('git', ['-C', repo, 'config', 'core.hooksPath', '/dev/null'])
186
- await runOk('git', ['-C', repo, 'config', 'user.email', 't@t.dev'])
187
- await runOk('git', ['-C', repo, 'config', 'user.name', 'T'])
188
- await writeFile(join(repo, 'a.txt'), 'x\n')
189
- await runOk('git', ['-C', repo, 'add', '-A'])
190
- await runOk('git', ['-C', repo, 'commit', '-q', '-m', 'init'])
191
- const h1 = await incumbentSurfaceHash(repo, 'main')
192
- const h2 = await incumbentSurfaceHash(repo, 'main')
193
- expect(h1).toBe(h2)
194
- expect(h1).toMatch(/^[0-9a-f]{16}$/)
195
- // A moved tip changes the hash — the fail-loud property the artifact rides.
196
- await writeFile(join(repo, 'a.txt'), 'y\n')
197
- await runOk('git', ['-C', repo, 'add', '-A'])
198
- await runOk('git', ['-C', repo, 'commit', '-q', '-m', 'move'])
199
- expect(await incumbentSurfaceHash(repo, 'main')).not.toBe(h1)
200
- })
201
- })