@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,729 +0,0 @@
1
- /**
2
- * Experiment runner CLI — the typed replacement for the experiment's
3
- * `orchestrate.sh` + `run-instance.sh`:
4
- *
5
- * tsx src/swe-arena/run-experiment.mts <config.json>
6
- * tsx src/swe-arena/run-experiment.mts --dry-run-parity [workDir]
7
- *
8
- * Per instance, sequentially: ledger-skip resume → endpoint capacity gates
9
- * (supervisor arms gate on the ROUTER path too — probing z.ai alone was the
10
- * proven blind spot) → solo arm → serialized judge → supervisor arm →
11
- * serialized judge → append one typed LedgerRow (M1 schema) to the ledger.
12
- *
13
- * DRY-RUN PARITY (the M2 gate): `--dry-run-parity` executes NO arms and spends
14
- * NO model tokens. It replays patch extraction + official judging for two
15
- * committed fixture patches (pallets__flask-5014 SOLO — resolved;
16
- * pydata__xarray-4687 SUP — unresolved) through materialize → apply →
17
- * extractPatch → serialized-judge, and checks the verdicts against the pinned
18
- * M1 fixtures. Docker time only.
19
- *
20
- * TODO(operator approval): full 12-instance parity re-run — re-execute both
21
- * arms on the same 12 instances through this typed path and diff the resulting
22
- * ledger against fixtures/ledger.jsonl. Costs ~$1 in model spend + ~4h wall;
23
- * do not launch without an explicit operator go.
24
- */
25
-
26
- import { appendFile, mkdir, readFile, rm, writeFile } from 'node:fs/promises'
27
- import { tmpdir } from 'node:os'
28
- import { dirname, isAbsolute, join, resolve } from 'node:path'
29
- import { pathToFileURL, fileURLToPath } from 'node:url'
30
- import { createSweBenchAdapter } from '../benchmarks/swe-bench.ts'
31
- import { exportBaseTree } from './factory-judge-child.mts'
32
- import { runFactoryCommand } from './factory-command-container.ts'
33
- import { loadFactoryInstances, type LoadedFactoryInstance } from './fixtures.ts'
34
- import { run, runOk } from './proc.ts'
35
- import {
36
- extractPatch,
37
- loadExcludes,
38
- runSoloArm,
39
- runSupervisorArm,
40
- type ExecutableArmSpec,
41
- type SecretsEnv,
42
- type SoloArmResult,
43
- type SoloArmSpec,
44
- type SupervisorArmResult,
45
- type SupervisorArmSpec,
46
- } from './arms.ts'
47
- import type { FactoryJudgeResult } from './factory-judge-child.mts'
48
- import { applyPatchWithFallback } from './calibrate.ts'
49
- import { gatesForArmKind, waitForCapacity } from './capacity.ts'
50
- import { materializeWorkspace } from './materialize.ts'
51
- import {
52
- createSerializedJudge,
53
- type JudgeVerdict,
54
- type SerializedJudge,
55
- } from './serialized-judge.ts'
56
- import {
57
- reportSupervisorRound,
58
- writeSupervisorRunReportSafe,
59
- } from '@tangle-network/agent-eval/supervisor-run'
60
- import type { LedgerRow } from './types.ts'
61
-
62
- const fixturesDir = fileURLToPath(new URL('./fixtures', import.meta.url))
63
-
64
- // ---------------------------------------------------------------------------
65
- // Instance images (fixtures/instances.json, vendored from the experiment).
66
- // ---------------------------------------------------------------------------
67
-
68
- export interface InstanceImageEntry {
69
- repo: string
70
- base_commit: string
71
- image: string
72
- environment_setup_commit: string | null
73
- }
74
-
75
- export async function loadInstanceImages(path?: string): Promise<Record<string, InstanceImageEntry>> {
76
- const raw = await readFile(path ?? join(fixturesDir, 'instances.json'), 'utf8')
77
- return JSON.parse(raw) as Record<string, InstanceImageEntry>
78
- }
79
-
80
- // ---------------------------------------------------------------------------
81
- // Ledger row assembly + resume.
82
- // ---------------------------------------------------------------------------
83
-
84
- /**
85
- * One paired LedgerRow from the two arm results + judge verdicts — the exact
86
- * field mapping run-instance.sh wrote. Throws on an inconclusive verdict
87
- * (resolved: null): an infra failure must abort the pair, never be written
88
- * into a boolean column.
89
- */
90
- export function buildLedgerRow(
91
- solo: SoloArmResult,
92
- soloVerdict: JudgeVerdict,
93
- sup: SupervisorArmResult,
94
- supVerdict: JudgeVerdict,
95
- ): LedgerRow {
96
- if (solo.iid !== sup.iid) throw new Error(`ledger row: arm iid mismatch ${solo.iid} vs ${sup.iid}`)
97
- if (soloVerdict.resolved === null || supVerdict.resolved === null) {
98
- throw new Error(
99
- `ledger row ${solo.iid}: inconclusive judge verdict (solo=${soloVerdict.resolved}, sup=${supVerdict.resolved}) — not writing a fabricated boolean`,
100
- )
101
- }
102
- const SUP_STATUSES = ['completed', 'running', 'failed', 'cancelled', null] as const
103
- const SUP_VERDICTS = ['delivered', 'no-winner', 'best-effort', null] as const
104
- if (!SUP_STATUSES.includes(sup.sup_status as (typeof SUP_STATUSES)[number])) {
105
- throw new Error(`ledger row ${solo.iid}: unknown sup_status ${JSON.stringify(sup.sup_status)} — loops contract changed?`)
106
- }
107
- if (!SUP_VERDICTS.includes(sup.sup_verdict as (typeof SUP_VERDICTS)[number])) {
108
- throw new Error(`ledger row ${solo.iid}: unknown sup_verdict ${JSON.stringify(sup.sup_verdict)} — loops contract changed?`)
109
- }
110
- return {
111
- iid: solo.iid,
112
- solo_resolved: soloVerdict.resolved,
113
- sup_resolved: supVerdict.resolved,
114
- solo_verify_pass: solo.verify_pass,
115
- sup_verify_pass: sup.verify_pass,
116
- solo_patch_lines: solo.patch_lines,
117
- sup_patch_lines: sup.patch_lines,
118
- solo_wall_s: solo.wall_s,
119
- sup_wall_s: sup.wall_s,
120
- solo_tokens: solo.usage.total_io,
121
- solo_usage: solo.usage,
122
- sup_spentTokens: sup.spentTokens,
123
- sup_spentUsd: sup.spentUsd,
124
- sup_spawned: sup.spawned,
125
- sup_workers: sup.workers,
126
- sup_settled: sup.settled,
127
- sup_subtasks: sup.subtasks,
128
- sup_delivered: sup.delivered,
129
- sup_status: sup.sup_status as LedgerRow['sup_status'],
130
- sup_verdict: sup.sup_verdict as LedgerRow['sup_verdict'],
131
- solo_oc_rc: solo.oc_rc,
132
- sup_driver_rc: sup.driver_rc,
133
- solo_patch: solo.patchPath,
134
- sup_patch: sup.patchPath,
135
- }
136
- }
137
-
138
- /** iids already present in the ledger (resume-skip, orchestrate.sh semantics). */
139
- export async function ledgerIids(ledgerPath: string): Promise<Set<string>> {
140
- const raw = await readFile(ledgerPath, 'utf8').catch(() => '')
141
- const iids = new Set<string>()
142
- for (const line of raw.split('\n')) {
143
- if (!line.trim()) continue
144
- try {
145
- const row = JSON.parse(line) as { iid?: string }
146
- if (typeof row.iid === 'string') iids.add(row.iid)
147
- } catch {
148
- throw new Error(`corrupt ledger line in ${ledgerPath}: ${line.slice(0, 120)}`)
149
- }
150
- }
151
- return iids
152
- }
153
-
154
- // ---------------------------------------------------------------------------
155
- // Experiment config + loop.
156
- // ---------------------------------------------------------------------------
157
-
158
- export interface ExperimentConfig {
159
- instances: string[]
160
- /** Exactly one solo and one supervisor arm (the paired-ledger contract). */
161
- arms: ExecutableArmSpec[]
162
- ledgerPath: string
163
- outDir: string
164
- secretsDir: string
165
- envFiles: string[]
166
- /** Per-instance self-repro verify scripts: <verifyDir>/<iid>.sh. */
167
- verifyDir: string
168
- /** Override fixtures/instances.json (image map). */
169
- instanceImagesPath?: string
170
- judgeTimeoutMs?: number
171
- gateWaitCeilingMs?: number
172
- /** Probe model id. Defaults per-endpoint in capacity.ts. */
173
- capacityModel?: string
174
- /** Pause between instances (orchestrate.sh: 15s, gentle on the shared key). */
175
- cooldownMs?: number
176
- /**
177
- * Run log the per-cell run-report headline is appended to. Defaults to
178
- * `<outDir>/run.log`; the headline is always echoed to stdout as well, so a
179
- * shell-redirected log gets it either way.
180
- */
181
- runLogPath?: string
182
- }
183
-
184
- function armPair(arms: ExecutableArmSpec[]): { solo: SoloArmSpec; sup: SupervisorArmSpec } {
185
- const solo = arms.filter((a): a is SoloArmSpec => a.kind === 'solo')
186
- const sup = arms.filter((a): a is SupervisorArmSpec => a.kind === 'supervisor')
187
- if (solo.length !== 1 || sup.length !== 1) {
188
- throw new Error(`expected exactly one solo + one supervisor arm, got ${arms.map((a) => a.kind).join(', ')}`)
189
- }
190
- return { solo: solo[0], sup: sup[0] }
191
- }
192
-
193
- export async function runExperiment(config: ExperimentConfig): Promise<void> {
194
- const { solo, sup } = armPair(config.arms)
195
- const secrets: SecretsEnv = { secretsDir: config.secretsDir, envFiles: config.envFiles }
196
- const excludes = await loadExcludes()
197
- const images = await loadInstanceImages(config.instanceImagesPath)
198
- const judge = createSerializedJudge({
199
- ...(config.judgeTimeoutMs !== undefined ? { timeoutMs: config.judgeTimeoutMs } : {}),
200
- })
201
- const adapter = createSweBenchAdapter()
202
- const log = (msg: string): void => console.log(`[${new Date().toISOString().slice(11, 19)}] ${msg}`)
203
-
204
- const done = await ledgerIids(config.ledgerPath)
205
- const pending = config.instances.filter((iid) => !done.has(iid))
206
- for (const iid of config.instances.filter((i) => done.has(i))) log(`SKIP ${iid} (already in ledger)`)
207
- if (pending.length === 0) {
208
- log('nothing to do — all instances already in ledger')
209
- return
210
- }
211
-
212
- // One dataset load for all pending instances (problem statements + metadata).
213
- const tasks = await adapter.loadTasks({ ids: pending, split: 'test' })
214
- const taskById = new Map(tasks.map((t) => [t.id, t]))
215
-
216
- for (const iid of pending) {
217
- const task = taskById.get(iid)
218
- if (!task) throw new Error(`instance ${iid} not found in SWE-bench_Verified`)
219
- const entry = images[iid]
220
- if (!entry) throw new Error(`instance ${iid} has no image mapping (instances.json)`)
221
- const problemStatement = String(task.metadata?.problem_statement ?? '')
222
- if (!problemStatement) throw new Error(`instance ${iid}: empty problem_statement`)
223
-
224
- // Capacity gates: worker path always; router path because a supervisor arm runs.
225
- const gateOpts = {
226
- ...(config.gateWaitCeilingMs !== undefined ? { waitCeilingMs: config.gateWaitCeilingMs } : {}),
227
- ...(config.capacityModel !== undefined ? { model: config.capacityModel } : {}),
228
- onStatus: log,
229
- }
230
- for (const gate of gatesForArmKind('supervisor', secrets, gateOpts)) {
231
- if (!(await waitForCapacity(gate))) {
232
- log(`NO CAPACITY on ${gate.name} within ceiling — stopping before ${iid} (resume later)`)
233
- return
234
- }
235
- }
236
-
237
- const ctx = {
238
- instanceId: iid,
239
- image: entry.image,
240
- baseCommit: entry.base_commit,
241
- problemStatement,
242
- verifyCmd: `bash ${join(config.verifyDir, `${iid}.sh`)}`,
243
- outDir: config.outDir,
244
- secrets,
245
- excludes,
246
- }
247
-
248
- log(`>>> ${iid} SOLO arm`)
249
- const soloResult = await runSoloArm(solo, ctx)
250
- const soloVerdict = await judge.judge(iid, soloResult.patchPath, 'solo')
251
- log(`${iid} SOLO judged: ${JSON.stringify(soloVerdict)}`)
252
-
253
- log(`>>> ${iid} SUP arm`)
254
- const supResult = await runSupervisorArm(sup, ctx)
255
- const supVerdict = await judge.judge(iid, supResult.patchPath, 'sup')
256
- log(`${iid} SUP judged: ${JSON.stringify(supVerdict)}`)
257
-
258
- const row = buildLedgerRow(soloResult, soloVerdict, supResult, supVerdict)
259
- await appendFile(config.ledgerPath, JSON.stringify(row) + '\n')
260
- log(`LEDGER_ROW ${iid} solo=${row.solo_resolved} sup=${row.sup_resolved}`)
261
-
262
- // Deterministic run observability: never hand-grep a journal for steers/waves/
263
- // idle/cost again. Best-effort — a reporting failure can't lose a finished cell.
264
- await writeSupervisorRunReportSafe(join(config.outDir, 'runs', iid, sup.name), {
265
- appendHeadlineTo: config.runLogPath ?? join(config.outDir, 'run.log'),
266
- ledgerPath: config.ledgerPath,
267
- })
268
-
269
- await new Promise((r) => setTimeout(r, config.cooldownMs ?? 15_000))
270
- }
271
-
272
- await reportSupervisorRound(config.outDir, {
273
- appendHeadlineTo: config.runLogPath ?? join(config.outDir, 'run.log'),
274
- ledgerPath: config.ledgerPath,
275
- title: 'Round rollup — paired solo/supervisor experiment',
276
- echo: true,
277
- })
278
- }
279
-
280
- // ---------------------------------------------------------------------------
281
- // Dry-run parity — the M2 gate. No arms, no tokens; docker only.
282
- // ---------------------------------------------------------------------------
283
-
284
- export interface ParityCaseSpec {
285
- iid: string
286
- arm: 'solo' | 'sup'
287
- /** Committed patch fixture, relative to fixtures/ (e.g. patches/x.solo.patch). */
288
- patchFixture: string
289
- }
290
-
291
- export interface ParityCaseResult {
292
- iid: string
293
- arm: 'solo' | 'sup'
294
- applyRc: number
295
- fixtureFiles: string[]
296
- extractedFiles: string[]
297
- extractedPatchLines: number
298
- verdict: JudgeVerdict
299
- }
300
-
301
- /** The two pinned parity cases: one resolved SOLO patch, one unresolved SUP patch. */
302
- export const PARITY_CASES: ParityCaseSpec[] = [
303
- { iid: 'pallets__flask-5014', arm: 'solo', patchFixture: 'patches/pallets__flask-5014.solo.patch' },
304
- { iid: 'pydata__xarray-4687', arm: 'sup', patchFixture: 'patches/pydata__xarray-4687.sup.patch' },
305
- ]
306
-
307
- /** Changed paths of a unified diff (b/ side), for extraction-parity checks. */
308
- export function diffChangedFiles(patch: string): string[] {
309
- const files = new Set<string>()
310
- for (const m of patch.matchAll(/^diff --git a\/.+ b\/(.+)$/gm)) files.add(m[1])
311
- return [...files].sort()
312
- }
313
-
314
- /**
315
- * Replay extraction + judging for committed patches WITHOUT running any arm:
316
- * materialize the instance workspace from its image, apply the committed
317
- * patch, re-extract it via the arms.ts extraction path, then grade the
318
- * re-extracted patch with the serialized judge. Byte-identical output is not
319
- * required (git normalizes); the changed-file set and the official verdict are.
320
- */
321
- export async function replayPatchParity(
322
- cases: ParityCaseSpec[],
323
- opts: { workDir: string; judge?: SerializedJudge; keepWorkspaces?: boolean },
324
- ): Promise<ParityCaseResult[]> {
325
- const judge = opts.judge ?? createSerializedJudge()
326
- const excludes = await loadExcludes()
327
- const images = await loadInstanceImages()
328
- const results: ParityCaseResult[] = []
329
- for (const c of cases) {
330
- const entry = images[c.iid]
331
- if (!entry) throw new Error(`parity: no image mapping for ${c.iid}`)
332
- const fixturePatchPath = join(fixturesDir, c.patchFixture)
333
- const fixturePatch = await readFile(fixturePatchPath, 'utf8')
334
- const ws = join(opts.workDir, `parity-${c.iid}-${c.arm}`)
335
- try {
336
- await materializeWorkspace({
337
- instanceId: c.iid,
338
- image: entry.image,
339
- baseCommit: entry.base_commit,
340
- dest: ws,
341
- })
342
- const applyRc = await applyPatchWithFallback(ws, fixturePatchPath)
343
- if (applyRc !== 0) throw new Error(`parity ${c.iid}: committed patch failed to apply (rc=${applyRc})`)
344
- const extracted = await extractPatch(ws, entry.base_commit, excludes)
345
- const extractedPath = join(opts.workDir, `parity-${c.iid}.${c.arm}.extracted.patch`)
346
- await writeFile(extractedPath, extracted)
347
- const verdict = await judge.judge(c.iid, extractedPath, `parity-${c.arm}`)
348
- results.push({
349
- iid: c.iid,
350
- arm: c.arm,
351
- applyRc,
352
- fixtureFiles: diffChangedFiles(fixturePatch),
353
- extractedFiles: diffChangedFiles(extracted),
354
- extractedPatchLines: extracted.length === 0 ? 0 : extracted.split('\n').length - 1,
355
- verdict,
356
- })
357
- } finally {
358
- if (!opts.keepWorkspaces) await rm(ws, { recursive: true, force: true })
359
- }
360
- }
361
- return results
362
- }
363
-
364
- // ---------------------------------------------------------------------------
365
- // Factory-bench: worker workspace + experiment loop.
366
- //
367
- // The worker cell for a factory instance is the `git archive` export of the
368
- // base commit re-initialized as a FRESH git repo with one synthetic commit —
369
- // worker tooling that expects git works, but `git log`/refs cannot leak the
370
- // real repo's future history (the PR's impl and tests live only on the
371
- // judge-side mirror). SPEC.md (the rewritten PM-ticket spec) is part of that
372
- // initial commit. Everything downstream — arm runners, budgets, serialized
373
- // judge queue/ceiling, ledger resume — is the same machinery as swe-arena.
374
- // ---------------------------------------------------------------------------
375
-
376
- /** Ref name the synthetic initial commit is pinned to; the arm's diff base. */
377
- export const FACTORY_BASE_REF = 'factory-base'
378
-
379
- export interface FactoryWorkspace {
380
- /** Sha of the synthetic initial commit (== FACTORY_BASE_REF). */
381
- syntheticBase: string
382
- }
383
-
384
- /**
385
- * Materialize a worker workspace for a factory instance: archive-export the
386
- * base tree, add SPEC.md, re-init as a synthetic-history repo (single commit,
387
- * no remotes), then pre-run `setup_cmds` so the worker starts on installed
388
- * deps. The real repo's refs/objects are unreachable by construction — the
389
- * leak test greps the workspace for them after setup.
390
- */
391
- export async function materializeFactoryWorkspace(
392
- inst: LoadedFactoryInstance,
393
- dest: string,
394
- opts: { setup?: boolean } = {},
395
- ): Promise<FactoryWorkspace> {
396
- await rm(dest, { recursive: true, force: true })
397
- await mkdir(dirname(dest), { recursive: true })
398
- await exportBaseTree(inst.repo_local_mirror, inst.base_commit, dest)
399
- await writeFile(join(dest, 'SPEC.md'), inst.spec)
400
-
401
- await runOk('git', ['-C', dest, 'init', '-q', '-b', 'work'])
402
- // This synthetic benchmark repository must not execute the host's personal Git hooks.
403
- await runOk('git', ['-C', dest, 'config', 'core.hooksPath', '/dev/null'])
404
- await runOk('git', ['-C', dest, 'config', 'user.email', 'factory-bench@local'])
405
- await runOk('git', ['-C', dest, 'config', 'user.name', 'factory-bench'])
406
- await runOk('git', ['-C', dest, 'add', '-A'])
407
- await runOk('git', ['-C', dest, 'commit', '-q', '-m', 'baseline workspace'])
408
- await runOk('git', ['-C', dest, 'branch', '-f', FACTORY_BASE_REF, 'HEAD'])
409
- const syntheticBase = (await runOk('git', ['-C', dest, 'rev-parse', 'HEAD'])).stdout.trim()
410
- if (syntheticBase === inst.base_commit) {
411
- throw new Error(`factory workspace ${inst.id}: synthetic base equals the real base commit — history leaked`)
412
- }
413
-
414
- if (opts.setup !== false && inst.setup_cmds.length > 0) {
415
- for (const cmd of inst.setup_cmds) {
416
- const res = await runFactoryCommand(dest, cmd, {
417
- image: inst.command_image,
418
- network: 'enabled',
419
- timeoutMs: inst.timeout_s * 1000,
420
- })
421
- if (res.code !== 0) {
422
- throw new Error(
423
- `factory workspace ${inst.id}: setup_cmd failed (rc=${res.code}): ${cmd}\n${(res.stderr || res.stdout).slice(-2000)}`,
424
- )
425
- }
426
- }
427
- }
428
- return { syntheticBase }
429
- }
430
-
431
- /** One appended line of the factory ledger (JSONL, resume key = iid + rep). */
432
- export interface FactoryLedgerRow {
433
- at: string
434
- iid: string
435
- rep: number
436
- arm: string
437
- resolved: boolean
438
- /** passed / calibrated total — the dense partial-credit signal. */
439
- score: number
440
- passed: number | null
441
- total: number | null
442
- verify_pass: boolean
443
- patch_lines: number
444
- wall_s: number
445
- judge_secs: number | null
446
- judge_attempts: number | null
447
- driver_rc: number
448
- sup_status: string | null
449
- sup_verdict: string | null
450
- delivered: boolean | null
451
- spentTokens: number | null
452
- spentUsd: number | null
453
- spawned: number
454
- workers: number
455
- settled: number
456
- patchPath: string
457
- runDir: string
458
- }
459
-
460
- /** `iid#r<rep>` keys already in a factory ledger (resume-skip). */
461
- export async function factoryLedgerKeys(ledgerPath: string): Promise<Set<string>> {
462
- const raw = await readFile(ledgerPath, 'utf8').catch(() => '')
463
- const keys = new Set<string>()
464
- for (const line of raw.split('\n')) {
465
- if (!line.trim()) continue
466
- let row: { iid?: string; rep?: number }
467
- try {
468
- row = JSON.parse(line) as { iid?: string; rep?: number }
469
- } catch {
470
- throw new Error(`corrupt factory ledger line in ${ledgerPath}: ${line.slice(0, 120)}`)
471
- }
472
- if (typeof row.iid === 'string' && typeof row.rep === 'number') keys.add(`${row.iid}#r${row.rep}`)
473
- }
474
- return keys
475
- }
476
-
477
- export interface FactoryArmConfig {
478
- workerModel: string
479
- driverModel: string
480
- budget?: number
481
- maxSandboxes?: number
482
- maxUsd?: number
483
- maxDepth?: number
484
- timeoutMs?: number
485
- envKnobs?: Record<string, string>
486
- }
487
-
488
- export interface FactoryExperimentConfig {
489
- /** Instance-dir root; relative paths resolve against the config file. */
490
- instancesDir: string
491
- /** Manifest ids to run (subset of instancesDir). */
492
- instances: string[]
493
- repsPerInstance: number
494
- armName: string
495
- arm: FactoryArmConfig
496
- /** The loops checkout in the supervisor seat (baseline = loops main). */
497
- loopsRepo: string
498
- ledgerPath: string
499
- outDir: string
500
- secretsDir: string
501
- envFiles: string[]
502
- /**
503
- * Worker-side self-check per instance (the supervisor's internal verify
504
- * gate). NEVER the judge tests — those stay hidden. Default `true` (no gate).
505
- */
506
- verifyCmds?: Record<string, string>
507
- judgeTimeoutMs?: number
508
- cooldownMs?: number
509
- gateWaitCeilingMs?: number
510
- capacityModel?: string
511
- /** Run log the per-cell run-report headline is appended to (default `<outDir>/run.log`). */
512
- runLogPath?: string
513
- }
514
-
515
- const factoryJudgeChildPath = fileURLToPath(new URL('./factory-judge-child.mts', import.meta.url))
516
- const benchRootDir = fileURLToPath(new URL('../..', import.meta.url))
517
-
518
- /**
519
- * Serialized judge whose child is factory-judge-child.mts — same JUDGE_RESULT
520
- * line protocol, queue, retry, and SIGKILL ceiling as the swebench judge. The
521
- * 1800s ceiling floor is kept as the backstop; the child self-enforces the
522
- * manifest's (much smaller) per-command timeout_s inside it.
523
- */
524
- export function createFactoryJudge(
525
- instances: LoadedFactoryInstance[],
526
- opts: { timeoutMs?: number } = {},
527
- ): SerializedJudge {
528
- const dirById = new Map(instances.map((i) => [i.id, i.dir]))
529
- return createSerializedJudge({
530
- ...(opts.timeoutMs !== undefined ? { timeoutMs: opts.timeoutMs } : {}),
531
- lockFile: join(tmpdir(), 'factory-arena-judge.lock'),
532
- command: (iid, patchPath) => {
533
- const dir = dirById.get(iid)
534
- if (!dir) throw new Error(`factory judge: unknown instance id ${iid}`)
535
- return { bin: 'node', argv: ['--import', 'tsx', factoryJudgeChildPath, dir, patchPath], cwd: benchRootDir }
536
- },
537
- })
538
- }
539
-
540
- /**
541
- * Factory gen0 loop: per (instance × rep), sequentially — ledger resume →
542
- * capacity gates → supervisor arm on a factory workspace → factory judge →
543
- * one FactoryLedgerRow appended. Structure mirrors runExperiment.
544
- */
545
- export async function runFactoryExperiment(
546
- config: FactoryExperimentConfig,
547
- opts: { configDir?: string; only?: string[]; repsOverride?: number } = {},
548
- ): Promise<void> {
549
- const log = (msg: string): void => console.log(`[${new Date().toISOString().slice(11, 19)}] ${msg}`)
550
- const baseDir = opts.configDir ?? process.cwd()
551
- const instancesDir = isAbsolute(config.instancesDir) ? config.instancesDir : resolve(baseDir, config.instancesDir)
552
- const all = loadFactoryInstances(instancesDir)
553
- const byId = new Map(all.map((i) => [i.id, i]))
554
- const wanted = (opts.only ?? config.instances).map((id) => {
555
- const inst = byId.get(id)
556
- if (!inst) throw new Error(`factory config: instance ${id} not found under ${instancesDir}`)
557
- return inst
558
- })
559
- const reps = opts.repsOverride ?? config.repsPerInstance
560
- if (!Number.isInteger(reps) || reps < 1) throw new Error(`repsPerInstance must be an integer ≥ 1, got ${reps}`)
561
-
562
- const secrets: SecretsEnv = { secretsDir: config.secretsDir, envFiles: config.envFiles }
563
- const judge = createFactoryJudge(all, {
564
- ...(config.judgeTimeoutMs !== undefined ? { timeoutMs: config.judgeTimeoutMs } : {}),
565
- })
566
- await mkdir(config.outDir, { recursive: true })
567
- await mkdir(dirname(config.ledgerPath), { recursive: true })
568
- const done = await factoryLedgerKeys(config.ledgerPath)
569
-
570
- for (const inst of wanted) {
571
- for (let rep = 0; rep < reps; rep += 1) {
572
- const key = `${inst.id}#r${rep}`
573
- if (done.has(key)) {
574
- log(`SKIP ${key} (already in ledger)`)
575
- continue
576
- }
577
-
578
- const gateOpts = {
579
- ...(config.gateWaitCeilingMs !== undefined ? { waitCeilingMs: config.gateWaitCeilingMs } : {}),
580
- ...(config.capacityModel !== undefined ? { model: config.capacityModel } : {}),
581
- onStatus: log,
582
- }
583
- for (const gate of gatesForArmKind('supervisor', secrets, gateOpts)) {
584
- if (!(await waitForCapacity(gate))) {
585
- log(`NO CAPACITY on ${gate.name} within ceiling — stopping before ${key} (resume later)`)
586
- return
587
- }
588
- }
589
-
590
- const spec: SupervisorArmSpec = {
591
- kind: 'supervisor',
592
- name: config.armName,
593
- workerModel: config.arm.workerModel,
594
- driverModel: config.arm.driverModel,
595
- budget: config.arm.budget,
596
- maxSandboxes: config.arm.maxSandboxes,
597
- maxUsd: config.arm.maxUsd,
598
- maxDepth: config.arm.maxDepth,
599
- ...(config.arm.envKnobs ? { envKnobs: config.arm.envKnobs } : {}),
600
- loopsRepo: config.loopsRepo,
601
- timeoutMs: config.arm.timeoutMs,
602
- }
603
- const armOutDir = join(config.outDir, `rep-${rep}`)
604
- log(`>>> ${config.armName} ${inst.id} rep=${rep}`)
605
- const armRes = await runSupervisorArm(spec, {
606
- instanceId: inst.id,
607
- image: `factory-archive:${inst.id}`,
608
- // The synthetic-history ref, NOT the real base sha: patch extraction
609
- // diffs against the workspace's own single commit.
610
- baseCommit: FACTORY_BASE_REF,
611
- materialize: async (dest) => {
612
- await materializeFactoryWorkspace(inst, dest)
613
- },
614
- problemStatement: inst.spec,
615
- verifyCmd: config.verifyCmds?.[inst.id] ?? 'true',
616
- outDir: armOutDir,
617
- secrets,
618
- // SPEC.md is workspace furniture, not worker product.
619
- excludes: [':(exclude)SPEC.md'],
620
- })
621
-
622
- const factoryRunDir = join(armOutDir, 'runs', inst.id, config.armName)
623
- const { ws: _ws, ...armSummary } = armRes
624
- await writeFile(join(factoryRunDir, 'result.json'), JSON.stringify(armSummary, null, 1)).catch(() => {})
625
-
626
- const verdict = await judge.judge(inst.id, armRes.patchPath, `${config.armName}-r${rep}`)
627
- log(`${inst.id} r${rep} judged: ${JSON.stringify(verdict)}`)
628
- await writeFile(join(factoryRunDir, 'judge.json'), JSON.stringify(verdict, null, 1)).catch(() => {})
629
- if (verdict.resolved === null) {
630
- throw new Error(`inconclusive factory judge verdict for ${key} (${verdict.error ?? 'unknown'}) — not writing a fabricated boolean`)
631
- }
632
- const fv = verdict as JudgeVerdict & Partial<FactoryJudgeResult>
633
- const row: FactoryLedgerRow = {
634
- at: new Date().toISOString(),
635
- iid: inst.id,
636
- rep,
637
- arm: config.armName,
638
- resolved: verdict.resolved,
639
- score: typeof fv.score === 'number' ? fv.score : 0,
640
- passed: typeof fv.passed === 'number' ? fv.passed : null,
641
- total: typeof fv.total === 'number' ? fv.total : null,
642
- verify_pass: armRes.verify_pass,
643
- patch_lines: armRes.patch_lines,
644
- wall_s: armRes.wall_s,
645
- judge_secs: fv.secs ?? null,
646
- judge_attempts: fv.attempts ?? null,
647
- driver_rc: armRes.driver_rc,
648
- sup_status: armRes.sup_status,
649
- sup_verdict: armRes.sup_verdict,
650
- delivered: armRes.delivered,
651
- spentTokens: armRes.spentTokens,
652
- spentUsd: armRes.spentUsd,
653
- spawned: armRes.spawned,
654
- workers: armRes.workers,
655
- settled: armRes.settled,
656
- patchPath: armRes.patchPath,
657
- runDir: factoryRunDir,
658
- }
659
- await appendFile(config.ledgerPath, JSON.stringify(row) + '\n')
660
- log(`LEDGER_ROW ${key} resolved=${row.resolved} score=${row.score} (${row.passed}/${row.total})`)
661
-
662
- await writeSupervisorRunReportSafe(row.runDir, {
663
- appendHeadlineTo: config.runLogPath ?? join(config.outDir, 'run.log'),
664
- ledgerPath: config.ledgerPath,
665
- patchPath: armRes.patchPath,
666
- })
667
-
668
- await new Promise((r) => setTimeout(r, config.cooldownMs ?? 15_000))
669
- }
670
- }
671
-
672
- await reportSupervisorRound(config.outDir, {
673
- appendHeadlineTo: config.runLogPath ?? join(config.outDir, 'run.log'),
674
- ledgerPath: config.ledgerPath,
675
- title: `Round rollup — factory ${config.armName}`,
676
- echo: true,
677
- })
678
- }
679
-
680
- // ---------------------------------------------------------------------------
681
- // CLI.
682
- // ---------------------------------------------------------------------------
683
-
684
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
685
-
686
- if (isMain) {
687
- const [arg, extra] = process.argv.slice(2)
688
- if (arg === '--dry-run-parity') {
689
- const workDir = extra ?? join(process.env.TMPDIR ?? '/tmp', 'swe-arena-parity')
690
- await mkdir(workDir, { recursive: true })
691
- const results = await replayPatchParity(PARITY_CASES, { workDir })
692
- for (const r of results) {
693
- console.log(
694
- `PARITY ${r.iid} [${r.arm}] resolved=${r.verdict.resolved} score=${r.verdict.score} ` +
695
- `files(fixture=${r.fixtureFiles.join(',')} extracted=${r.extractedFiles.join(',')})`,
696
- )
697
- }
698
- } else if (arg === '--factory') {
699
- if (!extra) {
700
- console.error('usage: tsx src/swe-arena/run-experiment.mts --factory <config.json> [--only <iid> ...] [--reps <n>]')
701
- process.exit(2)
702
- }
703
- const rest = process.argv.slice(4)
704
- const only: string[] = []
705
- let repsOverride: number | undefined
706
- for (let i = 0; i < rest.length; i += 1) {
707
- if (rest[i] === '--only' && rest[i + 1]) only.push(rest[(i += 1)]!)
708
- else if (rest[i] === '--reps' && rest[i + 1]) repsOverride = Number(rest[(i += 1)])
709
- else throw new Error(`unknown --factory flag: ${rest[i]}`)
710
- }
711
- const configPath = resolve(extra)
712
- const config = JSON.parse(await readFile(configPath, 'utf8')) as FactoryExperimentConfig
713
- await runFactoryExperiment(config, {
714
- configDir: dirname(configPath),
715
- ...(only.length > 0 ? { only } : {}),
716
- ...(repsOverride !== undefined ? { repsOverride } : {}),
717
- })
718
- } else if (arg && !arg.startsWith('--')) {
719
- const config = JSON.parse(await readFile(arg, 'utf8')) as ExperimentConfig
720
- await runExperiment(config)
721
- } else {
722
- console.error(
723
- 'usage: tsx src/swe-arena/run-experiment.mts <config.json>\n' +
724
- ' tsx src/swe-arena/run-experiment.mts --factory <config.json> [--only <iid> ...] [--reps <n>]\n' +
725
- ' tsx src/swe-arena/run-experiment.mts --dry-run-parity [workDir]',
726
- )
727
- process.exit(2)
728
- }
729
- }