@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,414 +0,0 @@
1
- /**
2
- * Serialized official judge — the typed port of the experiment's `judge.sh`,
3
- * wrapping `adapter.judge` (via the tracked judge-child.mts) with the three
4
- * protections the bash experiment had to learn the hard way:
5
- *
6
- * 1. ONE JUDGE AT A TIME. The stale-container pre-clean does `docker rm -f`
7
- * by instance short-name; two overlapping judges of the same instance nuke
8
- * each other's container mid-run and each records a spurious
9
- * `resolved:false` (proven false-negative on psf__requests-1766). The lock
10
- * here is two-layer: an in-process promise chain plus a kernel-backed
11
- * `flock`, so the pre-clean can only ever remove a genuinely stale
12
- * container — never a live grade.
13
- *
14
- * 2. CEILING ≥ 1800s. The original 700s ceiling caused 3 spurious timeouts on
15
- * psf__requests-2317 (a successful grade took 1181s). The floor is
16
- * enforced, not advisory: a shorter ceiling for the REAL judge is a
17
- * protocol violation and throws.
18
- *
19
- * 3. ONE RETRY on empty/unparseable judge output (single-run judge flake was
20
- * observed live: byte-identical patches split-verdicted). A second failure
21
- * returns `resolved: null` — inconclusive, never a fabricated verdict —
22
- * matching the RejudgeRow semantics M1 pinned.
23
- */
24
-
25
- import { spawn } from 'node:child_process'
26
- import { open, readFile, type FileHandle } from 'node:fs/promises'
27
- import { tmpdir } from 'node:os'
28
- import { join } from 'node:path'
29
- import { fileURLToPath } from 'node:url'
30
- import { run } from './proc'
31
-
32
- /** Verdict shape — field names match the experiment's JUDGE_RESULT/RejudgeRow rows. */
33
- export interface JudgeVerdict {
34
- iid: string
35
- /** `null` = inconclusive (judge failed twice); never a guessed boolean. */
36
- resolved: boolean | null
37
- score?: number
38
- secs?: number
39
- patch_bytes?: number
40
- note?: string
41
- error?: string
42
- /** First 200 chars of the failing output, for post-mortem (mirrors rejudge rows). */
43
- raw?: string
44
- /** How many judge child runs it took (1 = clean, 2 = retried). */
45
- attempts?: number
46
- }
47
-
48
- export interface JudgeCommand {
49
- bin: string
50
- argv: string[]
51
- cwd: string
52
- }
53
-
54
- export interface SerializedJudgeOptions {
55
- /**
56
- * Hard ceiling per judge child. Default 1_800_000 ms; values below the floor
57
- * throw unless `unsafeAllowShortTimeout` (test hook) is set.
58
- */
59
- timeoutMs?: number
60
- /** Maximum time spent waiting to acquire both judge locks. */
61
- lockWaitTimeoutMs?: number
62
- /** Hard ceiling for each stale-container pre-clean command. */
63
- preCleanTimeoutMs?: number
64
- /** Test-only escape hatch for the 1800s floor. NEVER set on a real judge. */
65
- unsafeAllowShortTimeout?: boolean
66
- /** Cross-process kernel lock file. One per docker daemon. */
67
- lockFile?: string
68
- /** SWEBENCH_CACHE_LEVEL for the child. `instance` = we manage image rotation. */
69
- cacheLevel?: string
70
- /** Override the judge child invocation (tests inject fakes here). */
71
- command?: (iid: string, patchPath: string) => JudgeCommand
72
- }
73
-
74
- export const JUDGE_TIMEOUT_FLOOR_MS = 1_800_000
75
- export const JUDGE_PRE_CLEAN_TIMEOUT_MS = 60_000
76
- /** Process-group settlement and scheduler margin beyond all command ceilings. */
77
- export const JUDGE_LOCK_SETTLEMENT_MARGIN_MS = 30_000
78
- /** One live holder may use two attempts, each with docker ps + docker rm. */
79
- export const JUDGE_LOCK_WAIT_TIMEOUT_MS =
80
- 2 * (JUDGE_TIMEOUT_FLOOR_MS + 2 * JUDGE_PRE_CLEAN_TIMEOUT_MS) + JUDGE_LOCK_SETTLEMENT_MARGIN_MS
81
-
82
- const benchRoot = fileURLToPath(new URL('../..', import.meta.url))
83
- const judgeChild = fileURLToPath(new URL('./judge-child.mts', import.meta.url))
84
-
85
- const defaultCommand = (iid: string, patchPath: string): JudgeCommand => ({
86
- bin: 'node',
87
- argv: ['--import', 'tsx', judgeChild, iid, patchPath],
88
- cwd: benchRoot,
89
- })
90
-
91
- /** judge.sh's `sed 's/.*__//'` — the docker container filter for pre-clean. */
92
- export function instanceShortName(iid: string): string {
93
- return iid.replace(/.*__/, '')
94
- }
95
-
96
- // ---------------------------------------------------------------------------
97
- // Locking: in-process promise chain + cross-process kernel flock.
98
- // ---------------------------------------------------------------------------
99
-
100
- let inProcessChain: Promise<unknown> = Promise.resolve()
101
-
102
- export interface JudgeLockOptions {
103
- /** Maximum total wait across the in-process queue and kernel lock. */
104
- timeoutMs?: number
105
- signal?: AbortSignal
106
- }
107
-
108
- function abortReason(signal: AbortSignal): Error {
109
- if (signal.reason instanceof Error) return signal.reason
110
- const error = new Error('serialized-judge: aborted while waiting for the judge lock')
111
- error.name = 'AbortError'
112
- return error
113
- }
114
-
115
- function throwIfAborted(signal?: AbortSignal): void {
116
- if (signal?.aborted) throw abortReason(signal)
117
- }
118
-
119
- function lockTimeoutError(timeoutMs: number): Error {
120
- return new Error(`serialized-judge: lock wait exceeded ${timeoutMs}ms`)
121
- }
122
-
123
- function waitFor<T>(promise: Promise<T>, deadline: number, timeoutMs: number, signal?: AbortSignal): Promise<T> {
124
- throwIfAborted(signal)
125
- const remainingMs = deadline - Date.now()
126
- if (remainingMs <= 0) return Promise.reject(lockTimeoutError(timeoutMs))
127
-
128
- return new Promise<T>((resolve, reject) => {
129
- let settled = false
130
- const finish = (callback: () => void) => {
131
- if (settled) return
132
- settled = true
133
- clearTimeout(timer)
134
- signal?.removeEventListener('abort', onAbort)
135
- callback()
136
- }
137
- const onAbort = () => finish(() => reject(abortReason(signal!)))
138
- const timer = setTimeout(() => finish(() => reject(lockTimeoutError(timeoutMs))), remainingMs)
139
- signal?.addEventListener('abort', onAbort, { once: true })
140
- if (signal?.aborted) onAbort()
141
- promise.then(
142
- (value) => finish(() => resolve(value)),
143
- (error: unknown) => finish(() => reject(error)),
144
- )
145
- })
146
- }
147
-
148
- /**
149
- * Ask util-linux `flock` to lock an fd inherited from this process. Linux
150
- * associates a flock with the open file description, so the lock remains held
151
- * by `handle` after the short helper exits and is released atomically on close.
152
- * There is no owner record to initialize, inspect, or unlink: process death and
153
- * fd close are the only stale-owner recovery path, both enforced by the kernel.
154
- */
155
- function tryAcquireFlock(
156
- handle: FileHandle,
157
- deadline: number,
158
- timeoutMs: number,
159
- signal?: AbortSignal,
160
- ): Promise<boolean> {
161
- throwIfAborted(signal)
162
- const remainingMs = deadline - Date.now()
163
- if (remainingMs <= 0) return Promise.reject(lockTimeoutError(timeoutMs))
164
-
165
- return new Promise<boolean>((resolve, reject) => {
166
- const waitSeconds = Math.max(0.001, remainingMs / 1_000).toFixed(3)
167
- const child = spawn('flock', ['--exclusive', '--wait', waitSeconds, '3'], {
168
- stdio: ['ignore', 'ignore', 'pipe', handle.fd],
169
- })
170
- let stderr = ''
171
- let pendingError: Error | undefined
172
- let settled = false
173
- const finish = (callback: () => void) => {
174
- if (settled) return
175
- settled = true
176
- clearTimeout(timer)
177
- signal?.removeEventListener('abort', onAbort)
178
- callback()
179
- }
180
- const stopWith = (error: Error) => {
181
- if (pendingError || settled) return
182
- pendingError = error
183
- child.kill('SIGKILL')
184
- }
185
- const onAbort = () => stopWith(abortReason(signal!))
186
- const timer = setTimeout(() => stopWith(lockTimeoutError(timeoutMs)), remainingMs)
187
-
188
- signal?.addEventListener('abort', onAbort, { once: true })
189
- if (signal?.aborted) onAbort()
190
- child.stderr?.on('data', (chunk: Buffer) => {
191
- if (stderr.length < 2_000) stderr += chunk.toString('utf8')
192
- })
193
- child.on('error', (error) => finish(() => reject(pendingError ?? error)))
194
- child.on('close', (code, closeSignal) => {
195
- if (pendingError) {
196
- finish(() => reject(pendingError!))
197
- return
198
- }
199
- if (code === 0) {
200
- finish(() => resolve(true))
201
- return
202
- }
203
- // util-linux flock uses rc=1 when --wait expires without ownership.
204
- if (code === 1) {
205
- finish(() => resolve(false))
206
- return
207
- }
208
- const detail = stderr.trim()
209
- finish(() => reject(new Error(
210
- `serialized-judge: flock failed with ${closeSignal ?? `exit ${code ?? 'unknown'}`}${detail ? `: ${detail}` : ''}`,
211
- )))
212
- })
213
- })
214
- }
215
-
216
- async function acquireFlock(
217
- lockFile: string,
218
- deadline: number,
219
- timeoutMs: number,
220
- signal?: AbortSignal,
221
- ): Promise<() => Promise<void>> {
222
- for (;;) {
223
- throwIfAborted(signal)
224
- if (Date.now() >= deadline) throw lockTimeoutError(timeoutMs)
225
- const handle = await open(lockFile, 'a')
226
- try {
227
- if (await tryAcquireFlock(handle, deadline, timeoutMs, signal)) {
228
- let released = false
229
- return async () => {
230
- if (released) return
231
- released = true
232
- await handle.close()
233
- }
234
- }
235
- } catch (error) {
236
- await handle.close().catch(() => {})
237
- throw error
238
- }
239
- await handle.close()
240
- if (Date.now() >= deadline) throw lockTimeoutError(timeoutMs)
241
- }
242
- }
243
-
244
- /**
245
- * Run `fn` holding BOTH locks. Exported so calibrate/parity paths can pin any
246
- * docker-touching critical section to the same mutex the judge uses.
247
- */
248
- export function withJudgeLock<T>(
249
- lockFile: string,
250
- fn: () => Promise<T>,
251
- opts: JudgeLockOptions = {},
252
- ): Promise<T> {
253
- const timeoutMs = opts.timeoutMs ?? JUDGE_LOCK_WAIT_TIMEOUT_MS
254
- if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) {
255
- return Promise.reject(new Error('serialized-judge: lock timeoutMs must be a positive finite number'))
256
- }
257
- const deadline = Date.now() + timeoutMs
258
- const predecessor = inProcessChain.catch(() => {})
259
- const task = (async () => {
260
- // Cancellation stops this waiter, not the predecessor. A later waiter
261
- // still has to acquire the kernel lock, so the live holder is never bypassed.
262
- await waitFor(predecessor, deadline, timeoutMs, opts.signal)
263
- const release = await acquireFlock(lockFile, deadline, timeoutMs, opts.signal)
264
- try {
265
- throwIfAborted(opts.signal)
266
- return await fn()
267
- } finally {
268
- await release()
269
- }
270
- })()
271
- inProcessChain = task.catch(() => {})
272
- return task
273
- }
274
-
275
- // ---------------------------------------------------------------------------
276
- // The judge itself.
277
- // ---------------------------------------------------------------------------
278
-
279
- export interface SerializedJudge {
280
- judge(iid: string, patchPath: string, tag?: string, signal?: AbortSignal): Promise<JudgeVerdict>
281
- readonly lockFile: string
282
- }
283
-
284
- function parseJudgeResult(stdout: string): Omit<JudgeVerdict, 'attempts'> | undefined {
285
- const line = stdout.split('\n').find((l) => l.includes('JUDGE_RESULT'))
286
- if (!line) return undefined
287
- try {
288
- const parsed = JSON.parse(line.slice(line.indexOf('JUDGE_RESULT') + 'JUDGE_RESULT'.length).trim())
289
- if (typeof parsed !== 'object' || parsed === null || typeof parsed.resolved !== 'boolean') return undefined
290
- return parsed as Omit<JudgeVerdict, 'attempts'>
291
- } catch {
292
- return undefined
293
- }
294
- }
295
-
296
- export function createSerializedJudge(opts: SerializedJudgeOptions = {}): SerializedJudge {
297
- const timeoutMs = opts.timeoutMs ?? JUDGE_TIMEOUT_FLOOR_MS
298
- const usingRealJudge = opts.command === undefined
299
- if (timeoutMs < JUDGE_TIMEOUT_FLOOR_MS && (usingRealJudge || !opts.unsafeAllowShortTimeout)) {
300
- throw new Error(
301
- `serialized-judge: timeoutMs=${timeoutMs} is below the ${JUDGE_TIMEOUT_FLOOR_MS}ms floor ` +
302
- `(700s provably caused 3 spurious timeouts on psf__requests-2317)`,
303
- )
304
- }
305
- const lockFile = opts.lockFile ?? join(tmpdir(), 'swe-arena-judge.lock')
306
- const preCleanTimeoutMs = opts.preCleanTimeoutMs ?? JUDGE_PRE_CLEAN_TIMEOUT_MS
307
- const lockWaitTimeoutMs =
308
- opts.lockWaitTimeoutMs ??
309
- 2 * (timeoutMs + 2 * preCleanTimeoutMs) + JUDGE_LOCK_SETTLEMENT_MARGIN_MS
310
- if (!Number.isFinite(lockWaitTimeoutMs) || lockWaitTimeoutMs <= 0) {
311
- throw new Error('serialized-judge: lockWaitTimeoutMs must be a positive finite number')
312
- }
313
- if (!Number.isFinite(preCleanTimeoutMs) || preCleanTimeoutMs <= 0) {
314
- throw new Error('serialized-judge: preCleanTimeoutMs must be a positive finite number')
315
- }
316
- const cacheLevel = opts.cacheLevel ?? 'instance'
317
- const command = opts.command ?? defaultCommand
318
- // A command override is a hermetic test hook and does not touch Docker.
319
- // Tests that exercise cleanup opt in by setting preCleanTimeoutMs.
320
- const shouldPreClean = usingRealJudge || opts.preCleanTimeoutMs !== undefined
321
-
322
- async function runDockerCleanup(argv: string[], signal?: AbortSignal): Promise<Awaited<ReturnType<typeof run>>> {
323
- let result: Awaited<ReturnType<typeof run>>
324
- try {
325
- result = await run('docker', argv, { timeoutMs: preCleanTimeoutMs, signal })
326
- } catch (error) {
327
- throw new Error(`serialized-judge: docker pre-clean could not start: ${(error as Error).message}`, {
328
- cause: error,
329
- })
330
- }
331
- throwIfAborted(signal)
332
- if (result.timedOut) {
333
- throw new Error(`serialized-judge: docker ${argv[0]} pre-clean timed out after ${preCleanTimeoutMs}ms`)
334
- }
335
- if (result.code !== 0) {
336
- const detail = (result.stderr || result.stdout).trim().slice(0, 1_500)
337
- throw new Error(
338
- `serialized-judge: docker ${argv[0]} pre-clean exited ${result.code}${detail ? `: ${detail}` : ''}`,
339
- )
340
- }
341
- return result
342
- }
343
-
344
- async function preClean(iid: string, signal?: AbortSignal): Promise<void> {
345
- throwIfAborted(signal)
346
- const short = instanceShortName(iid)
347
- const ps = await runDockerCleanup(['ps', '-aq', '--filter', `name=${short}`], signal)
348
- throwIfAborted(signal)
349
- const ids = ps.stdout.split('\n').map((s) => s.trim()).filter(Boolean)
350
- if (ids.length > 0) {
351
- await runDockerCleanup(['rm', '-f', ...ids], signal)
352
- throwIfAborted(signal)
353
- }
354
- }
355
-
356
- async function attempt(
357
- iid: string,
358
- patchPath: string,
359
- signal?: AbortSignal,
360
- ): Promise<{ verdict?: Omit<JudgeVerdict, 'attempts'>; raw: string }> {
361
- // Pre-clean INSIDE the mutex: with judging serialized this can only ever
362
- // remove a stale container, never a live grade.
363
- if (shouldPreClean) await preClean(iid, signal)
364
- throwIfAborted(signal)
365
- const { bin, argv, cwd } = command(iid, patchPath)
366
- const res = await run(bin, argv, {
367
- cwd,
368
- timeoutMs,
369
- signal,
370
- env: { ...process.env, SWEBENCH_CACHE_LEVEL: cacheLevel },
371
- })
372
- throwIfAborted(signal)
373
- const raw = res.timedOut
374
- ? `timeout after ${timeoutMs}ms`
375
- : res.code === 0
376
- ? res.stdout || res.stderr
377
- : `exit ${res.code}: ${res.stdout || res.stderr}`
378
- return {
379
- verdict: !res.timedOut && res.code === 0 ? parseJudgeResult(res.stdout) : undefined,
380
- raw,
381
- }
382
- }
383
-
384
- return {
385
- lockFile,
386
- async judge(iid, patchPath, _tag = 'x', signal) {
387
- throwIfAborted(signal)
388
- // Empty patch never reaches docker — same short-circuit as judge.sh.
389
- const patch = await readFile(patchPath, 'utf8')
390
- throwIfAborted(signal)
391
- if (patch.trim().length === 0) {
392
- return { iid, resolved: false, score: 0, note: 'empty-patch', attempts: 0 }
393
- }
394
- return withJudgeLock(
395
- lockFile,
396
- async () => {
397
- const first = await attempt(iid, patchPath, signal)
398
- if (first.verdict) return { ...first.verdict, attempts: 1 }
399
- throwIfAborted(signal)
400
- const second = await attempt(iid, patchPath, signal)
401
- if (second.verdict) return { ...second.verdict, attempts: 2 }
402
- return {
403
- iid,
404
- resolved: null,
405
- error: 'parse-or-timeout',
406
- raw: second.raw.slice(0, 200),
407
- attempts: 2,
408
- }
409
- },
410
- { timeoutMs: lockWaitTimeoutMs, signal },
411
- )
412
- },
413
- }
414
- }
@@ -1,218 +0,0 @@
1
- /**
2
- * swe-arena — typed replay of the committed SOLO-vs-SUPERVISOR head-to-head
3
- * artifacts (SWE-bench Verified, glm-5.2 both arms).
4
- *
5
- * MILESTONE 1: these types mirror the fixture files byte-for-byte semantics.
6
- * They are the proof-of-faithfulness layer: `reconcile.ts` + `analyze.ts`
7
- * must reproduce the reference `fixtures/analyze.py` output exactly (pinned in
8
- * `replay.test.mts`) before any typed execution path is built on top.
9
- */
10
-
11
- /**
12
- * The fields of one SWE-bench Verified instance we actually consume from
13
- * `task-meta.json` (generated by the experiment's `load_meta.py` from
14
- * princeton-nlp/SWE-bench_Verified). `FAIL_TO_PASS` / `PASS_TO_PASS` are
15
- * JSON-encoded string arrays as shipped by the HF dataset — kept as raw
16
- * strings here; decode at the point of use.
17
- */
18
- export interface SweInstance {
19
- instance_id: string
20
- repo: string
21
- base_commit: string
22
- problem_statement: string
23
- patch: string
24
- test_patch: string
25
- /** JSON-encoded string[] (raw HF dataset encoding). */
26
- FAIL_TO_PASS: string
27
- /** JSON-encoded string[] (raw HF dataset encoding). */
28
- PASS_TO_PASS: string
29
- version: string | null
30
- environment_setup_commit: string | null
31
- }
32
-
33
- /**
34
- * One factory-bench instance — a merged feature PR from our own repo history
35
- * turned into a gradable end-to-end feature-building task (see
36
- * supervisor-lab/factory-bench/docs/design.md). The worker sees only the tree
37
- * at `base_commit` (archive export, synthetic git history) plus the rewritten
38
- * spec; the PR's own added test files are the hidden judge, overlaid from
39
- * `judge_ref` at judge time. Field names mirror the instance dirs'
40
- * `manifest.json` byte-for-byte, same policy as SweInstance vs task-meta.json.
41
- */
42
- export interface FactoryInstance {
43
- /** `factory.<repo>.<pr>` */
44
- id: string
45
- /** `owner/name` */
46
- repo: string
47
- /** Judge-side local mirror; NEVER exposed to the worker workspace. */
48
- repo_local_mirror: string
49
- /** The worker's world — the PR's base commit. */
50
- base_commit: string
51
- /** Merge commit the judge tests are read from (`git show <judge_ref>:<path>`). */
52
- judge_ref: string
53
- /** Instance-dir-relative spec file (PM-ticket grade rewrite of the PR body). */
54
- spec_md: string
55
- /** Hidden judge test files, overlaid at judge time only. */
56
- judge_tests: string[]
57
- /** Flaky/env-dependent tests excluded at calibration, reasons in calibration.md. */
58
- excluded_tests: string[]
59
- /** Immutable Node container image used for setup and judge commands. */
60
- command_image: string
61
- setup_cmds: string[]
62
- judge_cmds: string[]
63
- /** e.g. "all 30 judge tests pass; partial score = passed/30" — the /NN is parsed. */
64
- resolved_criterion: string
65
- timeout_s: number
66
- worker_visible_paths_note?: string
67
- runtime?: string
68
- calibration?: { gold: string; base: string; receipts: string }
69
- }
70
-
71
- /**
72
- * Discriminated instance union for the seams that used to assume SweInstance.
73
- * A tagged wrapper (not a structural union) because both shapes mirror their
74
- * on-disk artifacts byte-for-byte and neither may grow a discriminant field.
75
- */
76
- export type ArenaInstance =
77
- | { kind: 'swe'; instance: SweInstance }
78
- | { kind: 'factory'; instance: FactoryInstance }
79
-
80
- /** The ledger/judge identity: SWE `instance_id` or factory `id`. */
81
- export function arenaInstanceId(a: ArenaInstance): string {
82
- return a.kind === 'swe' ? a.instance.instance_id : a.instance.id
83
- }
84
-
85
- /** Per-run opencode usage breakdown captured on the SOLO arm. */
86
- export interface SoloUsage {
87
- steps: number
88
- in: number
89
- out: number
90
- reasoning: number
91
- cache_w: number
92
- cache_r: number
93
- max_ctx: number
94
- oc_cost: number
95
- total_io: number
96
- }
97
-
98
- /**
99
- * One paired row of `ledger.jsonl` — exactly the schema the experiment wrote.
100
- * `null` values are real telemetry gaps (e.g. `sup_spentTokens: null` when the
101
- * supervisor driver exited rc=3 mid-run), not absent data to be defaulted.
102
- */
103
- export interface LedgerRow {
104
- iid: string
105
- solo_resolved: boolean
106
- sup_resolved: boolean
107
- solo_verify_pass: boolean
108
- sup_verify_pass: boolean
109
- solo_patch_lines: number
110
- sup_patch_lines: number
111
- solo_wall_s: number
112
- sup_wall_s: number
113
- solo_tokens: number
114
- solo_usage: SoloUsage
115
- sup_spentTokens: number | null
116
- sup_spentUsd: number | null
117
- sup_spawned: number
118
- /** Missing on rows written before the field was added (pallets__flask-5014). */
119
- sup_workers?: number
120
- sup_settled: number
121
- sup_subtasks: string[]
122
- sup_delivered: boolean | null
123
- /**
124
- * The fixture ledger observed only completed/running; the driver can also
125
- * settle failed/cancelled (M2 widened the union — the typed execution path
126
- * records those honestly instead of coercing them).
127
- */
128
- sup_status: 'completed' | 'running' | 'failed' | 'cancelled' | null
129
- sup_verdict: 'delivered' | 'no-winner' | 'best-effort' | null
130
- solo_oc_rc: number
131
- sup_driver_rc: number
132
- solo_patch: string
133
- sup_patch: string
134
- /** Free-form correction note (e.g. psf__requests-1766 sup re-judge). */
135
- _note?: string
136
- }
137
-
138
- /**
139
- * Tags carried by `rejudge.jsonl` rows. Gold-family tags grade the OFFICIAL
140
- * gold patch (calibrating the judge itself); final/my-reverify tags are the
141
- * personal authoritative re-judgements of arm patches that override the
142
- * automated ledger verdicts.
143
- */
144
- export type RejudgeTag =
145
- | 'gold-control'
146
- | 'gold'
147
- | 'gold2'
148
- | 'solo-final'
149
- | 'sup-final'
150
- | 'solo-final2'
151
- | 'sup-final2'
152
- | 'my-reverify'
153
-
154
- /** One row of `rejudge.jsonl`. `resolved: null` = the judge run failed to parse (inconclusive). */
155
- export interface RejudgeRow {
156
- iid: string
157
- tag: RejudgeTag
158
- patch: string
159
- resolved: boolean | null
160
- score?: number
161
- secs?: number
162
- patch_bytes?: number
163
- error?: string
164
- raw?: string
165
- }
166
-
167
- /** Evolution-round labels: SUP2/3/4 = rematch.jsonl / rematch2.jsonl / rematch3.jsonl. */
168
- export type RematchArm = 'SUP2' | 'SUP3' | 'SUP4'
169
-
170
- /** One row of a rematch*.jsonl evolution round. */
171
- export interface RematchRow {
172
- iid: string
173
- arm: RematchArm
174
- resolved: boolean
175
- verify_pass: boolean
176
- patch_lines: number
177
- wall_s: number
178
- spawned: number
179
- workers: number
180
- sup_status: 'completed' | 'running'
181
- sup_verdict: 'delivered' | 'no-winner' | 'best-effort' | null
182
- delivered: boolean | null
183
- spentTokens: number | null
184
- }
185
-
186
- /** Declarative arm identity for typed execution paths built on this module. */
187
- export interface ArmSpec {
188
- name: string
189
- kind: 'solo' | 'supervisor'
190
- env: Record<string, string>
191
- provenance: { repo: string; commit: string }
192
- }
193
-
194
- /** One pre-registered holdout instance from `holdout.json`. */
195
- export interface HoldoutEntry {
196
- iid: string
197
- repo: string
198
- gold_official_resolved: boolean
199
- verify_calibrated: boolean
200
- selected_at_commit: string
201
- }
202
-
203
- /**
204
- * The holdout registry: instances selected BEFORE any arm ran (see
205
- * `holdout-preregister.log`), each with its gold patch verified against the
206
- * calibrated judge. Only entries passing both checks are usable.
207
- */
208
- export interface HoldoutRegistry {
209
- entries: HoldoutEntry[]
210
- /** Commit of the supervisor runtime (`loops`) at selection time. */
211
- selectedAtCommit: string
212
- }
213
-
214
- /** Per-instance supervisor worker-session spend from `worker-tokens.json`. */
215
- export interface WorkerTokens {
216
- worker_sessions: number
217
- worker_tok: number
218
- }