@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,714 +0,0 @@
1
- /**
2
- * Unit tests for the round-4 outer loop's pure protocol logic: the declared
3
- * change-space enforcement, porcelain path parsing, the keep-if-better verdict
4
- * (protocol_v2), the staircase row schema, and the frozen-arm assertion.
5
- * Pure — no arms, no docker, no tokens.
6
- */
7
-
8
- import { existsSync } from 'node:fs'
9
- import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
10
- import { tmpdir } from 'node:os'
11
- import { join } from 'node:path'
12
- import { makeProposalFinding } from '@tangle-network/agent-eval'
13
- import { describe, expect, it } from 'vitest'
14
- import { runOk } from './proc.ts'
15
- import {
16
- addEvalWorktree,
17
- DISPATCH_CLEANUP_GRACE_MS,
18
- DEFAULT_GATE_WAIT_CEILING_MS,
19
- FIXTURES_VERIFY_DIR,
20
- FROZEN_ARM,
21
- INSTANCE_LOCK_FILENAME,
22
- LOOPS_CHANGE_SPACE,
23
- RAW_TRACE_DIAGNOSIS_PATH,
24
- STAIRCASE_SCHEMA,
25
- SUPERVISOR_GATE_COUNT,
26
- acquireInstanceLock,
27
- assertFrozenArm,
28
- assertLaunchEnv,
29
- baselineDriftWarnings,
30
- campaignDispatchCeilingMs,
31
- changeSpaceInstruction,
32
- changeSpaceViolations,
33
- decideVerdict,
34
- defaultRound4Config,
35
- instanceVerdictsFromCells,
36
- isPidAlive,
37
- loopsCandidateVerifier,
38
- normalizeRepoPath,
39
- parseStaircaseRow,
40
- porcelainChangedPaths,
41
- purgeIgnoredArtifacts,
42
- removeEvalWorktree,
43
- replicateCoverageComplete,
44
- resolvedInstanceCount,
45
- round4BuildPrompt,
46
- runWithPostGateClock,
47
- type EvidenceCell,
48
- type ReplicateRun,
49
- type StaircaseRow,
50
- } from './outer-loop.mts'
51
-
52
- describe('changeSpaceViolations', () => {
53
- it('accepts the declared change-space, including nested extension paths', () => {
54
- expect(
55
- changeSpaceViolations([
56
- 'extensions/pi/loops.ts',
57
- 'extensions/pi/prompts/worker-coding-system.md',
58
- 'extensions/pi/deep/new-module.ts',
59
- 'src/worker-evidence.ts',
60
- 'src/best-effort.ts',
61
- 'src/worker-clone.ts',
62
- RAW_TRACE_DIAGNOSIS_PATH,
63
- './src/worker-clone.ts',
64
- ]),
65
- ).toEqual([])
66
- })
67
-
68
- it('rejects everything outside the declared space', () => {
69
- expect(
70
- changeSpaceViolations([
71
- 'src/runner.ts',
72
- 'src/strategy-loop.ts',
73
- 'package.json',
74
- 'extensions/other/loops.ts',
75
- 'tests/top-model.test.ts',
76
- 'src/worker-evidence.ts.bak',
77
- ]),
78
- ).toEqual([
79
- 'src/runner.ts',
80
- 'src/strategy-loop.ts',
81
- 'package.json',
82
- 'extensions/other/loops.ts',
83
- 'tests/top-model.test.ts',
84
- 'src/worker-evidence.ts.bak',
85
- ])
86
- })
87
-
88
- it('is not fooled by prefix-sharing directories (extensions/pi2 is out)', () => {
89
- expect(changeSpaceViolations(['extensions/pi2/loops.ts'])).toEqual(['extensions/pi2/loops.ts'])
90
- })
91
-
92
- it('fails closed on traversal, absolute, and empty paths', () => {
93
- expect(changeSpaceViolations(['extensions/pi/../../package.json'])).toEqual([
94
- 'extensions/pi/../../package.json',
95
- ])
96
- expect(changeSpaceViolations(['/etc/passwd'])).toEqual(['/etc/passwd'])
97
- expect(changeSpaceViolations([''])).toEqual([''])
98
- })
99
-
100
- it('normalizes quoted and backslashed paths before matching', () => {
101
- expect(normalizeRepoPath('"extensions/pi/a b.ts"')).toBe('extensions/pi/a b.ts')
102
- expect(normalizeRepoPath('extensions\\pi\\loops.ts')).toBe('extensions/pi/loops.ts')
103
- expect(normalizeRepoPath('../outside.ts')).toBeNull()
104
- expect(changeSpaceViolations(['"extensions/pi/a b.ts"'])).toEqual([])
105
- })
106
- })
107
-
108
- describe('porcelainChangedPaths', () => {
109
- it('parses modified, untracked, and rename entries (both rename sides)', () => {
110
- const stdout = [
111
- ' M extensions/pi/loops.ts',
112
- '?? .improve/raw-trace-diagnosis.md',
113
- 'R src/worker-clone.ts -> src/worker-clone-2.ts',
114
- 'A src/best-effort.ts',
115
- '',
116
- ].join('\n')
117
- expect(porcelainChangedPaths(stdout)).toEqual([
118
- 'extensions/pi/loops.ts',
119
- '.improve/raw-trace-diagnosis.md',
120
- 'src/worker-clone.ts',
121
- 'src/worker-clone-2.ts',
122
- 'src/best-effort.ts',
123
- ])
124
- })
125
-
126
- it('a rename OUT of the change-space is caught end-to-end', () => {
127
- const paths = porcelainChangedPaths('R src/worker-clone.ts -> src/worker-clone-moved.ts\n')
128
- expect(changeSpaceViolations(paths)).toEqual(['src/worker-clone-moved.ts'])
129
- })
130
- })
131
-
132
- describe('decideVerdict (protocol_v2 keep-if-better)', () => {
133
- const base = {
134
- violations: [] as string[],
135
- coverageComplete: true,
136
- resolvedCount: 2,
137
- parentResolvedCount: 1,
138
- costRatio: 1.0,
139
- costGuardRatio: 1.2,
140
- }
141
- it('accepts a gaining, in-space, in-budget candidate', () => {
142
- expect(decideVerdict(base)).toBe('accepted')
143
- })
144
- it('rejects out-of-space before anything else', () => {
145
- expect(decideVerdict({ ...base, violations: ['package.json'] })).toBe('rejected-out-of-space')
146
- })
147
- it('rejects incomplete coverage (an errored cell can never promote)', () => {
148
- expect(decideVerdict({ ...base, coverageComplete: false })).toBe('rejected-incomplete')
149
- })
150
- it('requires a STRICT improvement-set gain (tie = reject)', () => {
151
- expect(decideVerdict({ ...base, resolvedCount: 1 })).toBe('rejected-no-gain')
152
- expect(decideVerdict({ ...base, resolvedCount: 0 })).toBe('rejected-no-gain')
153
- })
154
- it('rejects on the +20% cost guard and on unprovable cost', () => {
155
- expect(decideVerdict({ ...base, costRatio: 1.21 })).toBe('rejected-cost')
156
- expect(decideVerdict({ ...base, costRatio: null })).toBe('rejected-cost')
157
- expect(decideVerdict({ ...base, costRatio: 1.2 })).toBe('accepted')
158
- })
159
- })
160
-
161
- describe('staircase row schema', () => {
162
- const row: StaircaseRow = {
163
- schema: STAIRCASE_SCHEMA,
164
- round: 4,
165
- generation: 0,
166
- runId: 'r4-abc123',
167
- at: '2026-07-15T00:00:00.000Z',
168
- candidate: 'sha256:cand',
169
- candidateCommit: 'deadbeef00',
170
- parent: 'sha256:parent',
171
- parentResolvedCount: 1,
172
- label: 'placement-aware settle',
173
- rationale: 'diagnosis: fix placement mismatch (3/3 analysts)',
174
- changedFiles: ['extensions/pi/loops.ts'],
175
- changeSpaceViolations: [],
176
- perInstance: [
177
- {
178
- iid: 'django__django-11532',
179
- rep: 0,
180
- resolved: true,
181
- verify_pass: true,
182
- patch_lines: 47,
183
- wall_s: 900,
184
- spentTokens: 54623,
185
- recoveredTokens: 61000,
186
- judgeAttempts: 1,
187
- costUsd: 0.12,
188
- },
189
- ],
190
- resolvedCount: 2,
191
- coverageComplete: true,
192
- wallS: 2700,
193
- baselineWallS: 2500,
194
- costRatio: 1.08,
195
- costGuardRatio: 1.2,
196
- internallyPromoted: true,
197
- verdict: 'accepted',
198
- holdout: 'operator-approval-required',
199
- armProvenance: { repo: '/tmp/eval-wt', commit: 'deadbeef00' },
200
- diffPath: '/tmp/out/candidates/deadbeef00.patch',
201
- diffSha256: 'sha256:aaaa',
202
- }
203
-
204
- it('round-trips through JSONL', () => {
205
- const parsed = parseStaircaseRow(JSON.stringify(row))
206
- expect(parsed).toEqual(row)
207
- })
208
-
209
- it('rejects schema drift, bad verdicts, and missing fields', () => {
210
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, schema: 'v0' }))).toThrow(/unknown schema/)
211
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, verdict: 'kept' }))).toThrow(/unknown verdict/)
212
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, resolvedCount: '2' }))).toThrow(/must be a number/)
213
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, runId: '' }))).toThrow(/runId/)
214
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, perInstance: 'x' }))).toThrow(/perInstance/)
215
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, costRatio: 'high' }))).toThrow(/costRatio/)
216
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, internallyPromoted: 'yes' }))).toThrow(/booleans/)
217
- })
218
-
219
- it('accepts a null-cost rejected dot (telemetry gap is data, not a zero)', () => {
220
- const dot = { ...row, costRatio: null, verdict: 'rejected-cost' as const, internallyPromoted: false }
221
- expect(parseStaircaseRow(JSON.stringify(dot)).costRatio).toBeNull()
222
- })
223
-
224
- it('accepts a gen-3 prefilter kill dot with its killReason', () => {
225
- const dot = {
226
- ...row,
227
- candidate: 'prefilter-kill:aaaabbbbcccc',
228
- candidateCommit: null,
229
- verdict: 'rejected-prefilter' as const,
230
- internallyPromoted: false,
231
- coverageComplete: false,
232
- resolvedCount: 0,
233
- perInstance: [],
234
- costRatio: null,
235
- killReason: 'smoke: smoke astropy__astropy-13033: resolved=false — below the mechanism bar',
236
- armProvenance: null,
237
- }
238
- const parsed = parseStaircaseRow(JSON.stringify(dot))
239
- expect(parsed.verdict).toBe('rejected-prefilter')
240
- expect(parsed.killReason).toContain('below the mechanism bar')
241
- })
242
- })
243
-
244
- describe('frozen arm + default config', () => {
245
- it('passes on the round-3 frozen arm and the default config', () => {
246
- expect(() => assertFrozenArm(FROZEN_ARM)).not.toThrow()
247
- expect(() => assertFrozenArm(defaultRound4Config().arm)).not.toThrow()
248
- })
249
- it('throws on any immutable-arm drift (protocol_v2)', () => {
250
- expect(() => assertFrozenArm({ ...FROZEN_ARM, workerModel: 'gpt-5.5' })).toThrow(/immutable/)
251
- expect(() => assertFrozenArm({ ...FROZEN_ARM, maxUsd: 16 })).toThrow(/maxUsd/)
252
- expect(() => assertFrozenArm({ ...FROZEN_ARM, budget: 80 })).toThrow(/budget/)
253
- })
254
- it('default config: improvement set and holdout are the pre-registered, disjoint sets', () => {
255
- const config = defaultRound4Config()
256
- expect(config.instances).toEqual([
257
- 'astropy__astropy-13033',
258
- 'django__django-11532',
259
- 'matplotlib__matplotlib-20826',
260
- ])
261
- expect(config.holdoutInstances).toHaveLength(6)
262
- expect(config.instances.filter((i) => config.holdoutInstances.includes(i))).toEqual([])
263
- expect(config.roundsDir).toBe('/home/drew/code/supervisor-lab/.evolve/rounds')
264
- expect(config.analystModels.every((m) => m === 'glm-5.2')).toBe(true)
265
- })
266
- it('default config: verify scripts come from the COMMITTED fixtures dir and reps=2', () => {
267
- const config = defaultRound4Config()
268
- // The scratchpad copy died with a host reboot; the committed dir is the durable home.
269
- expect(config.verifyDir).toBe(FIXTURES_VERIFY_DIR)
270
- expect(config.verifyDir).toContain('fixtures/verify')
271
- // Single-rep scoring flips instance outcomes run-to-run — round 4 runs 2.
272
- expect(config.repsPerInstance).toBe(2)
273
- })
274
- it('default config: the premeasured baseline artifact path is required-with-default', () => {
275
- const config = defaultRound4Config()
276
- expect(config.premeasuredBaselinePath).toContain('/r4/premeasured-baseline.json')
277
- })
278
- it('default config: author-shot timeout doubled after 3 gen-1 timeouts; outDir name is overridable', () => {
279
- const config = defaultRound4Config()
280
- // 20-min shots died 3× under degraded capacity ("author shot timed out").
281
- expect(config.proposerTimeoutMs).toBe(2_400_000)
282
- expect(defaultRound4Config(undefined, { outDirName: 'r4-gen2' }).outDir.endsWith('/r4-gen2')).toBe(true)
283
- expect(config.outDir.endsWith('/r4')).toBe(true)
284
- })
285
- })
286
-
287
- describe('dispatch clocks (gate holds are never billed to the cell)', () => {
288
- it('starts neither capacity checks nor work for an already-aborted caller', async () => {
289
- const controller = new AbortController()
290
- controller.abort(new Error('cancelled before cell'))
291
- let gates = 0
292
- let work = 0
293
- await expect(runWithPostGateClock({
294
- awaitGates: async () => { gates += 1 },
295
- work: async () => { work += 1; return 'x' },
296
- timeoutMs: 1_000,
297
- signal: controller.signal,
298
- })).rejects.toThrow('cancelled before cell')
299
- expect(gates).toBe(0)
300
- expect(work).toBe(0)
301
- })
302
-
303
- it('passes parent cancellation through work and waits for its cleanup', async () => {
304
- const controller = new AbortController()
305
- let cleanupFinished = false
306
- let markStarted!: () => void
307
- const started = new Promise<void>((resolve) => { markStarted = resolve })
308
- const running = runWithPostGateClock({
309
- awaitGates: async (signal) => signal?.throwIfAborted(),
310
- work: (signal) => new Promise<string>((resolve) => {
311
- markStarted()
312
- signal.addEventListener('abort', () => {
313
- setTimeout(() => {
314
- cleanupFinished = true
315
- resolve('settled')
316
- }, 25)
317
- }, { once: true })
318
- }),
319
- timeoutMs: 1_000,
320
- signal: controller.signal,
321
- })
322
- await started
323
- controller.abort(new Error('operator interrupted'))
324
- await expect(running).rejects.toThrow('operator interrupted')
325
- expect(cleanupFinished).toBe(true)
326
- })
327
-
328
- it('campaignDispatchCeilingMs includes gate holds, both judge attempts, and cleanup', () => {
329
- expect(campaignDispatchCeilingMs({ dispatchTimeoutMs: 7_200_000 })).toBe(
330
- 7_200_000 + SUPERVISOR_GATE_COUNT * DEFAULT_GATE_WAIT_CEILING_MS + 2 * 1_800_000 + DISPATCH_CLEANUP_GRACE_MS,
331
- )
332
- expect(campaignDispatchCeilingMs({
333
- dispatchTimeoutMs: 1_000,
334
- gateWaitCeilingMs: 500,
335
- judgeTimeoutMs: 2_000,
336
- })).toBe(
337
- 1_000 + 2 * 500 + 2 * 2_000 + DISPATCH_CLEANUP_GRACE_MS,
338
- )
339
- })
340
-
341
- it('a gate hold LONGER than the work clock does not abort the cell (the pre-crash bug)', async () => {
342
- // Pre-crash failure shape: 58-min capacity hold billed to the 7200s clock.
343
- // Here: gate hold 120ms > work clock 60ms; the work itself takes 10ms.
344
- const result = await runWithPostGateClock({
345
- awaitGates: () => new Promise<void>((r) => setTimeout(r, 120)),
346
- work: () => new Promise<string>((r) => setTimeout(() => r('done'), 10)),
347
- timeoutMs: 60,
348
- })
349
- expect(result).toBe('done')
350
- })
351
-
352
- it('work exceeding the post-gate clock still fails loud', async () => {
353
- await expect(
354
- runWithPostGateClock({
355
- awaitGates: () => Promise.resolve(),
356
- work: () => new Promise<string>((r) => setTimeout(() => r('late'), 200)),
357
- timeoutMs: 30,
358
- label: 'R4 deadbeef00 astropy__astropy-13033 r0',
359
- }),
360
- ).rejects.toThrow(/post-gate dispatch exceeded 30ms .*astropy__astropy-13033/)
361
- })
362
-
363
- it('waits for abort cleanup before reporting a post-gate timeout', async () => {
364
- let cleanupFinished = false
365
- const started = Date.now()
366
- await expect(
367
- runWithPostGateClock({
368
- awaitGates: () => Promise.resolve(),
369
- work: (signal) => new Promise<string>((resolve) => {
370
- signal.addEventListener('abort', () => {
371
- setTimeout(() => {
372
- cleanupFinished = true
373
- resolve('settled after cleanup')
374
- }, 30)
375
- }, { once: true })
376
- }),
377
- timeoutMs: 20,
378
- label: 'cleanup proof',
379
- }),
380
- ).rejects.toThrow(/post-gate dispatch exceeded 20ms .*cleanup proof/)
381
- expect(cleanupFinished).toBe(true)
382
- expect(Date.now() - started).toBeGreaterThanOrEqual(45)
383
- })
384
-
385
- it('preserves a process-cleanup failure after the dispatch clock expires', async () => {
386
- await expect(
387
- runWithPostGateClock({
388
- awaitGates: () => Promise.resolve(),
389
- work: (signal) => new Promise<never>((_resolve, reject) => {
390
- signal.addEventListener('abort', () => {
391
- reject(new Error('process group 123 survived SIGKILL'))
392
- }, { once: true })
393
- }),
394
- timeoutMs: 20,
395
- label: 'cleanup failure proof',
396
- }),
397
- ).rejects.toThrow(/post-gate dispatch exceeded 20ms .*process group 123 survived SIGKILL/)
398
- })
399
-
400
- it('a gate failure rejects before the work clock ever starts', async () => {
401
- let workStarted = false
402
- await expect(
403
- runWithPostGateClock({
404
- awaitGates: () => Promise.reject(new Error('no capacity on router within ceiling')),
405
- work: async () => {
406
- workStarted = true
407
- return 'x'
408
- },
409
- timeoutMs: 1_000,
410
- }),
411
- ).rejects.toThrow(/no capacity/)
412
- expect(workStarted).toBe(false)
413
- })
414
- })
415
-
416
- describe('replicate semantics (repsPerInstance)', () => {
417
- const iids = ['a', 'b', 'c']
418
- const run = (iid: string, resolved: boolean | null): ReplicateRun => ({ iid, resolved })
419
-
420
- it('an instance resolves only when ALL replicates resolve (AND, fail-closed)', () => {
421
- const runs = [
422
- run('a', true), run('a', true), // both reps resolved → counts
423
- run('b', true), run('b', false), // flaky split → does NOT count
424
- run('c', false), run('c', false),
425
- ]
426
- expect(resolvedInstanceCount(runs, iids, 2)).toBe(1)
427
- })
428
-
429
- it('missing replicates never count as resolved', () => {
430
- expect(resolvedInstanceCount([run('a', true)], iids, 2)).toBe(0)
431
- expect(resolvedInstanceCount([run('a', true)], iids, 1)).toBe(1)
432
- })
433
-
434
- it('coverage requires every replicate of every instance with a conclusive verdict', () => {
435
- const full = iids.flatMap((iid) => [run(iid, true), run(iid, false)])
436
- expect(replicateCoverageComplete(full, iids, 2)).toBe(true)
437
- expect(replicateCoverageComplete(full.slice(1), iids, 2)).toBe(false)
438
- const inconclusive = [...full.slice(0, 5), run('c', null)]
439
- expect(replicateCoverageComplete(inconclusive, iids, 2)).toBe(false)
440
- })
441
- })
442
-
443
- describe('premeasured-baseline drift (the validated artifact is the only denominator)', () => {
444
- const iids = ['astropy__astropy-13033', 'django__django-11532', 'matplotlib__matplotlib-20826']
445
- // AND-verdicts of the premeasured artifact's campaign (measured 1/3).
446
- const expected = {
447
- 'astropy__astropy-13033': false,
448
- 'django__django-11532': false,
449
- 'matplotlib__matplotlib-20826': true,
450
- }
451
- const run = (iid: string, resolved: boolean | null): ReplicateRun => ({ iid, resolved })
452
- const cell = (iid: string, rep: number, resolved: boolean | null): EvidenceCell => ({
453
- scenarioId: iid,
454
- rep,
455
- artifact:
456
- resolved === null
457
- ? null
458
- : {
459
- kind: 'swe-arm',
460
- iid,
461
- commit: 'basecommit0',
462
- resolved,
463
- verifyPass: resolved,
464
- patchLines: 1,
465
- wallS: 10,
466
- spentTokens: null,
467
- spentUsd: null,
468
- recoveredTokens: null,
469
- workerTokIn: null,
470
- workerTokOut: null,
471
- judgeAttempts: null,
472
- judgeWallS: null,
473
- runDir: '/tmp/none',
474
- patchPath: '/tmp/none.patch',
475
- },
476
- ...(resolved === null ? { error: 'inconclusive' } : {}),
477
- })
478
-
479
- it('instanceVerdictsFromCells ANDs replicates and omits incomplete instances', () => {
480
- const cells = [
481
- cell(iids[0]!, 0, false), cell(iids[0]!, 1, false),
482
- cell(iids[1]!, 0, true), cell(iids[1]!, 1, false), // flaky → AND false
483
- cell(iids[2]!, 0, true), // partial → omitted
484
- ]
485
- expect(instanceVerdictsFromCells(cells, iids, 2)).toEqual({
486
- 'astropy__astropy-13033': false,
487
- 'django__django-11532': false,
488
- })
489
- const inconclusive = [cell(iids[0]!, 0, true), cell(iids[0]!, 1, null)]
490
- expect(instanceVerdictsFromCells(inconclusive, [iids[0]!], 2)).toEqual({})
491
- })
492
-
493
- it('drift warnings fire on a reps-complete contradiction, both directions', () => {
494
- const runs = [
495
- run(iids[0]!, true), run(iids[0]!, true), // premeasured false, cached true → drift
496
- run(iids[1]!, false), run(iids[1]!, false), // premeasured false, cached false → quiet
497
- run(iids[2]!, true), run(iids[2]!, false), // premeasured true, cached false → drift
498
- ]
499
- const warnings = baselineDriftWarnings(expected, runs, iids, 2)
500
- expect(warnings).toHaveLength(2)
501
- expect(warnings[0]).toContain('astropy__astropy-13033: premeasured=false')
502
- expect(warnings[1]).toContain('matplotlib__matplotlib-20826: premeasured=true')
503
- expect(warnings.every((w) => w.includes('premeasured artifact rules'))).toBe(true)
504
- })
505
-
506
- it('a partial or inconclusive cached record has no AND-verdict — no drift claim', () => {
507
- // Partial coverage (1 of 2 reps per instance): no AND-verdict exists yet,
508
- // even where the single present cell disagrees with the artifact.
509
- const partial = [run(iids[1]!, true), run(iids[2]!, false)]
510
- expect(baselineDriftWarnings(expected, partial, iids, 2)).toEqual([])
511
- const inconclusive = [run(iids[2]!, true), run(iids[2]!, null)]
512
- expect(baselineDriftWarnings(expected, inconclusive, iids, 2)).toEqual([])
513
- })
514
- })
515
-
516
- describe('launch guards', () => {
517
- it('assertLaunchEnv refuses when either key is absent or blank, naming dotenvx', () => {
518
- expect(() => assertLaunchEnv({})).toThrow(/TANGLE_API_KEY \+ ZAI_API_KEY absent .*dotenvx/)
519
- expect(() => assertLaunchEnv({ TANGLE_API_KEY: 'x' })).toThrow(/ZAI_API_KEY/)
520
- expect(() => assertLaunchEnv({ TANGLE_API_KEY: 'x', ZAI_API_KEY: ' ' })).toThrow(/ZAI_API_KEY/)
521
- expect(() => assertLaunchEnv({ TANGLE_API_KEY: 'x', ZAI_API_KEY: 'y' })).not.toThrow()
522
- })
523
-
524
- it('isPidAlive: own pid is alive; an absurd pid is not', () => {
525
- expect(isPidAlive(process.pid)).toBe(true)
526
- expect(isPidAlive(2 ** 30)).toBe(false)
527
- })
528
-
529
- it('instance lock: acquire, refuse a live second instance, release', async () => {
530
- const dir = await mkdtemp(join(tmpdir(), 'r4-lock-'))
531
- try {
532
- const lock = await acquireInstanceLock(dir)
533
- expect(lock.path).toBe(join(dir, INSTANCE_LOCK_FILENAME))
534
- expect((await readFile(lock.path, 'utf8')).trim()).toBe(String(process.pid))
535
- // A DIFFERENT live pid (init/pid 1 is always alive) must be refused.
536
- await writeFile(lock.path, '1\n')
537
- await expect(acquireInstanceLock(dir)).rejects.toThrow(/pid 1.*refusing to race/s)
538
- await writeFile(lock.path, `${process.pid}\n`)
539
- await lock.release()
540
- expect(existsSync(lock.path)).toBe(false)
541
- } finally {
542
- await rm(dir, { recursive: true, force: true })
543
- }
544
- })
545
-
546
- it('instance lock: a stale lock (dead pid or garbage) is reclaimed', async () => {
547
- const dir = await mkdtemp(join(tmpdir(), 'r4-lock-stale-'))
548
- try {
549
- await writeFile(join(dir, INSTANCE_LOCK_FILENAME), `${2 ** 30}\n`)
550
- const lock = await acquireInstanceLock(dir)
551
- expect((await readFile(lock.path, 'utf8')).trim()).toBe(String(process.pid))
552
- await lock.release()
553
- await writeFile(join(dir, INSTANCE_LOCK_FILENAME), 'not-a-pid\n')
554
- const lock2 = await acquireInstanceLock(dir)
555
- expect((await readFile(lock2.path, 'utf8')).trim()).toBe(String(process.pid))
556
- await lock2.release()
557
- } finally {
558
- await rm(dir, { recursive: true, force: true })
559
- }
560
- })
561
-
562
- it('release only removes a lock this instance still owns', async () => {
563
- const dir = await mkdtemp(join(tmpdir(), 'r4-lock-own-'))
564
- try {
565
- const lock = await acquireInstanceLock(dir)
566
- await writeFile(lock.path, '424242\n') // another instance reclaimed it
567
- await lock.release()
568
- expect((await readFile(lock.path, 'utf8')).trim()).toBe('424242')
569
- } finally {
570
- await rm(dir, { recursive: true, force: true })
571
- }
572
- })
573
- })
574
-
575
- describe('round4BuildPrompt', () => {
576
- it('declares the change-space and renders findings', () => {
577
- const prompt = round4BuildPrompt({
578
- findings: [
579
- makeProposalFinding({
580
- analyst_id: 'test',
581
- severity: 'high',
582
- area: 'mechanism',
583
- claim: 'fix placement mismatch',
584
- recommended_action: 'settle where maintainers expect',
585
- evidence_refs: [],
586
- confidence: 1,
587
- proposal_origin: 'search',
588
- }),
589
- ],
590
- })
591
- expect(prompt).toContain('DECLARED CHANGE-SPACE')
592
- expect(prompt).toContain('extensions/pi/**')
593
- expect(prompt).toContain('src/worker-evidence.ts')
594
- expect(prompt).toContain('fix placement mismatch')
595
- expect(prompt).toContain('→ settle where maintainers expect')
596
- // No raw-trace findings ⇒ no evidence-file requirement block (the
597
- // change-space instruction still NAMES the artifact path as allowed).
598
- expect(prompt).not.toContain('Raw trace evidence requirement')
599
- })
600
-
601
- it('adds the raw-trace evidence contract when raw-trace findings are present', () => {
602
- const prompt = round4BuildPrompt({
603
- findings: [
604
- makeProposalFinding({
605
- analyst_id: 'raw-trace-distiller',
606
- severity: 'high',
607
- area: 'raw-trace-context',
608
- claim: 'traces at /run/gen-0',
609
- evidence_refs: [],
610
- confidence: 1,
611
- proposal_origin: 'search',
612
- }),
613
- ],
614
- })
615
- expect(prompt).toContain('Raw trace evidence requirement')
616
- expect(prompt).toContain(RAW_TRACE_DIAGNOSIS_PATH)
617
- })
618
-
619
- it('changeSpaceInstruction names every allowed root exactly once', () => {
620
- const text = changeSpaceInstruction(LOOPS_CHANGE_SPACE)
621
- for (const f of LOOPS_CHANGE_SPACE.files) expect(text).toContain(f)
622
- for (const p of LOOPS_CHANGE_SPACE.prefixes) expect(text).toContain(`${p}**`)
623
- })
624
- })
625
-
626
- describe('candidate worktree hygiene (finalize precondition + eval isolation)', () => {
627
- const git = (dir: string, ...argv: string[]) =>
628
- runOk('git', ['-C', dir, '-c', 'user.email=t@test', '-c', 'user.name=t', '-c', 'core.hooksPath=/dev/null', ...argv])
629
-
630
- /** Base repo + candidate worktree with proposer-style state: a tracked edit,
631
- * an untracked non-ignored deliverable, and gitignored install dirt (the
632
- * exact mix round-4 gen-0 cand-1 died on). */
633
- async function makeRepoWithDirtyCandidate(): Promise<{ base: string; wt: string; root: string }> {
634
- const root = await mkdtemp(join(tmpdir(), 'r4-wt-hygiene-'))
635
- const base = join(root, 'base')
636
- await mkdir(join(base, 'src'), { recursive: true })
637
- await writeFile(join(base, '.gitignore'), 'node_modules/\n*.log\n')
638
- await writeFile(join(base, 'src', 'a.ts'), 'export const a = 1\n')
639
- await runOk('git', ['-C', base, 'init', '-q'])
640
- await git(base, 'add', '-A')
641
- await git(base, 'commit', '-q', '-m', 'base')
642
- const wt = join(root, 'wt')
643
- await git(base, 'worktree', 'add', '-q', '-b', 'improve/test-cand', wt, 'HEAD')
644
- // Proposer session: intentional edit + evidence artifact + install dirt.
645
- await writeFile(join(wt, 'src', 'a.ts'), 'export const a = 2\n')
646
- await mkdir(join(wt, '.improve'), { recursive: true })
647
- await writeFile(join(wt, '.improve', 'raw-trace-diagnosis.md'), '# diagnosis\n')
648
- await mkdir(join(wt, 'node_modules', 'pkg'), { recursive: true })
649
- await writeFile(join(wt, 'node_modules', 'pkg', 'index.js'), 'module.exports = 1\n')
650
- await writeFile(join(wt, 'node_modules', '.modules.yaml'), 'store: real-install\n')
651
- await writeFile(join(wt, 'debug.log'), 'stray ignored file\n')
652
- return { base, wt, root }
653
- }
654
-
655
- it('purgeIgnoredArtifacts removes ignored dirt but keeps tracked edits and untracked deliverables', async () => {
656
- const { wt, root } = await makeRepoWithDirtyCandidate()
657
- try {
658
- await purgeIgnoredArtifacts(wt)
659
- expect(existsSync(join(wt, 'node_modules'))).toBe(false)
660
- expect(existsSync(join(wt, 'debug.log'))).toBe(false)
661
- expect(existsSync(join(wt, '.improve', 'raw-trace-diagnosis.md'))).toBe(true)
662
- expect(await readFile(join(wt, 'src', 'a.ts'), 'utf8')).toBe('export const a = 2\n')
663
- // The exact finalize precondition the substrate enforces: zero ignored extras.
664
- const ignored = await git(wt, 'ls-files', '--others', '--ignored', '--exclude-standard')
665
- expect(ignored.stdout.trim()).toBe('')
666
- } finally {
667
- await rm(root, { recursive: true, force: true })
668
- }
669
- })
670
-
671
- it('a pre-aborted candidate verifier does not purge or inspect the worktree', async () => {
672
- const { wt, root } = await makeRepoWithDirtyCandidate()
673
- const controller = new AbortController()
674
- controller.abort(new Error('candidate verification cancelled'))
675
- try {
676
- await expect(
677
- loopsCandidateVerifier(root)(wt, controller.signal),
678
- ).rejects.toThrow(/candidate verification cancelled/)
679
- expect(existsSync(join(wt, 'node_modules'))).toBe(true)
680
- expect(existsSync(join(wt, 'debug.log'))).toBe(true)
681
- } finally {
682
- await rm(root, { recursive: true, force: true })
683
- }
684
- })
685
-
686
- it('a finalized candidate worktree tree-hash is identical before and after a mocked evaluation', async () => {
687
- const { base, wt, root } = await makeRepoWithDirtyCandidate()
688
- try {
689
- await purgeIgnoredArtifacts(wt)
690
- // Finalize the candidate: everything intentional is committed.
691
- await git(wt, 'add', '-A')
692
- await git(wt, 'commit', '-q', '-m', 'agentic: candidate finalized')
693
- const commit = (await git(wt, 'rev-parse', 'HEAD')).stdout.trim()
694
- const treeBefore = (await git(wt, 'rev-parse', 'HEAD^{tree}')).stdout.trim()
695
-
696
- // Mocked evaluation: a detached eval worktree at the candidate commit
697
- // takes ALL the runtime dirt; the CodeSurface worktree is never touched.
698
- const evalWt = join(root, 'eval-wt')
699
- await addEvalWorktree(base, commit, evalWt)
700
- await mkdir(join(evalWt, '.loops'), { recursive: true })
701
- await writeFile(join(evalWt, '.loops', 'state.json'), '{"run":"mock"}\n')
702
- await writeFile(join(evalWt, 'run.log'), 'arm eval output\n')
703
- await removeEvalWorktree(base, evalWt)
704
-
705
- expect(existsSync(evalWt)).toBe(false)
706
- expect((await git(wt, 'rev-parse', 'HEAD^{tree}')).stdout.trim()).toBe(treeBefore)
707
- expect((await git(wt, 'status', '--porcelain=v1', '--untracked-files=all')).stdout.trim()).toBe('')
708
- const ignored = await git(wt, 'ls-files', '--others', '--ignored', '--exclude-standard')
709
- expect(ignored.stdout.trim()).toBe('')
710
- } finally {
711
- await rm(root, { recursive: true, force: true })
712
- }
713
- })
714
- })