@tangle-network/agent-bench 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,714 +0,0 @@
1
- /**
2
- * Unit tests for the round-4 outer loop's pure protocol logic: the declared
3
- * change-space enforcement, porcelain path parsing, the keep-if-better verdict
4
- * (protocol_v2), the staircase row schema, and the frozen-arm assertion.
5
- * Pure — no arms, no docker, no tokens.
6
- */
7
-
8
- import { existsSync } from 'node:fs'
9
- import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
10
- import { tmpdir } from 'node:os'
11
- import { join } from 'node:path'
12
- import { makeProposalFinding } from '@tangle-network/agent-eval'
13
- import { describe, expect, it } from 'vitest'
14
- import { runOk } from './proc.ts'
15
- import {
16
- addEvalWorktree,
17
- DISPATCH_CLEANUP_GRACE_MS,
18
- DEFAULT_GATE_WAIT_CEILING_MS,
19
- FIXTURES_VERIFY_DIR,
20
- FROZEN_ARM,
21
- INSTANCE_LOCK_FILENAME,
22
- LOOPS_CHANGE_SPACE,
23
- RAW_TRACE_DIAGNOSIS_PATH,
24
- STAIRCASE_SCHEMA,
25
- SUPERVISOR_GATE_COUNT,
26
- acquireInstanceLock,
27
- assertFrozenArm,
28
- assertLaunchEnv,
29
- baselineDriftWarnings,
30
- campaignDispatchCeilingMs,
31
- changeSpaceInstruction,
32
- changeSpaceViolations,
33
- decideVerdict,
34
- defaultRound4Config,
35
- instanceVerdictsFromCells,
36
- isPidAlive,
37
- loopsCandidateVerifier,
38
- normalizeRepoPath,
39
- parseStaircaseRow,
40
- porcelainChangedPaths,
41
- purgeIgnoredArtifacts,
42
- removeEvalWorktree,
43
- replicateCoverageComplete,
44
- resolvedInstanceCount,
45
- round4BuildPrompt,
46
- runWithPostGateClock,
47
- type EvidenceCell,
48
- type ReplicateRun,
49
- type StaircaseRow,
50
- } from './outer-loop.mts'
51
-
52
- describe('changeSpaceViolations', () => {
53
- it('accepts the declared change-space, including nested extension paths', () => {
54
- expect(
55
- changeSpaceViolations([
56
- 'extensions/pi/loops.ts',
57
- 'extensions/pi/prompts/worker-coding-system.md',
58
- 'extensions/pi/deep/new-module.ts',
59
- 'src/worker-evidence.ts',
60
- 'src/best-effort.ts',
61
- 'src/worker-clone.ts',
62
- RAW_TRACE_DIAGNOSIS_PATH,
63
- './src/worker-clone.ts',
64
- ]),
65
- ).toEqual([])
66
- })
67
-
68
- it('rejects everything outside the declared space', () => {
69
- expect(
70
- changeSpaceViolations([
71
- 'src/runner.ts',
72
- 'src/strategy-loop.ts',
73
- 'package.json',
74
- 'extensions/other/loops.ts',
75
- 'tests/top-model.test.ts',
76
- 'src/worker-evidence.ts.bak',
77
- ]),
78
- ).toEqual([
79
- 'src/runner.ts',
80
- 'src/strategy-loop.ts',
81
- 'package.json',
82
- 'extensions/other/loops.ts',
83
- 'tests/top-model.test.ts',
84
- 'src/worker-evidence.ts.bak',
85
- ])
86
- })
87
-
88
- it('is not fooled by prefix-sharing directories (extensions/pi2 is out)', () => {
89
- expect(changeSpaceViolations(['extensions/pi2/loops.ts'])).toEqual(['extensions/pi2/loops.ts'])
90
- })
91
-
92
- it('fails closed on traversal, absolute, and empty paths', () => {
93
- expect(changeSpaceViolations(['extensions/pi/../../package.json'])).toEqual([
94
- 'extensions/pi/../../package.json',
95
- ])
96
- expect(changeSpaceViolations(['/etc/passwd'])).toEqual(['/etc/passwd'])
97
- expect(changeSpaceViolations([''])).toEqual([''])
98
- })
99
-
100
- it('normalizes quoted and backslashed paths before matching', () => {
101
- expect(normalizeRepoPath('"extensions/pi/a b.ts"')).toBe('extensions/pi/a b.ts')
102
- expect(normalizeRepoPath('extensions\\pi\\loops.ts')).toBe('extensions/pi/loops.ts')
103
- expect(normalizeRepoPath('../outside.ts')).toBeNull()
104
- expect(changeSpaceViolations(['"extensions/pi/a b.ts"'])).toEqual([])
105
- })
106
- })
107
-
108
- describe('porcelainChangedPaths', () => {
109
- it('parses modified, untracked, and rename entries (both rename sides)', () => {
110
- const stdout = [
111
- ' M extensions/pi/loops.ts',
112
- '?? .improve/raw-trace-diagnosis.md',
113
- 'R src/worker-clone.ts -> src/worker-clone-2.ts',
114
- 'A src/best-effort.ts',
115
- '',
116
- ].join('\n')
117
- expect(porcelainChangedPaths(stdout)).toEqual([
118
- 'extensions/pi/loops.ts',
119
- '.improve/raw-trace-diagnosis.md',
120
- 'src/worker-clone.ts',
121
- 'src/worker-clone-2.ts',
122
- 'src/best-effort.ts',
123
- ])
124
- })
125
-
126
- it('a rename OUT of the change-space is caught end-to-end', () => {
127
- const paths = porcelainChangedPaths('R src/worker-clone.ts -> src/worker-clone-moved.ts\n')
128
- expect(changeSpaceViolations(paths)).toEqual(['src/worker-clone-moved.ts'])
129
- })
130
- })
131
-
132
- describe('decideVerdict (protocol_v2 keep-if-better)', () => {
133
- const base = {
134
- violations: [] as string[],
135
- coverageComplete: true,
136
- resolvedCount: 2,
137
- parentResolvedCount: 1,
138
- costRatio: 1.0,
139
- costGuardRatio: 1.2,
140
- }
141
- it('accepts a gaining, in-space, in-budget candidate', () => {
142
- expect(decideVerdict(base)).toBe('accepted')
143
- })
144
- it('rejects out-of-space before anything else', () => {
145
- expect(decideVerdict({ ...base, violations: ['package.json'] })).toBe('rejected-out-of-space')
146
- })
147
- it('rejects incomplete coverage (an errored cell can never promote)', () => {
148
- expect(decideVerdict({ ...base, coverageComplete: false })).toBe('rejected-incomplete')
149
- })
150
- it('requires a STRICT improvement-set gain (tie = reject)', () => {
151
- expect(decideVerdict({ ...base, resolvedCount: 1 })).toBe('rejected-no-gain')
152
- expect(decideVerdict({ ...base, resolvedCount: 0 })).toBe('rejected-no-gain')
153
- })
154
- it('rejects on the +20% cost guard and on unprovable cost', () => {
155
- expect(decideVerdict({ ...base, costRatio: 1.21 })).toBe('rejected-cost')
156
- expect(decideVerdict({ ...base, costRatio: null })).toBe('rejected-cost')
157
- expect(decideVerdict({ ...base, costRatio: 1.2 })).toBe('accepted')
158
- })
159
- })
160
-
161
- describe('staircase row schema', () => {
162
- const row: StaircaseRow = {
163
- schema: STAIRCASE_SCHEMA,
164
- round: 4,
165
- generation: 0,
166
- runId: 'r4-abc123',
167
- at: '2026-07-15T00:00:00.000Z',
168
- candidate: 'sha256:cand',
169
- candidateCommit: 'deadbeef00',
170
- parent: 'sha256:parent',
171
- parentResolvedCount: 1,
172
- label: 'placement-aware settle',
173
- rationale: 'diagnosis: fix placement mismatch (3/3 analysts)',
174
- changedFiles: ['extensions/pi/loops.ts'],
175
- changeSpaceViolations: [],
176
- perInstance: [
177
- {
178
- iid: 'django__django-11532',
179
- rep: 0,
180
- resolved: true,
181
- verify_pass: true,
182
- patch_lines: 47,
183
- wall_s: 900,
184
- spentTokens: 54623,
185
- recoveredTokens: 61000,
186
- judgeAttempts: 1,
187
- costUsd: 0.12,
188
- },
189
- ],
190
- resolvedCount: 2,
191
- coverageComplete: true,
192
- wallS: 2700,
193
- baselineWallS: 2500,
194
- costRatio: 1.08,
195
- costGuardRatio: 1.2,
196
- internallyPromoted: true,
197
- verdict: 'accepted',
198
- holdout: 'operator-approval-required',
199
- armProvenance: { repo: '/tmp/eval-wt', commit: 'deadbeef00' },
200
- diffPath: '/tmp/out/candidates/deadbeef00.patch',
201
- diffSha256: 'sha256:aaaa',
202
- }
203
-
204
- it('round-trips through JSONL', () => {
205
- const parsed = parseStaircaseRow(JSON.stringify(row))
206
- expect(parsed).toEqual(row)
207
- })
208
-
209
- it('rejects schema drift, bad verdicts, and missing fields', () => {
210
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, schema: 'v0' }))).toThrow(/unknown schema/)
211
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, verdict: 'kept' }))).toThrow(/unknown verdict/)
212
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, resolvedCount: '2' }))).toThrow(/must be a number/)
213
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, runId: '' }))).toThrow(/runId/)
214
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, perInstance: 'x' }))).toThrow(/perInstance/)
215
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, costRatio: 'high' }))).toThrow(/costRatio/)
216
- expect(() => parseStaircaseRow(JSON.stringify({ ...row, internallyPromoted: 'yes' }))).toThrow(/booleans/)
217
- })
218
-
219
- it('accepts a null-cost rejected dot (telemetry gap is data, not a zero)', () => {
220
- const dot = { ...row, costRatio: null, verdict: 'rejected-cost' as const, internallyPromoted: false }
221
- expect(parseStaircaseRow(JSON.stringify(dot)).costRatio).toBeNull()
222
- })
223
-
224
- it('accepts a gen-3 prefilter kill dot with its killReason', () => {
225
- const dot = {
226
- ...row,
227
- candidate: 'prefilter-kill:aaaabbbbcccc',
228
- candidateCommit: null,
229
- verdict: 'rejected-prefilter' as const,
230
- internallyPromoted: false,
231
- coverageComplete: false,
232
- resolvedCount: 0,
233
- perInstance: [],
234
- costRatio: null,
235
- killReason: 'smoke: smoke astropy__astropy-13033: resolved=false — below the mechanism bar',
236
- armProvenance: null,
237
- }
238
- const parsed = parseStaircaseRow(JSON.stringify(dot))
239
- expect(parsed.verdict).toBe('rejected-prefilter')
240
- expect(parsed.killReason).toContain('below the mechanism bar')
241
- })
242
- })
243
-
244
- describe('frozen arm + default config', () => {
245
- it('passes on the round-3 frozen arm and the default config', () => {
246
- expect(() => assertFrozenArm(FROZEN_ARM)).not.toThrow()
247
- expect(() => assertFrozenArm(defaultRound4Config().arm)).not.toThrow()
248
- })
249
- it('throws on any immutable-arm drift (protocol_v2)', () => {
250
- expect(() => assertFrozenArm({ ...FROZEN_ARM, workerModel: 'gpt-5.5' })).toThrow(/immutable/)
251
- expect(() => assertFrozenArm({ ...FROZEN_ARM, maxUsd: 16 })).toThrow(/maxUsd/)
252
- expect(() => assertFrozenArm({ ...FROZEN_ARM, budget: 80 })).toThrow(/budget/)
253
- })
254
- it('default config: improvement set and holdout are the pre-registered, disjoint sets', () => {
255
- const config = defaultRound4Config()
256
- expect(config.instances).toEqual([
257
- 'astropy__astropy-13033',
258
- 'django__django-11532',
259
- 'matplotlib__matplotlib-20826',
260
- ])
261
- expect(config.holdoutInstances).toHaveLength(6)
262
- expect(config.instances.filter((i) => config.holdoutInstances.includes(i))).toEqual([])
263
- expect(config.roundsDir).toBe('/home/drew/code/supervisor-lab/.evolve/rounds')
264
- expect(config.analystModels.every((m) => m === 'glm-5.2')).toBe(true)
265
- })
266
- it('default config: verify scripts come from the COMMITTED fixtures dir and reps=2', () => {
267
- const config = defaultRound4Config()
268
- // The scratchpad copy died with a host reboot; the committed dir is the durable home.
269
- expect(config.verifyDir).toBe(FIXTURES_VERIFY_DIR)
270
- expect(config.verifyDir).toContain('fixtures/verify')
271
- // Single-rep scoring flips instance outcomes run-to-run — round 4 runs 2.
272
- expect(config.repsPerInstance).toBe(2)
273
- })
274
- it('default config: the premeasured baseline artifact path is required-with-default', () => {
275
- const config = defaultRound4Config()
276
- expect(config.premeasuredBaselinePath).toContain('/r4/premeasured-baseline.json')
277
- })
278
- it('default config: author-shot timeout doubled after 3 gen-1 timeouts; outDir name is overridable', () => {
279
- const config = defaultRound4Config()
280
- // 20-min shots died 3× under degraded capacity ("author shot timed out").
281
- expect(config.proposerTimeoutMs).toBe(2_400_000)
282
- expect(defaultRound4Config(undefined, { outDirName: 'r4-gen2' }).outDir.endsWith('/r4-gen2')).toBe(true)
283
- expect(config.outDir.endsWith('/r4')).toBe(true)
284
- })
285
- })
286
-
287
- describe('dispatch clocks (gate holds are never billed to the cell)', () => {
288
- it('starts neither capacity checks nor work for an already-aborted caller', async () => {
289
- const controller = new AbortController()
290
- controller.abort(new Error('cancelled before cell'))
291
- let gates = 0
292
- let work = 0
293
- await expect(runWithPostGateClock({
294
- awaitGates: async () => { gates += 1 },
295
- work: async () => { work += 1; return 'x' },
296
- timeoutMs: 1_000,
297
- signal: controller.signal,
298
- })).rejects.toThrow('cancelled before cell')
299
- expect(gates).toBe(0)
300
- expect(work).toBe(0)
301
- })
302
-
303
- it('passes parent cancellation through work and waits for its cleanup', async () => {
304
- const controller = new AbortController()
305
- let cleanupFinished = false
306
- let markStarted!: () => void
307
- const started = new Promise<void>((resolve) => { markStarted = resolve })
308
- const running = runWithPostGateClock({
309
- awaitGates: async (signal) => signal?.throwIfAborted(),
310
- work: (signal) => new Promise<string>((resolve) => {
311
- markStarted()
312
- signal.addEventListener('abort', () => {
313
- setTimeout(() => {
314
- cleanupFinished = true
315
- resolve('settled')
316
- }, 25)
317
- }, { once: true })
318
- }),
319
- timeoutMs: 1_000,
320
- signal: controller.signal,
321
- })
322
- await started
323
- controller.abort(new Error('operator interrupted'))
324
- await expect(running).rejects.toThrow('operator interrupted')
325
- expect(cleanupFinished).toBe(true)
326
- })
327
-
328
- it('campaignDispatchCeilingMs includes gate holds, both judge attempts, and cleanup', () => {
329
- expect(campaignDispatchCeilingMs({ dispatchTimeoutMs: 7_200_000 })).toBe(
330
- 7_200_000 + SUPERVISOR_GATE_COUNT * DEFAULT_GATE_WAIT_CEILING_MS + 2 * 1_800_000 + DISPATCH_CLEANUP_GRACE_MS,
331
- )
332
- expect(campaignDispatchCeilingMs({
333
- dispatchTimeoutMs: 1_000,
334
- gateWaitCeilingMs: 500,
335
- judgeTimeoutMs: 2_000,
336
- })).toBe(
337
- 1_000 + 2 * 500 + 2 * 2_000 + DISPATCH_CLEANUP_GRACE_MS,
338
- )
339
- })
340
-
341
- it('a gate hold LONGER than the work clock does not abort the cell (the pre-crash bug)', async () => {
342
- // Pre-crash failure shape: 58-min capacity hold billed to the 7200s clock.
343
- // Here: gate hold 120ms > work clock 60ms; the work itself takes 10ms.
344
- const result = await runWithPostGateClock({
345
- awaitGates: () => new Promise<void>((r) => setTimeout(r, 120)),
346
- work: () => new Promise<string>((r) => setTimeout(() => r('done'), 10)),
347
- timeoutMs: 60,
348
- })
349
- expect(result).toBe('done')
350
- })
351
-
352
- it('work exceeding the post-gate clock still fails loud', async () => {
353
- await expect(
354
- runWithPostGateClock({
355
- awaitGates: () => Promise.resolve(),
356
- work: () => new Promise<string>((r) => setTimeout(() => r('late'), 200)),
357
- timeoutMs: 30,
358
- label: 'R4 deadbeef00 astropy__astropy-13033 r0',
359
- }),
360
- ).rejects.toThrow(/post-gate dispatch exceeded 30ms .*astropy__astropy-13033/)
361
- })
362
-
363
- it('waits for abort cleanup before reporting a post-gate timeout', async () => {
364
- let cleanupFinished = false
365
- const started = Date.now()
366
- await expect(
367
- runWithPostGateClock({
368
- awaitGates: () => Promise.resolve(),
369
- work: (signal) => new Promise<string>((resolve) => {
370
- signal.addEventListener('abort', () => {
371
- setTimeout(() => {
372
- cleanupFinished = true
373
- resolve('settled after cleanup')
374
- }, 30)
375
- }, { once: true })
376
- }),
377
- timeoutMs: 20,
378
- label: 'cleanup proof',
379
- }),
380
- ).rejects.toThrow(/post-gate dispatch exceeded 20ms .*cleanup proof/)
381
- expect(cleanupFinished).toBe(true)
382
- expect(Date.now() - started).toBeGreaterThanOrEqual(45)
383
- })
384
-
385
- it('preserves a process-cleanup failure after the dispatch clock expires', async () => {
386
- await expect(
387
- runWithPostGateClock({
388
- awaitGates: () => Promise.resolve(),
389
- work: (signal) => new Promise<never>((_resolve, reject) => {
390
- signal.addEventListener('abort', () => {
391
- reject(new Error('process group 123 survived SIGKILL'))
392
- }, { once: true })
393
- }),
394
- timeoutMs: 20,
395
- label: 'cleanup failure proof',
396
- }),
397
- ).rejects.toThrow(/post-gate dispatch exceeded 20ms .*process group 123 survived SIGKILL/)
398
- })
399
-
400
- it('a gate failure rejects before the work clock ever starts', async () => {
401
- let workStarted = false
402
- await expect(
403
- runWithPostGateClock({
404
- awaitGates: () => Promise.reject(new Error('no capacity on router within ceiling')),
405
- work: async () => {
406
- workStarted = true
407
- return 'x'
408
- },
409
- timeoutMs: 1_000,
410
- }),
411
- ).rejects.toThrow(/no capacity/)
412
- expect(workStarted).toBe(false)
413
- })
414
- })
415
-
416
- describe('replicate semantics (repsPerInstance)', () => {
417
- const iids = ['a', 'b', 'c']
418
- const run = (iid: string, resolved: boolean | null): ReplicateRun => ({ iid, resolved })
419
-
420
- it('an instance resolves only when ALL replicates resolve (AND, fail-closed)', () => {
421
- const runs = [
422
- run('a', true), run('a', true), // both reps resolved → counts
423
- run('b', true), run('b', false), // flaky split → does NOT count
424
- run('c', false), run('c', false),
425
- ]
426
- expect(resolvedInstanceCount(runs, iids, 2)).toBe(1)
427
- })
428
-
429
- it('missing replicates never count as resolved', () => {
430
- expect(resolvedInstanceCount([run('a', true)], iids, 2)).toBe(0)
431
- expect(resolvedInstanceCount([run('a', true)], iids, 1)).toBe(1)
432
- })
433
-
434
- it('coverage requires every replicate of every instance with a conclusive verdict', () => {
435
- const full = iids.flatMap((iid) => [run(iid, true), run(iid, false)])
436
- expect(replicateCoverageComplete(full, iids, 2)).toBe(true)
437
- expect(replicateCoverageComplete(full.slice(1), iids, 2)).toBe(false)
438
- const inconclusive = [...full.slice(0, 5), run('c', null)]
439
- expect(replicateCoverageComplete(inconclusive, iids, 2)).toBe(false)
440
- })
441
- })
442
-
443
- describe('premeasured-baseline drift (the validated artifact is the only denominator)', () => {
444
- const iids = ['astropy__astropy-13033', 'django__django-11532', 'matplotlib__matplotlib-20826']
445
- // AND-verdicts of the premeasured artifact's campaign (measured 1/3).
446
- const expected = {
447
- 'astropy__astropy-13033': false,
448
- 'django__django-11532': false,
449
- 'matplotlib__matplotlib-20826': true,
450
- }
451
- const run = (iid: string, resolved: boolean | null): ReplicateRun => ({ iid, resolved })
452
- const cell = (iid: string, rep: number, resolved: boolean | null): EvidenceCell => ({
453
- scenarioId: iid,
454
- rep,
455
- artifact:
456
- resolved === null
457
- ? null
458
- : {
459
- kind: 'swe-arm',
460
- iid,
461
- commit: 'basecommit0',
462
- resolved,
463
- verifyPass: resolved,
464
- patchLines: 1,
465
- wallS: 10,
466
- spentTokens: null,
467
- spentUsd: null,
468
- recoveredTokens: null,
469
- workerTokIn: null,
470
- workerTokOut: null,
471
- judgeAttempts: null,
472
- judgeWallS: null,
473
- runDir: '/tmp/none',
474
- patchPath: '/tmp/none.patch',
475
- },
476
- ...(resolved === null ? { error: 'inconclusive' } : {}),
477
- })
478
-
479
- it('instanceVerdictsFromCells ANDs replicates and omits incomplete instances', () => {
480
- const cells = [
481
- cell(iids[0]!, 0, false), cell(iids[0]!, 1, false),
482
- cell(iids[1]!, 0, true), cell(iids[1]!, 1, false), // flaky → AND false
483
- cell(iids[2]!, 0, true), // partial → omitted
484
- ]
485
- expect(instanceVerdictsFromCells(cells, iids, 2)).toEqual({
486
- 'astropy__astropy-13033': false,
487
- 'django__django-11532': false,
488
- })
489
- const inconclusive = [cell(iids[0]!, 0, true), cell(iids[0]!, 1, null)]
490
- expect(instanceVerdictsFromCells(inconclusive, [iids[0]!], 2)).toEqual({})
491
- })
492
-
493
- it('drift warnings fire on a reps-complete contradiction, both directions', () => {
494
- const runs = [
495
- run(iids[0]!, true), run(iids[0]!, true), // premeasured false, cached true → drift
496
- run(iids[1]!, false), run(iids[1]!, false), // premeasured false, cached false → quiet
497
- run(iids[2]!, true), run(iids[2]!, false), // premeasured true, cached false → drift
498
- ]
499
- const warnings = baselineDriftWarnings(expected, runs, iids, 2)
500
- expect(warnings).toHaveLength(2)
501
- expect(warnings[0]).toContain('astropy__astropy-13033: premeasured=false')
502
- expect(warnings[1]).toContain('matplotlib__matplotlib-20826: premeasured=true')
503
- expect(warnings.every((w) => w.includes('premeasured artifact rules'))).toBe(true)
504
- })
505
-
506
- it('a partial or inconclusive cached record has no AND-verdict — no drift claim', () => {
507
- // Partial coverage (1 of 2 reps per instance): no AND-verdict exists yet,
508
- // even where the single present cell disagrees with the artifact.
509
- const partial = [run(iids[1]!, true), run(iids[2]!, false)]
510
- expect(baselineDriftWarnings(expected, partial, iids, 2)).toEqual([])
511
- const inconclusive = [run(iids[2]!, true), run(iids[2]!, null)]
512
- expect(baselineDriftWarnings(expected, inconclusive, iids, 2)).toEqual([])
513
- })
514
- })
515
-
516
- describe('launch guards', () => {
517
- it('assertLaunchEnv refuses when either key is absent or blank, naming dotenvx', () => {
518
- expect(() => assertLaunchEnv({})).toThrow(/TANGLE_API_KEY \+ ZAI_API_KEY absent .*dotenvx/)
519
- expect(() => assertLaunchEnv({ TANGLE_API_KEY: 'x' })).toThrow(/ZAI_API_KEY/)
520
- expect(() => assertLaunchEnv({ TANGLE_API_KEY: 'x', ZAI_API_KEY: ' ' })).toThrow(/ZAI_API_KEY/)
521
- expect(() => assertLaunchEnv({ TANGLE_API_KEY: 'x', ZAI_API_KEY: 'y' })).not.toThrow()
522
- })
523
-
524
- it('isPidAlive: own pid is alive; an absurd pid is not', () => {
525
- expect(isPidAlive(process.pid)).toBe(true)
526
- expect(isPidAlive(2 ** 30)).toBe(false)
527
- })
528
-
529
- it('instance lock: acquire, refuse a live second instance, release', async () => {
530
- const dir = await mkdtemp(join(tmpdir(), 'r4-lock-'))
531
- try {
532
- const lock = await acquireInstanceLock(dir)
533
- expect(lock.path).toBe(join(dir, INSTANCE_LOCK_FILENAME))
534
- expect((await readFile(lock.path, 'utf8')).trim()).toBe(String(process.pid))
535
- // A DIFFERENT live pid (init/pid 1 is always alive) must be refused.
536
- await writeFile(lock.path, '1\n')
537
- await expect(acquireInstanceLock(dir)).rejects.toThrow(/pid 1.*refusing to race/s)
538
- await writeFile(lock.path, `${process.pid}\n`)
539
- await lock.release()
540
- expect(existsSync(lock.path)).toBe(false)
541
- } finally {
542
- await rm(dir, { recursive: true, force: true })
543
- }
544
- })
545
-
546
- it('instance lock: a stale lock (dead pid or garbage) is reclaimed', async () => {
547
- const dir = await mkdtemp(join(tmpdir(), 'r4-lock-stale-'))
548
- try {
549
- await writeFile(join(dir, INSTANCE_LOCK_FILENAME), `${2 ** 30}\n`)
550
- const lock = await acquireInstanceLock(dir)
551
- expect((await readFile(lock.path, 'utf8')).trim()).toBe(String(process.pid))
552
- await lock.release()
553
- await writeFile(join(dir, INSTANCE_LOCK_FILENAME), 'not-a-pid\n')
554
- const lock2 = await acquireInstanceLock(dir)
555
- expect((await readFile(lock2.path, 'utf8')).trim()).toBe(String(process.pid))
556
- await lock2.release()
557
- } finally {
558
- await rm(dir, { recursive: true, force: true })
559
- }
560
- })
561
-
562
- it('release only removes a lock this instance still owns', async () => {
563
- const dir = await mkdtemp(join(tmpdir(), 'r4-lock-own-'))
564
- try {
565
- const lock = await acquireInstanceLock(dir)
566
- await writeFile(lock.path, '424242\n') // another instance reclaimed it
567
- await lock.release()
568
- expect((await readFile(lock.path, 'utf8')).trim()).toBe('424242')
569
- } finally {
570
- await rm(dir, { recursive: true, force: true })
571
- }
572
- })
573
- })
574
-
575
- describe('round4BuildPrompt', () => {
576
- it('declares the change-space and renders findings', () => {
577
- const prompt = round4BuildPrompt({
578
- findings: [
579
- makeProposalFinding({
580
- analyst_id: 'test',
581
- severity: 'high',
582
- area: 'mechanism',
583
- claim: 'fix placement mismatch',
584
- recommended_action: 'settle where maintainers expect',
585
- evidence_refs: [],
586
- confidence: 1,
587
- proposal_origin: 'search',
588
- }),
589
- ],
590
- })
591
- expect(prompt).toContain('DECLARED CHANGE-SPACE')
592
- expect(prompt).toContain('extensions/pi/**')
593
- expect(prompt).toContain('src/worker-evidence.ts')
594
- expect(prompt).toContain('fix placement mismatch')
595
- expect(prompt).toContain('→ settle where maintainers expect')
596
- // No raw-trace findings ⇒ no evidence-file requirement block (the
597
- // change-space instruction still NAMES the artifact path as allowed).
598
- expect(prompt).not.toContain('Raw trace evidence requirement')
599
- })
600
-
601
- it('adds the raw-trace evidence contract when raw-trace findings are present', () => {
602
- const prompt = round4BuildPrompt({
603
- findings: [
604
- makeProposalFinding({
605
- analyst_id: 'raw-trace-distiller',
606
- severity: 'high',
607
- area: 'raw-trace-context',
608
- claim: 'traces at /run/gen-0',
609
- evidence_refs: [],
610
- confidence: 1,
611
- proposal_origin: 'search',
612
- }),
613
- ],
614
- })
615
- expect(prompt).toContain('Raw trace evidence requirement')
616
- expect(prompt).toContain(RAW_TRACE_DIAGNOSIS_PATH)
617
- })
618
-
619
- it('changeSpaceInstruction names every allowed root exactly once', () => {
620
- const text = changeSpaceInstruction(LOOPS_CHANGE_SPACE)
621
- for (const f of LOOPS_CHANGE_SPACE.files) expect(text).toContain(f)
622
- for (const p of LOOPS_CHANGE_SPACE.prefixes) expect(text).toContain(`${p}**`)
623
- })
624
- })
625
-
626
- describe('candidate worktree hygiene (finalize precondition + eval isolation)', () => {
627
- const git = (dir: string, ...argv: string[]) =>
628
- runOk('git', ['-C', dir, '-c', 'user.email=t@test', '-c', 'user.name=t', '-c', 'core.hooksPath=/dev/null', ...argv])
629
-
630
- /** Base repo + candidate worktree with proposer-style state: a tracked edit,
631
- * an untracked non-ignored deliverable, and gitignored install dirt (the
632
- * exact mix round-4 gen-0 cand-1 died on). */
633
- async function makeRepoWithDirtyCandidate(): Promise<{ base: string; wt: string; root: string }> {
634
- const root = await mkdtemp(join(tmpdir(), 'r4-wt-hygiene-'))
635
- const base = join(root, 'base')
636
- await mkdir(join(base, 'src'), { recursive: true })
637
- await writeFile(join(base, '.gitignore'), 'node_modules/\n*.log\n')
638
- await writeFile(join(base, 'src', 'a.ts'), 'export const a = 1\n')
639
- await runOk('git', ['-C', base, 'init', '-q'])
640
- await git(base, 'add', '-A')
641
- await git(base, 'commit', '-q', '-m', 'base')
642
- const wt = join(root, 'wt')
643
- await git(base, 'worktree', 'add', '-q', '-b', 'improve/test-cand', wt, 'HEAD')
644
- // Proposer session: intentional edit + evidence artifact + install dirt.
645
- await writeFile(join(wt, 'src', 'a.ts'), 'export const a = 2\n')
646
- await mkdir(join(wt, '.improve'), { recursive: true })
647
- await writeFile(join(wt, '.improve', 'raw-trace-diagnosis.md'), '# diagnosis\n')
648
- await mkdir(join(wt, 'node_modules', 'pkg'), { recursive: true })
649
- await writeFile(join(wt, 'node_modules', 'pkg', 'index.js'), 'module.exports = 1\n')
650
- await writeFile(join(wt, 'node_modules', '.modules.yaml'), 'store: real-install\n')
651
- await writeFile(join(wt, 'debug.log'), 'stray ignored file\n')
652
- return { base, wt, root }
653
- }
654
-
655
- it('purgeIgnoredArtifacts removes ignored dirt but keeps tracked edits and untracked deliverables', async () => {
656
- const { wt, root } = await makeRepoWithDirtyCandidate()
657
- try {
658
- await purgeIgnoredArtifacts(wt)
659
- expect(existsSync(join(wt, 'node_modules'))).toBe(false)
660
- expect(existsSync(join(wt, 'debug.log'))).toBe(false)
661
- expect(existsSync(join(wt, '.improve', 'raw-trace-diagnosis.md'))).toBe(true)
662
- expect(await readFile(join(wt, 'src', 'a.ts'), 'utf8')).toBe('export const a = 2\n')
663
- // The exact finalize precondition the substrate enforces: zero ignored extras.
664
- const ignored = await git(wt, 'ls-files', '--others', '--ignored', '--exclude-standard')
665
- expect(ignored.stdout.trim()).toBe('')
666
- } finally {
667
- await rm(root, { recursive: true, force: true })
668
- }
669
- })
670
-
671
- it('a pre-aborted candidate verifier does not purge or inspect the worktree', async () => {
672
- const { wt, root } = await makeRepoWithDirtyCandidate()
673
- const controller = new AbortController()
674
- controller.abort(new Error('candidate verification cancelled'))
675
- try {
676
- await expect(
677
- loopsCandidateVerifier(root)(wt, controller.signal),
678
- ).rejects.toThrow(/candidate verification cancelled/)
679
- expect(existsSync(join(wt, 'node_modules'))).toBe(true)
680
- expect(existsSync(join(wt, 'debug.log'))).toBe(true)
681
- } finally {
682
- await rm(root, { recursive: true, force: true })
683
- }
684
- })
685
-
686
- it('a finalized candidate worktree tree-hash is identical before and after a mocked evaluation', async () => {
687
- const { base, wt, root } = await makeRepoWithDirtyCandidate()
688
- try {
689
- await purgeIgnoredArtifacts(wt)
690
- // Finalize the candidate: everything intentional is committed.
691
- await git(wt, 'add', '-A')
692
- await git(wt, 'commit', '-q', '-m', 'agentic: candidate finalized')
693
- const commit = (await git(wt, 'rev-parse', 'HEAD')).stdout.trim()
694
- const treeBefore = (await git(wt, 'rev-parse', 'HEAD^{tree}')).stdout.trim()
695
-
696
- // Mocked evaluation: a detached eval worktree at the candidate commit
697
- // takes ALL the runtime dirt; the CodeSurface worktree is never touched.
698
- const evalWt = join(root, 'eval-wt')
699
- await addEvalWorktree(base, commit, evalWt)
700
- await mkdir(join(evalWt, '.loops'), { recursive: true })
701
- await writeFile(join(evalWt, '.loops', 'state.json'), '{"run":"mock"}\n')
702
- await writeFile(join(evalWt, 'run.log'), 'arm eval output\n')
703
- await removeEvalWorktree(base, evalWt)
704
-
705
- expect(existsSync(evalWt)).toBe(false)
706
- expect((await git(wt, 'rev-parse', 'HEAD^{tree}')).stdout.trim()).toBe(treeBefore)
707
- expect((await git(wt, 'status', '--porcelain=v1', '--untracked-files=all')).stdout.trim()).toBe('')
708
- const ignored = await git(wt, 'ls-files', '--others', '--ignored', '--exclude-standard')
709
- expect(ignored.stdout.trim()).toBe('')
710
- } finally {
711
- await rm(root, { recursive: true, force: true })
712
- }
713
- })
714
- })