@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,300 +0,0 @@
1
- /**
2
- * Proof-of-faithfulness gate for the swe-arena replay module.
3
- *
4
- * Every expected value below was produced by running the reference
5
- * implementation (`fixtures/analyze.py`) against the committed fixtures on
6
- * 2026-07-15 and captured verbatim — the typed port must reproduce it exactly
7
- * before any typed execution path is built on top.
8
- *
9
- * KNOWN, DELIBERATE DIVERGENCE (flagged, both sides pinned): analyze.py's
10
- * printed "SUP total tokens: 560,554 / 0.83x" counts the supervisor BRAIN
11
- * only (journal metered events). The true SUP-arm spend adds worker-session
12
- * tokens (worker-tokens.json): 560,554 + 1,161,836 = 1,722,390 → 2.55x
13
- * SOLO. Both rollups are pinned; neither replaces the other.
14
- */
15
-
16
- import { createHash } from 'node:crypto'
17
- import { describe, expect, it } from 'vitest'
18
- import {
19
- costRollup,
20
- ledgerOutcomes,
21
- pairedSignTest,
22
- reconciledOutcomes,
23
- roundsProgression,
24
- splitPairs,
25
- } from './analyze.ts'
26
- import {
27
- loadHoldout,
28
- loadLedger,
29
- loadPreregisterLog,
30
- loadRejudge,
31
- loadRematchRounds,
32
- loadSupJournalTrue,
33
- loadWorkerTokens,
34
- readFixture,
35
- } from './fixtures.ts'
36
- import { applyRejudge, reconcile } from './reconcile.ts'
37
- import { buildReplay, renderReplay } from './replay.mts'
38
-
39
- const DISCORDANT = [
40
- 'astropy__astropy-13033',
41
- 'django__django-11532',
42
- 'matplotlib__matplotlib-20826',
43
- ]
44
-
45
- describe('fixture integrity', () => {
46
- // The fixtures ARE the experiment record. Any byte drift invalidates every
47
- // pinned number below, so drift must fail loudly here first.
48
- const sha256 = (name: string): string =>
49
- createHash('sha256').update(readFixture(name)).digest('hex')
50
-
51
- it.each([
52
- ['ledger.jsonl', '0014a1a2e7432a6551edbaf6668357cdbee8adb15b92e4f48ff530daf6ab8529'],
53
- ['rejudge.jsonl', '71ecadde0925d573573a7bb0fe171d26da386149f693035d3ca47c983a5f1700'],
54
- ['rematch.jsonl', 'aac48fdaa3c80f08367c220e406e1bb28b09c4f7d74fd17e4b76f50b5caaa5e4'],
55
- ['rematch2.jsonl', 'fe8347eae3e3f48c838cca0aa37e835730520482b705038c19450c98edd54d54'],
56
- ['rematch3.jsonl', '4eaeb9916634f3ffea7f66cb3bf747040a346524602a74a537efcdbc828ff020'],
57
- ['holdout.json', '56938f7c509621bcadbd50b1d6dacbee0f63b350506795e45d785c9de37d93f2'],
58
- ['worker-tokens.json', 'aa2f394120984a1c21ea7b70ed830a990477c08a1b38b6db743024660b43b129'],
59
- ['sup-journal-true.json', 'a95b48bdd7ab46011355c6a2b2844d67aa5811235ed3b8a5c9a4c9342eff9784'],
60
- ['analyze.py', '5512f4e1732ae4f0f92d7cb92840e5e99e54073c27f2b8ebd293b75346d9ebfc'],
61
- ['holdout-preregister.log', 'e11fbf34840765d8cd452c870905acec67e5a6b9dc05c2f76c64f71c2ad70884'],
62
- ])('%s is byte-identical to the captured artifact', (name, expected) => {
63
- expect(sha256(name)).toBe(expected)
64
- })
65
-
66
- it('loads the expected row counts', () => {
67
- expect(loadLedger()).toHaveLength(12)
68
- expect(loadRejudge()).toHaveLength(15)
69
- expect(loadRematchRounds().map((r) => r.length)).toEqual([3, 3, 3])
70
- })
71
-
72
- it('psf__requests-1766 carries the corrected-sup-verdict note', () => {
73
- const row = loadLedger().find((r) => r.iid === 'psf__requests-1766')
74
- expect(row?.sup_resolved).toBe(true)
75
- expect(row?._note).toMatch(/re-judged 2x stable resolved=true/)
76
- })
77
- })
78
-
79
- describe('reconcile: rejudge overrides + gold gate', () => {
80
- const table = reconcile(loadLedger(), loadRejudge())
81
-
82
- it('excludes exactly the gold-ungradeable instances 2317 + 2931', () => {
83
- expect(table.excluded.map((e) => e.iid)).toEqual(['psf__requests-2317', 'psf__requests-2931'])
84
- expect(table.excluded.map((e) => e.excludeReason)).toEqual([
85
- 'gold patch unresolved by judge (gold2) — instance ungradeable',
86
- 'gold patch unresolved by judge (gold) — instance ungradeable',
87
- ])
88
- expect(table.valid).toHaveLength(10)
89
- })
90
-
91
- it('gold-control (flask) grades true and stays in the denominator', () => {
92
- const flask = table.valid.find((r) => r.iid === 'pallets__flask-5014')
93
- expect(flask?.goldStatus).toBe('gradeable')
94
- expect(flask?.excluded).toBe(false)
95
- })
96
-
97
- it('parse-error rejudge rows (resolved=null) never override; final2 retries do', () => {
98
- const r2317 = applyRejudge(loadLedger(), loadRejudge()).find(
99
- (r) => r.iid === 'psf__requests-2317',
100
- )
101
- // solo-final and sup-final on 2317 both failed to parse; the *-final2
102
- // retries are the authoritative rows.
103
- expect(r2317?.solo).toEqual({ resolved: false, source: 'solo-final2' })
104
- expect(r2317?.sup).toEqual({ resolved: false, source: 'sup-final2' })
105
- // gold parse error superseded by the conclusive gold2 retry → ungradeable.
106
- expect(r2317?.goldStatus).toBe('ungradeable')
107
- })
108
-
109
- it('personal re-judges override the automated verdict for their arm only', () => {
110
- const byIid = new Map(applyRejudge(loadLedger(), loadRejudge()).map((r) => [r.iid, r]))
111
- for (const iid of DISCORDANT) {
112
- expect(byIid.get(iid)?.solo).toEqual({ resolved: true, source: 'solo-final' })
113
- expect(byIid.get(iid)?.sup.source).toBe('ledger')
114
- }
115
- expect(byIid.get('pydata__xarray-4687')?.solo).toEqual({
116
- resolved: false,
117
- source: 'solo-final',
118
- })
119
- expect(byIid.get('pydata__xarray-4687')?.sup).toEqual({ resolved: false, source: 'sup-final' })
120
- })
121
- })
122
-
123
- describe('analyze: raw ledger reproduction of analyze.py', () => {
124
- const outcomes = ledgerOutcomes(loadLedger())
125
- const split = splitPairs(outcomes)
126
- const sign = pairedSignTest(outcomes)
127
-
128
- it('SOLO 7/12, SUP 4/12', () => {
129
- expect(outcomes.filter((o) => o.solo)).toHaveLength(7)
130
- expect(outcomes.filter((o) => o.sup)).toHaveLength(4)
131
- })
132
-
133
- it('discordant pairs: 3 SOLO-only, 0 SUP-only', () => {
134
- expect(split.soloOnly).toEqual(DISCORDANT)
135
- expect(split.supOnly).toEqual([])
136
- expect(split.both).toHaveLength(4)
137
- expect(split.neither).toEqual([
138
- 'psf__requests-2317',
139
- 'psf__requests-2931',
140
- 'pydata__xarray-4687',
141
- 'pytest-dev__pytest-6197',
142
- 'sphinx-doc__sphinx-9658',
143
- ])
144
- })
145
-
146
- it('exact two-sided sign test p = 0.25 (agent-eval mcnemar)', () => {
147
- expect(sign.b).toBe(0)
148
- expect(sign.c).toBe(3)
149
- expect(sign.nDiscordant).toBe(3)
150
- expect(sign.pValue).toBeCloseTo(0.25, 10)
151
- expect(sign.pValue.toFixed(4)).toBe('0.2500')
152
- })
153
- })
154
-
155
- describe('analyze: valid-denominator verdict', () => {
156
- const outcomes = reconciledOutcomes(reconcile(loadLedger(), loadRejudge()).valid)
157
-
158
- it('SOLO 7/10, SUP 4/10 after excluding 2931 + 2317', () => {
159
- expect(outcomes).toHaveLength(10)
160
- expect(outcomes.filter((o) => o.solo)).toHaveLength(7)
161
- expect(outcomes.filter((o) => o.sup)).toHaveLength(4)
162
- })
163
-
164
- it('same discordant set and p=0.25 (excluded pairs were concordant-neither)', () => {
165
- const split = splitPairs(outcomes)
166
- expect(split.soloOnly).toEqual(DISCORDANT)
167
- expect(split.supOnly).toEqual([])
168
- expect(pairedSignTest(outcomes).pValue).toBeCloseTo(0.25, 10)
169
- })
170
- })
171
-
172
- describe('analyze: cost rollups', () => {
173
- const cost = costRollup(loadLedger(), loadSupJournalTrue(), loadWorkerTokens())
174
-
175
- it('SOLO total 675,412 tokens (analyze.py oracle)', () => {
176
- expect(cost.soloTokens).toBe(675412)
177
- })
178
-
179
- it('SUP brain total 560,554 tokens — what analyze.py prints as "SUP total"', () => {
180
- expect(cost.supBrainTokens).toBe(560554)
181
- expect(cost.brainTokenRatio.toFixed(2)).toBe('0.83')
182
- })
183
-
184
- it('TRUE SUP total 1,722,390 = brain 560,554 + workers 1,161,836 → 2.55x SOLO', () => {
185
- // The loudly-flagged divergence: analyze.py never printed worker tokens,
186
- // so its 0.83x understates the SUP arm's true spend by the worker share.
187
- expect(cost.supWorkerTokens).toBe(1161836)
188
- expect(cost.supTotalTokens).toBe(1722390)
189
- expect(cost.totalTokenRatio).toBeCloseTo(2.5501, 4)
190
- expect(cost.totalTokenRatio.toFixed(2)).toBe('2.55')
191
- })
192
-
193
- it('USD via blended SUP rate (analyze.py oracle strings)', () => {
194
- expect(cost.supUsd.toFixed(4)).toBe('0.1754')
195
- expect((cost.blendedRatePerTok * 1e6).toFixed(3)).toBe('0.313')
196
- expect(cost.soloUsdDerived.toFixed(4)).toBe('0.2114')
197
- })
198
-
199
- it('wall time: SOLO 3769s vs SUP 9124s → 2.42x', () => {
200
- expect(cost.soloWallS).toBe(3769)
201
- expect(cost.supWallS).toBe(9124)
202
- expect(cost.wallRatio.toFixed(2)).toBe('2.42')
203
- })
204
-
205
- it('telemetry gaps: runtime zeroed spentTokens on the 4 no-winner/crashed runs', () => {
206
- expect(cost.telemetryGaps).toEqual([
207
- 'astropy__astropy-13033',
208
- 'django__django-11532',
209
- 'matplotlib__matplotlib-20826',
210
- 'pytest-dev__pytest-6197',
211
- ])
212
- })
213
- })
214
-
215
- describe('analyze: supervisor evolution rounds', () => {
216
- const rounds = roundsProgression(loadLedger(), loadRematchRounds())
217
-
218
- it('matplotlib: 0-line unresolved at SUP round 0 → RESOLVED at SUP4', () => {
219
- const states = rounds.get('matplotlib__matplotlib-20826')
220
- expect(states?.map((s) => ({ round: s.round, resolved: s.resolved, patchLines: s.patchLines }))).toEqual([
221
- { round: 'SUP', resolved: false, patchLines: 0 },
222
- { round: 'SUP2', resolved: false, patchLines: 0 },
223
- { round: 'SUP3', resolved: false, patchLines: 0 },
224
- { round: 'SUP4', resolved: true, patchLines: 23 },
225
- ])
226
- })
227
-
228
- it('astropy + django: never resolved across SUP..SUP4', () => {
229
- for (const iid of ['astropy__astropy-13033', 'django__django-11532']) {
230
- const states = rounds.get(iid)
231
- expect(states).toHaveLength(4)
232
- expect(states?.every((s) => !s.resolved)).toBe(true)
233
- }
234
- })
235
- })
236
-
237
- describe('holdout registry', () => {
238
- const holdout = loadHoldout()
239
- const log = loadPreregisterLog()
240
-
241
- it('6 instances, all gold-verified and judge-calibrated, single selection commit', () => {
242
- expect(holdout.entries).toHaveLength(6)
243
- expect(holdout.selectedAtCommit).toBe('4a06fc54bd180789d1402c843a63ed245df9f8eb')
244
- expect(holdout.entries.every((e) => e.gold_official_resolved && e.verify_calibrated)).toBe(true)
245
- })
246
-
247
- it('every entry was preregistered before any arm ran, then registered', () => {
248
- const iids = holdout.entries.map((e) => e.iid)
249
- for (const iid of iids) {
250
- const pre = log.findIndex((l) => l.includes('PREREGISTER') && l.includes(iid))
251
- const reg = log.findIndex((l) => l.includes('REGISTERED') && l.includes(iid))
252
- expect(pre).toBeGreaterThanOrEqual(0)
253
- expect(reg).toBeGreaterThan(pre)
254
- expect(log[pre]).toContain('status=selected-before-any-arm-run')
255
- }
256
- })
257
- })
258
-
259
- describe('replay CLI output', () => {
260
- // Lines below are verbatim from running fixtures/analyze.py on 2026-07-15
261
- // (trailing whitespace trimmed — the per-instance table pads columns).
262
- const lines = renderReplay(buildReplay())
263
- .split('\n')
264
- .map((l) => l.trimEnd())
265
-
266
- it.each([
267
- ['SOLO resolved: 7/12 = 58.3%'],
268
- ['SUP resolved: 4/12 = 33.3%'],
269
- ['delta (SUP-SOLO): -3 instances (-25.0 pts)'],
270
- [" SOLO-only wins (SOLO✓ SUP✗): 3 ['astropy__astropy-13033', 'django__django-11532', 'matplotlib__matplotlib-20826']"],
271
- [' SUP-only wins (SUP✓ SOLO✗): 0 []'],
272
- [' exact two-sided sign test on discordant pairs: p=0.2500'],
273
- ['COST (measured tokens; USD via shared blended rate $0.313/1M from SUP accounting):'],
274
- [' SOLO total tokens: 675,412 -> derived $0.2114'],
275
- [' SUP total tokens: 560,554 -> runtime $0.1754'],
276
- [' SUP/SOLO token ratio: 0.83x'],
277
- [' SUP/SOLO cost ratio (token-derived): 0.83x'],
278
- [" [telemetry] instances where runtime spentTokens != journal-true (no-winner zeroing): ['astropy__astropy-13033', 'django__django-11532', 'matplotlib__matplotlib-20826', 'pytest-dev__pytest-6197']"],
279
- [' WALL: SOLO 3769s total vs SUP 9124s total -> SUP 2.42x wall'],
280
- ['pallets__flask-5014 True True T T ? 30738 15731 0.0124 116 261'],
281
- ['astropy__astropy-13033 True False T F 3 59496 0 0.0000 350 1550'],
282
- ['pytest-dev__pytest-6197 False False F F 0 133032 0 0.0000 755 248'],
283
- ])('reproduces analyze.py line: %s', (expected) => {
284
- expect(lines).toContain(expected)
285
- })
286
-
287
- it.each([
288
- ['SOLO resolved: 7/10'],
289
- ['SUP resolved: 4/10'],
290
- [' EXCLUDED psf__requests-2317: gold patch unresolved by judge (gold2) — instance ungradeable'],
291
- [' EXCLUDED psf__requests-2931: gold patch unresolved by judge (gold) — instance ungradeable'],
292
- ['exact two-sided sign test: p=0.2500'],
293
- [' brain 560,554 + workers 1,161,836 = 1,722,390 tokens'],
294
- [' SUP/SOLO true token ratio: 2.55x (brain-only ratio: 0.83x)'],
295
- [' matplotlib__matplotlib-20826 SUP:unresolved(0L,null) -> SUP2:unresolved(0L,no-winner) -> SUP3:unresolved(0L,null) -> SUP4:RESOLVED(23L,delivered)'],
296
- ['HOLDOUT REGISTRY (pre-registered at loops@4a06fc54bd, untouched):'],
297
- ])('prints reconciled line: %s', (expected) => {
298
- expect(lines).toContain(expected)
299
- })
300
- })