@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,176 +0,0 @@
1
- /**
2
- * GEN-4 proposer-model provenance — pin each author seat's MODEL IDENTITY in
3
- * the run record at t=0, before any author shot fires.
4
- *
5
- * Why: a proposer spec may pin a model explicitly (`spec.model` → the harness
6
- * CLI's `-m` flag via the author profile's `model.default`), but the claude
7
- * seat deliberately does NOT pass `-m` — the CLI runs on its logged-in
8
- * account, whose resolved model comes from its own settings. Verified against
9
- * claude CLI 2.1.217: `--model` IS supported headless (`-p`), but the run's
10
- * provenance must not depend on a flag we chose not to send. So the capture
11
- * records what is VERIFIABLE at launch for every configured harness:
12
- *
13
- * - `<harness> --version` output (the exact CLI build that authored),
14
- * - the claude settings default model (`~/.claude/settings.json` `model`),
15
- * - `codex login status` (the codex seat must be authed or the run refuses
16
- * at t=0 — a mid-run auth failure would silently kill one candidate slot),
17
- * - the spec's pinned model id (null when the seat rides the CLI default).
18
- *
19
- * The capture FAILS LOUD on a missing/broken harness binary: populationSize
20
- * must equal proposers.length, so a dead seat cannot be skipped at runtime —
21
- * the config decides membership, this guard proves it at launch.
22
- */
23
-
24
- import { readFileSync } from 'node:fs'
25
- import { homedir } from 'node:os'
26
- import { join } from 'node:path'
27
- import { localHarnessExecutable } from '@tangle-network/agent-runtime/mcp'
28
- import {
29
- DEFAULT_GEPA_PYTHON,
30
- isGepaSeat,
31
- probeGepaRuntime,
32
- } from './gepa-seat.mts'
33
- import { run } from './proc.ts'
34
- import type { ProposerSpec } from './proposer-fanout.mts'
35
-
36
- export interface ProposerModelProvenance {
37
- name: string
38
- /** Absent on an engine seat (see `engine`). */
39
- harness: ProposerSpec['harness']
40
- /** Explicit model pin from the spec (threaded as `-m`), or null when the
41
- * seat runs the CLI's own resolved default. */
42
- pinnedModel: string | null
43
- /** `<harness> --version` stdout (trimmed). For a GEPA seat this is
44
- * the bridge python's `--version` output — the runtime that authors. */
45
- harnessVersion: string
46
- /** claude seats only: the settings default model the logged-in CLI resolves
47
- * when no `-m` is passed. Null when unreadable (recorded, never fatal —
48
- * the version capture is the hard gate). */
49
- settingsModel: string | null
50
- /** codex seats only: `codex login status` stdout (trimmed). */
51
- authStatus: string | null
52
- merge: boolean
53
- /** GEPA seat: the engine name from the spec. */
54
- engine?: 'gepa' | 'omni'
55
- /** GEPA seat: the one change-space file GEPA optimizes. */
56
- surface?: string
57
- /** GEPA seat: installed GEPA version ('source' for a source pin). */
58
- gepaVersion?: string
59
- /** GEPA seat: the Python bridge module the seat runs. */
60
- bridge?: string
61
- }
62
-
63
- export interface ProvenanceCaptureRecord {
64
- schema: 'swe-arena.proposer-provenance.v1'
65
- capturedAt: string
66
- proposers: ProposerModelProvenance[]
67
- }
68
-
69
- /** Exec seam — test-injectable. Mirrors `run` from proc.ts. */
70
- export type VersionExec = (
71
- command: string,
72
- args: string[],
73
- ) => Promise<{ code: number | null; stdout: string; stderr: string }>
74
-
75
- const defaultExec: VersionExec = async (command, args) => {
76
- const res = await run(command, args, { timeoutMs: 30_000 })
77
- return { code: res.code, stdout: res.stdout, stderr: res.stderr }
78
- }
79
-
80
- /** Read the claude CLI's settings default model. Pure over injected reader. */
81
- export function claudeSettingsModel(
82
- readFile: (path: string) => string = (p) => readFileSync(p, 'utf8'),
83
- settingsPath = join(homedir(), '.claude', 'settings.json'),
84
- ): string | null {
85
- try {
86
- const parsed = JSON.parse(readFile(settingsPath)) as { model?: unknown }
87
- return typeof parsed.model === 'string' && parsed.model.length > 0 ? parsed.model : null
88
- } catch {
89
- return null
90
- }
91
- }
92
-
93
- /** Capture per-proposer model provenance. Throws when any configured harness
94
- * binary is missing/broken, when a codex seat is not logged in, or when
95
- * a GEPA seat's Python bridge or engine cannot import
96
- * (`probeGepaRuntime` carries the exact install instructions). A dead seat
97
- * fails the launch at t=0, never a mid-run candidate slot. The TypeScript
98
- * adapter is a pinned package dependency and therefore checked at install. */
99
- export async function captureProposerProvenance(
100
- proposers: ProposerSpec[],
101
- deps: {
102
- exec?: VersionExec
103
- readSettingsModel?: () => string | null
104
- } = {},
105
- ): Promise<ProvenanceCaptureRecord> {
106
- const exec = deps.exec ?? defaultExec
107
- const readSettingsModel = deps.readSettingsModel ?? (() => claudeSettingsModel())
108
- const versionByHarness = new Map<string, string>()
109
- const authByHarness = new Map<string, string>()
110
- const gepaBySeat = new Map<string, { pythonVersion: string; gepaVersion: string }>()
111
- const gepaSeats = proposers.filter(isGepaSeat)
112
- if (gepaSeats.length > 0) {
113
- for (const seat of gepaSeats) {
114
- gepaBySeat.set(seat.name, await probeGepaRuntime(seat.python ?? DEFAULT_GEPA_PYTHON, exec, seat.name))
115
- }
116
- }
117
- const harnesses = [...new Set(proposers.map((p) => p.harness))].filter(
118
- (h): h is NonNullable<ProposerSpec['harness']> => h !== undefined,
119
- )
120
- for (const harness of harnesses) {
121
- // The harness id is not the binary name (`claude-code` runs `claude`); read the executable
122
- // from the runtime's harness table rather than spawning the id.
123
- const executable = harness === 'pi' ? 'pi' : localHarnessExecutable(harness)
124
- const res = await exec(executable, ['--version'])
125
- if (res.code !== 0) {
126
- throw new Error(
127
- `proposer provenance: '${executable} --version' failed (rc=${res.code}) — the ${harness} seat cannot author. ` +
128
- `stderr: ${res.stderr.slice(0, 300)}`,
129
- )
130
- }
131
- versionByHarness.set(harness, res.stdout.trim())
132
- if (harness === 'codex') {
133
- const auth = await exec('codex', ['login', 'status'])
134
- const authOut = auth.stdout + auth.stderr
135
- const authed = /logged in/i.test(authOut) && !/not logged in/i.test(authOut)
136
- if (auth.code !== 0 || !authed) {
137
- throw new Error(
138
- `proposer provenance: codex seat configured but 'codex login status' says not authed ` +
139
- `(rc=${auth.code}, out=${(auth.stdout + auth.stderr).trim().slice(0, 200)})`,
140
- )
141
- }
142
- authByHarness.set('codex', (auth.stdout + auth.stderr).trim())
143
- }
144
- }
145
- return {
146
- schema: 'swe-arena.proposer-provenance.v1',
147
- capturedAt: new Date().toISOString(),
148
- proposers: proposers.map((spec): ProposerModelProvenance => {
149
- if (isGepaSeat(spec)) {
150
- const probe = gepaBySeat.get(spec.name)!
151
- return {
152
- name: spec.name,
153
- harness: undefined,
154
- pinnedModel: null,
155
- harnessVersion: probe.pythonVersion,
156
- settingsModel: null,
157
- authStatus: null,
158
- merge: false,
159
- engine: spec.engine,
160
- surface: spec.surface,
161
- gepaVersion: probe.gepaVersion,
162
- bridge: 'agent_eval_rpc.gepa_bridge',
163
- }
164
- }
165
- return {
166
- name: spec.name,
167
- harness: spec.harness,
168
- pinnedModel: spec.model ?? null,
169
- harnessVersion: versionByHarness.get(spec.harness!)!,
170
- settingsModel: spec.harness === 'claude-code' && !spec.model ? readSettingsModel() : null,
171
- authStatus: (spec.harness !== undefined ? authByHarness.get(spec.harness) : undefined) ?? null,
172
- merge: spec.merge === true,
173
- }
174
- }),
175
- }
176
- }
@@ -1,106 +0,0 @@
1
- import { describe, expect, it } from 'vitest'
2
- import {
3
- captureProposerProvenance,
4
- claudeSettingsModel,
5
- type VersionExec,
6
- } from './proposer-provenance.mts'
7
- import type { ProposerSpec } from './proposer-fanout.mts'
8
-
9
- const ok = (stdout: string) => ({ code: 0, stdout, stderr: '' })
10
-
11
- const gen4ish: ProposerSpec[] = [
12
- { name: 'claude-author', profile: 'default-author.profile.json', harness: 'claude-code' },
13
- { name: 'glm-author', harness: 'opencode', model: 'zai-coding-plan/glm-5.2' },
14
- { name: 'codex-author', harness: 'codex' },
15
- { name: 'merge-author', profile: 'default-author.profile.json', harness: 'claude-code', merge: true },
16
- ]
17
-
18
- describe('captureProposerProvenance', () => {
19
- const exec: VersionExec = async (command, args) => {
20
- if (args[0] === '--version') return ok(`${command}-version 9.9.9`)
21
- if (command === 'codex' && args[0] === 'login') return ok('Logged in using ChatGPT')
22
- throw new Error(`unexpected exec ${command} ${args.join(' ')}`)
23
- }
24
-
25
- it('records pinned model, harness version, settings model, and codex auth per seat', async () => {
26
- const record = await captureProposerProvenance(gen4ish, {
27
- exec,
28
- readSettingsModel: () => 'claude-fable-5',
29
- })
30
- expect(record.schema).toBe('swe-arena.proposer-provenance.v1')
31
- const byName = Object.fromEntries(record.proposers.map((p) => [p.name, p]))
32
- // Claude seat: no pin — the CLI's resolved settings model is the record.
33
- expect(byName['claude-author']).toMatchObject({
34
- harness: 'claude-code',
35
- pinnedModel: null,
36
- settingsModel: 'claude-fable-5',
37
- harnessVersion: 'claude-version 9.9.9',
38
- authStatus: null,
39
- merge: false,
40
- })
41
- // Pinned opencode seat: explicit model id, settings model not consulted.
42
- expect(byName['glm-author']).toMatchObject({
43
- harness: 'opencode',
44
- pinnedModel: 'zai-coding-plan/glm-5.2',
45
- settingsModel: null,
46
- harnessVersion: 'opencode-version 9.9.9',
47
- })
48
- // Codex seat: version + auth status captured.
49
- expect(byName['codex-author']).toMatchObject({
50
- harness: 'codex',
51
- pinnedModel: null,
52
- authStatus: 'Logged in using ChatGPT',
53
- })
54
- expect(byName['merge-author']!.merge).toBe(true)
55
- })
56
-
57
- it('fails loud when a configured harness binary is missing', async () => {
58
- const broken: VersionExec = async (command, args) =>
59
- command === 'codex' ? { code: 127, stdout: '', stderr: 'not found' } : exec(command, args)
60
- await expect(
61
- captureProposerProvenance(gen4ish, { exec: broken, readSettingsModel: () => null }),
62
- ).rejects.toThrow(/'codex --version' failed/)
63
- })
64
-
65
- it('fails loud when the codex seat is not logged in', async () => {
66
- const loggedOut: VersionExec = async (command, args) => {
67
- if (args[0] === '--version') return ok(`${command} 1.0.0`)
68
- return ok('Not logged in')
69
- }
70
- await expect(
71
- captureProposerProvenance([{ name: 'codex-author', harness: 'codex' }], {
72
- exec: loggedOut,
73
- readSettingsModel: () => null,
74
- }),
75
- ).rejects.toThrow(/not authed/)
76
- })
77
-
78
- it('runs one version probe per harness, not per proposer', async () => {
79
- const calls: string[] = []
80
- const counting: VersionExec = async (command, args) => {
81
- calls.push(`${command} ${args.join(' ')}`)
82
- if (args[0] === '--version') return ok(`${command} 1`)
83
- return ok('Logged in using ChatGPT')
84
- }
85
- await captureProposerProvenance(gen4ish, { exec: counting, readSettingsModel: () => null })
86
- expect(calls.filter((c) => c === 'claude --version')).toHaveLength(1)
87
- expect(calls.filter((c) => c === 'codex --version')).toHaveLength(1)
88
- expect(calls.filter((c) => c === 'codex login status')).toHaveLength(1)
89
- })
90
- })
91
-
92
- describe('claudeSettingsModel', () => {
93
- it('reads the settings model field', () => {
94
- expect(claudeSettingsModel(() => JSON.stringify({ model: 'claude-fable-5' }), '/x')).toBe('claude-fable-5')
95
- })
96
-
97
- it('returns null on unreadable/missing/blank settings', () => {
98
- expect(
99
- claudeSettingsModel(() => {
100
- throw new Error('ENOENT')
101
- }, '/x'),
102
- ).toBeNull()
103
- expect(claudeSettingsModel(() => JSON.stringify({}), '/x')).toBeNull()
104
- expect(claudeSettingsModel(() => JSON.stringify({ model: '' }), '/x')).toBeNull()
105
- })
106
- })
Binary file
@@ -1,183 +0,0 @@
1
- /**
2
- * Replay CLI: `tsx src/swe-arena/replay.mts`
3
- *
4
- * Reproduces the reference `fixtures/analyze.py` output from the committed
5
- * fixtures (section 1 matches its printed lines exactly — pinned in
6
- * replay.test.mts), then prints what the reference script never did:
7
- * the reconciled valid-denominator verdict, the true SUP spend including
8
- * worker tokens, the supervisor evolution rounds, and the holdout registry.
9
- */
10
-
11
- import { pathToFileURL } from 'node:url'
12
- import {
13
- loadHoldout,
14
- loadLedger,
15
- loadPreregisterLog,
16
- loadRejudge,
17
- loadRematchRounds,
18
- loadSupJournalTrue,
19
- loadWorkerTokens,
20
- } from './fixtures.ts'
21
- import {
22
- costRollup,
23
- ledgerOutcomes,
24
- pairedSignTest,
25
- reconciledOutcomes,
26
- roundsProgression,
27
- splitPairs,
28
- type CostRollup,
29
- type DiscordantSplit,
30
- type RoundState,
31
- type SignTestResult,
32
- } from './analyze.ts'
33
- import { reconcile, type PairedTable } from './reconcile.ts'
34
- import type { HoldoutRegistry, LedgerRow, RematchRow } from './types.ts'
35
-
36
- /** Python-style list repr (single quotes) so section 1 matches analyze.py byte-for-byte. */
37
- const pyList = (xs: string[]): string => `[${xs.map((x) => `'${x}'`).join(', ')}]`
38
- const signed = (x: number, digits?: number): string =>
39
- (x >= 0 ? '+' : '') + (digits === undefined ? String(x) : x.toFixed(digits))
40
-
41
- export interface Replay {
42
- ledger: LedgerRow[]
43
- raw: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
44
- cost: CostRollup
45
- table: PairedTable
46
- valid: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
47
- rounds: Map<string, RoundState[]>
48
- rematchRounds: RematchRow[][]
49
- holdout: HoldoutRegistry
50
- preregisterLog: string[]
51
- }
52
-
53
- export function buildReplay(): Replay {
54
- const ledger = loadLedger()
55
- const rejudge = loadRejudge()
56
- const rematchRounds = loadRematchRounds()
57
-
58
- const rawOutcomes = ledgerOutcomes(ledger)
59
- const table = reconcile(ledger, rejudge)
60
- const validOutcomes = reconciledOutcomes(table.valid)
61
-
62
- return {
63
- ledger,
64
- raw: {
65
- split: splitPairs(rawOutcomes),
66
- sign: pairedSignTest(rawOutcomes),
67
- soloResolved: rawOutcomes.filter((o) => o.solo).length,
68
- supResolved: rawOutcomes.filter((o) => o.sup).length,
69
- },
70
- cost: costRollup(ledger, loadSupJournalTrue(), loadWorkerTokens()),
71
- table,
72
- valid: {
73
- split: splitPairs(validOutcomes),
74
- sign: pairedSignTest(validOutcomes),
75
- soloResolved: validOutcomes.filter((o) => o.solo).length,
76
- supResolved: validOutcomes.filter((o) => o.sup).length,
77
- },
78
- rounds: roundsProgression(ledger, rematchRounds),
79
- rematchRounds,
80
- holdout: loadHoldout(),
81
- preregisterLog: loadPreregisterLog(),
82
- }
83
- }
84
-
85
- /** Section 1 — byte-faithful reproduction of analyze.py's printed analysis. */
86
- export function renderReference(r: Replay): string[] {
87
- const rows = [...r.ledger].sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
88
- const n = rows.length
89
- const { raw, cost } = r
90
- const bar = '='.repeat(80)
91
- const lines: string[] = []
92
- lines.push(bar)
93
- lines.push(`PAIRED HEAD-TO-HEAD: glm-5.2 SOLO vs glm-5.2 SUPERVISOR (N=${n} paired instances)`)
94
- lines.push(bar)
95
- lines.push(`SOLO resolved: ${raw.soloResolved}/${n} = ${((100 * raw.soloResolved) / n).toFixed(1)}%`)
96
- lines.push(`SUP resolved: ${raw.supResolved}/${n} = ${((100 * raw.supResolved) / n).toFixed(1)}%`)
97
- const delta = raw.supResolved - raw.soloResolved
98
- lines.push(`delta (SUP-SOLO): ${signed(delta)} instances (${signed((100 * delta) / n, 1)} pts)`)
99
- lines.push('')
100
- lines.push('DISCORDANT PAIRS (the signal):')
101
- lines.push(` SUP-only wins (SUP✓ SOLO✗): ${raw.split.supOnly.length} ${pyList(raw.split.supOnly)}`)
102
- lines.push(` SOLO-only wins (SOLO✓ SUP✗): ${raw.split.soloOnly.length} ${pyList(raw.split.soloOnly)}`)
103
- lines.push(` both resolved: ${raw.split.both.length} | neither: ${raw.split.neither.length} ${pyList(raw.split.neither)}`)
104
- lines.push(` exact two-sided sign test on discordant pairs: p=${raw.sign.pValue.toFixed(4)}`)
105
- lines.push('')
106
- lines.push(
107
- `COST (measured tokens; USD via shared blended rate $${(cost.blendedRatePerTok * 1e6).toFixed(3)}/1M from SUP accounting):`,
108
- )
109
- lines.push(` SOLO total tokens: ${cost.soloTokens.toLocaleString('en-US')} -> derived $${cost.soloUsdDerived.toFixed(4)}`)
110
- lines.push(` SUP total tokens: ${cost.supBrainTokens.toLocaleString('en-US')} -> runtime $${cost.supUsd.toFixed(4)}`)
111
- lines.push(` SUP/SOLO token ratio: ${cost.brainTokenRatio.toFixed(2)}x`)
112
- lines.push(` SUP/SOLO cost ratio (token-derived): ${(cost.supUsd / cost.soloUsdDerived).toFixed(2)}x`)
113
- lines.push(` [telemetry] instances where runtime spentTokens != journal-true (no-winner zeroing): ${pyList(cost.telemetryGaps)}`)
114
- lines.push(` WALL: SOLO ${cost.soloWallS}s total vs SUP ${cost.supWallS}s total -> SUP ${cost.wallRatio.toFixed(2)}x wall`)
115
- lines.push('')
116
- lines.push(bar)
117
- lines.push('PER-INSTANCE')
118
- lines.push(bar)
119
- lines.push(
120
- `${'instance'.padEnd(32)} ${'SOLO'.padEnd(5)} ${'SUP'.padEnd(5)} ${'v_s'.padEnd(3)} ${'v_p'.padEnd(3)} ${'wrk'.padEnd(3)} ${'soloTok'.padEnd(8)} ${'supTok'.padEnd(8)} ${'supUSD'.padEnd(7)} ${'soloW'.padEnd(5)} ${'supW'.padEnd(5)}`,
121
- )
122
- for (const row of rows) {
123
- const pyBool = (v: boolean): string => (v ? 'True' : 'False')
124
- lines.push(
125
- `${row.iid.padEnd(32)} ${pyBool(row.solo_resolved).slice(0, 5).padEnd(5)} ${pyBool(row.sup_resolved).slice(0, 5).padEnd(5)} ` +
126
- `${pyBool(row.solo_verify_pass)[0].padEnd(3)} ${pyBool(row.sup_verify_pass)[0].padEnd(3)} ` +
127
- `${String(row.sup_workers ?? '?').padEnd(3)} ${String(row.solo_tokens).padEnd(8)} ${String(row.sup_spentTokens ?? 0).padEnd(8)} ` +
128
- `${(row.sup_spentUsd ?? 0).toFixed(4)} ${String(row.solo_wall_s).padEnd(5)} ${String(row.sup_wall_s).padEnd(5)}`,
129
- )
130
- }
131
- return lines
132
- }
133
-
134
- /** Sections 2-5 — the analysis that lived in session lore, now typed. */
135
- export function renderReconciled(r: Replay): string[] {
136
- const bar = '='.repeat(80)
137
- const lines: string[] = []
138
- const { table, valid, cost } = r
139
- const n = table.valid.length
140
-
141
- lines.push(bar)
142
- lines.push('RECONCILED VERDICT (re-judged, gold-gated denominator)')
143
- lines.push(bar)
144
- for (const e of table.excluded) lines.push(` EXCLUDED ${e.iid}: ${e.excludeReason}`)
145
- for (const v of table.valid) {
146
- const src = [v.solo.source !== 'ledger' ? `solo:${v.solo.source}` : null, v.sup.source !== 'ledger' ? `sup:${v.sup.source}` : null]
147
- .filter(Boolean)
148
- .join(' ')
149
- if (src) lines.push(` RE-JUDGED ${v.iid}: ${src}`)
150
- }
151
- lines.push(`SOLO resolved: ${valid.soloResolved}/${n}`)
152
- lines.push(`SUP resolved: ${valid.supResolved}/${n}`)
153
- lines.push(`discordant: SUP-only ${pyList(valid.split.supOnly)} | SOLO-only ${pyList(valid.split.soloOnly)}`)
154
- lines.push(`exact two-sided sign test: p=${valid.sign.pValue.toFixed(4)}`)
155
- lines.push('')
156
- lines.push('TRUE SUP SPEND (brain + workers; analyze.py printed brain only):')
157
- lines.push(` brain ${cost.supBrainTokens.toLocaleString('en-US')} + workers ${cost.supWorkerTokens.toLocaleString('en-US')} = ${cost.supTotalTokens.toLocaleString('en-US')} tokens`)
158
- lines.push(` SUP/SOLO true token ratio: ${cost.totalTokenRatio.toFixed(2)}x (brain-only ratio: ${cost.brainTokenRatio.toFixed(2)}x)`)
159
- lines.push('')
160
- lines.push('SUP EVOLUTION ROUNDS (SUP = original head-to-head run):')
161
- for (const [iid, states] of r.rounds) {
162
- const cells = states.map(
163
- (s) => `${s.round}:${s.resolved ? 'RESOLVED' : 'unresolved'}(${s.patchLines}L,${s.verdict ?? 'null'})`,
164
- )
165
- lines.push(` ${iid.padEnd(32)} ${cells.join(' -> ')}`)
166
- }
167
- lines.push('')
168
- lines.push(`HOLDOUT REGISTRY (pre-registered at loops@${r.holdout.selectedAtCommit.slice(0, 10)}, untouched):`)
169
- for (const e of r.holdout.entries) {
170
- lines.push(` ${e.iid.padEnd(36)} gold_official_resolved=${e.gold_official_resolved} verify_calibrated=${e.verify_calibrated}`)
171
- }
172
- return lines
173
- }
174
-
175
- export function renderReplay(r: Replay = buildReplay()): string {
176
- return [...renderReference(r), '', ...renderReconciled(r)].join('\n')
177
- }
178
-
179
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
180
-
181
- if (isMain) {
182
- console.log(renderReplay())
183
- }