@tangle-network/agent-bench 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,176 +0,0 @@
1
- /**
2
- * GEN-4 proposer-model provenance — pin each author seat's MODEL IDENTITY in
3
- * the run record at t=0, before any author shot fires.
4
- *
5
- * Why: a proposer spec may pin a model explicitly (`spec.model` → the harness
6
- * CLI's `-m` flag via the author profile's `model.default`), but the claude
7
- * seat deliberately does NOT pass `-m` — the CLI runs on its logged-in
8
- * account, whose resolved model comes from its own settings. Verified against
9
- * claude CLI 2.1.217: `--model` IS supported headless (`-p`), but the run's
10
- * provenance must not depend on a flag we chose not to send. So the capture
11
- * records what is VERIFIABLE at launch for every configured harness:
12
- *
13
- * - `<harness> --version` output (the exact CLI build that authored),
14
- * - the claude settings default model (`~/.claude/settings.json` `model`),
15
- * - `codex login status` (the codex seat must be authed or the run refuses
16
- * at t=0 — a mid-run auth failure would silently kill one candidate slot),
17
- * - the spec's pinned model id (null when the seat rides the CLI default).
18
- *
19
- * The capture FAILS LOUD on a missing/broken harness binary: populationSize
20
- * must equal proposers.length, so a dead seat cannot be skipped at runtime —
21
- * the config decides membership, this guard proves it at launch.
22
- */
23
-
24
- import { readFileSync } from 'node:fs'
25
- import { homedir } from 'node:os'
26
- import { join } from 'node:path'
27
- import { localHarnessExecutable } from '@tangle-network/agent-runtime/mcp'
28
- import {
29
- DEFAULT_GEPA_PYTHON,
30
- isGepaSeat,
31
- probeGepaRuntime,
32
- } from './gepa-seat.mts'
33
- import { run } from './proc.ts'
34
- import type { ProposerSpec } from './proposer-fanout.mts'
35
-
36
- export interface ProposerModelProvenance {
37
- name: string
38
- /** Absent on an engine seat (see `engine`). */
39
- harness: ProposerSpec['harness']
40
- /** Explicit model pin from the spec (threaded as `-m`), or null when the
41
- * seat runs the CLI's own resolved default. */
42
- pinnedModel: string | null
43
- /** `<harness> --version` stdout (trimmed). For a GEPA seat this is
44
- * the bridge python's `--version` output — the runtime that authors. */
45
- harnessVersion: string
46
- /** claude seats only: the settings default model the logged-in CLI resolves
47
- * when no `-m` is passed. Null when unreadable (recorded, never fatal —
48
- * the version capture is the hard gate). */
49
- settingsModel: string | null
50
- /** codex seats only: `codex login status` stdout (trimmed). */
51
- authStatus: string | null
52
- merge: boolean
53
- /** GEPA seat: the engine name from the spec. */
54
- engine?: 'gepa' | 'omni'
55
- /** GEPA seat: the one change-space file GEPA optimizes. */
56
- surface?: string
57
- /** GEPA seat: installed GEPA version ('source' for a source pin). */
58
- gepaVersion?: string
59
- /** GEPA seat: the Python bridge module the seat runs. */
60
- bridge?: string
61
- }
62
-
63
- export interface ProvenanceCaptureRecord {
64
- schema: 'swe-arena.proposer-provenance.v1'
65
- capturedAt: string
66
- proposers: ProposerModelProvenance[]
67
- }
68
-
69
- /** Exec seam — test-injectable. Mirrors `run` from proc.ts. */
70
- export type VersionExec = (
71
- command: string,
72
- args: string[],
73
- ) => Promise<{ code: number | null; stdout: string; stderr: string }>
74
-
75
- const defaultExec: VersionExec = async (command, args) => {
76
- const res = await run(command, args, { timeoutMs: 30_000 })
77
- return { code: res.code, stdout: res.stdout, stderr: res.stderr }
78
- }
79
-
80
- /** Read the claude CLI's settings default model. Pure over injected reader. */
81
- export function claudeSettingsModel(
82
- readFile: (path: string) => string = (p) => readFileSync(p, 'utf8'),
83
- settingsPath = join(homedir(), '.claude', 'settings.json'),
84
- ): string | null {
85
- try {
86
- const parsed = JSON.parse(readFile(settingsPath)) as { model?: unknown }
87
- return typeof parsed.model === 'string' && parsed.model.length > 0 ? parsed.model : null
88
- } catch {
89
- return null
90
- }
91
- }
92
-
93
- /** Capture per-proposer model provenance. Throws when any configured harness
94
- * binary is missing/broken, when a codex seat is not logged in, or when
95
- * a GEPA seat's Python bridge or engine cannot import
96
- * (`probeGepaRuntime` carries the exact install instructions). A dead seat
97
- * fails the launch at t=0, never a mid-run candidate slot. The TypeScript
98
- * adapter is a pinned package dependency and therefore checked at install. */
99
- export async function captureProposerProvenance(
100
- proposers: ProposerSpec[],
101
- deps: {
102
- exec?: VersionExec
103
- readSettingsModel?: () => string | null
104
- } = {},
105
- ): Promise<ProvenanceCaptureRecord> {
106
- const exec = deps.exec ?? defaultExec
107
- const readSettingsModel = deps.readSettingsModel ?? (() => claudeSettingsModel())
108
- const versionByHarness = new Map<string, string>()
109
- const authByHarness = new Map<string, string>()
110
- const gepaBySeat = new Map<string, { pythonVersion: string; gepaVersion: string }>()
111
- const gepaSeats = proposers.filter(isGepaSeat)
112
- if (gepaSeats.length > 0) {
113
- for (const seat of gepaSeats) {
114
- gepaBySeat.set(seat.name, await probeGepaRuntime(seat.python ?? DEFAULT_GEPA_PYTHON, exec, seat.name))
115
- }
116
- }
117
- const harnesses = [...new Set(proposers.map((p) => p.harness))].filter(
118
- (h): h is NonNullable<ProposerSpec['harness']> => h !== undefined,
119
- )
120
- for (const harness of harnesses) {
121
- // The harness id is not the binary name (`claude-code` runs `claude`); read the executable
122
- // from the runtime's harness table rather than spawning the id.
123
- const executable = harness === 'pi' ? 'pi' : localHarnessExecutable(harness)
124
- const res = await exec(executable, ['--version'])
125
- if (res.code !== 0) {
126
- throw new Error(
127
- `proposer provenance: '${executable} --version' failed (rc=${res.code}) — the ${harness} seat cannot author. ` +
128
- `stderr: ${res.stderr.slice(0, 300)}`,
129
- )
130
- }
131
- versionByHarness.set(harness, res.stdout.trim())
132
- if (harness === 'codex') {
133
- const auth = await exec('codex', ['login', 'status'])
134
- const authOut = auth.stdout + auth.stderr
135
- const authed = /logged in/i.test(authOut) && !/not logged in/i.test(authOut)
136
- if (auth.code !== 0 || !authed) {
137
- throw new Error(
138
- `proposer provenance: codex seat configured but 'codex login status' says not authed ` +
139
- `(rc=${auth.code}, out=${(auth.stdout + auth.stderr).trim().slice(0, 200)})`,
140
- )
141
- }
142
- authByHarness.set('codex', (auth.stdout + auth.stderr).trim())
143
- }
144
- }
145
- return {
146
- schema: 'swe-arena.proposer-provenance.v1',
147
- capturedAt: new Date().toISOString(),
148
- proposers: proposers.map((spec): ProposerModelProvenance => {
149
- if (isGepaSeat(spec)) {
150
- const probe = gepaBySeat.get(spec.name)!
151
- return {
152
- name: spec.name,
153
- harness: undefined,
154
- pinnedModel: null,
155
- harnessVersion: probe.pythonVersion,
156
- settingsModel: null,
157
- authStatus: null,
158
- merge: false,
159
- engine: spec.engine,
160
- surface: spec.surface,
161
- gepaVersion: probe.gepaVersion,
162
- bridge: 'agent_eval_rpc.gepa_bridge',
163
- }
164
- }
165
- return {
166
- name: spec.name,
167
- harness: spec.harness,
168
- pinnedModel: spec.model ?? null,
169
- harnessVersion: versionByHarness.get(spec.harness!)!,
170
- settingsModel: spec.harness === 'claude-code' && !spec.model ? readSettingsModel() : null,
171
- authStatus: (spec.harness !== undefined ? authByHarness.get(spec.harness) : undefined) ?? null,
172
- merge: spec.merge === true,
173
- }
174
- }),
175
- }
176
- }
@@ -1,106 +0,0 @@
1
- import { describe, expect, it } from 'vitest'
2
- import {
3
- captureProposerProvenance,
4
- claudeSettingsModel,
5
- type VersionExec,
6
- } from './proposer-provenance.mts'
7
- import type { ProposerSpec } from './proposer-fanout.mts'
8
-
9
- const ok = (stdout: string) => ({ code: 0, stdout, stderr: '' })
10
-
11
- const gen4ish: ProposerSpec[] = [
12
- { name: 'claude-author', profile: 'default-author.profile.json', harness: 'claude-code' },
13
- { name: 'glm-author', harness: 'opencode', model: 'zai-coding-plan/glm-5.2' },
14
- { name: 'codex-author', harness: 'codex' },
15
- { name: 'merge-author', profile: 'default-author.profile.json', harness: 'claude-code', merge: true },
16
- ]
17
-
18
- describe('captureProposerProvenance', () => {
19
- const exec: VersionExec = async (command, args) => {
20
- if (args[0] === '--version') return ok(`${command}-version 9.9.9`)
21
- if (command === 'codex' && args[0] === 'login') return ok('Logged in using ChatGPT')
22
- throw new Error(`unexpected exec ${command} ${args.join(' ')}`)
23
- }
24
-
25
- it('records pinned model, harness version, settings model, and codex auth per seat', async () => {
26
- const record = await captureProposerProvenance(gen4ish, {
27
- exec,
28
- readSettingsModel: () => 'claude-fable-5',
29
- })
30
- expect(record.schema).toBe('swe-arena.proposer-provenance.v1')
31
- const byName = Object.fromEntries(record.proposers.map((p) => [p.name, p]))
32
- // Claude seat: no pin — the CLI's resolved settings model is the record.
33
- expect(byName['claude-author']).toMatchObject({
34
- harness: 'claude-code',
35
- pinnedModel: null,
36
- settingsModel: 'claude-fable-5',
37
- harnessVersion: 'claude-version 9.9.9',
38
- authStatus: null,
39
- merge: false,
40
- })
41
- // Pinned opencode seat: explicit model id, settings model not consulted.
42
- expect(byName['glm-author']).toMatchObject({
43
- harness: 'opencode',
44
- pinnedModel: 'zai-coding-plan/glm-5.2',
45
- settingsModel: null,
46
- harnessVersion: 'opencode-version 9.9.9',
47
- })
48
- // Codex seat: version + auth status captured.
49
- expect(byName['codex-author']).toMatchObject({
50
- harness: 'codex',
51
- pinnedModel: null,
52
- authStatus: 'Logged in using ChatGPT',
53
- })
54
- expect(byName['merge-author']!.merge).toBe(true)
55
- })
56
-
57
- it('fails loud when a configured harness binary is missing', async () => {
58
- const broken: VersionExec = async (command, args) =>
59
- command === 'codex' ? { code: 127, stdout: '', stderr: 'not found' } : exec(command, args)
60
- await expect(
61
- captureProposerProvenance(gen4ish, { exec: broken, readSettingsModel: () => null }),
62
- ).rejects.toThrow(/'codex --version' failed/)
63
- })
64
-
65
- it('fails loud when the codex seat is not logged in', async () => {
66
- const loggedOut: VersionExec = async (command, args) => {
67
- if (args[0] === '--version') return ok(`${command} 1.0.0`)
68
- return ok('Not logged in')
69
- }
70
- await expect(
71
- captureProposerProvenance([{ name: 'codex-author', harness: 'codex' }], {
72
- exec: loggedOut,
73
- readSettingsModel: () => null,
74
- }),
75
- ).rejects.toThrow(/not authed/)
76
- })
77
-
78
- it('runs one version probe per harness, not per proposer', async () => {
79
- const calls: string[] = []
80
- const counting: VersionExec = async (command, args) => {
81
- calls.push(`${command} ${args.join(' ')}`)
82
- if (args[0] === '--version') return ok(`${command} 1`)
83
- return ok('Logged in using ChatGPT')
84
- }
85
- await captureProposerProvenance(gen4ish, { exec: counting, readSettingsModel: () => null })
86
- expect(calls.filter((c) => c === 'claude --version')).toHaveLength(1)
87
- expect(calls.filter((c) => c === 'codex --version')).toHaveLength(1)
88
- expect(calls.filter((c) => c === 'codex login status')).toHaveLength(1)
89
- })
90
- })
91
-
92
- describe('claudeSettingsModel', () => {
93
- it('reads the settings model field', () => {
94
- expect(claudeSettingsModel(() => JSON.stringify({ model: 'claude-fable-5' }), '/x')).toBe('claude-fable-5')
95
- })
96
-
97
- it('returns null on unreadable/missing/blank settings', () => {
98
- expect(
99
- claudeSettingsModel(() => {
100
- throw new Error('ENOENT')
101
- }, '/x'),
102
- ).toBeNull()
103
- expect(claudeSettingsModel(() => JSON.stringify({}), '/x')).toBeNull()
104
- expect(claudeSettingsModel(() => JSON.stringify({ model: '' }), '/x')).toBeNull()
105
- })
106
- })
Binary file
@@ -1,183 +0,0 @@
1
- /**
2
- * Replay CLI: `tsx src/swe-arena/replay.mts`
3
- *
4
- * Reproduces the reference `fixtures/analyze.py` output from the committed
5
- * fixtures (section 1 matches its printed lines exactly — pinned in
6
- * replay.test.mts), then prints what the reference script never did:
7
- * the reconciled valid-denominator verdict, the true SUP spend including
8
- * worker tokens, the supervisor evolution rounds, and the holdout registry.
9
- */
10
-
11
- import { pathToFileURL } from 'node:url'
12
- import {
13
- loadHoldout,
14
- loadLedger,
15
- loadPreregisterLog,
16
- loadRejudge,
17
- loadRematchRounds,
18
- loadSupJournalTrue,
19
- loadWorkerTokens,
20
- } from './fixtures.ts'
21
- import {
22
- costRollup,
23
- ledgerOutcomes,
24
- pairedSignTest,
25
- reconciledOutcomes,
26
- roundsProgression,
27
- splitPairs,
28
- type CostRollup,
29
- type DiscordantSplit,
30
- type RoundState,
31
- type SignTestResult,
32
- } from './analyze.ts'
33
- import { reconcile, type PairedTable } from './reconcile.ts'
34
- import type { HoldoutRegistry, LedgerRow, RematchRow } from './types.ts'
35
-
36
- /** Python-style list repr (single quotes) so section 1 matches analyze.py byte-for-byte. */
37
- const pyList = (xs: string[]): string => `[${xs.map((x) => `'${x}'`).join(', ')}]`
38
- const signed = (x: number, digits?: number): string =>
39
- (x >= 0 ? '+' : '') + (digits === undefined ? String(x) : x.toFixed(digits))
40
-
41
- export interface Replay {
42
- ledger: LedgerRow[]
43
- raw: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
44
- cost: CostRollup
45
- table: PairedTable
46
- valid: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
47
- rounds: Map<string, RoundState[]>
48
- rematchRounds: RematchRow[][]
49
- holdout: HoldoutRegistry
50
- preregisterLog: string[]
51
- }
52
-
53
- export function buildReplay(): Replay {
54
- const ledger = loadLedger()
55
- const rejudge = loadRejudge()
56
- const rematchRounds = loadRematchRounds()
57
-
58
- const rawOutcomes = ledgerOutcomes(ledger)
59
- const table = reconcile(ledger, rejudge)
60
- const validOutcomes = reconciledOutcomes(table.valid)
61
-
62
- return {
63
- ledger,
64
- raw: {
65
- split: splitPairs(rawOutcomes),
66
- sign: pairedSignTest(rawOutcomes),
67
- soloResolved: rawOutcomes.filter((o) => o.solo).length,
68
- supResolved: rawOutcomes.filter((o) => o.sup).length,
69
- },
70
- cost: costRollup(ledger, loadSupJournalTrue(), loadWorkerTokens()),
71
- table,
72
- valid: {
73
- split: splitPairs(validOutcomes),
74
- sign: pairedSignTest(validOutcomes),
75
- soloResolved: validOutcomes.filter((o) => o.solo).length,
76
- supResolved: validOutcomes.filter((o) => o.sup).length,
77
- },
78
- rounds: roundsProgression(ledger, rematchRounds),
79
- rematchRounds,
80
- holdout: loadHoldout(),
81
- preregisterLog: loadPreregisterLog(),
82
- }
83
- }
84
-
85
- /** Section 1 — byte-faithful reproduction of analyze.py's printed analysis. */
86
- export function renderReference(r: Replay): string[] {
87
- const rows = [...r.ledger].sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
88
- const n = rows.length
89
- const { raw, cost } = r
90
- const bar = '='.repeat(80)
91
- const lines: string[] = []
92
- lines.push(bar)
93
- lines.push(`PAIRED HEAD-TO-HEAD: glm-5.2 SOLO vs glm-5.2 SUPERVISOR (N=${n} paired instances)`)
94
- lines.push(bar)
95
- lines.push(`SOLO resolved: ${raw.soloResolved}/${n} = ${((100 * raw.soloResolved) / n).toFixed(1)}%`)
96
- lines.push(`SUP resolved: ${raw.supResolved}/${n} = ${((100 * raw.supResolved) / n).toFixed(1)}%`)
97
- const delta = raw.supResolved - raw.soloResolved
98
- lines.push(`delta (SUP-SOLO): ${signed(delta)} instances (${signed((100 * delta) / n, 1)} pts)`)
99
- lines.push('')
100
- lines.push('DISCORDANT PAIRS (the signal):')
101
- lines.push(` SUP-only wins (SUP✓ SOLO✗): ${raw.split.supOnly.length} ${pyList(raw.split.supOnly)}`)
102
- lines.push(` SOLO-only wins (SOLO✓ SUP✗): ${raw.split.soloOnly.length} ${pyList(raw.split.soloOnly)}`)
103
- lines.push(` both resolved: ${raw.split.both.length} | neither: ${raw.split.neither.length} ${pyList(raw.split.neither)}`)
104
- lines.push(` exact two-sided sign test on discordant pairs: p=${raw.sign.pValue.toFixed(4)}`)
105
- lines.push('')
106
- lines.push(
107
- `COST (measured tokens; USD via shared blended rate $${(cost.blendedRatePerTok * 1e6).toFixed(3)}/1M from SUP accounting):`,
108
- )
109
- lines.push(` SOLO total tokens: ${cost.soloTokens.toLocaleString('en-US')} -> derived $${cost.soloUsdDerived.toFixed(4)}`)
110
- lines.push(` SUP total tokens: ${cost.supBrainTokens.toLocaleString('en-US')} -> runtime $${cost.supUsd.toFixed(4)}`)
111
- lines.push(` SUP/SOLO token ratio: ${cost.brainTokenRatio.toFixed(2)}x`)
112
- lines.push(` SUP/SOLO cost ratio (token-derived): ${(cost.supUsd / cost.soloUsdDerived).toFixed(2)}x`)
113
- lines.push(` [telemetry] instances where runtime spentTokens != journal-true (no-winner zeroing): ${pyList(cost.telemetryGaps)}`)
114
- lines.push(` WALL: SOLO ${cost.soloWallS}s total vs SUP ${cost.supWallS}s total -> SUP ${cost.wallRatio.toFixed(2)}x wall`)
115
- lines.push('')
116
- lines.push(bar)
117
- lines.push('PER-INSTANCE')
118
- lines.push(bar)
119
- lines.push(
120
- `${'instance'.padEnd(32)} ${'SOLO'.padEnd(5)} ${'SUP'.padEnd(5)} ${'v_s'.padEnd(3)} ${'v_p'.padEnd(3)} ${'wrk'.padEnd(3)} ${'soloTok'.padEnd(8)} ${'supTok'.padEnd(8)} ${'supUSD'.padEnd(7)} ${'soloW'.padEnd(5)} ${'supW'.padEnd(5)}`,
121
- )
122
- for (const row of rows) {
123
- const pyBool = (v: boolean): string => (v ? 'True' : 'False')
124
- lines.push(
125
- `${row.iid.padEnd(32)} ${pyBool(row.solo_resolved).slice(0, 5).padEnd(5)} ${pyBool(row.sup_resolved).slice(0, 5).padEnd(5)} ` +
126
- `${pyBool(row.solo_verify_pass)[0].padEnd(3)} ${pyBool(row.sup_verify_pass)[0].padEnd(3)} ` +
127
- `${String(row.sup_workers ?? '?').padEnd(3)} ${String(row.solo_tokens).padEnd(8)} ${String(row.sup_spentTokens ?? 0).padEnd(8)} ` +
128
- `${(row.sup_spentUsd ?? 0).toFixed(4)} ${String(row.solo_wall_s).padEnd(5)} ${String(row.sup_wall_s).padEnd(5)}`,
129
- )
130
- }
131
- return lines
132
- }
133
-
134
- /** Sections 2-5 — the analysis that lived in session lore, now typed. */
135
- export function renderReconciled(r: Replay): string[] {
136
- const bar = '='.repeat(80)
137
- const lines: string[] = []
138
- const { table, valid, cost } = r
139
- const n = table.valid.length
140
-
141
- lines.push(bar)
142
- lines.push('RECONCILED VERDICT (re-judged, gold-gated denominator)')
143
- lines.push(bar)
144
- for (const e of table.excluded) lines.push(` EXCLUDED ${e.iid}: ${e.excludeReason}`)
145
- for (const v of table.valid) {
146
- const src = [v.solo.source !== 'ledger' ? `solo:${v.solo.source}` : null, v.sup.source !== 'ledger' ? `sup:${v.sup.source}` : null]
147
- .filter(Boolean)
148
- .join(' ')
149
- if (src) lines.push(` RE-JUDGED ${v.iid}: ${src}`)
150
- }
151
- lines.push(`SOLO resolved: ${valid.soloResolved}/${n}`)
152
- lines.push(`SUP resolved: ${valid.supResolved}/${n}`)
153
- lines.push(`discordant: SUP-only ${pyList(valid.split.supOnly)} | SOLO-only ${pyList(valid.split.soloOnly)}`)
154
- lines.push(`exact two-sided sign test: p=${valid.sign.pValue.toFixed(4)}`)
155
- lines.push('')
156
- lines.push('TRUE SUP SPEND (brain + workers; analyze.py printed brain only):')
157
- lines.push(` brain ${cost.supBrainTokens.toLocaleString('en-US')} + workers ${cost.supWorkerTokens.toLocaleString('en-US')} = ${cost.supTotalTokens.toLocaleString('en-US')} tokens`)
158
- lines.push(` SUP/SOLO true token ratio: ${cost.totalTokenRatio.toFixed(2)}x (brain-only ratio: ${cost.brainTokenRatio.toFixed(2)}x)`)
159
- lines.push('')
160
- lines.push('SUP EVOLUTION ROUNDS (SUP = original head-to-head run):')
161
- for (const [iid, states] of r.rounds) {
162
- const cells = states.map(
163
- (s) => `${s.round}:${s.resolved ? 'RESOLVED' : 'unresolved'}(${s.patchLines}L,${s.verdict ?? 'null'})`,
164
- )
165
- lines.push(` ${iid.padEnd(32)} ${cells.join(' -> ')}`)
166
- }
167
- lines.push('')
168
- lines.push(`HOLDOUT REGISTRY (pre-registered at loops@${r.holdout.selectedAtCommit.slice(0, 10)}, untouched):`)
169
- for (const e of r.holdout.entries) {
170
- lines.push(` ${e.iid.padEnd(36)} gold_official_resolved=${e.gold_official_resolved} verify_calibrated=${e.verify_calibrated}`)
171
- }
172
- return lines
173
- }
174
-
175
- export function renderReplay(r: Replay = buildReplay()): string {
176
- return [...renderReference(r), '', ...renderReconciled(r)].join('\n')
177
- }
178
-
179
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
180
-
181
- if (isMain) {
182
- console.log(renderReplay())
183
- }