@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,610 +0,0 @@
1
- /**
2
- * swe-arena → rollout backfill — DOMAIN GLUE ONLY. Bench owns the JOIN
3
- * (manifest + cached cells + judge.json + result.json workerCwds + proposer
4
- * shot receipts → harness stores); the `tangle.rollout.v1` schema, line
5
- * validation, serialization, readers, exporters, and release pipeline are
6
- * owned by `@tangle-network/agent-eval/rollout` — bench never writes a
7
- * rollout row through anything but that API.
8
- *
9
- * Pure READ over one round's outDir
10
- * (the same artifact tree `swe-arena/manifest.mts` joins) plus the harness
11
- * stores, emitting `tangle.rollout.v1` lines with capture:"backfill":
12
- *
13
- * - one SUPERVISOR line per campaign cell (instance × rep × candidate):
14
- * reward = the official-judge verdict off cached-result.json, verdict =
15
- * raw judge.json, cost = the durable episode receipt. The loops brain
16
- * persists per-request STATS only (brain.jsonl has no message content),
17
- * so supervisor transcripts are honest gap lines until settle-time
18
- * capture lands.
19
- * - one WORKER line per opencode worker session, joined via result.json
20
- * recoveredSpend.workerCwds → opencode sqlite session.directory, with the
21
- * FULL transcript inlined and reward inherited from the episode.
22
- * - one PROPOSER line per proposer shot receipt, joined to the Claude
23
- * project transcript of the shot's worktree cwd by time window.
24
- *
25
- * A failed join is a labeled gap line (messages: [], provenance.gap) — never
26
- * a silent drop: the ledger must account for every invocation it knows of.
27
- *
28
- * tsx src/rollout-ledger/backfill-swe-arena.mts <outDir> <out.jsonl>
29
- */
30
-
31
- import { randomUUID } from 'node:crypto'
32
- import { readdir, readFile } from 'node:fs/promises'
33
- import { join } from 'node:path'
34
- import { pathToFileURL } from 'node:url'
35
- import type { DatabaseSync } from 'node:sqlite'
36
- import { loadCampaignCellRecords } from '../swe-arena/cell-evidence.mts'
37
- import { buildRolloutManifest, type RolloutEntry } from '../swe-arena/manifest.mts'
38
- import {
39
- DEFAULT_CLAUDE_PROJECTS_DIR,
40
- DEFAULT_OPENCODE_DB,
41
- findClaudeTranscripts,
42
- findOpencodeSessionsByDirectory,
43
- openOpencodeDb,
44
- readClaudeTranscript,
45
- readOpencodeSessionMessages,
46
- ROLLOUT_SCHEMA,
47
- type OpencodeSessionRow,
48
- type RolloutLine,
49
- writeRolloutLedger,
50
- } from '@tangle-network/agent-eval/rollout'
51
- import { candidateId } from './settle-capture.mts'
52
-
53
- const OFFICIAL_JUDGE = 'swe-arena-official-judge'
54
-
55
- const isRecord = (v: unknown): v is Record<string, unknown> =>
56
- typeof v === 'object' && v !== null && !Array.isArray(v)
57
-
58
- /** Full cached-result.json record (superset of the EvidenceCell slice). */
59
- interface CellRecord {
60
- scenarioId: string
61
- rep: number
62
- artifact: Record<string, unknown> | null
63
- error?: string
64
- judgeScores?: Record<string, { composite?: number }>
65
- costUsd?: number
66
- tokenUsage?: { input?: number; output?: number }
67
- resolvedModel?: string
68
- seed?: number
69
- cached?: boolean
70
- }
71
-
72
- /** The same cell caches `loadCampaignCells` reads, kept verbatim: the backfill
73
- * needs fields (judgeScores, resolvedModel, seed) the EvidenceCell slice drops.
74
- * Traversal and the fail-loud identity check are owned by cell-evidence — one
75
- * rule for what counts as a campaign cell.
76
- *
77
- * Only scenarioId and rep are proven by that check; `artifact` is normalized to
78
- * a record-or-null here so the `artifact !== null` guards downstream cannot be
79
- * handed an `undefined` that passes them. Every remaining field is optional and
80
- * read behind its own typeof guard. */
81
- async function readCellRecords(campaignDir: string): Promise<CellRecord[]> {
82
- const records = (await loadCampaignCellRecords(campaignDir)).map((record) => ({
83
- ...record,
84
- artifact: isRecord(record.artifact) ? record.artifact : null,
85
- })) as unknown as CellRecord[]
86
- return records.sort((a, b) => a.scenarioId.localeCompare(b.scenarioId) || a.rep - b.rep)
87
- }
88
-
89
- async function readJson(path: string): Promise<unknown | null> {
90
- const raw = await readFile(path, 'utf8').catch(() => null)
91
- if (raw === null) return null
92
- try {
93
- return JSON.parse(raw) as unknown
94
- } catch {
95
- return null
96
- }
97
- }
98
-
99
- function splitModelRef(ref: string | undefined): { provider: string | null; model: string | null } {
100
- if (typeof ref !== 'string' || ref.length === 0) return { provider: null, model: null }
101
- const slash = ref.indexOf('/')
102
- if (slash === -1) return { provider: null, model: ref }
103
- return { provider: ref.slice(0, slash), model: ref.slice(slash + 1) }
104
- }
105
-
106
- export interface BackfillStats {
107
- lines: number
108
- byRole: Record<string, number>
109
- fullTranscripts: number
110
- gapLines: number
111
- rewardLabeled: number
112
- workerSessionsJoined: number
113
- /** Worker cwds with no opencode session row — a join failure. */
114
- workerCwdsMissed: number
115
- /** Sessions found but carrying no readable message parts — a store failure. */
116
- workerSessionsEmpty: number
117
- proposerShotsJoined: number
118
- proposerShotsMissed: number
119
- opencodeDbAvailable: boolean
120
- }
121
-
122
- export interface BackfillResult {
123
- lines: RolloutLine[]
124
- stats: BackfillStats
125
- }
126
-
127
- export interface BackfillOptions {
128
- opencodeDb?: string
129
- claudeProjectsDir?: string
130
- /** Injected clock for deterministic tests. */
131
- now?: () => Date
132
- }
133
-
134
- interface Ctx {
135
- outDir: string
136
- runId: string
137
- instanceCount: number
138
- db: DatabaseSync | null
139
- claudeProjectsDir: string
140
- capturedAt: string
141
- stats: BackfillStats
142
- }
143
-
144
- function baseLine(ctx: Ctx): Pick<RolloutLine, 'schema' | 'run_id' | 'experiment_id' | 'provenance'> {
145
- return {
146
- schema: ROLLOUT_SCHEMA,
147
- run_id: ctx.runId,
148
- experiment_id: null,
149
- provenance: { captured_at: ctx.capturedAt, capture: 'backfill' },
150
- }
151
- }
152
-
153
- function push(ctx: Ctx, lines: RolloutLine[], line: RolloutLine): void {
154
- lines.push(line)
155
- ctx.stats.lines += 1
156
- ctx.stats.byRole[line.role] = (ctx.stats.byRole[line.role] ?? 0) + 1
157
- if (line.messages.length > 0) ctx.stats.fullTranscripts += 1
158
- else ctx.stats.gapLines += 1
159
- if (line.outcome.reward !== null) ctx.stats.rewardLabeled += 1
160
- }
161
-
162
- // ---------------------------------------------------------------------------
163
- // Supervisor episode + worker sessions for one campaign cell.
164
- // ---------------------------------------------------------------------------
165
-
166
- async function emitCellLines(
167
- ctx: Ctx,
168
- lines: RolloutLine[],
169
- entry: RolloutEntry,
170
- cell: CellRecord,
171
- ): Promise<void> {
172
- const artifact = cell.artifact
173
- const runDir = artifact !== null && typeof artifact.runDir === 'string' ? artifact.runDir : null
174
- const judgeVerdict = runDir !== null ? await readJson(join(runDir, 'judge.json')) : null
175
- const resultJson = runDir !== null ? await readJson(join(runDir, 'result.json')) : null
176
- const result = isRecord(resultJson) ? resultJson : null
177
-
178
- const composite = cell.judgeScores?.[OFFICIAL_JUDGE]?.composite
179
- const reward =
180
- typeof composite === 'number'
181
- ? composite
182
- : artifact !== null && typeof artifact.resolved === 'boolean'
183
- ? (artifact.resolved ? 1 : 0)
184
- : null
185
-
186
- const { provider, model } = splitModelRef(cell.resolvedModel)
187
- const supStatus = result !== null && typeof result.sup_status === 'string' ? result.sup_status : null
188
- const supervisorId = randomUUID()
189
-
190
- push(ctx, lines, {
191
- ...baseLine(ctx),
192
- rollout_id: supervisorId,
193
- parent_rollout_id: null,
194
- candidate_id: candidateId(entry.state.generation, entry.state.candidateIndex),
195
- generation: entry.state.generation,
196
- candidate_index: entry.state.candidateIndex,
197
- role: 'supervisor',
198
- task: {
199
- suite: 'swe-bench-verified',
200
- instance_id: cell.scenarioId,
201
- split: 'search',
202
- seed: typeof cell.seed === 'number' ? cell.seed : null,
203
- rep: cell.rep,
204
- },
205
- policy: {
206
- harness: 'pi-loops',
207
- harness_version: null,
208
- model,
209
- provider,
210
- profile_commit:
211
- artifact !== null && typeof artifact.commit === 'string' ? artifact.commit : entry.state.commit,
212
- sampling: null,
213
- },
214
- messages: [],
215
- tool_defs: [],
216
- outcome: {
217
- reward,
218
- reward_source: reward === null ? null : OFFICIAL_JUDGE,
219
- verdict: judgeVerdict,
220
- realness_gated: false,
221
- metrics: {
222
- resolved: artifact?.resolved ?? null,
223
- verify_pass: artifact?.verifyPass ?? null,
224
- patch_lines: artifact?.patchLines ?? null,
225
- judge_attempts: artifact?.judgeAttempts ?? null,
226
- judge_wall_s: artifact?.judgeWallS ?? null,
227
- // Supervisor-tree spend (state.json spend tree); the durable episode
228
- // receipt usd is on cost.usd. Worker-session tokens live on the
229
- // worker lines — recovered_tokens here is the EPISODE total, so do
230
- // not sum tokens across roles without dedup.
231
- spent_tokens: artifact?.spentTokens ?? null,
232
- spent_usd: artifact?.spentUsd ?? null,
233
- recovered_tokens: artifact?.recoveredTokens ?? null,
234
- sup_status: supStatus,
235
- sup_verdict: result !== null && typeof result.sup_verdict === 'string' ? result.sup_verdict : null,
236
- spawned: result !== null && typeof result.spawned === 'number' ? result.spawned : null,
237
- workers: result !== null && typeof result.workers === 'number' ? result.workers : null,
238
- settled: result !== null && typeof result.settled === 'number' ? result.settled : null,
239
- subtasks: result !== null && Array.isArray(result.subtasks) ? result.subtasks : null,
240
- cached: cell.cached ?? null,
241
- },
242
- is_completed: artifact !== null && cell.error === undefined,
243
- is_truncated: supStatus === 'running',
244
- error: cell.error ?? null,
245
- },
246
- cost: {
247
- // The campaign CostLedger receipt for the whole episode.
248
- usd: typeof cell.costUsd === 'number' ? cell.costUsd : null,
249
- // Brain-side token split is not recorded (brain.jsonl is stats-only);
250
- // totals live in metrics.spent_tokens / recovered_tokens.
251
- tokens_in: null,
252
- tokens_out: null,
253
- tokens_reasoning: null,
254
- cache_read: null,
255
- cache_write: null,
256
- wall_s: artifact !== null && typeof artifact.wallS === 'number' ? artifact.wallS : null,
257
- },
258
- artifacts: {
259
- patch_path: artifact !== null && typeof artifact.patchPath === 'string' ? artifact.patchPath : null,
260
- run_dir: runDir,
261
- transcript_ref: runDir !== null ? join(runDir, 'brain.jsonl') : null,
262
- },
263
- provenance: {
264
- captured_at: ctx.capturedAt,
265
- capture: 'backfill',
266
- gap: 'supervisor brain transcript not persisted (brain.jsonl carries per-request stats only)',
267
- },
268
- })
269
-
270
- // Worker sessions: result.json recoveredSpend.workerCwds → opencode store.
271
- const recovered = result !== null && isRecord(result.recoveredSpend) ? result.recoveredSpend : null
272
- // A respawned worker reuses its clone cwd, so the recovered list can name the
273
- // same directory twice. Joining it twice would mint two rollout ids over one
274
- // opencode session — duplicate training signal, inflated worker counts.
275
- const workerCwds =
276
- recovered !== null && Array.isArray(recovered.workerCwds)
277
- ? [...new Set(recovered.workerCwds.filter((c): c is string => typeof c === 'string'))]
278
- : []
279
- for (const cwd of workerCwds) {
280
- const sessions = ctx.db === null ? [] : findOpencodeSessionsByDirectory(ctx.db, cwd)
281
- if (sessions.length === 0) {
282
- ctx.stats.workerCwdsMissed += 1
283
- push(ctx, lines, workerLine(ctx, entry, cell, supervisorId, reward, cwd, null, []))
284
- continue
285
- }
286
- for (const session of sessions) {
287
- const messages = ctx.db === null ? [] : readOpencodeSessionMessages(ctx.db, session.id)
288
- // Two distinct failures: no session row for the cwd (join failure) versus
289
- // a session row whose parts are unreadable (store integrity). One counter
290
- // for both would hide store corruption behind a cwd-join metric.
291
- if (messages.length > 0) ctx.stats.workerSessionsJoined += 1
292
- else ctx.stats.workerSessionsEmpty += 1
293
- push(ctx, lines, workerLine(ctx, entry, cell, supervisorId, reward, cwd, session, messages))
294
- }
295
- }
296
- }
297
-
298
- function workerLine(
299
- ctx: Ctx,
300
- entry: RolloutEntry,
301
- cell: CellRecord,
302
- supervisorId: string,
303
- reward: number | null,
304
- cwd: string,
305
- session: OpencodeSessionRow | null,
306
- messages: RolloutLine['messages'],
307
- ): RolloutLine {
308
- const artifact = cell.artifact
309
- return {
310
- ...baseLine(ctx),
311
- rollout_id: randomUUID(),
312
- parent_rollout_id: supervisorId,
313
- candidate_id: candidateId(entry.state.generation, entry.state.candidateIndex),
314
- generation: entry.state.generation,
315
- candidate_index: entry.state.candidateIndex,
316
- role: 'worker',
317
- task: {
318
- suite: 'swe-bench-verified',
319
- instance_id: cell.scenarioId,
320
- split: 'search',
321
- seed: typeof cell.seed === 'number' ? cell.seed : null,
322
- rep: cell.rep,
323
- },
324
- policy: {
325
- harness: 'opencode',
326
- harness_version: null,
327
- model: session?.model?.id ?? null,
328
- provider: session?.model?.providerID ?? null,
329
- profile_commit:
330
- artifact !== null && typeof artifact.commit === 'string' ? artifact.commit : entry.state.commit,
331
- sampling: null,
332
- },
333
- messages,
334
- tool_defs: [],
335
- outcome: {
336
- // Episode-level credit: the worker has no verdict of its own.
337
- reward,
338
- reward_source: reward === null ? null : `${OFFICIAL_JUDGE}/inherited`,
339
- verdict: null,
340
- realness_gated: false,
341
- metrics: {
342
- worker_cwd: cwd,
343
- session_agent: session?.agent ?? null,
344
- session_parent_id: session?.parentId ?? null,
345
- has_session: session !== null,
346
- },
347
- // A session row with no readable parts is a gap line, not a completed
348
- // invocation — training filters that key on is_completed alone would
349
- // otherwise pull empty transcripts.
350
- is_completed: session !== null && messages.length > 0,
351
- is_truncated: false,
352
- error: null,
353
- },
354
- cost: {
355
- usd: session?.costUsd ?? null,
356
- tokens_in: session?.tokensInput ?? null,
357
- tokens_out: session?.tokensOutput ?? null,
358
- tokens_reasoning: session?.tokensReasoning ?? null,
359
- cache_read: session?.tokensCacheRead ?? null,
360
- cache_write: session?.tokensCacheWrite ?? null,
361
- wall_s: session !== null ? Math.round((session.timeUpdated - session.timeCreated) / 1000) : null,
362
- },
363
- artifacts: {
364
- patch_path: null,
365
- run_dir: artifact !== null && typeof artifact.runDir === 'string' ? artifact.runDir : null,
366
- transcript_ref: session !== null ? `opencode:${session.id}` : null,
367
- },
368
- provenance: {
369
- captured_at: ctx.capturedAt,
370
- capture: 'backfill',
371
- ...(messages.length === 0
372
- ? {
373
- gap:
374
- session === null
375
- ? ctx.db === null
376
- ? `opencode store unavailable; worker cwd ${cwd} unrecoverable`
377
- : `no opencode session found for worker cwd ${cwd}`
378
- : `opencode session ${session.id} has no readable message parts`,
379
- }
380
- : {}),
381
- },
382
- }
383
- }
384
-
385
- // ---------------------------------------------------------------------------
386
- // Proposer shots — receipts under proposer-shots/ (flat or per-author dirs),
387
- // joined to Claude transcripts of the shot worktree by time window.
388
- // ---------------------------------------------------------------------------
389
-
390
- interface ShotReceiptFile {
391
- path: string
392
- author: string | null
393
- generation: number
394
- candidateIndex: number
395
- shot: number
396
- }
397
-
398
- async function listShotReceipts(outDir: string): Promise<ShotReceiptFile[]> {
399
- const shotDir = join(outDir, 'proposer-shots')
400
- const found: ShotReceiptFile[] = []
401
- const parse = (name: string, author: string | null, path: string): void => {
402
- const match = /^gen(\d+)-cand(\d+)-shot(\d+)\.json$/.exec(name)
403
- if (match === null) return
404
- found.push({
405
- path,
406
- author,
407
- generation: Number(match[1]),
408
- candidateIndex: Number(match[2]),
409
- shot: Number(match[3]),
410
- })
411
- }
412
- for (const entry of await readdir(shotDir, { withFileTypes: true }).catch(() => [])) {
413
- if (entry.isDirectory()) {
414
- const authorDir = join(shotDir, entry.name)
415
- for (const name of await readdir(authorDir).catch(() => [])) {
416
- parse(name, entry.name, join(authorDir, name))
417
- }
418
- } else {
419
- parse(entry.name, null, join(shotDir, entry.name))
420
- }
421
- }
422
- return found.sort((a, b) => a.path.localeCompare(b.path))
423
- }
424
-
425
- async function emitProposerLines(ctx: Ctx, lines: RolloutLine[], entries: RolloutEntry[]): Promise<void> {
426
- const receipts = await listShotReceipts(ctx.outDir)
427
- const entryFor = new Map<string, RolloutEntry>()
428
- for (const entry of entries) {
429
- entryFor.set(`${entry.state.generation}:${entry.state.candidateIndex}`, entry)
430
- }
431
-
432
- for (const shot of receipts) {
433
- const parsed = await readJson(shot.path)
434
- const receipt = isRecord(parsed) && isRecord(parsed.receipt) ? parsed.receipt : null
435
- const entry = entryFor.get(`${shot.generation}:${shot.candidateIndex}`) ?? null
436
-
437
- // The proposer worktree the arena dispatched this shot into.
438
- const worktreeCwd =
439
- shot.author !== null
440
- ? join(ctx.outDir, 'proposer-wt', `gen${shot.generation}-cand${shot.candidateIndex}-${shot.author}`)
441
- : join(ctx.outDir, 'proposer-wt', `gen${shot.generation}-cand${shot.candidateIndex}`)
442
-
443
- const startedAt = receipt !== null && typeof receipt.startedAt === 'string' ? Date.parse(receipt.startedAt) : NaN
444
- const completedAt =
445
- receipt !== null && typeof receipt.completedAt === 'string' ? Date.parse(receipt.completedAt) : NaN
446
-
447
- let transcript: Awaited<ReturnType<typeof readClaudeTranscript>> | null = null
448
- let transcriptRef: string | null = null
449
- const candidates = await findClaudeTranscripts(worktreeCwd, ctx.claudeProjectsDir)
450
- const windowed: Array<{ path: string; startMs: number }> = []
451
- for (const ref of candidates) {
452
- const t = await readClaudeTranscript(ref.path)
453
- if (t.startedAt === null) continue
454
- const startMs = Date.parse(t.startedAt)
455
- const inWindow =
456
- Number.isNaN(startedAt) || Number.isNaN(completedAt)
457
- ? candidates.length === 1
458
- : startMs >= startedAt - 120_000 && startMs <= completedAt + 120_000
459
- if (inWindow) windowed.push({ path: ref.path, startMs })
460
- }
461
- if (windowed.length > 0) {
462
- windowed.sort(
463
- (a, b) => Math.abs(a.startMs - (startedAt || a.startMs)) - Math.abs(b.startMs - (startedAt || b.startMs)),
464
- )
465
- transcriptRef = windowed[0].path
466
- transcript = await readClaudeTranscript(transcriptRef)
467
- ctx.stats.proposerShotsJoined += 1
468
- } else {
469
- ctx.stats.proposerShotsMissed += 1
470
- }
471
-
472
- // The proposer's reward is its candidate's official resolved fraction —
473
- // fail-closed null when the candidate has no scored entry.
474
- const reward =
475
- entry !== null && ctx.instanceCount > 0 ? entry.outcome.resolvedCount / ctx.instanceCount : null
476
-
477
- push(ctx, lines, {
478
- ...baseLine(ctx),
479
- rollout_id: randomUUID(),
480
- parent_rollout_id: null,
481
- candidate_id: candidateId(shot.generation, shot.candidateIndex),
482
- generation: shot.generation,
483
- candidate_index: shot.candidateIndex,
484
- role: 'proposer',
485
- task: {
486
- suite: 'swe-arena-proposer',
487
- instance_id:
488
- shot.author !== null
489
- ? `${candidateId(shot.generation, shot.candidateIndex)}-${shot.author}`
490
- : candidateId(shot.generation, shot.candidateIndex),
491
- split: 'search',
492
- seed: null,
493
- rep: shot.shot,
494
- },
495
- policy: {
496
- harness: receipt !== null && typeof receipt.harness === 'string' ? receipt.harness : null,
497
- harness_version: null,
498
- model:
499
- transcript?.model ?? (receipt !== null && typeof receipt.model === 'string' ? receipt.model : null),
500
- provider: null,
501
- profile_commit: entry?.state.commit ?? null,
502
- sampling: null,
503
- },
504
- messages: transcript?.messages ?? [],
505
- tool_defs: [],
506
- outcome: {
507
- reward,
508
- reward_source: reward === null ? null : `${OFFICIAL_JUDGE}/candidate-resolved-fraction`,
509
- verdict: null,
510
- realness_gated: false,
511
- metrics: {
512
- resolved_count: entry?.outcome.resolvedCount ?? null,
513
- instance_count: ctx.instanceCount,
514
- exit_code: receipt !== null && typeof receipt.exitCode === 'number' ? receipt.exitCode : null,
515
- timed_out: receipt !== null && typeof receipt.timedOut === 'boolean' ? receipt.timedOut : null,
516
- author: shot.author,
517
- max_shots: receipt !== null && typeof receipt.maxShots === 'number' ? receipt.maxShots : null,
518
- prompt_sha256: receipt !== null && typeof receipt.promptSha256 === 'string' ? receipt.promptSha256 : null,
519
- },
520
- is_completed: receipt !== null && receipt.exitCode === 0,
521
- is_truncated: receipt !== null && receipt.timedOut === true,
522
- error: receipt !== null && typeof receipt.error === 'string' ? receipt.error : null,
523
- },
524
- cost: {
525
- usd: receipt !== null && typeof receipt.costUsd === 'number' ? receipt.costUsd : null,
526
- tokens_in: transcript !== null ? transcript.usage.tokensIn : null,
527
- tokens_out: transcript !== null ? transcript.usage.tokensOut : null,
528
- tokens_reasoning: null,
529
- cache_read: transcript !== null ? transcript.usage.cacheRead : null,
530
- cache_write: transcript !== null ? transcript.usage.cacheWrite : null,
531
- wall_s:
532
- receipt !== null && typeof receipt.durationMs === 'number' ? Math.round(receipt.durationMs / 1000) : null,
533
- },
534
- artifacts: {
535
- patch_path: entry?.action.diffPath ?? null,
536
- run_dir: worktreeCwd,
537
- transcript_ref: transcriptRef ?? shot.path,
538
- },
539
- provenance: {
540
- captured_at: ctx.capturedAt,
541
- capture: 'backfill',
542
- ...(transcript === null
543
- ? { gap: `no claude transcript matched worktree ${worktreeCwd} in the shot's time window` }
544
- : {}),
545
- },
546
- })
547
- }
548
- }
549
-
550
- // ---------------------------------------------------------------------------
551
- // Entry point.
552
- // ---------------------------------------------------------------------------
553
-
554
- export async function backfillSweArena(outDir: string, opts: BackfillOptions = {}): Promise<BackfillResult> {
555
- const manifest = await buildRolloutManifest(outDir)
556
- const provenance = isRecord(manifest.provenance) ? manifest.provenance : null
557
- const runId = provenance !== null && typeof provenance.runId === 'string' ? provenance.runId : outDir
558
-
559
- const db = await openOpencodeDb(opts.opencodeDb ?? DEFAULT_OPENCODE_DB)
560
- const ctx: Ctx = {
561
- outDir,
562
- runId,
563
- instanceCount: manifest.instances.length,
564
- db,
565
- claudeProjectsDir: opts.claudeProjectsDir ?? DEFAULT_CLAUDE_PROJECTS_DIR,
566
- capturedAt: (opts.now?.() ?? new Date()).toISOString(),
567
- stats: {
568
- lines: 0,
569
- byRole: {},
570
- fullTranscripts: 0,
571
- gapLines: 0,
572
- rewardLabeled: 0,
573
- workerSessionsJoined: 0,
574
- workerCwdsMissed: 0,
575
- workerSessionsEmpty: 0,
576
- proposerShotsJoined: 0,
577
- proposerShotsMissed: 0,
578
- opencodeDbAvailable: db !== null,
579
- },
580
- }
581
-
582
- const lines: RolloutLine[] = []
583
- const entries: RolloutEntry[] = manifest.generations.flatMap((g) => g.rollouts)
584
- try {
585
- for (const entry of entries) {
586
- for (const cell of await readCellRecords(entry.state.campaignDir)) {
587
- await emitCellLines(ctx, lines, entry, cell)
588
- }
589
- }
590
- await emitProposerLines(ctx, lines, entries)
591
- } finally {
592
- db?.close()
593
- }
594
-
595
- return { lines, stats: ctx.stats }
596
- }
597
-
598
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
599
-
600
- if (isMain) {
601
- const [outDir, ledgerPath] = process.argv.slice(2)
602
- if (!outDir || !ledgerPath) {
603
- console.error('usage: tsx src/rollout-ledger/backfill-swe-arena.mts <outDir> <out.jsonl>')
604
- process.exit(2)
605
- }
606
- const { lines, stats } = await backfillSweArena(outDir)
607
- await writeRolloutLedger(ledgerPath, lines)
608
- console.log(JSON.stringify(stats, null, 2))
609
- console.log(`rollout ledger → ${ledgerPath} (${lines.length} lines)`)
610
- }