@tangle-network/agent-bench 0.11.3 → 0.13.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,610 +0,0 @@
1
- /**
2
- * swe-arena → rollout backfill — DOMAIN GLUE ONLY. Bench owns the JOIN
3
- * (manifest + cached cells + judge.json + result.json workerCwds + proposer
4
- * shot receipts → harness stores); the `tangle.rollout.v1` schema, line
5
- * validation, serialization, readers, exporters, and release pipeline are
6
- * owned by `@tangle-network/agent-eval/rollout` — bench never writes a
7
- * rollout row through anything but that API.
8
- *
9
- * Pure READ over one round's outDir
10
- * (the same artifact tree `swe-arena/manifest.mts` joins) plus the harness
11
- * stores, emitting `tangle.rollout.v1` lines with capture:"backfill":
12
- *
13
- * - one SUPERVISOR line per campaign cell (instance × rep × candidate):
14
- * reward = the official-judge verdict off cached-result.json, verdict =
15
- * raw judge.json, cost = the durable episode receipt. The loops brain
16
- * persists per-request STATS only (brain.jsonl has no message content),
17
- * so supervisor transcripts are honest gap lines until settle-time
18
- * capture lands.
19
- * - one WORKER line per opencode worker session, joined via result.json
20
- * recoveredSpend.workerCwds → opencode sqlite session.directory, with the
21
- * FULL transcript inlined and reward inherited from the episode.
22
- * - one PROPOSER line per proposer shot receipt, joined to the Claude
23
- * project transcript of the shot's worktree cwd by time window.
24
- *
25
- * A failed join is a labeled gap line (messages: [], provenance.gap) — never
26
- * a silent drop: the ledger must account for every invocation it knows of.
27
- *
28
- * tsx src/rollout-ledger/backfill-swe-arena.mts <outDir> <out.jsonl>
29
- */
30
-
31
- import { randomUUID } from 'node:crypto'
32
- import { readdir, readFile } from 'node:fs/promises'
33
- import { join } from 'node:path'
34
- import { pathToFileURL } from 'node:url'
35
- import type { DatabaseSync } from 'node:sqlite'
36
- import { loadCampaignCellRecords } from '../swe-arena/cell-evidence.mts'
37
- import { buildRolloutManifest, type RolloutEntry } from '../swe-arena/manifest.mts'
38
- import {
39
- DEFAULT_CLAUDE_PROJECTS_DIR,
40
- DEFAULT_OPENCODE_DB,
41
- findClaudeTranscripts,
42
- findOpencodeSessionsByDirectory,
43
- openOpencodeDb,
44
- readClaudeTranscript,
45
- readOpencodeSessionMessages,
46
- ROLLOUT_SCHEMA,
47
- type OpencodeSessionRow,
48
- type RolloutLine,
49
- writeRolloutLedger,
50
- } from '@tangle-network/agent-eval/rollout'
51
- import { candidateId } from './settle-capture.mts'
52
-
53
- const OFFICIAL_JUDGE = 'swe-arena-official-judge'
54
-
55
- const isRecord = (v: unknown): v is Record<string, unknown> =>
56
- typeof v === 'object' && v !== null && !Array.isArray(v)
57
-
58
- /** Full cached-result.json record (superset of the EvidenceCell slice). */
59
- interface CellRecord {
60
- scenarioId: string
61
- rep: number
62
- artifact: Record<string, unknown> | null
63
- error?: string
64
- judgeScores?: Record<string, { composite?: number }>
65
- costUsd?: number
66
- tokenUsage?: { input?: number; output?: number }
67
- resolvedModel?: string
68
- seed?: number
69
- cached?: boolean
70
- }
71
-
72
- /** The same cell caches `loadCampaignCells` reads, kept verbatim: the backfill
73
- * needs fields (judgeScores, resolvedModel, seed) the EvidenceCell slice drops.
74
- * Traversal and the fail-loud identity check are owned by cell-evidence — one
75
- * rule for what counts as a campaign cell.
76
- *
77
- * Only scenarioId and rep are proven by that check; `artifact` is normalized to
78
- * a record-or-null here so the `artifact !== null` guards downstream cannot be
79
- * handed an `undefined` that passes them. Every remaining field is optional and
80
- * read behind its own typeof guard. */
81
- async function readCellRecords(campaignDir: string): Promise<CellRecord[]> {
82
- const records = (await loadCampaignCellRecords(campaignDir)).map((record) => ({
83
- ...record,
84
- artifact: isRecord(record.artifact) ? record.artifact : null,
85
- })) as unknown as CellRecord[]
86
- return records.sort((a, b) => a.scenarioId.localeCompare(b.scenarioId) || a.rep - b.rep)
87
- }
88
-
89
- async function readJson(path: string): Promise<unknown | null> {
90
- const raw = await readFile(path, 'utf8').catch(() => null)
91
- if (raw === null) return null
92
- try {
93
- return JSON.parse(raw) as unknown
94
- } catch {
95
- return null
96
- }
97
- }
98
-
99
- function splitModelRef(ref: string | undefined): { provider: string | null; model: string | null } {
100
- if (typeof ref !== 'string' || ref.length === 0) return { provider: null, model: null }
101
- const slash = ref.indexOf('/')
102
- if (slash === -1) return { provider: null, model: ref }
103
- return { provider: ref.slice(0, slash), model: ref.slice(slash + 1) }
104
- }
105
-
106
- export interface BackfillStats {
107
- lines: number
108
- byRole: Record<string, number>
109
- fullTranscripts: number
110
- gapLines: number
111
- rewardLabeled: number
112
- workerSessionsJoined: number
113
- /** Worker cwds with no opencode session row — a join failure. */
114
- workerCwdsMissed: number
115
- /** Sessions found but carrying no readable message parts — a store failure. */
116
- workerSessionsEmpty: number
117
- proposerShotsJoined: number
118
- proposerShotsMissed: number
119
- opencodeDbAvailable: boolean
120
- }
121
-
122
- export interface BackfillResult {
123
- lines: RolloutLine[]
124
- stats: BackfillStats
125
- }
126
-
127
- export interface BackfillOptions {
128
- opencodeDb?: string
129
- claudeProjectsDir?: string
130
- /** Injected clock for deterministic tests. */
131
- now?: () => Date
132
- }
133
-
134
- interface Ctx {
135
- outDir: string
136
- runId: string
137
- instanceCount: number
138
- db: DatabaseSync | null
139
- claudeProjectsDir: string
140
- capturedAt: string
141
- stats: BackfillStats
142
- }
143
-
144
- function baseLine(ctx: Ctx): Pick<RolloutLine, 'schema' | 'run_id' | 'experiment_id' | 'provenance'> {
145
- return {
146
- schema: ROLLOUT_SCHEMA,
147
- run_id: ctx.runId,
148
- experiment_id: null,
149
- provenance: { captured_at: ctx.capturedAt, capture: 'backfill' },
150
- }
151
- }
152
-
153
- function push(ctx: Ctx, lines: RolloutLine[], line: RolloutLine): void {
154
- lines.push(line)
155
- ctx.stats.lines += 1
156
- ctx.stats.byRole[line.role] = (ctx.stats.byRole[line.role] ?? 0) + 1
157
- if (line.messages.length > 0) ctx.stats.fullTranscripts += 1
158
- else ctx.stats.gapLines += 1
159
- if (line.outcome.reward !== null) ctx.stats.rewardLabeled += 1
160
- }
161
-
162
- // ---------------------------------------------------------------------------
163
- // Supervisor episode + worker sessions for one campaign cell.
164
- // ---------------------------------------------------------------------------
165
-
166
- async function emitCellLines(
167
- ctx: Ctx,
168
- lines: RolloutLine[],
169
- entry: RolloutEntry,
170
- cell: CellRecord,
171
- ): Promise<void> {
172
- const artifact = cell.artifact
173
- const runDir = artifact !== null && typeof artifact.runDir === 'string' ? artifact.runDir : null
174
- const judgeVerdict = runDir !== null ? await readJson(join(runDir, 'judge.json')) : null
175
- const resultJson = runDir !== null ? await readJson(join(runDir, 'result.json')) : null
176
- const result = isRecord(resultJson) ? resultJson : null
177
-
178
- const composite = cell.judgeScores?.[OFFICIAL_JUDGE]?.composite
179
- const reward =
180
- typeof composite === 'number'
181
- ? composite
182
- : artifact !== null && typeof artifact.resolved === 'boolean'
183
- ? (artifact.resolved ? 1 : 0)
184
- : null
185
-
186
- const { provider, model } = splitModelRef(cell.resolvedModel)
187
- const supStatus = result !== null && typeof result.sup_status === 'string' ? result.sup_status : null
188
- const supervisorId = randomUUID()
189
-
190
- push(ctx, lines, {
191
- ...baseLine(ctx),
192
- rollout_id: supervisorId,
193
- parent_rollout_id: null,
194
- candidate_id: candidateId(entry.state.generation, entry.state.candidateIndex),
195
- generation: entry.state.generation,
196
- candidate_index: entry.state.candidateIndex,
197
- role: 'supervisor',
198
- task: {
199
- suite: 'swe-bench-verified',
200
- instance_id: cell.scenarioId,
201
- split: 'search',
202
- seed: typeof cell.seed === 'number' ? cell.seed : null,
203
- rep: cell.rep,
204
- },
205
- policy: {
206
- harness: 'pi-loops',
207
- harness_version: null,
208
- model,
209
- provider,
210
- profile_commit:
211
- artifact !== null && typeof artifact.commit === 'string' ? artifact.commit : entry.state.commit,
212
- sampling: null,
213
- },
214
- messages: [],
215
- tool_defs: [],
216
- outcome: {
217
- reward,
218
- reward_source: reward === null ? null : OFFICIAL_JUDGE,
219
- verdict: judgeVerdict,
220
- realness_gated: false,
221
- metrics: {
222
- resolved: artifact?.resolved ?? null,
223
- verify_pass: artifact?.verifyPass ?? null,
224
- patch_lines: artifact?.patchLines ?? null,
225
- judge_attempts: artifact?.judgeAttempts ?? null,
226
- judge_wall_s: artifact?.judgeWallS ?? null,
227
- // Supervisor-tree spend (state.json spend tree); the durable episode
228
- // receipt usd is on cost.usd. Worker-session tokens live on the
229
- // worker lines — recovered_tokens here is the EPISODE total, so do
230
- // not sum tokens across roles without dedup.
231
- spent_tokens: artifact?.spentTokens ?? null,
232
- spent_usd: artifact?.spentUsd ?? null,
233
- recovered_tokens: artifact?.recoveredTokens ?? null,
234
- sup_status: supStatus,
235
- sup_verdict: result !== null && typeof result.sup_verdict === 'string' ? result.sup_verdict : null,
236
- spawned: result !== null && typeof result.spawned === 'number' ? result.spawned : null,
237
- workers: result !== null && typeof result.workers === 'number' ? result.workers : null,
238
- settled: result !== null && typeof result.settled === 'number' ? result.settled : null,
239
- subtasks: result !== null && Array.isArray(result.subtasks) ? result.subtasks : null,
240
- cached: cell.cached ?? null,
241
- },
242
- is_completed: artifact !== null && cell.error === undefined,
243
- is_truncated: supStatus === 'running',
244
- error: cell.error ?? null,
245
- },
246
- cost: {
247
- // The campaign CostLedger receipt for the whole episode.
248
- usd: typeof cell.costUsd === 'number' ? cell.costUsd : null,
249
- // Brain-side token split is not recorded (brain.jsonl is stats-only);
250
- // totals live in metrics.spent_tokens / recovered_tokens.
251
- tokens_in: null,
252
- tokens_out: null,
253
- tokens_reasoning: null,
254
- cache_read: null,
255
- cache_write: null,
256
- wall_s: artifact !== null && typeof artifact.wallS === 'number' ? artifact.wallS : null,
257
- },
258
- artifacts: {
259
- patch_path: artifact !== null && typeof artifact.patchPath === 'string' ? artifact.patchPath : null,
260
- run_dir: runDir,
261
- transcript_ref: runDir !== null ? join(runDir, 'brain.jsonl') : null,
262
- },
263
- provenance: {
264
- captured_at: ctx.capturedAt,
265
- capture: 'backfill',
266
- gap: 'supervisor brain transcript not persisted (brain.jsonl carries per-request stats only)',
267
- },
268
- })
269
-
270
- // Worker sessions: result.json recoveredSpend.workerCwds → opencode store.
271
- const recovered = result !== null && isRecord(result.recoveredSpend) ? result.recoveredSpend : null
272
- // A respawned worker reuses its clone cwd, so the recovered list can name the
273
- // same directory twice. Joining it twice would mint two rollout ids over one
274
- // opencode session — duplicate training signal, inflated worker counts.
275
- const workerCwds =
276
- recovered !== null && Array.isArray(recovered.workerCwds)
277
- ? [...new Set(recovered.workerCwds.filter((c): c is string => typeof c === 'string'))]
278
- : []
279
- for (const cwd of workerCwds) {
280
- const sessions = ctx.db === null ? [] : findOpencodeSessionsByDirectory(ctx.db, cwd)
281
- if (sessions.length === 0) {
282
- ctx.stats.workerCwdsMissed += 1
283
- push(ctx, lines, workerLine(ctx, entry, cell, supervisorId, reward, cwd, null, []))
284
- continue
285
- }
286
- for (const session of sessions) {
287
- const messages = ctx.db === null ? [] : readOpencodeSessionMessages(ctx.db, session.id)
288
- // Two distinct failures: no session row for the cwd (join failure) versus
289
- // a session row whose parts are unreadable (store integrity). One counter
290
- // for both would hide store corruption behind a cwd-join metric.
291
- if (messages.length > 0) ctx.stats.workerSessionsJoined += 1
292
- else ctx.stats.workerSessionsEmpty += 1
293
- push(ctx, lines, workerLine(ctx, entry, cell, supervisorId, reward, cwd, session, messages))
294
- }
295
- }
296
- }
297
-
298
- function workerLine(
299
- ctx: Ctx,
300
- entry: RolloutEntry,
301
- cell: CellRecord,
302
- supervisorId: string,
303
- reward: number | null,
304
- cwd: string,
305
- session: OpencodeSessionRow | null,
306
- messages: RolloutLine['messages'],
307
- ): RolloutLine {
308
- const artifact = cell.artifact
309
- return {
310
- ...baseLine(ctx),
311
- rollout_id: randomUUID(),
312
- parent_rollout_id: supervisorId,
313
- candidate_id: candidateId(entry.state.generation, entry.state.candidateIndex),
314
- generation: entry.state.generation,
315
- candidate_index: entry.state.candidateIndex,
316
- role: 'worker',
317
- task: {
318
- suite: 'swe-bench-verified',
319
- instance_id: cell.scenarioId,
320
- split: 'search',
321
- seed: typeof cell.seed === 'number' ? cell.seed : null,
322
- rep: cell.rep,
323
- },
324
- policy: {
325
- harness: 'opencode',
326
- harness_version: null,
327
- model: session?.model?.id ?? null,
328
- provider: session?.model?.providerID ?? null,
329
- profile_commit:
330
- artifact !== null && typeof artifact.commit === 'string' ? artifact.commit : entry.state.commit,
331
- sampling: null,
332
- },
333
- messages,
334
- tool_defs: [],
335
- outcome: {
336
- // Episode-level credit: the worker has no verdict of its own.
337
- reward,
338
- reward_source: reward === null ? null : `${OFFICIAL_JUDGE}/inherited`,
339
- verdict: null,
340
- realness_gated: false,
341
- metrics: {
342
- worker_cwd: cwd,
343
- session_agent: session?.agent ?? null,
344
- session_parent_id: session?.parentId ?? null,
345
- has_session: session !== null,
346
- },
347
- // A session row with no readable parts is a gap line, not a completed
348
- // invocation — training filters that key on is_completed alone would
349
- // otherwise pull empty transcripts.
350
- is_completed: session !== null && messages.length > 0,
351
- is_truncated: false,
352
- error: null,
353
- },
354
- cost: {
355
- usd: session?.costUsd ?? null,
356
- tokens_in: session?.tokensInput ?? null,
357
- tokens_out: session?.tokensOutput ?? null,
358
- tokens_reasoning: session?.tokensReasoning ?? null,
359
- cache_read: session?.tokensCacheRead ?? null,
360
- cache_write: session?.tokensCacheWrite ?? null,
361
- wall_s: session !== null ? Math.round((session.timeUpdated - session.timeCreated) / 1000) : null,
362
- },
363
- artifacts: {
364
- patch_path: null,
365
- run_dir: artifact !== null && typeof artifact.runDir === 'string' ? artifact.runDir : null,
366
- transcript_ref: session !== null ? `opencode:${session.id}` : null,
367
- },
368
- provenance: {
369
- captured_at: ctx.capturedAt,
370
- capture: 'backfill',
371
- ...(messages.length === 0
372
- ? {
373
- gap:
374
- session === null
375
- ? ctx.db === null
376
- ? `opencode store unavailable; worker cwd ${cwd} unrecoverable`
377
- : `no opencode session found for worker cwd ${cwd}`
378
- : `opencode session ${session.id} has no readable message parts`,
379
- }
380
- : {}),
381
- },
382
- }
383
- }
384
-
385
- // ---------------------------------------------------------------------------
386
- // Proposer shots — receipts under proposer-shots/ (flat or per-author dirs),
387
- // joined to Claude transcripts of the shot worktree by time window.
388
- // ---------------------------------------------------------------------------
389
-
390
- interface ShotReceiptFile {
391
- path: string
392
- author: string | null
393
- generation: number
394
- candidateIndex: number
395
- shot: number
396
- }
397
-
398
- async function listShotReceipts(outDir: string): Promise<ShotReceiptFile[]> {
399
- const shotDir = join(outDir, 'proposer-shots')
400
- const found: ShotReceiptFile[] = []
401
- const parse = (name: string, author: string | null, path: string): void => {
402
- const match = /^gen(\d+)-cand(\d+)-shot(\d+)\.json$/.exec(name)
403
- if (match === null) return
404
- found.push({
405
- path,
406
- author,
407
- generation: Number(match[1]),
408
- candidateIndex: Number(match[2]),
409
- shot: Number(match[3]),
410
- })
411
- }
412
- for (const entry of await readdir(shotDir, { withFileTypes: true }).catch(() => [])) {
413
- if (entry.isDirectory()) {
414
- const authorDir = join(shotDir, entry.name)
415
- for (const name of await readdir(authorDir).catch(() => [])) {
416
- parse(name, entry.name, join(authorDir, name))
417
- }
418
- } else {
419
- parse(entry.name, null, join(shotDir, entry.name))
420
- }
421
- }
422
- return found.sort((a, b) => a.path.localeCompare(b.path))
423
- }
424
-
425
- async function emitProposerLines(ctx: Ctx, lines: RolloutLine[], entries: RolloutEntry[]): Promise<void> {
426
- const receipts = await listShotReceipts(ctx.outDir)
427
- const entryFor = new Map<string, RolloutEntry>()
428
- for (const entry of entries) {
429
- entryFor.set(`${entry.state.generation}:${entry.state.candidateIndex}`, entry)
430
- }
431
-
432
- for (const shot of receipts) {
433
- const parsed = await readJson(shot.path)
434
- const receipt = isRecord(parsed) && isRecord(parsed.receipt) ? parsed.receipt : null
435
- const entry = entryFor.get(`${shot.generation}:${shot.candidateIndex}`) ?? null
436
-
437
- // The proposer worktree the arena dispatched this shot into.
438
- const worktreeCwd =
439
- shot.author !== null
440
- ? join(ctx.outDir, 'proposer-wt', `gen${shot.generation}-cand${shot.candidateIndex}-${shot.author}`)
441
- : join(ctx.outDir, 'proposer-wt', `gen${shot.generation}-cand${shot.candidateIndex}`)
442
-
443
- const startedAt = receipt !== null && typeof receipt.startedAt === 'string' ? Date.parse(receipt.startedAt) : NaN
444
- const completedAt =
445
- receipt !== null && typeof receipt.completedAt === 'string' ? Date.parse(receipt.completedAt) : NaN
446
-
447
- let transcript: Awaited<ReturnType<typeof readClaudeTranscript>> | null = null
448
- let transcriptRef: string | null = null
449
- const candidates = await findClaudeTranscripts(worktreeCwd, ctx.claudeProjectsDir)
450
- const windowed: Array<{ path: string; startMs: number }> = []
451
- for (const ref of candidates) {
452
- const t = await readClaudeTranscript(ref.path)
453
- if (t.startedAt === null) continue
454
- const startMs = Date.parse(t.startedAt)
455
- const inWindow =
456
- Number.isNaN(startedAt) || Number.isNaN(completedAt)
457
- ? candidates.length === 1
458
- : startMs >= startedAt - 120_000 && startMs <= completedAt + 120_000
459
- if (inWindow) windowed.push({ path: ref.path, startMs })
460
- }
461
- if (windowed.length > 0) {
462
- windowed.sort(
463
- (a, b) => Math.abs(a.startMs - (startedAt || a.startMs)) - Math.abs(b.startMs - (startedAt || b.startMs)),
464
- )
465
- transcriptRef = windowed[0].path
466
- transcript = await readClaudeTranscript(transcriptRef)
467
- ctx.stats.proposerShotsJoined += 1
468
- } else {
469
- ctx.stats.proposerShotsMissed += 1
470
- }
471
-
472
- // The proposer's reward is its candidate's official resolved fraction —
473
- // fail-closed null when the candidate has no scored entry.
474
- const reward =
475
- entry !== null && ctx.instanceCount > 0 ? entry.outcome.resolvedCount / ctx.instanceCount : null
476
-
477
- push(ctx, lines, {
478
- ...baseLine(ctx),
479
- rollout_id: randomUUID(),
480
- parent_rollout_id: null,
481
- candidate_id: candidateId(shot.generation, shot.candidateIndex),
482
- generation: shot.generation,
483
- candidate_index: shot.candidateIndex,
484
- role: 'proposer',
485
- task: {
486
- suite: 'swe-arena-proposer',
487
- instance_id:
488
- shot.author !== null
489
- ? `${candidateId(shot.generation, shot.candidateIndex)}-${shot.author}`
490
- : candidateId(shot.generation, shot.candidateIndex),
491
- split: 'search',
492
- seed: null,
493
- rep: shot.shot,
494
- },
495
- policy: {
496
- harness: receipt !== null && typeof receipt.harness === 'string' ? receipt.harness : null,
497
- harness_version: null,
498
- model:
499
- transcript?.model ?? (receipt !== null && typeof receipt.model === 'string' ? receipt.model : null),
500
- provider: null,
501
- profile_commit: entry?.state.commit ?? null,
502
- sampling: null,
503
- },
504
- messages: transcript?.messages ?? [],
505
- tool_defs: [],
506
- outcome: {
507
- reward,
508
- reward_source: reward === null ? null : `${OFFICIAL_JUDGE}/candidate-resolved-fraction`,
509
- verdict: null,
510
- realness_gated: false,
511
- metrics: {
512
- resolved_count: entry?.outcome.resolvedCount ?? null,
513
- instance_count: ctx.instanceCount,
514
- exit_code: receipt !== null && typeof receipt.exitCode === 'number' ? receipt.exitCode : null,
515
- timed_out: receipt !== null && typeof receipt.timedOut === 'boolean' ? receipt.timedOut : null,
516
- author: shot.author,
517
- max_shots: receipt !== null && typeof receipt.maxShots === 'number' ? receipt.maxShots : null,
518
- prompt_sha256: receipt !== null && typeof receipt.promptSha256 === 'string' ? receipt.promptSha256 : null,
519
- },
520
- is_completed: receipt !== null && receipt.exitCode === 0,
521
- is_truncated: receipt !== null && receipt.timedOut === true,
522
- error: receipt !== null && typeof receipt.error === 'string' ? receipt.error : null,
523
- },
524
- cost: {
525
- usd: receipt !== null && typeof receipt.costUsd === 'number' ? receipt.costUsd : null,
526
- tokens_in: transcript !== null ? transcript.usage.tokensIn : null,
527
- tokens_out: transcript !== null ? transcript.usage.tokensOut : null,
528
- tokens_reasoning: null,
529
- cache_read: transcript !== null ? transcript.usage.cacheRead : null,
530
- cache_write: transcript !== null ? transcript.usage.cacheWrite : null,
531
- wall_s:
532
- receipt !== null && typeof receipt.durationMs === 'number' ? Math.round(receipt.durationMs / 1000) : null,
533
- },
534
- artifacts: {
535
- patch_path: entry?.action.diffPath ?? null,
536
- run_dir: worktreeCwd,
537
- transcript_ref: transcriptRef ?? shot.path,
538
- },
539
- provenance: {
540
- captured_at: ctx.capturedAt,
541
- capture: 'backfill',
542
- ...(transcript === null
543
- ? { gap: `no claude transcript matched worktree ${worktreeCwd} in the shot's time window` }
544
- : {}),
545
- },
546
- })
547
- }
548
- }
549
-
550
- // ---------------------------------------------------------------------------
551
- // Entry point.
552
- // ---------------------------------------------------------------------------
553
-
554
- export async function backfillSweArena(outDir: string, opts: BackfillOptions = {}): Promise<BackfillResult> {
555
- const manifest = await buildRolloutManifest(outDir)
556
- const provenance = isRecord(manifest.provenance) ? manifest.provenance : null
557
- const runId = provenance !== null && typeof provenance.runId === 'string' ? provenance.runId : outDir
558
-
559
- const db = await openOpencodeDb(opts.opencodeDb ?? DEFAULT_OPENCODE_DB)
560
- const ctx: Ctx = {
561
- outDir,
562
- runId,
563
- instanceCount: manifest.instances.length,
564
- db,
565
- claudeProjectsDir: opts.claudeProjectsDir ?? DEFAULT_CLAUDE_PROJECTS_DIR,
566
- capturedAt: (opts.now?.() ?? new Date()).toISOString(),
567
- stats: {
568
- lines: 0,
569
- byRole: {},
570
- fullTranscripts: 0,
571
- gapLines: 0,
572
- rewardLabeled: 0,
573
- workerSessionsJoined: 0,
574
- workerCwdsMissed: 0,
575
- workerSessionsEmpty: 0,
576
- proposerShotsJoined: 0,
577
- proposerShotsMissed: 0,
578
- opencodeDbAvailable: db !== null,
579
- },
580
- }
581
-
582
- const lines: RolloutLine[] = []
583
- const entries: RolloutEntry[] = manifest.generations.flatMap((g) => g.rollouts)
584
- try {
585
- for (const entry of entries) {
586
- for (const cell of await readCellRecords(entry.state.campaignDir)) {
587
- await emitCellLines(ctx, lines, entry, cell)
588
- }
589
- }
590
- await emitProposerLines(ctx, lines, entries)
591
- } finally {
592
- db?.close()
593
- }
594
-
595
- return { lines, stats: ctx.stats }
596
- }
597
-
598
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
599
-
600
- if (isMain) {
601
- const [outDir, ledgerPath] = process.argv.slice(2)
602
- if (!outDir || !ledgerPath) {
603
- console.error('usage: tsx src/rollout-ledger/backfill-swe-arena.mts <outDir> <out.jsonl>')
604
- process.exit(2)
605
- }
606
- const { lines, stats } = await backfillSweArena(outDir)
607
- await writeRolloutLedger(ledgerPath, lines)
608
- console.log(JSON.stringify(stats, null, 2))
609
- console.log(`rollout ledger → ${ledgerPath} (${lines.length} lines)`)
610
- }