@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,448 +0,0 @@
1
- /**
2
- * Settle-time rollout capture with LABEL v2 — `tangle.rollout.v1` lines
3
- * emitted LIVE, right after each campaign cell is judged, while the harness
4
- * stores still hold the transcripts (the whole reason settle-time capture
5
- * beats backfill: the opencode sqlite db and claude project transcripts are
6
- * mutable/garbage-collected).
7
- *
8
- * LABEL v2 semantics (reward_source tags distinguish them from the v1
9
- * backfill's inherited labels — both coexist in one ledger):
10
- *
11
- * - WORKERS get CONTRIBUTION-AWARE reward: reward=1 only when the cell
12
- * resolved AND that worker's patch was the delivered one (delivered-patch
13
- * identity from the supervisor run dir's per-worker evidence,
14
- * workers/<label>.patch); sibling bystanders in a resolved cell get
15
- * reward=0 with metrics.bystander=true. In an unresolved cell every worker
16
- * gets reward=0. A resolved cell whose delivered patch matches NO worker
17
- * patch is a labeled identity gap: reward=null,
18
- * metrics.delivered_match='unknown' — never a fabricated credit.
19
- * - The SUPERVISOR line keeps the official-judge verdict as its reward
20
- * (unchanged semantics; brain transcripts remain honest gap lines until
21
- * the loops brain persists message content — journal + stats are joined
22
- * into metrics).
23
- * - PROPOSERS get BASELINE-RELATIVE reward: candidate_score −
24
- * baseline_score (resolved fractions), so an improvement is positive and
25
- * a regression negative — emitted once per candidate when the round's
26
- * scores are final.
27
- */
28
-
29
- import { randomUUID } from 'node:crypto'
30
- import { existsSync } from 'node:fs'
31
- import { readFile, readdir } from 'node:fs/promises'
32
- import { join } from 'node:path'
33
- import type { DatabaseSync } from 'node:sqlite'
34
- import {
35
- appendRolloutLines,
36
- DEFAULT_OPENCODE_DB,
37
- findOpencodeSessionsByDirectory,
38
- openOpencodeDb,
39
- readOpencodeSessionMessages,
40
- ROLLOUT_SCHEMA,
41
- type ChatMessage,
42
- type OpencodeSessionRow,
43
- type RolloutLine,
44
- } from '@tangle-network/agent-eval/rollout'
45
-
46
- export const OFFICIAL_JUDGE = 'swe-arena-official-judge'
47
- export const WORKER_REWARD_SOURCE_V2 = `${OFFICIAL_JUDGE}/contribution-v2`
48
- export const PROPOSER_REWARD_SOURCE_V2 = `${OFFICIAL_JUDGE}/baseline-relative-v2`
49
-
50
- /** Stable candidate identity from the improvement-loop coordinates. */
51
- export function candidateId(generation: number, candidateIndex: number): string {
52
- return candidateIndex === -1 ? 'baseline' : `gen${generation}-cand${candidateIndex}`
53
- }
54
-
55
- // ---------------------------------------------------------------------------
56
- // Worker evidence — the supervisor run dir's per-worker record.
57
- // ---------------------------------------------------------------------------
58
-
59
- export interface WorkerEvidenceRecord {
60
- label: string
61
- /** Worker clone cwd from the ndjson `started` event (opencode join key). */
62
- cwd: string | null
63
- /** Patch persisted at settle time (workers/<label>.patch); null = none. */
64
- patch: string | null
65
- }
66
-
67
- /** Read workers/<label>.{ndjson,patch} under one supervisor run dir. */
68
- export async function readWorkerEvidence(supRunDir: string): Promise<WorkerEvidenceRecord[]> {
69
- const workersDir = join(supRunDir, 'workers')
70
- const names = await readdir(workersDir).catch(() => [])
71
- const labels = new Set<string>()
72
- for (const name of names) {
73
- if (name.endsWith('.ndjson') && !name.endsWith('.inbox.ndjson')) labels.add(name.slice(0, -'.ndjson'.length))
74
- else if (name.endsWith('.patch')) labels.add(name.slice(0, -'.patch'.length))
75
- }
76
- const records: WorkerEvidenceRecord[] = []
77
- for (const label of [...labels].sort()) {
78
- let cwd: string | null = null
79
- const nd = await readFile(join(workersDir, `${label}.ndjson`), 'utf8').catch(() => '')
80
- for (const line of nd.split('\n')) {
81
- if (!line.trim()) continue
82
- try {
83
- const o = JSON.parse(line) as Record<string, unknown>
84
- if (o.kind === 'started' && typeof o.cwd === 'string') cwd = o.cwd
85
- } catch {
86
- // A corrupt event line cannot invalidate the rest of the evidence.
87
- }
88
- }
89
- const patch = await readFile(join(workersDir, `${label}.patch`), 'utf8').catch(() => null)
90
- records.push({ label, cwd, patch: patch !== null && patch.trim().length > 0 ? patch : null })
91
- }
92
- return records
93
- }
94
-
95
- /** Labels whose persisted patch IS the delivered patch (trimmed equality —
96
- * the supervisor delivers a settled worker's patch verbatim). */
97
- export function deliveredWorkerLabels(deliveredPatch: string, workers: readonly WorkerEvidenceRecord[]): string[] {
98
- const delivered = deliveredPatch.trim()
99
- if (delivered.length === 0) return []
100
- return workers.filter((w) => w.patch !== null && w.patch.trim() === delivered).map((w) => w.label)
101
- }
102
-
103
- export interface WorkerRewardV2 {
104
- reward: number | null
105
- bystander: boolean
106
- /** 'delivered' | 'bystander' | 'unresolved' | 'unknown' (identity gap). */
107
- deliveredMatch: 'delivered' | 'bystander' | 'unresolved' | 'unknown'
108
- }
109
-
110
- /** The v2 contribution rule for one worker in one cell. */
111
- export function workerRewardV2(input: {
112
- /** Official judge verdict for the cell; null = inconclusive. */
113
- resolved: boolean | null
114
- /** This worker's label is among the delivered-patch matches. */
115
- isDelivered: boolean
116
- /** At least one worker matched the delivered patch (identity known). */
117
- identityKnown: boolean
118
- }): WorkerRewardV2 {
119
- if (input.resolved !== true) {
120
- return { reward: input.resolved === false ? 0 : null, bystander: false, deliveredMatch: 'unresolved' }
121
- }
122
- if (!input.identityKnown) {
123
- return { reward: null, bystander: false, deliveredMatch: 'unknown' }
124
- }
125
- if (input.isDelivered) return { reward: 1, bystander: false, deliveredMatch: 'delivered' }
126
- return { reward: 0, bystander: true, deliveredMatch: 'bystander' }
127
- }
128
-
129
- /** Baseline-relative proposer reward (resolved fractions; improvement > 0). */
130
- export function proposerRewardV2(candResolved: number, baselineResolved: number, instanceCount: number): number {
131
- if (!(instanceCount > 0)) throw new Error('proposerRewardV2: instanceCount must be > 0')
132
- return (candResolved - baselineResolved) / instanceCount
133
- }
134
-
135
- // ---------------------------------------------------------------------------
136
- // The live capture handle.
137
- // ---------------------------------------------------------------------------
138
-
139
- export interface SettleCaptureOptions {
140
- /** Ledger JSONL destination (appended live, one flush per cell). */
141
- ledgerPath: string
142
- runId: string
143
- instanceCount: number
144
- opencodeDb?: string
145
- /** Injected clock for deterministic tests. */
146
- now?: () => Date
147
- log?: (msg: string) => void
148
- }
149
-
150
- export interface CellCaptureArgs {
151
- generation: number
152
- candidateIndex: number
153
- iid: string
154
- rep: number
155
- seed: number | null
156
- /** Our public/private sub-split of the train instances (metrics label). */
157
- splitVisibility: 'public' | 'private' | null
158
- commit: string | null
159
- /** Official judge verdict (null = inconclusive). */
160
- resolved: boolean | null
161
- /** Raw judge verdict record, verbatim. */
162
- judgeVerdict: unknown
163
- runDir: string | null
164
- patchPath: string | null
165
- /** Supervisor run dir (<ws>/.loops/supervisor/<id>); null = unlocatable. */
166
- supRunDir: string | null
167
- /** The delivered patch text (empty = no delivery). */
168
- deliveredPatch: string
169
- workerModel: string | null
170
- metrics: Record<string, unknown>
171
- cost: { usd: number | null; wallS: number | null; spentTokens: number | null }
172
- }
173
-
174
- export interface ProposerCaptureArgs {
175
- generation: number
176
- candidateIndex: number
177
- proposer: string
178
- harness: string | null
179
- commit: string | null
180
- candResolved: number
181
- baselineResolved: number
182
- shotReceiptPaths: string[]
183
- diffPath: string | null
184
- }
185
-
186
- export interface SettleCapture {
187
- readonly path: string
188
- captureCell(args: CellCaptureArgs): Promise<{ lines: number }>
189
- captureProposer(args: ProposerCaptureArgs): Promise<{ lines: number }>
190
- }
191
-
192
- export function createSettleCapture(opts: SettleCaptureOptions): SettleCapture {
193
- const now = opts.now ?? (() => new Date())
194
- const log = opts.log ?? (() => {})
195
- const dbPath = opts.opencodeDb ?? DEFAULT_OPENCODE_DB
196
-
197
- const base = (
198
- capturedAt: string,
199
- ): Pick<RolloutLine, 'schema' | 'run_id' | 'experiment_id'> & { provenance: RolloutLine['provenance'] } => ({
200
- schema: ROLLOUT_SCHEMA,
201
- run_id: opts.runId,
202
- experiment_id: null,
203
- provenance: { captured_at: capturedAt, capture: 'settle-time' },
204
- })
205
-
206
- return {
207
- path: opts.ledgerPath,
208
-
209
- async captureCell(args: CellCaptureArgs): Promise<{ lines: number }> {
210
- const capturedAt = now().toISOString()
211
- const supervisorId = randomUUID()
212
- const reward = args.resolved === null ? null : args.resolved ? 1 : 0
213
- const lines: RolloutLine[] = []
214
-
215
- lines.push({
216
- ...base(capturedAt),
217
- rollout_id: supervisorId,
218
- parent_rollout_id: null,
219
- candidate_id: candidateId(args.generation, args.candidateIndex),
220
- generation: args.generation,
221
- candidate_index: args.candidateIndex,
222
- role: 'supervisor',
223
- task: { suite: 'swe-bench-verified', instance_id: args.iid, split: 'search', seed: args.seed, rep: args.rep },
224
- policy: {
225
- harness: 'pi-loops',
226
- harness_version: null,
227
- model: args.workerModel,
228
- provider: null,
229
- profile_commit: args.commit,
230
- sampling: null,
231
- },
232
- messages: [],
233
- tool_defs: [],
234
- outcome: {
235
- reward,
236
- reward_source: reward === null ? null : OFFICIAL_JUDGE,
237
- verdict: args.judgeVerdict,
238
- realness_gated: false,
239
- metrics: {
240
- ...args.metrics,
241
- ...(args.splitVisibility !== null ? { split_visibility: args.splitVisibility } : {}),
242
- },
243
- is_completed: args.resolved !== null,
244
- is_truncated: false,
245
- error: null,
246
- },
247
- cost: {
248
- usd: args.cost.usd,
249
- tokens_in: null,
250
- tokens_out: null,
251
- tokens_reasoning: null,
252
- cache_read: null,
253
- cache_write: null,
254
- wall_s: args.cost.wallS,
255
- },
256
- artifacts: {
257
- patch_path: args.patchPath,
258
- run_dir: args.runDir,
259
- transcript_ref: args.runDir !== null ? join(args.runDir, 'brain.jsonl') : null,
260
- },
261
- provenance: {
262
- captured_at: capturedAt,
263
- capture: 'settle-time',
264
- gap: 'supervisor brain transcript not persisted (brain.jsonl carries per-request stats only)',
265
- },
266
- })
267
-
268
- // Worker lines with v2 contribution labels.
269
- const workers = args.supRunDir !== null ? await readWorkerEvidence(args.supRunDir) : []
270
- if (workers.length > 0) {
271
- const delivered = new Set(deliveredWorkerLabels(args.deliveredPatch, workers))
272
- const identityKnown = delivered.size > 0
273
- let db: DatabaseSync | null = null
274
- try {
275
- db = await openOpencodeDb(dbPath)
276
- for (const worker of workers) {
277
- const v2 = workerRewardV2({
278
- resolved: args.resolved,
279
- isDelivered: delivered.has(worker.label),
280
- identityKnown,
281
- })
282
- let session: OpencodeSessionRow | null = null
283
- let messages: ChatMessage[] = []
284
- if (db !== null && worker.cwd !== null) {
285
- const sessions = findOpencodeSessionsByDirectory(db, worker.cwd)
286
- session = sessions[0] ?? null
287
- if (session !== null) messages = readOpencodeSessionMessages(db, session.id)
288
- }
289
- lines.push({
290
- ...base(capturedAt),
291
- rollout_id: randomUUID(),
292
- parent_rollout_id: supervisorId,
293
- candidate_id: candidateId(args.generation, args.candidateIndex),
294
- generation: args.generation,
295
- candidate_index: args.candidateIndex,
296
- role: 'worker',
297
- task: {
298
- suite: 'swe-bench-verified',
299
- instance_id: args.iid,
300
- split: 'search',
301
- seed: args.seed,
302
- rep: args.rep,
303
- },
304
- policy: {
305
- harness: 'opencode',
306
- harness_version: null,
307
- model: session?.model?.id ?? null,
308
- provider: session?.model?.providerID ?? null,
309
- profile_commit: args.commit,
310
- sampling: null,
311
- },
312
- messages,
313
- tool_defs: [],
314
- outcome: {
315
- reward: v2.reward,
316
- reward_source: v2.reward === null && v2.deliveredMatch !== 'unknown' ? null : WORKER_REWARD_SOURCE_V2,
317
- verdict: null,
318
- realness_gated: false,
319
- metrics: {
320
- worker_label: worker.label,
321
- worker_cwd: worker.cwd,
322
- bystander: v2.bystander,
323
- delivered_match: v2.deliveredMatch,
324
- has_patch: worker.patch !== null,
325
- has_session: session !== null,
326
- ...(args.splitVisibility !== null ? { split_visibility: args.splitVisibility } : {}),
327
- },
328
- // Matches the backfill rule: a session with no readable parts is
329
- // a gap line, not a completed invocation.
330
- is_completed: session !== null && messages.length > 0,
331
- is_truncated: false,
332
- error: null,
333
- },
334
- cost: {
335
- usd: session?.costUsd ?? null,
336
- tokens_in: session?.tokensInput ?? null,
337
- tokens_out: session?.tokensOutput ?? null,
338
- tokens_reasoning: session?.tokensReasoning ?? null,
339
- cache_read: session?.tokensCacheRead ?? null,
340
- cache_write: session?.tokensCacheWrite ?? null,
341
- wall_s: session !== null ? Math.round((session.timeUpdated - session.timeCreated) / 1000) : null,
342
- },
343
- artifacts: {
344
- patch_path: null,
345
- run_dir: args.runDir,
346
- transcript_ref: session !== null ? `opencode:${session.id}` : null,
347
- },
348
- provenance: {
349
- captured_at: capturedAt,
350
- capture: 'settle-time',
351
- ...(messages.length === 0
352
- ? {
353
- gap:
354
- worker.cwd === null
355
- ? `worker ${worker.label} recorded no clone cwd in its ndjson`
356
- : session === null
357
- ? db === null
358
- ? `opencode store unavailable; worker cwd ${worker.cwd} unrecoverable`
359
- : `no opencode session found for worker cwd ${worker.cwd}`
360
- : `opencode session ${session.id} has no readable message parts`,
361
- }
362
- : {}),
363
- },
364
- })
365
- }
366
- } finally {
367
- db?.close()
368
- }
369
- }
370
-
371
- await appendRolloutLines(opts.ledgerPath, lines)
372
- log(`rollout-ledger: +${lines.length} line(s) (cell ${args.iid} r${args.rep}) → ${opts.ledgerPath}`)
373
- return { lines: lines.length }
374
- },
375
-
376
- async captureProposer(args: ProposerCaptureArgs): Promise<{ lines: number }> {
377
- const capturedAt = now().toISOString()
378
- const reward = proposerRewardV2(args.candResolved, args.baselineResolved, opts.instanceCount)
379
- const line: RolloutLine = {
380
- ...base(capturedAt),
381
- rollout_id: randomUUID(),
382
- parent_rollout_id: null,
383
- candidate_id: candidateId(args.generation, args.candidateIndex),
384
- generation: args.generation,
385
- candidate_index: args.candidateIndex,
386
- role: 'proposer',
387
- task: {
388
- suite: 'swe-arena-proposer',
389
- instance_id: `${candidateId(args.generation, args.candidateIndex)}-${args.proposer}`,
390
- split: 'search',
391
- seed: null,
392
- rep: 0,
393
- },
394
- policy: {
395
- harness: args.harness,
396
- harness_version: null,
397
- model: null,
398
- provider: null,
399
- profile_commit: args.commit,
400
- sampling: null,
401
- },
402
- messages: [],
403
- tool_defs: [],
404
- outcome: {
405
- reward,
406
- reward_source: PROPOSER_REWARD_SOURCE_V2,
407
- verdict: null,
408
- realness_gated: false,
409
- metrics: {
410
- resolved_count: args.candResolved,
411
- baseline_resolved_count: args.baselineResolved,
412
- instance_count: opts.instanceCount,
413
- proposer: args.proposer,
414
- shot_receipts: args.shotReceiptPaths,
415
- },
416
- is_completed: true,
417
- is_truncated: false,
418
- error: null,
419
- },
420
- cost: { usd: null, tokens_in: null, tokens_out: null, tokens_reasoning: null, cache_read: null, cache_write: null, wall_s: null },
421
- artifacts: {
422
- patch_path: args.diffPath !== null && existsSync(args.diffPath) ? args.diffPath : null,
423
- run_dir: null,
424
- transcript_ref: args.shotReceiptPaths[0] ?? null,
425
- },
426
- provenance: {
427
- captured_at: capturedAt,
428
- capture: 'settle-time',
429
- gap: 'proposer transcript joined at backfill time (settle-time line carries the v2 reward label)',
430
- },
431
- }
432
- await appendRolloutLines(opts.ledgerPath, [line])
433
- log(`rollout-ledger: +1 proposer line (${args.proposer}, reward ${reward.toFixed(3)}) → ${opts.ledgerPath}`)
434
- return { lines: 1 }
435
- },
436
- }
437
- }
438
-
439
- /** Parse (generation, candidateIndex) off a campaign cell artifact path — the
440
- * campaign DIRECTORY is the blessed attribution source (never dispatch
441
- * order): `<improveRunDir>/gen-<g>/candidate-<k>/<cell>/…` for candidates,
442
- * `<improveRunDir>/baseline/<cell>/…` (→ −1/−1) for the baseline. */
443
- export function campaignCoordsFromCellPath(path: string): { generation: number; candidateIndex: number } | null {
444
- const cand = /\/gen-(\d+)\/candidate-(\d+)\//.exec(path)
445
- if (cand !== null) return { generation: Number(cand[1]), candidateIndex: Number(cand[2]) }
446
- if (/\/baseline\//.test(path)) return { generation: -1, candidateIndex: -1 }
447
- return null
448
- }