@tangle-network/agent-bench 0.11.3 → 0.13.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,2854 +0,0 @@
1
- /**
2
- * Round-4 outer loop — agent-runtime's `improve()` in the OPTIMIZER SEAT,
3
- * proposing code changes to the loops pi supervisor, evaluated by this typed
4
- * swe-arena harness. Replaces the human/Claude-driven rounds 1-3 recorded in
5
- * supervisor-lab `.evolve/state.json`.
6
- *
7
- * tsx src/swe-arena/outer-loop.mts <config.json> # SPENDS: fires arms + judges
8
- * tsx src/swe-arena/outer-loop.mts --write-config <path> # emit the default round-4 config
9
- * tsx src/swe-arena/outer-loop.mts --calibration-smoke [supRunDir] [--analysts N] [--model M]
10
- *
11
- * One `runRound()` = one `improve()` call with `surface: 'code'`:
12
- *
13
- * (a) DIAGNOSE — the `analyzeGeneration` seam runs the blind diagnosis
14
- * ensemble (diagnosis-ensemble.ts) over the PREVIOUS round's failure
15
- * artifacts (round-3 SUP4 run dirs seeded via config) plus every fresh
16
- * arm run this round produced, and UNIONS the fused findings with
17
- * `rawTraceDistiller` path-context so the coding agent also greps the raw
18
- * traces itself (`rawTraceContext: true` names the mechanism; an explicit
19
- * `analyzeGeneration` wins, so the distiller is composed in directly).
20
- * (b) PROPOSE — Runtime's code candidate driver + a change-space-constrained
21
- * `agenticGenerator` edit an isolated git worktree of loops. The DECLARED
22
- * CHANGE-SPACE is enforced twice: in the generator's verifier (feedback →
23
- * next shot) and fail-closed in the dispatch below (an out-of-space
24
- * candidate never reaches a model token).
25
- * (c) EVALUATE — each candidate surface is a loops commit; the dispatch adds
26
- * a detached eval worktree at that commit, points the supervisor arm's
27
- * extension path at it (armProvenance records the commit), runs the
28
- * 3-instance improvement set through arms.ts + the serialized official
29
- * judge. Score = resolved count; cost guard = wall ratio vs baseline.
30
- * (d) ACCEPT/REJECT — keep-if-better per protocol_v2. The loop NEVER ships:
31
- * `budget.holdout: 'deferred'` makes the lib dispatch zero holdout
32
- * cells, force `hold`, and omit `lift` — the pre-registered 6-instance
33
- * holdout costs real money and runs only in a separate, operator-
34
- * approved run. The would-be-KEEP operator brief is computed post-run
35
- * from campaign cells; every candidate + verdict persists as staircase
36
- * rows in `<roundsDir>/gen-<N>.jsonl`.
37
- *
38
- * BASELINE: the gate's only denominator is the stored premeasured baseline
39
- * artifact ({surfaceHash, campaign}) that the lib validates (surface hash,
40
- * seed, reps, split digest, coverage) before skipping the baseline campaign.
41
- * A missing artifact = the bootstrap run: the baseline is measured
42
- * (cache-resumable) and the artifact written for every later run.
43
- * capabilities.mts fails loud on a stale substrate install that would
44
- * silently drop the passthrough.
45
- *
46
- * SCORING SOURCE: operator-brief evidence + staircase rows derive from the
47
- * LIB's campaign cells (`improve()` result campaigns in memory; the per-cell
48
- * `cached-result.json` caches on disk survive resume) — see cell-evidence.mts.
49
- * The in-process RoundRecorder is dispatch-time only: fail-closed
50
- * change-space enforcement + candidate diff writing. It is NOT a scoring
51
- * source — that recorder role mislabeled a resumed run's baseline
52
- * (r4-mroh3rkt) because cached cells replay without dispatching.
53
- *
54
- * Immutable per protocol_v2 (enforced, not advisory): judge + verify scripts,
55
- * task prompts, model ids, budgets. `assertFrozenArm` pins the arm to the
56
- * round-3 values; the serialized judge enforces its own 1800s floor; the
57
- * change space keeps candidates inside extensions/pi/** and the three named
58
- * src files (plus the `.improve/` raw-trace diagnosis artifact the agentic
59
- * generator's evidence gate requires).
60
- */
61
-
62
- import { appendFile, mkdir, readdir, readFile, rm, symlink, unlink, writeFile } from 'node:fs/promises'
63
- import { existsSync } from 'node:fs'
64
- import process from 'node:process'
65
- import { join } from 'node:path'
66
- import { fileURLToPath, pathToFileURL } from 'node:url'
67
- import {
68
- agenticGenerator,
69
- improve,
70
- rawTraceDistiller,
71
- type CandidateGenerator,
72
- type ImproveCodeRunOptions,
73
- type Verifier,
74
- } from '@tangle-network/agent-runtime'
75
- import { canonicalCandidateDigest, type AgentProfile } from '@tangle-network/agent-interface'
76
- import {
77
- makeProposalFinding,
78
- type AnalystFinding,
79
- type ProposalFinding,
80
- } from '@tangle-network/agent-eval'
81
- import {
82
- FsLabeledScenarioStore,
83
- surfaceHash,
84
- type CampaignResult,
85
- type CodeSurface,
86
- type DispatchContext,
87
- type JudgeConfig,
88
- type MutableSurface,
89
- type PremeasuredOptimizationBaseline,
90
- type Scenario,
91
- } from '@tangle-network/agent-eval/campaign'
92
- import { runVenvPython } from '../benchmarks/_harness.ts'
93
- import { createSweBenchAdapter } from '../benchmarks/swe-bench.ts'
94
- import {
95
- fileTreeImplementationRef,
96
- pythonDistributionImplementationRef,
97
- } from './implementation-ref.ts'
98
- import {
99
- baselineDriftWarnings,
100
- cellsFromCampaign,
101
- gateEvidenceFromCells,
102
- instanceVerdictsFromCells,
103
- loadCampaignCells,
104
- perInstanceFromCells,
105
- replicateCoverageComplete,
106
- replicateRunsFromCells,
107
- resolvedInstanceCount,
108
- sumWallSFromCells,
109
- decideVerdict,
110
- type R4Artifact,
111
- type StaircasePerInstance,
112
- type StaircaseVerdict,
113
- } from './cell-evidence.mts'
114
- import { assertSubstratePassthroughs } from './capabilities.mts'
115
- import {
116
- loadExcludes,
117
- runSupervisorArm,
118
- type SecretsEnv,
119
- type SupervisorArmSpec,
120
- type SupervisorArmResult,
121
- } from './arms.ts'
122
- import { gatesForArmKind, waitForCapacity, ZAI_CODING_ENDPOINT } from './capacity.ts'
123
- import {
124
- defaultAnalysts,
125
- fusedToAnalystFindings,
126
- runDiagnosisEnsemble,
127
- surfacesPlacementRegex,
128
- type AnalystSpec,
129
- type SupRunArtifacts,
130
- } from './diagnosis-ensemble.ts'
131
- import {
132
- defaultProposers,
133
- fanOutLoopsGenerator,
134
- materializeParetoParents,
135
- proposerShotHooks,
136
- resolveAuthorProfile,
137
- type ParetoParentContext,
138
- type ParetoParentSeed,
139
- type PrefilterConfig,
140
- type PrefilterKill,
141
- type ProposerSpec,
142
- type SmokeRunner,
143
- type SmokeVerdict,
144
- } from './proposer-fanout.mts'
145
- import { captureProposerProvenance } from './proposer-provenance.mts'
146
- import { CRASH_ORPHAN_REASON, reconcileCrashOrphansOnDisk } from './ledger-orphans.mts'
147
- import {
148
- AUTHOR_BRIEFING_VERSION,
149
- resolveAuthorBriefing,
150
- writeEvidenceIndex,
151
- type BriefingContext,
152
- } from './briefing.mts'
153
- import {
154
- readCommittedPredicate,
155
- runActivationPredicate,
156
- ACTIVATION_PREDICATE_RELPATH,
157
- type ActivationRecord,
158
- } from './activation.mts'
159
- import {
160
- loadOrCreateScoreSplit,
161
- subScores,
162
- type ScoreSplit,
163
- type ScoreSplitConfig,
164
- } from './score-split.mts'
165
- import {
166
- reportSupervisorRound,
167
- writeSupervisorRunReportSafe,
168
- } from '@tangle-network/agent-eval/supervisor-run'
169
- import {
170
- campaignCoordsFromCellPath,
171
- createSettleCapture,
172
- type SettleCapture,
173
- } from '../rollout-ledger/settle-capture.mts'
174
- import { findSupervisorRunDir } from './arms.ts'
175
- import { installProcessSignalAbort, run, runOk } from './proc.ts'
176
- import { loadInstanceImages } from './run-experiment.mts'
177
- import {
178
- createSerializedJudge,
179
- JUDGE_TIMEOUT_FLOOR_MS,
180
- type SerializedJudge,
181
- } from './serialized-judge.ts'
182
-
183
- // ---------------------------------------------------------------------------
184
- // The DECLARED CHANGE-SPACE (protocol_v2). Pure + unit-tested.
185
- // ---------------------------------------------------------------------------
186
-
187
- export interface ChangeSpace {
188
- /** Directory prefixes (repo-relative, trailing '/') where edits are allowed. */
189
- prefixes: string[]
190
- /** Exact repo-relative files where edits are allowed. */
191
- files: string[]
192
- /** Non-code artifact prefixes allowed to change (the agentic generator's
193
- * raw-trace evidence gate REQUIRES `.improve/raw-trace-diagnosis.md`, which
194
- * finalize commits — evidence metadata, not supervisor code). */
195
- metadataPrefixes: string[]
196
- }
197
-
198
- export const LOOPS_CHANGE_SPACE: ChangeSpace = {
199
- prefixes: ['extensions/pi/'],
200
- files: ['src/worker-evidence.ts', 'src/best-effort.ts', 'src/worker-clone.ts'],
201
- metadataPrefixes: ['.improve/'],
202
- }
203
-
204
- /** Normalize a repo-relative path; `null` = un-normalizable (always a violation). */
205
- export function normalizeRepoPath(p: string): string | null {
206
- let s = p.trim().replace(/\\/g, '/')
207
- if (s.startsWith('"') && s.endsWith('"') && s.length >= 2) {
208
- // git quotes paths containing spaces/specials; minimal unquote.
209
- s = s.slice(1, -1).replace(/\\"/g, '"')
210
- }
211
- while (s.startsWith('./')) s = s.slice(2)
212
- if (s.length === 0) return null
213
- if (s.startsWith('/')) return null // absolute — never a repo-relative candidate path
214
- const segments = s.split('/')
215
- if (segments.some((seg) => seg === '..' || seg === '')) return null // traversal / '//' — fail closed
216
- return s
217
- }
218
-
219
- /** Paths that fall OUTSIDE the declared change-space (empty ⇒ compliant). */
220
- export function changeSpaceViolations(paths: string[], space: ChangeSpace = LOOPS_CHANGE_SPACE): string[] {
221
- const violations: string[] = []
222
- for (const raw of paths) {
223
- const p = normalizeRepoPath(raw)
224
- if (p === null) {
225
- violations.push(raw)
226
- continue
227
- }
228
- const allowed =
229
- space.files.includes(p) ||
230
- space.prefixes.some((pre) => p.startsWith(pre)) ||
231
- space.metadataPrefixes.some((pre) => p.startsWith(pre))
232
- if (!allowed) violations.push(p)
233
- }
234
- return violations
235
- }
236
-
237
- /** Changed paths from `git status --porcelain=v1 --untracked-files=all`.
238
- * Renames contribute BOTH sides (removing an out-of-space file is a change). */
239
- export function porcelainChangedPaths(stdout: string): string[] {
240
- const paths: string[] = []
241
- for (const line of stdout.split('\n')) {
242
- if (line.trim().length === 0) continue
243
- const entry = line.slice(3)
244
- const arrow = entry.indexOf(' -> ')
245
- if (arrow !== -1) {
246
- paths.push(entry.slice(0, arrow).trim(), entry.slice(arrow + 4).trim())
247
- } else {
248
- paths.push(entry.trim())
249
- }
250
- }
251
- return paths.filter((p) => p.length > 0)
252
- }
253
-
254
- // ---------------------------------------------------------------------------
255
- // Dispatch clocks. The campaign's dispatchTimeoutMs races the ENTIRE dispatch
256
- // — including the endpoint capacity-gate wait — so a legitimate multi-hour
257
- // capacity hold was billed to the cell's work budget (measured: a 58-min gate
258
- // hold pushed the astropy baseline cell over the 7200s clock and the whole
259
- // candidate became 'rejected-incomplete'). Fix: the cell's REAL work clock
260
- // (`runWithPostGateClock`) starts only after the gates clear, and the campaign
261
- // clock is widened to cover worst-case gate holds so it can never fire during
262
- // a legitimate wait. Both clocks still fail loud — a hung arm is bounded by
263
- // dispatchTimeoutMs post-gate, and the widened campaign clock is the backstop.
264
- // ---------------------------------------------------------------------------
265
-
266
- /** Supervisor arms gate on BOTH endpoints (worker z.ai path + brain router path). */
267
- export const SUPERVISOR_GATE_COUNT = 2
268
-
269
- /** capacity.ts's default waitCeilingMs (orchestrate.sh: 300 min/gate). */
270
- export const DEFAULT_GATE_WAIT_CEILING_MS = 300 * 60_000
271
-
272
- /** Extra time for process/worktree cleanup after the post-gate clock aborts.
273
- * Judge time is budgeted separately because one verdict may require two full
274
- * attempts. The campaign must not abandon either attempt or cleanup. */
275
- export const DISPATCH_CLEANUP_GRACE_MS = 5 * 60_000
276
-
277
- /** The widened ceiling handed to the campaign: per-cell work budget PLUS the
278
- * worst-case capacity-gate holds (gates run sequentially, each with its own
279
- * ceiling). The campaign clock starts at dispatch entry — before the gates —
280
- * so it must cover them; `waitForCapacity` itself fails the cell at each
281
- * gate's own ceiling, so total cell time stays bounded. */
282
- export function campaignDispatchCeilingMs(
283
- config: Pick<OuterLoopConfig, 'dispatchTimeoutMs' | 'gateWaitCeilingMs' | 'judgeTimeoutMs'>,
284
- gateCount = SUPERVISOR_GATE_COUNT,
285
- ): number {
286
- const judgeSettlementMs = config.judgeTimeoutMs ?? JUDGE_TIMEOUT_FLOOR_MS
287
- return (
288
- config.dispatchTimeoutMs +
289
- gateCount * (config.gateWaitCeilingMs ?? DEFAULT_GATE_WAIT_CEILING_MS) +
290
- 2 * judgeSettlementMs +
291
- DISPATCH_CLEANUP_GRACE_MS
292
- )
293
- }
294
-
295
- /** Run `work` under `timeoutMs`, with the clock started AFTER `awaitGates`
296
- * resolves — a capacity hold is never billed to the cell's work budget.
297
- * Gate failures (no capacity within a gate's own ceiling) still reject. */
298
- export async function runWithPostGateClock<T>(opts: {
299
- awaitGates: (signal?: AbortSignal) => Promise<void>
300
- work: (signal: AbortSignal) => Promise<T>
301
- timeoutMs: number
302
- label?: string
303
- /** Caller cancellation remains active during both capacity waiting and work. */
304
- signal?: AbortSignal
305
- }): Promise<T> {
306
- opts.signal?.throwIfAborted()
307
- await opts.awaitGates(opts.signal)
308
- opts.signal?.throwIfAborted()
309
- const abort = new AbortController()
310
- const linked = linkAbortSignals([abort.signal, ...(opts.signal ? [opts.signal] : [])])
311
- let timer: NodeJS.Timeout | undefined
312
- let timedOut = false
313
- const timeoutError = new Error(
314
- `post-gate dispatch exceeded ${opts.timeoutMs}ms${opts.label ? ` (${opts.label})` : ''} — failed loud, gate wait unbilled`,
315
- )
316
- try {
317
- if (opts.timeoutMs > 0) {
318
- timer = setTimeout(() => {
319
- timedOut = true
320
- abort.abort(timeoutError)
321
- }, opts.timeoutMs)
322
- timer.unref?.()
323
- }
324
- const result = await opts.work(linked.signal)
325
- if (timedOut) throw timeoutError
326
- opts.signal?.throwIfAborted()
327
- return result
328
- } catch (err) {
329
- if (timedOut && err !== timeoutError) {
330
- const cleanupFailure = err instanceof Error ? err.message : String(err)
331
- throw new Error(`${timeoutError.message}; cleanup failed: ${cleanupFailure}`, { cause: err })
332
- }
333
- if (timedOut) throw timeoutError
334
- if (opts.signal?.aborted) {
335
- if (err !== opts.signal.reason) {
336
- const cleanupFailure = err instanceof Error ? err.message : String(err)
337
- const interrupted = opts.signal.reason instanceof Error
338
- ? opts.signal.reason.message
339
- : String(opts.signal.reason ?? 'caller aborted')
340
- throw new Error(`${interrupted}; cleanup failed: ${cleanupFailure}`, { cause: err })
341
- }
342
- throw opts.signal.reason
343
- }
344
- throw err
345
- } finally {
346
- if (timer) clearTimeout(timer)
347
- linked.dispose()
348
- }
349
- }
350
-
351
- function linkAbortSignals(signals: AbortSignal[]): { signal: AbortSignal; dispose: () => void } {
352
- const controller = new AbortController()
353
- const listeners = new Map<AbortSignal, () => void>()
354
- for (const signal of signals) {
355
- const onAbort = () => controller.abort(signal.reason)
356
- listeners.set(signal, onAbort)
357
- if (signal.aborted) {
358
- onAbort()
359
- break
360
- }
361
- signal.addEventListener('abort', onAbort, { once: true })
362
- }
363
- return {
364
- signal: controller.signal,
365
- dispose: () => {
366
- for (const [signal, listener] of listeners) signal.removeEventListener('abort', listener)
367
- },
368
- }
369
- }
370
-
371
- function withParentCancellation<T extends CandidateGenerator>(generator: T, signal?: AbortSignal): T {
372
- if (!signal) return generator
373
- return {
374
- ...generator,
375
- async generate(args) {
376
- signal.throwIfAborted()
377
- const linked = linkAbortSignals([args.signal, signal])
378
- try {
379
- const result = await generator.generate({ ...args, signal: linked.signal })
380
- signal.throwIfAborted()
381
- return result
382
- } finally {
383
- linked.dispose()
384
- }
385
- },
386
- } as T
387
- }
388
-
389
- // ---------------------------------------------------------------------------
390
- // Scoring primitives — replicate semantics, the pinned baseline, and the
391
- // protocol_v2 verdict — live in cell-evidence.mts (pure over lib campaign
392
- // cells). Re-exported here so existing consumers/tests keep one import home.
393
- // ---------------------------------------------------------------------------
394
-
395
- export {
396
- baselineDriftWarnings,
397
- cellsFromCampaign,
398
- decideVerdict,
399
- gateEvidenceFromCells,
400
- instanceVerdictsFromCells,
401
- loadCampaignCells,
402
- loadCandidateCellGroups,
403
- perInstanceFromCells,
404
- replicateCoverageComplete,
405
- replicateRunsFromCells,
406
- resolvedInstanceCount,
407
- sumWallSFromCells,
408
- type EvidenceCell,
409
- type R4Artifact,
410
- type ReplicateRun,
411
- type StaircasePerInstance,
412
- type StaircaseVerdict,
413
- } from './cell-evidence.mts'
414
-
415
- // ---------------------------------------------------------------------------
416
- // Launch guards. (a) The arms + judge + proposer all die confusingly hours in
417
- // when the two API keys are absent (the launcher forgot dotenvx) — refuse at
418
- // t=0 instead. (b) Two outer-loops sharing an outDir corrupt the campaign
419
- // runDir and the arm-run caches — a pid-file lock with a staleness check makes
420
- // the race impossible.
421
- // ---------------------------------------------------------------------------
422
-
423
- export const REQUIRED_LAUNCH_ENV = ['TANGLE_API_KEY', 'ZAI_API_KEY'] as const
424
-
425
- export function assertLaunchEnv(env: Record<string, string | undefined> = process.env): void {
426
- const missing = REQUIRED_LAUNCH_ENV.filter((k) => !env[k] || env[k]!.trim().length === 0)
427
- if (missing.length > 0) {
428
- throw new Error(
429
- `outer-loop: ${missing.join(' + ')} absent from env — launch through dotenvx (dotenvx run -f agent-state.env -f tangle-router.env -- ...)`,
430
- )
431
- }
432
- }
433
-
434
- /** True when `pid` is a live process (EPERM = alive but not ours — still live). */
435
- export function isPidAlive(pid: number): boolean {
436
- try {
437
- process.kill(pid, 0)
438
- return true
439
- } catch (err) {
440
- return (err as NodeJS.ErrnoException).code === 'EPERM'
441
- }
442
- }
443
-
444
- export const INSTANCE_LOCK_FILENAME = 'outer-loop.pid'
445
-
446
- export interface InstanceLock {
447
- path: string
448
- release: () => Promise<void>
449
- }
450
-
451
- /** Single-instance pid-file lock in `outDir`. `wx` creation is the atomic
452
- * claim; an existing file is honored only while its pid is alive (a crashed
453
- * loop's stale lock — dead pid or garbage — is reclaimed). Pid reuse can in
454
- * principle false-positive a stale lock as live; that fails SAFE (refuses to
455
- * start) and clears on the next reboot cycle. */
456
- export async function acquireInstanceLock(outDir: string, pid: number = process.pid): Promise<InstanceLock> {
457
- await mkdir(outDir, { recursive: true })
458
- const lockPath = join(outDir, INSTANCE_LOCK_FILENAME)
459
- for (let attempt = 0; attempt < 2; attempt++) {
460
- try {
461
- await writeFile(lockPath, `${pid}\n`, { flag: 'wx' })
462
- return {
463
- path: lockPath,
464
- release: async () => {
465
- const raw = (await readFile(lockPath, 'utf8').catch(() => '')).trim()
466
- if (raw === String(pid)) await unlink(lockPath).catch(() => {})
467
- },
468
- }
469
- } catch (err) {
470
- if ((err as NodeJS.ErrnoException).code !== 'EEXIST') throw err
471
- const raw = (await readFile(lockPath, 'utf8').catch(() => '')).trim()
472
- const holder = Number.parseInt(raw, 10)
473
- if (Number.isInteger(holder) && holder > 0 && holder !== pid && isPidAlive(holder)) {
474
- throw new Error(
475
- `outer-loop: another outer-loop (pid ${holder}) holds ${lockPath} — single-instance lock, refusing to race`,
476
- )
477
- }
478
- await unlink(lockPath).catch(() => {}) // stale: dead pid or garbage content
479
- }
480
- }
481
- throw new Error(`outer-loop: could not acquire ${lockPath} after clearing a stale lock`)
482
- }
483
-
484
- // ---------------------------------------------------------------------------
485
- // Staircase rows — accepted successors + rejected dots, one JSONL row each.
486
- // ---------------------------------------------------------------------------
487
-
488
- export const STAIRCASE_SCHEMA = 'swe-arena.staircase.v1'
489
-
490
- export interface StaircaseRow {
491
- schema: typeof STAIRCASE_SCHEMA
492
- round: number
493
- generation: number
494
- runId: string
495
- at: string
496
- /** Candidate surface hash (agent-eval surface identity). */
497
- candidate: string
498
- candidateCommit: string | null
499
- /** Incumbent surface hash the candidate mutated. */
500
- parent: string
501
- parentResolvedCount: number
502
- label?: string
503
- rationale?: string
504
- changedFiles: string[]
505
- changeSpaceViolations: string[]
506
- perInstance: StaircasePerInstance[]
507
- resolvedCount: number
508
- coverageComplete: boolean
509
- wallS: number
510
- baselineWallS: number
511
- costRatio: number | null
512
- costGuardRatio: number
513
- /** Whether runOptimization's internal keep-if-better advanced the incumbent
514
- * to this candidate (composite-only rule; may diverge from `verdict` when
515
- * the protocol cost guard rejects a gaining candidate — divergence is the
516
- * signal, so both are recorded). */
517
- internallyPromoted: boolean
518
- verdict: StaircaseVerdict
519
- /** Present only on `rejected-prefilter` dots: which pre-filter stage killed
520
- * the candidate and why (e.g. `smoke: pallets__flask-5014 unresolved`). */
521
- killReason?: string
522
- holdout: 'operator-approval-required' | 'not-run'
523
- armProvenance: { repo: string; commit: string } | null
524
- diffPath: string | null
525
- diffSha256: string | null
526
- /** GEN-5 public/private sub-scores (selection stays on the combined count;
527
- * the private sub-score is never surfaced to proposers). */
528
- split?: {
529
- publicInstances: string[]
530
- privateInstances: string[]
531
- publicResolvedCount: number
532
- privateResolvedCount: number
533
- }
534
- /** GEN-5 activation-gate outcome for this candidate. */
535
- activation?: ActivationRecord
536
- }
537
-
538
- const STAIRCASE_VERDICTS: ReadonlySet<string> = new Set([
539
- 'accepted',
540
- 'rejected-no-gain',
541
- 'rejected-cost',
542
- 'rejected-out-of-space',
543
- 'rejected-incomplete',
544
- 'rejected-prefilter',
545
- 'quarantined-inactive',
546
- ])
547
-
548
- /** Parse + validate one staircase JSONL row. Throws on schema drift. */
549
- export function parseStaircaseRow(line: string): StaircaseRow {
550
- const row = JSON.parse(line) as StaircaseRow
551
- if (row.schema !== STAIRCASE_SCHEMA) throw new Error(`staircase row: unknown schema ${JSON.stringify(row.schema)}`)
552
- for (const field of ['round', 'generation', 'resolvedCount', 'parentResolvedCount', 'wallS', 'baselineWallS', 'costGuardRatio'] as const) {
553
- if (typeof row[field] !== 'number') throw new Error(`staircase row: ${field} must be a number`)
554
- }
555
- for (const field of ['runId', 'at', 'candidate', 'parent'] as const) {
556
- if (typeof row[field] !== 'string' || row[field].length === 0) throw new Error(`staircase row: ${field} must be a non-empty string`)
557
- }
558
- if (!Array.isArray(row.perInstance)) throw new Error('staircase row: perInstance must be an array')
559
- if (!Array.isArray(row.changedFiles) || !Array.isArray(row.changeSpaceViolations)) {
560
- throw new Error('staircase row: changedFiles/changeSpaceViolations must be arrays')
561
- }
562
- if (!STAIRCASE_VERDICTS.has(row.verdict)) throw new Error(`staircase row: unknown verdict ${JSON.stringify(row.verdict)}`)
563
- if (typeof row.coverageComplete !== 'boolean' || typeof row.internallyPromoted !== 'boolean') {
564
- throw new Error('staircase row: coverageComplete/internallyPromoted must be booleans')
565
- }
566
- if (row.costRatio !== null && typeof row.costRatio !== 'number') throw new Error('staircase row: costRatio must be number|null')
567
- return row
568
- }
569
-
570
- // ---------------------------------------------------------------------------
571
- // Config.
572
- // ---------------------------------------------------------------------------
573
-
574
- /** Round 1-3 artifact home (this session's scratchpad). Config-overridable —
575
- * a future round supplies its own artifact roots. */
576
- export const DEFAULT_HH_SCRATCHPAD =
577
- '/tmp/claude-1000/-home-drew-code-supervisor-lab/f06fd156-042a-4ef9-bd88-f2ec7f52b90c/scratchpad/hh'
578
-
579
- export interface SeedArtifactRun {
580
- iid: string
581
- arm: string
582
- dir: string
583
- patchPath?: string
584
- /** Official-judge outcome for the seed run (round-3 values pinned in config). */
585
- resolved: boolean | null
586
- }
587
-
588
- export interface FrozenArmParams {
589
- workerModel: string
590
- driverModel: string
591
- budget: number
592
- maxSandboxes: number
593
- maxUsd: number
594
- maxDepth: number
595
- timeoutMs: number
596
- envKnobs?: Record<string, string>
597
- }
598
-
599
- /** The round-3 (SUP4) arm — protocol_v2 immutables. */
600
- export const FROZEN_ARM: FrozenArmParams = {
601
- workerModel: 'zai-coding-plan/glm-5.2',
602
- driverModel: 'glm-5.2',
603
- budget: 40,
604
- maxSandboxes: 4,
605
- maxUsd: 8,
606
- maxDepth: 3,
607
- timeoutMs: 2_800_000,
608
- }
609
-
610
- export interface OuterLoopConfig {
611
- round: number
612
- /** Improvement set — the arena `improve()` trains on. */
613
- instances: string[]
614
- /** Pre-registered holdout. RECORDED here so the flag + operator instruction
615
- * are self-contained; this driver NEVER runs them. */
616
- holdoutInstances: string[]
617
- loopsRepo: string
618
- loopsBaseRef: string
619
- armName: string
620
- arm: FrozenArmParams
621
- verifyDir: string
622
- outDir: string
623
- /** Staircase home, e.g. /home/drew/code/supervisor-lab/.evolve/rounds. */
624
- roundsDir: string
625
- secretsDir: string
626
- envFiles: string[]
627
- instanceImagesPath?: string
628
- judgeTimeoutMs?: number
629
- gateWaitCeilingMs?: number
630
- capacityModel?: string
631
- generations: number
632
- populationSize: number
633
- /** Replicate cells per (candidate × instance). Default 1. Instances count as
634
- * resolved only when ALL replicates resolve (see resolvedInstanceCount) —
635
- * single-rep scoring flips instance outcomes run-to-run. */
636
- repsPerInstance?: number
637
- /** Stored `PremeasuredOptimizationBaseline` JSON ({surfaceHash, campaign})
638
- * from a prior run's baseline campaign — REQUIRED, the gate's only
639
- * denominator. The LIB validates the artifact (surface hash, seed, reps,
640
- * split digest, coverage) before skipping the baseline campaign, so a
641
- * wrong artifact fails loud at t≈0. BOOTSTRAP: when the file does not
642
- * exist yet, this run MEASURES the baseline (cache-resumable) and WRITES
643
- * the artifact here for every later run to consume. */
644
- premeasuredBaselinePath: string
645
- /** DEPTH for the agentic generator — forwarded as
646
- * budget.maxImprovementShots; the LIB owns the dial (capabilities.mts
647
- * fails loud on a substrate that would drop it). */
648
- maxShots: number
649
- proposerHarness: 'pi'
650
- /** Complete profile used by the single-author path and as the run's admitted author identity. */
651
- proposerProfile: string
652
- proposerTimeoutMs: number
653
- /** GEN-3 proposer fan-out: N proposers author candidates CONCURRENTLY, each
654
- * an AgentProfile-pinned harness invocation (see proposer-fanout.mts).
655
- * When set, `populationSize` MUST equal `proposers.length` (one candidate
656
- * slot per proposer — enforced at launch). Unset = the legacy
657
- * single-author generator (`proposerHarness` + bare invocation).
658
- * A spec with `engine` set is a GEPA seat (gepa-seat.mts); the
659
- * agent-eval external-GEPA adapter optimizes ONE change-space file as a
660
- * string against the pre-filter smoke cell; requires `prefilter.enabled`. */
661
- proposers?: ProposerSpec[]
662
- /** GEN-3 cheap pre-filter: per candidate, change-space + tsc (the authoring
663
- * verifier) plus ONE smoke arm cell before any full-evaluation spend.
664
- * Killed candidates become `rejected-prefilter` staircase dots. */
665
- prefilter?: PrefilterConfig
666
- /** GEN-4 Pareto parents: prior-run frontier candidates (loops commits +
667
- * measured per-instance results) seeded into every author's prompt and
668
- * the merge seat's explicit input. Seeded at OUR buildPrompt seam, not the
669
- * lib's `ctx.paretoParents` — the lib frontier is within-run only and a
670
- * prior campaign cannot be injected without its runDir + ledger receipts
671
- * (see proposer-fanout.mts). */
672
- paretoParents?: ParetoParentSeed[]
673
- /** Replicates per holdout instance in the operator-approved certification
674
- * run (holdout-certify.mts). Default 2 — the gen-2 winner failed 3/6 vs
675
- * 4/6 on a 1-rep holdout with exactly one discordant cell, a known
676
- * single-rep noise class. */
677
- holdoutRepsPerInstance?: number
678
- /** SAME-PROTOCOL parent measurement the certification bar compares against:
679
- * an explicit {iid -> AND-verdict} map measured under the identical
680
- * reps/fail-closed protocol, or 'measure' — the incumbent runs the same
681
- * 2-rep holdout first in the certification run. */
682
- holdoutBaseline?: Record<string, boolean> | 'measure'
683
- /** GEN-5 public/private score split (score-split.mts): proposers + the
684
- * pre-filter see only PUBLIC instances' scores/evidence; selection stays
685
- * on the combined set. Unset = everything public (pre-gen-5 behavior). */
686
- scoreSplit?: ScoreSplitConfig
687
- /** GEN-5 MAP+TOOLBOX briefing (briefing.mts): write the per-run evidence
688
- * index and append the toolbox/permission briefing (change-space
689
- * overridable) to every author prompt. */
690
- briefing?: typeof AUTHOR_BRIEFING_VERSION
691
- /** GEN-5 activation gate (activation.mts): require a machine-checkable
692
- * activation predicate per candidate (prefilter-enforced) and quarantine
693
- * candidates whose mechanism never fired in their own campaign traces. */
694
- activationGate?: boolean
695
- /** GEN-5 settle-time rollout-ledger capture (rollout-ledger/settle-capture.mts):
696
- * emit tangle.rollout.v1 lines live after each cell judges, with label-v2
697
- * rewards. Default path: <outDir>/rollout-ledger.jsonl. */
698
- rolloutLedger?: { enabled: boolean; path?: string; opencodeDb?: string }
699
- /** GEN-5 evidence map: prior run outDirs whose arm-runs/judge/candidate
700
- * evidence the authors may mine (rendered into the evidence index). */
701
- priorEvidenceDirs?: string[]
702
- /** Router model ids for the blind diagnosis ensemble (config, never a
703
- * hardcoded unrouted model). */
704
- analystModels: string[]
705
- /** Previous round's failure artifacts, diagnosed before generation 0. */
706
- seedArtifactRuns: SeedArtifactRun[]
707
- costGuardRatio: number
708
- dispatchTimeoutMs: number
709
- }
710
-
711
- export function assertFrozenArm(arm: FrozenArmParams): void {
712
- const drift: string[] = []
713
- for (const key of ['workerModel', 'driverModel', 'budget', 'maxSandboxes', 'maxUsd', 'maxDepth'] as const) {
714
- if (arm[key] !== FROZEN_ARM[key]) drift.push(`${key}: ${JSON.stringify(arm[key])} != ${JSON.stringify(FROZEN_ARM[key])}`)
715
- }
716
- if (drift.length > 0) {
717
- throw new Error(
718
- `protocol_v2 violation: arm params are immutable (round-3 frozen values) — ${drift.join('; ')}`,
719
- )
720
- }
721
- }
722
-
723
- /** Committed per-instance verify scripts (fixtures/verify/<iid>.sh) — the
724
- * durable home; the experiment's scratchpad copy did not survive a reboot. */
725
- export const FIXTURES_VERIFY_DIR = fileURLToPath(new URL('./fixtures/verify', import.meta.url))
726
-
727
- export function defaultRound4Config(
728
- hh = DEFAULT_HH_SCRATCHPAD,
729
- opts: { outDirName?: string } = {},
730
- ): OuterLoopConfig {
731
- const round3 = [
732
- { iid: 'astropy__astropy-13033', resolved: false },
733
- { iid: 'django__django-11532', resolved: false },
734
- { iid: 'matplotlib__matplotlib-20826', resolved: true },
735
- ]
736
- return {
737
- round: 4,
738
- instances: round3.map((r) => r.iid),
739
- holdoutInstances: [
740
- 'astropy__astropy-14182',
741
- 'django__django-12774',
742
- 'django__django-14140',
743
- 'scikit-learn__scikit-learn-14894',
744
- 'sympy__sympy-20438',
745
- 'pytest-dev__pytest-7236',
746
- ],
747
- loopsRepo: '/home/drew/code/loops',
748
- loopsBaseRef: 'feat/supervisor-evidence-flow',
749
- armName: 'R4',
750
- arm: { ...FROZEN_ARM },
751
- verifyDir: FIXTURES_VERIFY_DIR,
752
- outDir: join(hh, opts.outDirName ?? 'r4'),
753
- roundsDir: '/home/drew/code/supervisor-lab/.evolve/rounds',
754
- secretsDir: '/home/drew/company/devops/secrets',
755
- envFiles: ['agent-state.env', 'tangle-router.env'],
756
- generations: 1,
757
- populationSize: 2,
758
- repsPerInstance: 2,
759
- // The reps-confirmed baseline artifact (gen-1 measured: astropy F/F,
760
- // django T/F → F fail-closed, matplotlib T/T = 1/3) lives here once the
761
- // bootstrap run writes it; the lib validates it on every consumption.
762
- premeasuredBaselinePath: join(hh, 'r4', 'premeasured-baseline.json'),
763
- maxShots: 3,
764
- proposerHarness: 'pi',
765
- proposerProfile: 'default-author.profile.json',
766
- // Per author SHOT (agenticGenerator timeoutMs). 20 min timed out 3× under
767
- // degraded capacity in gen-1 ("author shot timed out") — doubled to 40 min.
768
- proposerTimeoutMs: 2_400_000,
769
- analystModels: ['glm-5.2', 'glm-5.2', 'glm-5.2'],
770
- seedArtifactRuns: round3.map((r) => ({
771
- iid: r.iid,
772
- arm: 'SUP4',
773
- dir: join(hh, 'runs', r.iid, 'SUP4'),
774
- patchPath: join(hh, 'patches', `${r.iid}.sup4.patch`),
775
- resolved: r.resolved,
776
- })),
777
- costGuardRatio: 1.2,
778
- dispatchTimeoutMs: 7_200_000,
779
- }
780
- }
781
-
782
- // ---------------------------------------------------------------------------
783
- // GEN-3 configuration — proposer fan-out + pre-filter + the widened
784
- // improvement set + the 2-rep holdout protocol.
785
- // ---------------------------------------------------------------------------
786
-
787
- /** The gen-3 improvement set: the round-3 trio plus the three BOTH-FAIL
788
- * instances from the original head-to-head (solo glm-5.2 ALSO failed them —
789
- * any resolution beats solo, not just the parent). All six carry committed,
790
- * dual-calibrated verify fixtures (repro base-fail/gold-pass + gold
791
- * official-resolved). */
792
- export const GEN3_IMPROVEMENT_SET = [
793
- 'astropy__astropy-13033',
794
- 'django__django-11532',
795
- 'matplotlib__matplotlib-20826',
796
- 'pydata__xarray-4687',
797
- 'pytest-dev__pytest-6197',
798
- 'sphinx-doc__sphinx-9658',
799
- ] as const
800
-
801
- /** Never-registered spare pool, pre-named in case a gen-3 instance has to be
802
- * replaced (calibration regression, image loss). */
803
- export const GEN3_SPARE_POOL = [
804
- 'sympy__sympy-17318',
805
- 'scikit-learn__scikit-learn-14087',
806
- 'astropy__astropy-14508',
807
- ] as const
808
-
809
- /** Resolve the pre-filter smoke instance. 'cheapest-of-set' picks the
810
- * improvement-set instance with the smallest summed baseline wall seconds
811
- * (from the premeasured artifact's cells); with no baseline measurement yet
812
- * it falls back to the first instance. An explicit iid passes through. */
813
- export function resolveSmokeInstance(
814
- smokeInstance: string,
815
- instances: readonly string[],
816
- baselineCells: import('./cell-evidence.mts').EvidenceCell[] | null,
817
- ): string {
818
- if (smokeInstance !== 'cheapest-of-set') return smokeInstance
819
- if (instances.length === 0) throw new Error('resolveSmokeInstance: empty improvement set')
820
- if (baselineCells === null || baselineCells.length === 0) return instances[0]!
821
- const wall = new Map<string, number>()
822
- for (const cell of baselineCells) {
823
- if (cell.artifact === null || cell.artifact.kind !== 'swe-arm') continue
824
- wall.set(cell.scenarioId, (wall.get(cell.scenarioId) ?? 0) + cell.artifact.wallS)
825
- }
826
- let best: string | null = null
827
- let bestWall = Number.POSITIVE_INFINITY
828
- for (const iid of instances) {
829
- const w = wall.get(iid)
830
- if (w !== undefined && w < bestWall) {
831
- best = iid
832
- bestWall = w
833
- }
834
- }
835
- return best ?? instances[0]!
836
- }
837
-
838
- /**
839
- * The gen-3 config: protocol round 4 continues (frozen arm, same holdout
840
- * registry, same roundsDir staircase) with the gen-3 machinery on:
841
- *
842
- * - THREE parallel Pi proposers using the exact GLM author profile that
843
- * differ by diagnosis slice/lens — fan-out diversity without unproven
844
- * harness seats; `populationSize` = `proposers.length`.
845
- * - Pre-filter enabled at the mechanism bar on the cheapest-of-set smoke
846
- * instance ('pallets__flask-5014' becomes the designated smoke once its
847
- * verify fixture is authored + calibrated; it has none committed yet).
848
- * - The 6-instance improvement set. The premeasured-baseline artifact path
849
- * is NEW (gen3/): the lib validates a premeasured campaign against the
850
- * FULL scenario split digest, so the 3-instance round-4 artifact cannot
851
- * seed a 6-instance split — the first gen-3 run is the bootstrap that
852
- * measures all six (cache-resumable) and writes the artifact; the three
853
- * new instances are thereby measured on the first round.
854
- * - Holdout protocol pinned at 2 reps, parent measured under the SAME
855
- * protocol ('measure'), operator valve unchanged (holdout: 'deferred').
856
- */
857
- export function defaultGen3Config(
858
- hh = DEFAULT_HH_SCRATCHPAD,
859
- opts: { outDirName?: string } = {},
860
- ): OuterLoopConfig {
861
- const base = defaultRound4Config(hh, opts)
862
- const outDirName = opts.outDirName ?? 'gen3'
863
- const proposers: ProposerSpec[] = [
864
- {
865
- name: 'default-author',
866
- profile: 'default-author.profile.json',
867
- harness: 'pi',
868
- model: 'glm-5.2',
869
- },
870
- {
871
- name: 'mechanics-author',
872
- profile: 'default-author.profile.json',
873
- harness: 'pi',
874
- model: 'glm-5.2',
875
- diagnosisSlice: 'mechanics',
876
- lens: 'Focus on MECHANICS: worker lifecycle, sandbox/clone contracts, settlement and delivery paths. Prefer code-path fixes over prompt wording.',
877
- },
878
- {
879
- name: 'prompts-author',
880
- profile: 'default-author.profile.json',
881
- harness: 'pi',
882
- model: 'glm-5.2',
883
- diagnosisSlice: 'prompts',
884
- lens: 'Focus on PROMPTS: worker/brain instruction wording, placement guidance, self-check discipline. Prefer prompt/instruction changes over code-path rewrites.',
885
- },
886
- ]
887
- return {
888
- ...base,
889
- instances: [...GEN3_IMPROVEMENT_SET],
890
- outDir: join(hh, outDirName),
891
- premeasuredBaselinePath: join(hh, outDirName, 'premeasured-baseline.json'),
892
- populationSize: proposers.length,
893
- proposers,
894
- prefilter: { enabled: true, smokeInstance: 'cheapest-of-set', requireResolved: false },
895
- holdoutRepsPerInstance: 2,
896
- holdoutBaseline: 'measure',
897
- }
898
- }
899
-
900
- // ---------------------------------------------------------------------------
901
- // GEN-4 configuration — pinned per-proposer models (recorded in provenance),
902
- // Pareto-parent seeding from the gen-3 frontier, and a dedicated merge seat.
903
- // ---------------------------------------------------------------------------
904
-
905
- /** The gen-3 frontier (run r4-mrwc0awe): winner + runner-up, both 2/6 vs the
906
- * 1/6 baseline on DIFFERENT instances — complementary lessons, the merge
907
- * seat's input. Per-instance verdicts are the fail-closed all-reps values
908
- * from `.evolve/rounds/gen-0.jsonl`. */
909
- export const GEN3_PARETO_PARENTS: ParetoParentSeed[] = [
910
- {
911
- commit: 'cc0d95584c7ea14324cd57c21fe946c7c0f53827',
912
- label: 'default-author',
913
- resolvedInstances: ['pydata__xarray-4687', 'sphinx-doc__sphinx-9658'],
914
- note:
915
- 'mechanical patch-risk scan (patchRiskWarnings in src/worker-evidence.ts, threaded through ' +
916
- 'extensions/pi/loops.ts) + never-reword / test-seam / run-the-neighbors worker rules + 3-check ' +
917
- 'reviewer; pytest-dev__pytest-6197 split 0/1 across reps (near-miss)',
918
- },
919
- {
920
- commit: 'a7a2a982e51551de3a8e796ee2efc448ed405e6a',
921
- label: 'prompts-author',
922
- resolvedInstances: ['django__django-11532', 'pydata__xarray-4687'],
923
- note:
924
- 'prompt-only: hidden-suite bullet in the supervisor GOAL-authoring section (the django seam), ' +
925
- 'frozen-behavior worker section + run-the-repo-tests discipline, 2-check reviewer; ' +
926
- 'sphinx-doc__sphinx-9658 split 0/1 across reps (near-miss)',
927
- },
928
- ]
929
-
930
- /**
931
- * The gen-4 config: protocol round 4 continues (frozen arm, same holdout
932
- * registry, same roundsDir staircase) with three changes as a unit:
933
- *
934
- * 1. PINNED PER-PROPOSER MODELS — GLM 5.2 and DeepSeek V4 Flash run through
935
- * Pi and Tangle Router; the merge seat uses the same exact GLM profile.
936
- * 2. PARETO PARENTS — the gen-3 winner + runner-up diffs and their measured
937
- * per-instance results seed every author's prompt; the merge seat's task
938
- * is their coherent union. Seeded at the buildPrompt seam (our seam): the
939
- * lib's `ctx.paretoParents` frontier is within-run only, and a prior
940
- * campaign cannot cross runs without its runDir + ledger receipts.
941
- * 3. PINNED BASELINE — the premeasured artifact at `hh/gen4/` is BUILT from
942
- * gen-3's measured baseline cells (premeasured-from-cells.mts; gen-3
943
- * measured astropy F, django F, matplotlib F, xarray F, pytest F,
944
- * sphinx T — matplotlib/django false under current weather), so gen-4
945
- * spends nothing re-measuring and fails loud if the loops tip moved.
946
- */
947
- export function defaultGen4Config(
948
- hh = DEFAULT_HH_SCRATCHPAD,
949
- opts: { outDirName?: string; includeDeepseek?: boolean } = {},
950
- ): OuterLoopConfig {
951
- const base = defaultGen3Config(hh, { outDirName: opts.outDirName ?? 'gen4' })
952
- const proposers: ProposerSpec[] = [
953
- {
954
- name: 'glm-author',
955
- profile: 'default-author.profile.json',
956
- harness: 'pi',
957
- model: 'glm-5.2',
958
- },
959
- ...(opts.includeDeepseek === false
960
- ? []
961
- : [
962
- {
963
- name: 'deepseek-author',
964
- profile: 'deepseek-author.profile.json',
965
- harness: 'pi',
966
- model: 'deepseek-v4-flash',
967
- } satisfies ProposerSpec,
968
- ]),
969
- {
970
- name: 'merge-author',
971
- profile: 'default-author.profile.json',
972
- harness: 'pi',
973
- model: 'glm-5.2',
974
- merge: true,
975
- },
976
- ]
977
- return {
978
- ...base,
979
- populationSize: proposers.length,
980
- proposers,
981
- paretoParents: [...GEN3_PARETO_PARENTS],
982
- }
983
- }
984
-
985
- // ---------------------------------------------------------------------------
986
- // GEN-5 configuration — gen-4's shape (4 proposers incl. the merge seat,
987
- // Pareto parents, premeasured baseline carried forward per the same
988
- // cell-derivation, 2 reps, deferred holdout) PLUS the gen-5 integration
989
- // bundle as a unit:
990
- //
991
- // 1. MAP+TOOLBOX briefing — per-run evidence index + toolbox/permission
992
- // briefing (change-space overridable at extensions/pi/author-briefing.md);
993
- // the 3-analyst diagnosis stays as ONE input among the named tools.
994
- // 2. PUBLIC/PRIVATE SPLIT — 4 public / 2 private of the 6 instances,
995
- // deterministically seeded by runId and persisted per outDir; proposers +
996
- // prefilter see public only, selection stays combined. Small-n caveat
997
- // documented in score-split.mts.
998
- // 3. ACTIVATION GATE — required machine-checkable predicate per candidate;
999
- // never-fired mechanisms are quarantined even on an improved score.
1000
- // 4. SETTLE-TIME ROLLOUT LEDGER — tangle.rollout.v1 lines live per cell,
1001
- // label v2 (contribution-aware workers, baseline-relative proposers).
1002
- // ---------------------------------------------------------------------------
1003
-
1004
- export function defaultGen5Config(
1005
- hh = DEFAULT_HH_SCRATCHPAD,
1006
- opts: { outDirName?: string; includeDeepseek?: boolean } = {},
1007
- ): OuterLoopConfig {
1008
- const base = defaultGen4Config(hh, {
1009
- outDirName: opts.outDirName ?? 'gen5',
1010
- ...(opts.includeDeepseek !== undefined ? { includeDeepseek: opts.includeDeepseek } : {}),
1011
- })
1012
- return {
1013
- ...base,
1014
- scoreSplit: { publicCount: 4 },
1015
- briefing: AUTHOR_BRIEFING_VERSION,
1016
- activationGate: true,
1017
- rolloutLedger: { enabled: true },
1018
- priorEvidenceDirs: [join(hh, 'gen4'), join(hh, 'gen3')],
1019
- }
1020
- }
1021
-
1022
- // ---------------------------------------------------------------------------
1023
- // Round recorder — dispatch-time change-space fail-closed + diff writing ONLY.
1024
- // NOT a scoring source: scoring reads the lib's campaign cells
1025
- // (cell-evidence.mts). The prior recorder role — accumulating per-instance
1026
- // results keyed by dispatch order — mislabeled a resumed run's baseline
1027
- // (r4-mroh3rkt: cached cells replay without dispatching, so "first dispatched
1028
- // surface" was a CANDIDATE and the summary published its cells as
1029
- // "baseline 0/3" while the measured baseline was 1/3).
1030
- // ---------------------------------------------------------------------------
1031
-
1032
- interface CandidateRecord {
1033
- surfaceKey: string
1034
- commit: string
1035
- baseCommit: string
1036
- tag: string
1037
- changedFiles: string[]
1038
- violations: string[]
1039
- diffPath: string | null
1040
- diffSha256: string | null
1041
- /** Dispatch-time forensics: which loops checkout ran the arm. Null for a
1042
- * candidate whose cells were all replayed from cache (never dispatched
1043
- * in this process). */
1044
- armProvenance: { repo: string; commit: string } | null
1045
- }
1046
-
1047
- class RoundRecorder {
1048
- readonly byKey = new Map<string, CandidateRecord>()
1049
- constructor(
1050
- private readonly loopsRepo: string,
1051
- private readonly candidatesDir: string,
1052
- ) {}
1053
-
1054
- byCommit(commit: string): CandidateRecord | undefined {
1055
- for (const rec of this.byKey.values()) if (rec.commit === commit) return rec
1056
- return undefined
1057
- }
1058
-
1059
- /** Describe a candidate surface: changed files, change-space violations, and
1060
- * the written diff. Idempotent and callable POST-RUN too (candidate commits
1061
- * survive in the loops object store after worktree cleanup), so resumed
1062
- * candidates that never dispatched here still get full staircase rows. */
1063
- async ensure(surface: CodeSurface): Promise<CandidateRecord> {
1064
- const key = surfaceHash(surface)
1065
- const existing = this.byKey.get(key)
1066
- if (existing) return existing
1067
- const names = await runOk('git', [
1068
- '-C', this.loopsRepo,
1069
- 'diff', '--name-only', surface.baseCommit, surface.candidateCommit,
1070
- ])
1071
- const changedFiles = names.stdout.split('\n').map((s) => s.trim()).filter(Boolean)
1072
- const violations = changeSpaceViolations(changedFiles)
1073
- const tag = surface.candidateCommit.slice(0, 10)
1074
- let diffPath: string | null = null
1075
- if (surface.candidateCommit !== surface.baseCommit) {
1076
- const diff = await runOk('git', ['-C', this.loopsRepo, 'diff', surface.baseCommit, surface.candidateCommit])
1077
- await mkdir(this.candidatesDir, { recursive: true })
1078
- diffPath = join(this.candidatesDir, `${tag}.patch`)
1079
- await writeFile(diffPath, diff.stdout)
1080
- }
1081
- const rec: CandidateRecord = {
1082
- surfaceKey: key,
1083
- commit: surface.candidateCommit,
1084
- baseCommit: surface.baseCommit,
1085
- tag,
1086
- changedFiles,
1087
- violations,
1088
- diffPath,
1089
- diffSha256: surface.patch.sha256,
1090
- armProvenance: null,
1091
- }
1092
- this.byKey.set(key, rec)
1093
- return rec
1094
- }
1095
- }
1096
-
1097
- // ---------------------------------------------------------------------------
1098
- // Eval worktrees — a candidate commit gets its own loops checkout so the
1099
- // candidate worktree managed by the improvement driver stays PRISTINE (its
1100
- // finalize-time verification rejects any extra file, node_modules included).
1101
- // ---------------------------------------------------------------------------
1102
-
1103
- export async function addEvalWorktree(
1104
- loopsRepo: string,
1105
- commit: string,
1106
- dest: string,
1107
- signal?: AbortSignal,
1108
- ): Promise<void> {
1109
- signal?.throwIfAborted()
1110
- await run('git', ['-C', loopsRepo, 'worktree', 'remove', '--force', '--', dest], { timeoutMs: 60_000, signal })
1111
- signal?.throwIfAborted()
1112
- await rm(dest, { recursive: true, force: true })
1113
- signal?.throwIfAborted()
1114
- await run('git', ['-C', loopsRepo, 'worktree', 'prune'], { timeoutMs: 60_000, signal })
1115
- signal?.throwIfAborted()
1116
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', dest, commit], { timeoutMs: 60_000, signal })
1117
- signal?.throwIfAborted()
1118
- // The loops driver needs deps; a worktree has none. Shared install is safe:
1119
- // arms never write into the loops checkout (state goes to ws/.loops + runDir).
1120
- await symlink(join(loopsRepo, 'node_modules'), join(dest, 'node_modules'), 'dir')
1121
- }
1122
-
1123
- export async function removeEvalWorktree(loopsRepo: string, dest: string): Promise<void> {
1124
- await unlink(join(dest, 'node_modules')).catch(() => {})
1125
- const res = await run('git', ['-C', loopsRepo, 'worktree', 'remove', '--force', '--', dest], { timeoutMs: 60_000 })
1126
- if (res.code !== 0) {
1127
- await rm(dest, { recursive: true, force: true })
1128
- await run('git', ['-C', loopsRepo, 'worktree', 'prune'], { timeoutMs: 60_000 })
1129
- }
1130
- }
1131
-
1132
- // ---------------------------------------------------------------------------
1133
- // The constrained proposer: agenticGenerator + change-space verifier + the
1134
- // round-4 task prompt. `improve(surface:'code')` requires the generator via
1135
- // `code.generator` so the runtime owns candidate-worktree cleanup.
1136
- // ---------------------------------------------------------------------------
1137
-
1138
- /** Mirrors the agentic generator's raw-trace evidence contract — the exact
1139
- * artifact path its gate checks for. */
1140
- export const RAW_TRACE_DIAGNOSIS_PATH = '.improve/raw-trace-diagnosis.md'
1141
-
1142
- function asSearchProposalFinding(finding: AnalystFinding): ProposalFinding {
1143
- return { ...finding, proposal_origin: 'search' }
1144
- }
1145
-
1146
- export function changeSpaceInstruction(space: ChangeSpace = LOOPS_CHANGE_SPACE): string {
1147
- return [
1148
- 'DECLARED CHANGE-SPACE (hard constraint, enforced by an automated gate):',
1149
- `- You may ONLY edit files under: ${space.prefixes.map((p) => `${p}**`).join(', ')}`,
1150
- `- and these exact files: ${space.files.join(', ')}`,
1151
- `- plus the diagnosis artifact ${RAW_TRACE_DIAGNOSIS_PATH}.`,
1152
- '- Everything else is IMMUTABLE for this experiment: the official judge, the per-instance verify scripts,',
1153
- ' task prompts, model ids, and budgets live outside your reach and candidates whose diff touches any',
1154
- ' other path are REJECTED before they are ever evaluated.',
1155
- ].join('\n')
1156
- }
1157
-
1158
- export function round4BuildPrompt(args: { findings: ReadonlyArray<ProposalFinding> }): string {
1159
- const lines: string[] = [
1160
- 'You are the optimizer of the "loops" pi SUPERVISOR — an agent that plans, spawns sandboxed coding',
1161
- 'workers, and settles a delivered patch for SWE-bench Verified instances (glm-5.2 in both seats, frozen).',
1162
- 'Round-3 state: the supervisor resolves 1/3 of its improvement set (matplotlib resolved; astropy + django',
1163
- 'deliver self-verify-passing patches the OFFICIAL maintainer test suite still rejects).',
1164
- '',
1165
- 'GOAL: raise the official resolved count on the improvement set WITHOUT raising cost/arm by more than 20%.',
1166
- 'Make the smallest coherent change to the supervisor implementation that addresses the diagnosis below,',
1167
- 'then stop. Do not commit — leave changes in the working tree.',
1168
- '',
1169
- changeSpaceInstruction(),
1170
- '',
1171
- 'Diagnosis findings (blind multi-analyst ensemble + raw-trace context):',
1172
- ]
1173
- for (const f of args.findings) {
1174
- const severity = typeof f.severity === 'string' ? f.severity : 'info'
1175
- const subject = typeof f.subject === 'string' ? ` [${f.subject}]` : ''
1176
- const claim = typeof f.claim === 'string' ? f.claim : JSON.stringify(f)
1177
- lines.push(`- (${severity})${subject} ${claim}`)
1178
- if (typeof f.recommended_action === 'string') lines.push(` → ${f.recommended_action}`)
1179
- }
1180
- const hasRawTrace = args.findings.some(
1181
- (f) => f.analyst_id === 'raw-trace-distiller' || f.area === 'raw-trace-context',
1182
- )
1183
- if (hasRawTrace) {
1184
- lines.push(
1185
- '',
1186
- 'Raw trace evidence requirement:',
1187
- '- Inspect at least one raw trace path named above before editing.',
1188
- `- Write ${RAW_TRACE_DIAGNOSIS_PATH} in this worktree.`,
1189
- '- Include the exact trace path(s) inspected, the failure mechanism, and the code change made.',
1190
- '- A candidate without this file, or with only this file changed, is discarded.',
1191
- )
1192
- }
1193
- return lines.join('\n')
1194
- }
1195
-
1196
- /** Purge gitignored artifacts from a candidate worktree with `git clean -Xdff`.
1197
- *
1198
- * The proposer agent may run a dependency install inside its worktree to
1199
- * verify its own change (measured: round-4 gen-0 cand-1 left a real pnpm
1200
- * `node_modules/` — 38k paths — after editing loops.ts). Ignored paths are
1201
- * invisible to the change-space check (`git status` honors .gitignore), but
1202
- * the improvement driver's finalize-time surface verification rejects ANY
1203
- * extra path, ignored included (`ls-files --others --ignored`), killing the
1204
- * whole run. `-X` deletes only ignored paths, so tracked edits and untracked
1205
- * non-ignored deliverables (e.g. .improve/raw-trace-diagnosis.md) survive;
1206
- * the doubled `-f` clears nested git dirs some packages ship. */
1207
- export async function purgeIgnoredArtifacts(
1208
- worktreePath: string,
1209
- signal?: AbortSignal,
1210
- ): Promise<void> {
1211
- await runOk('git', ['-C', worktreePath, 'clean', '-Xdff'], {
1212
- ...(signal ? { signal } : {}),
1213
- })
1214
- }
1215
-
1216
- /** Verifier run after each generator shot: ignored-dirt purge first (the
1217
- * finalize precondition), then change-space compliance (cheap,
1218
- * feedback-rich), then `tsc --noEmit` with the main repo's
1219
- * node_modules linked in TEMPORARILY (the link must not survive — the
1220
- * driver's finalize-time surface verification rejects any extra path). */
1221
- export function loopsCandidateVerifier(loopsRepo: string): Verifier {
1222
- return async (worktreePath: string, signal?: AbortSignal) => {
1223
- signal?.throwIfAborted()
1224
- await purgeIgnoredArtifacts(worktreePath, signal)
1225
- signal?.throwIfAborted()
1226
- const status = await runOk(
1227
- 'git',
1228
- ['-C', worktreePath, 'status', '--porcelain=v1', '--untracked-files=all'],
1229
- { ...(signal ? { signal } : {}) },
1230
- )
1231
- signal?.throwIfAborted()
1232
- const violations = changeSpaceViolations(porcelainChangedPaths(status.stdout))
1233
- if (violations.length > 0) {
1234
- return {
1235
- ok: false,
1236
- feedback:
1237
- `CHANGE-SPACE VIOLATION — these paths are outside the declared change-space:\n` +
1238
- violations.map((v) => ` - ${v}`).join('\n') +
1239
- `\n${changeSpaceInstruction()}\nRevert or relocate those edits (git checkout -- <path> / rm for untracked).`,
1240
- }
1241
- }
1242
- const nm = join(worktreePath, 'node_modules')
1243
- let linked = false
1244
- signal?.throwIfAborted()
1245
- if (!existsSync(nm)) {
1246
- await symlink(join(loopsRepo, 'node_modules'), nm, 'dir')
1247
- linked = true
1248
- }
1249
- try {
1250
- const tsc = join(loopsRepo, 'node_modules', '.bin', 'tsc')
1251
- const res = await run(tsc, ['--noEmit'], {
1252
- cwd: worktreePath,
1253
- timeoutMs: 300_000,
1254
- ...(signal ? { signal } : {}),
1255
- })
1256
- signal?.throwIfAborted()
1257
- if (res.code !== 0) {
1258
- return {
1259
- ok: false,
1260
- feedback: `tsc --noEmit failed (rc=${res.code}${res.timedOut ? ', timeout' : ''}):\n${(res.stdout + res.stderr).slice(0, 4000)}`,
1261
- }
1262
- }
1263
- return { ok: true }
1264
- } finally {
1265
- if (linked) await unlink(nm).catch(() => {})
1266
- }
1267
- }
1268
- }
1269
-
1270
- /** Ambient auth vars that hijack the claude CLI away from its claude.ai login.
1271
- * The run is launched under dotenvx, and agent-state.env injects an
1272
- * ANTHROPIC_API_KEY meant for other tooling; the claude CLI prefers env-key
1273
- * auth over the logged-in account and exits 1 immediately when that key's org
1274
- * is over its usage cap (reproduced 2026-07-20: `claude -p` under the run env
1275
- * → rc=1, "API Error: 400 You have reached your specified API usage limits";
1276
- * same command with these vars unset → rc=0). The author shot must run on the
1277
- * CLI's own login, so the leaked auth is stripped for the shot subprocess
1278
- * only — the rest of the run keeps its env untouched. */
1279
- const CLAUDE_AMBIENT_AUTH_VARS = ['ANTHROPIC_API_KEY', 'ANTHROPIC_AUTH_TOKEN', 'ANTHROPIC_BASE_URL'] as const
1280
-
1281
- /** Same failure class for the gen-4 codex seat: agent-state.env injects an
1282
- * OPENAI_API_KEY meant for other tooling, and the codex CLI prefers env-key
1283
- * auth over its ChatGPT login. The codex author shot must run on the CLI's
1284
- * own login (`codex login status` is provenance-gated at launch), so the
1285
- * leaked auth is stripped for the shot subprocess only. */
1286
- const CODEX_AMBIENT_AUTH_VARS = ['OPENAI_API_KEY', 'OPENAI_BASE_URL'] as const
1287
-
1288
- export function proposerShotEnv(harness: NonNullable<ProposerSpec['harness']>): NodeJS.ProcessEnv {
1289
- const env: NodeJS.ProcessEnv = { ...process.env }
1290
- if (harness === 'claude-code') {
1291
- for (const name of CLAUDE_AMBIENT_AUTH_VARS) delete env[name]
1292
- }
1293
- if (harness === 'codex') {
1294
- for (const name of CODEX_AMBIENT_AUTH_VARS) delete env[name]
1295
- }
1296
- return env
1297
- }
1298
-
1299
- function configuredAuthorProfile(config: OuterLoopConfig): AgentProfile {
1300
- const spec =
1301
- config.proposers?.find((proposer) => proposer.engine === undefined) ??
1302
- ({
1303
- name: 'single-author',
1304
- profile: config.proposerProfile,
1305
- harness: config.proposerHarness,
1306
- } satisfies ProposerSpec)
1307
- const profile = resolveAuthorProfile(spec)
1308
- if (!profile) throw new Error(`author ${spec.name}: an exact AgentProfile is required`)
1309
- return profile
1310
- }
1311
-
1312
- function authorExecutorForWorktree(worktreePath: string) {
1313
- const bridgeUrl = process.env.CLI_BRIDGE_URL ?? process.env.BRIDGE_URL
1314
- const bridgeBearer = process.env.CLI_BRIDGE_BEARER ?? process.env.BRIDGE_BEARER
1315
- if (!bridgeUrl || !bridgeBearer) {
1316
- throw new Error(
1317
- 'authoring requires CLI_BRIDGE_URL/BRIDGE_URL and CLI_BRIDGE_BEARER/BRIDGE_BEARER',
1318
- )
1319
- }
1320
- return {
1321
- backend: 'bridge' as const,
1322
- bridgeUrl,
1323
- bridgeBearer,
1324
- cwd: worktreePath,
1325
- }
1326
- }
1327
-
1328
- export function constrainedLoopsGenerator(config: OuterLoopConfig): CandidateGenerator {
1329
- const shotDir = join(config.outDir, 'proposer-shots')
1330
- const inner = agenticGenerator({
1331
- profile: configuredAuthorProfile(config),
1332
- executorForWorktree: authorExecutorForWorktree,
1333
- timeoutMs: config.proposerTimeoutMs,
1334
- buildPrompt: round4BuildPrompt,
1335
- verify: loopsCandidateVerifier(config.loopsRepo),
1336
- onShotCompleted: proposerShotHooks({
1337
- shotDir,
1338
- }),
1339
- })
1340
- return {
1341
- kind: `round4-constrained:${inner.kind}`,
1342
- proposesWithoutFindings: true,
1343
- generate: (args) => inner.generate(args),
1344
- }
1345
- }
1346
-
1347
- // ---------------------------------------------------------------------------
1348
- // runRound. (The evaluated R4Artifact type lives in cell-evidence.mts with
1349
- // the scoring that consumes it.)
1350
- // ---------------------------------------------------------------------------
1351
-
1352
- /** Ledger model id for the dockerized official judge's $0 receipts. */
1353
- export const OFFICIAL_JUDGE_MODEL = 'swe-bench-official-judge'
1354
-
1355
- function asCodeSurface(surface: MutableSurface): CodeSurface {
1356
- if (typeof surface !== 'object' || surface === null || surface.kind !== 'code') {
1357
- throw new Error('outer-loop: expected a CodeSurface (improve surface:"code" contract)')
1358
- }
1359
- return surface
1360
- }
1361
-
1362
- const log = (msg: string): void => console.log(`[${new Date().toISOString().slice(11, 19)}] ${msg}`)
1363
-
1364
- export async function runRound(config: OuterLoopConfig, signal?: AbortSignal): Promise<void> {
1365
- signal?.throwIfAborted()
1366
- assertFrozenArm(config.arm)
1367
- if (config.instances.length === 0) throw new Error('outer-loop: empty improvement set')
1368
- const overlap = config.instances.filter((i) => config.holdoutInstances.includes(i))
1369
- if (overlap.length > 0) {
1370
- throw new Error(`outer-loop: improvement set leaks into the pre-registered holdout: ${overlap.join(', ')}`)
1371
- }
1372
- const reps = config.repsPerInstance ?? 1
1373
- if (!Number.isInteger(reps) || reps < 1) {
1374
- throw new Error(`outer-loop: repsPerInstance must be a positive integer, got ${JSON.stringify(config.repsPerInstance)}`)
1375
- }
1376
- if (config.proposers !== undefined) {
1377
- if (config.proposers.length === 0) throw new Error('outer-loop: config.proposers must not be empty when set')
1378
- if (config.proposers.length !== config.populationSize) {
1379
- throw new Error(
1380
- `outer-loop: populationSize ${config.populationSize} != proposers.length ${config.proposers.length} — ` +
1381
- 'the fan-out assigns exactly one candidate slot per proposer',
1382
- )
1383
- }
1384
- }
1385
- // Stale-install guard: the resolved substrate must thread the passthroughs
1386
- // this run depends on. Fails loud — a silent drop would re-spend the
1387
- // premeasured baseline and pin the depth dial (see capabilities.mts).
1388
- assertSubstratePassthroughs(log)
1389
-
1390
- // GEN-4 model-identity provenance at t=0: harness CLI versions, the claude
1391
- // seat's resolved settings model, codex auth, and every explicit model pin.
1392
- // Fails loud on a missing/unauthed harness binary — populationSize equals
1393
- // proposers.length, so a dead seat cannot be skipped mid-run.
1394
- if (config.proposers !== undefined) {
1395
- const provenance = await captureProposerProvenance(config.proposers)
1396
- await mkdir(config.outDir, { recursive: true })
1397
- await writeFile(join(config.outDir, 'proposer-provenance.json'), JSON.stringify(provenance, null, 2))
1398
- for (const p of provenance.proposers) {
1399
- log(
1400
- `proposer ${p.name} (${p.harness}${p.merge ? ', merge seat' : ''}): ` +
1401
- `model=${p.pinnedModel ?? `cli-default${p.settingsModel ? `:${p.settingsModel}` : ''}`} ` +
1402
- `version=${p.harnessVersion.split('\n')[0]}`,
1403
- )
1404
- }
1405
- }
1406
-
1407
- // GEN-4 Pareto parents: materialize the configured prior-run frontier
1408
- // (commit existence + full diffs) before any authoring.
1409
- const paretoParents: ParetoParentContext[] =
1410
- config.paretoParents !== undefined && config.paretoParents.length > 0
1411
- ? await materializeParetoParents(config.loopsRepo, config.paretoParents)
1412
- : []
1413
- if (paretoParents.length > 0) {
1414
- log(`pareto parents: ${paretoParents.map((p) => `${p.label}@${p.commit.slice(0, 10)}`).join(', ')}`)
1415
- }
1416
-
1417
- // The gate's only denominator: a stored prior baseline campaign the LIB
1418
- // validates (surface hash, seed, reps, split digest, coverage) before
1419
- // skipping the baseline campaign. A missing artifact = the BOOTSTRAP run —
1420
- // the baseline is measured (cache-resumable) and the artifact written at
1421
- // the end of this run.
1422
- if (typeof config.premeasuredBaselinePath !== 'string' || config.premeasuredBaselinePath.length === 0) {
1423
- throw new Error('outer-loop: config.premeasuredBaselinePath is required (the bootstrap run writes the artifact there)')
1424
- }
1425
- let premeasured: PremeasuredOptimizationBaseline<R4Artifact, Scenario> | undefined
1426
- if (existsSync(config.premeasuredBaselinePath)) {
1427
- premeasured = JSON.parse(
1428
- await readFile(config.premeasuredBaselinePath, 'utf8'),
1429
- ) as PremeasuredOptimizationBaseline<R4Artifact, Scenario>
1430
- if (!premeasured || typeof premeasured.surfaceHash !== 'string' || !premeasured.campaign) {
1431
- throw new Error(`premeasuredBaselinePath: ${config.premeasuredBaselinePath} is not a {surfaceHash, campaign} record`)
1432
- }
1433
- log(`premeasured baseline: ${config.premeasuredBaselinePath} (surface ${premeasured.surfaceHash})`)
1434
- } else {
1435
- log(
1436
- `premeasured baseline artifact missing at ${config.premeasuredBaselinePath} — BOOTSTRAP run: ` +
1437
- 'the baseline campaign will be measured (cache-resumable) and the artifact written there for later runs',
1438
- )
1439
- }
1440
-
1441
- const secrets: SecretsEnv = { secretsDir: config.secretsDir, envFiles: config.envFiles }
1442
- const excludes = await loadExcludes()
1443
- const images = await loadInstanceImages(config.instanceImagesPath)
1444
- const adapter = createSweBenchAdapter()
1445
- const runId = `r${config.round}-${Date.now().toString(36)}`
1446
- await mkdir(config.outDir, { recursive: true })
1447
-
1448
- // GEN-5 public/private split — deterministic (seeded by runId), PERSISTED
1449
- // per outDir so a resume can never rotate private instances into view.
1450
- // Scored identically; selection stays combined; proposers + prefilter see
1451
- // public only.
1452
- const split: ScoreSplit | null =
1453
- config.scoreSplit !== undefined
1454
- ? await loadOrCreateScoreSplit({
1455
- outDir: config.outDir,
1456
- runId,
1457
- instances: config.instances,
1458
- publicCount: config.scoreSplit.publicCount,
1459
- })
1460
- : null
1461
- const privateIids = new Set(split?.privateInstances ?? [])
1462
- if (split !== null) {
1463
- log(
1464
- `score split (seeded by ${split.seededBy}): public [${split.publicInstances.join(', ')}] + ` +
1465
- `${split.privateInstances.length} private instance(s) (identities withheld from proposers; ` +
1466
- `selection uses public+private combined; small-n caveat: 2 private of 6 is a direction check, not certification)`,
1467
- )
1468
- }
1469
-
1470
- // The pre-filter's smoke instance may sit outside the improvement set (e.g.
1471
- // a designated cheap instance) — it needs the same problem/image/verify
1472
- // validation and rides the same loaded-task map. Under the gen-5 split the
1473
- // smoke choice is restricted to PUBLIC instances (the prefilter surfaces
1474
- // its verdict to the kill log the authors can mine).
1475
- const smokeIid =
1476
- config.proposers !== undefined && config.prefilter?.enabled
1477
- ? resolveSmokeInstance(
1478
- config.prefilter.smokeInstance,
1479
- split !== null ? split.publicInstances : config.instances,
1480
- premeasured ? cellsFromCampaign(premeasured.campaign) : null,
1481
- )
1482
- : null
1483
- const taskIds = [...new Set([...config.instances, ...(smokeIid !== null ? [smokeIid] : [])])]
1484
- const tasks = await adapter.loadTasks({ ids: taskIds, split: 'test' })
1485
- const problemById = new Map<string, string>()
1486
- for (const iid of taskIds) {
1487
- const task = tasks.find((t) => t.id === iid)
1488
- if (!task) throw new Error(`outer-loop: ${iid} not found in SWE-bench_Verified`)
1489
- const problem = String(task.metadata?.problem_statement ?? '')
1490
- if (!problem) throw new Error(`outer-loop: ${iid} has an empty problem_statement`)
1491
- if (!images[iid]) throw new Error(`outer-loop: ${iid} has no image mapping`)
1492
- const verifyScript = join(config.verifyDir, `${iid}.sh`)
1493
- if (!existsSync(verifyScript)) throw new Error(`outer-loop: missing verify script ${verifyScript}`)
1494
- problemById.set(iid, problem)
1495
- }
1496
- if (smokeIid !== null) log(`prefilter smoke instance: ${smokeIid}`)
1497
-
1498
- const judge: SerializedJudge = createSerializedJudge(
1499
- config.judgeTimeoutMs !== undefined ? { timeoutMs: config.judgeTimeoutMs } : {},
1500
- )
1501
- await mkdir(config.roundsDir, { recursive: true })
1502
- const recorder = new RoundRecorder(config.loopsRepo, join(config.outDir, 'candidates'))
1503
- const analysts: AnalystSpec[] = config.analystModels.map((model, i) => ({ id: `${model}#${i + 1}`, model }))
1504
-
1505
- // GEN-5 MAP+TOOLBOX briefing: persist the Pareto parent diffs, write the
1506
- // per-run evidence index (a map — one line per evidence path, private
1507
- // instances excluded), and resolve the briefing text (the change-space
1508
- // override at extensions/pi/author-briefing.md wins over the default).
1509
- let briefingCtx: BriefingContext | undefined
1510
- if (config.briefing === AUTHOR_BRIEFING_VERSION) {
1511
- const parentPatches: Array<{ label: string; path: string }> = []
1512
- if (paretoParents.length > 0) {
1513
- const parentsDir = join(config.outDir, 'pareto-parents')
1514
- await mkdir(parentsDir, { recursive: true })
1515
- for (const parent of paretoParents) {
1516
- const patchPath = join(parentsDir, `${parent.label}.patch`)
1517
- await writeFile(patchPath, parent.diff)
1518
- parentPatches.push({ label: parent.label, path: patchPath })
1519
- }
1520
- }
1521
- const index = await writeEvidenceIndex({
1522
- outDir: config.outDir,
1523
- roundsDir: config.roundsDir,
1524
- seedArtifactRuns: config.seedArtifactRuns,
1525
- ...(config.priorEvidenceDirs !== undefined ? { priorEvidenceDirs: config.priorEvidenceDirs } : {}),
1526
- paretoParentPatches: parentPatches,
1527
- split,
1528
- })
1529
- const briefing = await resolveAuthorBriefing(config.loopsRepo, config.loopsBaseRef)
1530
- briefingCtx = { indexPath: index.path, briefingText: briefing.text, briefingSource: briefing.source }
1531
- log(
1532
- `briefing ${AUTHOR_BRIEFING_VERSION}: evidence index → ${index.path} (${index.rows.length} row(s)); ` +
1533
- `briefing text source: ${briefing.source}`,
1534
- )
1535
- }
1536
-
1537
- // GEN-5 settle-time rollout ledger — tangle.rollout.v1 lines appended live
1538
- // after each cell judges (label v2); capture failure logs loud but never
1539
- // kills a cell.
1540
- const settleCapture: SettleCapture | null =
1541
- config.rolloutLedger?.enabled === true
1542
- ? createSettleCapture({
1543
- ledgerPath: config.rolloutLedger.path ?? join(config.outDir, 'rollout-ledger.jsonl'),
1544
- runId,
1545
- instanceCount: config.instances.length,
1546
- ...(config.rolloutLedger.opencodeDb !== undefined ? { opencodeDb: config.rolloutLedger.opencodeDb } : {}),
1547
- log,
1548
- })
1549
- : null
1550
- if (settleCapture !== null) log(`rollout-ledger: settle-time capture ON → ${settleCapture.path}`)
1551
-
1552
- const sweScenarios: Scenario[] = config.instances.map((iid) => ({ id: iid, kind: 'swe-instance' }))
1553
-
1554
- // Capacity gates on BOTH paths the supervisor arm rides (worker + router).
1555
- // Shared by every arm dispatch — the improvement cells AND the pre-filter
1556
- // smoke cell. A cell's WORK clock (config.dispatchTimeoutMs) starts only
1557
- // after these clear — a capacity hold is never billed to the work budget.
1558
- const awaitGates = async (gateSignal: AbortSignal | undefined = signal): Promise<void> => {
1559
- for (const gate of gatesForArmKind('supervisor', secrets, {
1560
- ...(config.gateWaitCeilingMs !== undefined ? { waitCeilingMs: config.gateWaitCeilingMs } : {}),
1561
- ...(config.capacityModel !== undefined ? { model: config.capacityModel } : {}),
1562
- onStatus: log,
1563
- })) {
1564
- if (!(await waitForCapacity(gate, gateSignal))) throw new Error(`no capacity on ${gate.name} within ceiling`)
1565
- }
1566
- }
1567
-
1568
- // ── the pre-filter smoke runner: ONE supervisor arm cell + official judge
1569
- // on the smoke instance, run against the proposer's scratch worktree BEFORE
1570
- // any full-evaluation spend. A crashed smoke KILLS the candidate (recorded
1571
- // in the kill reason) rather than the round — the pre-filter is allowed to
1572
- // be strict; a survivor still faces the full gate. ────────────────────
1573
- const smokeRunner: SmokeRunner | undefined =
1574
- smokeIid === null
1575
- ? undefined
1576
- : async ({
1577
- scratchPath,
1578
- generation,
1579
- proposer,
1580
- evaluationKey,
1581
- costLedger,
1582
- }): Promise<SmokeVerdict> => {
1583
- const iid = smokeIid
1584
- if (!/^[a-zA-Z0-9_-]+$/.test(evaluationKey)) {
1585
- throw new Error(`outer-loop: invalid smoke evaluation key ${JSON.stringify(evaluationKey)}`)
1586
- }
1587
- const requireResolved = config.prefilter?.requireResolved === true
1588
- const entry = images[iid]!
1589
- const armOutDir = join(
1590
- config.outDir,
1591
- 'prefilter-smoke',
1592
- `gen${generation}-${proposer.name}-${evaluationKey}`,
1593
- )
1594
- const nm = join(scratchPath, 'node_modules')
1595
- let linked = false
1596
- const t0 = Date.now()
1597
- try {
1598
- if (!existsSync(nm)) {
1599
- await symlink(join(config.loopsRepo, 'node_modules'), nm, 'dir')
1600
- linked = true
1601
- }
1602
- const work = async (signal: AbortSignal): Promise<{ armRes: SupervisorArmResult; resolved: boolean | null }> => {
1603
- const spec: SupervisorArmSpec = {
1604
- kind: 'supervisor',
1605
- name: config.armName,
1606
- workerModel: config.arm.workerModel,
1607
- driverModel: config.arm.driverModel,
1608
- budget: config.arm.budget,
1609
- maxSandboxes: config.arm.maxSandboxes,
1610
- maxUsd: config.arm.maxUsd,
1611
- maxDepth: config.arm.maxDepth,
1612
- ...(config.arm.envKnobs ? { envKnobs: config.arm.envKnobs } : {}),
1613
- loopsRepo: scratchPath,
1614
- extensionPath: join(scratchPath, 'extensions', 'pi', 'loops.ts'),
1615
- timeoutMs: config.arm.timeoutMs,
1616
- }
1617
- log(`>>> prefilter smoke ${proposer.name} ${iid} gen=${generation}`)
1618
- const armRes = await runSupervisorArm(spec, {
1619
- instanceId: iid,
1620
- image: entry.image,
1621
- baseCommit: entry.base_commit,
1622
- problemStatement: problemById.get(iid)!,
1623
- verifyCmd: `bash ${join(config.verifyDir, `${iid}.sh`)}`,
1624
- outDir: armOutDir,
1625
- secrets,
1626
- excludes,
1627
- signal,
1628
- })
1629
- if (signal.aborted) throw signal.reason
1630
- const verdict = await judge.judge(
1631
- iid,
1632
- armRes.patchPath,
1633
- `prefilter-g${generation}-${proposer.name}`,
1634
- signal,
1635
- )
1636
- return { armRes, resolved: verdict.resolved }
1637
- }
1638
- const runWork = (): Promise<{ armRes: SupervisorArmResult; resolved: boolean | null }> =>
1639
- runWithPostGateClock({
1640
- awaitGates,
1641
- work,
1642
- timeoutMs: config.dispatchTimeoutMs,
1643
- label: `prefilter smoke ${proposer.name} ${iid}`,
1644
- signal,
1645
- })
1646
- let outcome: { armRes: SupervisorArmResult; resolved: boolean | null }
1647
- if (costLedger) {
1648
- // The smoke's real arm spend reaches the run ledger like any cell.
1649
- const paid = await costLedger.runPaidCall({
1650
- channel: 'agent',
1651
- phase: 'search.prefilter',
1652
- actor: `prefilter-smoke:${iid}:g${generation}:${proposer.name}:${evaluationKey}`,
1653
- model: config.arm.workerModel,
1654
- execute: runWork,
1655
- receipt: ({ armRes }) => {
1656
- const spend = armRes.recoveredSpend
1657
- const usageKnown = (spend?.workerTokIn ?? null) !== null || (spend?.workerTokOut ?? null) !== null
1658
- return {
1659
- model: config.arm.workerModel,
1660
- inputTokens: spend?.workerTokIn ?? 0,
1661
- outputTokens: spend?.workerTokOut ?? 0,
1662
- ...(usageKnown ? {} : { usageUnknown: true }),
1663
- ...(armRes.spentUsd !== null ? { actualCostUsd: armRes.spentUsd } : {}),
1664
- }
1665
- },
1666
- })
1667
- if (!paid.succeeded) throw paid.error
1668
- outcome = paid.value
1669
- } else {
1670
- outcome = await runWork()
1671
- }
1672
- const wallS = Math.round((Date.now() - t0) / 1000)
1673
- const patchDelivered = outcome.armRes.patch_lines > 0
1674
- const conclusive = outcome.resolved !== null
1675
- const pass = requireResolved ? outcome.resolved === true : patchDelivered && conclusive
1676
- const verdictLine =
1677
- `smoke ${iid}: resolved=${outcome.resolved} patch_lines=${outcome.armRes.patch_lines} ` +
1678
- `verify_pass=${outcome.armRes.verify_pass} wall_s=${outcome.armRes.wall_s}`
1679
- const result: SmokeVerdict = {
1680
- iid,
1681
- pass,
1682
- reason: pass
1683
- ? verdictLine
1684
- : `${verdictLine} — below the ${requireResolved ? 'resolved' : 'mechanism (patch + conclusive judge)'} bar`,
1685
- resolved: outcome.resolved,
1686
- patchLines: outcome.armRes.patch_lines,
1687
- wallS,
1688
- // The GEPA seat's inner-score tiebreak.
1689
- verifyPass: outcome.armRes.verify_pass,
1690
- }
1691
- await mkdir(armOutDir, { recursive: true })
1692
- await writeFile(join(armOutDir, 'smoke.json'), JSON.stringify(result, null, 2))
1693
- return result
1694
- } catch (cause) {
1695
- if (signal?.aborted) throw signal.reason
1696
- const wallS = Math.round((Date.now() - t0) / 1000)
1697
- const result: SmokeVerdict = {
1698
- iid,
1699
- pass: false,
1700
- reason: `smoke errored: ${(cause as Error).message}`,
1701
- resolved: null,
1702
- patchLines: 0,
1703
- wallS,
1704
- }
1705
- await mkdir(armOutDir, { recursive: true })
1706
- await writeFile(join(armOutDir, 'smoke.json'), JSON.stringify(result, null, 2)).catch(() => {})
1707
- return result
1708
- } finally {
1709
- if (linked) await unlink(nm).catch(() => {})
1710
- }
1711
- }
1712
-
1713
- let runnerImplementationRef: string | undefined
1714
- let judgeImplementationRef: string | undefined
1715
- const hasGepaSeat = config.proposers?.some((proposer) => proposer.engine !== undefined) === true
1716
- if (hasGepaSeat && smokeIid !== null && smokeRunner !== undefined) {
1717
- const runtimeRoot = fileURLToPath(new URL('../../..', import.meta.url))
1718
- const benchRoot = fileURLToPath(new URL('../..', import.meta.url))
1719
- const sourceCommit = (
1720
- await runOk('git', ['-C', config.loopsRepo, 'rev-parse', 'HEAD'])
1721
- ).stdout.trim()
1722
- const entry = images[smokeIid]!
1723
- const verifyScript = await readFile(join(config.verifyDir, `${smokeIid}.sh`), 'utf8')
1724
- const implementationArtifacts = {
1725
- benchSource: await fileTreeImplementationRef(join(benchRoot, 'src')),
1726
- runtimeSource: await fileTreeImplementationRef(join(runtimeRoot, 'src')),
1727
- runtimeDist: await fileTreeImplementationRef(join(runtimeRoot, 'dist')),
1728
- packageFiles: canonicalCandidateDigest({
1729
- runtimePackage: await readFile(join(runtimeRoot, 'package.json'), 'utf8'),
1730
- benchPackage: await readFile(join(benchRoot, 'package.json'), 'utf8'),
1731
- lockfile: await readFile(join(runtimeRoot, 'pnpm-lock.yaml'), 'utf8'),
1732
- }),
1733
- swebench: await pythonDistributionImplementationRef(
1734
- 'swebench',
1735
- (script, args) => runVenvPython(script, args),
1736
- ),
1737
- }
1738
- const runnerSourceRef = canonicalCandidateDigest({
1739
- benchSource: implementationArtifacts.benchSource,
1740
- runtimeSource: implementationArtifacts.runtimeSource,
1741
- runtimeDist: implementationArtifacts.runtimeDist,
1742
- packageFiles: implementationArtifacts.packageFiles,
1743
- })
1744
- const judgeSourceRef = canonicalCandidateDigest({
1745
- benchSource: implementationArtifacts.benchSource,
1746
- packageFiles: implementationArtifacts.packageFiles,
1747
- swebench: implementationArtifacts.swebench,
1748
- })
1749
- runnerImplementationRef = canonicalCandidateDigest({
1750
- implementation: 'swe-arena-smoke-runner',
1751
- sourceCommit,
1752
- sourceRef: runnerSourceRef,
1753
- smoke: {
1754
- iid: smokeIid,
1755
- image: entry.image,
1756
- baseCommit: entry.base_commit,
1757
- problemStatement: problemById.get(smokeIid)!,
1758
- verifyScript,
1759
- requireResolved: config.prefilter?.requireResolved === true,
1760
- },
1761
- arm: {
1762
- name: config.armName,
1763
- workerModel: config.arm.workerModel,
1764
- driverModel: config.arm.driverModel,
1765
- budget: config.arm.budget,
1766
- maxSandboxes: config.arm.maxSandboxes,
1767
- maxUsd: config.arm.maxUsd,
1768
- maxDepth: config.arm.maxDepth,
1769
- timeoutMs: config.arm.timeoutMs,
1770
- envKnobs: config.arm.envKnobs ?? null,
1771
- },
1772
- dispatchTimeoutMs: config.dispatchTimeoutMs,
1773
- capacityModel: config.capacityModel ?? null,
1774
- gateWaitCeilingMs: config.gateWaitCeilingMs ?? null,
1775
- excludes,
1776
- })
1777
- judgeImplementationRef = canonicalCandidateDigest({
1778
- implementation: 'serialized-swebench-judge',
1779
- sourceCommit,
1780
- sourceRef: judgeSourceRef,
1781
- timeoutMs: config.judgeTimeoutMs ?? JUDGE_TIMEOUT_FLOOR_MS,
1782
- cacheLevel: 'instance',
1783
- })
1784
- }
1785
-
1786
- // ── dispatch: one (surface × scenario) cell ──────────────────────────
1787
- const agent = async (surface: MutableSurface, scenario: Scenario, ctx: DispatchContext): Promise<R4Artifact> => {
1788
- const cs = asCodeSurface(surface)
1789
- const rec = await recorder.ensure(cs)
1790
-
1791
- const iid = scenario.id
1792
- // FAIL-CLOSED change-space enforcement: an out-of-space candidate must
1793
- // never reach a model token or a docker container. The thrown cell is the
1794
- // record (the lib stores it with `error` set — no side bookkeeping).
1795
- if (rec.violations.length > 0) {
1796
- throw new Error(`change-space violation (${rec.violations.length} path(s)): ${rec.violations.join(', ')}`)
1797
- }
1798
-
1799
- const runCell = async (signal: AbortSignal): Promise<R4Artifact> => {
1800
- const entry = images[iid]!
1801
- const evalWt = join(config.outDir, 'eval-wt', `${rec.tag}-${iid}-r${ctx.rep}`)
1802
- const armOutDir = join(config.outDir, 'arm-runs', rec.tag, `rep-${ctx.rep}`)
1803
- try {
1804
- await addEvalWorktree(config.loopsRepo, cs.candidateCommit, evalWt, signal)
1805
- const spec: SupervisorArmSpec = {
1806
- kind: 'supervisor',
1807
- name: config.armName,
1808
- workerModel: config.arm.workerModel,
1809
- driverModel: config.arm.driverModel,
1810
- budget: config.arm.budget,
1811
- maxSandboxes: config.arm.maxSandboxes,
1812
- maxUsd: config.arm.maxUsd,
1813
- maxDepth: config.arm.maxDepth,
1814
- ...(config.arm.envKnobs ? { envKnobs: config.arm.envKnobs } : {}),
1815
- loopsRepo: evalWt,
1816
- extensionPath: join(evalWt, 'extensions', 'pi', 'loops.ts'),
1817
- timeoutMs: config.arm.timeoutMs,
1818
- }
1819
- log(`>>> ${config.armName} ${rec.tag} ${iid} rep=${ctx.rep}`)
1820
- const armRes: SupervisorArmResult = await runSupervisorArm(spec, {
1821
- instanceId: iid,
1822
- image: entry.image,
1823
- baseCommit: entry.base_commit,
1824
- problemStatement: problemById.get(iid)!,
1825
- verifyCmd: `bash ${join(config.verifyDir, `${iid}.sh`)}`,
1826
- outDir: armOutDir,
1827
- secrets,
1828
- excludes,
1829
- signal,
1830
- })
1831
- if (signal.aborted) throw signal.reason
1832
- const runDir = join(armOutDir, 'runs', iid, config.armName)
1833
- const { ws: _ws, ...armSummary } = armRes
1834
- await writeFile(join(runDir, 'result.json'), JSON.stringify(armSummary, null, 1))
1835
-
1836
- // The official judge is a docker test-suite run — real wall time, zero
1837
- // LLM spend. Its OWN paid call (channel 'judge', $0 actual) keeps the
1838
- // run's spend tree attributing judge work per cell without inventing a
1839
- // token cost; the wall lands on the artifact + judge.json.
1840
- const judgeT0 = Date.now()
1841
- const judgePaid = await ctx.cost.runPaidCall({
1842
- channel: 'judge',
1843
- actor: `official-judge:${iid}#r${ctx.rep}`,
1844
- model: OFFICIAL_JUDGE_MODEL,
1845
- execute: () => {
1846
- if (signal.aborted) throw signal.reason
1847
- return judge.judge(iid, armRes.patchPath, `${config.armName}-${rec.tag}`, signal)
1848
- },
1849
- receipt: () => ({ model: OFFICIAL_JUDGE_MODEL, inputTokens: 0, outputTokens: 0, actualCostUsd: 0 }),
1850
- })
1851
- if (!judgePaid.succeeded) throw judgePaid.error
1852
- const verdict = judgePaid.value
1853
- const judgeWallS = Math.round((Date.now() - judgeT0) / 1000)
1854
- await writeFile(join(runDir, 'judge.json'), JSON.stringify({ ...verdict, wallS: judgeWallS }, null, 1))
1855
- log(`${config.armName} ${rec.tag} ${iid} judged: resolved=${verdict.resolved} (attempts=${verdict.attempts})`)
1856
-
1857
- // Deterministic run observability, per cell: steer count, waves, concurrency,
1858
- // idle, evidence→respawn, cost by role. The headline lands in the run log so
1859
- // the answers are in the tail without a follow-up command.
1860
- await writeSupervisorRunReportSafe(runDir, {
1861
- appendHeadlineTo: join(config.outDir, 'run.log'),
1862
- patchPath: armRes.patchPath,
1863
- })
1864
-
1865
- const spend = armRes.recoveredSpend
1866
- const recovered =
1867
- armRes.spentTokens === null && (spend?.workerTokSqlite ?? null) === null
1868
- ? null
1869
- : (armRes.spentTokens ?? 0) + (spend?.workerTokSqlite ?? 0)
1870
- rec.armProvenance = { repo: armRes.provenance.repo, commit: armRes.provenance.commit }
1871
- await appendFile(
1872
- join(config.outDir, 'progress.jsonl'),
1873
- JSON.stringify({
1874
- at: new Date().toISOString(),
1875
- runId,
1876
- candidate: rec.tag,
1877
- iid,
1878
- rep: ctx.rep,
1879
- runDir,
1880
- resolved: verdict.resolved,
1881
- verify_pass: armRes.verify_pass,
1882
- wall_s: armRes.wall_s,
1883
- spentTokens: armRes.spentTokens,
1884
- spentUsd: armRes.spentUsd,
1885
- recoveredTokens: recovered,
1886
- }) + '\n',
1887
- )
1888
- const summaryPath = await ctx.artifacts.writeJson('arm-summary.json', {
1889
- runDir,
1890
- patchPath: armRes.patchPath,
1891
- verdict,
1892
- })
1893
-
1894
- // GEN-5 settle-time rollout capture: emit supervisor + worker lines
1895
- // NOW, while the opencode store still holds the worker transcripts.
1896
- // Attribution comes from the campaign cell path (never dispatch
1897
- // order); a capture failure logs loud but never kills the cell.
1898
- if (settleCapture !== null) {
1899
- try {
1900
- const coords = campaignCoordsFromCellPath(summaryPath)
1901
- if (coords === null) {
1902
- log(`rollout-ledger: cannot derive campaign coords from ${summaryPath} — cell ${iid} r${ctx.rep} skipped`)
1903
- } else {
1904
- const supRunDir = await findSupervisorRunDir(armRes.ws)
1905
- const deliveredPatch = await readFile(armRes.patchPath, 'utf8').catch(() => '')
1906
- await settleCapture.captureCell({
1907
- generation: coords.generation,
1908
- candidateIndex: coords.candidateIndex,
1909
- iid,
1910
- rep: ctx.rep,
1911
- seed: ctx.seed,
1912
- splitVisibility: split === null ? null : privateIids.has(iid) ? 'private' : 'public',
1913
- commit: cs.candidateCommit,
1914
- resolved: verdict.resolved,
1915
- judgeVerdict: { ...verdict, wallS: judgeWallS },
1916
- runDir,
1917
- patchPath: armRes.patchPath,
1918
- supRunDir,
1919
- deliveredPatch,
1920
- workerModel: config.arm.workerModel,
1921
- metrics: {
1922
- resolved: verdict.resolved,
1923
- verify_pass: armRes.verify_pass,
1924
- patch_lines: armRes.patch_lines,
1925
- judge_attempts: verdict.attempts ?? null,
1926
- judge_wall_s: judgeWallS,
1927
- spent_tokens: armRes.spentTokens,
1928
- spent_usd: armRes.spentUsd,
1929
- recovered_tokens: recovered,
1930
- sup_status: armRes.sup_status,
1931
- sup_verdict: armRes.sup_verdict,
1932
- spawned: armRes.spawned,
1933
- workers: armRes.workers,
1934
- settled: armRes.settled,
1935
- },
1936
- cost: { usd: armRes.spentUsd, wallS: armRes.wall_s, spentTokens: armRes.spentTokens },
1937
- })
1938
- }
1939
- } catch (cause) {
1940
- log(`rollout-ledger: settle-time capture FAILED for ${iid} r${ctx.rep}: ${(cause as Error).message}`)
1941
- }
1942
- }
1943
-
1944
- if (verdict.resolved === null) {
1945
- // Inconclusive judge (double flake / infra) — the cell must FAIL, not
1946
- // score a fabricated boolean; the candidate becomes coverage-incomplete.
1947
- throw new Error(`inconclusive judge verdict for ${iid} (${verdict.error ?? 'unknown'})`)
1948
- }
1949
- return {
1950
- kind: 'swe-arm',
1951
- iid,
1952
- commit: cs.candidateCommit,
1953
- resolved: verdict.resolved,
1954
- verifyPass: armRes.verify_pass,
1955
- patchLines: armRes.patch_lines,
1956
- wallS: armRes.wall_s,
1957
- spentTokens: armRes.spentTokens,
1958
- spentUsd: armRes.spentUsd,
1959
- recoveredTokens: recovered,
1960
- workerTokIn: spend?.workerTokIn ?? null,
1961
- workerTokOut: spend?.workerTokOut ?? null,
1962
- judgeAttempts: verdict.attempts ?? null,
1963
- judgeWallS,
1964
- runDir,
1965
- patchPath: armRes.patchPath,
1966
- }
1967
- } finally {
1968
- await removeEvalWorktree(config.loopsRepo, evalWt)
1969
- }
1970
- }
1971
-
1972
- // The arm's real spend reaches the LIB's CostLedger here: one agent-channel
1973
- // paid call per cell whose receipt carries the recovered worker-session
1974
- // token split (opencode sqlite join) and the runtime spend-tree dollars
1975
- // (state.json spentUsd). run-campaign commits it into cell.costUsd /
1976
- // cell.tokenUsage + durable cost-ledger.jsonl receipts — the stub/$0
1977
- // rounds this replaces.
1978
- const paid = await ctx.cost.runPaidCall<R4Artifact>({
1979
- actor: `${config.armName}:${rec.tag}:${iid}#r${ctx.rep}`,
1980
- model: config.arm.workerModel,
1981
- execute: () =>
1982
- runWithPostGateClock({
1983
- awaitGates,
1984
- work: runCell,
1985
- timeoutMs: config.dispatchTimeoutMs,
1986
- label: `${config.armName} ${rec.tag} ${iid} r${ctx.rep}`,
1987
- signal,
1988
- }),
1989
- receipt: (artifact) => {
1990
- if (artifact.kind !== 'swe-arm') throw new Error('swe cell produced a non-arm artifact')
1991
- const usageKnown = artifact.workerTokIn !== null || artifact.workerTokOut !== null
1992
- return {
1993
- model: config.arm.workerModel,
1994
- inputTokens: artifact.workerTokIn ?? 0,
1995
- outputTokens: artifact.workerTokOut ?? 0,
1996
- ...(usageKnown ? {} : { usageUnknown: true }),
1997
- // The runtime spend-tree usd is the measured bill; without it the
1998
- // receipt stays honestly unpriced (costUnknown) rather than $0.
1999
- ...(artifact.spentUsd !== null ? { actualCostUsd: artifact.spentUsd } : {}),
2000
- }
2001
- },
2002
- })
2003
- if (!paid.succeeded) throw paid.error
2004
- return paid.value
2005
- }
2006
-
2007
- // ── judge config: a deterministic READ of the official verdict the dispatch
2008
- // already obtained under the serialized-judge lock. ───────────────────
2009
- const judgeConfig: JudgeConfig<R4Artifact, Scenario> = {
2010
- name: 'swe-arena-official-judge',
2011
- dimensions: [{ key: 'resolved', description: 'official SWE-bench judge verdict' }],
2012
- score: ({ artifact }) => {
2013
- const v = artifact.resolved ? 1 : 0
2014
- return {
2015
- composite: v,
2016
- dimensions: { resolved: v },
2017
- notes: `official judge: ${artifact.iid} resolved=${artifact.resolved} (verify_pass=${artifact.verifyPass}, patch_lines=${artifact.patchLines}, wall_s=${artifact.wallS})`,
2018
- }
2019
- },
2020
- }
2021
-
2022
- // ── diagnosis at the analyzeGeneration seam ──────────────────────────
2023
- const rawTrace = rawTraceDistiller<Scenario, R4Artifact>({ fallbackFindings: [] })
2024
- const steeringFinding = makeProposalFinding({
2025
- analyst_id: 'round4-protocol',
2026
- severity: 'high',
2027
- area: 'constraint',
2028
- confidence: 1,
2029
- claim:
2030
- 'Declared change-space: ONLY extensions/pi/** and src/{worker-evidence,best-effort,worker-clone}.ts may change ' +
2031
- '(plus the .improve/ diagnosis artifact). Judge, verify scripts, task prompts, model ids and budgets are immutable.',
2032
- recommended_action: 'Keep every edit inside the change-space; out-of-space candidate diffs are rejected before evaluation.',
2033
- evidence_refs: [],
2034
- proposal_origin: 'search',
2035
- })
2036
- const analyzeGeneration: NonNullable<
2037
- ImproveCodeRunOptions<Scenario, R4Artifact>['analyzeGeneration']
2038
- > = async (input): Promise<ProposalFinding[]> => {
2039
- signal?.throwIfAborted()
2040
- const runs: SupRunArtifacts[] = []
2041
- if (input.generation === -1) {
2042
- for (const seed of config.seedArtifactRuns) {
2043
- if (privateIids.has(seed.iid)) continue // gen-5 split: never surfaced to proposers
2044
- if (!existsSync(seed.dir)) {
2045
- // A wiped scratchpad (host reboot) must not feed EMPTY bundles to the
2046
- // analysts as if they were real artifacts — skip loudly.
2047
- log(`seed artifact dir missing — skipped from diagnosis: ${seed.dir}`)
2048
- continue
2049
- }
2050
- runs.push({
2051
- iid: seed.iid,
2052
- arm: seed.arm,
2053
- dir: seed.dir,
2054
- ...(seed.patchPath ? { patchPath: seed.patchPath } : {}),
2055
- judge: { resolved: seed.resolved, note: 'previous round (seeded)' },
2056
- })
2057
- }
2058
- }
2059
- // Candidate failure artifacts come from the LIB's campaign cells (the
2060
- // artifacts name their own runDir/patch) — resume-replayed cells included,
2061
- // which the old recorder-based lookup silently dropped.
2062
- const worstFirst = [...input.candidates]
2063
- .sort((a, b) => (a.composite ?? -Infinity) - (b.composite ?? -Infinity))
2064
- .slice(0, 4)
2065
- for (const cand of worstFirst) {
2066
- const cells = cellsFromCampaign(cand.campaign)
2067
- for (const cell of cells) {
2068
- const a = cell.artifact
2069
- if (a === null || a.kind !== 'swe-arm' || !a.runDir) continue
2070
- if (privateIids.has(a.iid)) continue // gen-5 split: never surfaced to proposers
2071
- runs.push({
2072
- iid: a.iid,
2073
- arm: config.armName,
2074
- dir: a.runDir,
2075
- ...(a.patchPath ? { patchPath: a.patchPath } : {}),
2076
- judge: { resolved: cell.error ? null : a.resolved },
2077
- })
2078
- }
2079
- }
2080
- let ensembleFindings: ProposalFinding[] = []
2081
- if (runs.length > 0) {
2082
- try {
2083
- const scratch = join(config.outDir, 'diagnosis', `gen-${input.generation}`)
2084
- const ensemble = await runDiagnosisEnsemble({
2085
- analysts,
2086
- runs,
2087
- secrets,
2088
- scratchDir: scratch,
2089
- onStatus: log,
2090
- signal,
2091
- })
2092
- signal?.throwIfAborted()
2093
- await writeFile(
2094
- join(config.outDir, 'diagnosis', `gen-${input.generation}.json`),
2095
- JSON.stringify({ reports: ensemble.reports, fused: ensemble.fused }, null, 2),
2096
- )
2097
- ensembleFindings = fusedToAnalystFindings(ensemble.fused, {
2098
- dirs: [...new Set(runs.map((r) => r.dir))],
2099
- totalAnalysts: analysts.length,
2100
- }).map(asSearchProposalFinding)
2101
- } catch (cause) {
2102
- if (signal?.aborted) throw signal.reason
2103
- // A dead router must not kill the round: the raw-trace context below
2104
- // still grounds the proposer; the failure is logged, never silent.
2105
- log(`diagnosis ensemble FAILED for gen ${input.generation}: ${(cause as Error).message}`)
2106
- }
2107
- }
2108
- // GEN-5 split: the raw-trace distiller must not hand private-instance
2109
- // cells' path context to the authors either — censor them out of the
2110
- // candidates' campaigns before distillation.
2111
- const censoredInput =
2112
- split === null
2113
- ? input
2114
- : {
2115
- ...input,
2116
- candidates: input.candidates.map((cand) => {
2117
- return {
2118
- ...cand,
2119
- campaign: {
2120
- ...cand.campaign,
2121
- cells: cand.campaign.cells.filter((cell) => !privateIids.has(cell.scenarioId)),
2122
- },
2123
- }
2124
- }),
2125
- }
2126
- signal?.throwIfAborted()
2127
- const rawFindings = await rawTrace(censoredInput)
2128
- signal?.throwIfAborted()
2129
- return [steeringFinding, ...ensembleFindings, ...rawFindings]
2130
- }
2131
-
2132
- // ── protocol_v2: NEVER ships from inside the loop. `budget.holdout:
2133
- // 'deferred'` makes the LIB dispatch zero holdout cells, force `hold`, omit
2134
- // `lift`, and record `holdout: 'deferred'` in the provenance record; the
2135
- // pre-registered holdout run happens later, with operator approval. The
2136
- // would-be-KEEP operator brief is computed post-run from campaign cells
2137
- // (see the summary below). ───────────────────────────────────────────
2138
- const holdoutReps = config.holdoutRepsPerInstance ?? 2
2139
- const holdoutInstruction =
2140
- `holdout (${config.holdoutInstances.length} pre-registered instances: ${config.holdoutInstances.join(', ')}) ` +
2141
- `was NOT run — operator approval required. To certify a would-be KEEP under the ${holdoutReps}-rep ` +
2142
- 'fail-closed protocol (same-protocol parent comparison): ' +
2143
- 'tsx src/swe-arena/holdout-certify.mts <config.json> --candidate <winner-loops-commit>'
2144
- const improveRunDir = join(config.outDir, 'improve-run')
2145
-
2146
- // ── crash recovery: a killed run leaves its in-flight paid call 'pending'
2147
- // in the durable cost ledger, and the ledger's fail-closed guard then
2148
- // refuses ALL new paid work on resume. Under the outDir instance lock
2149
- // (sole runner), every pending call restored from disk is provably from a
2150
- // dead process — settle each as a $0 failure receipt (reason
2151
- // 'process-crash-orphan') so the guard passes without erasing the crash
2152
- // from the durable record. ───────────────────────────────────────────
2153
- for (const receipt of reconcileCrashOrphansOnDisk(improveRunDir)) {
2154
- log(
2155
- `cost-ledger: reconciled crash-orphaned call '${receipt.callId}' ` +
2156
- `(${receipt.actor}, ${receipt.phase}) as ${CRASH_ORPHAN_REASON}`,
2157
- )
2158
- }
2159
-
2160
- // ── generator: the gen-3 proposer fan-out (parallel AgentProfile-pinned
2161
- // authors + pre-filter) when `proposers` is configured; the legacy
2162
- // single-author generator otherwise. ─────────────────────────────────
2163
- const fanout =
2164
- config.proposers !== undefined
2165
- ? fanOutLoopsGenerator(config, {
2166
- ...(smokeRunner ? { smokeRunner } : {}),
2167
- ...(runnerImplementationRef ? { runnerImplementationRef } : {}),
2168
- ...(judgeImplementationRef ? { judgeImplementationRef } : {}),
2169
- ...(paretoParents.length > 0 ? { parents: paretoParents } : {}),
2170
- ...(briefingCtx !== undefined ? { briefing: briefingCtx } : {}),
2171
- // The GEPA seat's inner evaluator uses the same public-only
2172
- // smoke instance; the split guards the never-surfaced invariant at
2173
- // the bridge boundary too.
2174
- ...(smokeIid !== null ? { smokeInstanceId: smokeIid } : {}),
2175
- scoreSplit: split,
2176
- log,
2177
- })
2178
- : null
2179
- const generator = withParentCancellation(fanout ?? constrainedLoopsGenerator(config), signal)
2180
-
2181
- // ── the improve() call: the optimizer seat ───────────────────────────
2182
- // Typed from improve()'s own parameter: the monorepo hoists two
2183
- // agent-interface majors, so a nominal import can resolve to the wrong one.
2184
- signal?.throwIfAborted()
2185
- log(`round ${config.round} runId=${runId}: improve(surface:'code') over ${config.loopsRepo}@${config.loopsBaseRef}`)
2186
- const result = await improve<Scenario, R4Artifact>({
2187
- surface: 'code',
2188
- // analyzeGeneration wins over this flag; the composite above embeds
2189
- // rawTraceDistiller directly so the raw-trace mechanism stays active.
2190
- rawTraceContext: true,
2191
- analyzeGeneration,
2192
- code: {
2193
- repoRoot: config.loopsRepo,
2194
- baseRef: config.loopsBaseRef,
2195
- worktreeDir: join(config.outDir, 'loops-worktrees'),
2196
- profile: configuredAuthorProfile(config),
2197
- generator,
2198
- },
2199
- scenarios: sweScenarios,
2200
- judge: judgeConfig,
2201
- agent,
2202
- budget: {
2203
- generations: config.generations,
2204
- populationSize: config.populationSize,
2205
- maxConcurrency: 1,
2206
- reps,
2207
- maxImprovementShots: config.maxShots,
2208
- // Deferred with no reserved set: ALL improvement-set scenarios train;
2209
- // the held-out comparison lives in the separate operator-approved run.
2210
- holdout: 'deferred',
2211
- },
2212
- // TRAINING RECORDER: every scored (artifact, judge score) lands in the
2213
- // lib's labeled-scenario store as a JSONL corpus under outDir (growth is
2214
- // outDir-scoped; a handful of cells per round). Records carry the default
2215
- // 'unverified' trust — corpus-grade, NOT gold-eligible, which is right
2216
- // until an operator-confirmed holdout verdict upgrades them.
2217
- labeledStore: new FsLabeledScenarioStore({ root: join(config.outDir, 'labeled-store') }),
2218
- captureSource: 'eval-run',
2219
- // Arm/judge/proposer spend reaches the campaign meter through real paid
2220
- // calls (worker receipt per swe cell, $0 judge receipts, imported
2221
- // proposer-shot receipts). 'warn' not 'assert': the official judge's $0
2222
- // receipts are correct-by-design and must not kill the round as "stubs".
2223
- expectUsage: 'warn',
2224
- // Widened: covers worst-case capacity-gate holds; the REAL per-cell work
2225
- // clock (config.dispatchTimeoutMs) starts post-gate inside the dispatch.
2226
- dispatchTimeoutMs: campaignDispatchCeilingMs(config),
2227
- runDir: improveRunDir,
2228
- ...(premeasured ? { premeasuredBaseline: premeasured } : {}),
2229
- })
2230
- signal?.throwIfAborted()
2231
-
2232
- // ── staircase rows + round summary — scored from the LIB's campaign cells
2233
- // (baselineCampaign + per-generation candidate campaigns), which replay
2234
- // correctly attributed on resume. The recorder only contributes the
2235
- // dispatch-time diff/change-space description (recomputed post-run via
2236
- // ensure() for candidates that were replayed, never dispatched here). ────
2237
- try {
2238
- const loop = result.raw.raw
2239
- const baselineCells = cellsFromCampaign(loop.baselineCampaign)
2240
- const baselineWallS = sumWallSFromCells(baselineCells)
2241
- const measuredBaselineCount = resolvedInstanceCount(
2242
- replicateRunsFromCells(baselineCells),
2243
- config.instances,
2244
- reps,
2245
- )
2246
- const campaignBySurface = new Map<string, CampaignResult<R4Artifact, Scenario>>()
2247
- for (const gen of loop.generations) {
2248
- for (const s of gen.surfaces) campaignBySurface.set(s.surfaceHash, s.campaign)
2249
- }
2250
- const resolvedCountOf = (campaign: CampaignResult<R4Artifact, Scenario>): number =>
2251
- resolvedInstanceCount(replicateRunsFromCells(cellsFromCampaign(campaign)), config.instances, reps)
2252
-
2253
- // BASELINE-DRIFT: a resumed runDir can still hold baseline cells cached by
2254
- // an OLDER (pre-artifact) run. When they contradict the lib-validated
2255
- // premeasured artifact, log loud — the artifact rules, never silently.
2256
- if (premeasured) {
2257
- const cachedBaseline = await loadCampaignCells(join(improveRunDir, 'baseline'))
2258
- if (cachedBaseline.length > 0) {
2259
- const expected = instanceVerdictsFromCells(baselineCells, config.instances, reps)
2260
- for (const w of baselineDriftWarnings(
2261
- expected,
2262
- replicateRunsFromCells(cachedBaseline),
2263
- config.instances,
2264
- reps,
2265
- )) {
2266
- log(`BASELINE-DRIFT: ${w}`)
2267
- }
2268
- }
2269
- }
2270
-
2271
- // Collected per-candidate facts for activation and proposer rewards.
2272
- const activationBySurface = new Map<string, ActivationRecord>()
2273
- interface ProposerOutcomeFact {
2274
- generation: number
2275
- candidateIndex: number
2276
- label: string
2277
- commit: string | null
2278
- resolvedCount: number
2279
- diffPath: string | null
2280
- }
2281
- const proposerFacts: ProposerOutcomeFact[] = []
2282
-
2283
- for (let g = 0; g < loop.generations.length; g++) {
2284
- const gen = loop.generations[g]!
2285
- const rows: StaircaseRow[] = []
2286
- for (let candIndex = 0; candIndex < gen.record.candidates.length; candIndex++) {
2287
- const cand = gen.record.candidates[candIndex]!
2288
- const surface = gen.surfaces.find((s) => s.surfaceHash === cand.surfaceHash)?.surface
2289
- const cs = surface && typeof surface === 'object' && surface.kind === 'code' ? surface : null
2290
- const desc = cs ? await recorder.ensure(cs) : undefined
2291
- const campaign = campaignBySurface.get(cand.surfaceHash)
2292
- const cells = campaign ? cellsFromCampaign(campaign) : []
2293
- const runs = replicateRunsFromCells(cells)
2294
- const perInstance = perInstanceFromCells(cells)
2295
- const candResolved = resolvedInstanceCount(runs, config.instances, reps)
2296
- const wallS = sumWallSFromCells(cells)
2297
- const coverageComplete =
2298
- cand.eligibleForPromotion === true && replicateCoverageComplete(runs, config.instances, reps)
2299
- const costRatio = baselineWallS > 0 ? wallS / baselineWallS : null
2300
- // Parent's AND-resolved count. A parent hash with no candidate
2301
- // campaign IS the baseline incumbent — its count comes from the
2302
- // baseline campaign (the lib-validated premeasured artifact, or the
2303
- // bootstrap run's measurement; both survive resume, no dispatch-order
2304
- // guess).
2305
- const parentCampaign = cand.parentSurfaceHash
2306
- ? campaignBySurface.get(cand.parentSurfaceHash)
2307
- : undefined
2308
- const parentResolvedCount =
2309
- parentCampaign !== undefined ? resolvedCountOf(parentCampaign) : measuredBaselineCount
2310
- const violations = desc?.violations ?? []
2311
-
2312
- // GEN-5 activation gate: run the candidate's own committed predicate
2313
- // over its own cell run dirs. Fail-closed — a missing/unparseable
2314
- // predicate (the prefilter should have killed it) quarantines.
2315
- let activation: ActivationRecord | undefined
2316
- if (config.activationGate === true && cs !== null) {
2317
- const committed = await readCommittedPredicate(config.loopsRepo, cs.candidateCommit)
2318
- if (committed === null || !committed.parsed.ok) {
2319
- const why =
2320
- committed === null
2321
- ? `no ${ACTIVATION_PREDICATE_RELPATH} at ${cs.candidateCommit.slice(0, 10)}`
2322
- : `unparseable activation predicate: ${committed.parsed.ok ? '' : committed.parsed.error}`
2323
- activation = {
2324
- present: false,
2325
- description: null,
2326
- fired: false,
2327
- evidence: [],
2328
- warnings: [`${why} — fail-closed quarantine`],
2329
- }
2330
- } else {
2331
- const runDirs = [
2332
- ...new Set(
2333
- cells
2334
- .map((c) => (c.artifact !== null && c.artifact.kind === 'swe-arm' ? c.artifact.runDir : null))
2335
- .filter((d): d is string => typeof d === 'string' && d.length > 0),
2336
- ),
2337
- ]
2338
- const res = await runActivationPredicate(committed.parsed.predicate, runDirs)
2339
- activation = {
2340
- present: true,
2341
- description: committed.parsed.predicate.description,
2342
- fired: res.fired,
2343
- evidence: res.evidence,
2344
- warnings: res.warnings,
2345
- }
2346
- }
2347
- activationBySurface.set(cand.surfaceHash, activation)
2348
- log(
2349
- `activation ${cand.label ?? cand.surfaceHash.slice(0, 10)}: present=${activation.present} ` +
2350
- `fired=${activation.fired}${activation.fired ? ` — ${activation.evidence[0] ?? ''}` : ''}` +
2351
- `${activation.warnings.length > 0 ? ` (warnings: ${activation.warnings.join('; ')})` : ''}`,
2352
- )
2353
- }
2354
-
2355
- // GEN-5 split sub-scores: both halves logged per candidate; the
2356
- // selection rule stays combined (candResolved over ALL instances).
2357
- const verdicts = instanceVerdictsFromCells(cells, config.instances, reps)
2358
- const splitScores = split !== null ? subScores(verdicts, split) : null
2359
- if (split !== null && splitScores !== null) {
2360
- log(
2361
- `split scores ${cand.label ?? cand.surfaceHash.slice(0, 10)}: ` +
2362
- `public ${splitScores.publicResolvedCount}/${split.publicInstances.length}, ` +
2363
- `private ${splitScores.privateResolvedCount}/${split.privateInstances.length} (combined ${candResolved}/${config.instances.length})`,
2364
- )
2365
- }
2366
-
2367
- const verdict = decideVerdict({
2368
- violations,
2369
- coverageComplete,
2370
- resolvedCount: candResolved,
2371
- parentResolvedCount,
2372
- costRatio,
2373
- costGuardRatio: config.costGuardRatio,
2374
- ...(activation !== undefined ? { activationFired: activation.fired } : {}),
2375
- })
2376
- rows.push({
2377
- schema: STAIRCASE_SCHEMA,
2378
- round: config.round,
2379
- generation: g,
2380
- runId,
2381
- at: new Date().toISOString(),
2382
- candidate: cand.surfaceHash,
2383
- candidateCommit: cs?.candidateCommit ?? null,
2384
- parent: cand.parentSurfaceHash ?? 'baseline',
2385
- parentResolvedCount,
2386
- ...(cand.label ? { label: cand.label } : {}),
2387
- ...(cand.rationale ? { rationale: cand.rationale } : {}),
2388
- changedFiles: desc?.changedFiles ?? [],
2389
- changeSpaceViolations: violations,
2390
- perInstance,
2391
- resolvedCount: candResolved,
2392
- coverageComplete,
2393
- wallS,
2394
- baselineWallS,
2395
- costRatio,
2396
- costGuardRatio: config.costGuardRatio,
2397
- internallyPromoted: gen.record.promoted.includes(cand.surfaceHash),
2398
- verdict,
2399
- holdout: 'operator-approval-required',
2400
- armProvenance: desc?.armProvenance ?? null,
2401
- diffPath: desc?.diffPath ?? null,
2402
- diffSha256: desc?.diffSha256 ?? null,
2403
- ...(split !== null && splitScores !== null
2404
- ? {
2405
- split: {
2406
- publicInstances: split.publicInstances,
2407
- privateInstances: split.privateInstances,
2408
- ...splitScores,
2409
- },
2410
- }
2411
- : {}),
2412
- ...(activation !== undefined ? { activation } : {}),
2413
- })
2414
-
2415
- const label = cand.label ?? cs?.candidateCommit?.slice(0, 10) ?? cand.surfaceHash.slice(0, 10)
2416
- proposerFacts.push({
2417
- generation: g,
2418
- candidateIndex: candIndex,
2419
- label,
2420
- commit: cs?.candidateCommit ?? null,
2421
- resolvedCount: candResolved,
2422
- diffPath: desc?.diffPath ?? null,
2423
- })
2424
- }
2425
- const genFile = join(config.roundsDir, `gen-${g}.jsonl`)
2426
- for (const row of rows) await appendFile(genFile, JSON.stringify(row) + '\n')
2427
- log(`staircase: ${rows.length} row(s) → ${genFile} (${rows.map((r) => r.verdict).join(', ')})`)
2428
- }
2429
-
2430
- // Pre-filter kills: candidates the fan-out killed BEFORE evaluation never
2431
- // became surfaces (zero arm cells), so the loop has no row for them —
2432
- // each becomes an explicit `rejected-prefilter` staircase dot with its
2433
- // kill reason and forensics patch.
2434
- if (fanout) {
2435
- const kills = fanout.drainPrefilterKills()
2436
- for (const kill of kills) {
2437
- const row: StaircaseRow = {
2438
- schema: STAIRCASE_SCHEMA,
2439
- round: config.round,
2440
- generation: kill.generation,
2441
- runId,
2442
- at: new Date().toISOString(),
2443
- candidate: `prefilter-kill:${kill.diffSha256?.slice('sha256:'.length, 'sha256:'.length + 12) ?? kill.proposer}`,
2444
- candidateCommit: null,
2445
- parent: premeasured?.surfaceHash ?? 'baseline',
2446
- parentResolvedCount: measuredBaselineCount,
2447
- label: kill.proposer,
2448
- rationale: `prefilter kill at stage '${kill.stage}' (${kill.harness ?? 'engine'})`,
2449
- changedFiles: [],
2450
- changeSpaceViolations: kill.stage === 'change-space' ? [kill.reason] : [],
2451
- perInstance: [],
2452
- resolvedCount: 0,
2453
- coverageComplete: false,
2454
- wallS: kill.smoke?.wallS ?? 0,
2455
- baselineWallS,
2456
- costRatio: null,
2457
- costGuardRatio: config.costGuardRatio,
2458
- internallyPromoted: false,
2459
- verdict: 'rejected-prefilter',
2460
- killReason: `${kill.stage}: ${kill.reason}`,
2461
- holdout: 'operator-approval-required',
2462
- armProvenance: null,
2463
- diffPath: kill.patchPath,
2464
- diffSha256: kill.diffSha256,
2465
- }
2466
- const genFile = join(config.roundsDir, `gen-${kill.generation}.jsonl`)
2467
- await appendFile(genFile, JSON.stringify(row) + '\n')
2468
- log(`staircase: prefilter kill dot (${kill.proposer}, ${kill.stage}) → ${genFile}`)
2469
- }
2470
- }
2471
-
2472
- // GEN-5 label-v2 proposer rewards: baseline-relative (candidate − baseline
2473
- // resolved fraction, improvement positive), one settle-time ledger line
2474
- // per evaluated candidate now that the round's scores are final.
2475
- if (settleCapture !== null) {
2476
- const sanitizeName = (s: string): string => s.replace(/[^a-zA-Z0-9_-]/g, '_')
2477
- for (const fact of proposerFacts) {
2478
- try {
2479
- const flatDir = join(config.outDir, 'proposer-shots')
2480
- const pattern = new RegExp(`^gen${fact.generation}-cand${fact.candidateIndex}-shot\\d+\\.json$`)
2481
- const receiptPaths: string[] = []
2482
- for (const dir of [flatDir, join(flatDir, sanitizeName(fact.label))]) {
2483
- for (const name of (await readdir(dir).catch(() => [])).sort()) {
2484
- if (pattern.test(name)) receiptPaths.push(join(dir, name))
2485
- }
2486
- }
2487
- await settleCapture.captureProposer({
2488
- generation: fact.generation,
2489
- candidateIndex: fact.candidateIndex,
2490
- proposer: fact.label,
2491
- harness: config.proposers?.find((p) => p.name === fact.label)?.harness ?? null,
2492
- commit: fact.commit,
2493
- candResolved: fact.resolvedCount,
2494
- baselineResolved: measuredBaselineCount,
2495
- shotReceiptPaths: receiptPaths,
2496
- diffPath: fact.diffPath,
2497
- })
2498
- } catch (cause) {
2499
- log(`rollout-ledger: proposer capture FAILED for ${fact.label}: ${(cause as Error).message}`)
2500
- }
2501
- }
2502
- }
2503
-
2504
- const winnerSurface = result.raw.winner.surface
2505
- const winnerCs =
2506
- typeof winnerSurface === 'object' && winnerSurface !== null && winnerSurface.kind === 'code'
2507
- ? winnerSurface
2508
- : null
2509
- const winnerRec = winnerCs ? await recorder.ensure(winnerCs) : undefined
2510
- let winnerPatch: string | null = null
2511
- if (winnerRec?.diffPath) {
2512
- winnerPatch = join(config.outDir, 'winner.patch')
2513
- await writeFile(winnerPatch, await readFile(winnerRec.diffPath, 'utf8'))
2514
- }
2515
-
2516
- // BOOTSTRAP: persist this run's measured baseline campaign as the
2517
- // premeasured artifact every later run consumes (and the lib re-validates
2518
- // by surface hash / seed / reps / split digest). The baseline surface
2519
- // hash comes from the Pareto frontier's generation −1 entry — the lib's
2520
- // own record of the baseline measurement.
2521
- if (!premeasured) {
2522
- const baselineHash = loop.paretoFrontier.find((p) => p.generation === -1)?.surfaceHash
2523
- if (baselineHash === undefined) {
2524
- log('bootstrap: no generation −1 Pareto entry — premeasured baseline artifact NOT written')
2525
- } else {
2526
- const artifact: PremeasuredOptimizationBaseline<R4Artifact, Scenario> = {
2527
- surfaceHash: baselineHash,
2528
- campaign: loop.baselineCampaign,
2529
- }
2530
- await writeFile(config.premeasuredBaselinePath, JSON.stringify(artifact, null, 1))
2531
- log(`bootstrap: premeasured baseline artifact → ${config.premeasuredBaselinePath} (surface ${baselineHash})`)
2532
- }
2533
- }
2534
-
2535
- // The would-be-KEEP operator brief: winner vs baseline on the improvement
2536
- // set, from campaign cells. The lib's deferred-holdout gate always holds;
2537
- // this evidence tells the operator whether the pre-registered holdout run
2538
- // is worth approving.
2539
- const winnerHash = winnerCs ? surfaceHash(winnerCs) : null
2540
- const winnerCampaign = winnerHash !== null ? campaignBySurface.get(winnerHash) : undefined
2541
- const winnerActivation = winnerHash !== null ? activationBySurface.get(winnerHash) : undefined
2542
- const improvementSet =
2543
- winnerCampaign !== undefined && winnerRec !== undefined
2544
- ? gateEvidenceFromCells({
2545
- winnerCells: cellsFromCampaign(winnerCampaign),
2546
- baselineCells,
2547
- violations: winnerRec.violations,
2548
- iids: config.instances,
2549
- reps,
2550
- costGuardRatio: config.costGuardRatio,
2551
- ...(winnerActivation !== undefined ? { activationFired: winnerActivation.fired } : {}),
2552
- })
2553
- : null
2554
- const wouldKeep = improvementSet !== null && improvementSet.verdict === 'accepted'
2555
- if (improvementSet) {
2556
- log(
2557
- `improvement set: winner ${improvementSet.candResolved}/${config.instances.length} vs baseline ` +
2558
- `${improvementSet.baseResolved}/${config.instances.length}; wall ${improvementSet.candWallS}s vs ` +
2559
- `${improvementSet.baseWallS}s (ratio ${improvementSet.costRatio === null ? 'n/a' : improvementSet.costRatio.toFixed(2)}, ` +
2560
- `guard ${config.costGuardRatio}); protocol verdict: ${improvementSet.verdict}${wouldKeep ? ' (WOULD-BE KEEP)' : ''}`,
2561
- )
2562
- } else {
2563
- log('improvement set: winner == baseline (no candidate campaign) — nothing to promote')
2564
- }
2565
-
2566
- const summary = {
2567
- schema: 'swe-arena.round-summary.v2',
2568
- round: config.round,
2569
- runId,
2570
- at: new Date().toISOString(),
2571
- loops: { repo: config.loopsRepo, baseRef: config.loopsBaseRef },
2572
- // The gate's denominator: the lib-validated premeasured artifact, or
2573
- // this bootstrap run's freshly measured (and persisted) campaign.
2574
- baseline: {
2575
- resolvedCount: measuredBaselineCount,
2576
- wallS: baselineWallS,
2577
- perInstance: perInstanceFromCells(baselineCells),
2578
- premeasured: premeasured !== undefined,
2579
- artifactPath: config.premeasuredBaselinePath,
2580
- ...(premeasured ? { surfaceHash: premeasured.surfaceHash } : {}),
2581
- },
2582
- winner: winnerCs
2583
- ? {
2584
- surfaceHash: winnerHash,
2585
- commit: winnerCs.candidateCommit,
2586
- label: result.raw.winner.label ?? null,
2587
- rationale: result.raw.winner.rationale ?? null,
2588
- patch: winnerPatch,
2589
- }
2590
- : null,
2591
- // The lib's verdict + reasons: deferred holdout forces `hold` with zero
2592
- // holdout cells dispatched and no fabricated lift.
2593
- gateDecision: result.decision,
2594
- gateReasons: loop.gateResult.reasons,
2595
- // Improvement-set (search-split) evidence — NOT a held-out measurement.
2596
- improvementSet: improvementSet === null ? null : { ...improvementSet, wouldKeep },
2597
- // GEN-5: the public/private split (sub-scores live per candidate in the
2598
- // staircase rows; selection stays combined; private never surfaced to
2599
- // proposers — the 2-of-6 private half is a direction check, not a
2600
- // certification).
2601
- scoreSplit:
2602
- split === null
2603
- ? null
2604
- : {
2605
- seededBy: split.seededBy,
2606
- publicInstances: split.publicInstances,
2607
- privateInstances: split.privateInstances,
2608
- },
2609
- // GEN-5: activation-gate outcomes per candidate surface.
2610
- activationGate:
2611
- config.activationGate === true
2612
- ? {
2613
- enabled: true,
2614
- byCandidate: [...activationBySurface.entries()].map(([surface, a]) => ({
2615
- surface,
2616
- present: a.present,
2617
- fired: a.fired,
2618
- description: a.description,
2619
- })),
2620
- }
2621
- : { enabled: false },
2622
- // GEN-5: MAP+TOOLBOX briefing provenance.
2623
- briefing:
2624
- briefingCtx === undefined
2625
- ? null
2626
- : { version: AUTHOR_BRIEFING_VERSION, indexPath: briefingCtx.indexPath, textSource: briefingCtx.briefingSource },
2627
- // GEN-5: settle-time rollout ledger location (tangle.rollout.v1, label v2).
2628
- rolloutLedger: settleCapture === null ? null : { path: settleCapture.path, capture: 'settle-time', labels: 'v2' },
2629
- // Honest run-wide spend from the lib's CostLedger: per-channel rollups
2630
- // (agent = arm cells, judge = official-judge calls, driver = proposer
2631
- // shots), token totals, and accounting-completeness flags.
2632
- cost: {
2633
- totalCostUsd: result.raw.totalCostUsd,
2634
- inputTokens: result.raw.cost.inputTokens,
2635
- outputTokens: result.raw.cost.outputTokens,
2636
- byChannel: result.raw.cost.byChannel,
2637
- fullyPriced: result.raw.cost.fullyPriced,
2638
- usageComplete: result.raw.cost.usageComplete,
2639
- accountingComplete: result.raw.cost.accountingComplete,
2640
- incompleteReasons: result.raw.cost.incompleteReasons,
2641
- receipts: result.raw.receipts.length,
2642
- },
2643
- holdout: {
2644
- instances: config.holdoutInstances,
2645
- mode: 'deferred',
2646
- status: 'operator-approval-required',
2647
- // The certification protocol the operator run must use — 2-rep
2648
- // fail-closed with a same-protocol parent (gen-2 postmortem).
2649
- protocol: {
2650
- repsPerInstance: holdoutReps,
2651
- resolvedRule: 'all-reps',
2652
- parentBaseline: config.holdoutBaseline ?? 'measure',
2653
- },
2654
- instruction: holdoutInstruction,
2655
- },
2656
- }
2657
- const summaryPath = join(config.roundsDir, `round${config.round}-summary-${runId}.json`)
2658
- await writeFile(summaryPath, JSON.stringify(summary, null, 2))
2659
- log(`round summary → ${summaryPath}`)
2660
-
2661
- // Round rollup at gate time: every cell's orchestration/economics in one table,
2662
- // written next to the round summary and echoed into the run log.
2663
- await reportSupervisorRound(join(config.outDir, 'arm-runs'), {
2664
- appendHeadlineTo: join(config.outDir, 'run.log'),
2665
- reportDir: config.roundsDir,
2666
- title: `Round ${config.round} rollup — ${runId}`,
2667
- echo: true,
2668
- }).catch((err: unknown) => {
2669
- log(`round rollup failed: ${err instanceof Error ? err.message : String(err)}`)
2670
- })
2671
-
2672
- log(`gate: ${result.decision} — ${loop.gateResult.reasons[0] ?? ''}`)
2673
- } finally {
2674
- await result.dispose()
2675
- }
2676
- signal?.throwIfAborted()
2677
- }
2678
-
2679
- // ---------------------------------------------------------------------------
2680
- // Calibration smoke — the ensemble over the REAL round-2 django SUP2 run
2681
- // (known truth: the worker authored a LOCAL idna helper inside the mail module
2682
- // while the gold fix adds punycode() in django/utils/encoding.py — a fix
2683
- // PLACEMENT failure). Cheap (a few k tokens/analyst); grades whether each
2684
- // blind analyst independently surfaces placement.
2685
- // ---------------------------------------------------------------------------
2686
-
2687
- export interface SmokeArgs {
2688
- supRunDir?: string
2689
- patchPath?: string
2690
- analysts?: number
2691
- model?: string
2692
- /** 'router' (default) or 'zai' — the z.ai coding endpoint is the proven
2693
- * fallback when router.tangle.tools 524-storms (a measured infra class). */
2694
- endpoint?: 'router' | 'zai'
2695
- /** Transport retries per analyst. Default 4 in the smoke (storms pass). */
2696
- retries?: number
2697
- secrets?: SecretsEnv
2698
- scratchDir?: string
2699
- }
2700
-
2701
- export async function calibrationSmoke(args: SmokeArgs = {}): Promise<{
2702
- perAnalyst: Array<{ analystId: string; ok: boolean; surfacesPlacement: boolean; findings: number; error?: string }>
2703
- fusedTop: string[]
2704
- }> {
2705
- const supRunDir = args.supRunDir ?? join(DEFAULT_HH_SCRATCHPAD, 'runs', 'django__django-11532', 'SUP2')
2706
- const patchPath = args.patchPath ?? join(DEFAULT_HH_SCRATCHPAD, 'patches', 'django__django-11532.sup2.patch')
2707
- const secrets: SecretsEnv = args.secrets ?? {
2708
- secretsDir: '/home/drew/company/devops/secrets',
2709
- envFiles: ['agent-state.env', 'tangle-router.env'],
2710
- }
2711
- const scratchDir = args.scratchDir ?? join(supRunDir, '..', '..', '..', 'r4', 'calibration-smoke')
2712
- const analysts: AnalystSpec[] = defaultAnalysts(args.analysts ?? 3, args.model ?? 'glm-5.2').map((s) =>
2713
- args.endpoint === 'zai' ? { ...s, url: ZAI_CODING_ENDPOINT, apiKeyEnv: 'ZAI_API_KEY' } : s,
2714
- )
2715
- const runs: SupRunArtifacts[] = [
2716
- {
2717
- iid: 'django__django-11532',
2718
- arm: 'SUP2',
2719
- dir: supRunDir,
2720
- ...(existsSync(patchPath) ? { patchPath } : {}),
2721
- judge: { resolved: false, note: 'round-2 official judge: unresolved while the self-verify passed' },
2722
- },
2723
- ]
2724
- const ensemble = await runDiagnosisEnsemble({
2725
- analysts,
2726
- runs,
2727
- secrets,
2728
- scratchDir,
2729
- retriesPerAnalyst: args.retries ?? 4,
2730
- retryDelayMs: 15_000,
2731
- onStatus: log,
2732
- })
2733
- const placement = surfacesPlacementRegex()
2734
- const perAnalyst = ensemble.reports.map((r) => ({
2735
- analystId: r.analystId,
2736
- ok: r.ok,
2737
- surfacesPlacement: r.findings.some((f) =>
2738
- placement.test(`${f.failure_class} ${f.evidence_quote} ${f.proposed_direction}`),
2739
- ),
2740
- findings: r.findings.length,
2741
- ...(r.error ? { error: r.error } : {}),
2742
- }))
2743
- console.log('\n=== CALIBRATION SMOKE (django__django-11532 SUP2, truth = fix placement) ===')
2744
- console.log(`bundle: ${ensemble.bundleChars} chars; analysts: ${analysts.map((a) => a.model).join(', ')}`)
2745
- for (const r of ensemble.reports) {
2746
- const grade = perAnalyst.find((p) => p.analystId === r.analystId)!
2747
- console.log(`\n--- ${r.analystId} ok=${r.ok} placement-surfaced=${grade.surfacesPlacement}${r.error ? ` error=${r.error}` : ''}` +
2748
- (r.tokens ? ` tokens(in=${r.tokens.input},out=${r.tokens.output})` : ''))
2749
- for (const f of r.findings) {
2750
- console.log(` [${f.confidence.toFixed(2)}] ${f.failure_class} → ${f.proposed_direction.slice(0, 160)}`)
2751
- if (f.evidence_quote) console.log(` evidence: ${f.evidence_quote.slice(0, 160)}`)
2752
- }
2753
- }
2754
- console.log('\n--- fused (agreement-ranked) ---')
2755
- for (const f of ensemble.fused) {
2756
- console.log(
2757
- ` agreement=${f.agreement}${f.competingHypothesis ? ' [competing hypothesis]' : ''} conf=${f.meanConfidence.toFixed(2)} — ${f.failure_class}`,
2758
- )
2759
- }
2760
- const outPath = join(scratchDir, 'calibration-smoke.json')
2761
- await mkdir(scratchDir, { recursive: true })
2762
- await writeFile(outPath, JSON.stringify({ perAnalyst, ensemble }, null, 2))
2763
- console.log(`\nfull output → ${outPath}`)
2764
- return { perAnalyst, fusedTop: ensemble.fused.map((f) => f.failure_class) }
2765
- }
2766
-
2767
- // ---------------------------------------------------------------------------
2768
- // CLI.
2769
- // ---------------------------------------------------------------------------
2770
-
2771
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
2772
-
2773
- if (isMain) {
2774
- const argv = process.argv.slice(2)
2775
- // Config execution owns long-lived workers and therefore needs cooperative
2776
- // cleanup. Utility modes keep the terminal's default signal behavior because
2777
- // their analyst APIs do not yet accept AbortSignal; installing a handler there
2778
- // would swallow Ctrl-C while the model call continued.
2779
- const interrupt = argv[0] && !argv[0].startsWith('--')
2780
- ? installProcessSignalAbort('outer-loop')
2781
- : undefined
2782
- try {
2783
- const flag = (name: string): string | undefined => {
2784
- const i = argv.indexOf(name)
2785
- return i !== -1 ? argv[i + 1] : undefined
2786
- }
2787
- if (argv[0] === '--write-config') {
2788
- const path = argv[1]
2789
- if (!path || path.startsWith('--')) {
2790
- console.error('usage: outer-loop.mts --write-config <path> [--out-name <dirname>] [--gen3|--gen4|--gen5]')
2791
- process.exit(2)
2792
- }
2793
- const outDirName = flag('--out-name')
2794
- const gen3 = argv.includes('--gen3')
2795
- const gen4 = argv.includes('--gen4')
2796
- const gen5 = argv.includes('--gen5')
2797
- let config: OuterLoopConfig
2798
- let flavor: string
2799
- if (gen4 || gen5) {
2800
- const make = gen5 ? defaultGen5Config : defaultGen4Config
2801
- config = make(undefined, { ...(outDirName ? { outDirName } : {}) })
2802
- flavor = gen5 ? 'gen-5' : 'gen-4'
2803
- } else {
2804
- const make = gen3 ? defaultGen3Config : defaultRound4Config
2805
- config = make(undefined, outDirName ? { outDirName } : {})
2806
- flavor = gen3 ? 'gen-3' : 'round-4'
2807
- }
2808
- await writeFile(path, JSON.stringify(config, null, 2) + '\n')
2809
- console.log(`default ${flavor} config → ${path}`)
2810
- } else if (argv[0] === '--calibration-smoke') {
2811
- const dir = argv[1] && !argv[1].startsWith('--') ? argv[1] : undefined
2812
- const n = flag('--analysts')
2813
- const model = flag('--model')
2814
- const endpoint = flag('--endpoint')
2815
- const retries = flag('--retries')
2816
- if (endpoint !== undefined && endpoint !== 'router' && endpoint !== 'zai') {
2817
- console.error(`--endpoint must be 'router' or 'zai', got ${JSON.stringify(endpoint)}`)
2818
- process.exit(2)
2819
- }
2820
- await calibrationSmoke({
2821
- ...(dir ? { supRunDir: dir } : {}),
2822
- ...(n ? { analysts: Number(n) } : {}),
2823
- ...(model ? { model } : {}),
2824
- ...(endpoint ? { endpoint } : {}),
2825
- ...(retries ? { retries: Number(retries) } : {}),
2826
- })
2827
- } else if (argv[0] && !argv[0].startsWith('--')) {
2828
- const config = JSON.parse(await readFile(argv[0], 'utf8')) as OuterLoopConfig
2829
- // Launch guards BEFORE any spend: keys present (dotenvx forgotten = hours
2830
- // of confusing downstream failures) and exactly one loop per outDir.
2831
- assertLaunchEnv()
2832
- interrupt!.signal.throwIfAborted()
2833
- const lock = await acquireInstanceLock(config.outDir)
2834
- try {
2835
- await runRound(config, interrupt!.signal)
2836
- } finally {
2837
- await lock.release()
2838
- }
2839
- } else {
2840
- console.error(
2841
- 'usage: tsx src/swe-arena/outer-loop.mts <config.json> # SPENDS: arms + judges + proposer\n' +
2842
- ' tsx src/swe-arena/outer-loop.mts --write-config <path> [--out-name <dirname>] [--gen3|--gen4|--gen5]\n' +
2843
- ' tsx src/swe-arena/outer-loop.mts --calibration-smoke [supRunDir] [--analysts N] [--model M] [--endpoint router|zai] [--retries N]',
2844
- )
2845
- process.exit(2)
2846
- }
2847
- } catch (cause) {
2848
- if (!interrupt?.signal.aborted) throw cause
2849
- const detail = cause instanceof Error ? cause.message : String(cause)
2850
- console.error(`outer-loop stopped after cleanup: ${detail}`)
2851
- } finally {
2852
- interrupt?.dispose()
2853
- }
2854
- }