@tangle-network/agent-bench 0.3.6 → 0.3.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (265) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/HARNESS.md +43 -0
  3. package/dist/adapters.js +24 -24
  4. package/dist/benchmarks/_harness.d.ts +1 -1
  5. package/dist/benchmarks/_harness.js +1 -1
  6. package/dist/benchmarks/aec-bench.js +2 -2
  7. package/dist/benchmarks/agentbench.js +2 -2
  8. package/dist/benchmarks/appworld.js +2 -2
  9. package/dist/benchmarks/bfcl.js +2 -2
  10. package/dist/benchmarks/commit0.js +2 -2
  11. package/dist/benchmarks/crag.js +2 -2
  12. package/dist/benchmarks/dabstep.js +2 -2
  13. package/dist/benchmarks/enterpriseops-gym.js +2 -2
  14. package/dist/benchmarks/finresearchbench.js +2 -2
  15. package/dist/benchmarks/humaneval.d.ts +10 -1
  16. package/dist/benchmarks/humaneval.js +5 -3
  17. package/dist/benchmarks/nomiracl.js +2 -2
  18. package/dist/benchmarks/open-rag-bench.js +2 -2
  19. package/dist/benchmarks/programbench.js +2 -2
  20. package/dist/benchmarks/ragbench.js +2 -2
  21. package/dist/benchmarks/swe-bench.js +2 -2
  22. package/dist/benchmarks/t2-ragbench.js +2 -2
  23. package/dist/benchmarks/tau-bench-shared.js +2 -2
  24. package/dist/benchmarks/tau2-bench.js +3 -3
  25. package/dist/benchmarks/tau3-banking.js +3 -3
  26. package/dist/benchmarks/terminal-bench.js +2 -2
  27. package/dist/benchmarks/toollm.js +2 -2
  28. package/dist/benchmarks/webarena-verified.js +2 -2
  29. package/dist/{chunk-CKUVRZ2T.js → chunk-3U5TXJZS.js} +2 -2
  30. package/dist/{chunk-PPYSEKFM.js → chunk-5H5XV76F.js} +73 -15
  31. package/dist/chunk-5H5XV76F.js.map +1 -0
  32. package/dist/{chunk-YCGY7UIZ.js → chunk-7GRVHU22.js} +2 -2
  33. package/dist/{chunk-Z7ML6L77.js → chunk-HWST3SED.js} +2 -2
  34. package/dist/{chunk-SYDW647C.js → chunk-IA2FBTWC.js} +2 -2
  35. package/dist/{chunk-R67DFVLO.js → chunk-IFVINJ4B.js} +2 -2
  36. package/dist/{chunk-R36V2VP7.js → chunk-IZ5M6OAC.js} +2 -2
  37. package/dist/{chunk-ODT47UAY.js → chunk-K3BQGZCT.js} +2 -2
  38. package/dist/{chunk-IFAV6KEM.js → chunk-KP5KD6EN.js} +2 -2
  39. package/dist/{chunk-ZEWMTR5M.js → chunk-MQMRLGOG.js} +2 -2
  40. package/dist/{chunk-TSWPNOYM.js → chunk-NQG5XDSB.js} +2 -2
  41. package/dist/{chunk-7WSD27QQ.js → chunk-PB64GYIG.js} +2 -2
  42. package/dist/{chunk-HBSWHQNJ.js → chunk-RCYQEFNX.js} +3 -3
  43. package/dist/{chunk-J3KDJNX2.js → chunk-RH5F53JT.js} +2 -2
  44. package/dist/{chunk-UAIOHCUK.js → chunk-SFLA7OH3.js} +3 -3
  45. package/dist/{chunk-Y6O2OCUO.js → chunk-SHM6MRRF.js} +2 -2
  46. package/dist/{chunk-KDIKRJGB.js → chunk-SHYIRB7I.js} +2 -2
  47. package/dist/{chunk-HHXFIHXC.js → chunk-SVR2LKYI.js} +2 -2
  48. package/dist/{chunk-5SBJCB6W.js → chunk-V7AEBY6U.js} +22 -22
  49. package/dist/{chunk-LRRD7NAG.js → chunk-WSKWVEQB.js} +18 -2
  50. package/dist/chunk-WSKWVEQB.js.map +1 -0
  51. package/dist/{chunk-2PVVP7GN.js → chunk-XKEFIFIC.js} +2 -2
  52. package/dist/{chunk-JRWWGMK7.js → chunk-XYA4XSNU.js} +2 -2
  53. package/dist/{chunk-X5YKXC6V.js → chunk-YSMEKBTD.js} +2 -2
  54. package/dist/{chunk-2XU6OGEN.js → chunk-Z4TZ76N7.js} +2 -2
  55. package/dist/index.js +24 -24
  56. package/package.json +6 -5
  57. package/scripts/run-package-tests.mjs +30 -8
  58. package/scripts/verify-packed-consumer.mjs +1 -1
  59. package/scripts/verify-pier-agent.mts +1 -0
  60. package/src/benchmarks/_harness.ts +20 -2
  61. package/src/benchmarks/humaneval.test.mts +122 -0
  62. package/src/benchmarks/humaneval.ts +100 -27
  63. package/src/david-attribution.mts +78 -0
  64. package/src/david-goliath.mts +149 -0
  65. package/src/hev-improve.mts +25 -6
  66. package/src/humaneval-object-ablation.mts +201 -0
  67. package/src/live-improve-campaign-mbpp.mts +641 -0
  68. package/src/live-improve-campaign.mts +500 -0
  69. package/src/mbpp-structural.mts +12 -7
  70. package/src/quant-arena/README.md +144 -0
  71. package/src/quant-arena/backtest.test.mts +135 -0
  72. package/src/quant-arena/backtest.ts +218 -0
  73. package/src/quant-arena/data.test.mts +44 -0
  74. package/src/quant-arena/data.ts +141 -0
  75. package/src/quant-arena/driver.test.mts +253 -0
  76. package/src/quant-arena/driver.ts +219 -0
  77. package/src/quant-arena/fixtures/data/PROVENANCE.md +26 -0
  78. package/src/quant-arena/fixtures/data/holdout/IDX.csv +523 -0
  79. package/src/quant-arena/fixtures/data/holdout/S01.csv +523 -0
  80. package/src/quant-arena/fixtures/data/holdout/S02.csv +523 -0
  81. package/src/quant-arena/fixtures/data/holdout/S03.csv +523 -0
  82. package/src/quant-arena/fixtures/data/holdout/S04.csv +523 -0
  83. package/src/quant-arena/fixtures/data/holdout/S05.csv +523 -0
  84. package/src/quant-arena/fixtures/data/holdout/S06.csv +523 -0
  85. package/src/quant-arena/fixtures/data/holdout/S07.csv +523 -0
  86. package/src/quant-arena/fixtures/data/holdout/S08.csv +523 -0
  87. package/src/quant-arena/fixtures/data/holdout/S09.csv +523 -0
  88. package/src/quant-arena/fixtures/data/holdout/S10.csv +523 -0
  89. package/src/quant-arena/fixtures/data/insample/IDX.csv +2087 -0
  90. package/src/quant-arena/fixtures/data/insample/S01.csv +2087 -0
  91. package/src/quant-arena/fixtures/data/insample/S02.csv +2087 -0
  92. package/src/quant-arena/fixtures/data/insample/S03.csv +2087 -0
  93. package/src/quant-arena/fixtures/data/insample/S04.csv +2087 -0
  94. package/src/quant-arena/fixtures/data/insample/S05.csv +2087 -0
  95. package/src/quant-arena/fixtures/data/insample/S06.csv +2087 -0
  96. package/src/quant-arena/fixtures/data/insample/S07.csv +2087 -0
  97. package/src/quant-arena/fixtures/data/insample/S08.csv +2087 -0
  98. package/src/quant-arena/fixtures/data/insample/S09.csv +2087 -0
  99. package/src/quant-arena/fixtures/data/insample/S10.csv +2087 -0
  100. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +16 -0
  101. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +5 -0
  102. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +171 -0
  103. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +119 -0
  104. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +119 -0
  105. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +105 -0
  106. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +102 -0
  107. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +4 -0
  108. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +2 -0
  109. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +84 -0
  110. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +117 -0
  111. package/src/quant-arena/holdout-certify.mts +206 -0
  112. package/src/quant-arena/holdout-certify.test.mts +82 -0
  113. package/src/quant-arena/leak-audit.test.mts +79 -0
  114. package/src/quant-arena/leak-audit.ts +95 -0
  115. package/src/quant-arena/make-fixtures.mts +161 -0
  116. package/src/quant-arena/multiplicity.test.mts +68 -0
  117. package/src/quant-arena/multiplicity.ts +87 -0
  118. package/src/quant-arena/nautilus-certify.ts +31 -0
  119. package/src/quant-arena/oms.ts +90 -0
  120. package/src/quant-arena/profiles/quant-researcher.profile.json +7 -0
  121. package/src/quant-arena/python/pyproject.toml +8 -0
  122. package/src/quant-arena/python/uv.lock +1297 -0
  123. package/src/quant-arena/python/vbt-worker.py +192 -0
  124. package/src/quant-arena/quant-loop.mts +813 -0
  125. package/src/quant-arena/quant-loop.test.mts +75 -0
  126. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +11 -0
  127. package/src/quant-arena/strategies/equal-weight/strategy.ts +20 -0
  128. package/src/quant-arena/strategies/sma-crossover/strategy.ts +42 -0
  129. package/src/quant-arena/types.ts +133 -0
  130. package/src/quant-arena/vbt-client.ts +321 -0
  131. package/src/quant-arena/vbt-parity.test.mts +183 -0
  132. package/src/quant-arena/windows.test.mts +45 -0
  133. package/src/quant-arena/windows.ts +54 -0
  134. package/src/rollout-ledger/backfill-swe-arena.mts +606 -0
  135. package/src/rollout-ledger/backfill-swe-arena.test.mts +338 -0
  136. package/src/rollout-ledger/settle-capture.mts +442 -0
  137. package/src/rollout-ledger/settle-capture.test.mts +270 -0
  138. package/src/stream-observe.py +45 -0
  139. package/src/stream-observe.tpl.html +247 -0
  140. package/src/supervisor-arena.mts +816 -0
  141. package/src/swe-arena/activation.mts +228 -0
  142. package/src/swe-arena/activation.test.mts +303 -0
  143. package/src/swe-arena/analyze.ts +211 -0
  144. package/src/swe-arena/arms.ts +804 -0
  145. package/src/swe-arena/bootstrap-meta.mts +188 -0
  146. package/src/swe-arena/bootstrap-meta.test.mts +51 -0
  147. package/src/swe-arena/briefing.mts +217 -0
  148. package/src/swe-arena/briefing.test.mts +178 -0
  149. package/src/swe-arena/calibrate.ts +217 -0
  150. package/src/swe-arena/capabilities.mts +76 -0
  151. package/src/swe-arena/capabilities.test.mts +57 -0
  152. package/src/swe-arena/capacity.ts +194 -0
  153. package/src/swe-arena/cell-evidence.mts +437 -0
  154. package/src/swe-arena/cell-evidence.test.mts +248 -0
  155. package/src/swe-arena/diagnosis-ensemble.test.mts +210 -0
  156. package/src/swe-arena/diagnosis-ensemble.ts +520 -0
  157. package/src/swe-arena/execution.test.mts +1170 -0
  158. package/src/swe-arena/factory-command-container.ts +284 -0
  159. package/src/swe-arena/factory-judge-child.mts +228 -0
  160. package/src/swe-arena/factory.test.mts +643 -0
  161. package/src/swe-arena/fixtures/analyze.py +80 -0
  162. package/src/swe-arena/fixtures/excludes.txt +8 -0
  163. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +51 -0
  164. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +29 -0
  165. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +64 -0
  166. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +48 -0
  167. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +29 -0
  168. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +48 -0
  169. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +47 -0
  170. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +30 -0
  171. package/src/swe-arena/fixtures/factory/loops-28/spec.md +50 -0
  172. package/src/swe-arena/fixtures/gen1-salvage/README.md +45 -0
  173. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +116 -0
  174. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +293 -0
  175. package/src/swe-arena/fixtures/holdout-preregister.log +12 -0
  176. package/src/swe-arena/fixtures/holdout.json +44 -0
  177. package/src/swe-arena/fixtures/instances.json +146 -0
  178. package/src/swe-arena/fixtures/ledger.jsonl +12 -0
  179. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +36 -0
  180. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +33 -0
  181. package/src/swe-arena/fixtures/rejudge.jsonl +15 -0
  182. package/src/swe-arena/fixtures/rematch.jsonl +3 -0
  183. package/src/swe-arena/fixtures/rematch2.jsonl +3 -0
  184. package/src/swe-arena/fixtures/rematch3.jsonl +3 -0
  185. package/src/swe-arena/fixtures/run-report/README.md +43 -0
  186. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +173 -0
  187. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +100 -0
  188. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +551 -0
  189. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +64 -0
  190. package/src/swe-arena/fixtures/sup-journal-true.json +19 -0
  191. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +48 -0
  192. package/src/swe-arena/fixtures/verify/django__django-11532.sh +50 -0
  193. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +76 -0
  194. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +44 -0
  195. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +32 -0
  196. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +51 -0
  197. package/src/swe-arena/fixtures/worker-tokens.json +42 -0
  198. package/src/swe-arena/fixtures.ts +237 -0
  199. package/src/swe-arena/gepa-seat.mts +583 -0
  200. package/src/swe-arena/gepa-seat.test.mts +635 -0
  201. package/src/swe-arena/holdout-certify.mts +408 -0
  202. package/src/swe-arena/holdout-certify.test.mts +160 -0
  203. package/src/swe-arena/judge-child.mts +37 -0
  204. package/src/swe-arena/ledger-orphans.mts +77 -0
  205. package/src/swe-arena/ledger-orphans.test.mts +147 -0
  206. package/src/swe-arena/lineage-record.mts +164 -0
  207. package/src/swe-arena/lineage-record.test.mts +115 -0
  208. package/src/swe-arena/manifest.mts +293 -0
  209. package/src/swe-arena/manifest.test.mts +169 -0
  210. package/src/swe-arena/materialize.ts +142 -0
  211. package/src/swe-arena/outer-loop.mts +2795 -0
  212. package/src/swe-arena/outer-loop.test.mts +696 -0
  213. package/src/swe-arena/parity.test.mts +87 -0
  214. package/src/swe-arena/premeasured-from-cells.mts +281 -0
  215. package/src/swe-arena/premeasured-from-cells.test.mts +180 -0
  216. package/src/swe-arena/proc.test.mts +174 -0
  217. package/src/swe-arena/proc.ts +260 -0
  218. package/src/swe-arena/profiles/default-author.profile.json +4 -0
  219. package/src/swe-arena/proposer-fanout.mts +770 -0
  220. package/src/swe-arena/proposer-fanout.test.mts +619 -0
  221. package/src/swe-arena/proposer-provenance.mts +177 -0
  222. package/src/swe-arena/proposer-provenance.test.mts +106 -0
  223. package/src/swe-arena/reconcile.ts +0 -0
  224. package/src/swe-arena/replay.mts +183 -0
  225. package/src/swe-arena/replay.test.mts +300 -0
  226. package/src/swe-arena/run-experiment.mts +727 -0
  227. package/src/swe-arena/run-report.mts +75 -0
  228. package/src/swe-arena/run-supervisor.mjs +297 -0
  229. package/src/swe-arena/run-supervisor.test.mts +500 -0
  230. package/src/swe-arena/score-split.mts +140 -0
  231. package/src/swe-arena/score-split.test.mts +123 -0
  232. package/src/swe-arena/serialized-judge.ts +414 -0
  233. package/src/swe-arena/types.ts +218 -0
  234. package/src/swe-code-improve.mts +328 -0
  235. package/src/swe-emit-patch.mts +104 -0
  236. package/src/swe-improve.mts +232 -0
  237. package/src/swe-jail.ts +2 -2
  238. package/src/swe-local-proof.mts +169 -0
  239. package/src/swe-repro-calibrate.mts +446 -0
  240. package/src/swe-stream.mts +1497 -0
  241. package/src/swe-structural.mts +245 -837
  242. package/dist/chunk-LRRD7NAG.js.map +0 -1
  243. package/dist/chunk-PPYSEKFM.js.map +0 -1
  244. /package/dist/{chunk-CKUVRZ2T.js.map → chunk-3U5TXJZS.js.map} +0 -0
  245. /package/dist/{chunk-YCGY7UIZ.js.map → chunk-7GRVHU22.js.map} +0 -0
  246. /package/dist/{chunk-Z7ML6L77.js.map → chunk-HWST3SED.js.map} +0 -0
  247. /package/dist/{chunk-SYDW647C.js.map → chunk-IA2FBTWC.js.map} +0 -0
  248. /package/dist/{chunk-R67DFVLO.js.map → chunk-IFVINJ4B.js.map} +0 -0
  249. /package/dist/{chunk-R36V2VP7.js.map → chunk-IZ5M6OAC.js.map} +0 -0
  250. /package/dist/{chunk-ODT47UAY.js.map → chunk-K3BQGZCT.js.map} +0 -0
  251. /package/dist/{chunk-IFAV6KEM.js.map → chunk-KP5KD6EN.js.map} +0 -0
  252. /package/dist/{chunk-ZEWMTR5M.js.map → chunk-MQMRLGOG.js.map} +0 -0
  253. /package/dist/{chunk-TSWPNOYM.js.map → chunk-NQG5XDSB.js.map} +0 -0
  254. /package/dist/{chunk-7WSD27QQ.js.map → chunk-PB64GYIG.js.map} +0 -0
  255. /package/dist/{chunk-HBSWHQNJ.js.map → chunk-RCYQEFNX.js.map} +0 -0
  256. /package/dist/{chunk-J3KDJNX2.js.map → chunk-RH5F53JT.js.map} +0 -0
  257. /package/dist/{chunk-UAIOHCUK.js.map → chunk-SFLA7OH3.js.map} +0 -0
  258. /package/dist/{chunk-Y6O2OCUO.js.map → chunk-SHM6MRRF.js.map} +0 -0
  259. /package/dist/{chunk-KDIKRJGB.js.map → chunk-SHYIRB7I.js.map} +0 -0
  260. /package/dist/{chunk-HHXFIHXC.js.map → chunk-SVR2LKYI.js.map} +0 -0
  261. /package/dist/{chunk-5SBJCB6W.js.map → chunk-V7AEBY6U.js.map} +0 -0
  262. /package/dist/{chunk-2PVVP7GN.js.map → chunk-XKEFIFIC.js.map} +0 -0
  263. /package/dist/{chunk-JRWWGMK7.js.map → chunk-XYA4XSNU.js.map} +0 -0
  264. /package/dist/{chunk-X5YKXC6V.js.map → chunk-YSMEKBTD.js.map} +0 -0
  265. /package/dist/{chunk-2XU6OGEN.js.map → chunk-Z4TZ76N7.js.map} +0 -0
@@ -0,0 +1,177 @@
1
+ /**
2
+ * GEN-4 proposer-model provenance — pin each author seat's MODEL IDENTITY in
3
+ * the run record at t=0, before any author shot fires.
4
+ *
5
+ * Why: a proposer spec may pin a model explicitly (`spec.model` → the harness
6
+ * CLI's `-m` flag via the author profile's `model.default`), but the claude
7
+ * seat deliberately does NOT pass `-m` — the CLI runs on its logged-in
8
+ * account, whose resolved model comes from its own settings. Verified against
9
+ * claude CLI 2.1.217: `--model` IS supported headless (`-p`), but the run's
10
+ * provenance must not depend on a flag we chose not to send. So the capture
11
+ * records what is VERIFIABLE at launch for every configured harness:
12
+ *
13
+ * - `<harness> --version` output (the exact CLI build that authored),
14
+ * - the claude settings default model (`~/.claude/settings.json` `model`),
15
+ * - `codex login status` (the codex seat must be authed or the run refuses
16
+ * at t=0 — a mid-run auth failure would silently kill one candidate slot),
17
+ * - the spec's pinned model id (null when the seat rides the CLI default).
18
+ *
19
+ * The capture FAILS LOUD on a missing/broken harness binary: populationSize
20
+ * must equal proposers.length, so a dead seat cannot be skipped at runtime —
21
+ * the config decides membership, this guard proves it at launch.
22
+ */
23
+
24
+ import { readFileSync } from 'node:fs'
25
+ import { homedir } from 'node:os'
26
+ import { join } from 'node:path'
27
+ import {
28
+ DEFAULT_GEPA_PYTHON,
29
+ isGepaSeat,
30
+ loadGepaMethodFactory,
31
+ probeGepaRuntime,
32
+ type CampaignModuleImport,
33
+ } from './gepa-seat.mts'
34
+ import { run } from './proc.ts'
35
+ import type { ProposerSpec } from './proposer-fanout.mts'
36
+
37
+ export interface ProposerModelProvenance {
38
+ name: string
39
+ /** Absent on a GEN-6 engine seat (see `engine`). */
40
+ harness: ProposerSpec['harness']
41
+ /** Explicit model pin from the spec (threaded as `-m`), or null when the
42
+ * seat runs the CLI's own resolved default. */
43
+ pinnedModel: string | null
44
+ /** `<harness> --version` stdout (trimmed). For a GEN-6 gepa seat this is
45
+ * the bridge python's `--version` output — the runtime that authors. */
46
+ harnessVersion: string
47
+ /** claude seats only: the settings default model the logged-in CLI resolves
48
+ * when no `-m` is passed. Null when unreadable (recorded, never fatal —
49
+ * the version capture is the hard gate). */
50
+ settingsModel: string | null
51
+ /** codex seats only: `codex login status` stdout (trimmed). */
52
+ authStatus: string | null
53
+ merge: boolean
54
+ /** GEN-6 gepa seat: the engine name from the spec. */
55
+ engine?: 'gepa' | 'omni'
56
+ /** GEN-6 gepa seat: the ONE change-space file GEPA optimizes. */
57
+ surface?: string
58
+ /** GEN-6 gepa seat: installed gepa version ('source' for a source pin). */
59
+ gepaVersion?: string
60
+ /** GEN-6 gepa seat: the Python bridge module the seat runs. */
61
+ bridge?: string
62
+ }
63
+
64
+ export interface ProvenanceCaptureRecord {
65
+ schema: 'swe-arena.proposer-provenance.v1'
66
+ capturedAt: string
67
+ proposers: ProposerModelProvenance[]
68
+ }
69
+
70
+ /** Exec seam — test-injectable. Mirrors `run` from proc.ts. */
71
+ export type VersionExec = (
72
+ command: string,
73
+ args: string[],
74
+ ) => Promise<{ code: number | null; stdout: string; stderr: string }>
75
+
76
+ const defaultExec: VersionExec = async (command, args) => {
77
+ const res = await run(command, args, { timeoutMs: 30_000 })
78
+ return { code: res.code, stdout: res.stdout, stderr: res.stderr }
79
+ }
80
+
81
+ /** Read the claude CLI's settings default model. Pure over injected reader. */
82
+ export function claudeSettingsModel(
83
+ readFile: (path: string) => string = (p) => readFileSync(p, 'utf8'),
84
+ settingsPath = join(homedir(), '.claude', 'settings.json'),
85
+ ): string | null {
86
+ try {
87
+ const parsed = JSON.parse(readFile(settingsPath)) as { model?: unknown }
88
+ return typeof parsed.model === 'string' && parsed.model.length > 0 ? parsed.model : null
89
+ } catch {
90
+ return null
91
+ }
92
+ }
93
+
94
+ /** Capture per-proposer model provenance. Throws when any configured harness
95
+ * binary is missing/broken, when a codex seat is not logged in, or — GEN-6 —
96
+ * when a gepa seat's runtime is incomplete: the installed agent-eval must
97
+ * export `gepaOptimizationMethod` and the Python bridge + GEPA engine must
98
+ * import (probeGepaRuntime carries the exact install instructions). A dead
99
+ * seat fails the launch at t=0, never a mid-run candidate slot. */
100
+ export async function captureProposerProvenance(
101
+ proposers: ProposerSpec[],
102
+ deps: {
103
+ exec?: VersionExec
104
+ readSettingsModel?: () => string | null
105
+ importCampaign?: CampaignModuleImport
106
+ } = {},
107
+ ): Promise<ProvenanceCaptureRecord> {
108
+ const exec = deps.exec ?? defaultExec
109
+ const readSettingsModel = deps.readSettingsModel ?? (() => claudeSettingsModel())
110
+ const versionByHarness = new Map<string, string>()
111
+ const authByHarness = new Map<string, string>()
112
+ const gepaBySeat = new Map<string, { pythonVersion: string; gepaVersion: string }>()
113
+ const gepaSeats = proposers.filter(isGepaSeat)
114
+ if (gepaSeats.length > 0) {
115
+ // Node side first: the adapter export (fails loud with upgrade hint).
116
+ await loadGepaMethodFactory(...(deps.importCampaign ? [deps.importCampaign] : []))
117
+ for (const seat of gepaSeats) {
118
+ gepaBySeat.set(seat.name, await probeGepaRuntime(seat.python ?? DEFAULT_GEPA_PYTHON, exec, seat.name))
119
+ }
120
+ }
121
+ const harnesses = [...new Set(proposers.map((p) => p.harness))].filter(
122
+ (h): h is NonNullable<ProposerSpec['harness']> => h !== undefined,
123
+ )
124
+ for (const harness of harnesses) {
125
+ const res = await exec(harness, ['--version'])
126
+ if (res.code !== 0) {
127
+ throw new Error(
128
+ `proposer provenance: '${harness} --version' failed (rc=${res.code}) — the ${harness} seat cannot author. ` +
129
+ `stderr: ${res.stderr.slice(0, 300)}`,
130
+ )
131
+ }
132
+ versionByHarness.set(harness, res.stdout.trim())
133
+ if (harness === 'codex') {
134
+ const auth = await exec('codex', ['login', 'status'])
135
+ const authOut = auth.stdout + auth.stderr
136
+ const authed = /logged in/i.test(authOut) && !/not logged in/i.test(authOut)
137
+ if (auth.code !== 0 || !authed) {
138
+ throw new Error(
139
+ `proposer provenance: codex seat configured but 'codex login status' says not authed ` +
140
+ `(rc=${auth.code}, out=${(auth.stdout + auth.stderr).trim().slice(0, 200)})`,
141
+ )
142
+ }
143
+ authByHarness.set('codex', (auth.stdout + auth.stderr).trim())
144
+ }
145
+ }
146
+ return {
147
+ schema: 'swe-arena.proposer-provenance.v1',
148
+ capturedAt: new Date().toISOString(),
149
+ proposers: proposers.map((spec): ProposerModelProvenance => {
150
+ if (isGepaSeat(spec)) {
151
+ const probe = gepaBySeat.get(spec.name)!
152
+ return {
153
+ name: spec.name,
154
+ harness: undefined,
155
+ pinnedModel: null,
156
+ harnessVersion: probe.pythonVersion,
157
+ settingsModel: null,
158
+ authStatus: null,
159
+ merge: false,
160
+ engine: spec.engine,
161
+ surface: spec.surface,
162
+ gepaVersion: probe.gepaVersion,
163
+ bridge: 'agent_eval_rpc.gepa_bridge',
164
+ }
165
+ }
166
+ return {
167
+ name: spec.name,
168
+ harness: spec.harness,
169
+ pinnedModel: spec.model ?? null,
170
+ harnessVersion: versionByHarness.get(spec.harness!)!,
171
+ settingsModel: spec.harness === 'claude' && !spec.model ? readSettingsModel() : null,
172
+ authStatus: (spec.harness !== undefined ? authByHarness.get(spec.harness) : undefined) ?? null,
173
+ merge: spec.merge === true,
174
+ }
175
+ }),
176
+ }
177
+ }
@@ -0,0 +1,106 @@
1
+ import { describe, expect, it } from 'vitest'
2
+ import {
3
+ captureProposerProvenance,
4
+ claudeSettingsModel,
5
+ type VersionExec,
6
+ } from './proposer-provenance.mts'
7
+ import type { ProposerSpec } from './proposer-fanout.mts'
8
+
9
+ const ok = (stdout: string) => ({ code: 0, stdout, stderr: '' })
10
+
11
+ const gen4ish: ProposerSpec[] = [
12
+ { name: 'claude-author', profile: 'default-author.profile.json', harness: 'claude' },
13
+ { name: 'glm-author', harness: 'opencode', model: 'zai-coding-plan/glm-5.2' },
14
+ { name: 'codex-author', harness: 'codex' },
15
+ { name: 'merge-author', profile: 'default-author.profile.json', harness: 'claude', merge: true },
16
+ ]
17
+
18
+ describe('captureProposerProvenance', () => {
19
+ const exec: VersionExec = async (command, args) => {
20
+ if (args[0] === '--version') return ok(`${command}-version 9.9.9`)
21
+ if (command === 'codex' && args[0] === 'login') return ok('Logged in using ChatGPT')
22
+ throw new Error(`unexpected exec ${command} ${args.join(' ')}`)
23
+ }
24
+
25
+ it('records pinned model, harness version, settings model, and codex auth per seat', async () => {
26
+ const record = await captureProposerProvenance(gen4ish, {
27
+ exec,
28
+ readSettingsModel: () => 'claude-fable-5',
29
+ })
30
+ expect(record.schema).toBe('swe-arena.proposer-provenance.v1')
31
+ const byName = Object.fromEntries(record.proposers.map((p) => [p.name, p]))
32
+ // Claude seat: no pin — the CLI's resolved settings model is the record.
33
+ expect(byName['claude-author']).toMatchObject({
34
+ harness: 'claude',
35
+ pinnedModel: null,
36
+ settingsModel: 'claude-fable-5',
37
+ harnessVersion: 'claude-version 9.9.9',
38
+ authStatus: null,
39
+ merge: false,
40
+ })
41
+ // Pinned opencode seat: explicit model id, settings model not consulted.
42
+ expect(byName['glm-author']).toMatchObject({
43
+ harness: 'opencode',
44
+ pinnedModel: 'zai-coding-plan/glm-5.2',
45
+ settingsModel: null,
46
+ harnessVersion: 'opencode-version 9.9.9',
47
+ })
48
+ // Codex seat: version + auth status captured.
49
+ expect(byName['codex-author']).toMatchObject({
50
+ harness: 'codex',
51
+ pinnedModel: null,
52
+ authStatus: 'Logged in using ChatGPT',
53
+ })
54
+ expect(byName['merge-author']!.merge).toBe(true)
55
+ })
56
+
57
+ it('fails loud when a configured harness binary is missing', async () => {
58
+ const broken: VersionExec = async (command, args) =>
59
+ command === 'codex' ? { code: 127, stdout: '', stderr: 'not found' } : exec(command, args)
60
+ await expect(
61
+ captureProposerProvenance(gen4ish, { exec: broken, readSettingsModel: () => null }),
62
+ ).rejects.toThrow(/'codex --version' failed/)
63
+ })
64
+
65
+ it('fails loud when the codex seat is not logged in', async () => {
66
+ const loggedOut: VersionExec = async (command, args) => {
67
+ if (args[0] === '--version') return ok(`${command} 1.0.0`)
68
+ return ok('Not logged in')
69
+ }
70
+ await expect(
71
+ captureProposerProvenance([{ name: 'codex-author', harness: 'codex' }], {
72
+ exec: loggedOut,
73
+ readSettingsModel: () => null,
74
+ }),
75
+ ).rejects.toThrow(/not authed/)
76
+ })
77
+
78
+ it('runs one version probe per harness, not per proposer', async () => {
79
+ const calls: string[] = []
80
+ const counting: VersionExec = async (command, args) => {
81
+ calls.push(`${command} ${args.join(' ')}`)
82
+ if (args[0] === '--version') return ok(`${command} 1`)
83
+ return ok('Logged in using ChatGPT')
84
+ }
85
+ await captureProposerProvenance(gen4ish, { exec: counting, readSettingsModel: () => null })
86
+ expect(calls.filter((c) => c === 'claude --version')).toHaveLength(1)
87
+ expect(calls.filter((c) => c === 'codex --version')).toHaveLength(1)
88
+ expect(calls.filter((c) => c === 'codex login status')).toHaveLength(1)
89
+ })
90
+ })
91
+
92
+ describe('claudeSettingsModel', () => {
93
+ it('reads the settings model field', () => {
94
+ expect(claudeSettingsModel(() => JSON.stringify({ model: 'claude-fable-5' }), '/x')).toBe('claude-fable-5')
95
+ })
96
+
97
+ it('returns null on unreadable/missing/blank settings', () => {
98
+ expect(
99
+ claudeSettingsModel(() => {
100
+ throw new Error('ENOENT')
101
+ }, '/x'),
102
+ ).toBeNull()
103
+ expect(claudeSettingsModel(() => JSON.stringify({}), '/x')).toBeNull()
104
+ expect(claudeSettingsModel(() => JSON.stringify({ model: '' }), '/x')).toBeNull()
105
+ })
106
+ })
Binary file
@@ -0,0 +1,183 @@
1
+ /**
2
+ * Replay CLI: `tsx src/swe-arena/replay.mts`
3
+ *
4
+ * Reproduces the reference `fixtures/analyze.py` output from the committed
5
+ * fixtures (section 1 matches its printed lines exactly — pinned in
6
+ * replay.test.mts), then prints what the reference script never did:
7
+ * the reconciled valid-denominator verdict, the true SUP spend including
8
+ * worker tokens, the supervisor evolution rounds, and the holdout registry.
9
+ */
10
+
11
+ import { pathToFileURL } from 'node:url'
12
+ import {
13
+ loadHoldout,
14
+ loadLedger,
15
+ loadPreregisterLog,
16
+ loadRejudge,
17
+ loadRematchRounds,
18
+ loadSupJournalTrue,
19
+ loadWorkerTokens,
20
+ } from './fixtures.ts'
21
+ import {
22
+ costRollup,
23
+ ledgerOutcomes,
24
+ pairedSignTest,
25
+ reconciledOutcomes,
26
+ roundsProgression,
27
+ splitPairs,
28
+ type CostRollup,
29
+ type DiscordantSplit,
30
+ type RoundState,
31
+ type SignTestResult,
32
+ } from './analyze.ts'
33
+ import { reconcile, type PairedTable } from './reconcile.ts'
34
+ import type { HoldoutRegistry, LedgerRow, RematchRow } from './types.ts'
35
+
36
+ /** Python-style list repr (single quotes) so section 1 matches analyze.py byte-for-byte. */
37
+ const pyList = (xs: string[]): string => `[${xs.map((x) => `'${x}'`).join(', ')}]`
38
+ const signed = (x: number, digits?: number): string =>
39
+ (x >= 0 ? '+' : '') + (digits === undefined ? String(x) : x.toFixed(digits))
40
+
41
+ export interface Replay {
42
+ ledger: LedgerRow[]
43
+ raw: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
44
+ cost: CostRollup
45
+ table: PairedTable
46
+ valid: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
47
+ rounds: Map<string, RoundState[]>
48
+ rematchRounds: RematchRow[][]
49
+ holdout: HoldoutRegistry
50
+ preregisterLog: string[]
51
+ }
52
+
53
+ export function buildReplay(): Replay {
54
+ const ledger = loadLedger()
55
+ const rejudge = loadRejudge()
56
+ const rematchRounds = loadRematchRounds()
57
+
58
+ const rawOutcomes = ledgerOutcomes(ledger)
59
+ const table = reconcile(ledger, rejudge)
60
+ const validOutcomes = reconciledOutcomes(table.valid)
61
+
62
+ return {
63
+ ledger,
64
+ raw: {
65
+ split: splitPairs(rawOutcomes),
66
+ sign: pairedSignTest(rawOutcomes),
67
+ soloResolved: rawOutcomes.filter((o) => o.solo).length,
68
+ supResolved: rawOutcomes.filter((o) => o.sup).length,
69
+ },
70
+ cost: costRollup(ledger, loadSupJournalTrue(), loadWorkerTokens()),
71
+ table,
72
+ valid: {
73
+ split: splitPairs(validOutcomes),
74
+ sign: pairedSignTest(validOutcomes),
75
+ soloResolved: validOutcomes.filter((o) => o.solo).length,
76
+ supResolved: validOutcomes.filter((o) => o.sup).length,
77
+ },
78
+ rounds: roundsProgression(ledger, rematchRounds),
79
+ rematchRounds,
80
+ holdout: loadHoldout(),
81
+ preregisterLog: loadPreregisterLog(),
82
+ }
83
+ }
84
+
85
+ /** Section 1 — byte-faithful reproduction of analyze.py's printed analysis. */
86
+ export function renderReference(r: Replay): string[] {
87
+ const rows = [...r.ledger].sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
88
+ const n = rows.length
89
+ const { raw, cost } = r
90
+ const bar = '='.repeat(80)
91
+ const lines: string[] = []
92
+ lines.push(bar)
93
+ lines.push(`PAIRED HEAD-TO-HEAD: glm-5.2 SOLO vs glm-5.2 SUPERVISOR (N=${n} paired instances)`)
94
+ lines.push(bar)
95
+ lines.push(`SOLO resolved: ${raw.soloResolved}/${n} = ${((100 * raw.soloResolved) / n).toFixed(1)}%`)
96
+ lines.push(`SUP resolved: ${raw.supResolved}/${n} = ${((100 * raw.supResolved) / n).toFixed(1)}%`)
97
+ const delta = raw.supResolved - raw.soloResolved
98
+ lines.push(`delta (SUP-SOLO): ${signed(delta)} instances (${signed((100 * delta) / n, 1)} pts)`)
99
+ lines.push('')
100
+ lines.push('DISCORDANT PAIRS (the signal):')
101
+ lines.push(` SUP-only wins (SUP✓ SOLO✗): ${raw.split.supOnly.length} ${pyList(raw.split.supOnly)}`)
102
+ lines.push(` SOLO-only wins (SOLO✓ SUP✗): ${raw.split.soloOnly.length} ${pyList(raw.split.soloOnly)}`)
103
+ lines.push(` both resolved: ${raw.split.both.length} | neither: ${raw.split.neither.length} ${pyList(raw.split.neither)}`)
104
+ lines.push(` exact two-sided sign test on discordant pairs: p=${raw.sign.pValue.toFixed(4)}`)
105
+ lines.push('')
106
+ lines.push(
107
+ `COST (measured tokens; USD via shared blended rate $${(cost.blendedRatePerTok * 1e6).toFixed(3)}/1M from SUP accounting):`,
108
+ )
109
+ lines.push(` SOLO total tokens: ${cost.soloTokens.toLocaleString('en-US')} -> derived $${cost.soloUsdDerived.toFixed(4)}`)
110
+ lines.push(` SUP total tokens: ${cost.supBrainTokens.toLocaleString('en-US')} -> runtime $${cost.supUsd.toFixed(4)}`)
111
+ lines.push(` SUP/SOLO token ratio: ${cost.brainTokenRatio.toFixed(2)}x`)
112
+ lines.push(` SUP/SOLO cost ratio (token-derived): ${(cost.supUsd / cost.soloUsdDerived).toFixed(2)}x`)
113
+ lines.push(` [telemetry] instances where runtime spentTokens != journal-true (no-winner zeroing): ${pyList(cost.telemetryGaps)}`)
114
+ lines.push(` WALL: SOLO ${cost.soloWallS}s total vs SUP ${cost.supWallS}s total -> SUP ${cost.wallRatio.toFixed(2)}x wall`)
115
+ lines.push('')
116
+ lines.push(bar)
117
+ lines.push('PER-INSTANCE')
118
+ lines.push(bar)
119
+ lines.push(
120
+ `${'instance'.padEnd(32)} ${'SOLO'.padEnd(5)} ${'SUP'.padEnd(5)} ${'v_s'.padEnd(3)} ${'v_p'.padEnd(3)} ${'wrk'.padEnd(3)} ${'soloTok'.padEnd(8)} ${'supTok'.padEnd(8)} ${'supUSD'.padEnd(7)} ${'soloW'.padEnd(5)} ${'supW'.padEnd(5)}`,
121
+ )
122
+ for (const row of rows) {
123
+ const pyBool = (v: boolean): string => (v ? 'True' : 'False')
124
+ lines.push(
125
+ `${row.iid.padEnd(32)} ${pyBool(row.solo_resolved).slice(0, 5).padEnd(5)} ${pyBool(row.sup_resolved).slice(0, 5).padEnd(5)} ` +
126
+ `${pyBool(row.solo_verify_pass)[0].padEnd(3)} ${pyBool(row.sup_verify_pass)[0].padEnd(3)} ` +
127
+ `${String(row.sup_workers ?? '?').padEnd(3)} ${String(row.solo_tokens).padEnd(8)} ${String(row.sup_spentTokens ?? 0).padEnd(8)} ` +
128
+ `${(row.sup_spentUsd ?? 0).toFixed(4)} ${String(row.solo_wall_s).padEnd(5)} ${String(row.sup_wall_s).padEnd(5)}`,
129
+ )
130
+ }
131
+ return lines
132
+ }
133
+
134
+ /** Sections 2-5 — the analysis that lived in session lore, now typed. */
135
+ export function renderReconciled(r: Replay): string[] {
136
+ const bar = '='.repeat(80)
137
+ const lines: string[] = []
138
+ const { table, valid, cost } = r
139
+ const n = table.valid.length
140
+
141
+ lines.push(bar)
142
+ lines.push('RECONCILED VERDICT (re-judged, gold-gated denominator)')
143
+ lines.push(bar)
144
+ for (const e of table.excluded) lines.push(` EXCLUDED ${e.iid}: ${e.excludeReason}`)
145
+ for (const v of table.valid) {
146
+ const src = [v.solo.source !== 'ledger' ? `solo:${v.solo.source}` : null, v.sup.source !== 'ledger' ? `sup:${v.sup.source}` : null]
147
+ .filter(Boolean)
148
+ .join(' ')
149
+ if (src) lines.push(` RE-JUDGED ${v.iid}: ${src}`)
150
+ }
151
+ lines.push(`SOLO resolved: ${valid.soloResolved}/${n}`)
152
+ lines.push(`SUP resolved: ${valid.supResolved}/${n}`)
153
+ lines.push(`discordant: SUP-only ${pyList(valid.split.supOnly)} | SOLO-only ${pyList(valid.split.soloOnly)}`)
154
+ lines.push(`exact two-sided sign test: p=${valid.sign.pValue.toFixed(4)}`)
155
+ lines.push('')
156
+ lines.push('TRUE SUP SPEND (brain + workers; analyze.py printed brain only):')
157
+ lines.push(` brain ${cost.supBrainTokens.toLocaleString('en-US')} + workers ${cost.supWorkerTokens.toLocaleString('en-US')} = ${cost.supTotalTokens.toLocaleString('en-US')} tokens`)
158
+ lines.push(` SUP/SOLO true token ratio: ${cost.totalTokenRatio.toFixed(2)}x (brain-only ratio: ${cost.brainTokenRatio.toFixed(2)}x)`)
159
+ lines.push('')
160
+ lines.push('SUP EVOLUTION ROUNDS (SUP = original head-to-head run):')
161
+ for (const [iid, states] of r.rounds) {
162
+ const cells = states.map(
163
+ (s) => `${s.round}:${s.resolved ? 'RESOLVED' : 'unresolved'}(${s.patchLines}L,${s.verdict ?? 'null'})`,
164
+ )
165
+ lines.push(` ${iid.padEnd(32)} ${cells.join(' -> ')}`)
166
+ }
167
+ lines.push('')
168
+ lines.push(`HOLDOUT REGISTRY (pre-registered at loops@${r.holdout.selectedAtCommit.slice(0, 10)}, untouched):`)
169
+ for (const e of r.holdout.entries) {
170
+ lines.push(` ${e.iid.padEnd(36)} gold_official_resolved=${e.gold_official_resolved} verify_calibrated=${e.verify_calibrated}`)
171
+ }
172
+ return lines
173
+ }
174
+
175
+ export function renderReplay(r: Replay = buildReplay()): string {
176
+ return [...renderReference(r), '', ...renderReconciled(r)].join('\n')
177
+ }
178
+
179
+ const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
180
+
181
+ if (isMain) {
182
+ console.log(renderReplay())
183
+ }