@tangle-network/agent-bench 0.3.6 → 0.3.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/HARNESS.md +43 -0
- package/dist/adapters.js +24 -24
- package/dist/benchmarks/_harness.d.ts +1 -1
- package/dist/benchmarks/_harness.js +1 -1
- package/dist/benchmarks/aec-bench.js +2 -2
- package/dist/benchmarks/agentbench.js +2 -2
- package/dist/benchmarks/appworld.js +2 -2
- package/dist/benchmarks/bfcl.js +2 -2
- package/dist/benchmarks/commit0.js +2 -2
- package/dist/benchmarks/crag.js +2 -2
- package/dist/benchmarks/dabstep.js +2 -2
- package/dist/benchmarks/enterpriseops-gym.js +2 -2
- package/dist/benchmarks/finresearchbench.js +2 -2
- package/dist/benchmarks/humaneval.d.ts +10 -1
- package/dist/benchmarks/humaneval.js +5 -3
- package/dist/benchmarks/nomiracl.js +2 -2
- package/dist/benchmarks/open-rag-bench.js +2 -2
- package/dist/benchmarks/programbench.js +2 -2
- package/dist/benchmarks/ragbench.js +2 -2
- package/dist/benchmarks/swe-bench.js +2 -2
- package/dist/benchmarks/t2-ragbench.js +2 -2
- package/dist/benchmarks/tau-bench-shared.js +2 -2
- package/dist/benchmarks/tau2-bench.js +3 -3
- package/dist/benchmarks/tau3-banking.js +3 -3
- package/dist/benchmarks/terminal-bench.js +2 -2
- package/dist/benchmarks/toollm.js +2 -2
- package/dist/benchmarks/webarena-verified.js +2 -2
- package/dist/{chunk-CKUVRZ2T.js → chunk-3U5TXJZS.js} +2 -2
- package/dist/{chunk-PPYSEKFM.js → chunk-5H5XV76F.js} +73 -15
- package/dist/chunk-5H5XV76F.js.map +1 -0
- package/dist/{chunk-YCGY7UIZ.js → chunk-7GRVHU22.js} +2 -2
- package/dist/{chunk-Z7ML6L77.js → chunk-HWST3SED.js} +2 -2
- package/dist/{chunk-SYDW647C.js → chunk-IA2FBTWC.js} +2 -2
- package/dist/{chunk-R67DFVLO.js → chunk-IFVINJ4B.js} +2 -2
- package/dist/{chunk-R36V2VP7.js → chunk-IZ5M6OAC.js} +2 -2
- package/dist/{chunk-ODT47UAY.js → chunk-K3BQGZCT.js} +2 -2
- package/dist/{chunk-IFAV6KEM.js → chunk-KP5KD6EN.js} +2 -2
- package/dist/{chunk-ZEWMTR5M.js → chunk-MQMRLGOG.js} +2 -2
- package/dist/{chunk-TSWPNOYM.js → chunk-NQG5XDSB.js} +2 -2
- package/dist/{chunk-7WSD27QQ.js → chunk-PB64GYIG.js} +2 -2
- package/dist/{chunk-HBSWHQNJ.js → chunk-RCYQEFNX.js} +3 -3
- package/dist/{chunk-J3KDJNX2.js → chunk-RH5F53JT.js} +2 -2
- package/dist/{chunk-UAIOHCUK.js → chunk-SFLA7OH3.js} +3 -3
- package/dist/{chunk-Y6O2OCUO.js → chunk-SHM6MRRF.js} +2 -2
- package/dist/{chunk-KDIKRJGB.js → chunk-SHYIRB7I.js} +2 -2
- package/dist/{chunk-HHXFIHXC.js → chunk-SVR2LKYI.js} +2 -2
- package/dist/{chunk-5SBJCB6W.js → chunk-V7AEBY6U.js} +22 -22
- package/dist/{chunk-LRRD7NAG.js → chunk-WSKWVEQB.js} +18 -2
- package/dist/chunk-WSKWVEQB.js.map +1 -0
- package/dist/{chunk-2PVVP7GN.js → chunk-XKEFIFIC.js} +2 -2
- package/dist/{chunk-JRWWGMK7.js → chunk-XYA4XSNU.js} +2 -2
- package/dist/{chunk-X5YKXC6V.js → chunk-YSMEKBTD.js} +2 -2
- package/dist/{chunk-2XU6OGEN.js → chunk-Z4TZ76N7.js} +2 -2
- package/dist/index.js +24 -24
- package/package.json +6 -5
- package/scripts/run-package-tests.mjs +30 -8
- package/scripts/verify-packed-consumer.mjs +1 -1
- package/scripts/verify-pier-agent.mts +1 -0
- package/src/benchmarks/_harness.ts +20 -2
- package/src/benchmarks/humaneval.test.mts +122 -0
- package/src/benchmarks/humaneval.ts +100 -27
- package/src/david-attribution.mts +78 -0
- package/src/david-goliath.mts +149 -0
- package/src/hev-improve.mts +25 -6
- package/src/humaneval-object-ablation.mts +201 -0
- package/src/live-improve-campaign-mbpp.mts +641 -0
- package/src/live-improve-campaign.mts +500 -0
- package/src/mbpp-structural.mts +12 -7
- package/src/quant-arena/README.md +144 -0
- package/src/quant-arena/backtest.test.mts +135 -0
- package/src/quant-arena/backtest.ts +218 -0
- package/src/quant-arena/data.test.mts +44 -0
- package/src/quant-arena/data.ts +141 -0
- package/src/quant-arena/driver.test.mts +253 -0
- package/src/quant-arena/driver.ts +219 -0
- package/src/quant-arena/fixtures/data/PROVENANCE.md +26 -0
- package/src/quant-arena/fixtures/data/holdout/IDX.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S01.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S02.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S03.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S04.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S05.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S06.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S07.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S08.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S09.csv +523 -0
- package/src/quant-arena/fixtures/data/holdout/S10.csv +523 -0
- package/src/quant-arena/fixtures/data/insample/IDX.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S01.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S02.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S03.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S04.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S05.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S06.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S07.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S08.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S09.csv +2087 -0
- package/src/quant-arena/fixtures/data/insample/S10.csv +2087 -0
- package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +16 -0
- package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +5 -0
- package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +171 -0
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +119 -0
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +119 -0
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +105 -0
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +102 -0
- package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +4 -0
- package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +2 -0
- package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +84 -0
- package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +117 -0
- package/src/quant-arena/holdout-certify.mts +206 -0
- package/src/quant-arena/holdout-certify.test.mts +82 -0
- package/src/quant-arena/leak-audit.test.mts +79 -0
- package/src/quant-arena/leak-audit.ts +95 -0
- package/src/quant-arena/make-fixtures.mts +161 -0
- package/src/quant-arena/multiplicity.test.mts +68 -0
- package/src/quant-arena/multiplicity.ts +87 -0
- package/src/quant-arena/nautilus-certify.ts +31 -0
- package/src/quant-arena/oms.ts +90 -0
- package/src/quant-arena/profiles/quant-researcher.profile.json +7 -0
- package/src/quant-arena/python/pyproject.toml +8 -0
- package/src/quant-arena/python/uv.lock +1297 -0
- package/src/quant-arena/python/vbt-worker.py +192 -0
- package/src/quant-arena/quant-loop.mts +813 -0
- package/src/quant-arena/quant-loop.test.mts +75 -0
- package/src/quant-arena/strategies/buy-hold-index/strategy.ts +11 -0
- package/src/quant-arena/strategies/equal-weight/strategy.ts +20 -0
- package/src/quant-arena/strategies/sma-crossover/strategy.ts +42 -0
- package/src/quant-arena/types.ts +133 -0
- package/src/quant-arena/vbt-client.ts +321 -0
- package/src/quant-arena/vbt-parity.test.mts +183 -0
- package/src/quant-arena/windows.test.mts +45 -0
- package/src/quant-arena/windows.ts +54 -0
- package/src/rollout-ledger/backfill-swe-arena.mts +606 -0
- package/src/rollout-ledger/backfill-swe-arena.test.mts +338 -0
- package/src/rollout-ledger/settle-capture.mts +442 -0
- package/src/rollout-ledger/settle-capture.test.mts +270 -0
- package/src/stream-observe.py +45 -0
- package/src/stream-observe.tpl.html +247 -0
- package/src/supervisor-arena.mts +816 -0
- package/src/swe-arena/activation.mts +228 -0
- package/src/swe-arena/activation.test.mts +303 -0
- package/src/swe-arena/analyze.ts +211 -0
- package/src/swe-arena/arms.ts +804 -0
- package/src/swe-arena/bootstrap-meta.mts +188 -0
- package/src/swe-arena/bootstrap-meta.test.mts +51 -0
- package/src/swe-arena/briefing.mts +217 -0
- package/src/swe-arena/briefing.test.mts +178 -0
- package/src/swe-arena/calibrate.ts +217 -0
- package/src/swe-arena/capabilities.mts +76 -0
- package/src/swe-arena/capabilities.test.mts +57 -0
- package/src/swe-arena/capacity.ts +194 -0
- package/src/swe-arena/cell-evidence.mts +437 -0
- package/src/swe-arena/cell-evidence.test.mts +248 -0
- package/src/swe-arena/diagnosis-ensemble.test.mts +210 -0
- package/src/swe-arena/diagnosis-ensemble.ts +520 -0
- package/src/swe-arena/execution.test.mts +1170 -0
- package/src/swe-arena/factory-command-container.ts +284 -0
- package/src/swe-arena/factory-judge-child.mts +228 -0
- package/src/swe-arena/factory.test.mts +643 -0
- package/src/swe-arena/fixtures/analyze.py +80 -0
- package/src/swe-arena/fixtures/excludes.txt +8 -0
- package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +51 -0
- package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +29 -0
- package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +64 -0
- package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +48 -0
- package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +29 -0
- package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +48 -0
- package/src/swe-arena/fixtures/factory/loops-28/calibration.md +47 -0
- package/src/swe-arena/fixtures/factory/loops-28/manifest.json +30 -0
- package/src/swe-arena/fixtures/factory/loops-28/spec.md +50 -0
- package/src/swe-arena/fixtures/gen1-salvage/README.md +45 -0
- package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +116 -0
- package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +293 -0
- package/src/swe-arena/fixtures/holdout-preregister.log +12 -0
- package/src/swe-arena/fixtures/holdout.json +44 -0
- package/src/swe-arena/fixtures/instances.json +146 -0
- package/src/swe-arena/fixtures/ledger.jsonl +12 -0
- package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +36 -0
- package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +33 -0
- package/src/swe-arena/fixtures/rejudge.jsonl +15 -0
- package/src/swe-arena/fixtures/rematch.jsonl +3 -0
- package/src/swe-arena/fixtures/rematch2.jsonl +3 -0
- package/src/swe-arena/fixtures/rematch3.jsonl +3 -0
- package/src/swe-arena/fixtures/run-report/README.md +43 -0
- package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +173 -0
- package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +100 -0
- package/src/swe-arena/fixtures/run-report/gen3-rollup.json +551 -0
- package/src/swe-arena/fixtures/run-report/gen3-rollup.md +64 -0
- package/src/swe-arena/fixtures/sup-journal-true.json +19 -0
- package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +48 -0
- package/src/swe-arena/fixtures/verify/django__django-11532.sh +50 -0
- package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +76 -0
- package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +44 -0
- package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +32 -0
- package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +51 -0
- package/src/swe-arena/fixtures/worker-tokens.json +42 -0
- package/src/swe-arena/fixtures.ts +237 -0
- package/src/swe-arena/gepa-seat.mts +583 -0
- package/src/swe-arena/gepa-seat.test.mts +635 -0
- package/src/swe-arena/holdout-certify.mts +408 -0
- package/src/swe-arena/holdout-certify.test.mts +160 -0
- package/src/swe-arena/judge-child.mts +37 -0
- package/src/swe-arena/ledger-orphans.mts +77 -0
- package/src/swe-arena/ledger-orphans.test.mts +147 -0
- package/src/swe-arena/lineage-record.mts +164 -0
- package/src/swe-arena/lineage-record.test.mts +115 -0
- package/src/swe-arena/manifest.mts +293 -0
- package/src/swe-arena/manifest.test.mts +169 -0
- package/src/swe-arena/materialize.ts +142 -0
- package/src/swe-arena/outer-loop.mts +2795 -0
- package/src/swe-arena/outer-loop.test.mts +696 -0
- package/src/swe-arena/parity.test.mts +87 -0
- package/src/swe-arena/premeasured-from-cells.mts +281 -0
- package/src/swe-arena/premeasured-from-cells.test.mts +180 -0
- package/src/swe-arena/proc.test.mts +174 -0
- package/src/swe-arena/proc.ts +260 -0
- package/src/swe-arena/profiles/default-author.profile.json +4 -0
- package/src/swe-arena/proposer-fanout.mts +770 -0
- package/src/swe-arena/proposer-fanout.test.mts +619 -0
- package/src/swe-arena/proposer-provenance.mts +177 -0
- package/src/swe-arena/proposer-provenance.test.mts +106 -0
- package/src/swe-arena/reconcile.ts +0 -0
- package/src/swe-arena/replay.mts +183 -0
- package/src/swe-arena/replay.test.mts +300 -0
- package/src/swe-arena/run-experiment.mts +727 -0
- package/src/swe-arena/run-report.mts +75 -0
- package/src/swe-arena/run-supervisor.mjs +297 -0
- package/src/swe-arena/run-supervisor.test.mts +500 -0
- package/src/swe-arena/score-split.mts +140 -0
- package/src/swe-arena/score-split.test.mts +123 -0
- package/src/swe-arena/serialized-judge.ts +414 -0
- package/src/swe-arena/types.ts +218 -0
- package/src/swe-code-improve.mts +328 -0
- package/src/swe-emit-patch.mts +104 -0
- package/src/swe-improve.mts +232 -0
- package/src/swe-jail.ts +2 -2
- package/src/swe-local-proof.mts +169 -0
- package/src/swe-repro-calibrate.mts +446 -0
- package/src/swe-stream.mts +1497 -0
- package/src/swe-structural.mts +245 -837
- package/dist/chunk-LRRD7NAG.js.map +0 -1
- package/dist/chunk-PPYSEKFM.js.map +0 -1
- /package/dist/{chunk-CKUVRZ2T.js.map → chunk-3U5TXJZS.js.map} +0 -0
- /package/dist/{chunk-YCGY7UIZ.js.map → chunk-7GRVHU22.js.map} +0 -0
- /package/dist/{chunk-Z7ML6L77.js.map → chunk-HWST3SED.js.map} +0 -0
- /package/dist/{chunk-SYDW647C.js.map → chunk-IA2FBTWC.js.map} +0 -0
- /package/dist/{chunk-R67DFVLO.js.map → chunk-IFVINJ4B.js.map} +0 -0
- /package/dist/{chunk-R36V2VP7.js.map → chunk-IZ5M6OAC.js.map} +0 -0
- /package/dist/{chunk-ODT47UAY.js.map → chunk-K3BQGZCT.js.map} +0 -0
- /package/dist/{chunk-IFAV6KEM.js.map → chunk-KP5KD6EN.js.map} +0 -0
- /package/dist/{chunk-ZEWMTR5M.js.map → chunk-MQMRLGOG.js.map} +0 -0
- /package/dist/{chunk-TSWPNOYM.js.map → chunk-NQG5XDSB.js.map} +0 -0
- /package/dist/{chunk-7WSD27QQ.js.map → chunk-PB64GYIG.js.map} +0 -0
- /package/dist/{chunk-HBSWHQNJ.js.map → chunk-RCYQEFNX.js.map} +0 -0
- /package/dist/{chunk-J3KDJNX2.js.map → chunk-RH5F53JT.js.map} +0 -0
- /package/dist/{chunk-UAIOHCUK.js.map → chunk-SFLA7OH3.js.map} +0 -0
- /package/dist/{chunk-Y6O2OCUO.js.map → chunk-SHM6MRRF.js.map} +0 -0
- /package/dist/{chunk-KDIKRJGB.js.map → chunk-SHYIRB7I.js.map} +0 -0
- /package/dist/{chunk-HHXFIHXC.js.map → chunk-SVR2LKYI.js.map} +0 -0
- /package/dist/{chunk-5SBJCB6W.js.map → chunk-V7AEBY6U.js.map} +0 -0
- /package/dist/{chunk-2PVVP7GN.js.map → chunk-XKEFIFIC.js.map} +0 -0
- /package/dist/{chunk-JRWWGMK7.js.map → chunk-XYA4XSNU.js.map} +0 -0
- /package/dist/{chunk-X5YKXC6V.js.map → chunk-YSMEKBTD.js.map} +0 -0
- /package/dist/{chunk-2XU6OGEN.js.map → chunk-Z4TZ76N7.js.map} +0 -0
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* GEN-4 proposer-model provenance — pin each author seat's MODEL IDENTITY in
|
|
3
|
+
* the run record at t=0, before any author shot fires.
|
|
4
|
+
*
|
|
5
|
+
* Why: a proposer spec may pin a model explicitly (`spec.model` → the harness
|
|
6
|
+
* CLI's `-m` flag via the author profile's `model.default`), but the claude
|
|
7
|
+
* seat deliberately does NOT pass `-m` — the CLI runs on its logged-in
|
|
8
|
+
* account, whose resolved model comes from its own settings. Verified against
|
|
9
|
+
* claude CLI 2.1.217: `--model` IS supported headless (`-p`), but the run's
|
|
10
|
+
* provenance must not depend on a flag we chose not to send. So the capture
|
|
11
|
+
* records what is VERIFIABLE at launch for every configured harness:
|
|
12
|
+
*
|
|
13
|
+
* - `<harness> --version` output (the exact CLI build that authored),
|
|
14
|
+
* - the claude settings default model (`~/.claude/settings.json` `model`),
|
|
15
|
+
* - `codex login status` (the codex seat must be authed or the run refuses
|
|
16
|
+
* at t=0 — a mid-run auth failure would silently kill one candidate slot),
|
|
17
|
+
* - the spec's pinned model id (null when the seat rides the CLI default).
|
|
18
|
+
*
|
|
19
|
+
* The capture FAILS LOUD on a missing/broken harness binary: populationSize
|
|
20
|
+
* must equal proposers.length, so a dead seat cannot be skipped at runtime —
|
|
21
|
+
* the config decides membership, this guard proves it at launch.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { readFileSync } from 'node:fs'
|
|
25
|
+
import { homedir } from 'node:os'
|
|
26
|
+
import { join } from 'node:path'
|
|
27
|
+
import {
|
|
28
|
+
DEFAULT_GEPA_PYTHON,
|
|
29
|
+
isGepaSeat,
|
|
30
|
+
loadGepaMethodFactory,
|
|
31
|
+
probeGepaRuntime,
|
|
32
|
+
type CampaignModuleImport,
|
|
33
|
+
} from './gepa-seat.mts'
|
|
34
|
+
import { run } from './proc.ts'
|
|
35
|
+
import type { ProposerSpec } from './proposer-fanout.mts'
|
|
36
|
+
|
|
37
|
+
export interface ProposerModelProvenance {
|
|
38
|
+
name: string
|
|
39
|
+
/** Absent on a GEN-6 engine seat (see `engine`). */
|
|
40
|
+
harness: ProposerSpec['harness']
|
|
41
|
+
/** Explicit model pin from the spec (threaded as `-m`), or null when the
|
|
42
|
+
* seat runs the CLI's own resolved default. */
|
|
43
|
+
pinnedModel: string | null
|
|
44
|
+
/** `<harness> --version` stdout (trimmed). For a GEN-6 gepa seat this is
|
|
45
|
+
* the bridge python's `--version` output — the runtime that authors. */
|
|
46
|
+
harnessVersion: string
|
|
47
|
+
/** claude seats only: the settings default model the logged-in CLI resolves
|
|
48
|
+
* when no `-m` is passed. Null when unreadable (recorded, never fatal —
|
|
49
|
+
* the version capture is the hard gate). */
|
|
50
|
+
settingsModel: string | null
|
|
51
|
+
/** codex seats only: `codex login status` stdout (trimmed). */
|
|
52
|
+
authStatus: string | null
|
|
53
|
+
merge: boolean
|
|
54
|
+
/** GEN-6 gepa seat: the engine name from the spec. */
|
|
55
|
+
engine?: 'gepa' | 'omni'
|
|
56
|
+
/** GEN-6 gepa seat: the ONE change-space file GEPA optimizes. */
|
|
57
|
+
surface?: string
|
|
58
|
+
/** GEN-6 gepa seat: installed gepa version ('source' for a source pin). */
|
|
59
|
+
gepaVersion?: string
|
|
60
|
+
/** GEN-6 gepa seat: the Python bridge module the seat runs. */
|
|
61
|
+
bridge?: string
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export interface ProvenanceCaptureRecord {
|
|
65
|
+
schema: 'swe-arena.proposer-provenance.v1'
|
|
66
|
+
capturedAt: string
|
|
67
|
+
proposers: ProposerModelProvenance[]
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Exec seam — test-injectable. Mirrors `run` from proc.ts. */
|
|
71
|
+
export type VersionExec = (
|
|
72
|
+
command: string,
|
|
73
|
+
args: string[],
|
|
74
|
+
) => Promise<{ code: number | null; stdout: string; stderr: string }>
|
|
75
|
+
|
|
76
|
+
const defaultExec: VersionExec = async (command, args) => {
|
|
77
|
+
const res = await run(command, args, { timeoutMs: 30_000 })
|
|
78
|
+
return { code: res.code, stdout: res.stdout, stderr: res.stderr }
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Read the claude CLI's settings default model. Pure over injected reader. */
|
|
82
|
+
export function claudeSettingsModel(
|
|
83
|
+
readFile: (path: string) => string = (p) => readFileSync(p, 'utf8'),
|
|
84
|
+
settingsPath = join(homedir(), '.claude', 'settings.json'),
|
|
85
|
+
): string | null {
|
|
86
|
+
try {
|
|
87
|
+
const parsed = JSON.parse(readFile(settingsPath)) as { model?: unknown }
|
|
88
|
+
return typeof parsed.model === 'string' && parsed.model.length > 0 ? parsed.model : null
|
|
89
|
+
} catch {
|
|
90
|
+
return null
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Capture per-proposer model provenance. Throws when any configured harness
|
|
95
|
+
* binary is missing/broken, when a codex seat is not logged in, or — GEN-6 —
|
|
96
|
+
* when a gepa seat's runtime is incomplete: the installed agent-eval must
|
|
97
|
+
* export `gepaOptimizationMethod` and the Python bridge + GEPA engine must
|
|
98
|
+
* import (probeGepaRuntime carries the exact install instructions). A dead
|
|
99
|
+
* seat fails the launch at t=0, never a mid-run candidate slot. */
|
|
100
|
+
export async function captureProposerProvenance(
|
|
101
|
+
proposers: ProposerSpec[],
|
|
102
|
+
deps: {
|
|
103
|
+
exec?: VersionExec
|
|
104
|
+
readSettingsModel?: () => string | null
|
|
105
|
+
importCampaign?: CampaignModuleImport
|
|
106
|
+
} = {},
|
|
107
|
+
): Promise<ProvenanceCaptureRecord> {
|
|
108
|
+
const exec = deps.exec ?? defaultExec
|
|
109
|
+
const readSettingsModel = deps.readSettingsModel ?? (() => claudeSettingsModel())
|
|
110
|
+
const versionByHarness = new Map<string, string>()
|
|
111
|
+
const authByHarness = new Map<string, string>()
|
|
112
|
+
const gepaBySeat = new Map<string, { pythonVersion: string; gepaVersion: string }>()
|
|
113
|
+
const gepaSeats = proposers.filter(isGepaSeat)
|
|
114
|
+
if (gepaSeats.length > 0) {
|
|
115
|
+
// Node side first: the adapter export (fails loud with upgrade hint).
|
|
116
|
+
await loadGepaMethodFactory(...(deps.importCampaign ? [deps.importCampaign] : []))
|
|
117
|
+
for (const seat of gepaSeats) {
|
|
118
|
+
gepaBySeat.set(seat.name, await probeGepaRuntime(seat.python ?? DEFAULT_GEPA_PYTHON, exec, seat.name))
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
const harnesses = [...new Set(proposers.map((p) => p.harness))].filter(
|
|
122
|
+
(h): h is NonNullable<ProposerSpec['harness']> => h !== undefined,
|
|
123
|
+
)
|
|
124
|
+
for (const harness of harnesses) {
|
|
125
|
+
const res = await exec(harness, ['--version'])
|
|
126
|
+
if (res.code !== 0) {
|
|
127
|
+
throw new Error(
|
|
128
|
+
`proposer provenance: '${harness} --version' failed (rc=${res.code}) — the ${harness} seat cannot author. ` +
|
|
129
|
+
`stderr: ${res.stderr.slice(0, 300)}`,
|
|
130
|
+
)
|
|
131
|
+
}
|
|
132
|
+
versionByHarness.set(harness, res.stdout.trim())
|
|
133
|
+
if (harness === 'codex') {
|
|
134
|
+
const auth = await exec('codex', ['login', 'status'])
|
|
135
|
+
const authOut = auth.stdout + auth.stderr
|
|
136
|
+
const authed = /logged in/i.test(authOut) && !/not logged in/i.test(authOut)
|
|
137
|
+
if (auth.code !== 0 || !authed) {
|
|
138
|
+
throw new Error(
|
|
139
|
+
`proposer provenance: codex seat configured but 'codex login status' says not authed ` +
|
|
140
|
+
`(rc=${auth.code}, out=${(auth.stdout + auth.stderr).trim().slice(0, 200)})`,
|
|
141
|
+
)
|
|
142
|
+
}
|
|
143
|
+
authByHarness.set('codex', (auth.stdout + auth.stderr).trim())
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
return {
|
|
147
|
+
schema: 'swe-arena.proposer-provenance.v1',
|
|
148
|
+
capturedAt: new Date().toISOString(),
|
|
149
|
+
proposers: proposers.map((spec): ProposerModelProvenance => {
|
|
150
|
+
if (isGepaSeat(spec)) {
|
|
151
|
+
const probe = gepaBySeat.get(spec.name)!
|
|
152
|
+
return {
|
|
153
|
+
name: spec.name,
|
|
154
|
+
harness: undefined,
|
|
155
|
+
pinnedModel: null,
|
|
156
|
+
harnessVersion: probe.pythonVersion,
|
|
157
|
+
settingsModel: null,
|
|
158
|
+
authStatus: null,
|
|
159
|
+
merge: false,
|
|
160
|
+
engine: spec.engine,
|
|
161
|
+
surface: spec.surface,
|
|
162
|
+
gepaVersion: probe.gepaVersion,
|
|
163
|
+
bridge: 'agent_eval_rpc.gepa_bridge',
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
return {
|
|
167
|
+
name: spec.name,
|
|
168
|
+
harness: spec.harness,
|
|
169
|
+
pinnedModel: spec.model ?? null,
|
|
170
|
+
harnessVersion: versionByHarness.get(spec.harness!)!,
|
|
171
|
+
settingsModel: spec.harness === 'claude' && !spec.model ? readSettingsModel() : null,
|
|
172
|
+
authStatus: (spec.harness !== undefined ? authByHarness.get(spec.harness) : undefined) ?? null,
|
|
173
|
+
merge: spec.merge === true,
|
|
174
|
+
}
|
|
175
|
+
}),
|
|
176
|
+
}
|
|
177
|
+
}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { describe, expect, it } from 'vitest'
|
|
2
|
+
import {
|
|
3
|
+
captureProposerProvenance,
|
|
4
|
+
claudeSettingsModel,
|
|
5
|
+
type VersionExec,
|
|
6
|
+
} from './proposer-provenance.mts'
|
|
7
|
+
import type { ProposerSpec } from './proposer-fanout.mts'
|
|
8
|
+
|
|
9
|
+
const ok = (stdout: string) => ({ code: 0, stdout, stderr: '' })
|
|
10
|
+
|
|
11
|
+
const gen4ish: ProposerSpec[] = [
|
|
12
|
+
{ name: 'claude-author', profile: 'default-author.profile.json', harness: 'claude' },
|
|
13
|
+
{ name: 'glm-author', harness: 'opencode', model: 'zai-coding-plan/glm-5.2' },
|
|
14
|
+
{ name: 'codex-author', harness: 'codex' },
|
|
15
|
+
{ name: 'merge-author', profile: 'default-author.profile.json', harness: 'claude', merge: true },
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
describe('captureProposerProvenance', () => {
|
|
19
|
+
const exec: VersionExec = async (command, args) => {
|
|
20
|
+
if (args[0] === '--version') return ok(`${command}-version 9.9.9`)
|
|
21
|
+
if (command === 'codex' && args[0] === 'login') return ok('Logged in using ChatGPT')
|
|
22
|
+
throw new Error(`unexpected exec ${command} ${args.join(' ')}`)
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
it('records pinned model, harness version, settings model, and codex auth per seat', async () => {
|
|
26
|
+
const record = await captureProposerProvenance(gen4ish, {
|
|
27
|
+
exec,
|
|
28
|
+
readSettingsModel: () => 'claude-fable-5',
|
|
29
|
+
})
|
|
30
|
+
expect(record.schema).toBe('swe-arena.proposer-provenance.v1')
|
|
31
|
+
const byName = Object.fromEntries(record.proposers.map((p) => [p.name, p]))
|
|
32
|
+
// Claude seat: no pin — the CLI's resolved settings model is the record.
|
|
33
|
+
expect(byName['claude-author']).toMatchObject({
|
|
34
|
+
harness: 'claude',
|
|
35
|
+
pinnedModel: null,
|
|
36
|
+
settingsModel: 'claude-fable-5',
|
|
37
|
+
harnessVersion: 'claude-version 9.9.9',
|
|
38
|
+
authStatus: null,
|
|
39
|
+
merge: false,
|
|
40
|
+
})
|
|
41
|
+
// Pinned opencode seat: explicit model id, settings model not consulted.
|
|
42
|
+
expect(byName['glm-author']).toMatchObject({
|
|
43
|
+
harness: 'opencode',
|
|
44
|
+
pinnedModel: 'zai-coding-plan/glm-5.2',
|
|
45
|
+
settingsModel: null,
|
|
46
|
+
harnessVersion: 'opencode-version 9.9.9',
|
|
47
|
+
})
|
|
48
|
+
// Codex seat: version + auth status captured.
|
|
49
|
+
expect(byName['codex-author']).toMatchObject({
|
|
50
|
+
harness: 'codex',
|
|
51
|
+
pinnedModel: null,
|
|
52
|
+
authStatus: 'Logged in using ChatGPT',
|
|
53
|
+
})
|
|
54
|
+
expect(byName['merge-author']!.merge).toBe(true)
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
it('fails loud when a configured harness binary is missing', async () => {
|
|
58
|
+
const broken: VersionExec = async (command, args) =>
|
|
59
|
+
command === 'codex' ? { code: 127, stdout: '', stderr: 'not found' } : exec(command, args)
|
|
60
|
+
await expect(
|
|
61
|
+
captureProposerProvenance(gen4ish, { exec: broken, readSettingsModel: () => null }),
|
|
62
|
+
).rejects.toThrow(/'codex --version' failed/)
|
|
63
|
+
})
|
|
64
|
+
|
|
65
|
+
it('fails loud when the codex seat is not logged in', async () => {
|
|
66
|
+
const loggedOut: VersionExec = async (command, args) => {
|
|
67
|
+
if (args[0] === '--version') return ok(`${command} 1.0.0`)
|
|
68
|
+
return ok('Not logged in')
|
|
69
|
+
}
|
|
70
|
+
await expect(
|
|
71
|
+
captureProposerProvenance([{ name: 'codex-author', harness: 'codex' }], {
|
|
72
|
+
exec: loggedOut,
|
|
73
|
+
readSettingsModel: () => null,
|
|
74
|
+
}),
|
|
75
|
+
).rejects.toThrow(/not authed/)
|
|
76
|
+
})
|
|
77
|
+
|
|
78
|
+
it('runs one version probe per harness, not per proposer', async () => {
|
|
79
|
+
const calls: string[] = []
|
|
80
|
+
const counting: VersionExec = async (command, args) => {
|
|
81
|
+
calls.push(`${command} ${args.join(' ')}`)
|
|
82
|
+
if (args[0] === '--version') return ok(`${command} 1`)
|
|
83
|
+
return ok('Logged in using ChatGPT')
|
|
84
|
+
}
|
|
85
|
+
await captureProposerProvenance(gen4ish, { exec: counting, readSettingsModel: () => null })
|
|
86
|
+
expect(calls.filter((c) => c === 'claude --version')).toHaveLength(1)
|
|
87
|
+
expect(calls.filter((c) => c === 'codex --version')).toHaveLength(1)
|
|
88
|
+
expect(calls.filter((c) => c === 'codex login status')).toHaveLength(1)
|
|
89
|
+
})
|
|
90
|
+
})
|
|
91
|
+
|
|
92
|
+
describe('claudeSettingsModel', () => {
|
|
93
|
+
it('reads the settings model field', () => {
|
|
94
|
+
expect(claudeSettingsModel(() => JSON.stringify({ model: 'claude-fable-5' }), '/x')).toBe('claude-fable-5')
|
|
95
|
+
})
|
|
96
|
+
|
|
97
|
+
it('returns null on unreadable/missing/blank settings', () => {
|
|
98
|
+
expect(
|
|
99
|
+
claudeSettingsModel(() => {
|
|
100
|
+
throw new Error('ENOENT')
|
|
101
|
+
}, '/x'),
|
|
102
|
+
).toBeNull()
|
|
103
|
+
expect(claudeSettingsModel(() => JSON.stringify({}), '/x')).toBeNull()
|
|
104
|
+
expect(claudeSettingsModel(() => JSON.stringify({ model: '' }), '/x')).toBeNull()
|
|
105
|
+
})
|
|
106
|
+
})
|
|
Binary file
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Replay CLI: `tsx src/swe-arena/replay.mts`
|
|
3
|
+
*
|
|
4
|
+
* Reproduces the reference `fixtures/analyze.py` output from the committed
|
|
5
|
+
* fixtures (section 1 matches its printed lines exactly — pinned in
|
|
6
|
+
* replay.test.mts), then prints what the reference script never did:
|
|
7
|
+
* the reconciled valid-denominator verdict, the true SUP spend including
|
|
8
|
+
* worker tokens, the supervisor evolution rounds, and the holdout registry.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { pathToFileURL } from 'node:url'
|
|
12
|
+
import {
|
|
13
|
+
loadHoldout,
|
|
14
|
+
loadLedger,
|
|
15
|
+
loadPreregisterLog,
|
|
16
|
+
loadRejudge,
|
|
17
|
+
loadRematchRounds,
|
|
18
|
+
loadSupJournalTrue,
|
|
19
|
+
loadWorkerTokens,
|
|
20
|
+
} from './fixtures.ts'
|
|
21
|
+
import {
|
|
22
|
+
costRollup,
|
|
23
|
+
ledgerOutcomes,
|
|
24
|
+
pairedSignTest,
|
|
25
|
+
reconciledOutcomes,
|
|
26
|
+
roundsProgression,
|
|
27
|
+
splitPairs,
|
|
28
|
+
type CostRollup,
|
|
29
|
+
type DiscordantSplit,
|
|
30
|
+
type RoundState,
|
|
31
|
+
type SignTestResult,
|
|
32
|
+
} from './analyze.ts'
|
|
33
|
+
import { reconcile, type PairedTable } from './reconcile.ts'
|
|
34
|
+
import type { HoldoutRegistry, LedgerRow, RematchRow } from './types.ts'
|
|
35
|
+
|
|
36
|
+
/** Python-style list repr (single quotes) so section 1 matches analyze.py byte-for-byte. */
|
|
37
|
+
const pyList = (xs: string[]): string => `[${xs.map((x) => `'${x}'`).join(', ')}]`
|
|
38
|
+
const signed = (x: number, digits?: number): string =>
|
|
39
|
+
(x >= 0 ? '+' : '') + (digits === undefined ? String(x) : x.toFixed(digits))
|
|
40
|
+
|
|
41
|
+
export interface Replay {
|
|
42
|
+
ledger: LedgerRow[]
|
|
43
|
+
raw: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
|
|
44
|
+
cost: CostRollup
|
|
45
|
+
table: PairedTable
|
|
46
|
+
valid: { split: DiscordantSplit; sign: SignTestResult; soloResolved: number; supResolved: number }
|
|
47
|
+
rounds: Map<string, RoundState[]>
|
|
48
|
+
rematchRounds: RematchRow[][]
|
|
49
|
+
holdout: HoldoutRegistry
|
|
50
|
+
preregisterLog: string[]
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export function buildReplay(): Replay {
|
|
54
|
+
const ledger = loadLedger()
|
|
55
|
+
const rejudge = loadRejudge()
|
|
56
|
+
const rematchRounds = loadRematchRounds()
|
|
57
|
+
|
|
58
|
+
const rawOutcomes = ledgerOutcomes(ledger)
|
|
59
|
+
const table = reconcile(ledger, rejudge)
|
|
60
|
+
const validOutcomes = reconciledOutcomes(table.valid)
|
|
61
|
+
|
|
62
|
+
return {
|
|
63
|
+
ledger,
|
|
64
|
+
raw: {
|
|
65
|
+
split: splitPairs(rawOutcomes),
|
|
66
|
+
sign: pairedSignTest(rawOutcomes),
|
|
67
|
+
soloResolved: rawOutcomes.filter((o) => o.solo).length,
|
|
68
|
+
supResolved: rawOutcomes.filter((o) => o.sup).length,
|
|
69
|
+
},
|
|
70
|
+
cost: costRollup(ledger, loadSupJournalTrue(), loadWorkerTokens()),
|
|
71
|
+
table,
|
|
72
|
+
valid: {
|
|
73
|
+
split: splitPairs(validOutcomes),
|
|
74
|
+
sign: pairedSignTest(validOutcomes),
|
|
75
|
+
soloResolved: validOutcomes.filter((o) => o.solo).length,
|
|
76
|
+
supResolved: validOutcomes.filter((o) => o.sup).length,
|
|
77
|
+
},
|
|
78
|
+
rounds: roundsProgression(ledger, rematchRounds),
|
|
79
|
+
rematchRounds,
|
|
80
|
+
holdout: loadHoldout(),
|
|
81
|
+
preregisterLog: loadPreregisterLog(),
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** Section 1 — byte-faithful reproduction of analyze.py's printed analysis. */
|
|
86
|
+
export function renderReference(r: Replay): string[] {
|
|
87
|
+
const rows = [...r.ledger].sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
|
|
88
|
+
const n = rows.length
|
|
89
|
+
const { raw, cost } = r
|
|
90
|
+
const bar = '='.repeat(80)
|
|
91
|
+
const lines: string[] = []
|
|
92
|
+
lines.push(bar)
|
|
93
|
+
lines.push(`PAIRED HEAD-TO-HEAD: glm-5.2 SOLO vs glm-5.2 SUPERVISOR (N=${n} paired instances)`)
|
|
94
|
+
lines.push(bar)
|
|
95
|
+
lines.push(`SOLO resolved: ${raw.soloResolved}/${n} = ${((100 * raw.soloResolved) / n).toFixed(1)}%`)
|
|
96
|
+
lines.push(`SUP resolved: ${raw.supResolved}/${n} = ${((100 * raw.supResolved) / n).toFixed(1)}%`)
|
|
97
|
+
const delta = raw.supResolved - raw.soloResolved
|
|
98
|
+
lines.push(`delta (SUP-SOLO): ${signed(delta)} instances (${signed((100 * delta) / n, 1)} pts)`)
|
|
99
|
+
lines.push('')
|
|
100
|
+
lines.push('DISCORDANT PAIRS (the signal):')
|
|
101
|
+
lines.push(` SUP-only wins (SUP✓ SOLO✗): ${raw.split.supOnly.length} ${pyList(raw.split.supOnly)}`)
|
|
102
|
+
lines.push(` SOLO-only wins (SOLO✓ SUP✗): ${raw.split.soloOnly.length} ${pyList(raw.split.soloOnly)}`)
|
|
103
|
+
lines.push(` both resolved: ${raw.split.both.length} | neither: ${raw.split.neither.length} ${pyList(raw.split.neither)}`)
|
|
104
|
+
lines.push(` exact two-sided sign test on discordant pairs: p=${raw.sign.pValue.toFixed(4)}`)
|
|
105
|
+
lines.push('')
|
|
106
|
+
lines.push(
|
|
107
|
+
`COST (measured tokens; USD via shared blended rate $${(cost.blendedRatePerTok * 1e6).toFixed(3)}/1M from SUP accounting):`,
|
|
108
|
+
)
|
|
109
|
+
lines.push(` SOLO total tokens: ${cost.soloTokens.toLocaleString('en-US')} -> derived $${cost.soloUsdDerived.toFixed(4)}`)
|
|
110
|
+
lines.push(` SUP total tokens: ${cost.supBrainTokens.toLocaleString('en-US')} -> runtime $${cost.supUsd.toFixed(4)}`)
|
|
111
|
+
lines.push(` SUP/SOLO token ratio: ${cost.brainTokenRatio.toFixed(2)}x`)
|
|
112
|
+
lines.push(` SUP/SOLO cost ratio (token-derived): ${(cost.supUsd / cost.soloUsdDerived).toFixed(2)}x`)
|
|
113
|
+
lines.push(` [telemetry] instances where runtime spentTokens != journal-true (no-winner zeroing): ${pyList(cost.telemetryGaps)}`)
|
|
114
|
+
lines.push(` WALL: SOLO ${cost.soloWallS}s total vs SUP ${cost.supWallS}s total -> SUP ${cost.wallRatio.toFixed(2)}x wall`)
|
|
115
|
+
lines.push('')
|
|
116
|
+
lines.push(bar)
|
|
117
|
+
lines.push('PER-INSTANCE')
|
|
118
|
+
lines.push(bar)
|
|
119
|
+
lines.push(
|
|
120
|
+
`${'instance'.padEnd(32)} ${'SOLO'.padEnd(5)} ${'SUP'.padEnd(5)} ${'v_s'.padEnd(3)} ${'v_p'.padEnd(3)} ${'wrk'.padEnd(3)} ${'soloTok'.padEnd(8)} ${'supTok'.padEnd(8)} ${'supUSD'.padEnd(7)} ${'soloW'.padEnd(5)} ${'supW'.padEnd(5)}`,
|
|
121
|
+
)
|
|
122
|
+
for (const row of rows) {
|
|
123
|
+
const pyBool = (v: boolean): string => (v ? 'True' : 'False')
|
|
124
|
+
lines.push(
|
|
125
|
+
`${row.iid.padEnd(32)} ${pyBool(row.solo_resolved).slice(0, 5).padEnd(5)} ${pyBool(row.sup_resolved).slice(0, 5).padEnd(5)} ` +
|
|
126
|
+
`${pyBool(row.solo_verify_pass)[0].padEnd(3)} ${pyBool(row.sup_verify_pass)[0].padEnd(3)} ` +
|
|
127
|
+
`${String(row.sup_workers ?? '?').padEnd(3)} ${String(row.solo_tokens).padEnd(8)} ${String(row.sup_spentTokens ?? 0).padEnd(8)} ` +
|
|
128
|
+
`${(row.sup_spentUsd ?? 0).toFixed(4)} ${String(row.solo_wall_s).padEnd(5)} ${String(row.sup_wall_s).padEnd(5)}`,
|
|
129
|
+
)
|
|
130
|
+
}
|
|
131
|
+
return lines
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** Sections 2-5 — the analysis that lived in session lore, now typed. */
|
|
135
|
+
export function renderReconciled(r: Replay): string[] {
|
|
136
|
+
const bar = '='.repeat(80)
|
|
137
|
+
const lines: string[] = []
|
|
138
|
+
const { table, valid, cost } = r
|
|
139
|
+
const n = table.valid.length
|
|
140
|
+
|
|
141
|
+
lines.push(bar)
|
|
142
|
+
lines.push('RECONCILED VERDICT (re-judged, gold-gated denominator)')
|
|
143
|
+
lines.push(bar)
|
|
144
|
+
for (const e of table.excluded) lines.push(` EXCLUDED ${e.iid}: ${e.excludeReason}`)
|
|
145
|
+
for (const v of table.valid) {
|
|
146
|
+
const src = [v.solo.source !== 'ledger' ? `solo:${v.solo.source}` : null, v.sup.source !== 'ledger' ? `sup:${v.sup.source}` : null]
|
|
147
|
+
.filter(Boolean)
|
|
148
|
+
.join(' ')
|
|
149
|
+
if (src) lines.push(` RE-JUDGED ${v.iid}: ${src}`)
|
|
150
|
+
}
|
|
151
|
+
lines.push(`SOLO resolved: ${valid.soloResolved}/${n}`)
|
|
152
|
+
lines.push(`SUP resolved: ${valid.supResolved}/${n}`)
|
|
153
|
+
lines.push(`discordant: SUP-only ${pyList(valid.split.supOnly)} | SOLO-only ${pyList(valid.split.soloOnly)}`)
|
|
154
|
+
lines.push(`exact two-sided sign test: p=${valid.sign.pValue.toFixed(4)}`)
|
|
155
|
+
lines.push('')
|
|
156
|
+
lines.push('TRUE SUP SPEND (brain + workers; analyze.py printed brain only):')
|
|
157
|
+
lines.push(` brain ${cost.supBrainTokens.toLocaleString('en-US')} + workers ${cost.supWorkerTokens.toLocaleString('en-US')} = ${cost.supTotalTokens.toLocaleString('en-US')} tokens`)
|
|
158
|
+
lines.push(` SUP/SOLO true token ratio: ${cost.totalTokenRatio.toFixed(2)}x (brain-only ratio: ${cost.brainTokenRatio.toFixed(2)}x)`)
|
|
159
|
+
lines.push('')
|
|
160
|
+
lines.push('SUP EVOLUTION ROUNDS (SUP = original head-to-head run):')
|
|
161
|
+
for (const [iid, states] of r.rounds) {
|
|
162
|
+
const cells = states.map(
|
|
163
|
+
(s) => `${s.round}:${s.resolved ? 'RESOLVED' : 'unresolved'}(${s.patchLines}L,${s.verdict ?? 'null'})`,
|
|
164
|
+
)
|
|
165
|
+
lines.push(` ${iid.padEnd(32)} ${cells.join(' -> ')}`)
|
|
166
|
+
}
|
|
167
|
+
lines.push('')
|
|
168
|
+
lines.push(`HOLDOUT REGISTRY (pre-registered at loops@${r.holdout.selectedAtCommit.slice(0, 10)}, untouched):`)
|
|
169
|
+
for (const e of r.holdout.entries) {
|
|
170
|
+
lines.push(` ${e.iid.padEnd(36)} gold_official_resolved=${e.gold_official_resolved} verify_calibrated=${e.verify_calibrated}`)
|
|
171
|
+
}
|
|
172
|
+
return lines
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
export function renderReplay(r: Replay = buildReplay()): string {
|
|
176
|
+
return [...renderReference(r), '', ...renderReconciled(r)].join('\n')
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
|
|
180
|
+
|
|
181
|
+
if (isMain) {
|
|
182
|
+
console.log(renderReplay())
|
|
183
|
+
}
|