@tangle-network/agent-bench 0.11.2 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -0
- package/HARNESS.md +6 -2
- package/README.md +1 -4
- package/dist/benchmarks/swe-bench.js +4 -9
- package/dist/benchmarks/swe-bench.js.map +1 -1
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +2 -2
- package/src/benchmarks/swe-bench.test.mts +49 -0
- package/src/benchmarks/swe-bench.ts +4 -9
- package/src/quant-arena/README.md +0 -144
- package/src/quant-arena/backtest.test.mts +0 -135
- package/src/quant-arena/backtest.ts +0 -218
- package/src/quant-arena/data.test.mts +0 -44
- package/src/quant-arena/data.ts +0 -141
- package/src/quant-arena/driver.test.mts +0 -253
- package/src/quant-arena/driver.ts +0 -219
- package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
- package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
- package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
- package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
- package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
- package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
- package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
- package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
- package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
- package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
- package/src/quant-arena/holdout-certify.mts +0 -206
- package/src/quant-arena/holdout-certify.test.mts +0 -82
- package/src/quant-arena/leak-audit.test.mts +0 -79
- package/src/quant-arena/leak-audit.ts +0 -95
- package/src/quant-arena/make-fixtures.mts +0 -161
- package/src/quant-arena/multiplicity.test.mts +0 -68
- package/src/quant-arena/multiplicity.ts +0 -87
- package/src/quant-arena/nautilus-certify.ts +0 -31
- package/src/quant-arena/oms.ts +0 -90
- package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
- package/src/quant-arena/python/pyproject.toml +0 -8
- package/src/quant-arena/python/uv.lock +0 -1297
- package/src/quant-arena/python/vbt-worker.py +0 -192
- package/src/quant-arena/quant-loop.mts +0 -840
- package/src/quant-arena/quant-loop.test.mts +0 -75
- package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
- package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
- package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
- package/src/quant-arena/types.ts +0 -133
- package/src/quant-arena/vbt-client.ts +0 -321
- package/src/quant-arena/vbt-parity.test.mts +0 -183
- package/src/quant-arena/windows.test.mts +0 -45
- package/src/quant-arena/windows.ts +0 -54
- package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
- package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
- package/src/rollout-ledger/settle-capture.mts +0 -448
- package/src/rollout-ledger/settle-capture.test.mts +0 -270
- package/src/swe-arena/activation.mts +0 -225
- package/src/swe-arena/activation.test.mts +0 -300
- package/src/swe-arena/analyze.ts +0 -211
- package/src/swe-arena/arms.ts +0 -862
- package/src/swe-arena/bootstrap-meta.mts +0 -188
- package/src/swe-arena/bootstrap-meta.test.mts +0 -51
- package/src/swe-arena/briefing.mts +0 -217
- package/src/swe-arena/briefing.test.mts +0 -179
- package/src/swe-arena/calibrate.ts +0 -217
- package/src/swe-arena/capabilities.mts +0 -76
- package/src/swe-arena/capabilities.test.mts +0 -57
- package/src/swe-arena/capacity.ts +0 -198
- package/src/swe-arena/cell-evidence.mts +0 -437
- package/src/swe-arena/cell-evidence.test.mts +0 -248
- package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
- package/src/swe-arena/diagnosis-ensemble.ts +0 -523
- package/src/swe-arena/execution.test.mts +0 -1171
- package/src/swe-arena/factory-command-container.ts +0 -284
- package/src/swe-arena/factory-judge-child.mts +0 -228
- package/src/swe-arena/factory.test.mts +0 -645
- package/src/swe-arena/fixtures/analyze.py +0 -80
- package/src/swe-arena/fixtures/excludes.txt +0 -8
- package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
- package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
- package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
- package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
- package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
- package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
- package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
- package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
- package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
- package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
- package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
- package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
- package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
- package/src/swe-arena/fixtures/holdout.json +0 -44
- package/src/swe-arena/fixtures/instances.json +0 -146
- package/src/swe-arena/fixtures/ledger.jsonl +0 -12
- package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
- package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
- package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
- package/src/swe-arena/fixtures/rematch.jsonl +0 -3
- package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
- package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
- package/src/swe-arena/fixtures/run-report/README.md +0 -43
- package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
- package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
- package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
- package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
- package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
- package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
- package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
- package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
- package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
- package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
- package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
- package/src/swe-arena/fixtures/worker-tokens.json +0 -42
- package/src/swe-arena/fixtures.ts +0 -237
- package/src/swe-arena/gepa-seat.mts +0 -886
- package/src/swe-arena/gepa-seat.test.mts +0 -1136
- package/src/swe-arena/holdout-certify.mts +0 -408
- package/src/swe-arena/holdout-certify.test.mts +0 -160
- package/src/swe-arena/implementation-ref.test.mts +0 -64
- package/src/swe-arena/implementation-ref.ts +0 -62
- package/src/swe-arena/judge-child.mts +0 -37
- package/src/swe-arena/ledger-orphans.mts +0 -77
- package/src/swe-arena/ledger-orphans.test.mts +0 -149
- package/src/swe-arena/manifest.mts +0 -293
- package/src/swe-arena/manifest.test.mts +0 -169
- package/src/swe-arena/materialize.ts +0 -142
- package/src/swe-arena/outer-loop.mts +0 -2854
- package/src/swe-arena/outer-loop.test.mts +0 -714
- package/src/swe-arena/parity.test.mts +0 -87
- package/src/swe-arena/premeasured-from-cells.mts +0 -296
- package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
- package/src/swe-arena/proc.test.mts +0 -172
- package/src/swe-arena/proc.ts +0 -260
- package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
- package/src/swe-arena/profiles/default-author.profile.json +0 -12
- package/src/swe-arena/proposer-fanout.mts +0 -736
- package/src/swe-arena/proposer-fanout.test.mts +0 -660
- package/src/swe-arena/proposer-provenance.mts +0 -176
- package/src/swe-arena/proposer-provenance.test.mts +0 -106
- package/src/swe-arena/reconcile.ts +0 -0
- package/src/swe-arena/replay.mts +0 -183
- package/src/swe-arena/replay.test.mts +0 -300
- package/src/swe-arena/run-experiment.mts +0 -729
- package/src/swe-arena/run-report.mts +0 -75
- package/src/swe-arena/run-supervisor.mjs +0 -297
- package/src/swe-arena/run-supervisor.test.mts +0 -539
- package/src/swe-arena/score-split.mts +0 -140
- package/src/swe-arena/score-split.test.mts +0 -123
- package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
- package/src/swe-arena/scratch-worktree.test.mts +0 -56
- package/src/swe-arena/scratch-worktree.ts +0 -64
- package/src/swe-arena/serialized-judge.ts +0 -414
- package/src/swe-arena/types.ts +0 -218
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,33 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.13.0
|
|
4
|
+
|
|
5
|
+
Remove the SWE and quant research campaign drivers, their historical fixtures, and the campaign-specific rollout ledger bridge from the source package.
|
|
6
|
+
These drivers depended on the retired `.loops` layout.
|
|
7
|
+
Public benchmark adapters and exports remain available.
|
|
8
|
+
|
|
9
|
+
## 0.12.2
|
|
10
|
+
|
|
11
|
+
Consume the agent-eval 0.182.0 and agent-knowledge 17.0.1 cohort with Runtime 0.227.0.
|
|
12
|
+
|
|
13
|
+
## 0.12.1
|
|
14
|
+
|
|
15
|
+
Keep the published Sandbox dependency range aligned with the supported 0.39.x cohort.
|
|
16
|
+
|
|
17
|
+
## 0.11.5
|
|
18
|
+
|
|
19
|
+
Consume Runtime 0.218, including asynchronous coordination address resolution for provider managers.
|
|
20
|
+
|
|
21
|
+
## 0.11.4
|
|
22
|
+
|
|
23
|
+
Admit Sandbox 0.39 through the shared dependency catalog and consume Runtime's retained result reconciliation fix.
|
|
24
|
+
|
|
25
|
+
## 0.11.3
|
|
26
|
+
|
|
27
|
+
Uses Runtime 0.209.0, Eval 0.180.0, and Knowledge 15.0.3 together, including named resource accounting and source-attributed resource reports.
|
|
28
|
+
SWE-bench setup, prompts, and patch extraction use the writable session workspace instead of a root-level directory.
|
|
29
|
+
Official grading behavior is unchanged.
|
|
30
|
+
|
|
3
31
|
## 0.11.2
|
|
4
32
|
|
|
5
33
|
Supports SWE-bench 5.x by detecting removed cache and namespace evaluator flags while retaining the 4.x path.
|
package/HARNESS.md
CHANGED
|
@@ -109,19 +109,23 @@ Same-session continuation does not prove a barrier before every native request,
|
|
|
109
109
|
Offline fake-sandbox tests prove the managed contract and correction consumption only.
|
|
110
110
|
They establish no live learning gain or provider session restoration.
|
|
111
111
|
|
|
112
|
-
### Retained strategy
|
|
112
|
+
### Retained strategy fixture
|
|
113
113
|
|
|
114
114
|
```bash
|
|
115
115
|
cd bench
|
|
116
116
|
pnpm tsx src/swe-self-improve.mts
|
|
117
117
|
```
|
|
118
118
|
|
|
119
|
-
This
|
|
119
|
+
This fixture uses `runStrategyEvolution` with SWE-bench tasks and a frozen holdout.
|
|
120
120
|
It does not exercise `improve`, and it deletes its temporary run directory on exit.
|
|
121
121
|
It therefore cannot provide retained improvement or lineage evidence.
|
|
122
122
|
Use `examples/improve` for the maintained offline API fixture.
|
|
123
123
|
Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
|
|
124
124
|
|
|
125
|
+
The former `swe-arena`, `quant-arena`, and rollout-ledger campaign drivers were removed from this package.
|
|
126
|
+
They had no public exports or maintained package command, duplicated lab orchestration, and depended on the retired `.loops` run layout.
|
|
127
|
+
Historical campaign evidence belongs in its owning lab; new benchmark code here must satisfy the admission rule below.
|
|
128
|
+
|
|
125
129
|
### Offline diagnostics
|
|
126
130
|
|
|
127
131
|
```bash
|
package/README.md
CHANGED
|
@@ -41,10 +41,7 @@ pnpm install # tsx + link parent
|
|
|
41
41
|
The judge needs only Docker; workers need a model key (Tangle router `TANGLE_API_KEY`, or a direct provider).
|
|
42
42
|
|
|
43
43
|
Live optimizer scripts require explicit token prices so cost records cannot be guessed.
|
|
44
|
-
|
|
45
|
-
For either prefix, set `INPUT_USD_PER_MILLION`, `CACHED_INPUT_USD_PER_MILLION`, `CACHE_WRITE_USD_PER_MILLION`, and `OUTPUT_USD_PER_MILLION`.
|
|
46
|
-
The SWE arena seat reads its model from `GEPA_OPTIMIZER_MODEL`, its URL from `GEPA_OPTIMIZER_BASE_URL`, and its key from `GEPA_OPTIMIZER_API_KEY`.
|
|
47
|
-
It falls back to the configured driver model, `ROUTER_BASE`, and `TANGLE_API_KEY`.
|
|
44
|
+
Set `REFLECT_INPUT_USD_PER_MILLION`, `REFLECT_CACHED_INPUT_USD_PER_MILLION`, `REFLECT_CACHE_WRITE_USD_PER_MILLION`, and `REFLECT_OUTPUT_USD_PER_MILLION`.
|
|
48
45
|
Missing prices fail before an optimizer model call.
|
|
49
46
|
|
|
50
47
|
Retain every official per-test log and report before the temporary evaluator directory is removed:
|
|
@@ -14,13 +14,8 @@ import { join } from "node:path";
|
|
|
14
14
|
* SWE-specific pieces: the patch OutputAdapter, the dataset dump, and the
|
|
15
15
|
* predictions-file → run_evaluation argv → report-shape mapping.
|
|
16
16
|
*/
|
|
17
|
-
/**
|
|
18
|
-
|
|
19
|
-
* source of truth shared by the prompt template (which tells the agent to clone
|
|
20
|
-
* here) and `boxExtract` (which runs `git diff` here after the shot) — so the
|
|
21
|
-
* harness always knows exactly where the agent's edits live, for any instance.
|
|
22
|
-
*/
|
|
23
|
-
const SWE_REPO_DIR = "/work";
|
|
17
|
+
/** Root-level directories are not writable in every sandbox; use the session workspace. */
|
|
18
|
+
const SWE_REPO_DIR = "./swe-bench-repo";
|
|
24
19
|
/**
|
|
25
20
|
* The SWE deliverable's FALLBACK parser, from the agent's event STREAM.
|
|
26
21
|
*
|
|
@@ -189,10 +184,10 @@ print(json.dumps(out))
|
|
|
189
184
|
prompt: [
|
|
190
185
|
`Repository: ${r.repo} @ ${r.base_commit}`,
|
|
191
186
|
"",
|
|
192
|
-
`The repository is ALREADY cloned at ${SWE_REPO_DIR}, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
187
|
+
`The repository is ALREADY cloned at ${SWE_REPO_DIR} relative to your initial session workspace, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
193
188
|
"",
|
|
194
189
|
"Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.",
|
|
195
|
-
"Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your
|
|
190
|
+
"Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your source edits directly from the repo, including uncommitted changes.",
|
|
196
191
|
"",
|
|
197
192
|
"--- Issue ---",
|
|
198
193
|
String(r.problem_statement ?? "")
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"swe-bench.js","names":[],"sources":["../../src/benchmarks/swe-bench.ts"],"sourcesContent":["/**\n * SWE-bench Verified adapter. Worker artifact = a unified-diff patch. Judge =\n * the official `swebench` harness: apply the patch in the instance's Docker\n * image, run FAIL_TO_PASS + PASS_TO_PASS, report `resolved`. Fully deterministic\n * — no LLM judge.\n *\n * Requires: the bench `.venv` with `swebench` installed + a running Docker\n * daemon (per-instance images are pulled/built on first run).\n *\n * Process/Docker/report plumbing is shared via ./_harness; this file owns the\n * SWE-specific pieces: the patch OutputAdapter, the dataset dump, and the\n * predictions-file → run_evaluation argv → report-shape mapping.\n */\n\nimport { join } from 'node:path'\nimport type { OutputAdapter } from '@tangle-network/agent-runtime/kernel'\nimport {\n preflightVenvImports,\n readJsonReport,\n runStagedJudge,\n runVenvPython,\n safeRunId,\n stageFile,\n type StagedRunCaptureSpec,\n} from './_harness'\nimport type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'\n\n/**\n * Fixed in-box path the agent clones the instance repo into. It is the SINGLE\n * source of truth shared by the prompt template (which tells the agent to clone\n * here) and `boxExtract` (which runs `git diff` here after the shot) — so the\n * harness always knows exactly where the agent's edits live, for any instance.\n */\nconst SWE_REPO_DIR = '/work'\n\n/**\n * The SWE deliverable's FALLBACK parser, from the agent's event STREAM.\n *\n * The PRIMARY deliverable is `boxExtract` below: a `git diff` of the agent's\n * actual edits, read from the cloned repo's STATE inside the box (standard\n * SWE-bench practice). This event-stream parse only runs when that diff is empty\n * — a model that edited the source correctly but never printed a fenced diff (the\n * exact failure this replaces) still scores off its real changes, not its prose.\n */\nexport const swePatchOutput: OutputAdapter<string> = {\n parse(events) {\n let text = ''\n for (const ev of events) {\n const d = (ev as { data?: Record<string, unknown> })?.data\n const t = d?.finalText ?? d?.text ?? d?.result\n if (typeof t === 'string' && t.length > 0) text = t\n }\n // Last ```diff/```patch fenced block (the contract the prompt asks for);\n // fall back to the raw text so a fence-less but valid diff still reaches the judge.\n const fences = [...text.matchAll(/```(?:diff|patch)?\\s*\\n([\\s\\S]*?)```/g)]\n const last = fences.at(-1)?.[1]\n return (last ?? text).trim()\n },\n}\n\nconst DATASET = 'princeton-nlp/SWE-bench_Verified'\nexport type SweBenchCacheLevel = 'none' | 'base' | 'env' | 'instance'\n\nexport interface SweBenchArtifactCaptureContext {\n readonly taskId: string\n readonly runId: string\n /** One-based sequence unique within this adapter instance. */\n readonly attemptSequence: number\n}\n\nexport interface SweBenchAdapterOptions {\n readonly timeoutMs?: number\n readonly cacheLevel?: SweBenchCacheLevel\n /**\n * Return a unique destination for any attempt whose complete official\n * evaluator directory and process logs should be retained.\n */\n readonly captureEvaluatorArtifacts?: (\n context: SweBenchArtifactCaptureContext,\n ) => StagedRunCaptureSpec | undefined\n}\n\nconst SWE_CACHE_LEVELS = new Set<SweBenchCacheLevel>(['none', 'base', 'env', 'instance'])\n\nfunction scorerNamespace(): 'swebench' | 'none' {\n const namespace = process.env.SWEBENCH_NAMESPACE ?? 'swebench'\n if (namespace !== 'swebench' && namespace !== 'none') {\n throw new Error(`SWEBENCH_NAMESPACE must be swebench|none, got \"${namespace}\"`)\n }\n return namespace\n}\nconst TEST_FILE_EXCLUDES = [\n \"':(exclude,glob)**/tests/**'\",\n \"':(exclude,glob)**/test/**'\",\n \"':(exclude,glob)test_*.py'\",\n \"':(exclude,glob)**/test_*.py'\",\n \"':(exclude,glob)*_test.py'\",\n \"':(exclude,glob)**/*_test.py'\",\n \"':(exclude,glob)conftest.py'\",\n \"':(exclude,glob)**/conftest.py'\",\n].join(' ')\n\ninterface SweReport {\n resolved_instances?: number\n resolved_ids?: string[]\n unresolved_ids?: string[]\n empty_patch_ids?: string[]\n completed_ids?: string[]\n incomplete_ids?: string[]\n error_ids?: string[]\n submitted_ids?: string[]\n}\n\nfunction stringIds(report: Record<string, unknown>, key: keyof SweReport): string[] {\n const value = report[key]\n if (value === undefined) return []\n if (!Array.isArray(value) || value.some((entry) => typeof entry !== 'string')) {\n throw new Error(`swe-bench: malformed ${key}`)\n }\n return value\n}\n\n/** Convert one official report into a score without turning evaluator failures into agent failures. */\nexport function scoreSweReport(taskId: string, value: unknown): BenchScore {\n if (!value || typeof value !== 'object' || Array.isArray(value)) {\n throw new Error('swe-bench: report must be an object')\n }\n const report = value as Record<string, unknown>\n const statusIds = {\n resolved: stringIds(report, 'resolved_ids'),\n unresolved: stringIds(report, 'unresolved_ids'),\n emptyPatch: stringIds(report, 'empty_patch_ids'),\n completed: stringIds(report, 'completed_ids'),\n incomplete: stringIds(report, 'incomplete_ids'),\n error: stringIds(report, 'error_ids'),\n }\n const submitted = stringIds(report, 'submitted_ids')\n const mentioned = Object.values(statusIds).flat()\n if (\n mentioned.some((id) => id !== taskId)\n || (submitted.length > 0 && (submitted.length !== 1 || submitted[0] !== taskId))\n ) {\n throw new Error(`swe-bench: report identity mismatch for ${taskId}`)\n }\n if (statusIds.error.includes(taskId) || statusIds.incomplete.includes(taskId)) {\n throw new Error(`swe-bench: evaluator failed for ${taskId}`)\n }\n const outcomes = [\n statusIds.resolved.includes(taskId),\n statusIds.unresolved.includes(taskId),\n statusIds.emptyPatch.includes(taskId),\n ]\n if (outcomes.filter(Boolean).length !== 1) {\n throw new Error(`swe-bench: report has no unique outcome for ${taskId}`)\n }\n if ((outcomes[0] || outcomes[1]) && !statusIds.completed.includes(taskId)) {\n throw new Error(`swe-bench: report lacks a completed evaluation for ${taskId}`)\n }\n const resolved = outcomes[0]\n return { resolved, score: resolved ? 1 : 0, detail: JSON.stringify(report) }\n}\n\nexport function sweEvaluationArgv(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}): string[] {\n return sweEvaluationArgvForHarness(args, true)\n}\n\nfunction sweEvaluationArgvForHarness(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}, legacyFlags: boolean): string[] {\n const common = [\n '-m', 'swebench.harness.run_evaluation',\n '--dataset_name', DATASET,\n '--predictions_path', args.predictionsPath,\n '--run_id', args.runId,\n '--instance_ids', args.instanceId,\n '--max_workers', '1',\n ]\n if (!legacyFlags) return common\n return [...common, '--namespace', args.namespace ?? scorerNamespace(), '--cache_level', args.cacheLevel]\n}\n\nfunction shellQuote(value: string): string {\n return `'${value.replace(/'/g, `'\\\\''`)}'`\n}\n\nfunction sweMetadata(task: BenchTask): { repo: string; base: string } {\n const repo = String(task.metadata?.repo ?? '')\n const base = String(task.metadata?.base_commit ?? '')\n if (!/^[A-Za-z0-9_.-]+\\/[A-Za-z0-9_.-]+$/.test(repo)) {\n throw new Error(`swe-bench: invalid repo metadata for ${task.id}: ${repo}`)\n }\n if (!/^[0-9a-f]{7,40}$/i.test(base)) {\n throw new Error(`swe-bench: invalid base_commit metadata for ${task.id}: ${base}`)\n }\n return { repo, base }\n}\n\nexport function createSweBenchAdapter(options: SweBenchAdapterOptions = {}): BenchmarkAdapter {\n if (\n options.timeoutMs !== undefined\n && (!Number.isSafeInteger(options.timeoutMs) || options.timeoutMs <= 0)\n ) throw new Error('swe-bench: timeoutMs must be a positive integer')\n const cacheLevel = options.cacheLevel ?? 'env'\n if (!SWE_CACHE_LEVELS.has(cacheLevel)) throw new Error('swe-bench: invalid cacheLevel')\n if (\n options.captureEvaluatorArtifacts !== undefined\n && typeof options.captureEvaluatorArtifacts !== 'function'\n ) throw new Error('swe-bench: captureEvaluatorArtifacts must be a function')\n let attemptSequence = 0\n let legacyFlags: boolean | undefined\n return {\n name: 'swe-bench-verified',\n output: swePatchOutput,\n\n // Extract the patch from repo STATE, not printed text: stage every edit the\n // agent made in the cloned repo and diff it against the checked-out\n // base_commit (`HEAD`). Test files are excluded — the judge applies the gold\n // `test_patch` itself, so an agent edit to a test would collide on apply. The\n // paths come out `a/<repo-relative>` (cwd = repo root), matching the gold\n // patch format the swebench judge's `git apply` expects.\n // Pre-stage: clone the instance repo at base_commit into SWE_REPO_DIR so the\n // agent only edits (the harness owns the checkout — a stochastic model can't be\n // trusted to clone to an exact path). `--quiet` keeps the exec output small.\n boxSetup(task) {\n const { repo, base } = sweMetadata(task)\n return {\n command: `rm -rf ${shellQuote(SWE_REPO_DIR)} && git clone --quiet ${shellQuote(`https://github.com/${repo}`)} ${shellQuote(SWE_REPO_DIR)} && git -C ${shellQuote(SWE_REPO_DIR)} checkout --quiet ${shellQuote(base)}`,\n }\n },\n boxExtract() {\n return {\n command: `git -C ${shellQuote(SWE_REPO_DIR)} add -A && git -C ${shellQuote(SWE_REPO_DIR)} diff --cached -- . ${TEST_FILE_EXCLUDES}`,\n }\n },\n\n async preflight() {\n await preflightVenvImports({\n modules: ['swebench'],\n requireDocker: true,\n fix:\n `Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install 'swebench>=4,<6' ; ` +\n `(2) ensure the Docker daemon is running (the judge builds per-instance images).`,\n })\n legacyFlags = await detectLegacyFlags()\n },\n\n async loadTasks(opts: LoadOptions = {}) {\n const limit = opts.limit ?? 10\n const split = opts.split ?? 'test'\n // Dump instances as JSON via the datasets loader (HF download on first run).\n const script = `\nimport json, sys\nfrom datasets import load_dataset\nds = load_dataset(${JSON.stringify(DATASET)}, split=${JSON.stringify(split)})\nids = set(json.loads(sys.argv[1])) if len(sys.argv) > 1 and sys.argv[1] else None\nout = []\nfor r in ds:\n if ids is not None and r[\"instance_id\"] not in ids:\n continue\n out.append({\n \"instance_id\": r[\"instance_id\"], \"repo\": r[\"repo\"], \"base_commit\": r[\"base_commit\"],\n \"problem_statement\": r[\"problem_statement\"], \"patch\": r[\"patch\"], \"test_patch\": r[\"test_patch\"],\n \"FAIL_TO_PASS\": r[\"FAIL_TO_PASS\"], \"PASS_TO_PASS\": r[\"PASS_TO_PASS\"],\n \"version\": r.get(\"version\"), \"environment_setup_commit\": r.get(\"environment_setup_commit\"),\n })\n if ids is None and len(out) >= ${limit}:\n break\nprint(json.dumps(out))\n`\n const stdout = await runVenvPython(script, [opts.ids ? JSON.stringify(opts.ids) : ''])\n const rows = JSON.parse(stdout) as Array<Record<string, unknown>>\n return rows.map(\n (r): BenchTask => ({\n id: String(r.instance_id),\n split,\n prompt: [\n `Repository: ${r.repo} @ ${r.base_commit}`,\n '',\n `The repository is ALREADY cloned at ${SWE_REPO_DIR}, checked out at commit ${r.base_commit}. Work there directly (\\`cd ${SWE_REPO_DIR}\\`); do not re-clone.`,\n '',\n 'Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.',\n 'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your committed edits directly from the repo.',\n '',\n '--- Issue ---',\n String(r.problem_statement ?? ''),\n ].join('\\n'),\n metadata: r,\n }),\n )\n },\n\n async goldArtifact(task: BenchTask) {\n const gold = task.metadata?.patch\n return typeof gold === 'string' ? gold : undefined\n },\n\n async judge(task: BenchTask, artifact: string): Promise<BenchScore> {\n const useLegacyFlags = await ensureLegacyFlags()\n const runId = safeRunId('bench', task.id)\n const capture = options.captureEvaluatorArtifacts?.({\n taskId: task.id,\n runId,\n attemptSequence: ++attemptSequence,\n })\n return runStagedJudge({\n tmpPrefix: 'swebench-',\n ...(options.timeoutMs === undefined ? {} : { timeoutMs: options.timeoutMs }),\n ...(capture === undefined ? {} : { capture }),\n // Debug: retain the staged dir (holds swebench's per-instance apply/run\n // logs) for post-mortem when SWEBENCH_KEEP_TMP is set. Off by default.\n ...(process.env.SWEBENCH_KEEP_TMP ? { keepTmp: true } : {}),\n async stage(dir) {\n await stageFile(\n join(dir, 'preds.json'),\n JSON.stringify([\n { instance_id: task.id, model_name_or_path: 'agent-runtime-bench', model_patch: artifact },\n ]),\n )\n },\n // The official evaluation harness. Pulls/builds the instance image, applies\n // the patch, runs the test spec, writes a per-run report JSON in cwd.\n argv: (dir) => sweEvaluationArgvForHarness({\n predictionsPath: join(dir, 'preds.json'),\n runId,\n instanceId: task.id,\n cacheLevel,\n namespace: scorerNamespace(),\n }, useLegacyFlags),\n async parseReport(dir) {\n // Report file: agent-runtime-bench.<run_id>.json\n const report = await readJsonReport<SweReport>(join(dir, `agent-runtime-bench.${runId}.json`))\n return scoreSweReport(task.id, report)\n },\n })\n },\n }\n\n async function detectLegacyFlags(): Promise<boolean> {\n const script = `\nimport subprocess, sys\nresult = subprocess.run([sys.executable, '-m', 'swebench.harness.run_evaluation', '--help'], capture_output=True, text=True)\nif result.returncode != 0:\n raise SystemExit(result.stderr or result.stdout or str(result.returncode))\nprint('legacy' if '--namespace' in result.stdout and '--cache_level' in result.stdout else 'modern')\n`\n return (await runVenvPython(script)).trim() === 'legacy'\n }\n\n async function ensureLegacyFlags(): Promise<boolean> {\n if (legacyFlags === undefined) legacyFlags = await detectLegacyFlags()\n return legacyFlags\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAiCA,MAAM,eAAe;;;;;;;;;;AAWrB,MAAa,iBAAwC,EACnD,MAAM,QAAQ;CACZ,IAAI,OAAO;CACX,KAAK,MAAM,MAAM,QAAQ;EACvB,MAAM,IAAK,IAA2C;EACtD,MAAM,IAAI,GAAG,aAAa,GAAG,QAAQ,GAAG;EACxC,IAAI,OAAO,MAAM,YAAY,EAAE,SAAS,GAAG,OAAO;CACpD;CAKA,QADa,CADG,GAAG,KAAK,SAAS,uCAAuC,CACtD,CAAC,CAAC,GAAG,EAAE,CAAC,GAAG,MACb,KAAA,CAAM,KAAK;AAC7B,EACF;AAEA,MAAM,UAAU;AAsBhB,MAAM,mCAAmB,IAAI,IAAwB;CAAC;CAAQ;CAAQ;CAAO;AAAU,CAAC;AAExF,SAAS,kBAAuC;CAC9C,MAAM,YAAY,QAAQ,IAAI,sBAAsB;CACpD,IAAI,cAAc,cAAc,cAAc,QAC5C,MAAM,IAAI,MAAM,kDAAkD,UAAU,EAAE;CAEhF,OAAO;AACT;AACA,MAAM,qBAAqB;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC,CAAC,KAAK,GAAG;AAaV,SAAS,UAAU,QAAiC,KAAgC;CAClF,MAAM,QAAQ,OAAO;CACrB,IAAI,UAAU,KAAA,GAAW,OAAO,CAAC;CACjC,IAAI,CAAC,MAAM,QAAQ,KAAK,KAAK,MAAM,MAAM,UAAU,OAAO,UAAU,QAAQ,GAC1E,MAAM,IAAI,MAAM,wBAAwB,KAAK;CAE/C,OAAO;AACT;;AAGA,SAAgB,eAAe,QAAgB,OAA4B;CACzE,IAAI,CAAC,SAAS,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GAC5D,MAAM,IAAI,MAAM,qCAAqC;CAEvD,MAAM,SAAS;CACf,MAAM,YAAY;EAChB,UAAU,UAAU,QAAQ,cAAc;EAC1C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,YAAY,UAAU,QAAQ,iBAAiB;EAC/C,WAAW,UAAU,QAAQ,eAAe;EAC5C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,OAAO,UAAU,QAAQ,WAAW;CACtC;CACA,MAAM,YAAY,UAAU,QAAQ,eAAe;CAEnD,IADkB,OAAO,OAAO,SAAS,CAAC,CAAC,KAEjC,CAAC,CAAC,MAAM,OAAO,OAAO,MAAM,KAChC,UAAU,SAAS,MAAM,UAAU,WAAW,KAAK,UAAU,OAAO,SAExE,MAAM,IAAI,MAAM,2CAA2C,QAAQ;CAErE,IAAI,UAAU,MAAM,SAAS,MAAM,KAAK,UAAU,WAAW,SAAS,MAAM,GAC1E,MAAM,IAAI,MAAM,mCAAmC,QAAQ;CAE7D,MAAM,WAAW;EACf,UAAU,SAAS,SAAS,MAAM;EAClC,UAAU,WAAW,SAAS,MAAM;EACpC,UAAU,WAAW,SAAS,MAAM;CACtC;CACA,IAAI,SAAS,OAAO,OAAO,CAAC,CAAC,WAAW,GACtC,MAAM,IAAI,MAAM,+CAA+C,QAAQ;CAEzE,KAAK,SAAS,MAAM,SAAS,OAAO,CAAC,UAAU,UAAU,SAAS,MAAM,GACtE,MAAM,IAAI,MAAM,sDAAsD,QAAQ;CAEhF,MAAM,WAAW,SAAS;CAC1B,OAAO;EAAE;EAAU,OAAO,WAAW,IAAI;EAAG,QAAQ,KAAK,UAAU,MAAM;CAAE;AAC7E;AAEA,SAAgB,kBAAkB,MAMrB;CACX,OAAO,4BAA4B,MAAM,IAAI;AAC/C;AAEA,SAAS,4BAA4B,MAMlC,aAAgC;CACjC,MAAM,SAAS;EACb;EAAM;EACN;EAAkB;EAClB;EAAsB,KAAK;EAC3B;EAAY,KAAK;EACjB;EAAkB,KAAK;EACvB;EAAiB;CACnB;CACA,IAAI,CAAC,aAAa,OAAO;CACzB,OAAO;EAAC,GAAG;EAAQ;EAAe,KAAK,aAAa,gBAAgB;EAAG;EAAiB,KAAK;CAAU;AACzG;AAEA,SAAS,WAAW,OAAuB;CACzC,OAAO,IAAI,MAAM,QAAQ,MAAM,OAAO,EAAE;AAC1C;AAEA,SAAS,YAAY,MAAiD;CACpE,MAAM,OAAO,OAAO,KAAK,UAAU,QAAQ,EAAE;CAC7C,MAAM,OAAO,OAAO,KAAK,UAAU,eAAe,EAAE;CACpD,IAAI,CAAC,qCAAqC,KAAK,IAAI,GACjD,MAAM,IAAI,MAAM,wCAAwC,KAAK,GAAG,IAAI,MAAM;CAE5E,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAChC,MAAM,IAAI,MAAM,+CAA+C,KAAK,GAAG,IAAI,MAAM;CAEnF,OAAO;EAAE;EAAM;CAAK;AACtB;AAEA,SAAgB,sBAAsB,UAAkC,CAAC,GAAqB;CAC5F,IACE,QAAQ,cAAc,KAAA,MAClB,CAAC,OAAO,cAAc,QAAQ,SAAS,KAAK,QAAQ,aAAa,IACrE,MAAM,IAAI,MAAM,iDAAiD;CACnE,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,iBAAiB,IAAI,UAAU,GAAG,MAAM,IAAI,MAAM,+BAA+B;CACtF,IACE,QAAQ,8BAA8B,KAAA,KACnC,OAAO,QAAQ,8BAA8B,YAChD,MAAM,IAAI,MAAM,yDAAyD;CAC3E,IAAI,kBAAkB;CACtB,IAAI;CACJ,OAAO;EACL,MAAM;EACN,QAAQ;EAWR,SAAS,MAAM;GACb,MAAM,EAAE,MAAM,SAAS,YAAY,IAAI;GACvC,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,wBAAwB,WAAW,sBAAsB,MAAM,EAAE,GAAG,WAAW,YAAY,EAAE,aAAa,WAAW,YAAY,EAAE,oBAAoB,WAAW,IAAI,IACpN;EACF;EACA,aAAa;GACX,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,oBAAoB,WAAW,YAAY,EAAE,sBAAsB,qBACjH;EACF;EAEA,MAAM,YAAY;GAChB,MAAM,qBAAqB;IACzB,SAAS,CAAC,UAAU;IACpB,eAAe;IACf,KACE;GAEJ,CAAC;GACD,cAAc,MAAM,kBAAkB;EACxC;EAEA,MAAM,UAAU,OAAoB,CAAC,GAAG;GACtC,MAAM,QAAQ,KAAK,SAAS;GAC5B,MAAM,QAAQ,KAAK,SAAS;GAqB5B,MAAM,SAAS,MAAM,cAAc;;;oBAhBrB,KAAK,UAAU,OAAO,EAAE,UAAU,KAAK,UAAU,KAAK,EAAE;;;;;;;;;;;;qCAYvC,MAAM;;;GAIM,CAAC,KAAK,MAAM,KAAK,UAAU,KAAK,GAAG,IAAI,EAAE,CAAC;GAErF,OADa,KAAK,MAAM,MACd,CAAC,CAAC,KACT,OAAkB;IACjB,IAAI,OAAO,EAAE,WAAW;IACxB;IACA,QAAQ;KACN,eAAe,EAAE,KAAK,KAAK,EAAE;KAC7B;KACA,uCAAuC,aAAa,0BAA0B,EAAE,YAAY,8BAA8B,aAAa;KACvI;KACA;KACA;KACA;KACA;KACA,OAAO,EAAE,qBAAqB,EAAE;IAClC,CAAC,CAAC,KAAK,IAAI;IACX,UAAU;GACZ,EACF;EACF;EAEA,MAAM,aAAa,MAAiB;GAClC,MAAM,OAAO,KAAK,UAAU;GAC5B,OAAO,OAAO,SAAS,WAAW,OAAO,KAAA;EAC3C;EAEA,MAAM,MAAM,MAAiB,UAAuC;GAClE,MAAM,iBAAiB,MAAM,kBAAkB;GAC/C,MAAM,QAAQ,UAAU,SAAS,KAAK,EAAE;GACxC,MAAM,UAAU,QAAQ,4BAA4B;IAClD,QAAQ,KAAK;IACb;IACA,iBAAiB,EAAE;GACrB,CAAC;GACD,OAAO,eAAe;IACpB,WAAW;IACX,GAAI,QAAQ,cAAc,KAAA,IAAY,CAAC,IAAI,EAAE,WAAW,QAAQ,UAAU;IAC1E,GAAI,YAAY,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ;IAG3C,GAAI,QAAQ,IAAI,oBAAoB,EAAE,SAAS,KAAK,IAAI,CAAC;IACzD,MAAM,MAAM,KAAK;KACf,MAAM,UACJ,KAAK,KAAK,YAAY,GACtB,KAAK,UAAU,CACb;MAAE,aAAa,KAAK;MAAI,oBAAoB;MAAuB,aAAa;KAAS,CAC3F,CAAC,CACH;IACF;IAGA,OAAO,QAAQ,4BAA4B;KACzC,iBAAiB,KAAK,KAAK,YAAY;KACvC;KACA,YAAY,KAAK;KACjB;KACA,WAAW,gBAAgB;IAC7B,GAAG,cAAc;IACjB,MAAM,YAAY,KAAK;KAErB,MAAM,SAAS,MAAM,eAA0B,KAAK,KAAK,uBAAuB,MAAM,MAAM,CAAC;KAC7F,OAAO,eAAe,KAAK,IAAI,MAAM;IACvC;GACF,CAAC;EACH;CACF;CAEA,eAAe,oBAAsC;EAQnD,QAAQ,MAAM,cAAc;;;;;;CAAM,EAAA,CAAG,KAAK,MAAM;CAClD;CAEA,eAAe,oBAAsC;EACnD,IAAI,gBAAgB,KAAA,GAAW,cAAc,MAAM,kBAAkB;EACrE,OAAO;CACT;AACF"}
|
|
1
|
+
{"version":3,"file":"swe-bench.js","names":[],"sources":["../../src/benchmarks/swe-bench.ts"],"sourcesContent":["/**\n * SWE-bench Verified adapter. Worker artifact = a unified-diff patch. Judge =\n * the official `swebench` harness: apply the patch in the instance's Docker\n * image, run FAIL_TO_PASS + PASS_TO_PASS, report `resolved`. Fully deterministic\n * — no LLM judge.\n *\n * Requires: the bench `.venv` with `swebench` installed + a running Docker\n * daemon (per-instance images are pulled/built on first run).\n *\n * Process/Docker/report plumbing is shared via ./_harness; this file owns the\n * SWE-specific pieces: the patch OutputAdapter, the dataset dump, and the\n * predictions-file → run_evaluation argv → report-shape mapping.\n */\n\nimport { join } from 'node:path'\nimport type { OutputAdapter } from '@tangle-network/agent-runtime/kernel'\nimport {\n preflightVenvImports,\n readJsonReport,\n runStagedJudge,\n runVenvPython,\n safeRunId,\n stageFile,\n type StagedRunCaptureSpec,\n} from './_harness'\nimport type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'\n\n/** Root-level directories are not writable in every sandbox; use the session workspace. */\nconst SWE_REPO_DIR = './swe-bench-repo'\n\n/**\n * The SWE deliverable's FALLBACK parser, from the agent's event STREAM.\n *\n * The PRIMARY deliverable is `boxExtract` below: a `git diff` of the agent's\n * actual edits, read from the cloned repo's STATE inside the box (standard\n * SWE-bench practice). This event-stream parse only runs when that diff is empty\n * — a model that edited the source correctly but never printed a fenced diff (the\n * exact failure this replaces) still scores off its real changes, not its prose.\n */\nexport const swePatchOutput: OutputAdapter<string> = {\n parse(events) {\n let text = ''\n for (const ev of events) {\n const d = (ev as { data?: Record<string, unknown> })?.data\n const t = d?.finalText ?? d?.text ?? d?.result\n if (typeof t === 'string' && t.length > 0) text = t\n }\n // Last ```diff/```patch fenced block (the contract the prompt asks for);\n // fall back to the raw text so a fence-less but valid diff still reaches the judge.\n const fences = [...text.matchAll(/```(?:diff|patch)?\\s*\\n([\\s\\S]*?)```/g)]\n const last = fences.at(-1)?.[1]\n return (last ?? text).trim()\n },\n}\n\nconst DATASET = 'princeton-nlp/SWE-bench_Verified'\nexport type SweBenchCacheLevel = 'none' | 'base' | 'env' | 'instance'\n\nexport interface SweBenchArtifactCaptureContext {\n readonly taskId: string\n readonly runId: string\n /** One-based sequence unique within this adapter instance. */\n readonly attemptSequence: number\n}\n\nexport interface SweBenchAdapterOptions {\n readonly timeoutMs?: number\n readonly cacheLevel?: SweBenchCacheLevel\n /**\n * Return a unique destination for any attempt whose complete official\n * evaluator directory and process logs should be retained.\n */\n readonly captureEvaluatorArtifacts?: (\n context: SweBenchArtifactCaptureContext,\n ) => StagedRunCaptureSpec | undefined\n}\n\nconst SWE_CACHE_LEVELS = new Set<SweBenchCacheLevel>(['none', 'base', 'env', 'instance'])\n\nfunction scorerNamespace(): 'swebench' | 'none' {\n const namespace = process.env.SWEBENCH_NAMESPACE ?? 'swebench'\n if (namespace !== 'swebench' && namespace !== 'none') {\n throw new Error(`SWEBENCH_NAMESPACE must be swebench|none, got \"${namespace}\"`)\n }\n return namespace\n}\nconst TEST_FILE_EXCLUDES = [\n \"':(exclude,glob)**/tests/**'\",\n \"':(exclude,glob)**/test/**'\",\n \"':(exclude,glob)test_*.py'\",\n \"':(exclude,glob)**/test_*.py'\",\n \"':(exclude,glob)*_test.py'\",\n \"':(exclude,glob)**/*_test.py'\",\n \"':(exclude,glob)conftest.py'\",\n \"':(exclude,glob)**/conftest.py'\",\n].join(' ')\n\ninterface SweReport {\n resolved_instances?: number\n resolved_ids?: string[]\n unresolved_ids?: string[]\n empty_patch_ids?: string[]\n completed_ids?: string[]\n incomplete_ids?: string[]\n error_ids?: string[]\n submitted_ids?: string[]\n}\n\nfunction stringIds(report: Record<string, unknown>, key: keyof SweReport): string[] {\n const value = report[key]\n if (value === undefined) return []\n if (!Array.isArray(value) || value.some((entry) => typeof entry !== 'string')) {\n throw new Error(`swe-bench: malformed ${key}`)\n }\n return value\n}\n\n/** Convert one official report into a score without turning evaluator failures into agent failures. */\nexport function scoreSweReport(taskId: string, value: unknown): BenchScore {\n if (!value || typeof value !== 'object' || Array.isArray(value)) {\n throw new Error('swe-bench: report must be an object')\n }\n const report = value as Record<string, unknown>\n const statusIds = {\n resolved: stringIds(report, 'resolved_ids'),\n unresolved: stringIds(report, 'unresolved_ids'),\n emptyPatch: stringIds(report, 'empty_patch_ids'),\n completed: stringIds(report, 'completed_ids'),\n incomplete: stringIds(report, 'incomplete_ids'),\n error: stringIds(report, 'error_ids'),\n }\n const submitted = stringIds(report, 'submitted_ids')\n const mentioned = Object.values(statusIds).flat()\n if (\n mentioned.some((id) => id !== taskId)\n || (submitted.length > 0 && (submitted.length !== 1 || submitted[0] !== taskId))\n ) {\n throw new Error(`swe-bench: report identity mismatch for ${taskId}`)\n }\n if (statusIds.error.includes(taskId) || statusIds.incomplete.includes(taskId)) {\n throw new Error(`swe-bench: evaluator failed for ${taskId}`)\n }\n const outcomes = [\n statusIds.resolved.includes(taskId),\n statusIds.unresolved.includes(taskId),\n statusIds.emptyPatch.includes(taskId),\n ]\n if (outcomes.filter(Boolean).length !== 1) {\n throw new Error(`swe-bench: report has no unique outcome for ${taskId}`)\n }\n if ((outcomes[0] || outcomes[1]) && !statusIds.completed.includes(taskId)) {\n throw new Error(`swe-bench: report lacks a completed evaluation for ${taskId}`)\n }\n const resolved = outcomes[0]\n return { resolved, score: resolved ? 1 : 0, detail: JSON.stringify(report) }\n}\n\nexport function sweEvaluationArgv(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}): string[] {\n return sweEvaluationArgvForHarness(args, true)\n}\n\nfunction sweEvaluationArgvForHarness(args: {\n readonly predictionsPath: string\n readonly runId: string\n readonly instanceId: string\n readonly cacheLevel: SweBenchCacheLevel\n readonly namespace?: 'swebench' | 'none'\n}, legacyFlags: boolean): string[] {\n const common = [\n '-m', 'swebench.harness.run_evaluation',\n '--dataset_name', DATASET,\n '--predictions_path', args.predictionsPath,\n '--run_id', args.runId,\n '--instance_ids', args.instanceId,\n '--max_workers', '1',\n ]\n if (!legacyFlags) return common\n return [...common, '--namespace', args.namespace ?? scorerNamespace(), '--cache_level', args.cacheLevel]\n}\n\nfunction shellQuote(value: string): string {\n return `'${value.replace(/'/g, `'\\\\''`)}'`\n}\n\nfunction sweMetadata(task: BenchTask): { repo: string; base: string } {\n const repo = String(task.metadata?.repo ?? '')\n const base = String(task.metadata?.base_commit ?? '')\n if (!/^[A-Za-z0-9_.-]+\\/[A-Za-z0-9_.-]+$/.test(repo)) {\n throw new Error(`swe-bench: invalid repo metadata for ${task.id}: ${repo}`)\n }\n if (!/^[0-9a-f]{7,40}$/i.test(base)) {\n throw new Error(`swe-bench: invalid base_commit metadata for ${task.id}: ${base}`)\n }\n return { repo, base }\n}\n\nexport function createSweBenchAdapter(options: SweBenchAdapterOptions = {}): BenchmarkAdapter {\n if (\n options.timeoutMs !== undefined\n && (!Number.isSafeInteger(options.timeoutMs) || options.timeoutMs <= 0)\n ) throw new Error('swe-bench: timeoutMs must be a positive integer')\n const cacheLevel = options.cacheLevel ?? 'env'\n if (!SWE_CACHE_LEVELS.has(cacheLevel)) throw new Error('swe-bench: invalid cacheLevel')\n if (\n options.captureEvaluatorArtifacts !== undefined\n && typeof options.captureEvaluatorArtifacts !== 'function'\n ) throw new Error('swe-bench: captureEvaluatorArtifacts must be a function')\n let attemptSequence = 0\n let legacyFlags: boolean | undefined\n return {\n name: 'swe-bench-verified',\n output: swePatchOutput,\n\n // Extract the patch from repo STATE, not printed text: stage every edit the\n // agent made in the cloned repo and diff it against the checked-out\n // base_commit (`HEAD`). Test files are excluded — the judge applies the gold\n // `test_patch` itself, so an agent edit to a test would collide on apply. The\n // paths come out `a/<repo-relative>` (cwd = repo root), matching the gold\n // patch format the swebench judge's `git apply` expects.\n // Pre-stage: clone the instance repo at base_commit into SWE_REPO_DIR so the\n // agent only edits (the harness owns the checkout — a stochastic model can't be\n // trusted to clone to an exact path). `--quiet` keeps the exec output small.\n boxSetup(task) {\n const { repo, base } = sweMetadata(task)\n return {\n command: `rm -rf ${shellQuote(SWE_REPO_DIR)} && git clone --quiet ${shellQuote(`https://github.com/${repo}`)} ${shellQuote(SWE_REPO_DIR)} && git -C ${shellQuote(SWE_REPO_DIR)} checkout --quiet ${shellQuote(base)}`,\n }\n },\n boxExtract() {\n return {\n command: `git -C ${shellQuote(SWE_REPO_DIR)} add -A && git -C ${shellQuote(SWE_REPO_DIR)} diff --cached -- . ${TEST_FILE_EXCLUDES}`,\n }\n },\n\n async preflight() {\n await preflightVenvImports({\n modules: ['swebench'],\n requireDocker: true,\n fix:\n `Fix: (1) python3 -m venv bench/.venv && bench/.venv/bin/pip install 'swebench>=4,<6' ; ` +\n `(2) ensure the Docker daemon is running (the judge builds per-instance images).`,\n })\n legacyFlags = await detectLegacyFlags()\n },\n\n async loadTasks(opts: LoadOptions = {}) {\n const limit = opts.limit ?? 10\n const split = opts.split ?? 'test'\n // Dump instances as JSON via the datasets loader (HF download on first run).\n const script = `\nimport json, sys\nfrom datasets import load_dataset\nds = load_dataset(${JSON.stringify(DATASET)}, split=${JSON.stringify(split)})\nids = set(json.loads(sys.argv[1])) if len(sys.argv) > 1 and sys.argv[1] else None\nout = []\nfor r in ds:\n if ids is not None and r[\"instance_id\"] not in ids:\n continue\n out.append({\n \"instance_id\": r[\"instance_id\"], \"repo\": r[\"repo\"], \"base_commit\": r[\"base_commit\"],\n \"problem_statement\": r[\"problem_statement\"], \"patch\": r[\"patch\"], \"test_patch\": r[\"test_patch\"],\n \"FAIL_TO_PASS\": r[\"FAIL_TO_PASS\"], \"PASS_TO_PASS\": r[\"PASS_TO_PASS\"],\n \"version\": r.get(\"version\"), \"environment_setup_commit\": r.get(\"environment_setup_commit\"),\n })\n if ids is None and len(out) >= ${limit}:\n break\nprint(json.dumps(out))\n`\n const stdout = await runVenvPython(script, [opts.ids ? JSON.stringify(opts.ids) : ''])\n const rows = JSON.parse(stdout) as Array<Record<string, unknown>>\n return rows.map(\n (r): BenchTask => ({\n id: String(r.instance_id),\n split,\n prompt: [\n `Repository: ${r.repo} @ ${r.base_commit}`,\n '',\n `The repository is ALREADY cloned at ${SWE_REPO_DIR} relative to your initial session workspace, checked out at commit ${r.base_commit}. Work there directly (\\`cd ${SWE_REPO_DIR}\\`); do not re-clone.`,\n '',\n 'Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.',\n 'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your source edits directly from the repo, including uncommitted changes.',\n '',\n '--- Issue ---',\n String(r.problem_statement ?? ''),\n ].join('\\n'),\n metadata: r,\n }),\n )\n },\n\n async goldArtifact(task: BenchTask) {\n const gold = task.metadata?.patch\n return typeof gold === 'string' ? gold : undefined\n },\n\n async judge(task: BenchTask, artifact: string): Promise<BenchScore> {\n const useLegacyFlags = await ensureLegacyFlags()\n const runId = safeRunId('bench', task.id)\n const capture = options.captureEvaluatorArtifacts?.({\n taskId: task.id,\n runId,\n attemptSequence: ++attemptSequence,\n })\n return runStagedJudge({\n tmpPrefix: 'swebench-',\n ...(options.timeoutMs === undefined ? {} : { timeoutMs: options.timeoutMs }),\n ...(capture === undefined ? {} : { capture }),\n // Debug: retain the staged dir (holds swebench's per-instance apply/run\n // logs) for post-mortem when SWEBENCH_KEEP_TMP is set. Off by default.\n ...(process.env.SWEBENCH_KEEP_TMP ? { keepTmp: true } : {}),\n async stage(dir) {\n await stageFile(\n join(dir, 'preds.json'),\n JSON.stringify([\n { instance_id: task.id, model_name_or_path: 'agent-runtime-bench', model_patch: artifact },\n ]),\n )\n },\n // The official evaluation harness. Pulls/builds the instance image, applies\n // the patch, runs the test spec, writes a per-run report JSON in cwd.\n argv: (dir) => sweEvaluationArgvForHarness({\n predictionsPath: join(dir, 'preds.json'),\n runId,\n instanceId: task.id,\n cacheLevel,\n namespace: scorerNamespace(),\n }, useLegacyFlags),\n async parseReport(dir) {\n // Report file: agent-runtime-bench.<run_id>.json\n const report = await readJsonReport<SweReport>(join(dir, `agent-runtime-bench.${runId}.json`))\n return scoreSweReport(task.id, report)\n },\n })\n },\n }\n\n async function detectLegacyFlags(): Promise<boolean> {\n const script = `\nimport subprocess, sys\nresult = subprocess.run([sys.executable, '-m', 'swebench.harness.run_evaluation', '--help'], capture_output=True, text=True)\nif result.returncode != 0:\n raise SystemExit(result.stderr or result.stdout or str(result.returncode))\nprint('legacy' if '--namespace' in result.stdout and '--cache_level' in result.stdout else 'modern')\n`\n return (await runVenvPython(script)).trim() === 'legacy'\n }\n\n async function ensureLegacyFlags(): Promise<boolean> {\n if (legacyFlags === undefined) legacyFlags = await detectLegacyFlags()\n return legacyFlags\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;AA4BA,MAAM,eAAe;;;;;;;;;;AAWrB,MAAa,iBAAwC,EACnD,MAAM,QAAQ;CACZ,IAAI,OAAO;CACX,KAAK,MAAM,MAAM,QAAQ;EACvB,MAAM,IAAK,IAA2C;EACtD,MAAM,IAAI,GAAG,aAAa,GAAG,QAAQ,GAAG;EACxC,IAAI,OAAO,MAAM,YAAY,EAAE,SAAS,GAAG,OAAO;CACpD;CAKA,QADa,CADG,GAAG,KAAK,SAAS,uCAAuC,CACtD,CAAC,CAAC,GAAG,EAAE,CAAC,GAAG,MACb,KAAA,CAAM,KAAK;AAC7B,EACF;AAEA,MAAM,UAAU;AAsBhB,MAAM,mCAAmB,IAAI,IAAwB;CAAC;CAAQ;CAAQ;CAAO;AAAU,CAAC;AAExF,SAAS,kBAAuC;CAC9C,MAAM,YAAY,QAAQ,IAAI,sBAAsB;CACpD,IAAI,cAAc,cAAc,cAAc,QAC5C,MAAM,IAAI,MAAM,kDAAkD,UAAU,EAAE;CAEhF,OAAO;AACT;AACA,MAAM,qBAAqB;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC,CAAC,KAAK,GAAG;AAaV,SAAS,UAAU,QAAiC,KAAgC;CAClF,MAAM,QAAQ,OAAO;CACrB,IAAI,UAAU,KAAA,GAAW,OAAO,CAAC;CACjC,IAAI,CAAC,MAAM,QAAQ,KAAK,KAAK,MAAM,MAAM,UAAU,OAAO,UAAU,QAAQ,GAC1E,MAAM,IAAI,MAAM,wBAAwB,KAAK;CAE/C,OAAO;AACT;;AAGA,SAAgB,eAAe,QAAgB,OAA4B;CACzE,IAAI,CAAC,SAAS,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GAC5D,MAAM,IAAI,MAAM,qCAAqC;CAEvD,MAAM,SAAS;CACf,MAAM,YAAY;EAChB,UAAU,UAAU,QAAQ,cAAc;EAC1C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,YAAY,UAAU,QAAQ,iBAAiB;EAC/C,WAAW,UAAU,QAAQ,eAAe;EAC5C,YAAY,UAAU,QAAQ,gBAAgB;EAC9C,OAAO,UAAU,QAAQ,WAAW;CACtC;CACA,MAAM,YAAY,UAAU,QAAQ,eAAe;CAEnD,IADkB,OAAO,OAAO,SAAS,CAAC,CAAC,KAEjC,CAAC,CAAC,MAAM,OAAO,OAAO,MAAM,KAChC,UAAU,SAAS,MAAM,UAAU,WAAW,KAAK,UAAU,OAAO,SAExE,MAAM,IAAI,MAAM,2CAA2C,QAAQ;CAErE,IAAI,UAAU,MAAM,SAAS,MAAM,KAAK,UAAU,WAAW,SAAS,MAAM,GAC1E,MAAM,IAAI,MAAM,mCAAmC,QAAQ;CAE7D,MAAM,WAAW;EACf,UAAU,SAAS,SAAS,MAAM;EAClC,UAAU,WAAW,SAAS,MAAM;EACpC,UAAU,WAAW,SAAS,MAAM;CACtC;CACA,IAAI,SAAS,OAAO,OAAO,CAAC,CAAC,WAAW,GACtC,MAAM,IAAI,MAAM,+CAA+C,QAAQ;CAEzE,KAAK,SAAS,MAAM,SAAS,OAAO,CAAC,UAAU,UAAU,SAAS,MAAM,GACtE,MAAM,IAAI,MAAM,sDAAsD,QAAQ;CAEhF,MAAM,WAAW,SAAS;CAC1B,OAAO;EAAE;EAAU,OAAO,WAAW,IAAI;EAAG,QAAQ,KAAK,UAAU,MAAM;CAAE;AAC7E;AAEA,SAAgB,kBAAkB,MAMrB;CACX,OAAO,4BAA4B,MAAM,IAAI;AAC/C;AAEA,SAAS,4BAA4B,MAMlC,aAAgC;CACjC,MAAM,SAAS;EACb;EAAM;EACN;EAAkB;EAClB;EAAsB,KAAK;EAC3B;EAAY,KAAK;EACjB;EAAkB,KAAK;EACvB;EAAiB;CACnB;CACA,IAAI,CAAC,aAAa,OAAO;CACzB,OAAO;EAAC,GAAG;EAAQ;EAAe,KAAK,aAAa,gBAAgB;EAAG;EAAiB,KAAK;CAAU;AACzG;AAEA,SAAS,WAAW,OAAuB;CACzC,OAAO,IAAI,MAAM,QAAQ,MAAM,OAAO,EAAE;AAC1C;AAEA,SAAS,YAAY,MAAiD;CACpE,MAAM,OAAO,OAAO,KAAK,UAAU,QAAQ,EAAE;CAC7C,MAAM,OAAO,OAAO,KAAK,UAAU,eAAe,EAAE;CACpD,IAAI,CAAC,qCAAqC,KAAK,IAAI,GACjD,MAAM,IAAI,MAAM,wCAAwC,KAAK,GAAG,IAAI,MAAM;CAE5E,IAAI,CAAC,oBAAoB,KAAK,IAAI,GAChC,MAAM,IAAI,MAAM,+CAA+C,KAAK,GAAG,IAAI,MAAM;CAEnF,OAAO;EAAE;EAAM;CAAK;AACtB;AAEA,SAAgB,sBAAsB,UAAkC,CAAC,GAAqB;CAC5F,IACE,QAAQ,cAAc,KAAA,MAClB,CAAC,OAAO,cAAc,QAAQ,SAAS,KAAK,QAAQ,aAAa,IACrE,MAAM,IAAI,MAAM,iDAAiD;CACnE,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,iBAAiB,IAAI,UAAU,GAAG,MAAM,IAAI,MAAM,+BAA+B;CACtF,IACE,QAAQ,8BAA8B,KAAA,KACnC,OAAO,QAAQ,8BAA8B,YAChD,MAAM,IAAI,MAAM,yDAAyD;CAC3E,IAAI,kBAAkB;CACtB,IAAI;CACJ,OAAO;EACL,MAAM;EACN,QAAQ;EAWR,SAAS,MAAM;GACb,MAAM,EAAE,MAAM,SAAS,YAAY,IAAI;GACvC,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,wBAAwB,WAAW,sBAAsB,MAAM,EAAE,GAAG,WAAW,YAAY,EAAE,aAAa,WAAW,YAAY,EAAE,oBAAoB,WAAW,IAAI,IACpN;EACF;EACA,aAAa;GACX,OAAO,EACL,SAAS,UAAU,WAAW,YAAY,EAAE,oBAAoB,WAAW,YAAY,EAAE,sBAAsB,qBACjH;EACF;EAEA,MAAM,YAAY;GAChB,MAAM,qBAAqB;IACzB,SAAS,CAAC,UAAU;IACpB,eAAe;IACf,KACE;GAEJ,CAAC;GACD,cAAc,MAAM,kBAAkB;EACxC;EAEA,MAAM,UAAU,OAAoB,CAAC,GAAG;GACtC,MAAM,QAAQ,KAAK,SAAS;GAC5B,MAAM,QAAQ,KAAK,SAAS;GAqB5B,MAAM,SAAS,MAAM,cAAc;;;oBAhBrB,KAAK,UAAU,OAAO,EAAE,UAAU,KAAK,UAAU,KAAK,EAAE;;;;;;;;;;;;qCAYvC,MAAM;;;GAIM,CAAC,KAAK,MAAM,KAAK,UAAU,KAAK,GAAG,IAAI,EAAE,CAAC;GAErF,OADa,KAAK,MAAM,MACd,CAAC,CAAC,KACT,OAAkB;IACjB,IAAI,OAAO,EAAE,WAAW;IACxB;IACA,QAAQ;KACN,eAAe,EAAE,KAAK,KAAK,EAAE;KAC7B;KACA,uCAAuC,aAAa,qEAAqE,EAAE,YAAY,8BAA8B,aAAa;KAClL;KACA;KACA;KACA;KACA;KACA,OAAO,EAAE,qBAAqB,EAAE;IAClC,CAAC,CAAC,KAAK,IAAI;IACX,UAAU;GACZ,EACF;EACF;EAEA,MAAM,aAAa,MAAiB;GAClC,MAAM,OAAO,KAAK,UAAU;GAC5B,OAAO,OAAO,SAAS,WAAW,OAAO,KAAA;EAC3C;EAEA,MAAM,MAAM,MAAiB,UAAuC;GAClE,MAAM,iBAAiB,MAAM,kBAAkB;GAC/C,MAAM,QAAQ,UAAU,SAAS,KAAK,EAAE;GACxC,MAAM,UAAU,QAAQ,4BAA4B;IAClD,QAAQ,KAAK;IACb;IACA,iBAAiB,EAAE;GACrB,CAAC;GACD,OAAO,eAAe;IACpB,WAAW;IACX,GAAI,QAAQ,cAAc,KAAA,IAAY,CAAC,IAAI,EAAE,WAAW,QAAQ,UAAU;IAC1E,GAAI,YAAY,KAAA,IAAY,CAAC,IAAI,EAAE,QAAQ;IAG3C,GAAI,QAAQ,IAAI,oBAAoB,EAAE,SAAS,KAAK,IAAI,CAAC;IACzD,MAAM,MAAM,KAAK;KACf,MAAM,UACJ,KAAK,KAAK,YAAY,GACtB,KAAK,UAAU,CACb;MAAE,aAAa,KAAK;MAAI,oBAAoB;MAAuB,aAAa;KAAS,CAC3F,CAAC,CACH;IACF;IAGA,OAAO,QAAQ,4BAA4B;KACzC,iBAAiB,KAAK,KAAK,YAAY;KACvC;KACA,YAAY,KAAK;KACjB;KACA,WAAW,gBAAgB;IAC7B,GAAG,cAAc;IACjB,MAAM,YAAY,KAAK;KAErB,MAAM,SAAS,MAAM,eAA0B,KAAK,KAAK,uBAAuB,MAAM,MAAM,CAAC;KAC7F,OAAO,eAAe,KAAK,IAAI,MAAM;IACvC;GACF,CAAC;EACH;CACF;CAEA,eAAe,oBAAsC;EAQnD,QAAQ,MAAM,cAAc;;;;;;CAAM,EAAA,CAAG,KAAK,MAAM;CAClD;CAEA,eAAe,oBAAsC;EACnD,IAAI,gBAAgB,KAAA,GAAW,cAAc,MAAM,kBAAkB;EACrE,OAAO;CACT;AACF"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.13.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.182.0 <0.183.0",
|
|
29
29
|
"@tangle-network/agent-interface": "^2.6.0",
|
|
30
|
-
"@tangle-network/agent-knowledge": "^
|
|
31
|
-
"@tangle-network/sandbox": ">=0.36.4 <0.
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
30
|
+
"@tangle-network/agent-knowledge": "^17.0.1",
|
|
31
|
+
"@tangle-network/sandbox": ">=0.36.4 <0.40.0",
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.229.0"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
|
@@ -86,8 +86,8 @@ async function main() {
|
|
|
86
86
|
if (tests.length === 0) throw new Error('no package tests found under src/')
|
|
87
87
|
|
|
88
88
|
// Two test runtimes coexist under src/: node:test files run under `node --test`;
|
|
89
|
-
//
|
|
90
|
-
//
|
|
89
|
+
// Vitest files need the Vitest worker, so partition by the framework each file
|
|
90
|
+
// actually imports.
|
|
91
91
|
const nodeTests = []
|
|
92
92
|
const vitestTests = []
|
|
93
93
|
for (const file of tests) {
|
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
import assert from 'node:assert/strict'
|
|
2
2
|
import test from 'node:test'
|
|
3
|
+
import { execFileSync } from 'node:child_process'
|
|
4
|
+
import { mkdtempSync, mkdirSync, writeFileSync, rmSync, existsSync } from 'node:fs'
|
|
5
|
+
import { tmpdir } from 'node:os'
|
|
6
|
+
import { join } from 'node:path'
|
|
3
7
|
import { createSweBenchAdapter, scoreSweReport, sweEvaluationArgv } from './swe-bench'
|
|
4
8
|
|
|
5
9
|
const taskId = 'django__django-12345'
|
|
@@ -59,3 +63,48 @@ test('SWE evaluation command preserves the requested instance image', () => {
|
|
|
59
63
|
/invalid cacheLevel/,
|
|
60
64
|
)
|
|
61
65
|
})
|
|
66
|
+
|
|
67
|
+
test('SWE setup and extraction stay in the session workspace and exclude test edits', () => {
|
|
68
|
+
const root = mkdtempSync(join(tmpdir(), 'swe-workspace-'))
|
|
69
|
+
try {
|
|
70
|
+
const origin = join(root, 'origin')
|
|
71
|
+
const workspace = join(root, 'session')
|
|
72
|
+
mkdirSync(origin)
|
|
73
|
+
mkdirSync(workspace)
|
|
74
|
+
const git = (args: string[], input?: string) => execFileSync('git', args, { cwd: origin, input, encoding: 'utf8' }).trim()
|
|
75
|
+
git(['init', '--quiet'])
|
|
76
|
+
const blob = git(['hash-object', '-w', '--stdin'], 'before\n')
|
|
77
|
+
const tree = git(['mktree'], `100644 blob ${blob}\tsource.py\n`)
|
|
78
|
+
// Construct fixture history without changing the developer's Git identity or configuration.
|
|
79
|
+
const base = git(['hash-object', '-t', 'commit', '-w', '--stdin'],
|
|
80
|
+
`tree ${tree}\nauthor Fixture <fixture@example.invalid> 1 +0000\ncommitter Fixture <fixture@example.invalid> 1 +0000\n\nfixture\n`)
|
|
81
|
+
git(['update-ref', 'HEAD', base])
|
|
82
|
+
const task = { id: taskId, prompt: 'fix', metadata: { repo: 'fixture/repo', base_commit: base } }
|
|
83
|
+
const adapter = createSweBenchAdapter()
|
|
84
|
+
const setup = adapter.boxSetup!(task)
|
|
85
|
+
const extract = adapter.boxExtract!(task)
|
|
86
|
+
assert.match(setup.command, /^rm -rf '\.\//)
|
|
87
|
+
assert.equal(setup.cwd, undefined)
|
|
88
|
+
assert.equal(extract.cwd, undefined)
|
|
89
|
+
const env = {
|
|
90
|
+
...process.env,
|
|
91
|
+
GIT_CONFIG_COUNT: '2',
|
|
92
|
+
GIT_CONFIG_KEY_0: `url.file://${origin}.insteadOf`,
|
|
93
|
+
GIT_CONFIG_VALUE_0: 'https://github.com/fixture/repo',
|
|
94
|
+
GIT_CONFIG_KEY_1: 'protocol.file.allow',
|
|
95
|
+
GIT_CONFIG_VALUE_1: 'always',
|
|
96
|
+
}
|
|
97
|
+
execFileSync('sh', ['-c', setup.command], { cwd: workspace, env })
|
|
98
|
+
const repo = join(workspace, 'swe-bench-repo')
|
|
99
|
+
assert.ok(existsSync(join(repo, '.git')))
|
|
100
|
+
writeFileSync(join(repo, 'source.py'), 'after\n')
|
|
101
|
+
mkdirSync(join(repo, 'tests'))
|
|
102
|
+
writeFileSync(join(repo, 'tests', 'test_fix.py'), 'hidden-test-edit\n')
|
|
103
|
+
const patch = execFileSync('sh', ['-c', extract.command], { cwd: workspace, env, encoding: 'utf8' })
|
|
104
|
+
assert.match(patch, /diff --git a\/source.py b\/source.py/)
|
|
105
|
+
assert.match(patch, /\+after/)
|
|
106
|
+
assert.doesNotMatch(patch, /hidden-test-edit|test_fix/)
|
|
107
|
+
} finally {
|
|
108
|
+
rmSync(root, { recursive: true, force: true })
|
|
109
|
+
}
|
|
110
|
+
})
|
|
@@ -25,13 +25,8 @@ import {
|
|
|
25
25
|
} from './_harness'
|
|
26
26
|
import type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'
|
|
27
27
|
|
|
28
|
-
/**
|
|
29
|
-
|
|
30
|
-
* source of truth shared by the prompt template (which tells the agent to clone
|
|
31
|
-
* here) and `boxExtract` (which runs `git diff` here after the shot) — so the
|
|
32
|
-
* harness always knows exactly where the agent's edits live, for any instance.
|
|
33
|
-
*/
|
|
34
|
-
const SWE_REPO_DIR = '/work'
|
|
28
|
+
/** Root-level directories are not writable in every sandbox; use the session workspace. */
|
|
29
|
+
const SWE_REPO_DIR = './swe-bench-repo'
|
|
35
30
|
|
|
36
31
|
/**
|
|
37
32
|
* The SWE deliverable's FALLBACK parser, from the agent's event STREAM.
|
|
@@ -286,10 +281,10 @@ print(json.dumps(out))
|
|
|
286
281
|
prompt: [
|
|
287
282
|
`Repository: ${r.repo} @ ${r.base_commit}`,
|
|
288
283
|
'',
|
|
289
|
-
`The repository is ALREADY cloned at ${SWE_REPO_DIR}, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
284
|
+
`The repository is ALREADY cloned at ${SWE_REPO_DIR} relative to your initial session workspace, checked out at commit ${r.base_commit}. Work there directly (\`cd ${SWE_REPO_DIR}\`); do not re-clone.`,
|
|
290
285
|
'',
|
|
291
286
|
'Resolve this issue by editing the repository SOURCE so the failing tests pass without breaking the passing ones. Do NOT edit test files — the evaluation runs hidden tests on a fresh checkout, so editing tests does not count. Keep the change minimal and confined to the cloned repo.',
|
|
292
|
-
'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your
|
|
287
|
+
'Work iteratively: reproduce the issue, implement the fix in the source, and re-run the relevant tests until they pass. You do NOT need to print the diff — the harness reads your source edits directly from the repo, including uncommitted changes.',
|
|
293
288
|
'',
|
|
294
289
|
'--- Issue ---',
|
|
295
290
|
String(r.problem_statement ?? ''),
|
|
@@ -1,144 +0,0 @@
|
|
|
1
|
-
# QUANT-ARENA — a self-improving trading-strategy lab
|
|
2
|
-
|
|
3
|
-
A small, fully auditable research loop: an AI strategy author writes candidate strategies, every candidate is screened for look-ahead bias, backtested walk-forward on overlapping in-sample windows against pinned benchmarks, and judged by an acceptance rule whose bar **rises with every strategy tried**.
|
|
4
|
-
Every attempt — accepted, rejected, or killed for leaking — becomes a permanent row in a lab notebook.
|
|
5
|
-
The final two years of data are a locked out-of-sample set that only a separate certification command may touch, once.
|
|
6
|
-
|
|
7
|
-
Everything is plain TypeScript you can read in an afternoon: the backtester is one file with zero dependencies, the acceptance math is one file, the data is committed CSV.
|
|
8
|
-
|
|
9
|
-
## 1. Quickstart
|
|
10
|
-
|
|
11
|
-
```bash
|
|
12
|
-
# from bench/ (needs node >= 20, the `claude` CLI logged in, and `uv` on PATH
|
|
13
|
-
# — scoring runs in a pinned python environment, see "Two engines" below)
|
|
14
|
-
npx tsx src/quant-arena/quant-loop.mts --out /tmp/quant-demo --candidates 2
|
|
15
|
-
```
|
|
16
|
-
|
|
17
|
-
That runs a full research campaign: 2 strategy authors x 2 candidates each — the committed capture of exactly this command cost $1.65 of model spend across 8 metered calls; backtests are free.
|
|
18
|
-
No API? `--skip-llm-audit` keeps everything but the adversarial code review; the mechanical look-ahead check still runs.
|
|
19
|
-
|
|
20
|
-
Run the unit tests (backtester hand-computed cases, look-ahead detection, acceptance math, window reproducibility):
|
|
21
|
-
|
|
22
|
-
```bash
|
|
23
|
-
npx vitest run src/quant-arena
|
|
24
|
-
```
|
|
25
|
-
|
|
26
|
-
## 2. What happens when you run it
|
|
27
|
-
|
|
28
|
-
1. **Data loads.** ~8 years of daily bars for 11 tickers (an index `IDX` plus `S01`-`S10`) from `fixtures/data/insample/`. The final 2 years live in `fixtures/data/holdout/` and are **not** loaded — see step 8. The series are synthetic (regime-switching factor model, seeded, regenerable) because the free real-data source we checked licenses personal use only; `fixtures/data/PROVENANCE.md` has the details and how to drop in your own CSVs.
|
|
29
|
-
2. **Evaluation windows are drawn.** 8 overlapping 504-day (~2-year) blocks, block-bootstrap sampled from the in-sample years with a fixed seed — every candidate in the campaign is scored on the same windows, and reruns reproduce bit-identically.
|
|
30
|
-
3. **Benchmarks run.** Three pinned incumbents: buy-and-hold the index, equal-weight monthly rebalance, and a 20/100 moving-average crossover. Their per-window Sharpe ratios define the bar: "best benchmark" is the per-window maximum.
|
|
31
|
-
4. **Strategy authors write code.** Each author is a Claude call with a pinned identity (one plain, one with a quant-researcher system prompt). It gets the strategy contract, the universe summary, the benchmarks' per-window Sharpes, and the current acceptance bar — and must reply with one self-contained TypeScript module exporting `onBar(ctx)`: the harness calls it once per trading day with the bars **up to that day only** (the arrays are physically sliced, so reading the future is structurally impossible), plus the strategy's current holdings and equity, and it answers with target portfolio weights or "hold". Strategies never place orders — a single shared rebalancer (`oms.ts`) turns everyone's target weights into orders under the same sizing rule, LEAN-style `(targetWeight x equity - currentPosition) / price`, long-only. Model spend is metered into a durable cost log (`cost-ledger.jsonl`) with per-call receipts.
|
|
32
|
-
5. **Look-ahead screening, stage 1 (mechanical).** The candidate is re-run on data truncated at several cutoff days. Signals up to each cutoff must be bit-identical to the full-data run — any divergence proves the code read the future, and the candidate is killed with the divergence quoted.
|
|
33
|
-
6. **Look-ahead screening, stage 2 (adversarial).** A second, cheap model reads the source with one job: find look-ahead — indexing past `t`, whole-series statistics feeding per-day decisions, hardcoded dates that smell like memorization. Verdict is JSON; anything but a clean verdict kills the candidate, and an unparseable reply kills it too (the rule fails closed).
|
|
34
|
-
7. **Backtest and verdict.** Survivors are backtested over the whole in-sample period (next-day-open fills, 15 bps one-way costs, no shorting, no leverage) and scored per window. Two engines run: the one-file TypeScript reference engine first, as a fail-closed contract check, then the industry-standard **vectorbt** engine (a persistent python worker in a version-locked environment) produces the official numbers. A parity test suite holds the two engines to agreement on golden fixtures — exact on a no-trade book, within machine precision whenever the book holds cash, and within a documented 0.5% on fully-invested books (the engines differ only in whether fees may be financed by a slightly negative cash balance). The acceptance rule (section 3) decides. Accepted or not, the try is appended to `notebook.jsonl` with its window scores, audit evidence, code hash, and authoring cost.
|
|
35
|
-
8. **Certification, later and by hand.** When you believe a winner, run it once against the untouched final 2 years:
|
|
36
|
-
|
|
37
|
-
```bash
|
|
38
|
-
npx tsx src/quant-arena/holdout-certify.mts --strategy <path>/strategy.ts --out /tmp/quant-demo
|
|
39
|
-
```
|
|
40
|
-
|
|
41
|
-
It backtests in-sample + out-of-sample on one axis (so lookbacks are warm), scores only the out-of-sample days against the same three benchmarks, and appends the in-sample vs out-of-sample comparison to the notebook. A second run for the same strategy hash refuses without `--force` — an out-of-sample set answers once; re-rolling it until it agrees turns it into another in-sample set.
|
|
42
|
-
|
|
43
|
-
## 3. Why the acceptance rule is strict
|
|
44
|
-
|
|
45
|
-
If you test enough random strategies against the same data, the best one looks brilliant by luck alone.
|
|
46
|
-
Under the assumption of zero skill, the expected best Sharpe among N independent tries grows roughly like sqrt(2 ln N) — try 50 strategies and luck alone buys the winner a substantial edge.
|
|
47
|
-
So the bar a candidate must clear is not fixed: it is `0.10 + 0.15 * sqrt(2 ln N)` of mean excess Sharpe, where N counts **every** candidate ever tried in the campaign, including ones killed for leaking (this is a simplified, auditable version of the Deflated Sharpe Ratio of Bailey & Lopez de Prado, Journal of Portfolio Management, 2014).
|
|
48
|
-
A candidate must ALSO beat the best benchmark in at least 6 of the 8 windows, because one lucky two-year stretch should never carry a decision.
|
|
49
|
-
Every try is a permanent notebook row, so N can never be quietly reset — the price of another shot at the data is a higher bar for everyone after it.
|
|
50
|
-
|
|
51
|
-
## 4. Reading the notebook
|
|
52
|
-
|
|
53
|
-
`notebook.jsonl` is append-only JSON lines. Three row types:
|
|
54
|
-
|
|
55
|
-
- `quant-arena.baselines.v1` — the campaign header: seed, cost assumptions, the 8 windows with dates, and each benchmark's per-window Sharpe.
|
|
56
|
-
- `quant-arena.candidate.v1` — one per try. The fields that matter:
|
|
57
|
-
- `nTried` — this try's position in the campaign; sets its acceptance bar.
|
|
58
|
-
- `leakAudit.truncation` / `leakAudit.llm` — both screening verdicts with evidence.
|
|
59
|
-
- `eval.perWindow` — Sharpe vs best-benchmark Sharpe for each window, with dates.
|
|
60
|
-
- `verdict` — `accepted`, `rejected-no-edge` (failed the acceptance rule), `rejected-leak`, `rejected-contract` (didn't satisfy the module contract), or `rejected-error`.
|
|
61
|
-
- `reasons` — the decision spelled out, numbers included.
|
|
62
|
-
- `quant-arena.certification.v1` — the one-shot out-of-sample result, in-sample stats side by side.
|
|
63
|
-
|
|
64
|
-
A real excerpt from the committed demo campaign (`fixtures/demo-campaign/`): try #4 cleared both look-ahead screens, then lost to the benchmarks in 7 of 8 windows — and after four tries the bar it would have needed had already risen to 0.35 (condensed for width):
|
|
65
|
-
|
|
66
|
-
```
|
|
67
|
-
{
|
|
68
|
-
"candidateId": "cand-004-quant-researcher",
|
|
69
|
-
"proposer": "quant-researcher",
|
|
70
|
-
"nTried": 4,
|
|
71
|
-
"leakAudit": {
|
|
72
|
-
"truncation": {
|
|
73
|
-
"clean": true
|
|
74
|
-
},
|
|
75
|
-
"llm": {
|
|
76
|
-
"verdict": "clean"
|
|
77
|
-
}
|
|
78
|
-
},
|
|
79
|
-
"eval": {
|
|
80
|
-
"wins": 1,
|
|
81
|
-
"requiredWins": 6,
|
|
82
|
-
"meanExcessSharpe": -0.085,
|
|
83
|
-
"threshold": 0.35,
|
|
84
|
-
"perWindow": [
|
|
85
|
-
{
|
|
86
|
-
"startDate": "2017-04-06",
|
|
87
|
-
"endDate": "2019-03-12",
|
|
88
|
-
"sharpe": 0.94,
|
|
89
|
-
"bestBaselineSharpe": 1.21,
|
|
90
|
-
"excess": -0.27
|
|
91
|
-
},
|
|
92
|
-
{
|
|
93
|
-
"startDate": "2018-04-18",
|
|
94
|
-
"endDate": "2020-03-23",
|
|
95
|
-
"sharpe": 0.56,
|
|
96
|
-
"bestBaselineSharpe": 0.61,
|
|
97
|
-
"excess": -0.05
|
|
98
|
-
},
|
|
99
|
-
{
|
|
100
|
-
"startDate": "2018-08-06",
|
|
101
|
-
"endDate": "2020-07-09",
|
|
102
|
-
"sharpe": 0.2,
|
|
103
|
-
"bestBaselineSharpe": 0.29,
|
|
104
|
-
"excess": -0.08
|
|
105
|
-
},
|
|
106
|
-
"... 5 more windows"
|
|
107
|
-
]
|
|
108
|
-
},
|
|
109
|
-
"verdict": "rejected-no-edge",
|
|
110
|
-
"reasons": [
|
|
111
|
-
"consistency: beat the best baseline in only 1/8 windows (need 6)",
|
|
112
|
-
"multiplicity: mean excess Sharpe -0.085 < required 0.350 (bar after 4 tried candidates)"
|
|
113
|
-
]
|
|
114
|
-
}
|
|
115
|
-
```
|
|
116
|
-
|
|
117
|
-
## 5. Plugging in your own backtester and data
|
|
118
|
-
|
|
119
|
-
Scoring goes through one narrow seam: a worker process that takes `{open prices, close prices, target-weight rows, costs, windows}` as JSON lines on stdin and answers `{per-window total return / max drawdown / Sharpe / trade count, full equity curve}` on stdout — see the protocol comment at the top of `python/vbt-worker.py` and the client in `vbt-client.ts`.
|
|
120
|
-
The shipped worker is vectorbt (`Portfolio.from_orders`, target-percent sizing, shared cash, sells before buys), version-locked by `python/pyproject.toml` + `python/uv.lock`; to swap in your own engine, speak the same protocol and keep the fill model (decide at close, fill at next open, bps fees on traded dollars) or re-derive the parity fixtures in `vbt-parity.test.mts` for your model.
|
|
121
|
-
Point the data loader at your own Stooq-format CSVs (one file per ticker, `IDX.csv` as the benchmark asset, an `insample/` and a `holdout/` directory).
|
|
122
|
-
Keep the physical in-sample/out-of-sample split and the once-only certification rule — they are the point, not an implementation detail.
|
|
123
|
-
A third engine is planned but not built: event-driven certification of a winner's order stream through Nautilus Trader (`nautilus-certify.ts` is the named stub).
|
|
124
|
-
|
|
125
|
-
## Files
|
|
126
|
-
|
|
127
|
-
| file | what it is |
|
|
128
|
-
| --- | --- |
|
|
129
|
-
| `types.ts` | the strategy contract (v2 `onBar` + the order types), including the no-look-ahead rule |
|
|
130
|
-
| `driver.ts` | the incremental harness: feeds `onBar` day by day with physically truncated history; wraps old batch strategies unchanged |
|
|
131
|
-
| `oms.ts` | the one shared rebalancer: target weights -> orders (strategies never place orders) |
|
|
132
|
-
| `backtest.ts` | the TypeScript reference engine: next-open fills, bps costs, no shorting — zero dependencies; contract prefilter |
|
|
133
|
-
| `vbt-client.ts` + `python/vbt-worker.py` | the official scorer: persistent vectorbt worker, pinned env (`python/uv.lock`), crash-safe request handling |
|
|
134
|
-
| `vbt-parity.test.mts` | the two engines held to agreement on golden fixtures (prints both curves on any disagreement) |
|
|
135
|
-
| `nautilus-certify.ts` | named stub for the planned event-driven certification engine (not implemented) |
|
|
136
|
-
| `windows.ts` | seeded block-bootstrap evaluation windows |
|
|
137
|
-
| `multiplicity.ts` | the rising acceptance bar (documented formula + citation) |
|
|
138
|
-
| `leak-audit.ts` | the mechanical truncation-invariance check |
|
|
139
|
-
| `quant-loop.mts` | the campaign: author -> screen -> backtest -> verdict -> notebook |
|
|
140
|
-
| `holdout-certify.mts` | the once-only out-of-sample certification |
|
|
141
|
-
| `strategies/` | the three pinned benchmarks |
|
|
142
|
-
| `fixtures/data/` | committed daily bars + provenance; `holdout/` is the locked final 2 years |
|
|
143
|
-
| `fixtures/demo-campaign/` | a real captured campaign against the v1 batch contract: notebook, authored strategies, cost receipts |
|
|
144
|
-
| `fixtures/demo-campaign-v2/` | a real captured campaign against the v2 `onBar` contract (1 candidate, honestly rejected: 0/8 windows) |
|
|
@@ -1,135 +0,0 @@
|
|
|
1
|
-
import { describe, expect, it } from 'vitest'
|
|
2
|
-
import { fillSchedule, runBacktest, statsForRange } from './backtest.ts'
|
|
3
|
-
import type { Bar } from './types.ts'
|
|
4
|
-
|
|
5
|
-
/** Flat-price bar helper. */
|
|
6
|
-
const bar = (date: string, open: number, close = open): Bar => ({
|
|
7
|
-
date,
|
|
8
|
-
open,
|
|
9
|
-
high: Math.max(open, close),
|
|
10
|
-
low: Math.min(open, close),
|
|
11
|
-
close,
|
|
12
|
-
volume: 1000,
|
|
13
|
-
})
|
|
14
|
-
|
|
15
|
-
const dates = (n: number): string[] =>
|
|
16
|
-
Array.from({ length: n }, (_, i) => `2020-01-${String(i + 1).padStart(2, '0')}`)
|
|
17
|
-
|
|
18
|
-
const flatSeries = (n: number, price: number): Bar[] => dates(n).map((d) => bar(d, price))
|
|
19
|
-
|
|
20
|
-
const NO_COST = { costBps: 0, slippageBps: 0 }
|
|
21
|
-
|
|
22
|
-
describe('runBacktest hand-computed toy cases', () => {
|
|
23
|
-
it('all-cash (no signals) stays at equity 1 with zero trades', () => {
|
|
24
|
-
const result = runBacktest([flatSeries(5, 100)], [], NO_COST)
|
|
25
|
-
expect(result.equity).toEqual([1, 1, 1, 1, 1])
|
|
26
|
-
expect(result.stats.tradeCount).toBe(0)
|
|
27
|
-
expect(result.stats.totalReturn).toBe(0)
|
|
28
|
-
})
|
|
29
|
-
|
|
30
|
-
it('full weight on a flat ticker loses exactly the round-trip-free fee', () => {
|
|
31
|
-
// 10bps cost + 5bps slippage on 1.0 traded dollars = 15bps once.
|
|
32
|
-
const result = runBacktest([flatSeries(5, 100)], [{ t: 0, weights: [1] }], { costBps: 10, slippageBps: 5 })
|
|
33
|
-
expect(result.stats.tradeCount).toBe(1)
|
|
34
|
-
expect(result.stats.totalReturn).toBeCloseTo(-0.0015, 12)
|
|
35
|
-
expect(result.equity[4]).toBeCloseTo(1 - 0.0015, 12)
|
|
36
|
-
})
|
|
37
|
-
|
|
38
|
-
it('captures a known move: buy at open, ride to close', () => {
|
|
39
|
-
// Day1: open 100 -> close 110 with full weight decided at day0 close.
|
|
40
|
-
const series = [bar('2020-01-01', 100), bar('2020-01-02', 100, 110), bar('2020-01-03', 110)]
|
|
41
|
-
const result = runBacktest([series], [{ t: 0, weights: [1] }], NO_COST)
|
|
42
|
-
expect(result.equity[1]).toBeCloseTo(1.1, 12)
|
|
43
|
-
expect(result.stats.totalReturn).toBeCloseTo(0.1, 12)
|
|
44
|
-
})
|
|
45
|
-
|
|
46
|
-
it('fills at the NEXT open, not the signal-day close (no free look-ahead)', () => {
|
|
47
|
-
// Price gaps 100 -> 200 overnight after the signal. A same-close fill would
|
|
48
|
-
// double money on the gap; a next-open fill must NOT.
|
|
49
|
-
const series = [bar('2020-01-01', 100, 100), bar('2020-01-02', 200, 200), bar('2020-01-03', 300, 300)]
|
|
50
|
-
const result = runBacktest([series], [{ t: 0, weights: [1] }], NO_COST)
|
|
51
|
-
// Bought at 200 (day2 open); close day2 = 200 -> equity 1; day3 300/200 = 1.5.
|
|
52
|
-
expect(result.equity[1]).toBeCloseTo(1, 12)
|
|
53
|
-
expect(result.equity[2]).toBeCloseTo(1.5, 12)
|
|
54
|
-
})
|
|
55
|
-
|
|
56
|
-
it('max drawdown on a crafted path: peak 1.2 -> trough 0.9 = 25%', () => {
|
|
57
|
-
const series = [
|
|
58
|
-
bar('2020-01-01', 100),
|
|
59
|
-
bar('2020-01-02', 100, 120),
|
|
60
|
-
bar('2020-01-03', 120, 90),
|
|
61
|
-
bar('2020-01-04', 90, 100),
|
|
62
|
-
]
|
|
63
|
-
const result = runBacktest([series], [{ t: 0, weights: [1] }], NO_COST)
|
|
64
|
-
expect(result.equity).toEqual([1, 1.2, 0.9, 1.0].map((v) => expect.closeTo(v, 12) as unknown as number))
|
|
65
|
-
expect(result.stats.maxDrawdown).toBeCloseTo((1.2 - 0.9) / 1.2, 12)
|
|
66
|
-
})
|
|
67
|
-
|
|
68
|
-
it('splits weights across two tickers and holds cash for the rest', () => {
|
|
69
|
-
const a = [bar('2020-01-01', 100), bar('2020-01-02', 100, 110)]
|
|
70
|
-
const b = [bar('2020-01-01', 50), bar('2020-01-02', 50, 45)]
|
|
71
|
-
const result = runBacktest([a, b], [{ t: 0, weights: [0.5, 0.25] }], NO_COST)
|
|
72
|
-
// 0.5 * +10% + 0.25 * -10% + 0.25 cash = 1 + 0.05 - 0.025 = 1.025
|
|
73
|
-
expect(result.equity[1]).toBeCloseTo(1.025, 12)
|
|
74
|
-
expect(result.stats.tradeCount).toBe(2)
|
|
75
|
-
})
|
|
76
|
-
|
|
77
|
-
it('is deterministic: identical inputs give identical equity paths', () => {
|
|
78
|
-
const series = [flatSeries(30, 100).map((b, i) => bar(b.date, 100 + i, 101 + i))]
|
|
79
|
-
const signals = [
|
|
80
|
-
{ t: 0, weights: [0.7] },
|
|
81
|
-
{ t: 10, weights: [0.2] },
|
|
82
|
-
{ t: 20, weights: [1] },
|
|
83
|
-
]
|
|
84
|
-
const r1 = runBacktest(series as Bar[][], signals, { costBps: 10, slippageBps: 5 })
|
|
85
|
-
const r2 = runBacktest(series as Bar[][], signals, { costBps: 10, slippageBps: 5 })
|
|
86
|
-
expect(r1.equity).toEqual(r2.equity)
|
|
87
|
-
expect(r1.stats).toEqual(r2.stats)
|
|
88
|
-
})
|
|
89
|
-
|
|
90
|
-
it('rejects shorting, leverage, and misaligned universes (fail-closed)', () => {
|
|
91
|
-
const series = [flatSeries(3, 100)]
|
|
92
|
-
expect(() => runBacktest(series, [{ t: 0, weights: [-0.1] }], NO_COST)).toThrow(/no shorting/)
|
|
93
|
-
expect(() => runBacktest(series, [{ t: 0, weights: [1.2] }], NO_COST)).toThrow(/no leverage/)
|
|
94
|
-
expect(() => runBacktest([flatSeries(3, 100), flatSeries(4, 50)], [], NO_COST)).toThrow(/unaligned/)
|
|
95
|
-
expect(() => runBacktest(series, [{ t: 5, weights: [1] }], NO_COST)).toThrow(/outside/)
|
|
96
|
-
})
|
|
97
|
-
|
|
98
|
-
it('rebalances ONLY on emitted signals — positions drift in between', () => {
|
|
99
|
-
// One signal at t=0, then a price runup: the winning position is NOT
|
|
100
|
-
// trimmed back to its target weight on later days (no hidden churn).
|
|
101
|
-
const a = [bar('2020-01-01', 100), bar('2020-01-02', 100), bar('2020-01-03', 100, 200), bar('2020-01-04', 200)]
|
|
102
|
-
const result = runBacktest([a], [{ t: 0, weights: [0.5] }], NO_COST)
|
|
103
|
-
expect(result.stats.tradeCount).toBe(1)
|
|
104
|
-
// 0.5 in the ticker doubled -> equity 1.5, still only one fill ever.
|
|
105
|
-
expect(result.equity[3]).toBeCloseTo(1.5, 12)
|
|
106
|
-
const schedule = fillSchedule([{ t: 1, weights: [0.5] }], 1, 4)
|
|
107
|
-
expect(schedule.get(1)).toEqual([0.5])
|
|
108
|
-
expect(schedule.has(0)).toBe(false)
|
|
109
|
-
})
|
|
110
|
-
})
|
|
111
|
-
|
|
112
|
-
describe('statsForRange window slicing', () => {
|
|
113
|
-
it('scores only the window and carries positions in', () => {
|
|
114
|
-
// Flat first half, +10% single day in the second half.
|
|
115
|
-
const series = [
|
|
116
|
-
bar('2020-01-01', 100),
|
|
117
|
-
bar('2020-01-02', 100),
|
|
118
|
-
bar('2020-01-03', 100),
|
|
119
|
-
bar('2020-01-04', 100, 110),
|
|
120
|
-
bar('2020-01-05', 110),
|
|
121
|
-
]
|
|
122
|
-
const result = runBacktest([series], [{ t: 0, weights: [1] }], NO_COST)
|
|
123
|
-
const firstHalf = statsForRange(result, 0, 3)
|
|
124
|
-
const secondHalf = statsForRange(result, 2, 5)
|
|
125
|
-
expect(firstHalf.totalReturn).toBeCloseTo(0, 12)
|
|
126
|
-
expect(secondHalf.totalReturn).toBeCloseTo(0.1, 12)
|
|
127
|
-
expect(secondHalf.days).toBe(3)
|
|
128
|
-
})
|
|
129
|
-
|
|
130
|
-
it('rejects degenerate ranges', () => {
|
|
131
|
-
const result = runBacktest([flatSeries(5, 100)], [], NO_COST)
|
|
132
|
-
expect(() => statsForRange(result, 3, 3)).toThrow(/bad range/)
|
|
133
|
-
expect(() => statsForRange(result, 0, 99)).toThrow(/bad range/)
|
|
134
|
-
})
|
|
135
|
-
})
|