@tangle-network/agent-bench 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,840 +0,0 @@
1
- /**
2
- * QUANT-ARENA campaign loop — the improvement loop embodied for trading
3
- * strategies. One command runs: strategy authors (Runtime profile-pinned)
4
- * propose candidate strategies (v2 `onBar` contract, driven incrementally by
5
- * driver.ts) -> every candidate passes a two-stage leak audit -> survivors
6
- * are scored on K bootstrap in-sample windows against the pinned baselines
7
- * -> a multiplicity-adjusted acceptance rule decides -> every try becomes a
8
- * permanent lab-notebook row (notebook.jsonl).
9
- *
10
- * Scoring engines: the OFFICIAL per-window scores come from the vectorbt
11
- * worker (vbt-client.ts -> python/vbt-worker.py). The TS engine
12
- * (backtest.ts) runs first as contract prefilter + leak-audit substrate
13
- * only — it throws on shorting/leverage violations and supplies turnover
14
- * (which the worker protocol does not carry), but its Sharpe/return numbers
15
- * are never the acceptance currency.
16
- *
17
- * tsx src/quant-arena/quant-loop.mts --out <dir> [--candidates 2] [--seed 20260722]
18
- * [--author-model glm-5.2] [--audit-model glm-5.2] [--skip-llm-audit]
19
- *
20
- * Kernel reuse (import, not copy — see src/swe-arena/):
21
- * - cost accounting: the lib's durable CostLedger (createRunCostLedger) +
22
- * crash-orphan reconcile (ledger-orphans.mts) — author/audit shots are
23
- * metered paid calls with receipts in <out>/cost-ledger.jsonl.
24
- * - evidence cells: one cached-result.json per (strategy x window) in the
25
- * swe-arena cell shape, readable by cell-evidence.mts's loadCampaignCells.
26
- * - rollout manifest: a pure reader/join over cells + ledger receipts,
27
- * mirroring swe-arena/manifest.mts (loadLedgerReceipts imported from it).
28
- * - proposer identity: AgentProfile-pinned authors via proposer-fanout.mts's
29
- * loadAuthorProfile + the same ambient-auth-stripped shot env.
30
- *
31
- * The HOLDOUT (final 2 years) is never read here — see holdout-certify.mts.
32
- */
33
-
34
- import { createHash } from 'node:crypto'
35
- import { appendFile, mkdir, readFile, writeFile } from 'node:fs/promises'
36
- import { existsSync } from 'node:fs'
37
- import { join } from 'node:path'
38
- import { fileURLToPath, pathToFileURL } from 'node:url'
39
- import process from 'node:process'
40
- import { createRunCostLedger, fsCampaignStorage } from '@tangle-network/agent-eval/campaign'
41
- import { agentProfileSchema, type AgentProfile } from '@tangle-network/agent-interface'
42
- import { collectAgentTurn, createExecutor, streamAgentTurn } from '@tangle-network/agent-runtime/kernel'
43
- import { loadCampaignCells } from '../swe-arena/cell-evidence.mts'
44
- import { reconcileCrashOrphansOnDisk } from '../swe-arena/ledger-orphans.mts'
45
- import { loadLedgerReceipts } from '../swe-arena/manifest.mts'
46
- import { resolveAuthorProfile, type ProposerSpec } from '../swe-arena/proposer-fanout.mts'
47
- import { runBacktest, statsForRange, type BacktestConfig, type RangeStats } from './backtest.ts'
48
- import { loadInSample, type AlignedBars } from './data.ts'
49
- import { loadStrategyFile } from './driver.ts'
50
- import { truncationInvariance, type TruncationReport } from './leak-audit.ts'
51
- import { decideAcceptance, requiredExcessSharpe, type AcceptanceDecision } from './multiplicity.ts'
52
- import { scoreSignals, VbtWorker, type VbtWindowStats } from './vbt-client.ts'
53
- import { bootstrapWindows, type EvalWindow } from './windows.ts'
54
- import type { GenerateSignals, Signal } from './types.ts'
55
- import * as buyHoldIndex from './strategies/buy-hold-index/strategy.ts'
56
- import * as equalWeight from './strategies/equal-weight/strategy.ts'
57
- import * as smaCrossover from './strategies/sma-crossover/strategy.ts'
58
-
59
- export const QUANT_PROFILES_DIR = fileURLToPath(new URL('./profiles', import.meta.url))
60
- const DEFAULT_AUTHOR_PROFILE = fileURLToPath(
61
- new URL('../swe-arena/profiles/default-author.profile.json', import.meta.url),
62
- )
63
-
64
- // ---------------------------------------------------------------------------
65
- // Config.
66
- // ---------------------------------------------------------------------------
67
-
68
- export interface QuantLoopConfig {
69
- outDir: string
70
- candidatesPerProposer: number
71
- seed: number
72
- windows: number
73
- windowDays: number
74
- warmupDays: number
75
- costBps: number
76
- slippageBps: number
77
- authorModel: string
78
- auditModel: string
79
- skipLlmAudit: boolean
80
- authorTimeoutMs: number
81
- auditTimeoutMs: number
82
- proposers: ProposerSpec[]
83
- }
84
-
85
- export const PINNED_BASELINES: Record<string, GenerateSignals> = {
86
- 'buy-hold-index': buyHoldIndex.generateSignals,
87
- 'equal-weight': equalWeight.generateSignals,
88
- 'sma-crossover': smaCrossover.generateSignals,
89
- }
90
-
91
- /** The two demo author seats: the plain author and the quant lens. */
92
- export function defaultQuantProposers(): ProposerSpec[] {
93
- return [
94
- {
95
- name: 'default-author',
96
- profile: DEFAULT_AUTHOR_PROFILE,
97
- harness: 'pi',
98
- model: 'glm-5.2',
99
- },
100
- {
101
- name: 'quant-researcher',
102
- profile: join(QUANT_PROFILES_DIR, 'quant-researcher.profile.json'),
103
- harness: 'pi',
104
- model: 'glm-5.2',
105
- lens:
106
- 'Favor ONE economically-motivated effect (trend, mean reversion, vol targeting) with few parameters. ' +
107
- 'State the regime in which it should work and keep turnover low enough that 15bps a side cannot eat the edge.',
108
- },
109
- ]
110
- }
111
-
112
- export function defaultConfig(outDir: string): QuantLoopConfig {
113
- return {
114
- outDir,
115
- candidatesPerProposer: 2,
116
- seed: 20260722,
117
- windows: 8,
118
- windowDays: 504,
119
- warmupDays: 120,
120
- costBps: 10,
121
- slippageBps: 5,
122
- authorModel: 'glm-5.2',
123
- auditModel: 'glm-5.2',
124
- skipLlmAudit: false,
125
- authorTimeoutMs: 480_000,
126
- auditTimeoutMs: 240_000,
127
- proposers: defaultQuantProposers(),
128
- }
129
- }
130
-
131
- // ---------------------------------------------------------------------------
132
- // Notebook rows — the permanent lab notebook. Append-only JSONL.
133
- // ---------------------------------------------------------------------------
134
-
135
- export const NOTEBOOK_CANDIDATE_SCHEMA = 'quant-arena.candidate.v1'
136
- export const NOTEBOOK_BASELINES_SCHEMA = 'quant-arena.baselines.v1'
137
-
138
- export type CandidateVerdict =
139
- | 'accepted'
140
- | 'rejected-no-edge'
141
- | 'rejected-leak'
142
- | 'rejected-contract'
143
- | 'rejected-error'
144
-
145
- export interface WindowScore {
146
- start: number
147
- end: number
148
- startDate: string
149
- endDate: string
150
- sharpe: number
151
- bestBaselineSharpe: number
152
- excess: number
153
- }
154
-
155
- export interface CandidateRow {
156
- schema: typeof NOTEBOOK_CANDIDATE_SCHEMA
157
- at: string
158
- candidateId: string
159
- proposer: string
160
- authorModel: string
161
- strategyPath: string | null
162
- sha256: string | null
163
- authoringCostUsd: number | null
164
- /** Total candidates tried this campaign INCLUDING this one — the
165
- * multiplicity denominator. Monotone; never resets within a notebook. */
166
- nTried: number
167
- leakAudit: {
168
- truncation: TruncationReport | null
169
- llm: { verdict: 'clean' | 'leak' | 'inconclusive'; evidence: string } | 'skipped' | null
170
- }
171
- eval: {
172
- perWindow: WindowScore[]
173
- meanExcessSharpe: number
174
- wins: number
175
- requiredWins: number
176
- threshold: number
177
- } | null
178
- inSampleFull: RangeStats | null
179
- verdict: CandidateVerdict
180
- reasons: string[]
181
- }
182
-
183
- export async function loadNotebookRows(notebookPath: string): Promise<Array<Record<string, unknown>>> {
184
- if (!existsSync(notebookPath)) return []
185
- const raw = await readFile(notebookPath, 'utf8')
186
- return raw
187
- .split('\n')
188
- .filter((l) => l.trim().length > 0)
189
- .map((l) => JSON.parse(l) as Record<string, unknown>)
190
- }
191
-
192
- const log = (msg: string): void => console.log(`[${new Date().toISOString().slice(11, 19)}] ${msg}`)
193
-
194
- // ---------------------------------------------------------------------------
195
- // Exact-profile shots (author + auditor) — metered paid calls through Runtime + the run ledger.
196
- // ---------------------------------------------------------------------------
197
-
198
- interface ProfileShotOutcome {
199
- text: string
200
- model: string
201
- inputTokens: number
202
- outputTokens: number
203
- cachedTokens: number
204
- costUsd: number | null
205
- }
206
-
207
- type Ledger = ReturnType<typeof createRunCostLedger>
208
-
209
- function withModel(profile: AgentProfile, model: string, name = profile.name): AgentProfile {
210
- return agentProfileSchema.parse({
211
- ...profile,
212
- ...(name ? { name } : {}),
213
- model: { ...profile.model, default: model },
214
- })
215
- }
216
-
217
- function quantAuditProfile(model: string): AgentProfile {
218
- return agentProfileSchema.parse({
219
- name: 'quant-leak-auditor',
220
- harness: 'pi',
221
- model: { provider: 'tangle-router', default: model },
222
- prompt: {
223
- systemPrompt:
224
- 'Audit the supplied trading strategy for look-ahead bias and nondeterminism. Follow the requested JSON response contract exactly.',
225
- },
226
- })
227
- }
228
-
229
- async function profileShot(opts: {
230
- prompt: string
231
- profile: AgentProfile
232
- timeoutMs: number
233
- cwd: string
234
- }): Promise<ProfileShotOutcome> {
235
- const bridgeUrl = process.env.CLI_BRIDGE_URL ?? process.env.BRIDGE_URL
236
- const bridgeBearer = process.env.CLI_BRIDGE_BEARER ?? process.env.BRIDGE_BEARER
237
- if (!bridgeUrl || !bridgeBearer) {
238
- throw new Error(
239
- 'quant profile shots require CLI_BRIDGE_URL/BRIDGE_URL and CLI_BRIDGE_BEARER/BRIDGE_BEARER',
240
- )
241
- }
242
- const factory = createExecutor({
243
- backend: 'bridge',
244
- bridgeUrl,
245
- bridgeBearer,
246
- cwd: opts.cwd,
247
- timeoutMs: opts.timeoutMs,
248
- })
249
- const turn = await collectAgentTurn(
250
- streamAgentTurn(
251
- { kind: 'executor', factory, profile: opts.profile },
252
- { prompt: opts.prompt },
253
- { timeoutMs: opts.timeoutMs },
254
- ),
255
- )
256
- if (turn.status !== 'completed') {
257
- throw new Error(turn.error?.message ?? `quant profile shot ended with ${turn.status}`)
258
- }
259
- const cachedTokens = Number(turn.usage.promptCache?.readTokens ?? 0)
260
- return {
261
- text: turn.finalText,
262
- model: turn.usage.model ?? opts.profile.model?.default ?? 'unknown',
263
- inputTokens: turn.usage.input,
264
- outputTokens: turn.usage.output,
265
- cachedTokens: Number.isFinite(cachedTokens) ? cachedTokens : 0,
266
- costUsd: turn.usage.costUsd ?? null,
267
- }
268
- }
269
-
270
- async function meteredProfileShot(
271
- ledger: Ledger,
272
- meta: { phase: string; actor: string; tags: Record<string, string> },
273
- opts: Parameters<typeof profileShot>[0],
274
- ): Promise<{ outcome: ProfileShotOutcome; costUsd: number | null }> {
275
- const model = opts.profile.model?.default
276
- if (!model) throw new Error('meteredProfileShot: profile.model.default is required')
277
- const paid = await ledger.runPaidCall<ProfileShotOutcome>({
278
- channel: 'driver',
279
- phase: meta.phase,
280
- actor: meta.actor,
281
- model,
282
- tags: meta.tags,
283
- execute: () => profileShot(opts),
284
- receipt: (v) => ({
285
- model: v.model,
286
- inputTokens: v.inputTokens,
287
- outputTokens: v.outputTokens,
288
- cachedTokens: v.cachedTokens,
289
- ...(v.costUsd !== null ? { actualCostUsd: v.costUsd } : {}),
290
- }),
291
- })
292
- if (!paid.succeeded) throw paid.error
293
- return { outcome: paid.value, costUsd: paid.value.costUsd }
294
- }
295
-
296
- // ---------------------------------------------------------------------------
297
- // Authoring: prompt, extraction, hermeticity guard.
298
- // ---------------------------------------------------------------------------
299
-
300
- const CONTRACT_TEXT = `THE STRATEGY CONTRACT (v2 — incremental)
301
- - Write ONE self-contained TypeScript module. NO import/require/fs/network/process — declare any types you need locally.
302
- - Export exactly: export function onBar(ctx: StrategyContext): TargetPosition[] | null
303
- where StrategyContext = { symbols: string[]; t: number; history: Bar[][]; weights: number[]; equity: number },
304
- Bar = { date: string; open: number; high: number; low: number; close: number; volume: number },
305
- and TargetPosition = { symbol: string; weight: number }.
306
- - The lab calls onBar once per trading day, in order. ctx.history[k] holds the daily bars of ctx.symbols[k] from
307
- day 0 THROUGH TODAY ONLY (ctx.history[k].length === ctx.t + 1) — bars after today do not exist in the array.
308
- ctx.history[0] / ctx.symbols[0] is the benchmark index. ctx.weights and ctx.equity are your current drifted
309
- portfolio state (equity starts at 1).
310
- - Return TargetPosition[] to rebalance: weight = target fraction of equity per symbol; any symbol you omit is
311
- sold to 0. Return null to hold (positions drift with prices). A rebalance fills at the NEXT day's open.
312
- - You never construct orders — the lab's shared rebalancer turns your target weights into orders.
313
- - No shorting, no leverage: every weight >= 0 and the weights sum to <= 1 (rest is cash at 0%). Violations kill
314
- the candidate — fail-closed, not clamped.
315
- - DETERMINISM / NO LOOK-AHEAD: onBar must be a pure function of ctx (no RNG, no clock, no hidden state). The lab
316
- re-runs your code on truncated data; if any decision up to the cutoff changes, the candidate is killed. No
317
- hardcoded calendar dates that memorize this dataset.
318
- - Every fill pays 15bps one-way (cost + slippage) on traded dollars — churn is expensive.`
319
-
320
- export function buildAuthorPrompt(args: {
321
- universe: AlignedBars
322
- windows: EvalWindow[]
323
- baselineTable: string
324
- threshold: number
325
- nTried: number
326
- lens?: string
327
- }): string {
328
- const { universe, windows, baselineTable, threshold, nTried } = args
329
- return [
330
- 'You are proposing ONE candidate trading strategy for a research lab with a strict acceptance rule.',
331
- '',
332
- CONTRACT_TEXT,
333
- '',
334
- `UNIVERSE: ${universe.tickers.length} tickers (${universe.tickers.join(', ')}); bars[0] = ${universe.tickers[0]} (the index).`,
335
- `IN-SAMPLE: ${universe.dates.length} daily bars, ${universe.dates[0]} .. ${universe.dates[universe.dates.length - 1]}.`,
336
- '',
337
- 'ACCEPTANCE RULE (what you must beat):',
338
- `- Scored on ${windows.length} overlapping ${windows[0]!.end - windows[0]!.start}-day in-sample windows.`,
339
- '- You must beat the BEST pinned baseline Sharpe in at least 6 of 8 windows, AND',
340
- `- your mean excess Sharpe must clear ${threshold.toFixed(3)} (the bar rises with every candidate tried; you are try #${nTried}).`,
341
- '',
342
- 'PINNED BASELINES (annualized Sharpe per window; "best" is the per-window max):',
343
- baselineTable,
344
- '',
345
- ...(args.lens ? ['YOUR AUTHORING LENS:', args.lens, ''] : []),
346
- 'Reply with EXACTLY ONE fenced ```ts code block containing the module and nothing else after it.',
347
- ].join('\n')
348
- }
349
-
350
- /** Pull the strategy module out of the author's reply and enforce the
351
- * self-contained rule. Fail-closed: anything ambiguous is a rejection. */
352
- export function extractStrategySource(text: string): { ok: true; code: string } | { ok: false; reason: string } {
353
- const blocks = [...text.matchAll(/```(?:ts|typescript)?\s*\n([\s\S]*?)```/g)].map((m) => m[1]!)
354
- const withExport = blocks.filter((b) => /export\s+function\s+onBar\s*\(/.test(b))
355
- if (withExport.length === 0) {
356
- return { ok: false, reason: 'no fenced code block exporting `onBar` in the reply (v2 contract)' }
357
- }
358
- const code = withExport[withExport.length - 1]!
359
- const stripped = code.replace(/\/\*[\s\S]*?\*\//g, '').replace(/\/\/.*$/gm, '')
360
- const banned = /\b(import|require|fetch|process|globalThis|Deno|XMLHttpRequest|eval)\b/.exec(stripped)
361
- if (banned) {
362
- return { ok: false, reason: `not self-contained: uses banned identifier '${banned[1]}'` }
363
- }
364
- return { ok: true, code }
365
- }
366
-
367
- // ---------------------------------------------------------------------------
368
- // Adversarial LLM leak audit (the deterministic truncation check lives in
369
- // leak-audit.ts; both must pass).
370
- // ---------------------------------------------------------------------------
371
-
372
- const AUDIT_PROMPT_HEADER = `You are an adversarial reviewer with ONE job: find look-ahead bias or nondeterminism
373
- in the trading strategy below. The contract: onBar(ctx) is called once per day; ctx.history[k] holds ONLY bars
374
- 0..ctx.t (the harness slices the arrays), decisions must be pure functions of ctx, and fills happen at the next
375
- day's open. Hunt for:
376
- - hardcoded calendar dates or magic day indexes that smell like memorizing this dataset,
377
- - nondeterminism: Math.random, Date.now, or state carried between onBar calls that a re-run would not rebuild,
378
- - decisions that would change when the future is truncated,
379
- - any attempt to reach data beyond ctx.history (indexing past the array end, reconstructing future prices).
380
- Deciding at close of ctx.t and being filled at t+1's open is LEGAL — do not flag it. Whole-history statistics over
381
- ctx.history are LEGAL (the array ends at today) — do not flag them.
382
- Reply with JSON ONLY: {"verdict":"clean"} or {"verdict":"leak","evidence":"<quote the offending code and why>"}.
383
-
384
- STRATEGY SOURCE:
385
- `
386
-
387
- export function parseAuditVerdict(text: string): { verdict: 'clean' | 'leak'; evidence: string } | null {
388
- const matches = [...text.matchAll(/\{[\s\S]*?"verdict"[\s\S]*?\}/g)]
389
- for (const m of matches.reverse()) {
390
- try {
391
- const parsed = JSON.parse(m[0]) as { verdict?: string; evidence?: string }
392
- if (parsed.verdict === 'clean') return { verdict: 'clean', evidence: '' }
393
- if (parsed.verdict === 'leak') return { verdict: 'leak', evidence: parsed.evidence ?? '(no evidence quoted)' }
394
- } catch {
395
- continue
396
- }
397
- }
398
- return null
399
- }
400
-
401
- async function llmLeakAudit(
402
- ledger: Ledger,
403
- config: QuantLoopConfig,
404
- candidateId: string,
405
- code: string,
406
- ): Promise<{ verdict: 'clean' | 'leak' | 'inconclusive'; evidence: string }> {
407
- for (let attempt = 0; attempt < 2; attempt++) {
408
- const { outcome } = await meteredProfileShot(
409
- ledger,
410
- { phase: 'audit.leak', actor: 'leak-auditor:runtime', tags: { candidateId, attempt: String(attempt) } },
411
- {
412
- prompt: AUDIT_PROMPT_HEADER + '```ts\n' + code + '\n```',
413
- profile: quantAuditProfile(config.auditModel),
414
- timeoutMs: config.auditTimeoutMs,
415
- cwd: config.outDir,
416
- },
417
- )
418
- const verdict = parseAuditVerdict(outcome.text)
419
- if (verdict !== null) return verdict
420
- }
421
- return { verdict: 'inconclusive', evidence: 'auditor reply unparseable twice — fail-closed' }
422
- }
423
-
424
- // ---------------------------------------------------------------------------
425
- // Evidence cells (kernel cell shape) + rollout manifest.
426
- // ---------------------------------------------------------------------------
427
-
428
- async function writeWindowCells(
429
- campaignRoot: string,
430
- label: string,
431
- perWindow: Array<{ window: EvalWindow; stats: RangeStats; bestBaselineSharpe: number | null }>,
432
- ): Promise<string> {
433
- const dir = join(campaignRoot, label)
434
- for (let i = 0; i < perWindow.length; i++) {
435
- const { window, stats, bestBaselineSharpe } = perWindow[i]!
436
- const cellDir = join(dir, `window-${i}-rep-0`)
437
- await mkdir(cellDir, { recursive: true })
438
- const cell = {
439
- scenarioId: `window-${window.start}-${window.end}`,
440
- rep: 0,
441
- artifact: {
442
- kind: 'quant-window',
443
- strategy: label,
444
- windowStart: window.start,
445
- windowEnd: window.end,
446
- sharpe: stats.sharpe,
447
- totalReturn: stats.totalReturn,
448
- maxDrawdown: stats.maxDrawdown,
449
- tradeCount: stats.tradeCount,
450
- turnover: stats.turnover,
451
- bestBaselineSharpe,
452
- excessSharpe: bestBaselineSharpe === null ? null : stats.sharpe - bestBaselineSharpe,
453
- },
454
- costUsd: 0,
455
- tokenUsage: { input: 0, output: 0 },
456
- cached: true,
457
- }
458
- await writeFile(join(cellDir, 'cached-result.json'), JSON.stringify(cell, null, 2))
459
- }
460
- return dir
461
- }
462
-
463
- export const QUANT_ROLLOUT_SCHEMA = 'quant-arena.rollout.v1'
464
-
465
- /** Pure reader/join over what the campaign already wrote — the same posture
466
- * as swe-arena/manifest.mts, over the same cell + ledger primitives. */
467
- export async function writeQuantRolloutManifest(outDir: string): Promise<string> {
468
- const campaignRoot = join(outDir, 'campaign')
469
- const receipts = await loadLedgerReceipts(outDir)
470
- const notebook = await loadNotebookRows(join(outDir, 'notebook.jsonl'))
471
- const entries: Array<Record<string, unknown>> = []
472
- const { readdir } = await import('node:fs/promises')
473
- for (const name of (await readdir(campaignRoot).catch(() => [])).sort()) {
474
- const cells = await loadCampaignCells(join(campaignRoot, name))
475
- if (cells.length === 0) continue
476
- entries.push({
477
- label: name,
478
- campaignDir: join(campaignRoot, name),
479
- cells: cells.length,
480
- scenarios: cells.map((c) => c.scenarioId).sort(),
481
- })
482
- }
483
- const manifest = {
484
- schema: QUANT_ROLLOUT_SCHEMA,
485
- outDir,
486
- at: new Date().toISOString(),
487
- notebookRows: notebook.length,
488
- strategies: entries,
489
- receipts: receipts.map((r) => ({
490
- callId: r.callId,
491
- phase: (r as Record<string, unknown>).phase ?? null,
492
- actor: (r as Record<string, unknown>).actor ?? null,
493
- model: (r as Record<string, unknown>).model ?? null,
494
- costUsd: (r as Record<string, unknown>).costUsd ?? null,
495
- })),
496
- }
497
- const path = join(outDir, 'rollout-manifest.json')
498
- await writeFile(path, JSON.stringify(manifest, null, 2))
499
- return path
500
- }
501
-
502
- // ---------------------------------------------------------------------------
503
- // The campaign.
504
- // ---------------------------------------------------------------------------
505
-
506
- /** RangeStats assembled from the official (vectorbt) numbers plus turnover,
507
- * which only the TS prefilter tracks — labeled at the one place it mixes. */
508
- function rangeStatsFromVbt(stats: VbtWindowStats, start: number, end: number, turnover: number): RangeStats {
509
- return {
510
- start,
511
- end,
512
- days: end - start,
513
- totalReturn: stats.totalReturn,
514
- maxDrawdown: stats.maxDD,
515
- sharpe: stats.sharpe,
516
- tradeCount: stats.trades,
517
- turnover,
518
- }
519
- }
520
-
521
- interface ScoredStrategy {
522
- signals: Signal[]
523
- /** Official per-window stats (vectorbt; turnover from the TS prefilter). */
524
- perWindow: RangeStats[]
525
- /** Official full-range stats. */
526
- full: RangeStats
527
- }
528
-
529
- /** Score one decision record: TS engine first as fail-closed contract
530
- * prefilter (throws on shorting/leverage/malformed signals), then the
531
- * vectorbt worker for the official numbers. */
532
- async function scoreStrategySignals(
533
- worker: VbtWorker,
534
- universe: AlignedBars,
535
- signals: Signal[],
536
- windows: EvalWindow[],
537
- btConfig: BacktestConfig,
538
- ): Promise<ScoredStrategy> {
539
- const prefilter = runBacktest(universe.bars, signals, btConfig)
540
- const ranges = windows.map((w) => [w.start, w.end] as [number, number])
541
- const vbt = await scoreSignals(worker, universe.bars, signals, btConfig, ranges)
542
- const perWindow = windows.map((w, i) =>
543
- rangeStatsFromVbt(vbt.windows[i]!, w.start, w.end, statsForRange(prefilter, w.start, w.end).turnover),
544
- )
545
- const full = rangeStatsFromVbt(vbt.full, 0, universe.dates.length, prefilter.stats.turnover)
546
- return { signals, perWindow, full }
547
- }
548
-
549
- interface BaselineEvidence {
550
- perWindowSharpe: Record<string, number[]>
551
- bestPerWindow: number[]
552
- fullSample: Record<string, RangeStats>
553
- perWindowStats: Record<string, RangeStats[]>
554
- }
555
-
556
- async function evaluateBaselines(
557
- worker: VbtWorker,
558
- universe: AlignedBars,
559
- windows: EvalWindow[],
560
- btConfig: BacktestConfig,
561
- ): Promise<BaselineEvidence> {
562
- const perWindowSharpe: Record<string, number[]> = {}
563
- const fullSample: Record<string, RangeStats> = {}
564
- const perWindowStats: Record<string, RangeStats[]> = {}
565
- for (const [name, strategy] of Object.entries(PINNED_BASELINES)) {
566
- const scored = await scoreStrategySignals(worker, universe, strategy(universe.bars), windows, btConfig)
567
- fullSample[name] = scored.full
568
- perWindowStats[name] = scored.perWindow
569
- perWindowSharpe[name] = scored.perWindow.map((s) => s.sharpe)
570
- }
571
- const bestPerWindow = windows.map((_, i) =>
572
- Math.max(...Object.values(perWindowSharpe).map((sharpes) => sharpes[i]!)),
573
- )
574
- return { perWindowSharpe, bestPerWindow, fullSample, perWindowStats }
575
- }
576
-
577
- function baselineTable(evidence: BaselineEvidence): string {
578
- const lines: string[] = []
579
- for (const [name, sharpes] of Object.entries(evidence.perWindowSharpe)) {
580
- lines.push(` ${name.padEnd(15)} ${sharpes.map((s) => s.toFixed(2).padStart(6)).join(' ')}`)
581
- }
582
- lines.push(` ${'BEST'.padEnd(15)} ${evidence.bestPerWindow.map((s) => s.toFixed(2).padStart(6)).join(' ')}`)
583
- return lines.join('\n')
584
- }
585
-
586
- export async function runQuantCampaign(config: QuantLoopConfig): Promise<CandidateRow[]> {
587
- if (!VbtWorker.isAvailable()) {
588
- throw new Error(
589
- 'quant-arena: the official scorer is the vectorbt worker, which needs `uv` on PATH ' +
590
- '(src/quant-arena/python/uv.lock pins the environment). The TS engine is a prefilter ' +
591
- 'only and cannot stand in — install uv, there is no fallback scorer.',
592
- )
593
- }
594
- const worker = new VbtWorker()
595
- try {
596
- return await runQuantCampaignWithWorker(worker, config)
597
- } finally {
598
- await worker.close()
599
- }
600
- }
601
-
602
- async function runQuantCampaignWithWorker(worker: VbtWorker, config: QuantLoopConfig): Promise<CandidateRow[]> {
603
- await mkdir(config.outDir, { recursive: true })
604
- const notebookPath = join(config.outDir, 'notebook.jsonl')
605
- const reconciled = reconcileCrashOrphansOnDisk(config.outDir)
606
- if (reconciled.length > 0) log(`reconciled ${reconciled.length} crash-orphaned ledger call(s)`)
607
- const ledger = createRunCostLedger({ storage: fsCampaignStorage(), runDir: config.outDir })
608
-
609
- const universe = await loadInSample()
610
- const T = universe.dates.length
611
- const windows = bootstrapWindows(T, {
612
- k: config.windows,
613
- windowDays: config.windowDays,
614
- seed: config.seed,
615
- warmupDays: config.warmupDays,
616
- })
617
- const btConfig: BacktestConfig = { costBps: config.costBps, slippageBps: config.slippageBps }
618
- log(`in-sample: ${T} days x ${universe.tickers.length} tickers; ${windows.length} windows of ${config.windowDays}d (seed ${config.seed})`)
619
- log(`scoring engine: vectorbt ${await worker.ping()} (persistent worker, numba warm)`)
620
-
621
- const baselines = await evaluateBaselines(worker, universe, windows, btConfig)
622
- await appendFile(
623
- notebookPath,
624
- JSON.stringify({
625
- schema: NOTEBOOK_BASELINES_SCHEMA,
626
- at: new Date().toISOString(),
627
- seed: config.seed,
628
- costBps: config.costBps,
629
- slippageBps: config.slippageBps,
630
- windows: windows.map((w) => ({
631
- start: w.start,
632
- end: w.end,
633
- startDate: universe.dates[w.start],
634
- endDate: universe.dates[w.end - 1],
635
- })),
636
- perWindowSharpe: baselines.perWindowSharpe,
637
- bestPerWindow: baselines.bestPerWindow,
638
- fullSample: baselines.fullSample,
639
- }) + '\n',
640
- )
641
- const campaignRoot = join(config.outDir, 'campaign')
642
- for (const name of Object.keys(PINNED_BASELINES)) {
643
- await writeWindowCells(
644
- campaignRoot,
645
- `baseline-${name}`,
646
- windows.map((w, i) => ({
647
- window: w,
648
- stats: baselines.perWindowStats[name]![i]!,
649
- bestBaselineSharpe: baselines.bestPerWindow[i]!,
650
- })),
651
- )
652
- }
653
-
654
- const priorRows = await loadNotebookRows(notebookPath)
655
- let nTried = priorRows.filter((r) => r.schema === NOTEBOOK_CANDIDATE_SCHEMA).length
656
- const rows: CandidateRow[] = []
657
-
658
- for (const proposer of config.proposers) {
659
- const sourceProfile = resolveAuthorProfile(proposer)
660
- if (!sourceProfile) throw new Error(`quant proposer ${proposer.name}: exact profile is required`)
661
- const profile = withModel(sourceProfile, config.authorModel, `quant-${proposer.name}`)
662
- for (let shot = 0; shot < config.candidatesPerProposer; shot++) {
663
- nTried += 1
664
- const candidateId = `cand-${String(nTried).padStart(3, '0')}-${proposer.name}`
665
- const threshold = requiredExcessSharpe(nTried)
666
- log(`--- ${candidateId}: authoring (try #${nTried}, bar ${threshold.toFixed(3)})`)
667
-
668
- const row: CandidateRow = {
669
- schema: NOTEBOOK_CANDIDATE_SCHEMA,
670
- at: new Date().toISOString(),
671
- candidateId,
672
- proposer: proposer.name,
673
- authorModel: profile.model?.default ?? config.authorModel,
674
- strategyPath: null,
675
- sha256: null,
676
- authoringCostUsd: null,
677
- nTried,
678
- leakAudit: { truncation: null, llm: null },
679
- eval: null,
680
- inSampleFull: null,
681
- verdict: 'rejected-error',
682
- reasons: [],
683
- }
684
-
685
- try {
686
- const prompt = buildAuthorPrompt({
687
- universe,
688
- windows,
689
- baselineTable: baselineTable(baselines),
690
- threshold,
691
- nTried,
692
- ...(proposer.lens ? { lens: proposer.lens } : {}),
693
- })
694
- const { outcome, costUsd } = await meteredProfileShot(
695
- ledger,
696
- { phase: 'search.proposal', actor: `proposer-shot:${proposer.name}`, tags: { candidateId } },
697
- {
698
- prompt,
699
- profile,
700
- timeoutMs: config.authorTimeoutMs,
701
- cwd: config.outDir,
702
- },
703
- )
704
- row.authoringCostUsd = costUsd
705
-
706
- const extracted = extractStrategySource(outcome.text)
707
- if (!extracted.ok) {
708
- row.verdict = 'rejected-contract'
709
- row.reasons = [extracted.reason]
710
- } else {
711
- const strategyDir = join(config.outDir, 'strategies', candidateId)
712
- await mkdir(strategyDir, { recursive: true })
713
- const strategyPath = join(strategyDir, 'strategy.ts')
714
- await writeFile(strategyPath, extracted.code)
715
- row.strategyPath = strategyPath
716
- row.sha256 = `sha256:${createHash('sha256').update(extracted.code).digest('hex')}`
717
-
718
- let strategy: GenerateSignals | null = null
719
- try {
720
- // v2 (`onBar`) modules are wrapped through the incremental
721
- // driver, which structurally truncates history per bar.
722
- const loaded = await loadStrategyFile(strategyPath, {
723
- symbols: universe.tickers,
724
- costBps: config.costBps,
725
- slippageBps: config.slippageBps,
726
- })
727
- strategy = loaded.generateSignals
728
- } catch (cause) {
729
- row.verdict = 'rejected-contract'
730
- row.reasons = [(cause as Error).message.slice(0, 300)]
731
- }
732
- if (strategy !== null) {
733
- // Leak audit stage 1: deterministic truncation invariance.
734
- const truncation = truncationInvariance(strategy, universe.bars, { warmupDays: config.warmupDays })
735
- row.leakAudit.truncation = truncation
736
- // Leak audit stage 2: adversarial source read.
737
- const llm = config.skipLlmAudit
738
- ? ('skipped' as const)
739
- : await llmLeakAudit(ledger, config, candidateId, extracted.code)
740
- row.leakAudit.llm = llm
741
- const llmBad = llm !== 'skipped' && llm.verdict !== 'clean'
742
- if (!truncation.clean || llmBad) {
743
- row.verdict = 'rejected-leak'
744
- row.reasons = [
745
- ...(truncation.clean
746
- ? []
747
- : [`truncation divergence at cutoff ${truncation.divergence!.cutoff}: ${truncation.divergence!.detail}`]),
748
- ...(llmBad ? [`adversarial audit: ${(llm as { verdict: string; evidence: string }).evidence || (llm as { verdict: string }).verdict}`] : []),
749
- ]
750
- } else {
751
- // Eval: TS prefilter (contract enforcement) + official
752
- // vectorbt scores, per window against the pinned bar.
753
- const scored = await scoreStrategySignals(worker, universe, strategy(universe.bars), windows, btConfig)
754
- row.inSampleFull = scored.full
755
- const perWindow: WindowScore[] = windows.map((w, i) => {
756
- const stats = scored.perWindow[i]!
757
- return {
758
- start: w.start,
759
- end: w.end,
760
- startDate: universe.dates[w.start]!,
761
- endDate: universe.dates[w.end - 1]!,
762
- sharpe: stats.sharpe,
763
- bestBaselineSharpe: baselines.bestPerWindow[i]!,
764
- excess: stats.sharpe - baselines.bestPerWindow[i]!,
765
- }
766
- })
767
- await writeWindowCells(
768
- campaignRoot,
769
- candidateId,
770
- windows.map((w, i) => ({
771
- window: w,
772
- stats: scored.perWindow[i]!,
773
- bestBaselineSharpe: baselines.bestPerWindow[i]!,
774
- })),
775
- )
776
- const decision: AcceptanceDecision = decideAcceptance({
777
- perWindowExcess: perWindow.map((w) => w.excess),
778
- nTried,
779
- })
780
- row.eval = {
781
- perWindow,
782
- meanExcessSharpe: decision.meanExcessSharpe,
783
- wins: decision.wins,
784
- requiredWins: decision.requiredWins,
785
- threshold: decision.threshold,
786
- }
787
- row.verdict = decision.accepted ? 'accepted' : 'rejected-no-edge'
788
- row.reasons = decision.reasons
789
- }
790
- }
791
- }
792
- } catch (cause) {
793
- row.verdict = 'rejected-error'
794
- row.reasons = [(cause as Error).message.slice(0, 500)]
795
- }
796
-
797
- await appendFile(notebookPath, JSON.stringify(row) + '\n')
798
- rows.push(row)
799
- log(`${candidateId}: ${row.verdict}${row.reasons.length > 0 ? ` — ${row.reasons[0]}` : ''}`)
800
- }
801
- }
802
-
803
- const manifestPath = await writeQuantRolloutManifest(config.outDir)
804
- const summary = ledger.summary()
805
- log(`campaign done: ${rows.length} candidates, ${rows.filter((r) => r.verdict === 'accepted').length} accepted`)
806
- log(`spend: $${summary.totalCostUsd.toFixed(4)} across ${summary.totalCalls} metered calls (${summary.inputTokens} in / ${summary.outputTokens} out tokens)`)
807
- log(`notebook -> ${notebookPath}`)
808
- log(`rollout manifest -> ${manifestPath}`)
809
- return rows
810
- }
811
-
812
- // ---------------------------------------------------------------------------
813
- // CLI.
814
- // ---------------------------------------------------------------------------
815
-
816
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
817
-
818
- if (isMain) {
819
- const argv = process.argv.slice(2)
820
- const flag = (name: string): string | undefined => {
821
- const i = argv.indexOf(name)
822
- return i !== -1 ? argv[i + 1] : undefined
823
- }
824
- const outDir = flag('--out')
825
- if (!outDir) {
826
- console.error(
827
- 'usage: tsx src/quant-arena/quant-loop.mts --out <dir> [--candidates 2] [--seed 20260722] ' +
828
- '[--author-model glm-5.2] [--audit-model glm-5.2] [--skip-llm-audit] # SPENDS: author + audit shots',
829
- )
830
- process.exit(2)
831
- }
832
- const config = defaultConfig(outDir)
833
- if (flag('--candidates')) config.candidatesPerProposer = Number(flag('--candidates'))
834
- if (flag('--seed')) config.seed = Number(flag('--seed'))
835
- if (flag('--author-model')) config.authorModel = flag('--author-model')!
836
- if (flag('--audit-model')) config.auditModel = flag('--audit-model')!
837
- if (argv.includes('--skip-llm-audit')) config.skipLlmAudit = true
838
- const rows = await runQuantCampaign(config)
839
- process.exitCode = rows.some((r) => r.verdict === 'rejected-error') ? 1 : 0
840
- }