@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,840 +0,0 @@
1
- /**
2
- * QUANT-ARENA campaign loop — the improvement loop embodied for trading
3
- * strategies. One command runs: strategy authors (Runtime profile-pinned)
4
- * propose candidate strategies (v2 `onBar` contract, driven incrementally by
5
- * driver.ts) -> every candidate passes a two-stage leak audit -> survivors
6
- * are scored on K bootstrap in-sample windows against the pinned baselines
7
- * -> a multiplicity-adjusted acceptance rule decides -> every try becomes a
8
- * permanent lab-notebook row (notebook.jsonl).
9
- *
10
- * Scoring engines: the OFFICIAL per-window scores come from the vectorbt
11
- * worker (vbt-client.ts -> python/vbt-worker.py). The TS engine
12
- * (backtest.ts) runs first as contract prefilter + leak-audit substrate
13
- * only — it throws on shorting/leverage violations and supplies turnover
14
- * (which the worker protocol does not carry), but its Sharpe/return numbers
15
- * are never the acceptance currency.
16
- *
17
- * tsx src/quant-arena/quant-loop.mts --out <dir> [--candidates 2] [--seed 20260722]
18
- * [--author-model glm-5.2] [--audit-model glm-5.2] [--skip-llm-audit]
19
- *
20
- * Kernel reuse (import, not copy — see src/swe-arena/):
21
- * - cost accounting: the lib's durable CostLedger (createRunCostLedger) +
22
- * crash-orphan reconcile (ledger-orphans.mts) — author/audit shots are
23
- * metered paid calls with receipts in <out>/cost-ledger.jsonl.
24
- * - evidence cells: one cached-result.json per (strategy x window) in the
25
- * swe-arena cell shape, readable by cell-evidence.mts's loadCampaignCells.
26
- * - rollout manifest: a pure reader/join over cells + ledger receipts,
27
- * mirroring swe-arena/manifest.mts (loadLedgerReceipts imported from it).
28
- * - proposer identity: AgentProfile-pinned authors via proposer-fanout.mts's
29
- * loadAuthorProfile + the same ambient-auth-stripped shot env.
30
- *
31
- * The HOLDOUT (final 2 years) is never read here — see holdout-certify.mts.
32
- */
33
-
34
- import { createHash } from 'node:crypto'
35
- import { appendFile, mkdir, readFile, writeFile } from 'node:fs/promises'
36
- import { existsSync } from 'node:fs'
37
- import { join } from 'node:path'
38
- import { fileURLToPath, pathToFileURL } from 'node:url'
39
- import process from 'node:process'
40
- import { createRunCostLedger, fsCampaignStorage } from '@tangle-network/agent-eval/campaign'
41
- import { agentProfileSchema, type AgentProfile } from '@tangle-network/agent-interface'
42
- import { collectAgentTurn, createExecutor, streamAgentTurn } from '@tangle-network/agent-runtime/kernel'
43
- import { loadCampaignCells } from '../swe-arena/cell-evidence.mts'
44
- import { reconcileCrashOrphansOnDisk } from '../swe-arena/ledger-orphans.mts'
45
- import { loadLedgerReceipts } from '../swe-arena/manifest.mts'
46
- import { resolveAuthorProfile, type ProposerSpec } from '../swe-arena/proposer-fanout.mts'
47
- import { runBacktest, statsForRange, type BacktestConfig, type RangeStats } from './backtest.ts'
48
- import { loadInSample, type AlignedBars } from './data.ts'
49
- import { loadStrategyFile } from './driver.ts'
50
- import { truncationInvariance, type TruncationReport } from './leak-audit.ts'
51
- import { decideAcceptance, requiredExcessSharpe, type AcceptanceDecision } from './multiplicity.ts'
52
- import { scoreSignals, VbtWorker, type VbtWindowStats } from './vbt-client.ts'
53
- import { bootstrapWindows, type EvalWindow } from './windows.ts'
54
- import type { GenerateSignals, Signal } from './types.ts'
55
- import * as buyHoldIndex from './strategies/buy-hold-index/strategy.ts'
56
- import * as equalWeight from './strategies/equal-weight/strategy.ts'
57
- import * as smaCrossover from './strategies/sma-crossover/strategy.ts'
58
-
59
- export const QUANT_PROFILES_DIR = fileURLToPath(new URL('./profiles', import.meta.url))
60
- const DEFAULT_AUTHOR_PROFILE = fileURLToPath(
61
- new URL('../swe-arena/profiles/default-author.profile.json', import.meta.url),
62
- )
63
-
64
- // ---------------------------------------------------------------------------
65
- // Config.
66
- // ---------------------------------------------------------------------------
67
-
68
- export interface QuantLoopConfig {
69
- outDir: string
70
- candidatesPerProposer: number
71
- seed: number
72
- windows: number
73
- windowDays: number
74
- warmupDays: number
75
- costBps: number
76
- slippageBps: number
77
- authorModel: string
78
- auditModel: string
79
- skipLlmAudit: boolean
80
- authorTimeoutMs: number
81
- auditTimeoutMs: number
82
- proposers: ProposerSpec[]
83
- }
84
-
85
- export const PINNED_BASELINES: Record<string, GenerateSignals> = {
86
- 'buy-hold-index': buyHoldIndex.generateSignals,
87
- 'equal-weight': equalWeight.generateSignals,
88
- 'sma-crossover': smaCrossover.generateSignals,
89
- }
90
-
91
- /** The two demo author seats: the plain author and the quant lens. */
92
- export function defaultQuantProposers(): ProposerSpec[] {
93
- return [
94
- {
95
- name: 'default-author',
96
- profile: DEFAULT_AUTHOR_PROFILE,
97
- harness: 'pi',
98
- model: 'glm-5.2',
99
- },
100
- {
101
- name: 'quant-researcher',
102
- profile: join(QUANT_PROFILES_DIR, 'quant-researcher.profile.json'),
103
- harness: 'pi',
104
- model: 'glm-5.2',
105
- lens:
106
- 'Favor ONE economically-motivated effect (trend, mean reversion, vol targeting) with few parameters. ' +
107
- 'State the regime in which it should work and keep turnover low enough that 15bps a side cannot eat the edge.',
108
- },
109
- ]
110
- }
111
-
112
- export function defaultConfig(outDir: string): QuantLoopConfig {
113
- return {
114
- outDir,
115
- candidatesPerProposer: 2,
116
- seed: 20260722,
117
- windows: 8,
118
- windowDays: 504,
119
- warmupDays: 120,
120
- costBps: 10,
121
- slippageBps: 5,
122
- authorModel: 'glm-5.2',
123
- auditModel: 'glm-5.2',
124
- skipLlmAudit: false,
125
- authorTimeoutMs: 480_000,
126
- auditTimeoutMs: 240_000,
127
- proposers: defaultQuantProposers(),
128
- }
129
- }
130
-
131
- // ---------------------------------------------------------------------------
132
- // Notebook rows — the permanent lab notebook. Append-only JSONL.
133
- // ---------------------------------------------------------------------------
134
-
135
- export const NOTEBOOK_CANDIDATE_SCHEMA = 'quant-arena.candidate.v1'
136
- export const NOTEBOOK_BASELINES_SCHEMA = 'quant-arena.baselines.v1'
137
-
138
- export type CandidateVerdict =
139
- | 'accepted'
140
- | 'rejected-no-edge'
141
- | 'rejected-leak'
142
- | 'rejected-contract'
143
- | 'rejected-error'
144
-
145
- export interface WindowScore {
146
- start: number
147
- end: number
148
- startDate: string
149
- endDate: string
150
- sharpe: number
151
- bestBaselineSharpe: number
152
- excess: number
153
- }
154
-
155
- export interface CandidateRow {
156
- schema: typeof NOTEBOOK_CANDIDATE_SCHEMA
157
- at: string
158
- candidateId: string
159
- proposer: string
160
- authorModel: string
161
- strategyPath: string | null
162
- sha256: string | null
163
- authoringCostUsd: number | null
164
- /** Total candidates tried this campaign INCLUDING this one — the
165
- * multiplicity denominator. Monotone; never resets within a notebook. */
166
- nTried: number
167
- leakAudit: {
168
- truncation: TruncationReport | null
169
- llm: { verdict: 'clean' | 'leak' | 'inconclusive'; evidence: string } | 'skipped' | null
170
- }
171
- eval: {
172
- perWindow: WindowScore[]
173
- meanExcessSharpe: number
174
- wins: number
175
- requiredWins: number
176
- threshold: number
177
- } | null
178
- inSampleFull: RangeStats | null
179
- verdict: CandidateVerdict
180
- reasons: string[]
181
- }
182
-
183
- export async function loadNotebookRows(notebookPath: string): Promise<Array<Record<string, unknown>>> {
184
- if (!existsSync(notebookPath)) return []
185
- const raw = await readFile(notebookPath, 'utf8')
186
- return raw
187
- .split('\n')
188
- .filter((l) => l.trim().length > 0)
189
- .map((l) => JSON.parse(l) as Record<string, unknown>)
190
- }
191
-
192
- const log = (msg: string): void => console.log(`[${new Date().toISOString().slice(11, 19)}] ${msg}`)
193
-
194
- // ---------------------------------------------------------------------------
195
- // Exact-profile shots (author + auditor) — metered paid calls through Runtime + the run ledger.
196
- // ---------------------------------------------------------------------------
197
-
198
- interface ProfileShotOutcome {
199
- text: string
200
- model: string
201
- inputTokens: number
202
- outputTokens: number
203
- cachedTokens: number
204
- costUsd: number | null
205
- }
206
-
207
- type Ledger = ReturnType<typeof createRunCostLedger>
208
-
209
- function withModel(profile: AgentProfile, model: string, name = profile.name): AgentProfile {
210
- return agentProfileSchema.parse({
211
- ...profile,
212
- ...(name ? { name } : {}),
213
- model: { ...profile.model, default: model },
214
- })
215
- }
216
-
217
- function quantAuditProfile(model: string): AgentProfile {
218
- return agentProfileSchema.parse({
219
- name: 'quant-leak-auditor',
220
- harness: 'pi',
221
- model: { provider: 'tangle-router', default: model },
222
- prompt: {
223
- systemPrompt:
224
- 'Audit the supplied trading strategy for look-ahead bias and nondeterminism. Follow the requested JSON response contract exactly.',
225
- },
226
- })
227
- }
228
-
229
- async function profileShot(opts: {
230
- prompt: string
231
- profile: AgentProfile
232
- timeoutMs: number
233
- cwd: string
234
- }): Promise<ProfileShotOutcome> {
235
- const bridgeUrl = process.env.CLI_BRIDGE_URL ?? process.env.BRIDGE_URL
236
- const bridgeBearer = process.env.CLI_BRIDGE_BEARER ?? process.env.BRIDGE_BEARER
237
- if (!bridgeUrl || !bridgeBearer) {
238
- throw new Error(
239
- 'quant profile shots require CLI_BRIDGE_URL/BRIDGE_URL and CLI_BRIDGE_BEARER/BRIDGE_BEARER',
240
- )
241
- }
242
- const factory = createExecutor({
243
- backend: 'bridge',
244
- bridgeUrl,
245
- bridgeBearer,
246
- cwd: opts.cwd,
247
- timeoutMs: opts.timeoutMs,
248
- })
249
- const turn = await collectAgentTurn(
250
- streamAgentTurn(
251
- { kind: 'executor', factory, profile: opts.profile },
252
- { prompt: opts.prompt },
253
- { timeoutMs: opts.timeoutMs },
254
- ),
255
- )
256
- if (turn.status !== 'completed') {
257
- throw new Error(turn.error?.message ?? `quant profile shot ended with ${turn.status}`)
258
- }
259
- const cachedTokens = Number(turn.usage.promptCache?.readTokens ?? 0)
260
- return {
261
- text: turn.finalText,
262
- model: turn.usage.model ?? opts.profile.model?.default ?? 'unknown',
263
- inputTokens: turn.usage.input,
264
- outputTokens: turn.usage.output,
265
- cachedTokens: Number.isFinite(cachedTokens) ? cachedTokens : 0,
266
- costUsd: turn.usage.costUsd ?? null,
267
- }
268
- }
269
-
270
- async function meteredProfileShot(
271
- ledger: Ledger,
272
- meta: { phase: string; actor: string; tags: Record<string, string> },
273
- opts: Parameters<typeof profileShot>[0],
274
- ): Promise<{ outcome: ProfileShotOutcome; costUsd: number | null }> {
275
- const model = opts.profile.model?.default
276
- if (!model) throw new Error('meteredProfileShot: profile.model.default is required')
277
- const paid = await ledger.runPaidCall<ProfileShotOutcome>({
278
- channel: 'driver',
279
- phase: meta.phase,
280
- actor: meta.actor,
281
- model,
282
- tags: meta.tags,
283
- execute: () => profileShot(opts),
284
- receipt: (v) => ({
285
- model: v.model,
286
- inputTokens: v.inputTokens,
287
- outputTokens: v.outputTokens,
288
- cachedTokens: v.cachedTokens,
289
- ...(v.costUsd !== null ? { actualCostUsd: v.costUsd } : {}),
290
- }),
291
- })
292
- if (!paid.succeeded) throw paid.error
293
- return { outcome: paid.value, costUsd: paid.value.costUsd }
294
- }
295
-
296
- // ---------------------------------------------------------------------------
297
- // Authoring: prompt, extraction, hermeticity guard.
298
- // ---------------------------------------------------------------------------
299
-
300
- const CONTRACT_TEXT = `THE STRATEGY CONTRACT (v2 — incremental)
301
- - Write ONE self-contained TypeScript module. NO import/require/fs/network/process — declare any types you need locally.
302
- - Export exactly: export function onBar(ctx: StrategyContext): TargetPosition[] | null
303
- where StrategyContext = { symbols: string[]; t: number; history: Bar[][]; weights: number[]; equity: number },
304
- Bar = { date: string; open: number; high: number; low: number; close: number; volume: number },
305
- and TargetPosition = { symbol: string; weight: number }.
306
- - The lab calls onBar once per trading day, in order. ctx.history[k] holds the daily bars of ctx.symbols[k] from
307
- day 0 THROUGH TODAY ONLY (ctx.history[k].length === ctx.t + 1) — bars after today do not exist in the array.
308
- ctx.history[0] / ctx.symbols[0] is the benchmark index. ctx.weights and ctx.equity are your current drifted
309
- portfolio state (equity starts at 1).
310
- - Return TargetPosition[] to rebalance: weight = target fraction of equity per symbol; any symbol you omit is
311
- sold to 0. Return null to hold (positions drift with prices). A rebalance fills at the NEXT day's open.
312
- - You never construct orders — the lab's shared rebalancer turns your target weights into orders.
313
- - No shorting, no leverage: every weight >= 0 and the weights sum to <= 1 (rest is cash at 0%). Violations kill
314
- the candidate — fail-closed, not clamped.
315
- - DETERMINISM / NO LOOK-AHEAD: onBar must be a pure function of ctx (no RNG, no clock, no hidden state). The lab
316
- re-runs your code on truncated data; if any decision up to the cutoff changes, the candidate is killed. No
317
- hardcoded calendar dates that memorize this dataset.
318
- - Every fill pays 15bps one-way (cost + slippage) on traded dollars — churn is expensive.`
319
-
320
- export function buildAuthorPrompt(args: {
321
- universe: AlignedBars
322
- windows: EvalWindow[]
323
- baselineTable: string
324
- threshold: number
325
- nTried: number
326
- lens?: string
327
- }): string {
328
- const { universe, windows, baselineTable, threshold, nTried } = args
329
- return [
330
- 'You are proposing ONE candidate trading strategy for a research lab with a strict acceptance rule.',
331
- '',
332
- CONTRACT_TEXT,
333
- '',
334
- `UNIVERSE: ${universe.tickers.length} tickers (${universe.tickers.join(', ')}); bars[0] = ${universe.tickers[0]} (the index).`,
335
- `IN-SAMPLE: ${universe.dates.length} daily bars, ${universe.dates[0]} .. ${universe.dates[universe.dates.length - 1]}.`,
336
- '',
337
- 'ACCEPTANCE RULE (what you must beat):',
338
- `- Scored on ${windows.length} overlapping ${windows[0]!.end - windows[0]!.start}-day in-sample windows.`,
339
- '- You must beat the BEST pinned baseline Sharpe in at least 6 of 8 windows, AND',
340
- `- your mean excess Sharpe must clear ${threshold.toFixed(3)} (the bar rises with every candidate tried; you are try #${nTried}).`,
341
- '',
342
- 'PINNED BASELINES (annualized Sharpe per window; "best" is the per-window max):',
343
- baselineTable,
344
- '',
345
- ...(args.lens ? ['YOUR AUTHORING LENS:', args.lens, ''] : []),
346
- 'Reply with EXACTLY ONE fenced ```ts code block containing the module and nothing else after it.',
347
- ].join('\n')
348
- }
349
-
350
- /** Pull the strategy module out of the author's reply and enforce the
351
- * self-contained rule. Fail-closed: anything ambiguous is a rejection. */
352
- export function extractStrategySource(text: string): { ok: true; code: string } | { ok: false; reason: string } {
353
- const blocks = [...text.matchAll(/```(?:ts|typescript)?\s*\n([\s\S]*?)```/g)].map((m) => m[1]!)
354
- const withExport = blocks.filter((b) => /export\s+function\s+onBar\s*\(/.test(b))
355
- if (withExport.length === 0) {
356
- return { ok: false, reason: 'no fenced code block exporting `onBar` in the reply (v2 contract)' }
357
- }
358
- const code = withExport[withExport.length - 1]!
359
- const stripped = code.replace(/\/\*[\s\S]*?\*\//g, '').replace(/\/\/.*$/gm, '')
360
- const banned = /\b(import|require|fetch|process|globalThis|Deno|XMLHttpRequest|eval)\b/.exec(stripped)
361
- if (banned) {
362
- return { ok: false, reason: `not self-contained: uses banned identifier '${banned[1]}'` }
363
- }
364
- return { ok: true, code }
365
- }
366
-
367
- // ---------------------------------------------------------------------------
368
- // Adversarial LLM leak audit (the deterministic truncation check lives in
369
- // leak-audit.ts; both must pass).
370
- // ---------------------------------------------------------------------------
371
-
372
- const AUDIT_PROMPT_HEADER = `You are an adversarial reviewer with ONE job: find look-ahead bias or nondeterminism
373
- in the trading strategy below. The contract: onBar(ctx) is called once per day; ctx.history[k] holds ONLY bars
374
- 0..ctx.t (the harness slices the arrays), decisions must be pure functions of ctx, and fills happen at the next
375
- day's open. Hunt for:
376
- - hardcoded calendar dates or magic day indexes that smell like memorizing this dataset,
377
- - nondeterminism: Math.random, Date.now, or state carried between onBar calls that a re-run would not rebuild,
378
- - decisions that would change when the future is truncated,
379
- - any attempt to reach data beyond ctx.history (indexing past the array end, reconstructing future prices).
380
- Deciding at close of ctx.t and being filled at t+1's open is LEGAL — do not flag it. Whole-history statistics over
381
- ctx.history are LEGAL (the array ends at today) — do not flag them.
382
- Reply with JSON ONLY: {"verdict":"clean"} or {"verdict":"leak","evidence":"<quote the offending code and why>"}.
383
-
384
- STRATEGY SOURCE:
385
- `
386
-
387
- export function parseAuditVerdict(text: string): { verdict: 'clean' | 'leak'; evidence: string } | null {
388
- const matches = [...text.matchAll(/\{[\s\S]*?"verdict"[\s\S]*?\}/g)]
389
- for (const m of matches.reverse()) {
390
- try {
391
- const parsed = JSON.parse(m[0]) as { verdict?: string; evidence?: string }
392
- if (parsed.verdict === 'clean') return { verdict: 'clean', evidence: '' }
393
- if (parsed.verdict === 'leak') return { verdict: 'leak', evidence: parsed.evidence ?? '(no evidence quoted)' }
394
- } catch {
395
- continue
396
- }
397
- }
398
- return null
399
- }
400
-
401
- async function llmLeakAudit(
402
- ledger: Ledger,
403
- config: QuantLoopConfig,
404
- candidateId: string,
405
- code: string,
406
- ): Promise<{ verdict: 'clean' | 'leak' | 'inconclusive'; evidence: string }> {
407
- for (let attempt = 0; attempt < 2; attempt++) {
408
- const { outcome } = await meteredProfileShot(
409
- ledger,
410
- { phase: 'audit.leak', actor: 'leak-auditor:runtime', tags: { candidateId, attempt: String(attempt) } },
411
- {
412
- prompt: AUDIT_PROMPT_HEADER + '```ts\n' + code + '\n```',
413
- profile: quantAuditProfile(config.auditModel),
414
- timeoutMs: config.auditTimeoutMs,
415
- cwd: config.outDir,
416
- },
417
- )
418
- const verdict = parseAuditVerdict(outcome.text)
419
- if (verdict !== null) return verdict
420
- }
421
- return { verdict: 'inconclusive', evidence: 'auditor reply unparseable twice — fail-closed' }
422
- }
423
-
424
- // ---------------------------------------------------------------------------
425
- // Evidence cells (kernel cell shape) + rollout manifest.
426
- // ---------------------------------------------------------------------------
427
-
428
- async function writeWindowCells(
429
- campaignRoot: string,
430
- label: string,
431
- perWindow: Array<{ window: EvalWindow; stats: RangeStats; bestBaselineSharpe: number | null }>,
432
- ): Promise<string> {
433
- const dir = join(campaignRoot, label)
434
- for (let i = 0; i < perWindow.length; i++) {
435
- const { window, stats, bestBaselineSharpe } = perWindow[i]!
436
- const cellDir = join(dir, `window-${i}-rep-0`)
437
- await mkdir(cellDir, { recursive: true })
438
- const cell = {
439
- scenarioId: `window-${window.start}-${window.end}`,
440
- rep: 0,
441
- artifact: {
442
- kind: 'quant-window',
443
- strategy: label,
444
- windowStart: window.start,
445
- windowEnd: window.end,
446
- sharpe: stats.sharpe,
447
- totalReturn: stats.totalReturn,
448
- maxDrawdown: stats.maxDrawdown,
449
- tradeCount: stats.tradeCount,
450
- turnover: stats.turnover,
451
- bestBaselineSharpe,
452
- excessSharpe: bestBaselineSharpe === null ? null : stats.sharpe - bestBaselineSharpe,
453
- },
454
- costUsd: 0,
455
- tokenUsage: { input: 0, output: 0 },
456
- cached: true,
457
- }
458
- await writeFile(join(cellDir, 'cached-result.json'), JSON.stringify(cell, null, 2))
459
- }
460
- return dir
461
- }
462
-
463
- export const QUANT_ROLLOUT_SCHEMA = 'quant-arena.rollout.v1'
464
-
465
- /** Pure reader/join over what the campaign already wrote — the same posture
466
- * as swe-arena/manifest.mts, over the same cell + ledger primitives. */
467
- export async function writeQuantRolloutManifest(outDir: string): Promise<string> {
468
- const campaignRoot = join(outDir, 'campaign')
469
- const receipts = await loadLedgerReceipts(outDir)
470
- const notebook = await loadNotebookRows(join(outDir, 'notebook.jsonl'))
471
- const entries: Array<Record<string, unknown>> = []
472
- const { readdir } = await import('node:fs/promises')
473
- for (const name of (await readdir(campaignRoot).catch(() => [])).sort()) {
474
- const cells = await loadCampaignCells(join(campaignRoot, name))
475
- if (cells.length === 0) continue
476
- entries.push({
477
- label: name,
478
- campaignDir: join(campaignRoot, name),
479
- cells: cells.length,
480
- scenarios: cells.map((c) => c.scenarioId).sort(),
481
- })
482
- }
483
- const manifest = {
484
- schema: QUANT_ROLLOUT_SCHEMA,
485
- outDir,
486
- at: new Date().toISOString(),
487
- notebookRows: notebook.length,
488
- strategies: entries,
489
- receipts: receipts.map((r) => ({
490
- callId: r.callId,
491
- phase: (r as Record<string, unknown>).phase ?? null,
492
- actor: (r as Record<string, unknown>).actor ?? null,
493
- model: (r as Record<string, unknown>).model ?? null,
494
- costUsd: (r as Record<string, unknown>).costUsd ?? null,
495
- })),
496
- }
497
- const path = join(outDir, 'rollout-manifest.json')
498
- await writeFile(path, JSON.stringify(manifest, null, 2))
499
- return path
500
- }
501
-
502
- // ---------------------------------------------------------------------------
503
- // The campaign.
504
- // ---------------------------------------------------------------------------
505
-
506
- /** RangeStats assembled from the official (vectorbt) numbers plus turnover,
507
- * which only the TS prefilter tracks — labeled at the one place it mixes. */
508
- function rangeStatsFromVbt(stats: VbtWindowStats, start: number, end: number, turnover: number): RangeStats {
509
- return {
510
- start,
511
- end,
512
- days: end - start,
513
- totalReturn: stats.totalReturn,
514
- maxDrawdown: stats.maxDD,
515
- sharpe: stats.sharpe,
516
- tradeCount: stats.trades,
517
- turnover,
518
- }
519
- }
520
-
521
- interface ScoredStrategy {
522
- signals: Signal[]
523
- /** Official per-window stats (vectorbt; turnover from the TS prefilter). */
524
- perWindow: RangeStats[]
525
- /** Official full-range stats. */
526
- full: RangeStats
527
- }
528
-
529
- /** Score one decision record: TS engine first as fail-closed contract
530
- * prefilter (throws on shorting/leverage/malformed signals), then the
531
- * vectorbt worker for the official numbers. */
532
- async function scoreStrategySignals(
533
- worker: VbtWorker,
534
- universe: AlignedBars,
535
- signals: Signal[],
536
- windows: EvalWindow[],
537
- btConfig: BacktestConfig,
538
- ): Promise<ScoredStrategy> {
539
- const prefilter = runBacktest(universe.bars, signals, btConfig)
540
- const ranges = windows.map((w) => [w.start, w.end] as [number, number])
541
- const vbt = await scoreSignals(worker, universe.bars, signals, btConfig, ranges)
542
- const perWindow = windows.map((w, i) =>
543
- rangeStatsFromVbt(vbt.windows[i]!, w.start, w.end, statsForRange(prefilter, w.start, w.end).turnover),
544
- )
545
- const full = rangeStatsFromVbt(vbt.full, 0, universe.dates.length, prefilter.stats.turnover)
546
- return { signals, perWindow, full }
547
- }
548
-
549
- interface BaselineEvidence {
550
- perWindowSharpe: Record<string, number[]>
551
- bestPerWindow: number[]
552
- fullSample: Record<string, RangeStats>
553
- perWindowStats: Record<string, RangeStats[]>
554
- }
555
-
556
- async function evaluateBaselines(
557
- worker: VbtWorker,
558
- universe: AlignedBars,
559
- windows: EvalWindow[],
560
- btConfig: BacktestConfig,
561
- ): Promise<BaselineEvidence> {
562
- const perWindowSharpe: Record<string, number[]> = {}
563
- const fullSample: Record<string, RangeStats> = {}
564
- const perWindowStats: Record<string, RangeStats[]> = {}
565
- for (const [name, strategy] of Object.entries(PINNED_BASELINES)) {
566
- const scored = await scoreStrategySignals(worker, universe, strategy(universe.bars), windows, btConfig)
567
- fullSample[name] = scored.full
568
- perWindowStats[name] = scored.perWindow
569
- perWindowSharpe[name] = scored.perWindow.map((s) => s.sharpe)
570
- }
571
- const bestPerWindow = windows.map((_, i) =>
572
- Math.max(...Object.values(perWindowSharpe).map((sharpes) => sharpes[i]!)),
573
- )
574
- return { perWindowSharpe, bestPerWindow, fullSample, perWindowStats }
575
- }
576
-
577
- function baselineTable(evidence: BaselineEvidence): string {
578
- const lines: string[] = []
579
- for (const [name, sharpes] of Object.entries(evidence.perWindowSharpe)) {
580
- lines.push(` ${name.padEnd(15)} ${sharpes.map((s) => s.toFixed(2).padStart(6)).join(' ')}`)
581
- }
582
- lines.push(` ${'BEST'.padEnd(15)} ${evidence.bestPerWindow.map((s) => s.toFixed(2).padStart(6)).join(' ')}`)
583
- return lines.join('\n')
584
- }
585
-
586
- export async function runQuantCampaign(config: QuantLoopConfig): Promise<CandidateRow[]> {
587
- if (!VbtWorker.isAvailable()) {
588
- throw new Error(
589
- 'quant-arena: the official scorer is the vectorbt worker, which needs `uv` on PATH ' +
590
- '(src/quant-arena/python/uv.lock pins the environment). The TS engine is a prefilter ' +
591
- 'only and cannot stand in — install uv, there is no fallback scorer.',
592
- )
593
- }
594
- const worker = new VbtWorker()
595
- try {
596
- return await runQuantCampaignWithWorker(worker, config)
597
- } finally {
598
- await worker.close()
599
- }
600
- }
601
-
602
- async function runQuantCampaignWithWorker(worker: VbtWorker, config: QuantLoopConfig): Promise<CandidateRow[]> {
603
- await mkdir(config.outDir, { recursive: true })
604
- const notebookPath = join(config.outDir, 'notebook.jsonl')
605
- const reconciled = reconcileCrashOrphansOnDisk(config.outDir)
606
- if (reconciled.length > 0) log(`reconciled ${reconciled.length} crash-orphaned ledger call(s)`)
607
- const ledger = createRunCostLedger({ storage: fsCampaignStorage(), runDir: config.outDir })
608
-
609
- const universe = await loadInSample()
610
- const T = universe.dates.length
611
- const windows = bootstrapWindows(T, {
612
- k: config.windows,
613
- windowDays: config.windowDays,
614
- seed: config.seed,
615
- warmupDays: config.warmupDays,
616
- })
617
- const btConfig: BacktestConfig = { costBps: config.costBps, slippageBps: config.slippageBps }
618
- log(`in-sample: ${T} days x ${universe.tickers.length} tickers; ${windows.length} windows of ${config.windowDays}d (seed ${config.seed})`)
619
- log(`scoring engine: vectorbt ${await worker.ping()} (persistent worker, numba warm)`)
620
-
621
- const baselines = await evaluateBaselines(worker, universe, windows, btConfig)
622
- await appendFile(
623
- notebookPath,
624
- JSON.stringify({
625
- schema: NOTEBOOK_BASELINES_SCHEMA,
626
- at: new Date().toISOString(),
627
- seed: config.seed,
628
- costBps: config.costBps,
629
- slippageBps: config.slippageBps,
630
- windows: windows.map((w) => ({
631
- start: w.start,
632
- end: w.end,
633
- startDate: universe.dates[w.start],
634
- endDate: universe.dates[w.end - 1],
635
- })),
636
- perWindowSharpe: baselines.perWindowSharpe,
637
- bestPerWindow: baselines.bestPerWindow,
638
- fullSample: baselines.fullSample,
639
- }) + '\n',
640
- )
641
- const campaignRoot = join(config.outDir, 'campaign')
642
- for (const name of Object.keys(PINNED_BASELINES)) {
643
- await writeWindowCells(
644
- campaignRoot,
645
- `baseline-${name}`,
646
- windows.map((w, i) => ({
647
- window: w,
648
- stats: baselines.perWindowStats[name]![i]!,
649
- bestBaselineSharpe: baselines.bestPerWindow[i]!,
650
- })),
651
- )
652
- }
653
-
654
- const priorRows = await loadNotebookRows(notebookPath)
655
- let nTried = priorRows.filter((r) => r.schema === NOTEBOOK_CANDIDATE_SCHEMA).length
656
- const rows: CandidateRow[] = []
657
-
658
- for (const proposer of config.proposers) {
659
- const sourceProfile = resolveAuthorProfile(proposer)
660
- if (!sourceProfile) throw new Error(`quant proposer ${proposer.name}: exact profile is required`)
661
- const profile = withModel(sourceProfile, config.authorModel, `quant-${proposer.name}`)
662
- for (let shot = 0; shot < config.candidatesPerProposer; shot++) {
663
- nTried += 1
664
- const candidateId = `cand-${String(nTried).padStart(3, '0')}-${proposer.name}`
665
- const threshold = requiredExcessSharpe(nTried)
666
- log(`--- ${candidateId}: authoring (try #${nTried}, bar ${threshold.toFixed(3)})`)
667
-
668
- const row: CandidateRow = {
669
- schema: NOTEBOOK_CANDIDATE_SCHEMA,
670
- at: new Date().toISOString(),
671
- candidateId,
672
- proposer: proposer.name,
673
- authorModel: profile.model?.default ?? config.authorModel,
674
- strategyPath: null,
675
- sha256: null,
676
- authoringCostUsd: null,
677
- nTried,
678
- leakAudit: { truncation: null, llm: null },
679
- eval: null,
680
- inSampleFull: null,
681
- verdict: 'rejected-error',
682
- reasons: [],
683
- }
684
-
685
- try {
686
- const prompt = buildAuthorPrompt({
687
- universe,
688
- windows,
689
- baselineTable: baselineTable(baselines),
690
- threshold,
691
- nTried,
692
- ...(proposer.lens ? { lens: proposer.lens } : {}),
693
- })
694
- const { outcome, costUsd } = await meteredProfileShot(
695
- ledger,
696
- { phase: 'search.proposal', actor: `proposer-shot:${proposer.name}`, tags: { candidateId } },
697
- {
698
- prompt,
699
- profile,
700
- timeoutMs: config.authorTimeoutMs,
701
- cwd: config.outDir,
702
- },
703
- )
704
- row.authoringCostUsd = costUsd
705
-
706
- const extracted = extractStrategySource(outcome.text)
707
- if (!extracted.ok) {
708
- row.verdict = 'rejected-contract'
709
- row.reasons = [extracted.reason]
710
- } else {
711
- const strategyDir = join(config.outDir, 'strategies', candidateId)
712
- await mkdir(strategyDir, { recursive: true })
713
- const strategyPath = join(strategyDir, 'strategy.ts')
714
- await writeFile(strategyPath, extracted.code)
715
- row.strategyPath = strategyPath
716
- row.sha256 = `sha256:${createHash('sha256').update(extracted.code).digest('hex')}`
717
-
718
- let strategy: GenerateSignals | null = null
719
- try {
720
- // v2 (`onBar`) modules are wrapped through the incremental
721
- // driver, which structurally truncates history per bar.
722
- const loaded = await loadStrategyFile(strategyPath, {
723
- symbols: universe.tickers,
724
- costBps: config.costBps,
725
- slippageBps: config.slippageBps,
726
- })
727
- strategy = loaded.generateSignals
728
- } catch (cause) {
729
- row.verdict = 'rejected-contract'
730
- row.reasons = [(cause as Error).message.slice(0, 300)]
731
- }
732
- if (strategy !== null) {
733
- // Leak audit stage 1: deterministic truncation invariance.
734
- const truncation = truncationInvariance(strategy, universe.bars, { warmupDays: config.warmupDays })
735
- row.leakAudit.truncation = truncation
736
- // Leak audit stage 2: adversarial source read.
737
- const llm = config.skipLlmAudit
738
- ? ('skipped' as const)
739
- : await llmLeakAudit(ledger, config, candidateId, extracted.code)
740
- row.leakAudit.llm = llm
741
- const llmBad = llm !== 'skipped' && llm.verdict !== 'clean'
742
- if (!truncation.clean || llmBad) {
743
- row.verdict = 'rejected-leak'
744
- row.reasons = [
745
- ...(truncation.clean
746
- ? []
747
- : [`truncation divergence at cutoff ${truncation.divergence!.cutoff}: ${truncation.divergence!.detail}`]),
748
- ...(llmBad ? [`adversarial audit: ${(llm as { verdict: string; evidence: string }).evidence || (llm as { verdict: string }).verdict}`] : []),
749
- ]
750
- } else {
751
- // Eval: TS prefilter (contract enforcement) + official
752
- // vectorbt scores, per window against the pinned bar.
753
- const scored = await scoreStrategySignals(worker, universe, strategy(universe.bars), windows, btConfig)
754
- row.inSampleFull = scored.full
755
- const perWindow: WindowScore[] = windows.map((w, i) => {
756
- const stats = scored.perWindow[i]!
757
- return {
758
- start: w.start,
759
- end: w.end,
760
- startDate: universe.dates[w.start]!,
761
- endDate: universe.dates[w.end - 1]!,
762
- sharpe: stats.sharpe,
763
- bestBaselineSharpe: baselines.bestPerWindow[i]!,
764
- excess: stats.sharpe - baselines.bestPerWindow[i]!,
765
- }
766
- })
767
- await writeWindowCells(
768
- campaignRoot,
769
- candidateId,
770
- windows.map((w, i) => ({
771
- window: w,
772
- stats: scored.perWindow[i]!,
773
- bestBaselineSharpe: baselines.bestPerWindow[i]!,
774
- })),
775
- )
776
- const decision: AcceptanceDecision = decideAcceptance({
777
- perWindowExcess: perWindow.map((w) => w.excess),
778
- nTried,
779
- })
780
- row.eval = {
781
- perWindow,
782
- meanExcessSharpe: decision.meanExcessSharpe,
783
- wins: decision.wins,
784
- requiredWins: decision.requiredWins,
785
- threshold: decision.threshold,
786
- }
787
- row.verdict = decision.accepted ? 'accepted' : 'rejected-no-edge'
788
- row.reasons = decision.reasons
789
- }
790
- }
791
- }
792
- } catch (cause) {
793
- row.verdict = 'rejected-error'
794
- row.reasons = [(cause as Error).message.slice(0, 500)]
795
- }
796
-
797
- await appendFile(notebookPath, JSON.stringify(row) + '\n')
798
- rows.push(row)
799
- log(`${candidateId}: ${row.verdict}${row.reasons.length > 0 ? ` — ${row.reasons[0]}` : ''}`)
800
- }
801
- }
802
-
803
- const manifestPath = await writeQuantRolloutManifest(config.outDir)
804
- const summary = ledger.summary()
805
- log(`campaign done: ${rows.length} candidates, ${rows.filter((r) => r.verdict === 'accepted').length} accepted`)
806
- log(`spend: $${summary.totalCostUsd.toFixed(4)} across ${summary.totalCalls} metered calls (${summary.inputTokens} in / ${summary.outputTokens} out tokens)`)
807
- log(`notebook -> ${notebookPath}`)
808
- log(`rollout manifest -> ${manifestPath}`)
809
- return rows
810
- }
811
-
812
- // ---------------------------------------------------------------------------
813
- // CLI.
814
- // ---------------------------------------------------------------------------
815
-
816
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
817
-
818
- if (isMain) {
819
- const argv = process.argv.slice(2)
820
- const flag = (name: string): string | undefined => {
821
- const i = argv.indexOf(name)
822
- return i !== -1 ? argv[i + 1] : undefined
823
- }
824
- const outDir = flag('--out')
825
- if (!outDir) {
826
- console.error(
827
- 'usage: tsx src/quant-arena/quant-loop.mts --out <dir> [--candidates 2] [--seed 20260722] ' +
828
- '[--author-model glm-5.2] [--audit-model glm-5.2] [--skip-llm-audit] # SPENDS: author + audit shots',
829
- )
830
- process.exit(2)
831
- }
832
- const config = defaultConfig(outDir)
833
- if (flag('--candidates')) config.candidatesPerProposer = Number(flag('--candidates'))
834
- if (flag('--seed')) config.seed = Number(flag('--seed'))
835
- if (flag('--author-model')) config.authorModel = flag('--author-model')!
836
- if (flag('--audit-model')) config.auditModel = flag('--audit-model')!
837
- if (argv.includes('--skip-llm-audit')) config.skipLlmAudit = true
838
- const rows = await runQuantCampaign(config)
839
- process.exitCode = rows.some((r) => r.verdict === 'rejected-error') ? 1 : 0
840
- }