@tangle-network/agent-bench 0.11.3 → 0.13.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,217 +0,0 @@
1
- /**
2
- * Dual calibration — an instance may enter an experiment only if BOTH gates
3
- * hold, mirroring the experiment's `calibrate.sh` (repro gate) plus the
4
- * gold-family judge rows M1 reconciles on (official-judge gate):
5
- *
6
- * 1. REPRO GATE (`verifyCalibrated`): on a pristine image-materialized
7
- * workspace the self-repro verify command must FAIL at base_commit and
8
- * PASS once the official gold patch is applied (git apply, then
9
- * `patch --fuzz=3` fallback — several Verified gold patches only apply
10
- * fuzzily to their own base). A verify that can't see the gold fix can't
11
- * grade an arm's fix.
12
- *
13
- * 2. OFFICIAL-JUDGE GOLD GATE (`goldOfficialResolved`): the official swebench
14
- * judge (via serialized-judge → adapter.judge) must resolve the gold patch
15
- * itself. psf__requests-2931/-2317 proved a judge can be blind on an
16
- * instance whose verify calibrates fine — those became the excluded
17
- * "gold-ungradeable" rows in the M1 denominator.
18
- */
19
-
20
- import { rm, writeFile } from 'node:fs/promises'
21
- import { join } from 'node:path'
22
- import { pathToFileURL } from 'node:url'
23
- import { judgeFactoryPatch } from './factory-judge-child.mts'
24
- import { loadFactoryInstance, loadFactoryInstances, type LoadedFactoryInstance } from './fixtures'
25
- import { materializeWorkspace } from './materialize'
26
- import { run, runOk, shq } from './proc'
27
- import type { SerializedJudge } from './serialized-judge'
28
-
29
- export interface CalibrateOptions {
30
- instanceId: string
31
- image: string
32
- baseCommit: string
33
- /** The official gold patch text (task-meta `patch`). */
34
- goldPatch: string
35
- /**
36
- * Self-repro verify command, run via `bash -c` with cwd = the workspace
37
- * (calibrate.sh ran `bash verify/<iid>.sh` from inside the tree).
38
- */
39
- verifyCmd: string
40
- /** Scratch root; two throwaway workspaces are created and removed under it. */
41
- workDir: string
42
- /** Judge for the official gold gate. */
43
- judge: SerializedJudge
44
- /** Ceiling for one verify run (repro scripts self-limit at 180s; this is a backstop). */
45
- verifyTimeoutMs?: number
46
- /** Keep the calibration workspaces for post-mortem. Default: removed. */
47
- keepWorkspaces?: boolean
48
- }
49
-
50
- export interface CalibrationResult {
51
- iid: string
52
- /** rc of verify on pristine base (must be nonzero). */
53
- baseRc: number
54
- /** rc of applying the gold patch (0 via git apply or the fuzz fallback). */
55
- goldApplyRc: number
56
- /** rc of verify with gold applied (must be zero). */
57
- goldRc: number
58
- /** base FAILS and gold PASSES. */
59
- verifyCalibrated: boolean
60
- /** Official judge resolves the gold patch. */
61
- goldOfficialResolved: boolean
62
- /** verifyCalibrated && goldOfficialResolved — the experiment admission bar. */
63
- experimentValid: boolean
64
- }
65
-
66
- async function runVerify(verifyCmd: string, ws: string, timeoutMs: number): Promise<number> {
67
- const res = await run('bash', ['-c', verifyCmd], { cwd: ws, timeoutMs })
68
- return res.code
69
- }
70
-
71
- /**
72
- * Apply a patch file with calibrate.sh's exact fallback chain:
73
- * `git apply --whitespace=nowarn`, then `patch -p1 --fuzz=3` on failure.
74
- * Returns the rc of the LAST attempt (0 = applied).
75
- */
76
- export async function applyPatchWithFallback(ws: string, patchFile: string): Promise<number> {
77
- const gitApply = await run('git', ['apply', '--whitespace=nowarn', patchFile], { cwd: ws })
78
- if (gitApply.code === 0) return 0
79
- const fuzz = await run('bash', ['-c', `patch -p1 --fuzz=3 < ${shq(patchFile)}`], { cwd: ws })
80
- return fuzz.code
81
- }
82
-
83
- export async function calibrateInstance(opts: CalibrateOptions): Promise<CalibrationResult> {
84
- const { instanceId, image, baseCommit, verifyCmd, judge } = opts
85
- const verifyTimeoutMs = opts.verifyTimeoutMs ?? 600_000
86
- const baseWs = join(opts.workDir, `cal-base-${instanceId}`)
87
- const goldWs = join(opts.workDir, `cal-gold-${instanceId}`)
88
- const goldPatchFile = join(opts.workDir, `${instanceId}.gold.patch`)
89
-
90
- try {
91
- await materializeWorkspace({ instanceId, image, baseCommit, dest: baseWs })
92
- const baseRc = await runVerify(verifyCmd, baseWs, verifyTimeoutMs)
93
-
94
- await materializeWorkspace({ instanceId, image, baseCommit, dest: goldWs })
95
- await writeFile(goldPatchFile, opts.goldPatch)
96
- const goldApplyRc = await applyPatchWithFallback(goldWs, goldPatchFile)
97
- const goldRc = await runVerify(verifyCmd, goldWs, verifyTimeoutMs)
98
-
99
- const verifyCalibrated = baseRc !== 0 && goldRc === 0
100
-
101
- const goldVerdict = await judge.judge(instanceId, goldPatchFile, 'gold')
102
- const goldOfficialResolved = goldVerdict.resolved === true
103
-
104
- return {
105
- iid: instanceId,
106
- baseRc,
107
- goldApplyRc,
108
- goldRc,
109
- verifyCalibrated,
110
- goldOfficialResolved,
111
- experimentValid: verifyCalibrated && goldOfficialResolved,
112
- }
113
- } finally {
114
- if (!opts.keepWorkspaces) {
115
- await rm(baseWs, { recursive: true, force: true })
116
- await rm(goldWs, { recursive: true, force: true })
117
- }
118
- }
119
- }
120
-
121
- // ---------------------------------------------------------------------------
122
- // Factory-bench admission gate — the same "calibrate through the OFFICIAL
123
- // judge" lesson, generalized: gold (the real PR's impl-only diff) must judge
124
- // resolved, and the bare base (empty patch) must judge unresolved. Both runs
125
- // go through the SAME judge code the arena uses (judgeFactoryPatch — the
126
- // factory-judge-child body), deliberately bypassing serialized-judge's
127
- // empty-patch short-circuit so the base direction really executes the judge
128
- // tests on the bare tree instead of trivially returning false.
129
- // ---------------------------------------------------------------------------
130
-
131
- export interface FactoryCalibrationResult {
132
- iid: string
133
- /** Gold = impl-only PR diff. Must be resolved with full score. */
134
- goldResolved: boolean
135
- goldPassed: number
136
- /** Base = empty patch. Must be unresolved. */
137
- baseResolved: boolean
138
- basePassed: number
139
- total: number
140
- /** goldResolved && !baseResolved — the pool admission bar. */
141
- admitted: boolean
142
- }
143
-
144
- /**
145
- * The real PR's impl-only patch: full first-parent diff base→judge_ref minus
146
- * the judge test files (they are the hidden judge, not the deliverable).
147
- */
148
- export async function goldImplPatch(inst: LoadedFactoryInstance): Promise<string> {
149
- const res = await runOk('git', [
150
- '-C', inst.repo_local_mirror,
151
- 'diff', inst.base_commit, inst.judge_ref,
152
- '--', '.',
153
- ...inst.judge_tests.map((t) => `:(exclude)${t}`),
154
- ])
155
- if (res.stdout.trim().length === 0) {
156
- throw new Error(`calibrate ${inst.id}: impl-only gold diff is empty — judge_tests exclude everything?`)
157
- }
158
- return res.stdout
159
- }
160
-
161
- /** Run both admission directions for one instance. Throws only on infra failure. */
162
- export async function calibrateFactoryInstance(inst: LoadedFactoryInstance): Promise<FactoryCalibrationResult> {
163
- const gold = await judgeFactoryPatch(inst, await goldImplPatch(inst))
164
- const base = await judgeFactoryPatch(inst, '')
165
- return {
166
- iid: inst.id,
167
- goldResolved: gold.result.resolved,
168
- goldPassed: gold.result.passed,
169
- baseResolved: base.result.resolved,
170
- basePassed: base.result.passed,
171
- total: inst.judgeTestTotal,
172
- admitted: gold.result.resolved && !base.result.resolved,
173
- }
174
- }
175
-
176
- // ---------------------------------------------------------------------------
177
- // CLI: tsx src/swe-arena/calibrate.ts --factory <instancesDirOrInstanceDir> [id ...]
178
- // Rejection is LOUD: any instance failing either direction exits nonzero.
179
- // ---------------------------------------------------------------------------
180
-
181
- const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
182
-
183
- if (isMain) {
184
- const [mode, root, ...ids] = process.argv.slice(2)
185
- if (mode !== '--factory' || !root) {
186
- console.error('usage: tsx src/swe-arena/calibrate.ts --factory <instancesDir|instanceDir> [id ...]')
187
- process.exit(2)
188
- }
189
- let instances: LoadedFactoryInstance[]
190
- try {
191
- instances = loadFactoryInstances(root)
192
- } catch {
193
- instances = [loadFactoryInstance(root)]
194
- }
195
- if (ids.length > 0) {
196
- const byId = new Map(instances.map((i) => [i.id, i]))
197
- instances = ids.map((id) => {
198
- const inst = byId.get(id)
199
- if (!inst) throw new Error(`unknown instance id ${id} (have: ${[...byId.keys()].join(', ')})`)
200
- return inst
201
- })
202
- }
203
- let rejected = 0
204
- for (const inst of instances) {
205
- const r = await calibrateFactoryInstance(inst)
206
- const verdict = r.admitted ? 'ADMITTED' : 'REJECTED'
207
- console.log(
208
- `CALIBRATE ${r.iid}: gold ${r.goldPassed}/${r.total} resolved=${r.goldResolved}; ` +
209
- `base ${r.basePassed}/${r.total} resolved=${r.baseResolved} → ${verdict}`,
210
- )
211
- if (!r.admitted) rejected += 1
212
- }
213
- if (rejected > 0) {
214
- console.error(`calibration gate: ${rejected} instance(s) REJECTED (gold must pass AND base must fail)`)
215
- process.exit(1)
216
- }
217
- }
@@ -1,76 +0,0 @@
1
- /**
2
- * Substrate passthrough guard for the improve-loop options this harness
3
- * depends on: `selfImprove` forwarding `premeasuredBaseline` and
4
- * `budget.maxImprovementShots` into the loop (merged to agent-eval main).
5
- *
6
- * The remaining risk is a STALE INSTALL: the bench consumes agent-eval via a
7
- * pnpm `file:` dependency, which snapshots the checkout at install time — a
8
- * rebuilt-but-never-reinstalled substrate silently reverts to a bundle whose
9
- * `selfImprove` DROPS both options. Silent drop is the worst failure mode
10
- * here (a "premeasured" baseline would quietly re-run and re-spend; the depth
11
- * dial would quietly pin to the lib default), so the guard FAILS LOUD instead
12
- * of falling back.
13
- *
14
- * The probe reads the RESOLVED `@tangle-network/agent-eval/contract` module
15
- * text (the bundle that contains the compiled `selfImprove`) and requires
16
- * both option names. Verified against the pre-merge build: that bundle
17
- * contained ZERO occurrences of either literal (selfImprove never named them;
18
- * `runOptimization`'s own seam compiles into a different chunk), so the probe
19
- * cannot false-positive on a stale substrate.
20
- */
21
-
22
- import { readFileSync } from 'node:fs'
23
- import { createRequire } from 'node:module'
24
- import { fileURLToPath } from 'node:url'
25
-
26
- export interface ImproveLoopPassthroughCaps {
27
- /** `selfImprove` forwards `premeasuredBaseline` into the loop. */
28
- premeasuredBaseline: boolean
29
- /** `selfImprove` forwards `budget.maxImprovementShots` into the loop. */
30
- maxImprovementShots: boolean
31
- }
32
-
33
- /** Pure probe over the contract module text — unit-testable. */
34
- export function detectPassthroughCaps(contractModuleText: string): ImproveLoopPassthroughCaps {
35
- return {
36
- premeasuredBaseline: contractModuleText.includes('premeasuredBaseline'),
37
- maxImprovementShots: contractModuleText.includes('maxImprovementShots'),
38
- }
39
- }
40
-
41
- /** Resolve the installed agent-eval contract bundle's file path. */
42
- export function resolveContractModulePath(): string {
43
- const require = createRequire(import.meta.url)
44
- const url = import.meta.resolve?.('@tangle-network/agent-eval/contract')
45
- if (typeof url === 'string' && url.startsWith('file:')) return fileURLToPath(url)
46
- return require.resolve('@tangle-network/agent-eval/contract')
47
- }
48
-
49
- /** Fail-loud stale-install guard: throws unless the resolved substrate names
50
- * BOTH passthrough options in its contract bundle. An unreadable bundle also
51
- * throws — nothing here ever downgrades to a silent fallback. */
52
- export function assertSubstratePassthroughs(log: (msg: string) => void = () => {}): void {
53
- let path: string
54
- let caps: ImproveLoopPassthroughCaps
55
- try {
56
- path = resolveContractModulePath()
57
- caps = detectPassthroughCaps(readFileSync(path, 'utf8'))
58
- } catch (cause) {
59
- throw new Error(
60
- 'substrate passthrough probe failed — cannot prove the installed agent-eval forwards ' +
61
- `premeasuredBaseline/maxImprovementShots: ${(cause as Error).message}`,
62
- { cause },
63
- )
64
- }
65
- log(
66
- `substrate caps (${path}): premeasuredBaseline=${caps.premeasuredBaseline} maxImprovementShots=${caps.maxImprovementShots}`,
67
- )
68
- const missing = (Object.keys(caps) as Array<keyof ImproveLoopPassthroughCaps>).filter((k) => !caps[k])
69
- if (missing.length > 0) {
70
- throw new Error(
71
- `stale substrate install: the resolved agent-eval contract bundle (${path}) never names ` +
72
- `${missing.join(' + ')}, so selfImprove would silently drop the option(s). ` +
73
- 'Rebuild the checkout (cd ~/code/agent-eval && git pull && pnpm build), then `pnpm install --force` in the bench.',
74
- )
75
- }
76
- }
@@ -1,57 +0,0 @@
1
- /**
2
- * Substrate passthrough guard: the pure probe over module text plus the live
3
- * contract check against the installed agent-eval bundle. The guard must pass
4
- * cleanly on a substrate that threads premeasuredBaseline +
5
- * maxImprovementShots and THROW LOUD (never fall back) on one that drops
6
- * them — so a run against a stale install dies at t≈0, before it can
7
- * silently re-spend its baseline.
8
- */
9
-
10
- import { readFileSync } from 'node:fs'
11
- import { describe, expect, it } from 'vitest'
12
- import {
13
- assertSubstratePassthroughs,
14
- detectPassthroughCaps,
15
- resolveContractModulePath,
16
- } from './capabilities.mts'
17
-
18
- describe('detectPassthroughCaps (pure)', () => {
19
- it('an option name absent from the bundle reads as no capability', () => {
20
- expect(detectPassthroughCaps('')).toEqual({ premeasuredBaseline: false, maxImprovementShots: false })
21
- expect(detectPassthroughCaps('function selfImprove(opts) { return runSelfImprove(opts) }')).toEqual({
22
- premeasuredBaseline: false,
23
- maxImprovementShots: false,
24
- })
25
- })
26
-
27
- it('the forwarding literals flip their capability independently', () => {
28
- expect(detectPassthroughCaps('premeasuredBaseline: opts.premeasuredBaseline,')).toEqual({
29
- premeasuredBaseline: true,
30
- maxImprovementShots: false,
31
- })
32
- expect(detectPassthroughCaps('maxImprovementShots: budget.maxImprovementShots,')).toEqual({
33
- premeasuredBaseline: false,
34
- maxImprovementShots: true,
35
- })
36
- })
37
- })
38
-
39
- describe('live substrate guard', () => {
40
- it('passes on a substrate that names both passthroughs, throws loud otherwise', () => {
41
- const path = resolveContractModulePath()
42
- expect(path).toMatch(/agent-eval/)
43
- const caps = detectPassthroughCaps(readFileSync(path, 'utf8'))
44
- // The guard's contract holds against whatever substrate is installed:
45
- // both literals present ⇒ clean pass with the caps logged; anything less
46
- // ⇒ a loud actionable error naming the stale bundle (never a silent
47
- // fallback). A run against a stale install dies HERE, at t≈0.
48
- if (caps.premeasuredBaseline && caps.maxImprovementShots) {
49
- const logged: string[] = []
50
- expect(() => assertSubstratePassthroughs((msg) => logged.push(msg))).not.toThrow()
51
- expect(logged.join('\n')).toContain('premeasuredBaseline=true')
52
- expect(logged.join('\n')).toContain('maxImprovementShots=true')
53
- } else {
54
- expect(() => assertSubstratePassthroughs()).toThrow(/stale substrate install/)
55
- }
56
- })
57
- })
@@ -1,198 +0,0 @@
1
- /**
2
- * Endpoint capacity gate — the typed port of `probe-capacity.sh`, generalized
3
- * after a proven blind spot: the bash probe watched ONLY the z.ai coding
4
- * endpoint while the supervisor BRAIN rides router.tangle.tools — three
5
- * evolution rounds went infra-null because the gate said "capacity" while the
6
- * router 503-stormed. Rule encoded here: gate every arm on the endpoint that
7
- * arm actually calls; supervisor arms MUST include the router-path probe.
8
- *
9
- * The containing experiment is launched through dotenvx, so probes read the already-scoped key
10
- * from this process and enter Runtime through one exact AgentProfile.
11
- */
12
-
13
- import { runBenchRouterTurn } from '../router-turn'
14
- import type { SecretsEnv } from './arms'
15
-
16
- export type CapacityProbe = (signal?: AbortSignal) => Promise<boolean>
17
-
18
- export interface EndpointCapacityGate {
19
- /** Human label for status lines (e.g. 'z.ai-coding', 'router'). */
20
- name: string
21
- probe: CapacityProbe
22
- /** Window passes when >= k of n probes succeed (bash: 3 of 4). */
23
- kOfN: { k: number; n: number }
24
- /** Consecutive passing windows required before opening. Default 1. */
25
- steadyM?: number
26
- /** Total wait budget; exceeded → gate reports no-capacity (orchestrate: 300 min). */
27
- waitCeilingMs: number
28
- /** Pause between probes inside a window (bash: 1s). */
29
- probeIntervalMs?: number
30
- /** Pause between windows while waiting (orchestrate: 30s). */
31
- retryDelayMs?: number
32
- onStatus?: (msg: string) => void
33
- }
34
-
35
- export function sleepWithSignal(ms: number, signal?: AbortSignal): Promise<void> {
36
- signal?.throwIfAborted()
37
- return new Promise((resolve, reject) => {
38
- let timer: NodeJS.Timeout | undefined
39
- const onAbort = () => {
40
- if (timer) clearTimeout(timer)
41
- signal?.removeEventListener('abort', onAbort)
42
- reject(signal?.reason ?? new Error('capacity wait aborted'))
43
- }
44
- timer = setTimeout(() => {
45
- signal?.removeEventListener('abort', onAbort)
46
- resolve()
47
- }, ms)
48
- signal?.addEventListener('abort', onAbort, { once: true })
49
- })
50
- }
51
-
52
- /** One k-of-n probe window. Exported for direct reuse (bash probe-capacity.sh body). */
53
- export async function probeWindow(
54
- gate: EndpointCapacityGate,
55
- signal?: AbortSignal,
56
- ): Promise<{ ok: number; passed: boolean }> {
57
- const { k, n } = gate.kOfN
58
- let ok = 0
59
- for (let i = 0; i < n; i++) {
60
- signal?.throwIfAborted()
61
- try {
62
- if (await gate.probe(signal)) ok += 1
63
- } catch {
64
- // Endpoint failures count as a failed probe; caller cancellation does not.
65
- signal?.throwIfAborted()
66
- }
67
- signal?.throwIfAborted()
68
- if (i < n - 1) await sleepWithSignal(gate.probeIntervalMs ?? 1000, signal)
69
- }
70
- return { ok, passed: ok >= k }
71
- }
72
-
73
- /**
74
- * Block until the endpoint shows steady capacity (steadyM consecutive passing
75
- * k-of-n windows) or the ceiling elapses. Returns whether capacity was found —
76
- * callers decide whether a closed gate skips the instance or aborts the run.
77
- */
78
- export async function waitForCapacity(gate: EndpointCapacityGate, signal?: AbortSignal): Promise<boolean> {
79
- signal?.throwIfAborted()
80
- const steadyM = gate.steadyM ?? 1
81
- const deadline = Date.now() + gate.waitCeilingMs
82
- let consecutive = 0
83
- for (;;) {
84
- signal?.throwIfAborted()
85
- const { ok, passed } = await probeWindow(gate, signal)
86
- signal?.throwIfAborted()
87
- gate.onStatus?.(`[${gate.name}] capacity: ${ok}/${gate.kOfN.n}${passed ? '' : ' (below k)'} steady=${passed ? consecutive + 1 : 0}/${steadyM}`)
88
- if (passed) {
89
- consecutive += 1
90
- if (consecutive >= steadyM) return true
91
- } else {
92
- consecutive = 0
93
- }
94
- if (Date.now() >= deadline) return false
95
- await sleepWithSignal(gate.retryDelayMs ?? 30_000, signal)
96
- }
97
- }
98
-
99
- // ---------------------------------------------------------------------------
100
- // Probe functions.
101
- // ---------------------------------------------------------------------------
102
-
103
- export interface HttpProbeSpec {
104
- url: string
105
- provider: string
106
- /** NAME of the env var holding the bearer key (resolved in the dotenvx child). */
107
- apiKeyEnv: string
108
- model: string
109
- /** curl --max-time, seconds. Default 40 (probe-capacity.sh). */
110
- maxTimeS?: number
111
- /**
112
- * max_tokens in the probe body. Default 8000 — glm-5.2 returns empty content
113
- * below that (measured), and an empty-content 200 would be a lying probe.
114
- */
115
- maxTokens?: number
116
- }
117
-
118
- export const ZAI_CODING_ENDPOINT = 'https://api.z.ai/api/coding/paas/v4'
119
- export const ROUTER_ENDPOINT = 'https://router.tangle.tools/v1'
120
-
121
- /**
122
- * Generic chat-completions probe through Runtime: true only when the selected model emits `OK`
123
- * within the time budget.
124
- */
125
- export function httpCapacityProbe(spec: HttpProbeSpec): CapacityProbe {
126
- if (!/^[A-Z_][A-Z0-9_]*$/.test(spec.apiKeyEnv)) {
127
- throw new Error(`invalid apiKeyEnv name: ${spec.apiKeyEnv}`)
128
- }
129
- const maxTime = spec.maxTimeS ?? 40
130
- return async (signal?: AbortSignal) => {
131
- signal?.throwIfAborted()
132
- const routerKey = process.env[spec.apiKeyEnv]
133
- if (!routerKey) throw new Error(`${spec.apiKeyEnv} is required; launch through dotenvx`)
134
- const turn = await runBenchRouterTurn(
135
- {
136
- routerBaseUrl: spec.url.replace(/\/chat\/completions\/?$/u, ''),
137
- routerKey,
138
- profile: {
139
- name: `capacity-${spec.model}`,
140
- harness: 'cli-base',
141
- model: {
142
- provider: spec.provider,
143
- default: spec.model,
144
- metadata: { temperature: 0 },
145
- maxVisibleOutputTokens: spec.maxTokens ?? 8000,
146
- },
147
- prompt: { systemPrompt: 'Reply with the single word OK.' },
148
- },
149
- timeoutMs: maxTime * 1000,
150
- signal,
151
- },
152
- 'Capacity probe.',
153
- )
154
- return turn.finalText.trim() === 'OK'
155
- }
156
- }
157
-
158
- /** probe-capacity.sh's z.ai coding-plan probe (the WORKER path). */
159
- export function zaiCodingProbe(secrets: SecretsEnv, model = 'glm-5.2'): CapacityProbe {
160
- void secrets
161
- return httpCapacityProbe({
162
- url: ZAI_CODING_ENDPOINT,
163
- apiKeyEnv: 'ZAI_API_KEY',
164
- provider: 'zai',
165
- model,
166
- })
167
- }
168
-
169
- /**
170
- * Router-path probe (the BRAIN path — router.tangle.tools with TANGLE_API_KEY).
171
- * Supervisor arms must gate on this; probing only z.ai is the proven blind spot.
172
- */
173
- export function routerProbe(secrets: SecretsEnv, model = 'glm-5.2'): CapacityProbe {
174
- void secrets
175
- return httpCapacityProbe({
176
- url: ROUTER_ENDPOINT,
177
- apiKeyEnv: 'TANGLE_API_KEY',
178
- provider: 'tangle-router',
179
- model,
180
- })
181
- }
182
-
183
- /** The gates an arm must pass, by kind: solo → worker path; supervisor → BOTH paths. */
184
- export function gatesForArmKind(
185
- kind: 'solo' | 'supervisor',
186
- secrets: SecretsEnv,
187
- opts: { waitCeilingMs?: number; model?: string; onStatus?: (msg: string) => void } = {},
188
- ): EndpointCapacityGate[] {
189
- const base = {
190
- kOfN: { k: 3, n: 4 },
191
- waitCeilingMs: opts.waitCeilingMs ?? 300 * 60_000,
192
- ...(opts.onStatus ? { onStatus: opts.onStatus } : {}),
193
- }
194
- const worker: EndpointCapacityGate = { name: 'z.ai-coding', probe: zaiCodingProbe(secrets, opts.model), ...base }
195
- if (kind === 'solo') return [worker]
196
- const brain: EndpointCapacityGate = { name: 'router', probe: routerProbe(secrets, opts.model), ...base }
197
- return [worker, brain]
198
- }