@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,300 +0,0 @@
1
- /**
2
- * Gen-5 activation gate: predicate validation, execution over run artifacts
3
- * (grep + script, ws/ skip with ws/.loops searched), the prefilter kill for a
4
- * missing/invalid predicate, and the quarantine verdict path (an improved
5
- * score with a never-fired mechanism is quarantined, not promoted).
6
- */
7
-
8
- import { existsSync } from 'node:fs'
9
- import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
10
- import { tmpdir } from 'node:os'
11
- import { join } from 'node:path'
12
- import type { ProposalFinding } from '@tangle-network/agent-eval'
13
- import { afterEach, beforeEach, describe, expect, it } from 'vitest'
14
- import {
15
- ACTIVATION_PREDICATE_RELPATH,
16
- activationPredicateInstruction,
17
- parseActivationPredicate,
18
- runActivationPredicate,
19
- } from './activation.mts'
20
- import { decideVerdict } from './cell-evidence.mts'
21
- import { defaultRound4Config, parseStaircaseRow, STAIRCASE_SCHEMA, type OuterLoopConfig, type StaircaseRow } from './outer-loop.mts'
22
- import { fanOutLoopsGenerator, type ProposerSpec } from './proposer-fanout.mts'
23
- import { runOk } from './proc.ts'
24
-
25
- describe('parseActivationPredicate', () => {
26
- it('accepts a valid grep predicate (with optional files filters)', () => {
27
- const parsed = parseActivationPredicate(
28
- JSON.stringify({
29
- description: 'patchRiskWarnings emitted in at least one settle',
30
- kind: 'grep',
31
- pattern: 'patchRiskWarnings|patch-risk',
32
- files: ['journal.jsonl', 'workers/'],
33
- }),
34
- )
35
- expect(parsed.ok).toBe(true)
36
- })
37
-
38
- it('accepts a valid script predicate', () => {
39
- const parsed = parseActivationPredicate(
40
- JSON.stringify({ description: 'x', kind: 'script', script: 'grep -rq X .' }),
41
- )
42
- expect(parsed.ok).toBe(true)
43
- })
44
-
45
- it.each([
46
- ['not json', 'not valid JSON'],
47
- ['[]', 'JSON object'],
48
- [JSON.stringify({ description: '', kind: 'grep', pattern: 'a' }), 'description'],
49
- [JSON.stringify({ description: 'x', kind: 'grep' }), 'pattern'],
50
- [JSON.stringify({ description: 'x', kind: 'grep', pattern: '(' }), 'RegExp'],
51
- [JSON.stringify({ description: 'x', kind: 'grep', pattern: 'a', files: [1] }), 'files'],
52
- [JSON.stringify({ description: 'x', kind: 'script' }), 'script'],
53
- [JSON.stringify({ description: 'x', kind: 'sql' }), 'kind'],
54
- ])('rejects %s', (raw, want) => {
55
- const parsed = parseActivationPredicate(raw)
56
- expect(parsed.ok).toBe(false)
57
- if (!parsed.ok) expect(parsed.error).toContain(want)
58
- })
59
-
60
- it('the prompt instruction carries the template and the quarantine warning', () => {
61
- const text = activationPredicateInstruction()
62
- expect(text).toContain(ACTIVATION_PREDICATE_RELPATH)
63
- expect(text).toContain('QUARANTINED')
64
- expect(text).toContain('"kind": "grep"')
65
- })
66
- })
67
-
68
- describe('runActivationPredicate', () => {
69
- let runA: string
70
- let runB: string
71
-
72
- beforeEach(async () => {
73
- runA = await mkdtemp(join(tmpdir(), 'act-a-'))
74
- runB = await mkdtemp(join(tmpdir(), 'act-b-'))
75
- await writeFile(join(runA, 'driver.log'), 'boot\nno mechanism here\n')
76
- await writeFile(join(runB, 'driver.log'), 'boot\npatchRiskWarnings: 2 warnings emitted\n')
77
- })
78
-
79
- afterEach(async () => {
80
- await rm(runA, { recursive: true, force: true })
81
- await rm(runB, { recursive: true, force: true })
82
- })
83
-
84
- const grep = (pattern: string, files?: string[]) => ({
85
- description: 'd',
86
- kind: 'grep' as const,
87
- pattern,
88
- ...(files ? { files } : {}),
89
- })
90
-
91
- it('fires with path:line evidence when the pattern appears in ANY run dir', async () => {
92
- const res = await runActivationPredicate(grep('patchRiskWarnings'), [runA, runB])
93
- expect(res.fired).toBe(true)
94
- expect(res.checkedRunDirs).toBe(2)
95
- expect(res.evidence[0]).toContain(join(runB, 'driver.log'))
96
- expect(res.evidence[0]).toContain('patchRiskWarnings')
97
- })
98
-
99
- it('does not fire when the pattern never appears (fail-closed)', async () => {
100
- const res = await runActivationPredicate(grep('neverEverPresent'), [runA, runB])
101
- expect(res.fired).toBe(false)
102
- expect(res.evidence).toEqual([])
103
- })
104
-
105
- it('honors files filters (relative-path substring)', async () => {
106
- await writeFile(join(runB, 'other.txt'), 'patchRiskWarnings\n')
107
- const onlyOther = await runActivationPredicate(grep('patchRiskWarnings', ['other.txt']), [runB])
108
- expect(onlyOther.fired).toBe(true)
109
- const onlyMissing = await runActivationPredicate(grep('patchRiskWarnings', ['nope.bin']), [runB])
110
- expect(onlyMissing.fired).toBe(false)
111
- })
112
-
113
- it('skips the ws/ workspace subtree EXCEPT ws/.loops (supervisor artifacts)', async () => {
114
- // Marker only inside ws/ (a checked-out repo) → must NOT count as fired.
115
- await mkdir(join(runA, 'ws', 'src'), { recursive: true })
116
- await writeFile(join(runA, 'ws', 'src', 'code.py'), 'patchRiskWarnings in the repo source\n')
117
- const inWs = await runActivationPredicate(grep('patchRiskWarnings'), [runA])
118
- expect(inWs.fired).toBe(false)
119
- // Marker in ws/.loops → the supervisor's own artifacts, searched.
120
- await mkdir(join(runA, 'ws', '.loops', 'supervisor', 's1'), { recursive: true })
121
- await writeFile(join(runA, 'ws', '.loops', 'supervisor', 's1', 'journal.jsonl'), '{"k":"patchRiskWarnings"}\n')
122
- const inLoops = await runActivationPredicate(grep('patchRiskWarnings'), [runA])
123
- expect(inLoops.fired).toBe(true)
124
- })
125
-
126
- it('script kind fires on rc=0 in any run dir, not-fired otherwise', async () => {
127
- const script = (s: string) => ({ description: 'd', kind: 'script' as const, script: s })
128
- const hit = await runActivationPredicate(script('grep -q patchRiskWarnings driver.log'), [runA, runB])
129
- expect(hit.fired).toBe(true)
130
- expect(hit.evidence[0]).toContain(runB)
131
- const miss = await runActivationPredicate(script('grep -q neverEverPresent driver.log'), [runA, runB])
132
- expect(miss.fired).toBe(false)
133
- })
134
- })
135
-
136
- describe('quarantine verdict path', () => {
137
- const base = {
138
- violations: [],
139
- coverageComplete: true,
140
- resolvedCount: 3,
141
- parentResolvedCount: 1,
142
- costRatio: 1.0,
143
- costGuardRatio: 1.2,
144
- }
145
-
146
- it('quarantines a candidate whose mechanism never fired EVEN when its score improved', () => {
147
- expect(decideVerdict({ ...base, activationFired: false })).toBe('quarantined-inactive')
148
- })
149
-
150
- it('accepts an improved candidate whose mechanism fired; gate-not-applicable behaves as before', () => {
151
- expect(decideVerdict({ ...base, activationFired: true })).toBe('accepted')
152
- expect(decideVerdict({ ...base, activationFired: null })).toBe('accepted')
153
- expect(decideVerdict(base)).toBe('accepted')
154
- })
155
-
156
- it('out-of-space still wins over quarantine; quarantine wins over no-gain/cost/coverage', () => {
157
- expect(decideVerdict({ ...base, violations: ['judge.py'], activationFired: false })).toBe('rejected-out-of-space')
158
- expect(decideVerdict({ ...base, coverageComplete: false, activationFired: false })).toBe('quarantined-inactive')
159
- expect(decideVerdict({ ...base, resolvedCount: 1, activationFired: false })).toBe('quarantined-inactive')
160
- })
161
-
162
- it('parseStaircaseRow accepts a quarantined row with gen-5 split + activation fields', () => {
163
- const row: StaircaseRow = {
164
- schema: STAIRCASE_SCHEMA,
165
- round: 4,
166
- generation: 0,
167
- runId: 'r4-x',
168
- at: new Date(0).toISOString(),
169
- candidate: 'hash',
170
- candidateCommit: 'c'.repeat(40),
171
- parent: 'baseline',
172
- parentResolvedCount: 1,
173
- changedFiles: [],
174
- changeSpaceViolations: [],
175
- perInstance: [],
176
- resolvedCount: 3,
177
- coverageComplete: true,
178
- wallS: 10,
179
- baselineWallS: 10,
180
- costRatio: 1,
181
- costGuardRatio: 1.2,
182
- internallyPromoted: true,
183
- verdict: 'quarantined-inactive',
184
- holdout: 'operator-approval-required',
185
- armProvenance: null,
186
- diffPath: null,
187
- diffSha256: null,
188
- split: {
189
- publicInstances: ['a', 'b'],
190
- privateInstances: ['c'],
191
- publicResolvedCount: 2,
192
- privateResolvedCount: 1,
193
- },
194
- activation: { present: true, description: 'd', fired: false, evidence: [], warnings: [] },
195
- }
196
- expect(parseStaircaseRow(JSON.stringify(row))).toEqual(row)
197
- })
198
- })
199
-
200
- // ---------------------------------------------------------------------------
201
- // Prefilter enforcement — a candidate without a parseable predicate is killed
202
- // before any evaluation spend.
203
- // ---------------------------------------------------------------------------
204
-
205
- describe('activation-predicate prefilter', () => {
206
- let loopsRepo: string
207
- let outDir: string
208
- let driverWt: string
209
-
210
- beforeEach(async () => {
211
- loopsRepo = await mkdtemp(join(tmpdir(), 'act-repo-'))
212
- outDir = await mkdtemp(join(tmpdir(), 'act-out-'))
213
- await runOk('git', ['init', '-q', '-b', 'main', loopsRepo])
214
- await runOk('git', ['-C', loopsRepo, 'config', 'core.hooksPath', '/dev/null'])
215
- await runOk('git', ['-C', loopsRepo, 'config', 'user.email', 't@t.dev'])
216
- await runOk('git', ['-C', loopsRepo, 'config', 'user.name', 'T'])
217
- await writeFile(join(loopsRepo, 'src.ts'), 'base\n')
218
- await runOk('git', ['-C', loopsRepo, 'add', '-A'])
219
- await runOk('git', ['-C', loopsRepo, 'commit', '-q', '-m', 'init'])
220
- driverWt = join(outDir, 'driver-wt')
221
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', driverWt, 'HEAD'])
222
- })
223
-
224
- afterEach(async () => {
225
- await rm(outDir, { recursive: true, force: true })
226
- await rm(loopsRepo, { recursive: true, force: true })
227
- })
228
-
229
- const config = (proposers: ProposerSpec[]): OuterLoopConfig => ({
230
- ...defaultRound4Config(),
231
- loopsRepo,
232
- outDir,
233
- populationSize: proposers.length,
234
- proposers,
235
- activationGate: true,
236
- })
237
-
238
- const generatorArgs = (candidateIndex: number) => ({
239
- worktreePath: driverWt,
240
- findings: [] as ProposalFinding[],
241
- maxShots: 1,
242
- signal: new AbortController().signal,
243
- generation: 0,
244
- candidateIndex,
245
- })
246
-
247
- it('kills a candidate without .improve/activation.json (stage activation-predicate) and passes one WITH it', async () => {
248
- const proposers: ProposerSpec[] = [
249
- { name: 'with-predicate', harness: 'claude-code' },
250
- { name: 'without-predicate', harness: 'claude-code' },
251
- { name: 'invalid-predicate', harness: 'claude-code' },
252
- ]
253
- const gen = fanOutLoopsGenerator(config(proposers), {
254
- author: async (proposer, args) => {
255
- await mkdir(join(args.worktreePath, 'extensions', 'pi'), { recursive: true })
256
- await writeFile(join(args.worktreePath, 'extensions', 'pi', `${proposer.name}.ts`), 'x\n')
257
- if (proposer.name === 'with-predicate') {
258
- await mkdir(join(args.worktreePath, '.improve'), { recursive: true })
259
- await writeFile(
260
- join(args.worktreePath, ACTIVATION_PREDICATE_RELPATH),
261
- JSON.stringify({ description: 'd', kind: 'grep', pattern: 'x' }),
262
- )
263
- } else if (proposer.name === 'invalid-predicate') {
264
- await mkdir(join(args.worktreePath, '.improve'), { recursive: true })
265
- await writeFile(join(args.worktreePath, ACTIVATION_PREDICATE_RELPATH), '{}')
266
- }
267
- return { applied: true, summary: `${proposer.name} edit` }
268
- },
269
- })
270
-
271
- const survivor = await gen.generate(generatorArgs(0))
272
- const missing = await gen.generate(generatorArgs(1))
273
- const invalid = await gen.generate(generatorArgs(2))
274
-
275
- expect(survivor.applied).toBe(true)
276
- expect(missing.applied).toBe(false)
277
- expect(invalid.applied).toBe(false)
278
- const kills = gen.drainPrefilterKills()
279
- expect(kills).toHaveLength(2)
280
- expect(kills.find((k) => k.proposer === 'without-predicate')).toMatchObject({ stage: 'activation-predicate' })
281
- expect(kills.find((k) => k.proposer === 'without-predicate')!.reason).toContain('missing')
282
- expect(kills.find((k) => k.proposer === 'invalid-predicate')!.reason).toContain('invalid')
283
- // The survivor's predicate landed on the driver worktree with the patch.
284
- expect(existsSync(join(driverWt, ACTIVATION_PREDICATE_RELPATH))).toBe(true)
285
- })
286
-
287
- it('does not require a predicate when the gate is off (gen-4 behavior unchanged)', async () => {
288
- const cfg = config([{ name: 'legacy', harness: 'claude-code' }])
289
- cfg.activationGate = false
290
- const gen = fanOutLoopsGenerator(cfg, {
291
- author: async (_p, args) => {
292
- await mkdir(join(args.worktreePath, 'extensions', 'pi'), { recursive: true })
293
- await writeFile(join(args.worktreePath, 'extensions', 'pi', 'l.ts'), 'x\n')
294
- return { applied: true, summary: 'edit' }
295
- },
296
- })
297
- expect((await gen.generate(generatorArgs(0))).applied).toBe(true)
298
- expect(gen.drainPrefilterKills()).toEqual([])
299
- })
300
- })
@@ -1,211 +0,0 @@
1
- /**
2
- * Typed port of the reference `fixtures/analyze.py` statistics: discordant
3
- * pair extraction, exact two-sided sign test, and cost rollups.
4
- *
5
- * The sign test is NOT re-implemented: `mcnemar` from
6
- * `@tangle-network/agent-eval` computes the identical exact two-sided
7
- * binomial p on discordant pairs (`min(1, 2·P(X ≤ min(b,c)))`, X ~
8
- * Binomial(b+c, 0.5)) — the same formula as `analyze.py::sign_test`.
9
- *
10
- * Token accounting has TWO deliberately distinct rollups:
11
- * - `supBrainTokens` — what analyze.py prints as "SUP total tokens":
12
- * supervisor-journal metered spend (the supervisor "brain"), journal-true
13
- * where captured, else the runtime's `sup_spentTokens`.
14
- * - `supTotalTokens` — brain + worker-session tokens (worker-tokens.json).
15
- * This is the true SUP-arm spend (1,722,390 for the 12-run set) and the
16
- * number the 2.55x SUP/SOLO ratio refers to. analyze.py never printed it;
17
- * its printed 0.83x ratio is brain-only.
18
- */
19
-
20
- import { mcnemar } from '@tangle-network/agent-eval'
21
- import type { ReconciledInstance } from './reconcile'
22
- import type { LedgerRow, RematchRow, WorkerTokens } from './types'
23
-
24
- export interface PairedOutcome {
25
- iid: string
26
- solo: boolean
27
- sup: boolean
28
- }
29
-
30
- /** Raw ledger outcomes — analyze.py semantics (strict `is True`), sorted by iid. */
31
- export function ledgerOutcomes(ledger: LedgerRow[]): PairedOutcome[] {
32
- return [...ledger]
33
- .sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
34
- .map((r) => ({ iid: r.iid, solo: r.solo_resolved === true, sup: r.sup_resolved === true }))
35
- }
36
-
37
- /** Reconciled (re-judged, gold-gated) outcomes, sorted by iid. */
38
- export function reconciledOutcomes(instances: ReconciledInstance[]): PairedOutcome[] {
39
- return [...instances]
40
- .sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
41
- .map((r) => ({ iid: r.iid, solo: r.solo.resolved, sup: r.sup.resolved }))
42
- }
43
-
44
- export interface DiscordantSplit {
45
- /** SUP resolved, SOLO not. */
46
- supOnly: string[]
47
- /** SOLO resolved, SUP not. */
48
- soloOnly: string[]
49
- both: string[]
50
- neither: string[]
51
- }
52
-
53
- export function splitPairs(outcomes: PairedOutcome[]): DiscordantSplit {
54
- const supOnly: string[] = []
55
- const soloOnly: string[] = []
56
- const both: string[] = []
57
- const neither: string[] = []
58
- for (const o of outcomes) {
59
- if (o.sup && !o.solo) supOnly.push(o.iid)
60
- else if (o.solo && !o.sup) soloOnly.push(o.iid)
61
- else if (o.solo && o.sup) both.push(o.iid)
62
- else neither.push(o.iid)
63
- }
64
- return { supOnly, soloOnly, both, neither }
65
- }
66
-
67
- export interface SignTestResult {
68
- n: number
69
- /** Discordant pairs where SUP (treatment) won. */
70
- b: number
71
- /** Discordant pairs where SOLO (control) won. */
72
- c: number
73
- nDiscordant: number
74
- /** Exact two-sided binomial sign-test p on discordant pairs. */
75
- pValue: number
76
- }
77
-
78
- /** Exact paired sign test, SOLO as control and SUP as treatment. */
79
- export function pairedSignTest(outcomes: PairedOutcome[]): SignTestResult {
80
- const result = mcnemar(
81
- outcomes.map((o) => o.solo),
82
- outcomes.map((o) => o.sup),
83
- )
84
- const { n, b, c, nDiscordant, pValue } = result
85
- return { n, b, c, nDiscordant, pValue }
86
- }
87
-
88
- export interface CostRollup {
89
- soloTokens: number
90
- /** analyze.py's "SUP total tokens": journal-true brain spend, else sup_spentTokens. */
91
- supBrainTokens: number
92
- /** Σ worker-session tokens (worker-tokens.json; instances without workers contribute 0). */
93
- supWorkerTokens: number
94
- /** Brain + workers — the true SUP-arm token spend. */
95
- supTotalTokens: number
96
- supUsd: number
97
- /** Blended $ per token from SUP runtime accounting (analyze.py's `rate`). */
98
- blendedRatePerTok: number
99
- soloUsdDerived: number
100
- /** supBrainTokens / soloTokens — the ratio analyze.py prints (0.83x). */
101
- brainTokenRatio: number
102
- /** supTotalTokens / soloTokens — the true-spend ratio (2.55x). */
103
- totalTokenRatio: number
104
- soloWallS: number
105
- supWallS: number
106
- wallRatio: number
107
- /** iids where the runtime's sup_spentTokens disagrees with journal-true (no-winner zeroing). */
108
- telemetryGaps: string[]
109
- }
110
-
111
- /**
112
- * Port of analyze.py's cost section over the full ledger (all 12 paired runs
113
- * — cost is a property of the runs, not of the gradeable subset), extended
114
- * with the worker-token rollup.
115
- */
116
- export function costRollup(
117
- ledger: LedgerRow[],
118
- supJournalTrue: Record<string, number | null>,
119
- workerTokens: Record<string, WorkerTokens>,
120
- ): CostRollup {
121
- const rows = [...ledger].sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
122
- let soloTokens = 0
123
- let supBrainTokens = 0
124
- let supUsd = 0
125
- let soloWallS = 0
126
- let supWallS = 0
127
- const telemetryGaps: string[] = []
128
- for (const r of rows) {
129
- soloTokens += r.solo_tokens
130
- const journalTrue = supJournalTrue[r.iid] ?? null
131
- supBrainTokens += journalTrue ?? r.sup_spentTokens ?? 0
132
- supUsd += r.sup_spentUsd ?? 0
133
- soloWallS += r.solo_wall_s
134
- supWallS += r.sup_wall_s
135
- if (journalTrue !== null && (r.sup_spentTokens ?? 0) !== journalTrue) telemetryGaps.push(r.iid)
136
- }
137
- const supWorkerTokens = Object.values(workerTokens).reduce((s, w) => s + w.worker_tok, 0)
138
- const supTotalTokens = supBrainTokens + supWorkerTokens
139
- const blendedRatePerTok = supBrainTokens > 0 ? supUsd / supBrainTokens : 0
140
- return {
141
- soloTokens,
142
- supBrainTokens,
143
- supWorkerTokens,
144
- supTotalTokens,
145
- supUsd,
146
- blendedRatePerTok,
147
- soloUsdDerived: soloTokens * blendedRatePerTok,
148
- brainTokenRatio: soloTokens > 0 ? supBrainTokens / soloTokens : 0,
149
- totalTokenRatio: soloTokens > 0 ? supTotalTokens / soloTokens : 0,
150
- soloWallS,
151
- supWallS,
152
- wallRatio: soloWallS > 0 ? supWallS / soloWallS : 0,
153
- telemetryGaps,
154
- }
155
- }
156
-
157
- /** Round label for the original head-to-head SUP run in the ledger. */
158
- export type SupRound = 'SUP' | 'SUP2' | 'SUP3' | 'SUP4'
159
-
160
- export interface RoundState {
161
- round: SupRound
162
- resolved: boolean
163
- verifyPass: boolean
164
- patchLines: number
165
- verdict: 'delivered' | 'no-winner' | 'best-effort' | null
166
- delivered: boolean | null
167
- spentTokens: number | null
168
- }
169
-
170
- /**
171
- * Supervisor evolution progression: the original SUP run (round 0, from the
172
- * ledger) followed by each rematch round, for every instance that appears in
173
- * at least one rematch file.
174
- */
175
- export function roundsProgression(
176
- ledger: LedgerRow[],
177
- rounds: RematchRow[][],
178
- ): Map<string, RoundState[]> {
179
- const byIid = new Map<string, RoundState[]>()
180
- const rematchIids = new Set(rounds.flat().map((r) => r.iid))
181
- for (const r of ledger) {
182
- if (!rematchIids.has(r.iid)) continue
183
- byIid.set(r.iid, [
184
- {
185
- round: 'SUP',
186
- resolved: r.sup_resolved === true,
187
- verifyPass: r.sup_verify_pass === true,
188
- patchLines: r.sup_patch_lines,
189
- verdict: r.sup_verdict,
190
- delivered: r.sup_delivered,
191
- spentTokens: r.sup_spentTokens,
192
- },
193
- ])
194
- }
195
- for (const round of rounds) {
196
- for (const r of round) {
197
- const states = byIid.get(r.iid)
198
- if (!states) throw new Error(`rematch row for ${r.iid} has no ledger baseline`)
199
- states.push({
200
- round: r.arm,
201
- resolved: r.resolved === true,
202
- verifyPass: r.verify_pass === true,
203
- patchLines: r.patch_lines,
204
- verdict: r.sup_verdict,
205
- delivered: r.delivered,
206
- spentTokens: r.spentTokens,
207
- })
208
- }
209
- }
210
- return byIid
211
- }