@tangle-network/agent-bench 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,300 +0,0 @@
1
- /**
2
- * Gen-5 activation gate: predicate validation, execution over run artifacts
3
- * (grep + script, ws/ skip with ws/.loops searched), the prefilter kill for a
4
- * missing/invalid predicate, and the quarantine verdict path (an improved
5
- * score with a never-fired mechanism is quarantined, not promoted).
6
- */
7
-
8
- import { existsSync } from 'node:fs'
9
- import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
10
- import { tmpdir } from 'node:os'
11
- import { join } from 'node:path'
12
- import type { ProposalFinding } from '@tangle-network/agent-eval'
13
- import { afterEach, beforeEach, describe, expect, it } from 'vitest'
14
- import {
15
- ACTIVATION_PREDICATE_RELPATH,
16
- activationPredicateInstruction,
17
- parseActivationPredicate,
18
- runActivationPredicate,
19
- } from './activation.mts'
20
- import { decideVerdict } from './cell-evidence.mts'
21
- import { defaultRound4Config, parseStaircaseRow, STAIRCASE_SCHEMA, type OuterLoopConfig, type StaircaseRow } from './outer-loop.mts'
22
- import { fanOutLoopsGenerator, type ProposerSpec } from './proposer-fanout.mts'
23
- import { runOk } from './proc.ts'
24
-
25
- describe('parseActivationPredicate', () => {
26
- it('accepts a valid grep predicate (with optional files filters)', () => {
27
- const parsed = parseActivationPredicate(
28
- JSON.stringify({
29
- description: 'patchRiskWarnings emitted in at least one settle',
30
- kind: 'grep',
31
- pattern: 'patchRiskWarnings|patch-risk',
32
- files: ['journal.jsonl', 'workers/'],
33
- }),
34
- )
35
- expect(parsed.ok).toBe(true)
36
- })
37
-
38
- it('accepts a valid script predicate', () => {
39
- const parsed = parseActivationPredicate(
40
- JSON.stringify({ description: 'x', kind: 'script', script: 'grep -rq X .' }),
41
- )
42
- expect(parsed.ok).toBe(true)
43
- })
44
-
45
- it.each([
46
- ['not json', 'not valid JSON'],
47
- ['[]', 'JSON object'],
48
- [JSON.stringify({ description: '', kind: 'grep', pattern: 'a' }), 'description'],
49
- [JSON.stringify({ description: 'x', kind: 'grep' }), 'pattern'],
50
- [JSON.stringify({ description: 'x', kind: 'grep', pattern: '(' }), 'RegExp'],
51
- [JSON.stringify({ description: 'x', kind: 'grep', pattern: 'a', files: [1] }), 'files'],
52
- [JSON.stringify({ description: 'x', kind: 'script' }), 'script'],
53
- [JSON.stringify({ description: 'x', kind: 'sql' }), 'kind'],
54
- ])('rejects %s', (raw, want) => {
55
- const parsed = parseActivationPredicate(raw)
56
- expect(parsed.ok).toBe(false)
57
- if (!parsed.ok) expect(parsed.error).toContain(want)
58
- })
59
-
60
- it('the prompt instruction carries the template and the quarantine warning', () => {
61
- const text = activationPredicateInstruction()
62
- expect(text).toContain(ACTIVATION_PREDICATE_RELPATH)
63
- expect(text).toContain('QUARANTINED')
64
- expect(text).toContain('"kind": "grep"')
65
- })
66
- })
67
-
68
- describe('runActivationPredicate', () => {
69
- let runA: string
70
- let runB: string
71
-
72
- beforeEach(async () => {
73
- runA = await mkdtemp(join(tmpdir(), 'act-a-'))
74
- runB = await mkdtemp(join(tmpdir(), 'act-b-'))
75
- await writeFile(join(runA, 'driver.log'), 'boot\nno mechanism here\n')
76
- await writeFile(join(runB, 'driver.log'), 'boot\npatchRiskWarnings: 2 warnings emitted\n')
77
- })
78
-
79
- afterEach(async () => {
80
- await rm(runA, { recursive: true, force: true })
81
- await rm(runB, { recursive: true, force: true })
82
- })
83
-
84
- const grep = (pattern: string, files?: string[]) => ({
85
- description: 'd',
86
- kind: 'grep' as const,
87
- pattern,
88
- ...(files ? { files } : {}),
89
- })
90
-
91
- it('fires with path:line evidence when the pattern appears in ANY run dir', async () => {
92
- const res = await runActivationPredicate(grep('patchRiskWarnings'), [runA, runB])
93
- expect(res.fired).toBe(true)
94
- expect(res.checkedRunDirs).toBe(2)
95
- expect(res.evidence[0]).toContain(join(runB, 'driver.log'))
96
- expect(res.evidence[0]).toContain('patchRiskWarnings')
97
- })
98
-
99
- it('does not fire when the pattern never appears (fail-closed)', async () => {
100
- const res = await runActivationPredicate(grep('neverEverPresent'), [runA, runB])
101
- expect(res.fired).toBe(false)
102
- expect(res.evidence).toEqual([])
103
- })
104
-
105
- it('honors files filters (relative-path substring)', async () => {
106
- await writeFile(join(runB, 'other.txt'), 'patchRiskWarnings\n')
107
- const onlyOther = await runActivationPredicate(grep('patchRiskWarnings', ['other.txt']), [runB])
108
- expect(onlyOther.fired).toBe(true)
109
- const onlyMissing = await runActivationPredicate(grep('patchRiskWarnings', ['nope.bin']), [runB])
110
- expect(onlyMissing.fired).toBe(false)
111
- })
112
-
113
- it('skips the ws/ workspace subtree EXCEPT ws/.loops (supervisor artifacts)', async () => {
114
- // Marker only inside ws/ (a checked-out repo) → must NOT count as fired.
115
- await mkdir(join(runA, 'ws', 'src'), { recursive: true })
116
- await writeFile(join(runA, 'ws', 'src', 'code.py'), 'patchRiskWarnings in the repo source\n')
117
- const inWs = await runActivationPredicate(grep('patchRiskWarnings'), [runA])
118
- expect(inWs.fired).toBe(false)
119
- // Marker in ws/.loops → the supervisor's own artifacts, searched.
120
- await mkdir(join(runA, 'ws', '.loops', 'supervisor', 's1'), { recursive: true })
121
- await writeFile(join(runA, 'ws', '.loops', 'supervisor', 's1', 'journal.jsonl'), '{"k":"patchRiskWarnings"}\n')
122
- const inLoops = await runActivationPredicate(grep('patchRiskWarnings'), [runA])
123
- expect(inLoops.fired).toBe(true)
124
- })
125
-
126
- it('script kind fires on rc=0 in any run dir, not-fired otherwise', async () => {
127
- const script = (s: string) => ({ description: 'd', kind: 'script' as const, script: s })
128
- const hit = await runActivationPredicate(script('grep -q patchRiskWarnings driver.log'), [runA, runB])
129
- expect(hit.fired).toBe(true)
130
- expect(hit.evidence[0]).toContain(runB)
131
- const miss = await runActivationPredicate(script('grep -q neverEverPresent driver.log'), [runA, runB])
132
- expect(miss.fired).toBe(false)
133
- })
134
- })
135
-
136
- describe('quarantine verdict path', () => {
137
- const base = {
138
- violations: [],
139
- coverageComplete: true,
140
- resolvedCount: 3,
141
- parentResolvedCount: 1,
142
- costRatio: 1.0,
143
- costGuardRatio: 1.2,
144
- }
145
-
146
- it('quarantines a candidate whose mechanism never fired EVEN when its score improved', () => {
147
- expect(decideVerdict({ ...base, activationFired: false })).toBe('quarantined-inactive')
148
- })
149
-
150
- it('accepts an improved candidate whose mechanism fired; gate-not-applicable behaves as before', () => {
151
- expect(decideVerdict({ ...base, activationFired: true })).toBe('accepted')
152
- expect(decideVerdict({ ...base, activationFired: null })).toBe('accepted')
153
- expect(decideVerdict(base)).toBe('accepted')
154
- })
155
-
156
- it('out-of-space still wins over quarantine; quarantine wins over no-gain/cost/coverage', () => {
157
- expect(decideVerdict({ ...base, violations: ['judge.py'], activationFired: false })).toBe('rejected-out-of-space')
158
- expect(decideVerdict({ ...base, coverageComplete: false, activationFired: false })).toBe('quarantined-inactive')
159
- expect(decideVerdict({ ...base, resolvedCount: 1, activationFired: false })).toBe('quarantined-inactive')
160
- })
161
-
162
- it('parseStaircaseRow accepts a quarantined row with gen-5 split + activation fields', () => {
163
- const row: StaircaseRow = {
164
- schema: STAIRCASE_SCHEMA,
165
- round: 4,
166
- generation: 0,
167
- runId: 'r4-x',
168
- at: new Date(0).toISOString(),
169
- candidate: 'hash',
170
- candidateCommit: 'c'.repeat(40),
171
- parent: 'baseline',
172
- parentResolvedCount: 1,
173
- changedFiles: [],
174
- changeSpaceViolations: [],
175
- perInstance: [],
176
- resolvedCount: 3,
177
- coverageComplete: true,
178
- wallS: 10,
179
- baselineWallS: 10,
180
- costRatio: 1,
181
- costGuardRatio: 1.2,
182
- internallyPromoted: true,
183
- verdict: 'quarantined-inactive',
184
- holdout: 'operator-approval-required',
185
- armProvenance: null,
186
- diffPath: null,
187
- diffSha256: null,
188
- split: {
189
- publicInstances: ['a', 'b'],
190
- privateInstances: ['c'],
191
- publicResolvedCount: 2,
192
- privateResolvedCount: 1,
193
- },
194
- activation: { present: true, description: 'd', fired: false, evidence: [], warnings: [] },
195
- }
196
- expect(parseStaircaseRow(JSON.stringify(row))).toEqual(row)
197
- })
198
- })
199
-
200
- // ---------------------------------------------------------------------------
201
- // Prefilter enforcement — a candidate without a parseable predicate is killed
202
- // before any evaluation spend.
203
- // ---------------------------------------------------------------------------
204
-
205
- describe('activation-predicate prefilter', () => {
206
- let loopsRepo: string
207
- let outDir: string
208
- let driverWt: string
209
-
210
- beforeEach(async () => {
211
- loopsRepo = await mkdtemp(join(tmpdir(), 'act-repo-'))
212
- outDir = await mkdtemp(join(tmpdir(), 'act-out-'))
213
- await runOk('git', ['init', '-q', '-b', 'main', loopsRepo])
214
- await runOk('git', ['-C', loopsRepo, 'config', 'core.hooksPath', '/dev/null'])
215
- await runOk('git', ['-C', loopsRepo, 'config', 'user.email', 't@t.dev'])
216
- await runOk('git', ['-C', loopsRepo, 'config', 'user.name', 'T'])
217
- await writeFile(join(loopsRepo, 'src.ts'), 'base\n')
218
- await runOk('git', ['-C', loopsRepo, 'add', '-A'])
219
- await runOk('git', ['-C', loopsRepo, 'commit', '-q', '-m', 'init'])
220
- driverWt = join(outDir, 'driver-wt')
221
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', driverWt, 'HEAD'])
222
- })
223
-
224
- afterEach(async () => {
225
- await rm(outDir, { recursive: true, force: true })
226
- await rm(loopsRepo, { recursive: true, force: true })
227
- })
228
-
229
- const config = (proposers: ProposerSpec[]): OuterLoopConfig => ({
230
- ...defaultRound4Config(),
231
- loopsRepo,
232
- outDir,
233
- populationSize: proposers.length,
234
- proposers,
235
- activationGate: true,
236
- })
237
-
238
- const generatorArgs = (candidateIndex: number) => ({
239
- worktreePath: driverWt,
240
- findings: [] as ProposalFinding[],
241
- maxShots: 1,
242
- signal: new AbortController().signal,
243
- generation: 0,
244
- candidateIndex,
245
- })
246
-
247
- it('kills a candidate without .improve/activation.json (stage activation-predicate) and passes one WITH it', async () => {
248
- const proposers: ProposerSpec[] = [
249
- { name: 'with-predicate', harness: 'claude-code' },
250
- { name: 'without-predicate', harness: 'claude-code' },
251
- { name: 'invalid-predicate', harness: 'claude-code' },
252
- ]
253
- const gen = fanOutLoopsGenerator(config(proposers), {
254
- author: async (proposer, args) => {
255
- await mkdir(join(args.worktreePath, 'extensions', 'pi'), { recursive: true })
256
- await writeFile(join(args.worktreePath, 'extensions', 'pi', `${proposer.name}.ts`), 'x\n')
257
- if (proposer.name === 'with-predicate') {
258
- await mkdir(join(args.worktreePath, '.improve'), { recursive: true })
259
- await writeFile(
260
- join(args.worktreePath, ACTIVATION_PREDICATE_RELPATH),
261
- JSON.stringify({ description: 'd', kind: 'grep', pattern: 'x' }),
262
- )
263
- } else if (proposer.name === 'invalid-predicate') {
264
- await mkdir(join(args.worktreePath, '.improve'), { recursive: true })
265
- await writeFile(join(args.worktreePath, ACTIVATION_PREDICATE_RELPATH), '{}')
266
- }
267
- return { applied: true, summary: `${proposer.name} edit` }
268
- },
269
- })
270
-
271
- const survivor = await gen.generate(generatorArgs(0))
272
- const missing = await gen.generate(generatorArgs(1))
273
- const invalid = await gen.generate(generatorArgs(2))
274
-
275
- expect(survivor.applied).toBe(true)
276
- expect(missing.applied).toBe(false)
277
- expect(invalid.applied).toBe(false)
278
- const kills = gen.drainPrefilterKills()
279
- expect(kills).toHaveLength(2)
280
- expect(kills.find((k) => k.proposer === 'without-predicate')).toMatchObject({ stage: 'activation-predicate' })
281
- expect(kills.find((k) => k.proposer === 'without-predicate')!.reason).toContain('missing')
282
- expect(kills.find((k) => k.proposer === 'invalid-predicate')!.reason).toContain('invalid')
283
- // The survivor's predicate landed on the driver worktree with the patch.
284
- expect(existsSync(join(driverWt, ACTIVATION_PREDICATE_RELPATH))).toBe(true)
285
- })
286
-
287
- it('does not require a predicate when the gate is off (gen-4 behavior unchanged)', async () => {
288
- const cfg = config([{ name: 'legacy', harness: 'claude-code' }])
289
- cfg.activationGate = false
290
- const gen = fanOutLoopsGenerator(cfg, {
291
- author: async (_p, args) => {
292
- await mkdir(join(args.worktreePath, 'extensions', 'pi'), { recursive: true })
293
- await writeFile(join(args.worktreePath, 'extensions', 'pi', 'l.ts'), 'x\n')
294
- return { applied: true, summary: 'edit' }
295
- },
296
- })
297
- expect((await gen.generate(generatorArgs(0))).applied).toBe(true)
298
- expect(gen.drainPrefilterKills()).toEqual([])
299
- })
300
- })
@@ -1,211 +0,0 @@
1
- /**
2
- * Typed port of the reference `fixtures/analyze.py` statistics: discordant
3
- * pair extraction, exact two-sided sign test, and cost rollups.
4
- *
5
- * The sign test is NOT re-implemented: `mcnemar` from
6
- * `@tangle-network/agent-eval` computes the identical exact two-sided
7
- * binomial p on discordant pairs (`min(1, 2·P(X ≤ min(b,c)))`, X ~
8
- * Binomial(b+c, 0.5)) — the same formula as `analyze.py::sign_test`.
9
- *
10
- * Token accounting has TWO deliberately distinct rollups:
11
- * - `supBrainTokens` — what analyze.py prints as "SUP total tokens":
12
- * supervisor-journal metered spend (the supervisor "brain"), journal-true
13
- * where captured, else the runtime's `sup_spentTokens`.
14
- * - `supTotalTokens` — brain + worker-session tokens (worker-tokens.json).
15
- * This is the true SUP-arm spend (1,722,390 for the 12-run set) and the
16
- * number the 2.55x SUP/SOLO ratio refers to. analyze.py never printed it;
17
- * its printed 0.83x ratio is brain-only.
18
- */
19
-
20
- import { mcnemar } from '@tangle-network/agent-eval'
21
- import type { ReconciledInstance } from './reconcile'
22
- import type { LedgerRow, RematchRow, WorkerTokens } from './types'
23
-
24
- export interface PairedOutcome {
25
- iid: string
26
- solo: boolean
27
- sup: boolean
28
- }
29
-
30
- /** Raw ledger outcomes — analyze.py semantics (strict `is True`), sorted by iid. */
31
- export function ledgerOutcomes(ledger: LedgerRow[]): PairedOutcome[] {
32
- return [...ledger]
33
- .sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
34
- .map((r) => ({ iid: r.iid, solo: r.solo_resolved === true, sup: r.sup_resolved === true }))
35
- }
36
-
37
- /** Reconciled (re-judged, gold-gated) outcomes, sorted by iid. */
38
- export function reconciledOutcomes(instances: ReconciledInstance[]): PairedOutcome[] {
39
- return [...instances]
40
- .sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
41
- .map((r) => ({ iid: r.iid, solo: r.solo.resolved, sup: r.sup.resolved }))
42
- }
43
-
44
- export interface DiscordantSplit {
45
- /** SUP resolved, SOLO not. */
46
- supOnly: string[]
47
- /** SOLO resolved, SUP not. */
48
- soloOnly: string[]
49
- both: string[]
50
- neither: string[]
51
- }
52
-
53
- export function splitPairs(outcomes: PairedOutcome[]): DiscordantSplit {
54
- const supOnly: string[] = []
55
- const soloOnly: string[] = []
56
- const both: string[] = []
57
- const neither: string[] = []
58
- for (const o of outcomes) {
59
- if (o.sup && !o.solo) supOnly.push(o.iid)
60
- else if (o.solo && !o.sup) soloOnly.push(o.iid)
61
- else if (o.solo && o.sup) both.push(o.iid)
62
- else neither.push(o.iid)
63
- }
64
- return { supOnly, soloOnly, both, neither }
65
- }
66
-
67
- export interface SignTestResult {
68
- n: number
69
- /** Discordant pairs where SUP (treatment) won. */
70
- b: number
71
- /** Discordant pairs where SOLO (control) won. */
72
- c: number
73
- nDiscordant: number
74
- /** Exact two-sided binomial sign-test p on discordant pairs. */
75
- pValue: number
76
- }
77
-
78
- /** Exact paired sign test, SOLO as control and SUP as treatment. */
79
- export function pairedSignTest(outcomes: PairedOutcome[]): SignTestResult {
80
- const result = mcnemar(
81
- outcomes.map((o) => o.solo),
82
- outcomes.map((o) => o.sup),
83
- )
84
- const { n, b, c, nDiscordant, pValue } = result
85
- return { n, b, c, nDiscordant, pValue }
86
- }
87
-
88
- export interface CostRollup {
89
- soloTokens: number
90
- /** analyze.py's "SUP total tokens": journal-true brain spend, else sup_spentTokens. */
91
- supBrainTokens: number
92
- /** Σ worker-session tokens (worker-tokens.json; instances without workers contribute 0). */
93
- supWorkerTokens: number
94
- /** Brain + workers — the true SUP-arm token spend. */
95
- supTotalTokens: number
96
- supUsd: number
97
- /** Blended $ per token from SUP runtime accounting (analyze.py's `rate`). */
98
- blendedRatePerTok: number
99
- soloUsdDerived: number
100
- /** supBrainTokens / soloTokens — the ratio analyze.py prints (0.83x). */
101
- brainTokenRatio: number
102
- /** supTotalTokens / soloTokens — the true-spend ratio (2.55x). */
103
- totalTokenRatio: number
104
- soloWallS: number
105
- supWallS: number
106
- wallRatio: number
107
- /** iids where the runtime's sup_spentTokens disagrees with journal-true (no-winner zeroing). */
108
- telemetryGaps: string[]
109
- }
110
-
111
- /**
112
- * Port of analyze.py's cost section over the full ledger (all 12 paired runs
113
- * — cost is a property of the runs, not of the gradeable subset), extended
114
- * with the worker-token rollup.
115
- */
116
- export function costRollup(
117
- ledger: LedgerRow[],
118
- supJournalTrue: Record<string, number | null>,
119
- workerTokens: Record<string, WorkerTokens>,
120
- ): CostRollup {
121
- const rows = [...ledger].sort((a, b) => (a.iid < b.iid ? -1 : a.iid > b.iid ? 1 : 0))
122
- let soloTokens = 0
123
- let supBrainTokens = 0
124
- let supUsd = 0
125
- let soloWallS = 0
126
- let supWallS = 0
127
- const telemetryGaps: string[] = []
128
- for (const r of rows) {
129
- soloTokens += r.solo_tokens
130
- const journalTrue = supJournalTrue[r.iid] ?? null
131
- supBrainTokens += journalTrue ?? r.sup_spentTokens ?? 0
132
- supUsd += r.sup_spentUsd ?? 0
133
- soloWallS += r.solo_wall_s
134
- supWallS += r.sup_wall_s
135
- if (journalTrue !== null && (r.sup_spentTokens ?? 0) !== journalTrue) telemetryGaps.push(r.iid)
136
- }
137
- const supWorkerTokens = Object.values(workerTokens).reduce((s, w) => s + w.worker_tok, 0)
138
- const supTotalTokens = supBrainTokens + supWorkerTokens
139
- const blendedRatePerTok = supBrainTokens > 0 ? supUsd / supBrainTokens : 0
140
- return {
141
- soloTokens,
142
- supBrainTokens,
143
- supWorkerTokens,
144
- supTotalTokens,
145
- supUsd,
146
- blendedRatePerTok,
147
- soloUsdDerived: soloTokens * blendedRatePerTok,
148
- brainTokenRatio: soloTokens > 0 ? supBrainTokens / soloTokens : 0,
149
- totalTokenRatio: soloTokens > 0 ? supTotalTokens / soloTokens : 0,
150
- soloWallS,
151
- supWallS,
152
- wallRatio: soloWallS > 0 ? supWallS / soloWallS : 0,
153
- telemetryGaps,
154
- }
155
- }
156
-
157
- /** Round label for the original head-to-head SUP run in the ledger. */
158
- export type SupRound = 'SUP' | 'SUP2' | 'SUP3' | 'SUP4'
159
-
160
- export interface RoundState {
161
- round: SupRound
162
- resolved: boolean
163
- verifyPass: boolean
164
- patchLines: number
165
- verdict: 'delivered' | 'no-winner' | 'best-effort' | null
166
- delivered: boolean | null
167
- spentTokens: number | null
168
- }
169
-
170
- /**
171
- * Supervisor evolution progression: the original SUP run (round 0, from the
172
- * ledger) followed by each rematch round, for every instance that appears in
173
- * at least one rematch file.
174
- */
175
- export function roundsProgression(
176
- ledger: LedgerRow[],
177
- rounds: RematchRow[][],
178
- ): Map<string, RoundState[]> {
179
- const byIid = new Map<string, RoundState[]>()
180
- const rematchIids = new Set(rounds.flat().map((r) => r.iid))
181
- for (const r of ledger) {
182
- if (!rematchIids.has(r.iid)) continue
183
- byIid.set(r.iid, [
184
- {
185
- round: 'SUP',
186
- resolved: r.sup_resolved === true,
187
- verifyPass: r.sup_verify_pass === true,
188
- patchLines: r.sup_patch_lines,
189
- verdict: r.sup_verdict,
190
- delivered: r.sup_delivered,
191
- spentTokens: r.sup_spentTokens,
192
- },
193
- ])
194
- }
195
- for (const round of rounds) {
196
- for (const r of round) {
197
- const states = byIid.get(r.iid)
198
- if (!states) throw new Error(`rematch row for ${r.iid} has no ledger baseline`)
199
- states.push({
200
- round: r.arm,
201
- resolved: r.resolved === true,
202
- verifyPass: r.verify_pass === true,
203
- patchLines: r.patch_lines,
204
- verdict: r.sup_verdict,
205
- delivered: r.delivered,
206
- spentTokens: r.spentTokens,
207
- })
208
- }
209
- }
210
- return byIid
211
- }