@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,270 +0,0 @@
1
- /**
2
- * Settle-time capture with LABEL v2: the contribution rule (delivered worker
3
- * vs sibling bystander vs identity gap), baseline-relative proposer rewards,
4
- * campaign-path attribution, and end-to-end line validity on a synthetic
5
- * supervisor run dir (opencode store absent → labeled gap lines).
6
- */
7
-
8
- import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
9
- import { tmpdir } from 'node:os'
10
- import { join } from 'node:path'
11
- import { afterEach, beforeEach, describe, expect, it } from 'vitest'
12
- import { readRolloutLedger } from '@tangle-network/agent-eval/rollout'
13
- import {
14
- campaignCoordsFromCellPath,
15
- createSettleCapture,
16
- deliveredWorkerLabels,
17
- PROPOSER_REWARD_SOURCE_V2,
18
- proposerRewardV2,
19
- readWorkerEvidence,
20
- WORKER_REWARD_SOURCE_V2,
21
- workerRewardV2,
22
- type CellCaptureArgs,
23
- } from './settle-capture.mts'
24
-
25
- describe('workerRewardV2 (contribution rule)', () => {
26
- it('rewards ONLY the delivering worker in a resolved cell', () => {
27
- expect(workerRewardV2({ resolved: true, isDelivered: true, identityKnown: true })).toEqual({
28
- reward: 1,
29
- bystander: false,
30
- deliveredMatch: 'delivered',
31
- })
32
- })
33
-
34
- it('marks resolved-cell siblings as bystanders with reward 0', () => {
35
- expect(workerRewardV2({ resolved: true, isDelivered: false, identityKnown: true })).toEqual({
36
- reward: 0,
37
- bystander: true,
38
- deliveredMatch: 'bystander',
39
- })
40
- })
41
-
42
- it('gives every worker 0 in an unresolved cell (no bystander flag)', () => {
43
- expect(workerRewardV2({ resolved: false, isDelivered: false, identityKnown: true })).toEqual({
44
- reward: 0,
45
- bystander: false,
46
- deliveredMatch: 'unresolved',
47
- })
48
- })
49
-
50
- it('labels an identity gap as reward null (never fabricated credit) and an inconclusive cell as null', () => {
51
- expect(workerRewardV2({ resolved: true, isDelivered: false, identityKnown: false })).toEqual({
52
- reward: null,
53
- bystander: false,
54
- deliveredMatch: 'unknown',
55
- })
56
- expect(workerRewardV2({ resolved: null, isDelivered: false, identityKnown: false }).reward).toBeNull()
57
- })
58
- })
59
-
60
- describe('deliveredWorkerLabels', () => {
61
- const workers = [
62
- { label: 'w1', cwd: '/tmp/w1', patch: 'diff --git a/x b/x\n+fix\n' },
63
- { label: 'w2', cwd: '/tmp/w2', patch: 'diff --git a/y b/y\n+other\n' },
64
- { label: 'w3', cwd: null, patch: null },
65
- ]
66
-
67
- it('matches by trimmed patch equality', () => {
68
- expect(deliveredWorkerLabels('diff --git a/x b/x\n+fix\n\n', workers)).toEqual(['w1'])
69
- })
70
-
71
- it('returns [] for an empty delivery or no match', () => {
72
- expect(deliveredWorkerLabels('', workers)).toEqual([])
73
- expect(deliveredWorkerLabels('diff --git a/z b/z\n+mystery\n', workers)).toEqual([])
74
- })
75
- })
76
-
77
- describe('proposerRewardV2 (baseline-relative)', () => {
78
- it('is positive for an improvement, negative for a regression, zero for a tie', () => {
79
- expect(proposerRewardV2(3, 1, 6)).toBeCloseTo(2 / 6)
80
- expect(proposerRewardV2(0, 1, 6)).toBeCloseTo(-1 / 6)
81
- expect(proposerRewardV2(1, 1, 6)).toBe(0)
82
- })
83
-
84
- it('rejects a zero instance count', () => {
85
- expect(() => proposerRewardV2(1, 0, 0)).toThrow(/instanceCount/)
86
- })
87
- })
88
-
89
- describe('campaignCoordsFromCellPath (directory attribution, never dispatch order)', () => {
90
- it('parses candidate and baseline cells; unknown shapes return null', () => {
91
- expect(campaignCoordsFromCellPath('/x/improve-run/gen-0/candidate-2/cell-1/arm-summary.json')).toEqual({
92
- generation: 0,
93
- candidateIndex: 2,
94
- })
95
- expect(campaignCoordsFromCellPath('/x/improve-run/baseline/cell-1/arm-summary.json')).toEqual({
96
- generation: -1,
97
- candidateIndex: -1,
98
- })
99
- expect(campaignCoordsFromCellPath('/x/somewhere/else.json')).toBeNull()
100
- })
101
- })
102
-
103
- describe('readWorkerEvidence', () => {
104
- let supRunDir: string
105
-
106
- beforeEach(async () => {
107
- supRunDir = await mkdtemp(join(tmpdir(), 'sup-'))
108
- const workers = join(supRunDir, 'workers')
109
- await mkdir(workers, { recursive: true })
110
- await writeFile(
111
- join(workers, 'w1.ndjson'),
112
- `${JSON.stringify({ kind: 'started', cwd: '/tmp/clone-w1' })}\n${JSON.stringify({ kind: 'settled' })}\n`,
113
- )
114
- await writeFile(join(workers, 'w1.patch'), 'diff --git a/x b/x\n+fix\n')
115
- await writeFile(join(workers, 'w2.ndjson'), `${JSON.stringify({ kind: 'started', cwd: '/tmp/clone-w2' })}\n`)
116
- // Inbox files must not create phantom workers.
117
- await writeFile(join(workers, 'w1.inbox.ndjson'), '{}\n')
118
- })
119
-
120
- afterEach(async () => {
121
- await rm(supRunDir, { recursive: true, force: true })
122
- })
123
-
124
- it('joins per-worker cwd + patch; a worker without a patch reads patch:null', async () => {
125
- const records = await readWorkerEvidence(supRunDir)
126
- expect(records).toEqual([
127
- { label: 'w1', cwd: '/tmp/clone-w1', patch: 'diff --git a/x b/x\n+fix\n' },
128
- { label: 'w2', cwd: '/tmp/clone-w2', patch: null },
129
- ])
130
- })
131
-
132
- it('returns [] for a run dir without worker evidence', async () => {
133
- expect(await readWorkerEvidence(join(supRunDir, 'nope'))).toEqual([])
134
- })
135
- })
136
-
137
- describe('createSettleCapture end-to-end (opencode store absent)', () => {
138
- let root: string
139
- let supRunDir: string
140
- let ledgerPath: string
141
-
142
- const cellArgs = (over: Partial<CellCaptureArgs> = {}): CellCaptureArgs => ({
143
- generation: 0,
144
- candidateIndex: 1,
145
- iid: 'pydata__xarray-4687',
146
- rep: 0,
147
- seed: 42,
148
- splitVisibility: 'public',
149
- commit: 'c'.repeat(40),
150
- resolved: true,
151
- judgeVerdict: { resolved: true, attempts: 1 },
152
- runDir: join(root, 'runs', 'pydata__xarray-4687', 'R4'),
153
- patchPath: join(root, 'delivered.patch'),
154
- supRunDir,
155
- deliveredPatch: 'diff --git a/x b/x\n+fix\n',
156
- workerModel: 'zai-coding-plan/glm-5.2',
157
- metrics: { resolved: true, verify_pass: true },
158
- cost: { usd: 1.25, wallS: 900, spentTokens: 1000 },
159
- ...over,
160
- })
161
-
162
- beforeEach(async () => {
163
- root = await mkdtemp(join(tmpdir(), 'settle-'))
164
- ledgerPath = join(root, 'rollout-ledger.jsonl')
165
- supRunDir = join(root, 'ws', '.loops', 'supervisor', 's1')
166
- const workers = join(supRunDir, 'workers')
167
- await mkdir(workers, { recursive: true })
168
- await writeFile(join(workers, 'w1.ndjson'), `${JSON.stringify({ kind: 'started', cwd: '/tmp/clone-w1' })}\n`)
169
- await writeFile(join(workers, 'w1.patch'), 'diff --git a/x b/x\n+fix\n')
170
- await writeFile(join(workers, 'w2.ndjson'), `${JSON.stringify({ kind: 'started', cwd: '/tmp/clone-w2' })}\n`)
171
- await writeFile(join(workers, 'w2.patch'), 'diff --git a/y b/y\n+other\n')
172
- })
173
-
174
- afterEach(async () => {
175
- await rm(root, { recursive: true, force: true })
176
- })
177
-
178
- const capture = () =>
179
- createSettleCapture({
180
- ledgerPath,
181
- runId: 'r4-test',
182
- instanceCount: 6,
183
- opencodeDb: join(root, 'no-such.db'),
184
- now: () => new Date('2026-07-23T00:00:00Z'),
185
- })
186
-
187
- it('emits schema-valid supervisor + worker lines with v2 labels: delivered worker 1, bystander 0', async () => {
188
- const { lines } = await capture().captureCell(cellArgs())
189
- expect(lines).toBe(3)
190
- const ledger = await readRolloutLedger(ledgerPath) // validates every line
191
- expect(ledger).toHaveLength(3)
192
-
193
- const sup = ledger.find((l) => l.role === 'supervisor')!
194
- expect(sup.outcome.reward).toBe(1)
195
- expect(sup.outcome.reward_source).toBe('swe-arena-official-judge')
196
- expect(sup.generation).toBe(0)
197
- expect(sup.candidate_index).toBe(1)
198
- expect(sup.candidate_id).toBe('gen0-cand1')
199
- // The canonical trainable split, not the legacy `train` alias.
200
- expect(sup.task.split).toBe('search')
201
- expect(sup.outcome.metrics.split_visibility).toBe('public')
202
- expect(sup.provenance.capture).toBe('settle-time')
203
-
204
- const workers = ledger.filter((l) => l.role === 'worker')
205
- expect(workers).toHaveLength(2)
206
- for (const w of workers) {
207
- expect(w.parent_rollout_id).toBe(sup.rollout_id)
208
- expect(w.candidate_id).toBe('gen0-cand1')
209
- expect(w.task.split).toBe('search')
210
- expect(w.outcome.reward_source).toBe(WORKER_REWARD_SOURCE_V2)
211
- // Store absent → labeled gap, never a silent drop.
212
- expect(w.messages).toEqual([])
213
- expect(w.provenance.gap).toBeTruthy()
214
- // No session at all, so no completed invocation to claim.
215
- expect(w.outcome.metrics.has_session).toBe(false)
216
- expect(w.outcome.is_completed).toBe(false)
217
- }
218
- const delivered = workers.find((w) => w.outcome.metrics.worker_label === 'w1')!
219
- const bystander = workers.find((w) => w.outcome.metrics.worker_label === 'w2')!
220
- expect(delivered.outcome.reward).toBe(1)
221
- expect(delivered.outcome.metrics.bystander).toBe(false)
222
- expect(delivered.outcome.metrics.delivered_match).toBe('delivered')
223
- expect(bystander.outcome.reward).toBe(0)
224
- expect(bystander.outcome.metrics.bystander).toBe(true)
225
- expect(bystander.outcome.metrics.delivered_match).toBe('bystander')
226
- })
227
-
228
- it('gives all workers 0 in an unresolved cell', async () => {
229
- await capture().captureCell(cellArgs({ resolved: false }))
230
- const workers = (await readRolloutLedger(ledgerPath)).filter((l) => l.role === 'worker')
231
- expect(workers.map((w) => w.outcome.reward)).toEqual([0, 0])
232
- expect(workers.every((w) => w.outcome.metrics.bystander === false)).toBe(true)
233
- })
234
-
235
- it('labels an unmatched delivery as an identity gap: worker rewards null', async () => {
236
- await capture().captureCell(cellArgs({ deliveredPatch: 'diff --git a/z b/z\n+mystery\n' }))
237
- const workers = (await readRolloutLedger(ledgerPath)).filter((l) => l.role === 'worker')
238
- expect(workers.map((w) => w.outcome.reward)).toEqual([null, null])
239
- expect(workers.every((w) => w.outcome.metrics.delivered_match === 'unknown')).toBe(true)
240
- })
241
-
242
- it('emits a baseline-relative proposer line (improvement positive, v2 source)', async () => {
243
- await capture().captureProposer({
244
- generation: 0,
245
- candidateIndex: 1,
246
- proposer: 'glm-author',
247
- harness: 'opencode',
248
- commit: 'c'.repeat(40),
249
- candResolved: 3,
250
- baselineResolved: 1,
251
- shotReceiptPaths: [],
252
- diffPath: null,
253
- })
254
- const [line] = await readRolloutLedger(ledgerPath)
255
- expect(line!.role).toBe('proposer')
256
- expect(line!.outcome.reward).toBeCloseTo(2 / 6)
257
- expect(line!.outcome.reward_source).toBe(PROPOSER_REWARD_SOURCE_V2)
258
- expect(line!.outcome.metrics.baseline_resolved_count).toBe(1)
259
- expect(line!.task.instance_id).toBe('gen0-cand1-glm-author')
260
- })
261
-
262
- it('appends across cells (one ledger, many flushes) and keeps every line valid', async () => {
263
- const c = capture()
264
- await c.captureCell(cellArgs())
265
- await c.captureCell(cellArgs({ rep: 1, resolved: false, splitVisibility: 'private' }))
266
- const ledger = await readRolloutLedger(ledgerPath)
267
- expect(ledger).toHaveLength(6)
268
- expect(ledger.filter((l) => l.outcome.metrics.split_visibility === 'private')).toHaveLength(3)
269
- })
270
- })
@@ -1,225 +0,0 @@
1
- /**
2
- * Gen-5 activation gate (SOTA adoption #2, GSME-style "verify the mechanism
3
- * fired before the score counts").
4
- *
5
- * Every proposer's deliverable must include a MACHINE-CHECKABLE ACTIVATION
6
- * PREDICATE at `.improve/activation.json` (inside the change-space's metadata
7
- * prefix, so the improvement driver's finalize commits it with the candidate):
8
- * a grep pattern or script over the candidate's OWN campaign run artifacts
9
- * that proves its mechanism actually fired (e.g. "the new prompt section
10
- * rendered in worker prompts", "patchRiskWarnings emitted in >=1 settle").
11
- *
12
- * Enforcement is two-stage, both fail-closed:
13
- * 1. PREFILTER — a candidate without a parseable predicate is killed before
14
- * any evaluation spend (proposer-fanout.mts, stage 'activation-predicate').
15
- * 2. POST-EVAL — the evaluator runs the predicate over the candidate's own
16
- * cell run dirs; a candidate whose mechanism NEVER fired is QUARANTINED
17
- * (staircase verdict 'quarantined-inactive': recorded, never promoted)
18
- * even when its score improved — a score with an inactive mechanism is
19
- * indistinguishable from luck or from gaming the visible instances.
20
- */
21
-
22
- import { readFile, readdir, stat } from 'node:fs/promises'
23
- import { join, relative } from 'node:path'
24
- import { run } from './proc.ts'
25
-
26
- export const ACTIVATION_PREDICATE_RELPATH = '.improve/activation.json'
27
-
28
- export interface ActivationPredicate {
29
- /** One sentence: which mechanism this proves fired. */
30
- description: string
31
- kind: 'grep' | 'script'
32
- /** kind 'grep': JS RegExp source tested line-by-line over run artifacts. */
33
- pattern?: string
34
- /** kind 'grep': optional relative-path substring filters (a file is searched
35
- * when its run-dir-relative path contains ANY entry). Empty/absent = all. */
36
- files?: string[]
37
- /** kind 'script': bash script; run once per run dir with cwd=<runDir> and
38
- * $RUN_DIR set; exit 0 in ANY run dir = mechanism fired. */
39
- script?: string
40
- }
41
-
42
- export type ParsedPredicate = { ok: true; predicate: ActivationPredicate } | { ok: false; error: string }
43
-
44
- export function parseActivationPredicate(raw: string): ParsedPredicate {
45
- let value: unknown
46
- try {
47
- value = JSON.parse(raw)
48
- } catch (cause) {
49
- return { ok: false, error: `not valid JSON: ${(cause as Error).message}` }
50
- }
51
- if (typeof value !== 'object' || value === null || Array.isArray(value)) {
52
- return { ok: false, error: 'must be a JSON object' }
53
- }
54
- const p = value as Record<string, unknown>
55
- if (typeof p.description !== 'string' || p.description.trim().length === 0) {
56
- return { ok: false, error: 'description must be a non-empty string' }
57
- }
58
- if (p.kind === 'grep') {
59
- if (typeof p.pattern !== 'string' || p.pattern.length === 0) {
60
- return { ok: false, error: 'kind "grep" requires a non-empty pattern' }
61
- }
62
- try {
63
- new RegExp(p.pattern)
64
- } catch (cause) {
65
- return { ok: false, error: `pattern is not a valid RegExp: ${(cause as Error).message}` }
66
- }
67
- if (p.files !== undefined && (!Array.isArray(p.files) || p.files.some((f) => typeof f !== 'string'))) {
68
- return { ok: false, error: 'files must be an array of strings when present' }
69
- }
70
- } else if (p.kind === 'script') {
71
- if (typeof p.script !== 'string' || p.script.trim().length === 0) {
72
- return { ok: false, error: 'kind "script" requires a non-empty script' }
73
- }
74
- } else {
75
- return { ok: false, error: `kind must be "grep" or "script", got ${JSON.stringify(p.kind)}` }
76
- }
77
- return { ok: true, predicate: p as unknown as ActivationPredicate }
78
- }
79
-
80
- /** The prompt-visible contract: template + worked example. */
81
- export function activationPredicateInstruction(): string {
82
- return [
83
- 'ACTIVATION PREDICATE (required deliverable — a candidate without one is rejected before evaluation):',
84
- `Write ${ACTIVATION_PREDICATE_RELPATH} in this worktree: a machine-checkable proof that YOUR mechanism`,
85
- "actually fired during evaluation. The evaluator runs it over your candidate's own run artifacts",
86
- '(each arm run dir: driver.log, brain.jsonl, result.json, ws/.loops/** journal + worker evidence);',
87
- 'if it never fires, your candidate is QUARANTINED even when its score improved.',
88
- '',
89
- 'Template (kind "grep" — a RegExp tested over the run artifacts):',
90
- ' {',
91
- ' "description": "patchRiskWarnings emitted in at least one settle",',
92
- ' "kind": "grep",',
93
- ' "pattern": "patchRiskWarnings|patch-risk",',
94
- ' "files": ["journal.jsonl", "workers/"]',
95
- ' }',
96
- 'Or kind "script": {"description":"...","kind":"script","script":"grep -rq NEW_SECTION ws/.loops"}',
97
- '(exit 0 in any run dir = fired).',
98
- 'Pick a pattern that can ONLY appear when your mechanism ran — not one that matches the diff itself.',
99
- ].join('\n')
100
- }
101
-
102
- // ---------------------------------------------------------------------------
103
- // Predicate execution over run dirs.
104
- // ---------------------------------------------------------------------------
105
-
106
- export interface ActivationResult {
107
- fired: boolean
108
- /** Bounded evidence: matching `path:line` refs (grep) or script stdout tails. */
109
- evidence: string[]
110
- checkedRunDirs: number
111
- checkedFiles: number
112
- /** Bounded notes on skipped inputs (oversized files, walk caps). */
113
- warnings: string[]
114
- }
115
-
116
- const MAX_FILES_PER_RUN_DIR = 4000
117
- const MAX_FILE_BYTES = 32 * 1024 * 1024
118
- const MAX_EVIDENCE = 10
119
- const SCRIPT_TIMEOUT_MS = 120_000
120
-
121
- /** Walk one run dir. The `ws/` workspace subtree (a whole checked-out repo) is
122
- * skipped EXCEPT `ws/.loops/**` — the supervisor's own artifacts live there
123
- * and are exactly where mechanism traces land. */
124
- async function walkRunDir(runDir: string, warnings: string[]): Promise<string[]> {
125
- const files: string[] = []
126
- const queue: string[] = [runDir]
127
- while (queue.length > 0) {
128
- const dir = queue.shift()!
129
- const entries = await readdir(dir, { withFileTypes: true }).catch(() => [])
130
- for (const entry of entries) {
131
- const abs = join(dir, entry.name)
132
- const rel = relative(runDir, abs)
133
- if (entry.isDirectory()) {
134
- if (rel === 'ws') {
135
- queue.push(join(abs, '.loops'))
136
- continue
137
- }
138
- queue.push(abs)
139
- } else if (entry.isFile()) {
140
- files.push(abs)
141
- if (files.length >= MAX_FILES_PER_RUN_DIR) {
142
- warnings.push(`${runDir}: file walk capped at ${MAX_FILES_PER_RUN_DIR} files`)
143
- return files
144
- }
145
- }
146
- }
147
- }
148
- return files
149
- }
150
-
151
- /** Run the predicate over the candidate's run dirs. Fail-closed: an unreadable
152
- * artifact contributes nothing (with a warning) — it can never count as
153
- * "fired". */
154
- export async function runActivationPredicate(
155
- predicate: ActivationPredicate,
156
- runDirs: readonly string[],
157
- ): Promise<ActivationResult> {
158
- const result: ActivationResult = { fired: false, evidence: [], checkedRunDirs: 0, checkedFiles: 0, warnings: [] }
159
- for (const runDir of runDirs) {
160
- result.checkedRunDirs += 1
161
- if (predicate.kind === 'script') {
162
- const res = await run('bash', ['-c', predicate.script!], {
163
- cwd: runDir,
164
- timeoutMs: SCRIPT_TIMEOUT_MS,
165
- env: { ...process.env, RUN_DIR: runDir },
166
- })
167
- if (res.code === 0) {
168
- result.fired = true
169
- if (result.evidence.length < MAX_EVIDENCE) {
170
- result.evidence.push(`${runDir}: script rc=0${res.stdout.trim() ? ` — ${res.stdout.trim().slice(0, 200)}` : ''}`)
171
- }
172
- } else if (res.timedOut) {
173
- result.warnings.push(`${runDir}: script timed out after ${SCRIPT_TIMEOUT_MS}ms (counted as not-fired)`)
174
- }
175
- continue
176
- }
177
- const regex = new RegExp(predicate.pattern!)
178
- const filters = (predicate.files ?? []).filter((f) => f.length > 0)
179
- for (const file of await walkRunDir(runDir, result.warnings)) {
180
- const rel = relative(runDir, file)
181
- if (filters.length > 0 && !filters.some((f) => rel.includes(f))) continue
182
- const info = await stat(file).catch(() => null)
183
- if (info === null) continue
184
- if (info.size > MAX_FILE_BYTES) {
185
- result.warnings.push(`${rel}: skipped (${info.size} bytes > ${MAX_FILE_BYTES})`)
186
- continue
187
- }
188
- result.checkedFiles += 1
189
- const content = await readFile(file, 'utf8').catch(() => null)
190
- if (content === null) continue
191
- const lines = content.split('\n')
192
- for (let i = 0; i < lines.length; i++) {
193
- if (regex.test(lines[i]!)) {
194
- result.fired = true
195
- if (result.evidence.length < MAX_EVIDENCE) {
196
- result.evidence.push(`${file}:${i + 1}: ${lines[i]!.trim().slice(0, 200)}`)
197
- }
198
- break // one match per file is enough evidence
199
- }
200
- }
201
- }
202
- }
203
- return result
204
- }
205
-
206
- /** Read + parse the predicate committed at a candidate's loops commit. */
207
- export async function readCommittedPredicate(
208
- loopsRepo: string,
209
- commit: string,
210
- ): Promise<{ raw: string; parsed: ParsedPredicate } | null> {
211
- const show = await run('git', ['-C', loopsRepo, 'show', `${commit}:${ACTIVATION_PREDICATE_RELPATH}`])
212
- if (show.code !== 0) return null
213
- return { raw: show.stdout, parsed: parseActivationPredicate(show.stdout) }
214
- }
215
-
216
- /** The staircase row's activation record. */
217
- export interface ActivationRecord {
218
- /** Whether a parseable predicate was present on the candidate commit. */
219
- present: boolean
220
- description: string | null
221
- /** null = not evaluated (no predicate / gate disabled / baseline). */
222
- fired: boolean | null
223
- evidence: string[]
224
- warnings: string[]
225
- }