@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,1136 +0,0 @@
1
- import { spawnSync } from 'node:child_process'
2
- import { existsSync } from 'node:fs'
3
- import { mkdir, mkdtemp, readFile, readdir, rm, writeFile } from 'node:fs/promises'
4
- import { createServer } from 'node:http'
5
- import { tmpdir } from 'node:os'
6
- import { join } from 'node:path'
7
- import {
8
- gepaOptimizationMethod,
9
- createRunCostLedger,
10
- fsCampaignStorage,
11
- type DispatchContext,
12
- type OptimizationMethodProvenance,
13
- } from '@tangle-network/agent-eval/campaign'
14
- import { afterEach, beforeEach, describe, expect, it } from 'vitest'
15
- import { ACTIVATION_PREDICATE_RELPATH, parseActivationPredicate } from './activation.mts'
16
- import {
17
- DEFAULT_GEPA_PYTHON,
18
- DEFAULT_MAX_METRIC_CALLS,
19
- GEPA_INNER_RUNS_DIRNAME,
20
- GEPA_PYTHON_INSTALL_HINT,
21
- gepaBridgeScenarios,
22
- gepaSeatEvaluationId,
23
- innerSmokeComposite,
24
- innerSmokeJudge,
25
- isGepaSeat,
26
- mechanicalActivationPredicate,
27
- probeGepaRuntime,
28
- recipeEvaluationBudget,
29
- recipeForSeat,
30
- recordGepaSeatInnerRun,
31
- validateGepaSeat,
32
- type GepaMethodFactory,
33
- type GepaSeatInnerRun,
34
- type GepaSeatSpec,
35
- type ProbeExec,
36
- } from './gepa-seat.mts'
37
- import { defaultRound4Config, type OuterLoopConfig } from './outer-loop.mts'
38
- import { fanOutLoopsGenerator, type ProposerSpec, type SmokeRunner, type SmokeVerdict } from './proposer-fanout.mts'
39
- import { captureProposerProvenance } from './proposer-provenance.mts'
40
- import { officialOptimizerModel } from '../official-optimizer-config.mts'
41
- import { runOk } from './proc.ts'
42
-
43
- const SURFACE = 'extensions/pi/prompts/worker-coding-system.md'
44
- const IMPLEMENTATION_REFS = {
45
- runnerImplementationRef: `sha256:${'a'.repeat(64)}`,
46
- judgeImplementationRef: `sha256:${'b'.repeat(64)}`,
47
- } as const
48
-
49
- const seat = (over: Partial<ProposerSpec> = {}): ProposerSpec => ({
50
- name: 'gepa-author',
51
- engine: 'gepa',
52
- surface: SURFACE,
53
- ...over,
54
- })
55
-
56
- // ---------------------------------------------------------------------------
57
- // Spec validation.
58
- // ---------------------------------------------------------------------------
59
-
60
- describe('validateGepaSeat', () => {
61
- it('accepts an engine seat and isGepaSeat discriminates on engine', () => {
62
- expect(() => validateGepaSeat(seat())).not.toThrow()
63
- expect(() => validateGepaSeat(seat({ engine: 'omni', maxMetricCalls: 8 }))).not.toThrow()
64
- expect(isGepaSeat(seat())).toBe(true)
65
- expect(isGepaSeat({ name: 'x', harness: 'claude-code' })).toBe(false)
66
- })
67
-
68
- it('requires a surface inside the declared change-space', () => {
69
- expect(() => validateGepaSeat(seat({ surface: undefined }))).toThrow(/surface is required/)
70
- expect(() => validateGepaSeat(seat({ surface: 'judge.py' }))).toThrow(/outside the declared change-space/)
71
- expect(() => validateGepaSeat(seat({ surface: '../escape.md' }))).toThrow(/outside the declared change-space/)
72
- })
73
-
74
- it('rejects harness-seat fields on an engine seat instead of silently ignoring them', () => {
75
- expect(() => validateGepaSeat(seat({ harness: 'claude-code' }))).toThrow(/'harness' belongs to harness-authored/)
76
- expect(() => validateGepaSeat(seat({ merge: true }))).toThrow(/'merge'/)
77
- expect(() => validateGepaSeat(seat({ model: 'x' }))).toThrow(/'model'/)
78
- expect(() => validateGepaSeat(seat({ profile: 'p.json' }))).toThrow(/'profile'/)
79
- })
80
-
81
- it('bounds the budget: positive integer calls, omni needs >= 4, positive cost cap', () => {
82
- expect(() => validateGepaSeat(seat({ maxMetricCalls: 0 }))).toThrow(/maxMetricCalls/)
83
- expect(() => validateGepaSeat(seat({ maxMetricCalls: 2.5 }))).toThrow(/maxMetricCalls/)
84
- expect(() => validateGepaSeat(seat({ engine: 'omni', maxMetricCalls: 3 }))).toThrow(/omni.*needs maxMetricCalls >= 4/)
85
- expect(() => validateGepaSeat(seat({ maxProposerCostUsd: 0 }))).toThrow(/maxProposerCostUsd/)
86
- expect(() => validateGepaSeat(seat({ engine: 'nope' as never }))).toThrow(/engine must be one of/)
87
- })
88
- })
89
-
90
- // ---------------------------------------------------------------------------
91
- // Recipe / budget cap.
92
- // ---------------------------------------------------------------------------
93
-
94
- describe('recipeForSeat', () => {
95
- it.each(['gepa', 'omni'] as const)('transports the real %s optimizer configuration', async (engine) => {
96
- const runDir = await mkdtemp(join(tmpdir(), 'gepa-seat-transport-'))
97
- const requests: unknown[] = []
98
- try {
99
- const optimizer = officialOptimizerModel({
100
- env: { OPT_INPUT_USD_PER_MILLION: '1', OPT_CACHED_INPUT_USD_PER_MILLION: '1', OPT_CACHE_WRITE_USD_PER_MILLION: '1', OPT_OUTPUT_USD_PER_MILLION: '1' },
101
- envPrefix: 'OPT', model: 'fixture-model', baseUrl: 'http://127.0.0.1:1/v1', apiKey: 'fixture-key',
102
- maxCostUsd: 1, maxOutputTokensPerRequest: 100, anthropicEndpoint: engine === 'omni',
103
- complete: async (request) => {
104
- requests.push(request)
105
- return { model: 'fixture-model', choices: [{ message: { content: 'ok' }, finish_reason: 'stop' }], usage: { prompt_tokens: 2, completion_tokens: 1, cost: 0.000003 } }
106
- },
107
- })
108
- const spec = seat({ engine })
109
- validateGepaSeat(spec)
110
- const runtime = {
111
- python: { implementation: 'CPython', version: '3.12.0' },
112
- bridge: { package: 'agent-eval-rpc', version: 'fixture', sourceUrl: 'https://github.com/tangle-network/agent-eval', revision: 'fixture', sourceSha256: 'a'.repeat(64) },
113
- optimizer: { package: 'gepa', version: 'fixture', sourceUrl: 'https://github.com/gepa-ai/gepa', revision: 'fixture', sourceSha256: 'b'.repeat(64) },
114
- engineModules: [],
115
- }
116
- // Only the child optimizer is substituted; the real Eval proxy calls Runtime.
117
- const bridge = join(runDir, 'transport.mjs')
118
- await writeFile(bridge, `
119
- import fs from 'node:fs'
120
- const input = JSON.parse(fs.readFileSync(process.argv[process.argv.indexOf('--input') + 1], 'utf8'))
121
- if (input.operation === 'inspect') {
122
- fs.writeFileSync(process.argv[process.argv.indexOf('--output') + 1], JSON.stringify({ runtime: ${JSON.stringify(runtime)} }))
123
- } else {
124
- const omni = input.recipe.kind === 'omni'
125
- if (omni && input.recipe.explore.some(run => run.engine !== 'gepa' && run.engineConfig.model !== input.modelProxy.model)) throw new Error('wrong engine model')
126
- const send = maxTokens => fetch(input.modelProxy.baseUrl + (omni ? '/messages' : '/chat/completions'), {
127
- method: 'POST',
128
- headers: { 'content-type': 'application/json', authorization: 'Bearer ' + input.modelProxy.apiKey, 'x-api-key': input.modelProxy.apiKey, 'anthropic-version': '2023-06-01' },
129
- body: JSON.stringify({ model: input.modelProxy.model, max_tokens: maxTokens, messages: [{ role: 'user', content: 'transport proof' }] }),
130
- })
131
- const overBudget = await send(101)
132
- if (overBudget.ok) throw new Error('output ceiling was not enforced')
133
- await overBudget.text()
134
- const response = await send(10)
135
- if (!response.ok) throw new Error('model transport failed: ' + response.status + ' ' + await response.text())
136
- await response.json()
137
- throw new Error('transport-handshake-complete')
138
- }
139
- `)
140
- const method = gepaOptimizationMethod({
141
- recipe: recipeForSeat(spec, optimizer.model), optimizer,
142
- objective: 'Check transport only', evaluationId: 'bench-transport',
143
- runner: { command: process.execPath, args: [bridge] },
144
- })
145
- await expect(method.optimize({
146
- baselineSurface: 'seed', trainScenarios: [{ id: 'train', kind: 'fixture' }], selectionScenarios: [{ id: 'selection', kind: 'fixture' }],
147
- dispatchWithSurface: async () => 'unused',
148
- judges: [{ name: 'fixture', dimensions: [{ key: 'score', description: 'fixture' }], score: async () => ({ dimensions: { score: 1 }, composite: 1 }) }],
149
- runDir, seed: 42, runOptions: {}, costLedger: createRunCostLedger({ storage: fsCampaignStorage(), runDir: join(runDir, 'cost') }),
150
- })).rejects.toThrow('transport-handshake-complete')
151
- expect(requests).toHaveLength(1)
152
- } finally {
153
- await rm(runDir, { recursive: true, force: true })
154
- }
155
- })
156
-
157
- it("'gepa' is one bounded engine run carrying the full budget (default 10)", () => {
158
- const recipe = recipeForSeat(seat() as GepaSeatSpec)
159
- expect(recipe).toMatchObject({ kind: 'engine', run: { engine: 'gepa', maxEvaluations: DEFAULT_MAX_METRIC_CALLS } })
160
- expect(recipeEvaluationBudget(recipe)).toBe(DEFAULT_MAX_METRIC_CALLS)
161
- })
162
-
163
- it("'omni' uses the official recipe and its four bounded runs preserve the budget", () => {
164
- const recipe = recipeForSeat(seat({ engine: 'omni', maxMetricCalls: 10 }) as GepaSeatSpec)
165
- expect(recipe.kind).toBe('omni')
166
- if (recipe.kind !== 'omni') throw new Error('unreachable')
167
- expect(recipe.explore.map((r) => r.engine)).toEqual(['gepa', 'autoresearch', 'meta_harness'])
168
- expect(recipe.continueWith.engine).toBe('gepa')
169
- expect(recipeEvaluationBudget(recipe)).toBe(10)
170
- for (const run of [...recipe.explore, recipe.continueWith]) {
171
- expect(run.maxEvaluations).toBeGreaterThan(0)
172
- expect(run.maxProposerCostUsd).toBeGreaterThan(0)
173
- }
174
- })
175
-
176
- it('every budget from 4 upward is preserved exactly by the omni split', () => {
177
- for (const calls of [4, 5, 8, 12, 24]) {
178
- const recipe = recipeForSeat(seat({ engine: 'omni', maxMetricCalls: calls }) as GepaSeatSpec)
179
- expect(recipeEvaluationBudget(recipe)).toBe(calls)
180
- }
181
- })
182
- })
183
-
184
- // ---------------------------------------------------------------------------
185
- // Public-only bridge examples.
186
- // ---------------------------------------------------------------------------
187
-
188
- describe('gepaBridgeScenarios (public-only invariant)', () => {
189
- const split = { privateInstances: ['django__django-11532', 'sphinx-doc__sphinx-9658'] }
190
-
191
- it('serializes ONLY the public smoke instance (train + a distinct-id selection alias)', () => {
192
- const { train, selection } = gepaBridgeScenarios('astropy__astropy-13033', split)
193
- expect(train).toEqual([{ id: 'astropy__astropy-13033', kind: 'swe-smoke', smokeIid: 'astropy__astropy-13033' }])
194
- expect(selection[0]!.id).toBe('astropy__astropy-13033::selection')
195
- expect(selection[0]!.smokeIid).toBe('astropy__astropy-13033')
196
- // Disjoint ids — the adapter's scenario map requires uniqueness.
197
- expect(train[0]!.id).not.toBe(selection[0]!.id)
198
- const serialized = JSON.stringify([...train, ...selection])
199
- for (const iid of split.privateInstances) expect(serialized).not.toContain(iid)
200
- })
201
-
202
- it('fails loud when the smoke instance is private — private ids never cross the bridge', () => {
203
- expect(() => gepaBridgeScenarios('django__django-11532', split)).toThrow(/PRIVATE under the score split/)
204
- })
205
-
206
- it('passes through with no split configured (pre-gen-5 behavior)', () => {
207
- expect(gepaBridgeScenarios('astropy__astropy-13033', null).train).toHaveLength(1)
208
- })
209
- })
210
-
211
- // ---------------------------------------------------------------------------
212
- // Inner score.
213
- // ---------------------------------------------------------------------------
214
-
215
- describe('inner smoke score', () => {
216
- const verdict = (over: Partial<SmokeVerdict>): SmokeVerdict => ({
217
- iid: 'astropy__astropy-13033',
218
- pass: true,
219
- reason: 'ok',
220
- resolved: false,
221
- patchLines: 3,
222
- wallS: 60,
223
- ...over,
224
- })
225
-
226
- it('resolve dominates; verify-pass is a bounded tiebreak that can never beat a resolve', () => {
227
- expect(innerSmokeComposite(verdict({ resolved: false, verifyPass: false }))).toBe(0)
228
- expect(innerSmokeComposite(verdict({ resolved: false, verifyPass: true }))).toBe(0.25)
229
- expect(innerSmokeComposite(verdict({ resolved: true, verifyPass: false }))).toBe(1)
230
- expect(innerSmokeComposite(verdict({ resolved: true, verifyPass: true }))).toBe(1.25)
231
- // Older verdicts without the field score as no verify signal.
232
- expect(innerSmokeComposite(verdict({ resolved: true }))).toBe(1)
233
- })
234
-
235
- it('the judge reports both dimensions and the composite', async () => {
236
- const judge = innerSmokeJudge()
237
- const score = await judge.score({
238
- artifact: verdict({ resolved: true, verifyPass: true, reason: 'smoke line' }),
239
- scenario: { id: 'x', kind: 'swe-smoke', smokeIid: 'x' },
240
- signal: new AbortController().signal,
241
- })
242
- expect(score).toMatchObject({ dimensions: { resolved: 1, verifyPass: 1 }, composite: 1.25, notes: 'smoke line' })
243
- })
244
- })
245
-
246
- // ---------------------------------------------------------------------------
247
- // Optional Python runtime: loud failures with exact instructions.
248
- // ---------------------------------------------------------------------------
249
-
250
- describe('probeGepaRuntime', () => {
251
- const execFailingOn =
252
- (failFragment: string, stderr: string): ProbeExec =>
253
- async (_cmd, args) => {
254
- const line = args.join(' ')
255
- if (line.includes(failFragment)) return { code: 1, stdout: '', stderr }
256
- if (line === '--version') return { code: 0, stdout: 'Python 3.12.3', stderr: '' }
257
- return { code: 0, stdout: '0.2.0', stderr: '' }
258
- }
259
-
260
- it('fails loud with pip install instructions when the bridge module is missing', async () => {
261
- const exec = execFailingOn('agent_eval_rpc.gepa_bridge', "ModuleNotFoundError: No module named 'agent_eval_rpc'")
262
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(
263
- GEPA_PYTHON_INSTALL_HINT,
264
- )
265
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(/not installed/)
266
- })
267
-
268
- it('fails loud when the installed gepa lacks the multi-engine optimize_anything API', async () => {
269
- const exec = execFailingOn('OptimizeAnythingConfig', "ImportError: cannot import name 'OptimizeAnythingConfig'")
270
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(/optimize_anything API/)
271
- })
272
-
273
- it('fails loud when python itself is missing', async () => {
274
- const exec: ProbeExec = async () => ({ code: 127, stdout: '', stderr: 'not found' })
275
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(/--version' failed/)
276
- })
277
-
278
- it('returns python + gepa versions on a complete runtime', async () => {
279
- const exec: ProbeExec = async (_cmd, args) =>
280
- args.join(' ') === '--version'
281
- ? { code: 0, stdout: 'Python 3.12.3\n', stderr: '' }
282
- : { code: 0, stdout: 'source\n', stderr: '' }
283
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).resolves.toEqual({
284
- pythonVersion: 'Python 3.12.3',
285
- gepaVersion: 'source',
286
- })
287
- })
288
- })
289
-
290
- // ---------------------------------------------------------------------------
291
- // Provenance capture at t=0.
292
- // ---------------------------------------------------------------------------
293
-
294
- describe('captureProposerProvenance with a gepa seat', () => {
295
- const okExec: ProbeExec = async (cmd, args) => {
296
- const line = args.join(' ')
297
- if (line === '--version') {
298
- return { code: 0, stdout: cmd === 'python3' ? 'Python 3.12.3' : `${cmd} 1.0.0`, stderr: '' }
299
- }
300
- return { code: 0, stdout: 'source', stderr: '' }
301
- }
302
- it('records engine, surface, gepa version, bridge module, and the python runtime as harnessVersion', async () => {
303
- const record = await captureProposerProvenance([{ name: 'claude-author', harness: 'claude-code' }, seat()], {
304
- exec: okExec,
305
- readSettingsModel: () => 'settings-model',
306
- })
307
- const gepa = record.proposers.find((p) => p.name === 'gepa-author')!
308
- expect(gepa).toMatchObject({
309
- engine: 'gepa',
310
- surface: SURFACE,
311
- gepaVersion: 'source',
312
- bridge: 'agent_eval_rpc.gepa_bridge',
313
- harnessVersion: 'Python 3.12.3',
314
- pinnedModel: null,
315
- merge: false,
316
- })
317
- expect(gepa.harness).toBeUndefined()
318
- // The claude seat is untouched by the gepa capture path.
319
- expect(record.proposers.find((p) => p.name === 'claude-author')).toMatchObject({
320
- harness: 'claude-code',
321
- settingsModel: 'settings-model',
322
- })
323
- })
324
-
325
- it('fails LOUD at t=0 when the python runtime is missing — with install instructions', async () => {
326
- const exec: ProbeExec = async (_cmd, args) =>
327
- args.join(' ').includes('gepa_bridge')
328
- ? { code: 1, stdout: '', stderr: 'ModuleNotFoundError' }
329
- : { code: 0, stdout: 'Python 3.12.3', stderr: '' }
330
- await expect(captureProposerProvenance([seat()], { exec })).rejects.toThrow(
331
- GEPA_PYTHON_INSTALL_HINT,
332
- )
333
- })
334
- })
335
-
336
- // ---------------------------------------------------------------------------
337
- // Mechanical activation predicate.
338
- // ---------------------------------------------------------------------------
339
-
340
- describe('mechanicalActivationPredicate', () => {
341
- it('targets the longest added line and produces a parseable grep predicate', () => {
342
- const seed = 'alpha\nshared line stays here\n'
343
- const winner = 'alpha\nshared line stays here\nAlways run the neighboring test file before finalizing.\nshort\n'
344
- const predicate = mechanicalActivationPredicate(seed, winner, SURFACE)!
345
- expect(predicate.kind).toBe('grep')
346
- expect(predicate.pattern).toContain('Always run the neighboring test file')
347
- const parsed = parseActivationPredicate(JSON.stringify(predicate))
348
- expect(parsed.ok).toBe(true)
349
- // The pattern is regex-escaped: it must match its own source line.
350
- expect(new RegExp(predicate.pattern!).test('Always run the neighboring test file before finalizing.')).toBe(true)
351
- })
352
-
353
- it('escapes regex metacharacters in the added line', () => {
354
- const predicate = mechanicalActivationPredicate('', 'Use pattern (a|b).* with $VAR [strictly].\n', SURFACE)!
355
- expect(new RegExp(predicate.pattern!).test('Use pattern (a|b).* with $VAR [strictly].')).toBe(true)
356
- expect(new RegExp(predicate.pattern!).test('Use pattern axb1* with 2VAR strictly.')).toBe(false)
357
- })
358
-
359
- it('returns null when no added line is distinctive enough', () => {
360
- expect(mechanicalActivationPredicate('a\nb\n', 'a\nb\nshort\n', SURFACE)).toBeNull()
361
- expect(mechanicalActivationPredicate('same\n', 'same\n', SURFACE)).toBeNull()
362
- })
363
- })
364
-
365
- describe('recordGepaSeatInnerRun', () => {
366
- const sourceHash = 'a'.repeat(64)
367
- const bridgeHash = 'b'.repeat(64)
368
- const moduleHash = 'c'.repeat(64)
369
- const provenance = (runId: string): OptimizationMethodProvenance => ({
370
- source: {
371
- kind: 'package',
372
- evidence: 'observed',
373
- package: 'gepa',
374
- version: '0.1.4',
375
- sourceUrl: 'https://github.com/gepa-ai/gepa.git',
376
- revision: 'f919db0a622e2e9f9204779b81fe00cc1b2d808f',
377
- sourceSha256: sourceHash,
378
- },
379
- bridge: {
380
- kind: 'package',
381
- evidence: 'observed',
382
- package: 'agent-eval-rpc',
383
- version: '0.126.1',
384
- sourceSha256: bridgeHash,
385
- },
386
- modules: [{ module: 'example.engine', sourceSha256: moduleHash }],
387
- python: { implementation: 'CPython', version: '3.12.3' },
388
- runId,
389
- compatibleRunId: 'compatible-run',
390
- resumed: false,
391
- evaluationCount: 3,
392
- tokenUsage: {
393
- inputTokens: 120,
394
- cachedInputTokens: 20,
395
- outputTokens: 30,
396
- reasoningTokens: 10,
397
- totalTokens: 150,
398
- calls: 2,
399
- },
400
- artifactDir: `/tmp/${runId}`,
401
- })
402
- const run = (runId: string, over: Partial<GepaSeatInnerRun> = {}): GepaSeatInnerRun => {
403
- const result = provenance(runId)
404
- if (
405
- result.source.revision === undefined ||
406
- result.source.sourceSha256 === undefined ||
407
- result.bridge?.sourceSha256 === undefined ||
408
- result.modules === undefined ||
409
- result.python === undefined ||
410
- result.compatibleRunId === undefined ||
411
- result.tokenUsage === undefined
412
- ) {
413
- throw new Error('invalid test provenance')
414
- }
415
- return {
416
- seat: 'gepa-author',
417
- engine: 'gepa',
418
- surface: SURFACE,
419
- generation: 0,
420
- budget: 10,
421
- innerCallCount: 2,
422
- innerScores: [],
423
- bestComposite: 1,
424
- source: {
425
- ...result.source,
426
- revision: result.source.revision,
427
- sourceSha256: result.source.sourceSha256,
428
- },
429
- bridge: { ...result.bridge, sourceSha256: result.bridge.sourceSha256 },
430
- modules: result.modules,
431
- python: result.python,
432
- runId: result.runId,
433
- compatibleRunId: result.compatibleRunId,
434
- resumed: result.resumed,
435
- evaluationCount: result.evaluationCount,
436
- tokenUsage: result.tokenUsage,
437
- artifactDir: result.artifactDir,
438
- totalCostUsd: 0.25,
439
- costProvenance: { kind: 'observed', usd: 0.25 },
440
- accountingComplete: true,
441
- incompleteReasons: [],
442
- durationMs: 5,
443
- ...over,
444
- }
445
- }
446
-
447
- it('writes collision-free immutable records in parallel and preserves the launch record', async () => {
448
- const dir = await mkdtemp(join(tmpdir(), 'gepa-prov-'))
449
- try {
450
- const launchRecord = JSON.stringify({ capturedAt: '2026-07-24T00:00:00.000Z', proposers: [] }, null, 2)
451
- await writeFile(join(dir, 'proposer-provenance.json'), launchRecord)
452
- const paths = await Promise.all(
453
- Array.from({ length: 24 }, (_, index) =>
454
- recordGepaSeatInnerRun(dir, run(`run-${index}`, { generation: index })),
455
- ),
456
- )
457
- expect(new Set(paths).size).toBe(24)
458
- expect(await readFile(join(dir, 'proposer-provenance.json'), 'utf8')).toBe(launchRecord)
459
-
460
- const names = (await readdir(join(dir, GEPA_INNER_RUNS_DIRNAME))).filter((name) => name.endsWith('.json'))
461
- expect(names).toHaveLength(24)
462
- const records = await Promise.all(
463
- names.map(async (name) => JSON.parse(await readFile(join(dir, GEPA_INNER_RUNS_DIRNAME, name), 'utf8'))),
464
- )
465
- expect(new Set(records.map((record) => record.runId))).toEqual(
466
- new Set(Array.from({ length: 24 }, (_, index) => `run-${index}`)),
467
- )
468
- expect(records.every((record) => record.source.sourceSha256 === sourceHash)).toBe(true)
469
- } finally {
470
- await rm(dir, { recursive: true, force: true })
471
- }
472
- })
473
-
474
- it('fails on malformed existing launch or run data', async () => {
475
- const launchDir = await mkdtemp(join(tmpdir(), 'gepa-prov-bad-launch-'))
476
- const runDir = await mkdtemp(join(tmpdir(), 'gepa-prov-bad-run-'))
477
- try {
478
- await writeFile(join(launchDir, 'proposer-provenance.json'), '{not json')
479
- await expect(recordGepaSeatInnerRun(launchDir, run('new-run'))).rejects.toThrow(/malformed JSON/)
480
-
481
- await mkdir(join(runDir, GEPA_INNER_RUNS_DIRNAME), { recursive: true })
482
- await writeFile(join(runDir, GEPA_INNER_RUNS_DIRNAME, 'broken.json'), '[]')
483
- await expect(recordGepaSeatInnerRun(runDir, run('new-run'))).rejects.toThrow(/must contain a JSON object/)
484
- } finally {
485
- await rm(launchDir, { recursive: true, force: true })
486
- await rm(runDir, { recursive: true, force: true })
487
- }
488
- })
489
- })
490
-
491
- // ---------------------------------------------------------------------------
492
- // The seat inside the fan-out generator, against a real temp git repo.
493
- // ---------------------------------------------------------------------------
494
-
495
- const fakeCtx = {} as unknown as DispatchContext
496
- const testOptimizerEnv: NodeJS.ProcessEnv = {
497
- TEST_OPTIMIZER_INPUT_USD_PER_MILLION: '1',
498
- TEST_OPTIMIZER_CACHED_INPUT_USD_PER_MILLION: '0.1',
499
- TEST_OPTIMIZER_CACHE_WRITE_USD_PER_MILLION: '1.25',
500
- TEST_OPTIMIZER_OUTPUT_USD_PER_MILLION: '5',
501
- TEST_OPTIMIZER_MAX_REQUESTS: '10',
502
- TEST_OPTIMIZER_MAX_REQUEST_BYTES: '100000',
503
- TEST_OPTIMIZER_MAX_RESPONSE_BYTES: '100000',
504
- }
505
-
506
- const testOptimizer = officialOptimizerModel({
507
- env: testOptimizerEnv,
508
- envPrefix: 'TEST_OPTIMIZER',
509
- model: 'optimizer-model',
510
- baseUrl: 'http://127.0.0.1:1/v1',
511
- apiKey: 'optimizer-key',
512
- maxCostUsd: 1,
513
- maxOutputTokensPerRequest: 2_000,
514
- callRef: 'test:optimizer-model',
515
- })
516
-
517
- const fullProvenance = (
518
- runId = 'gepa-run',
519
- over: Partial<OptimizationMethodProvenance> = {},
520
- ): OptimizationMethodProvenance => ({
521
- source: {
522
- kind: 'package',
523
- evidence: 'observed',
524
- package: 'gepa',
525
- version: '0.1.4',
526
- sourceUrl: 'https://github.com/gepa-ai/gepa.git',
527
- revision: 'f919db0a622e2e9f9204779b81fe00cc1b2d808f',
528
- sourceSha256: '1'.repeat(64),
529
- },
530
- bridge: {
531
- kind: 'package',
532
- evidence: 'observed',
533
- package: 'agent-eval-rpc',
534
- version: '0.126.1',
535
- sourceSha256: '2'.repeat(64),
536
- },
537
- modules: [{ module: 'custom_gepa_engines', sourceSha256: '3'.repeat(64) }],
538
- python: { implementation: 'CPython', version: '3.12.3' },
539
- runId,
540
- compatibleRunId: 'compatible-gepa-run',
541
- resumed: false,
542
- evaluationCount: 3,
543
- tokenUsage: {
544
- inputTokens: 100,
545
- cachedInputTokens: 10,
546
- cacheWriteInputTokens: 5,
547
- outputTokens: 25,
548
- reasoningTokens: 8,
549
- totalTokens: 125,
550
- calls: 2,
551
- },
552
- artifactDir: `/tmp/${runId}`,
553
- ...over,
554
- })
555
-
556
- /** Mimics the adapter's loop: score the seed and each provided candidate via
557
- * the seat's dispatch + judge, return the best-scoring candidate — exactly
558
- * the contract gepaOptimizationMethod fulfills through the Python bridge. */
559
- const fakeGepaFactory =
560
- (candidates: string[], observed?: { config?: unknown }): GepaMethodFactory =>
561
- (config) => {
562
- if (observed) observed.config = config
563
- return {
564
- name: config.name ?? 'fake-gepa',
565
- async optimize(input) {
566
- const judge = input.judges[0]!
567
- const scenario = input.trainScenarios[0]!
568
- let best = { surface: input.baselineSurface as string, composite: -Infinity }
569
- for (const candidate of [input.baselineSurface as string, ...candidates]) {
570
- const artifact = await input.dispatchWithSurface(candidate, scenario, fakeCtx)
571
- const score = await judge.score({ artifact, scenario, signal: new AbortController().signal })
572
- if (score.composite > best.composite) best = { surface: candidate, composite: score.composite }
573
- }
574
- return {
575
- winnerSurface: best.surface,
576
- cost: {
577
- totalCostUsd: 0.125,
578
- costProvenance: { kind: 'observed', usd: 0.125 },
579
- accountingComplete: true,
580
- incompleteReasons: [],
581
- },
582
- durationMs: 1,
583
- provenance: fullProvenance(),
584
- }
585
- },
586
- }
587
- }
588
-
589
- describe('fanOutLoopsGenerator with the gepa seat', () => {
590
- let loopsRepo: string
591
- let outDir: string
592
- let driverWt: string
593
-
594
- const git = async (args: string[], cwd: string): Promise<string> =>
595
- (await runOk('git', ['-C', cwd, ...args])).stdout.trim()
596
-
597
- const SEED = '# worker coding system\nkeep tests green\n'
598
-
599
- beforeEach(async () => {
600
- loopsRepo = await mkdtemp(join(tmpdir(), 'gepa-repo-'))
601
- outDir = await mkdtemp(join(tmpdir(), 'gepa-out-'))
602
- await runOk('git', ['init', '-q', '-b', 'main', loopsRepo])
603
- await runOk('git', ['-C', loopsRepo, 'config', 'core.hooksPath', '/dev/null'])
604
- await git(['config', 'user.email', 't@t.dev'], loopsRepo)
605
- await git(['config', 'user.name', 'T'], loopsRepo)
606
- await mkdir(join(loopsRepo, 'extensions', 'pi', 'prompts'), { recursive: true })
607
- await writeFile(join(loopsRepo, SURFACE), SEED)
608
- await git(['add', '-A'], loopsRepo)
609
- await git(['commit', '-q', '-m', 'init'], loopsRepo)
610
- driverWt = join(outDir, 'driver-wt')
611
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', driverWt, 'HEAD'])
612
- })
613
-
614
- afterEach(async () => {
615
- await rm(outDir, { recursive: true, force: true })
616
- await rm(loopsRepo, { recursive: true, force: true })
617
- })
618
-
619
- const baseConfig = (proposers: ProposerSpec[], over: Partial<OuterLoopConfig> = {}): OuterLoopConfig => ({
620
- ...defaultRound4Config(),
621
- loopsRepo,
622
- outDir,
623
- populationSize: proposers.length,
624
- proposers,
625
- prefilter: { enabled: true, smokeInstance: 'cheapest-of-set', requireResolved: false },
626
- ...over,
627
- })
628
-
629
- const generatorArgs = (candidateIndex: number) => ({
630
- worktreePath: driverWt,
631
- findings: [],
632
- maxShots: 1,
633
- signal: new AbortController().signal,
634
- generation: 0,
635
- candidateIndex,
636
- })
637
-
638
- const smokeVerdict = (over: Partial<SmokeVerdict> = {}): SmokeVerdict => ({
639
- iid: 'astropy__astropy-13033',
640
- pass: true,
641
- reason: 'smoke ok',
642
- resolved: false,
643
- patchLines: 3,
644
- wallS: 5,
645
- verifyPass: false,
646
- ...over,
647
- })
648
-
649
- it('construction fails loud without the smoke runner (the inner evaluator)', () => {
650
- expect(() => fanOutLoopsGenerator(baseConfig([seat()]))).toThrow(/inner evaluator/)
651
- expect(() =>
652
- fanOutLoopsGenerator(baseConfig([seat()]), {
653
- smokeRunner: async () => smokeVerdict(),
654
- smokeInstanceId: 'astropy__astropy-13033',
655
- }),
656
- ).toThrow(/immutable runner and judge references/)
657
- expect(() => fanOutLoopsGenerator(baseConfig([{ name: 'no-seat-kind' }]))).toThrow(/neither a harness nor an engine/)
658
- })
659
-
660
- it('materializes each candidate into the scratch surface, applies the winner through the normal prefilter path, and records provenance', async () => {
661
- const WINNER = `${SEED}Always run the neighboring test file before finalizing.\n`
662
- const LOSER = `${SEED}delete all tests\n`
663
- const seen: Array<{
664
- content: string
665
- scratch: string
666
- evaluationKey: string
667
- hasCostLedger: boolean
668
- }> = []
669
- const smokeRunner: SmokeRunner = async ({
670
- scratchPath,
671
- evaluationKey,
672
- costLedger,
673
- }) => {
674
- const content = await readFile(join(scratchPath, SURFACE), 'utf8')
675
- seen.push({
676
- content,
677
- scratch: scratchPath,
678
- evaluationKey,
679
- hasCostLedger: costLedger !== undefined,
680
- })
681
- // The winner candidate resolves; the seed gets verify-pass only; the
682
- // loser gets nothing — exercising resolve-dominates + tiebreak.
683
- if (content === WINNER) return smokeVerdict({ resolved: true, verifyPass: true })
684
- if (content === SEED) return smokeVerdict({ verifyPass: true })
685
- return smokeVerdict()
686
- }
687
- const observed: { config?: unknown } = {}
688
- const config = baseConfig([seat({ maxMetricCalls: 5 })], { activationGate: true })
689
- const gen = fanOutLoopsGenerator(config, {
690
- ...IMPLEMENTATION_REFS,
691
- smokeRunner,
692
- smokeInstanceId: 'astropy__astropy-13033',
693
- scoreSplit: { privateInstances: ['django__django-11532'] },
694
- gepaMethodFactory: fakeGepaFactory([LOSER, WINNER], observed),
695
- gepaOptimizer: testOptimizer,
696
- })
697
-
698
- const result = await gen.generate(generatorArgs(0))
699
-
700
- expect(result).toMatchObject({ applied: true, label: 'gepa-author' })
701
- expect(result.rationale).toContain('engine gepa')
702
- // 3 isolated inner calls (seed, loser, winner) + 1 stage-B prefilter smoke
703
- // on the final candidate, all outside the driver worktree.
704
- expect(seen).toHaveLength(4)
705
- for (const call of seen) expect(call.scratch).not.toBe(driverWt)
706
- expect(seen.map((c) => c.content)).toEqual([SEED, LOSER, WINNER, WINNER])
707
- expect(new Set(seen.slice(0, 3).map((call) => call.scratch)).size).toBe(3)
708
- expect(seen.slice(0, 3).every((call) => call.evaluationKey.startsWith('inner-'))).toBe(true)
709
- expect(seen[3]!.evaluationKey).toBe('candidate-0')
710
- expect(seen.slice(0, 3).every((call) => call.hasCostLedger)).toBe(true)
711
- // The winner landed on the driver worktree with the mechanical predicate.
712
- expect(await readFile(join(driverWt, SURFACE), 'utf8')).toBe(WINNER)
713
- const predicate = parseActivationPredicate(await readFile(join(driverWt, ACTIVATION_PREDICATE_RELPATH), 'utf8'))
714
- expect(predicate.ok).toBe(true)
715
- // Only the surface + predicate changed.
716
- const changed = (await runOk('git', ['-C', driverWt, 'status', '--porcelain', '--untracked-files=all'])).stdout
717
- .split('\n')
718
- .map((l) => l.slice(3).trim())
719
- .filter(Boolean)
720
- expect(changed.sort()).toEqual([ACTIVATION_PREDICATE_RELPATH, SURFACE].sort())
721
- // Budget threaded into the adapter recipe.
722
- const incumbentCommit = await git(['rev-parse', 'HEAD'], driverWt)
723
- expect(observed.config).toMatchObject({
724
- evaluationId: gepaSeatEvaluationId({
725
- smokeInstanceId: 'astropy__astropy-13033',
726
- dispatchTimeoutMs: config.dispatchTimeoutMs,
727
- incumbentCommit,
728
- ...IMPLEMENTATION_REFS,
729
- }),
730
- recipe: { kind: 'engine', run: { engine: 'gepa', maxEvaluations: 5 } },
731
- optimizer: testOptimizer,
732
- resume: 'if-compatible',
733
- trustResumeState: true,
734
- })
735
- // Inner-run provenance: the seat-local file plus one immutable shared record.
736
- const inner = JSON.parse(await readFile(join(outDir, 'gepa-seat', 'gen0-gepa-author', 'inner-provenance.json'), 'utf8'))
737
- expect(inner).toMatchObject({
738
- seat: 'gepa-author',
739
- engine: 'gepa',
740
- budget: 5,
741
- innerCallCount: 3,
742
- bestComposite: 1.25,
743
- source: {
744
- package: 'gepa',
745
- version: '0.1.4',
746
- revision: 'f919db0a622e2e9f9204779b81fe00cc1b2d808f',
747
- sourceSha256: '1'.repeat(64),
748
- },
749
- bridge: {
750
- package: 'agent-eval-rpc',
751
- version: '0.126.1',
752
- sourceSha256: '2'.repeat(64),
753
- },
754
- modules: [{ module: 'custom_gepa_engines', sourceSha256: '3'.repeat(64) }],
755
- python: { implementation: 'CPython', version: '3.12.3' },
756
- runId: 'gepa-run',
757
- compatibleRunId: 'compatible-gepa-run',
758
- resumed: false,
759
- evaluationCount: 3,
760
- tokenUsage: {
761
- inputTokens: 100,
762
- cachedInputTokens: 10,
763
- cacheWriteInputTokens: 5,
764
- outputTokens: 25,
765
- reasoningTokens: 8,
766
- totalTokens: 125,
767
- calls: 2,
768
- },
769
- artifactDir: '/tmp/gepa-run',
770
- totalCostUsd: 0.125,
771
- accountingComplete: true,
772
- incompleteReasons: [],
773
- })
774
- expect(inner.innerScores.map((s: { composite: number }) => s.composite)).toEqual([0.25, 0, 1.25])
775
- const records = await readdir(join(outDir, GEPA_INNER_RUNS_DIRNAME))
776
- expect(records.filter((name) => name.endsWith('.json'))).toHaveLength(1)
777
- expect(JSON.parse(await readFile(join(outDir, GEPA_INNER_RUNS_DIRNAME, records[0]!), 'utf8'))).toEqual(inner)
778
- expect(gen.drainPrefilterKills()).toEqual([])
779
- })
780
-
781
- it('isolates concurrent candidate evaluations while preserving parallel execution', async () => {
782
- const candidateA = `${SEED}candidate A keeps its own workspace\n`
783
- const candidateB = `${SEED}candidate B keeps its own workspace\n`
784
- let releaseBoth!: () => void
785
- const bothStarted = new Promise<void>((resolve) => {
786
- releaseBoth = resolve
787
- })
788
- let started = 0
789
- const innerSeen: Array<{ content: string; scratchPath: string }> = []
790
- const smokeRunner: SmokeRunner = async ({
791
- scratchPath,
792
- evaluationKey,
793
- }) => {
794
- if (evaluationKey.startsWith('inner-')) {
795
- started += 1
796
- if (started === 2) releaseBoth()
797
- await bothStarted
798
- }
799
- const content = await readFile(join(scratchPath, SURFACE), 'utf8')
800
- if (evaluationKey.startsWith('inner-')) {
801
- innerSeen.push({ content, scratchPath })
802
- }
803
- return smokeVerdict({ resolved: content === candidateA })
804
- }
805
- const parallelFactory: GepaMethodFactory = () => ({
806
- name: 'parallel-gepa',
807
- async optimize(input) {
808
- const scenario = input.trainScenarios[0]!
809
- await Promise.all([
810
- input.dispatchWithSurface(candidateA, scenario, fakeCtx),
811
- input.dispatchWithSurface(candidateB, scenario, fakeCtx),
812
- ])
813
- return {
814
- winnerSurface: candidateA,
815
- cost: {
816
- totalCostUsd: 0,
817
- costProvenance: { kind: 'observed', usd: 0 },
818
- accountingComplete: true,
819
- incompleteReasons: [],
820
- },
821
- durationMs: 1,
822
- provenance: fullProvenance('parallel-gepa', { evaluationCount: 2 }),
823
- }
824
- },
825
- })
826
- const gen = fanOutLoopsGenerator(baseConfig([seat({ maxMetricCalls: 2 })]), {
827
- ...IMPLEMENTATION_REFS,
828
- smokeRunner,
829
- smokeInstanceId: 'astropy__astropy-13033',
830
- scoreSplit: null,
831
- gepaMethodFactory: parallelFactory,
832
- })
833
-
834
- const result = await gen.generate(generatorArgs(0))
835
-
836
- expect(result.applied).toBe(true)
837
- expect(innerSeen.map((entry) => entry.content).sort()).toEqual(
838
- [candidateA, candidateB].sort(),
839
- )
840
- expect(new Set(innerSeen.map((entry) => entry.scratchPath)).size).toBe(2)
841
- })
842
-
843
- it('uses explicit immutable refs so captured runner or judge behavior cannot share resume state', async () => {
844
- const config = baseConfig([seat()])
845
- const incumbentCommit = await git(['rev-parse', 'HEAD'], driverWt)
846
- const makeRunner = (resolved: boolean): SmokeRunner =>
847
- async () => smokeVerdict({ resolved })
848
- const runnerV1 = makeRunner(false)
849
- const runnerV2 = makeRunner(true)
850
- const common = {
851
- smokeInstanceId: 'astropy__astropy-13033',
852
- dispatchTimeoutMs: config.dispatchTimeoutMs,
853
- incumbentCommit,
854
- }
855
- const original = gepaSeatEvaluationId({ ...common, ...IMPLEMENTATION_REFS })
856
- const changedRunner = gepaSeatEvaluationId({
857
- ...common,
858
- ...IMPLEMENTATION_REFS,
859
- runnerImplementationRef: `sha256:${'c'.repeat(64)}`,
860
- })
861
- const changedJudge = gepaSeatEvaluationId({
862
- ...common,
863
- ...IMPLEMENTATION_REFS,
864
- judgeImplementationRef: `sha256:${'d'.repeat(64)}`,
865
- })
866
- const changedIncumbent = gepaSeatEvaluationId({
867
- ...common,
868
- ...IMPLEMENTATION_REFS,
869
- incumbentCommit: 'e'.repeat(40),
870
- })
871
- const smokeArgs: Parameters<SmokeRunner>[0] = {
872
- scratchPath: driverWt,
873
- generation: 0,
874
- proposer: seat(),
875
- evaluationKey: 'identity-test',
876
- }
877
-
878
- expect(runnerV1.toString()).toBe(runnerV2.toString())
879
- expect((await runnerV1(smokeArgs)).resolved).toBe(false)
880
- expect((await runnerV2(smokeArgs)).resolved).toBe(true)
881
- expect(original).toMatch(
882
- /^swe-arena-gepa-seat\|smoke=astropy__astropy-13033\|incumbent=[a-f0-9]{40,64}\|runner=sha256:[a-f0-9]{64}\|judge=sha256:[a-f0-9]{64}\|dispatchTimeoutMs=\d+$/,
883
- )
884
- expect(changedRunner).not.toBe(original)
885
- expect(changedJudge).not.toBe(original)
886
- expect(changedIncumbent).not.toBe(original)
887
- expect(() =>
888
- gepaSeatEvaluationId({
889
- ...common,
890
- ...IMPLEMENTATION_REFS,
891
- runnerImplementationRef: 'runner-v2',
892
- }),
893
- ).toThrow(/runnerImplementationRef must be an immutable sha256 reference/)
894
- })
895
-
896
- it('records but rejects a winner when cost accounting is incomplete', async () => {
897
- const WINNER = `${SEED}Always run the neighboring test file before finalizing.\n`
898
- const incomplete: GepaMethodFactory = () => ({
899
- name: 'incomplete-cost',
900
- async optimize(input) {
901
- await input.dispatchWithSurface(WINNER, input.trainScenarios[0]!, fakeCtx)
902
- return {
903
- winnerSurface: WINNER,
904
- cost: {
905
- totalCostUsd: 0.25,
906
- costProvenance: { kind: 'uncaptured', usd: null },
907
- accountingComplete: false,
908
- incompleteReasons: ['optimizer model receipt missing'],
909
- },
910
- durationMs: 1,
911
- provenance: fullProvenance('incomplete-run', { evaluationCount: 1 }),
912
- }
913
- },
914
- })
915
- const gen = fanOutLoopsGenerator(baseConfig([seat()]), {
916
- ...IMPLEMENTATION_REFS,
917
- smokeRunner: async () => smokeVerdict({ resolved: true }),
918
- smokeInstanceId: 'astropy__astropy-13033',
919
- scoreSplit: null,
920
- gepaMethodFactory: incomplete,
921
- })
922
-
923
- await expect(gen.generate(generatorArgs(0))).rejects.toThrow(
924
- /cost accounting is incomplete: optimizer model receipt missing/,
925
- )
926
- expect(await readFile(join(driverWt, SURFACE), 'utf8')).toBe(SEED)
927
- const inner = JSON.parse(
928
- await readFile(join(outDir, 'gepa-seat', 'gen0-gepa-author', 'inner-provenance.json'), 'utf8'),
929
- )
930
- expect(inner).toMatchObject({
931
- runId: 'incomplete-run',
932
- totalCostUsd: 0.25,
933
- accountingComplete: false,
934
- incompleteReasons: ['optimizer model receipt missing'],
935
- })
936
- })
937
-
938
- it('enforces the inner-call budget cap fail-closed', async () => {
939
- const runaway: GepaMethodFactory = () => ({
940
- name: 'runaway',
941
- async optimize(input) {
942
- const results = await Promise.allSettled(
943
- Array.from({ length: 4 }, (_, index) =>
944
- input.dispatchWithSurface(
945
- `${SEED}candidate ${index}\n`,
946
- input.trainScenarios[0]!,
947
- fakeCtx,
948
- ),
949
- ),
950
- )
951
- const rejected = results.find(
952
- (result): result is PromiseRejectedResult => result.status === 'rejected',
953
- )
954
- if (rejected) throw rejected.reason
955
- return {
956
- winnerSurface: SEED,
957
- cost: {
958
- totalCostUsd: 0,
959
- costProvenance: { kind: 'uncaptured', usd: null },
960
- accountingComplete: false,
961
- incompleteReasons: [],
962
- },
963
- durationMs: 1,
964
- provenance: fullProvenance('runaway'),
965
- }
966
- },
967
- })
968
- const gen = fanOutLoopsGenerator(baseConfig([seat({ maxMetricCalls: 3 })]), {
969
- ...IMPLEMENTATION_REFS,
970
- smokeRunner: async () => smokeVerdict(),
971
- smokeInstanceId: 'astropy__astropy-13033',
972
- scoreSplit: null,
973
- gepaMethodFactory: runaway,
974
- })
975
- await expect(gen.generate(generatorArgs(0))).rejects.toThrow(/inner-call budget 3 exhausted/)
976
- })
977
-
978
- it('refuses to feed a PRIVATE smoke verdict to the bridge', async () => {
979
- const gen = fanOutLoopsGenerator(baseConfig([seat()]), {
980
- ...IMPLEMENTATION_REFS,
981
- smokeRunner: async () => smokeVerdict({ iid: 'django__django-11532' }),
982
- smokeInstanceId: 'astropy__astropy-13033',
983
- scoreSplit: { privateInstances: ['django__django-11532'] },
984
- gepaMethodFactory: fakeGepaFactory([`${SEED}x line long enough\n`]),
985
- })
986
- await expect(gen.generate(generatorArgs(0))).rejects.toThrow(/PRIVATE instance django__django-11532/)
987
- })
988
-
989
- it('declines the slot without a kill when GEPA returns the seed unchanged', async () => {
990
- const gen = fanOutLoopsGenerator(baseConfig([seat()]), {
991
- ...IMPLEMENTATION_REFS,
992
- smokeRunner: async () => smokeVerdict(),
993
- smokeInstanceId: 'astropy__astropy-13033',
994
- scoreSplit: null,
995
- gepaMethodFactory: fakeGepaFactory([]),
996
- })
997
- const result = await gen.generate(generatorArgs(0))
998
- expect(result.applied).toBe(false)
999
- expect(result.summary).toContain('equals the seed')
1000
- expect(gen.drainPrefilterKills()).toEqual([])
1001
- expect((await runOk('git', ['-C', driverWt, 'status', '--porcelain'])).stdout.trim()).toBe('')
1002
- })
1003
- })
1004
-
1005
- // ---------------------------------------------------------------------------
1006
- // Integration: one real Node-to-Python-to-score roundtrip through the installed
1007
- // bridge. The TypeScript adapter is a compile-time package dependency.
1008
- // ---------------------------------------------------------------------------
1009
-
1010
- const pythonBridgeReady = (python: string): { ok: boolean; reason: string } => {
1011
- const probe = spawnSync(python, [
1012
- '-c',
1013
- 'import agent_eval_rpc.gepa_bridge; from gepa.optimize_anything import optimize_anything, OptimizeAnythingConfig',
1014
- ])
1015
- if (probe.status !== 0) {
1016
- return { ok: false, reason: `python bridge unavailable: ${String(probe.stderr).trim().split('\n').pop()}` }
1017
- }
1018
- return { ok: true, reason: '' }
1019
- }
1020
-
1021
- describe('integration: real adapter roundtrip', () => {
1022
- it('runs a metered optimizer through the real bridge and applies its winner', async (ctx) => {
1023
- const python = process.env.AGENT_EVAL_TEST_PYTHON ?? DEFAULT_GEPA_PYTHON
1024
- const pythonRuntime = pythonBridgeReady(python)
1025
- if (!pythonRuntime.ok) {
1026
- ctx.skip(`skip-with-reason: ${pythonRuntime.reason}`)
1027
- return
1028
- }
1029
-
1030
- const winner = 'tiny synthetic surface\nAlways run the neighboring test before finalizing.'
1031
- const modelServer = createServer((_request, response) => {
1032
- response.writeHead(200, { 'content-type': 'application/json' })
1033
- response.end(
1034
- JSON.stringify({
1035
- choices: [
1036
- {
1037
- message: {
1038
- role: 'assistant',
1039
- content: `\`\`\`\n${winner}\`\`\``,
1040
- },
1041
- },
1042
- ],
1043
- usage: {
1044
- prompt_tokens: 20,
1045
- completion_tokens: 20,
1046
- total_tokens: 40,
1047
- },
1048
- }),
1049
- )
1050
- })
1051
- await new Promise<void>((resolve, reject) => {
1052
- modelServer.once('error', reject)
1053
- modelServer.listen(0, '127.0.0.1', resolve)
1054
- })
1055
- const address = modelServer.address()
1056
- if (!address || typeof address === 'string') {
1057
- throw new Error('test optimizer server did not bind')
1058
- }
1059
-
1060
- const loopsRepo = await mkdtemp(join(tmpdir(), 'gepa-int-repo-'))
1061
- const outDir = await mkdtemp(join(tmpdir(), 'gepa-int-out-'))
1062
- try {
1063
- await runOk('git', ['init', '-q', '-b', 'main', loopsRepo])
1064
- await runOk('git', ['-C', loopsRepo, 'config', 'core.hooksPath', '/dev/null'])
1065
- await runOk('git', ['-C', loopsRepo, 'config', 'user.email', 't@t.dev'])
1066
- await runOk('git', ['-C', loopsRepo, 'config', 'user.name', 'T'])
1067
- await mkdir(join(loopsRepo, 'extensions', 'pi', 'prompts'), { recursive: true })
1068
- await writeFile(join(loopsRepo, SURFACE), 'tiny synthetic surface\n')
1069
- await runOk('git', ['-C', loopsRepo, 'add', '-A'])
1070
- await runOk('git', ['-C', loopsRepo, 'commit', '-q', '-m', 'init'])
1071
- const driverWt = join(outDir, 'driver-wt')
1072
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', driverWt, 'HEAD'])
1073
-
1074
- const gen = fanOutLoopsGenerator(
1075
- {
1076
- ...defaultRound4Config(),
1077
- loopsRepo,
1078
- outDir,
1079
- populationSize: 1,
1080
- proposers: [seat({ maxMetricCalls: 10, python })],
1081
- prefilter: { enabled: true, smokeInstance: 'cheapest-of-set', requireResolved: false },
1082
- },
1083
- {
1084
- ...IMPLEMENTATION_REFS,
1085
- smokeRunner: async ({ scratchPath }) => {
1086
- const candidate = await readFile(join(scratchPath, SURFACE), 'utf8')
1087
- return {
1088
- iid: 'astropy__astropy-13033',
1089
- pass: true,
1090
- reason: 'stub smoke (integration)',
1091
- resolved: candidate === winner,
1092
- patchLines: 1,
1093
- wallS: 0,
1094
- verifyPass: true,
1095
- }
1096
- },
1097
- smokeInstanceId: 'astropy__astropy-13033',
1098
- scoreSplit: null,
1099
- gepaOptimizer: officialOptimizerModel({
1100
- env: testOptimizerEnv,
1101
- envPrefix: 'TEST_OPTIMIZER',
1102
- model: 'test-optimizer',
1103
- baseUrl: `http://127.0.0.1:${address.port}/v1`,
1104
- apiKey: 'local-test-key',
1105
- maxCostUsd: 1,
1106
- maxOutputTokensPerRequest: 2_000,
1107
- callRef: 'test:integration-optimizer',
1108
- }),
1109
- },
1110
- )
1111
- const result = await gen.generate({
1112
- worktreePath: driverWt,
1113
- findings: [],
1114
- maxShots: 1,
1115
- signal: new AbortController().signal,
1116
- generation: 0,
1117
- candidateIndex: 0,
1118
- })
1119
- const inner = JSON.parse(
1120
- await readFile(join(outDir, 'gepa-seat', 'gen0-gepa-author', 'inner-provenance.json'), 'utf8'),
1121
- )
1122
- expect(result.applied, `${result.summary}\n${JSON.stringify(inner.innerScores, null, 2)}`).toBe(true)
1123
- expect(await readFile(join(driverWt, SURFACE), 'utf8')).toBe(winner)
1124
- expect(inner.innerCallCount).toBeGreaterThanOrEqual(1)
1125
- expect(inner.accountingComplete).toBe(true)
1126
- expect(inner.incompleteReasons).toEqual([])
1127
- expect(inner.tokenUsage.calls).toBeGreaterThan(0)
1128
- } finally {
1129
- await new Promise<void>((resolve, reject) =>
1130
- modelServer.close((error) => (error ? reject(error) : resolve())),
1131
- )
1132
- await rm(outDir, { recursive: true, force: true })
1133
- await rm(loopsRepo, { recursive: true, force: true })
1134
- }
1135
- }, 300_000)
1136
- })