@tangle-network/agent-bench 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,1136 +0,0 @@
1
- import { spawnSync } from 'node:child_process'
2
- import { existsSync } from 'node:fs'
3
- import { mkdir, mkdtemp, readFile, readdir, rm, writeFile } from 'node:fs/promises'
4
- import { createServer } from 'node:http'
5
- import { tmpdir } from 'node:os'
6
- import { join } from 'node:path'
7
- import {
8
- gepaOptimizationMethod,
9
- createRunCostLedger,
10
- fsCampaignStorage,
11
- type DispatchContext,
12
- type OptimizationMethodProvenance,
13
- } from '@tangle-network/agent-eval/campaign'
14
- import { afterEach, beforeEach, describe, expect, it } from 'vitest'
15
- import { ACTIVATION_PREDICATE_RELPATH, parseActivationPredicate } from './activation.mts'
16
- import {
17
- DEFAULT_GEPA_PYTHON,
18
- DEFAULT_MAX_METRIC_CALLS,
19
- GEPA_INNER_RUNS_DIRNAME,
20
- GEPA_PYTHON_INSTALL_HINT,
21
- gepaBridgeScenarios,
22
- gepaSeatEvaluationId,
23
- innerSmokeComposite,
24
- innerSmokeJudge,
25
- isGepaSeat,
26
- mechanicalActivationPredicate,
27
- probeGepaRuntime,
28
- recipeEvaluationBudget,
29
- recipeForSeat,
30
- recordGepaSeatInnerRun,
31
- validateGepaSeat,
32
- type GepaMethodFactory,
33
- type GepaSeatInnerRun,
34
- type GepaSeatSpec,
35
- type ProbeExec,
36
- } from './gepa-seat.mts'
37
- import { defaultRound4Config, type OuterLoopConfig } from './outer-loop.mts'
38
- import { fanOutLoopsGenerator, type ProposerSpec, type SmokeRunner, type SmokeVerdict } from './proposer-fanout.mts'
39
- import { captureProposerProvenance } from './proposer-provenance.mts'
40
- import { officialOptimizerModel } from '../official-optimizer-config.mts'
41
- import { runOk } from './proc.ts'
42
-
43
- const SURFACE = 'extensions/pi/prompts/worker-coding-system.md'
44
- const IMPLEMENTATION_REFS = {
45
- runnerImplementationRef: `sha256:${'a'.repeat(64)}`,
46
- judgeImplementationRef: `sha256:${'b'.repeat(64)}`,
47
- } as const
48
-
49
- const seat = (over: Partial<ProposerSpec> = {}): ProposerSpec => ({
50
- name: 'gepa-author',
51
- engine: 'gepa',
52
- surface: SURFACE,
53
- ...over,
54
- })
55
-
56
- // ---------------------------------------------------------------------------
57
- // Spec validation.
58
- // ---------------------------------------------------------------------------
59
-
60
- describe('validateGepaSeat', () => {
61
- it('accepts an engine seat and isGepaSeat discriminates on engine', () => {
62
- expect(() => validateGepaSeat(seat())).not.toThrow()
63
- expect(() => validateGepaSeat(seat({ engine: 'omni', maxMetricCalls: 8 }))).not.toThrow()
64
- expect(isGepaSeat(seat())).toBe(true)
65
- expect(isGepaSeat({ name: 'x', harness: 'claude-code' })).toBe(false)
66
- })
67
-
68
- it('requires a surface inside the declared change-space', () => {
69
- expect(() => validateGepaSeat(seat({ surface: undefined }))).toThrow(/surface is required/)
70
- expect(() => validateGepaSeat(seat({ surface: 'judge.py' }))).toThrow(/outside the declared change-space/)
71
- expect(() => validateGepaSeat(seat({ surface: '../escape.md' }))).toThrow(/outside the declared change-space/)
72
- })
73
-
74
- it('rejects harness-seat fields on an engine seat instead of silently ignoring them', () => {
75
- expect(() => validateGepaSeat(seat({ harness: 'claude-code' }))).toThrow(/'harness' belongs to harness-authored/)
76
- expect(() => validateGepaSeat(seat({ merge: true }))).toThrow(/'merge'/)
77
- expect(() => validateGepaSeat(seat({ model: 'x' }))).toThrow(/'model'/)
78
- expect(() => validateGepaSeat(seat({ profile: 'p.json' }))).toThrow(/'profile'/)
79
- })
80
-
81
- it('bounds the budget: positive integer calls, omni needs >= 4, positive cost cap', () => {
82
- expect(() => validateGepaSeat(seat({ maxMetricCalls: 0 }))).toThrow(/maxMetricCalls/)
83
- expect(() => validateGepaSeat(seat({ maxMetricCalls: 2.5 }))).toThrow(/maxMetricCalls/)
84
- expect(() => validateGepaSeat(seat({ engine: 'omni', maxMetricCalls: 3 }))).toThrow(/omni.*needs maxMetricCalls >= 4/)
85
- expect(() => validateGepaSeat(seat({ maxProposerCostUsd: 0 }))).toThrow(/maxProposerCostUsd/)
86
- expect(() => validateGepaSeat(seat({ engine: 'nope' as never }))).toThrow(/engine must be one of/)
87
- })
88
- })
89
-
90
- // ---------------------------------------------------------------------------
91
- // Recipe / budget cap.
92
- // ---------------------------------------------------------------------------
93
-
94
- describe('recipeForSeat', () => {
95
- it.each(['gepa', 'omni'] as const)('transports the real %s optimizer configuration', async (engine) => {
96
- const runDir = await mkdtemp(join(tmpdir(), 'gepa-seat-transport-'))
97
- const requests: unknown[] = []
98
- try {
99
- const optimizer = officialOptimizerModel({
100
- env: { OPT_INPUT_USD_PER_MILLION: '1', OPT_CACHED_INPUT_USD_PER_MILLION: '1', OPT_CACHE_WRITE_USD_PER_MILLION: '1', OPT_OUTPUT_USD_PER_MILLION: '1' },
101
- envPrefix: 'OPT', model: 'fixture-model', baseUrl: 'http://127.0.0.1:1/v1', apiKey: 'fixture-key',
102
- maxCostUsd: 1, maxOutputTokensPerRequest: 100, anthropicEndpoint: engine === 'omni',
103
- complete: async (request) => {
104
- requests.push(request)
105
- return { model: 'fixture-model', choices: [{ message: { content: 'ok' }, finish_reason: 'stop' }], usage: { prompt_tokens: 2, completion_tokens: 1, cost: 0.000003 } }
106
- },
107
- })
108
- const spec = seat({ engine })
109
- validateGepaSeat(spec)
110
- const runtime = {
111
- python: { implementation: 'CPython', version: '3.12.0' },
112
- bridge: { package: 'agent-eval-rpc', version: 'fixture', sourceUrl: 'https://github.com/tangle-network/agent-eval', revision: 'fixture', sourceSha256: 'a'.repeat(64) },
113
- optimizer: { package: 'gepa', version: 'fixture', sourceUrl: 'https://github.com/gepa-ai/gepa', revision: 'fixture', sourceSha256: 'b'.repeat(64) },
114
- engineModules: [],
115
- }
116
- // Only the child optimizer is substituted; the real Eval proxy calls Runtime.
117
- const bridge = join(runDir, 'transport.mjs')
118
- await writeFile(bridge, `
119
- import fs from 'node:fs'
120
- const input = JSON.parse(fs.readFileSync(process.argv[process.argv.indexOf('--input') + 1], 'utf8'))
121
- if (input.operation === 'inspect') {
122
- fs.writeFileSync(process.argv[process.argv.indexOf('--output') + 1], JSON.stringify({ runtime: ${JSON.stringify(runtime)} }))
123
- } else {
124
- const omni = input.recipe.kind === 'omni'
125
- if (omni && input.recipe.explore.some(run => run.engine !== 'gepa' && run.engineConfig.model !== input.modelProxy.model)) throw new Error('wrong engine model')
126
- const send = maxTokens => fetch(input.modelProxy.baseUrl + (omni ? '/messages' : '/chat/completions'), {
127
- method: 'POST',
128
- headers: { 'content-type': 'application/json', authorization: 'Bearer ' + input.modelProxy.apiKey, 'x-api-key': input.modelProxy.apiKey, 'anthropic-version': '2023-06-01' },
129
- body: JSON.stringify({ model: input.modelProxy.model, max_tokens: maxTokens, messages: [{ role: 'user', content: 'transport proof' }] }),
130
- })
131
- const overBudget = await send(101)
132
- if (overBudget.ok) throw new Error('output ceiling was not enforced')
133
- await overBudget.text()
134
- const response = await send(10)
135
- if (!response.ok) throw new Error('model transport failed: ' + response.status + ' ' + await response.text())
136
- await response.json()
137
- throw new Error('transport-handshake-complete')
138
- }
139
- `)
140
- const method = gepaOptimizationMethod({
141
- recipe: recipeForSeat(spec, optimizer.model), optimizer,
142
- objective: 'Check transport only', evaluationId: 'bench-transport',
143
- runner: { command: process.execPath, args: [bridge] },
144
- })
145
- await expect(method.optimize({
146
- baselineSurface: 'seed', trainScenarios: [{ id: 'train', kind: 'fixture' }], selectionScenarios: [{ id: 'selection', kind: 'fixture' }],
147
- dispatchWithSurface: async () => 'unused',
148
- judges: [{ name: 'fixture', dimensions: [{ key: 'score', description: 'fixture' }], score: async () => ({ dimensions: { score: 1 }, composite: 1 }) }],
149
- runDir, seed: 42, runOptions: {}, costLedger: createRunCostLedger({ storage: fsCampaignStorage(), runDir: join(runDir, 'cost') }),
150
- })).rejects.toThrow('transport-handshake-complete')
151
- expect(requests).toHaveLength(1)
152
- } finally {
153
- await rm(runDir, { recursive: true, force: true })
154
- }
155
- })
156
-
157
- it("'gepa' is one bounded engine run carrying the full budget (default 10)", () => {
158
- const recipe = recipeForSeat(seat() as GepaSeatSpec)
159
- expect(recipe).toMatchObject({ kind: 'engine', run: { engine: 'gepa', maxEvaluations: DEFAULT_MAX_METRIC_CALLS } })
160
- expect(recipeEvaluationBudget(recipe)).toBe(DEFAULT_MAX_METRIC_CALLS)
161
- })
162
-
163
- it("'omni' uses the official recipe and its four bounded runs preserve the budget", () => {
164
- const recipe = recipeForSeat(seat({ engine: 'omni', maxMetricCalls: 10 }) as GepaSeatSpec)
165
- expect(recipe.kind).toBe('omni')
166
- if (recipe.kind !== 'omni') throw new Error('unreachable')
167
- expect(recipe.explore.map((r) => r.engine)).toEqual(['gepa', 'autoresearch', 'meta_harness'])
168
- expect(recipe.continueWith.engine).toBe('gepa')
169
- expect(recipeEvaluationBudget(recipe)).toBe(10)
170
- for (const run of [...recipe.explore, recipe.continueWith]) {
171
- expect(run.maxEvaluations).toBeGreaterThan(0)
172
- expect(run.maxProposerCostUsd).toBeGreaterThan(0)
173
- }
174
- })
175
-
176
- it('every budget from 4 upward is preserved exactly by the omni split', () => {
177
- for (const calls of [4, 5, 8, 12, 24]) {
178
- const recipe = recipeForSeat(seat({ engine: 'omni', maxMetricCalls: calls }) as GepaSeatSpec)
179
- expect(recipeEvaluationBudget(recipe)).toBe(calls)
180
- }
181
- })
182
- })
183
-
184
- // ---------------------------------------------------------------------------
185
- // Public-only bridge examples.
186
- // ---------------------------------------------------------------------------
187
-
188
- describe('gepaBridgeScenarios (public-only invariant)', () => {
189
- const split = { privateInstances: ['django__django-11532', 'sphinx-doc__sphinx-9658'] }
190
-
191
- it('serializes ONLY the public smoke instance (train + a distinct-id selection alias)', () => {
192
- const { train, selection } = gepaBridgeScenarios('astropy__astropy-13033', split)
193
- expect(train).toEqual([{ id: 'astropy__astropy-13033', kind: 'swe-smoke', smokeIid: 'astropy__astropy-13033' }])
194
- expect(selection[0]!.id).toBe('astropy__astropy-13033::selection')
195
- expect(selection[0]!.smokeIid).toBe('astropy__astropy-13033')
196
- // Disjoint ids — the adapter's scenario map requires uniqueness.
197
- expect(train[0]!.id).not.toBe(selection[0]!.id)
198
- const serialized = JSON.stringify([...train, ...selection])
199
- for (const iid of split.privateInstances) expect(serialized).not.toContain(iid)
200
- })
201
-
202
- it('fails loud when the smoke instance is private — private ids never cross the bridge', () => {
203
- expect(() => gepaBridgeScenarios('django__django-11532', split)).toThrow(/PRIVATE under the score split/)
204
- })
205
-
206
- it('passes through with no split configured (pre-gen-5 behavior)', () => {
207
- expect(gepaBridgeScenarios('astropy__astropy-13033', null).train).toHaveLength(1)
208
- })
209
- })
210
-
211
- // ---------------------------------------------------------------------------
212
- // Inner score.
213
- // ---------------------------------------------------------------------------
214
-
215
- describe('inner smoke score', () => {
216
- const verdict = (over: Partial<SmokeVerdict>): SmokeVerdict => ({
217
- iid: 'astropy__astropy-13033',
218
- pass: true,
219
- reason: 'ok',
220
- resolved: false,
221
- patchLines: 3,
222
- wallS: 60,
223
- ...over,
224
- })
225
-
226
- it('resolve dominates; verify-pass is a bounded tiebreak that can never beat a resolve', () => {
227
- expect(innerSmokeComposite(verdict({ resolved: false, verifyPass: false }))).toBe(0)
228
- expect(innerSmokeComposite(verdict({ resolved: false, verifyPass: true }))).toBe(0.25)
229
- expect(innerSmokeComposite(verdict({ resolved: true, verifyPass: false }))).toBe(1)
230
- expect(innerSmokeComposite(verdict({ resolved: true, verifyPass: true }))).toBe(1.25)
231
- // Older verdicts without the field score as no verify signal.
232
- expect(innerSmokeComposite(verdict({ resolved: true }))).toBe(1)
233
- })
234
-
235
- it('the judge reports both dimensions and the composite', async () => {
236
- const judge = innerSmokeJudge()
237
- const score = await judge.score({
238
- artifact: verdict({ resolved: true, verifyPass: true, reason: 'smoke line' }),
239
- scenario: { id: 'x', kind: 'swe-smoke', smokeIid: 'x' },
240
- signal: new AbortController().signal,
241
- })
242
- expect(score).toMatchObject({ dimensions: { resolved: 1, verifyPass: 1 }, composite: 1.25, notes: 'smoke line' })
243
- })
244
- })
245
-
246
- // ---------------------------------------------------------------------------
247
- // Optional Python runtime: loud failures with exact instructions.
248
- // ---------------------------------------------------------------------------
249
-
250
- describe('probeGepaRuntime', () => {
251
- const execFailingOn =
252
- (failFragment: string, stderr: string): ProbeExec =>
253
- async (_cmd, args) => {
254
- const line = args.join(' ')
255
- if (line.includes(failFragment)) return { code: 1, stdout: '', stderr }
256
- if (line === '--version') return { code: 0, stdout: 'Python 3.12.3', stderr: '' }
257
- return { code: 0, stdout: '0.2.0', stderr: '' }
258
- }
259
-
260
- it('fails loud with pip install instructions when the bridge module is missing', async () => {
261
- const exec = execFailingOn('agent_eval_rpc.gepa_bridge', "ModuleNotFoundError: No module named 'agent_eval_rpc'")
262
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(
263
- GEPA_PYTHON_INSTALL_HINT,
264
- )
265
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(/not installed/)
266
- })
267
-
268
- it('fails loud when the installed gepa lacks the multi-engine optimize_anything API', async () => {
269
- const exec = execFailingOn('OptimizeAnythingConfig', "ImportError: cannot import name 'OptimizeAnythingConfig'")
270
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(/optimize_anything API/)
271
- })
272
-
273
- it('fails loud when python itself is missing', async () => {
274
- const exec: ProbeExec = async () => ({ code: 127, stdout: '', stderr: 'not found' })
275
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).rejects.toThrow(/--version' failed/)
276
- })
277
-
278
- it('returns python + gepa versions on a complete runtime', async () => {
279
- const exec: ProbeExec = async (_cmd, args) =>
280
- args.join(' ') === '--version'
281
- ? { code: 0, stdout: 'Python 3.12.3\n', stderr: '' }
282
- : { code: 0, stdout: 'source\n', stderr: '' }
283
- await expect(probeGepaRuntime('python3', exec, 'gepa-author')).resolves.toEqual({
284
- pythonVersion: 'Python 3.12.3',
285
- gepaVersion: 'source',
286
- })
287
- })
288
- })
289
-
290
- // ---------------------------------------------------------------------------
291
- // Provenance capture at t=0.
292
- // ---------------------------------------------------------------------------
293
-
294
- describe('captureProposerProvenance with a gepa seat', () => {
295
- const okExec: ProbeExec = async (cmd, args) => {
296
- const line = args.join(' ')
297
- if (line === '--version') {
298
- return { code: 0, stdout: cmd === 'python3' ? 'Python 3.12.3' : `${cmd} 1.0.0`, stderr: '' }
299
- }
300
- return { code: 0, stdout: 'source', stderr: '' }
301
- }
302
- it('records engine, surface, gepa version, bridge module, and the python runtime as harnessVersion', async () => {
303
- const record = await captureProposerProvenance([{ name: 'claude-author', harness: 'claude-code' }, seat()], {
304
- exec: okExec,
305
- readSettingsModel: () => 'settings-model',
306
- })
307
- const gepa = record.proposers.find((p) => p.name === 'gepa-author')!
308
- expect(gepa).toMatchObject({
309
- engine: 'gepa',
310
- surface: SURFACE,
311
- gepaVersion: 'source',
312
- bridge: 'agent_eval_rpc.gepa_bridge',
313
- harnessVersion: 'Python 3.12.3',
314
- pinnedModel: null,
315
- merge: false,
316
- })
317
- expect(gepa.harness).toBeUndefined()
318
- // The claude seat is untouched by the gepa capture path.
319
- expect(record.proposers.find((p) => p.name === 'claude-author')).toMatchObject({
320
- harness: 'claude-code',
321
- settingsModel: 'settings-model',
322
- })
323
- })
324
-
325
- it('fails LOUD at t=0 when the python runtime is missing — with install instructions', async () => {
326
- const exec: ProbeExec = async (_cmd, args) =>
327
- args.join(' ').includes('gepa_bridge')
328
- ? { code: 1, stdout: '', stderr: 'ModuleNotFoundError' }
329
- : { code: 0, stdout: 'Python 3.12.3', stderr: '' }
330
- await expect(captureProposerProvenance([seat()], { exec })).rejects.toThrow(
331
- GEPA_PYTHON_INSTALL_HINT,
332
- )
333
- })
334
- })
335
-
336
- // ---------------------------------------------------------------------------
337
- // Mechanical activation predicate.
338
- // ---------------------------------------------------------------------------
339
-
340
- describe('mechanicalActivationPredicate', () => {
341
- it('targets the longest added line and produces a parseable grep predicate', () => {
342
- const seed = 'alpha\nshared line stays here\n'
343
- const winner = 'alpha\nshared line stays here\nAlways run the neighboring test file before finalizing.\nshort\n'
344
- const predicate = mechanicalActivationPredicate(seed, winner, SURFACE)!
345
- expect(predicate.kind).toBe('grep')
346
- expect(predicate.pattern).toContain('Always run the neighboring test file')
347
- const parsed = parseActivationPredicate(JSON.stringify(predicate))
348
- expect(parsed.ok).toBe(true)
349
- // The pattern is regex-escaped: it must match its own source line.
350
- expect(new RegExp(predicate.pattern!).test('Always run the neighboring test file before finalizing.')).toBe(true)
351
- })
352
-
353
- it('escapes regex metacharacters in the added line', () => {
354
- const predicate = mechanicalActivationPredicate('', 'Use pattern (a|b).* with $VAR [strictly].\n', SURFACE)!
355
- expect(new RegExp(predicate.pattern!).test('Use pattern (a|b).* with $VAR [strictly].')).toBe(true)
356
- expect(new RegExp(predicate.pattern!).test('Use pattern axb1* with 2VAR strictly.')).toBe(false)
357
- })
358
-
359
- it('returns null when no added line is distinctive enough', () => {
360
- expect(mechanicalActivationPredicate('a\nb\n', 'a\nb\nshort\n', SURFACE)).toBeNull()
361
- expect(mechanicalActivationPredicate('same\n', 'same\n', SURFACE)).toBeNull()
362
- })
363
- })
364
-
365
- describe('recordGepaSeatInnerRun', () => {
366
- const sourceHash = 'a'.repeat(64)
367
- const bridgeHash = 'b'.repeat(64)
368
- const moduleHash = 'c'.repeat(64)
369
- const provenance = (runId: string): OptimizationMethodProvenance => ({
370
- source: {
371
- kind: 'package',
372
- evidence: 'observed',
373
- package: 'gepa',
374
- version: '0.1.4',
375
- sourceUrl: 'https://github.com/gepa-ai/gepa.git',
376
- revision: 'f919db0a622e2e9f9204779b81fe00cc1b2d808f',
377
- sourceSha256: sourceHash,
378
- },
379
- bridge: {
380
- kind: 'package',
381
- evidence: 'observed',
382
- package: 'agent-eval-rpc',
383
- version: '0.126.1',
384
- sourceSha256: bridgeHash,
385
- },
386
- modules: [{ module: 'example.engine', sourceSha256: moduleHash }],
387
- python: { implementation: 'CPython', version: '3.12.3' },
388
- runId,
389
- compatibleRunId: 'compatible-run',
390
- resumed: false,
391
- evaluationCount: 3,
392
- tokenUsage: {
393
- inputTokens: 120,
394
- cachedInputTokens: 20,
395
- outputTokens: 30,
396
- reasoningTokens: 10,
397
- totalTokens: 150,
398
- calls: 2,
399
- },
400
- artifactDir: `/tmp/${runId}`,
401
- })
402
- const run = (runId: string, over: Partial<GepaSeatInnerRun> = {}): GepaSeatInnerRun => {
403
- const result = provenance(runId)
404
- if (
405
- result.source.revision === undefined ||
406
- result.source.sourceSha256 === undefined ||
407
- result.bridge?.sourceSha256 === undefined ||
408
- result.modules === undefined ||
409
- result.python === undefined ||
410
- result.compatibleRunId === undefined ||
411
- result.tokenUsage === undefined
412
- ) {
413
- throw new Error('invalid test provenance')
414
- }
415
- return {
416
- seat: 'gepa-author',
417
- engine: 'gepa',
418
- surface: SURFACE,
419
- generation: 0,
420
- budget: 10,
421
- innerCallCount: 2,
422
- innerScores: [],
423
- bestComposite: 1,
424
- source: {
425
- ...result.source,
426
- revision: result.source.revision,
427
- sourceSha256: result.source.sourceSha256,
428
- },
429
- bridge: { ...result.bridge, sourceSha256: result.bridge.sourceSha256 },
430
- modules: result.modules,
431
- python: result.python,
432
- runId: result.runId,
433
- compatibleRunId: result.compatibleRunId,
434
- resumed: result.resumed,
435
- evaluationCount: result.evaluationCount,
436
- tokenUsage: result.tokenUsage,
437
- artifactDir: result.artifactDir,
438
- totalCostUsd: 0.25,
439
- costProvenance: { kind: 'observed', usd: 0.25 },
440
- accountingComplete: true,
441
- incompleteReasons: [],
442
- durationMs: 5,
443
- ...over,
444
- }
445
- }
446
-
447
- it('writes collision-free immutable records in parallel and preserves the launch record', async () => {
448
- const dir = await mkdtemp(join(tmpdir(), 'gepa-prov-'))
449
- try {
450
- const launchRecord = JSON.stringify({ capturedAt: '2026-07-24T00:00:00.000Z', proposers: [] }, null, 2)
451
- await writeFile(join(dir, 'proposer-provenance.json'), launchRecord)
452
- const paths = await Promise.all(
453
- Array.from({ length: 24 }, (_, index) =>
454
- recordGepaSeatInnerRun(dir, run(`run-${index}`, { generation: index })),
455
- ),
456
- )
457
- expect(new Set(paths).size).toBe(24)
458
- expect(await readFile(join(dir, 'proposer-provenance.json'), 'utf8')).toBe(launchRecord)
459
-
460
- const names = (await readdir(join(dir, GEPA_INNER_RUNS_DIRNAME))).filter((name) => name.endsWith('.json'))
461
- expect(names).toHaveLength(24)
462
- const records = await Promise.all(
463
- names.map(async (name) => JSON.parse(await readFile(join(dir, GEPA_INNER_RUNS_DIRNAME, name), 'utf8'))),
464
- )
465
- expect(new Set(records.map((record) => record.runId))).toEqual(
466
- new Set(Array.from({ length: 24 }, (_, index) => `run-${index}`)),
467
- )
468
- expect(records.every((record) => record.source.sourceSha256 === sourceHash)).toBe(true)
469
- } finally {
470
- await rm(dir, { recursive: true, force: true })
471
- }
472
- })
473
-
474
- it('fails on malformed existing launch or run data', async () => {
475
- const launchDir = await mkdtemp(join(tmpdir(), 'gepa-prov-bad-launch-'))
476
- const runDir = await mkdtemp(join(tmpdir(), 'gepa-prov-bad-run-'))
477
- try {
478
- await writeFile(join(launchDir, 'proposer-provenance.json'), '{not json')
479
- await expect(recordGepaSeatInnerRun(launchDir, run('new-run'))).rejects.toThrow(/malformed JSON/)
480
-
481
- await mkdir(join(runDir, GEPA_INNER_RUNS_DIRNAME), { recursive: true })
482
- await writeFile(join(runDir, GEPA_INNER_RUNS_DIRNAME, 'broken.json'), '[]')
483
- await expect(recordGepaSeatInnerRun(runDir, run('new-run'))).rejects.toThrow(/must contain a JSON object/)
484
- } finally {
485
- await rm(launchDir, { recursive: true, force: true })
486
- await rm(runDir, { recursive: true, force: true })
487
- }
488
- })
489
- })
490
-
491
- // ---------------------------------------------------------------------------
492
- // The seat inside the fan-out generator, against a real temp git repo.
493
- // ---------------------------------------------------------------------------
494
-
495
- const fakeCtx = {} as unknown as DispatchContext
496
- const testOptimizerEnv: NodeJS.ProcessEnv = {
497
- TEST_OPTIMIZER_INPUT_USD_PER_MILLION: '1',
498
- TEST_OPTIMIZER_CACHED_INPUT_USD_PER_MILLION: '0.1',
499
- TEST_OPTIMIZER_CACHE_WRITE_USD_PER_MILLION: '1.25',
500
- TEST_OPTIMIZER_OUTPUT_USD_PER_MILLION: '5',
501
- TEST_OPTIMIZER_MAX_REQUESTS: '10',
502
- TEST_OPTIMIZER_MAX_REQUEST_BYTES: '100000',
503
- TEST_OPTIMIZER_MAX_RESPONSE_BYTES: '100000',
504
- }
505
-
506
- const testOptimizer = officialOptimizerModel({
507
- env: testOptimizerEnv,
508
- envPrefix: 'TEST_OPTIMIZER',
509
- model: 'optimizer-model',
510
- baseUrl: 'http://127.0.0.1:1/v1',
511
- apiKey: 'optimizer-key',
512
- maxCostUsd: 1,
513
- maxOutputTokensPerRequest: 2_000,
514
- callRef: 'test:optimizer-model',
515
- })
516
-
517
- const fullProvenance = (
518
- runId = 'gepa-run',
519
- over: Partial<OptimizationMethodProvenance> = {},
520
- ): OptimizationMethodProvenance => ({
521
- source: {
522
- kind: 'package',
523
- evidence: 'observed',
524
- package: 'gepa',
525
- version: '0.1.4',
526
- sourceUrl: 'https://github.com/gepa-ai/gepa.git',
527
- revision: 'f919db0a622e2e9f9204779b81fe00cc1b2d808f',
528
- sourceSha256: '1'.repeat(64),
529
- },
530
- bridge: {
531
- kind: 'package',
532
- evidence: 'observed',
533
- package: 'agent-eval-rpc',
534
- version: '0.126.1',
535
- sourceSha256: '2'.repeat(64),
536
- },
537
- modules: [{ module: 'custom_gepa_engines', sourceSha256: '3'.repeat(64) }],
538
- python: { implementation: 'CPython', version: '3.12.3' },
539
- runId,
540
- compatibleRunId: 'compatible-gepa-run',
541
- resumed: false,
542
- evaluationCount: 3,
543
- tokenUsage: {
544
- inputTokens: 100,
545
- cachedInputTokens: 10,
546
- cacheWriteInputTokens: 5,
547
- outputTokens: 25,
548
- reasoningTokens: 8,
549
- totalTokens: 125,
550
- calls: 2,
551
- },
552
- artifactDir: `/tmp/${runId}`,
553
- ...over,
554
- })
555
-
556
- /** Mimics the adapter's loop: score the seed and each provided candidate via
557
- * the seat's dispatch + judge, return the best-scoring candidate — exactly
558
- * the contract gepaOptimizationMethod fulfills through the Python bridge. */
559
- const fakeGepaFactory =
560
- (candidates: string[], observed?: { config?: unknown }): GepaMethodFactory =>
561
- (config) => {
562
- if (observed) observed.config = config
563
- return {
564
- name: config.name ?? 'fake-gepa',
565
- async optimize(input) {
566
- const judge = input.judges[0]!
567
- const scenario = input.trainScenarios[0]!
568
- let best = { surface: input.baselineSurface as string, composite: -Infinity }
569
- for (const candidate of [input.baselineSurface as string, ...candidates]) {
570
- const artifact = await input.dispatchWithSurface(candidate, scenario, fakeCtx)
571
- const score = await judge.score({ artifact, scenario, signal: new AbortController().signal })
572
- if (score.composite > best.composite) best = { surface: candidate, composite: score.composite }
573
- }
574
- return {
575
- winnerSurface: best.surface,
576
- cost: {
577
- totalCostUsd: 0.125,
578
- costProvenance: { kind: 'observed', usd: 0.125 },
579
- accountingComplete: true,
580
- incompleteReasons: [],
581
- },
582
- durationMs: 1,
583
- provenance: fullProvenance(),
584
- }
585
- },
586
- }
587
- }
588
-
589
- describe('fanOutLoopsGenerator with the gepa seat', () => {
590
- let loopsRepo: string
591
- let outDir: string
592
- let driverWt: string
593
-
594
- const git = async (args: string[], cwd: string): Promise<string> =>
595
- (await runOk('git', ['-C', cwd, ...args])).stdout.trim()
596
-
597
- const SEED = '# worker coding system\nkeep tests green\n'
598
-
599
- beforeEach(async () => {
600
- loopsRepo = await mkdtemp(join(tmpdir(), 'gepa-repo-'))
601
- outDir = await mkdtemp(join(tmpdir(), 'gepa-out-'))
602
- await runOk('git', ['init', '-q', '-b', 'main', loopsRepo])
603
- await runOk('git', ['-C', loopsRepo, 'config', 'core.hooksPath', '/dev/null'])
604
- await git(['config', 'user.email', 't@t.dev'], loopsRepo)
605
- await git(['config', 'user.name', 'T'], loopsRepo)
606
- await mkdir(join(loopsRepo, 'extensions', 'pi', 'prompts'), { recursive: true })
607
- await writeFile(join(loopsRepo, SURFACE), SEED)
608
- await git(['add', '-A'], loopsRepo)
609
- await git(['commit', '-q', '-m', 'init'], loopsRepo)
610
- driverWt = join(outDir, 'driver-wt')
611
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', driverWt, 'HEAD'])
612
- })
613
-
614
- afterEach(async () => {
615
- await rm(outDir, { recursive: true, force: true })
616
- await rm(loopsRepo, { recursive: true, force: true })
617
- })
618
-
619
- const baseConfig = (proposers: ProposerSpec[], over: Partial<OuterLoopConfig> = {}): OuterLoopConfig => ({
620
- ...defaultRound4Config(),
621
- loopsRepo,
622
- outDir,
623
- populationSize: proposers.length,
624
- proposers,
625
- prefilter: { enabled: true, smokeInstance: 'cheapest-of-set', requireResolved: false },
626
- ...over,
627
- })
628
-
629
- const generatorArgs = (candidateIndex: number) => ({
630
- worktreePath: driverWt,
631
- findings: [],
632
- maxShots: 1,
633
- signal: new AbortController().signal,
634
- generation: 0,
635
- candidateIndex,
636
- })
637
-
638
- const smokeVerdict = (over: Partial<SmokeVerdict> = {}): SmokeVerdict => ({
639
- iid: 'astropy__astropy-13033',
640
- pass: true,
641
- reason: 'smoke ok',
642
- resolved: false,
643
- patchLines: 3,
644
- wallS: 5,
645
- verifyPass: false,
646
- ...over,
647
- })
648
-
649
- it('construction fails loud without the smoke runner (the inner evaluator)', () => {
650
- expect(() => fanOutLoopsGenerator(baseConfig([seat()]))).toThrow(/inner evaluator/)
651
- expect(() =>
652
- fanOutLoopsGenerator(baseConfig([seat()]), {
653
- smokeRunner: async () => smokeVerdict(),
654
- smokeInstanceId: 'astropy__astropy-13033',
655
- }),
656
- ).toThrow(/immutable runner and judge references/)
657
- expect(() => fanOutLoopsGenerator(baseConfig([{ name: 'no-seat-kind' }]))).toThrow(/neither a harness nor an engine/)
658
- })
659
-
660
- it('materializes each candidate into the scratch surface, applies the winner through the normal prefilter path, and records provenance', async () => {
661
- const WINNER = `${SEED}Always run the neighboring test file before finalizing.\n`
662
- const LOSER = `${SEED}delete all tests\n`
663
- const seen: Array<{
664
- content: string
665
- scratch: string
666
- evaluationKey: string
667
- hasCostLedger: boolean
668
- }> = []
669
- const smokeRunner: SmokeRunner = async ({
670
- scratchPath,
671
- evaluationKey,
672
- costLedger,
673
- }) => {
674
- const content = await readFile(join(scratchPath, SURFACE), 'utf8')
675
- seen.push({
676
- content,
677
- scratch: scratchPath,
678
- evaluationKey,
679
- hasCostLedger: costLedger !== undefined,
680
- })
681
- // The winner candidate resolves; the seed gets verify-pass only; the
682
- // loser gets nothing — exercising resolve-dominates + tiebreak.
683
- if (content === WINNER) return smokeVerdict({ resolved: true, verifyPass: true })
684
- if (content === SEED) return smokeVerdict({ verifyPass: true })
685
- return smokeVerdict()
686
- }
687
- const observed: { config?: unknown } = {}
688
- const config = baseConfig([seat({ maxMetricCalls: 5 })], { activationGate: true })
689
- const gen = fanOutLoopsGenerator(config, {
690
- ...IMPLEMENTATION_REFS,
691
- smokeRunner,
692
- smokeInstanceId: 'astropy__astropy-13033',
693
- scoreSplit: { privateInstances: ['django__django-11532'] },
694
- gepaMethodFactory: fakeGepaFactory([LOSER, WINNER], observed),
695
- gepaOptimizer: testOptimizer,
696
- })
697
-
698
- const result = await gen.generate(generatorArgs(0))
699
-
700
- expect(result).toMatchObject({ applied: true, label: 'gepa-author' })
701
- expect(result.rationale).toContain('engine gepa')
702
- // 3 isolated inner calls (seed, loser, winner) + 1 stage-B prefilter smoke
703
- // on the final candidate, all outside the driver worktree.
704
- expect(seen).toHaveLength(4)
705
- for (const call of seen) expect(call.scratch).not.toBe(driverWt)
706
- expect(seen.map((c) => c.content)).toEqual([SEED, LOSER, WINNER, WINNER])
707
- expect(new Set(seen.slice(0, 3).map((call) => call.scratch)).size).toBe(3)
708
- expect(seen.slice(0, 3).every((call) => call.evaluationKey.startsWith('inner-'))).toBe(true)
709
- expect(seen[3]!.evaluationKey).toBe('candidate-0')
710
- expect(seen.slice(0, 3).every((call) => call.hasCostLedger)).toBe(true)
711
- // The winner landed on the driver worktree with the mechanical predicate.
712
- expect(await readFile(join(driverWt, SURFACE), 'utf8')).toBe(WINNER)
713
- const predicate = parseActivationPredicate(await readFile(join(driverWt, ACTIVATION_PREDICATE_RELPATH), 'utf8'))
714
- expect(predicate.ok).toBe(true)
715
- // Only the surface + predicate changed.
716
- const changed = (await runOk('git', ['-C', driverWt, 'status', '--porcelain', '--untracked-files=all'])).stdout
717
- .split('\n')
718
- .map((l) => l.slice(3).trim())
719
- .filter(Boolean)
720
- expect(changed.sort()).toEqual([ACTIVATION_PREDICATE_RELPATH, SURFACE].sort())
721
- // Budget threaded into the adapter recipe.
722
- const incumbentCommit = await git(['rev-parse', 'HEAD'], driverWt)
723
- expect(observed.config).toMatchObject({
724
- evaluationId: gepaSeatEvaluationId({
725
- smokeInstanceId: 'astropy__astropy-13033',
726
- dispatchTimeoutMs: config.dispatchTimeoutMs,
727
- incumbentCommit,
728
- ...IMPLEMENTATION_REFS,
729
- }),
730
- recipe: { kind: 'engine', run: { engine: 'gepa', maxEvaluations: 5 } },
731
- optimizer: testOptimizer,
732
- resume: 'if-compatible',
733
- trustResumeState: true,
734
- })
735
- // Inner-run provenance: the seat-local file plus one immutable shared record.
736
- const inner = JSON.parse(await readFile(join(outDir, 'gepa-seat', 'gen0-gepa-author', 'inner-provenance.json'), 'utf8'))
737
- expect(inner).toMatchObject({
738
- seat: 'gepa-author',
739
- engine: 'gepa',
740
- budget: 5,
741
- innerCallCount: 3,
742
- bestComposite: 1.25,
743
- source: {
744
- package: 'gepa',
745
- version: '0.1.4',
746
- revision: 'f919db0a622e2e9f9204779b81fe00cc1b2d808f',
747
- sourceSha256: '1'.repeat(64),
748
- },
749
- bridge: {
750
- package: 'agent-eval-rpc',
751
- version: '0.126.1',
752
- sourceSha256: '2'.repeat(64),
753
- },
754
- modules: [{ module: 'custom_gepa_engines', sourceSha256: '3'.repeat(64) }],
755
- python: { implementation: 'CPython', version: '3.12.3' },
756
- runId: 'gepa-run',
757
- compatibleRunId: 'compatible-gepa-run',
758
- resumed: false,
759
- evaluationCount: 3,
760
- tokenUsage: {
761
- inputTokens: 100,
762
- cachedInputTokens: 10,
763
- cacheWriteInputTokens: 5,
764
- outputTokens: 25,
765
- reasoningTokens: 8,
766
- totalTokens: 125,
767
- calls: 2,
768
- },
769
- artifactDir: '/tmp/gepa-run',
770
- totalCostUsd: 0.125,
771
- accountingComplete: true,
772
- incompleteReasons: [],
773
- })
774
- expect(inner.innerScores.map((s: { composite: number }) => s.composite)).toEqual([0.25, 0, 1.25])
775
- const records = await readdir(join(outDir, GEPA_INNER_RUNS_DIRNAME))
776
- expect(records.filter((name) => name.endsWith('.json'))).toHaveLength(1)
777
- expect(JSON.parse(await readFile(join(outDir, GEPA_INNER_RUNS_DIRNAME, records[0]!), 'utf8'))).toEqual(inner)
778
- expect(gen.drainPrefilterKills()).toEqual([])
779
- })
780
-
781
- it('isolates concurrent candidate evaluations while preserving parallel execution', async () => {
782
- const candidateA = `${SEED}candidate A keeps its own workspace\n`
783
- const candidateB = `${SEED}candidate B keeps its own workspace\n`
784
- let releaseBoth!: () => void
785
- const bothStarted = new Promise<void>((resolve) => {
786
- releaseBoth = resolve
787
- })
788
- let started = 0
789
- const innerSeen: Array<{ content: string; scratchPath: string }> = []
790
- const smokeRunner: SmokeRunner = async ({
791
- scratchPath,
792
- evaluationKey,
793
- }) => {
794
- if (evaluationKey.startsWith('inner-')) {
795
- started += 1
796
- if (started === 2) releaseBoth()
797
- await bothStarted
798
- }
799
- const content = await readFile(join(scratchPath, SURFACE), 'utf8')
800
- if (evaluationKey.startsWith('inner-')) {
801
- innerSeen.push({ content, scratchPath })
802
- }
803
- return smokeVerdict({ resolved: content === candidateA })
804
- }
805
- const parallelFactory: GepaMethodFactory = () => ({
806
- name: 'parallel-gepa',
807
- async optimize(input) {
808
- const scenario = input.trainScenarios[0]!
809
- await Promise.all([
810
- input.dispatchWithSurface(candidateA, scenario, fakeCtx),
811
- input.dispatchWithSurface(candidateB, scenario, fakeCtx),
812
- ])
813
- return {
814
- winnerSurface: candidateA,
815
- cost: {
816
- totalCostUsd: 0,
817
- costProvenance: { kind: 'observed', usd: 0 },
818
- accountingComplete: true,
819
- incompleteReasons: [],
820
- },
821
- durationMs: 1,
822
- provenance: fullProvenance('parallel-gepa', { evaluationCount: 2 }),
823
- }
824
- },
825
- })
826
- const gen = fanOutLoopsGenerator(baseConfig([seat({ maxMetricCalls: 2 })]), {
827
- ...IMPLEMENTATION_REFS,
828
- smokeRunner,
829
- smokeInstanceId: 'astropy__astropy-13033',
830
- scoreSplit: null,
831
- gepaMethodFactory: parallelFactory,
832
- })
833
-
834
- const result = await gen.generate(generatorArgs(0))
835
-
836
- expect(result.applied).toBe(true)
837
- expect(innerSeen.map((entry) => entry.content).sort()).toEqual(
838
- [candidateA, candidateB].sort(),
839
- )
840
- expect(new Set(innerSeen.map((entry) => entry.scratchPath)).size).toBe(2)
841
- })
842
-
843
- it('uses explicit immutable refs so captured runner or judge behavior cannot share resume state', async () => {
844
- const config = baseConfig([seat()])
845
- const incumbentCommit = await git(['rev-parse', 'HEAD'], driverWt)
846
- const makeRunner = (resolved: boolean): SmokeRunner =>
847
- async () => smokeVerdict({ resolved })
848
- const runnerV1 = makeRunner(false)
849
- const runnerV2 = makeRunner(true)
850
- const common = {
851
- smokeInstanceId: 'astropy__astropy-13033',
852
- dispatchTimeoutMs: config.dispatchTimeoutMs,
853
- incumbentCommit,
854
- }
855
- const original = gepaSeatEvaluationId({ ...common, ...IMPLEMENTATION_REFS })
856
- const changedRunner = gepaSeatEvaluationId({
857
- ...common,
858
- ...IMPLEMENTATION_REFS,
859
- runnerImplementationRef: `sha256:${'c'.repeat(64)}`,
860
- })
861
- const changedJudge = gepaSeatEvaluationId({
862
- ...common,
863
- ...IMPLEMENTATION_REFS,
864
- judgeImplementationRef: `sha256:${'d'.repeat(64)}`,
865
- })
866
- const changedIncumbent = gepaSeatEvaluationId({
867
- ...common,
868
- ...IMPLEMENTATION_REFS,
869
- incumbentCommit: 'e'.repeat(40),
870
- })
871
- const smokeArgs: Parameters<SmokeRunner>[0] = {
872
- scratchPath: driverWt,
873
- generation: 0,
874
- proposer: seat(),
875
- evaluationKey: 'identity-test',
876
- }
877
-
878
- expect(runnerV1.toString()).toBe(runnerV2.toString())
879
- expect((await runnerV1(smokeArgs)).resolved).toBe(false)
880
- expect((await runnerV2(smokeArgs)).resolved).toBe(true)
881
- expect(original).toMatch(
882
- /^swe-arena-gepa-seat\|smoke=astropy__astropy-13033\|incumbent=[a-f0-9]{40,64}\|runner=sha256:[a-f0-9]{64}\|judge=sha256:[a-f0-9]{64}\|dispatchTimeoutMs=\d+$/,
883
- )
884
- expect(changedRunner).not.toBe(original)
885
- expect(changedJudge).not.toBe(original)
886
- expect(changedIncumbent).not.toBe(original)
887
- expect(() =>
888
- gepaSeatEvaluationId({
889
- ...common,
890
- ...IMPLEMENTATION_REFS,
891
- runnerImplementationRef: 'runner-v2',
892
- }),
893
- ).toThrow(/runnerImplementationRef must be an immutable sha256 reference/)
894
- })
895
-
896
- it('records but rejects a winner when cost accounting is incomplete', async () => {
897
- const WINNER = `${SEED}Always run the neighboring test file before finalizing.\n`
898
- const incomplete: GepaMethodFactory = () => ({
899
- name: 'incomplete-cost',
900
- async optimize(input) {
901
- await input.dispatchWithSurface(WINNER, input.trainScenarios[0]!, fakeCtx)
902
- return {
903
- winnerSurface: WINNER,
904
- cost: {
905
- totalCostUsd: 0.25,
906
- costProvenance: { kind: 'uncaptured', usd: null },
907
- accountingComplete: false,
908
- incompleteReasons: ['optimizer model receipt missing'],
909
- },
910
- durationMs: 1,
911
- provenance: fullProvenance('incomplete-run', { evaluationCount: 1 }),
912
- }
913
- },
914
- })
915
- const gen = fanOutLoopsGenerator(baseConfig([seat()]), {
916
- ...IMPLEMENTATION_REFS,
917
- smokeRunner: async () => smokeVerdict({ resolved: true }),
918
- smokeInstanceId: 'astropy__astropy-13033',
919
- scoreSplit: null,
920
- gepaMethodFactory: incomplete,
921
- })
922
-
923
- await expect(gen.generate(generatorArgs(0))).rejects.toThrow(
924
- /cost accounting is incomplete: optimizer model receipt missing/,
925
- )
926
- expect(await readFile(join(driverWt, SURFACE), 'utf8')).toBe(SEED)
927
- const inner = JSON.parse(
928
- await readFile(join(outDir, 'gepa-seat', 'gen0-gepa-author', 'inner-provenance.json'), 'utf8'),
929
- )
930
- expect(inner).toMatchObject({
931
- runId: 'incomplete-run',
932
- totalCostUsd: 0.25,
933
- accountingComplete: false,
934
- incompleteReasons: ['optimizer model receipt missing'],
935
- })
936
- })
937
-
938
- it('enforces the inner-call budget cap fail-closed', async () => {
939
- const runaway: GepaMethodFactory = () => ({
940
- name: 'runaway',
941
- async optimize(input) {
942
- const results = await Promise.allSettled(
943
- Array.from({ length: 4 }, (_, index) =>
944
- input.dispatchWithSurface(
945
- `${SEED}candidate ${index}\n`,
946
- input.trainScenarios[0]!,
947
- fakeCtx,
948
- ),
949
- ),
950
- )
951
- const rejected = results.find(
952
- (result): result is PromiseRejectedResult => result.status === 'rejected',
953
- )
954
- if (rejected) throw rejected.reason
955
- return {
956
- winnerSurface: SEED,
957
- cost: {
958
- totalCostUsd: 0,
959
- costProvenance: { kind: 'uncaptured', usd: null },
960
- accountingComplete: false,
961
- incompleteReasons: [],
962
- },
963
- durationMs: 1,
964
- provenance: fullProvenance('runaway'),
965
- }
966
- },
967
- })
968
- const gen = fanOutLoopsGenerator(baseConfig([seat({ maxMetricCalls: 3 })]), {
969
- ...IMPLEMENTATION_REFS,
970
- smokeRunner: async () => smokeVerdict(),
971
- smokeInstanceId: 'astropy__astropy-13033',
972
- scoreSplit: null,
973
- gepaMethodFactory: runaway,
974
- })
975
- await expect(gen.generate(generatorArgs(0))).rejects.toThrow(/inner-call budget 3 exhausted/)
976
- })
977
-
978
- it('refuses to feed a PRIVATE smoke verdict to the bridge', async () => {
979
- const gen = fanOutLoopsGenerator(baseConfig([seat()]), {
980
- ...IMPLEMENTATION_REFS,
981
- smokeRunner: async () => smokeVerdict({ iid: 'django__django-11532' }),
982
- smokeInstanceId: 'astropy__astropy-13033',
983
- scoreSplit: { privateInstances: ['django__django-11532'] },
984
- gepaMethodFactory: fakeGepaFactory([`${SEED}x line long enough\n`]),
985
- })
986
- await expect(gen.generate(generatorArgs(0))).rejects.toThrow(/PRIVATE instance django__django-11532/)
987
- })
988
-
989
- it('declines the slot without a kill when GEPA returns the seed unchanged', async () => {
990
- const gen = fanOutLoopsGenerator(baseConfig([seat()]), {
991
- ...IMPLEMENTATION_REFS,
992
- smokeRunner: async () => smokeVerdict(),
993
- smokeInstanceId: 'astropy__astropy-13033',
994
- scoreSplit: null,
995
- gepaMethodFactory: fakeGepaFactory([]),
996
- })
997
- const result = await gen.generate(generatorArgs(0))
998
- expect(result.applied).toBe(false)
999
- expect(result.summary).toContain('equals the seed')
1000
- expect(gen.drainPrefilterKills()).toEqual([])
1001
- expect((await runOk('git', ['-C', driverWt, 'status', '--porcelain'])).stdout.trim()).toBe('')
1002
- })
1003
- })
1004
-
1005
- // ---------------------------------------------------------------------------
1006
- // Integration: one real Node-to-Python-to-score roundtrip through the installed
1007
- // bridge. The TypeScript adapter is a compile-time package dependency.
1008
- // ---------------------------------------------------------------------------
1009
-
1010
- const pythonBridgeReady = (python: string): { ok: boolean; reason: string } => {
1011
- const probe = spawnSync(python, [
1012
- '-c',
1013
- 'import agent_eval_rpc.gepa_bridge; from gepa.optimize_anything import optimize_anything, OptimizeAnythingConfig',
1014
- ])
1015
- if (probe.status !== 0) {
1016
- return { ok: false, reason: `python bridge unavailable: ${String(probe.stderr).trim().split('\n').pop()}` }
1017
- }
1018
- return { ok: true, reason: '' }
1019
- }
1020
-
1021
- describe('integration: real adapter roundtrip', () => {
1022
- it('runs a metered optimizer through the real bridge and applies its winner', async (ctx) => {
1023
- const python = process.env.AGENT_EVAL_TEST_PYTHON ?? DEFAULT_GEPA_PYTHON
1024
- const pythonRuntime = pythonBridgeReady(python)
1025
- if (!pythonRuntime.ok) {
1026
- ctx.skip(`skip-with-reason: ${pythonRuntime.reason}`)
1027
- return
1028
- }
1029
-
1030
- const winner = 'tiny synthetic surface\nAlways run the neighboring test before finalizing.'
1031
- const modelServer = createServer((_request, response) => {
1032
- response.writeHead(200, { 'content-type': 'application/json' })
1033
- response.end(
1034
- JSON.stringify({
1035
- choices: [
1036
- {
1037
- message: {
1038
- role: 'assistant',
1039
- content: `\`\`\`\n${winner}\`\`\``,
1040
- },
1041
- },
1042
- ],
1043
- usage: {
1044
- prompt_tokens: 20,
1045
- completion_tokens: 20,
1046
- total_tokens: 40,
1047
- },
1048
- }),
1049
- )
1050
- })
1051
- await new Promise<void>((resolve, reject) => {
1052
- modelServer.once('error', reject)
1053
- modelServer.listen(0, '127.0.0.1', resolve)
1054
- })
1055
- const address = modelServer.address()
1056
- if (!address || typeof address === 'string') {
1057
- throw new Error('test optimizer server did not bind')
1058
- }
1059
-
1060
- const loopsRepo = await mkdtemp(join(tmpdir(), 'gepa-int-repo-'))
1061
- const outDir = await mkdtemp(join(tmpdir(), 'gepa-int-out-'))
1062
- try {
1063
- await runOk('git', ['init', '-q', '-b', 'main', loopsRepo])
1064
- await runOk('git', ['-C', loopsRepo, 'config', 'core.hooksPath', '/dev/null'])
1065
- await runOk('git', ['-C', loopsRepo, 'config', 'user.email', 't@t.dev'])
1066
- await runOk('git', ['-C', loopsRepo, 'config', 'user.name', 'T'])
1067
- await mkdir(join(loopsRepo, 'extensions', 'pi', 'prompts'), { recursive: true })
1068
- await writeFile(join(loopsRepo, SURFACE), 'tiny synthetic surface\n')
1069
- await runOk('git', ['-C', loopsRepo, 'add', '-A'])
1070
- await runOk('git', ['-C', loopsRepo, 'commit', '-q', '-m', 'init'])
1071
- const driverWt = join(outDir, 'driver-wt')
1072
- await runOk('git', ['-C', loopsRepo, 'worktree', 'add', '--detach', driverWt, 'HEAD'])
1073
-
1074
- const gen = fanOutLoopsGenerator(
1075
- {
1076
- ...defaultRound4Config(),
1077
- loopsRepo,
1078
- outDir,
1079
- populationSize: 1,
1080
- proposers: [seat({ maxMetricCalls: 10, python })],
1081
- prefilter: { enabled: true, smokeInstance: 'cheapest-of-set', requireResolved: false },
1082
- },
1083
- {
1084
- ...IMPLEMENTATION_REFS,
1085
- smokeRunner: async ({ scratchPath }) => {
1086
- const candidate = await readFile(join(scratchPath, SURFACE), 'utf8')
1087
- return {
1088
- iid: 'astropy__astropy-13033',
1089
- pass: true,
1090
- reason: 'stub smoke (integration)',
1091
- resolved: candidate === winner,
1092
- patchLines: 1,
1093
- wallS: 0,
1094
- verifyPass: true,
1095
- }
1096
- },
1097
- smokeInstanceId: 'astropy__astropy-13033',
1098
- scoreSplit: null,
1099
- gepaOptimizer: officialOptimizerModel({
1100
- env: testOptimizerEnv,
1101
- envPrefix: 'TEST_OPTIMIZER',
1102
- model: 'test-optimizer',
1103
- baseUrl: `http://127.0.0.1:${address.port}/v1`,
1104
- apiKey: 'local-test-key',
1105
- maxCostUsd: 1,
1106
- maxOutputTokensPerRequest: 2_000,
1107
- callRef: 'test:integration-optimizer',
1108
- }),
1109
- },
1110
- )
1111
- const result = await gen.generate({
1112
- worktreePath: driverWt,
1113
- findings: [],
1114
- maxShots: 1,
1115
- signal: new AbortController().signal,
1116
- generation: 0,
1117
- candidateIndex: 0,
1118
- })
1119
- const inner = JSON.parse(
1120
- await readFile(join(outDir, 'gepa-seat', 'gen0-gepa-author', 'inner-provenance.json'), 'utf8'),
1121
- )
1122
- expect(result.applied, `${result.summary}\n${JSON.stringify(inner.innerScores, null, 2)}`).toBe(true)
1123
- expect(await readFile(join(driverWt, SURFACE), 'utf8')).toBe(winner)
1124
- expect(inner.innerCallCount).toBeGreaterThanOrEqual(1)
1125
- expect(inner.accountingComplete).toBe(true)
1126
- expect(inner.incompleteReasons).toEqual([])
1127
- expect(inner.tokenUsage.calls).toBeGreaterThan(0)
1128
- } finally {
1129
- await new Promise<void>((resolve, reject) =>
1130
- modelServer.close((error) => (error ? reject(error) : resolve())),
1131
- )
1132
- await rm(outDir, { recursive: true, force: true })
1133
- await rm(loopsRepo, { recursive: true, force: true })
1134
- }
1135
- }, 300_000)
1136
- })