@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,645 +0,0 @@
1
- /**
2
- * Factory-bench runner tests: a synthetic two-commit "mirror"
3
- * repo stands in for the real local mirrors, and its judge harness prints
4
- * vitest-format summary lines, so the full archive → apply → overlay → run →
5
- * parse pipeline is exercised through the same container boundary as real
6
- * factory instances.
7
- */
8
-
9
- import { mkdtemp, mkdir, readFile, rm, writeFile } from 'node:fs/promises'
10
- import { existsSync } from 'node:fs'
11
- import { tmpdir } from 'node:os'
12
- import { join } from 'node:path'
13
- import { fileURLToPath } from 'node:url'
14
- import { afterAll, beforeAll, describe, expect, it } from 'vitest'
15
- import {
16
- calibrateFactoryInstance,
17
- goldImplPatch,
18
- } from './calibrate.ts'
19
- import {
20
- judgeFactoryPatch,
21
- parseFactoryJudgeResult,
22
- parseVitestSummary,
23
- } from './factory-judge-child.mts'
24
- import { runFactoryCommand } from './factory-command-container.ts'
25
- import {
26
- FACTORY_INSTANCES_DIR,
27
- loadFactoryInstance,
28
- loadFactoryInstances,
29
- type LoadedFactoryInstance,
30
- } from './fixtures.ts'
31
- import { run, runOk } from './proc.ts'
32
- import { FACTORY_BASE_REF, factoryLedgerKeys, materializeFactoryWorkspace } from './run-experiment.mts'
33
-
34
- // ---------------------------------------------------------------------------
35
- // Synthetic mirror: base commit (README + package.json, no impl), then three
36
- // judge_ref variants layered on top —
37
- // good: impl (add+mul) + judge tests that pin both → calibrates
38
- // trivial: judge tests that pass with NO impl → base passes → reject
39
- // too-hard: judge tests demanding an un-shipped sub() → gold fails → reject
40
- // The judge harness is plain node (no vitest dependency) that PRINTS vitest's
41
- // summary format — the protocol under test is the parse, not vitest itself.
42
- // ---------------------------------------------------------------------------
43
-
44
- const JUDGE_HARNESS = `
45
- let passed = 0
46
- let failed = 0
47
- const check = (name, fn) => {
48
- try {
49
- if (fn()) passed += 1
50
- else failed += 1
51
- } catch {
52
- failed += 1
53
- }
54
- }
55
- export async function runChecks(checks) {
56
- let lib
57
- try {
58
- lib = await import('../lib.mjs')
59
- } catch {
60
- console.log(' Test Files 1 failed (1)')
61
- console.log(' Tests no tests')
62
- process.exit(1)
63
- }
64
- for (const [name, fn] of checks) check(name, () => fn(lib))
65
- const total = passed + failed
66
- const parts = []
67
- if (failed > 0) parts.push(failed + ' failed')
68
- if (passed > 0) parts.push(passed + ' passed')
69
- console.log(' Test Files ' + (failed > 0 ? '1 failed (1)' : '1 passed (1)'))
70
- console.log(' Tests ' + parts.join(' | ') + ' (' + total + ')')
71
- process.exit(failed > 0 ? 1 : 0)
72
- }
73
- `
74
-
75
- const GOOD_TESTS = `
76
- import { runChecks } from './harness.mjs'
77
- await runChecks([
78
- ['add', (lib) => lib.add(1, 2) === 3],
79
- ['mul', (lib) => lib.mul(2, 3) === 6],
80
- ])
81
- `
82
-
83
- const TRIVIAL_TESTS = `
84
- console.log(' Test Files 1 passed (1)')
85
- console.log(' Tests 2 passed (2)')
86
- `
87
-
88
- const TOO_HARD_TESTS = `
89
- import { runChecks } from './harness.mjs'
90
- await runChecks([
91
- ['add', (lib) => lib.add(1, 2) === 3],
92
- ['sub', (lib) => lib.sub(3, 1) === 2],
93
- ])
94
- `
95
-
96
- const SYNTHETIC_COMMAND_IMAGE =
97
- 'node:24-bookworm-slim@sha256:cb4e8f7c443347358b7875e717c29e27bf9befc8f5a26cf18af3c3dec80e58c5'
98
-
99
- function credentialProbe(hostRoot: string): string {
100
- const knownAmbientFiles = [
101
- join(hostRoot, 'ambient-home', '.config', 'gh', 'hosts.yml'),
102
- join(hostRoot, 'ambient-home', '.aws', 'credentials'),
103
- join(hostRoot, 'ambient-home', '.npmrc'),
104
- join(hostRoot, 'ambient-xdg', 'gh', 'hosts.yml'),
105
- '/home/drew/.config/gh/hosts.yml',
106
- '/home/drew/.aws/credentials',
107
- '/home/drew/.npmrc',
108
- '/home/runner/.config/gh/hosts.yml',
109
- '/home/runner/.aws/credentials',
110
- '/home/runner/.npmrc',
111
- '/root/.config/gh/hosts.yml',
112
- '/root/.aws/credentials',
113
- '/root/.npmrc',
114
- ]
115
- return `
116
- import { lstatSync, readFileSync, writeFileSync } from 'node:fs'
117
- import { join } from 'node:path'
118
-
119
- const secretNames = [
120
- 'ARBITRARY_FACTORY_SECRET',
121
- 'GH_TOKEN',
122
- 'TANGLETOOLS_GH_TOKEN',
123
- 'DREW_GH_TOKEN',
124
- 'AWS_SECRET_ACCESS_KEY',
125
- 'NPM_TOKEN',
126
- 'NODE_AUTH_TOKEN',
127
- 'SSH_AUTH_SOCK',
128
- 'GPG_AGENT_INFO',
129
- 'DBUS_SESSION_BUS_ADDRESS',
130
- ]
131
- const knownAmbientFiles = ${JSON.stringify(knownAmbientFiles)}
132
- const leakedEnv = secretNames.filter((name) => process.env[name] !== undefined)
133
- const procLeaks = []
134
- const procRootFiles = []
135
- let pid = process.pid
136
- for (let depth = 0; depth < 32; depth += 1) {
137
- let stat
138
- try {
139
- stat = readFileSync('/proc/' + pid + '/stat', 'utf8')
140
- } catch {
141
- break
142
- }
143
- try {
144
- const environ = readFileSync('/proc/' + pid + '/environ')
145
- const names = secretNames.filter((name) => environ.includes(Buffer.from(name + '=')))
146
- if (names.length > 0) procLeaks.push({ depth, pid, names })
147
- } catch {}
148
- for (const path of knownAmbientFiles) {
149
- const procPath = '/proc/' + pid + '/root' + path
150
- try {
151
- if (readFileSync(procPath).length > 0) procRootFiles.push(procPath)
152
- } catch {}
153
- }
154
- const close = stat.lastIndexOf(')')
155
- const parent = Number(stat.slice(close + 2).split(' ')[1])
156
- if (!Number.isSafeInteger(parent) || parent < 1 || parent === pid) break
157
- pid = parent
158
- }
159
- const candidateFiles = [
160
- process.env.HOME && join(process.env.HOME, '.config', 'gh', 'hosts.yml'),
161
- process.env.HOME && join(process.env.HOME, '.aws', 'credentials'),
162
- process.env.HOME && join(process.env.HOME, '.npmrc'),
163
- process.env.XDG_CONFIG_HOME && join(process.env.XDG_CONFIG_HOME, 'gh', 'hosts.yml'),
164
- process.env.NPM_CONFIG_USERCONFIG,
165
- process.env.npm_config_userconfig,
166
- process.env.NPM_CONFIG_GLOBALCONFIG,
167
- process.env.npm_config_globalconfig,
168
- ...knownAmbientFiles,
169
- ...procRootFiles,
170
- ].filter(Boolean)
171
- const credentialFiles = []
172
- for (const path of new Set(candidateFiles)) {
173
- try {
174
- const content = readFileSync(path, 'utf8')
175
- if (content.trim()) credentialFiles.push(path)
176
- } catch {}
177
- }
178
- const hiddenJudgeFiles = ['tests/judge.mjs', 'tests/harness.mjs'].filter((path) => {
179
- try {
180
- return readFileSync(path).length > 0
181
- } catch {
182
- return false
183
- }
184
- })
185
- const credentialSockets = ['/var/run/docker.sock', '/run/docker.sock'].filter((path) => {
186
- try {
187
- return lstatSync(path).isSocket()
188
- } catch {
189
- return false
190
- }
191
- })
192
- let rootWritable = false
193
- try {
194
- writeFileSync('/factory-root-write-probe', 'unexpected')
195
- rootWritable = true
196
- } catch {}
197
- writeFileSync(
198
- 'credential-probe-result.json',
199
- JSON.stringify({
200
- leakedEnv,
201
- procLeaks,
202
- procRootFiles,
203
- credentialFiles,
204
- hiddenJudgeFiles,
205
- credentialSockets,
206
- rootWritable,
207
- uid: process.getuid?.(),
208
- home: process.env.HOME,
209
- xdgConfigHome: process.env.XDG_CONFIG_HOME,
210
- npmUserConfig: process.env.NPM_CONFIG_USERCONFIG ?? process.env.npm_config_userconfig,
211
- }),
212
- )
213
- if (
214
- leakedEnv.length > 0 ||
215
- procLeaks.length > 0 ||
216
- procRootFiles.length > 0 ||
217
- credentialFiles.length > 0 ||
218
- hiddenJudgeFiles.length > 0 ||
219
- credentialSockets.length > 0 ||
220
- rootWritable ||
221
- process.getuid?.() === 0
222
- ) process.exit(86)
223
- `
224
- }
225
-
226
- const IMPL = `export const add = (a, b) => a + b\nexport const mul = (a, b) => a * b\n`
227
-
228
- interface SyntheticMirror {
229
- mirror: string
230
- baseCommit: string
231
- refs: { good: string; trivial: string; tooHard: string }
232
- }
233
-
234
- async function git(cwd: string, ...argv: string[]): Promise<string> {
235
- const res = await runOk('git', ['-C', cwd, ...argv])
236
- return res.stdout.trim()
237
- }
238
-
239
- async function commitAll(cwd: string, msg: string): Promise<string> {
240
- await git(cwd, 'add', '-A')
241
- await runOk('git', ['-C', cwd, '-c', 'user.email=t@t', '-c', 'user.name=t', 'commit', '-q', '-m', msg])
242
- return git(cwd, 'rev-parse', 'HEAD')
243
- }
244
-
245
- async function makeSyntheticMirror(root: string): Promise<SyntheticMirror> {
246
- const mirror = join(root, 'mirror')
247
- await mkdir(mirror, { recursive: true })
248
- await runOk('git', ['-C', mirror, 'init', '-q', '-b', 'main'])
249
- await runOk('git', ['-C', mirror, 'config', 'core.hooksPath', '/dev/null'])
250
- await writeFile(join(mirror, 'README.md'), '# synthetic\n')
251
- await writeFile(
252
- join(mirror, 'package.json'),
253
- JSON.stringify({
254
- name: 'synthetic',
255
- type: 'module',
256
- scripts: { preinstall: 'node credential-probe.mjs' },
257
- }),
258
- )
259
- await writeFile(join(mirror, 'credential-probe.mjs'), credentialProbe(root))
260
- await writeFile(join(mirror, '.gitignore'), 'node_modules\n')
261
- const baseCommit = await commitAll(mirror, 'base')
262
-
263
- const variant = async (branch: string, tests: string, withImpl: boolean): Promise<string> => {
264
- await git(mirror, 'checkout', '-q', '-b', branch, baseCommit)
265
- await mkdir(join(mirror, 'tests'), { recursive: true })
266
- await writeFile(join(mirror, 'tests', 'harness.mjs'), JUDGE_HARNESS)
267
- await writeFile(join(mirror, 'tests', 'judge.mjs'), tests)
268
- if (withImpl) await writeFile(join(mirror, 'lib.mjs'), IMPL)
269
- return commitAll(mirror, `merge ${branch}`)
270
- }
271
- const good = await variant('pr-good', GOOD_TESTS, true)
272
- const trivial = await variant('pr-trivial', TRIVIAL_TESTS, true)
273
- const tooHard = await variant('pr-too-hard', TOO_HARD_TESTS, true)
274
- return { mirror, baseCommit, refs: { good, trivial, tooHard } }
275
- }
276
-
277
- async function makeInstanceDir(
278
- root: string,
279
- name: string,
280
- m: SyntheticMirror,
281
- judgeRef: string,
282
- overrides: Record<string, unknown> = {},
283
- ): Promise<string> {
284
- const dir = join(root, name)
285
- await mkdir(dir, { recursive: true })
286
- await writeFile(
287
- join(dir, 'manifest.json'),
288
- JSON.stringify({
289
- id: `factory.synthetic.${name}`,
290
- repo: 'synthetic/repo',
291
- repo_local_mirror: m.mirror,
292
- base_commit: m.baseCommit,
293
- judge_ref: judgeRef,
294
- spec_md: 'spec.md',
295
- judge_tests: ['tests/judge.mjs', 'tests/harness.mjs'],
296
- excluded_tests: [],
297
- command_image: SYNTHETIC_COMMAND_IMAGE,
298
- setup_cmds: [],
299
- judge_cmds: ['node tests/judge.mjs'],
300
- resolved_criterion: 'all 2 judge tests pass; partial score = passed/2',
301
- timeout_s: 60,
302
- ...overrides,
303
- }),
304
- )
305
- await writeFile(join(dir, 'spec.md'), '# Implement lib.mjs\n\nExport `add(a,b)` and `mul(a,b)` from `lib.mjs`.\n')
306
- return dir
307
- }
308
-
309
- let root: string
310
- let mirror: SyntheticMirror
311
- let goodInst: LoadedFactoryInstance
312
-
313
- beforeAll(async () => {
314
- root = await mkdtemp(join(tmpdir(), 'factory-test-'))
315
- mirror = await makeSyntheticMirror(root)
316
- goodInst = loadFactoryInstance(await makeInstanceDir(root, 'good', mirror, mirror.refs.good))
317
- })
318
-
319
- afterAll(async () => {
320
- await rm(root, { recursive: true, force: true })
321
- })
322
-
323
- // ---------------------------------------------------------------------------
324
- // parseVitestSummary — partial-credit arithmetic feeds off these counts.
325
- // ---------------------------------------------------------------------------
326
-
327
- describe('parseVitestSummary', () => {
328
- it('parses an all-pass summary', () => {
329
- expect(parseVitestSummary(' Test Files 2 passed (2)\n Tests 30 passed (30)\n')).toEqual({
330
- passed: 30,
331
- failed: 0,
332
- reportedTotal: 30,
333
- })
334
- })
335
-
336
- it('parses a mixed summary', () => {
337
- expect(parseVitestSummary(' Tests 17 failed | 13 passed (30)\n')).toEqual({
338
- passed: 13,
339
- failed: 17,
340
- reportedTotal: 30,
341
- })
342
- })
343
-
344
- it('treats a collection failure ("no tests") as a zero-pass verdict', () => {
345
- expect(parseVitestSummary(' Test Files 2 failed (2)\n Tests no tests\n')).toEqual({
346
- passed: 0,
347
- failed: 0,
348
- reportedTotal: 0,
349
- })
350
- })
351
-
352
- it('parses through ANSI color codes', () => {
353
- const colored = '\u001b[2m Tests \u001b[22m \u001b[1m\u001b[32m21 passed\u001b[39m\u001b[22m\u001b[90m (21)\u001b[39m\n'
354
- expect(parseVitestSummary(colored)).toEqual({ passed: 21, failed: 0, reportedTotal: 21 })
355
- })
356
-
357
- it('never mistakes the Test Files line for the Tests line, and returns undefined without a summary', () => {
358
- expect(parseVitestSummary(' Test Files 2 failed (2)\n')).toBeUndefined()
359
- expect(parseVitestSummary('some crash output\n')).toBeUndefined()
360
- })
361
- })
362
-
363
- // The isolated Docker controller uses the Linux daemon socket, never personal Docker contexts.
364
- describe.skipIf(process.platform !== 'linux')('factory command credential isolation', () => {
365
- it('blocks arbitrary env, auth sockets, npm config, and home credentials from a package lifecycle script', async () => {
366
- const ambientHome = join(root, 'ambient-home')
367
- const ambientXdg = join(root, 'ambient-xdg')
368
- await mkdir(join(ambientHome, '.config', 'gh'), { recursive: true })
369
- await mkdir(join(ambientHome, '.aws'), { recursive: true })
370
- await mkdir(join(ambientXdg, 'gh'), { recursive: true })
371
- await writeFile(join(ambientHome, '.config', 'gh', 'hosts.yml'), 'oauth_token: ambient-secret\n')
372
- await writeFile(join(ambientHome, '.aws', 'credentials'), 'aws_secret_access_key=ambient-secret\n')
373
- await writeFile(join(ambientHome, '.npmrc'), '//registry.npmjs.org/:_authToken=ambient-secret\n')
374
- await writeFile(join(ambientXdg, 'gh', 'hosts.yml'), 'oauth_token: ambient-xdg-secret\n')
375
-
376
- const injected: Record<string, string> = {
377
- HOME: ambientHome,
378
- XDG_CONFIG_HOME: ambientXdg,
379
- ARBITRARY_FACTORY_SECRET: 'arbitrary-secret',
380
- GH_TOKEN: 'github-secret',
381
- TANGLETOOLS_GH_TOKEN: 'tangletools-secret',
382
- DREW_GH_TOKEN: 'drew-secret',
383
- AWS_SECRET_ACCESS_KEY: 'aws-secret',
384
- NPM_TOKEN: 'npm-secret',
385
- NODE_AUTH_TOKEN: 'node-auth-secret',
386
- NPM_CONFIG_USERCONFIG: join(ambientHome, '.npmrc'),
387
- SSH_AUTH_SOCK: join(ambientHome, 'agent.sock'),
388
- GPG_AGENT_INFO: join(ambientHome, 'gpg-agent'),
389
- DBUS_SESSION_BUS_ADDRESS: 'unix:path=/tmp/operator-bus',
390
- }
391
- const saved = new Map(Object.keys(injected).map((name) => [name, process.env[name]]))
392
- for (const [name, value] of Object.entries(injected)) process.env[name] = value
393
-
394
- const workDir = join(root, 'credential-probe-workspace')
395
- try {
396
- const instance = loadFactoryInstance(
397
- await makeInstanceDir(root, 'credential-probe', mirror, mirror.refs.good, {
398
- setup_cmds: ['npm install --no-audit --no-fund'],
399
- }),
400
- )
401
- const patch = await makePatch(goodInst, IMPL)
402
- const { result } = await judgeFactoryPatch(instance, patch, { workDir, keepWorkspace: true })
403
- expect(result).toMatchObject({ resolved: true, passed: 2, total: 2 })
404
-
405
- const report = JSON.parse(await readFile(join(workDir, 'credential-probe-result.json'), 'utf8')) as {
406
- leakedEnv: string[]
407
- procLeaks: Array<{ depth: number; pid: number; names: string[] }>
408
- procRootFiles: string[]
409
- credentialFiles: string[]
410
- hiddenJudgeFiles: string[]
411
- credentialSockets: string[]
412
- rootWritable: boolean
413
- uid: number
414
- home: string
415
- xdgConfigHome: string
416
- npmUserConfig: string
417
- }
418
- expect(report.leakedEnv).toEqual([])
419
- expect(report.procLeaks).toEqual([])
420
- expect(report.procRootFiles).toEqual([])
421
- expect(report.credentialFiles).toEqual([])
422
- expect(report.hiddenJudgeFiles).toEqual([])
423
- expect(report.credentialSockets).toEqual([])
424
- expect(report.rootWritable).toBe(false)
425
- expect(report.uid).toBeGreaterThan(0)
426
- expect(report.home).not.toBe(ambientHome)
427
- expect(report.xdgConfigHome).not.toBe(ambientXdg)
428
- expect(report.npmUserConfig).not.toBe(join(ambientHome, '.npmrc'))
429
- } finally {
430
- for (const [name, value] of saved) {
431
- if (value === undefined) delete process.env[name]
432
- else process.env[name] = value
433
- }
434
- }
435
- }, 60_000)
436
-
437
- it('does not reuse writable Corepack, npm, or pnpm caches between commands', async () => {
438
- const workspace = join(root, 'cache-isolation-workspace')
439
- await mkdir(workspace, { recursive: true })
440
- const first = await runFactoryCommand(
441
- workspace,
442
- 'touch "$COREPACK_HOME/poison" "$NPM_CONFIG_CACHE/poison" "$npm_config_store_dir/poison" && test -e "$COREPACK_HOME/poison" && test -e "$NPM_CONFIG_CACHE/poison" && test -e "$npm_config_store_dir/poison"',
443
- {
444
- image: SYNTHETIC_COMMAND_IMAGE,
445
- network: 'none',
446
- timeoutMs: 30_000,
447
- },
448
- )
449
- expect(first.code).toBe(0)
450
-
451
- const second = await runFactoryCommand(
452
- workspace,
453
- 'test ! -e "$COREPACK_HOME/poison" && test ! -e "$NPM_CONFIG_CACHE/poison" && test ! -e "$npm_config_store_dir/poison"',
454
- {
455
- image: SYNTHETIC_COMMAND_IMAGE,
456
- network: 'none',
457
- timeoutMs: 30_000,
458
- },
459
- )
460
- expect(second.code).toBe(0)
461
- }, 60_000)
462
- })
463
-
464
- // ---------------------------------------------------------------------------
465
- // Manifest loader — fail-loud shape validation.
466
- // ---------------------------------------------------------------------------
467
-
468
- describe('loadFactoryInstance', () => {
469
- it('loads a valid manifest and derives the calibrated denominator', () => {
470
- expect(goodInst.id).toBe('factory.synthetic.good')
471
- expect(goodInst.judgeTestTotal).toBe(2)
472
- expect(goodInst.spec).toContain('lib.mjs')
473
- })
474
-
475
- it('rejects a manifest missing a required field, naming it', async () => {
476
- const dir = await makeInstanceDir(root, 'bad-missing', mirror, mirror.refs.good, { base_commit: undefined })
477
- expect(() => loadFactoryInstance(dir)).toThrow(/base_commit/)
478
- })
479
-
480
- it('rejects a resolved_criterion without a passed/<N> denominator', async () => {
481
- const dir = await makeInstanceDir(root, 'bad-criterion', mirror, mirror.refs.good, {
482
- resolved_criterion: 'all judge tests pass',
483
- })
484
- expect(() => loadFactoryInstance(dir)).toThrow(/passed\/<N>/)
485
- })
486
-
487
- it('rejects a mutable command image tag', async () => {
488
- const dir = await makeInstanceDir(root, 'bad-image', mirror, mirror.refs.good, {
489
- command_image: 'node:24-bookworm-slim',
490
- })
491
- expect(() => loadFactoryInstance(dir)).toThrow(/pinned.*sha256/)
492
- })
493
-
494
- it('loads the 3 shipped pilot instances with their calibrated totals', () => {
495
- const pilots = loadFactoryInstances(FACTORY_INSTANCES_DIR)
496
- expect(pilots.every((pilot) => /@sha256:[0-9a-f]{64}$/.test(pilot.command_image))).toBe(
497
- true,
498
- )
499
- expect(pilots.map((p) => [p.id, p.judgeTestTotal])).toEqual([
500
- ['factory.agent-eval.309', 30],
501
- ['factory.agent-runtime.232', 21],
502
- ['factory.loops.28', 20],
503
- ])
504
- })
505
- })
506
-
507
- // ---------------------------------------------------------------------------
508
- // Judge child pipeline on the synthetic fixture.
509
- // ---------------------------------------------------------------------------
510
-
511
- describe.skipIf(process.platform !== 'linux')('judgeFactoryPatch', () => {
512
- it('gold (impl-only PR diff) resolves with full score', async () => {
513
- const { result } = await judgeFactoryPatch(goodInst, await goldImplPatch(goodInst))
514
- expect(result).toMatchObject({ resolved: true, score: 1, passed: 2, total: 2 })
515
- }, 60_000)
516
-
517
- it('empty patch (bare base) fails: judge tests cannot even collect', async () => {
518
- const { result } = await judgeFactoryPatch(goodInst, '')
519
- expect(result).toMatchObject({ resolved: false, score: 0, passed: 0, total: 2 })
520
- }, 60_000)
521
-
522
- it('partial impl earns partial credit, never resolved', async () => {
523
- // add correct, mul wrong → 1/2.
524
- const patch = await makePatch(goodInst, 'export const add = (a, b) => a + b\nexport const mul = (a, b) => a + b\n')
525
- const { result } = await judgeFactoryPatch(goodInst, patch)
526
- expect(result).toMatchObject({ resolved: false, score: 0.5, passed: 1, total: 2 })
527
- }, 60_000)
528
-
529
- it('an unappliable patch is the candidate\'s failure (resolved false), not infra', async () => {
530
- const garbage = 'diff --git a/nope.txt b/nope.txt\n--- a/nope.txt\n+++ b/nope.txt\n@@ -1 +1 @@\n-x\n+y\n'
531
- const { result } = await judgeFactoryPatch(goodInst, garbage)
532
- expect(result.resolved).toBe(false)
533
- expect(result.error).toMatch(/apply-failed/)
534
- })
535
-
536
- it('the child CLI prints one JUDGE_RESULT line the serialized-judge protocol parses', async () => {
537
- const patchFile = join(root, 'gold.patch')
538
- await writeFile(patchFile, await goldImplPatch(goodInst))
539
- const here = fileURLToPath(new URL('.', import.meta.url))
540
- const childPath = join(here, 'factory-judge-child.mts')
541
- const res = await runOk('node', ['--import', 'tsx', childPath, goodInst.dir, patchFile], {
542
- cwd: join(here, '..', '..'),
543
- })
544
- const parsed = parseFactoryJudgeResult(res.stdout)
545
- expect(parsed).toMatchObject({ iid: 'factory.synthetic.good', resolved: true, passed: 2, total: 2 })
546
- }, 60_000)
547
- })
548
-
549
- /** A worker-style patch: apply content in a scratch clone of base, git-diff it. */
550
- async function makePatch(inst: LoadedFactoryInstance, libSource: string): Promise<string> {
551
- const ws = await mkdtemp(join(tmpdir(), 'factory-patch-'))
552
- try {
553
- const { syntheticBase } = await materializeFactoryWorkspace(inst, join(ws, 'w'))
554
- await writeFile(join(ws, 'w', 'lib.mjs'), libSource)
555
- await git(join(ws, 'w'), 'add', '-A')
556
- return (await runOk('git', ['-C', join(ws, 'w'), 'diff', '--cached', syntheticBase])).stdout
557
- } finally {
558
- await rm(ws, { recursive: true, force: true })
559
- }
560
- }
561
-
562
- // ---------------------------------------------------------------------------
563
- // Worker workspace: synthetic history, no future refs leakable.
564
- // ---------------------------------------------------------------------------
565
-
566
- describe('materializeFactoryWorkspace', () => {
567
- it('produces a fresh-history repo with SPEC.md and no trace of the real repo', async () => {
568
- const dest = join(root, 'worker-ws')
569
- const { syntheticBase } = await materializeFactoryWorkspace(goodInst, dest)
570
-
571
- // SPEC.md present; judge tests absent (they only exist at judge_ref).
572
- expect(existsSync(join(dest, 'SPEC.md'))).toBe(true)
573
- expect(existsSync(join(dest, 'tests', 'judge.mjs'))).toBe(false)
574
-
575
- // Exactly one synthetic commit; not the real base sha; base ref pinned.
576
- const log = await git(dest, 'log', '--oneline')
577
- expect(log.split('\n')).toHaveLength(1)
578
- expect(syntheticBase).not.toBe(goodInst.base_commit)
579
- expect(await git(dest, 'rev-parse', FACTORY_BASE_REF)).toBe(syntheticBase)
580
-
581
- // No remotes, and the real repo's objects are unreachable.
582
- expect(await git(dest, 'remote')).toBe('')
583
- const baseObj = await run('git', ['-C', dest, 'cat-file', '-e', goodInst.base_commit])
584
- expect(baseObj.code).not.toBe(0)
585
- const judgeObj = await run('git', ['-C', dest, 'cat-file', '-e', goodInst.judge_ref])
586
- expect(judgeObj.code).not.toBe(0)
587
-
588
- // Leak grep: neither the real shas nor the mirror path appear anywhere in
589
- // the workspace (including .git internals).
590
- for (const needle of [goodInst.base_commit, goodInst.judge_ref, goodInst.repo_local_mirror]) {
591
- const grep = await run('bash', ['-c', `grep -rF ${JSON.stringify(needle)} ${JSON.stringify(dest)}`])
592
- expect(grep.stdout).toBe('')
593
- expect(grep.code).not.toBe(0)
594
- }
595
- })
596
- })
597
-
598
- // ---------------------------------------------------------------------------
599
- // Calibration admission gate — both rejection directions.
600
- // ---------------------------------------------------------------------------
601
-
602
- describe.skipIf(process.platform !== 'linux')('calibrateFactoryInstance', () => {
603
- it('admits a well-formed instance (gold passes, base fails)', async () => {
604
- const r = await calibrateFactoryInstance(goodInst)
605
- expect(r).toMatchObject({
606
- goldResolved: true,
607
- goldPassed: 2,
608
- baseResolved: false,
609
- basePassed: 0,
610
- total: 2,
611
- admitted: true,
612
- })
613
- })
614
-
615
- it('rejects when the base already passes (judge has no signal)', async () => {
616
- const inst = loadFactoryInstance(await makeInstanceDir(root, 'trivial', mirror, mirror.refs.trivial))
617
- const r = await calibrateFactoryInstance(inst)
618
- expect(r.baseResolved).toBe(true)
619
- expect(r.admitted).toBe(false)
620
- })
621
-
622
- it('rejects when gold cannot pass its own judge', async () => {
623
- const inst = loadFactoryInstance(await makeInstanceDir(root, 'too-hard', mirror, mirror.refs.tooHard))
624
- const r = await calibrateFactoryInstance(inst)
625
- expect(r.goldResolved).toBe(false)
626
- expect(r.admitted).toBe(false)
627
- })
628
- })
629
-
630
- // ---------------------------------------------------------------------------
631
- // Ledger resume keys.
632
- // ---------------------------------------------------------------------------
633
-
634
- describe('factoryLedgerKeys', () => {
635
- it('keys rows by iid and rep, and fails loud on corrupt lines', async () => {
636
- const ledger = join(root, 'ledger.jsonl')
637
- await writeFile(
638
- ledger,
639
- JSON.stringify({ iid: 'factory.x.1', rep: 0 }) + '\n' + JSON.stringify({ iid: 'factory.x.1', rep: 1 }) + '\n',
640
- )
641
- expect(await factoryLedgerKeys(ledger)).toEqual(new Set(['factory.x.1#r0', 'factory.x.1#r1']))
642
- await writeFile(ledger, '{corrupt\n')
643
- await expect(factoryLedgerKeys(ledger)).rejects.toThrow(/corrupt/)
644
- })
645
- })