@tangle-network/agent-bench 0.13.13 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,10 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.14.0
4
+
5
+ Require Eval 0.201 through 0.203, Interface 2.15, Knowledge 18, and Sandbox 0.58.4 through the shared catalog.
6
+ Consume Runtime 0.287 through the workspace dependency.
7
+
3
8
  ## 0.13.13
4
9
 
5
10
  Require Knowledge 17.1.6 and Eval `>=0.191.0 <0.194.0` through the shared dependency catalog.
package/HARNESS.md CHANGED
@@ -118,9 +118,9 @@ cd bench
118
118
  pnpm tsx src/swe-self-improve.mts
119
119
  ```
120
120
 
121
- This fixture uses `runStrategyEvolution` with SWE-bench tasks and a frozen holdout.
122
- It does not exercise `improve`, and it deletes its temporary run directory on exit.
123
- It therefore cannot provide retained improvement or lineage evidence.
121
+ This fixture uses `runStrategyEvolution` with SWE-bench tasks split into train, selection and sealed test slices.
122
+ It does not exercise `improve`.
123
+ It keeps its run directory, whose search ledger is the checkpoint and the lineage record of one search.
124
124
  Use `examples/improve` for the maintained offline API fixture.
125
125
  Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
126
126
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-bench",
3
- "version": "0.13.13",
3
+ "version": "0.14.0",
4
4
  "type": "module",
5
5
  "description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
6
6
  "repository": {
@@ -25,11 +25,11 @@
25
25
  }
26
26
  },
27
27
  "dependencies": {
28
- "@tangle-network/agent-eval": ">=0.191.0 <0.194.0",
29
- "@tangle-network/agent-interface": "^2.11.0",
30
- "@tangle-network/agent-knowledge": "^17.1.6",
31
- "@tangle-network/sandbox": ">=0.36.4 <0.48.0",
32
- "@tangle-network/agent-runtime": "^0.277.0"
28
+ "@tangle-network/agent-eval": ">=0.201.0 <0.204.0",
29
+ "@tangle-network/agent-interface": "^2.15.0",
30
+ "@tangle-network/agent-knowledge": "^18.0.0",
31
+ "@tangle-network/sandbox": ">=0.58.4 <0.59.0",
32
+ "@tangle-network/agent-runtime": "^0.287.0"
33
33
  },
34
34
  "devDependencies": {
35
35
  "@arethetypeswrong/cli": "0.18.5",
package/src/selector.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Deployable, non-oracle selector (docs/roadmap-rsi.md Phase 1).
2
+ * Deployable, non-oracle selector.
3
3
  *
4
4
  * best-of-N only pays if you can pick the good attempt WITHOUT the judge. Today
5
5
  * the loop's winner is judge-selected (verdict.score) — an oracle upper bound, not
@@ -277,8 +277,9 @@ async function resolveInstanceImage(ws: Ws, expected?: SweImageIdentity): Promis
277
277
  }
278
278
 
279
279
  /** Build the SWE-bench Environment + a DISJOINT-slice task supplier over the Verified split. The
280
- * supplier keys tasks by dataset offset so `runStrategyEvolution`'s train [0,trainN) and holdout
281
- * [trainN+off,…) never overlap. Verified is loaded once; instances carry their repo/base_commit. */
280
+ * supplier keys tasks by dataset offset, so consecutive slices (train, selection and test for
281
+ * `runStrategyEvolution`) never overlap. Verified is loaded once; instances carry their
282
+ * repo/base_commit. */
282
283
  export async function createSweBenchEnvironment(
283
284
  poolN = 80,
284
285
  opts: {
@@ -1,20 +1,22 @@
1
1
  /**
2
- * SWE-bench self-improvement — the PROPER, no-cheating run: a frontier worker over the SWE-bench
3
- * `Environment`, with `runStrategyEvolution` enforcing the train→freeze→holdout split (the substrate
4
- * draws a disjoint holdout slice and gates once — adaptive reuse is impossible). CONTAMINATION CAVEAT
5
- * applies (public fixes may be memorized) — reported, never claimed clean.
2
+ * SWE-bench self-improvement: a frontier worker over the SWE-bench `Environment`, with
3
+ * `runStrategyEvolution` searching strategies on disjoint train, selection and test slices. The
4
+ * author reads train results only, strategies are ranked on the private selection slice, and the
5
+ * claim runs once on the sealed test slice. CONTAMINATION CAVEAT applies (public fixes may be
6
+ * memorized) — reported, never claimed clean.
6
7
  *
7
8
  * CALIBRATE first (cost gate): TANGLE_API_KEY=… CALIBRATE=1 N=3 tsx bench/src/swe-self-improve.mts
8
- * Full run: TANGLE_API_KEY=… TRAIN_N=6 HOLDOUT_N=8 GENERATIONS=2 tsx bench/src/swe-self-improve.mts
9
+ * Full run: TANGLE_API_KEY=… TRAIN_N=4 SELECTION_N=12 TEST_N=12 tsx bench/src/swe-self-improve.mts
10
+ *
11
+ * The run directory is kept (`OUT_DIR`, default `.swe-run`): its ledger is the checkpoint, so the
12
+ * same command continues an interrupted search.
9
13
  */
10
- import { mkdtempSync, rmSync } from 'node:fs'
11
14
  import { join } from 'node:path'
12
- import type { AgentProfile } from '@tangle-network/agent-interface'
15
+ import { type AgentProfile, canonicalCandidateDigest } from '@tangle-network/agent-interface'
13
16
  import {
14
17
  refine,
15
18
  runAgentic,
16
19
  runStrategyEvolution,
17
- sample,
18
20
  strategyAuthorSystemPrompt,
19
21
  } from '@tangle-network/agent-runtime/kernel'
20
22
  import { createSweBenchEnvironment } from './swe-bench-env'
@@ -72,46 +74,52 @@ async function main(): Promise<void> {
72
74
  return
73
75
  }
74
76
 
75
- const report = await (async () => {
76
- const outDir = mkdtempSync(join(process.cwd(), '.swe-run-'))
77
- try {
78
- return await runStrategyEvolution({
79
- environment,
80
- tasks,
81
- trainN: Number(process.env.TRAIN_N ?? 6),
82
- holdoutN: Number(process.env.HOLDOUT_N ?? 8),
83
- worker: { routerBaseUrl, routerKey, workerProfile },
84
- author: {
85
- profile: authorProfile(authorModel, 'swe-strategy-author'),
86
- executor: { backend: 'router', routerBaseUrl, routerKey },
87
- fallbackProfile: authorProfile(
88
- process.env.AUTHOR_FALLBACK ?? 'deepseek-v4-flash',
89
- 'swe-strategy-author-fallback',
90
- ),
91
- },
92
- baselines: [sample, refine],
93
- budget: Number(process.env.BUDGET ?? 2),
94
- generations: Number(process.env.GENERATIONS ?? 2),
95
- populationSize: Number(process.env.POP ?? 2),
96
- outDir,
97
- })
98
- } finally {
99
- rmSync(outDir, { recursive: true, force: true })
100
- }
101
- })()
77
+ const trainN = Number(process.env.TRAIN_N ?? 4)
78
+ const selectionN = Number(process.env.SELECTION_N ?? 12)
79
+ const testN = Number(process.env.TEST_N ?? 12)
80
+ const all = await tasks(0, trainN + selectionN + testN)
81
+ const report = await runStrategyEvolution({
82
+ environment,
83
+ train: all.slice(0, trainN),
84
+ selection: all.slice(trainN, trainN + selectionN),
85
+ test: all.slice(trainN + selectionN),
86
+ claim: {
87
+ use: 'comparison',
88
+ population: { id: 'swe-bench-verified', description: 'SWE-bench Verified instances' },
89
+ samplingFrame: 'consecutive SWE-bench Verified instances from the loaded pool',
90
+ independentUnit: 'id',
91
+ generalization: 'new-units',
92
+ minimumEffect: Number(process.env.MIN_EFFECT ?? 0.15),
93
+ },
94
+ executionRef: canonicalCandidateDigest({
95
+ environment: 'bench/src/swe-bench-env.ts',
96
+ harness: 'swebench-docker',
97
+ innerTurns,
98
+ }),
99
+ worker: { routerBaseUrl, routerKey, workerProfile },
100
+ author: {
101
+ profile: authorProfile(authorModel, 'swe-strategy-author'),
102
+ executor: { backend: 'router', routerBaseUrl, routerKey },
103
+ fallbackProfile: authorProfile(
104
+ process.env.AUTHOR_FALLBACK ?? 'deepseek-v4-flash',
105
+ 'swe-strategy-author-fallback',
106
+ ),
107
+ },
108
+ root: refine,
109
+ budget: Number(process.env.BUDGET ?? 2),
110
+ maxExpansions: Number(process.env.EXPANSIONS ?? 4),
111
+ outDir: join(process.cwd(), process.env.OUT_DIR ?? '.swe-run'),
112
+ })
102
113
 
103
- const v = report.verdict
104
- console.log('\n═══ SWE-bench SELF-IMPROVEMENT — certified on a FROZEN holdout (CONTAMINATION-flagged) ═══')
105
- console.log(`worker=${workerModel} author=${authorModel}`)
106
- console.log(`gen0 champion: ${report.gen0Champion.name}`)
107
- console.log(`final champion: ${report.finalChampion.name}`)
108
- console.log(`PROMOTED: ${v.promoted} (${v.reason})`)
109
- console.log(`held-out lift: mean ${v.lift.mean.toFixed(3)} 95% CI [${v.lift.low.toFixed(3)}, ${v.lift.high.toFixed(3)}] n=${v.n}`)
110
- console.log(
111
- v.promoted
112
- ? '\n>>> The search taught the agent a strategy that resolves MORE real bugs it never trained on, beyond luck. (Report the contamination caveat: public fixes may be memorized.)'
113
- : '\n>>> No promotion: the evolved strategy did not beat gen0 on the fresh holdout beyond noise (honest null).',
114
- )
114
+ const shipped = report.claim.finalists.find((f) => f.nodeId === report.claim.selected)?.test
115
+ console.log('\n═══ SWE-bench SELF-IMPROVEMENT — claimed on a SEALED test slice (CONTAMINATION-flagged) ═══')
116
+ console.log(`worker=${workerModel} author=${authorModel} ledger=${report.ledger}`)
117
+ console.log(`strategies: ${report.strategies.map((s) => `${s.name} (${s.status})`).join(', ')}`)
118
+ console.log(`selected: ${report.selected.name}`)
119
+ console.log(`DECISION: ${report.decision} (${report.reason})`)
120
+ if (shipped) {
121
+ console.log(`test lift: ${shipped.delta.toFixed(3)} [${shipped.interval[0].toFixed(3)}, ${shipped.interval[1].toFixed(3)}] n=${shipped.pairs}`)
122
+ }
115
123
  }
116
124
 
117
125
  main().catch((e) => {