@tangle-network/agent-bench 0.13.13 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +5 -0
- package/HARNESS.md +3 -3
- package/package.json +6 -6
- package/src/selector.ts +1 -1
- package/src/swe-bench-env.ts +3 -2
- package/src/swe-self-improve.mts +55 -47
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,10 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.14.0
|
|
4
|
+
|
|
5
|
+
Require Eval 0.201 through 0.203, Interface 2.15, Knowledge 18, and Sandbox 0.58.4 through the shared catalog.
|
|
6
|
+
Consume Runtime 0.287 through the workspace dependency.
|
|
7
|
+
|
|
3
8
|
## 0.13.13
|
|
4
9
|
|
|
5
10
|
Require Knowledge 17.1.6 and Eval `>=0.191.0 <0.194.0` through the shared dependency catalog.
|
package/HARNESS.md
CHANGED
|
@@ -118,9 +118,9 @@ cd bench
|
|
|
118
118
|
pnpm tsx src/swe-self-improve.mts
|
|
119
119
|
```
|
|
120
120
|
|
|
121
|
-
This fixture uses `runStrategyEvolution` with SWE-bench tasks and
|
|
122
|
-
It does not exercise `improve
|
|
123
|
-
It
|
|
121
|
+
This fixture uses `runStrategyEvolution` with SWE-bench tasks split into train, selection and sealed test slices.
|
|
122
|
+
It does not exercise `improve`.
|
|
123
|
+
It keeps its run directory, whose search ledger is the checkpoint and the lineage record of one search.
|
|
124
124
|
Use `examples/improve` for the maintained offline API fixture.
|
|
125
125
|
Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
|
|
126
126
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.14.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
29
|
-
"@tangle-network/agent-interface": "^2.
|
|
30
|
-
"@tangle-network/agent-knowledge": "^
|
|
31
|
-
"@tangle-network/sandbox": ">=0.
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.201.0 <0.204.0",
|
|
29
|
+
"@tangle-network/agent-interface": "^2.15.0",
|
|
30
|
+
"@tangle-network/agent-knowledge": "^18.0.0",
|
|
31
|
+
"@tangle-network/sandbox": ">=0.58.4 <0.59.0",
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.287.0"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
package/src/selector.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Deployable, non-oracle selector
|
|
2
|
+
* Deployable, non-oracle selector.
|
|
3
3
|
*
|
|
4
4
|
* best-of-N only pays if you can pick the good attempt WITHOUT the judge. Today
|
|
5
5
|
* the loop's winner is judge-selected (verdict.score) — an oracle upper bound, not
|
package/src/swe-bench-env.ts
CHANGED
|
@@ -277,8 +277,9 @@ async function resolveInstanceImage(ws: Ws, expected?: SweImageIdentity): Promis
|
|
|
277
277
|
}
|
|
278
278
|
|
|
279
279
|
/** Build the SWE-bench Environment + a DISJOINT-slice task supplier over the Verified split. The
|
|
280
|
-
* supplier keys tasks by dataset offset so
|
|
281
|
-
*
|
|
280
|
+
* supplier keys tasks by dataset offset, so consecutive slices (train, selection and test for
|
|
281
|
+
* `runStrategyEvolution`) never overlap. Verified is loaded once; instances carry their
|
|
282
|
+
* repo/base_commit. */
|
|
282
283
|
export async function createSweBenchEnvironment(
|
|
283
284
|
poolN = 80,
|
|
284
285
|
opts: {
|
package/src/swe-self-improve.mts
CHANGED
|
@@ -1,20 +1,22 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* SWE-bench self-improvement
|
|
3
|
-
* `
|
|
4
|
-
*
|
|
5
|
-
* applies (public fixes may be
|
|
2
|
+
* SWE-bench self-improvement: a frontier worker over the SWE-bench `Environment`, with
|
|
3
|
+
* `runStrategyEvolution` searching strategies on disjoint train, selection and test slices. The
|
|
4
|
+
* author reads train results only, strategies are ranked on the private selection slice, and the
|
|
5
|
+
* claim runs once on the sealed test slice. CONTAMINATION CAVEAT applies (public fixes may be
|
|
6
|
+
* memorized) — reported, never claimed clean.
|
|
6
7
|
*
|
|
7
8
|
* CALIBRATE first (cost gate): TANGLE_API_KEY=… CALIBRATE=1 N=3 tsx bench/src/swe-self-improve.mts
|
|
8
|
-
* Full run: TANGLE_API_KEY=… TRAIN_N=
|
|
9
|
+
* Full run: TANGLE_API_KEY=… TRAIN_N=4 SELECTION_N=12 TEST_N=12 tsx bench/src/swe-self-improve.mts
|
|
10
|
+
*
|
|
11
|
+
* The run directory is kept (`OUT_DIR`, default `.swe-run`): its ledger is the checkpoint, so the
|
|
12
|
+
* same command continues an interrupted search.
|
|
9
13
|
*/
|
|
10
|
-
import { mkdtempSync, rmSync } from 'node:fs'
|
|
11
14
|
import { join } from 'node:path'
|
|
12
|
-
import type
|
|
15
|
+
import { type AgentProfile, canonicalCandidateDigest } from '@tangle-network/agent-interface'
|
|
13
16
|
import {
|
|
14
17
|
refine,
|
|
15
18
|
runAgentic,
|
|
16
19
|
runStrategyEvolution,
|
|
17
|
-
sample,
|
|
18
20
|
strategyAuthorSystemPrompt,
|
|
19
21
|
} from '@tangle-network/agent-runtime/kernel'
|
|
20
22
|
import { createSweBenchEnvironment } from './swe-bench-env'
|
|
@@ -72,46 +74,52 @@ async function main(): Promise<void> {
|
|
|
72
74
|
return
|
|
73
75
|
}
|
|
74
76
|
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
77
|
+
const trainN = Number(process.env.TRAIN_N ?? 4)
|
|
78
|
+
const selectionN = Number(process.env.SELECTION_N ?? 12)
|
|
79
|
+
const testN = Number(process.env.TEST_N ?? 12)
|
|
80
|
+
const all = await tasks(0, trainN + selectionN + testN)
|
|
81
|
+
const report = await runStrategyEvolution({
|
|
82
|
+
environment,
|
|
83
|
+
train: all.slice(0, trainN),
|
|
84
|
+
selection: all.slice(trainN, trainN + selectionN),
|
|
85
|
+
test: all.slice(trainN + selectionN),
|
|
86
|
+
claim: {
|
|
87
|
+
use: 'comparison',
|
|
88
|
+
population: { id: 'swe-bench-verified', description: 'SWE-bench Verified instances' },
|
|
89
|
+
samplingFrame: 'consecutive SWE-bench Verified instances from the loaded pool',
|
|
90
|
+
independentUnit: 'id',
|
|
91
|
+
generalization: 'new-units',
|
|
92
|
+
minimumEffect: Number(process.env.MIN_EFFECT ?? 0.15),
|
|
93
|
+
},
|
|
94
|
+
executionRef: canonicalCandidateDigest({
|
|
95
|
+
environment: 'bench/src/swe-bench-env.ts',
|
|
96
|
+
harness: 'swebench-docker',
|
|
97
|
+
innerTurns,
|
|
98
|
+
}),
|
|
99
|
+
worker: { routerBaseUrl, routerKey, workerProfile },
|
|
100
|
+
author: {
|
|
101
|
+
profile: authorProfile(authorModel, 'swe-strategy-author'),
|
|
102
|
+
executor: { backend: 'router', routerBaseUrl, routerKey },
|
|
103
|
+
fallbackProfile: authorProfile(
|
|
104
|
+
process.env.AUTHOR_FALLBACK ?? 'deepseek-v4-flash',
|
|
105
|
+
'swe-strategy-author-fallback',
|
|
106
|
+
),
|
|
107
|
+
},
|
|
108
|
+
root: refine,
|
|
109
|
+
budget: Number(process.env.BUDGET ?? 2),
|
|
110
|
+
maxExpansions: Number(process.env.EXPANSIONS ?? 4),
|
|
111
|
+
outDir: join(process.cwd(), process.env.OUT_DIR ?? '.swe-run'),
|
|
112
|
+
})
|
|
102
113
|
|
|
103
|
-
const
|
|
104
|
-
console.log('\n═══ SWE-bench SELF-IMPROVEMENT —
|
|
105
|
-
console.log(`worker=${workerModel} author=${authorModel}`)
|
|
106
|
-
console.log(`
|
|
107
|
-
console.log(`
|
|
108
|
-
console.log(`
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
? '\n>>> The search taught the agent a strategy that resolves MORE real bugs it never trained on, beyond luck. (Report the contamination caveat: public fixes may be memorized.)'
|
|
113
|
-
: '\n>>> No promotion: the evolved strategy did not beat gen0 on the fresh holdout beyond noise (honest null).',
|
|
114
|
-
)
|
|
114
|
+
const shipped = report.claim.finalists.find((f) => f.nodeId === report.claim.selected)?.test
|
|
115
|
+
console.log('\n═══ SWE-bench SELF-IMPROVEMENT — claimed on a SEALED test slice (CONTAMINATION-flagged) ═══')
|
|
116
|
+
console.log(`worker=${workerModel} author=${authorModel} ledger=${report.ledger}`)
|
|
117
|
+
console.log(`strategies: ${report.strategies.map((s) => `${s.name} (${s.status})`).join(', ')}`)
|
|
118
|
+
console.log(`selected: ${report.selected.name}`)
|
|
119
|
+
console.log(`DECISION: ${report.decision} (${report.reason})`)
|
|
120
|
+
if (shipped) {
|
|
121
|
+
console.log(`test lift: ${shipped.delta.toFixed(3)} [${shipped.interval[0].toFixed(3)}, ${shipped.interval[1].toFixed(3)}] n=${shipped.pairs}`)
|
|
122
|
+
}
|
|
115
123
|
}
|
|
116
124
|
|
|
117
125
|
main().catch((e) => {
|