@tangle-network/agent-bench 0.13.11 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/HARNESS.md +5 -3
- package/package.json +6 -6
- package/src/selector.ts +1 -1
- package/src/swe-bench-env.ts +3 -2
- package/src/swe-local-proof.mts +72 -3
- package/src/swe-self-improve.mts +55 -47
- package/src/terminal-compare.ts +9 -13
- package/tb_agents/opencode_router_agent.py +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,20 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.14.0
|
|
4
|
+
|
|
5
|
+
Require Eval 0.201 through 0.203, Interface 2.15, Knowledge 18, and Sandbox 0.58.4 through the shared catalog.
|
|
6
|
+
Consume Runtime 0.287 through the workspace dependency.
|
|
7
|
+
|
|
8
|
+
## 0.13.13
|
|
9
|
+
|
|
10
|
+
Require Knowledge 17.1.6 and Eval `>=0.191.0 <0.194.0` through the shared dependency catalog.
|
|
11
|
+
Benchmark behavior is unchanged.
|
|
12
|
+
|
|
13
|
+
## 0.13.12
|
|
14
|
+
|
|
15
|
+
Accept Eval 0.185 and 0.186 and require Knowledge 17.1.2 through the shared dependency catalog.
|
|
16
|
+
Benchmark behavior is unchanged.
|
|
17
|
+
|
|
3
18
|
## 0.13.11
|
|
4
19
|
|
|
5
20
|
Accept Eval 0.184 and require Knowledge 17.1.1 through the shared dependency catalog.
|
package/HARNESS.md
CHANGED
|
@@ -28,6 +28,8 @@ Use these labels literally. Do not promote one level into another in prose.
|
|
|
28
28
|
|
|
29
29
|
A localization score, output-shape check, LLM quality judge, or toy deterministic reward can be useful for development. None is a substitute for the benchmark's outcome evaluator.
|
|
30
30
|
|
|
31
|
+
The [September 24 public benchmark proof](evidence/public-agent-benchmarks-20260924/README.md) retains three officially graded agent tasks, traces, and validity limits across Terminal-Bench and SWE-bench Verified.
|
|
32
|
+
|
|
31
33
|
## Supported commands
|
|
32
34
|
|
|
33
35
|
### Package and integration contracts
|
|
@@ -116,9 +118,9 @@ cd bench
|
|
|
116
118
|
pnpm tsx src/swe-self-improve.mts
|
|
117
119
|
```
|
|
118
120
|
|
|
119
|
-
This fixture uses `runStrategyEvolution` with SWE-bench tasks and
|
|
120
|
-
It does not exercise `improve
|
|
121
|
-
It
|
|
121
|
+
This fixture uses `runStrategyEvolution` with SWE-bench tasks split into train, selection and sealed test slices.
|
|
122
|
+
It does not exercise `improve`.
|
|
123
|
+
It keeps its run directory, whose search ledger is the checkpoint and the lineage record of one search.
|
|
122
124
|
Use `examples/improve` for the maintained offline API fixture.
|
|
123
125
|
Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
|
|
124
126
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-bench",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.14.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Benchmark adapters and execution for agent-runtime across coding, tool-use, RAG, memory, browser, and terminal tasks.",
|
|
6
6
|
"repository": {
|
|
@@ -25,11 +25,11 @@
|
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
27
|
"dependencies": {
|
|
28
|
-
"@tangle-network/agent-eval": ">=0.
|
|
29
|
-
"@tangle-network/agent-interface": "^2.
|
|
30
|
-
"@tangle-network/agent-knowledge": "^
|
|
31
|
-
"@tangle-network/sandbox": ">=0.
|
|
32
|
-
"@tangle-network/agent-runtime": "^0.
|
|
28
|
+
"@tangle-network/agent-eval": ">=0.201.0 <0.204.0",
|
|
29
|
+
"@tangle-network/agent-interface": "^2.15.0",
|
|
30
|
+
"@tangle-network/agent-knowledge": "^18.0.0",
|
|
31
|
+
"@tangle-network/sandbox": ">=0.58.4 <0.59.0",
|
|
32
|
+
"@tangle-network/agent-runtime": "^0.287.0"
|
|
33
33
|
},
|
|
34
34
|
"devDependencies": {
|
|
35
35
|
"@arethetypeswrong/cli": "0.18.5",
|
package/src/selector.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Deployable, non-oracle selector
|
|
2
|
+
* Deployable, non-oracle selector.
|
|
3
3
|
*
|
|
4
4
|
* best-of-N only pays if you can pick the good attempt WITHOUT the judge. Today
|
|
5
5
|
* the loop's winner is judge-selected (verdict.score) — an oracle upper bound, not
|
package/src/swe-bench-env.ts
CHANGED
|
@@ -277,8 +277,9 @@ async function resolveInstanceImage(ws: Ws, expected?: SweImageIdentity): Promis
|
|
|
277
277
|
}
|
|
278
278
|
|
|
279
279
|
/** Build the SWE-bench Environment + a DISJOINT-slice task supplier over the Verified split. The
|
|
280
|
-
* supplier keys tasks by dataset offset so
|
|
281
|
-
*
|
|
280
|
+
* supplier keys tasks by dataset offset, so consecutive slices (train, selection and test for
|
|
281
|
+
* `runStrategyEvolution`) never overlap. Verified is loaded once; instances carry their
|
|
282
|
+
* repo/base_commit. */
|
|
282
283
|
export async function createSweBenchEnvironment(
|
|
283
284
|
poolN = 80,
|
|
284
285
|
opts: {
|
package/src/swe-local-proof.mts
CHANGED
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
* node_modules/.bin/tsx bench/src/swe-local-proof.mts
|
|
17
17
|
*/
|
|
18
18
|
import { execFile } from 'node:child_process'
|
|
19
|
-
import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
|
19
|
+
import { appendFileSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
|
20
20
|
import { tmpdir } from 'node:os'
|
|
21
21
|
import { join } from 'node:path'
|
|
22
22
|
import { promisify } from 'node:util'
|
|
@@ -40,12 +40,24 @@ async function main(): Promise<void> {
|
|
|
40
40
|
// WITH-TOOLS arm: RUN_TOOL=1 exposes the jailed `run` tool + the run-aware prompt. Default OFF ⇒
|
|
41
41
|
// the read/edit-only baseline (glm-5.2 7/12) unchanged.
|
|
42
42
|
const enableRun = ['1', 'true', 'yes'].includes((process.env.RUN_TOOL ?? '').toLowerCase())
|
|
43
|
+
const artifactDir = process.env.RUN_ARTIFACT_DIR
|
|
44
|
+
if (artifactDir) mkdirSync(artifactDir, { recursive: true })
|
|
43
45
|
|
|
44
46
|
console.log(`═══ SWE-bench Verified — SEE-able LOCAL proof (no tangle sandbox) ═══`)
|
|
45
47
|
console.log(`model=${model} ids=${ids.join(',')} innerTurns=${innerTurns} maxTokens=${maxTokens} budget=${budget} runTool=${enableRun}`)
|
|
46
48
|
console.log(`router=${routerBaseUrl}`)
|
|
47
49
|
|
|
48
|
-
const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, {
|
|
50
|
+
const { environment, tasks, adapter } = await createSweBenchEnvironment(ids.length, {
|
|
51
|
+
ids,
|
|
52
|
+
enableRun,
|
|
53
|
+
...(artifactDir ? {
|
|
54
|
+
adapterOptions: {
|
|
55
|
+
captureEvaluatorArtifacts: ({ taskId, attemptSequence }) => ({
|
|
56
|
+
destination: join(artifactDir, taskId, `judge-${attemptSequence}`),
|
|
57
|
+
}),
|
|
58
|
+
},
|
|
59
|
+
} : {}),
|
|
60
|
+
})
|
|
49
61
|
const workerProfile = withBenchProfile(
|
|
50
62
|
{
|
|
51
63
|
name: 'swe-local-proof-worker',
|
|
@@ -85,10 +97,27 @@ async function main(): Promise<void> {
|
|
|
85
97
|
const captured = new Map<string, Rec>()
|
|
86
98
|
const judged = new Map<string, BenchScore>()
|
|
87
99
|
const toolStats = new Map<string, { list: number; read: number; edit_ok: number; edit_fail: number; run: number; run_err: number }>()
|
|
100
|
+
let activeTaskId = ''
|
|
101
|
+
const recordTool = (name: string, args: unknown, result: unknown, error?: string) => {
|
|
102
|
+
if (!artifactDir || !activeTaskId) return
|
|
103
|
+
const taskDir = join(artifactDir, activeTaskId)
|
|
104
|
+
mkdirSync(taskDir, { recursive: true })
|
|
105
|
+
appendFileSync(join(taskDir, 'tools.jsonl'), JSON.stringify({
|
|
106
|
+
at: new Date().toISOString(), name, args,
|
|
107
|
+
...(error ? { error } : { result }),
|
|
108
|
+
}) + '\n')
|
|
109
|
+
}
|
|
88
110
|
const proxy: AgenticSurface = {
|
|
89
111
|
...environment,
|
|
90
112
|
async call(handle, name, args) {
|
|
91
|
-
|
|
113
|
+
let res: string
|
|
114
|
+
try {
|
|
115
|
+
res = await environment.call(handle, name, args)
|
|
116
|
+
} catch (error) {
|
|
117
|
+
recordTool(name, args, undefined, error instanceof Error ? error.message : String(error))
|
|
118
|
+
throw error
|
|
119
|
+
}
|
|
120
|
+
recordTool(name, args, res)
|
|
92
121
|
// Count tool usage per workspace so we can SEE whether the agent ever edited, and whether
|
|
93
122
|
// edits succeeded or bounced off old_string matching. handle.id keys the workspace.
|
|
94
123
|
const st = toolStats.get(handle.id) ?? { list: 0, read: 0, edit_ok: 0, edit_fail: 0, run: 0, run_err: 0 }
|
|
@@ -161,6 +190,7 @@ async function main(): Promise<void> {
|
|
|
161
190
|
|
|
162
191
|
let anyResolved = 0
|
|
163
192
|
for (const task of taskList) {
|
|
193
|
+
activeTaskId = task.id
|
|
164
194
|
const t0 = Date.now()
|
|
165
195
|
const r = await runAgentic({
|
|
166
196
|
surface: proxy,
|
|
@@ -171,6 +201,13 @@ async function main(): Promise<void> {
|
|
|
171
201
|
workerProfile,
|
|
172
202
|
analystProfile,
|
|
173
203
|
budget,
|
|
204
|
+
...(artifactDir ? { hooks: { onEvent(event: unknown) {
|
|
205
|
+
const taskDir = join(artifactDir, activeTaskId)
|
|
206
|
+
mkdirSync(taskDir, { recursive: true })
|
|
207
|
+
appendFileSync(join(taskDir, 'runtime-events.jsonl'), JSON.stringify(event, (_key, value) =>
|
|
208
|
+
value instanceof Error ? { name: value.name, message: value.message, stack: value.stack } : value,
|
|
209
|
+
) + '\n')
|
|
210
|
+
} } } : {}),
|
|
174
211
|
})
|
|
175
212
|
const rec = captured.get(task.id)
|
|
176
213
|
const st = toolStats.get(task.id)
|
|
@@ -186,6 +223,38 @@ async function main(): Promise<void> {
|
|
|
186
223
|
console.log(` patch APPLIED (git apply --check on clean base): ${rec?.applied ? 'YES' : 'NO'}${rec?.applyErr ? ` (${rec.applyErr})` : ''}`)
|
|
187
224
|
console.log(` swebench judge RESOLVED: ${resolved ? '1' : '0'}`)
|
|
188
225
|
if (rec?.score?.detail) console.log(` judge report: ${rec.score.detail.slice(0, 400)}`)
|
|
226
|
+
if (artifactDir) {
|
|
227
|
+
const taskDir = join(artifactDir, task.id)
|
|
228
|
+
mkdirSync(taskDir, { recursive: true })
|
|
229
|
+
writeFileSync(join(taskDir, 'patch.diff'), rec?.patch ?? '')
|
|
230
|
+
writeFileSync(join(taskDir, 'summary.json'), JSON.stringify({
|
|
231
|
+
benchmark: 'SWE-bench Verified',
|
|
232
|
+
taskId: task.id,
|
|
233
|
+
modelRequested: model,
|
|
234
|
+
prompt: task.userPrompt,
|
|
235
|
+
turnsLimit: innerTurns,
|
|
236
|
+
maxTokens,
|
|
237
|
+
budget,
|
|
238
|
+
enableRun,
|
|
239
|
+
shots: r.shots,
|
|
240
|
+
completions: r.completions,
|
|
241
|
+
tokens: r.tokens,
|
|
242
|
+
tokensKnown: r.tokensKnown,
|
|
243
|
+
billedCostUsd: null,
|
|
244
|
+
reportedUsd: r.usd,
|
|
245
|
+
reportedUsdKnown: r.usdKnown,
|
|
246
|
+
progression: r.progression,
|
|
247
|
+
runScore: r.score,
|
|
248
|
+
runResolved: r.resolved,
|
|
249
|
+
wallMs: Date.now() - t0,
|
|
250
|
+
toolStats: st ?? null,
|
|
251
|
+
patchBytes,
|
|
252
|
+
patchLines,
|
|
253
|
+
files,
|
|
254
|
+
patchApplies: rec?.applied ?? false,
|
|
255
|
+
score: rec?.score ?? null,
|
|
256
|
+
}, null, 2) + '\n')
|
|
257
|
+
}
|
|
189
258
|
}
|
|
190
259
|
console.log(`\n>>> resolved ${anyResolved}/${taskList.length}`)
|
|
191
260
|
}
|
package/src/swe-self-improve.mts
CHANGED
|
@@ -1,20 +1,22 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* SWE-bench self-improvement
|
|
3
|
-
* `
|
|
4
|
-
*
|
|
5
|
-
* applies (public fixes may be
|
|
2
|
+
* SWE-bench self-improvement: a frontier worker over the SWE-bench `Environment`, with
|
|
3
|
+
* `runStrategyEvolution` searching strategies on disjoint train, selection and test slices. The
|
|
4
|
+
* author reads train results only, strategies are ranked on the private selection slice, and the
|
|
5
|
+
* claim runs once on the sealed test slice. CONTAMINATION CAVEAT applies (public fixes may be
|
|
6
|
+
* memorized) — reported, never claimed clean.
|
|
6
7
|
*
|
|
7
8
|
* CALIBRATE first (cost gate): TANGLE_API_KEY=… CALIBRATE=1 N=3 tsx bench/src/swe-self-improve.mts
|
|
8
|
-
* Full run: TANGLE_API_KEY=… TRAIN_N=
|
|
9
|
+
* Full run: TANGLE_API_KEY=… TRAIN_N=4 SELECTION_N=12 TEST_N=12 tsx bench/src/swe-self-improve.mts
|
|
10
|
+
*
|
|
11
|
+
* The run directory is kept (`OUT_DIR`, default `.swe-run`): its ledger is the checkpoint, so the
|
|
12
|
+
* same command continues an interrupted search.
|
|
9
13
|
*/
|
|
10
|
-
import { mkdtempSync, rmSync } from 'node:fs'
|
|
11
14
|
import { join } from 'node:path'
|
|
12
|
-
import type
|
|
15
|
+
import { type AgentProfile, canonicalCandidateDigest } from '@tangle-network/agent-interface'
|
|
13
16
|
import {
|
|
14
17
|
refine,
|
|
15
18
|
runAgentic,
|
|
16
19
|
runStrategyEvolution,
|
|
17
|
-
sample,
|
|
18
20
|
strategyAuthorSystemPrompt,
|
|
19
21
|
} from '@tangle-network/agent-runtime/kernel'
|
|
20
22
|
import { createSweBenchEnvironment } from './swe-bench-env'
|
|
@@ -72,46 +74,52 @@ async function main(): Promise<void> {
|
|
|
72
74
|
return
|
|
73
75
|
}
|
|
74
76
|
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
77
|
+
const trainN = Number(process.env.TRAIN_N ?? 4)
|
|
78
|
+
const selectionN = Number(process.env.SELECTION_N ?? 12)
|
|
79
|
+
const testN = Number(process.env.TEST_N ?? 12)
|
|
80
|
+
const all = await tasks(0, trainN + selectionN + testN)
|
|
81
|
+
const report = await runStrategyEvolution({
|
|
82
|
+
environment,
|
|
83
|
+
train: all.slice(0, trainN),
|
|
84
|
+
selection: all.slice(trainN, trainN + selectionN),
|
|
85
|
+
test: all.slice(trainN + selectionN),
|
|
86
|
+
claim: {
|
|
87
|
+
use: 'comparison',
|
|
88
|
+
population: { id: 'swe-bench-verified', description: 'SWE-bench Verified instances' },
|
|
89
|
+
samplingFrame: 'consecutive SWE-bench Verified instances from the loaded pool',
|
|
90
|
+
independentUnit: 'id',
|
|
91
|
+
generalization: 'new-units',
|
|
92
|
+
minimumEffect: Number(process.env.MIN_EFFECT ?? 0.15),
|
|
93
|
+
},
|
|
94
|
+
executionRef: canonicalCandidateDigest({
|
|
95
|
+
environment: 'bench/src/swe-bench-env.ts',
|
|
96
|
+
harness: 'swebench-docker',
|
|
97
|
+
innerTurns,
|
|
98
|
+
}),
|
|
99
|
+
worker: { routerBaseUrl, routerKey, workerProfile },
|
|
100
|
+
author: {
|
|
101
|
+
profile: authorProfile(authorModel, 'swe-strategy-author'),
|
|
102
|
+
executor: { backend: 'router', routerBaseUrl, routerKey },
|
|
103
|
+
fallbackProfile: authorProfile(
|
|
104
|
+
process.env.AUTHOR_FALLBACK ?? 'deepseek-v4-flash',
|
|
105
|
+
'swe-strategy-author-fallback',
|
|
106
|
+
),
|
|
107
|
+
},
|
|
108
|
+
root: refine,
|
|
109
|
+
budget: Number(process.env.BUDGET ?? 2),
|
|
110
|
+
maxExpansions: Number(process.env.EXPANSIONS ?? 4),
|
|
111
|
+
outDir: join(process.cwd(), process.env.OUT_DIR ?? '.swe-run'),
|
|
112
|
+
})
|
|
102
113
|
|
|
103
|
-
const
|
|
104
|
-
console.log('\n═══ SWE-bench SELF-IMPROVEMENT —
|
|
105
|
-
console.log(`worker=${workerModel} author=${authorModel}`)
|
|
106
|
-
console.log(`
|
|
107
|
-
console.log(`
|
|
108
|
-
console.log(`
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
? '\n>>> The search taught the agent a strategy that resolves MORE real bugs it never trained on, beyond luck. (Report the contamination caveat: public fixes may be memorized.)'
|
|
113
|
-
: '\n>>> No promotion: the evolved strategy did not beat gen0 on the fresh holdout beyond noise (honest null).',
|
|
114
|
-
)
|
|
114
|
+
const shipped = report.claim.finalists.find((f) => f.nodeId === report.claim.selected)?.test
|
|
115
|
+
console.log('\n═══ SWE-bench SELF-IMPROVEMENT — claimed on a SEALED test slice (CONTAMINATION-flagged) ═══')
|
|
116
|
+
console.log(`worker=${workerModel} author=${authorModel} ledger=${report.ledger}`)
|
|
117
|
+
console.log(`strategies: ${report.strategies.map((s) => `${s.name} (${s.status})`).join(', ')}`)
|
|
118
|
+
console.log(`selected: ${report.selected.name}`)
|
|
119
|
+
console.log(`DECISION: ${report.decision} (${report.reason})`)
|
|
120
|
+
if (shipped) {
|
|
121
|
+
console.log(`test lift: ${shipped.delta.toFixed(3)} [${shipped.interval[0].toFixed(3)}, ${shipped.interval[1].toFixed(3)}] n=${shipped.pairs}`)
|
|
122
|
+
}
|
|
115
123
|
}
|
|
116
124
|
|
|
117
125
|
main().catch((e) => {
|
package/src/terminal-compare.ts
CHANGED
|
@@ -58,7 +58,7 @@ const PROFILE = agentProfileSchema.parse({
|
|
|
58
58
|
systemPrompt:
|
|
59
59
|
'Solve the Terminal-Bench task completely in the provided container and verify the result before finishing.',
|
|
60
60
|
},
|
|
61
|
-
|
|
61
|
+
permissions: {
|
|
62
62
|
edit: 'allow',
|
|
63
63
|
bash: 'allow',
|
|
64
64
|
webfetch: 'allow',
|
|
@@ -360,14 +360,9 @@ async function captureRunRecord(
|
|
|
360
360
|
await appendRunRecord(CORPUS, record)
|
|
361
361
|
}
|
|
362
362
|
|
|
363
|
-
/**
|
|
364
|
-
*
|
|
365
|
-
*
|
|
366
|
-
* contends on the shared docker build backend and can return nonzero. Warming the
|
|
367
|
-
* cache serially makes the parallel rounds hit a warm cache and never race.
|
|
368
|
-
* Dataset-agnostic: derives the compose path from the tb cache layout
|
|
369
|
-
* (<cache>/<name>/<version>/<task>/docker-compose.yaml).
|
|
370
|
-
*/
|
|
363
|
+
/** Warm each task image serially before the concurrent fan-out. Terminal-Bench
|
|
364
|
+
* supplies compose variables only inside its own runner, so build the task's
|
|
365
|
+
* Dockerfile directly and let its later compose build reuse the cached layers. */
|
|
371
366
|
async function prebuildImages(taskIds: string[]): Promise<void> {
|
|
372
367
|
const [name, version] = DATASET.split('==')
|
|
373
368
|
if (!name || !version) {
|
|
@@ -376,18 +371,19 @@ async function prebuildImages(taskIds: string[]): Promise<void> {
|
|
|
376
371
|
}
|
|
377
372
|
const cacheRoot = join(homedir(), '.cache', 'terminal-bench', name, version)
|
|
378
373
|
for (const taskId of taskIds) {
|
|
379
|
-
const
|
|
374
|
+
const taskDir = join(cacheRoot, taskId)
|
|
375
|
+
const dockerfile = join(taskDir, 'Dockerfile')
|
|
380
376
|
try {
|
|
381
|
-
await stat(
|
|
377
|
+
await stat(dockerfile)
|
|
382
378
|
} catch {
|
|
383
|
-
console.log(` prebuild: no
|
|
379
|
+
console.log(` prebuild: no Dockerfile at ${dockerfile}; tb will materialize ${taskId} on first run`)
|
|
384
380
|
continue
|
|
385
381
|
}
|
|
386
382
|
process.stdout.write(` prebuild: ${taskId} … `)
|
|
387
383
|
await new Promise<void>((resolve, reject) => {
|
|
388
384
|
execFile(
|
|
389
385
|
'docker',
|
|
390
|
-
['
|
|
386
|
+
['build', '--tag', `tbprebuild-${taskId}`.toLowerCase().replace(/[^a-z0-9_.-]/g, '-'), '--file', dockerfile, taskDir],
|
|
391
387
|
{ maxBuffer: 1024 * 1024 * 256, timeout: ROUND_CAP_MS },
|
|
392
388
|
(err) => (err ? reject(new Error(`prebuild ${taskId} failed: ${err.message}`)) : resolve()),
|
|
393
389
|
)
|
|
@@ -202,7 +202,7 @@ class OpenCodeRouterAgent(OpenCodeAgent):
|
|
|
202
202
|
"$schema": "https://opencode.ai/config.json",
|
|
203
203
|
# Headless benchmark runs cannot answer interactive permission prompts.
|
|
204
204
|
# Keep this identical for raw and supervisor arms.
|
|
205
|
-
"permission": self._profile.get("
|
|
205
|
+
"permission": self._profile.get("permissions", {}),
|
|
206
206
|
"provider": {
|
|
207
207
|
self._provider: {
|
|
208
208
|
"npm": "@ai-sdk/openai-compatible",
|