@tangle-network/agent-bench 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/HARNESS.md +1 -1
- package/dist/adapters.js +2 -2
- package/dist/benchmarks/appworld.d.ts +61 -2
- package/dist/benchmarks/appworld.js +111 -40
- package/dist/benchmarks/appworld.js.map +1 -1
- package/dist/benchmarks/cadbench.js +1 -1
- package/dist/benchmarks/cadgenbench.js +1 -1
- package/dist/benchmarks/finresearchbench.js +16 -22
- package/dist/benchmarks/finresearchbench.js.map +1 -1
- package/dist/benchmarks/finsearchcomp.js +16 -22
- package/dist/benchmarks/finsearchcomp.js.map +1 -1
- package/dist/benchmarks/frames.js +18 -20
- package/dist/benchmarks/frames.js.map +1 -1
- package/dist/benchmarks/simpleqa.js +15 -19
- package/dist/benchmarks/simpleqa.js.map +1 -1
- package/dist/benchmarks/trata-hedge.js +20 -23
- package/dist/benchmarks/trata-hedge.js.map +1 -1
- package/dist/{cadbench-BrpwOU6A.js → cadbench-BLSyxR1N.js} +21 -20
- package/dist/cadbench-BLSyxR1N.js.map +1 -0
- package/dist/{cadgenbench-DF7hYHdl.js → cadgenbench-x2OFkf8y.js} +2 -2
- package/dist/cadgenbench-x2OFkf8y.js.map +1 -0
- package/dist/index.d.ts +0 -1
- package/dist/index.js +10 -9
- package/dist/index.js.map +1 -1
- package/dist/router-turn-C2wMiDoo.js +31 -0
- package/dist/router-turn-C2wMiDoo.js.map +1 -0
- package/package.json +7 -7
- package/scripts/appworld_driver.py +1 -170
- package/scripts/run-package-tests.mjs +56 -36
- package/scripts/run-package-tests.test.mjs +33 -0
- package/scripts/trata-hedge/README.md +3 -3
- package/scripts/trata-hedge/run.sh +6 -2
- package/src/aec-gate.mts +30 -9
- package/src/agent-graphs-gen2.mts +523 -0
- package/src/agent-graphs-gen3.mts +660 -0
- package/src/agent-graphs-improve/offline-seams.mts +128 -0
- package/src/agent-graphs-improve.mts +747 -0
- package/src/atom-humaneval.mts +76 -55
- package/src/atom-mcp-e2e.mts +27 -13
- package/src/benchmarks/appworld.test.mts +119 -1
- package/src/benchmarks/appworld.ts +156 -33
- package/src/benchmarks/cadbench.ts +21 -8
- package/src/benchmarks/finresearchbench.ts +20 -16
- package/src/benchmarks/finsearchcomp.ts +20 -20
- package/src/benchmarks/frames.ts +19 -18
- package/src/benchmarks/simpleqa.ts +19 -17
- package/src/benchmarks/trata-hedge.ts +26 -18
- package/src/browser/ui-reviewer.ts +1 -1
- package/src/clbench-context-gate.mts +45 -9
- package/src/cloud-loop.mts +24 -6
- package/src/commit0-env-run.mts +6 -3
- package/src/commit0-env.ts +2 -3
- package/src/commit0-gate.mts +53 -24
- package/src/corpus.ts +3 -3
- package/src/david-attribution.mts +28 -9
- package/src/david-goliath.mts +31 -12
- package/src/egress-probe.mts +3 -3
- package/src/examples/lean-proof-gate.mts +10 -3
- package/src/examples/math-demo.mts +10 -3
- package/src/examples/strategy-demo.mts +10 -4
- package/src/fleet.mts +30 -8
- package/src/gate-cli.mts +9 -4
- package/src/gate.test.mts +7 -2
- package/src/gate.ts +23 -75
- package/src/generate-eval/certify.ts +13 -5
- package/src/hev-eval.mts +39 -10
- package/src/hev-improve.mts +28 -16
- package/src/hev-structural.mts +28 -20
- package/src/humaneval-gate.mts +24 -8
- package/src/humaneval-object-ablation.mts +52 -14
- package/src/humaneval-repair-gate.mts +53 -14
- package/src/mbpp-structural.mts +28 -20
- package/src/mcp-mount-probe.mts +23 -11
- package/src/official-optimizer-config.mts +61 -12
- package/src/official-optimizer-config.test.mts +72 -3
- package/src/quant-arena/profiles/quant-researcher.profile.json +6 -1
- package/src/quant-arena/quant-loop.mts +88 -61
- package/src/research-shot.ts +21 -11
- package/src/resolve-client.ts +6 -8
- package/src/rollout-ledger/backfill-swe-arena.test.mts +7 -2
- package/src/router-executor.ts +1 -1
- package/src/router-turn.ts +135 -0
- package/src/run-benchmarks.ts +7 -2
- package/src/sandbox-run.ts +23 -41
- package/src/search-bench/bridge.ts +66 -8
- package/src/search-bench/parametric-check.mts +19 -4
- package/src/search-bench/profiles.ts +24 -9
- package/src/search-bench/run.mts +3 -4
- package/src/smoke-structural-rollout.mts +21 -5
- package/src/supervisor-arena.mts +28 -20
- package/src/swe-arena/arms.ts +73 -15
- package/src/swe-arena/capacity.ts +44 -41
- package/src/swe-arena/diagnosis-ensemble.ts +51 -48
- package/src/swe-arena/execution.test.mts +12 -12
- package/src/swe-arena/gepa-seat.mts +12 -5
- package/src/swe-arena/gepa-seat.test.mts +45 -33
- package/src/swe-arena/outer-loop.mts +94 -70
- package/src/swe-arena/premeasured-from-cells.mts +16 -1
- package/src/swe-arena/premeasured-from-cells.test.mts +21 -1
- package/src/swe-arena/profiles/deepseek-author.profile.json +12 -0
- package/src/swe-arena/profiles/default-author.profile.json +9 -1
- package/src/swe-arena/proposer-fanout.mts +52 -99
- package/src/swe-arena/proposer-fanout.test.mts +92 -57
- package/src/swe-arena/proposer-provenance.mts +1 -1
- package/src/swe-code-improve.mts +30 -52
- package/src/swe-emit-patch.mts +26 -5
- package/src/swe-improve.mts +27 -8
- package/src/swe-jail.test.ts +128 -0
- package/src/swe-jail.ts +106 -22
- package/src/swe-local-proof.mts +27 -5
- package/src/swe-repro-calibrate.mts +5 -0
- package/src/swe-self-improve.mts +45 -10
- package/src/swe-stream.mts +68 -6
- package/src/swe-structural.mts +46 -7
- package/src/tb-container-executor.mts +38 -6
- package/src/tb-container-executor.test.mts +11 -3
- package/src/tb-supervisor-sidecar.mts +11 -27
- package/src/terminal-compare.ts +38 -18
- package/src/trata-gate.mts +26 -26
- package/src/trata-gepa.mts +25 -16
- package/src/trata-hedge-solve.mts +80 -0
- package/src/worker-blender.ts +21 -11
- package/src/worker-browser.ts +18 -5
- package/src/worker-build123d.ts +17 -8
- package/src/worker-cad.ts +31 -12
- package/src/worker.ts +1 -1
- package/tb_agents/opencode_router_agent.py +84 -42
- package/dist/cadbench-BrpwOU6A.js.map +0 -1
- package/dist/cadgenbench-DF7hYHdl.js.map +0 -1
- package/scripts/trata-hedge/solve.py +0 -83
package/src/gate.ts
CHANGED
|
@@ -32,28 +32,26 @@ import type {
|
|
|
32
32
|
AgentSpec,
|
|
33
33
|
Budget,
|
|
34
34
|
CombinatorShape,
|
|
35
|
-
DefaultVerdict,
|
|
36
35
|
EqualKArm,
|
|
37
36
|
EqualKVerdict,
|
|
38
37
|
ExecutorContext,
|
|
39
38
|
ExecutorRegistry,
|
|
40
39
|
Executor,
|
|
41
40
|
ExecutorFactory,
|
|
42
|
-
ExecutorResult,
|
|
43
41
|
Outcome,
|
|
44
42
|
Persona,
|
|
45
|
-
Runtime,
|
|
46
43
|
Spend,
|
|
47
44
|
SupervisedResult,
|
|
48
45
|
TrajectoryReport,
|
|
49
46
|
} from '@tangle-network/agent-runtime/kernel'
|
|
50
47
|
import {
|
|
51
48
|
definePersona,
|
|
52
|
-
|
|
49
|
+
createExecutor,
|
|
53
50
|
equalKOnCost,
|
|
54
51
|
fanout,
|
|
55
52
|
InMemoryResultBlobStore,
|
|
56
53
|
InMemorySpawnJournal,
|
|
54
|
+
mapExecutorResult,
|
|
57
55
|
runPersonified,
|
|
58
56
|
trajectoryReport,
|
|
59
57
|
} from '@tangle-network/agent-runtime/kernel'
|
|
@@ -73,10 +71,6 @@ export interface BenchSolverOptions {
|
|
|
73
71
|
readonly adapter: BenchmarkAdapter
|
|
74
72
|
readonly routerBaseUrl: string
|
|
75
73
|
readonly routerKey: string
|
|
76
|
-
readonly model: string
|
|
77
|
-
/** Sampling temperature. >0 is required for the blind arm to be more than k identical samples
|
|
78
|
-
* (k copies at temperature 0 collapse to one answer — no compute control). Default 0.7. */
|
|
79
|
-
readonly temperature?: number
|
|
80
74
|
}
|
|
81
75
|
|
|
82
76
|
const fnv = (prefix: string, value: unknown): string => {
|
|
@@ -114,58 +108,27 @@ function extractArtifact(adapter: BenchmarkAdapter, content: string): string {
|
|
|
114
108
|
* a judge throw rejects the leaf (the scope types it into a `down` settlement — never a silent 0).
|
|
115
109
|
*/
|
|
116
110
|
export function benchSolveLeaf(opts: BenchSolverOptions, spec: AgentSpec, ctx: ExecutorContext): Executor<unknown> {
|
|
117
|
-
const
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
let artifact: ExecutorResult<unknown> | undefined
|
|
125
|
-
|
|
126
|
-
return {
|
|
127
|
-
runtime: 'bench-router' as Runtime,
|
|
128
|
-
async execute(task, signal): Promise<ExecutorResult<unknown>> {
|
|
111
|
+
const inner = createExecutor({
|
|
112
|
+
backend: 'router',
|
|
113
|
+
routerBaseUrl: opts.routerBaseUrl,
|
|
114
|
+
routerKey: opts.routerKey,
|
|
115
|
+
})(spec, ctx)
|
|
116
|
+
return mapExecutorResult(inner, async (result, task) => {
|
|
129
117
|
const t = task as SolveTask
|
|
130
|
-
const
|
|
131
|
-
const
|
|
132
|
-
|
|
133
|
-
? [{ role: 'system', content: system }]
|
|
134
|
-
: []),
|
|
135
|
-
{ role: 'user', content: t.prompt },
|
|
136
|
-
]
|
|
137
|
-
const started = Date.now()
|
|
138
|
-
const linked = linkSignals(signal, controller.signal)
|
|
139
|
-
const chat = await routerChatWithUsage(
|
|
140
|
-
{ routerBaseUrl: opts.routerBaseUrl, routerKey: opts.routerKey, model: opts.model },
|
|
141
|
-
messages,
|
|
142
|
-
{ temperature: opts.temperature ?? 0.7, ...(linked ? { signal: linked } : {}) },
|
|
143
|
-
)
|
|
144
|
-
const candidate = extractArtifact(opts.adapter, chat.content)
|
|
118
|
+
const raw = result.out as { content?: unknown }
|
|
119
|
+
const content = typeof raw?.content === 'string' ? raw.content : ''
|
|
120
|
+
const candidate = extractArtifact(opts.adapter, content)
|
|
145
121
|
const score = await opts.adapter.judge(t.instance, candidate)
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
usd: chat.costUsd ?? 0,
|
|
155
|
-
ms: Date.now() - started,
|
|
122
|
+
return {
|
|
123
|
+
outRef: fnv('bench', { id: t.instance.id, candidate }),
|
|
124
|
+
out: candidate,
|
|
125
|
+
verdict: {
|
|
126
|
+
valid: score.resolved,
|
|
127
|
+
score: score.score,
|
|
128
|
+
...(score.detail ? { notes: score.detail } : {}),
|
|
129
|
+
},
|
|
156
130
|
}
|
|
157
|
-
|
|
158
|
-
return artifact
|
|
159
|
-
},
|
|
160
|
-
teardown(): Promise<{ destroyed: boolean }> {
|
|
161
|
-
controller.abort()
|
|
162
|
-
return Promise.resolve({ destroyed: true })
|
|
163
|
-
},
|
|
164
|
-
resultArtifact() {
|
|
165
|
-
if (!artifact) throw new Error('benchSolveLeaf: resultArtifact() read before execute()')
|
|
166
|
-
return artifact
|
|
167
|
-
},
|
|
168
|
-
}
|
|
131
|
+
})
|
|
169
132
|
}
|
|
170
133
|
|
|
171
134
|
/**
|
|
@@ -189,10 +152,9 @@ export function benchSolverRegistry(opts: BenchSolverOptions): ExecutorRegistry
|
|
|
189
152
|
}
|
|
190
153
|
}
|
|
191
154
|
|
|
192
|
-
/** Build the solver `Persona` from the developer's `AgentProfile` + a solve-and-grade
|
|
193
|
-
* The
|
|
194
|
-
*
|
|
195
|
-
* built-ins. */
|
|
155
|
+
/** Build the solver `Persona` from the developer's exact `AgentProfile` + a solve-and-grade
|
|
156
|
+
* registry. The profile still declares its execution identity; `AgentSpec.harness: null` only
|
|
157
|
+
* says that the supplied registry owns placement instead of a built-in executor. */
|
|
196
158
|
export function defineSolverPersona(
|
|
197
159
|
profile: AgentProfile,
|
|
198
160
|
registry: ExecutorRegistry,
|
|
@@ -233,8 +195,6 @@ export interface RunGateOptions {
|
|
|
233
195
|
readonly strategies: ReadonlyArray<string>
|
|
234
196
|
readonly routerBaseUrl: string
|
|
235
197
|
readonly routerKey: string
|
|
236
|
-
readonly model: string
|
|
237
|
-
readonly temperature?: number
|
|
238
198
|
/** How many benchmark instances to run (the paired n). */
|
|
239
199
|
readonly n?: number
|
|
240
200
|
readonly ids?: string[]
|
|
@@ -447,15 +407,3 @@ export async function runGate(opts: RunGateOptions): Promise<GateReport> {
|
|
|
447
407
|
}
|
|
448
408
|
|
|
449
409
|
/** Link two abort signals into one that fires when either does; `undefined` when neither is set. */
|
|
450
|
-
function linkSignals(a: AbortSignal, b: AbortSignal): AbortSignal | undefined {
|
|
451
|
-
if (a.aborted || b.aborted) {
|
|
452
|
-
const c = new AbortController()
|
|
453
|
-
c.abort()
|
|
454
|
-
return c.signal
|
|
455
|
-
}
|
|
456
|
-
const c = new AbortController()
|
|
457
|
-
const onAbort = () => c.abort()
|
|
458
|
-
a.addEventListener('abort', onAbort, { once: true })
|
|
459
|
-
b.addEventListener('abort', onAbort, { once: true })
|
|
460
|
-
return c.signal
|
|
461
|
-
}
|
|
@@ -24,8 +24,8 @@ import { execSync } from 'node:child_process'
|
|
|
24
24
|
import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
|
25
25
|
import { tmpdir } from 'node:os'
|
|
26
26
|
import { dirname, join, resolve } from 'node:path'
|
|
27
|
-
import { routerChatWithUsage } from '@tangle-network/agent-runtime/kernel'
|
|
28
27
|
import { scoreTask, taskToPrompt } from '../search-bench/tasks'
|
|
28
|
+
import { runBenchRouterTurn } from '../router-turn'
|
|
29
29
|
import { type EvalCertification, type GeneratedEval, generatedEvalSchemaVersion, parseCandidate } from './schema'
|
|
30
30
|
|
|
31
31
|
const certifierId = 'agent-runtime/generate-eval@1'
|
|
@@ -121,10 +121,18 @@ export async function discriminationGate(
|
|
|
121
121
|
const apiKey = opts.gateApiKey ?? process.env.EVAL_GATE_API_KEY ?? process.env.TANGLE_API_KEY
|
|
122
122
|
const model = opts.gateModel ?? process.env.EVAL_GATE_MODEL ?? 'deepseek-v4-flash'
|
|
123
123
|
if (!apiKey) throw new Error('discrimination gate needs EVAL_GATE_API_KEY (or TANGLE_API_KEY)')
|
|
124
|
-
const res = await
|
|
125
|
-
{
|
|
126
|
-
|
|
127
|
-
|
|
124
|
+
const res = await runBenchRouterTurn(
|
|
125
|
+
{
|
|
126
|
+
routerBaseUrl: baseUrl,
|
|
127
|
+
routerKey: apiKey,
|
|
128
|
+
profile: {
|
|
129
|
+
name: 'generated-eval-parametric-check',
|
|
130
|
+
model: { provider: 'tangle-router', default: model },
|
|
131
|
+
},
|
|
132
|
+
},
|
|
133
|
+
taskToPrompt(candidate),
|
|
134
|
+
)
|
|
135
|
+
const { score, reasons } = scoreTask(candidate, res.finalText)
|
|
128
136
|
return score === 0
|
|
129
137
|
? { passed: true, detail: `parametric ${model} failed as required (${reasons.join('; ')})` }
|
|
130
138
|
: { passed: false, detail: `parametric ${model} SOLVED the task from memory — not search-discriminating` }
|
package/src/hev-eval.mts
CHANGED
|
@@ -9,19 +9,41 @@
|
|
|
9
9
|
*/
|
|
10
10
|
import { readFileSync } from 'node:fs'
|
|
11
11
|
import { extractCode, loadHumanEval, runChecker, type HumanEvalTask } from './benchmarks/humaneval'
|
|
12
|
+
import { runBenchRouterTurn } from './router-turn'
|
|
12
13
|
|
|
13
14
|
const SEED_INSTRUCTION =
|
|
14
15
|
'Complete the following Python function. Output the COMPLETE function definition (signature, docstring optional, body) inside a single ```python code block. Include any imports the function needs. Do not write tests or example calls.'
|
|
15
16
|
|
|
16
|
-
async function complete(
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
17
|
+
async function complete(
|
|
18
|
+
base: string,
|
|
19
|
+
key: string,
|
|
20
|
+
model: string,
|
|
21
|
+
instruction: string,
|
|
22
|
+
prompt: string,
|
|
23
|
+
maxTokens: number,
|
|
24
|
+
): Promise<string> {
|
|
25
|
+
try {
|
|
26
|
+
const turn = await runBenchRouterTurn(
|
|
27
|
+
{
|
|
28
|
+
routerBaseUrl: base,
|
|
29
|
+
routerKey: key,
|
|
30
|
+
profile: {
|
|
31
|
+
name: 'humaneval-worker',
|
|
32
|
+
harness: 'cli-base',
|
|
33
|
+
model: {
|
|
34
|
+
provider: 'tangle-router',
|
|
35
|
+
default: model,
|
|
36
|
+
metadata: { temperature: 0.2, maxTokens },
|
|
37
|
+
},
|
|
38
|
+
prompt: { systemPrompt: instruction },
|
|
39
|
+
},
|
|
40
|
+
},
|
|
41
|
+
prompt,
|
|
42
|
+
)
|
|
43
|
+
return turn.finalText
|
|
44
|
+
} catch {
|
|
45
|
+
return ''
|
|
46
|
+
}
|
|
25
47
|
}
|
|
26
48
|
|
|
27
49
|
async function main(): Promise<void> {
|
|
@@ -55,7 +77,14 @@ async function main(): Promise<void> {
|
|
|
55
77
|
const t = tasks[i]
|
|
56
78
|
i += 1
|
|
57
79
|
if (!t) continue
|
|
58
|
-
const reply = await complete(
|
|
80
|
+
const reply = await complete(
|
|
81
|
+
base,
|
|
82
|
+
apiKey,
|
|
83
|
+
model,
|
|
84
|
+
instruction,
|
|
85
|
+
`\`\`\`python\n${t.prompt}\`\`\``,
|
|
86
|
+
maxTokens,
|
|
87
|
+
)
|
|
59
88
|
const { pass: p } = await runChecker(t, extractCode(reply))
|
|
60
89
|
if (p === 1) pass += 1
|
|
61
90
|
else fails.push(t.taskId)
|
package/src/hev-improve.mts
CHANGED
|
@@ -19,6 +19,7 @@ import {
|
|
|
19
19
|
import {
|
|
20
20
|
canonicalCandidateDigest,
|
|
21
21
|
type AgentProfile,
|
|
22
|
+
agentProfileSchema,
|
|
22
23
|
} from '@tangle-network/agent-interface'
|
|
23
24
|
import type { DispatchContext, JudgeConfig, Scenario } from '@tangle-network/agent-eval/contract'
|
|
24
25
|
import { extractCode, loadHumanEval, runChecker, type HumanEvalTask } from './benchmarks/humaneval'
|
|
@@ -27,6 +28,7 @@ import {
|
|
|
27
28
|
officialOptimizerModel,
|
|
28
29
|
requiredTokenPricing,
|
|
29
30
|
} from './official-optimizer-config.mjs'
|
|
31
|
+
import { runBenchRouterTurn, withBenchProfile } from './router-turn'
|
|
30
32
|
|
|
31
33
|
// The SEED instruction GEPA evolves. Byte-identical to humaneval.ts basePrompt's
|
|
32
34
|
// solveInstruction so the baseline arm reproduces the plain-prompt denominator.
|
|
@@ -39,21 +41,26 @@ interface Completion {
|
|
|
39
41
|
tokOut: number
|
|
40
42
|
}
|
|
41
43
|
|
|
42
|
-
async function complete(
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
const
|
|
50
|
-
|
|
51
|
-
|
|
44
|
+
async function complete(
|
|
45
|
+
base: string,
|
|
46
|
+
key: string,
|
|
47
|
+
profile: AgentProfile,
|
|
48
|
+
prompt: string,
|
|
49
|
+
maxTokens: number,
|
|
50
|
+
): Promise<Completion> {
|
|
51
|
+
const result = await runBenchRouterTurn(
|
|
52
|
+
{
|
|
53
|
+
routerBaseUrl: base,
|
|
54
|
+
routerKey: key,
|
|
55
|
+
profile: withBenchProfile(profile, { temperature: 0.2, maxTokens }),
|
|
56
|
+
},
|
|
57
|
+
prompt,
|
|
58
|
+
)
|
|
59
|
+
return {
|
|
60
|
+
text: result.finalText,
|
|
61
|
+
tokIn: result.usage.input,
|
|
62
|
+
tokOut: result.usage.output,
|
|
52
63
|
}
|
|
53
|
-
const text = d.choices?.[0]?.message?.content ?? ''
|
|
54
|
-
const tokIn = d.usage?.prompt_tokens ?? 0
|
|
55
|
-
const tokOut = d.usage?.completion_tokens ?? 0
|
|
56
|
-
return { text, tokIn, tokOut }
|
|
57
64
|
}
|
|
58
65
|
|
|
59
66
|
async function main(): Promise<void> {
|
|
@@ -115,13 +122,18 @@ async function main(): Promise<void> {
|
|
|
115
122
|
if (instr === undefined) throw new Error('agent: candidate profile has no system prompt')
|
|
116
123
|
const t = byId.get(scenario.id)
|
|
117
124
|
if (!t) throw new Error(`agent: unknown scenario ${scenario.id}`)
|
|
118
|
-
const prompt =
|
|
125
|
+
const prompt = `\`\`\`python\n${t.prompt}\`\`\``
|
|
126
|
+
const executionProfile: AgentProfile = agentProfileSchema.parse({
|
|
127
|
+
...candidate,
|
|
128
|
+
name: candidate.name ?? 'humaneval-improvement-worker',
|
|
129
|
+
model: { ...candidate.model, provider: 'tangle-router', default: workerModel },
|
|
130
|
+
})
|
|
119
131
|
const t0 = Date.now()
|
|
120
132
|
const paid = await ctx.cost.runPaidCall({
|
|
121
133
|
channel: 'agent',
|
|
122
134
|
actor: 'humaneval-worker',
|
|
123
135
|
model: workerModel,
|
|
124
|
-
execute: () => complete(base, key,
|
|
136
|
+
execute: () => complete(base, key, executionProfile, prompt, workerMaxTokens),
|
|
125
137
|
receipt: (result) => {
|
|
126
138
|
const usageUnknown = result.tokIn === 0 && result.tokOut === 0
|
|
127
139
|
return {
|
package/src/hev-structural.mts
CHANGED
|
@@ -46,6 +46,7 @@ import { tmpdir } from 'node:os'
|
|
|
46
46
|
import { join } from 'node:path'
|
|
47
47
|
import { type HumanEvalTask, extractCode, loadHumanEval } from './benchmarks/humaneval'
|
|
48
48
|
import { composeStrategies } from './directives'
|
|
49
|
+
import { runBenchRouterTurn } from './router-turn'
|
|
49
50
|
import { type PairedLift, pairedLift, pool } from './stats.mts'
|
|
50
51
|
|
|
51
52
|
const dockerImage = 'python:3.12-slim'
|
|
@@ -345,35 +346,42 @@ async function complete(cfg: ClientCfg, messages: Array<{ role: string; content:
|
|
|
345
346
|
let lastErr = ''
|
|
346
347
|
for (let attempt = 1; attempt <= 4; attempt += 1) {
|
|
347
348
|
if (attempt > 1) await new Promise((r) => setTimeout(r, 2000 * 2 ** attempt))
|
|
348
|
-
const ctl = new AbortController()
|
|
349
|
-
const timer = setTimeout(() => ctl.abort(), 240_000)
|
|
350
349
|
try {
|
|
351
|
-
const
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
350
|
+
const system = messages.find((message) => message.role === 'system')?.content
|
|
351
|
+
const result = await runBenchRouterTurn(
|
|
352
|
+
{
|
|
353
|
+
routerBaseUrl: cfg.base,
|
|
354
|
+
routerKey: cfg.key,
|
|
355
|
+
profile: {
|
|
356
|
+
name: 'humaneval-structural-worker',
|
|
357
|
+
harness: 'cli-base',
|
|
358
|
+
model: {
|
|
359
|
+
provider: 'tangle-router',
|
|
360
|
+
default: cfg.model,
|
|
361
|
+
metadata: { temperature: cfg.temperature, maxTokens: cfg.maxTokens },
|
|
362
|
+
},
|
|
363
|
+
...(system ? { prompt: { systemPrompt: system } } : {}),
|
|
364
|
+
},
|
|
365
|
+
timeoutMs: Number(process.env.LLM_TIMEOUT_MS ?? 240_000),
|
|
366
|
+
},
|
|
367
|
+
{ messages: messages.filter((message) => message.role !== 'system') },
|
|
368
|
+
)
|
|
369
|
+
if (result.usage.tokensKnown === false) throw new Error('provider omitted token usage')
|
|
370
|
+
const content = result.finalText
|
|
366
371
|
// Reasoning models starve `content` when reasoning exhausts max_tokens — an
|
|
367
372
|
// empty reply is a transient fault to retry, not a candidate to score.
|
|
368
373
|
if (content.trim() === '') {
|
|
369
374
|
lastErr = 'empty content'
|
|
370
375
|
continue
|
|
371
376
|
}
|
|
372
|
-
return {
|
|
377
|
+
return {
|
|
378
|
+
content,
|
|
379
|
+
attempts: attempt,
|
|
380
|
+
tokensIn: result.usage.input,
|
|
381
|
+
tokensOut: result.usage.output,
|
|
382
|
+
}
|
|
373
383
|
} catch (e) {
|
|
374
384
|
lastErr = e instanceof Error ? e.message : String(e)
|
|
375
|
-
} finally {
|
|
376
|
-
clearTimeout(timer)
|
|
377
385
|
}
|
|
378
386
|
}
|
|
379
387
|
throw new Error(`completion failed after retries: ${lastErr}`)
|
package/src/humaneval-gate.mts
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* it calls the router directly). That is the *degenerate* rollout (HARNESS.md's
|
|
16
16
|
* "Terminology"): it isolates the SELECTOR with the generator unable to self-correct,
|
|
17
17
|
* so it measures the selector's value at its MAXIMUM. A real rollout (an `AgentProfile`
|
|
18
|
-
* through `
|
|
18
|
+
* through `runAgentRounds`, `maxTurns>0` over a persistent workspace) self-verifies by
|
|
19
19
|
* iterating, which shrinks the external selector's job — that is the next experiment,
|
|
20
20
|
* not this one. A positive result here is the science (the selector works in a
|
|
21
21
|
* deployable-checker regime), not the product.
|
|
@@ -43,7 +43,12 @@
|
|
|
43
43
|
|
|
44
44
|
import { composeStrategies } from './directives'
|
|
45
45
|
import { basePrompt, type CheckResult, extractCode, type HumanEvalTask, loadHumanEval, runChecker } from './benchmarks/humaneval'
|
|
46
|
-
import {
|
|
46
|
+
import {
|
|
47
|
+
benchRouterProfile,
|
|
48
|
+
type BenchRouterTarget,
|
|
49
|
+
runBenchRouterTurn,
|
|
50
|
+
withBenchProfile,
|
|
51
|
+
} from './router-turn'
|
|
47
52
|
import { selfConsistencySelect, verifierGroundedSelect } from './selector'
|
|
48
53
|
import { type PairedLift, pairedLift, pool } from './stats.mts'
|
|
49
54
|
|
|
@@ -84,12 +89,18 @@ async function main(): Promise<void> {
|
|
|
84
89
|
if (!Number.isInteger(k) || k < 1) throw new Error(`K must be a positive integer, got ${process.env.K}`)
|
|
85
90
|
if (!Number.isInteger(offset) || offset < 0) throw new Error(`OFFSET must be a non-negative integer, got ${process.env.OFFSET}`)
|
|
86
91
|
|
|
87
|
-
const cfg:
|
|
92
|
+
const cfg: BenchRouterTarget = {
|
|
93
|
+
routerBaseUrl,
|
|
94
|
+
routerKey,
|
|
95
|
+
profile: benchRouterProfile('humaneval-gate-worker', model, {
|
|
96
|
+
temperature: Number(process.env.TEMPERATURE ?? '0.8'),
|
|
97
|
+
}),
|
|
98
|
+
}
|
|
88
99
|
|
|
89
100
|
console.log(`=== HumanEval deployable-verifier gate · N=${n} K=${k} offset=${offset} model=${model} ===`)
|
|
90
101
|
console.log(` router=${routerBaseUrl} docker=${dockerImage} (--network=none, timeout ${dockerTimeoutMs}ms)`)
|
|
91
102
|
console.log(
|
|
92
|
-
' regime: STATELESS single completions (
|
|
103
|
+
' regime: STATELESS single completions (one exact AgentProfile turn, no sandbox) — the selector no-self-correction LOWER BOUND, not a rollout/product number',
|
|
93
104
|
)
|
|
94
105
|
|
|
95
106
|
const tasks = await loadHumanEval(n, offset)
|
|
@@ -110,10 +121,15 @@ async function main(): Promise<void> {
|
|
|
110
121
|
console.log(`\n▶ solving ${units.length} attempts (${tasks.length} tasks × ${k} shots × 2 arms) via router, conc=${solveConcurrency}`)
|
|
111
122
|
|
|
112
123
|
const codes = await pool(units, solveConcurrency, async (u) => {
|
|
113
|
-
const res = await
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
124
|
+
const res = await runBenchRouterTurn(
|
|
125
|
+
{
|
|
126
|
+
routerBaseUrl: cfg.routerBaseUrl,
|
|
127
|
+
routerKey: cfg.routerKey,
|
|
128
|
+
profile: withBenchProfile(cfg.profile, { name: 'humaneval-gate-worker' }),
|
|
129
|
+
},
|
|
130
|
+
u.prompt,
|
|
131
|
+
)
|
|
132
|
+
return extractCode(res.finalText)
|
|
117
133
|
})
|
|
118
134
|
|
|
119
135
|
console.log(`▶ running ${codes.length} candidates through the Docker deployable checker, conc=${dockerConcurrency}`)
|
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
* container with a hard timeout — seconds per task. Paired
|
|
20
20
|
* McNemar over the per-task pass/fail difference gives the significance.
|
|
21
21
|
*
|
|
22
|
-
* Run from cwd=bench: env WORKER_MODEL=
|
|
22
|
+
* Run from cwd=bench: env WORKER_MODEL=deepseek-v4-flash N=60 K=3 \
|
|
23
23
|
* REPS=2 node_modules/.bin/tsx src/humaneval-object-ablation.mts
|
|
24
24
|
*/
|
|
25
25
|
import {
|
|
@@ -28,11 +28,13 @@ import {
|
|
|
28
28
|
runPythonProgram,
|
|
29
29
|
type HumanEvalTask,
|
|
30
30
|
} from './benchmarks/humaneval'
|
|
31
|
+
import { runBenchRouterTurn } from './router-turn'
|
|
31
32
|
|
|
32
33
|
const ROUTER = process.env.ROUTER_BASE ?? 'https://router.tangle.tools/v1'
|
|
33
34
|
const KEY = process.env.TANGLE_API_KEY
|
|
34
35
|
if (!KEY) throw new Error('TANGLE_API_KEY required')
|
|
35
|
-
const
|
|
36
|
+
const ROUTER_KEY = KEY
|
|
37
|
+
const MODEL = process.env.WORKER_MODEL ?? 'deepseek-v4-flash'
|
|
36
38
|
const N = Number(process.env.N ?? 60)
|
|
37
39
|
const OFFSET = Number(process.env.OFFSET ?? 0)
|
|
38
40
|
const K = Number(process.env.K ?? 3) // rounds/budget per task (equal for both arms)
|
|
@@ -40,23 +42,59 @@ const REPS = Number(process.env.REPS ?? 2)
|
|
|
40
42
|
const CONC = Number(process.env.CONCURRENCY ?? 6)
|
|
41
43
|
const EXEC_TIMEOUT = Number(process.env.EXEC_TIMEOUT_MS ?? 8000)
|
|
42
44
|
|
|
43
|
-
interface ChatMsg { role: string; content: string; tool_calls?: unknown; tool_call_id?: string; name?: string }
|
|
45
|
+
interface ChatMsg extends Record<string, unknown> { role: string; content: string; tool_calls?: unknown; tool_call_id?: string; name?: string }
|
|
44
46
|
interface Tool { type: 'function'; function: { name: string; description: string; parameters: unknown } }
|
|
45
47
|
|
|
46
48
|
async function router(messages: ChatMsg[], tools?: Tool[]): Promise<{ content: string; toolCalls: { id: string; name: string; args: Record<string, unknown> }[] }> {
|
|
47
|
-
const body: Record<string, unknown> = { model: MODEL, messages, temperature: 0.4 }
|
|
48
|
-
if (tools) { body.tools = tools; body.tool_choice = 'auto' }
|
|
49
49
|
for (let attempt = 0; ; attempt++) {
|
|
50
|
-
let res: Response
|
|
51
50
|
try {
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
51
|
+
const system = messages.find((message) => message.role === 'system')?.content
|
|
52
|
+
const result = await runBenchRouterTurn(
|
|
53
|
+
{
|
|
54
|
+
routerBaseUrl: ROUTER,
|
|
55
|
+
routerKey: ROUTER_KEY,
|
|
56
|
+
profile: {
|
|
57
|
+
name: 'humaneval-object-ablation-worker',
|
|
58
|
+
harness: 'cli-base',
|
|
59
|
+
model: {
|
|
60
|
+
provider: 'tangle-router',
|
|
61
|
+
default: MODEL,
|
|
62
|
+
metadata: {
|
|
63
|
+
temperature: 0.4,
|
|
64
|
+
...(tools ? { toolChoice: 'auto' } : {}),
|
|
65
|
+
},
|
|
66
|
+
},
|
|
67
|
+
...(system ? { prompt: { systemPrompt: system } } : {}),
|
|
68
|
+
...(tools
|
|
69
|
+
? { tools: Object.fromEntries(tools.map((tool) => [tool.function.name, true])) }
|
|
70
|
+
: {}),
|
|
71
|
+
},
|
|
72
|
+
...(tools ? { tools } : {}),
|
|
73
|
+
timeoutMs: Number(process.env.LLM_TIMEOUT_MS ?? 60_000),
|
|
74
|
+
},
|
|
75
|
+
{ messages: messages.filter((message) => message.role !== 'system') },
|
|
76
|
+
)
|
|
77
|
+
const toolCalls = result.toolCalls.map((call) => {
|
|
78
|
+
if (call.id === undefined) {
|
|
79
|
+
throw new Error(`router tool call '${call.name}' omitted its required id`)
|
|
80
|
+
}
|
|
81
|
+
let args: Record<string, unknown> = {}
|
|
82
|
+
try {
|
|
83
|
+
args = JSON.parse(call.arguments) as Record<string, unknown>
|
|
84
|
+
} catch {
|
|
85
|
+
// Keep the empty argument object; the tool returns a useful error.
|
|
86
|
+
}
|
|
87
|
+
return { id: call.id, name: call.name, args }
|
|
88
|
+
})
|
|
89
|
+
return { content: result.finalText, toolCalls }
|
|
90
|
+
} catch (error) {
|
|
91
|
+
const message = error instanceof Error ? error.message : String(error)
|
|
92
|
+
const status = Number(/router (\d+)/.exec(message)?.[1])
|
|
93
|
+
const transient =
|
|
94
|
+
!Number.isFinite(status) || [408, 429, 500, 502, 503, 504, 520, 522, 524].includes(status)
|
|
95
|
+
if (!transient || attempt >= 5) throw error
|
|
96
|
+
await sleep(800 * 2 ** attempt)
|
|
97
|
+
}
|
|
60
98
|
}
|
|
61
99
|
}
|
|
62
100
|
const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms))
|