@tangle-network/agent-bench 0.9.4 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/HARNESS.md +10 -10
- package/dist/index.d.ts +12 -3
- package/dist/index.js +55 -9
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +12 -2
- package/scripts/run-package-tests.test.mjs +26 -1
- package/src/run-benchmarks.test.mts +84 -3
- package/src/run-benchmarks.ts +52 -9
package/HARNESS.md
CHANGED
|
@@ -62,27 +62,27 @@ Use `LOOP_ATTEMPTS=N` only when the benchmark's own visible feedback is allowed
|
|
|
62
62
|
`runBenchmarks()` returns each judged artifact, worker events, and observed usage in `perTask`.
|
|
63
63
|
Retry usage includes every attempt; missing receipts leave the measured subtotal explicitly incomplete.
|
|
64
64
|
Judge failures retain completed worker evidence.
|
|
65
|
+
Each task separates `execution` from `measurement` availability.
|
|
66
|
+
Captured empty output and explicit failed turns remain measured failures when the evaluator runs successfully.
|
|
67
|
+
Read, extraction, and judge failures leave measurement unavailable while retaining observed usage.
|
|
68
|
+
Missing dispatch evidence remains unknown; `ok: false` never establishes permission to retry.
|
|
65
69
|
Errors propagated by `close()` remain in `detail` beside the settled task outcome.
|
|
66
70
|
The current Runtime lineage suppresses sandbox deletion errors, so a returned result does not confirm resource deletion.
|
|
67
71
|
The caller's abort signal stops queued shots and reaches active sandbox turns.
|
|
68
72
|
`modelApiKey` supplies sandbox inference authorization separately from the `routerKey` used for sandbox control.
|
|
69
73
|
|
|
70
|
-
###
|
|
74
|
+
### Retained strategy driver
|
|
71
75
|
|
|
72
76
|
```bash
|
|
73
77
|
cd bench
|
|
74
78
|
pnpm tsx src/swe-self-improve.mts
|
|
75
79
|
```
|
|
76
80
|
|
|
77
|
-
This
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
- explicit train, selection, and frozen final-test partitions;
|
|
83
|
-
- Runtime's `improve()` boundary and complete cost receipts.
|
|
84
|
-
|
|
85
|
-
It proves the integrated execution path can support a real value campaign. A paid powered result still belongs in Discovery Lab.
|
|
81
|
+
This driver uses `runStrategyEvolution` with SWE-bench tasks and a frozen holdout.
|
|
82
|
+
It does not exercise `improve`, and it deletes its temporary run directory on exit.
|
|
83
|
+
It therefore cannot provide retained improvement or lineage evidence.
|
|
84
|
+
Use `examples/improve` for the maintained offline API fixture.
|
|
85
|
+
Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
|
|
86
86
|
|
|
87
87
|
### Offline diagnostics
|
|
88
88
|
|
package/dist/index.d.ts
CHANGED
|
@@ -11,7 +11,7 @@ import { createT2RagBenchAdapter } from "./benchmarks/t2-ragbench.js";
|
|
|
11
11
|
import { AgentProfile, SandboxClient, sumSandboxUsage } from "@tangle-network/agent-runtime/kernel";
|
|
12
12
|
import { SandboxEvent } from "@tangle-network/sandbox";
|
|
13
13
|
import { AgentCandidateBenchmarkGraderPort, AgentCandidateExecutionClaimStore, AgentCandidateExecutorPort, AgentCandidateExecutorRequest, AgentCandidateExecutorStopRequest, AgentCandidateOutputArtifactPort, AgentCandidateRunFinalization, PreparedAgentCandidateExecution } from "@tangle-network/agent-runtime/candidate-execution";
|
|
14
|
-
import { TraceStore } from "@tangle-network/agent-eval";
|
|
14
|
+
import { RunTerminalOutcome, TraceStore } from "@tangle-network/agent-eval";
|
|
15
15
|
//#region src/resolve-client.d.ts
|
|
16
16
|
interface ResolveBenchClientOptions {
|
|
17
17
|
/** The selector (the `BACKEND` env value): `router` = off-box; anything else = in-box Sandbox. */
|
|
@@ -56,6 +56,13 @@ interface BenchShotResult {
|
|
|
56
56
|
/** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
|
|
57
57
|
readonly usage?: ReturnType<typeof sumSandboxUsage>;
|
|
58
58
|
readonly events?: readonly SandboxEvent[];
|
|
59
|
+
/** Observed dispatch and terminal state, independent of artifact quality. */
|
|
60
|
+
readonly execution?: {
|
|
61
|
+
readonly phase: 'not-started' | 'started' | 'unknown';
|
|
62
|
+
readonly terminalOutcome: RunTerminalOutcome;
|
|
63
|
+
};
|
|
64
|
+
/** Whether the artifact was captured without a read or extraction failure. */
|
|
65
|
+
readonly artifactAvailable?: boolean;
|
|
59
66
|
}
|
|
60
67
|
/** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
|
|
61
68
|
type BenchShot = (input: {
|
|
@@ -122,9 +129,11 @@ interface BenchCellTaskResult {
|
|
|
122
129
|
readonly rep: number;
|
|
123
130
|
readonly resolved: boolean;
|
|
124
131
|
readonly score: number;
|
|
125
|
-
/**
|
|
126
|
-
* denominator so a harness outage can't masquerade as a 0% capability result. */
|
|
132
|
+
/** Whether execution completed successfully and produced a readable, nonempty artifact. */
|
|
127
133
|
readonly ok: boolean;
|
|
134
|
+
readonly execution?: BenchShotResult['execution'];
|
|
135
|
+
/** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
|
|
136
|
+
readonly measurement?: 'available' | 'unavailable';
|
|
128
137
|
readonly detail?: string;
|
|
129
138
|
readonly wallMs: number;
|
|
130
139
|
/** Exact bytes given to the benchmark judge, retained even when judging fails. */
|
package/dist/index.js
CHANGED
|
@@ -304,11 +304,15 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
304
304
|
};
|
|
305
305
|
const controller = new AbortController();
|
|
306
306
|
const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : void 0;
|
|
307
|
+
const observedEvents = [];
|
|
307
308
|
const runOptions = {
|
|
308
309
|
agentRun,
|
|
309
310
|
signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
|
|
310
311
|
runId: `bench:${adapter.name}:${task.id}:${uniq}`,
|
|
311
|
-
scenarioId: task.id
|
|
312
|
+
scenarioId: task.id,
|
|
313
|
+
onSandboxEvent: (event) => {
|
|
314
|
+
observedEvents.push(event);
|
|
315
|
+
}
|
|
312
316
|
};
|
|
313
317
|
const boxSetup = adapter.boxSetup;
|
|
314
318
|
if (boxSetup) runOptions.beforeStart = async ({ box, sessionId }) => {
|
|
@@ -327,12 +331,23 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
327
331
|
};
|
|
328
332
|
try {
|
|
329
333
|
run = await openSandboxRun(client, runOptions, deliverable);
|
|
334
|
+
result = {
|
|
335
|
+
...result,
|
|
336
|
+
execution: {
|
|
337
|
+
phase: "unknown",
|
|
338
|
+
terminalOutcome: "unknown"
|
|
339
|
+
}
|
|
340
|
+
};
|
|
330
341
|
const turn = await run.start(prompt ?? task.prompt);
|
|
331
342
|
result = {
|
|
332
343
|
artifact: "",
|
|
333
344
|
ok: false,
|
|
334
345
|
usage: sumSandboxUsage(turn.events),
|
|
335
|
-
events: turn.events
|
|
346
|
+
events: turn.events,
|
|
347
|
+
execution: {
|
|
348
|
+
phase: "started",
|
|
349
|
+
terminalOutcome: turn.outcome.success ? "succeeded" : turn.outcome.status === "failed" ? "failed" : "incomplete"
|
|
350
|
+
}
|
|
336
351
|
};
|
|
337
352
|
let artifact = (turn.out ?? "").trim();
|
|
338
353
|
let boxExtractError;
|
|
@@ -379,19 +394,29 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
379
394
|
result = {
|
|
380
395
|
artifact,
|
|
381
396
|
ok: turn.outcome.success && artifact.length > 0 && turn.readError === void 0 && boxExtractError === void 0,
|
|
397
|
+
execution: result.execution,
|
|
398
|
+
artifactAvailable: turn.readError === void 0 && boxExtractError === void 0,
|
|
382
399
|
usage: result.usage,
|
|
383
400
|
events: turn.events,
|
|
384
401
|
...detail ? { detail } : {}
|
|
385
402
|
};
|
|
386
403
|
} catch (err) {
|
|
404
|
+
const events = err instanceof SandboxRunAbortError ? err.events : observedEvents;
|
|
387
405
|
result = {
|
|
388
406
|
...result,
|
|
389
407
|
ok: false,
|
|
408
|
+
artifactAvailable: false,
|
|
409
|
+
execution: {
|
|
410
|
+
phase: events.length > 0 ? "started" : "unknown",
|
|
411
|
+
terminalOutcome: result.execution?.terminalOutcome ?? "unknown"
|
|
412
|
+
},
|
|
390
413
|
detail: err instanceof Error ? err.message : String(err),
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
414
|
+
usage: {
|
|
415
|
+
...sumSandboxUsage(events),
|
|
416
|
+
tokensKnown: false,
|
|
417
|
+
usdKnown: false
|
|
418
|
+
},
|
|
419
|
+
events
|
|
395
420
|
};
|
|
396
421
|
} finally {
|
|
397
422
|
if (timer) clearTimeout(timer);
|
|
@@ -496,6 +521,11 @@ async function loopedShot(input, shot, attempts) {
|
|
|
496
521
|
return {
|
|
497
522
|
artifact: completed.at(-1)?.artifact ?? "",
|
|
498
523
|
ok: false,
|
|
524
|
+
execution: pendingShot ? {
|
|
525
|
+
phase: "unknown",
|
|
526
|
+
terminalOutcome: "unknown"
|
|
527
|
+
} : completed.at(-1)?.execution,
|
|
528
|
+
artifactAvailable: false,
|
|
499
529
|
usage: combinedUsage(pendingShot ? [...completed, {
|
|
500
530
|
artifact: "",
|
|
501
531
|
ok: false
|
|
@@ -505,8 +535,10 @@ async function loopedShot(input, shot, attempts) {
|
|
|
505
535
|
};
|
|
506
536
|
}
|
|
507
537
|
const best = result.rounds.reduce((winner, candidate) => {
|
|
508
|
-
|
|
509
|
-
|
|
538
|
+
const candidateShot = shots.get(candidate.round);
|
|
539
|
+
const winnerShot = shots.get(winner.round);
|
|
540
|
+
const rank = (shot) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0;
|
|
541
|
+
if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner;
|
|
510
542
|
const a = scores.get(winner.round);
|
|
511
543
|
const b = scores.get(candidate.round);
|
|
512
544
|
if (!a) return candidate;
|
|
@@ -519,6 +551,8 @@ async function loopedShot(input, shot, attempts) {
|
|
|
519
551
|
return {
|
|
520
552
|
artifact: best.artifact,
|
|
521
553
|
ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
|
|
554
|
+
execution: shots.get(best.round)?.execution,
|
|
555
|
+
artifactAvailable: shots.get(best.round)?.artifactAvailable,
|
|
522
556
|
usage: combinedUsage([...shots.values()]),
|
|
523
557
|
events: [...shots.values()].flatMap((shot) => shot.events ?? []),
|
|
524
558
|
detail: JSON.stringify({
|
|
@@ -641,6 +675,7 @@ async function runBenchmarks(opts) {
|
|
|
641
675
|
const startedAt = Date.now();
|
|
642
676
|
let result;
|
|
643
677
|
let out;
|
|
678
|
+
let invoked = false;
|
|
644
679
|
try {
|
|
645
680
|
opts.signal?.throwIfAborted();
|
|
646
681
|
const shotInput = {
|
|
@@ -657,6 +692,7 @@ async function runBenchmarks(opts) {
|
|
|
657
692
|
...opts.signal ? { signal: opts.signal } : {},
|
|
658
693
|
...opts.resolveClient ? { resolveClient: opts.resolveClient } : {}
|
|
659
694
|
};
|
|
695
|
+
invoked = true;
|
|
660
696
|
out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput);
|
|
661
697
|
const score = await job.adapter.judge(job.task, out.artifact);
|
|
662
698
|
result = {
|
|
@@ -667,6 +703,11 @@ async function runBenchmarks(opts) {
|
|
|
667
703
|
resolved: out.ok && score.resolved,
|
|
668
704
|
score: out.ok ? score.score : 0,
|
|
669
705
|
ok: out.ok,
|
|
706
|
+
execution: out.execution ?? {
|
|
707
|
+
phase: out.ok ? "started" : "unknown",
|
|
708
|
+
terminalOutcome: out.ok ? "succeeded" : "unknown"
|
|
709
|
+
},
|
|
710
|
+
measurement: out.artifactAvailable ?? out.ok ? "available" : "unavailable",
|
|
670
711
|
...out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {},
|
|
671
712
|
wallMs: Date.now() - startedAt,
|
|
672
713
|
artifact: out.artifact,
|
|
@@ -682,6 +723,11 @@ async function runBenchmarks(opts) {
|
|
|
682
723
|
resolved: false,
|
|
683
724
|
score: 0,
|
|
684
725
|
ok: false,
|
|
726
|
+
execution: out?.execution ?? {
|
|
727
|
+
phase: !invoked ? "not-started" : out?.ok ? "started" : "unknown",
|
|
728
|
+
terminalOutcome: out?.ok ? "succeeded" : "unknown"
|
|
729
|
+
},
|
|
730
|
+
measurement: "unavailable",
|
|
685
731
|
detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
|
|
686
732
|
wallMs: Date.now() - startedAt,
|
|
687
733
|
...out === void 0 ? {} : { artifact: out.artifact },
|
|
@@ -714,7 +760,7 @@ function aggregate(perTask) {
|
|
|
714
760
|
scoreSum: 0
|
|
715
761
|
};
|
|
716
762
|
e.n += 1;
|
|
717
|
-
if (
|
|
763
|
+
if ((r.measurement ?? (r.ok ? "available" : "unavailable")) === "unavailable") e.errored += 1;
|
|
718
764
|
else {
|
|
719
765
|
if (r.resolved) e.resolved += 1;
|
|
720
766
|
e.scoreSum += r.score;
|