@tangle-network/agent-bench 0.9.4 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/HARNESS.md +53 -15
- package/README.md +22 -0
- package/dist/index.d.ts +45 -5
- package/dist/index.js +180 -16
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +12 -2
- package/scripts/run-package-tests.test.mjs +26 -1
- package/scripts/verify-packed-consumer.mjs +47 -1
- package/src/index.ts +3 -0
- package/src/run-benchmarks.test.mts +262 -4
- package/src/run-benchmarks.ts +195 -12
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.11.0
|
|
4
|
+
|
|
5
|
+
Adds caller-controlled start and resume inside managed benchmark shots.
|
|
6
|
+
Retains setup, final extraction, grading, and cleanup while recording every prompt and its observed usage.
|
|
7
|
+
The callback composes with outer retry attempts; learning policy and policy inference accounting remain consumer responsibilities.
|
|
8
|
+
|
|
3
9
|
## 0.9.4
|
|
4
10
|
|
|
5
11
|
Requires Runtime 0.202.0 so Bench consumers receive explicit failed-execution settlement and retained failure recovery.
|
package/HARNESS.md
CHANGED
|
@@ -8,9 +8,11 @@
|
|
|
8
8
|
|---|---|
|
|
9
9
|
| **agent-runtime** | exact execution, reusable benchmark adapters, packed-consumer checks, one full-fidelity integration fixture |
|
|
10
10
|
| **discovery** | research questions, preregistrations, acceptance criteria, negative results, and decisions about what is worth testing |
|
|
11
|
-
| **
|
|
11
|
+
| **supervisor-lab** | registered adaptive-agent and retained-learning comparisons, including carried versus revised profiles |
|
|
12
|
+
| **discovery-lab** | research campaigns and evidence for Discovery, including immutable inputs, run records, and result archives |
|
|
12
13
|
|
|
13
|
-
A benchmark implementation may begin here while it is becoming a reusable adapter. Once the question is “does method X improve benchmark Y?”, the campaign belongs in
|
|
14
|
+
A benchmark implementation may begin here while it is becoming a reusable adapter. Once the question is “does method X improve benchmark Y?”, the campaign belongs in the consuming lab.
|
|
15
|
+
Supervisor Lab owns adaptive-learning comparisons; Discovery Lab owns research campaigns for Discovery.
|
|
14
16
|
|
|
15
17
|
## Evidence levels
|
|
16
18
|
|
|
@@ -20,8 +22,8 @@ Use these labels literally. Do not promote one level into another in prose.
|
|
|
20
22
|
|---|---|---|
|
|
21
23
|
| **contract proof** | packages install; identities, budgets, callbacks, resume, and receipts have the expected shape | root `pnpm verify:official-optimizers`, `pnpm verify:bench` |
|
|
22
24
|
| **evaluator proof** | the benchmark's own evaluator can distinguish known fail/pass artifacts in the exact environment | adapter preflight and gold/self-check |
|
|
23
|
-
| **reproduction proof** | an upstream method is run at a pinned revision on its claimed benchmark under a matched protocol |
|
|
24
|
-
| **value proof** | the integrated method beats the preregistered baseline on frozen evidence with uncertainty and complete cost accounting |
|
|
25
|
+
| **reproduction proof** | an upstream method is run at a pinned revision on its claimed benchmark under a matched protocol | Consuming lab reproduction manifest and runner |
|
|
26
|
+
| **value proof** | the integrated method beats the preregistered baseline on frozen evidence with uncertainty and complete cost accounting | Consuming lab result receipt |
|
|
25
27
|
| **production proof** | a promoted artifact transfers to real traffic under a canary or controlled rollout | product repository / platform telemetry |
|
|
26
28
|
|
|
27
29
|
A localization score, output-shape check, LLM quality judge, or toy deterministic reward can be useful for development. None is a substitute for the benchmark's outcome evaluator.
|
|
@@ -62,27 +64,63 @@ Use `LOOP_ATTEMPTS=N` only when the benchmark's own visible feedback is allowed
|
|
|
62
64
|
`runBenchmarks()` returns each judged artifact, worker events, and observed usage in `perTask`.
|
|
63
65
|
Retry usage includes every attempt; missing receipts leave the measured subtotal explicitly incomplete.
|
|
64
66
|
Judge failures retain completed worker evidence.
|
|
67
|
+
Each task separates `execution` from `measurement` availability.
|
|
68
|
+
Captured empty output and explicit failed turns remain measured failures when the evaluator runs successfully.
|
|
69
|
+
Read, extraction, and judge failures leave measurement unavailable while retaining observed usage.
|
|
70
|
+
Missing dispatch evidence remains unknown; `ok: false` never establishes permission to retry.
|
|
65
71
|
Errors propagated by `close()` remain in `detail` beside the settled task outcome.
|
|
66
72
|
The current Runtime lineage suppresses sandbox deletion errors, so a returned result does not confirm resource deletion.
|
|
67
73
|
The caller's abort signal stops queued shots and reaches active sandbox turns.
|
|
68
74
|
`modelApiKey` supplies sandbox inference authorization separately from the `routerKey` used for sandbox control.
|
|
69
75
|
|
|
70
|
-
###
|
|
76
|
+
### Caller-controlled prompts within a task
|
|
77
|
+
|
|
78
|
+
`runBenchmarks({ execute })` invokes the callback inside each managed sandbox shot.
|
|
79
|
+
The context supplies the task prompt, executed profile, benchmark and task identities, attempt number, and abort signal.
|
|
80
|
+
Its managed `run` exposes Runtime's `start`, `resume`, `box`, and `sessionId`.
|
|
81
|
+
Use the live box for permitted working checks in the same session.
|
|
82
|
+
Submit worker prompts through managed `start` and `resume` so Bench captures their outcomes and usage.
|
|
83
|
+
Direct sandbox prompt calls bypass this accounting.
|
|
84
|
+
The callback must leave session lifecycle, extraction, and cleanup to Bench.
|
|
85
|
+
It receives no adapter, final grader, or task metadata containing gold material.
|
|
86
|
+
This callback is trusted consumer code, not an isolation boundary for arbitrary code.
|
|
87
|
+
|
|
88
|
+
Return after the final prompt completes.
|
|
89
|
+
Bench extracts the last captured prompt's artifact and applies the adapter's final grading outside the callback.
|
|
90
|
+
Start and resume calls must be sequential.
|
|
91
|
+
Bench waits for an unawaited active invocation before extraction and refuses calls after the callback settles.
|
|
92
|
+
The callback must use its signal to cancel external work.
|
|
93
|
+
Bench stops awaiting policy work when cancelled, but cannot stop external effects that ignore cancellation.
|
|
94
|
+
|
|
95
|
+
Each task's `prompts` retains ordered method, prompt, attempt, session identity, outcome, events, usage, and capture errors.
|
|
96
|
+
Prompt indices start at zero; attempt numbers start at one.
|
|
97
|
+
Usage sums each prompt separately, including failed and interrupted prompts.
|
|
98
|
+
A partial capture retains observed counters and marks accounting incomplete.
|
|
99
|
+
These counters cover sandbox workers only; consumers must account for policy and working-evaluator inference separately.
|
|
100
|
+
Consumers must bind their callback source, configuration, profiles, and initial state to their execution identity.
|
|
101
|
+
|
|
102
|
+
`execute` runs inside every `loopAttempts` shot.
|
|
103
|
+
Those outer attempts still create fresh sandboxes and use the existing checker feedback policy.
|
|
104
|
+
Use one outer attempt when final grading must remain unavailable to adaptation.
|
|
105
|
+
A custom `runShot` receives `execute` and owns whether it consumes the callback.
|
|
106
|
+
|
|
107
|
+
A sandbox prompt can contain multiple native model requests.
|
|
108
|
+
Same-session continuation does not prove a barrier before every native request, profile reload, coordinator restart, or fresh-session state transfer.
|
|
109
|
+
Offline fake-sandbox tests prove the managed contract and correction consumption only.
|
|
110
|
+
They establish no live learning gain or provider session restoration.
|
|
111
|
+
|
|
112
|
+
### Retained strategy driver
|
|
71
113
|
|
|
72
114
|
```bash
|
|
73
115
|
cd bench
|
|
74
116
|
pnpm tsx src/swe-self-improve.mts
|
|
75
117
|
```
|
|
76
118
|
|
|
77
|
-
This
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
- explicit train, selection, and frozen final-test partitions;
|
|
83
|
-
- Runtime's `improve()` boundary and complete cost receipts.
|
|
84
|
-
|
|
85
|
-
It proves the integrated execution path can support a real value campaign. A paid powered result still belongs in Discovery Lab.
|
|
119
|
+
This driver uses `runStrategyEvolution` with SWE-bench tasks and a frozen holdout.
|
|
120
|
+
It does not exercise `improve`, and it deletes its temporary run directory on exit.
|
|
121
|
+
It therefore cannot provide retained improvement or lineage evidence.
|
|
122
|
+
Use `examples/improve` for the maintained offline API fixture.
|
|
123
|
+
Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
|
|
86
124
|
|
|
87
125
|
### Offline diagnostics
|
|
88
126
|
|
|
@@ -133,4 +171,4 @@ A new file under `bench/src` must be one of:
|
|
|
133
171
|
- a package-consumer or evaluator calibration test;
|
|
134
172
|
- one canonical full-fidelity fixture that exercises a public Runtime contract.
|
|
135
173
|
|
|
136
|
-
A one-off campaign, generation-N optimizer script, bespoke dashboard, or historical result belongs in
|
|
174
|
+
A one-off campaign, generation-N optimizer script, bespoke dashboard, or historical result belongs in the consuming lab. If an older file has no package script, no importer, and no unique reusable primitive, delete it rather than adding another index entry.
|
package/README.md
CHANGED
|
@@ -100,3 +100,25 @@ Pier owns the task container and verifier; protected model usage and traces stay
|
|
|
100
100
|
`FilePierCandidateTrialController` atomically reserves a unique Pier job, then persists the supervisor PID, process-session identity, and that job's exact Docker projects so a fresh evaluator process can stop and remove an abandoned trial.
|
|
101
101
|
Run `PIER_REPO=/path/to/pier pnpm verify:pier` for the zero-model failure/pass and fresh-process recovery proof, and see `HARNESS.md` for the exact invocation and failure contract.
|
|
102
102
|
From an installed npm package, expose the shipped Python module with `export PYTHONPATH="$(npm root)/@tangle-network/agent-bench${PYTHONPATH:+:$PYTHONPATH}"` before invoking Pier.
|
|
103
|
+
|
|
104
|
+
## Control execution within a managed shot
|
|
105
|
+
|
|
106
|
+
Supply `execute` to control whole sandbox prompts while Bench owns setup, extraction, grading, and cleanup.
|
|
107
|
+
Your policy can inspect permitted working checks through the existing sandbox handle.
|
|
108
|
+
|
|
109
|
+
```ts
|
|
110
|
+
import { runBenchmarks, type BenchExecution } from '@tangle-network/agent-bench'
|
|
111
|
+
import { executionOptions, chooseNextPrompt } from './policy.js'
|
|
112
|
+
|
|
113
|
+
const execute: BenchExecution = async ({ run, prompt, signal }) => {
|
|
114
|
+
const first = await run.start(prompt)
|
|
115
|
+
const correction = await chooseNextPrompt({ first, box: run.box, sessionId: run.sessionId, signal })
|
|
116
|
+
if (correction !== undefined) await run.resume(correction)
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const report = await runBenchmarks({ ...executionOptions, execute })
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The consumer supplies and identifies `chooseNextPrompt`; Bench provides no learning policy.
|
|
123
|
+
`perTask[].prompts` retains each invocation and its observed worker cost, including partial failures.
|
|
124
|
+
See [the execution contract](./HARNESS.md#caller-controlled-prompts-within-a-task) for accounting, retry composition, and lifecycle limits.
|
package/dist/index.d.ts
CHANGED
|
@@ -8,10 +8,10 @@ import { FINAL_ANSWER_SENTINEL, RagAnswerScore, RagContext, answerScoreToBenchSc
|
|
|
8
8
|
import { createRagBenchAdapter } from "./benchmarks/ragbench.js";
|
|
9
9
|
import { SweBenchAdapterOptions, SweBenchArtifactCaptureContext, SweBenchCacheLevel, createSweBenchAdapter, scoreSweReport, sweEvaluationArgv, swePatchOutput } from "./benchmarks/swe-bench.js";
|
|
10
10
|
import { createT2RagBenchAdapter } from "./benchmarks/t2-ragbench.js";
|
|
11
|
-
import { AgentProfile, SandboxClient, sumSandboxUsage } from "@tangle-network/agent-runtime/kernel";
|
|
11
|
+
import { AgentProfile, SandboxClient, SandboxRun, TurnResult, sumSandboxUsage } from "@tangle-network/agent-runtime/kernel";
|
|
12
12
|
import { SandboxEvent } from "@tangle-network/sandbox";
|
|
13
13
|
import { AgentCandidateBenchmarkGraderPort, AgentCandidateExecutionClaimStore, AgentCandidateExecutorPort, AgentCandidateExecutorRequest, AgentCandidateExecutorStopRequest, AgentCandidateOutputArtifactPort, AgentCandidateRunFinalization, PreparedAgentCandidateExecution } from "@tangle-network/agent-runtime/candidate-execution";
|
|
14
|
-
import { TraceStore } from "@tangle-network/agent-eval";
|
|
14
|
+
import { RunTerminalOutcome, TraceStore } from "@tangle-network/agent-eval";
|
|
15
15
|
//#region src/resolve-client.d.ts
|
|
16
16
|
interface ResolveBenchClientOptions {
|
|
17
17
|
/** The selector (the `BACKEND` env value): `router` = off-box; anything else = in-box Sandbox. */
|
|
@@ -48,6 +48,31 @@ interface BenchCell {
|
|
|
48
48
|
/** The agent under test. Defaults to a minimal `{ name, metadata.backendType }` profile. */
|
|
49
49
|
readonly profile?: AgentProfile;
|
|
50
50
|
}
|
|
51
|
+
/** Caller-owned work inside a managed benchmark shot. Final grading stays outside this callback. */
|
|
52
|
+
interface BenchExecutionContext {
|
|
53
|
+
readonly prompt: string;
|
|
54
|
+
readonly profile: AgentProfile;
|
|
55
|
+
readonly benchmark: string;
|
|
56
|
+
readonly taskId: string;
|
|
57
|
+
readonly attempt: number;
|
|
58
|
+
readonly signal: AbortSignal;
|
|
59
|
+
/** Runtime owns session continuity. Bench owns capture, extraction, and close. */
|
|
60
|
+
readonly run: Pick<SandboxRun<string>, 'start' | 'resume' | 'box' | 'sessionId'>;
|
|
61
|
+
}
|
|
62
|
+
type BenchExecution = (context: BenchExecutionContext) => Promise<void>;
|
|
63
|
+
/** One submitted sandbox prompt, including partial evidence when its invocation throws. */
|
|
64
|
+
interface BenchPromptResult {
|
|
65
|
+
readonly attempt: number;
|
|
66
|
+
readonly index: number;
|
|
67
|
+
readonly method: 'start' | 'resume';
|
|
68
|
+
readonly prompt: string;
|
|
69
|
+
readonly sessionId?: string;
|
|
70
|
+
readonly outcome?: TurnResult<string>['outcome'];
|
|
71
|
+
readonly events: readonly SandboxEvent[];
|
|
72
|
+
readonly usage: ReturnType<typeof sumSandboxUsage>;
|
|
73
|
+
readonly readError?: string;
|
|
74
|
+
readonly error?: string;
|
|
75
|
+
}
|
|
51
76
|
/** A worker's artifact and observed execution evidence, before external grading. */
|
|
52
77
|
interface BenchShotResult {
|
|
53
78
|
readonly artifact: string;
|
|
@@ -56,6 +81,14 @@ interface BenchShotResult {
|
|
|
56
81
|
/** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
|
|
57
82
|
readonly usage?: ReturnType<typeof sumSandboxUsage>;
|
|
58
83
|
readonly events?: readonly SandboxEvent[];
|
|
84
|
+
readonly prompts?: readonly BenchPromptResult[];
|
|
85
|
+
/** Observed dispatch and terminal state, independent of artifact quality. */
|
|
86
|
+
readonly execution?: {
|
|
87
|
+
readonly phase: 'not-started' | 'started' | 'unknown';
|
|
88
|
+
readonly terminalOutcome: RunTerminalOutcome;
|
|
89
|
+
};
|
|
90
|
+
/** Whether the artifact was captured without a read or extraction failure. */
|
|
91
|
+
readonly artifactAvailable?: boolean;
|
|
59
92
|
}
|
|
60
93
|
/** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
|
|
61
94
|
type BenchShot = (input: {
|
|
@@ -66,6 +99,8 @@ type BenchShot = (input: {
|
|
|
66
99
|
readonly prompt?: string;
|
|
67
100
|
/** 1-based attempt index for looped runs. */
|
|
68
101
|
readonly attempt?: number;
|
|
102
|
+
/** Custom shots own whether they consume this managed execution callback. */
|
|
103
|
+
readonly execute?: BenchExecution;
|
|
69
104
|
readonly routerBaseUrl: string;
|
|
70
105
|
readonly routerKey: string;
|
|
71
106
|
/** Optional inference credential for the box; routerKey continues to authorize sandbox control. */
|
|
@@ -109,6 +144,8 @@ interface RunBenchmarksOptions {
|
|
|
109
144
|
/** Self-verify each benchmark's judge against its gold artifact on the first task before spending
|
|
110
145
|
* model tokens; a benchmark whose judge rejects its own gold is recorded unavailable. Default true. */
|
|
111
146
|
readonly verifyJudge?: boolean;
|
|
147
|
+
/** Caller policy inside each managed shot, including every refine-loop attempt. */
|
|
148
|
+
readonly execute?: BenchExecution;
|
|
112
149
|
/** Test seam: a deterministic shot runner. Defaults to the `openSandboxRun` leaf. */
|
|
113
150
|
readonly runShot?: BenchShot;
|
|
114
151
|
/** Test seam: resolve a benchmark key to an adapter. Defaults to the registry `resolveAdapter`. */
|
|
@@ -122,9 +159,11 @@ interface BenchCellTaskResult {
|
|
|
122
159
|
readonly rep: number;
|
|
123
160
|
readonly resolved: boolean;
|
|
124
161
|
readonly score: number;
|
|
125
|
-
/**
|
|
126
|
-
* denominator so a harness outage can't masquerade as a 0% capability result. */
|
|
162
|
+
/** Whether execution completed successfully and produced a readable, nonempty artifact. */
|
|
127
163
|
readonly ok: boolean;
|
|
164
|
+
readonly execution?: BenchShotResult['execution'];
|
|
165
|
+
/** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
|
|
166
|
+
readonly measurement?: 'available' | 'unavailable';
|
|
128
167
|
readonly detail?: string;
|
|
129
168
|
readonly wallMs: number;
|
|
130
169
|
/** Exact bytes given to the benchmark judge, retained even when judging fails. */
|
|
@@ -132,6 +171,7 @@ interface BenchCellTaskResult {
|
|
|
132
171
|
readonly usage?: ReturnType<typeof sumSandboxUsage>;
|
|
133
172
|
/** Worker events only; benchmark grading remains outside this trace. */
|
|
134
173
|
readonly events?: readonly SandboxEvent[];
|
|
174
|
+
readonly prompts?: readonly BenchPromptResult[];
|
|
135
175
|
}
|
|
136
176
|
interface BenchLeaderboardRow {
|
|
137
177
|
readonly benchmark: string;
|
|
@@ -319,5 +359,5 @@ declare class FilePierCandidateTrialController implements PierCandidateTrialCont
|
|
|
319
359
|
//#region src/pier-result-grader.d.ts
|
|
320
360
|
declare function createPierResultGrader(descriptor: Pick<PierCandidateGraderPort, 'name' | 'version' | 'artifact'>): PierCandidateGraderPort;
|
|
321
361
|
//#endregion
|
|
322
|
-
export { ADAPTERS, type BenchCell, type BenchCellTaskResult, type BenchLeaderboardRow, type BenchScore, type BenchShot, type BenchShotResult, type BenchTask, type BenchmarkAdapter, type ExecutePreparedPierCandidateOptions, FINAL_ANSWER_SENTINEL, FilePierCandidateTrialController, type FilePierCandidateTrialControllerOptions, type JudgeArtifactFileReceipt, type JudgeArtifactReceipt, type LoadOptions, type PierCandidateGraderPort, type PierCandidateOfficialResult, type PierCandidateProcessSpec, type PierCandidateTerminationAcknowledgement, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierCandidateTrialIdentity, type PierCandidateTrialResult, type PierDockerConnection, type RagAnswerScore, type RagContext, type RunBenchmarksOptions, type RunBenchmarksReport, StagedJudgeError, type StagedPierCandidateExecution, type StagedRunCaptureSpec, type StagedRunSpec, type SweBenchAdapterOptions, type SweBenchArtifactCaptureContext, type SweBenchCacheLevel, answerScoreToBenchScore, contextBlock, contextsFrom, createCragAdapter, createNoMiraclAdapter, createOpenRagBenchAdapter, createPierCandidateRecoveryExecutor, createPierResultGrader, createRagBenchAdapter, createSweBenchAdapter, createT2RagBenchAdapter, executePreparedPierCandidate, normalizeAnswer, parseCitations, parseFinalAnswer, printBenchmarksReport, ragAnswerOutput, resolveAdapter, runBenchmarks, runStagedJudge, scoreAnswerArtifact, scoreSweReport, sweEvaluationArgv, swePatchOutput, tokenF1 };
|
|
362
|
+
export { ADAPTERS, type BenchCell, type BenchCellTaskResult, type BenchExecution, type BenchExecutionContext, type BenchLeaderboardRow, type BenchPromptResult, type BenchScore, type BenchShot, type BenchShotResult, type BenchTask, type BenchmarkAdapter, type ExecutePreparedPierCandidateOptions, FINAL_ANSWER_SENTINEL, FilePierCandidateTrialController, type FilePierCandidateTrialControllerOptions, type JudgeArtifactFileReceipt, type JudgeArtifactReceipt, type LoadOptions, type PierCandidateGraderPort, type PierCandidateOfficialResult, type PierCandidateProcessSpec, type PierCandidateTerminationAcknowledgement, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierCandidateTrialIdentity, type PierCandidateTrialResult, type PierDockerConnection, type RagAnswerScore, type RagContext, type RunBenchmarksOptions, type RunBenchmarksReport, StagedJudgeError, type StagedPierCandidateExecution, type StagedRunCaptureSpec, type StagedRunSpec, type SweBenchAdapterOptions, type SweBenchArtifactCaptureContext, type SweBenchCacheLevel, answerScoreToBenchScore, contextBlock, contextsFrom, createCragAdapter, createNoMiraclAdapter, createOpenRagBenchAdapter, createPierCandidateRecoveryExecutor, createPierResultGrader, createRagBenchAdapter, createSweBenchAdapter, createT2RagBenchAdapter, executePreparedPierCandidate, normalizeAnswer, parseCitations, parseFinalAnswer, printBenchmarksReport, ragAnswerOutput, resolveAdapter, runBenchmarks, runStagedJudge, scoreAnswerArtifact, scoreSweReport, sweEvaluationArgv, swePatchOutput, tokenF1 };
|
|
323
363
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.js
CHANGED
|
@@ -252,7 +252,7 @@ function finalText(events) {
|
|
|
252
252
|
}
|
|
253
253
|
/** The default real-agent shot: one `openSandboxRun` over the cell's harness+model, deliverable
|
|
254
254
|
* extracted by the adapter's parser (or final text), abortable on `timeoutMs`. */
|
|
255
|
-
const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
255
|
+
const openSandboxShot = async ({ adapter, task, cell, prompt, attempt = 1, execute, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
256
256
|
signal?.throwIfAborted();
|
|
257
257
|
const client = (resolveClient ?? resolveBenchClient)({
|
|
258
258
|
backend: cell.backend ?? "router",
|
|
@@ -304,11 +304,15 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
304
304
|
};
|
|
305
305
|
const controller = new AbortController();
|
|
306
306
|
const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : void 0;
|
|
307
|
+
const observedEvents = [];
|
|
307
308
|
const runOptions = {
|
|
308
309
|
agentRun,
|
|
309
310
|
signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
|
|
310
311
|
runId: `bench:${adapter.name}:${task.id}:${uniq}`,
|
|
311
|
-
scenarioId: task.id
|
|
312
|
+
scenarioId: task.id,
|
|
313
|
+
onSandboxEvent: (event) => {
|
|
314
|
+
observedEvents.push(event);
|
|
315
|
+
}
|
|
312
316
|
};
|
|
313
317
|
const boxSetup = adapter.boxSetup;
|
|
314
318
|
if (boxSetup) runOptions.beforeStart = async ({ box, sessionId }) => {
|
|
@@ -325,14 +329,119 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
325
329
|
artifact: "",
|
|
326
330
|
ok: false
|
|
327
331
|
};
|
|
332
|
+
const prompts = [];
|
|
333
|
+
const execution = { accepting: true };
|
|
328
334
|
try {
|
|
329
335
|
run = await openSandboxRun(client, runOptions, deliverable);
|
|
330
|
-
|
|
336
|
+
result = {
|
|
337
|
+
...result,
|
|
338
|
+
execution: {
|
|
339
|
+
phase: "unknown",
|
|
340
|
+
terminalOutcome: "unknown"
|
|
341
|
+
}
|
|
342
|
+
};
|
|
343
|
+
const managedRun = run;
|
|
344
|
+
const invoke = (method, input) => {
|
|
345
|
+
if (!execution.accepting || execution.active) {
|
|
346
|
+
execution.coordinationError = /* @__PURE__ */ new Error(!execution.accepting ? "benchmark execution callback has settled" : "benchmark execution requires sequential start/resume calls");
|
|
347
|
+
throw execution.coordinationError;
|
|
348
|
+
}
|
|
349
|
+
runOptions.signal.throwIfAborted();
|
|
350
|
+
const index = prompts.length;
|
|
351
|
+
const eventStart = observedEvents.length;
|
|
352
|
+
execution.lastTurn = void 0;
|
|
353
|
+
const capture = async () => {
|
|
354
|
+
let turn;
|
|
355
|
+
let failure;
|
|
356
|
+
try {
|
|
357
|
+
turn = await managedRun[method](input);
|
|
358
|
+
return turn;
|
|
359
|
+
} catch (error) {
|
|
360
|
+
failure = error;
|
|
361
|
+
throw error;
|
|
362
|
+
} finally {
|
|
363
|
+
const events = observedEvents.slice(eventStart);
|
|
364
|
+
if (turn) execution.lastTurn = {
|
|
365
|
+
...turn,
|
|
366
|
+
events,
|
|
367
|
+
outcome: { ...turn.outcome }
|
|
368
|
+
};
|
|
369
|
+
let sessionId;
|
|
370
|
+
try {
|
|
371
|
+
sessionId = managedRun.sessionId;
|
|
372
|
+
} catch {}
|
|
373
|
+
const readError = turn?.readError ?? (failure instanceof SandboxRunAbortError ? failure.readError : void 0);
|
|
374
|
+
prompts.push({
|
|
375
|
+
attempt,
|
|
376
|
+
index,
|
|
377
|
+
method,
|
|
378
|
+
prompt: input,
|
|
379
|
+
...sessionId === void 0 ? {} : { sessionId },
|
|
380
|
+
...turn === void 0 ? {} : { outcome: { ...turn.outcome } },
|
|
381
|
+
events,
|
|
382
|
+
usage: turn === void 0 ? {
|
|
383
|
+
...sumSandboxUsage(events),
|
|
384
|
+
tokensKnown: false,
|
|
385
|
+
usdKnown: false
|
|
386
|
+
} : sumSandboxUsage(events),
|
|
387
|
+
...readError === void 0 ? {} : { readError },
|
|
388
|
+
...turn !== void 0 ? {} : { error: failure instanceof Error ? failure.message : String(failure) }
|
|
389
|
+
});
|
|
390
|
+
}
|
|
391
|
+
};
|
|
392
|
+
const pending = capture();
|
|
393
|
+
execution.active = pending;
|
|
394
|
+
pending.then(() => {
|
|
395
|
+
execution.active = void 0;
|
|
396
|
+
}, () => {
|
|
397
|
+
execution.active = void 0;
|
|
398
|
+
});
|
|
399
|
+
return pending;
|
|
400
|
+
};
|
|
401
|
+
const context = {
|
|
402
|
+
prompt: prompt ?? task.prompt,
|
|
403
|
+
profile: structuredClone(profile),
|
|
404
|
+
benchmark: adapter.name,
|
|
405
|
+
taskId: task.id,
|
|
406
|
+
attempt,
|
|
407
|
+
signal: runOptions.signal,
|
|
408
|
+
run: {
|
|
409
|
+
start: (input) => invoke("start", input),
|
|
410
|
+
resume: (input) => invoke("resume", input),
|
|
411
|
+
get box() {
|
|
412
|
+
return managedRun.box;
|
|
413
|
+
},
|
|
414
|
+
get sessionId() {
|
|
415
|
+
return managedRun.sessionId;
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
};
|
|
419
|
+
try {
|
|
420
|
+
await executeWithSignal(async () => {
|
|
421
|
+
if (execute) await execute(context);
|
|
422
|
+
else await context.run.start(context.prompt);
|
|
423
|
+
}, runOptions.signal);
|
|
424
|
+
} catch (error) {
|
|
425
|
+
if (execution.active) controller.abort();
|
|
426
|
+
throw error;
|
|
427
|
+
} finally {
|
|
428
|
+
execution.accepting = false;
|
|
429
|
+
await execution.active?.catch(() => void 0);
|
|
430
|
+
}
|
|
431
|
+
if (execution.coordinationError) throw execution.coordinationError;
|
|
432
|
+
runOptions.signal.throwIfAborted();
|
|
433
|
+
const turn = execution.lastTurn;
|
|
434
|
+
if (!turn) throw new Error(prompts.at(-1)?.error ?? "benchmark execution returned without a completed prompt");
|
|
331
435
|
result = {
|
|
332
436
|
artifact: "",
|
|
333
437
|
ok: false,
|
|
334
|
-
usage:
|
|
335
|
-
events:
|
|
438
|
+
usage: promptUsage(prompts),
|
|
439
|
+
events: observedEvents,
|
|
440
|
+
prompts,
|
|
441
|
+
execution: {
|
|
442
|
+
phase: "started",
|
|
443
|
+
terminalOutcome: turn.outcome.success ? "succeeded" : turn.outcome.status === "failed" ? "failed" : "incomplete"
|
|
444
|
+
}
|
|
336
445
|
};
|
|
337
446
|
let artifact = (turn.out ?? "").trim();
|
|
338
447
|
let boxExtractError;
|
|
@@ -379,19 +488,27 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
379
488
|
result = {
|
|
380
489
|
artifact,
|
|
381
490
|
ok: turn.outcome.success && artifact.length > 0 && turn.readError === void 0 && boxExtractError === void 0,
|
|
491
|
+
execution: result.execution,
|
|
492
|
+
artifactAvailable: turn.readError === void 0 && boxExtractError === void 0,
|
|
382
493
|
usage: result.usage,
|
|
383
|
-
events:
|
|
494
|
+
events: observedEvents,
|
|
495
|
+
prompts,
|
|
384
496
|
...detail ? { detail } : {}
|
|
385
497
|
};
|
|
386
498
|
} catch (err) {
|
|
499
|
+
const events = observedEvents;
|
|
387
500
|
result = {
|
|
388
501
|
...result,
|
|
389
502
|
ok: false,
|
|
503
|
+
artifactAvailable: false,
|
|
504
|
+
execution: {
|
|
505
|
+
phase: events.length > 0 ? "started" : "unknown",
|
|
506
|
+
terminalOutcome: result.execution?.terminalOutcome ?? "unknown"
|
|
507
|
+
},
|
|
390
508
|
detail: err instanceof Error ? err.message : String(err),
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
} : {}
|
|
509
|
+
usage: promptUsage(prompts),
|
|
510
|
+
events,
|
|
511
|
+
prompts
|
|
395
512
|
};
|
|
396
513
|
} finally {
|
|
397
514
|
if (timer) clearTimeout(timer);
|
|
@@ -407,6 +524,27 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
407
524
|
}
|
|
408
525
|
return result;
|
|
409
526
|
};
|
|
527
|
+
/** Stop waiting on policy work when cancelled; the caller must also cancel its external effects. */
|
|
528
|
+
async function executeWithSignal(execute, signal) {
|
|
529
|
+
signal.throwIfAborted();
|
|
530
|
+
let onAbort;
|
|
531
|
+
const aborted = new Promise((_resolve, reject) => {
|
|
532
|
+
onAbort = () => reject(signal.reason ?? /* @__PURE__ */ new Error("aborted"));
|
|
533
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
534
|
+
});
|
|
535
|
+
try {
|
|
536
|
+
await Promise.race([Promise.resolve().then(execute), aborted]);
|
|
537
|
+
} finally {
|
|
538
|
+
if (onAbort) signal.removeEventListener("abort", onAbort);
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
function promptUsage(prompts) {
|
|
542
|
+
return combinedUsage(prompts.map((prompt) => ({
|
|
543
|
+
artifact: "",
|
|
544
|
+
ok: false,
|
|
545
|
+
usage: prompt.usage
|
|
546
|
+
})));
|
|
547
|
+
}
|
|
410
548
|
function parseMaybeJson(value) {
|
|
411
549
|
try {
|
|
412
550
|
return JSON.parse(value);
|
|
@@ -496,17 +634,25 @@ async function loopedShot(input, shot, attempts) {
|
|
|
496
634
|
return {
|
|
497
635
|
artifact: completed.at(-1)?.artifact ?? "",
|
|
498
636
|
ok: false,
|
|
637
|
+
execution: pendingShot ? {
|
|
638
|
+
phase: "unknown",
|
|
639
|
+
terminalOutcome: "unknown"
|
|
640
|
+
} : completed.at(-1)?.execution,
|
|
641
|
+
artifactAvailable: false,
|
|
499
642
|
usage: combinedUsage(pendingShot ? [...completed, {
|
|
500
643
|
artifact: "",
|
|
501
644
|
ok: false
|
|
502
645
|
}] : completed),
|
|
503
646
|
events: completed.flatMap((shot) => shot.events ?? []),
|
|
647
|
+
prompts: completed.flatMap((shot) => shot.prompts ?? []),
|
|
504
648
|
detail: err instanceof Error ? err.message : String(err)
|
|
505
649
|
};
|
|
506
650
|
}
|
|
507
651
|
const best = result.rounds.reduce((winner, candidate) => {
|
|
508
|
-
|
|
509
|
-
|
|
652
|
+
const candidateShot = shots.get(candidate.round);
|
|
653
|
+
const winnerShot = shots.get(winner.round);
|
|
654
|
+
const rank = (shot) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0;
|
|
655
|
+
if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner;
|
|
510
656
|
const a = scores.get(winner.round);
|
|
511
657
|
const b = scores.get(candidate.round);
|
|
512
658
|
if (!a) return candidate;
|
|
@@ -519,8 +665,11 @@ async function loopedShot(input, shot, attempts) {
|
|
|
519
665
|
return {
|
|
520
666
|
artifact: best.artifact,
|
|
521
667
|
ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
|
|
668
|
+
execution: shots.get(best.round)?.execution,
|
|
669
|
+
artifactAvailable: shots.get(best.round)?.artifactAvailable,
|
|
522
670
|
usage: combinedUsage([...shots.values()]),
|
|
523
671
|
events: [...shots.values()].flatMap((shot) => shot.events ?? []),
|
|
672
|
+
prompts: [...shots.values()].flatMap((shot) => shot.prompts ?? []),
|
|
524
673
|
detail: JSON.stringify({
|
|
525
674
|
mode: "refine-loop",
|
|
526
675
|
attempts: result.rounds.length,
|
|
@@ -641,6 +790,7 @@ async function runBenchmarks(opts) {
|
|
|
641
790
|
const startedAt = Date.now();
|
|
642
791
|
let result;
|
|
643
792
|
let out;
|
|
793
|
+
let invoked = false;
|
|
644
794
|
try {
|
|
645
795
|
opts.signal?.throwIfAborted();
|
|
646
796
|
const shotInput = {
|
|
@@ -655,8 +805,10 @@ async function runBenchmarks(opts) {
|
|
|
655
805
|
...opts.sandboxBaseUrl ? { sandboxBaseUrl: opts.sandboxBaseUrl } : {},
|
|
656
806
|
...opts.timeoutMs ? { timeoutMs: opts.timeoutMs } : {},
|
|
657
807
|
...opts.signal ? { signal: opts.signal } : {},
|
|
658
|
-
...opts.resolveClient ? { resolveClient: opts.resolveClient } : {}
|
|
808
|
+
...opts.resolveClient ? { resolveClient: opts.resolveClient } : {},
|
|
809
|
+
...opts.execute ? { execute: opts.execute } : {}
|
|
659
810
|
};
|
|
811
|
+
invoked = true;
|
|
660
812
|
out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput);
|
|
661
813
|
const score = await job.adapter.judge(job.task, out.artifact);
|
|
662
814
|
result = {
|
|
@@ -667,11 +819,17 @@ async function runBenchmarks(opts) {
|
|
|
667
819
|
resolved: out.ok && score.resolved,
|
|
668
820
|
score: out.ok ? score.score : 0,
|
|
669
821
|
ok: out.ok,
|
|
822
|
+
execution: out.execution ?? {
|
|
823
|
+
phase: out.ok ? "started" : "unknown",
|
|
824
|
+
terminalOutcome: out.ok ? "succeeded" : "unknown"
|
|
825
|
+
},
|
|
826
|
+
measurement: out.artifactAvailable ?? out.ok ? "available" : "unavailable",
|
|
670
827
|
...out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {},
|
|
671
828
|
wallMs: Date.now() - startedAt,
|
|
672
829
|
artifact: out.artifact,
|
|
673
830
|
...out.usage === void 0 ? {} : { usage: out.usage },
|
|
674
|
-
...out.events === void 0 ? {} : { events: out.events }
|
|
831
|
+
...out.events === void 0 ? {} : { events: out.events },
|
|
832
|
+
...out.prompts === void 0 ? {} : { prompts: out.prompts }
|
|
675
833
|
};
|
|
676
834
|
} catch (err) {
|
|
677
835
|
result = {
|
|
@@ -682,11 +840,17 @@ async function runBenchmarks(opts) {
|
|
|
682
840
|
resolved: false,
|
|
683
841
|
score: 0,
|
|
684
842
|
ok: false,
|
|
843
|
+
execution: out?.execution ?? {
|
|
844
|
+
phase: !invoked ? "not-started" : out?.ok ? "started" : "unknown",
|
|
845
|
+
terminalOutcome: out?.ok ? "succeeded" : "unknown"
|
|
846
|
+
},
|
|
847
|
+
measurement: "unavailable",
|
|
685
848
|
detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
|
|
686
849
|
wallMs: Date.now() - startedAt,
|
|
687
850
|
...out === void 0 ? {} : { artifact: out.artifact },
|
|
688
851
|
...out?.usage === void 0 ? {} : { usage: out.usage },
|
|
689
|
-
...out?.events === void 0 ? {} : { events: out.events }
|
|
852
|
+
...out?.events === void 0 ? {} : { events: out.events },
|
|
853
|
+
...out?.prompts === void 0 ? {} : { prompts: out.prompts }
|
|
690
854
|
};
|
|
691
855
|
}
|
|
692
856
|
perTask.push(result);
|
|
@@ -714,7 +878,7 @@ function aggregate(perTask) {
|
|
|
714
878
|
scoreSum: 0
|
|
715
879
|
};
|
|
716
880
|
e.n += 1;
|
|
717
|
-
if (
|
|
881
|
+
if ((r.measurement ?? (r.ok ? "available" : "unavailable")) === "unavailable") e.errored += 1;
|
|
718
882
|
else {
|
|
719
883
|
if (r.resolved) e.resolved += 1;
|
|
720
884
|
e.scoreSum += r.score;
|