@tangle-network/agent-bench 0.10.0 → 0.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/HARNESS.md +43 -5
- package/README.md +22 -0
- package/dist/index.d.ts +33 -2
- package/dist/index.js +133 -15
- package/dist/index.js.map +1 -1
- package/package.json +5 -5
- package/scripts/verify-packed-consumer.mjs +47 -1
- package/src/index.ts +3 -0
- package/src/run-benchmarks.test.mts +178 -1
- package/src/run-benchmarks.ts +147 -7
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.11.1
|
|
4
|
+
|
|
5
|
+
The published benchmark package now admits sandbox SDK 0.38.x, including consumer-ready runtime-edge readiness.
|
|
6
|
+
|
|
7
|
+
## 0.11.0
|
|
8
|
+
|
|
9
|
+
Adds caller-controlled start and resume inside managed benchmark shots.
|
|
10
|
+
Retains setup, final extraction, grading, and cleanup while recording every prompt and its observed usage.
|
|
11
|
+
The callback composes with outer retry attempts; learning policy and policy inference accounting remain consumer responsibilities.
|
|
12
|
+
|
|
3
13
|
## 0.9.4
|
|
4
14
|
|
|
5
15
|
Requires Runtime 0.202.0 so Bench consumers receive explicit failed-execution settlement and retained failure recovery.
|
package/HARNESS.md
CHANGED
|
@@ -8,9 +8,11 @@
|
|
|
8
8
|
|---|---|
|
|
9
9
|
| **agent-runtime** | exact execution, reusable benchmark adapters, packed-consumer checks, one full-fidelity integration fixture |
|
|
10
10
|
| **discovery** | research questions, preregistrations, acceptance criteria, negative results, and decisions about what is worth testing |
|
|
11
|
-
| **
|
|
11
|
+
| **supervisor-lab** | registered adaptive-agent and retained-learning comparisons, including carried versus revised profiles |
|
|
12
|
+
| **discovery-lab** | research campaigns and evidence for Discovery, including immutable inputs, run records, and result archives |
|
|
12
13
|
|
|
13
|
-
A benchmark implementation may begin here while it is becoming a reusable adapter. Once the question is “does method X improve benchmark Y?”, the campaign belongs in
|
|
14
|
+
A benchmark implementation may begin here while it is becoming a reusable adapter. Once the question is “does method X improve benchmark Y?”, the campaign belongs in the consuming lab.
|
|
15
|
+
Supervisor Lab owns adaptive-learning comparisons; Discovery Lab owns research campaigns for Discovery.
|
|
14
16
|
|
|
15
17
|
## Evidence levels
|
|
16
18
|
|
|
@@ -20,8 +22,8 @@ Use these labels literally. Do not promote one level into another in prose.
|
|
|
20
22
|
|---|---|---|
|
|
21
23
|
| **contract proof** | packages install; identities, budgets, callbacks, resume, and receipts have the expected shape | root `pnpm verify:official-optimizers`, `pnpm verify:bench` |
|
|
22
24
|
| **evaluator proof** | the benchmark's own evaluator can distinguish known fail/pass artifacts in the exact environment | adapter preflight and gold/self-check |
|
|
23
|
-
| **reproduction proof** | an upstream method is run at a pinned revision on its claimed benchmark under a matched protocol |
|
|
24
|
-
| **value proof** | the integrated method beats the preregistered baseline on frozen evidence with uncertainty and complete cost accounting |
|
|
25
|
+
| **reproduction proof** | an upstream method is run at a pinned revision on its claimed benchmark under a matched protocol | Consuming lab reproduction manifest and runner |
|
|
26
|
+
| **value proof** | the integrated method beats the preregistered baseline on frozen evidence with uncertainty and complete cost accounting | Consuming lab result receipt |
|
|
25
27
|
| **production proof** | a promoted artifact transfers to real traffic under a canary or controlled rollout | product repository / platform telemetry |
|
|
26
28
|
|
|
27
29
|
A localization score, output-shape check, LLM quality judge, or toy deterministic reward can be useful for development. None is a substitute for the benchmark's outcome evaluator.
|
|
@@ -71,6 +73,42 @@ The current Runtime lineage suppresses sandbox deletion errors, so a returned re
|
|
|
71
73
|
The caller's abort signal stops queued shots and reaches active sandbox turns.
|
|
72
74
|
`modelApiKey` supplies sandbox inference authorization separately from the `routerKey` used for sandbox control.
|
|
73
75
|
|
|
76
|
+
### Caller-controlled prompts within a task
|
|
77
|
+
|
|
78
|
+
`runBenchmarks({ execute })` invokes the callback inside each managed sandbox shot.
|
|
79
|
+
The context supplies the task prompt, executed profile, benchmark and task identities, attempt number, and abort signal.
|
|
80
|
+
Its managed `run` exposes Runtime's `start`, `resume`, `box`, and `sessionId`.
|
|
81
|
+
Use the live box for permitted working checks in the same session.
|
|
82
|
+
Submit worker prompts through managed `start` and `resume` so Bench captures their outcomes and usage.
|
|
83
|
+
Direct sandbox prompt calls bypass this accounting.
|
|
84
|
+
The callback must leave session lifecycle, extraction, and cleanup to Bench.
|
|
85
|
+
It receives no adapter, final grader, or task metadata containing gold material.
|
|
86
|
+
This callback is trusted consumer code, not an isolation boundary for arbitrary code.
|
|
87
|
+
|
|
88
|
+
Return after the final prompt completes.
|
|
89
|
+
Bench extracts the last captured prompt's artifact and applies the adapter's final grading outside the callback.
|
|
90
|
+
Start and resume calls must be sequential.
|
|
91
|
+
Bench waits for an unawaited active invocation before extraction and refuses calls after the callback settles.
|
|
92
|
+
The callback must use its signal to cancel external work.
|
|
93
|
+
Bench stops awaiting policy work when cancelled, but cannot stop external effects that ignore cancellation.
|
|
94
|
+
|
|
95
|
+
Each task's `prompts` retains ordered method, prompt, attempt, session identity, outcome, events, usage, and capture errors.
|
|
96
|
+
Prompt indices start at zero; attempt numbers start at one.
|
|
97
|
+
Usage sums each prompt separately, including failed and interrupted prompts.
|
|
98
|
+
A partial capture retains observed counters and marks accounting incomplete.
|
|
99
|
+
These counters cover sandbox workers only; consumers must account for policy and working-evaluator inference separately.
|
|
100
|
+
Consumers must bind their callback source, configuration, profiles, and initial state to their execution identity.
|
|
101
|
+
|
|
102
|
+
`execute` runs inside every `loopAttempts` shot.
|
|
103
|
+
Those outer attempts still create fresh sandboxes and use the existing checker feedback policy.
|
|
104
|
+
Use one outer attempt when final grading must remain unavailable to adaptation.
|
|
105
|
+
A custom `runShot` receives `execute` and owns whether it consumes the callback.
|
|
106
|
+
|
|
107
|
+
A sandbox prompt can contain multiple native model requests.
|
|
108
|
+
Same-session continuation does not prove a barrier before every native request, profile reload, coordinator restart, or fresh-session state transfer.
|
|
109
|
+
Offline fake-sandbox tests prove the managed contract and correction consumption only.
|
|
110
|
+
They establish no live learning gain or provider session restoration.
|
|
111
|
+
|
|
74
112
|
### Retained strategy driver
|
|
75
113
|
|
|
76
114
|
```bash
|
|
@@ -133,4 +171,4 @@ A new file under `bench/src` must be one of:
|
|
|
133
171
|
- a package-consumer or evaluator calibration test;
|
|
134
172
|
- one canonical full-fidelity fixture that exercises a public Runtime contract.
|
|
135
173
|
|
|
136
|
-
A one-off campaign, generation-N optimizer script, bespoke dashboard, or historical result belongs in
|
|
174
|
+
A one-off campaign, generation-N optimizer script, bespoke dashboard, or historical result belongs in the consuming lab. If an older file has no package script, no importer, and no unique reusable primitive, delete it rather than adding another index entry.
|
package/README.md
CHANGED
|
@@ -100,3 +100,25 @@ Pier owns the task container and verifier; protected model usage and traces stay
|
|
|
100
100
|
`FilePierCandidateTrialController` atomically reserves a unique Pier job, then persists the supervisor PID, process-session identity, and that job's exact Docker projects so a fresh evaluator process can stop and remove an abandoned trial.
|
|
101
101
|
Run `PIER_REPO=/path/to/pier pnpm verify:pier` for the zero-model failure/pass and fresh-process recovery proof, and see `HARNESS.md` for the exact invocation and failure contract.
|
|
102
102
|
From an installed npm package, expose the shipped Python module with `export PYTHONPATH="$(npm root)/@tangle-network/agent-bench${PYTHONPATH:+:$PYTHONPATH}"` before invoking Pier.
|
|
103
|
+
|
|
104
|
+
## Control execution within a managed shot
|
|
105
|
+
|
|
106
|
+
Supply `execute` to control whole sandbox prompts while Bench owns setup, extraction, grading, and cleanup.
|
|
107
|
+
Your policy can inspect permitted working checks through the existing sandbox handle.
|
|
108
|
+
|
|
109
|
+
```ts
|
|
110
|
+
import { runBenchmarks, type BenchExecution } from '@tangle-network/agent-bench'
|
|
111
|
+
import { executionOptions, chooseNextPrompt } from './policy.js'
|
|
112
|
+
|
|
113
|
+
const execute: BenchExecution = async ({ run, prompt, signal }) => {
|
|
114
|
+
const first = await run.start(prompt)
|
|
115
|
+
const correction = await chooseNextPrompt({ first, box: run.box, sessionId: run.sessionId, signal })
|
|
116
|
+
if (correction !== undefined) await run.resume(correction)
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const report = await runBenchmarks({ ...executionOptions, execute })
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The consumer supplies and identifies `chooseNextPrompt`; Bench provides no learning policy.
|
|
123
|
+
`perTask[].prompts` retains each invocation and its observed worker cost, including partial failures.
|
|
124
|
+
See [the execution contract](./HARNESS.md#caller-controlled-prompts-within-a-task) for accounting, retry composition, and lifecycle limits.
|
package/dist/index.d.ts
CHANGED
|
@@ -8,7 +8,7 @@ import { FINAL_ANSWER_SENTINEL, RagAnswerScore, RagContext, answerScoreToBenchSc
|
|
|
8
8
|
import { createRagBenchAdapter } from "./benchmarks/ragbench.js";
|
|
9
9
|
import { SweBenchAdapterOptions, SweBenchArtifactCaptureContext, SweBenchCacheLevel, createSweBenchAdapter, scoreSweReport, sweEvaluationArgv, swePatchOutput } from "./benchmarks/swe-bench.js";
|
|
10
10
|
import { createT2RagBenchAdapter } from "./benchmarks/t2-ragbench.js";
|
|
11
|
-
import { AgentProfile, SandboxClient, sumSandboxUsage } from "@tangle-network/agent-runtime/kernel";
|
|
11
|
+
import { AgentProfile, SandboxClient, SandboxRun, TurnResult, sumSandboxUsage } from "@tangle-network/agent-runtime/kernel";
|
|
12
12
|
import { SandboxEvent } from "@tangle-network/sandbox";
|
|
13
13
|
import { AgentCandidateBenchmarkGraderPort, AgentCandidateExecutionClaimStore, AgentCandidateExecutorPort, AgentCandidateExecutorRequest, AgentCandidateExecutorStopRequest, AgentCandidateOutputArtifactPort, AgentCandidateRunFinalization, PreparedAgentCandidateExecution } from "@tangle-network/agent-runtime/candidate-execution";
|
|
14
14
|
import { RunTerminalOutcome, TraceStore } from "@tangle-network/agent-eval";
|
|
@@ -48,6 +48,31 @@ interface BenchCell {
|
|
|
48
48
|
/** The agent under test. Defaults to a minimal `{ name, metadata.backendType }` profile. */
|
|
49
49
|
readonly profile?: AgentProfile;
|
|
50
50
|
}
|
|
51
|
+
/** Caller-owned work inside a managed benchmark shot. Final grading stays outside this callback. */
|
|
52
|
+
interface BenchExecutionContext {
|
|
53
|
+
readonly prompt: string;
|
|
54
|
+
readonly profile: AgentProfile;
|
|
55
|
+
readonly benchmark: string;
|
|
56
|
+
readonly taskId: string;
|
|
57
|
+
readonly attempt: number;
|
|
58
|
+
readonly signal: AbortSignal;
|
|
59
|
+
/** Runtime owns session continuity. Bench owns capture, extraction, and close. */
|
|
60
|
+
readonly run: Pick<SandboxRun<string>, 'start' | 'resume' | 'box' | 'sessionId'>;
|
|
61
|
+
}
|
|
62
|
+
type BenchExecution = (context: BenchExecutionContext) => Promise<void>;
|
|
63
|
+
/** One submitted sandbox prompt, including partial evidence when its invocation throws. */
|
|
64
|
+
interface BenchPromptResult {
|
|
65
|
+
readonly attempt: number;
|
|
66
|
+
readonly index: number;
|
|
67
|
+
readonly method: 'start' | 'resume';
|
|
68
|
+
readonly prompt: string;
|
|
69
|
+
readonly sessionId?: string;
|
|
70
|
+
readonly outcome?: TurnResult<string>['outcome'];
|
|
71
|
+
readonly events: readonly SandboxEvent[];
|
|
72
|
+
readonly usage: ReturnType<typeof sumSandboxUsage>;
|
|
73
|
+
readonly readError?: string;
|
|
74
|
+
readonly error?: string;
|
|
75
|
+
}
|
|
51
76
|
/** A worker's artifact and observed execution evidence, before external grading. */
|
|
52
77
|
interface BenchShotResult {
|
|
53
78
|
readonly artifact: string;
|
|
@@ -56,6 +81,7 @@ interface BenchShotResult {
|
|
|
56
81
|
/** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
|
|
57
82
|
readonly usage?: ReturnType<typeof sumSandboxUsage>;
|
|
58
83
|
readonly events?: readonly SandboxEvent[];
|
|
84
|
+
readonly prompts?: readonly BenchPromptResult[];
|
|
59
85
|
/** Observed dispatch and terminal state, independent of artifact quality. */
|
|
60
86
|
readonly execution?: {
|
|
61
87
|
readonly phase: 'not-started' | 'started' | 'unknown';
|
|
@@ -73,6 +99,8 @@ type BenchShot = (input: {
|
|
|
73
99
|
readonly prompt?: string;
|
|
74
100
|
/** 1-based attempt index for looped runs. */
|
|
75
101
|
readonly attempt?: number;
|
|
102
|
+
/** Custom shots own whether they consume this managed execution callback. */
|
|
103
|
+
readonly execute?: BenchExecution;
|
|
76
104
|
readonly routerBaseUrl: string;
|
|
77
105
|
readonly routerKey: string;
|
|
78
106
|
/** Optional inference credential for the box; routerKey continues to authorize sandbox control. */
|
|
@@ -116,6 +144,8 @@ interface RunBenchmarksOptions {
|
|
|
116
144
|
/** Self-verify each benchmark's judge against its gold artifact on the first task before spending
|
|
117
145
|
* model tokens; a benchmark whose judge rejects its own gold is recorded unavailable. Default true. */
|
|
118
146
|
readonly verifyJudge?: boolean;
|
|
147
|
+
/** Caller policy inside each managed shot, including every refine-loop attempt. */
|
|
148
|
+
readonly execute?: BenchExecution;
|
|
119
149
|
/** Test seam: a deterministic shot runner. Defaults to the `openSandboxRun` leaf. */
|
|
120
150
|
readonly runShot?: BenchShot;
|
|
121
151
|
/** Test seam: resolve a benchmark key to an adapter. Defaults to the registry `resolveAdapter`. */
|
|
@@ -141,6 +171,7 @@ interface BenchCellTaskResult {
|
|
|
141
171
|
readonly usage?: ReturnType<typeof sumSandboxUsage>;
|
|
142
172
|
/** Worker events only; benchmark grading remains outside this trace. */
|
|
143
173
|
readonly events?: readonly SandboxEvent[];
|
|
174
|
+
readonly prompts?: readonly BenchPromptResult[];
|
|
144
175
|
}
|
|
145
176
|
interface BenchLeaderboardRow {
|
|
146
177
|
readonly benchmark: string;
|
|
@@ -328,5 +359,5 @@ declare class FilePierCandidateTrialController implements PierCandidateTrialCont
|
|
|
328
359
|
//#region src/pier-result-grader.d.ts
|
|
329
360
|
declare function createPierResultGrader(descriptor: Pick<PierCandidateGraderPort, 'name' | 'version' | 'artifact'>): PierCandidateGraderPort;
|
|
330
361
|
//#endregion
|
|
331
|
-
export { ADAPTERS, type BenchCell, type BenchCellTaskResult, type BenchLeaderboardRow, type BenchScore, type BenchShot, type BenchShotResult, type BenchTask, type BenchmarkAdapter, type ExecutePreparedPierCandidateOptions, FINAL_ANSWER_SENTINEL, FilePierCandidateTrialController, type FilePierCandidateTrialControllerOptions, type JudgeArtifactFileReceipt, type JudgeArtifactReceipt, type LoadOptions, type PierCandidateGraderPort, type PierCandidateOfficialResult, type PierCandidateProcessSpec, type PierCandidateTerminationAcknowledgement, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierCandidateTrialIdentity, type PierCandidateTrialResult, type PierDockerConnection, type RagAnswerScore, type RagContext, type RunBenchmarksOptions, type RunBenchmarksReport, StagedJudgeError, type StagedPierCandidateExecution, type StagedRunCaptureSpec, type StagedRunSpec, type SweBenchAdapterOptions, type SweBenchArtifactCaptureContext, type SweBenchCacheLevel, answerScoreToBenchScore, contextBlock, contextsFrom, createCragAdapter, createNoMiraclAdapter, createOpenRagBenchAdapter, createPierCandidateRecoveryExecutor, createPierResultGrader, createRagBenchAdapter, createSweBenchAdapter, createT2RagBenchAdapter, executePreparedPierCandidate, normalizeAnswer, parseCitations, parseFinalAnswer, printBenchmarksReport, ragAnswerOutput, resolveAdapter, runBenchmarks, runStagedJudge, scoreAnswerArtifact, scoreSweReport, sweEvaluationArgv, swePatchOutput, tokenF1 };
|
|
362
|
+
export { ADAPTERS, type BenchCell, type BenchCellTaskResult, type BenchExecution, type BenchExecutionContext, type BenchLeaderboardRow, type BenchPromptResult, type BenchScore, type BenchShot, type BenchShotResult, type BenchTask, type BenchmarkAdapter, type ExecutePreparedPierCandidateOptions, FINAL_ANSWER_SENTINEL, FilePierCandidateTrialController, type FilePierCandidateTrialControllerOptions, type JudgeArtifactFileReceipt, type JudgeArtifactReceipt, type LoadOptions, type PierCandidateGraderPort, type PierCandidateOfficialResult, type PierCandidateProcessSpec, type PierCandidateTerminationAcknowledgement, type PierCandidateTrialController, type PierCandidateTrialHandle, type PierCandidateTrialIdentity, type PierCandidateTrialResult, type PierDockerConnection, type RagAnswerScore, type RagContext, type RunBenchmarksOptions, type RunBenchmarksReport, StagedJudgeError, type StagedPierCandidateExecution, type StagedRunCaptureSpec, type StagedRunSpec, type SweBenchAdapterOptions, type SweBenchArtifactCaptureContext, type SweBenchCacheLevel, answerScoreToBenchScore, contextBlock, contextsFrom, createCragAdapter, createNoMiraclAdapter, createOpenRagBenchAdapter, createPierCandidateRecoveryExecutor, createPierResultGrader, createRagBenchAdapter, createSweBenchAdapter, createT2RagBenchAdapter, executePreparedPierCandidate, normalizeAnswer, parseCitations, parseFinalAnswer, printBenchmarksReport, ragAnswerOutput, resolveAdapter, runBenchmarks, runStagedJudge, scoreAnswerArtifact, scoreSweReport, sweEvaluationArgv, swePatchOutput, tokenF1 };
|
|
332
363
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.js
CHANGED
|
@@ -252,7 +252,7 @@ function finalText(events) {
|
|
|
252
252
|
}
|
|
253
253
|
/** The default real-agent shot: one `openSandboxRun` over the cell's harness+model, deliverable
|
|
254
254
|
* extracted by the adapter's parser (or final text), abortable on `timeoutMs`. */
|
|
255
|
-
const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
255
|
+
const openSandboxShot = async ({ adapter, task, cell, prompt, attempt = 1, execute, routerBaseUrl, routerKey, modelApiKey, bridgeUrl, bridgeBearer, sandboxBaseUrl, timeoutMs, signal, resolveClient }) => {
|
|
256
256
|
signal?.throwIfAborted();
|
|
257
257
|
const client = (resolveClient ?? resolveBenchClient)({
|
|
258
258
|
backend: cell.backend ?? "router",
|
|
@@ -329,6 +329,8 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
329
329
|
artifact: "",
|
|
330
330
|
ok: false
|
|
331
331
|
};
|
|
332
|
+
const prompts = [];
|
|
333
|
+
const execution = { accepting: true };
|
|
332
334
|
try {
|
|
333
335
|
run = await openSandboxRun(client, runOptions, deliverable);
|
|
334
336
|
result = {
|
|
@@ -338,12 +340,104 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
338
340
|
terminalOutcome: "unknown"
|
|
339
341
|
}
|
|
340
342
|
};
|
|
341
|
-
const
|
|
343
|
+
const managedRun = run;
|
|
344
|
+
const invoke = (method, input) => {
|
|
345
|
+
if (!execution.accepting || execution.active) {
|
|
346
|
+
execution.coordinationError = /* @__PURE__ */ new Error(!execution.accepting ? "benchmark execution callback has settled" : "benchmark execution requires sequential start/resume calls");
|
|
347
|
+
throw execution.coordinationError;
|
|
348
|
+
}
|
|
349
|
+
runOptions.signal.throwIfAborted();
|
|
350
|
+
const index = prompts.length;
|
|
351
|
+
const eventStart = observedEvents.length;
|
|
352
|
+
execution.lastTurn = void 0;
|
|
353
|
+
const capture = async () => {
|
|
354
|
+
let turn;
|
|
355
|
+
let failure;
|
|
356
|
+
try {
|
|
357
|
+
turn = await managedRun[method](input);
|
|
358
|
+
return turn;
|
|
359
|
+
} catch (error) {
|
|
360
|
+
failure = error;
|
|
361
|
+
throw error;
|
|
362
|
+
} finally {
|
|
363
|
+
const events = observedEvents.slice(eventStart);
|
|
364
|
+
if (turn) execution.lastTurn = {
|
|
365
|
+
...turn,
|
|
366
|
+
events,
|
|
367
|
+
outcome: { ...turn.outcome }
|
|
368
|
+
};
|
|
369
|
+
let sessionId;
|
|
370
|
+
try {
|
|
371
|
+
sessionId = managedRun.sessionId;
|
|
372
|
+
} catch {}
|
|
373
|
+
const readError = turn?.readError ?? (failure instanceof SandboxRunAbortError ? failure.readError : void 0);
|
|
374
|
+
prompts.push({
|
|
375
|
+
attempt,
|
|
376
|
+
index,
|
|
377
|
+
method,
|
|
378
|
+
prompt: input,
|
|
379
|
+
...sessionId === void 0 ? {} : { sessionId },
|
|
380
|
+
...turn === void 0 ? {} : { outcome: { ...turn.outcome } },
|
|
381
|
+
events,
|
|
382
|
+
usage: turn === void 0 ? {
|
|
383
|
+
...sumSandboxUsage(events),
|
|
384
|
+
tokensKnown: false,
|
|
385
|
+
usdKnown: false
|
|
386
|
+
} : sumSandboxUsage(events),
|
|
387
|
+
...readError === void 0 ? {} : { readError },
|
|
388
|
+
...turn !== void 0 ? {} : { error: failure instanceof Error ? failure.message : String(failure) }
|
|
389
|
+
});
|
|
390
|
+
}
|
|
391
|
+
};
|
|
392
|
+
const pending = capture();
|
|
393
|
+
execution.active = pending;
|
|
394
|
+
pending.then(() => {
|
|
395
|
+
execution.active = void 0;
|
|
396
|
+
}, () => {
|
|
397
|
+
execution.active = void 0;
|
|
398
|
+
});
|
|
399
|
+
return pending;
|
|
400
|
+
};
|
|
401
|
+
const context = {
|
|
402
|
+
prompt: prompt ?? task.prompt,
|
|
403
|
+
profile: structuredClone(profile),
|
|
404
|
+
benchmark: adapter.name,
|
|
405
|
+
taskId: task.id,
|
|
406
|
+
attempt,
|
|
407
|
+
signal: runOptions.signal,
|
|
408
|
+
run: {
|
|
409
|
+
start: (input) => invoke("start", input),
|
|
410
|
+
resume: (input) => invoke("resume", input),
|
|
411
|
+
get box() {
|
|
412
|
+
return managedRun.box;
|
|
413
|
+
},
|
|
414
|
+
get sessionId() {
|
|
415
|
+
return managedRun.sessionId;
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
};
|
|
419
|
+
try {
|
|
420
|
+
await executeWithSignal(async () => {
|
|
421
|
+
if (execute) await execute(context);
|
|
422
|
+
else await context.run.start(context.prompt);
|
|
423
|
+
}, runOptions.signal);
|
|
424
|
+
} catch (error) {
|
|
425
|
+
if (execution.active) controller.abort();
|
|
426
|
+
throw error;
|
|
427
|
+
} finally {
|
|
428
|
+
execution.accepting = false;
|
|
429
|
+
await execution.active?.catch(() => void 0);
|
|
430
|
+
}
|
|
431
|
+
if (execution.coordinationError) throw execution.coordinationError;
|
|
432
|
+
runOptions.signal.throwIfAborted();
|
|
433
|
+
const turn = execution.lastTurn;
|
|
434
|
+
if (!turn) throw new Error(prompts.at(-1)?.error ?? "benchmark execution returned without a completed prompt");
|
|
342
435
|
result = {
|
|
343
436
|
artifact: "",
|
|
344
437
|
ok: false,
|
|
345
|
-
usage:
|
|
346
|
-
events:
|
|
438
|
+
usage: promptUsage(prompts),
|
|
439
|
+
events: observedEvents,
|
|
440
|
+
prompts,
|
|
347
441
|
execution: {
|
|
348
442
|
phase: "started",
|
|
349
443
|
terminalOutcome: turn.outcome.success ? "succeeded" : turn.outcome.status === "failed" ? "failed" : "incomplete"
|
|
@@ -397,11 +491,12 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
397
491
|
execution: result.execution,
|
|
398
492
|
artifactAvailable: turn.readError === void 0 && boxExtractError === void 0,
|
|
399
493
|
usage: result.usage,
|
|
400
|
-
events:
|
|
494
|
+
events: observedEvents,
|
|
495
|
+
prompts,
|
|
401
496
|
...detail ? { detail } : {}
|
|
402
497
|
};
|
|
403
498
|
} catch (err) {
|
|
404
|
-
const events =
|
|
499
|
+
const events = observedEvents;
|
|
405
500
|
result = {
|
|
406
501
|
...result,
|
|
407
502
|
ok: false,
|
|
@@ -411,12 +506,9 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
411
506
|
terminalOutcome: result.execution?.terminalOutcome ?? "unknown"
|
|
412
507
|
},
|
|
413
508
|
detail: err instanceof Error ? err.message : String(err),
|
|
414
|
-
usage:
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
usdKnown: false
|
|
418
|
-
},
|
|
419
|
-
events
|
|
509
|
+
usage: promptUsage(prompts),
|
|
510
|
+
events,
|
|
511
|
+
prompts
|
|
420
512
|
};
|
|
421
513
|
} finally {
|
|
422
514
|
if (timer) clearTimeout(timer);
|
|
@@ -432,6 +524,27 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
|
|
|
432
524
|
}
|
|
433
525
|
return result;
|
|
434
526
|
};
|
|
527
|
+
/** Stop waiting on policy work when cancelled; the caller must also cancel its external effects. */
|
|
528
|
+
async function executeWithSignal(execute, signal) {
|
|
529
|
+
signal.throwIfAborted();
|
|
530
|
+
let onAbort;
|
|
531
|
+
const aborted = new Promise((_resolve, reject) => {
|
|
532
|
+
onAbort = () => reject(signal.reason ?? /* @__PURE__ */ new Error("aborted"));
|
|
533
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
534
|
+
});
|
|
535
|
+
try {
|
|
536
|
+
await Promise.race([Promise.resolve().then(execute), aborted]);
|
|
537
|
+
} finally {
|
|
538
|
+
if (onAbort) signal.removeEventListener("abort", onAbort);
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
function promptUsage(prompts) {
|
|
542
|
+
return combinedUsage(prompts.map((prompt) => ({
|
|
543
|
+
artifact: "",
|
|
544
|
+
ok: false,
|
|
545
|
+
usage: prompt.usage
|
|
546
|
+
})));
|
|
547
|
+
}
|
|
435
548
|
function parseMaybeJson(value) {
|
|
436
549
|
try {
|
|
437
550
|
return JSON.parse(value);
|
|
@@ -531,6 +644,7 @@ async function loopedShot(input, shot, attempts) {
|
|
|
531
644
|
ok: false
|
|
532
645
|
}] : completed),
|
|
533
646
|
events: completed.flatMap((shot) => shot.events ?? []),
|
|
647
|
+
prompts: completed.flatMap((shot) => shot.prompts ?? []),
|
|
534
648
|
detail: err instanceof Error ? err.message : String(err)
|
|
535
649
|
};
|
|
536
650
|
}
|
|
@@ -555,6 +669,7 @@ async function loopedShot(input, shot, attempts) {
|
|
|
555
669
|
artifactAvailable: shots.get(best.round)?.artifactAvailable,
|
|
556
670
|
usage: combinedUsage([...shots.values()]),
|
|
557
671
|
events: [...shots.values()].flatMap((shot) => shot.events ?? []),
|
|
672
|
+
prompts: [...shots.values()].flatMap((shot) => shot.prompts ?? []),
|
|
558
673
|
detail: JSON.stringify({
|
|
559
674
|
mode: "refine-loop",
|
|
560
675
|
attempts: result.rounds.length,
|
|
@@ -690,7 +805,8 @@ async function runBenchmarks(opts) {
|
|
|
690
805
|
...opts.sandboxBaseUrl ? { sandboxBaseUrl: opts.sandboxBaseUrl } : {},
|
|
691
806
|
...opts.timeoutMs ? { timeoutMs: opts.timeoutMs } : {},
|
|
692
807
|
...opts.signal ? { signal: opts.signal } : {},
|
|
693
|
-
...opts.resolveClient ? { resolveClient: opts.resolveClient } : {}
|
|
808
|
+
...opts.resolveClient ? { resolveClient: opts.resolveClient } : {},
|
|
809
|
+
...opts.execute ? { execute: opts.execute } : {}
|
|
694
810
|
};
|
|
695
811
|
invoked = true;
|
|
696
812
|
out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput);
|
|
@@ -712,7 +828,8 @@ async function runBenchmarks(opts) {
|
|
|
712
828
|
wallMs: Date.now() - startedAt,
|
|
713
829
|
artifact: out.artifact,
|
|
714
830
|
...out.usage === void 0 ? {} : { usage: out.usage },
|
|
715
|
-
...out.events === void 0 ? {} : { events: out.events }
|
|
831
|
+
...out.events === void 0 ? {} : { events: out.events },
|
|
832
|
+
...out.prompts === void 0 ? {} : { prompts: out.prompts }
|
|
716
833
|
};
|
|
717
834
|
} catch (err) {
|
|
718
835
|
result = {
|
|
@@ -732,7 +849,8 @@ async function runBenchmarks(opts) {
|
|
|
732
849
|
wallMs: Date.now() - startedAt,
|
|
733
850
|
...out === void 0 ? {} : { artifact: out.artifact },
|
|
734
851
|
...out?.usage === void 0 ? {} : { usage: out.usage },
|
|
735
|
-
...out?.events === void 0 ? {} : { events: out.events }
|
|
852
|
+
...out?.events === void 0 ? {} : { events: out.events },
|
|
853
|
+
...out?.prompts === void 0 ? {} : { prompts: out.prompts }
|
|
736
854
|
};
|
|
737
855
|
}
|
|
738
856
|
perTask.push(result);
|