@tangle-network/agent-bench 0.9.4 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/HARNESS.md CHANGED
@@ -62,27 +62,27 @@ Use `LOOP_ATTEMPTS=N` only when the benchmark's own visible feedback is allowed
62
62
  `runBenchmarks()` returns each judged artifact, worker events, and observed usage in `perTask`.
63
63
  Retry usage includes every attempt; missing receipts leave the measured subtotal explicitly incomplete.
64
64
  Judge failures retain completed worker evidence.
65
+ Each task separates `execution` from `measurement` availability.
66
+ Captured empty output and explicit failed turns remain measured failures when the evaluator runs successfully.
67
+ Read, extraction, and judge failures leave measurement unavailable while retaining observed usage.
68
+ Missing dispatch evidence remains unknown; `ok: false` never establishes permission to retry.
65
69
  Errors propagated by `close()` remain in `detail` beside the settled task outcome.
66
70
  The current Runtime lineage suppresses sandbox deletion errors, so a returned result does not confirm resource deletion.
67
71
  The caller's abort signal stops queued shots and reaches active sandbox turns.
68
72
  `modelApiKey` supplies sandbox inference authorization separately from the `routerKey` used for sandbox control.
69
73
 
70
- ### Full-fidelity improvement fixture
74
+ ### Retained strategy driver
71
75
 
72
76
  ```bash
73
77
  cd bench
74
78
  pnpm tsx src/swe-self-improve.mts
75
79
  ```
76
80
 
77
- This is the canonical real-task Runtime fixture:
78
-
79
- - SWE-bench Verified instances;
80
- - repository state as the produced artifact;
81
- - the official Docker judge outside the candidate agent;
82
- - explicit train, selection, and frozen final-test partitions;
83
- - Runtime's `improve()` boundary and complete cost receipts.
84
-
85
- It proves the integrated execution path can support a real value campaign. A paid powered result still belongs in Discovery Lab.
81
+ This driver uses `runStrategyEvolution` with SWE-bench tasks and a frozen holdout.
82
+ It does not exercise `improve`, and it deletes its temporary run directory on exit.
83
+ It therefore cannot provide retained improvement or lineage evidence.
84
+ Use `examples/improve` for the maintained offline API fixture.
85
+ Use the consuming labs for registered learning campaigns with retained execution and comparison evidence.
86
86
 
87
87
  ### Offline diagnostics
88
88
 
package/dist/index.d.ts CHANGED
@@ -11,7 +11,7 @@ import { createT2RagBenchAdapter } from "./benchmarks/t2-ragbench.js";
11
11
  import { AgentProfile, SandboxClient, sumSandboxUsage } from "@tangle-network/agent-runtime/kernel";
12
12
  import { SandboxEvent } from "@tangle-network/sandbox";
13
13
  import { AgentCandidateBenchmarkGraderPort, AgentCandidateExecutionClaimStore, AgentCandidateExecutorPort, AgentCandidateExecutorRequest, AgentCandidateExecutorStopRequest, AgentCandidateOutputArtifactPort, AgentCandidateRunFinalization, PreparedAgentCandidateExecution } from "@tangle-network/agent-runtime/candidate-execution";
14
- import { TraceStore } from "@tangle-network/agent-eval";
14
+ import { RunTerminalOutcome, TraceStore } from "@tangle-network/agent-eval";
15
15
  //#region src/resolve-client.d.ts
16
16
  interface ResolveBenchClientOptions {
17
17
  /** The selector (the `BACKEND` env value): `router` = off-box; anything else = in-box Sandbox. */
@@ -56,6 +56,13 @@ interface BenchShotResult {
56
56
  /** Provider observations, including explicit unknown counters. Omitted when the shot reports none. */
57
57
  readonly usage?: ReturnType<typeof sumSandboxUsage>;
58
58
  readonly events?: readonly SandboxEvent[];
59
+ /** Observed dispatch and terminal state, independent of artifact quality. */
60
+ readonly execution?: {
61
+ readonly phase: 'not-started' | 'started' | 'unknown';
62
+ readonly terminalOutcome: RunTerminalOutcome;
63
+ };
64
+ /** Whether the artifact was captured without a read or extraction failure. */
65
+ readonly artifactAvailable?: boolean;
59
66
  }
60
67
  /** Runs one (adapter, task, cell) shot. Defaults to `openSandboxRun`. */
61
68
  type BenchShot = (input: {
@@ -122,9 +129,11 @@ interface BenchCellTaskResult {
122
129
  readonly rep: number;
123
130
  readonly resolved: boolean;
124
131
  readonly score: number;
125
- /** false = the shot threw or produced no artifact (infra/empty), excluded from the resolve
126
- * denominator so a harness outage can't masquerade as a 0% capability result. */
132
+ /** Whether execution completed successfully and produced a readable, nonempty artifact. */
127
133
  readonly ok: boolean;
134
+ readonly execution?: BenchShotResult['execution'];
135
+ /** Available failed attempts remain in comparisons; unavailable measurement is reported separately. */
136
+ readonly measurement?: 'available' | 'unavailable';
128
137
  readonly detail?: string;
129
138
  readonly wallMs: number;
130
139
  /** Exact bytes given to the benchmark judge, retained even when judging fails. */
package/dist/index.js CHANGED
@@ -304,11 +304,15 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
304
304
  };
305
305
  const controller = new AbortController();
306
306
  const timer = timeoutMs ? setTimeout(() => controller.abort(), timeoutMs) : void 0;
307
+ const observedEvents = [];
307
308
  const runOptions = {
308
309
  agentRun,
309
310
  signal: signal ? AbortSignal.any([controller.signal, signal]) : controller.signal,
310
311
  runId: `bench:${adapter.name}:${task.id}:${uniq}`,
311
- scenarioId: task.id
312
+ scenarioId: task.id,
313
+ onSandboxEvent: (event) => {
314
+ observedEvents.push(event);
315
+ }
312
316
  };
313
317
  const boxSetup = adapter.boxSetup;
314
318
  if (boxSetup) runOptions.beforeStart = async ({ box, sessionId }) => {
@@ -327,12 +331,23 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
327
331
  };
328
332
  try {
329
333
  run = await openSandboxRun(client, runOptions, deliverable);
334
+ result = {
335
+ ...result,
336
+ execution: {
337
+ phase: "unknown",
338
+ terminalOutcome: "unknown"
339
+ }
340
+ };
330
341
  const turn = await run.start(prompt ?? task.prompt);
331
342
  result = {
332
343
  artifact: "",
333
344
  ok: false,
334
345
  usage: sumSandboxUsage(turn.events),
335
- events: turn.events
346
+ events: turn.events,
347
+ execution: {
348
+ phase: "started",
349
+ terminalOutcome: turn.outcome.success ? "succeeded" : turn.outcome.status === "failed" ? "failed" : "incomplete"
350
+ }
336
351
  };
337
352
  let artifact = (turn.out ?? "").trim();
338
353
  let boxExtractError;
@@ -379,19 +394,29 @@ const openSandboxShot = async ({ adapter, task, cell, prompt, routerBaseUrl, rou
379
394
  result = {
380
395
  artifact,
381
396
  ok: turn.outcome.success && artifact.length > 0 && turn.readError === void 0 && boxExtractError === void 0,
397
+ execution: result.execution,
398
+ artifactAvailable: turn.readError === void 0 && boxExtractError === void 0,
382
399
  usage: result.usage,
383
400
  events: turn.events,
384
401
  ...detail ? { detail } : {}
385
402
  };
386
403
  } catch (err) {
404
+ const events = err instanceof SandboxRunAbortError ? err.events : observedEvents;
387
405
  result = {
388
406
  ...result,
389
407
  ok: false,
408
+ artifactAvailable: false,
409
+ execution: {
410
+ phase: events.length > 0 ? "started" : "unknown",
411
+ terminalOutcome: result.execution?.terminalOutcome ?? "unknown"
412
+ },
390
413
  detail: err instanceof Error ? err.message : String(err),
391
- ...err instanceof SandboxRunAbortError ? {
392
- usage: sumSandboxUsage(err.events),
393
- events: err.events
394
- } : {}
414
+ usage: {
415
+ ...sumSandboxUsage(events),
416
+ tokensKnown: false,
417
+ usdKnown: false
418
+ },
419
+ events
395
420
  };
396
421
  } finally {
397
422
  if (timer) clearTimeout(timer);
@@ -496,6 +521,11 @@ async function loopedShot(input, shot, attempts) {
496
521
  return {
497
522
  artifact: completed.at(-1)?.artifact ?? "",
498
523
  ok: false,
524
+ execution: pendingShot ? {
525
+ phase: "unknown",
526
+ terminalOutcome: "unknown"
527
+ } : completed.at(-1)?.execution,
528
+ artifactAvailable: false,
499
529
  usage: combinedUsage(pendingShot ? [...completed, {
500
530
  artifact: "",
501
531
  ok: false
@@ -505,8 +535,10 @@ async function loopedShot(input, shot, attempts) {
505
535
  };
506
536
  }
507
537
  const best = result.rounds.reduce((winner, candidate) => {
508
- if (shots.get(candidate.round)?.ok !== true) return winner;
509
- if (shots.get(winner.round)?.ok !== true) return candidate;
538
+ const candidateShot = shots.get(candidate.round);
539
+ const winnerShot = shots.get(winner.round);
540
+ const rank = (shot) => shot?.ok ? 2 : shot?.artifactAvailable ? 1 : 0;
541
+ if (rank(candidateShot) !== rank(winnerShot)) return rank(candidateShot) > rank(winnerShot) ? candidate : winner;
510
542
  const a = scores.get(winner.round);
511
543
  const b = scores.get(candidate.round);
512
544
  if (!a) return candidate;
@@ -519,6 +551,8 @@ async function loopedShot(input, shot, attempts) {
519
551
  return {
520
552
  artifact: best.artifact,
521
553
  ok: shots.get(best.round)?.ok === true && best.artifact.trim().length > 0,
554
+ execution: shots.get(best.round)?.execution,
555
+ artifactAvailable: shots.get(best.round)?.artifactAvailable,
522
556
  usage: combinedUsage([...shots.values()]),
523
557
  events: [...shots.values()].flatMap((shot) => shot.events ?? []),
524
558
  detail: JSON.stringify({
@@ -641,6 +675,7 @@ async function runBenchmarks(opts) {
641
675
  const startedAt = Date.now();
642
676
  let result;
643
677
  let out;
678
+ let invoked = false;
644
679
  try {
645
680
  opts.signal?.throwIfAborted();
646
681
  const shotInput = {
@@ -657,6 +692,7 @@ async function runBenchmarks(opts) {
657
692
  ...opts.signal ? { signal: opts.signal } : {},
658
693
  ...opts.resolveClient ? { resolveClient: opts.resolveClient } : {}
659
694
  };
695
+ invoked = true;
660
696
  out = loopAttempts > 1 ? await loopedShot(shotInput, shot, loopAttempts) : await shot(shotInput);
661
697
  const score = await job.adapter.judge(job.task, out.artifact);
662
698
  result = {
@@ -667,6 +703,11 @@ async function runBenchmarks(opts) {
667
703
  resolved: out.ok && score.resolved,
668
704
  score: out.ok ? score.score : 0,
669
705
  ok: out.ok,
706
+ execution: out.execution ?? {
707
+ phase: out.ok ? "started" : "unknown",
708
+ terminalOutcome: out.ok ? "succeeded" : "unknown"
709
+ },
710
+ measurement: out.artifactAvailable ?? out.ok ? "available" : "unavailable",
670
711
  ...out.detail ?? score.detail ? { detail: combineDetails(out.detail, score.detail) } : {},
671
712
  wallMs: Date.now() - startedAt,
672
713
  artifact: out.artifact,
@@ -682,6 +723,11 @@ async function runBenchmarks(opts) {
682
723
  resolved: false,
683
724
  score: 0,
684
725
  ok: false,
726
+ execution: out?.execution ?? {
727
+ phase: !invoked ? "not-started" : out?.ok ? "started" : "unknown",
728
+ terminalOutcome: out?.ok ? "succeeded" : "unknown"
729
+ },
730
+ measurement: "unavailable",
685
731
  detail: err instanceof Error ? err.message.slice(0, 200) : String(err),
686
732
  wallMs: Date.now() - startedAt,
687
733
  ...out === void 0 ? {} : { artifact: out.artifact },
@@ -714,7 +760,7 @@ function aggregate(perTask) {
714
760
  scoreSum: 0
715
761
  };
716
762
  e.n += 1;
717
- if (!r.ok) e.errored += 1;
763
+ if ((r.measurement ?? (r.ok ? "available" : "unavailable")) === "unavailable") e.errored += 1;
718
764
  else {
719
765
  if (r.resolved) e.resolved += 1;
720
766
  e.scoreSum += r.score;