@hona/openeval 0.5.9 → 0.5.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.9",
3
+ "version": "0.5.11",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -0,0 +1,47 @@
1
+ import type { BenchmarkDefinition, BenchmarkRun, PlanItem } from "../types";
2
+ import type { Results } from "../infra/sqlite";
3
+ import { candidateFingerprint, judgeFingerprint } from "./input-fingerprints";
4
+
5
+ /** Explicit IDs only: do not recollect fully scored or superseded recordings. */
6
+ export function retryPlan(
7
+ definition: BenchmarkDefinition,
8
+ runtime: BenchmarkRun["runtime"],
9
+ results: Results,
10
+ evalRunIds: readonly string[],
11
+ ): PlanItem[] {
12
+ if (!evalRunIds.length || new Set(evalRunIds).size !== evalRunIds.length)
13
+ throw new Error("Select unique EvalRun IDs to retry");
14
+ return evalRunIds.map(id => {
15
+ const previous = results.evalRun(id);
16
+ const slot = previous && results.slot(previous.slotId);
17
+ if (!previous || !slot?.active || slot.evalRunId !== previous.id)
18
+ throw new Error("Select the active EvalRun for each repetition");
19
+ const judge = slot.judgeRunId && results.judgeRun(slot.judgeRunId);
20
+ const unresolved = judge && judge.state === "completed" &&
21
+ Object.values(judge.judgment?.scores ?? {}).some(score => score.value === null);
22
+ if (previous.state === "running" || (previous.state === "completed" && !unresolved))
23
+ throw new Error("Select failed, stopped, timed-out, or completed-but-unresolved work");
24
+ const item = definition.evals.find(item => item.id === slot.evalId);
25
+ if (!item || item.prompt !== previous.input.prompt || item.sourceHash !== previous.input.sourceHash ||
26
+ runtime.imageId !== previous.input.imageId)
27
+ throw new Error("Task inputs or image changed; use scoped run for changed work");
28
+ // The current authored time limit may replace an older limit. Other inputs must match.
29
+ const recordedLimits = {
30
+ ...definition,
31
+ evals: definition.evals.map(value => value.id === item.id ? {
32
+ ...value, settings: { ...value.settings,
33
+ candidate: { ...value.settings.candidate, timeoutMs: previous.input.timeoutMs } },
34
+ } : value),
35
+ };
36
+ if (candidateFingerprint(recordedLimits, slot.evalId, slot.model, runtime) !== previous.input.candidateHash)
37
+ throw new Error("Candidate inputs other than the time limit changed");
38
+ return {
39
+ action: "candidate",
40
+ reason: previous.state === "completed" ? "Explicit retry of unresolved recorded work" : "Explicit retry of interrupted or failed work",
41
+ slot: { ...slot,
42
+ candidateHash: candidateFingerprint(definition, slot.evalId, slot.model, runtime),
43
+ judgeHash: judgeFingerprint(definition, slot.evalId),
44
+ },
45
+ };
46
+ });
47
+ }
@@ -43,10 +43,17 @@ const costOf = (run: EvalRun | JudgeRun, results?: Results): Cost => {
43
43
  complete: usd !== undefined,
44
44
  };
45
45
  };
46
- // Execution IDs are unique, including after merging runs.
46
+ // Restored records represent the same original execution, not additional model spend.
47
47
  const runCosts = (runs: (EvalRun | JudgeRun)[], results: Results) => {
48
48
  const unique = new Map(runs.map((run) => [run.id, run]));
49
- return sumCosts([...unique.values()].map((run) => costOf(run, results)));
49
+ const executions = new Map<string, EvalRun | JudgeRun>();
50
+ for (const run of unique.values()) {
51
+ const restoration = "restoration" in run ? run.restoration : undefined;
52
+ const key = restoration?.evalRunId ?? run.id;
53
+ const previous = executions.get(key);
54
+ if (!previous || restoration) executions.set(key, run);
55
+ }
56
+ return sumCosts([...executions.values()].map((run) => costOf(run, results)));
50
57
  };
51
58
  const entryOf = (
52
59
  run: BenchmarkRun,
@@ -18,26 +18,41 @@ export async function restoreEvalRun(
18
18
  using results = new Results(resolve(root, "runner.db"));
19
19
  const previous = results.evalRun(evalRunId);
20
20
  const slot = previous && results.slot(previous.slotId);
21
+ const original = previous?.restoration
22
+ ? results.evalRun(previous.restoration.evalRunId)
23
+ : previous;
21
24
  if (!previous || !slot || !slot.active || slot.evalRunId !== previous.id ||
22
- previous.state !== "failed" || !previous.interrupted || previous.evidence)
23
- throw new Error("Select an active interrupted EvalRun without finalized evidence");
25
+ !original || original.state !== "failed" || !original.interrupted || original.evidence ||
26
+ (previous !== original && (previous.state !== "completed" || !previous.restoration)))
27
+ throw new Error("Select an active interrupted EvalRun or its restored recording");
28
+ if (previous.restoration && (options.databaseHash !== previous.restoration.databaseHash ||
29
+ options.workspaceArchiveHash !== previous.restoration.workspaceArchiveHash))
30
+ throw new Error("A restored recording must retain the same original backup hashes");
24
31
  if (results.benchmark!.state === "running" || results.benchmark!.mergedInto)
25
32
  throw new Error("Wait for the active BenchmarkRun and use its aggregate directory");
26
- const events = (await readFile(resolve(root, "eval-runs", previous.id, "evidence/events.jsonl"), "utf8"))
33
+ const events = (await readFile(resolve(root, "eval-runs", original.id, "evidence/events.jsonl"), "utf8"))
27
34
  .trim().split("\n").filter(Boolean).map(line => JSON.parse(line).event);
28
35
  const creation = events.find(event => event.type === "session.created" && !event.data.parentID);
29
36
  if (!creation?.data.sessionID)
30
37
  throw new Error("The original recording does not identify its root session");
31
- const initial = resolve(root, "inputs", previous.input.evalId, previous.input.sourceHash, "workspace");
32
- if (!(await Bun.file(resolve(initial, "..", "ready")).exists()))
38
+ const frozen = resolve(root, "inputs", original.input.evalId, original.input.sourceHash, "workspace");
39
+ if (!(await Bun.file(resolve(frozen, "..", "ready")).exists()))
33
40
  throw new Error("The frozen starting workspace is missing");
41
+ const item = results.benchmark!.definition.evals.find(item => item.id === original.input.evalId);
42
+ if (!item || item.sourceHash !== original.input.sourceHash)
43
+ throw new Error("The frozen preparation no longer matches the original inputs");
34
44
  const staging = resolve(root, "eval-runs", `.restored-${randomUUID()}`);
35
45
  await mkdir(staging, { recursive: false });
36
46
  try {
47
+ const preparation = resolve(staging, "preparation");
48
+ const initial = await archive.prepareRestoredInitial(
49
+ results.benchmark!.definition, item, original.input.imageId, frozen, preparation,
50
+ );
37
51
  const restored = await archive.restoreArchivedCandidate(
38
- previous.input, creation.data.sessionID, initial, options, staging,
52
+ original.input, creation.data.sessionID, initial, options, staging,
39
53
  );
40
- const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(previous.startedAt);
54
+ await rm(preparation, { recursive: true, force: true });
55
+ const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(original.startedAt);
41
56
  if (!Number.isFinite(elapsedMs) || elapsedMs < 0)
42
57
  throw new Error("Native completion predates the original EvalRun");
43
58
  const evidence = { ...restored.evidence };
@@ -46,19 +61,19 @@ export async function restoreEvalRun(
46
61
  if (results.benchmark!.state === "running" ||
47
62
  results.slot(slot.id)?.evalRunId !== previous.id)
48
63
  throw new Error("The BenchmarkRun or selected execution changed during restoration");
49
- const run = results.startEval(slot, previous.input);
64
+ const run = results.startEval(slot, original.input);
50
65
  const destination = resolve(root, "eval-runs", run.id);
51
66
  renameSync(staging, destination);
52
67
  evidence.directory = relative(root, resolve(destination, restored.evidence.directory));
53
68
  const completed: EvalRun = {
54
69
  ...run,
55
70
  state: "completed",
56
- startedAt: previous.startedAt,
71
+ startedAt: original.startedAt,
57
72
  completedAt: restored.receipt.completedAt,
58
73
  elapsedMs,
59
74
  evidence,
60
75
  session: { ...restored.session, database: relative(root, resolve(destination, restored.session.database)) },
61
- restoration: { ...restored.receipt, evalRunId: previous.id },
76
+ restoration: { ...restored.receipt, evalRunId: original.id },
62
77
  metrics: measureRecording(restored.events.map((event, sequence) => ({
63
78
  event, sequence, time: new Date("created" in event ? event.created : Date.now()).toISOString(),
64
79
  })), restored.tools, evidence),
@@ -1,18 +1,40 @@
1
1
  import { resolve } from "node:path";
2
- import type { ExecutionObserver } from "../types";
2
+ import type { EvalRun, ExecutionObserver } from "../types";
3
3
  import { Results } from "../infra/sqlite";
4
4
  import { runtimeFingerprint } from "../infra/containers/oci";
5
- import { candidateFingerprint } from "./plan-benchmark";
6
5
  import { runEvalPipeline } from "./run-eval-pipeline";
7
6
  import { ExecutionBudget } from "./execution-budget";
8
- import { finishBenchmark } from "./run-benchmark";
7
+ import { executeQueue, finishBenchmark } from "./run-benchmark";
9
8
  import { recoverStoppedRunner, runnerStopped } from "./recover-run";
9
+ import { retryPlan } from "./plan-retries";
10
+ import { CostBudget, estimateWork } from "./cost-plan";
10
11
 
11
12
  export async function retryEvalRun(
12
13
  directory: string,
13
14
  evalRunId: string,
14
15
  onEvent?: ExecutionObserver,
15
16
  ) {
17
+ return (await retryEvalRuns(directory, [evalRunId], onEvent))[0]!;
18
+ }
19
+
20
+ /** Read-only preflight for a precise retry scope. No model calls or selection updates. */
21
+ export async function planEvalRunRetries(directory: string, evalRunIds: readonly string[]) {
22
+ using results = new Results(resolve(directory, "runner.db"), true);
23
+ const benchmark = results.benchmark!;
24
+ if (benchmark.mergedInto || benchmark.state === "running")
25
+ throw new Error("Wait for active work and select the aggregate run");
26
+ const runtime = await runtimeFingerprint(benchmark.definition.container, benchmark.definition.judge.verification);
27
+ const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
28
+ return { plan, estimate: estimateWork(benchmark.definition, results, plan) };
29
+ }
30
+
31
+ /** A single retry round; original evidence and judgments remain immutable. */
32
+ export async function retryEvalRuns(
33
+ directory: string,
34
+ evalRunIds: readonly string[],
35
+ onEvent?: ExecutionObserver,
36
+ options: { maxCostUSD?: number } = {},
37
+ ): Promise<EvalRun[]> {
16
38
  using results = new Results(resolve(directory, "runner.db"));
17
39
  if (results.benchmark!.mergedInto)
18
40
  throw new Error(
@@ -24,41 +46,26 @@ export async function retryEvalRun(
24
46
  recoverStoppedRunner(results);
25
47
  }
26
48
  const benchmark = results.benchmark!;
27
- const previous = results.evalRun(evalRunId);
28
- if (!previous || previous.state === "completed")
29
- throw new Error(
30
- "Select a failed eval run; completed evidence can be rejudged",
31
- );
32
- const slot = results.slot(previous.slotId)!;
33
- if (slot.evalRunId !== previous.id)
34
- throw new Error("Select the current eval run for this repetition");
35
49
  const runtime = await runtimeFingerprint(
36
50
  benchmark.definition.container,
37
51
  benchmark.definition.judge.verification,
38
52
  );
39
- if (
40
- candidateFingerprint(
41
- benchmark.definition,
42
- slot.evalId,
43
- slot.model,
44
- runtime,
45
- ) !== slot.candidateHash
46
- )
47
- throw new Error(
48
- "Declared candidate inputs changed; use run to schedule the affected work",
49
- );
50
- const definition = benchmark.definition.evals.find(
51
- (evalDefinition) => evalDefinition.id === slot.evalId,
52
- )!;
53
- const workspace = resolve(
54
- directory,
55
- "inputs",
56
- definition.id,
57
- definition.sourceHash,
58
- "workspace",
59
- );
60
- if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
61
- throw new Error("The saved starting workspace is missing");
53
+ const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
54
+ const workspaces = new Map<string, string>();
55
+ for (const item of plan) {
56
+ const definition = benchmark.definition.evals.find(value => value.id === item.slot.evalId)!;
57
+ const workspace = resolve(directory, "inputs", definition.id, definition.sourceHash, "workspace");
58
+ if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
59
+ throw new Error("The saved starting workspace is missing");
60
+ workspaces.set(item.slot.evalId, workspace);
61
+ }
62
+ const estimate = estimateWork(benchmark.definition, results, plan);
63
+ const existing = new Set([...results.evalRuns(), ...results.judgeRuns()].map(run => run.id));
64
+ const spent = () => [...results.evalRuns(), ...results.judgeRuns()]
65
+ .filter(run => !existing.has(run.id))
66
+ .reduce((sum, run) => sum + (run.session?.accounting?.costUSD ?? results.recordedCost(run.id) ?? 0), 0);
67
+ const costBudget = new CostBudget(options.maxCostUSD, spent);
68
+ const estimates = new Map(estimate.items.map(item => [item.slotId, item.estimatedUSD]));
62
69
  const context = {
63
70
  directory: resolve(directory),
64
71
  definition: benchmark.definition,
@@ -66,29 +73,58 @@ export async function retryEvalRun(
66
73
  results,
67
74
  onEvent,
68
75
  };
69
- results.saveBenchmark({
70
- ...benchmark,
71
- state: "running",
72
- scheduledSlotIds: [slot.id],
73
- updatedAt: new Date().toISOString(),
76
+ results.transaction(() => {
77
+ if (results.benchmark!.state === "running") throw new Error("Another operation claimed this run");
78
+ for (const item of plan) {
79
+ const selected = results.slot(item.slot.id);
80
+ if (!selected?.active || selected.evalRunId !== item.slot.evalRunId ||
81
+ selected.judgeRunId !== item.slot.judgeRunId)
82
+ throw new Error("The retry selection changed during preflight");
83
+ }
84
+ results.saveBenchmark({ ...benchmark, runtime, state: "running",
85
+ scheduledSlotIds: plan.map(item => item.slot.id), updatedAt: new Date().toISOString(),
86
+ execution: { startedAt: new Date().toISOString(),
87
+ onlyModels: [...new Set(plan.map(item => item.slot.model))],
88
+ onlyEvals: [...new Set(plan.map(item => item.slot.evalId))],
89
+ onlyRepetitions: [...new Set(plan.map(item => item.slot.repetition))],
90
+ budgetUSD: options.maxCostUSD, estimatedUSD: estimate.estimatedUSD,
91
+ spentUSD: 0, deferred: 0 } });
74
92
  });
75
93
  const timer = setInterval(
76
94
  () =>
77
95
  results.saveBenchmark({
78
96
  ...results.benchmark!,
79
97
  updatedAt: new Date().toISOString(),
98
+ scheduledSlotIds: plan.filter(item => item.action !== "deferred").map(item => item.slot.id),
99
+ execution: { ...results.benchmark!.execution, spentUSD: spent(),
100
+ deferred: plan.filter(item => item.action === "deferred").length },
80
101
  }),
81
102
  5000,
82
103
  );
83
104
  try {
84
- return await runEvalPipeline(
85
- context,
86
- slot,
87
- workspace,
88
- new ExecutionBudget(benchmark.definition.concurrency),
89
- );
105
+ const budget = new ExecutionBudget(benchmark.definition.concurrency);
106
+ const completed = new Map<string, EvalRun>();
107
+ const errors: unknown[] = [];
108
+ await executeQueue(plan, benchmark.definition.concurrency, async item => {
109
+ const release = costBudget.reserve(estimates.get(item.slot.id) ?? null);
110
+ if (!release) {
111
+ item.action = "deferred";
112
+ item.reason = "Retry scheduling budget reached or no estimate is available";
113
+ return;
114
+ }
115
+ try {
116
+ const run = await runEvalPipeline(context, item.slot, workspaces.get(item.slot.evalId)!, budget);
117
+ completed.set(item.slot.id, run);
118
+ } catch (error) {
119
+ errors.push(error);
120
+ } finally { release(); }
121
+ });
122
+ if (errors.length) throw new AggregateError(errors, "One or more retry operations failed");
123
+ return plan.flatMap(item => completed.has(item.slot.id) ? [completed.get(item.slot.id)!] : []);
90
124
  } finally {
91
125
  clearInterval(timer);
126
+ results.saveBenchmark({ ...results.benchmark!, execution: { ...results.benchmark!.execution,
127
+ spentUSD: spent(), deferred: plan.filter(item => item.action === "deferred").length } });
92
128
  finishBenchmark(context);
93
129
  }
94
130
  }
package/src/cli.ts CHANGED
@@ -7,7 +7,8 @@ import {
7
7
  buildImage,
8
8
  addModels,
9
9
  removeModels,
10
- retryEvalRun,
10
+ retryEvalRuns,
11
+ planEvalRunRetries,
11
12
  restoreEvalRun,
12
13
  judgeRun,
13
14
  serveResults,
@@ -48,7 +49,7 @@ Commands:
48
49
  add-models <run> Add models with repeated --model flags
49
50
  remove-models <run> Remove active models while retaining their evidence
50
51
  refresh-model-names <run> Refresh reporting names without running evals
51
- retry <run> <eval-run> Retry a candidate execution
52
+ retry <run> <eval-run>... Retry failed or unresolved candidate work
52
53
  rejudge <run> <eval-run> Judge the saved evidence again
53
54
  restore <run> <eval-run> Restore completed orphan work from hashed backups
54
55
 
@@ -63,6 +64,7 @@ Options:
63
64
  --only-repetition <n> Execute only this repetition (repeatable)
64
65
  --max-cost <usd> Scheduling budget for this invocation
65
66
  --final-only Judge only after candidates finish
67
+ --dry-run Plan explicit retries without running them
66
68
  --verification Build only the standard verification image (image command)
67
69
  --output <dir> New prepared-input directory (prepare command)
68
70
  --port <port> Viewer port (default: 4173)
@@ -184,15 +186,22 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
184
186
  console.log(JSON.stringify(await restoreEvalRun(resolve(args[1]), args[2], {
185
187
  database: resolve(database), databaseHash, workspaceArchive: resolve(workspaceArchive), workspaceArchiveHash,
186
188
  }), null, 2));
187
- } else if (command === "retry" || command === "rejudge") {
189
+ } else if (command === "retry") {
190
+ const flags = args.findIndex((arg, index) => index >= 2 && arg.startsWith("--"));
191
+ const ids = args.slice(2, flags < 0 ? args.length : flags);
192
+ if (!args[1] || !ids.length) throw new Error("retry requires a run directory and explicit EvalRun IDs");
193
+ const result = args.includes("--dry-run")
194
+ ? await planEvalRunRetries(resolve(args[1]), ids)
195
+ : await retryEvalRuns(resolve(args[1]), ids, undefined,
196
+ { maxCostUSD: option("--max-cost") === undefined ? undefined : Number(option("--max-cost")) });
197
+ console.log(JSON.stringify(!args.includes("--dry-run") && ids.length === 1 && Array.isArray(result)
198
+ ? result[0] : result, null, 2));
199
+ } else if (command === "rejudge") {
188
200
  if (!args[1] || !args[2])
189
201
  throw new Error(
190
202
  `${command} requires a benchmark run directory and eval run ID`,
191
203
  );
192
- const result =
193
- command === "retry"
194
- ? await retryEvalRun(resolve(args[1]), args[2])
195
- : await judgeRun(resolve(args[1]), args[2]);
204
+ const result = await judgeRun(resolve(args[1]), args[2]);
196
205
  console.log(JSON.stringify(result, null, 2));
197
206
  } else
198
207
  throw new Error(
package/src/index.ts CHANGED
@@ -67,7 +67,7 @@ export { removeModels } from "./app/remove-models";
67
67
  export { refreshModelNames } from "./app/refresh-model-names";
68
68
  export type { CostEstimate } from "./app/cost-plan";
69
69
  export { mergeBenchmarkRuns } from "./app/merge-runs";
70
- export { retryEvalRun } from "./app/retry-run";
70
+ export { retryEvalRun, retryEvalRuns, planEvalRunRetries } from "./app/retry-run";
71
71
  export { restoreEvalRun } from "./app/restore-eval-run";
72
72
  export { judgeRun, judgeRuns } from "./app/rejudge";
73
73
  export { judgeEvidence, recordEvidence } from "./app/judge-evidence";
@@ -1,13 +1,15 @@
1
1
  import { Database } from "bun:sqlite";
2
2
  import { mkdir, mkdtemp, rm } from "node:fs/promises";
3
3
  import { resolve, dirname } from "node:path";
4
- import type { EvalRunInput, OpenCodeStreamEvent } from "../types";
4
+ import type { BenchmarkDefinition, EvalDefinition, EvalRunInput, OpenCodeStreamEvent } from "../types";
5
5
  import { EvidenceCapture } from "./evidence";
6
6
  import { extractWorkspaceArchive } from "./containers/transfer";
7
7
  import { removeCredentials } from "./opencode/auth";
8
8
  import { readArchivedSession } from "./opencode/archive";
9
9
  import { parseModel, type SessionResult } from "./opencode/session";
10
10
  import { hash, writeJson } from "./files";
11
+ import { CandidateContainer } from "./containers/oci";
12
+ import { createSessionDatabase } from "./opencode/host";
11
13
 
12
14
  export type ArchiveRestore = {
13
15
  database: string;
@@ -16,6 +18,26 @@ export type ArchiveRestore = {
16
18
  workspaceArchiveHash: string;
17
19
  };
18
20
 
21
+ /** Replay only frozen author preparation, without credentials or a model prompt. */
22
+ export async function prepareRestoredInitial(
23
+ definition: BenchmarkDefinition,
24
+ item: EvalDefinition,
25
+ imageId: string,
26
+ workspace: string,
27
+ staging: string,
28
+ ) {
29
+ if (!item.settings.prepare?.length) return workspace;
30
+ await mkdir(staging, { recursive: true });
31
+ const database = resolve(staging, "opencode.db");
32
+ await createSessionDatabase(database, []);
33
+ await using container = await CandidateContainer.create(definition.container, imageId);
34
+ await container.prepare(workspace, database, definition.candidate.websearch,
35
+ item.settings.prepare, staging);
36
+ const initial = resolve(staging, "workspace");
37
+ await container.snapshot(initial, staging);
38
+ return initial;
39
+ }
40
+
19
41
  /** A backup must belong to the original execution, not merely resemble its answer. */
20
42
  export function verifyArchivedInput(
21
43
  input: EvalRunInput,
@@ -88,7 +110,7 @@ export async function restoreArchivedCandidate(
88
110
  completedAt: new Date(completedAt).toISOString(),
89
111
  databaseHash: options.databaseHash,
90
112
  workspaceArchiveHash: options.workspaceArchiveHash,
91
- initial: "frozen prepared input, not a newly observed candidate snapshot",
113
+ initial: "frozen input and author preparation replayed without model calls; not a newly observed candidate snapshot",
92
114
  };
93
115
  await writeJson(resolve(directory, "restoration.json"), receipt);
94
116
  return {