@hona/openeval 0.5.10 → 0.5.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.10",
3
+ "version": "0.5.11",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -0,0 +1,47 @@
1
+ import type { BenchmarkDefinition, BenchmarkRun, PlanItem } from "../types";
2
+ import type { Results } from "../infra/sqlite";
3
+ import { candidateFingerprint, judgeFingerprint } from "./input-fingerprints";
4
+
5
+ /** Explicit IDs only: do not recollect fully scored or superseded recordings. */
6
+ export function retryPlan(
7
+ definition: BenchmarkDefinition,
8
+ runtime: BenchmarkRun["runtime"],
9
+ results: Results,
10
+ evalRunIds: readonly string[],
11
+ ): PlanItem[] {
12
+ if (!evalRunIds.length || new Set(evalRunIds).size !== evalRunIds.length)
13
+ throw new Error("Select unique EvalRun IDs to retry");
14
+ return evalRunIds.map(id => {
15
+ const previous = results.evalRun(id);
16
+ const slot = previous && results.slot(previous.slotId);
17
+ if (!previous || !slot?.active || slot.evalRunId !== previous.id)
18
+ throw new Error("Select the active EvalRun for each repetition");
19
+ const judge = slot.judgeRunId && results.judgeRun(slot.judgeRunId);
20
+ const unresolved = judge && judge.state === "completed" &&
21
+ Object.values(judge.judgment?.scores ?? {}).some(score => score.value === null);
22
+ if (previous.state === "running" || (previous.state === "completed" && !unresolved))
23
+ throw new Error("Select failed, stopped, timed-out, or completed-but-unresolved work");
24
+ const item = definition.evals.find(item => item.id === slot.evalId);
25
+ if (!item || item.prompt !== previous.input.prompt || item.sourceHash !== previous.input.sourceHash ||
26
+ runtime.imageId !== previous.input.imageId)
27
+ throw new Error("Task inputs or image changed; use scoped run for changed work");
28
+ // The current authored time limit may replace an older limit. Other inputs must match.
29
+ const recordedLimits = {
30
+ ...definition,
31
+ evals: definition.evals.map(value => value.id === item.id ? {
32
+ ...value, settings: { ...value.settings,
33
+ candidate: { ...value.settings.candidate, timeoutMs: previous.input.timeoutMs } },
34
+ } : value),
35
+ };
36
+ if (candidateFingerprint(recordedLimits, slot.evalId, slot.model, runtime) !== previous.input.candidateHash)
37
+ throw new Error("Candidate inputs other than the time limit changed");
38
+ return {
39
+ action: "candidate",
40
+ reason: previous.state === "completed" ? "Explicit retry of unresolved recorded work" : "Explicit retry of interrupted or failed work",
41
+ slot: { ...slot,
42
+ candidateHash: candidateFingerprint(definition, slot.evalId, slot.model, runtime),
43
+ judgeHash: judgeFingerprint(definition, slot.evalId),
44
+ },
45
+ };
46
+ });
47
+ }
@@ -1,18 +1,40 @@
1
1
  import { resolve } from "node:path";
2
- import type { ExecutionObserver } from "../types";
2
+ import type { EvalRun, ExecutionObserver } from "../types";
3
3
  import { Results } from "../infra/sqlite";
4
4
  import { runtimeFingerprint } from "../infra/containers/oci";
5
- import { candidateFingerprint } from "./plan-benchmark";
6
5
  import { runEvalPipeline } from "./run-eval-pipeline";
7
6
  import { ExecutionBudget } from "./execution-budget";
8
- import { finishBenchmark } from "./run-benchmark";
7
+ import { executeQueue, finishBenchmark } from "./run-benchmark";
9
8
  import { recoverStoppedRunner, runnerStopped } from "./recover-run";
9
+ import { retryPlan } from "./plan-retries";
10
+ import { CostBudget, estimateWork } from "./cost-plan";
10
11
 
11
12
  export async function retryEvalRun(
12
13
  directory: string,
13
14
  evalRunId: string,
14
15
  onEvent?: ExecutionObserver,
15
16
  ) {
17
+ return (await retryEvalRuns(directory, [evalRunId], onEvent))[0]!;
18
+ }
19
+
20
+ /** Read-only preflight for a precise retry scope. No model calls or selection updates. */
21
+ export async function planEvalRunRetries(directory: string, evalRunIds: readonly string[]) {
22
+ using results = new Results(resolve(directory, "runner.db"), true);
23
+ const benchmark = results.benchmark!;
24
+ if (benchmark.mergedInto || benchmark.state === "running")
25
+ throw new Error("Wait for active work and select the aggregate run");
26
+ const runtime = await runtimeFingerprint(benchmark.definition.container, benchmark.definition.judge.verification);
27
+ const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
28
+ return { plan, estimate: estimateWork(benchmark.definition, results, plan) };
29
+ }
30
+
31
+ /** A single retry round; original evidence and judgments remain immutable. */
32
+ export async function retryEvalRuns(
33
+ directory: string,
34
+ evalRunIds: readonly string[],
35
+ onEvent?: ExecutionObserver,
36
+ options: { maxCostUSD?: number } = {},
37
+ ): Promise<EvalRun[]> {
16
38
  using results = new Results(resolve(directory, "runner.db"));
17
39
  if (results.benchmark!.mergedInto)
18
40
  throw new Error(
@@ -24,41 +46,26 @@ export async function retryEvalRun(
24
46
  recoverStoppedRunner(results);
25
47
  }
26
48
  const benchmark = results.benchmark!;
27
- const previous = results.evalRun(evalRunId);
28
- if (!previous || previous.state === "completed")
29
- throw new Error(
30
- "Select a failed eval run; completed evidence can be rejudged",
31
- );
32
- const slot = results.slot(previous.slotId)!;
33
- if (slot.evalRunId !== previous.id)
34
- throw new Error("Select the current eval run for this repetition");
35
49
  const runtime = await runtimeFingerprint(
36
50
  benchmark.definition.container,
37
51
  benchmark.definition.judge.verification,
38
52
  );
39
- if (
40
- candidateFingerprint(
41
- benchmark.definition,
42
- slot.evalId,
43
- slot.model,
44
- runtime,
45
- ) !== slot.candidateHash
46
- )
47
- throw new Error(
48
- "Declared candidate inputs changed; use run to schedule the affected work",
49
- );
50
- const definition = benchmark.definition.evals.find(
51
- (evalDefinition) => evalDefinition.id === slot.evalId,
52
- )!;
53
- const workspace = resolve(
54
- directory,
55
- "inputs",
56
- definition.id,
57
- definition.sourceHash,
58
- "workspace",
59
- );
60
- if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
61
- throw new Error("The saved starting workspace is missing");
53
+ const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
54
+ const workspaces = new Map<string, string>();
55
+ for (const item of plan) {
56
+ const definition = benchmark.definition.evals.find(value => value.id === item.slot.evalId)!;
57
+ const workspace = resolve(directory, "inputs", definition.id, definition.sourceHash, "workspace");
58
+ if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
59
+ throw new Error("The saved starting workspace is missing");
60
+ workspaces.set(item.slot.evalId, workspace);
61
+ }
62
+ const estimate = estimateWork(benchmark.definition, results, plan);
63
+ const existing = new Set([...results.evalRuns(), ...results.judgeRuns()].map(run => run.id));
64
+ const spent = () => [...results.evalRuns(), ...results.judgeRuns()]
65
+ .filter(run => !existing.has(run.id))
66
+ .reduce((sum, run) => sum + (run.session?.accounting?.costUSD ?? results.recordedCost(run.id) ?? 0), 0);
67
+ const costBudget = new CostBudget(options.maxCostUSD, spent);
68
+ const estimates = new Map(estimate.items.map(item => [item.slotId, item.estimatedUSD]));
62
69
  const context = {
63
70
  directory: resolve(directory),
64
71
  definition: benchmark.definition,
@@ -66,29 +73,58 @@ export async function retryEvalRun(
66
73
  results,
67
74
  onEvent,
68
75
  };
69
- results.saveBenchmark({
70
- ...benchmark,
71
- state: "running",
72
- scheduledSlotIds: [slot.id],
73
- updatedAt: new Date().toISOString(),
76
+ results.transaction(() => {
77
+ if (results.benchmark!.state === "running") throw new Error("Another operation claimed this run");
78
+ for (const item of plan) {
79
+ const selected = results.slot(item.slot.id);
80
+ if (!selected?.active || selected.evalRunId !== item.slot.evalRunId ||
81
+ selected.judgeRunId !== item.slot.judgeRunId)
82
+ throw new Error("The retry selection changed during preflight");
83
+ }
84
+ results.saveBenchmark({ ...benchmark, runtime, state: "running",
85
+ scheduledSlotIds: plan.map(item => item.slot.id), updatedAt: new Date().toISOString(),
86
+ execution: { startedAt: new Date().toISOString(),
87
+ onlyModels: [...new Set(plan.map(item => item.slot.model))],
88
+ onlyEvals: [...new Set(plan.map(item => item.slot.evalId))],
89
+ onlyRepetitions: [...new Set(plan.map(item => item.slot.repetition))],
90
+ budgetUSD: options.maxCostUSD, estimatedUSD: estimate.estimatedUSD,
91
+ spentUSD: 0, deferred: 0 } });
74
92
  });
75
93
  const timer = setInterval(
76
94
  () =>
77
95
  results.saveBenchmark({
78
96
  ...results.benchmark!,
79
97
  updatedAt: new Date().toISOString(),
98
+ scheduledSlotIds: plan.filter(item => item.action !== "deferred").map(item => item.slot.id),
99
+ execution: { ...results.benchmark!.execution, spentUSD: spent(),
100
+ deferred: plan.filter(item => item.action === "deferred").length },
80
101
  }),
81
102
  5000,
82
103
  );
83
104
  try {
84
- return await runEvalPipeline(
85
- context,
86
- slot,
87
- workspace,
88
- new ExecutionBudget(benchmark.definition.concurrency),
89
- );
105
+ const budget = new ExecutionBudget(benchmark.definition.concurrency);
106
+ const completed = new Map<string, EvalRun>();
107
+ const errors: unknown[] = [];
108
+ await executeQueue(plan, benchmark.definition.concurrency, async item => {
109
+ const release = costBudget.reserve(estimates.get(item.slot.id) ?? null);
110
+ if (!release) {
111
+ item.action = "deferred";
112
+ item.reason = "Retry scheduling budget reached or no estimate is available";
113
+ return;
114
+ }
115
+ try {
116
+ const run = await runEvalPipeline(context, item.slot, workspaces.get(item.slot.evalId)!, budget);
117
+ completed.set(item.slot.id, run);
118
+ } catch (error) {
119
+ errors.push(error);
120
+ } finally { release(); }
121
+ });
122
+ if (errors.length) throw new AggregateError(errors, "One or more retry operations failed");
123
+ return plan.flatMap(item => completed.has(item.slot.id) ? [completed.get(item.slot.id)!] : []);
90
124
  } finally {
91
125
  clearInterval(timer);
126
+ results.saveBenchmark({ ...results.benchmark!, execution: { ...results.benchmark!.execution,
127
+ spentUSD: spent(), deferred: plan.filter(item => item.action === "deferred").length } });
92
128
  finishBenchmark(context);
93
129
  }
94
130
  }
package/src/cli.ts CHANGED
@@ -7,7 +7,8 @@ import {
7
7
  buildImage,
8
8
  addModels,
9
9
  removeModels,
10
- retryEvalRun,
10
+ retryEvalRuns,
11
+ planEvalRunRetries,
11
12
  restoreEvalRun,
12
13
  judgeRun,
13
14
  serveResults,
@@ -48,7 +49,7 @@ Commands:
48
49
  add-models <run> Add models with repeated --model flags
49
50
  remove-models <run> Remove active models while retaining their evidence
50
51
  refresh-model-names <run> Refresh reporting names without running evals
51
- retry <run> <eval-run> Retry a candidate execution
52
+ retry <run> <eval-run>... Retry failed or unresolved candidate work
52
53
  rejudge <run> <eval-run> Judge the saved evidence again
53
54
  restore <run> <eval-run> Restore completed orphan work from hashed backups
54
55
 
@@ -63,6 +64,7 @@ Options:
63
64
  --only-repetition <n> Execute only this repetition (repeatable)
64
65
  --max-cost <usd> Scheduling budget for this invocation
65
66
  --final-only Judge only after candidates finish
67
+ --dry-run Plan explicit retries without running them
66
68
  --verification Build only the standard verification image (image command)
67
69
  --output <dir> New prepared-input directory (prepare command)
68
70
  --port <port> Viewer port (default: 4173)
@@ -184,15 +186,22 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
184
186
  console.log(JSON.stringify(await restoreEvalRun(resolve(args[1]), args[2], {
185
187
  database: resolve(database), databaseHash, workspaceArchive: resolve(workspaceArchive), workspaceArchiveHash,
186
188
  }), null, 2));
187
- } else if (command === "retry" || command === "rejudge") {
189
+ } else if (command === "retry") {
190
+ const flags = args.findIndex((arg, index) => index >= 2 && arg.startsWith("--"));
191
+ const ids = args.slice(2, flags < 0 ? args.length : flags);
192
+ if (!args[1] || !ids.length) throw new Error("retry requires a run directory and explicit EvalRun IDs");
193
+ const result = args.includes("--dry-run")
194
+ ? await planEvalRunRetries(resolve(args[1]), ids)
195
+ : await retryEvalRuns(resolve(args[1]), ids, undefined,
196
+ { maxCostUSD: option("--max-cost") === undefined ? undefined : Number(option("--max-cost")) });
197
+ console.log(JSON.stringify(!args.includes("--dry-run") && ids.length === 1 && Array.isArray(result)
198
+ ? result[0] : result, null, 2));
199
+ } else if (command === "rejudge") {
188
200
  if (!args[1] || !args[2])
189
201
  throw new Error(
190
202
  `${command} requires a benchmark run directory and eval run ID`,
191
203
  );
192
- const result =
193
- command === "retry"
194
- ? await retryEvalRun(resolve(args[1]), args[2])
195
- : await judgeRun(resolve(args[1]), args[2]);
204
+ const result = await judgeRun(resolve(args[1]), args[2]);
196
205
  console.log(JSON.stringify(result, null, 2));
197
206
  } else
198
207
  throw new Error(
package/src/index.ts CHANGED
@@ -67,7 +67,7 @@ export { removeModels } from "./app/remove-models";
67
67
  export { refreshModelNames } from "./app/refresh-model-names";
68
68
  export type { CostEstimate } from "./app/cost-plan";
69
69
  export { mergeBenchmarkRuns } from "./app/merge-runs";
70
- export { retryEvalRun } from "./app/retry-run";
70
+ export { retryEvalRun, retryEvalRuns, planEvalRunRetries } from "./app/retry-run";
71
71
  export { restoreEvalRun } from "./app/restore-eval-run";
72
72
  export { judgeRun, judgeRuns } from "./app/rejudge";
73
73
  export { judgeEvidence, recordEvidence } from "./app/judge-evidence";