@hona/openeval 0.5.10 → 0.5.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app/plan-retries.ts +47 -0
- package/src/app/retry-run.ts +81 -45
- package/src/cli.ts +16 -7
- package/src/index.ts +1 -1
package/package.json
CHANGED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import type { BenchmarkDefinition, BenchmarkRun, PlanItem } from "../types";
|
|
2
|
+
import type { Results } from "../infra/sqlite";
|
|
3
|
+
import { candidateFingerprint, judgeFingerprint } from "./input-fingerprints";
|
|
4
|
+
|
|
5
|
+
/** Explicit IDs only: do not recollect fully scored or superseded recordings. */
|
|
6
|
+
export function retryPlan(
|
|
7
|
+
definition: BenchmarkDefinition,
|
|
8
|
+
runtime: BenchmarkRun["runtime"],
|
|
9
|
+
results: Results,
|
|
10
|
+
evalRunIds: readonly string[],
|
|
11
|
+
): PlanItem[] {
|
|
12
|
+
if (!evalRunIds.length || new Set(evalRunIds).size !== evalRunIds.length)
|
|
13
|
+
throw new Error("Select unique EvalRun IDs to retry");
|
|
14
|
+
return evalRunIds.map(id => {
|
|
15
|
+
const previous = results.evalRun(id);
|
|
16
|
+
const slot = previous && results.slot(previous.slotId);
|
|
17
|
+
if (!previous || !slot?.active || slot.evalRunId !== previous.id)
|
|
18
|
+
throw new Error("Select the active EvalRun for each repetition");
|
|
19
|
+
const judge = slot.judgeRunId && results.judgeRun(slot.judgeRunId);
|
|
20
|
+
const unresolved = judge && judge.state === "completed" &&
|
|
21
|
+
Object.values(judge.judgment?.scores ?? {}).some(score => score.value === null);
|
|
22
|
+
if (previous.state === "running" || (previous.state === "completed" && !unresolved))
|
|
23
|
+
throw new Error("Select failed, stopped, timed-out, or completed-but-unresolved work");
|
|
24
|
+
const item = definition.evals.find(item => item.id === slot.evalId);
|
|
25
|
+
if (!item || item.prompt !== previous.input.prompt || item.sourceHash !== previous.input.sourceHash ||
|
|
26
|
+
runtime.imageId !== previous.input.imageId)
|
|
27
|
+
throw new Error("Task inputs or image changed; use scoped run for changed work");
|
|
28
|
+
// The current authored time limit may replace an older limit. Other inputs must match.
|
|
29
|
+
const recordedLimits = {
|
|
30
|
+
...definition,
|
|
31
|
+
evals: definition.evals.map(value => value.id === item.id ? {
|
|
32
|
+
...value, settings: { ...value.settings,
|
|
33
|
+
candidate: { ...value.settings.candidate, timeoutMs: previous.input.timeoutMs } },
|
|
34
|
+
} : value),
|
|
35
|
+
};
|
|
36
|
+
if (candidateFingerprint(recordedLimits, slot.evalId, slot.model, runtime) !== previous.input.candidateHash)
|
|
37
|
+
throw new Error("Candidate inputs other than the time limit changed");
|
|
38
|
+
return {
|
|
39
|
+
action: "candidate",
|
|
40
|
+
reason: previous.state === "completed" ? "Explicit retry of unresolved recorded work" : "Explicit retry of interrupted or failed work",
|
|
41
|
+
slot: { ...slot,
|
|
42
|
+
candidateHash: candidateFingerprint(definition, slot.evalId, slot.model, runtime),
|
|
43
|
+
judgeHash: judgeFingerprint(definition, slot.evalId),
|
|
44
|
+
},
|
|
45
|
+
};
|
|
46
|
+
});
|
|
47
|
+
}
|
package/src/app/retry-run.ts
CHANGED
|
@@ -1,18 +1,40 @@
|
|
|
1
1
|
import { resolve } from "node:path";
|
|
2
|
-
import type { ExecutionObserver } from "../types";
|
|
2
|
+
import type { EvalRun, ExecutionObserver } from "../types";
|
|
3
3
|
import { Results } from "../infra/sqlite";
|
|
4
4
|
import { runtimeFingerprint } from "../infra/containers/oci";
|
|
5
|
-
import { candidateFingerprint } from "./plan-benchmark";
|
|
6
5
|
import { runEvalPipeline } from "./run-eval-pipeline";
|
|
7
6
|
import { ExecutionBudget } from "./execution-budget";
|
|
8
|
-
import { finishBenchmark } from "./run-benchmark";
|
|
7
|
+
import { executeQueue, finishBenchmark } from "./run-benchmark";
|
|
9
8
|
import { recoverStoppedRunner, runnerStopped } from "./recover-run";
|
|
9
|
+
import { retryPlan } from "./plan-retries";
|
|
10
|
+
import { CostBudget, estimateWork } from "./cost-plan";
|
|
10
11
|
|
|
11
12
|
export async function retryEvalRun(
|
|
12
13
|
directory: string,
|
|
13
14
|
evalRunId: string,
|
|
14
15
|
onEvent?: ExecutionObserver,
|
|
15
16
|
) {
|
|
17
|
+
return (await retryEvalRuns(directory, [evalRunId], onEvent))[0]!;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** Read-only preflight for a precise retry scope. No model calls or selection updates. */
|
|
21
|
+
export async function planEvalRunRetries(directory: string, evalRunIds: readonly string[]) {
|
|
22
|
+
using results = new Results(resolve(directory, "runner.db"), true);
|
|
23
|
+
const benchmark = results.benchmark!;
|
|
24
|
+
if (benchmark.mergedInto || benchmark.state === "running")
|
|
25
|
+
throw new Error("Wait for active work and select the aggregate run");
|
|
26
|
+
const runtime = await runtimeFingerprint(benchmark.definition.container, benchmark.definition.judge.verification);
|
|
27
|
+
const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
|
|
28
|
+
return { plan, estimate: estimateWork(benchmark.definition, results, plan) };
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** A single retry round; original evidence and judgments remain immutable. */
|
|
32
|
+
export async function retryEvalRuns(
|
|
33
|
+
directory: string,
|
|
34
|
+
evalRunIds: readonly string[],
|
|
35
|
+
onEvent?: ExecutionObserver,
|
|
36
|
+
options: { maxCostUSD?: number } = {},
|
|
37
|
+
): Promise<EvalRun[]> {
|
|
16
38
|
using results = new Results(resolve(directory, "runner.db"));
|
|
17
39
|
if (results.benchmark!.mergedInto)
|
|
18
40
|
throw new Error(
|
|
@@ -24,41 +46,26 @@ export async function retryEvalRun(
|
|
|
24
46
|
recoverStoppedRunner(results);
|
|
25
47
|
}
|
|
26
48
|
const benchmark = results.benchmark!;
|
|
27
|
-
const previous = results.evalRun(evalRunId);
|
|
28
|
-
if (!previous || previous.state === "completed")
|
|
29
|
-
throw new Error(
|
|
30
|
-
"Select a failed eval run; completed evidence can be rejudged",
|
|
31
|
-
);
|
|
32
|
-
const slot = results.slot(previous.slotId)!;
|
|
33
|
-
if (slot.evalRunId !== previous.id)
|
|
34
|
-
throw new Error("Select the current eval run for this repetition");
|
|
35
49
|
const runtime = await runtimeFingerprint(
|
|
36
50
|
benchmark.definition.container,
|
|
37
51
|
benchmark.definition.judge.verification,
|
|
38
52
|
);
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
const
|
|
51
|
-
(
|
|
52
|
-
|
|
53
|
-
const
|
|
54
|
-
|
|
55
|
-
"inputs",
|
|
56
|
-
definition.id,
|
|
57
|
-
definition.sourceHash,
|
|
58
|
-
"workspace",
|
|
59
|
-
);
|
|
60
|
-
if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
|
|
61
|
-
throw new Error("The saved starting workspace is missing");
|
|
53
|
+
const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
|
|
54
|
+
const workspaces = new Map<string, string>();
|
|
55
|
+
for (const item of plan) {
|
|
56
|
+
const definition = benchmark.definition.evals.find(value => value.id === item.slot.evalId)!;
|
|
57
|
+
const workspace = resolve(directory, "inputs", definition.id, definition.sourceHash, "workspace");
|
|
58
|
+
if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
|
|
59
|
+
throw new Error("The saved starting workspace is missing");
|
|
60
|
+
workspaces.set(item.slot.evalId, workspace);
|
|
61
|
+
}
|
|
62
|
+
const estimate = estimateWork(benchmark.definition, results, plan);
|
|
63
|
+
const existing = new Set([...results.evalRuns(), ...results.judgeRuns()].map(run => run.id));
|
|
64
|
+
const spent = () => [...results.evalRuns(), ...results.judgeRuns()]
|
|
65
|
+
.filter(run => !existing.has(run.id))
|
|
66
|
+
.reduce((sum, run) => sum + (run.session?.accounting?.costUSD ?? results.recordedCost(run.id) ?? 0), 0);
|
|
67
|
+
const costBudget = new CostBudget(options.maxCostUSD, spent);
|
|
68
|
+
const estimates = new Map(estimate.items.map(item => [item.slotId, item.estimatedUSD]));
|
|
62
69
|
const context = {
|
|
63
70
|
directory: resolve(directory),
|
|
64
71
|
definition: benchmark.definition,
|
|
@@ -66,29 +73,58 @@ export async function retryEvalRun(
|
|
|
66
73
|
results,
|
|
67
74
|
onEvent,
|
|
68
75
|
};
|
|
69
|
-
results.
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
76
|
+
results.transaction(() => {
|
|
77
|
+
if (results.benchmark!.state === "running") throw new Error("Another operation claimed this run");
|
|
78
|
+
for (const item of plan) {
|
|
79
|
+
const selected = results.slot(item.slot.id);
|
|
80
|
+
if (!selected?.active || selected.evalRunId !== item.slot.evalRunId ||
|
|
81
|
+
selected.judgeRunId !== item.slot.judgeRunId)
|
|
82
|
+
throw new Error("The retry selection changed during preflight");
|
|
83
|
+
}
|
|
84
|
+
results.saveBenchmark({ ...benchmark, runtime, state: "running",
|
|
85
|
+
scheduledSlotIds: plan.map(item => item.slot.id), updatedAt: new Date().toISOString(),
|
|
86
|
+
execution: { startedAt: new Date().toISOString(),
|
|
87
|
+
onlyModels: [...new Set(plan.map(item => item.slot.model))],
|
|
88
|
+
onlyEvals: [...new Set(plan.map(item => item.slot.evalId))],
|
|
89
|
+
onlyRepetitions: [...new Set(plan.map(item => item.slot.repetition))],
|
|
90
|
+
budgetUSD: options.maxCostUSD, estimatedUSD: estimate.estimatedUSD,
|
|
91
|
+
spentUSD: 0, deferred: 0 } });
|
|
74
92
|
});
|
|
75
93
|
const timer = setInterval(
|
|
76
94
|
() =>
|
|
77
95
|
results.saveBenchmark({
|
|
78
96
|
...results.benchmark!,
|
|
79
97
|
updatedAt: new Date().toISOString(),
|
|
98
|
+
scheduledSlotIds: plan.filter(item => item.action !== "deferred").map(item => item.slot.id),
|
|
99
|
+
execution: { ...results.benchmark!.execution, spentUSD: spent(),
|
|
100
|
+
deferred: plan.filter(item => item.action === "deferred").length },
|
|
80
101
|
}),
|
|
81
102
|
5000,
|
|
82
103
|
);
|
|
83
104
|
try {
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
105
|
+
const budget = new ExecutionBudget(benchmark.definition.concurrency);
|
|
106
|
+
const completed = new Map<string, EvalRun>();
|
|
107
|
+
const errors: unknown[] = [];
|
|
108
|
+
await executeQueue(plan, benchmark.definition.concurrency, async item => {
|
|
109
|
+
const release = costBudget.reserve(estimates.get(item.slot.id) ?? null);
|
|
110
|
+
if (!release) {
|
|
111
|
+
item.action = "deferred";
|
|
112
|
+
item.reason = "Retry scheduling budget reached or no estimate is available";
|
|
113
|
+
return;
|
|
114
|
+
}
|
|
115
|
+
try {
|
|
116
|
+
const run = await runEvalPipeline(context, item.slot, workspaces.get(item.slot.evalId)!, budget);
|
|
117
|
+
completed.set(item.slot.id, run);
|
|
118
|
+
} catch (error) {
|
|
119
|
+
errors.push(error);
|
|
120
|
+
} finally { release(); }
|
|
121
|
+
});
|
|
122
|
+
if (errors.length) throw new AggregateError(errors, "One or more retry operations failed");
|
|
123
|
+
return plan.flatMap(item => completed.has(item.slot.id) ? [completed.get(item.slot.id)!] : []);
|
|
90
124
|
} finally {
|
|
91
125
|
clearInterval(timer);
|
|
126
|
+
results.saveBenchmark({ ...results.benchmark!, execution: { ...results.benchmark!.execution,
|
|
127
|
+
spentUSD: spent(), deferred: plan.filter(item => item.action === "deferred").length } });
|
|
92
128
|
finishBenchmark(context);
|
|
93
129
|
}
|
|
94
130
|
}
|
package/src/cli.ts
CHANGED
|
@@ -7,7 +7,8 @@ import {
|
|
|
7
7
|
buildImage,
|
|
8
8
|
addModels,
|
|
9
9
|
removeModels,
|
|
10
|
-
|
|
10
|
+
retryEvalRuns,
|
|
11
|
+
planEvalRunRetries,
|
|
11
12
|
restoreEvalRun,
|
|
12
13
|
judgeRun,
|
|
13
14
|
serveResults,
|
|
@@ -48,7 +49,7 @@ Commands:
|
|
|
48
49
|
add-models <run> Add models with repeated --model flags
|
|
49
50
|
remove-models <run> Remove active models while retaining their evidence
|
|
50
51
|
refresh-model-names <run> Refresh reporting names without running evals
|
|
51
|
-
retry <run> <eval-run
|
|
52
|
+
retry <run> <eval-run>... Retry failed or unresolved candidate work
|
|
52
53
|
rejudge <run> <eval-run> Judge the saved evidence again
|
|
53
54
|
restore <run> <eval-run> Restore completed orphan work from hashed backups
|
|
54
55
|
|
|
@@ -63,6 +64,7 @@ Options:
|
|
|
63
64
|
--only-repetition <n> Execute only this repetition (repeatable)
|
|
64
65
|
--max-cost <usd> Scheduling budget for this invocation
|
|
65
66
|
--final-only Judge only after candidates finish
|
|
67
|
+
--dry-run Plan explicit retries without running them
|
|
66
68
|
--verification Build only the standard verification image (image command)
|
|
67
69
|
--output <dir> New prepared-input directory (prepare command)
|
|
68
70
|
--port <port> Viewer port (default: 4173)
|
|
@@ -184,15 +186,22 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
184
186
|
console.log(JSON.stringify(await restoreEvalRun(resolve(args[1]), args[2], {
|
|
185
187
|
database: resolve(database), databaseHash, workspaceArchive: resolve(workspaceArchive), workspaceArchiveHash,
|
|
186
188
|
}), null, 2));
|
|
187
|
-
} else if (command === "retry"
|
|
189
|
+
} else if (command === "retry") {
|
|
190
|
+
const flags = args.findIndex((arg, index) => index >= 2 && arg.startsWith("--"));
|
|
191
|
+
const ids = args.slice(2, flags < 0 ? args.length : flags);
|
|
192
|
+
if (!args[1] || !ids.length) throw new Error("retry requires a run directory and explicit EvalRun IDs");
|
|
193
|
+
const result = args.includes("--dry-run")
|
|
194
|
+
? await planEvalRunRetries(resolve(args[1]), ids)
|
|
195
|
+
: await retryEvalRuns(resolve(args[1]), ids, undefined,
|
|
196
|
+
{ maxCostUSD: option("--max-cost") === undefined ? undefined : Number(option("--max-cost")) });
|
|
197
|
+
console.log(JSON.stringify(!args.includes("--dry-run") && ids.length === 1 && Array.isArray(result)
|
|
198
|
+
? result[0] : result, null, 2));
|
|
199
|
+
} else if (command === "rejudge") {
|
|
188
200
|
if (!args[1] || !args[2])
|
|
189
201
|
throw new Error(
|
|
190
202
|
`${command} requires a benchmark run directory and eval run ID`,
|
|
191
203
|
);
|
|
192
|
-
const result =
|
|
193
|
-
command === "retry"
|
|
194
|
-
? await retryEvalRun(resolve(args[1]), args[2])
|
|
195
|
-
: await judgeRun(resolve(args[1]), args[2]);
|
|
204
|
+
const result = await judgeRun(resolve(args[1]), args[2]);
|
|
196
205
|
console.log(JSON.stringify(result, null, 2));
|
|
197
206
|
} else
|
|
198
207
|
throw new Error(
|
package/src/index.ts
CHANGED
|
@@ -67,7 +67,7 @@ export { removeModels } from "./app/remove-models";
|
|
|
67
67
|
export { refreshModelNames } from "./app/refresh-model-names";
|
|
68
68
|
export type { CostEstimate } from "./app/cost-plan";
|
|
69
69
|
export { mergeBenchmarkRuns } from "./app/merge-runs";
|
|
70
|
-
export { retryEvalRun } from "./app/retry-run";
|
|
70
|
+
export { retryEvalRun, retryEvalRuns, planEvalRunRetries } from "./app/retry-run";
|
|
71
71
|
export { restoreEvalRun } from "./app/restore-eval-run";
|
|
72
72
|
export { judgeRun, judgeRuns } from "./app/rejudge";
|
|
73
73
|
export { judgeEvidence, recordEvidence } from "./app/judge-evidence";
|