@hona/openeval 0.5.9 → 0.5.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app/plan-retries.ts +47 -0
- package/src/app/read-results.ts +9 -2
- package/src/app/restore-eval-run.ts +25 -10
- package/src/app/retry-run.ts +81 -45
- package/src/cli.ts +16 -7
- package/src/index.ts +1 -1
- package/src/infra/restore-archive.ts +24 -2
package/package.json
CHANGED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import type { BenchmarkDefinition, BenchmarkRun, PlanItem } from "../types";
|
|
2
|
+
import type { Results } from "../infra/sqlite";
|
|
3
|
+
import { candidateFingerprint, judgeFingerprint } from "./input-fingerprints";
|
|
4
|
+
|
|
5
|
+
/** Explicit IDs only: do not recollect fully scored or superseded recordings. */
|
|
6
|
+
export function retryPlan(
|
|
7
|
+
definition: BenchmarkDefinition,
|
|
8
|
+
runtime: BenchmarkRun["runtime"],
|
|
9
|
+
results: Results,
|
|
10
|
+
evalRunIds: readonly string[],
|
|
11
|
+
): PlanItem[] {
|
|
12
|
+
if (!evalRunIds.length || new Set(evalRunIds).size !== evalRunIds.length)
|
|
13
|
+
throw new Error("Select unique EvalRun IDs to retry");
|
|
14
|
+
return evalRunIds.map(id => {
|
|
15
|
+
const previous = results.evalRun(id);
|
|
16
|
+
const slot = previous && results.slot(previous.slotId);
|
|
17
|
+
if (!previous || !slot?.active || slot.evalRunId !== previous.id)
|
|
18
|
+
throw new Error("Select the active EvalRun for each repetition");
|
|
19
|
+
const judge = slot.judgeRunId && results.judgeRun(slot.judgeRunId);
|
|
20
|
+
const unresolved = judge && judge.state === "completed" &&
|
|
21
|
+
Object.values(judge.judgment?.scores ?? {}).some(score => score.value === null);
|
|
22
|
+
if (previous.state === "running" || (previous.state === "completed" && !unresolved))
|
|
23
|
+
throw new Error("Select failed, stopped, timed-out, or completed-but-unresolved work");
|
|
24
|
+
const item = definition.evals.find(item => item.id === slot.evalId);
|
|
25
|
+
if (!item || item.prompt !== previous.input.prompt || item.sourceHash !== previous.input.sourceHash ||
|
|
26
|
+
runtime.imageId !== previous.input.imageId)
|
|
27
|
+
throw new Error("Task inputs or image changed; use scoped run for changed work");
|
|
28
|
+
// The current authored time limit may replace an older limit. Other inputs must match.
|
|
29
|
+
const recordedLimits = {
|
|
30
|
+
...definition,
|
|
31
|
+
evals: definition.evals.map(value => value.id === item.id ? {
|
|
32
|
+
...value, settings: { ...value.settings,
|
|
33
|
+
candidate: { ...value.settings.candidate, timeoutMs: previous.input.timeoutMs } },
|
|
34
|
+
} : value),
|
|
35
|
+
};
|
|
36
|
+
if (candidateFingerprint(recordedLimits, slot.evalId, slot.model, runtime) !== previous.input.candidateHash)
|
|
37
|
+
throw new Error("Candidate inputs other than the time limit changed");
|
|
38
|
+
return {
|
|
39
|
+
action: "candidate",
|
|
40
|
+
reason: previous.state === "completed" ? "Explicit retry of unresolved recorded work" : "Explicit retry of interrupted or failed work",
|
|
41
|
+
slot: { ...slot,
|
|
42
|
+
candidateHash: candidateFingerprint(definition, slot.evalId, slot.model, runtime),
|
|
43
|
+
judgeHash: judgeFingerprint(definition, slot.evalId),
|
|
44
|
+
},
|
|
45
|
+
};
|
|
46
|
+
});
|
|
47
|
+
}
|
package/src/app/read-results.ts
CHANGED
|
@@ -43,10 +43,17 @@ const costOf = (run: EvalRun | JudgeRun, results?: Results): Cost => {
|
|
|
43
43
|
complete: usd !== undefined,
|
|
44
44
|
};
|
|
45
45
|
};
|
|
46
|
-
//
|
|
46
|
+
// Restored records represent the same original execution, not additional model spend.
|
|
47
47
|
const runCosts = (runs: (EvalRun | JudgeRun)[], results: Results) => {
|
|
48
48
|
const unique = new Map(runs.map((run) => [run.id, run]));
|
|
49
|
-
|
|
49
|
+
const executions = new Map<string, EvalRun | JudgeRun>();
|
|
50
|
+
for (const run of unique.values()) {
|
|
51
|
+
const restoration = "restoration" in run ? run.restoration : undefined;
|
|
52
|
+
const key = restoration?.evalRunId ?? run.id;
|
|
53
|
+
const previous = executions.get(key);
|
|
54
|
+
if (!previous || restoration) executions.set(key, run);
|
|
55
|
+
}
|
|
56
|
+
return sumCosts([...executions.values()].map((run) => costOf(run, results)));
|
|
50
57
|
};
|
|
51
58
|
const entryOf = (
|
|
52
59
|
run: BenchmarkRun,
|
|
@@ -18,26 +18,41 @@ export async function restoreEvalRun(
|
|
|
18
18
|
using results = new Results(resolve(root, "runner.db"));
|
|
19
19
|
const previous = results.evalRun(evalRunId);
|
|
20
20
|
const slot = previous && results.slot(previous.slotId);
|
|
21
|
+
const original = previous?.restoration
|
|
22
|
+
? results.evalRun(previous.restoration.evalRunId)
|
|
23
|
+
: previous;
|
|
21
24
|
if (!previous || !slot || !slot.active || slot.evalRunId !== previous.id ||
|
|
22
|
-
|
|
23
|
-
|
|
25
|
+
!original || original.state !== "failed" || !original.interrupted || original.evidence ||
|
|
26
|
+
(previous !== original && (previous.state !== "completed" || !previous.restoration)))
|
|
27
|
+
throw new Error("Select an active interrupted EvalRun or its restored recording");
|
|
28
|
+
if (previous.restoration && (options.databaseHash !== previous.restoration.databaseHash ||
|
|
29
|
+
options.workspaceArchiveHash !== previous.restoration.workspaceArchiveHash))
|
|
30
|
+
throw new Error("A restored recording must retain the same original backup hashes");
|
|
24
31
|
if (results.benchmark!.state === "running" || results.benchmark!.mergedInto)
|
|
25
32
|
throw new Error("Wait for the active BenchmarkRun and use its aggregate directory");
|
|
26
|
-
const events = (await readFile(resolve(root, "eval-runs",
|
|
33
|
+
const events = (await readFile(resolve(root, "eval-runs", original.id, "evidence/events.jsonl"), "utf8"))
|
|
27
34
|
.trim().split("\n").filter(Boolean).map(line => JSON.parse(line).event);
|
|
28
35
|
const creation = events.find(event => event.type === "session.created" && !event.data.parentID);
|
|
29
36
|
if (!creation?.data.sessionID)
|
|
30
37
|
throw new Error("The original recording does not identify its root session");
|
|
31
|
-
const
|
|
32
|
-
if (!(await Bun.file(resolve(
|
|
38
|
+
const frozen = resolve(root, "inputs", original.input.evalId, original.input.sourceHash, "workspace");
|
|
39
|
+
if (!(await Bun.file(resolve(frozen, "..", "ready")).exists()))
|
|
33
40
|
throw new Error("The frozen starting workspace is missing");
|
|
41
|
+
const item = results.benchmark!.definition.evals.find(item => item.id === original.input.evalId);
|
|
42
|
+
if (!item || item.sourceHash !== original.input.sourceHash)
|
|
43
|
+
throw new Error("The frozen preparation no longer matches the original inputs");
|
|
34
44
|
const staging = resolve(root, "eval-runs", `.restored-${randomUUID()}`);
|
|
35
45
|
await mkdir(staging, { recursive: false });
|
|
36
46
|
try {
|
|
47
|
+
const preparation = resolve(staging, "preparation");
|
|
48
|
+
const initial = await archive.prepareRestoredInitial(
|
|
49
|
+
results.benchmark!.definition, item, original.input.imageId, frozen, preparation,
|
|
50
|
+
);
|
|
37
51
|
const restored = await archive.restoreArchivedCandidate(
|
|
38
|
-
|
|
52
|
+
original.input, creation.data.sessionID, initial, options, staging,
|
|
39
53
|
);
|
|
40
|
-
|
|
54
|
+
await rm(preparation, { recursive: true, force: true });
|
|
55
|
+
const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(original.startedAt);
|
|
41
56
|
if (!Number.isFinite(elapsedMs) || elapsedMs < 0)
|
|
42
57
|
throw new Error("Native completion predates the original EvalRun");
|
|
43
58
|
const evidence = { ...restored.evidence };
|
|
@@ -46,19 +61,19 @@ export async function restoreEvalRun(
|
|
|
46
61
|
if (results.benchmark!.state === "running" ||
|
|
47
62
|
results.slot(slot.id)?.evalRunId !== previous.id)
|
|
48
63
|
throw new Error("The BenchmarkRun or selected execution changed during restoration");
|
|
49
|
-
const run = results.startEval(slot,
|
|
64
|
+
const run = results.startEval(slot, original.input);
|
|
50
65
|
const destination = resolve(root, "eval-runs", run.id);
|
|
51
66
|
renameSync(staging, destination);
|
|
52
67
|
evidence.directory = relative(root, resolve(destination, restored.evidence.directory));
|
|
53
68
|
const completed: EvalRun = {
|
|
54
69
|
...run,
|
|
55
70
|
state: "completed",
|
|
56
|
-
startedAt:
|
|
71
|
+
startedAt: original.startedAt,
|
|
57
72
|
completedAt: restored.receipt.completedAt,
|
|
58
73
|
elapsedMs,
|
|
59
74
|
evidence,
|
|
60
75
|
session: { ...restored.session, database: relative(root, resolve(destination, restored.session.database)) },
|
|
61
|
-
restoration: { ...restored.receipt, evalRunId:
|
|
76
|
+
restoration: { ...restored.receipt, evalRunId: original.id },
|
|
62
77
|
metrics: measureRecording(restored.events.map((event, sequence) => ({
|
|
63
78
|
event, sequence, time: new Date("created" in event ? event.created : Date.now()).toISOString(),
|
|
64
79
|
})), restored.tools, evidence),
|
package/src/app/retry-run.ts
CHANGED
|
@@ -1,18 +1,40 @@
|
|
|
1
1
|
import { resolve } from "node:path";
|
|
2
|
-
import type { ExecutionObserver } from "../types";
|
|
2
|
+
import type { EvalRun, ExecutionObserver } from "../types";
|
|
3
3
|
import { Results } from "../infra/sqlite";
|
|
4
4
|
import { runtimeFingerprint } from "../infra/containers/oci";
|
|
5
|
-
import { candidateFingerprint } from "./plan-benchmark";
|
|
6
5
|
import { runEvalPipeline } from "./run-eval-pipeline";
|
|
7
6
|
import { ExecutionBudget } from "./execution-budget";
|
|
8
|
-
import { finishBenchmark } from "./run-benchmark";
|
|
7
|
+
import { executeQueue, finishBenchmark } from "./run-benchmark";
|
|
9
8
|
import { recoverStoppedRunner, runnerStopped } from "./recover-run";
|
|
9
|
+
import { retryPlan } from "./plan-retries";
|
|
10
|
+
import { CostBudget, estimateWork } from "./cost-plan";
|
|
10
11
|
|
|
11
12
|
export async function retryEvalRun(
|
|
12
13
|
directory: string,
|
|
13
14
|
evalRunId: string,
|
|
14
15
|
onEvent?: ExecutionObserver,
|
|
15
16
|
) {
|
|
17
|
+
return (await retryEvalRuns(directory, [evalRunId], onEvent))[0]!;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** Read-only preflight for a precise retry scope. No model calls or selection updates. */
|
|
21
|
+
export async function planEvalRunRetries(directory: string, evalRunIds: readonly string[]) {
|
|
22
|
+
using results = new Results(resolve(directory, "runner.db"), true);
|
|
23
|
+
const benchmark = results.benchmark!;
|
|
24
|
+
if (benchmark.mergedInto || benchmark.state === "running")
|
|
25
|
+
throw new Error("Wait for active work and select the aggregate run");
|
|
26
|
+
const runtime = await runtimeFingerprint(benchmark.definition.container, benchmark.definition.judge.verification);
|
|
27
|
+
const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
|
|
28
|
+
return { plan, estimate: estimateWork(benchmark.definition, results, plan) };
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** A single retry round; original evidence and judgments remain immutable. */
|
|
32
|
+
export async function retryEvalRuns(
|
|
33
|
+
directory: string,
|
|
34
|
+
evalRunIds: readonly string[],
|
|
35
|
+
onEvent?: ExecutionObserver,
|
|
36
|
+
options: { maxCostUSD?: number } = {},
|
|
37
|
+
): Promise<EvalRun[]> {
|
|
16
38
|
using results = new Results(resolve(directory, "runner.db"));
|
|
17
39
|
if (results.benchmark!.mergedInto)
|
|
18
40
|
throw new Error(
|
|
@@ -24,41 +46,26 @@ export async function retryEvalRun(
|
|
|
24
46
|
recoverStoppedRunner(results);
|
|
25
47
|
}
|
|
26
48
|
const benchmark = results.benchmark!;
|
|
27
|
-
const previous = results.evalRun(evalRunId);
|
|
28
|
-
if (!previous || previous.state === "completed")
|
|
29
|
-
throw new Error(
|
|
30
|
-
"Select a failed eval run; completed evidence can be rejudged",
|
|
31
|
-
);
|
|
32
|
-
const slot = results.slot(previous.slotId)!;
|
|
33
|
-
if (slot.evalRunId !== previous.id)
|
|
34
|
-
throw new Error("Select the current eval run for this repetition");
|
|
35
49
|
const runtime = await runtimeFingerprint(
|
|
36
50
|
benchmark.definition.container,
|
|
37
51
|
benchmark.definition.judge.verification,
|
|
38
52
|
);
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
const
|
|
51
|
-
(
|
|
52
|
-
|
|
53
|
-
const
|
|
54
|
-
|
|
55
|
-
"inputs",
|
|
56
|
-
definition.id,
|
|
57
|
-
definition.sourceHash,
|
|
58
|
-
"workspace",
|
|
59
|
-
);
|
|
60
|
-
if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
|
|
61
|
-
throw new Error("The saved starting workspace is missing");
|
|
53
|
+
const plan = retryPlan(benchmark.definition, runtime, results, evalRunIds);
|
|
54
|
+
const workspaces = new Map<string, string>();
|
|
55
|
+
for (const item of plan) {
|
|
56
|
+
const definition = benchmark.definition.evals.find(value => value.id === item.slot.evalId)!;
|
|
57
|
+
const workspace = resolve(directory, "inputs", definition.id, definition.sourceHash, "workspace");
|
|
58
|
+
if (!(await Bun.file(resolve(workspace, "..", "ready")).exists()))
|
|
59
|
+
throw new Error("The saved starting workspace is missing");
|
|
60
|
+
workspaces.set(item.slot.evalId, workspace);
|
|
61
|
+
}
|
|
62
|
+
const estimate = estimateWork(benchmark.definition, results, plan);
|
|
63
|
+
const existing = new Set([...results.evalRuns(), ...results.judgeRuns()].map(run => run.id));
|
|
64
|
+
const spent = () => [...results.evalRuns(), ...results.judgeRuns()]
|
|
65
|
+
.filter(run => !existing.has(run.id))
|
|
66
|
+
.reduce((sum, run) => sum + (run.session?.accounting?.costUSD ?? results.recordedCost(run.id) ?? 0), 0);
|
|
67
|
+
const costBudget = new CostBudget(options.maxCostUSD, spent);
|
|
68
|
+
const estimates = new Map(estimate.items.map(item => [item.slotId, item.estimatedUSD]));
|
|
62
69
|
const context = {
|
|
63
70
|
directory: resolve(directory),
|
|
64
71
|
definition: benchmark.definition,
|
|
@@ -66,29 +73,58 @@ export async function retryEvalRun(
|
|
|
66
73
|
results,
|
|
67
74
|
onEvent,
|
|
68
75
|
};
|
|
69
|
-
results.
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
76
|
+
results.transaction(() => {
|
|
77
|
+
if (results.benchmark!.state === "running") throw new Error("Another operation claimed this run");
|
|
78
|
+
for (const item of plan) {
|
|
79
|
+
const selected = results.slot(item.slot.id);
|
|
80
|
+
if (!selected?.active || selected.evalRunId !== item.slot.evalRunId ||
|
|
81
|
+
selected.judgeRunId !== item.slot.judgeRunId)
|
|
82
|
+
throw new Error("The retry selection changed during preflight");
|
|
83
|
+
}
|
|
84
|
+
results.saveBenchmark({ ...benchmark, runtime, state: "running",
|
|
85
|
+
scheduledSlotIds: plan.map(item => item.slot.id), updatedAt: new Date().toISOString(),
|
|
86
|
+
execution: { startedAt: new Date().toISOString(),
|
|
87
|
+
onlyModels: [...new Set(plan.map(item => item.slot.model))],
|
|
88
|
+
onlyEvals: [...new Set(plan.map(item => item.slot.evalId))],
|
|
89
|
+
onlyRepetitions: [...new Set(plan.map(item => item.slot.repetition))],
|
|
90
|
+
budgetUSD: options.maxCostUSD, estimatedUSD: estimate.estimatedUSD,
|
|
91
|
+
spentUSD: 0, deferred: 0 } });
|
|
74
92
|
});
|
|
75
93
|
const timer = setInterval(
|
|
76
94
|
() =>
|
|
77
95
|
results.saveBenchmark({
|
|
78
96
|
...results.benchmark!,
|
|
79
97
|
updatedAt: new Date().toISOString(),
|
|
98
|
+
scheduledSlotIds: plan.filter(item => item.action !== "deferred").map(item => item.slot.id),
|
|
99
|
+
execution: { ...results.benchmark!.execution, spentUSD: spent(),
|
|
100
|
+
deferred: plan.filter(item => item.action === "deferred").length },
|
|
80
101
|
}),
|
|
81
102
|
5000,
|
|
82
103
|
);
|
|
83
104
|
try {
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
105
|
+
const budget = new ExecutionBudget(benchmark.definition.concurrency);
|
|
106
|
+
const completed = new Map<string, EvalRun>();
|
|
107
|
+
const errors: unknown[] = [];
|
|
108
|
+
await executeQueue(plan, benchmark.definition.concurrency, async item => {
|
|
109
|
+
const release = costBudget.reserve(estimates.get(item.slot.id) ?? null);
|
|
110
|
+
if (!release) {
|
|
111
|
+
item.action = "deferred";
|
|
112
|
+
item.reason = "Retry scheduling budget reached or no estimate is available";
|
|
113
|
+
return;
|
|
114
|
+
}
|
|
115
|
+
try {
|
|
116
|
+
const run = await runEvalPipeline(context, item.slot, workspaces.get(item.slot.evalId)!, budget);
|
|
117
|
+
completed.set(item.slot.id, run);
|
|
118
|
+
} catch (error) {
|
|
119
|
+
errors.push(error);
|
|
120
|
+
} finally { release(); }
|
|
121
|
+
});
|
|
122
|
+
if (errors.length) throw new AggregateError(errors, "One or more retry operations failed");
|
|
123
|
+
return plan.flatMap(item => completed.has(item.slot.id) ? [completed.get(item.slot.id)!] : []);
|
|
90
124
|
} finally {
|
|
91
125
|
clearInterval(timer);
|
|
126
|
+
results.saveBenchmark({ ...results.benchmark!, execution: { ...results.benchmark!.execution,
|
|
127
|
+
spentUSD: spent(), deferred: plan.filter(item => item.action === "deferred").length } });
|
|
92
128
|
finishBenchmark(context);
|
|
93
129
|
}
|
|
94
130
|
}
|
package/src/cli.ts
CHANGED
|
@@ -7,7 +7,8 @@ import {
|
|
|
7
7
|
buildImage,
|
|
8
8
|
addModels,
|
|
9
9
|
removeModels,
|
|
10
|
-
|
|
10
|
+
retryEvalRuns,
|
|
11
|
+
planEvalRunRetries,
|
|
11
12
|
restoreEvalRun,
|
|
12
13
|
judgeRun,
|
|
13
14
|
serveResults,
|
|
@@ -48,7 +49,7 @@ Commands:
|
|
|
48
49
|
add-models <run> Add models with repeated --model flags
|
|
49
50
|
remove-models <run> Remove active models while retaining their evidence
|
|
50
51
|
refresh-model-names <run> Refresh reporting names without running evals
|
|
51
|
-
retry <run> <eval-run
|
|
52
|
+
retry <run> <eval-run>... Retry failed or unresolved candidate work
|
|
52
53
|
rejudge <run> <eval-run> Judge the saved evidence again
|
|
53
54
|
restore <run> <eval-run> Restore completed orphan work from hashed backups
|
|
54
55
|
|
|
@@ -63,6 +64,7 @@ Options:
|
|
|
63
64
|
--only-repetition <n> Execute only this repetition (repeatable)
|
|
64
65
|
--max-cost <usd> Scheduling budget for this invocation
|
|
65
66
|
--final-only Judge only after candidates finish
|
|
67
|
+
--dry-run Plan explicit retries without running them
|
|
66
68
|
--verification Build only the standard verification image (image command)
|
|
67
69
|
--output <dir> New prepared-input directory (prepare command)
|
|
68
70
|
--port <port> Viewer port (default: 4173)
|
|
@@ -184,15 +186,22 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
184
186
|
console.log(JSON.stringify(await restoreEvalRun(resolve(args[1]), args[2], {
|
|
185
187
|
database: resolve(database), databaseHash, workspaceArchive: resolve(workspaceArchive), workspaceArchiveHash,
|
|
186
188
|
}), null, 2));
|
|
187
|
-
} else if (command === "retry"
|
|
189
|
+
} else if (command === "retry") {
|
|
190
|
+
const flags = args.findIndex((arg, index) => index >= 2 && arg.startsWith("--"));
|
|
191
|
+
const ids = args.slice(2, flags < 0 ? args.length : flags);
|
|
192
|
+
if (!args[1] || !ids.length) throw new Error("retry requires a run directory and explicit EvalRun IDs");
|
|
193
|
+
const result = args.includes("--dry-run")
|
|
194
|
+
? await planEvalRunRetries(resolve(args[1]), ids)
|
|
195
|
+
: await retryEvalRuns(resolve(args[1]), ids, undefined,
|
|
196
|
+
{ maxCostUSD: option("--max-cost") === undefined ? undefined : Number(option("--max-cost")) });
|
|
197
|
+
console.log(JSON.stringify(!args.includes("--dry-run") && ids.length === 1 && Array.isArray(result)
|
|
198
|
+
? result[0] : result, null, 2));
|
|
199
|
+
} else if (command === "rejudge") {
|
|
188
200
|
if (!args[1] || !args[2])
|
|
189
201
|
throw new Error(
|
|
190
202
|
`${command} requires a benchmark run directory and eval run ID`,
|
|
191
203
|
);
|
|
192
|
-
const result =
|
|
193
|
-
command === "retry"
|
|
194
|
-
? await retryEvalRun(resolve(args[1]), args[2])
|
|
195
|
-
: await judgeRun(resolve(args[1]), args[2]);
|
|
204
|
+
const result = await judgeRun(resolve(args[1]), args[2]);
|
|
196
205
|
console.log(JSON.stringify(result, null, 2));
|
|
197
206
|
} else
|
|
198
207
|
throw new Error(
|
package/src/index.ts
CHANGED
|
@@ -67,7 +67,7 @@ export { removeModels } from "./app/remove-models";
|
|
|
67
67
|
export { refreshModelNames } from "./app/refresh-model-names";
|
|
68
68
|
export type { CostEstimate } from "./app/cost-plan";
|
|
69
69
|
export { mergeBenchmarkRuns } from "./app/merge-runs";
|
|
70
|
-
export { retryEvalRun } from "./app/retry-run";
|
|
70
|
+
export { retryEvalRun, retryEvalRuns, planEvalRunRetries } from "./app/retry-run";
|
|
71
71
|
export { restoreEvalRun } from "./app/restore-eval-run";
|
|
72
72
|
export { judgeRun, judgeRuns } from "./app/rejudge";
|
|
73
73
|
export { judgeEvidence, recordEvidence } from "./app/judge-evidence";
|
|
@@ -1,13 +1,15 @@
|
|
|
1
1
|
import { Database } from "bun:sqlite";
|
|
2
2
|
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
3
3
|
import { resolve, dirname } from "node:path";
|
|
4
|
-
import type { EvalRunInput, OpenCodeStreamEvent } from "../types";
|
|
4
|
+
import type { BenchmarkDefinition, EvalDefinition, EvalRunInput, OpenCodeStreamEvent } from "../types";
|
|
5
5
|
import { EvidenceCapture } from "./evidence";
|
|
6
6
|
import { extractWorkspaceArchive } from "./containers/transfer";
|
|
7
7
|
import { removeCredentials } from "./opencode/auth";
|
|
8
8
|
import { readArchivedSession } from "./opencode/archive";
|
|
9
9
|
import { parseModel, type SessionResult } from "./opencode/session";
|
|
10
10
|
import { hash, writeJson } from "./files";
|
|
11
|
+
import { CandidateContainer } from "./containers/oci";
|
|
12
|
+
import { createSessionDatabase } from "./opencode/host";
|
|
11
13
|
|
|
12
14
|
export type ArchiveRestore = {
|
|
13
15
|
database: string;
|
|
@@ -16,6 +18,26 @@ export type ArchiveRestore = {
|
|
|
16
18
|
workspaceArchiveHash: string;
|
|
17
19
|
};
|
|
18
20
|
|
|
21
|
+
/** Replay only frozen author preparation, without credentials or a model prompt. */
|
|
22
|
+
export async function prepareRestoredInitial(
|
|
23
|
+
definition: BenchmarkDefinition,
|
|
24
|
+
item: EvalDefinition,
|
|
25
|
+
imageId: string,
|
|
26
|
+
workspace: string,
|
|
27
|
+
staging: string,
|
|
28
|
+
) {
|
|
29
|
+
if (!item.settings.prepare?.length) return workspace;
|
|
30
|
+
await mkdir(staging, { recursive: true });
|
|
31
|
+
const database = resolve(staging, "opencode.db");
|
|
32
|
+
await createSessionDatabase(database, []);
|
|
33
|
+
await using container = await CandidateContainer.create(definition.container, imageId);
|
|
34
|
+
await container.prepare(workspace, database, definition.candidate.websearch,
|
|
35
|
+
item.settings.prepare, staging);
|
|
36
|
+
const initial = resolve(staging, "workspace");
|
|
37
|
+
await container.snapshot(initial, staging);
|
|
38
|
+
return initial;
|
|
39
|
+
}
|
|
40
|
+
|
|
19
41
|
/** A backup must belong to the original execution, not merely resemble its answer. */
|
|
20
42
|
export function verifyArchivedInput(
|
|
21
43
|
input: EvalRunInput,
|
|
@@ -88,7 +110,7 @@ export async function restoreArchivedCandidate(
|
|
|
88
110
|
completedAt: new Date(completedAt).toISOString(),
|
|
89
111
|
databaseHash: options.databaseHash,
|
|
90
112
|
workspaceArchiveHash: options.workspaceArchiveHash,
|
|
91
|
-
initial: "frozen
|
|
113
|
+
initial: "frozen input and author preparation replayed without model calls; not a newly observed candidate snapshot",
|
|
92
114
|
};
|
|
93
115
|
await writeJson(resolve(directory, "restoration.json"), receipt);
|
|
94
116
|
return {
|