@hona/openeval 0.5.8 → 0.5.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app/read-results.ts +9 -2
- package/src/app/restore-eval-run.ts +91 -0
- package/src/cli.ts +15 -1
- package/src/index.ts +1 -0
- package/src/infra/restore-archive.ts +132 -0
- package/src/types.ts +10 -0
package/package.json
CHANGED
package/src/app/read-results.ts
CHANGED
|
@@ -43,10 +43,17 @@ const costOf = (run: EvalRun | JudgeRun, results?: Results): Cost => {
|
|
|
43
43
|
complete: usd !== undefined,
|
|
44
44
|
};
|
|
45
45
|
};
|
|
46
|
-
//
|
|
46
|
+
// Restored records represent the same original execution, not additional model spend.
|
|
47
47
|
const runCosts = (runs: (EvalRun | JudgeRun)[], results: Results) => {
|
|
48
48
|
const unique = new Map(runs.map((run) => [run.id, run]));
|
|
49
|
-
|
|
49
|
+
const executions = new Map<string, EvalRun | JudgeRun>();
|
|
50
|
+
for (const run of unique.values()) {
|
|
51
|
+
const restoration = "restoration" in run ? run.restoration : undefined;
|
|
52
|
+
const key = restoration?.evalRunId ?? run.id;
|
|
53
|
+
const previous = executions.get(key);
|
|
54
|
+
if (!previous || restoration) executions.set(key, run);
|
|
55
|
+
}
|
|
56
|
+
return sumCosts([...executions.values()].map((run) => costOf(run, results)));
|
|
50
57
|
};
|
|
51
58
|
const entryOf = (
|
|
52
59
|
run: BenchmarkRun,
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
2
|
+
import { mkdir, readFile, rm } from "node:fs/promises";
|
|
3
|
+
import { renameSync } from "node:fs";
|
|
4
|
+
import { relative, resolve } from "node:path";
|
|
5
|
+
import type { EvalRun } from "../types";
|
|
6
|
+
import { Results } from "../infra/sqlite";
|
|
7
|
+
import * as archive from "../infra/restore-archive";
|
|
8
|
+
import type { ArchiveRestore } from "../infra/restore-archive";
|
|
9
|
+
import { measureRecording } from "../infra/recording/metrics";
|
|
10
|
+
|
|
11
|
+
/** Restore completed orphan work as a new record; retain the failed coordinator record. */
|
|
12
|
+
export async function restoreEvalRun(
|
|
13
|
+
directory: string,
|
|
14
|
+
evalRunId: string,
|
|
15
|
+
options: ArchiveRestore,
|
|
16
|
+
): Promise<EvalRun> {
|
|
17
|
+
const root = resolve(directory);
|
|
18
|
+
using results = new Results(resolve(root, "runner.db"));
|
|
19
|
+
const previous = results.evalRun(evalRunId);
|
|
20
|
+
const slot = previous && results.slot(previous.slotId);
|
|
21
|
+
const original = previous?.restoration
|
|
22
|
+
? results.evalRun(previous.restoration.evalRunId)
|
|
23
|
+
: previous;
|
|
24
|
+
if (!previous || !slot || !slot.active || slot.evalRunId !== previous.id ||
|
|
25
|
+
!original || original.state !== "failed" || !original.interrupted || original.evidence ||
|
|
26
|
+
(previous !== original && (previous.state !== "completed" || !previous.restoration)))
|
|
27
|
+
throw new Error("Select an active interrupted EvalRun or its restored recording");
|
|
28
|
+
if (previous.restoration && (options.databaseHash !== previous.restoration.databaseHash ||
|
|
29
|
+
options.workspaceArchiveHash !== previous.restoration.workspaceArchiveHash))
|
|
30
|
+
throw new Error("A restored recording must retain the same original backup hashes");
|
|
31
|
+
if (results.benchmark!.state === "running" || results.benchmark!.mergedInto)
|
|
32
|
+
throw new Error("Wait for the active BenchmarkRun and use its aggregate directory");
|
|
33
|
+
const events = (await readFile(resolve(root, "eval-runs", original.id, "evidence/events.jsonl"), "utf8"))
|
|
34
|
+
.trim().split("\n").filter(Boolean).map(line => JSON.parse(line).event);
|
|
35
|
+
const creation = events.find(event => event.type === "session.created" && !event.data.parentID);
|
|
36
|
+
if (!creation?.data.sessionID)
|
|
37
|
+
throw new Error("The original recording does not identify its root session");
|
|
38
|
+
const frozen = resolve(root, "inputs", original.input.evalId, original.input.sourceHash, "workspace");
|
|
39
|
+
if (!(await Bun.file(resolve(frozen, "..", "ready")).exists()))
|
|
40
|
+
throw new Error("The frozen starting workspace is missing");
|
|
41
|
+
const item = results.benchmark!.definition.evals.find(item => item.id === original.input.evalId);
|
|
42
|
+
if (!item || item.sourceHash !== original.input.sourceHash)
|
|
43
|
+
throw new Error("The frozen preparation no longer matches the original inputs");
|
|
44
|
+
const staging = resolve(root, "eval-runs", `.restored-${randomUUID()}`);
|
|
45
|
+
await mkdir(staging, { recursive: false });
|
|
46
|
+
try {
|
|
47
|
+
const preparation = resolve(staging, "preparation");
|
|
48
|
+
const initial = await archive.prepareRestoredInitial(
|
|
49
|
+
results.benchmark!.definition, item, original.input.imageId, frozen, preparation,
|
|
50
|
+
);
|
|
51
|
+
const restored = await archive.restoreArchivedCandidate(
|
|
52
|
+
original.input, creation.data.sessionID, initial, options, staging,
|
|
53
|
+
);
|
|
54
|
+
await rm(preparation, { recursive: true, force: true });
|
|
55
|
+
const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(original.startedAt);
|
|
56
|
+
if (!Number.isFinite(elapsedMs) || elapsedMs < 0)
|
|
57
|
+
throw new Error("Native completion predates the original EvalRun");
|
|
58
|
+
const evidence = { ...restored.evidence };
|
|
59
|
+
// This operation does not claim a live candidate. Validate selection again at publication.
|
|
60
|
+
return results.transaction(() => {
|
|
61
|
+
if (results.benchmark!.state === "running" ||
|
|
62
|
+
results.slot(slot.id)?.evalRunId !== previous.id)
|
|
63
|
+
throw new Error("The BenchmarkRun or selected execution changed during restoration");
|
|
64
|
+
const run = results.startEval(slot, original.input);
|
|
65
|
+
const destination = resolve(root, "eval-runs", run.id);
|
|
66
|
+
renameSync(staging, destination);
|
|
67
|
+
evidence.directory = relative(root, resolve(destination, restored.evidence.directory));
|
|
68
|
+
const completed: EvalRun = {
|
|
69
|
+
...run,
|
|
70
|
+
state: "completed",
|
|
71
|
+
startedAt: original.startedAt,
|
|
72
|
+
completedAt: restored.receipt.completedAt,
|
|
73
|
+
elapsedMs,
|
|
74
|
+
evidence,
|
|
75
|
+
session: { ...restored.session, database: relative(root, resolve(destination, restored.session.database)) },
|
|
76
|
+
restoration: { ...restored.receipt, evalRunId: original.id },
|
|
77
|
+
metrics: measureRecording(restored.events.map((event, sequence) => ({
|
|
78
|
+
event, sequence, time: new Date("created" in event ? event.created : Date.now()).toISOString(),
|
|
79
|
+
})), restored.tools, evidence),
|
|
80
|
+
};
|
|
81
|
+
for (const event of restored.events)
|
|
82
|
+
results.append({ executionId: run.id, stage: "candidate", time: new Date("created" in event ? event.created : Date.now()).toISOString(), event });
|
|
83
|
+
results.finishEval(completed);
|
|
84
|
+
results.saveBenchmark({ ...results.benchmark!, state: "incomplete", scheduledSlotIds: [], updatedAt: new Date().toISOString() });
|
|
85
|
+
return completed;
|
|
86
|
+
});
|
|
87
|
+
} finally {
|
|
88
|
+
// Only this operation's private staging directory; published evidence is never removed.
|
|
89
|
+
await rm(staging, { recursive: true, force: true });
|
|
90
|
+
}
|
|
91
|
+
}
|
package/src/cli.ts
CHANGED
|
@@ -8,6 +8,7 @@ import {
|
|
|
8
8
|
addModels,
|
|
9
9
|
removeModels,
|
|
10
10
|
retryEvalRun,
|
|
11
|
+
restoreEvalRun,
|
|
11
12
|
judgeRun,
|
|
12
13
|
serveResults,
|
|
13
14
|
mergeBenchmarkRuns,
|
|
@@ -49,6 +50,7 @@ Commands:
|
|
|
49
50
|
refresh-model-names <run> Refresh reporting names without running evals
|
|
50
51
|
retry <run> <eval-run> Retry a candidate execution
|
|
51
52
|
rejudge <run> <eval-run> Judge the saved evidence again
|
|
53
|
+
restore <run> <eval-run> Restore completed orphan work from hashed backups
|
|
52
54
|
|
|
53
55
|
Options:
|
|
54
56
|
--benchmark <dir> Benchmark directory (default: current directory)
|
|
@@ -64,6 +66,10 @@ Options:
|
|
|
64
66
|
--verification Build only the standard verification image (image command)
|
|
65
67
|
--output <dir> New prepared-input directory (prepare command)
|
|
66
68
|
--port <port> Viewer port (default: 4173)
|
|
69
|
+
--database <file> Native backup for restore
|
|
70
|
+
--database-hash <sha> SHA-256 of the native backup
|
|
71
|
+
--workspace-archive <file> Workspace tar backup for restore
|
|
72
|
+
--workspace-archive-hash <sha> SHA-256 of the workspace tar
|
|
67
73
|
|
|
68
74
|
Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
69
75
|
} else if (command === "--version") {
|
|
@@ -170,6 +176,14 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
170
176
|
} else if (command === "refresh-model-names") {
|
|
171
177
|
if (!args[1]) throw new Error("refresh-model-names requires a benchmark run directory");
|
|
172
178
|
console.log(JSON.stringify(await refreshModelNames(resolve(args[1])), null, 2));
|
|
179
|
+
} else if (command === "restore") {
|
|
180
|
+
const database = option("--database"), databaseHash = option("--database-hash");
|
|
181
|
+
const workspaceArchive = option("--workspace-archive"), workspaceArchiveHash = option("--workspace-archive-hash");
|
|
182
|
+
if (!args[1] || !args[2] || !database || !databaseHash || !workspaceArchive || !workspaceArchiveHash)
|
|
183
|
+
throw new Error("restore requires the run, interrupted EvalRun, native backup, workspace tar, and both SHA-256 hashes");
|
|
184
|
+
console.log(JSON.stringify(await restoreEvalRun(resolve(args[1]), args[2], {
|
|
185
|
+
database: resolve(database), databaseHash, workspaceArchive: resolve(workspaceArchive), workspaceArchiveHash,
|
|
186
|
+
}), null, 2));
|
|
173
187
|
} else if (command === "retry" || command === "rejudge") {
|
|
174
188
|
if (!args[1] || !args[2])
|
|
175
189
|
throw new Error(
|
|
@@ -182,7 +196,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
182
196
|
console.log(JSON.stringify(result, null, 2));
|
|
183
197
|
} else
|
|
184
198
|
throw new Error(
|
|
185
|
-
"Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, snapshot",
|
|
199
|
+
"Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, restore, snapshot",
|
|
186
200
|
);
|
|
187
201
|
} catch (error) {
|
|
188
202
|
console.error(error instanceof Error ? error.message : String(error));
|
package/src/index.ts
CHANGED
|
@@ -68,6 +68,7 @@ export { refreshModelNames } from "./app/refresh-model-names";
|
|
|
68
68
|
export type { CostEstimate } from "./app/cost-plan";
|
|
69
69
|
export { mergeBenchmarkRuns } from "./app/merge-runs";
|
|
70
70
|
export { retryEvalRun } from "./app/retry-run";
|
|
71
|
+
export { restoreEvalRun } from "./app/restore-eval-run";
|
|
71
72
|
export { judgeRun, judgeRuns } from "./app/rejudge";
|
|
72
73
|
export { judgeEvidence, recordEvidence } from "./app/judge-evidence";
|
|
73
74
|
export {
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
import { Database } from "bun:sqlite";
|
|
2
|
+
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
3
|
+
import { resolve, dirname } from "node:path";
|
|
4
|
+
import type { BenchmarkDefinition, EvalDefinition, EvalRunInput, OpenCodeStreamEvent } from "../types";
|
|
5
|
+
import { EvidenceCapture } from "./evidence";
|
|
6
|
+
import { extractWorkspaceArchive } from "./containers/transfer";
|
|
7
|
+
import { removeCredentials } from "./opencode/auth";
|
|
8
|
+
import { readArchivedSession } from "./opencode/archive";
|
|
9
|
+
import { parseModel, type SessionResult } from "./opencode/session";
|
|
10
|
+
import { hash, writeJson } from "./files";
|
|
11
|
+
import { CandidateContainer } from "./containers/oci";
|
|
12
|
+
import { createSessionDatabase } from "./opencode/host";
|
|
13
|
+
|
|
14
|
+
export type ArchiveRestore = {
|
|
15
|
+
database: string;
|
|
16
|
+
databaseHash: string;
|
|
17
|
+
workspaceArchive: string;
|
|
18
|
+
workspaceArchiveHash: string;
|
|
19
|
+
};
|
|
20
|
+
|
|
21
|
+
/** Replay only frozen author preparation, without credentials or a model prompt. */
|
|
22
|
+
export async function prepareRestoredInitial(
|
|
23
|
+
definition: BenchmarkDefinition,
|
|
24
|
+
item: EvalDefinition,
|
|
25
|
+
imageId: string,
|
|
26
|
+
workspace: string,
|
|
27
|
+
staging: string,
|
|
28
|
+
) {
|
|
29
|
+
if (!item.settings.prepare?.length) return workspace;
|
|
30
|
+
await mkdir(staging, { recursive: true });
|
|
31
|
+
const database = resolve(staging, "opencode.db");
|
|
32
|
+
await createSessionDatabase(database, []);
|
|
33
|
+
await using container = await CandidateContainer.create(definition.container, imageId);
|
|
34
|
+
await container.prepare(workspace, database, definition.candidate.websearch,
|
|
35
|
+
item.settings.prepare, staging);
|
|
36
|
+
const initial = resolve(staging, "workspace");
|
|
37
|
+
await container.snapshot(initial, staging);
|
|
38
|
+
return initial;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** A backup must belong to the original execution, not merely resemble its answer. */
|
|
42
|
+
export function verifyArchivedInput(
|
|
43
|
+
input: EvalRunInput,
|
|
44
|
+
sessionId: string,
|
|
45
|
+
archived: { result: SessionResult; outcome: string | undefined },
|
|
46
|
+
) {
|
|
47
|
+
const root = archived.result.sessions.find(session => session.info.id === sessionId);
|
|
48
|
+
const model = parseModel(input.model);
|
|
49
|
+
const prompt = root?.messages.find(message => message.type === "user")?.text;
|
|
50
|
+
if (!root || root.info.parentID || root.info.model?.providerID !== model.providerID ||
|
|
51
|
+
root.info.model?.id !== model.id || root.info.model?.variant !== model.variant ||
|
|
52
|
+
prompt !== input.prompt)
|
|
53
|
+
throw new Error("The native backup does not match the original session, model, and prompt");
|
|
54
|
+
if (archived.outcome !== "succeeded" || archived.result.state !== "completed")
|
|
55
|
+
throw new Error("Restore requires a naturally completed native session");
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Finalize a copied native recording. No candidate program, prompt, or model runs. */
|
|
59
|
+
export async function restoreArchivedCandidate(
|
|
60
|
+
input: EvalRunInput,
|
|
61
|
+
sessionId: string,
|
|
62
|
+
initialWorkspace: string,
|
|
63
|
+
options: ArchiveRestore,
|
|
64
|
+
directory: string,
|
|
65
|
+
) {
|
|
66
|
+
for (const [file, expected] of [
|
|
67
|
+
[options.database, options.databaseHash],
|
|
68
|
+
[options.workspaceArchive, options.workspaceArchiveHash],
|
|
69
|
+
]) {
|
|
70
|
+
if (!/^[a-f0-9]{64}$/.test(expected!) || hash(await Bun.file(file!).bytes()) !== expected)
|
|
71
|
+
throw new Error("The recovery backup hash does not match");
|
|
72
|
+
}
|
|
73
|
+
if (Bun.file(options.workspaceArchive).size > 256 * 1024 * 1024)
|
|
74
|
+
throw new Error("The recovery workspace archive exceeds 256 MiB");
|
|
75
|
+
const staging = await mkdtemp(resolve(dirname(directory), ".restore-"));
|
|
76
|
+
try {
|
|
77
|
+
await mkdir(directory, { recursive: true });
|
|
78
|
+
const database = resolve(directory, "opencode.db");
|
|
79
|
+
// SQLite serialization includes committed WAL pages without modifying the backup.
|
|
80
|
+
using source = new Database(options.database, { readonly: true });
|
|
81
|
+
await Bun.write(database, source.serialize());
|
|
82
|
+
removeCredentials(database);
|
|
83
|
+
const events: OpenCodeStreamEvent[] = [];
|
|
84
|
+
const archived = await readArchivedSession(database, sessionId, event => events.push(event));
|
|
85
|
+
verifyArchivedInput(input, sessionId, archived);
|
|
86
|
+
const creation = events.find((event): event is Extract<OpenCodeStreamEvent, { type: "session.created" }> =>
|
|
87
|
+
event.type === "session.created" && event.data.sessionID === sessionId);
|
|
88
|
+
if (!creation) throw new Error("The native backup has no root creation event");
|
|
89
|
+
const completedAt = archived.result.sessions.find(session => session.info.id === sessionId)!.info.time.idle;
|
|
90
|
+
if (typeof completedAt !== "number" || !Number.isFinite(completedAt))
|
|
91
|
+
throw new Error("The native session has no completion timestamp");
|
|
92
|
+
removeCredentials(database);
|
|
93
|
+
const workspace = resolve(staging, "workspace");
|
|
94
|
+
await mkdir(workspace);
|
|
95
|
+
await extractWorkspaceArchive(options.workspaceArchive, workspace, { sourceRoot: "/workspace" });
|
|
96
|
+
const capture = await EvidenceCapture.create(resolve(directory, "evidence"), initialWorkspace,
|
|
97
|
+
{ excludeDirectories: ["node_modules"] });
|
|
98
|
+
for (const event of events)
|
|
99
|
+
capture.event(event, new Date("created" in event ? event.created : Date.now()).toISOString());
|
|
100
|
+
const manifest = await capture.finish({
|
|
101
|
+
prompt: input.prompt,
|
|
102
|
+
response: { text: archived.result.text },
|
|
103
|
+
tools: archived.result.tools,
|
|
104
|
+
workspace,
|
|
105
|
+
});
|
|
106
|
+
await writeJson(resolve(directory, "session.json"), archived.result.sessions);
|
|
107
|
+
const receipt = {
|
|
108
|
+
sessionId,
|
|
109
|
+
restoredAt: new Date().toISOString(),
|
|
110
|
+
completedAt: new Date(completedAt).toISOString(),
|
|
111
|
+
databaseHash: options.databaseHash,
|
|
112
|
+
workspaceArchiveHash: options.workspaceArchiveHash,
|
|
113
|
+
initial: "frozen input and author preparation replayed without model calls; not a newly observed candidate snapshot",
|
|
114
|
+
};
|
|
115
|
+
await writeJson(resolve(directory, "restoration.json"), receipt);
|
|
116
|
+
return {
|
|
117
|
+
evidence: { directory: "evidence", hash: manifest.sha256 },
|
|
118
|
+
session: {
|
|
119
|
+
sessionId,
|
|
120
|
+
database: "opencode.db",
|
|
121
|
+
databaseHash: hash(await Bun.file(database).bytes()),
|
|
122
|
+
opencodeVersion: creation.data.version,
|
|
123
|
+
accounting: archived.result.accounting,
|
|
124
|
+
},
|
|
125
|
+
events,
|
|
126
|
+
tools: archived.result.tools,
|
|
127
|
+
receipt,
|
|
128
|
+
};
|
|
129
|
+
} finally {
|
|
130
|
+
await rm(staging, { recursive: true, force: true });
|
|
131
|
+
}
|
|
132
|
+
}
|
package/src/types.ts
CHANGED
|
@@ -232,6 +232,16 @@ export type EvalRun = {
|
|
|
232
232
|
stop?: EvalStop;
|
|
233
233
|
/** The runner stopped before this session finished; the planner collects it again. */
|
|
234
234
|
interrupted?: boolean;
|
|
235
|
+
/** A restored native session; the original interrupted coordinator record is retained. */
|
|
236
|
+
restoration?: {
|
|
237
|
+
evalRunId: string;
|
|
238
|
+
sessionId: string;
|
|
239
|
+
restoredAt: string;
|
|
240
|
+
completedAt: string;
|
|
241
|
+
databaseHash: string;
|
|
242
|
+
workspaceArchiveHash: string;
|
|
243
|
+
initial: string;
|
|
244
|
+
};
|
|
235
245
|
};
|
|
236
246
|
export type CriterionDefinition = { id: string; name: string };
|
|
237
247
|
export type EvidenceCitation = (
|