@hona/openeval 0.5.8 → 0.5.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.8",
3
+ "version": "0.5.10",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -43,10 +43,17 @@ const costOf = (run: EvalRun | JudgeRun, results?: Results): Cost => {
43
43
  complete: usd !== undefined,
44
44
  };
45
45
  };
46
- // Execution IDs are unique, including after merging runs.
46
+ // Restored records represent the same original execution, not additional model spend.
47
47
  const runCosts = (runs: (EvalRun | JudgeRun)[], results: Results) => {
48
48
  const unique = new Map(runs.map((run) => [run.id, run]));
49
- return sumCosts([...unique.values()].map((run) => costOf(run, results)));
49
+ const executions = new Map<string, EvalRun | JudgeRun>();
50
+ for (const run of unique.values()) {
51
+ const restoration = "restoration" in run ? run.restoration : undefined;
52
+ const key = restoration?.evalRunId ?? run.id;
53
+ const previous = executions.get(key);
54
+ if (!previous || restoration) executions.set(key, run);
55
+ }
56
+ return sumCosts([...executions.values()].map((run) => costOf(run, results)));
50
57
  };
51
58
  const entryOf = (
52
59
  run: BenchmarkRun,
@@ -0,0 +1,91 @@
1
+ import { randomUUID } from "node:crypto";
2
+ import { mkdir, readFile, rm } from "node:fs/promises";
3
+ import { renameSync } from "node:fs";
4
+ import { relative, resolve } from "node:path";
5
+ import type { EvalRun } from "../types";
6
+ import { Results } from "../infra/sqlite";
7
+ import * as archive from "../infra/restore-archive";
8
+ import type { ArchiveRestore } from "../infra/restore-archive";
9
+ import { measureRecording } from "../infra/recording/metrics";
10
+
11
+ /** Restore completed orphan work as a new record; retain the failed coordinator record. */
12
+ export async function restoreEvalRun(
13
+ directory: string,
14
+ evalRunId: string,
15
+ options: ArchiveRestore,
16
+ ): Promise<EvalRun> {
17
+ const root = resolve(directory);
18
+ using results = new Results(resolve(root, "runner.db"));
19
+ const previous = results.evalRun(evalRunId);
20
+ const slot = previous && results.slot(previous.slotId);
21
+ const original = previous?.restoration
22
+ ? results.evalRun(previous.restoration.evalRunId)
23
+ : previous;
24
+ if (!previous || !slot || !slot.active || slot.evalRunId !== previous.id ||
25
+ !original || original.state !== "failed" || !original.interrupted || original.evidence ||
26
+ (previous !== original && (previous.state !== "completed" || !previous.restoration)))
27
+ throw new Error("Select an active interrupted EvalRun or its restored recording");
28
+ if (previous.restoration && (options.databaseHash !== previous.restoration.databaseHash ||
29
+ options.workspaceArchiveHash !== previous.restoration.workspaceArchiveHash))
30
+ throw new Error("A restored recording must retain the same original backup hashes");
31
+ if (results.benchmark!.state === "running" || results.benchmark!.mergedInto)
32
+ throw new Error("Wait for the active BenchmarkRun and use its aggregate directory");
33
+ const events = (await readFile(resolve(root, "eval-runs", original.id, "evidence/events.jsonl"), "utf8"))
34
+ .trim().split("\n").filter(Boolean).map(line => JSON.parse(line).event);
35
+ const creation = events.find(event => event.type === "session.created" && !event.data.parentID);
36
+ if (!creation?.data.sessionID)
37
+ throw new Error("The original recording does not identify its root session");
38
+ const frozen = resolve(root, "inputs", original.input.evalId, original.input.sourceHash, "workspace");
39
+ if (!(await Bun.file(resolve(frozen, "..", "ready")).exists()))
40
+ throw new Error("The frozen starting workspace is missing");
41
+ const item = results.benchmark!.definition.evals.find(item => item.id === original.input.evalId);
42
+ if (!item || item.sourceHash !== original.input.sourceHash)
43
+ throw new Error("The frozen preparation no longer matches the original inputs");
44
+ const staging = resolve(root, "eval-runs", `.restored-${randomUUID()}`);
45
+ await mkdir(staging, { recursive: false });
46
+ try {
47
+ const preparation = resolve(staging, "preparation");
48
+ const initial = await archive.prepareRestoredInitial(
49
+ results.benchmark!.definition, item, original.input.imageId, frozen, preparation,
50
+ );
51
+ const restored = await archive.restoreArchivedCandidate(
52
+ original.input, creation.data.sessionID, initial, options, staging,
53
+ );
54
+ await rm(preparation, { recursive: true, force: true });
55
+ const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(original.startedAt);
56
+ if (!Number.isFinite(elapsedMs) || elapsedMs < 0)
57
+ throw new Error("Native completion predates the original EvalRun");
58
+ const evidence = { ...restored.evidence };
59
+ // This operation does not claim a live candidate. Validate selection again at publication.
60
+ return results.transaction(() => {
61
+ if (results.benchmark!.state === "running" ||
62
+ results.slot(slot.id)?.evalRunId !== previous.id)
63
+ throw new Error("The BenchmarkRun or selected execution changed during restoration");
64
+ const run = results.startEval(slot, original.input);
65
+ const destination = resolve(root, "eval-runs", run.id);
66
+ renameSync(staging, destination);
67
+ evidence.directory = relative(root, resolve(destination, restored.evidence.directory));
68
+ const completed: EvalRun = {
69
+ ...run,
70
+ state: "completed",
71
+ startedAt: original.startedAt,
72
+ completedAt: restored.receipt.completedAt,
73
+ elapsedMs,
74
+ evidence,
75
+ session: { ...restored.session, database: relative(root, resolve(destination, restored.session.database)) },
76
+ restoration: { ...restored.receipt, evalRunId: original.id },
77
+ metrics: measureRecording(restored.events.map((event, sequence) => ({
78
+ event, sequence, time: new Date("created" in event ? event.created : Date.now()).toISOString(),
79
+ })), restored.tools, evidence),
80
+ };
81
+ for (const event of restored.events)
82
+ results.append({ executionId: run.id, stage: "candidate", time: new Date("created" in event ? event.created : Date.now()).toISOString(), event });
83
+ results.finishEval(completed);
84
+ results.saveBenchmark({ ...results.benchmark!, state: "incomplete", scheduledSlotIds: [], updatedAt: new Date().toISOString() });
85
+ return completed;
86
+ });
87
+ } finally {
88
+ // Only this operation's private staging directory; published evidence is never removed.
89
+ await rm(staging, { recursive: true, force: true });
90
+ }
91
+ }
package/src/cli.ts CHANGED
@@ -8,6 +8,7 @@ import {
8
8
  addModels,
9
9
  removeModels,
10
10
  retryEvalRun,
11
+ restoreEvalRun,
11
12
  judgeRun,
12
13
  serveResults,
13
14
  mergeBenchmarkRuns,
@@ -49,6 +50,7 @@ Commands:
49
50
  refresh-model-names <run> Refresh reporting names without running evals
50
51
  retry <run> <eval-run> Retry a candidate execution
51
52
  rejudge <run> <eval-run> Judge the saved evidence again
53
+ restore <run> <eval-run> Restore completed orphan work from hashed backups
52
54
 
53
55
  Options:
54
56
  --benchmark <dir> Benchmark directory (default: current directory)
@@ -64,6 +66,10 @@ Options:
64
66
  --verification Build only the standard verification image (image command)
65
67
  --output <dir> New prepared-input directory (prepare command)
66
68
  --port <port> Viewer port (default: 4173)
69
+ --database <file> Native backup for restore
70
+ --database-hash <sha> SHA-256 of the native backup
71
+ --workspace-archive <file> Workspace tar backup for restore
72
+ --workspace-archive-hash <sha> SHA-256 of the workspace tar
67
73
 
68
74
  Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
69
75
  } else if (command === "--version") {
@@ -170,6 +176,14 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
170
176
  } else if (command === "refresh-model-names") {
171
177
  if (!args[1]) throw new Error("refresh-model-names requires a benchmark run directory");
172
178
  console.log(JSON.stringify(await refreshModelNames(resolve(args[1])), null, 2));
179
+ } else if (command === "restore") {
180
+ const database = option("--database"), databaseHash = option("--database-hash");
181
+ const workspaceArchive = option("--workspace-archive"), workspaceArchiveHash = option("--workspace-archive-hash");
182
+ if (!args[1] || !args[2] || !database || !databaseHash || !workspaceArchive || !workspaceArchiveHash)
183
+ throw new Error("restore requires the run, interrupted EvalRun, native backup, workspace tar, and both SHA-256 hashes");
184
+ console.log(JSON.stringify(await restoreEvalRun(resolve(args[1]), args[2], {
185
+ database: resolve(database), databaseHash, workspaceArchive: resolve(workspaceArchive), workspaceArchiveHash,
186
+ }), null, 2));
173
187
  } else if (command === "retry" || command === "rejudge") {
174
188
  if (!args[1] || !args[2])
175
189
  throw new Error(
@@ -182,7 +196,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
182
196
  console.log(JSON.stringify(result, null, 2));
183
197
  } else
184
198
  throw new Error(
185
- "Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, snapshot",
199
+ "Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, restore, snapshot",
186
200
  );
187
201
  } catch (error) {
188
202
  console.error(error instanceof Error ? error.message : String(error));
package/src/index.ts CHANGED
@@ -68,6 +68,7 @@ export { refreshModelNames } from "./app/refresh-model-names";
68
68
  export type { CostEstimate } from "./app/cost-plan";
69
69
  export { mergeBenchmarkRuns } from "./app/merge-runs";
70
70
  export { retryEvalRun } from "./app/retry-run";
71
+ export { restoreEvalRun } from "./app/restore-eval-run";
71
72
  export { judgeRun, judgeRuns } from "./app/rejudge";
72
73
  export { judgeEvidence, recordEvidence } from "./app/judge-evidence";
73
74
  export {
@@ -0,0 +1,132 @@
1
+ import { Database } from "bun:sqlite";
2
+ import { mkdir, mkdtemp, rm } from "node:fs/promises";
3
+ import { resolve, dirname } from "node:path";
4
+ import type { BenchmarkDefinition, EvalDefinition, EvalRunInput, OpenCodeStreamEvent } from "../types";
5
+ import { EvidenceCapture } from "./evidence";
6
+ import { extractWorkspaceArchive } from "./containers/transfer";
7
+ import { removeCredentials } from "./opencode/auth";
8
+ import { readArchivedSession } from "./opencode/archive";
9
+ import { parseModel, type SessionResult } from "./opencode/session";
10
+ import { hash, writeJson } from "./files";
11
+ import { CandidateContainer } from "./containers/oci";
12
+ import { createSessionDatabase } from "./opencode/host";
13
+
14
+ export type ArchiveRestore = {
15
+ database: string;
16
+ databaseHash: string;
17
+ workspaceArchive: string;
18
+ workspaceArchiveHash: string;
19
+ };
20
+
21
+ /** Replay only frozen author preparation, without credentials or a model prompt. */
22
+ export async function prepareRestoredInitial(
23
+ definition: BenchmarkDefinition,
24
+ item: EvalDefinition,
25
+ imageId: string,
26
+ workspace: string,
27
+ staging: string,
28
+ ) {
29
+ if (!item.settings.prepare?.length) return workspace;
30
+ await mkdir(staging, { recursive: true });
31
+ const database = resolve(staging, "opencode.db");
32
+ await createSessionDatabase(database, []);
33
+ await using container = await CandidateContainer.create(definition.container, imageId);
34
+ await container.prepare(workspace, database, definition.candidate.websearch,
35
+ item.settings.prepare, staging);
36
+ const initial = resolve(staging, "workspace");
37
+ await container.snapshot(initial, staging);
38
+ return initial;
39
+ }
40
+
41
+ /** A backup must belong to the original execution, not merely resemble its answer. */
42
+ export function verifyArchivedInput(
43
+ input: EvalRunInput,
44
+ sessionId: string,
45
+ archived: { result: SessionResult; outcome: string | undefined },
46
+ ) {
47
+ const root = archived.result.sessions.find(session => session.info.id === sessionId);
48
+ const model = parseModel(input.model);
49
+ const prompt = root?.messages.find(message => message.type === "user")?.text;
50
+ if (!root || root.info.parentID || root.info.model?.providerID !== model.providerID ||
51
+ root.info.model?.id !== model.id || root.info.model?.variant !== model.variant ||
52
+ prompt !== input.prompt)
53
+ throw new Error("The native backup does not match the original session, model, and prompt");
54
+ if (archived.outcome !== "succeeded" || archived.result.state !== "completed")
55
+ throw new Error("Restore requires a naturally completed native session");
56
+ }
57
+
58
+ /** Finalize a copied native recording. No candidate program, prompt, or model runs. */
59
+ export async function restoreArchivedCandidate(
60
+ input: EvalRunInput,
61
+ sessionId: string,
62
+ initialWorkspace: string,
63
+ options: ArchiveRestore,
64
+ directory: string,
65
+ ) {
66
+ for (const [file, expected] of [
67
+ [options.database, options.databaseHash],
68
+ [options.workspaceArchive, options.workspaceArchiveHash],
69
+ ]) {
70
+ if (!/^[a-f0-9]{64}$/.test(expected!) || hash(await Bun.file(file!).bytes()) !== expected)
71
+ throw new Error("The recovery backup hash does not match");
72
+ }
73
+ if (Bun.file(options.workspaceArchive).size > 256 * 1024 * 1024)
74
+ throw new Error("The recovery workspace archive exceeds 256 MiB");
75
+ const staging = await mkdtemp(resolve(dirname(directory), ".restore-"));
76
+ try {
77
+ await mkdir(directory, { recursive: true });
78
+ const database = resolve(directory, "opencode.db");
79
+ // SQLite serialization includes committed WAL pages without modifying the backup.
80
+ using source = new Database(options.database, { readonly: true });
81
+ await Bun.write(database, source.serialize());
82
+ removeCredentials(database);
83
+ const events: OpenCodeStreamEvent[] = [];
84
+ const archived = await readArchivedSession(database, sessionId, event => events.push(event));
85
+ verifyArchivedInput(input, sessionId, archived);
86
+ const creation = events.find((event): event is Extract<OpenCodeStreamEvent, { type: "session.created" }> =>
87
+ event.type === "session.created" && event.data.sessionID === sessionId);
88
+ if (!creation) throw new Error("The native backup has no root creation event");
89
+ const completedAt = archived.result.sessions.find(session => session.info.id === sessionId)!.info.time.idle;
90
+ if (typeof completedAt !== "number" || !Number.isFinite(completedAt))
91
+ throw new Error("The native session has no completion timestamp");
92
+ removeCredentials(database);
93
+ const workspace = resolve(staging, "workspace");
94
+ await mkdir(workspace);
95
+ await extractWorkspaceArchive(options.workspaceArchive, workspace, { sourceRoot: "/workspace" });
96
+ const capture = await EvidenceCapture.create(resolve(directory, "evidence"), initialWorkspace,
97
+ { excludeDirectories: ["node_modules"] });
98
+ for (const event of events)
99
+ capture.event(event, new Date("created" in event ? event.created : Date.now()).toISOString());
100
+ const manifest = await capture.finish({
101
+ prompt: input.prompt,
102
+ response: { text: archived.result.text },
103
+ tools: archived.result.tools,
104
+ workspace,
105
+ });
106
+ await writeJson(resolve(directory, "session.json"), archived.result.sessions);
107
+ const receipt = {
108
+ sessionId,
109
+ restoredAt: new Date().toISOString(),
110
+ completedAt: new Date(completedAt).toISOString(),
111
+ databaseHash: options.databaseHash,
112
+ workspaceArchiveHash: options.workspaceArchiveHash,
113
+ initial: "frozen input and author preparation replayed without model calls; not a newly observed candidate snapshot",
114
+ };
115
+ await writeJson(resolve(directory, "restoration.json"), receipt);
116
+ return {
117
+ evidence: { directory: "evidence", hash: manifest.sha256 },
118
+ session: {
119
+ sessionId,
120
+ database: "opencode.db",
121
+ databaseHash: hash(await Bun.file(database).bytes()),
122
+ opencodeVersion: creation.data.version,
123
+ accounting: archived.result.accounting,
124
+ },
125
+ events,
126
+ tools: archived.result.tools,
127
+ receipt,
128
+ };
129
+ } finally {
130
+ await rm(staging, { recursive: true, force: true });
131
+ }
132
+ }
package/src/types.ts CHANGED
@@ -232,6 +232,16 @@ export type EvalRun = {
232
232
  stop?: EvalStop;
233
233
  /** The runner stopped before this session finished; the planner collects it again. */
234
234
  interrupted?: boolean;
235
+ /** A restored native session; the original interrupted coordinator record is retained. */
236
+ restoration?: {
237
+ evalRunId: string;
238
+ sessionId: string;
239
+ restoredAt: string;
240
+ completedAt: string;
241
+ databaseHash: string;
242
+ workspaceArchiveHash: string;
243
+ initial: string;
244
+ };
235
245
  };
236
246
  export type CriterionDefinition = { id: string; name: string };
237
247
  export type EvidenceCitation = (