@hona/openeval 0.5.9 → 0.5.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.9",
3
+ "version": "0.5.10",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -43,10 +43,17 @@ const costOf = (run: EvalRun | JudgeRun, results?: Results): Cost => {
43
43
  complete: usd !== undefined,
44
44
  };
45
45
  };
46
- // Execution IDs are unique, including after merging runs.
46
+ // Restored records represent the same original execution, not additional model spend.
47
47
  const runCosts = (runs: (EvalRun | JudgeRun)[], results: Results) => {
48
48
  const unique = new Map(runs.map((run) => [run.id, run]));
49
- return sumCosts([...unique.values()].map((run) => costOf(run, results)));
49
+ const executions = new Map<string, EvalRun | JudgeRun>();
50
+ for (const run of unique.values()) {
51
+ const restoration = "restoration" in run ? run.restoration : undefined;
52
+ const key = restoration?.evalRunId ?? run.id;
53
+ const previous = executions.get(key);
54
+ if (!previous || restoration) executions.set(key, run);
55
+ }
56
+ return sumCosts([...executions.values()].map((run) => costOf(run, results)));
50
57
  };
51
58
  const entryOf = (
52
59
  run: BenchmarkRun,
@@ -18,26 +18,41 @@ export async function restoreEvalRun(
18
18
  using results = new Results(resolve(root, "runner.db"));
19
19
  const previous = results.evalRun(evalRunId);
20
20
  const slot = previous && results.slot(previous.slotId);
21
+ const original = previous?.restoration
22
+ ? results.evalRun(previous.restoration.evalRunId)
23
+ : previous;
21
24
  if (!previous || !slot || !slot.active || slot.evalRunId !== previous.id ||
22
- previous.state !== "failed" || !previous.interrupted || previous.evidence)
23
- throw new Error("Select an active interrupted EvalRun without finalized evidence");
25
+ !original || original.state !== "failed" || !original.interrupted || original.evidence ||
26
+ (previous !== original && (previous.state !== "completed" || !previous.restoration)))
27
+ throw new Error("Select an active interrupted EvalRun or its restored recording");
28
+ if (previous.restoration && (options.databaseHash !== previous.restoration.databaseHash ||
29
+ options.workspaceArchiveHash !== previous.restoration.workspaceArchiveHash))
30
+ throw new Error("A restored recording must retain the same original backup hashes");
24
31
  if (results.benchmark!.state === "running" || results.benchmark!.mergedInto)
25
32
  throw new Error("Wait for the active BenchmarkRun and use its aggregate directory");
26
- const events = (await readFile(resolve(root, "eval-runs", previous.id, "evidence/events.jsonl"), "utf8"))
33
+ const events = (await readFile(resolve(root, "eval-runs", original.id, "evidence/events.jsonl"), "utf8"))
27
34
  .trim().split("\n").filter(Boolean).map(line => JSON.parse(line).event);
28
35
  const creation = events.find(event => event.type === "session.created" && !event.data.parentID);
29
36
  if (!creation?.data.sessionID)
30
37
  throw new Error("The original recording does not identify its root session");
31
- const initial = resolve(root, "inputs", previous.input.evalId, previous.input.sourceHash, "workspace");
32
- if (!(await Bun.file(resolve(initial, "..", "ready")).exists()))
38
+ const frozen = resolve(root, "inputs", original.input.evalId, original.input.sourceHash, "workspace");
39
+ if (!(await Bun.file(resolve(frozen, "..", "ready")).exists()))
33
40
  throw new Error("The frozen starting workspace is missing");
41
+ const item = results.benchmark!.definition.evals.find(item => item.id === original.input.evalId);
42
+ if (!item || item.sourceHash !== original.input.sourceHash)
43
+ throw new Error("The frozen preparation no longer matches the original inputs");
34
44
  const staging = resolve(root, "eval-runs", `.restored-${randomUUID()}`);
35
45
  await mkdir(staging, { recursive: false });
36
46
  try {
47
+ const preparation = resolve(staging, "preparation");
48
+ const initial = await archive.prepareRestoredInitial(
49
+ results.benchmark!.definition, item, original.input.imageId, frozen, preparation,
50
+ );
37
51
  const restored = await archive.restoreArchivedCandidate(
38
- previous.input, creation.data.sessionID, initial, options, staging,
52
+ original.input, creation.data.sessionID, initial, options, staging,
39
53
  );
40
- const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(previous.startedAt);
54
+ await rm(preparation, { recursive: true, force: true });
55
+ const elapsedMs = Date.parse(restored.receipt.completedAt) - Date.parse(original.startedAt);
41
56
  if (!Number.isFinite(elapsedMs) || elapsedMs < 0)
42
57
  throw new Error("Native completion predates the original EvalRun");
43
58
  const evidence = { ...restored.evidence };
@@ -46,19 +61,19 @@ export async function restoreEvalRun(
46
61
  if (results.benchmark!.state === "running" ||
47
62
  results.slot(slot.id)?.evalRunId !== previous.id)
48
63
  throw new Error("The BenchmarkRun or selected execution changed during restoration");
49
- const run = results.startEval(slot, previous.input);
64
+ const run = results.startEval(slot, original.input);
50
65
  const destination = resolve(root, "eval-runs", run.id);
51
66
  renameSync(staging, destination);
52
67
  evidence.directory = relative(root, resolve(destination, restored.evidence.directory));
53
68
  const completed: EvalRun = {
54
69
  ...run,
55
70
  state: "completed",
56
- startedAt: previous.startedAt,
71
+ startedAt: original.startedAt,
57
72
  completedAt: restored.receipt.completedAt,
58
73
  elapsedMs,
59
74
  evidence,
60
75
  session: { ...restored.session, database: relative(root, resolve(destination, restored.session.database)) },
61
- restoration: { ...restored.receipt, evalRunId: previous.id },
76
+ restoration: { ...restored.receipt, evalRunId: original.id },
62
77
  metrics: measureRecording(restored.events.map((event, sequence) => ({
63
78
  event, sequence, time: new Date("created" in event ? event.created : Date.now()).toISOString(),
64
79
  })), restored.tools, evidence),
@@ -1,13 +1,15 @@
1
1
  import { Database } from "bun:sqlite";
2
2
  import { mkdir, mkdtemp, rm } from "node:fs/promises";
3
3
  import { resolve, dirname } from "node:path";
4
- import type { EvalRunInput, OpenCodeStreamEvent } from "../types";
4
+ import type { BenchmarkDefinition, EvalDefinition, EvalRunInput, OpenCodeStreamEvent } from "../types";
5
5
  import { EvidenceCapture } from "./evidence";
6
6
  import { extractWorkspaceArchive } from "./containers/transfer";
7
7
  import { removeCredentials } from "./opencode/auth";
8
8
  import { readArchivedSession } from "./opencode/archive";
9
9
  import { parseModel, type SessionResult } from "./opencode/session";
10
10
  import { hash, writeJson } from "./files";
11
+ import { CandidateContainer } from "./containers/oci";
12
+ import { createSessionDatabase } from "./opencode/host";
11
13
 
12
14
  export type ArchiveRestore = {
13
15
  database: string;
@@ -16,6 +18,26 @@ export type ArchiveRestore = {
16
18
  workspaceArchiveHash: string;
17
19
  };
18
20
 
21
+ /** Replay only frozen author preparation, without credentials or a model prompt. */
22
+ export async function prepareRestoredInitial(
23
+ definition: BenchmarkDefinition,
24
+ item: EvalDefinition,
25
+ imageId: string,
26
+ workspace: string,
27
+ staging: string,
28
+ ) {
29
+ if (!item.settings.prepare?.length) return workspace;
30
+ await mkdir(staging, { recursive: true });
31
+ const database = resolve(staging, "opencode.db");
32
+ await createSessionDatabase(database, []);
33
+ await using container = await CandidateContainer.create(definition.container, imageId);
34
+ await container.prepare(workspace, database, definition.candidate.websearch,
35
+ item.settings.prepare, staging);
36
+ const initial = resolve(staging, "workspace");
37
+ await container.snapshot(initial, staging);
38
+ return initial;
39
+ }
40
+
19
41
  /** A backup must belong to the original execution, not merely resemble its answer. */
20
42
  export function verifyArchivedInput(
21
43
  input: EvalRunInput,
@@ -88,7 +110,7 @@ export async function restoreArchivedCandidate(
88
110
  completedAt: new Date(completedAt).toISOString(),
89
111
  databaseHash: options.databaseHash,
90
112
  workspaceArchiveHash: options.workspaceArchiveHash,
91
- initial: "frozen prepared input, not a newly observed candidate snapshot",
113
+ initial: "frozen input and author preparation replayed without model calls; not a newly observed candidate snapshot",
92
114
  };
93
115
  await writeJson(resolve(directory, "restoration.json"), receipt);
94
116
  return {