@hona/openeval 0.5.5 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/VERIFICATION.md CHANGED
@@ -92,6 +92,13 @@ evidence. Do not put evaluator scripts in candidate workspaces.
92
92
 
93
93
  ## Readable judgments and controls
94
94
 
95
+ Retained code judgments stay scored when a metadata-only identity change leaves
96
+ the exact self-contained executable unchanged. The reader verifies both recorded
97
+ identities and rejects changed source, external package imports, unknown metadata,
98
+ or a different execution protocol. This check makes no model calls and rewrites
99
+ no evidence. A completed code verification is still required. Package/lockfile
100
+ metadata alone must not make an unchanged historical scorecard appear unscored.
101
+
95
102
  `openeval prepare --output <new-directory>` assembles real candidate inputs and
96
103
  runs declared preparation in the candidate image, then archives the prepared
97
104
  workspace. `--only-eval` scopes it. This makes zero model calls, creates no
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.5",
3
+ "version": "0.5.7",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -0,0 +1,10 @@
1
+ import type { ModelNames, BenchmarkRun } from "../types";
2
+
3
+ /** Reporting metadata only: no new attempts, selections, fingerprints, or heartbeat. */
4
+ export function withModelNames(benchmark: BenchmarkRun, names: ModelNames): BenchmarkRun {
5
+ const known = new Set(benchmark.definition.models.map(model => model.split("#")[0]));
6
+ for (const [model, name] of Object.entries(names))
7
+ if (!known.has(model) || typeof name !== "string" || !name.trim())
8
+ throw new Error("Model display names must name a recorded model and have non-empty text");
9
+ return { ...benchmark, modelNames: { ...benchmark.modelNames, ...names } };
10
+ }
@@ -0,0 +1,21 @@
1
+ import { resolve } from "node:path";
2
+ import type { ModelNames } from "../types";
3
+ import { Results } from "../infra/sqlite";
4
+ import { readModelNames } from "../infra/containers/catalog";
5
+ import { withModelNames } from "./model-name-metadata";
6
+
7
+ /** Refresh catalog names, or apply names read from an authoritative provider catalog. */
8
+ export async function refreshModelNames(directory: string, names?: ModelNames) {
9
+ using results = new Results(resolve(directory, "runner.db"));
10
+ const current = results.benchmark;
11
+ if (!current) throw new Error("No benchmark run in this directory");
12
+ if (current.mergedInto) throw new Error("Refresh names on the aggregate benchmark run");
13
+ const fresh = names ?? await readModelNames(current.definition, current.runtime.imageId);
14
+ // Re-read under the write lock: a live runner may have updated progress while
15
+ // catalog discovery was in flight. Never replace that progress or its heartbeat.
16
+ return results.transaction(() => {
17
+ const benchmark = withModelNames(results.benchmark!, fresh);
18
+ results.saveBenchmark(benchmark);
19
+ return { directory: resolve(directory), modelNames: benchmark.modelNames };
20
+ });
21
+ }
package/src/app/scores.ts CHANGED
@@ -1,7 +1,9 @@
1
1
  import type { BenchmarkDefinition, JudgeRun, Slot } from "../types";
2
+ import type { CodeJudgeDefinition } from "../judge-context";
2
3
  import { modelScore } from "../view";
3
4
  import { isScored } from "../judgment";
4
5
  import { categoryKey, inCategories } from "../criterion-categories";
6
+ import { recordedCodeMatches } from "../infra/judging/code-source";
5
7
 
6
8
  /** Scores are derived from a single selection snapshot; every eval has equal weight. */
7
9
  export function benchmarkScores(
@@ -14,12 +16,23 @@ export function benchmarkScores(
14
16
  const evals = definition.evals.filter(
15
17
  (item) => !evalId || item.id === evalId,
16
18
  );
17
- // A code judgment counts only when it ran the eval's current judge.ts to completion.
19
+ const matches = new WeakMap<CodeJudgeDefinition, WeakMap<CodeJudgeDefinition, boolean>>();
20
+ const equivalent = (expected: CodeJudgeDefinition, recorded: CodeJudgeDefinition | undefined) => {
21
+ if (!recorded) return false;
22
+ let previous = matches.get(expected);
23
+ if (!previous) { previous = new WeakMap(); matches.set(expected, previous); }
24
+ const cached = previous.get(recorded);
25
+ if (cached !== undefined) return cached;
26
+ const value = recordedCodeMatches(expected, recorded);
27
+ previous.set(recorded, value);
28
+ return value;
29
+ };
30
+ // A code judgment counts only after completed verification of the matching executable.
18
31
  const current = (item: (typeof evals)[number], slot: Slot) => {
19
32
  const judge = slot.judgeRunId ? index.get(slot.judgeRunId) : undefined;
20
33
  return !item.code ||
21
34
  (judge?.code?.state === "completed" &&
22
- judge.input.code?.hash === item.code.hash)
35
+ equivalent(item.code, judge.input.code))
23
36
  ? judge
24
37
  : undefined;
25
38
  };
package/src/cli.ts CHANGED
@@ -15,6 +15,7 @@ import {
15
15
  buildVerificationImage,
16
16
  VERIFICATION_IMAGE,
17
17
  prepareInputs,
18
+ refreshModelNames,
18
19
  } from "./index";
19
20
  import type { ModelRef } from "./index";
20
21
 
@@ -45,6 +46,7 @@ Commands:
45
46
  merge-runs <to> <from> Merge results into one aggregate
46
47
  add-models <run> Add models with repeated --model flags
47
48
  remove-models <run> Remove active models while retaining their evidence
49
+ refresh-model-names <run> Refresh reporting names without running evals
48
50
  retry <run> <eval-run> Retry a candidate execution
49
51
  rejudge <run> <eval-run> Judge the saved evidence again
50
52
 
@@ -165,6 +167,9 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
165
167
  retiredSlots: result.retiredSlots,
166
168
  status: result.benchmark.state,
167
169
  }, null, 2));
170
+ } else if (command === "refresh-model-names") {
171
+ if (!args[1]) throw new Error("refresh-model-names requires a benchmark run directory");
172
+ console.log(JSON.stringify(await refreshModelNames(resolve(args[1])), null, 2));
168
173
  } else if (command === "retry" || command === "rejudge") {
169
174
  if (!args[1] || !args[2])
170
175
  throw new Error(
@@ -177,7 +182,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
177
182
  console.log(JSON.stringify(result, null, 2));
178
183
  } else
179
184
  throw new Error(
180
- "Commands: image, plan, run, view, add-models, remove-models, merge-runs, retry, rejudge, snapshot",
185
+ "Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, snapshot",
181
186
  );
182
187
  } catch (error) {
183
188
  console.error(error instanceof Error ? error.message : String(error));
package/src/index.ts CHANGED
@@ -64,6 +64,7 @@ export {
64
64
  } from "./app/run-benchmark";
65
65
  export { addModels } from "./app/add-models";
66
66
  export { removeModels } from "./app/remove-models";
67
+ export { refreshModelNames } from "./app/refresh-model-names";
67
68
  export type { CostEstimate } from "./app/cost-plan";
68
69
  export { mergeBenchmarkRuns } from "./app/merge-runs";
69
70
  export { retryEvalRun } from "./app/retry-run";
@@ -10,6 +10,37 @@ const printer = new Bun.Transpiler({
10
10
  deadCodeElimination: false,
11
11
  });
12
12
 
13
+ function portableDirectories(source: string) {
14
+ return source.replace(
15
+ /^\s*var __(?:dirname|filename) = .*$/gm,
16
+ (line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
17
+ );
18
+ }
19
+
20
+ /** An unchanged self-contained executable can retain judgments whose older
21
+ * identity included host package manifests and lockfiles. This is a read-only
22
+ * identity check, not a rewrite of recorded code or a changed scoring rule.
23
+ * External packages still require their exact executable identity. */
24
+ export function recordedCodeMatches(
25
+ expected: CodeJudgeDefinition,
26
+ recorded: CodeJudgeDefinition | undefined,
27
+ ) {
28
+ if (!recorded) return false;
29
+ if (recorded.hash === expected.hash) return true;
30
+ if (!expected.source.trim() || expected.source !== recorded.source ||
31
+ Object.keys(expected.dependencies).length !== 0) return false;
32
+ const imports = new Bun.Transpiler({ loader: "js" }).scan(expected.source).imports;
33
+ if (imports.some(item => !item.path.startsWith("node:") && !item.path.startsWith("bun:"))) return false;
34
+ const metadata = Object.keys(recorded.dependencies);
35
+ if (!metadata.length || metadata.some(path =>
36
+ !/(?:^|[\\/])(?:package\.json|bun\.lockb?|package-lock\.json|pnpm-lock\.yaml|yarn\.lock)$/.test(path))) return false;
37
+ // Verify both identities rather than accepting an arbitrary mismatched hash.
38
+ return CODE_JUDGE_PROTOCOL === 1 &&
39
+ recorded.hash === fingerprint({ source: recorded.source, dependencies: recorded.dependencies }) &&
40
+ expected.hash === fingerprint({ protocol: CODE_JUDGE_PROTOCOL,
41
+ source: printer.transformSync(portableDirectories(expected.source)) });
42
+ }
43
+
13
44
  /** The package that provides an external import, as `name@version/path`. */
14
45
  async function packageSpecifier(path: string, from: string) {
15
46
  for (let directory = dirname(path); ; directory = dirname(directory)) {
@@ -93,10 +124,7 @@ export async function compileCodeJudge(
93
124
  );
94
125
  }
95
126
  // Bun writes __dirname and __filename as absolute paths. Like other runtime file inputs, they are not identity.
96
- portable = portable.replace(
97
- /^\s*var __(?:dirname|filename) = .*$/gm,
98
- (line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
99
- );
127
+ portable = portableDirectories(portable);
100
128
  return {
101
129
  file,
102
130
  source,