@hona/openeval 0.5.6 → 0.5.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/VERIFICATION.md CHANGED
@@ -93,7 +93,9 @@ evidence. Do not put evaluator scripts in candidate workspaces.
93
93
  ## Readable judgments and controls
94
94
 
95
95
  Retained code judgments stay scored when a metadata-only identity change leaves
96
- the exact self-contained executable unchanged. The reader verifies both recorded
96
+ the comment-free self-contained executable unchanged. Generated bundle comments
97
+ and debug IDs are excluded; string literals remain executable input.
98
+ The reader verifies both recorded
97
99
  identities and rejects changed source, external package imports, unknown metadata,
98
100
  or a different execution protocol. This check makes no model calls and rewrites
99
101
  no evidence. A completed code verification is still required. Package/lockfile
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.6",
3
+ "version": "0.5.8",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -0,0 +1,10 @@
1
+ import type { ModelNames, BenchmarkRun } from "../types";
2
+
3
+ /** Reporting metadata only: no new attempts, selections, fingerprints, or heartbeat. */
4
+ export function withModelNames(benchmark: BenchmarkRun, names: ModelNames): BenchmarkRun {
5
+ const known = new Set(benchmark.definition.models.map(model => model.split("#")[0]));
6
+ for (const [model, name] of Object.entries(names))
7
+ if (!known.has(model) || typeof name !== "string" || !name.trim())
8
+ throw new Error("Model display names must name a recorded model and have non-empty text");
9
+ return { ...benchmark, modelNames: { ...benchmark.modelNames, ...names } };
10
+ }
@@ -0,0 +1,21 @@
1
+ import { resolve } from "node:path";
2
+ import type { ModelNames } from "../types";
3
+ import { Results } from "../infra/sqlite";
4
+ import { readModelNames } from "../infra/containers/catalog";
5
+ import { withModelNames } from "./model-name-metadata";
6
+
7
+ /** Refresh catalog names, or apply names read from an authoritative provider catalog. */
8
+ export async function refreshModelNames(directory: string, names?: ModelNames) {
9
+ using results = new Results(resolve(directory, "runner.db"));
10
+ const current = results.benchmark;
11
+ if (!current) throw new Error("No benchmark run in this directory");
12
+ if (current.mergedInto) throw new Error("Refresh names on the aggregate benchmark run");
13
+ const fresh = names ?? await readModelNames(current.definition, current.runtime.imageId);
14
+ // Re-read under the write lock: a live runner may have updated progress while
15
+ // catalog discovery was in flight. Never replace that progress or its heartbeat.
16
+ return results.transaction(() => {
17
+ const benchmark = withModelNames(results.benchmark!, fresh);
18
+ results.saveBenchmark(benchmark);
19
+ return { directory: resolve(directory), modelNames: benchmark.modelNames };
20
+ });
21
+ }
package/src/cli.ts CHANGED
@@ -15,6 +15,7 @@ import {
15
15
  buildVerificationImage,
16
16
  VERIFICATION_IMAGE,
17
17
  prepareInputs,
18
+ refreshModelNames,
18
19
  } from "./index";
19
20
  import type { ModelRef } from "./index";
20
21
 
@@ -45,6 +46,7 @@ Commands:
45
46
  merge-runs <to> <from> Merge results into one aggregate
46
47
  add-models <run> Add models with repeated --model flags
47
48
  remove-models <run> Remove active models while retaining their evidence
49
+ refresh-model-names <run> Refresh reporting names without running evals
48
50
  retry <run> <eval-run> Retry a candidate execution
49
51
  rejudge <run> <eval-run> Judge the saved evidence again
50
52
 
@@ -165,6 +167,9 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
165
167
  retiredSlots: result.retiredSlots,
166
168
  status: result.benchmark.state,
167
169
  }, null, 2));
170
+ } else if (command === "refresh-model-names") {
171
+ if (!args[1]) throw new Error("refresh-model-names requires a benchmark run directory");
172
+ console.log(JSON.stringify(await refreshModelNames(resolve(args[1])), null, 2));
168
173
  } else if (command === "retry" || command === "rejudge") {
169
174
  if (!args[1] || !args[2])
170
175
  throw new Error(
@@ -177,7 +182,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
177
182
  console.log(JSON.stringify(result, null, 2));
178
183
  } else
179
184
  throw new Error(
180
- "Commands: image, plan, run, view, add-models, remove-models, merge-runs, retry, rejudge, snapshot",
185
+ "Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, snapshot",
181
186
  );
182
187
  } catch (error) {
183
188
  console.error(error instanceof Error ? error.message : String(error));
package/src/index.ts CHANGED
@@ -64,6 +64,7 @@ export {
64
64
  } from "./app/run-benchmark";
65
65
  export { addModels } from "./app/add-models";
66
66
  export { removeModels } from "./app/remove-models";
67
+ export { refreshModelNames } from "./app/refresh-model-names";
67
68
  export type { CostEstimate } from "./app/cost-plan";
68
69
  export { mergeBenchmarkRuns } from "./app/merge-runs";
69
70
  export { retryEvalRun } from "./app/retry-run";
@@ -18,7 +18,8 @@ function portableDirectories(source: string) {
18
18
  }
19
19
 
20
20
  /** An unchanged self-contained executable can retain judgments whose older
21
- * identity included host package manifests and lockfiles. This is a read-only
21
+ * identity included host package manifests and lockfiles. Generated comments
22
+ * and debug IDs are not executable input. This is a read-only
22
23
  * identity check, not a rewrite of recorded code or a changed scoring rule.
23
24
  * External packages still require their exact executable identity. */
24
25
  export function recordedCodeMatches(
@@ -27,7 +28,8 @@ export function recordedCodeMatches(
27
28
  ) {
28
29
  if (!recorded) return false;
29
30
  if (recorded.hash === expected.hash) return true;
30
- if (!expected.source.trim() || expected.source !== recorded.source ||
31
+ if (!expected.source.trim() ||
32
+ printer.transformSync(expected.source) !== printer.transformSync(recorded.source) ||
31
33
  Object.keys(expected.dependencies).length !== 0) return false;
32
34
  const imports = new Bun.Transpiler({ loader: "js" }).scan(expected.source).imports;
33
35
  if (imports.some(item => !item.path.startsWith("node:") && !item.path.startsWith("bun:"))) return false;