@hona/openeval 0.5.6 → 0.5.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/VERIFICATION.md +3 -1
- package/package.json +1 -1
- package/src/app/model-name-metadata.ts +10 -0
- package/src/app/refresh-model-names.ts +21 -0
- package/src/cli.ts +6 -1
- package/src/index.ts +1 -0
- package/src/infra/judging/code-source.ts +4 -2
package/VERIFICATION.md
CHANGED
|
@@ -93,7 +93,9 @@ evidence. Do not put evaluator scripts in candidate workspaces.
|
|
|
93
93
|
## Readable judgments and controls
|
|
94
94
|
|
|
95
95
|
Retained code judgments stay scored when a metadata-only identity change leaves
|
|
96
|
-
the
|
|
96
|
+
the comment-free self-contained executable unchanged. Generated bundle comments
|
|
97
|
+
and debug IDs are excluded; string literals remain executable input.
|
|
98
|
+
The reader verifies both recorded
|
|
97
99
|
identities and rejects changed source, external package imports, unknown metadata,
|
|
98
100
|
or a different execution protocol. This check makes no model calls and rewrites
|
|
99
101
|
no evidence. A completed code verification is still required. Package/lockfile
|
package/package.json
CHANGED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { ModelNames, BenchmarkRun } from "../types";
|
|
2
|
+
|
|
3
|
+
/** Reporting metadata only: no new attempts, selections, fingerprints, or heartbeat. */
|
|
4
|
+
export function withModelNames(benchmark: BenchmarkRun, names: ModelNames): BenchmarkRun {
|
|
5
|
+
const known = new Set(benchmark.definition.models.map(model => model.split("#")[0]));
|
|
6
|
+
for (const [model, name] of Object.entries(names))
|
|
7
|
+
if (!known.has(model) || typeof name !== "string" || !name.trim())
|
|
8
|
+
throw new Error("Model display names must name a recorded model and have non-empty text");
|
|
9
|
+
return { ...benchmark, modelNames: { ...benchmark.modelNames, ...names } };
|
|
10
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import type { ModelNames } from "../types";
|
|
3
|
+
import { Results } from "../infra/sqlite";
|
|
4
|
+
import { readModelNames } from "../infra/containers/catalog";
|
|
5
|
+
import { withModelNames } from "./model-name-metadata";
|
|
6
|
+
|
|
7
|
+
/** Refresh catalog names, or apply names read from an authoritative provider catalog. */
|
|
8
|
+
export async function refreshModelNames(directory: string, names?: ModelNames) {
|
|
9
|
+
using results = new Results(resolve(directory, "runner.db"));
|
|
10
|
+
const current = results.benchmark;
|
|
11
|
+
if (!current) throw new Error("No benchmark run in this directory");
|
|
12
|
+
if (current.mergedInto) throw new Error("Refresh names on the aggregate benchmark run");
|
|
13
|
+
const fresh = names ?? await readModelNames(current.definition, current.runtime.imageId);
|
|
14
|
+
// Re-read under the write lock: a live runner may have updated progress while
|
|
15
|
+
// catalog discovery was in flight. Never replace that progress or its heartbeat.
|
|
16
|
+
return results.transaction(() => {
|
|
17
|
+
const benchmark = withModelNames(results.benchmark!, fresh);
|
|
18
|
+
results.saveBenchmark(benchmark);
|
|
19
|
+
return { directory: resolve(directory), modelNames: benchmark.modelNames };
|
|
20
|
+
});
|
|
21
|
+
}
|
package/src/cli.ts
CHANGED
|
@@ -15,6 +15,7 @@ import {
|
|
|
15
15
|
buildVerificationImage,
|
|
16
16
|
VERIFICATION_IMAGE,
|
|
17
17
|
prepareInputs,
|
|
18
|
+
refreshModelNames,
|
|
18
19
|
} from "./index";
|
|
19
20
|
import type { ModelRef } from "./index";
|
|
20
21
|
|
|
@@ -45,6 +46,7 @@ Commands:
|
|
|
45
46
|
merge-runs <to> <from> Merge results into one aggregate
|
|
46
47
|
add-models <run> Add models with repeated --model flags
|
|
47
48
|
remove-models <run> Remove active models while retaining their evidence
|
|
49
|
+
refresh-model-names <run> Refresh reporting names without running evals
|
|
48
50
|
retry <run> <eval-run> Retry a candidate execution
|
|
49
51
|
rejudge <run> <eval-run> Judge the saved evidence again
|
|
50
52
|
|
|
@@ -165,6 +167,9 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
165
167
|
retiredSlots: result.retiredSlots,
|
|
166
168
|
status: result.benchmark.state,
|
|
167
169
|
}, null, 2));
|
|
170
|
+
} else if (command === "refresh-model-names") {
|
|
171
|
+
if (!args[1]) throw new Error("refresh-model-names requires a benchmark run directory");
|
|
172
|
+
console.log(JSON.stringify(await refreshModelNames(resolve(args[1])), null, 2));
|
|
168
173
|
} else if (command === "retry" || command === "rejudge") {
|
|
169
174
|
if (!args[1] || !args[2])
|
|
170
175
|
throw new Error(
|
|
@@ -177,7 +182,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
177
182
|
console.log(JSON.stringify(result, null, 2));
|
|
178
183
|
} else
|
|
179
184
|
throw new Error(
|
|
180
|
-
"Commands: image, plan, run, view, add-models, remove-models, merge-runs, retry, rejudge, snapshot",
|
|
185
|
+
"Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, snapshot",
|
|
181
186
|
);
|
|
182
187
|
} catch (error) {
|
|
183
188
|
console.error(error instanceof Error ? error.message : String(error));
|
package/src/index.ts
CHANGED
|
@@ -64,6 +64,7 @@ export {
|
|
|
64
64
|
} from "./app/run-benchmark";
|
|
65
65
|
export { addModels } from "./app/add-models";
|
|
66
66
|
export { removeModels } from "./app/remove-models";
|
|
67
|
+
export { refreshModelNames } from "./app/refresh-model-names";
|
|
67
68
|
export type { CostEstimate } from "./app/cost-plan";
|
|
68
69
|
export { mergeBenchmarkRuns } from "./app/merge-runs";
|
|
69
70
|
export { retryEvalRun } from "./app/retry-run";
|
|
@@ -18,7 +18,8 @@ function portableDirectories(source: string) {
|
|
|
18
18
|
}
|
|
19
19
|
|
|
20
20
|
/** An unchanged self-contained executable can retain judgments whose older
|
|
21
|
-
* identity included host package manifests and lockfiles.
|
|
21
|
+
* identity included host package manifests and lockfiles. Generated comments
|
|
22
|
+
* and debug IDs are not executable input. This is a read-only
|
|
22
23
|
* identity check, not a rewrite of recorded code or a changed scoring rule.
|
|
23
24
|
* External packages still require their exact executable identity. */
|
|
24
25
|
export function recordedCodeMatches(
|
|
@@ -27,7 +28,8 @@ export function recordedCodeMatches(
|
|
|
27
28
|
) {
|
|
28
29
|
if (!recorded) return false;
|
|
29
30
|
if (recorded.hash === expected.hash) return true;
|
|
30
|
-
if (!expected.source.trim() ||
|
|
31
|
+
if (!expected.source.trim() ||
|
|
32
|
+
printer.transformSync(expected.source) !== printer.transformSync(recorded.source) ||
|
|
31
33
|
Object.keys(expected.dependencies).length !== 0) return false;
|
|
32
34
|
const imports = new Bun.Transpiler({ loader: "js" }).scan(expected.source).imports;
|
|
33
35
|
if (imports.some(item => !item.path.startsWith("node:") && !item.path.startsWith("bun:"))) return false;
|