@hona/openeval 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/VERIFICATION.md +7 -0
- package/package.json +1 -1
- package/src/app/model-name-metadata.ts +10 -0
- package/src/app/refresh-model-names.ts +21 -0
- package/src/app/scores.ts +15 -2
- package/src/cli.ts +6 -1
- package/src/index.ts +1 -0
- package/src/infra/judging/code-source.ts +32 -4
package/VERIFICATION.md
CHANGED
|
@@ -92,6 +92,13 @@ evidence. Do not put evaluator scripts in candidate workspaces.
|
|
|
92
92
|
|
|
93
93
|
## Readable judgments and controls
|
|
94
94
|
|
|
95
|
+
Retained code judgments stay scored when a metadata-only identity change leaves
|
|
96
|
+
the exact self-contained executable unchanged. The reader verifies both recorded
|
|
97
|
+
identities and rejects changed source, external package imports, unknown metadata,
|
|
98
|
+
or a different execution protocol. This check makes no model calls and rewrites
|
|
99
|
+
no evidence. A completed code verification is still required. Package/lockfile
|
|
100
|
+
metadata alone must not make an unchanged historical scorecard appear unscored.
|
|
101
|
+
|
|
95
102
|
`openeval prepare --output <new-directory>` assembles real candidate inputs and
|
|
96
103
|
runs declared preparation in the candidate image, then archives the prepared
|
|
97
104
|
workspace. `--only-eval` scopes it. This makes zero model calls, creates no
|
package/package.json
CHANGED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { ModelNames, BenchmarkRun } from "../types";
|
|
2
|
+
|
|
3
|
+
/** Reporting metadata only: no new attempts, selections, fingerprints, or heartbeat. */
|
|
4
|
+
export function withModelNames(benchmark: BenchmarkRun, names: ModelNames): BenchmarkRun {
|
|
5
|
+
const known = new Set(benchmark.definition.models.map(model => model.split("#")[0]));
|
|
6
|
+
for (const [model, name] of Object.entries(names))
|
|
7
|
+
if (!known.has(model) || typeof name !== "string" || !name.trim())
|
|
8
|
+
throw new Error("Model display names must name a recorded model and have non-empty text");
|
|
9
|
+
return { ...benchmark, modelNames: { ...benchmark.modelNames, ...names } };
|
|
10
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import type { ModelNames } from "../types";
|
|
3
|
+
import { Results } from "../infra/sqlite";
|
|
4
|
+
import { readModelNames } from "../infra/containers/catalog";
|
|
5
|
+
import { withModelNames } from "./model-name-metadata";
|
|
6
|
+
|
|
7
|
+
/** Refresh catalog names, or apply names read from an authoritative provider catalog. */
|
|
8
|
+
export async function refreshModelNames(directory: string, names?: ModelNames) {
|
|
9
|
+
using results = new Results(resolve(directory, "runner.db"));
|
|
10
|
+
const current = results.benchmark;
|
|
11
|
+
if (!current) throw new Error("No benchmark run in this directory");
|
|
12
|
+
if (current.mergedInto) throw new Error("Refresh names on the aggregate benchmark run");
|
|
13
|
+
const fresh = names ?? await readModelNames(current.definition, current.runtime.imageId);
|
|
14
|
+
// Re-read under the write lock: a live runner may have updated progress while
|
|
15
|
+
// catalog discovery was in flight. Never replace that progress or its heartbeat.
|
|
16
|
+
return results.transaction(() => {
|
|
17
|
+
const benchmark = withModelNames(results.benchmark!, fresh);
|
|
18
|
+
results.saveBenchmark(benchmark);
|
|
19
|
+
return { directory: resolve(directory), modelNames: benchmark.modelNames };
|
|
20
|
+
});
|
|
21
|
+
}
|
package/src/app/scores.ts
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
import type { BenchmarkDefinition, JudgeRun, Slot } from "../types";
|
|
2
|
+
import type { CodeJudgeDefinition } from "../judge-context";
|
|
2
3
|
import { modelScore } from "../view";
|
|
3
4
|
import { isScored } from "../judgment";
|
|
4
5
|
import { categoryKey, inCategories } from "../criterion-categories";
|
|
6
|
+
import { recordedCodeMatches } from "../infra/judging/code-source";
|
|
5
7
|
|
|
6
8
|
/** Scores are derived from a single selection snapshot; every eval has equal weight. */
|
|
7
9
|
export function benchmarkScores(
|
|
@@ -14,12 +16,23 @@ export function benchmarkScores(
|
|
|
14
16
|
const evals = definition.evals.filter(
|
|
15
17
|
(item) => !evalId || item.id === evalId,
|
|
16
18
|
);
|
|
17
|
-
|
|
19
|
+
const matches = new WeakMap<CodeJudgeDefinition, WeakMap<CodeJudgeDefinition, boolean>>();
|
|
20
|
+
const equivalent = (expected: CodeJudgeDefinition, recorded: CodeJudgeDefinition | undefined) => {
|
|
21
|
+
if (!recorded) return false;
|
|
22
|
+
let previous = matches.get(expected);
|
|
23
|
+
if (!previous) { previous = new WeakMap(); matches.set(expected, previous); }
|
|
24
|
+
const cached = previous.get(recorded);
|
|
25
|
+
if (cached !== undefined) return cached;
|
|
26
|
+
const value = recordedCodeMatches(expected, recorded);
|
|
27
|
+
previous.set(recorded, value);
|
|
28
|
+
return value;
|
|
29
|
+
};
|
|
30
|
+
// A code judgment counts only after completed verification of the matching executable.
|
|
18
31
|
const current = (item: (typeof evals)[number], slot: Slot) => {
|
|
19
32
|
const judge = slot.judgeRunId ? index.get(slot.judgeRunId) : undefined;
|
|
20
33
|
return !item.code ||
|
|
21
34
|
(judge?.code?.state === "completed" &&
|
|
22
|
-
|
|
35
|
+
equivalent(item.code, judge.input.code))
|
|
23
36
|
? judge
|
|
24
37
|
: undefined;
|
|
25
38
|
};
|
package/src/cli.ts
CHANGED
|
@@ -15,6 +15,7 @@ import {
|
|
|
15
15
|
buildVerificationImage,
|
|
16
16
|
VERIFICATION_IMAGE,
|
|
17
17
|
prepareInputs,
|
|
18
|
+
refreshModelNames,
|
|
18
19
|
} from "./index";
|
|
19
20
|
import type { ModelRef } from "./index";
|
|
20
21
|
|
|
@@ -45,6 +46,7 @@ Commands:
|
|
|
45
46
|
merge-runs <to> <from> Merge results into one aggregate
|
|
46
47
|
add-models <run> Add models with repeated --model flags
|
|
47
48
|
remove-models <run> Remove active models while retaining their evidence
|
|
49
|
+
refresh-model-names <run> Refresh reporting names without running evals
|
|
48
50
|
retry <run> <eval-run> Retry a candidate execution
|
|
49
51
|
rejudge <run> <eval-run> Judge the saved evidence again
|
|
50
52
|
|
|
@@ -165,6 +167,9 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
165
167
|
retiredSlots: result.retiredSlots,
|
|
166
168
|
status: result.benchmark.state,
|
|
167
169
|
}, null, 2));
|
|
170
|
+
} else if (command === "refresh-model-names") {
|
|
171
|
+
if (!args[1]) throw new Error("refresh-model-names requires a benchmark run directory");
|
|
172
|
+
console.log(JSON.stringify(await refreshModelNames(resolve(args[1])), null, 2));
|
|
168
173
|
} else if (command === "retry" || command === "rejudge") {
|
|
169
174
|
if (!args[1] || !args[2])
|
|
170
175
|
throw new Error(
|
|
@@ -177,7 +182,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
177
182
|
console.log(JSON.stringify(result, null, 2));
|
|
178
183
|
} else
|
|
179
184
|
throw new Error(
|
|
180
|
-
"Commands: image, plan, run, view, add-models, remove-models, merge-runs, retry, rejudge, snapshot",
|
|
185
|
+
"Commands: image, plan, run, view, add-models, remove-models, refresh-model-names, merge-runs, retry, rejudge, snapshot",
|
|
181
186
|
);
|
|
182
187
|
} catch (error) {
|
|
183
188
|
console.error(error instanceof Error ? error.message : String(error));
|
package/src/index.ts
CHANGED
|
@@ -64,6 +64,7 @@ export {
|
|
|
64
64
|
} from "./app/run-benchmark";
|
|
65
65
|
export { addModels } from "./app/add-models";
|
|
66
66
|
export { removeModels } from "./app/remove-models";
|
|
67
|
+
export { refreshModelNames } from "./app/refresh-model-names";
|
|
67
68
|
export type { CostEstimate } from "./app/cost-plan";
|
|
68
69
|
export { mergeBenchmarkRuns } from "./app/merge-runs";
|
|
69
70
|
export { retryEvalRun } from "./app/retry-run";
|
|
@@ -10,6 +10,37 @@ const printer = new Bun.Transpiler({
|
|
|
10
10
|
deadCodeElimination: false,
|
|
11
11
|
});
|
|
12
12
|
|
|
13
|
+
function portableDirectories(source: string) {
|
|
14
|
+
return source.replace(
|
|
15
|
+
/^\s*var __(?:dirname|filename) = .*$/gm,
|
|
16
|
+
(line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
|
|
17
|
+
);
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** An unchanged self-contained executable can retain judgments whose older
|
|
21
|
+
* identity included host package manifests and lockfiles. This is a read-only
|
|
22
|
+
* identity check, not a rewrite of recorded code or a changed scoring rule.
|
|
23
|
+
* External packages still require their exact executable identity. */
|
|
24
|
+
export function recordedCodeMatches(
|
|
25
|
+
expected: CodeJudgeDefinition,
|
|
26
|
+
recorded: CodeJudgeDefinition | undefined,
|
|
27
|
+
) {
|
|
28
|
+
if (!recorded) return false;
|
|
29
|
+
if (recorded.hash === expected.hash) return true;
|
|
30
|
+
if (!expected.source.trim() || expected.source !== recorded.source ||
|
|
31
|
+
Object.keys(expected.dependencies).length !== 0) return false;
|
|
32
|
+
const imports = new Bun.Transpiler({ loader: "js" }).scan(expected.source).imports;
|
|
33
|
+
if (imports.some(item => !item.path.startsWith("node:") && !item.path.startsWith("bun:"))) return false;
|
|
34
|
+
const metadata = Object.keys(recorded.dependencies);
|
|
35
|
+
if (!metadata.length || metadata.some(path =>
|
|
36
|
+
!/(?:^|[\\/])(?:package\.json|bun\.lockb?|package-lock\.json|pnpm-lock\.yaml|yarn\.lock)$/.test(path))) return false;
|
|
37
|
+
// Verify both identities rather than accepting an arbitrary mismatched hash.
|
|
38
|
+
return CODE_JUDGE_PROTOCOL === 1 &&
|
|
39
|
+
recorded.hash === fingerprint({ source: recorded.source, dependencies: recorded.dependencies }) &&
|
|
40
|
+
expected.hash === fingerprint({ protocol: CODE_JUDGE_PROTOCOL,
|
|
41
|
+
source: printer.transformSync(portableDirectories(expected.source)) });
|
|
42
|
+
}
|
|
43
|
+
|
|
13
44
|
/** The package that provides an external import, as `name@version/path`. */
|
|
14
45
|
async function packageSpecifier(path: string, from: string) {
|
|
15
46
|
for (let directory = dirname(path); ; directory = dirname(directory)) {
|
|
@@ -93,10 +124,7 @@ export async function compileCodeJudge(
|
|
|
93
124
|
);
|
|
94
125
|
}
|
|
95
126
|
// Bun writes __dirname and __filename as absolute paths. Like other runtime file inputs, they are not identity.
|
|
96
|
-
portable = portable
|
|
97
|
-
/^\s*var __(?:dirname|filename) = .*$/gm,
|
|
98
|
-
(line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
|
|
99
|
-
);
|
|
127
|
+
portable = portableDirectories(portable);
|
|
100
128
|
return {
|
|
101
129
|
file,
|
|
102
130
|
source,
|