@hona/openeval 0.5.4 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/VERIFICATION.md +18 -2
- package/package.json +1 -1
- package/src/app/load-benchmark.ts +4 -1
- package/src/app/scores.ts +15 -2
- package/src/infra/judging/code-source.ts +32 -4
- package/src/infra/verification/image.ts +3 -0
- package/src/infra/verification/limits.ts +14 -0
- package/src/infra/verification/session.ts +3 -3
- package/src/judge-context.ts +3 -0
package/VERIFICATION.md
CHANGED
|
@@ -8,7 +8,7 @@ run it in a disposable OCI container rather than on the runner host.
|
|
|
8
8
|
import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
|
|
9
9
|
export default {
|
|
10
10
|
models: ["example/model"],
|
|
11
|
-
judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096, workspaceMiB: 2048 } },
|
|
11
|
+
judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096, workspaceMiB: 2048, inputMiB: 256 } },
|
|
12
12
|
} satisfies Benchmark;
|
|
13
13
|
```
|
|
14
14
|
|
|
@@ -56,7 +56,7 @@ prefix does not necessarily establish final-task failure.
|
|
|
56
56
|
- Commands, exit statuses, image/resource identity, logs, and requested regular
|
|
57
57
|
output files are retained with the JudgeRun. Viewer previews never execute HTML
|
|
58
58
|
or SVG from the delivered artifact.
|
|
59
|
-
- Workspace transfer
|
|
59
|
+
- Workspace transfer defaults to 128 MiB; each output archive is limited to
|
|
60
60
|
32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
|
|
61
61
|
- Containers self-expire. The parent also removes containers bearing only its
|
|
62
62
|
unique execution label after worker interruption. No shared/broad cleanup.
|
|
@@ -68,6 +68,15 @@ retained in verification receipts and code-judge identity. Omitting it preserves
|
|
|
68
68
|
existing identities and limits. Network, credentials, privileges, transfer caps,
|
|
69
69
|
and the read-only image filesystem are unchanged.
|
|
70
70
|
|
|
71
|
+
For larger source snapshots, set `judge.verification.inputMiB` to a positive
|
|
72
|
+
integer archive limit. This is separate from `workspaceMiB`: increasing storage
|
|
73
|
+
alone does not increase the transfer limit. The configured byte count must be a
|
|
74
|
+
safe integer. Non-default limits are retained in receipts and code-judge identity;
|
|
75
|
+
omitting the field or setting it to 128 preserves existing identities. Output and
|
|
76
|
+
log caps, filesystem permissions, network access, and container isolation stay
|
|
77
|
+
unchanged. Provision enough workspace storage for the actual restored files and
|
|
78
|
+
any verification-generated data.
|
|
79
|
+
|
|
71
80
|
The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
|
|
72
81
|
with Chromium. Import Playwright from
|
|
73
82
|
`/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
|
|
@@ -83,6 +92,13 @@ evidence. Do not put evaluator scripts in candidate workspaces.
|
|
|
83
92
|
|
|
84
93
|
## Readable judgments and controls
|
|
85
94
|
|
|
95
|
+
Retained code judgments stay scored when a metadata-only identity change leaves
|
|
96
|
+
the exact self-contained executable unchanged. The reader verifies both recorded
|
|
97
|
+
identities and rejects changed source, external package imports, unknown metadata,
|
|
98
|
+
or a different execution protocol. This check makes no model calls and rewrites
|
|
99
|
+
no evidence. A completed code verification is still required. Package/lockfile
|
|
100
|
+
metadata alone must not make an unchanged historical scorecard appear unscored.
|
|
101
|
+
|
|
86
102
|
`openeval prepare --output <new-directory>` assembles real candidate inputs and
|
|
87
103
|
runs declared preparation in the candidate image, then archives the prepared
|
|
88
104
|
workspace. `--only-eval` scopes it. This makes zero model calls, creates no
|
package/package.json
CHANGED
|
@@ -18,6 +18,7 @@ import {
|
|
|
18
18
|
import { compileCodeJudge } from "../infra/judging/code-source";
|
|
19
19
|
import { readCodeCriteria } from "../infra/judging/code-criteria";
|
|
20
20
|
import { RUNTIME_IMAGE } from "../infra/opencode/version";
|
|
21
|
+
import { DEFAULT_INPUT_MIB, verificationInputMiB } from "../infra/verification/limits";
|
|
21
22
|
import { monitorPolicy } from "./monitor-policy";
|
|
22
23
|
import {
|
|
23
24
|
fingerprint,
|
|
@@ -364,7 +365,7 @@ export async function loadBenchmark(
|
|
|
364
365
|
const verification = definition.judge?.verification;
|
|
365
366
|
if (verification && (
|
|
366
367
|
typeof verification !== "object" || Array.isArray(verification) ||
|
|
367
|
-
Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB", "workspaceMiB"].includes(key)) ||
|
|
368
|
+
Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB", "workspaceMiB", "inputMiB"].includes(key)) ||
|
|
368
369
|
typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
|
|
369
370
|
!["docker", "podman"].includes(verification.engine ?? engine)
|
|
370
371
|
)) throw new Error("judge.verification requires an image and valid container limits");
|
|
@@ -372,6 +373,7 @@ export async function loadBenchmark(
|
|
|
372
373
|
if (verification?.workspaceMiB !== undefined &&
|
|
373
374
|
(!Number.isSafeInteger(verification.workspaceMiB) || verification.workspaceMiB < 1 || verification.workspaceMiB > verificationMemory))
|
|
374
375
|
throw new Error("Verification workspaceMiB must be a positive integer within memoryMiB");
|
|
376
|
+
const inputMiB = verificationInputMiB(verification?.inputMiB);
|
|
375
377
|
for (const search of [
|
|
376
378
|
definition.candidate?.websearch,
|
|
377
379
|
definition.judge?.websearch,
|
|
@@ -409,6 +411,7 @@ export async function loadBenchmark(
|
|
|
409
411
|
cpus: positive(verification.cpus, 2, "Verification CPU count"),
|
|
410
412
|
memoryMiB: verificationMemory,
|
|
411
413
|
...(verification.workspaceMiB === undefined ? {} : { workspaceMiB: verification.workspaceMiB }),
|
|
414
|
+
...(inputMiB === DEFAULT_INPUT_MIB ? {} : { inputMiB }),
|
|
412
415
|
} } : {}),
|
|
413
416
|
},
|
|
414
417
|
container: {
|
package/src/app/scores.ts
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
import type { BenchmarkDefinition, JudgeRun, Slot } from "../types";
|
|
2
|
+
import type { CodeJudgeDefinition } from "../judge-context";
|
|
2
3
|
import { modelScore } from "../view";
|
|
3
4
|
import { isScored } from "../judgment";
|
|
4
5
|
import { categoryKey, inCategories } from "../criterion-categories";
|
|
6
|
+
import { recordedCodeMatches } from "../infra/judging/code-source";
|
|
5
7
|
|
|
6
8
|
/** Scores are derived from a single selection snapshot; every eval has equal weight. */
|
|
7
9
|
export function benchmarkScores(
|
|
@@ -14,12 +16,23 @@ export function benchmarkScores(
|
|
|
14
16
|
const evals = definition.evals.filter(
|
|
15
17
|
(item) => !evalId || item.id === evalId,
|
|
16
18
|
);
|
|
17
|
-
|
|
19
|
+
const matches = new WeakMap<CodeJudgeDefinition, WeakMap<CodeJudgeDefinition, boolean>>();
|
|
20
|
+
const equivalent = (expected: CodeJudgeDefinition, recorded: CodeJudgeDefinition | undefined) => {
|
|
21
|
+
if (!recorded) return false;
|
|
22
|
+
let previous = matches.get(expected);
|
|
23
|
+
if (!previous) { previous = new WeakMap(); matches.set(expected, previous); }
|
|
24
|
+
const cached = previous.get(recorded);
|
|
25
|
+
if (cached !== undefined) return cached;
|
|
26
|
+
const value = recordedCodeMatches(expected, recorded);
|
|
27
|
+
previous.set(recorded, value);
|
|
28
|
+
return value;
|
|
29
|
+
};
|
|
30
|
+
// A code judgment counts only after completed verification of the matching executable.
|
|
18
31
|
const current = (item: (typeof evals)[number], slot: Slot) => {
|
|
19
32
|
const judge = slot.judgeRunId ? index.get(slot.judgeRunId) : undefined;
|
|
20
33
|
return !item.code ||
|
|
21
34
|
(judge?.code?.state === "completed" &&
|
|
22
|
-
|
|
35
|
+
equivalent(item.code, judge.input.code))
|
|
23
36
|
? judge
|
|
24
37
|
: undefined;
|
|
25
38
|
};
|
|
@@ -10,6 +10,37 @@ const printer = new Bun.Transpiler({
|
|
|
10
10
|
deadCodeElimination: false,
|
|
11
11
|
});
|
|
12
12
|
|
|
13
|
+
function portableDirectories(source: string) {
|
|
14
|
+
return source.replace(
|
|
15
|
+
/^\s*var __(?:dirname|filename) = .*$/gm,
|
|
16
|
+
(line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
|
|
17
|
+
);
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** An unchanged self-contained executable can retain judgments whose older
|
|
21
|
+
* identity included host package manifests and lockfiles. This is a read-only
|
|
22
|
+
* identity check, not a rewrite of recorded code or a changed scoring rule.
|
|
23
|
+
* External packages still require their exact executable identity. */
|
|
24
|
+
export function recordedCodeMatches(
|
|
25
|
+
expected: CodeJudgeDefinition,
|
|
26
|
+
recorded: CodeJudgeDefinition | undefined,
|
|
27
|
+
) {
|
|
28
|
+
if (!recorded) return false;
|
|
29
|
+
if (recorded.hash === expected.hash) return true;
|
|
30
|
+
if (!expected.source.trim() || expected.source !== recorded.source ||
|
|
31
|
+
Object.keys(expected.dependencies).length !== 0) return false;
|
|
32
|
+
const imports = new Bun.Transpiler({ loader: "js" }).scan(expected.source).imports;
|
|
33
|
+
if (imports.some(item => !item.path.startsWith("node:") && !item.path.startsWith("bun:"))) return false;
|
|
34
|
+
const metadata = Object.keys(recorded.dependencies);
|
|
35
|
+
if (!metadata.length || metadata.some(path =>
|
|
36
|
+
!/(?:^|[\\/])(?:package\.json|bun\.lockb?|package-lock\.json|pnpm-lock\.yaml|yarn\.lock)$/.test(path))) return false;
|
|
37
|
+
// Verify both identities rather than accepting an arbitrary mismatched hash.
|
|
38
|
+
return CODE_JUDGE_PROTOCOL === 1 &&
|
|
39
|
+
recorded.hash === fingerprint({ source: recorded.source, dependencies: recorded.dependencies }) &&
|
|
40
|
+
expected.hash === fingerprint({ protocol: CODE_JUDGE_PROTOCOL,
|
|
41
|
+
source: printer.transformSync(portableDirectories(expected.source)) });
|
|
42
|
+
}
|
|
43
|
+
|
|
13
44
|
/** The package that provides an external import, as `name@version/path`. */
|
|
14
45
|
async function packageSpecifier(path: string, from: string) {
|
|
15
46
|
for (let directory = dirname(path); ; directory = dirname(directory)) {
|
|
@@ -93,10 +124,7 @@ export async function compileCodeJudge(
|
|
|
93
124
|
);
|
|
94
125
|
}
|
|
95
126
|
// Bun writes __dirname and __filename as absolute paths. Like other runtime file inputs, they are not identity.
|
|
96
|
-
portable = portable
|
|
97
|
-
/^\s*var __(?:dirname|filename) = .*$/gm,
|
|
98
|
-
(line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
|
|
99
|
-
);
|
|
127
|
+
portable = portableDirectories(portable);
|
|
100
128
|
return {
|
|
101
129
|
file,
|
|
102
130
|
source,
|
|
@@ -2,6 +2,7 @@ import { fileURLToPath } from "node:url";
|
|
|
2
2
|
import type { VerificationEnvironment, VerificationRuntime } from "../../judge-context";
|
|
3
3
|
import { engineCommand } from "../containers/docker";
|
|
4
4
|
import { treeHash } from "../files";
|
|
5
|
+
import { DEFAULT_INPUT_MIB, verificationInputMiB } from "./limits";
|
|
5
6
|
|
|
6
7
|
export const VERIFICATION_IMAGE = "openeval-verification:0.5.0";
|
|
7
8
|
const directory = fileURLToPath(new URL("./runtime/", import.meta.url));
|
|
@@ -25,6 +26,7 @@ export async function verificationRuntime(environment?: VerificationEnvironment)
|
|
|
25
26
|
(!Number.isSafeInteger(environment.workspaceMiB) || environment.workspaceMiB < 1 ||
|
|
26
27
|
environment.workspaceMiB > (environment.memoryMiB ?? 4096)))
|
|
27
28
|
throw new Error("Verification workspaceMiB must be a positive integer within memoryMiB");
|
|
29
|
+
const inputMiB = verificationInputMiB(environment.inputMiB);
|
|
28
30
|
const output = await engineCommand(engine, [
|
|
29
31
|
"image", "inspect", environment.image, "--format",
|
|
30
32
|
'{{.Id}}|{{index .Config.Labels "openeval.verification"}}',
|
|
@@ -37,5 +39,6 @@ export async function verificationRuntime(environment?: VerificationEnvironment)
|
|
|
37
39
|
engine, image: environment.image, imageId,
|
|
38
40
|
cpus: environment.cpus ?? 2, memoryMiB: environment.memoryMiB ?? 4096,
|
|
39
41
|
...(environment.workspaceMiB === undefined ? {} : { workspaceMiB: environment.workspaceMiB }),
|
|
42
|
+
...(inputMiB === DEFAULT_INPUT_MIB ? {} : { inputMiB }),
|
|
40
43
|
};
|
|
41
44
|
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
export const DEFAULT_INPUT_MIB = 128;
|
|
2
|
+
const MIB = 1024 * 1024;
|
|
3
|
+
|
|
4
|
+
/** Restored workspace archives stay bounded, including opt-in larger inputs. */
|
|
5
|
+
export function verificationInputMiB(value?: number): number {
|
|
6
|
+
const limit = value === undefined ? DEFAULT_INPUT_MIB : value;
|
|
7
|
+
if (!Number.isSafeInteger(limit) || limit < 1 || !Number.isSafeInteger(limit * MIB))
|
|
8
|
+
throw new Error("Verification inputMiB must be a positive integer with a safe byte count");
|
|
9
|
+
return limit;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export function verificationInputBytes(value?: number): number {
|
|
13
|
+
return verificationInputMiB(value) * MIB;
|
|
14
|
+
}
|
|
@@ -6,10 +6,10 @@ import type { Engine } from "../../types";
|
|
|
6
6
|
import { engineCommand } from "../containers/docker";
|
|
7
7
|
import { extractWorkspaceArchive } from "../containers/transfer";
|
|
8
8
|
import { contained, hash, relativePath, writeJson } from "../files";
|
|
9
|
+
import { verificationInputBytes, verificationInputMiB } from "./limits";
|
|
9
10
|
|
|
10
11
|
const LOG_BYTES = 1024 * 1024;
|
|
11
12
|
const OUTPUT_BYTES = 32 * 1024 * 1024;
|
|
12
|
-
const INPUT_BYTES = 128 * 1024 * 1024;
|
|
13
13
|
|
|
14
14
|
export function verificationPath(value: string): string {
|
|
15
15
|
relativePath(value, "Verification path");
|
|
@@ -116,8 +116,8 @@ export function verificationSession(options: {
|
|
|
116
116
|
const archive = resolve(staging, "workspace.tar");
|
|
117
117
|
const pack = Bun.spawn(["tar", "-cf", archive, "-C", workspace, "."], { stdout: "ignore", stderr: "pipe" });
|
|
118
118
|
const [packed, packError] = await Promise.all([pack.exited, new Response(pack.stderr).text()]);
|
|
119
|
-
if (packed || (await lstat(archive)).size >
|
|
120
|
-
throw new Error(packed ? `Verification transfer failed: ${packError}` :
|
|
119
|
+
if (packed || (await lstat(archive)).size > verificationInputBytes(runtime.inputMiB))
|
|
120
|
+
throw new Error(packed ? `Verification transfer failed: ${packError}` : `Restored verification workspace exceeds ${verificationInputMiB(runtime.inputMiB)} MiB`);
|
|
121
121
|
const name = `openeval-verify-${randomUUID()}`;
|
|
122
122
|
const startedAt = new Date().toISOString();
|
|
123
123
|
const deadlineAt = Date.now() + timeoutMs;
|
package/src/judge-context.ts
CHANGED
|
@@ -41,6 +41,8 @@ export type VerificationEnvironment = {
|
|
|
41
41
|
memoryMiB?: number;
|
|
42
42
|
/** Workspace tmpfs capacity in MiB; default 512. A custom value must fit within memoryMiB. */
|
|
43
43
|
workspaceMiB?: number;
|
|
44
|
+
/** Maximum restored workspace archive size in MiB; default 128. */
|
|
45
|
+
inputMiB?: number;
|
|
44
46
|
};
|
|
45
47
|
export type VerificationRuntime = {
|
|
46
48
|
engine: Engine;
|
|
@@ -49,6 +51,7 @@ export type VerificationRuntime = {
|
|
|
49
51
|
cpus: number;
|
|
50
52
|
memoryMiB: number;
|
|
51
53
|
workspaceMiB?: number;
|
|
54
|
+
inputMiB?: number;
|
|
52
55
|
};
|
|
53
56
|
export type VerificationRequest = {
|
|
54
57
|
revision?: Revision;
|