@hona/openeval 0.5.4 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/VERIFICATION.md CHANGED
@@ -8,7 +8,7 @@ run it in a disposable OCI container rather than on the runner host.
8
8
  import { VERIFICATION_IMAGE, type Benchmark } from "@hona/openeval";
9
9
  export default {
10
10
  models: ["example/model"],
11
- judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096, workspaceMiB: 2048 } },
11
+ judge: { verification: { image: VERIFICATION_IMAGE, cpus: 2, memoryMiB: 4096, workspaceMiB: 2048, inputMiB: 256 } },
12
12
  } satisfies Benchmark;
13
13
  ```
14
14
 
@@ -56,7 +56,7 @@ prefix does not necessarily establish final-task failure.
56
56
  - Commands, exit statuses, image/resource identity, logs, and requested regular
57
57
  output files are retained with the JudgeRun. Viewer previews never execute HTML
58
58
  or SVG from the delivered artifact.
59
- - Workspace transfer is limited to 128 MiB; each output archive is limited to
59
+ - Workspace transfer defaults to 128 MiB; each output archive is limited to
60
60
  32 MiB; logs are limited to 1 MiB per stream and are marked when truncated.
61
61
  - Containers self-expire. The parent also removes containers bearing only its
62
62
  unique execution label after worker interruption. No shared/broad cleanup.
@@ -68,6 +68,15 @@ retained in verification receipts and code-judge identity. Omitting it preserves
68
68
  existing identities and limits. Network, credentials, privileges, transfer caps,
69
69
  and the read-only image filesystem are unchanged.
70
70
 
71
+ For larger source snapshots, set `judge.verification.inputMiB` to a positive
72
+ integer archive limit. This is separate from `workspaceMiB`: increasing storage
73
+ alone does not increase the transfer limit. The configured byte count must be a
74
+ safe integer. Non-default limits are retained in receipts and code-judge identity;
75
+ omitting the field or setting it to 128 preserves existing identities. Output and
76
+ log caps, filesystem permissions, network access, and container isolation stay
77
+ unchanged. Provision enough workspace storage for the actual restored files and
78
+ any verification-generated data.
79
+
71
80
  The standard image includes Bun 1.4.2, Python, Git, Node, and Playwright 1.63.0
72
81
  with Chromium. Import Playwright from
73
82
  `/opt/verify/node_modules/playwright/index.mjs`. Dependencies needed by a task
@@ -83,6 +92,13 @@ evidence. Do not put evaluator scripts in candidate workspaces.
83
92
 
84
93
  ## Readable judgments and controls
85
94
 
95
+ Retained code judgments stay scored when a metadata-only identity change leaves
96
+ the exact self-contained executable unchanged. The reader verifies both recorded
97
+ identities and rejects changed source, external package imports, unknown metadata,
98
+ or a different execution protocol. This check makes no model calls and rewrites
99
+ no evidence. A completed code verification is still required. Package/lockfile
100
+ metadata alone must not make an unchanged historical scorecard appear unscored.
101
+
86
102
  `openeval prepare --output <new-directory>` assembles real candidate inputs and
87
103
  runs declared preparation in the candidate image, then archives the prepared
88
104
  workspace. `--only-eval` scopes it. This makes zero model calls, creates no
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.5.4",
3
+ "version": "0.5.6",
4
4
  "description": "Code and LLM evaluations with isolated agents, recorded OpenCode data, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -18,6 +18,7 @@ import {
18
18
  import { compileCodeJudge } from "../infra/judging/code-source";
19
19
  import { readCodeCriteria } from "../infra/judging/code-criteria";
20
20
  import { RUNTIME_IMAGE } from "../infra/opencode/version";
21
+ import { DEFAULT_INPUT_MIB, verificationInputMiB } from "../infra/verification/limits";
21
22
  import { monitorPolicy } from "./monitor-policy";
22
23
  import {
23
24
  fingerprint,
@@ -364,7 +365,7 @@ export async function loadBenchmark(
364
365
  const verification = definition.judge?.verification;
365
366
  if (verification && (
366
367
  typeof verification !== "object" || Array.isArray(verification) ||
367
- Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB", "workspaceMiB"].includes(key)) ||
368
+ Object.keys(verification).some(key => !["image", "engine", "cpus", "memoryMiB", "workspaceMiB", "inputMiB"].includes(key)) ||
368
369
  typeof verification.image !== "string" || !verification.image.trim() || /\s/.test(verification.image) ||
369
370
  !["docker", "podman"].includes(verification.engine ?? engine)
370
371
  )) throw new Error("judge.verification requires an image and valid container limits");
@@ -372,6 +373,7 @@ export async function loadBenchmark(
372
373
  if (verification?.workspaceMiB !== undefined &&
373
374
  (!Number.isSafeInteger(verification.workspaceMiB) || verification.workspaceMiB < 1 || verification.workspaceMiB > verificationMemory))
374
375
  throw new Error("Verification workspaceMiB must be a positive integer within memoryMiB");
376
+ const inputMiB = verificationInputMiB(verification?.inputMiB);
375
377
  for (const search of [
376
378
  definition.candidate?.websearch,
377
379
  definition.judge?.websearch,
@@ -409,6 +411,7 @@ export async function loadBenchmark(
409
411
  cpus: positive(verification.cpus, 2, "Verification CPU count"),
410
412
  memoryMiB: verificationMemory,
411
413
  ...(verification.workspaceMiB === undefined ? {} : { workspaceMiB: verification.workspaceMiB }),
414
+ ...(inputMiB === DEFAULT_INPUT_MIB ? {} : { inputMiB }),
412
415
  } } : {}),
413
416
  },
414
417
  container: {
package/src/app/scores.ts CHANGED
@@ -1,7 +1,9 @@
1
1
  import type { BenchmarkDefinition, JudgeRun, Slot } from "../types";
2
+ import type { CodeJudgeDefinition } from "../judge-context";
2
3
  import { modelScore } from "../view";
3
4
  import { isScored } from "../judgment";
4
5
  import { categoryKey, inCategories } from "../criterion-categories";
6
+ import { recordedCodeMatches } from "../infra/judging/code-source";
5
7
 
6
8
  /** Scores are derived from a single selection snapshot; every eval has equal weight. */
7
9
  export function benchmarkScores(
@@ -14,12 +16,23 @@ export function benchmarkScores(
14
16
  const evals = definition.evals.filter(
15
17
  (item) => !evalId || item.id === evalId,
16
18
  );
17
- // A code judgment counts only when it ran the eval's current judge.ts to completion.
19
+ const matches = new WeakMap<CodeJudgeDefinition, WeakMap<CodeJudgeDefinition, boolean>>();
20
+ const equivalent = (expected: CodeJudgeDefinition, recorded: CodeJudgeDefinition | undefined) => {
21
+ if (!recorded) return false;
22
+ let previous = matches.get(expected);
23
+ if (!previous) { previous = new WeakMap(); matches.set(expected, previous); }
24
+ const cached = previous.get(recorded);
25
+ if (cached !== undefined) return cached;
26
+ const value = recordedCodeMatches(expected, recorded);
27
+ previous.set(recorded, value);
28
+ return value;
29
+ };
30
+ // A code judgment counts only after completed verification of the matching executable.
18
31
  const current = (item: (typeof evals)[number], slot: Slot) => {
19
32
  const judge = slot.judgeRunId ? index.get(slot.judgeRunId) : undefined;
20
33
  return !item.code ||
21
34
  (judge?.code?.state === "completed" &&
22
- judge.input.code?.hash === item.code.hash)
35
+ equivalent(item.code, judge.input.code))
23
36
  ? judge
24
37
  : undefined;
25
38
  };
@@ -10,6 +10,37 @@ const printer = new Bun.Transpiler({
10
10
  deadCodeElimination: false,
11
11
  });
12
12
 
13
+ function portableDirectories(source: string) {
14
+ return source.replace(
15
+ /^\s*var __(?:dirname|filename) = .*$/gm,
16
+ (line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
17
+ );
18
+ }
19
+
20
+ /** An unchanged self-contained executable can retain judgments whose older
21
+ * identity included host package manifests and lockfiles. This is a read-only
22
+ * identity check, not a rewrite of recorded code or a changed scoring rule.
23
+ * External packages still require their exact executable identity. */
24
+ export function recordedCodeMatches(
25
+ expected: CodeJudgeDefinition,
26
+ recorded: CodeJudgeDefinition | undefined,
27
+ ) {
28
+ if (!recorded) return false;
29
+ if (recorded.hash === expected.hash) return true;
30
+ if (!expected.source.trim() || expected.source !== recorded.source ||
31
+ Object.keys(expected.dependencies).length !== 0) return false;
32
+ const imports = new Bun.Transpiler({ loader: "js" }).scan(expected.source).imports;
33
+ if (imports.some(item => !item.path.startsWith("node:") && !item.path.startsWith("bun:"))) return false;
34
+ const metadata = Object.keys(recorded.dependencies);
35
+ if (!metadata.length || metadata.some(path =>
36
+ !/(?:^|[\\/])(?:package\.json|bun\.lockb?|package-lock\.json|pnpm-lock\.yaml|yarn\.lock)$/.test(path))) return false;
37
+ // Verify both identities rather than accepting an arbitrary mismatched hash.
38
+ return CODE_JUDGE_PROTOCOL === 1 &&
39
+ recorded.hash === fingerprint({ source: recorded.source, dependencies: recorded.dependencies }) &&
40
+ expected.hash === fingerprint({ protocol: CODE_JUDGE_PROTOCOL,
41
+ source: printer.transformSync(portableDirectories(expected.source)) });
42
+ }
43
+
13
44
  /** The package that provides an external import, as `name@version/path`. */
14
45
  async function packageSpecifier(path: string, from: string) {
15
46
  for (let directory = dirname(path); ; directory = dirname(directory)) {
@@ -93,10 +124,7 @@ export async function compileCodeJudge(
93
124
  );
94
125
  }
95
126
  // Bun writes __dirname and __filename as absolute paths. Like other runtime file inputs, they are not identity.
96
- portable = portable.replace(
97
- /^\s*var __(?:dirname|filename) = .*$/gm,
98
- (line) => line.replace(/"(?:[^"\\]|\\.)*"/g, '""'),
99
- );
127
+ portable = portableDirectories(portable);
100
128
  return {
101
129
  file,
102
130
  source,
@@ -2,6 +2,7 @@ import { fileURLToPath } from "node:url";
2
2
  import type { VerificationEnvironment, VerificationRuntime } from "../../judge-context";
3
3
  import { engineCommand } from "../containers/docker";
4
4
  import { treeHash } from "../files";
5
+ import { DEFAULT_INPUT_MIB, verificationInputMiB } from "./limits";
5
6
 
6
7
  export const VERIFICATION_IMAGE = "openeval-verification:0.5.0";
7
8
  const directory = fileURLToPath(new URL("./runtime/", import.meta.url));
@@ -25,6 +26,7 @@ export async function verificationRuntime(environment?: VerificationEnvironment)
25
26
  (!Number.isSafeInteger(environment.workspaceMiB) || environment.workspaceMiB < 1 ||
26
27
  environment.workspaceMiB > (environment.memoryMiB ?? 4096)))
27
28
  throw new Error("Verification workspaceMiB must be a positive integer within memoryMiB");
29
+ const inputMiB = verificationInputMiB(environment.inputMiB);
28
30
  const output = await engineCommand(engine, [
29
31
  "image", "inspect", environment.image, "--format",
30
32
  '{{.Id}}|{{index .Config.Labels "openeval.verification"}}',
@@ -37,5 +39,6 @@ export async function verificationRuntime(environment?: VerificationEnvironment)
37
39
  engine, image: environment.image, imageId,
38
40
  cpus: environment.cpus ?? 2, memoryMiB: environment.memoryMiB ?? 4096,
39
41
  ...(environment.workspaceMiB === undefined ? {} : { workspaceMiB: environment.workspaceMiB }),
42
+ ...(inputMiB === DEFAULT_INPUT_MIB ? {} : { inputMiB }),
40
43
  };
41
44
  }
@@ -0,0 +1,14 @@
1
+ export const DEFAULT_INPUT_MIB = 128;
2
+ const MIB = 1024 * 1024;
3
+
4
+ /** Restored workspace archives stay bounded, including opt-in larger inputs. */
5
+ export function verificationInputMiB(value?: number): number {
6
+ const limit = value === undefined ? DEFAULT_INPUT_MIB : value;
7
+ if (!Number.isSafeInteger(limit) || limit < 1 || !Number.isSafeInteger(limit * MIB))
8
+ throw new Error("Verification inputMiB must be a positive integer with a safe byte count");
9
+ return limit;
10
+ }
11
+
12
+ export function verificationInputBytes(value?: number): number {
13
+ return verificationInputMiB(value) * MIB;
14
+ }
@@ -6,10 +6,10 @@ import type { Engine } from "../../types";
6
6
  import { engineCommand } from "../containers/docker";
7
7
  import { extractWorkspaceArchive } from "../containers/transfer";
8
8
  import { contained, hash, relativePath, writeJson } from "../files";
9
+ import { verificationInputBytes, verificationInputMiB } from "./limits";
9
10
 
10
11
  const LOG_BYTES = 1024 * 1024;
11
12
  const OUTPUT_BYTES = 32 * 1024 * 1024;
12
- const INPUT_BYTES = 128 * 1024 * 1024;
13
13
 
14
14
  export function verificationPath(value: string): string {
15
15
  relativePath(value, "Verification path");
@@ -116,8 +116,8 @@ export function verificationSession(options: {
116
116
  const archive = resolve(staging, "workspace.tar");
117
117
  const pack = Bun.spawn(["tar", "-cf", archive, "-C", workspace, "."], { stdout: "ignore", stderr: "pipe" });
118
118
  const [packed, packError] = await Promise.all([pack.exited, new Response(pack.stderr).text()]);
119
- if (packed || (await lstat(archive)).size > INPUT_BYTES)
120
- throw new Error(packed ? `Verification transfer failed: ${packError}` : "Restored verification workspace exceeds 128 MiB");
119
+ if (packed || (await lstat(archive)).size > verificationInputBytes(runtime.inputMiB))
120
+ throw new Error(packed ? `Verification transfer failed: ${packError}` : `Restored verification workspace exceeds ${verificationInputMiB(runtime.inputMiB)} MiB`);
121
121
  const name = `openeval-verify-${randomUUID()}`;
122
122
  const startedAt = new Date().toISOString();
123
123
  const deadlineAt = Date.now() + timeoutMs;
@@ -41,6 +41,8 @@ export type VerificationEnvironment = {
41
41
  memoryMiB?: number;
42
42
  /** Workspace tmpfs capacity in MiB; default 512. A custom value must fit within memoryMiB. */
43
43
  workspaceMiB?: number;
44
+ /** Maximum restored workspace archive size in MiB; default 128. */
45
+ inputMiB?: number;
44
46
  };
45
47
  export type VerificationRuntime = {
46
48
  engine: Engine;
@@ -49,6 +51,7 @@ export type VerificationRuntime = {
49
51
  cpus: number;
50
52
  memoryMiB: number;
51
53
  workspaceMiB?: number;
54
+ inputMiB?: number;
52
55
  };
53
56
  export type VerificationRequest = {
54
57
  revision?: Revision;