@hona/openeval 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -90,6 +90,18 @@ bunx --bun @hona/openeval view
90
90
  The viewer opens at **http://127.0.0.1:4173**. `run` resumes the same aggregate;
91
91
  scope flags select work while retaining existing scores.
92
92
 
93
+ To remove a model from an existing aggregate, remove its entry from
94
+ `benchmark.ts`, then retire its active selections:
95
+
96
+ ```sh
97
+ bunx --bun @hona/openeval snapshot ./results/RUN before-model-removal
98
+ bunx --bun @hona/openeval remove-models ./results/RUN --model provider/retired-model
99
+ ```
100
+
101
+ This retains the model's recorded executions, judgments, and artifacts. It
102
+ does not run candidates or judges, and requires a stopped benchmark run.
103
+ Models still declared in `benchmark.ts` can be added back by a later `run`.
104
+
93
105
  ```mermaid
94
106
  flowchart LR
95
107
  P["prompt.md"] --> C["Isolated candidate"] --> E["Recording"]
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hona/openeval",
3
- "version": "0.2.1",
3
+ "version": "0.2.2",
4
4
  "description": "Typed prompt-and-judge evaluations with isolated agents, recorded evidence, and a results viewer",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -0,0 +1,60 @@
1
+ import { resolve } from "node:path";
2
+ import type { ModelRef } from "../types";
3
+ import { Results } from "../infra/sqlite";
4
+ import { modelRef } from "./load-benchmark";
5
+ import { readBenchmarkRun } from "./read-run";
6
+ import { finishBenchmark } from "./run-benchmark";
7
+
8
+ /** Retire active model selections while preserving every execution and artifact. */
9
+ export async function removeModels(
10
+ directory: string,
11
+ models: readonly ModelRef[],
12
+ ) {
13
+ const root = resolve(directory);
14
+ const removed = [...new Set(models.map(modelRef))];
15
+ if (!removed.length) throw new Error("Select models to remove");
16
+ const before = readBenchmarkRun(root).benchmark;
17
+
18
+ using results = new Results(resolve(root, "runner.db"));
19
+ return results.transaction(() => {
20
+ const current = results.benchmark!;
21
+ if (current.id !== before.id || current.mergedInto)
22
+ throw new Error(
23
+ "Benchmark was replaced or merged; use its current aggregate",
24
+ );
25
+ if (current.state === "running")
26
+ throw new Error(
27
+ "Cannot remove models while the benchmark run is running",
28
+ );
29
+ if (removed.some((model) => !current.definition.models.includes(model)))
30
+ throw new Error("Select models present in this benchmark run");
31
+ const remaining = current.definition.models.filter(
32
+ (model) => !removed.includes(model),
33
+ );
34
+ if (!remaining.length)
35
+ throw new Error("Keep at least one model in the benchmark run");
36
+ const retired = results
37
+ .slots()
38
+ .filter((slot) => slot.active && removed.includes(slot.model));
39
+ for (const slot of retired) results.select({ ...slot, active: false });
40
+ const definition = { ...current.definition, models: remaining };
41
+ results.saveBenchmark({
42
+ ...current,
43
+ definition,
44
+ execution: { ...current.execution, deferred: 0 },
45
+ });
46
+ const benchmark = finishBenchmark({
47
+ directory: root,
48
+ definition,
49
+ runtime: current.runtime,
50
+ results,
51
+ });
52
+ results.saveBenchmark(benchmark, true);
53
+ return {
54
+ directory: root,
55
+ removed,
56
+ retiredSlots: retired.length,
57
+ benchmark,
58
+ };
59
+ });
60
+ }
package/src/cli.ts CHANGED
@@ -6,6 +6,7 @@ import {
6
6
  loadBenchmark,
7
7
  buildImage,
8
8
  addModels,
9
+ removeModels,
9
10
  retryEvalRun,
10
11
  judgeRun,
11
12
  serveResults,
@@ -39,6 +40,7 @@ Commands:
39
40
  snapshot <run> <name> Export a score snapshot
40
41
  merge-runs <to> <from> Merge results into one aggregate
41
42
  add-models <run> Add models with repeated --model flags
43
+ remove-models <run> Remove active models while retaining their evidence
42
44
  retry <run> <eval-run> Retry a candidate execution
43
45
  rejudge <run> <eval-run> Judge the saved evidence again
44
46
 
@@ -46,7 +48,7 @@ Options:
46
48
  --benchmark <dir> Benchmark directory (default: current directory)
47
49
  --run <dir> Resume a specific result directory
48
50
  --new Start a new result
49
- --model <provider/id> Add a model to the benchmark
51
+ --model <provider/id> Model to add or remove (repeatable)
50
52
  --only-model <ref> Execute only this model (repeatable)
51
53
  --only-eval <id> Execute only this eval (repeatable)
52
54
  --only-repetition <n> Execute only this repetition (repeatable)
@@ -125,6 +127,16 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
125
127
  if (!args[1])
126
128
  throw new Error("add-models requires a benchmark run directory");
127
129
  console.log((await addModels(resolve(args[1]), models())).directory);
130
+ } else if (command === "remove-models") {
131
+ if (!args[1])
132
+ throw new Error("remove-models requires a benchmark run directory");
133
+ const result = await removeModels(resolve(args[1]), models());
134
+ console.log(JSON.stringify({
135
+ directory: result.directory,
136
+ removed: result.removed,
137
+ retiredSlots: result.retiredSlots,
138
+ status: result.benchmark.state,
139
+ }, null, 2));
128
140
  } else if (command === "retry" || command === "rejudge") {
129
141
  if (!args[1] || !args[2])
130
142
  throw new Error(
@@ -137,7 +149,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
137
149
  console.log(JSON.stringify(result, null, 2));
138
150
  } else
139
151
  throw new Error(
140
- "Commands: image, plan, run, view, add-models, merge-runs, retry, rejudge, snapshot",
152
+ "Commands: image, plan, run, view, add-models, remove-models, merge-runs, retry, rejudge, snapshot",
141
153
  );
142
154
  } catch (error) {
143
155
  console.error(error instanceof Error ? error.message : String(error));
package/src/index.ts CHANGED
@@ -38,6 +38,7 @@ export {
38
38
  type RunBenchmarkOptions,
39
39
  } from "./app/run-benchmark";
40
40
  export { addModels } from "./app/add-models";
41
+ export { removeModels } from "./app/remove-models";
41
42
  export type { CostEstimate } from "./app/cost-plan";
42
43
  export { mergeBenchmarkRuns } from "./app/merge-runs";
43
44
  export { retryEvalRun } from "./app/retry-run";