@hona/openeval 0.2.1 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -0
- package/package.json +1 -1
- package/src/app/remove-models.ts +60 -0
- package/src/cli.ts +14 -2
- package/src/index.ts +1 -0
package/README.md
CHANGED
|
@@ -90,6 +90,18 @@ bunx --bun @hona/openeval view
|
|
|
90
90
|
The viewer opens at **http://127.0.0.1:4173**. `run` resumes the same aggregate;
|
|
91
91
|
scope flags select work while retaining existing scores.
|
|
92
92
|
|
|
93
|
+
To remove a model from an existing aggregate, remove its entry from
|
|
94
|
+
`benchmark.ts`, then retire its active selections:
|
|
95
|
+
|
|
96
|
+
```sh
|
|
97
|
+
bunx --bun @hona/openeval snapshot ./results/RUN before-model-removal
|
|
98
|
+
bunx --bun @hona/openeval remove-models ./results/RUN --model provider/retired-model
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
This retains the model's recorded executions, judgments, and artifacts. It
|
|
102
|
+
does not run candidates or judges, and requires a stopped benchmark run.
|
|
103
|
+
Models still declared in `benchmark.ts` can be added back by a later `run`.
|
|
104
|
+
|
|
93
105
|
```mermaid
|
|
94
106
|
flowchart LR
|
|
95
107
|
P["prompt.md"] --> C["Isolated candidate"] --> E["Recording"]
|
package/package.json
CHANGED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import type { ModelRef } from "../types";
|
|
3
|
+
import { Results } from "../infra/sqlite";
|
|
4
|
+
import { modelRef } from "./load-benchmark";
|
|
5
|
+
import { readBenchmarkRun } from "./read-run";
|
|
6
|
+
import { finishBenchmark } from "./run-benchmark";
|
|
7
|
+
|
|
8
|
+
/** Retire active model selections while preserving every execution and artifact. */
|
|
9
|
+
export async function removeModels(
|
|
10
|
+
directory: string,
|
|
11
|
+
models: readonly ModelRef[],
|
|
12
|
+
) {
|
|
13
|
+
const root = resolve(directory);
|
|
14
|
+
const removed = [...new Set(models.map(modelRef))];
|
|
15
|
+
if (!removed.length) throw new Error("Select models to remove");
|
|
16
|
+
const before = readBenchmarkRun(root).benchmark;
|
|
17
|
+
|
|
18
|
+
using results = new Results(resolve(root, "runner.db"));
|
|
19
|
+
return results.transaction(() => {
|
|
20
|
+
const current = results.benchmark!;
|
|
21
|
+
if (current.id !== before.id || current.mergedInto)
|
|
22
|
+
throw new Error(
|
|
23
|
+
"Benchmark was replaced or merged; use its current aggregate",
|
|
24
|
+
);
|
|
25
|
+
if (current.state === "running")
|
|
26
|
+
throw new Error(
|
|
27
|
+
"Cannot remove models while the benchmark run is running",
|
|
28
|
+
);
|
|
29
|
+
if (removed.some((model) => !current.definition.models.includes(model)))
|
|
30
|
+
throw new Error("Select models present in this benchmark run");
|
|
31
|
+
const remaining = current.definition.models.filter(
|
|
32
|
+
(model) => !removed.includes(model),
|
|
33
|
+
);
|
|
34
|
+
if (!remaining.length)
|
|
35
|
+
throw new Error("Keep at least one model in the benchmark run");
|
|
36
|
+
const retired = results
|
|
37
|
+
.slots()
|
|
38
|
+
.filter((slot) => slot.active && removed.includes(slot.model));
|
|
39
|
+
for (const slot of retired) results.select({ ...slot, active: false });
|
|
40
|
+
const definition = { ...current.definition, models: remaining };
|
|
41
|
+
results.saveBenchmark({
|
|
42
|
+
...current,
|
|
43
|
+
definition,
|
|
44
|
+
execution: { ...current.execution, deferred: 0 },
|
|
45
|
+
});
|
|
46
|
+
const benchmark = finishBenchmark({
|
|
47
|
+
directory: root,
|
|
48
|
+
definition,
|
|
49
|
+
runtime: current.runtime,
|
|
50
|
+
results,
|
|
51
|
+
});
|
|
52
|
+
results.saveBenchmark(benchmark, true);
|
|
53
|
+
return {
|
|
54
|
+
directory: root,
|
|
55
|
+
removed,
|
|
56
|
+
retiredSlots: retired.length,
|
|
57
|
+
benchmark,
|
|
58
|
+
};
|
|
59
|
+
});
|
|
60
|
+
}
|
package/src/cli.ts
CHANGED
|
@@ -6,6 +6,7 @@ import {
|
|
|
6
6
|
loadBenchmark,
|
|
7
7
|
buildImage,
|
|
8
8
|
addModels,
|
|
9
|
+
removeModels,
|
|
9
10
|
retryEvalRun,
|
|
10
11
|
judgeRun,
|
|
11
12
|
serveResults,
|
|
@@ -39,6 +40,7 @@ Commands:
|
|
|
39
40
|
snapshot <run> <name> Export a score snapshot
|
|
40
41
|
merge-runs <to> <from> Merge results into one aggregate
|
|
41
42
|
add-models <run> Add models with repeated --model flags
|
|
43
|
+
remove-models <run> Remove active models while retaining their evidence
|
|
42
44
|
retry <run> <eval-run> Retry a candidate execution
|
|
43
45
|
rejudge <run> <eval-run> Judge the saved evidence again
|
|
44
46
|
|
|
@@ -46,7 +48,7 @@ Options:
|
|
|
46
48
|
--benchmark <dir> Benchmark directory (default: current directory)
|
|
47
49
|
--run <dir> Resume a specific result directory
|
|
48
50
|
--new Start a new result
|
|
49
|
-
--model <provider/id>
|
|
51
|
+
--model <provider/id> Model to add or remove (repeatable)
|
|
50
52
|
--only-model <ref> Execute only this model (repeatable)
|
|
51
53
|
--only-eval <id> Execute only this eval (repeatable)
|
|
52
54
|
--only-repetition <n> Execute only this repetition (repeatable)
|
|
@@ -125,6 +127,16 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
125
127
|
if (!args[1])
|
|
126
128
|
throw new Error("add-models requires a benchmark run directory");
|
|
127
129
|
console.log((await addModels(resolve(args[1]), models())).directory);
|
|
130
|
+
} else if (command === "remove-models") {
|
|
131
|
+
if (!args[1])
|
|
132
|
+
throw new Error("remove-models requires a benchmark run directory");
|
|
133
|
+
const result = await removeModels(resolve(args[1]), models());
|
|
134
|
+
console.log(JSON.stringify({
|
|
135
|
+
directory: result.directory,
|
|
136
|
+
removed: result.removed,
|
|
137
|
+
retiredSlots: result.retiredSlots,
|
|
138
|
+
status: result.benchmark.state,
|
|
139
|
+
}, null, 2));
|
|
128
140
|
} else if (command === "retry" || command === "rejudge") {
|
|
129
141
|
if (!args[1] || !args[2])
|
|
130
142
|
throw new Error(
|
|
@@ -137,7 +149,7 @@ Requires Bun 1.4.2+, Docker, and an authenticated OpenCode installation.`);
|
|
|
137
149
|
console.log(JSON.stringify(result, null, 2));
|
|
138
150
|
} else
|
|
139
151
|
throw new Error(
|
|
140
|
-
"Commands: image, plan, run, view, add-models, merge-runs, retry, rejudge, snapshot",
|
|
152
|
+
"Commands: image, plan, run, view, add-models, remove-models, merge-runs, retry, rejudge, snapshot",
|
|
141
153
|
);
|
|
142
154
|
} catch (error) {
|
|
143
155
|
console.error(error instanceof Error ? error.message : String(error));
|
package/src/index.ts
CHANGED
|
@@ -38,6 +38,7 @@ export {
|
|
|
38
38
|
type RunBenchmarkOptions,
|
|
39
39
|
} from "./app/run-benchmark";
|
|
40
40
|
export { addModels } from "./app/add-models";
|
|
41
|
+
export { removeModels } from "./app/remove-models";
|
|
41
42
|
export type { CostEstimate } from "./app/cost-plan";
|
|
42
43
|
export { mergeBenchmarkRuns } from "./app/merge-runs";
|
|
43
44
|
export { retryEvalRun } from "./app/retry-run";
|