openmerit 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +270 -0
  3. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +38 -0
  4. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  5. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +32 -0
  6. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  7. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +26 -0
  8. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  9. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +26 -0
  10. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  11. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +38 -0
  12. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  13. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +38 -0
  14. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  15. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +20 -0
  16. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  17. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +38 -0
  18. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  19. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +26 -0
  20. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  21. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +20 -0
  22. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  23. package/benchmark/invoice_ocr/data/manifest.json +97 -0
  24. package/dist/benchmarks.js +98 -0
  25. package/dist/catalog.js +51 -0
  26. package/dist/cli.js +261 -0
  27. package/dist/daemon.js +305 -0
  28. package/dist/frontier.js +49 -0
  29. package/dist/invoice-eval.js +33 -0
  30. package/dist/invoice-score.js +124 -0
  31. package/dist/judge.js +43 -0
  32. package/dist/llm.js +66 -0
  33. package/dist/pi-trials.js +224 -0
  34. package/dist/policy.js +110 -0
  35. package/dist/recommend.js +63 -0
  36. package/dist/store.js +55 -0
  37. package/dist/strategist.js +64 -0
  38. package/dist/task-input.js +30 -0
  39. package/dist/traces.js +123 -0
  40. package/dist/trials.js +101 -0
  41. package/dist/types.js +2 -0
  42. package/examples/invoice-prompt.txt +19 -0
  43. package/examples/task.example.json +6 -0
  44. package/extension/openmerit.ts +705 -0
  45. package/instructions/OPENMERIT.md +51 -0
  46. package/instructions/openmerit.policy.json +33 -0
  47. package/package.json +77 -0
@@ -0,0 +1,97 @@
1
+ {
2
+ "dataset": "alamgirqazi/invoice-ocr-synthetic",
3
+ "dataset_revision": "44ddfd3c0c373b69f377910f83c4164a1e52d52c",
4
+ "license": "CC BY 4.0",
5
+ "rows": [
6
+ {
7
+ "invoice_id": "invoice_01",
8
+ "row_idx": 2,
9
+ "file_name": "synth_0002_8f01361b.png",
10
+ "quality": "clean",
11
+ "template": "freelancer_invoice",
12
+ "image": "invoice_01_row_2.jpg",
13
+ "ground_truth": "invoice_01_ground_truth.json"
14
+ },
15
+ {
16
+ "invoice_id": "invoice_02",
17
+ "row_idx": 5,
18
+ "file_name": "synth_0005_553deef3.png",
19
+ "quality": "clean",
20
+ "template": "formal_corporate",
21
+ "image": "invoice_02_row_5.jpg",
22
+ "ground_truth": "invoice_02_ground_truth.json"
23
+ },
24
+ {
25
+ "invoice_id": "invoice_03",
26
+ "row_idx": 6,
27
+ "file_name": "synth_0006_474081ca.png",
28
+ "quality": "clean",
29
+ "template": "modern_minimal",
30
+ "image": "invoice_03_row_6.jpg",
31
+ "ground_truth": "invoice_03_ground_truth.json"
32
+ },
33
+ {
34
+ "invoice_id": "invoice_04",
35
+ "row_idx": 7,
36
+ "file_name": "synth_0007_7292e354.png",
37
+ "quality": "clean",
38
+ "template": "utility_bill",
39
+ "image": "invoice_04_row_7.jpg",
40
+ "ground_truth": "invoice_04_ground_truth.json"
41
+ },
42
+ {
43
+ "invoice_id": "invoice_05",
44
+ "row_idx": 947,
45
+ "file_name": "synth_0009_8f6b430f.png",
46
+ "quality": "faded",
47
+ "template": "freelancer_invoice",
48
+ "image": "invoice_05_row_947.jpg",
49
+ "ground_truth": "invoice_05_ground_truth.json"
50
+ },
51
+ {
52
+ "invoice_id": "invoice_06",
53
+ "row_idx": 948,
54
+ "file_name": "synth_0010_73bd8d75.png",
55
+ "quality": "faded",
56
+ "template": "formal_corporate",
57
+ "image": "invoice_06_row_948.jpg",
58
+ "ground_truth": "invoice_06_ground_truth.json"
59
+ },
60
+ {
61
+ "invoice_id": "invoice_07",
62
+ "row_idx": 949,
63
+ "file_name": "synth_0011_8888d5c6.png",
64
+ "quality": "faded",
65
+ "template": "utility_bill",
66
+ "image": "invoice_07_row_949.jpg",
67
+ "ground_truth": "invoice_07_ground_truth.json"
68
+ },
69
+ {
70
+ "invoice_id": "invoice_08",
71
+ "row_idx": 1888,
72
+ "file_name": "synth_0012_72cf2460.png",
73
+ "quality": "bad_scan",
74
+ "template": "modern_minimal",
75
+ "image": "invoice_08_row_1888.jpg",
76
+ "ground_truth": "invoice_08_ground_truth.json"
77
+ },
78
+ {
79
+ "invoice_id": "invoice_09",
80
+ "row_idx": 1890,
81
+ "file_name": "synth_0014_d3972e7f.png",
82
+ "quality": "bad_scan",
83
+ "template": "freelancer_invoice",
84
+ "image": "invoice_09_row_1890.jpg",
85
+ "ground_truth": "invoice_09_ground_truth.json"
86
+ },
87
+ {
88
+ "invoice_id": "invoice_10",
89
+ "row_idx": 1892,
90
+ "file_name": "synth_0016_223f5434.png",
91
+ "quality": "bad_scan",
92
+ "template": "modern_minimal",
93
+ "image": "invoice_10_row_1892.jpg",
94
+ "ground_truth": "invoice_10_ground_truth.json"
95
+ }
96
+ ]
97
+ }
@@ -0,0 +1,98 @@
1
+ /**
2
+ * Public benchmark signals.
3
+ *
4
+ * Two jobs:
5
+ * 1. Map a task to the public benchmarks that matter for it (so the
6
+ * strategist and the user know why a candidate is "worth testing").
7
+ * 2. Provide a prior score for a model from a benchmark digest, so a newly
8
+ * released model can be ranked against the existing frontier before we
9
+ * spend money on shadow trials.
10
+ *
11
+ * The digest lives at ~/.openmerit/benchmarks/digest.json and is a simple
12
+ * map: { "vendor/model": { "benchmark_name": score_0_to_100, ... }, ... }.
13
+ * Refresh it from public leaderboards (LMArena, Artificial Analysis,
14
+ * SWE-bench, ...) however you like; a small seed ships below so the system
15
+ * works offline.
16
+ */
17
+ import { paths, readJson, writeJson } from "./store.js";
18
+ /** Task category -> public benchmarks that are predictive for it. */
19
+ export const TASK_BENCHMARKS = {
20
+ code: ["SWE-bench Verified", "HumanEval", "LMArena Code", "Aider Polyglot"],
21
+ agentic: ["SWE-bench Verified", "tau2-bench", "Terminal-Bench", "BrowseComp"],
22
+ knowledge: ["MMLU", "GPQA Diamond", "SimpleQA"],
23
+ math: ["MATH-500", "AIME", "HMMT"],
24
+ instruction: ["IFEval", "FollowBench"],
25
+ multimodal: ["MMMU", "MathVista", "DocVQA"],
26
+ long_context: ["RULER", "MRCR", "LongBench"],
27
+ general: ["LMArena Overall", "MMLU", "IFEval"],
28
+ };
29
+ /** Guess the task category from its text (deliberately simple keyword scan). */
30
+ export function categorizeTask(taskText) {
31
+ const t = taskText.toLowerCase();
32
+ const has = (...words) => words.some((w) => t.includes(w));
33
+ if (has("image", "invoice", "screenshot", "ocr", "vision", "pdf page", "photo"))
34
+ return "multimodal";
35
+ if (has("agent", "tool call", "terminal", "browse", "multi-step", "workflow"))
36
+ return "agentic";
37
+ if (has("function", "code", "python", "typescript", "bug", "refactor", "compile", "sql"))
38
+ return "code";
39
+ if (has("prove", "equation", "math", "integral", "probability", "aime"))
40
+ return "math";
41
+ if (has("summarize", "extract", "format", "json", "follow the format", "instruction"))
42
+ return "instruction";
43
+ if (has("who ", "what is", "explain", "history", "science"))
44
+ return "knowledge";
45
+ return "general";
46
+ }
47
+ /** Small offline seed; overwrite/extend via `openmerit benchmarks import <file>`. */
48
+ export const SEED_DIGEST = {
49
+ updatedAt: "2026-09-01T00:00:00.000Z",
50
+ scores: {},
51
+ };
52
+ export function loadDigest() {
53
+ return readJson(paths.benchmarksDigest(), SEED_DIGEST);
54
+ }
55
+ export function saveDigest(d) {
56
+ writeJson(paths.benchmarksDigest(), d);
57
+ }
58
+ /**
59
+ * Prior 0..1 for a model on a task category, from the digest. Averages the
60
+ * category's benchmarks the model has scores for; returns null when the
61
+ * digest knows nothing about this model.
62
+ */
63
+ export function benchmarkPrior(taskCategory, modelId, digest) {
64
+ const benches = TASK_BENCHMARKS[taskCategory] ?? TASK_BENCHMARKS.general;
65
+ const modelScores = digest.scores[modelId];
66
+ if (!modelScores)
67
+ return null;
68
+ const hits = benches
69
+ .map((b) => modelScores[b])
70
+ .filter((s) => typeof s === "number");
71
+ if (hits.length === 0)
72
+ return null;
73
+ return hits.reduce((a, b) => a + b, 0) / hits.length / 100;
74
+ }
75
+ /**
76
+ * Which benchmarks should we check when a NEW model appears for this task?
77
+ * Returns the list to display in "worth testing" reasoning.
78
+ */
79
+ export function relevantBenchmarks(taskText) {
80
+ const category = categorizeTask(taskText);
81
+ return { category, benchmarks: TASK_BENCHMARKS[category] ?? TASK_BENCHMARKS.general };
82
+ }
83
+ /** Live OpenRouter benchmark scores shortlist candidates; task trials still decide quality. */
84
+ export async function openRouterBenchmarkCandidates(key, category, catalog) {
85
+ const metric = category === "code" ? "coding_index"
86
+ : category === "agentic" ? "agentic_index" : "intelligence_index";
87
+ const response = await fetch("https://openrouter.ai/api/v1/benchmarks?source=artificial-analysis", {
88
+ headers: { Authorization: `Bearer ${key}` },
89
+ });
90
+ if (!response.ok)
91
+ throw new Error(`OpenRouter benchmarks HTTP ${response.status}`);
92
+ const payload = await response.json();
93
+ return (payload.data ?? [])
94
+ .map((row) => ({ id: String(row.model_permaslug ?? ""), score: Number(row[metric]) }))
95
+ .filter((row) => catalog.has(row.id) && Number.isFinite(row.score))
96
+ .sort((a, b) => b.score - a.score)
97
+ .slice(0, 20).map((row) => row.id);
98
+ }
@@ -0,0 +1,51 @@
1
+ /** OpenRouter catalog fetch + snapshot diffing: the "new model release" watch. */
2
+ import { readJson, paths, writeJson } from "./store.js";
3
+ const OR = "https://openrouter.ai/api/v1";
4
+ export async function fetchCatalog(key) {
5
+ const res = await fetch(`${OR}/models`, {
6
+ headers: { Authorization: `Bearer ${key}` },
7
+ });
8
+ if (!res.ok)
9
+ throw new Error(`OpenRouter /models HTTP ${res.status}`);
10
+ const raw = (await res.json());
11
+ const cat = new Map();
12
+ for (const m of raw.data ?? []) {
13
+ const mm = m;
14
+ const id = String(mm.id ?? "");
15
+ if (!id || id.startsWith("openrouter/"))
16
+ continue; // skip meta-routers
17
+ const pricing = (mm.pricing ?? {});
18
+ const architecture = (mm.architecture ?? {});
19
+ const modalities = architecture.input_modalities;
20
+ const pp = Number(pricing.prompt ?? 0);
21
+ const pc = Number(pricing.completion ?? 0);
22
+ if (!(pp >= 0 && pc >= 0))
23
+ continue; // skip bogus pricing
24
+ cat.set(id, {
25
+ id,
26
+ name: String(mm.name ?? id),
27
+ ctx: Number(mm.context_length ?? 0),
28
+ pp,
29
+ pc,
30
+ price: Math.round((((pp + pc) / 2) * 1e6 + Number.EPSILON) * 1e4) / 1e4,
31
+ inputModalities: Array.isArray(modalities) ? modalities.map(String) : [],
32
+ created: typeof mm.created === "number" ? mm.created : undefined,
33
+ });
34
+ }
35
+ return cat;
36
+ }
37
+ /** Diff the live catalog against the stored snapshot; returns newly appeared models. */
38
+ export function diffNewModels(cat, snapshot) {
39
+ const seen = new Set(snapshot.ids);
40
+ return [...cat.values()].filter((e) => !seen.has(e.id));
41
+ }
42
+ export function loadSnapshot() {
43
+ return readJson(paths.catalogSnapshot(), { takenAt: "", ids: [] });
44
+ }
45
+ export function saveSnapshot(cat) {
46
+ writeJson(paths.catalogSnapshot(), {
47
+ takenAt: new Date().toISOString(),
48
+ ids: [...cat.keys()].sort(),
49
+ entries: [...cat.values()].sort((a, b) => a.id.localeCompare(b.id)),
50
+ });
51
+ }
package/dist/cli.js ADDED
@@ -0,0 +1,261 @@
1
+ #!/usr/bin/env node
2
+ /** openmerit CLI: init / watch / trial / frontier / recommend / status. */
3
+ import { copyFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
4
+ import { dirname, join } from "node:path";
5
+ import { fileURLToPath } from "node:url";
6
+ import { loadPolicy } from "./policy.js";
7
+ import { loadKey } from "./llm.js";
8
+ import { fetchCatalog } from "./catalog.js";
9
+ import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
10
+ import { buildRecommendation } from "./recommend.js";
11
+ import { appendJsonl, paths, readJson, readJsonl, taskKey } from "./store.js";
12
+ import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
13
+ import { budgetOk, recordTrialSpend } from "./trials.js";
14
+ import { availablePiModels, recordedActiveTask, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
15
+ import { pickNext, STRAT_PREFS } from "./strategist.js";
16
+ import { JUDGE_PREFS } from "./judge.js";
17
+ import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
18
+ const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
19
+ function resolvePref(cat, prefs, label) {
20
+ for (const p of prefs)
21
+ if (p && cat.has(p))
22
+ return p;
23
+ for (const p of prefs) {
24
+ if (!p)
25
+ continue;
26
+ for (const id of [...cat.keys()].sort())
27
+ if (id.includes(p))
28
+ return id;
29
+ }
30
+ throw new Error(`could not resolve ${label} model`);
31
+ }
32
+ function cmdInit() {
33
+ const policyPath = paths.policy();
34
+ mkdirSync(dirname(policyPath), { recursive: true });
35
+ if (!existsSync(policyPath)) {
36
+ copyFileSync(join(ROOT, "instructions", "openmerit.policy.json"), policyPath);
37
+ console.log(`policy -> ${policyPath}`);
38
+ }
39
+ else {
40
+ console.log(`policy -> ${policyPath} (kept existing)`);
41
+ }
42
+ console.log(`instr. -> ${join(ROOT, "instructions", "OPENMERIT.md")} (add to your harness's AGENTS.md/context)`);
43
+ console.log(`ext. -> ${join(ROOT, "extension", "openmerit.ts")}`);
44
+ console.log(`pi npm -> pi install npm:openmerit`);
45
+ console.log(`pi local-> pi install "${ROOT}"`);
46
+ console.log("\nNext: set OPENROUTER_API_KEY (env or ~/.openmerit/.env), install one extension source into pi, then start pi; the extension launches comparisons automatically.");
47
+ }
48
+ async function cmdTrial(taskFile, rounds) {
49
+ const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
50
+ const policy = loadPolicy(paths.policy());
51
+ const key = loadKey();
52
+ console.log("fetching OpenRouter catalog...");
53
+ const cat = await fetchCatalog(key);
54
+ const piModels = availablePiModels();
55
+ for (const id of [...cat.keys()])
56
+ if (!piModels.has(id))
57
+ cat.delete(id);
58
+ if (!cat.has(cfg.initial_model))
59
+ throw new Error(`initial_model ${cfg.initial_model} is unavailable in pi's OpenRouter registry`);
60
+ console.log(`catalog: ${cat.size} models selectable in pi`);
61
+ const judge = resolvePref(cat, [cfg.judge_model ?? policy.judge_model, ...JUDGE_PREFS], "judge");
62
+ const strat = resolvePref(cat, [cfg.strategist_model ?? policy.strategist_model, ...STRAT_PREFS], "strategist");
63
+ const { category, benchmarks } = relevantBenchmarks(cfg.task);
64
+ let benchmarkCandidates = [];
65
+ try {
66
+ benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, cat);
67
+ }
68
+ catch (e) {
69
+ console.log(`OpenRouter benchmark shortlist unavailable: ${e.message}`);
70
+ }
71
+ console.log(`judge: ${judge} | strategist: ${strat} | benchmarks: ${benchmarks.join(", ")}`);
72
+ console.log(`OpenRouter benchmark candidates: ${benchmarkCandidates.slice(0, 5).join(", ") || "none"}`);
73
+ const tKey = taskKey(cfg.task);
74
+ const maxPrice = cfg.max_usd_per_m ?? policy.max_usd_per_m;
75
+ const tried = new Set();
76
+ const results = [];
77
+ const failedVendors = new Set();
78
+ for (let i = 0; i < rounds; i++) {
79
+ const budget = budgetOk(policy);
80
+ if (!budget.ok) {
81
+ console.log(`budget: ${budget.reason}; stopping`);
82
+ break;
83
+ }
84
+ let model;
85
+ let why;
86
+ if (i === 0) {
87
+ model = cfg.initial_model;
88
+ why = "initial model";
89
+ }
90
+ else {
91
+ const pick = await pickNext(key, strat, cfg.task, cfg.eval, benchmarks, results, cat, tried, maxPrice, failedVendors, benchmarkCandidates);
92
+ if (!pick)
93
+ break;
94
+ model = pick.model;
95
+ why = pick.why;
96
+ }
97
+ if (tried.has(model))
98
+ break;
99
+ console.log(`[round ${i + 1}/${rounds}] ${model} (${why}) ...`);
100
+ const observed = i === 0 ? recordedActiveTask(cfg.task, model) : null;
101
+ if (i === 0)
102
+ console.log(observed ? " using model A's active pi session trace" : " active trace unavailable; running model A in a fresh pi session");
103
+ const { point, costUsd, error, sessionId } = observed
104
+ ? await scorePiRun(key, judge, cfg.task, cfg.eval, model, cat.get(model), observed, "trace")
105
+ : await runPiTrial(key, judge, cfg.task, cfg.eval, model, cat.get(model));
106
+ tried.add(model);
107
+ results.push(point);
108
+ recordTrialSpend(costUsd);
109
+ appendJsonl(paths.trials(), { ...point, taskKey: tKey });
110
+ if (sessionId)
111
+ console.log(` pi trace session: ${sessionId}`);
112
+ if (error) {
113
+ failedVendors.add(model.split("/")[0]);
114
+ console.log(` FAILED: ${error.slice(0, 160)}`);
115
+ }
116
+ else {
117
+ console.log(` score=${point.score.toFixed(2)} $/M=${point.price} est_cost=$${costUsd.toFixed(4)} ${(point.why ?? "").slice(0, 110)}`);
118
+ }
119
+ }
120
+ const pts = results.filter((p) => p.score > 0);
121
+ if (pts.length === 0) {
122
+ console.log("no trials completed");
123
+ process.exit(1);
124
+ }
125
+ const chain = paretoFrontier(pts);
126
+ const best = pickBest(pts);
127
+ const fb = pickFallback(pts, best, policy.fallback.min_score);
128
+ console.log("\n=== pareto frontier (nondominated: score up, price down, latency down) ===");
129
+ for (const p of chain)
130
+ console.log(` ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M`);
131
+ console.log(`\nBEST FIT : ${best.model} (score ${best.score.toFixed(2)}, $${best.price.toFixed(2)}/M)`);
132
+ console.log(fb ? `FALLBACK : ${fb.model} (score ${fb.score.toFixed(2)}, $${fb.price.toFixed(2)}/M)` : "FALLBACK : n/a");
133
+ const rec = buildRecommendation(tKey, cfg.task.slice(0, 120), cfg.initial_model, pts, policy);
134
+ if (rec) {
135
+ appendJsonl(paths.recommendations(), rec);
136
+ console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${rec.policy.autoApply})`);
137
+ }
138
+ }
139
+ function loadFrontiers() {
140
+ const trials = readJsonl(paths.trials());
141
+ const observations = readJsonl(paths.observations());
142
+ const labelByTask = new Map();
143
+ for (const o of observations)
144
+ if (!labelByTask.has(o.taskKey))
145
+ labelByTask.set(o.taskKey, o.taskLabel);
146
+ const byTask = new Map();
147
+ for (const t of trials) {
148
+ if (!t.taskKey)
149
+ continue;
150
+ byTask.set(t.taskKey, [...(byTask.get(t.taskKey) ?? []), t]);
151
+ }
152
+ const policy = loadPolicy(paths.policy());
153
+ return [...byTask.entries()].map(([tKey, points]) => {
154
+ const scored = points.filter((p) => p.score > 0);
155
+ const best = pickBest(scored);
156
+ return {
157
+ taskKey: tKey,
158
+ taskLabel: labelByTask.get(tKey) ?? tKey,
159
+ points,
160
+ frontier: paretoFrontier(scored).map((p) => p.model),
161
+ bestFit: best?.model,
162
+ fallback: best ? pickFallback(scored, best, policy.fallback.min_score)?.model : undefined,
163
+ updatedAt: new Date().toISOString(),
164
+ };
165
+ });
166
+ }
167
+ function cmdFrontier(filter) {
168
+ const frontiers = loadFrontiers().filter((f) => !filter || f.taskKey.startsWith(filter));
169
+ if (frontiers.length === 0) {
170
+ console.log("no frontiers yet (run `openmerit trial` or let `openmerit watch` collect trials)");
171
+ return;
172
+ }
173
+ for (const f of frontiers) {
174
+ console.log(`\ntask ${f.taskKey} "${f.taskLabel}"`);
175
+ const pts = [...f.points].sort((a, b) => a.price - b.price);
176
+ for (const p of pts) {
177
+ const mark = f.frontier.includes(p.model) ? "*" : " ";
178
+ console.log(` ${mark} ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M` +
179
+ (p.latencyMs ? ` ${Math.round(p.latencyMs)}ms` : ""));
180
+ }
181
+ console.log(` best=${f.bestFit ?? "n/a"} fallback=${f.fallback ?? "n/a"} (* = pareto frontier)`);
182
+ }
183
+ }
184
+ function cmdStatus() {
185
+ const harness = readJson(paths.harnessState(), {});
186
+ const queue = readJson(paths.candidates(), { newModels: [] });
187
+ const latestById = new Map();
188
+ for (const r of readJsonl(paths.recommendations()))
189
+ latestById.set(r.id, r);
190
+ const recs = [...latestById.values()].filter((r) => r.status === "pending");
191
+ const frontiers = loadFrontiers();
192
+ console.log(`harness model : ${harness.currentModel ?? "unknown"} (as of ${harness.updatedAt ?? "n/a"})`);
193
+ console.log(`tasks tracked : ${frontiers.length}`);
194
+ console.log(`new models queued for trial: ${queue.newModels.length}${queue.newModels.length ? " — " + queue.newModels.slice(0, 5).map((m) => m.id).join(", ") + (queue.newModels.length > 5 ? "…" : "") : ""}`);
195
+ console.log(`pending recommendations: ${recs.length}`);
196
+ for (const r of recs.slice(-5)) {
197
+ console.log(` - ${r.recommended.model} <- ${r.currentModel ?? "?"} (gain ${r.evidence.scoreGain}, auto=${r.policy.autoApply})`);
198
+ }
199
+ }
200
+ async function main() {
201
+ const [cmd, ...args] = process.argv.slice(2);
202
+ const policy = loadPolicy(paths.policy());
203
+ switch (cmd) {
204
+ case "init":
205
+ cmdInit();
206
+ break;
207
+ case "session-trial": {
208
+ const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText] = args;
209
+ if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
210
+ throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
211
+ const sessionBytes = Number(bytesText);
212
+ if (!Number.isSafeInteger(sessionBytes) || sessionBytes <= 0)
213
+ throw new Error("invalid session byte limit");
214
+ const task = settledActiveTask({ sessionFile, settledAt, currentModel, settledTaskKey, cwd, sessionBytes });
215
+ if (!task)
216
+ throw new Error("the specified pi session has no completed, supported task matching this marker");
217
+ await autoTaskTick(loadKey(), policy, task);
218
+ break;
219
+ }
220
+ case "watch":
221
+ if (args.includes("--once"))
222
+ await tickOnce(policy);
223
+ else
224
+ await runDaemon(policy);
225
+ break;
226
+ case "trial": {
227
+ const file = args[0];
228
+ if (!file)
229
+ throw new Error("usage: openmerit trial <task.json> [--rounds N]");
230
+ const rIdx = args.indexOf("--rounds");
231
+ await cmdTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
232
+ break;
233
+ }
234
+ case "frontier":
235
+ cmdFrontier(args[0]);
236
+ break;
237
+ case "recommend":
238
+ recommendTick(policy);
239
+ break;
240
+ case "status":
241
+ cmdStatus();
242
+ break;
243
+ default:
244
+ console.log(`openmerit — external model-merit harness
245
+
246
+ usage: openmerit <command>
247
+
248
+ init set up ~/.openmerit (policy, state dirs)
249
+ watch [--once] observe completed pi tasks, trial candidates sequentially, and recommend session swaps
250
+ trial <task.json> iterative model search on one task spec [--rounds N]
251
+ frontier [taskKey] print pareto frontier(s)
252
+ recommend emit recommendations now
253
+ status harness model, queued models, pending recommendations`);
254
+ if (cmd && cmd !== "help" && cmd !== "--help")
255
+ process.exitCode = 1;
256
+ }
257
+ }
258
+ main().catch((e) => {
259
+ console.error(`openmerit: ${e.message}`);
260
+ process.exit(1);
261
+ });