openmerit 0.1.3 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/README.md +83 -312
- package/dist/core/src/index.d.ts +90 -0
- package/dist/core/src/index.js +1137 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +15 -0
- package/dist/pi/src/index.js +423 -0
- package/dist/protocol/src/index.d.ts +402 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +450 -0
- package/dist/protocol/src/schemas.js +224 -0
- package/docs/adapter-guide.md +172 -0
- package/docs/architecture.md +55 -0
- package/docs/automation.md +66 -0
- package/docs/getting-started.md +55 -0
- package/docs/lifecycle.md +30 -0
- package/docs/metrics-and-evidence.md +40 -0
- package/docs/operations.md +31 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +44 -0
- package/docs/roadmap.md +26 -0
- package/docs/security.md +23 -0
- package/docs/testing.md +36 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +45 -54
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -388
- package/dist/diagnostics.js +0 -194
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -203
- package/dist/pi-trials.js +0 -366
- package/dist/policy.js +0 -115
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -59
- package/dist/standalone.js +0 -220
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -820
- package/instructions/OPENMERIT.md +0 -54
- package/instructions/openmerit.policy.json +0 -33
- package/rules.md +0 -39
package/dist/benchmarks.js
DELETED
|
@@ -1,98 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Public benchmark signals.
|
|
3
|
-
*
|
|
4
|
-
* Two jobs:
|
|
5
|
-
* 1. Map a task to the public benchmarks that matter for it (so the
|
|
6
|
-
* strategist and the user know why a candidate is "worth testing").
|
|
7
|
-
* 2. Provide a prior score for a model from a benchmark digest, so a newly
|
|
8
|
-
* released model can be ranked against the existing frontier before we
|
|
9
|
-
* spend money on shadow trials.
|
|
10
|
-
*
|
|
11
|
-
* The digest lives at ~/.openmerit/benchmarks/digest.json and is a simple
|
|
12
|
-
* map: { "vendor/model": { "benchmark_name": score_0_to_100, ... }, ... }.
|
|
13
|
-
* Refresh it from public leaderboards (LMArena, Artificial Analysis,
|
|
14
|
-
* SWE-bench, ...) however you like; a small seed ships below so the system
|
|
15
|
-
* works offline.
|
|
16
|
-
*/
|
|
17
|
-
import { paths, readJson, writeJson } from "./store.js";
|
|
18
|
-
/** Task category -> public benchmarks that are predictive for it. */
|
|
19
|
-
export const TASK_BENCHMARKS = {
|
|
20
|
-
code: ["SWE-bench Verified", "HumanEval", "LMArena Code", "Aider Polyglot"],
|
|
21
|
-
agentic: ["SWE-bench Verified", "tau2-bench", "Terminal-Bench", "BrowseComp"],
|
|
22
|
-
knowledge: ["MMLU", "GPQA Diamond", "SimpleQA"],
|
|
23
|
-
math: ["MATH-500", "AIME", "HMMT"],
|
|
24
|
-
instruction: ["IFEval", "FollowBench"],
|
|
25
|
-
multimodal: ["MMMU", "MathVista", "DocVQA"],
|
|
26
|
-
long_context: ["RULER", "MRCR", "LongBench"],
|
|
27
|
-
general: ["LMArena Overall", "MMLU", "IFEval"],
|
|
28
|
-
};
|
|
29
|
-
/** Guess the task category from its text (deliberately simple keyword scan). */
|
|
30
|
-
export function categorizeTask(taskText) {
|
|
31
|
-
const t = taskText.toLowerCase();
|
|
32
|
-
const has = (...words) => words.some((w) => t.includes(w));
|
|
33
|
-
if (has("image", "invoice", "screenshot", "ocr", "vision", "pdf page", "photo"))
|
|
34
|
-
return "multimodal";
|
|
35
|
-
if (has("agent", "tool call", "terminal", "browse", "multi-step", "workflow"))
|
|
36
|
-
return "agentic";
|
|
37
|
-
if (has("function", "code", "python", "typescript", "bug", "refactor", "compile", "sql"))
|
|
38
|
-
return "code";
|
|
39
|
-
if (has("prove", "equation", "math", "integral", "probability", "aime"))
|
|
40
|
-
return "math";
|
|
41
|
-
if (has("summarize", "extract", "format", "json", "follow the format", "instruction"))
|
|
42
|
-
return "instruction";
|
|
43
|
-
if (has("who ", "what is", "explain", "history", "science"))
|
|
44
|
-
return "knowledge";
|
|
45
|
-
return "general";
|
|
46
|
-
}
|
|
47
|
-
/** Small offline seed; overwrite/extend via `openmerit benchmarks import <file>`. */
|
|
48
|
-
export const SEED_DIGEST = {
|
|
49
|
-
updatedAt: "2026-09-01T00:00:00.000Z",
|
|
50
|
-
scores: {},
|
|
51
|
-
};
|
|
52
|
-
export function loadDigest() {
|
|
53
|
-
return readJson(paths.benchmarksDigest(), SEED_DIGEST);
|
|
54
|
-
}
|
|
55
|
-
export function saveDigest(d) {
|
|
56
|
-
writeJson(paths.benchmarksDigest(), d);
|
|
57
|
-
}
|
|
58
|
-
/**
|
|
59
|
-
* Prior 0..1 for a model on a task category, from the digest. Averages the
|
|
60
|
-
* category's benchmarks the model has scores for; returns null when the
|
|
61
|
-
* digest knows nothing about this model.
|
|
62
|
-
*/
|
|
63
|
-
export function benchmarkPrior(taskCategory, modelId, digest) {
|
|
64
|
-
const benches = TASK_BENCHMARKS[taskCategory] ?? TASK_BENCHMARKS.general;
|
|
65
|
-
const modelScores = digest.scores[modelId];
|
|
66
|
-
if (!modelScores)
|
|
67
|
-
return null;
|
|
68
|
-
const hits = benches
|
|
69
|
-
.map((b) => modelScores[b])
|
|
70
|
-
.filter((s) => typeof s === "number");
|
|
71
|
-
if (hits.length === 0)
|
|
72
|
-
return null;
|
|
73
|
-
return hits.reduce((a, b) => a + b, 0) / hits.length / 100;
|
|
74
|
-
}
|
|
75
|
-
/**
|
|
76
|
-
* Which benchmarks should we check when a NEW model appears for this task?
|
|
77
|
-
* Returns the list to display in "worth testing" reasoning.
|
|
78
|
-
*/
|
|
79
|
-
export function relevantBenchmarks(taskText) {
|
|
80
|
-
const category = categorizeTask(taskText);
|
|
81
|
-
return { category, benchmarks: TASK_BENCHMARKS[category] ?? TASK_BENCHMARKS.general };
|
|
82
|
-
}
|
|
83
|
-
/** Live OpenRouter benchmark scores shortlist candidates; task trials still decide quality. */
|
|
84
|
-
export async function openRouterBenchmarkCandidates(key, category, catalog) {
|
|
85
|
-
const metric = category === "code" ? "coding_index"
|
|
86
|
-
: category === "agentic" ? "agentic_index" : "intelligence_index";
|
|
87
|
-
const response = await fetch("https://openrouter.ai/api/v1/benchmarks?source=artificial-analysis", {
|
|
88
|
-
headers: { Authorization: `Bearer ${key}` },
|
|
89
|
-
});
|
|
90
|
-
if (!response.ok)
|
|
91
|
-
throw new Error(`OpenRouter benchmarks HTTP ${response.status}`);
|
|
92
|
-
const payload = await response.json();
|
|
93
|
-
return (payload.data ?? [])
|
|
94
|
-
.map((row) => ({ id: String(row.model_permaslug ?? ""), score: Number(row[metric]) }))
|
|
95
|
-
.filter((row) => catalog.has(row.id) && Number.isFinite(row.score))
|
|
96
|
-
.sort((a, b) => b.score - a.score)
|
|
97
|
-
.slice(0, 20).map((row) => row.id);
|
|
98
|
-
}
|
package/dist/catalog.js
DELETED
|
@@ -1,61 +0,0 @@
|
|
|
1
|
-
/** OpenRouter catalog fetch + snapshot diffing: the "new model release" watch. */
|
|
2
|
-
import { readJson, paths, writeJson } from "./store.js";
|
|
3
|
-
const OR = "https://openrouter.ai/api/v1";
|
|
4
|
-
export class OpenRouterCatalogProvider {
|
|
5
|
-
key;
|
|
6
|
-
id = "openrouter";
|
|
7
|
-
constructor(key) {
|
|
8
|
-
this.key = key;
|
|
9
|
-
}
|
|
10
|
-
fetchCatalog() { return fetchCatalog(this.key); }
|
|
11
|
-
}
|
|
12
|
-
export async function fetchCatalog(key, provider = "openrouter") {
|
|
13
|
-
if (provider !== "openrouter")
|
|
14
|
-
return new Map();
|
|
15
|
-
const res = await fetch(`${OR}/models`, {
|
|
16
|
-
headers: { Authorization: `Bearer ${key}` },
|
|
17
|
-
});
|
|
18
|
-
if (!res.ok)
|
|
19
|
-
throw new Error(`OpenRouter /models HTTP ${res.status}`);
|
|
20
|
-
const raw = (await res.json());
|
|
21
|
-
const cat = new Map();
|
|
22
|
-
for (const m of raw.data ?? []) {
|
|
23
|
-
const mm = m;
|
|
24
|
-
const id = String(mm.id ?? "");
|
|
25
|
-
if (!id || id.startsWith("openrouter/"))
|
|
26
|
-
continue; // skip meta-routers
|
|
27
|
-
const pricing = (mm.pricing ?? {});
|
|
28
|
-
const architecture = (mm.architecture ?? {});
|
|
29
|
-
const modalities = architecture.input_modalities;
|
|
30
|
-
const pp = Number(pricing.prompt ?? 0);
|
|
31
|
-
const pc = Number(pricing.completion ?? 0);
|
|
32
|
-
if (!(pp >= 0 && pc >= 0))
|
|
33
|
-
continue; // skip bogus pricing
|
|
34
|
-
cat.set(id, {
|
|
35
|
-
id,
|
|
36
|
-
name: String(mm.name ?? id),
|
|
37
|
-
ctx: Number(mm.context_length ?? 0),
|
|
38
|
-
pp,
|
|
39
|
-
pc,
|
|
40
|
-
price: Math.round((((pp + pc) / 2) * 1e6 + Number.EPSILON) * 1e4) / 1e4,
|
|
41
|
-
inputModalities: Array.isArray(modalities) ? modalities.map(String) : [],
|
|
42
|
-
created: typeof mm.created === "number" ? mm.created : undefined,
|
|
43
|
-
});
|
|
44
|
-
}
|
|
45
|
-
return cat;
|
|
46
|
-
}
|
|
47
|
-
/** Diff the live catalog against the stored snapshot; returns newly appeared models. */
|
|
48
|
-
export function diffNewModels(cat, snapshot) {
|
|
49
|
-
const seen = new Set(snapshot.ids);
|
|
50
|
-
return [...cat.values()].filter((e) => !seen.has(e.id));
|
|
51
|
-
}
|
|
52
|
-
export function loadSnapshot() {
|
|
53
|
-
return readJson(paths.catalogSnapshot(), { takenAt: "", ids: [] });
|
|
54
|
-
}
|
|
55
|
-
export function saveSnapshot(cat) {
|
|
56
|
-
writeJson(paths.catalogSnapshot(), {
|
|
57
|
-
takenAt: new Date().toISOString(),
|
|
58
|
-
ids: [...cat.keys()].sort(),
|
|
59
|
-
entries: [...cat.values()].sort((a, b) => a.id.localeCompare(b.id)),
|
|
60
|
-
});
|
|
61
|
-
}
|
package/dist/cli.js
DELETED
|
@@ -1,188 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor / verify. */
|
|
3
|
-
import { copyFileSync, existsSync, mkdirSync } from "node:fs";
|
|
4
|
-
import { dirname, join } from "node:path";
|
|
5
|
-
import { fileURLToPath } from "node:url";
|
|
6
|
-
import { loadPolicy } from "./policy.js";
|
|
7
|
-
import { loadKey } from "./llm.js";
|
|
8
|
-
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
9
|
-
import { paths, readJson, readJsonl } from "./store.js";
|
|
10
|
-
import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
|
|
11
|
-
import { settledActiveTask } from "./pi-trials.js";
|
|
12
|
-
import { modelRoute, routeLabel } from "./routes.js";
|
|
13
|
-
import { runStandaloneTrial } from "./standalone.js";
|
|
14
|
-
import { collectDoctorReport, renderDiagnosticReport, runOfflineVerify } from "./diagnostics.js";
|
|
15
|
-
const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
16
|
-
function cmdInit() {
|
|
17
|
-
const policyPath = paths.policy();
|
|
18
|
-
mkdirSync(dirname(policyPath), { recursive: true });
|
|
19
|
-
if (!existsSync(policyPath)) {
|
|
20
|
-
copyFileSync(join(ROOT, "instructions", "openmerit.policy.json"), policyPath);
|
|
21
|
-
console.log(`policy -> ${policyPath}`);
|
|
22
|
-
}
|
|
23
|
-
else {
|
|
24
|
-
console.log(`policy -> ${policyPath} (kept existing)`);
|
|
25
|
-
}
|
|
26
|
-
console.log(`instr. -> ${join(ROOT, "instructions", "OPENMERIT.md")} (add to your harness's AGENTS.md/context)`);
|
|
27
|
-
console.log(`ext. -> ${join(ROOT, "extension", "openmerit.ts")}`);
|
|
28
|
-
console.log(`pi npm -> pi install npm:openmerit`);
|
|
29
|
-
console.log(`pi local-> pi install "${ROOT}"`);
|
|
30
|
-
console.log("\nNext: authenticate at least two models in pi, install one extension source, then start pi. OpenRouter is optional enrichment; candidate tools are disabled by default.");
|
|
31
|
-
}
|
|
32
|
-
function optionalOpenRouterKey() {
|
|
33
|
-
try {
|
|
34
|
-
return loadKey();
|
|
35
|
-
}
|
|
36
|
-
catch {
|
|
37
|
-
return null;
|
|
38
|
-
}
|
|
39
|
-
}
|
|
40
|
-
function printReport(report, json) {
|
|
41
|
-
console.log(json ? JSON.stringify(report, null, 2) : renderDiagnosticReport(report));
|
|
42
|
-
if (!report.ok)
|
|
43
|
-
process.exitCode = 1;
|
|
44
|
-
}
|
|
45
|
-
function cmdDoctor(json) {
|
|
46
|
-
printReport(collectDoctorReport(), json);
|
|
47
|
-
}
|
|
48
|
-
function cmdVerify(json) {
|
|
49
|
-
printReport(runOfflineVerify(), json);
|
|
50
|
-
}
|
|
51
|
-
function loadFrontiers() {
|
|
52
|
-
const trials = readJsonl(paths.trials());
|
|
53
|
-
const observations = readJsonl(paths.observations());
|
|
54
|
-
const labelByTask = new Map();
|
|
55
|
-
for (const o of observations)
|
|
56
|
-
if (!labelByTask.has(o.taskKey))
|
|
57
|
-
labelByTask.set(o.taskKey, o.taskLabel);
|
|
58
|
-
const byTask = new Map();
|
|
59
|
-
for (const t of trials) {
|
|
60
|
-
if (!t.taskKey)
|
|
61
|
-
continue;
|
|
62
|
-
byTask.set(t.taskKey, [...(byTask.get(t.taskKey) ?? []), t]);
|
|
63
|
-
}
|
|
64
|
-
const policy = loadPolicy(paths.policy());
|
|
65
|
-
return [...byTask.entries()].map(([tKey, points]) => {
|
|
66
|
-
const scored = points.filter((p) => p.score > 0);
|
|
67
|
-
const best = pickBest(scored);
|
|
68
|
-
return {
|
|
69
|
-
taskKey: tKey,
|
|
70
|
-
taskLabel: labelByTask.get(tKey) ?? tKey,
|
|
71
|
-
points,
|
|
72
|
-
frontier: paretoFrontier(scored).map((p) => p.model),
|
|
73
|
-
bestFit: best?.model,
|
|
74
|
-
fallback: best ? pickFallback(scored, best, policy.fallback.min_score)?.model : undefined,
|
|
75
|
-
updatedAt: new Date().toISOString(),
|
|
76
|
-
};
|
|
77
|
-
});
|
|
78
|
-
}
|
|
79
|
-
function cmdFrontier(filter) {
|
|
80
|
-
const frontiers = loadFrontiers().filter((f) => !filter || f.taskKey.startsWith(filter));
|
|
81
|
-
if (frontiers.length === 0) {
|
|
82
|
-
console.log("no frontiers yet (run `openmerit trial` or let `openmerit watch` collect trials)");
|
|
83
|
-
return;
|
|
84
|
-
}
|
|
85
|
-
for (const f of frontiers) {
|
|
86
|
-
console.log(`\ntask ${f.taskKey} "${f.taskLabel}"`);
|
|
87
|
-
const pts = [...f.points].sort((a, b) => a.price - b.price);
|
|
88
|
-
for (const p of pts) {
|
|
89
|
-
const mark = f.frontier.includes(p.model) ? "*" : " ";
|
|
90
|
-
console.log(` ${mark} ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M` +
|
|
91
|
-
(p.latencyMs ? ` ${Math.round(p.latencyMs)}ms` : ""));
|
|
92
|
-
}
|
|
93
|
-
console.log(` best=${f.bestFit ?? "n/a"} fallback=${f.fallback ?? "n/a"} (* = pareto frontier)`);
|
|
94
|
-
}
|
|
95
|
-
}
|
|
96
|
-
function cmdStatus() {
|
|
97
|
-
const harness = readJson(paths.harnessState(), {});
|
|
98
|
-
const queue = readJson(paths.candidates(), { newModels: [] });
|
|
99
|
-
const latestById = new Map();
|
|
100
|
-
for (const r of readJsonl(paths.recommendations()))
|
|
101
|
-
latestById.set(r.id, r);
|
|
102
|
-
const recs = [...latestById.values()].filter((r) => r.status === "pending");
|
|
103
|
-
const frontiers = loadFrontiers();
|
|
104
|
-
console.log(`harness model : ${harness.currentRoute ? routeLabel(harness.currentRoute) : harness.currentModel ?? "unknown"} ` +
|
|
105
|
-
`(as of ${harness.updatedAt ?? "n/a"})`);
|
|
106
|
-
console.log(`tasks tracked : ${frontiers.length}`);
|
|
107
|
-
console.log(`new models queued for trial: ${queue.newModels.length}${queue.newModels.length ? " — " + queue.newModels.slice(0, 5).map((m) => m.id).join(", ") + (queue.newModels.length > 5 ? "…" : "") : ""}`);
|
|
108
|
-
console.log(`pending recommendations: ${recs.length}`);
|
|
109
|
-
for (const r of recs.slice(-5)) {
|
|
110
|
-
console.log(` - ${r.recommended.model} <- ${r.currentModel ?? "?"} (gain ${r.evidence.scoreGain}, auto=${r.policy.autoApply})`);
|
|
111
|
-
}
|
|
112
|
-
}
|
|
113
|
-
async function main() {
|
|
114
|
-
const [cmd, ...args] = process.argv.slice(2);
|
|
115
|
-
switch (cmd) {
|
|
116
|
-
case "init":
|
|
117
|
-
cmdInit();
|
|
118
|
-
break;
|
|
119
|
-
case "session-trial": {
|
|
120
|
-
const policy = loadPolicy(paths.policy());
|
|
121
|
-
const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
|
|
122
|
-
if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
|
|
123
|
-
throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
|
|
124
|
-
const sessionBytes = Number(bytesText);
|
|
125
|
-
if (!Number.isSafeInteger(sessionBytes) || sessionBytes <= 0)
|
|
126
|
-
throw new Error("invalid session byte limit");
|
|
127
|
-
const state = readJson(paths.harnessState(), {});
|
|
128
|
-
const currentRoute = provider && providerModelId
|
|
129
|
-
? state.routes?.find((route) => route.provider === provider && route.modelId === providerModelId) ??
|
|
130
|
-
modelRoute(provider, providerModelId)
|
|
131
|
-
: state.currentRoute;
|
|
132
|
-
const task = settledActiveTask({ sessionFile, settledAt, currentModel, currentRoute,
|
|
133
|
-
routes: state.routes, settledTaskKey, cwd, sessionBytes });
|
|
134
|
-
if (!task)
|
|
135
|
-
throw new Error("the specified pi session has no completed, supported task matching this marker");
|
|
136
|
-
await autoTaskTick(optionalOpenRouterKey(), policy, task);
|
|
137
|
-
break;
|
|
138
|
-
}
|
|
139
|
-
case "watch":
|
|
140
|
-
if (args.includes("--once"))
|
|
141
|
-
await tickOnce(loadPolicy(paths.policy()));
|
|
142
|
-
else
|
|
143
|
-
await runDaemon(loadPolicy(paths.policy()));
|
|
144
|
-
break;
|
|
145
|
-
case "trial": {
|
|
146
|
-
const file = args[0];
|
|
147
|
-
if (!file)
|
|
148
|
-
throw new Error("usage: openmerit trial <task.json> [--rounds N]");
|
|
149
|
-
const rIdx = args.indexOf("--rounds");
|
|
150
|
-
await runStandaloneTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
|
|
151
|
-
break;
|
|
152
|
-
}
|
|
153
|
-
case "frontier":
|
|
154
|
-
cmdFrontier(args[0]);
|
|
155
|
-
break;
|
|
156
|
-
case "recommend":
|
|
157
|
-
recommendTick(loadPolicy(paths.policy()));
|
|
158
|
-
break;
|
|
159
|
-
case "status":
|
|
160
|
-
cmdStatus();
|
|
161
|
-
break;
|
|
162
|
-
case "doctor":
|
|
163
|
-
cmdDoctor(args.includes("--json"));
|
|
164
|
-
break;
|
|
165
|
-
case "verify":
|
|
166
|
-
cmdVerify(args.includes("--json"));
|
|
167
|
-
break;
|
|
168
|
-
default:
|
|
169
|
-
console.log(`openmerit — external model-merit harness
|
|
170
|
-
|
|
171
|
-
usage: openmerit <command>
|
|
172
|
-
|
|
173
|
-
init set up ~/.openmerit (policy, state dirs)
|
|
174
|
-
watch [--once] observe completed pi tasks, trial candidates sequentially, and recommend session swaps
|
|
175
|
-
trial <task.json> iterative model search on one task spec [--rounds N]
|
|
176
|
-
frontier [taskKey] print pareto frontier(s)
|
|
177
|
-
recommend emit recommendations now
|
|
178
|
-
status harness model, queued models, pending recommendations
|
|
179
|
-
doctor [--json] diagnose pi, policy, routes, state, and installation
|
|
180
|
-
verify [--json] run an offline, provider-free core self-test`);
|
|
181
|
-
if (cmd && cmd !== "help" && cmd !== "--help")
|
|
182
|
-
process.exitCode = 1;
|
|
183
|
-
}
|
|
184
|
-
}
|
|
185
|
-
main().catch((e) => {
|
|
186
|
-
console.error(`openmerit: ${e.message}`);
|
|
187
|
-
process.exit(1);
|
|
188
|
-
});
|