openmerit 0.1.3 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -312
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -388
  52. package/dist/diagnostics.js +0 -194
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -203
  60. package/dist/pi-trials.js +0 -366
  61. package/dist/policy.js +0 -115
  62. package/dist/providers.js +0 -1
  63. package/dist/recommend.js +0 -76
  64. package/dist/routes.js +0 -59
  65. package/dist/standalone.js +0 -220
  66. package/dist/store.js +0 -89
  67. package/dist/strategist.js +0 -68
  68. package/dist/task-input.js +0 -54
  69. package/dist/traces.js +0 -127
  70. package/dist/trials.js +0 -140
  71. package/dist/types.js +0 -2
  72. package/examples/invoice-prompt.txt +0 -19
  73. package/examples/task.example.json +0 -7
  74. package/extension/openmerit.ts +0 -820
  75. package/instructions/OPENMERIT.md +0 -54
  76. package/instructions/openmerit.policy.json +0 -33
  77. package/rules.md +0 -39
@@ -1,98 +0,0 @@
1
- /**
2
- * Public benchmark signals.
3
- *
4
- * Two jobs:
5
- * 1. Map a task to the public benchmarks that matter for it (so the
6
- * strategist and the user know why a candidate is "worth testing").
7
- * 2. Provide a prior score for a model from a benchmark digest, so a newly
8
- * released model can be ranked against the existing frontier before we
9
- * spend money on shadow trials.
10
- *
11
- * The digest lives at ~/.openmerit/benchmarks/digest.json and is a simple
12
- * map: { "vendor/model": { "benchmark_name": score_0_to_100, ... }, ... }.
13
- * Refresh it from public leaderboards (LMArena, Artificial Analysis,
14
- * SWE-bench, ...) however you like; a small seed ships below so the system
15
- * works offline.
16
- */
17
- import { paths, readJson, writeJson } from "./store.js";
18
- /** Task category -> public benchmarks that are predictive for it. */
19
- export const TASK_BENCHMARKS = {
20
- code: ["SWE-bench Verified", "HumanEval", "LMArena Code", "Aider Polyglot"],
21
- agentic: ["SWE-bench Verified", "tau2-bench", "Terminal-Bench", "BrowseComp"],
22
- knowledge: ["MMLU", "GPQA Diamond", "SimpleQA"],
23
- math: ["MATH-500", "AIME", "HMMT"],
24
- instruction: ["IFEval", "FollowBench"],
25
- multimodal: ["MMMU", "MathVista", "DocVQA"],
26
- long_context: ["RULER", "MRCR", "LongBench"],
27
- general: ["LMArena Overall", "MMLU", "IFEval"],
28
- };
29
- /** Guess the task category from its text (deliberately simple keyword scan). */
30
- export function categorizeTask(taskText) {
31
- const t = taskText.toLowerCase();
32
- const has = (...words) => words.some((w) => t.includes(w));
33
- if (has("image", "invoice", "screenshot", "ocr", "vision", "pdf page", "photo"))
34
- return "multimodal";
35
- if (has("agent", "tool call", "terminal", "browse", "multi-step", "workflow"))
36
- return "agentic";
37
- if (has("function", "code", "python", "typescript", "bug", "refactor", "compile", "sql"))
38
- return "code";
39
- if (has("prove", "equation", "math", "integral", "probability", "aime"))
40
- return "math";
41
- if (has("summarize", "extract", "format", "json", "follow the format", "instruction"))
42
- return "instruction";
43
- if (has("who ", "what is", "explain", "history", "science"))
44
- return "knowledge";
45
- return "general";
46
- }
47
- /** Small offline seed; overwrite/extend via `openmerit benchmarks import <file>`. */
48
- export const SEED_DIGEST = {
49
- updatedAt: "2026-09-01T00:00:00.000Z",
50
- scores: {},
51
- };
52
- export function loadDigest() {
53
- return readJson(paths.benchmarksDigest(), SEED_DIGEST);
54
- }
55
- export function saveDigest(d) {
56
- writeJson(paths.benchmarksDigest(), d);
57
- }
58
- /**
59
- * Prior 0..1 for a model on a task category, from the digest. Averages the
60
- * category's benchmarks the model has scores for; returns null when the
61
- * digest knows nothing about this model.
62
- */
63
- export function benchmarkPrior(taskCategory, modelId, digest) {
64
- const benches = TASK_BENCHMARKS[taskCategory] ?? TASK_BENCHMARKS.general;
65
- const modelScores = digest.scores[modelId];
66
- if (!modelScores)
67
- return null;
68
- const hits = benches
69
- .map((b) => modelScores[b])
70
- .filter((s) => typeof s === "number");
71
- if (hits.length === 0)
72
- return null;
73
- return hits.reduce((a, b) => a + b, 0) / hits.length / 100;
74
- }
75
- /**
76
- * Which benchmarks should we check when a NEW model appears for this task?
77
- * Returns the list to display in "worth testing" reasoning.
78
- */
79
- export function relevantBenchmarks(taskText) {
80
- const category = categorizeTask(taskText);
81
- return { category, benchmarks: TASK_BENCHMARKS[category] ?? TASK_BENCHMARKS.general };
82
- }
83
- /** Live OpenRouter benchmark scores shortlist candidates; task trials still decide quality. */
84
- export async function openRouterBenchmarkCandidates(key, category, catalog) {
85
- const metric = category === "code" ? "coding_index"
86
- : category === "agentic" ? "agentic_index" : "intelligence_index";
87
- const response = await fetch("https://openrouter.ai/api/v1/benchmarks?source=artificial-analysis", {
88
- headers: { Authorization: `Bearer ${key}` },
89
- });
90
- if (!response.ok)
91
- throw new Error(`OpenRouter benchmarks HTTP ${response.status}`);
92
- const payload = await response.json();
93
- return (payload.data ?? [])
94
- .map((row) => ({ id: String(row.model_permaslug ?? ""), score: Number(row[metric]) }))
95
- .filter((row) => catalog.has(row.id) && Number.isFinite(row.score))
96
- .sort((a, b) => b.score - a.score)
97
- .slice(0, 20).map((row) => row.id);
98
- }
package/dist/catalog.js DELETED
@@ -1,61 +0,0 @@
1
- /** OpenRouter catalog fetch + snapshot diffing: the "new model release" watch. */
2
- import { readJson, paths, writeJson } from "./store.js";
3
- const OR = "https://openrouter.ai/api/v1";
4
- export class OpenRouterCatalogProvider {
5
- key;
6
- id = "openrouter";
7
- constructor(key) {
8
- this.key = key;
9
- }
10
- fetchCatalog() { return fetchCatalog(this.key); }
11
- }
12
- export async function fetchCatalog(key, provider = "openrouter") {
13
- if (provider !== "openrouter")
14
- return new Map();
15
- const res = await fetch(`${OR}/models`, {
16
- headers: { Authorization: `Bearer ${key}` },
17
- });
18
- if (!res.ok)
19
- throw new Error(`OpenRouter /models HTTP ${res.status}`);
20
- const raw = (await res.json());
21
- const cat = new Map();
22
- for (const m of raw.data ?? []) {
23
- const mm = m;
24
- const id = String(mm.id ?? "");
25
- if (!id || id.startsWith("openrouter/"))
26
- continue; // skip meta-routers
27
- const pricing = (mm.pricing ?? {});
28
- const architecture = (mm.architecture ?? {});
29
- const modalities = architecture.input_modalities;
30
- const pp = Number(pricing.prompt ?? 0);
31
- const pc = Number(pricing.completion ?? 0);
32
- if (!(pp >= 0 && pc >= 0))
33
- continue; // skip bogus pricing
34
- cat.set(id, {
35
- id,
36
- name: String(mm.name ?? id),
37
- ctx: Number(mm.context_length ?? 0),
38
- pp,
39
- pc,
40
- price: Math.round((((pp + pc) / 2) * 1e6 + Number.EPSILON) * 1e4) / 1e4,
41
- inputModalities: Array.isArray(modalities) ? modalities.map(String) : [],
42
- created: typeof mm.created === "number" ? mm.created : undefined,
43
- });
44
- }
45
- return cat;
46
- }
47
- /** Diff the live catalog against the stored snapshot; returns newly appeared models. */
48
- export function diffNewModels(cat, snapshot) {
49
- const seen = new Set(snapshot.ids);
50
- return [...cat.values()].filter((e) => !seen.has(e.id));
51
- }
52
- export function loadSnapshot() {
53
- return readJson(paths.catalogSnapshot(), { takenAt: "", ids: [] });
54
- }
55
- export function saveSnapshot(cat) {
56
- writeJson(paths.catalogSnapshot(), {
57
- takenAt: new Date().toISOString(),
58
- ids: [...cat.keys()].sort(),
59
- entries: [...cat.values()].sort((a, b) => a.id.localeCompare(b.id)),
60
- });
61
- }
package/dist/cli.js DELETED
@@ -1,188 +0,0 @@
1
- #!/usr/bin/env node
2
- /** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor / verify. */
3
- import { copyFileSync, existsSync, mkdirSync } from "node:fs";
4
- import { dirname, join } from "node:path";
5
- import { fileURLToPath } from "node:url";
6
- import { loadPolicy } from "./policy.js";
7
- import { loadKey } from "./llm.js";
8
- import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
9
- import { paths, readJson, readJsonl } from "./store.js";
10
- import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
11
- import { settledActiveTask } from "./pi-trials.js";
12
- import { modelRoute, routeLabel } from "./routes.js";
13
- import { runStandaloneTrial } from "./standalone.js";
14
- import { collectDoctorReport, renderDiagnosticReport, runOfflineVerify } from "./diagnostics.js";
15
- const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
16
- function cmdInit() {
17
- const policyPath = paths.policy();
18
- mkdirSync(dirname(policyPath), { recursive: true });
19
- if (!existsSync(policyPath)) {
20
- copyFileSync(join(ROOT, "instructions", "openmerit.policy.json"), policyPath);
21
- console.log(`policy -> ${policyPath}`);
22
- }
23
- else {
24
- console.log(`policy -> ${policyPath} (kept existing)`);
25
- }
26
- console.log(`instr. -> ${join(ROOT, "instructions", "OPENMERIT.md")} (add to your harness's AGENTS.md/context)`);
27
- console.log(`ext. -> ${join(ROOT, "extension", "openmerit.ts")}`);
28
- console.log(`pi npm -> pi install npm:openmerit`);
29
- console.log(`pi local-> pi install "${ROOT}"`);
30
- console.log("\nNext: authenticate at least two models in pi, install one extension source, then start pi. OpenRouter is optional enrichment; candidate tools are disabled by default.");
31
- }
32
- function optionalOpenRouterKey() {
33
- try {
34
- return loadKey();
35
- }
36
- catch {
37
- return null;
38
- }
39
- }
40
- function printReport(report, json) {
41
- console.log(json ? JSON.stringify(report, null, 2) : renderDiagnosticReport(report));
42
- if (!report.ok)
43
- process.exitCode = 1;
44
- }
45
- function cmdDoctor(json) {
46
- printReport(collectDoctorReport(), json);
47
- }
48
- function cmdVerify(json) {
49
- printReport(runOfflineVerify(), json);
50
- }
51
- function loadFrontiers() {
52
- const trials = readJsonl(paths.trials());
53
- const observations = readJsonl(paths.observations());
54
- const labelByTask = new Map();
55
- for (const o of observations)
56
- if (!labelByTask.has(o.taskKey))
57
- labelByTask.set(o.taskKey, o.taskLabel);
58
- const byTask = new Map();
59
- for (const t of trials) {
60
- if (!t.taskKey)
61
- continue;
62
- byTask.set(t.taskKey, [...(byTask.get(t.taskKey) ?? []), t]);
63
- }
64
- const policy = loadPolicy(paths.policy());
65
- return [...byTask.entries()].map(([tKey, points]) => {
66
- const scored = points.filter((p) => p.score > 0);
67
- const best = pickBest(scored);
68
- return {
69
- taskKey: tKey,
70
- taskLabel: labelByTask.get(tKey) ?? tKey,
71
- points,
72
- frontier: paretoFrontier(scored).map((p) => p.model),
73
- bestFit: best?.model,
74
- fallback: best ? pickFallback(scored, best, policy.fallback.min_score)?.model : undefined,
75
- updatedAt: new Date().toISOString(),
76
- };
77
- });
78
- }
79
- function cmdFrontier(filter) {
80
- const frontiers = loadFrontiers().filter((f) => !filter || f.taskKey.startsWith(filter));
81
- if (frontiers.length === 0) {
82
- console.log("no frontiers yet (run `openmerit trial` or let `openmerit watch` collect trials)");
83
- return;
84
- }
85
- for (const f of frontiers) {
86
- console.log(`\ntask ${f.taskKey} "${f.taskLabel}"`);
87
- const pts = [...f.points].sort((a, b) => a.price - b.price);
88
- for (const p of pts) {
89
- const mark = f.frontier.includes(p.model) ? "*" : " ";
90
- console.log(` ${mark} ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M` +
91
- (p.latencyMs ? ` ${Math.round(p.latencyMs)}ms` : ""));
92
- }
93
- console.log(` best=${f.bestFit ?? "n/a"} fallback=${f.fallback ?? "n/a"} (* = pareto frontier)`);
94
- }
95
- }
96
- function cmdStatus() {
97
- const harness = readJson(paths.harnessState(), {});
98
- const queue = readJson(paths.candidates(), { newModels: [] });
99
- const latestById = new Map();
100
- for (const r of readJsonl(paths.recommendations()))
101
- latestById.set(r.id, r);
102
- const recs = [...latestById.values()].filter((r) => r.status === "pending");
103
- const frontiers = loadFrontiers();
104
- console.log(`harness model : ${harness.currentRoute ? routeLabel(harness.currentRoute) : harness.currentModel ?? "unknown"} ` +
105
- `(as of ${harness.updatedAt ?? "n/a"})`);
106
- console.log(`tasks tracked : ${frontiers.length}`);
107
- console.log(`new models queued for trial: ${queue.newModels.length}${queue.newModels.length ? " — " + queue.newModels.slice(0, 5).map((m) => m.id).join(", ") + (queue.newModels.length > 5 ? "…" : "") : ""}`);
108
- console.log(`pending recommendations: ${recs.length}`);
109
- for (const r of recs.slice(-5)) {
110
- console.log(` - ${r.recommended.model} <- ${r.currentModel ?? "?"} (gain ${r.evidence.scoreGain}, auto=${r.policy.autoApply})`);
111
- }
112
- }
113
- async function main() {
114
- const [cmd, ...args] = process.argv.slice(2);
115
- switch (cmd) {
116
- case "init":
117
- cmdInit();
118
- break;
119
- case "session-trial": {
120
- const policy = loadPolicy(paths.policy());
121
- const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
122
- if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
123
- throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
124
- const sessionBytes = Number(bytesText);
125
- if (!Number.isSafeInteger(sessionBytes) || sessionBytes <= 0)
126
- throw new Error("invalid session byte limit");
127
- const state = readJson(paths.harnessState(), {});
128
- const currentRoute = provider && providerModelId
129
- ? state.routes?.find((route) => route.provider === provider && route.modelId === providerModelId) ??
130
- modelRoute(provider, providerModelId)
131
- : state.currentRoute;
132
- const task = settledActiveTask({ sessionFile, settledAt, currentModel, currentRoute,
133
- routes: state.routes, settledTaskKey, cwd, sessionBytes });
134
- if (!task)
135
- throw new Error("the specified pi session has no completed, supported task matching this marker");
136
- await autoTaskTick(optionalOpenRouterKey(), policy, task);
137
- break;
138
- }
139
- case "watch":
140
- if (args.includes("--once"))
141
- await tickOnce(loadPolicy(paths.policy()));
142
- else
143
- await runDaemon(loadPolicy(paths.policy()));
144
- break;
145
- case "trial": {
146
- const file = args[0];
147
- if (!file)
148
- throw new Error("usage: openmerit trial <task.json> [--rounds N]");
149
- const rIdx = args.indexOf("--rounds");
150
- await runStandaloneTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
151
- break;
152
- }
153
- case "frontier":
154
- cmdFrontier(args[0]);
155
- break;
156
- case "recommend":
157
- recommendTick(loadPolicy(paths.policy()));
158
- break;
159
- case "status":
160
- cmdStatus();
161
- break;
162
- case "doctor":
163
- cmdDoctor(args.includes("--json"));
164
- break;
165
- case "verify":
166
- cmdVerify(args.includes("--json"));
167
- break;
168
- default:
169
- console.log(`openmerit — external model-merit harness
170
-
171
- usage: openmerit <command>
172
-
173
- init set up ~/.openmerit (policy, state dirs)
174
- watch [--once] observe completed pi tasks, trial candidates sequentially, and recommend session swaps
175
- trial <task.json> iterative model search on one task spec [--rounds N]
176
- frontier [taskKey] print pareto frontier(s)
177
- recommend emit recommendations now
178
- status harness model, queued models, pending recommendations
179
- doctor [--json] diagnose pi, policy, routes, state, and installation
180
- verify [--json] run an offline, provider-free core self-test`);
181
- if (cmd && cmd !== "help" && cmd !== "--help")
182
- process.exitCode = 1;
183
- }
184
- }
185
- main().catch((e) => {
186
- console.error(`openmerit: ${e.message}`);
187
- process.exit(1);
188
- });