openmerit 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +270 -0
  3. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +38 -0
  4. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  5. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +32 -0
  6. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  7. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +26 -0
  8. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  9. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +26 -0
  10. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  11. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +38 -0
  12. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  13. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +38 -0
  14. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  15. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +20 -0
  16. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  17. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +38 -0
  18. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  19. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +26 -0
  20. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  21. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +20 -0
  22. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  23. package/benchmark/invoice_ocr/data/manifest.json +97 -0
  24. package/dist/benchmarks.js +98 -0
  25. package/dist/catalog.js +51 -0
  26. package/dist/cli.js +261 -0
  27. package/dist/daemon.js +305 -0
  28. package/dist/frontier.js +49 -0
  29. package/dist/invoice-eval.js +33 -0
  30. package/dist/invoice-score.js +124 -0
  31. package/dist/judge.js +43 -0
  32. package/dist/llm.js +66 -0
  33. package/dist/pi-trials.js +224 -0
  34. package/dist/policy.js +110 -0
  35. package/dist/recommend.js +63 -0
  36. package/dist/store.js +55 -0
  37. package/dist/strategist.js +64 -0
  38. package/dist/task-input.js +30 -0
  39. package/dist/traces.js +123 -0
  40. package/dist/trials.js +101 -0
  41. package/dist/types.js +2 -0
  42. package/examples/invoice-prompt.txt +19 -0
  43. package/examples/task.example.json +6 -0
  44. package/extension/openmerit.ts +705 -0
  45. package/instructions/OPENMERIT.md +51 -0
  46. package/instructions/openmerit.policy.json +33 -0
  47. package/package.json +77 -0
package/dist/daemon.js ADDED
@@ -0,0 +1,305 @@
1
+ /**
2
+ * Trial engine shared by the pi extension's session-bound jobs and the optional
3
+ * CLI watcher. It reads traces, runs pi candidates, and writes recommendations
4
+ * for the extension to apply in the active session.
5
+ */
6
+ import { fetchCatalog, diffNewModels, loadSnapshot, saveSnapshot } from "./catalog.js";
7
+ import { categorizeTask, loadDigest, relevantBenchmarks, benchmarkPrior, openRouterBenchmarkCandidates } from "./benchmarks.js";
8
+ import { loadKey } from "./llm.js";
9
+ import { providerAllowed } from "./policy.js";
10
+ import { buildRecommendation } from "./recommend.js";
11
+ import { appendJsonl, paths, readJson, readJsonl, writeJson, } from "./store.js";
12
+ import { ingestNewTraces } from "./traces.js";
13
+ import { budgetOk, ensureRubric, recordTrialSpend, runTrial } from "./trials.js";
14
+ import { JUDGE_PREFS } from "./judge.js";
15
+ import { availablePiModels, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
16
+ import { taskInputKey } from "./task-input.js";
17
+ import { pickNext, STRAT_PREFS } from "./strategist.js";
18
+ function emitTrialProgress(progress) {
19
+ console.log(`[openmerit/progress] ${JSON.stringify(progress)}`);
20
+ }
21
+ function resolvePref(cat, prefs, label) {
22
+ for (const p of prefs) {
23
+ if (p && cat.has(p))
24
+ return p;
25
+ }
26
+ for (const p of prefs) {
27
+ if (!p)
28
+ continue;
29
+ for (const id of [...cat.keys()].sort()) {
30
+ if (id.includes(p))
31
+ return id;
32
+ }
33
+ }
34
+ throw new Error(`could not resolve ${label} model from preferences`);
35
+ }
36
+ /** Ingest new session-trace bytes into the observation log. */
37
+ export function tracesTick() {
38
+ const fresh = ingestNewTraces();
39
+ for (const o of fresh)
40
+ appendJsonl(paths.observations(), o);
41
+ if (fresh.length > 0)
42
+ console.log(`[openmerit] traces: ${fresh.length} new observation(s)`);
43
+ return fresh;
44
+ }
45
+ /** Diff the provider catalog; queue newly released models for trials. */
46
+ export async function catalogTick(key, policy) {
47
+ const cat = await fetchCatalog(key);
48
+ const snapshot = loadSnapshot();
49
+ const fresh = snapshot.ids.length === 0 ? [] : diffNewModels(cat, snapshot);
50
+ saveSnapshot(cat);
51
+ const allowed = fresh.filter((e) => e.price <= policy.max_usd_per_m && providerAllowed(policy, e.id));
52
+ if (allowed.length > 0) {
53
+ const queue = readJson(paths.candidates(), { newModels: [] });
54
+ const known = new Set(queue.newModels.map((m) => m.id));
55
+ for (const e of allowed) {
56
+ if (!known.has(e.id)) {
57
+ queue.newModels.push({ id: e.id, firstSeen: new Date().toISOString(), price: e.price });
58
+ }
59
+ }
60
+ writeJson(paths.candidates(), queue);
61
+ console.log(`[openmerit] catalog: ${allowed.length} new model(s) queued: ${allowed.map((e) => e.id).join(", ")}`);
62
+ }
63
+ return allowed.map((e) => e.id);
64
+ }
65
+ /**
66
+ * Run due shadow trials: for each recently observed task, trial queued new
67
+ * models (or a strategist pick when the queue is empty and the frontier is
68
+ * thin). Budget-capped per policy.
69
+ */
70
+ export async function trialTick(key, policy) {
71
+ const budget = budgetOk(policy);
72
+ if (!budget.ok) {
73
+ console.log(`[openmerit] trials paused: ${budget.reason}`);
74
+ return;
75
+ }
76
+ const observations = readJsonl(paths.observations());
77
+ const queue = readJson(paths.candidates(), { newModels: [] });
78
+ if (observations.length === 0 || queue.newModels.length === 0)
79
+ return;
80
+ const cat = await fetchCatalog(key);
81
+ const judge = resolvePref(cat, [policy.judge_model, ...JUDGE_PREFS], "judge");
82
+ // Most recent observation per task.
83
+ const latestByTask = new Map();
84
+ for (const o of observations)
85
+ latestByTask.set(o.taskKey, o);
86
+ const trials = readJsonl(paths.trials());
87
+ const rubricCache = new Map();
88
+ const digest = loadDigest();
89
+ for (const [tKey, obs] of latestByTask) {
90
+ const due = queue.newModels.filter((m) => !trials.some((t) => t.taskKey === tKey && t.model === m.id));
91
+ if (due.length === 0)
92
+ continue;
93
+ const { benchmarks } = relevantBenchmarks(obs.taskLabel);
94
+ const category = categorizeTask(obs.taskLabel);
95
+ // Rank due candidates by benchmark prior; cheapest first when unknown.
96
+ due.sort((a, b) => {
97
+ const pa = benchmarkPrior(category, a.id, digest);
98
+ const pb = benchmarkPrior(category, b.id, digest);
99
+ return (pb ?? -1) - (pa ?? -1) || a.price - b.price;
100
+ });
101
+ const candidate = due[0];
102
+ const b = budgetOk(policy);
103
+ if (!b.ok)
104
+ break;
105
+ console.log(`[openmerit] shadow trial: ${candidate.id} on task ${tKey} (${category}; benchmarks: ${benchmarks.join(", ")})`);
106
+ const rubric = await ensureRubric(key, judge, obs.taskLabel, rubricCache);
107
+ const { point, costUsd, error } = await runTrial(key, judge, obs.taskLabel, rubric, candidate.id, cat.get(candidate.id));
108
+ recordTrialSpend(costUsd);
109
+ appendJsonl(paths.trials(), { ...point, taskKey: tKey });
110
+ if (error)
111
+ console.log(`[openmerit] trial failed: ${error.slice(0, 120)}`);
112
+ else
113
+ console.log(`[openmerit] trial: ${candidate.id} score=${point.score.toFixed(2)} $${point.price.toFixed(2)}/M`);
114
+ }
115
+ }
116
+ /** Compare the one completed active pi task, sequentially, then target its session. */
117
+ export async function autoTaskTick(key, policy, settledTask) {
118
+ const task = settledTask === undefined ? settledActiveTask() : settledTask;
119
+ if (!task)
120
+ return null;
121
+ const marker = settledTask === undefined
122
+ ? `${task.sessionFile}:${task.settledAt}`
123
+ : `${task.sessionFile}:${task.sessionBytes ?? task.settledAt}`;
124
+ const processedPath = settledTask === undefined ? paths.watchProcessed() : paths.sessionJob(marker);
125
+ const prior = readJson(processedPath, null);
126
+ if (prior?.marker === marker && prior.status !== "failed")
127
+ return null;
128
+ if (!budgetOk(policy).ok)
129
+ return null;
130
+ // Claim the settled turn before provider calls so the next poll cannot duplicate it.
131
+ writeJson(processedPath, { marker, status: "running", updatedAt: new Date().toISOString() });
132
+ try {
133
+ const cat = await fetchCatalog(key);
134
+ const piModels = availablePiModels();
135
+ for (const id of [...cat.keys()]) {
136
+ const e = cat.get(id);
137
+ if (!piModels.has(id) || e.price > policy.max_usd_per_m || !providerAllowed(policy, id) ||
138
+ (task.images.length > 0 && !e.inputModalities?.includes("image")))
139
+ cat.delete(id);
140
+ }
141
+ if (!cat.has(task.model))
142
+ throw new Error(`active model ${task.model} is not trialable in pi/OpenRouter`);
143
+ const judge = resolvePref(cat, [policy.judge_model, ...JUDGE_PREFS], "judge");
144
+ const strategist = resolvePref(cat, [policy.strategist_model, ...STRAT_PREFS], "strategist");
145
+ const { category, benchmarks } = relevantBenchmarks(task.task);
146
+ let benchmarkCandidates = [];
147
+ try {
148
+ benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, cat);
149
+ }
150
+ catch (e) {
151
+ console.log(`[openmerit] benchmark shortlist unavailable: ${e.message}`);
152
+ }
153
+ const rubric = await ensureRubric(key, judge, task.task, new Map());
154
+ const tKey = taskInputKey(task.task, task.images);
155
+ const points = [];
156
+ const tried = new Set();
157
+ const failedVendors = new Set();
158
+ const total = Math.max(1, Math.floor(policy.watch.models_per_task));
159
+ if (settledTask !== undefined)
160
+ emitTrialProgress({ phase: "start", taskKey: tKey,
161
+ model: task.model, index: 1, total });
162
+ const baseline = await scorePiRun(key, judge, task.task, rubric, task.model, cat.get(task.model), task.run, "trace", task.images);
163
+ points.push(baseline.point);
164
+ tried.add(task.model);
165
+ appendJsonl(paths.trials(), { ...baseline.point, taskKey: tKey, sessionFile: task.sessionFile });
166
+ console.log(`[openmerit] task ${tKey}: A=${task.model} score=${baseline.point.score.toFixed(2)} from pi trace`);
167
+ if (settledTask !== undefined)
168
+ emitTrialProgress({ phase: "complete", taskKey: tKey,
169
+ model: task.model, index: 1, total, score: baseline.point.score,
170
+ price: baseline.point.price, latencyMs: baseline.point.latencyMs,
171
+ costUsd: baseline.costUsd });
172
+ for (let i = 1; i < total; i++) {
173
+ if (!budgetOk(policy).ok)
174
+ break;
175
+ const pick = await pickNext(key, strategist, task.task, rubric, benchmarks, points, cat, tried, policy.max_usd_per_m, failedVendors, benchmarkCandidates);
176
+ if (!pick)
177
+ break;
178
+ if (settledTask !== undefined)
179
+ emitTrialProgress({ phase: "start", taskKey: tKey,
180
+ model: pick.model, index: i + 1, total });
181
+ const { point, costUsd, error } = await runPiTrial(key, judge, task.task, rubric, pick.model, cat.get(pick.model), task.cwd, task.images);
182
+ tried.add(pick.model);
183
+ points.push(point);
184
+ recordTrialSpend(costUsd);
185
+ appendJsonl(paths.trials(), { ...point, taskKey: tKey, sessionFile: task.sessionFile });
186
+ if (error)
187
+ failedVendors.add(pick.model.split("/")[0]);
188
+ console.log(`[openmerit] task ${tKey}: ${pick.model} score=${point.score.toFixed(2)} cost=$${costUsd.toFixed(4)}${error ? ` error=${error.slice(0, 80)}` : ""}`);
189
+ if (settledTask !== undefined)
190
+ emitTrialProgress({ phase: "complete", taskKey: tKey,
191
+ model: pick.model, index: i + 1, total, score: point.score, price: point.price,
192
+ latencyMs: point.latencyMs, costUsd, error: error?.slice(0, 120) });
193
+ }
194
+ const rec = buildRecommendation(tKey, task.task.slice(0, 120), task.model, points, policy, task.sessionFile);
195
+ if (rec) {
196
+ appendJsonl(paths.recommendations(), rec);
197
+ console.log(`[openmerit] task ${tKey}: selected ${rec.recommended.model}; auto=${rec.policy.autoApply}`);
198
+ }
199
+ else
200
+ console.log(`[openmerit] task ${tKey}: active model remains best`);
201
+ writeJson(processedPath, { marker, status: rec ? "complete" : "no_swap",
202
+ updatedAt: new Date().toISOString() });
203
+ return rec;
204
+ }
205
+ catch (error) {
206
+ writeJson(processedPath, { marker, status: "failed",
207
+ updatedAt: new Date().toISOString() });
208
+ throw error;
209
+ }
210
+ }
211
+ /** Rebuild frontiers from trials and emit recommendations for the harness's current model. */
212
+ export function recommendTick(policy) {
213
+ const harness = readJson(paths.harnessState(), {});
214
+ const currentModel = harness.currentModel ?? null;
215
+ const trials = readJsonl(paths.trials());
216
+ const observations = readJsonl(paths.observations());
217
+ const labelByTask = new Map();
218
+ for (const o of observations)
219
+ if (!labelByTask.has(o.taskKey))
220
+ labelByTask.set(o.taskKey, o.taskLabel);
221
+ const byTask = new Map();
222
+ for (const t of trials) {
223
+ if (!t.taskKey)
224
+ continue;
225
+ const arr = byTask.get(t.taskKey) ?? [];
226
+ arr.push(t);
227
+ byTask.set(t.taskKey, arr);
228
+ }
229
+ // Dedupe on latest status per id: don't re-emit what is pending or was dismissed.
230
+ const all = readJsonl(paths.recommendations());
231
+ const latestById = new Map();
232
+ for (const r of all)
233
+ latestById.set(r.id, r);
234
+ const existing = new Set([...latestById.values()]
235
+ .filter((r) => r.status === "pending" || r.status === "dismissed")
236
+ .map((r) => `${r.taskKey}:${r.recommended.model}`));
237
+ const emitted = [];
238
+ for (const [tKey, points] of byTask) {
239
+ const rec = buildRecommendation(tKey, labelByTask.get(tKey) ?? tKey, currentModel, points, policy);
240
+ if (!rec)
241
+ continue;
242
+ const dedupeKey = `${rec.taskKey}:${rec.recommended.model}`;
243
+ if (existing.has(dedupeKey))
244
+ continue;
245
+ appendJsonl(paths.recommendations(), rec);
246
+ emitted.push(rec);
247
+ console.log(`[openmerit] recommendation: ${rec.recommended.model} <- ${currentModel ?? "unknown"} ` +
248
+ `(gain ${rec.evidence.scoreGain}, auto=${rec.policy.autoApply})`);
249
+ }
250
+ return emitted;
251
+ }
252
+ /** One full pass. */
253
+ export async function tickOnce(policy) {
254
+ tracesTick();
255
+ let key = null;
256
+ try {
257
+ key = loadKey();
258
+ }
259
+ catch (e) {
260
+ console.log(`[openmerit] ${e.message}; skipping provider ticks`);
261
+ }
262
+ if (key) {
263
+ await catalogTick(key, policy);
264
+ await autoTaskTick(key, policy);
265
+ }
266
+ }
267
+ /** Run forever, honoring policy watch intervals. */
268
+ export async function runDaemon(policy) {
269
+ console.log("[openmerit] background track started");
270
+ const tracesMs = Math.max(5, policy.watch.traces_interval_sec) * 1000;
271
+ let lastCatalog = 0;
272
+ // eslint-disable-next-line no-constant-condition
273
+ while (true) {
274
+ try {
275
+ tracesTick();
276
+ }
277
+ catch (e) {
278
+ console.log(`[openmerit] traces tick failed: ${e.message}`);
279
+ }
280
+ const now = Date.now();
281
+ const wantCatalog = now - lastCatalog >= policy.watch.catalog_interval_min * 60_000;
282
+ {
283
+ let key = null;
284
+ try {
285
+ key = loadKey();
286
+ }
287
+ catch {
288
+ /* no key; skip provider ticks */
289
+ }
290
+ if (key) {
291
+ try {
292
+ if (wantCatalog) {
293
+ await catalogTick(key, policy);
294
+ lastCatalog = now;
295
+ }
296
+ await autoTaskTick(key, policy);
297
+ }
298
+ catch (e) {
299
+ console.log(`[openmerit] provider tick failed: ${e.message}`);
300
+ }
301
+ }
302
+ }
303
+ await new Promise((r) => setTimeout(r, tracesMs));
304
+ }
305
+ }
@@ -0,0 +1,49 @@
1
+ /** Pareto frontier over (score up, price down, latency down) + best-fit/fallback picks. */
2
+ /** a dominates b if a is no worse on every objective and strictly better on one. */
3
+ export function dominates(a, b) {
4
+ const aLat = a.latencyMs ?? Number.POSITIVE_INFINITY;
5
+ const bLat = b.latencyMs ?? Number.POSITIVE_INFINITY;
6
+ const noWorse = a.score >= b.score && a.price <= b.price && aLat <= bLat;
7
+ const better = a.score > b.score || a.price < b.price || aLat < bLat;
8
+ return noWorse && better;
9
+ }
10
+ /**
11
+ * Non-dominated set, cheapest first. When no point carries latency this
12
+ * reduces to the classic "strictly increasing score as price rises" chain
13
+ * from the original model-search prototype.
14
+ */
15
+ export function paretoFrontier(points) {
16
+ const fr = points.filter((p) => !points.some((q) => q !== p && dominates(q, p)));
17
+ return fr.sort((a, b) => a.price - b.price || b.score - a.score);
18
+ }
19
+ /** Simple 2-objective chain (score vs price only), cheapest first. */
20
+ export function scorePriceChain(points) {
21
+ const pts = [...points].sort((a, b) => a.price - b.price || b.score - a.score);
22
+ const chain = [];
23
+ let best = -1;
24
+ for (const p of pts) {
25
+ if (p.score > best) {
26
+ chain.push(p);
27
+ best = p.score;
28
+ }
29
+ }
30
+ return chain;
31
+ }
32
+ /** Highest score; ties broken by lower price then lower latency. */
33
+ export function pickBest(points) {
34
+ return [...points].sort((a, b) => b.score - a.score || a.price - b.price || (a.latencyMs ?? 1e18) - (b.latencyMs ?? 1e18))[0];
35
+ }
36
+ /**
37
+ * Fallback for `best`: prefer the cheapest frontier point that still clears
38
+ * `minScore`; otherwise the next-best scoring point overall. Never returns
39
+ * `best` itself.
40
+ */
41
+ export function pickFallback(points, best, minScore) {
42
+ const others = points.filter((p) => p.model !== best.model);
43
+ if (others.length === 0)
44
+ return undefined;
45
+ const fr = paretoFrontier(others).filter((p) => p.score >= minScore);
46
+ if (fr.length > 0)
47
+ return fr[0]; // cheapest adequate frontier point
48
+ return pickBest(others);
49
+ }
@@ -0,0 +1,33 @@
1
+ /** Reuse the audited invoice benchmark scorer for its pinned image/prompt cases. */
2
+ import { createHash } from "node:crypto";
3
+ import { readFileSync } from "node:fs";
4
+ import { dirname, join } from "node:path";
5
+ import { fileURLToPath } from "node:url";
6
+ import { INVOICE_BENCHMARK_PROMPT, parseInvoiceJson, projectVisibleInvoice, scoreInvoice, } from "./invoice-score.js";
7
+ function root() { return join(dirname(fileURLToPath(import.meta.url)), ".."); }
8
+ export async function knownInvoiceScore(task, images, answer) {
9
+ if (images.length !== 1 || images[0].mimeType !== "image/jpeg")
10
+ return null;
11
+ const benchDir = join(root(), "benchmark", "invoice_ocr");
12
+ const piPrompt = readFileSync(join(root(), "examples", "invoice-prompt.txt"), "utf8").trim();
13
+ if (task.trim() !== INVOICE_BENCHMARK_PROMPT.trim() && task.trim() !== piPrompt)
14
+ return null;
15
+ const manifest = JSON.parse(readFileSync(join(benchDir, "data", "manifest.json"), "utf8"));
16
+ const inputHash = createHash("sha256").update(Buffer.from(images[0].data, "base64")).digest("hex");
17
+ const row = manifest.rows.find((r) => createHash("sha256")
18
+ .update(readFileSync(join(benchDir, "data", r.image))).digest("hex") === inputHash);
19
+ if (!row)
20
+ return null;
21
+ const expected = JSON.parse(readFileSync(join(benchDir, "data", row.ground_truth), "utf8"));
22
+ let actual = null;
23
+ try {
24
+ actual = parseInvoiceJson(answer);
25
+ }
26
+ catch { /* invalid JSON scores zero */ }
27
+ const result = scoreInvoice(projectVisibleInvoice(expected), projectVisibleInvoice(actual));
28
+ return {
29
+ score: result.exact_leaf_accuracy,
30
+ why: `pinned ${row.invoice_id}: audited exact-leaf ${result.exact_leaf_accuracy.toFixed(3)}, ` +
31
+ `exact document=${result.exact_document}`,
32
+ };
33
+ }
@@ -0,0 +1,124 @@
1
+ /** Runtime-safe copy of the audited invoice benchmark's exact scoring rules. */
2
+ export const INVOICE_BENCHMARK_PROMPT = `Extract the invoice into the required JSON schema using only the image.
3
+
4
+ Rules:
5
+ - Copy names, identifiers, and descriptions exactly as printed.
6
+ - Return dates as YYYY-MM-DD, regardless of the printed date format.
7
+ - Return currency as its three-letter ISO code.
8
+ - Return monetary values and quantities as JSON numbers without currency symbols or grouping separators.
9
+ - Preserve line-item order and include every printed line item exactly once.
10
+ - Use null only when a scalar field is not present. Never infer a missing value.
11
+ - document_total_net is the subtotal/net amount before tax; document_total_amount is the final amount including tax.
12
+ - Return only the schema-conforming JSON object.`;
13
+ export const EXCLUDED_INVOICE_QUALITY_FIELDS = new Set(["po_number"]);
14
+ function isPlainObject(value) {
15
+ return typeof value === "object" && value !== null && !Array.isArray(value);
16
+ }
17
+ /** Match the Python equality semantics used by the original audited scorer. */
18
+ function pyEqual(a, b) {
19
+ if (typeof a === "boolean" || typeof b === "boolean") {
20
+ if (typeof a === "boolean" && typeof b === "boolean")
21
+ return a === b;
22
+ if (typeof a === "number" && typeof b === "boolean")
23
+ return a === Number(b);
24
+ if (typeof a === "boolean" && typeof b === "number")
25
+ return Number(a) === b;
26
+ return false;
27
+ }
28
+ if (typeof a === "number" && typeof b === "number")
29
+ return a === b;
30
+ if (typeof a === "string" && typeof b === "string")
31
+ return a === b;
32
+ if (a === null || b === null)
33
+ return a === b;
34
+ if (Array.isArray(a) && Array.isArray(b)) {
35
+ return a.length === b.length && a.every((value, index) => pyEqual(value, b[index]));
36
+ }
37
+ if (isPlainObject(a) && isPlainObject(b)) {
38
+ const aKeys = Object.keys(a);
39
+ const bKeys = Object.keys(b);
40
+ return aKeys.length === bKeys.length && aKeys.every((key) => key in b && pyEqual(a[key], b[key]));
41
+ }
42
+ return false;
43
+ }
44
+ function flatten(value, path = "") {
45
+ if (isPlainObject(value)) {
46
+ let flattened = {};
47
+ for (const [key, child] of Object.entries(value)) {
48
+ const childPath = path ? `${path}.${key}` : key;
49
+ flattened = Object.assign(flattened, flatten(child, childPath));
50
+ }
51
+ return flattened;
52
+ }
53
+ if (Array.isArray(value)) {
54
+ if (value.length === 0)
55
+ return { [`${path}.__length__`]: 0 };
56
+ let flattened = { [`${path}.__length__`]: value.length };
57
+ for (let index = 0; index < value.length; index++) {
58
+ flattened = Object.assign(flattened, flatten(value[index], `${path}[${index}]`));
59
+ }
60
+ return flattened;
61
+ }
62
+ return { [path]: value };
63
+ }
64
+ function exactValue(expected, actual) {
65
+ if (typeof expected === "boolean" || typeof actual === "boolean") {
66
+ return typeof expected === typeof actual && expected === actual;
67
+ }
68
+ if (typeof expected === "number" && typeof actual === "number")
69
+ return expected === actual;
70
+ return typeof expected === typeof actual && expected === actual;
71
+ }
72
+ export function scoreInvoice(expected, actual) {
73
+ if (actual === null) {
74
+ const expectedFlat = flatten(expected);
75
+ return {
76
+ exact_document: false,
77
+ matched_leaves: 0,
78
+ total_union_leaves: Object.keys(expectedFlat).length,
79
+ exact_leaf_accuracy: 0,
80
+ scalar_accuracy: 0,
81
+ line_item_accuracy: 0,
82
+ };
83
+ }
84
+ const expectedFlat = flatten(expected);
85
+ const actualFlat = flatten(actual);
86
+ const paths = new Set([...Object.keys(expectedFlat), ...Object.keys(actualFlat)]);
87
+ let matches = 0;
88
+ for (const path of paths) {
89
+ if (path in expectedFlat && path in actualFlat && exactValue(expectedFlat[path], actualFlat[path])) {
90
+ matches += 1;
91
+ }
92
+ }
93
+ const scalarPaths = Object.keys(expectedFlat).filter((path) => !path.startsWith("line_items"));
94
+ const itemPaths = [...paths].filter((path) => path.startsWith("line_items"));
95
+ const scalarMatches = scalarPaths.filter((path) => path in actualFlat && exactValue(expectedFlat[path], actualFlat[path])).length;
96
+ const itemMatches = itemPaths.filter((path) => path in expectedFlat && path in actualFlat && exactValue(expectedFlat[path], actualFlat[path])).length;
97
+ return {
98
+ exact_document: pyEqual(expected, actual),
99
+ matched_leaves: matches,
100
+ total_union_leaves: paths.size,
101
+ exact_leaf_accuracy: paths.size ? matches / paths.size : 1,
102
+ scalar_accuracy: scalarPaths.length ? scalarMatches / scalarPaths.length : 1,
103
+ line_item_accuracy: itemPaths.length ? itemMatches / itemPaths.length : 1,
104
+ };
105
+ }
106
+ export function parseInvoiceJson(content) {
107
+ let stripped = content.trim();
108
+ if (stripped.startsWith("```")) {
109
+ const lines = stripped.split(/\r\n|\r|\n/);
110
+ stripped = lines.slice(1, -1).join("\n");
111
+ }
112
+ const parsed = JSON.parse(stripped);
113
+ if (!isPlainObject(parsed))
114
+ throw new TypeError("Model output is not a JSON object");
115
+ return parsed;
116
+ }
117
+ export function projectVisibleInvoice(document) {
118
+ if (document === null || document === undefined)
119
+ return document;
120
+ const projected = structuredClone(document);
121
+ for (const field of EXCLUDED_INVOICE_QUALITY_FIELDS)
122
+ delete projected[field];
123
+ return projected;
124
+ }
package/dist/judge.js ADDED
@@ -0,0 +1,43 @@
1
+ /** Judge: score a model's answer against a rubric. */
2
+ import { chat, chatWithImages } from "./llm.js";
3
+ export const JUDGE_PREFS = ["openai/gpt-4o-mini", "openai/gpt-4o"];
4
+ const JUDGE_PROMPT = `You are grading a model's answer.
5
+ TASK: {task}
6
+ RUBRIC: {rubric}
7
+ ANSWER: {answer}
8
+ Return ONLY json: {{"score": <0..1>, "why": "<one line>"}}`;
9
+ /** Extract the first {...} JSON object from model output. */
10
+ export function parseObj(txt) {
11
+ const m = txt.match(/\{.*\}/s) ?? txt.match(/\{[^}]*\}/s);
12
+ if (!m)
13
+ throw new Error("no json object found");
14
+ return JSON.parse(m[0]);
15
+ }
16
+ export async function judge(key, judgeModel, task, rubric, answer) {
17
+ const { content } = await chat(key, judgeModel, JUDGE_PROMPT.replace("{task}", task)
18
+ .replace("{rubric}", rubric)
19
+ .replace("{answer}", answer.slice(0, 16000)), 1024, 0);
20
+ try {
21
+ const j = parseObj(content);
22
+ return { score: Number(j.score ?? 0), why: String(j.why ?? "") };
23
+ }
24
+ catch {
25
+ return { score: 0, why: "unparseable judge output" };
26
+ }
27
+ }
28
+ /** Grade OCR answers against the actual uploaded pixels, not answer text alone. */
29
+ export async function judgeWithImages(key, judgeModel, task, rubric, answer, images) {
30
+ const prompt = `You are grading an answer to a visual task. Inspect the attached image carefully.\n` +
31
+ `TASK: ${task}\nRUBRIC: ${rubric}\nANSWER: ${answer.slice(0, 16000)}\n` +
32
+ `For invoice extraction, check every visible field and line item against the image. ` +
33
+ `Penalize missing, invented, or mistyped values and invalid JSON. ` +
34
+ `Return ONLY json: {"score": <0..1>, "why": "<one line>"}`;
35
+ const { content } = await chatWithImages(key, judgeModel, prompt, images, 1024);
36
+ try {
37
+ const j = parseObj(content);
38
+ return { score: Number(j.score ?? 0), why: String(j.why ?? "") };
39
+ }
40
+ catch {
41
+ return { score: 0, why: "unparseable visual judge output" };
42
+ }
43
+ }
package/dist/llm.js ADDED
@@ -0,0 +1,66 @@
1
+ /** Minimal OpenRouter chat client for catalog and trial requests. */
2
+ import { existsSync, readFileSync } from "node:fs";
3
+ import { paths } from "./store.js";
4
+ const OR = "https://openrouter.ai/api/v1";
5
+ /** Load OPENROUTER_API_KEY from env or ~/.openmerit/.env. */
6
+ export function loadKey() {
7
+ const envFile = paths.envFile();
8
+ if (existsSync(envFile)) {
9
+ for (const line of readFileSync(envFile, "utf8").split("\n")) {
10
+ const t = line.trim();
11
+ if (!t || t.startsWith("#") || !t.includes("="))
12
+ continue;
13
+ const [k, ...v] = t.split("=");
14
+ if (process.env[k.trim()] === undefined)
15
+ process.env[k.trim()] = v.join("=").trim();
16
+ }
17
+ }
18
+ const key = process.env.OPENROUTER_API_KEY;
19
+ if (!key)
20
+ throw new Error("missing OPENROUTER_API_KEY (env or ~/.openmerit/.env)");
21
+ return key;
22
+ }
23
+ async function request(key, path, body, tries = 3) {
24
+ for (let attempt = 0; attempt < tries; attempt++) {
25
+ const res = await fetch(OR + path, {
26
+ method: body === undefined ? "GET" : "POST",
27
+ headers: {
28
+ Authorization: `Bearer ${key}`,
29
+ ...(body === undefined ? {} : { "Content-Type": "application/json", "X-Title": "openmerit" }),
30
+ },
31
+ body: body === undefined ? undefined : JSON.stringify(body),
32
+ });
33
+ if (res.ok)
34
+ return (await res.json());
35
+ const detail = (await res.text()).slice(0, 400);
36
+ if ([429, 500, 502, 503].includes(res.status) && attempt < tries - 1) {
37
+ await new Promise((r) => setTimeout(r, 2000 * (attempt + 1)));
38
+ continue;
39
+ }
40
+ throw new Error(`OpenRouter HTTP ${res.status}: ${detail}`);
41
+ }
42
+ throw new Error("unreachable");
43
+ }
44
+ export async function chat(key, model, prompt, maxTokens, temperature) {
45
+ const r = await request(key, "/chat/completions", {
46
+ model,
47
+ messages: [{ role: "user", content: prompt }],
48
+ max_tokens: maxTokens,
49
+ temperature,
50
+ });
51
+ return { content: r.choices[0]?.message?.content ?? "", usage: r.usage ?? {} };
52
+ }
53
+ /** Send a visual grading request through the same OpenRouter account. */
54
+ export async function chatWithImages(key, model, prompt, images, maxTokens) {
55
+ const r = await request(key, "/chat/completions", {
56
+ model,
57
+ messages: [{ role: "user", content: [
58
+ { type: "text", text: prompt },
59
+ ...images.map((image) => ({ type: "image_url",
60
+ image_url: { url: `data:${image.mimeType};base64,${image.data}` } })),
61
+ ] }],
62
+ max_tokens: maxTokens,
63
+ temperature: 0,
64
+ });
65
+ return { content: r.choices[0]?.message?.content ?? "", usage: r.usage ?? {} };
66
+ }