openmerit 0.1.3 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/README.md +83 -312
- package/dist/core/src/index.d.ts +90 -0
- package/dist/core/src/index.js +1137 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +15 -0
- package/dist/pi/src/index.js +423 -0
- package/dist/protocol/src/index.d.ts +402 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +450 -0
- package/dist/protocol/src/schemas.js +224 -0
- package/docs/adapter-guide.md +172 -0
- package/docs/architecture.md +55 -0
- package/docs/automation.md +66 -0
- package/docs/getting-started.md +55 -0
- package/docs/lifecycle.md +30 -0
- package/docs/metrics-and-evidence.md +40 -0
- package/docs/operations.md +31 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +44 -0
- package/docs/roadmap.md +26 -0
- package/docs/security.md +23 -0
- package/docs/testing.md +36 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +45 -54
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -388
- package/dist/diagnostics.js +0 -194
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -203
- package/dist/pi-trials.js +0 -366
- package/dist/policy.js +0 -115
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -59
- package/dist/standalone.js +0 -220
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -820
- package/instructions/OPENMERIT.md +0 -54
- package/instructions/openmerit.policy.json +0 -33
- package/rules.md +0 -39
package/dist/trials.js
DELETED
|
@@ -1,140 +0,0 @@
|
|
|
1
|
-
/** Shadow trials: run candidate models against observed/declared tasks on the background track. */
|
|
2
|
-
import { directChatClient } from "./llm.js";
|
|
3
|
-
import { judge, parseObj } from "./judge.js";
|
|
4
|
-
import { paths, readJsonReport, writeJson } from "./store.js";
|
|
5
|
-
function today() {
|
|
6
|
-
return new Date().toISOString().slice(0, 10);
|
|
7
|
-
}
|
|
8
|
-
export function budgetOk(policy) {
|
|
9
|
-
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
10
|
-
if (!report.valid)
|
|
11
|
-
return { ok: false, reason: "trial ledger is malformed; run `openmerit doctor`" };
|
|
12
|
-
const ledger = report.value;
|
|
13
|
-
if (ledger.date !== today())
|
|
14
|
-
return { ok: true }; // new day resets
|
|
15
|
-
if (ledger.trials >= policy.budgets.max_trials_per_day)
|
|
16
|
-
return { ok: false, reason: `daily trial cap reached (${policy.budgets.max_trials_per_day})` };
|
|
17
|
-
if (ledger.usd >= policy.budgets.max_usd_per_day)
|
|
18
|
-
return { ok: false, reason: `daily budget exhausted ($${policy.budgets.max_usd_per_day})` };
|
|
19
|
-
return { ok: true };
|
|
20
|
-
}
|
|
21
|
-
/** Conservative admission estimate for one candidate answer (rough input tokens + 4K output). */
|
|
22
|
-
export function trialBudgetOk(policy, entry, inputChars) {
|
|
23
|
-
if (entry.priceKnown === false)
|
|
24
|
-
return { ok: false, reason: "route price is unknown" };
|
|
25
|
-
const inputTokens = Math.ceil(inputChars / 4);
|
|
26
|
-
const outputTokens = Math.min(entry.route?.maxTokens ?? 4096, 4096);
|
|
27
|
-
const estimatedUsd = entry.pp * inputTokens + entry.pc * outputTokens;
|
|
28
|
-
if (estimatedUsd > policy.budgets.max_usd_per_trial)
|
|
29
|
-
return {
|
|
30
|
-
ok: false,
|
|
31
|
-
estimatedUsd,
|
|
32
|
-
reason: `estimated trial cost $${estimatedUsd.toFixed(4)} exceeds $${policy.budgets.max_usd_per_trial.toFixed(4)} limit`,
|
|
33
|
-
};
|
|
34
|
-
return { ok: true, estimatedUsd };
|
|
35
|
-
}
|
|
36
|
-
export function recordTrialSpend(usd) {
|
|
37
|
-
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
38
|
-
if (!report.valid)
|
|
39
|
-
throw new Error("trial ledger is malformed; refusing to reset spend");
|
|
40
|
-
let ledger = report.value;
|
|
41
|
-
if (ledger.date !== today())
|
|
42
|
-
ledger = { date: today(), trials: 0, usd: 0 };
|
|
43
|
-
ledger.trials += 1;
|
|
44
|
-
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
45
|
-
writeJson(paths.ledger(), ledger);
|
|
46
|
-
}
|
|
47
|
-
/** Record non-candidate merit-loop spend without consuming a trial slot. */
|
|
48
|
-
export function recordMeritSpend(usd) {
|
|
49
|
-
if (!(usd > 0))
|
|
50
|
-
return;
|
|
51
|
-
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
52
|
-
if (!report.valid)
|
|
53
|
-
throw new Error("trial ledger is malformed; refusing to reset spend");
|
|
54
|
-
let ledger = report.value;
|
|
55
|
-
if (ledger.date !== today())
|
|
56
|
-
ledger = { date: today(), trials: 0, usd: 0 };
|
|
57
|
-
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
58
|
-
writeJson(paths.ledger(), ledger);
|
|
59
|
-
}
|
|
60
|
-
const RUBRIC_PROMPT = `Write a compact grading rubric for answers to the task below.
|
|
61
|
-
It must list the concrete criteria for a score of 1.0 and how to deduct.
|
|
62
|
-
TASK: {task}
|
|
63
|
-
Return ONLY json: {{"rubric": "<text>"}}`;
|
|
64
|
-
/** Derive (and cache) a grading rubric for a task that came from traces. */
|
|
65
|
-
export async function ensureRubric(key, judgeModel, task, cache, client = directChatClient(key)) {
|
|
66
|
-
const hit = cache.get(task);
|
|
67
|
-
if (hit)
|
|
68
|
-
return hit;
|
|
69
|
-
const { content } = await client.chat(judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
|
|
70
|
-
let rubric = "Score 1.0 for a fully correct, complete, usable answer; deduct for errors, omissions, or unusable output.";
|
|
71
|
-
try {
|
|
72
|
-
const obj = parseObj(content);
|
|
73
|
-
if (typeof obj.rubric === "string" && obj.rubric.trim())
|
|
74
|
-
rubric = obj.rubric;
|
|
75
|
-
}
|
|
76
|
-
catch {
|
|
77
|
-
/* keep default */
|
|
78
|
-
}
|
|
79
|
-
cache.set(task, rubric);
|
|
80
|
-
return rubric;
|
|
81
|
-
}
|
|
82
|
-
/** Run one shadow trial: generate with `model`, judge the output, return a frontier point + cost. */
|
|
83
|
-
export async function runTrial(key, judgeModel, task, rubric, model, entry, client = directChatClient(key)) {
|
|
84
|
-
const started = Date.now();
|
|
85
|
-
let gen;
|
|
86
|
-
let usage = {};
|
|
87
|
-
try {
|
|
88
|
-
const r = await client.chat(entry?.route ?? model, task, 4096, 0.2);
|
|
89
|
-
gen = r.content;
|
|
90
|
-
usage = r.usage;
|
|
91
|
-
}
|
|
92
|
-
catch (e) {
|
|
93
|
-
return {
|
|
94
|
-
point: {
|
|
95
|
-
schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
96
|
-
priceKnown: entry?.priceKnown ?? !!entry,
|
|
97
|
-
ts: new Date().toISOString(), source: "shadow_trial", why: String(e).slice(0, 160),
|
|
98
|
-
},
|
|
99
|
-
costUsd: 0,
|
|
100
|
-
error: String(e),
|
|
101
|
-
};
|
|
102
|
-
}
|
|
103
|
-
const latencyMs = Date.now() - started;
|
|
104
|
-
const costUsd = (entry?.pp ?? 0) * (usage.prompt_tokens ?? 0) + (entry?.pc ?? 0) * (usage.completion_tokens ?? 0);
|
|
105
|
-
if (!gen.trim()) {
|
|
106
|
-
return {
|
|
107
|
-
point: {
|
|
108
|
-
schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
109
|
-
priceKnown: entry?.priceKnown ?? !!entry, latencyMs,
|
|
110
|
-
ts: new Date().toISOString(), source: "shadow_trial", why: "empty response",
|
|
111
|
-
},
|
|
112
|
-
costUsd,
|
|
113
|
-
};
|
|
114
|
-
}
|
|
115
|
-
let score = 0;
|
|
116
|
-
let why = "judge error";
|
|
117
|
-
try {
|
|
118
|
-
const j = await judge(key, judgeModel, task, rubric, gen, client);
|
|
119
|
-
score = j.score;
|
|
120
|
-
why = j.why;
|
|
121
|
-
}
|
|
122
|
-
catch (e) {
|
|
123
|
-
why = `judge error: ${String(e).slice(0, 120)}`;
|
|
124
|
-
}
|
|
125
|
-
return {
|
|
126
|
-
point: {
|
|
127
|
-
schemaVersion: 1,
|
|
128
|
-
model,
|
|
129
|
-
route: entry?.route,
|
|
130
|
-
score,
|
|
131
|
-
price: entry?.price ?? 0,
|
|
132
|
-
priceKnown: entry?.priceKnown ?? !!entry,
|
|
133
|
-
latencyMs,
|
|
134
|
-
ts: new Date().toISOString(),
|
|
135
|
-
source: "shadow_trial",
|
|
136
|
-
why,
|
|
137
|
-
},
|
|
138
|
-
costUsd,
|
|
139
|
-
};
|
|
140
|
-
}
|
package/dist/types.js
DELETED
|
@@ -1,19 +0,0 @@
|
|
|
1
|
-
Extract the invoice into the required JSON schema using only the image.
|
|
2
|
-
|
|
3
|
-
Return a JSON object with exactly these fields: supplier_name, customer_name,
|
|
4
|
-
invoice_number, po_number, document_date, due_date, document_currency,
|
|
5
|
-
document_total_net, document_total_tax, document_total_amount, and line_items.
|
|
6
|
-
Scalar names, identifiers, dates, and currency are strings or null. Monetary
|
|
7
|
-
totals are JSON numbers or null. line_items is an array of objects, each with
|
|
8
|
-
exactly description (string or null), quantity (number or null), unit_price
|
|
9
|
-
(number or null), and amount (number or null). Do not add any other fields.
|
|
10
|
-
|
|
11
|
-
Rules:
|
|
12
|
-
- Copy names, identifiers, and descriptions exactly as printed.
|
|
13
|
-
- Return dates as YYYY-MM-DD, regardless of the printed date format.
|
|
14
|
-
- Return currency as its three-letter ISO code.
|
|
15
|
-
- Return monetary values and quantities as JSON numbers without currency symbols or grouping separators.
|
|
16
|
-
- Preserve line-item order and include every printed line item exactly once.
|
|
17
|
-
- Use null only when a scalar field is not present. Never infer a missing value.
|
|
18
|
-
- document_total_net is the subtotal/net amount before tax; document_total_amount is the final amount including tax.
|
|
19
|
-
- Return only the schema-conforming JSON object.
|
|
@@ -1,7 +0,0 @@
|
|
|
1
|
-
{
|
|
2
|
-
"task": "Write a Python function `def word_ladder(begin, end, words)` that returns the SHORTEST transformation sequence from `begin` to `end`, where consecutive words differ by exactly one character and every intermediate word must be in `words`. Return [] if no path exists. If begin == end return [begin]. All words are the same length, lowercase a-z. Output only runnable code, no explanation.",
|
|
3
|
-
"eval": "Score 1.0 requires ALL of: (a) algorithm guaranteed to find a shortest path (BFS or equivalent, NOT DFS/greedy); (b) returned path starts with begin and ends with end with all intermediates in words; (c) returns [] when impossible; (d) begin==end returns [begin]; (e) begin need not be in words; (f) syntactically valid, runnable Python, no external imports, function named word_ladder; (g) output contains only code. Deduct ~0.2 per missing criterion; non-shortest-path algorithms score at most 0.5.",
|
|
4
|
-
"initial_model": "openai/gpt-4o-mini",
|
|
5
|
-
"initial_route": "openai:gpt-4o-mini",
|
|
6
|
-
"max_usd_per_m": 20
|
|
7
|
-
}
|