openmerit 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +270 -0
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +32 -0
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +26 -0
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +26 -0
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +20 -0
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +26 -0
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +20 -0
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +97 -0
- package/dist/benchmarks.js +98 -0
- package/dist/catalog.js +51 -0
- package/dist/cli.js +261 -0
- package/dist/daemon.js +305 -0
- package/dist/frontier.js +49 -0
- package/dist/invoice-eval.js +33 -0
- package/dist/invoice-score.js +124 -0
- package/dist/judge.js +43 -0
- package/dist/llm.js +66 -0
- package/dist/pi-trials.js +224 -0
- package/dist/policy.js +110 -0
- package/dist/recommend.js +63 -0
- package/dist/store.js +55 -0
- package/dist/strategist.js +64 -0
- package/dist/task-input.js +30 -0
- package/dist/traces.js +123 -0
- package/dist/trials.js +101 -0
- package/dist/types.js +2 -0
- package/examples/invoice-prompt.txt +19 -0
- package/examples/task.example.json +6 -0
- package/extension/openmerit.ts +705 -0
- package/instructions/OPENMERIT.md +51 -0
- package/instructions/openmerit.policy.json +33 -0
- package/package.json +77 -0
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
/** Run a task through pi for each model, using pi's JSON event stream as the measurement source. */
|
|
2
|
+
import { spawn, spawnSync } from "node:child_process";
|
|
3
|
+
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import { join } from "node:path";
|
|
6
|
+
import { judge, judgeWithImages } from "./judge.js";
|
|
7
|
+
import { knownInvoiceScore } from "./invoice-eval.js";
|
|
8
|
+
import { paths, readJson } from "./store.js";
|
|
9
|
+
import { sessionImages, taskInputKey } from "./task-input.js";
|
|
10
|
+
function answerText(content) {
|
|
11
|
+
if (typeof content === "string")
|
|
12
|
+
return content;
|
|
13
|
+
if (!Array.isArray(content))
|
|
14
|
+
return "";
|
|
15
|
+
return content.filter((x) => !!x && typeof x === "object" && x.type === "text")
|
|
16
|
+
.map((x) => x.text ?? "").join("\n");
|
|
17
|
+
}
|
|
18
|
+
/** Use model A's completed result from the active pi session when it matches this task. */
|
|
19
|
+
export function recordedActiveTask(task, model, images = [], sessionFile, sessionBytes) {
|
|
20
|
+
const state = sessionFile ? null : readJson(paths.harnessState(), {});
|
|
21
|
+
const file = sessionFile ?? state?.sessionFile;
|
|
22
|
+
if (!file || !existsSync(file))
|
|
23
|
+
return null;
|
|
24
|
+
let user = null;
|
|
25
|
+
const assistants = [];
|
|
26
|
+
for (const line of readFileSync(file).subarray(0, sessionBytes).toString("utf8").split("\n")) {
|
|
27
|
+
if (!line.trim())
|
|
28
|
+
continue;
|
|
29
|
+
let entry;
|
|
30
|
+
try {
|
|
31
|
+
entry = JSON.parse(line);
|
|
32
|
+
}
|
|
33
|
+
catch {
|
|
34
|
+
continue;
|
|
35
|
+
}
|
|
36
|
+
if (entry.type !== "message" || !entry.message)
|
|
37
|
+
continue;
|
|
38
|
+
if (entry.message.role === "user") {
|
|
39
|
+
user = entry.message;
|
|
40
|
+
assistants.length = 0;
|
|
41
|
+
}
|
|
42
|
+
else if (user && entry.message.role === "assistant")
|
|
43
|
+
assistants.push(entry.message);
|
|
44
|
+
}
|
|
45
|
+
const userImages = user ? sessionImages(user.content) : null;
|
|
46
|
+
if (!user || !userImages || taskInputKey(answerText(user.content), userImages) !==
|
|
47
|
+
taskInputKey(task, images))
|
|
48
|
+
return null;
|
|
49
|
+
const matching = assistants.filter((m) => (m.provider === "openrouter" ? m.model : `${m.provider}/${m.model}`) === model);
|
|
50
|
+
if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
|
|
51
|
+
return null;
|
|
52
|
+
const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
|
|
53
|
+
if (!answer.trim())
|
|
54
|
+
return null;
|
|
55
|
+
const first = user.timestamp ?? 0;
|
|
56
|
+
const last = matching.at(-1)?.timestamp ?? first;
|
|
57
|
+
return {
|
|
58
|
+
answer,
|
|
59
|
+
costUsd: matching.reduce((n, m) => n + (m.usage?.cost?.total ?? 0), 0),
|
|
60
|
+
latencyMs: Math.max(0, last - first),
|
|
61
|
+
errors: 0,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
/** Read the exact latest completed text task from the pi extension's session marker. */
|
|
65
|
+
export function settledActiveTask(snapshot) {
|
|
66
|
+
const st = snapshot ?? readJson(paths.harnessState(), {});
|
|
67
|
+
if (!st.currentModel || !st.sessionFile || !st.settledTaskKey || !st.settledAt ||
|
|
68
|
+
!existsSync(st.sessionFile))
|
|
69
|
+
return null;
|
|
70
|
+
let userContent = null;
|
|
71
|
+
let usedTools = false;
|
|
72
|
+
for (const line of readFileSync(st.sessionFile).subarray(0, st.sessionBytes).toString("utf8").split("\n")) {
|
|
73
|
+
if (!line.trim())
|
|
74
|
+
continue;
|
|
75
|
+
let entry;
|
|
76
|
+
try {
|
|
77
|
+
entry = JSON.parse(line);
|
|
78
|
+
}
|
|
79
|
+
catch {
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
if (entry.type !== "message" || !entry.message)
|
|
83
|
+
continue;
|
|
84
|
+
if (entry.message.role === "user") {
|
|
85
|
+
userContent = entry.message.content;
|
|
86
|
+
usedTools = false;
|
|
87
|
+
}
|
|
88
|
+
else if (userContent && entry.message.role === "toolResult")
|
|
89
|
+
usedTools = true;
|
|
90
|
+
else if (userContent && entry.message.role === "assistant" &&
|
|
91
|
+
Array.isArray(entry.message.content) && entry.message.content.some((b) => b?.type === "toolCall"))
|
|
92
|
+
usedTools = true;
|
|
93
|
+
}
|
|
94
|
+
if (!userContent || usedTools)
|
|
95
|
+
return null;
|
|
96
|
+
const images = sessionImages(userContent);
|
|
97
|
+
if (!images)
|
|
98
|
+
return null;
|
|
99
|
+
const task = answerText(userContent);
|
|
100
|
+
if (!task.trim() || taskInputKey(task, images) !== st.settledTaskKey)
|
|
101
|
+
return null;
|
|
102
|
+
const run = recordedActiveTask(task, st.currentModel, images, st.sessionFile, st.sessionBytes);
|
|
103
|
+
if (!run)
|
|
104
|
+
return null;
|
|
105
|
+
return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
|
|
106
|
+
settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd() };
|
|
107
|
+
}
|
|
108
|
+
/** Limit candidates to models this installed pi can actually select. */
|
|
109
|
+
export function availablePiModels() {
|
|
110
|
+
const result = spawnSync("pi", ["--offline", "--list-models", "openrouter"], { encoding: "utf8" });
|
|
111
|
+
if (result.error || result.status !== 0)
|
|
112
|
+
throw new Error("could not list pi OpenRouter models");
|
|
113
|
+
return new Set(result.stdout.split("\n").slice(1)
|
|
114
|
+
.map((line) => line.trim().split(/\s+/)[1])
|
|
115
|
+
.filter((id) => !!id && !id.startsWith("~")));
|
|
116
|
+
}
|
|
117
|
+
/** Each invocation is a fresh pi run, so model B does not inherit model A's answer. */
|
|
118
|
+
export async function executePiTask(model, task, cwd = process.cwd(), images = []) {
|
|
119
|
+
const imageDir = images.length ? mkdtempSync(join(tmpdir(), "openmerit-pi-image-")) : null;
|
|
120
|
+
const suffix = {
|
|
121
|
+
"image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif",
|
|
122
|
+
};
|
|
123
|
+
const imageArgs = images.map((image, i) => {
|
|
124
|
+
if (!imageDir || !suffix[image.mimeType])
|
|
125
|
+
throw new Error(`unsupported pi image: ${image.mimeType}`);
|
|
126
|
+
const file = join(imageDir, `image-${i}.${suffix[image.mimeType]}`);
|
|
127
|
+
writeFileSync(file, Buffer.from(image.data, "base64"));
|
|
128
|
+
return `@${file}`;
|
|
129
|
+
});
|
|
130
|
+
const child = spawn("pi", [
|
|
131
|
+
"--provider", "openrouter", "--model", model, "--mode", "json",
|
|
132
|
+
"--offline", "--no-extensions", "--no-tools", "--print", ...imageArgs, task,
|
|
133
|
+
], { cwd, stdio: ["ignore", "pipe", "pipe"] });
|
|
134
|
+
let buffer = "";
|
|
135
|
+
let stderr = "";
|
|
136
|
+
let answer = "";
|
|
137
|
+
let costUsd = 0;
|
|
138
|
+
let errors = 0;
|
|
139
|
+
let sessionId;
|
|
140
|
+
const started = Date.now();
|
|
141
|
+
function consume(line) {
|
|
142
|
+
if (!line.trim())
|
|
143
|
+
return;
|
|
144
|
+
let event;
|
|
145
|
+
try {
|
|
146
|
+
event = JSON.parse(line);
|
|
147
|
+
}
|
|
148
|
+
catch {
|
|
149
|
+
return;
|
|
150
|
+
}
|
|
151
|
+
if (event.type === "session" && event.id)
|
|
152
|
+
sessionId = event.id;
|
|
153
|
+
if (event.type !== "message_end" || event.message?.role !== "assistant")
|
|
154
|
+
return;
|
|
155
|
+
answer = answerText(event.message.content) || answer;
|
|
156
|
+
costUsd += event.message.usage?.cost?.total ?? 0;
|
|
157
|
+
if (event.message.stopReason === "error")
|
|
158
|
+
errors++;
|
|
159
|
+
}
|
|
160
|
+
child.stdout.on("data", (chunk) => {
|
|
161
|
+
buffer += chunk.toString("utf8");
|
|
162
|
+
let n;
|
|
163
|
+
while ((n = buffer.indexOf("\n")) >= 0) {
|
|
164
|
+
consume(buffer.slice(0, n));
|
|
165
|
+
buffer = buffer.slice(n + 1);
|
|
166
|
+
}
|
|
167
|
+
});
|
|
168
|
+
child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
|
|
169
|
+
let exitCode;
|
|
170
|
+
try {
|
|
171
|
+
exitCode = await new Promise((resolve, reject) => {
|
|
172
|
+
child.on("error", reject);
|
|
173
|
+
child.on("close", (code) => resolve(code ?? 1));
|
|
174
|
+
});
|
|
175
|
+
}
|
|
176
|
+
finally {
|
|
177
|
+
if (imageDir)
|
|
178
|
+
rmSync(imageDir, { recursive: true, force: true });
|
|
179
|
+
}
|
|
180
|
+
consume(buffer);
|
|
181
|
+
if (exitCode !== 0 || !answer.trim()) {
|
|
182
|
+
throw new Error(`pi trial ${model} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
|
|
183
|
+
}
|
|
184
|
+
return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId };
|
|
185
|
+
}
|
|
186
|
+
/** Quality uses the same OpenMerit judge; cost and latency come from pi itself. */
|
|
187
|
+
export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = []) {
|
|
188
|
+
try {
|
|
189
|
+
const run = await executePiTask(model, task, cwd, images);
|
|
190
|
+
return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images);
|
|
191
|
+
}
|
|
192
|
+
catch (e) {
|
|
193
|
+
const error = String(e);
|
|
194
|
+
return {
|
|
195
|
+
point: { model, score: 0, price: entry?.price ?? 0,
|
|
196
|
+
ts: new Date().toISOString(), source: "pi_trial", why: error.slice(0, 160) },
|
|
197
|
+
costUsd: 0, error,
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = []) {
|
|
202
|
+
let score = 0;
|
|
203
|
+
let why = "judge unavailable";
|
|
204
|
+
try {
|
|
205
|
+
const pinned = images.length ? await knownInvoiceScore(task, images, run.answer) : null;
|
|
206
|
+
const j = pinned ?? (images.length
|
|
207
|
+
? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images)
|
|
208
|
+
: await judge(key, judgeModel, task, rubric, run.answer));
|
|
209
|
+
score = Math.max(0, Math.min(1, j.score));
|
|
210
|
+
why = j.why;
|
|
211
|
+
}
|
|
212
|
+
catch (e) {
|
|
213
|
+
why = `judge error: ${String(e).slice(0, 120)}`;
|
|
214
|
+
}
|
|
215
|
+
return {
|
|
216
|
+
point: {
|
|
217
|
+
model, score, price: entry?.price ?? 0,
|
|
218
|
+
latencyMs: run.latencyMs, ts: new Date().toISOString(), source,
|
|
219
|
+
why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}`,
|
|
220
|
+
},
|
|
221
|
+
costUsd: run.costUsd,
|
|
222
|
+
sessionId: run.sessionId,
|
|
223
|
+
};
|
|
224
|
+
}
|
package/dist/policy.js
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/** Policy loading, validation, and the swap gate. */
|
|
2
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
3
|
+
export const DEFAULT_POLICY = {
|
|
4
|
+
version: 1,
|
|
5
|
+
mode: "recommend",
|
|
6
|
+
auto_apply: {
|
|
7
|
+
enabled: false,
|
|
8
|
+
min_score_gain: 0.1,
|
|
9
|
+
max_price_ratio: 1.5,
|
|
10
|
+
require_frontier: true,
|
|
11
|
+
},
|
|
12
|
+
budgets: { max_usd_per_trial: 0.25, max_trials_per_day: 20, max_usd_per_day: 5.0 },
|
|
13
|
+
providers: { allow: ["*"], deny: [] },
|
|
14
|
+
watch: { catalog_interval_min: 360, traces_interval_sec: 5, trial_interval_min: 30, models_per_task: 3 },
|
|
15
|
+
fallback: { auto_update: true, min_score: 0.6, apply_on_error: true },
|
|
16
|
+
judge_model: null,
|
|
17
|
+
strategist_model: null,
|
|
18
|
+
max_usd_per_m: 20.0,
|
|
19
|
+
};
|
|
20
|
+
function num(v, fallback) {
|
|
21
|
+
return typeof v === "number" && Number.isFinite(v) ? v : fallback;
|
|
22
|
+
}
|
|
23
|
+
/** Load a policy file, filling defaults for missing keys. Throws on invalid JSON. */
|
|
24
|
+
export function loadPolicy(path) {
|
|
25
|
+
if (!existsSync(path))
|
|
26
|
+
return structuredClone(DEFAULT_POLICY);
|
|
27
|
+
const raw = JSON.parse(readFileSync(path, "utf8"));
|
|
28
|
+
const d = DEFAULT_POLICY;
|
|
29
|
+
return {
|
|
30
|
+
version: num(raw.version, d.version),
|
|
31
|
+
mode: raw.mode === "auto" ? "auto" : "recommend",
|
|
32
|
+
auto_apply: {
|
|
33
|
+
enabled: Boolean(raw.auto_apply?.enabled ?? d.auto_apply.enabled),
|
|
34
|
+
min_score_gain: num(raw.auto_apply?.min_score_gain, d.auto_apply.min_score_gain),
|
|
35
|
+
max_price_ratio: num(raw.auto_apply?.max_price_ratio, d.auto_apply.max_price_ratio),
|
|
36
|
+
require_frontier: Boolean(raw.auto_apply?.require_frontier ?? d.auto_apply.require_frontier),
|
|
37
|
+
},
|
|
38
|
+
budgets: {
|
|
39
|
+
max_usd_per_trial: num(raw.budgets?.max_usd_per_trial, d.budgets.max_usd_per_trial),
|
|
40
|
+
max_trials_per_day: num(raw.budgets?.max_trials_per_day, d.budgets.max_trials_per_day),
|
|
41
|
+
max_usd_per_day: num(raw.budgets?.max_usd_per_day, d.budgets.max_usd_per_day),
|
|
42
|
+
},
|
|
43
|
+
providers: {
|
|
44
|
+
allow: Array.isArray(raw.providers?.allow) ? raw.providers.allow.map(String) : d.providers.allow,
|
|
45
|
+
deny: Array.isArray(raw.providers?.deny) ? raw.providers.deny.map(String) : d.providers.deny,
|
|
46
|
+
},
|
|
47
|
+
watch: {
|
|
48
|
+
catalog_interval_min: num(raw.watch?.catalog_interval_min, d.watch.catalog_interval_min),
|
|
49
|
+
traces_interval_sec: num(raw.watch?.traces_interval_sec, d.watch.traces_interval_sec),
|
|
50
|
+
trial_interval_min: num(raw.watch?.trial_interval_min, d.watch.trial_interval_min),
|
|
51
|
+
models_per_task: num(raw.watch?.models_per_task, d.watch.models_per_task),
|
|
52
|
+
},
|
|
53
|
+
fallback: {
|
|
54
|
+
auto_update: Boolean(raw.fallback?.auto_update ?? d.fallback.auto_update),
|
|
55
|
+
min_score: num(raw.fallback?.min_score, d.fallback.min_score),
|
|
56
|
+
apply_on_error: Boolean(raw.fallback?.apply_on_error ?? d.fallback.apply_on_error),
|
|
57
|
+
},
|
|
58
|
+
judge_model: raw.judge_model ?? null,
|
|
59
|
+
strategist_model: raw.strategist_model ?? null,
|
|
60
|
+
max_usd_per_m: num(raw.max_usd_per_m, d.max_usd_per_m),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
/** Is a model id allowed by the provider allow/deny lists? ("*" matches all; "vendor/*" matches a vendor.) */
|
|
64
|
+
export function providerAllowed(policy, modelId) {
|
|
65
|
+
const vendor = modelId.split("/")[0];
|
|
66
|
+
const match = (pat) => pat === "*" || pat === modelId || (pat.endsWith("/*") && pat.slice(0, -2) === vendor);
|
|
67
|
+
if (policy.providers.deny.some(match))
|
|
68
|
+
return false;
|
|
69
|
+
return policy.providers.allow.some(match);
|
|
70
|
+
}
|
|
71
|
+
/**
|
|
72
|
+
* The swap gate: decide whether a recommendation may be applied without
|
|
73
|
+
* human approval. Always returns the reasons for the decision so the
|
|
74
|
+
* recommendation can explain itself.
|
|
75
|
+
*/
|
|
76
|
+
export function gate(policy, input) {
|
|
77
|
+
const reasons = [];
|
|
78
|
+
if (!input.providerOk) {
|
|
79
|
+
reasons.push("provider not allowed by policy");
|
|
80
|
+
return { autoApply: false, reasons };
|
|
81
|
+
}
|
|
82
|
+
if (input.baselineMeasured === false) {
|
|
83
|
+
reasons.push("current model has no measured baseline");
|
|
84
|
+
return { autoApply: false, reasons };
|
|
85
|
+
}
|
|
86
|
+
if (input.sessionBound === false) {
|
|
87
|
+
reasons.push("recommendation is not bound to a pi session");
|
|
88
|
+
return { autoApply: false, reasons };
|
|
89
|
+
}
|
|
90
|
+
if (policy.mode !== "auto" || !policy.auto_apply.enabled) {
|
|
91
|
+
reasons.push(`policy mode is "${policy.mode}" (auto_apply ${policy.auto_apply.enabled ? "enabled" : "disabled"})`);
|
|
92
|
+
return { autoApply: false, reasons };
|
|
93
|
+
}
|
|
94
|
+
let ok = true;
|
|
95
|
+
if (input.scoreGain < policy.auto_apply.min_score_gain) {
|
|
96
|
+
reasons.push(`score gain ${input.scoreGain.toFixed(3)} < min ${policy.auto_apply.min_score_gain}`);
|
|
97
|
+
ok = false;
|
|
98
|
+
}
|
|
99
|
+
if (input.priceRatio > policy.auto_apply.max_price_ratio) {
|
|
100
|
+
reasons.push(`price ratio ${input.priceRatio.toFixed(2)}x > max ${policy.auto_apply.max_price_ratio}x`);
|
|
101
|
+
ok = false;
|
|
102
|
+
}
|
|
103
|
+
if (policy.auto_apply.require_frontier && !input.onFrontier) {
|
|
104
|
+
reasons.push("candidate is not on the pareto frontier");
|
|
105
|
+
ok = false;
|
|
106
|
+
}
|
|
107
|
+
if (ok)
|
|
108
|
+
reasons.unshift("all auto-apply thresholds met");
|
|
109
|
+
return { autoApply: ok, reasons };
|
|
110
|
+
}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/** Build policy-gated recommendations from a task's frontier + the harness's current model. */
|
|
2
|
+
import { createHash } from "node:crypto";
|
|
3
|
+
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
4
|
+
import { gate, providerAllowed } from "./policy.js";
|
|
5
|
+
export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null) {
|
|
6
|
+
const scored = points.filter((p) => p.score > 0);
|
|
7
|
+
if (scored.length === 0)
|
|
8
|
+
return null;
|
|
9
|
+
const best = pickBest(scored);
|
|
10
|
+
if (!best)
|
|
11
|
+
return null;
|
|
12
|
+
if (currentModel && best.model === currentModel)
|
|
13
|
+
return null; // already on the best model
|
|
14
|
+
const current = currentModel ? scored.find((p) => p.model === currentModel) : undefined;
|
|
15
|
+
const scoreGain = current ? best.score - current.score : 0;
|
|
16
|
+
// pickBest may choose a cheaper or faster model at equal quality.
|
|
17
|
+
if (current && scoreGain < 0)
|
|
18
|
+
return null;
|
|
19
|
+
const priceRatio = current && current.price > 0 ? best.price / current.price : 1;
|
|
20
|
+
const frontierModels = paretoFrontier(scored).map((p) => p.model);
|
|
21
|
+
const onFrontier = frontierModels.includes(best.model);
|
|
22
|
+
const fb = pickFallback(scored, best, policy.fallback.min_score);
|
|
23
|
+
const decision = gate(policy, {
|
|
24
|
+
scoreGain,
|
|
25
|
+
priceRatio,
|
|
26
|
+
onFrontier,
|
|
27
|
+
providerOk: providerAllowed(policy, best.model),
|
|
28
|
+
baselineMeasured: !!current,
|
|
29
|
+
sessionBound: !!sessionFile,
|
|
30
|
+
});
|
|
31
|
+
const reason = `${best.model} scores ${best.score.toFixed(2)} vs ` +
|
|
32
|
+
(current ? `${current.score.toFixed(2)} for ${current.model}` : "no baseline measured") +
|
|
33
|
+
` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), at $${best.price.toFixed(2)}/M` +
|
|
34
|
+
(current ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
|
|
35
|
+
(best.why ? `. Judge: ${best.why}` : "");
|
|
36
|
+
return {
|
|
37
|
+
id: createHash("sha1")
|
|
38
|
+
.update(`${taskKey}:${best.model}:${Date.now()}`)
|
|
39
|
+
.digest("hex")
|
|
40
|
+
.slice(0, 10),
|
|
41
|
+
ts: new Date().toISOString(),
|
|
42
|
+
taskKey,
|
|
43
|
+
taskLabel,
|
|
44
|
+
sessionFile,
|
|
45
|
+
currentModel,
|
|
46
|
+
recommended: { model: best.model, reason },
|
|
47
|
+
fallback: fb
|
|
48
|
+
? {
|
|
49
|
+
model: fb.model,
|
|
50
|
+
reason: `score ${fb.score.toFixed(2)} at $${fb.price.toFixed(2)}/M` +
|
|
51
|
+
(fb.latencyMs ? `, ~${Math.round(fb.latencyMs)}ms` : ""),
|
|
52
|
+
}
|
|
53
|
+
: null,
|
|
54
|
+
evidence: {
|
|
55
|
+
scoreGain: Math.round(scoreGain * 1000) / 1000,
|
|
56
|
+
priceRatio: Math.round(priceRatio * 100) / 100,
|
|
57
|
+
onFrontier,
|
|
58
|
+
trials: scored.length,
|
|
59
|
+
},
|
|
60
|
+
policy: decision,
|
|
61
|
+
status: "pending",
|
|
62
|
+
};
|
|
63
|
+
}
|
package/dist/store.js
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/** State directory layout + JSONL persistence. All state lives under OPENMERIT_HOME (default ~/.openmerit). */
|
|
2
|
+
import { createHash } from "node:crypto";
|
|
3
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
4
|
+
import { homedir } from "node:os";
|
|
5
|
+
import { join } from "node:path";
|
|
6
|
+
export function stateDir() {
|
|
7
|
+
const dir = process.env.OPENMERIT_HOME ?? join(homedir(), ".openmerit");
|
|
8
|
+
if (!existsSync(dir))
|
|
9
|
+
mkdirSync(dir, { recursive: true });
|
|
10
|
+
return dir;
|
|
11
|
+
}
|
|
12
|
+
export const paths = {
|
|
13
|
+
policy: () => join(stateDir(), "policy.json"),
|
|
14
|
+
trials: () => join(stateDir(), "trials.jsonl"),
|
|
15
|
+
recommendations: () => join(stateDir(), "recommendations.jsonl"),
|
|
16
|
+
catalogSnapshot: () => join(stateDir(), "catalog", "snapshot.json"),
|
|
17
|
+
candidates: () => join(stateDir(), "catalog", "candidates.json"),
|
|
18
|
+
benchmarksDigest: () => join(stateDir(), "benchmarks", "digest.json"),
|
|
19
|
+
traceCursor: () => join(stateDir(), "traces", "cursor.json"),
|
|
20
|
+
observations: () => join(stateDir(), "traces", "observations.jsonl"),
|
|
21
|
+
harnessState: () => join(stateDir(), "harness-state.json"),
|
|
22
|
+
ledger: () => join(stateDir(), "ledger.json"),
|
|
23
|
+
watchProcessed: () => join(stateDir(), "watch", "processed.json"),
|
|
24
|
+
sessionJob: (marker) => join(stateDir(), "watch", "jobs", sha1(marker) + ".json"),
|
|
25
|
+
envFile: () => join(stateDir(), ".env"),
|
|
26
|
+
};
|
|
27
|
+
export function sha1(text) {
|
|
28
|
+
return createHash("sha1").update(text).digest("hex").slice(0, 12);
|
|
29
|
+
}
|
|
30
|
+
/** Stable task key from a task statement. */
|
|
31
|
+
export function taskKey(taskText) {
|
|
32
|
+
const normalized = taskText.toLowerCase().replace(/\s+/g, " ").trim();
|
|
33
|
+
return sha1(normalized);
|
|
34
|
+
}
|
|
35
|
+
export function appendJsonl(file, obj) {
|
|
36
|
+
mkdirSync(join(file, ".."), { recursive: true });
|
|
37
|
+
appendFileSync(file, JSON.stringify(obj) + "\n");
|
|
38
|
+
}
|
|
39
|
+
export function readJsonl(file) {
|
|
40
|
+
if (!existsSync(file))
|
|
41
|
+
return [];
|
|
42
|
+
return readFileSync(file, "utf8")
|
|
43
|
+
.split("\n")
|
|
44
|
+
.filter((l) => l.trim())
|
|
45
|
+
.map((l) => JSON.parse(l));
|
|
46
|
+
}
|
|
47
|
+
export function readJson(file, fallback) {
|
|
48
|
+
if (!existsSync(file))
|
|
49
|
+
return fallback;
|
|
50
|
+
return JSON.parse(readFileSync(file, "utf8"));
|
|
51
|
+
}
|
|
52
|
+
export function writeJson(file, obj) {
|
|
53
|
+
mkdirSync(join(file, ".."), { recursive: true });
|
|
54
|
+
writeFileSync(file, JSON.stringify(obj, null, 2) + "\n");
|
|
55
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/** Strategist: pick the next untried model to trial, benchmark-aware. */
|
|
2
|
+
import { chat } from "./llm.js";
|
|
3
|
+
import { parseObj } from "./judge.js";
|
|
4
|
+
export const STRAT_PREFS = [
|
|
5
|
+
"anthropic/claude-3.7-sonnet",
|
|
6
|
+
"anthropic/claude-3.5-sonnet",
|
|
7
|
+
"openai/gpt-4o",
|
|
8
|
+
];
|
|
9
|
+
const STRAT_PROMPT = `We are finding the best OpenRouter model for a task via iterative trials.
|
|
10
|
+
TASK: {task}
|
|
11
|
+
RUBRIC: {rubric}
|
|
12
|
+
RELEVANT PUBLIC BENCHMARKS FOR THIS TASK: {benchmarks}
|
|
13
|
+
RESULTS SO FAR (model=score): {results}
|
|
14
|
+
Pick ONE untried model from the catalog below that is likely to do well on this
|
|
15
|
+
task, based on the benchmarks relevant to it. Balance quality against price so
|
|
16
|
+
a pareto frontier emerges.
|
|
17
|
+
Return ONLY json: {{"next_model": "<catalog id>", "why": "<one line>"}}
|
|
18
|
+
CATALOG (id | blended $/1M tokens | ctx):
|
|
19
|
+
{catalog}`;
|
|
20
|
+
/** Pick the next model to trial, or null when the catalog is exhausted. */
|
|
21
|
+
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates) {
|
|
22
|
+
const ok = (c) => c.price <= maxPrice &&
|
|
23
|
+
!tried.has(c.id) &&
|
|
24
|
+
c.ctx >= 4096 &&
|
|
25
|
+
!failedVendors.has(c.id.split("/")[0]);
|
|
26
|
+
// Benchmark-shortlisted models get priority in the listing.
|
|
27
|
+
const pool = [...cat.values()].filter(ok);
|
|
28
|
+
if (pool.length === 0)
|
|
29
|
+
return null;
|
|
30
|
+
const priority = new Set(extraCandidates ?? []);
|
|
31
|
+
pool.sort((a, b) => {
|
|
32
|
+
const pa = priority.has(a.id) ? 0 : 1;
|
|
33
|
+
const pb = priority.has(b.id) ? 0 : 1;
|
|
34
|
+
return pa - pb || a.price - b.price;
|
|
35
|
+
});
|
|
36
|
+
const lines = pool
|
|
37
|
+
.map((c) => `${c.id} | ${c.price.toFixed(2)} | ${c.ctx}${priority.has(c.id) ? " | BENCHMARK" : ""}`)
|
|
38
|
+
.join("\n");
|
|
39
|
+
const res = results
|
|
40
|
+
.map((r) => `${r.model}=${r.score.toFixed(2)}`)
|
|
41
|
+
.join(" ") || "none yet";
|
|
42
|
+
const { content } = await chat(key, stratModel, STRAT_PROMPT.replace("{task}", task)
|
|
43
|
+
.replace("{rubric}", rubric)
|
|
44
|
+
.replace("{benchmarks}", benchmarks.join(", "))
|
|
45
|
+
.replace("{results}", res)
|
|
46
|
+
.replace("{catalog}", lines), 1024, 0.2);
|
|
47
|
+
let pick = null;
|
|
48
|
+
let why = "unparseable strategist output";
|
|
49
|
+
try {
|
|
50
|
+
const obj = parseObj(content);
|
|
51
|
+
pick = typeof obj.next_model === "string" ? obj.next_model : null;
|
|
52
|
+
why = String(obj.why ?? "");
|
|
53
|
+
}
|
|
54
|
+
catch {
|
|
55
|
+
/* fall through to cheap pick */
|
|
56
|
+
}
|
|
57
|
+
if (pick) {
|
|
58
|
+
const entry = cat.get(pick);
|
|
59
|
+
if (entry && ok(entry))
|
|
60
|
+
return { model: pick, why };
|
|
61
|
+
}
|
|
62
|
+
const cheapest = pool[0];
|
|
63
|
+
return { model: cheapest.id, why: `fallback pick: cheapest untried (strategist pick ${pick ?? "none"} unusable)` };
|
|
64
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/** A pi text task may carry images embedded in its saved session message. */
|
|
2
|
+
import { createHash } from "node:crypto";
|
|
3
|
+
import { taskKey } from "./store.js";
|
|
4
|
+
const SUPPORTED = new Set(["image/jpeg", "image/png", "image/webp", "image/gif"]);
|
|
5
|
+
export function sessionImages(content) {
|
|
6
|
+
if (!Array.isArray(content))
|
|
7
|
+
return [];
|
|
8
|
+
const images = [];
|
|
9
|
+
for (const part of content) {
|
|
10
|
+
if (!part || typeof part !== "object")
|
|
11
|
+
return null;
|
|
12
|
+
if (part.type === "text")
|
|
13
|
+
continue;
|
|
14
|
+
if (part.type !== "image" || typeof part.data !== "string" ||
|
|
15
|
+
typeof part.mimeType !== "string" || !SUPPORTED.has(part.mimeType) ||
|
|
16
|
+
!/^[A-Za-z0-9+/]+={0,2}$/.test(part.data))
|
|
17
|
+
return null;
|
|
18
|
+
images.push({ type: "image", data: part.data, mimeType: part.mimeType });
|
|
19
|
+
}
|
|
20
|
+
return images;
|
|
21
|
+
}
|
|
22
|
+
/** Text-only keys remain compatible; image bytes distinguish same-prompt documents. */
|
|
23
|
+
export function taskInputKey(text, images) {
|
|
24
|
+
if (images.length === 0)
|
|
25
|
+
return taskKey(text);
|
|
26
|
+
const normalized = text.toLowerCase().replace(/\s+/g, " ").trim();
|
|
27
|
+
const hashes = images.map((image) => `${image.mimeType}:` +
|
|
28
|
+
createHash("sha256").update(Buffer.from(image.data, "base64")).digest("hex"));
|
|
29
|
+
return taskKey(`${normalized}\nimages:${hashes.join(",")}`);
|
|
30
|
+
}
|
package/dist/traces.js
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Trace ingestion: read pi session files (JSONL entry trees) and extract
|
|
3
|
+
* per-task observations (model, cost, tokens, latency, errors). Incremental
|
|
4
|
+
* via a per-file byte-offset cursor, so the background track never re-reads.
|
|
5
|
+
*/
|
|
6
|
+
import { readdirSync, readFileSync, existsSync, statSync } from "node:fs";
|
|
7
|
+
import { homedir } from "node:os";
|
|
8
|
+
import { join } from "node:path";
|
|
9
|
+
import { paths, readJson, taskKey, writeJson } from "./store.js";
|
|
10
|
+
export function defaultSessionsDir() {
|
|
11
|
+
return join(homedir(), ".pi", "agent", "sessions");
|
|
12
|
+
}
|
|
13
|
+
function listSessionFiles(dir) {
|
|
14
|
+
if (!existsSync(dir))
|
|
15
|
+
return [];
|
|
16
|
+
const out = [];
|
|
17
|
+
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
|
18
|
+
const p = join(dir, entry.name);
|
|
19
|
+
if (entry.isDirectory())
|
|
20
|
+
out.push(...listSessionFiles(p));
|
|
21
|
+
else if (entry.name.endsWith(".jsonl"))
|
|
22
|
+
out.push(p);
|
|
23
|
+
}
|
|
24
|
+
return out;
|
|
25
|
+
}
|
|
26
|
+
function textOf(content) {
|
|
27
|
+
if (typeof content === "string")
|
|
28
|
+
return content;
|
|
29
|
+
if (Array.isArray(content)) {
|
|
30
|
+
return content
|
|
31
|
+
.map((b) => (b && typeof b === "object" && b.type === "text"
|
|
32
|
+
? String(b.text ?? "")
|
|
33
|
+
: ""))
|
|
34
|
+
.filter(Boolean)
|
|
35
|
+
.join(" ");
|
|
36
|
+
}
|
|
37
|
+
return "";
|
|
38
|
+
}
|
|
39
|
+
/** Extract observations from the entries of one session file. */
|
|
40
|
+
export function observationsFromEntries(entries, sessionFile) {
|
|
41
|
+
const obs = [];
|
|
42
|
+
let current = null;
|
|
43
|
+
for (const e of entries) {
|
|
44
|
+
if (e.type !== "message" || !e.message)
|
|
45
|
+
continue;
|
|
46
|
+
const m = e.message;
|
|
47
|
+
if (m.role === "user") {
|
|
48
|
+
if (current)
|
|
49
|
+
obs.push(current);
|
|
50
|
+
const text = textOf(m.content);
|
|
51
|
+
if (!text.trim()) {
|
|
52
|
+
current = null;
|
|
53
|
+
continue;
|
|
54
|
+
}
|
|
55
|
+
current = {
|
|
56
|
+
taskKey: taskKey(text),
|
|
57
|
+
taskLabel: text.replace(/\s+/g, " ").trim().slice(0, 120),
|
|
58
|
+
model: null,
|
|
59
|
+
costUsd: 0,
|
|
60
|
+
tokens: 0,
|
|
61
|
+
latencyMs: null,
|
|
62
|
+
errors: 0,
|
|
63
|
+
ts: e.timestamp ?? new Date().toISOString(),
|
|
64
|
+
sessionFile,
|
|
65
|
+
};
|
|
66
|
+
continue;
|
|
67
|
+
}
|
|
68
|
+
if (!current)
|
|
69
|
+
continue;
|
|
70
|
+
if (m.role === "assistant") {
|
|
71
|
+
if (!current.model && m.provider && m.model)
|
|
72
|
+
current.model = `${m.provider}/${m.model}`;
|
|
73
|
+
current.costUsd += m.usage?.cost?.total ?? 0;
|
|
74
|
+
current.tokens += m.usage?.totalTokens ?? 0;
|
|
75
|
+
if (m.stopReason === "error")
|
|
76
|
+
current.errors += 1;
|
|
77
|
+
if (typeof m.timestamp === "number") {
|
|
78
|
+
const start = Date.parse(current.ts);
|
|
79
|
+
if (!Number.isNaN(start))
|
|
80
|
+
current.latencyMs = Math.max(0, m.timestamp - start);
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
else if (m.role === "toolResult") {
|
|
84
|
+
const tr = m;
|
|
85
|
+
if (tr.isError)
|
|
86
|
+
current.errors += 1;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
if (current)
|
|
90
|
+
obs.push(current);
|
|
91
|
+
return obs;
|
|
92
|
+
}
|
|
93
|
+
/** Read new bytes of every session file since the last run; returns fresh observations. */
|
|
94
|
+
export function ingestNewTraces(sessionsDir = defaultSessionsDir()) {
|
|
95
|
+
const cursor = readJson(paths.traceCursor(), { offsets: {} });
|
|
96
|
+
const fresh = [];
|
|
97
|
+
for (const file of listSessionFiles(sessionsDir)) {
|
|
98
|
+
const size = statSync(file).size;
|
|
99
|
+
const from = cursor.offsets[file] ?? 0;
|
|
100
|
+
if (size <= from)
|
|
101
|
+
continue;
|
|
102
|
+
const buf = readFileSync(file);
|
|
103
|
+
const chunk = buf.subarray(from).toString("utf8");
|
|
104
|
+
// Drop a trailing partial line; it will be re-read next tick.
|
|
105
|
+
const lastNl = chunk.lastIndexOf("\n");
|
|
106
|
+
const complete = lastNl >= 0 ? chunk.slice(0, lastNl + 1) : "";
|
|
107
|
+
const entries = [];
|
|
108
|
+
for (const line of complete.split("\n")) {
|
|
109
|
+
if (!line.trim())
|
|
110
|
+
continue;
|
|
111
|
+
try {
|
|
112
|
+
entries.push(JSON.parse(line));
|
|
113
|
+
}
|
|
114
|
+
catch {
|
|
115
|
+
/* skip corrupt line */
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
fresh.push(...observationsFromEntries(entries, file));
|
|
119
|
+
cursor.offsets[file] = from + Buffer.byteLength(complete);
|
|
120
|
+
}
|
|
121
|
+
writeJson(paths.traceCursor(), cursor);
|
|
122
|
+
return fresh;
|
|
123
|
+
}
|