openmerit 0.1.3 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -312
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -388
  52. package/dist/diagnostics.js +0 -194
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -203
  60. package/dist/pi-trials.js +0 -366
  61. package/dist/policy.js +0 -115
  62. package/dist/providers.js +0 -1
  63. package/dist/recommend.js +0 -76
  64. package/dist/routes.js +0 -59
  65. package/dist/standalone.js +0 -220
  66. package/dist/store.js +0 -89
  67. package/dist/strategist.js +0 -68
  68. package/dist/task-input.js +0 -54
  69. package/dist/traces.js +0 -127
  70. package/dist/trials.js +0 -140
  71. package/dist/types.js +0 -2
  72. package/examples/invoice-prompt.txt +0 -19
  73. package/examples/task.example.json +0 -7
  74. package/extension/openmerit.ts +0 -820
  75. package/instructions/OPENMERIT.md +0 -54
  76. package/instructions/openmerit.policy.json +0 -33
  77. package/rules.md +0 -39
package/dist/pi-trials.js DELETED
@@ -1,366 +0,0 @@
1
- /** Run a task through pi for each model, using pi's JSON event stream as the measurement source. */
2
- import { spawn, spawnSync } from "node:child_process";
3
- import { cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
4
- import { tmpdir } from "node:os";
5
- import { dirname, join, relative } from "node:path";
6
- import { createHash } from "node:crypto";
7
- import { judge, judgeWithImages } from "./judge.js";
8
- import { knownInvoiceScore } from "./invoice-eval.js";
9
- import { paths, readJson } from "./store.js";
10
- import { sessionFiles, sessionImages, taskInputKey } from "./task-input.js";
11
- import { legacyOpenRouterRoute, meritModelId, modelRoute, routeKey } from "./routes.js";
12
- import { directChatClient } from "./llm.js";
13
- function answerText(content) {
14
- if (typeof content === "string")
15
- return content;
16
- if (!Array.isArray(content))
17
- return "";
18
- return content.filter((x) => !!x && typeof x === "object" && x.type === "text")
19
- .map((x) => x.text ?? "").join("\n");
20
- }
21
- /** Pi-backed harness adapter. Future harnesses can implement the same seam. */
22
- export class PiHarnessAdapter {
23
- runTask(model, input) {
24
- return executePiTask(model, input.task, input.cwd, input.images, input.files);
25
- }
26
- }
27
- const defaultHarness = new PiHarnessAdapter();
28
- const KNOWN_TRIAL_TOOLS = new Set([
29
- "read", "grep", "find", "ls", "bash", "powershell", "edit", "write",
30
- ]);
31
- /**
32
- * Candidate runs have no tools by default. Users may explicitly opt into a
33
- * list, but a copied cwd is not an OS sandbox for absolute paths or network.
34
- */
35
- export function piTrialToolArgs(setting = process.env.OPENMERIT_PI_TRIAL_TOOLS) {
36
- if (!setting?.trim())
37
- return ["--no-tools"];
38
- const requested = [...new Set(setting.split(",").map((tool) => tool.trim()).filter(Boolean))];
39
- if (requested.length === 1 && requested[0] === "none")
40
- return ["--no-tools"];
41
- const unknown = requested.filter((tool) => !KNOWN_TRIAL_TOOLS.has(tool));
42
- if (requested.length === 0 || unknown.length > 0) {
43
- throw new Error(`invalid OPENMERIT_PI_TRIAL_TOOLS${unknown.length ? `: ${unknown.join(", ")}` : ""}`);
44
- }
45
- return ["--tools", requested.join(",")];
46
- }
47
- function fileSnapshot(root) {
48
- const out = new Map();
49
- function walk(dir) {
50
- for (const name of readdirSync(dir)) {
51
- if (name === ".git" || name === "node_modules" || name === ".openmerit")
52
- continue;
53
- const file = join(dir, name);
54
- const rel = relative(root, file);
55
- const st = statSync(file);
56
- if (st.isDirectory())
57
- walk(file);
58
- else if (st.isFile() && st.size < 20_000_000) {
59
- out.set(rel, createHash("sha1").update(readFileSync(file)).digest("hex"));
60
- }
61
- }
62
- }
63
- walk(root);
64
- return out;
65
- }
66
- function isolatedWorkspace(cwd) {
67
- const root = mkdtempSync(join(tmpdir(), "openmerit-pi-workspace-"));
68
- cpSync(cwd, root, { recursive: true, filter: (src) => {
69
- const rel = relative(cwd, src);
70
- return !rel.split("/").some((part) => part === ".git" || part === "node_modules" || part === ".openmerit");
71
- } });
72
- return root;
73
- }
74
- /** Use model A's completed result from the active pi session when it matches this task. */
75
- export function recordedActiveTask(task, model, images = [], sessionFile, sessionBytes) {
76
- const state = sessionFile ? null : readJson(paths.harnessState(), {});
77
- const file = sessionFile ?? state?.sessionFile;
78
- if (!file || !existsSync(file))
79
- return null;
80
- let user = null;
81
- const assistants = [];
82
- let toolCalls = 0;
83
- let toolErrors = 0;
84
- for (const line of readFileSync(file).subarray(0, sessionBytes).toString("utf8").split("\n")) {
85
- if (!line.trim())
86
- continue;
87
- let entry;
88
- try {
89
- entry = JSON.parse(line);
90
- }
91
- catch {
92
- continue;
93
- }
94
- if (entry.type !== "message" || !entry.message)
95
- continue;
96
- if (entry.message.role === "user") {
97
- user = entry.message;
98
- assistants.length = 0;
99
- toolCalls = 0;
100
- toolErrors = 0;
101
- }
102
- else if (user && entry.message.role === "assistant") {
103
- assistants.push(entry.message);
104
- if (Array.isArray(entry.message.content))
105
- toolCalls += entry.message.content.filter((part) => part?.type === "toolCall").length;
106
- }
107
- else if (user && entry.message.role === "toolResult" &&
108
- entry.message.isError)
109
- toolErrors++;
110
- }
111
- const userImages = user ? sessionImages(user.content) : null;
112
- const files = sessionFiles(task);
113
- if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
114
- taskInputKey(task, images, files))
115
- return null;
116
- const matching = assistants.filter((m) => typeof model === "string"
117
- ? meritModelId(m.provider ?? "openrouter", m.model ?? "") === model
118
- : m.provider === model.provider && m.model === model.modelId);
119
- if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
120
- return null;
121
- const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
122
- if (!answer.trim())
123
- return null;
124
- const first = user.timestamp ?? 0;
125
- const last = matching.at(-1)?.timestamp ?? first;
126
- return {
127
- answer,
128
- costUsd: matching.reduce((n, m) => n + (m.usage?.cost?.total ?? 0), 0),
129
- latencyMs: Math.max(0, last - first),
130
- errors: 0,
131
- toolCalls,
132
- toolErrors,
133
- changedFiles: 0,
134
- };
135
- }
136
- /** Read the exact latest completed text task from the pi extension's session marker. */
137
- export function settledActiveTask(snapshot) {
138
- const st = snapshot ?? readJson(paths.harnessState(), {});
139
- if (!st.currentModel || !st.sessionFile || !st.settledTaskKey || !st.settledAt ||
140
- !existsSync(st.sessionFile))
141
- return null;
142
- let userContent = null;
143
- let usedTools = false;
144
- for (const line of readFileSync(st.sessionFile).subarray(0, st.sessionBytes).toString("utf8").split("\n")) {
145
- if (!line.trim())
146
- continue;
147
- let entry;
148
- try {
149
- entry = JSON.parse(line);
150
- }
151
- catch {
152
- continue;
153
- }
154
- if (entry.type !== "message" || !entry.message)
155
- continue;
156
- if (entry.message.role === "user") {
157
- userContent = entry.message.content;
158
- usedTools = false;
159
- }
160
- else if (userContent && entry.message.role === "toolResult")
161
- usedTools = true;
162
- else if (userContent && entry.message.role === "assistant" &&
163
- Array.isArray(entry.message.content) && entry.message.content.some((b) => b?.type === "toolCall"))
164
- usedTools = true;
165
- }
166
- if (!userContent)
167
- return null;
168
- const images = sessionImages(userContent);
169
- if (!images)
170
- return null;
171
- const task = answerText(userContent);
172
- const files = sessionFiles(task);
173
- if (!task.trim() || taskInputKey(task, images, files) !== st.settledTaskKey)
174
- return null;
175
- const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
176
- const run = recordedActiveTask(task, route, images, st.sessionFile, st.sessionBytes);
177
- if (!run)
178
- return null;
179
- const routes = st.routes?.length ? st.routes : [route];
180
- return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
181
- settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
182
- route, routes };
183
- }
184
- export function piExecutable() {
185
- return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
186
- }
187
- function compactNumber(value) {
188
- const match = value.match(/^(\d+(?:\.\d+)?)([KMG])?$/i);
189
- if (!match)
190
- return undefined;
191
- const scale = { K: 1_000, M: 1_000_000, G: 1_000_000_000 }[(match[2] ?? "").toUpperCase()] ?? 1;
192
- return Math.round(Number(match[1]) * scale);
193
- }
194
- /** Read provider/model routes from pi when no live extension snapshot is available. */
195
- export function availablePiRoutes(snapshot) {
196
- if (snapshot?.length)
197
- return snapshot;
198
- const result = spawnSync(piExecutable(), ["--offline", "--list-models"], { encoding: "utf8" });
199
- if (result.error || result.status !== 0)
200
- throw new Error("could not list models from pi");
201
- const routes = [];
202
- for (const line of result.stdout.split("\n").slice(1)) {
203
- const [provider, id, context, maxOut, , images] = line.trim().split(/\s+/);
204
- if (!provider || !id || id.startsWith("~"))
205
- continue;
206
- routes.push(modelRoute(provider, id, {
207
- input: images === "yes" ? ["text", "image"] : ["text"],
208
- contextWindow: compactNumber(context),
209
- maxTokens: compactNumber(maxOut),
210
- }));
211
- }
212
- return routes;
213
- }
214
- /** Backward-compatible logical-id view used by the standalone OpenRouter trial command. */
215
- export function availablePiModels() {
216
- return new Set(availablePiRoutes().map((route) => route.meritId));
217
- }
218
- /** Each invocation is a fresh pi run, so model B does not inherit model A's answer. */
219
- export async function executePiTask(model, task, cwd = process.cwd(), images = [], files = []) {
220
- const route = typeof model === "string" ? legacyOpenRouterRoute(model) : model;
221
- const imageDir = images.length ? mkdtempSync(join(tmpdir(), "openmerit-pi-image-")) : null;
222
- const suffix = {
223
- "image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif",
224
- };
225
- const imageArgs = images.map((image, i) => {
226
- if (!imageDir || !suffix[image.mimeType])
227
- throw new Error(`unsupported pi image: ${image.mimeType}`);
228
- const file = join(imageDir, `image-${i}.${suffix[image.mimeType]}`);
229
- writeFileSync(file, Buffer.from(image.data, "base64"));
230
- return `@${file}`;
231
- });
232
- const workspace = isolatedWorkspace(cwd);
233
- try {
234
- const attachmentDir = join(workspace, ".openmerit-attachments");
235
- mkdirSync(attachmentDir, { recursive: true });
236
- const attachmentPaths = new Map();
237
- const fileArgs = files.map((file, i) => {
238
- const safeName = `${String(i).padStart(3, "0")}-${file.name.replace(/[^A-Za-z0-9._-]/g, "_")}`;
239
- const target = join(attachmentDir, safeName);
240
- cpSync(file.path, target);
241
- attachmentPaths.set(file.path, target);
242
- return `@${target}`;
243
- });
244
- const candidateTask = [...attachmentPaths.entries()].reduce((text, [source, target]) => text.split(source).join(target), task);
245
- const before = fileSnapshot(workspace);
246
- const traceFile = paths.trialTrace(`${routeKey(route)}:${task}:${Date.now()}`);
247
- const traceLines = [];
248
- const child = spawn(piExecutable(), [
249
- "--provider", route.provider, "--model", route.modelId, "--mode", "json",
250
- "--offline", "--no-extensions", "--approve", ...piTrialToolArgs(),
251
- "--print", ...imageArgs, ...fileArgs, candidateTask,
252
- ], { cwd: workspace, stdio: ["ignore", "pipe", "pipe"] });
253
- let buffer = "";
254
- let stderr = "";
255
- let answer = "";
256
- let costUsd = 0;
257
- let errors = 0;
258
- let sessionId;
259
- let toolCalls = 0;
260
- let toolErrors = 0;
261
- const started = Date.now();
262
- function consume(line) {
263
- if (!line.trim())
264
- return;
265
- traceLines.push(line);
266
- let event;
267
- try {
268
- event = JSON.parse(line);
269
- }
270
- catch {
271
- return;
272
- }
273
- if (event.type === "session" && event.id)
274
- sessionId = event.id;
275
- const content = event.message?.content;
276
- if (event.message?.role === "assistant" && Array.isArray(content))
277
- toolCalls += content.filter((part) => part?.type === "toolCall").length;
278
- if (event.message?.role === "toolResult" && event.message.isError)
279
- toolErrors++;
280
- if (event.type !== "message_end" || event.message?.role !== "assistant")
281
- return;
282
- answer = answerText(event.message.content) || answer;
283
- costUsd += event.message.usage?.cost?.total ?? 0;
284
- if (event.message.stopReason === "error")
285
- errors++;
286
- }
287
- child.stdout.on("data", (chunk) => {
288
- buffer += chunk.toString("utf8");
289
- let n;
290
- while ((n = buffer.indexOf("\n")) >= 0) {
291
- consume(buffer.slice(0, n));
292
- buffer = buffer.slice(n + 1);
293
- }
294
- });
295
- child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
296
- const exitCode = await new Promise((resolve, reject) => {
297
- child.on("error", reject);
298
- child.on("close", (code) => resolve(code ?? 1));
299
- });
300
- consume(buffer);
301
- mkdirSync(dirname(traceFile), { recursive: true });
302
- writeFileSync(traceFile, traceLines.join("\n") + (traceLines.length ? "\n" : ""));
303
- const after = fileSnapshot(workspace);
304
- let changedFiles = 0;
305
- for (const [file, hash] of after)
306
- if (before.get(file) !== hash)
307
- changedFiles++;
308
- for (const file of before.keys())
309
- if (!after.has(file))
310
- changedFiles++;
311
- if (exitCode !== 0 || !answer.trim()) {
312
- throw new Error(`pi trial ${route.meritId} via ${route.provider} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
313
- }
314
- return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId,
315
- toolCalls, toolErrors, changedFiles, traceFile };
316
- }
317
- finally {
318
- if (imageDir)
319
- rmSync(imageDir, { recursive: true, force: true });
320
- rmSync(workspace, { recursive: true, force: true });
321
- }
322
- }
323
- /** Quality uses the same OpenMerit judge; cost and latency come from pi itself. */
324
- export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = [], files = [], harness = defaultHarness, client = directChatClient(key)) {
325
- try {
326
- const run = await harness.runTask(entry?.route ?? model, { task, cwd, images, files });
327
- return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images, client);
328
- }
329
- catch (e) {
330
- const error = String(e);
331
- return {
332
- point: { schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
333
- priceKnown: entry?.priceKnown ?? !!entry,
334
- ts: new Date().toISOString(), source: "pi_trial", toolCalls: 0, toolErrors: 1,
335
- changedFiles: 0, why: error.slice(0, 160) },
336
- costUsd: 0, error,
337
- };
338
- }
339
- }
340
- export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = [], client = directChatClient(key)) {
341
- let score = 0;
342
- let why = "judge unavailable";
343
- try {
344
- const pinned = images.length ? await knownInvoiceScore(task, images, run.answer) : null;
345
- const j = pinned ?? (images.length
346
- ? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images, client)
347
- : await judge(key, judgeModel, task, rubric, run.answer, client));
348
- score = Math.max(0, Math.min(1, j.score));
349
- why = j.why;
350
- }
351
- catch (e) {
352
- why = `judge error: ${String(e).slice(0, 120)}`;
353
- }
354
- return {
355
- point: {
356
- schemaVersion: 1, model, route: entry?.route, score, price: entry?.price ?? 0,
357
- priceKnown: entry?.priceKnown ?? !!entry,
358
- latencyMs: run.latencyMs, ts: new Date().toISOString(), source,
359
- why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}; ` +
360
- `tools=${run.toolCalls}, toolErrors=${run.toolErrors}, changedFiles=${run.changedFiles}`,
361
- toolCalls: run.toolCalls, toolErrors: run.toolErrors, changedFiles: run.changedFiles,
362
- },
363
- costUsd: run.costUsd,
364
- sessionId: run.sessionId,
365
- };
366
- }
package/dist/policy.js DELETED
@@ -1,115 +0,0 @@
1
- /** Policy loading, validation, and the swap gate. */
2
- import { existsSync, readFileSync } from "node:fs";
3
- export const DEFAULT_POLICY = {
4
- version: 1,
5
- mode: "recommend",
6
- auto_apply: {
7
- enabled: false,
8
- min_score_gain: 0.1,
9
- max_price_ratio: 1.5,
10
- require_frontier: true,
11
- },
12
- budgets: { max_usd_per_trial: 0.25, max_trials_per_day: 20, max_usd_per_day: 5.0 },
13
- providers: { allow: ["*"], deny: [] },
14
- watch: { catalog_interval_min: 360, traces_interval_sec: 5, trial_interval_min: 30, models_per_task: 3 },
15
- fallback: { auto_update: true, min_score: 0.6, apply_on_error: true },
16
- judge_model: null,
17
- strategist_model: null,
18
- max_usd_per_m: 20.0,
19
- };
20
- function num(v, fallback) {
21
- return typeof v === "number" && Number.isFinite(v) ? v : fallback;
22
- }
23
- /** Load a policy file, filling defaults for missing keys. Throws on invalid JSON. */
24
- export function loadPolicy(path) {
25
- if (!existsSync(path))
26
- return structuredClone(DEFAULT_POLICY);
27
- const raw = JSON.parse(readFileSync(path, "utf8"));
28
- const d = DEFAULT_POLICY;
29
- return {
30
- version: num(raw.version, d.version),
31
- mode: raw.mode === "auto" ? "auto" : "recommend",
32
- auto_apply: {
33
- enabled: Boolean(raw.auto_apply?.enabled ?? d.auto_apply.enabled),
34
- min_score_gain: num(raw.auto_apply?.min_score_gain, d.auto_apply.min_score_gain),
35
- max_price_ratio: num(raw.auto_apply?.max_price_ratio, d.auto_apply.max_price_ratio),
36
- require_frontier: Boolean(raw.auto_apply?.require_frontier ?? d.auto_apply.require_frontier),
37
- },
38
- budgets: {
39
- max_usd_per_trial: num(raw.budgets?.max_usd_per_trial, d.budgets.max_usd_per_trial),
40
- max_trials_per_day: num(raw.budgets?.max_trials_per_day, d.budgets.max_trials_per_day),
41
- max_usd_per_day: num(raw.budgets?.max_usd_per_day, d.budgets.max_usd_per_day),
42
- },
43
- providers: {
44
- allow: Array.isArray(raw.providers?.allow) ? raw.providers.allow.map(String) : d.providers.allow,
45
- deny: Array.isArray(raw.providers?.deny) ? raw.providers.deny.map(String) : d.providers.deny,
46
- },
47
- watch: {
48
- catalog_interval_min: num(raw.watch?.catalog_interval_min, d.watch.catalog_interval_min),
49
- traces_interval_sec: num(raw.watch?.traces_interval_sec, d.watch.traces_interval_sec),
50
- trial_interval_min: num(raw.watch?.trial_interval_min, d.watch.trial_interval_min),
51
- models_per_task: num(raw.watch?.models_per_task, d.watch.models_per_task),
52
- },
53
- fallback: {
54
- auto_update: Boolean(raw.fallback?.auto_update ?? d.fallback.auto_update),
55
- min_score: num(raw.fallback?.min_score, d.fallback.min_score),
56
- apply_on_error: Boolean(raw.fallback?.apply_on_error ?? d.fallback.apply_on_error),
57
- },
58
- judge_model: raw.judge_model ?? null,
59
- strategist_model: raw.strategist_model ?? null,
60
- max_usd_per_m: num(raw.max_usd_per_m, d.max_usd_per_m),
61
- };
62
- }
63
- /** Match either a logical vendor/model id or its concrete route provider. */
64
- export function providerAllowed(policy, modelId, routeProvider) {
65
- const vendor = modelId.split("/")[0];
66
- const match = (pat) => pat === "*" || pat === modelId || pat === routeProvider ||
67
- (pat.endsWith("/*") && (pat.slice(0, -2) === vendor || pat.slice(0, -2) === routeProvider));
68
- if (policy.providers.deny.some(match))
69
- return false;
70
- return policy.providers.allow.some(match);
71
- }
72
- /**
73
- * The swap gate: decide whether a recommendation may be applied without
74
- * human approval. Always returns the reasons for the decision so the
75
- * recommendation can explain itself.
76
- */
77
- export function gate(policy, input) {
78
- const reasons = [];
79
- if (!input.providerOk) {
80
- reasons.push("provider not allowed by policy");
81
- return { autoApply: false, reasons };
82
- }
83
- if (input.baselineMeasured === false) {
84
- reasons.push("current model has no measured baseline");
85
- return { autoApply: false, reasons };
86
- }
87
- if (input.sessionBound === false) {
88
- reasons.push("recommendation is not bound to a pi session");
89
- return { autoApply: false, reasons };
90
- }
91
- if (input.priceKnown === false) {
92
- reasons.push("candidate or current-model price is unknown");
93
- return { autoApply: false, reasons };
94
- }
95
- if (policy.mode !== "auto" || !policy.auto_apply.enabled) {
96
- reasons.push(`policy mode is "${policy.mode}" (auto_apply ${policy.auto_apply.enabled ? "enabled" : "disabled"})`);
97
- return { autoApply: false, reasons };
98
- }
99
- let ok = true;
100
- if (input.scoreGain < policy.auto_apply.min_score_gain) {
101
- reasons.push(`score gain ${input.scoreGain.toFixed(3)} < min ${policy.auto_apply.min_score_gain}`);
102
- ok = false;
103
- }
104
- if (input.priceRatio > policy.auto_apply.max_price_ratio) {
105
- reasons.push(`price ratio ${input.priceRatio.toFixed(2)}x > max ${policy.auto_apply.max_price_ratio}x`);
106
- ok = false;
107
- }
108
- if (policy.auto_apply.require_frontier && !input.onFrontier) {
109
- reasons.push("candidate is not on the pareto frontier");
110
- ok = false;
111
- }
112
- if (ok)
113
- reasons.unshift("all auto-apply thresholds met");
114
- return { autoApply: ok, reasons };
115
- }
package/dist/providers.js DELETED
@@ -1 +0,0 @@
1
- export {};
package/dist/recommend.js DELETED
@@ -1,76 +0,0 @@
1
- /** Build policy-gated recommendations from a task's frontier + the harness's current model. */
2
- import { createHash } from "node:crypto";
3
- import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
4
- import { gate, providerAllowed } from "./policy.js";
5
- import { pointKey, routeLabel } from "./routes.js";
6
- export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null, currentRoute = currentModel ? points.find((p) => p.model === currentModel)?.route ?? null : null) {
7
- const scored = points.filter((p) => p.score > 0);
8
- if (scored.length === 0)
9
- return null;
10
- const best = pickBest(scored);
11
- if (!best)
12
- return null;
13
- if (currentModel && best.model === currentModel &&
14
- (!currentRoute || pointKey(best) === `${currentRoute.provider}:${currentRoute.modelId}`))
15
- return null;
16
- const current = currentModel ? scored.find((p) => currentRoute
17
- ? pointKey(p) === `${currentRoute.provider}:${currentRoute.modelId}`
18
- : p.model === currentModel) : undefined;
19
- const scoreGain = current ? best.score - current.score : 0;
20
- // pickBest may choose a cheaper or faster model at equal quality.
21
- if (current && scoreGain < 0)
22
- return null;
23
- const priceKnown = best.priceKnown !== false && (!current || current.priceKnown !== false);
24
- const priceRatio = priceKnown && current && current.price > 0 ? best.price / current.price : 1;
25
- const frontierModels = paretoFrontier(scored).map(pointKey);
26
- const onFrontier = frontierModels.includes(pointKey(best));
27
- const fb = pickFallback(scored, best, policy.fallback.min_score);
28
- const decision = gate(policy, {
29
- scoreGain,
30
- priceRatio,
31
- priceKnown,
32
- onFrontier,
33
- providerOk: providerAllowed(policy, best.model, best.route?.provider),
34
- baselineMeasured: !!current,
35
- sessionBound: !!sessionFile,
36
- });
37
- const bestLabel = best.route ? routeLabel(best.route) : best.model;
38
- const reason = `${bestLabel} scores ${best.score.toFixed(2)} vs ` +
39
- (current ? `${current.score.toFixed(2)} for ${current.model}` : "no baseline measured") +
40
- ` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), ` +
41
- (best.priceKnown === false ? "with unknown price" : `at $${best.price.toFixed(2)}/M`) +
42
- (current && priceKnown ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
43
- (best.why ? `. Judge: ${best.why}` : "");
44
- return {
45
- schemaVersion: 1,
46
- id: createHash("sha1")
47
- .update(`${taskKey}:${pointKey(best)}:${Date.now()}`)
48
- .digest("hex")
49
- .slice(0, 10),
50
- ts: new Date().toISOString(),
51
- taskKey,
52
- taskLabel,
53
- sessionFile,
54
- currentModel,
55
- currentRoute,
56
- recommended: { model: best.model, route: best.route, reason },
57
- fallback: fb
58
- ? {
59
- model: fb.model,
60
- route: fb.route,
61
- reason: `score ${fb.score.toFixed(2)} ` +
62
- (fb.priceKnown === false ? "with unknown price" : `at $${fb.price.toFixed(2)}/M`) +
63
- (fb.latencyMs ? `, ~${Math.round(fb.latencyMs)}ms` : ""),
64
- }
65
- : null,
66
- evidence: {
67
- scoreGain: Math.round(scoreGain * 1000) / 1000,
68
- priceRatio: Math.round(priceRatio * 100) / 100,
69
- priceKnown,
70
- onFrontier,
71
- trials: scored.length,
72
- },
73
- policy: decision,
74
- status: "pending",
75
- };
76
- }
package/dist/routes.js DELETED
@@ -1,59 +0,0 @@
1
- /** Provider-neutral model-route normalization. */
2
- export function meritModelId(provider, modelId) {
3
- if (provider === "openrouter")
4
- return modelId;
5
- return modelId.startsWith(`${provider}/`) ? modelId : `${provider}/${modelId}`;
6
- }
7
- export function modelRoute(provider, modelId, details = {}) {
8
- return {
9
- provider,
10
- modelId,
11
- meritId: meritModelId(provider, modelId),
12
- input: details.input ?? ["text"],
13
- ...details,
14
- };
15
- }
16
- export function legacyOpenRouterRoute(meritId) {
17
- return modelRoute("openrouter", meritId);
18
- }
19
- export function routeKey(route) {
20
- return `${route.provider}:${route.modelId}`;
21
- }
22
- export function pointKey(point) {
23
- return point.route ? routeKey(point.route) : `openrouter:${point.model}`;
24
- }
25
- export function routeLabel(route) {
26
- return route.provider === "openrouter" ? route.meritId : `${route.meritId} via ${route.provider}`;
27
- }
28
- export function routePrice(route) {
29
- if (!route.cost || !Number.isFinite(route.cost.input) || !Number.isFinite(route.cost.output))
30
- return { price: 0, known: false };
31
- return { price: Math.round(((route.cost.input + route.cost.output) / 2 + Number.EPSILON) * 1e4) / 1e4,
32
- known: true };
33
- }
34
- export function catalogEntryFromRoute(route) {
35
- const priced = routePrice(route);
36
- return {
37
- id: route.meritId,
38
- name: routeLabel(route),
39
- ctx: route.contextWindow ?? 0,
40
- pp: route.cost ? route.cost.input / 1e6 : 0,
41
- pc: route.cost ? route.cost.output / 1e6 : 0,
42
- price: priced.price,
43
- priceKnown: priced.known,
44
- inputModalities: route.input,
45
- route,
46
- };
47
- }
48
- export function routeCatalog(routes) {
49
- const out = new Map();
50
- for (const route of routes)
51
- out.set(routeKey(route), catalogEntryFromRoute(route));
52
- return out;
53
- }
54
- /** Preserve the route while enriching an OpenRouter entry with its live catalog metadata. */
55
- export function enrichRouteEntry(entry, live) {
56
- if (!live)
57
- return entry;
58
- return { ...live, route: entry.route, id: entry.id, name: entry.name, priceKnown: true };
59
- }