openmerit 0.1.4 → 0.1.6-preview.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/README.md +121 -386
  3. package/dist/core/src/index.d.ts +101 -0
  4. package/dist/core/src/index.js +1649 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +32 -0
  8. package/dist/pi/src/index.js +794 -0
  9. package/dist/pi/src/scheduler.d.ts +11 -0
  10. package/dist/pi/src/scheduler.js +137 -0
  11. package/dist/pi/src/wakeup.d.ts +2 -0
  12. package/dist/pi/src/wakeup.js +108 -0
  13. package/dist/protocol/src/index.d.ts +484 -0
  14. package/dist/protocol/src/index.js +47 -0
  15. package/dist/protocol/src/schemas.d.ts +576 -0
  16. package/dist/protocol/src/schemas.js +280 -0
  17. package/dist/terminal/public/app.js +297 -0
  18. package/dist/terminal/public/brands/anthropic.png +0 -0
  19. package/dist/terminal/public/brands/baai.png +0 -0
  20. package/dist/terminal/public/brands/baseten.png +0 -0
  21. package/dist/terminal/public/brands/cerebras.png +0 -0
  22. package/dist/terminal/public/brands/cohere.png +0 -0
  23. package/dist/terminal/public/brands/deepseek.ico +0 -0
  24. package/dist/terminal/public/brands/google.png +0 -0
  25. package/dist/terminal/public/brands/groq.ico +0 -0
  26. package/dist/terminal/public/brands/lm-studio.png +0 -0
  27. package/dist/terminal/public/brands/meta.ico +0 -0
  28. package/dist/terminal/public/brands/mistral.png +0 -0
  29. package/dist/terminal/public/brands/nomic.png +0 -0
  30. package/dist/terminal/public/brands/ollama.png +0 -0
  31. package/dist/terminal/public/brands/openai.png +0 -0
  32. package/dist/terminal/public/brands/openrouter.png +0 -0
  33. package/dist/terminal/public/brands/qwen.png +0 -0
  34. package/dist/terminal/public/brands/vllm.ico +0 -0
  35. package/dist/terminal/public/brands/vllm.png +0 -0
  36. package/dist/terminal/public/favicon.svg +1 -0
  37. package/dist/terminal/public/flow.css +1 -0
  38. package/dist/terminal/public/flow.js +770 -0
  39. package/dist/terminal/public/index.html +21 -0
  40. package/dist/terminal/public/styles.css +779 -0
  41. package/dist/terminal/src/activity-merge.mjs +64 -0
  42. package/dist/terminal/src/browser.mjs +29 -0
  43. package/dist/terminal/src/cli.mjs +60 -0
  44. package/dist/terminal/src/collect.mjs +311 -0
  45. package/dist/terminal/src/discovery.mjs +93 -0
  46. package/dist/terminal/src/hardware.mjs +57 -0
  47. package/dist/terminal/src/project-activity.mjs +156 -0
  48. package/dist/terminal/src/sample.mjs +171 -0
  49. package/dist/terminal/src/server.mjs +56 -0
  50. package/dist/terminal/src/services.mjs +62 -0
  51. package/dist/terminal/src/topology.mjs +30 -0
  52. package/docs/adapter-guide.md +189 -0
  53. package/docs/architecture.md +59 -0
  54. package/docs/automation.md +74 -0
  55. package/docs/budgets.md +37 -0
  56. package/docs/commands.md +85 -0
  57. package/docs/demo-backfill.md +29 -0
  58. package/docs/demo-fieldkit.md +47 -0
  59. package/docs/demo-placement.md +30 -0
  60. package/docs/demo-spam.md +15 -0
  61. package/docs/demo-support.md +42 -0
  62. package/docs/demo.md +57 -0
  63. package/docs/first-trial.md +60 -0
  64. package/docs/getting-started.md +65 -0
  65. package/docs/index.md +40 -0
  66. package/docs/inference-terminal.md +439 -0
  67. package/docs/lifecycle.md +30 -0
  68. package/docs/memo.md +126 -0
  69. package/docs/metrics-and-evidence.md +48 -0
  70. package/docs/operations.md +40 -0
  71. package/docs/pareto-spec.md +76 -0
  72. package/docs/pi-extension.md +54 -0
  73. package/docs/roadmap.md +28 -0
  74. package/docs/security.md +37 -0
  75. package/docs/site-artwork-linocut.md +23 -0
  76. package/docs/site-artwork-miniature-diverse.md +28 -0
  77. package/docs/site-artwork-miniature.md +26 -0
  78. package/docs/site-demo.md +177 -0
  79. package/docs/site-design.md +94 -0
  80. package/docs/site-documentation.md +83 -0
  81. package/docs/site-dynamic-og.md +35 -0
  82. package/docs/site-faq-maintenance.md +115 -0
  83. package/docs/site-hero-resolution.md +60 -0
  84. package/docs/site-illustration-sequences.md +227 -0
  85. package/docs/site-inference-terminal.md +203 -0
  86. package/docs/site-memo.md +39 -0
  87. package/docs/site-og-image.md +38 -0
  88. package/docs/site-og-workshop.md +21 -0
  89. package/docs/site-section-artwork.md +56 -0
  90. package/docs/site-skill-review.md +57 -0
  91. package/docs/site-terminal-preview.md +85 -0
  92. package/docs/testing.md +118 -0
  93. package/docs/troubleshooting.md +55 -0
  94. package/docs/ux-reference.md +32 -0
  95. package/package.json +74 -42
  96. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  97. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  98. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  99. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  100. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  101. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  102. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  103. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  104. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  105. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  106. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  107. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  108. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  109. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  110. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  111. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  112. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  113. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  114. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  115. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  116. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  117. package/dist/benchmarks.js +0 -98
  118. package/dist/catalog.js +0 -61
  119. package/dist/cli.js +0 -188
  120. package/dist/daemon.js +0 -407
  121. package/dist/diagnostics.js +0 -227
  122. package/dist/frontier.js +0 -56
  123. package/dist/harness.js +0 -1
  124. package/dist/integrations.js +0 -19
  125. package/dist/invoice-eval.js +0 -33
  126. package/dist/invoice-score.js +0 -124
  127. package/dist/judge.js +0 -43
  128. package/dist/llm.js +0 -207
  129. package/dist/pi-config.js +0 -46
  130. package/dist/pi-trials.js +0 -373
  131. package/dist/policy.js +0 -185
  132. package/dist/providers.js +0 -1
  133. package/dist/recommend.js +0 -76
  134. package/dist/routes.js +0 -74
  135. package/dist/standalone.js +0 -224
  136. package/dist/store.js +0 -89
  137. package/dist/strategist.js +0 -68
  138. package/dist/task-input.js +0 -54
  139. package/dist/traces.js +0 -127
  140. package/dist/trials.js +0 -140
  141. package/dist/types.js +0 -2
  142. package/examples/invoice-prompt.txt +0 -19
  143. package/examples/task.example.json +0 -7
  144. package/extension/openmerit.ts +0 -947
  145. package/instructions/OPENMERIT.md +0 -63
  146. package/instructions/openmerit.policy.json +0 -37
  147. package/rules.md +0 -43
package/dist/pi-trials.js DELETED
@@ -1,373 +0,0 @@
1
- /** Run a task through pi for each model, using pi's JSON event stream as the measurement source. */
2
- import { spawn, spawnSync } from "node:child_process";
3
- import { cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
4
- import { tmpdir } from "node:os";
5
- import { dirname, join, relative } from "node:path";
6
- import { createHash } from "node:crypto";
7
- import { judge, judgeWithImages } from "./judge.js";
8
- import { knownInvoiceScore } from "./invoice-eval.js";
9
- import { paths, readJson } from "./store.js";
10
- import { sessionFiles, sessionImages, taskInputKey } from "./task-input.js";
11
- import { legacyOpenRouterRoute, meritModelId, modelRoute, routeKey } from "./routes.js";
12
- import { directChatClient } from "./llm.js";
13
- import { piProviderExtensionArgs } from "./pi-config.js";
14
- function answerText(content) {
15
- if (typeof content === "string")
16
- return content;
17
- if (!Array.isArray(content))
18
- return "";
19
- return content.filter((x) => !!x && typeof x === "object" && x.type === "text")
20
- .map((x) => x.text ?? "").join("\n");
21
- }
22
- /** Pi-backed harness adapter. Future harnesses can implement the same seam. */
23
- export class PiHarnessAdapter {
24
- providerExtensions;
25
- constructor(providerExtensions = []) {
26
- this.providerExtensions = providerExtensions;
27
- }
28
- runTask(model, input) {
29
- return executePiTask(model, input.task, input.cwd, input.images, input.files, this.providerExtensions);
30
- }
31
- }
32
- const defaultHarness = new PiHarnessAdapter();
33
- const KNOWN_TRIAL_TOOLS = new Set([
34
- "read", "grep", "find", "ls", "bash", "powershell", "edit", "write",
35
- ]);
36
- /**
37
- * Candidate runs have no tools by default. Users may explicitly opt into a
38
- * list, but a copied cwd is not an OS sandbox for absolute paths or network.
39
- */
40
- export function piTrialToolArgs(setting = process.env.OPENMERIT_PI_TRIAL_TOOLS) {
41
- if (!setting?.trim())
42
- return ["--no-tools"];
43
- const requested = [...new Set(setting.split(",").map((tool) => tool.trim()).filter(Boolean))];
44
- if (requested.length === 1 && requested[0] === "none")
45
- return ["--no-tools"];
46
- const unknown = requested.filter((tool) => !KNOWN_TRIAL_TOOLS.has(tool));
47
- if (requested.length === 0 || unknown.length > 0) {
48
- throw new Error(`invalid OPENMERIT_PI_TRIAL_TOOLS${unknown.length ? `: ${unknown.join(", ")}` : ""}`);
49
- }
50
- return ["--tools", requested.join(",")];
51
- }
52
- function fileSnapshot(root) {
53
- const out = new Map();
54
- function walk(dir) {
55
- for (const name of readdirSync(dir)) {
56
- if (name === ".git" || name === "node_modules" || name === ".openmerit")
57
- continue;
58
- const file = join(dir, name);
59
- const rel = relative(root, file);
60
- const st = statSync(file);
61
- if (st.isDirectory())
62
- walk(file);
63
- else if (st.isFile() && st.size < 20_000_000) {
64
- out.set(rel, createHash("sha1").update(readFileSync(file)).digest("hex"));
65
- }
66
- }
67
- }
68
- walk(root);
69
- return out;
70
- }
71
- function isolatedWorkspace(cwd) {
72
- const root = mkdtempSync(join(tmpdir(), "openmerit-pi-workspace-"));
73
- cpSync(cwd, root, { recursive: true, filter: (src) => {
74
- const rel = relative(cwd, src);
75
- return !rel.split("/").some((part) => part === ".git" || part === "node_modules" || part === ".openmerit");
76
- } });
77
- return root;
78
- }
79
- /** Use model A's completed result from the active pi session when it matches this task. */
80
- export function recordedActiveTask(task, model, images = [], sessionFile, sessionBytes) {
81
- const state = sessionFile ? null : readJson(paths.harnessState(), {});
82
- const file = sessionFile ?? state?.sessionFile;
83
- if (!file || !existsSync(file))
84
- return null;
85
- let user = null;
86
- const assistants = [];
87
- let toolCalls = 0;
88
- let toolErrors = 0;
89
- for (const line of readFileSync(file).subarray(0, sessionBytes).toString("utf8").split("\n")) {
90
- if (!line.trim())
91
- continue;
92
- let entry;
93
- try {
94
- entry = JSON.parse(line);
95
- }
96
- catch {
97
- continue;
98
- }
99
- if (entry.type !== "message" || !entry.message)
100
- continue;
101
- if (entry.message.role === "user") {
102
- user = entry.message;
103
- assistants.length = 0;
104
- toolCalls = 0;
105
- toolErrors = 0;
106
- }
107
- else if (user && entry.message.role === "assistant") {
108
- assistants.push(entry.message);
109
- if (Array.isArray(entry.message.content))
110
- toolCalls += entry.message.content.filter((part) => part?.type === "toolCall").length;
111
- }
112
- else if (user && entry.message.role === "toolResult" &&
113
- entry.message.isError)
114
- toolErrors++;
115
- }
116
- const userImages = user ? sessionImages(user.content) : null;
117
- const files = sessionFiles(task);
118
- if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
119
- taskInputKey(task, images, files))
120
- return null;
121
- const matching = assistants.filter((m) => typeof model === "string"
122
- ? meritModelId(m.provider ?? "openrouter", m.model ?? "") === model
123
- : m.provider === model.provider && m.model === model.modelId);
124
- if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
125
- return null;
126
- const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
127
- if (!answer.trim())
128
- return null;
129
- const first = user.timestamp ?? 0;
130
- const last = matching.at(-1)?.timestamp ?? first;
131
- return {
132
- answer,
133
- costUsd: matching.reduce((n, m) => n + (m.usage?.cost?.total ?? 0), 0),
134
- latencyMs: Math.max(0, last - first),
135
- errors: 0,
136
- toolCalls,
137
- toolErrors,
138
- changedFiles: 0,
139
- };
140
- }
141
- /** Read the exact latest completed text task from the pi extension's session marker. */
142
- export function settledActiveTask(snapshot) {
143
- const st = snapshot ?? readJson(paths.harnessState(), {});
144
- if (!st.currentModel || !st.sessionFile || !st.settledTaskKey || !st.settledAt ||
145
- !existsSync(st.sessionFile))
146
- return null;
147
- let userContent = null;
148
- let usedTools = false;
149
- for (const line of readFileSync(st.sessionFile).subarray(0, st.sessionBytes).toString("utf8").split("\n")) {
150
- if (!line.trim())
151
- continue;
152
- let entry;
153
- try {
154
- entry = JSON.parse(line);
155
- }
156
- catch {
157
- continue;
158
- }
159
- if (entry.type !== "message" || !entry.message)
160
- continue;
161
- if (entry.message.role === "user") {
162
- userContent = entry.message.content;
163
- usedTools = false;
164
- }
165
- else if (userContent && entry.message.role === "toolResult")
166
- usedTools = true;
167
- else if (userContent && entry.message.role === "assistant" &&
168
- Array.isArray(entry.message.content) && entry.message.content.some((b) => b?.type === "toolCall"))
169
- usedTools = true;
170
- }
171
- if (!userContent)
172
- return null;
173
- const images = sessionImages(userContent);
174
- if (!images)
175
- return null;
176
- const task = answerText(userContent);
177
- const files = sessionFiles(task);
178
- if (!task.trim() || taskInputKey(task, images, files) !== st.settledTaskKey)
179
- return null;
180
- const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
181
- const run = recordedActiveTask(task, route, images, st.sessionFile, st.sessionBytes);
182
- if (!run)
183
- return null;
184
- const routes = st.routes?.length ? st.routes : [route];
185
- return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
186
- settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
187
- route, routes };
188
- }
189
- export function piExecutable() {
190
- return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
191
- }
192
- function compactNumber(value) {
193
- const match = value.match(/^(\d+(?:\.\d+)?)([KMG])?$/i);
194
- if (!match)
195
- return undefined;
196
- const scale = { K: 1_000, M: 1_000_000, G: 1_000_000_000 }[(match[2] ?? "").toUpperCase()] ?? 1;
197
- return Math.round(Number(match[1]) * scale);
198
- }
199
- /** Read provider/model routes from pi when no live extension snapshot is available. */
200
- export function availablePiRoutes(snapshot, providerExtensions = []) {
201
- if (snapshot?.length)
202
- return snapshot;
203
- const result = spawnSync(piExecutable(), ["--offline", "--no-extensions",
204
- ...piProviderExtensionArgs(providerExtensions), "--list-models"], { encoding: "utf8" });
205
- if (result.error || result.status !== 0)
206
- throw new Error("could not list models from pi");
207
- const routes = [];
208
- for (const line of result.stdout.split("\n").slice(1)) {
209
- const [provider, id, context, maxOut, , images] = line.trim().split(/\s+/);
210
- if (!provider || !id || id.startsWith("~"))
211
- continue;
212
- routes.push(modelRoute(provider, id, {
213
- input: images === "yes" ? ["text", "image"] : ["text"],
214
- contextWindow: compactNumber(context),
215
- maxTokens: compactNumber(maxOut),
216
- }));
217
- }
218
- return routes;
219
- }
220
- /** Backward-compatible logical-id view used by the standalone OpenRouter trial command. */
221
- export function availablePiModels() {
222
- return new Set(availablePiRoutes().map((route) => route.meritId));
223
- }
224
- /** Each invocation is a fresh pi run, so model B does not inherit model A's answer. */
225
- export async function executePiTask(model, task, cwd = process.cwd(), images = [], files = [], providerExtensions = []) {
226
- const route = typeof model === "string" ? legacyOpenRouterRoute(model) : model;
227
- const imageDir = images.length ? mkdtempSync(join(tmpdir(), "openmerit-pi-image-")) : null;
228
- const suffix = {
229
- "image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif",
230
- };
231
- const imageArgs = images.map((image, i) => {
232
- if (!imageDir || !suffix[image.mimeType])
233
- throw new Error(`unsupported pi image: ${image.mimeType}`);
234
- const file = join(imageDir, `image-${i}.${suffix[image.mimeType]}`);
235
- writeFileSync(file, Buffer.from(image.data, "base64"));
236
- return `@${file}`;
237
- });
238
- const workspace = isolatedWorkspace(cwd);
239
- try {
240
- const attachmentDir = join(workspace, ".openmerit-attachments");
241
- mkdirSync(attachmentDir, { recursive: true });
242
- const attachmentPaths = new Map();
243
- const fileArgs = files.map((file, i) => {
244
- const safeName = `${String(i).padStart(3, "0")}-${file.name.replace(/[^A-Za-z0-9._-]/g, "_")}`;
245
- const target = join(attachmentDir, safeName);
246
- cpSync(file.path, target);
247
- attachmentPaths.set(file.path, target);
248
- return `@${target}`;
249
- });
250
- const candidateTask = [...attachmentPaths.entries()].reduce((text, [source, target]) => text.split(source).join(target), task);
251
- const before = fileSnapshot(workspace);
252
- const traceFile = paths.trialTrace(`${routeKey(route)}:${task}:${Date.now()}`);
253
- const traceLines = [];
254
- const child = spawn(piExecutable(), [
255
- "--provider", route.provider, "--model", route.modelId, "--mode", "json",
256
- "--offline", "--no-extensions", ...piProviderExtensionArgs(providerExtensions),
257
- "--approve", ...piTrialToolArgs(),
258
- "--print", ...imageArgs, ...fileArgs, candidateTask,
259
- ], { cwd: workspace, stdio: ["ignore", "pipe", "pipe"] });
260
- let buffer = "";
261
- let stderr = "";
262
- let answer = "";
263
- let costUsd = 0;
264
- let errors = 0;
265
- let sessionId;
266
- let toolCalls = 0;
267
- let toolErrors = 0;
268
- const started = Date.now();
269
- function consume(line) {
270
- if (!line.trim())
271
- return;
272
- traceLines.push(line);
273
- let event;
274
- try {
275
- event = JSON.parse(line);
276
- }
277
- catch {
278
- return;
279
- }
280
- if (event.type === "session" && event.id)
281
- sessionId = event.id;
282
- const content = event.message?.content;
283
- if (event.message?.role === "assistant" && Array.isArray(content))
284
- toolCalls += content.filter((part) => part?.type === "toolCall").length;
285
- if (event.message?.role === "toolResult" && event.message.isError)
286
- toolErrors++;
287
- if (event.type !== "message_end" || event.message?.role !== "assistant")
288
- return;
289
- answer = answerText(event.message.content) || answer;
290
- costUsd += event.message.usage?.cost?.total ?? 0;
291
- if (event.message.stopReason === "error")
292
- errors++;
293
- }
294
- child.stdout.on("data", (chunk) => {
295
- buffer += chunk.toString("utf8");
296
- let n;
297
- while ((n = buffer.indexOf("\n")) >= 0) {
298
- consume(buffer.slice(0, n));
299
- buffer = buffer.slice(n + 1);
300
- }
301
- });
302
- child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
303
- const exitCode = await new Promise((resolve, reject) => {
304
- child.on("error", reject);
305
- child.on("close", (code) => resolve(code ?? 1));
306
- });
307
- consume(buffer);
308
- mkdirSync(dirname(traceFile), { recursive: true });
309
- writeFileSync(traceFile, traceLines.join("\n") + (traceLines.length ? "\n" : ""));
310
- const after = fileSnapshot(workspace);
311
- let changedFiles = 0;
312
- for (const [file, hash] of after)
313
- if (before.get(file) !== hash)
314
- changedFiles++;
315
- for (const file of before.keys())
316
- if (!after.has(file))
317
- changedFiles++;
318
- if (exitCode !== 0 || !answer.trim()) {
319
- throw new Error(`pi trial ${route.meritId} via ${route.provider} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
320
- }
321
- return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId,
322
- toolCalls, toolErrors, changedFiles, traceFile };
323
- }
324
- finally {
325
- if (imageDir)
326
- rmSync(imageDir, { recursive: true, force: true });
327
- rmSync(workspace, { recursive: true, force: true });
328
- }
329
- }
330
- /** Quality uses the same OpenMerit judge; cost and latency come from pi itself. */
331
- export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = [], files = [], harness = defaultHarness, client = directChatClient(key)) {
332
- try {
333
- const run = await harness.runTask(entry?.route ?? model, { task, cwd, images, files });
334
- return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images, client);
335
- }
336
- catch (e) {
337
- const error = String(e);
338
- return {
339
- point: { schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
340
- priceKnown: entry?.priceKnown ?? !!entry,
341
- ts: new Date().toISOString(), source: "pi_trial", toolCalls: 0, toolErrors: 1,
342
- changedFiles: 0, why: error.slice(0, 160) },
343
- costUsd: 0, error,
344
- };
345
- }
346
- }
347
- export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = [], client = directChatClient(key)) {
348
- let score = 0;
349
- let why = "judge unavailable";
350
- try {
351
- const pinned = images.length ? await knownInvoiceScore(task, images, run.answer) : null;
352
- const j = pinned ?? (images.length
353
- ? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images, client)
354
- : await judge(key, judgeModel, task, rubric, run.answer, client));
355
- score = Math.max(0, Math.min(1, j.score));
356
- why = j.why;
357
- }
358
- catch (e) {
359
- why = `judge error: ${String(e).slice(0, 120)}`;
360
- }
361
- return {
362
- point: {
363
- schemaVersion: 1, model, route: entry?.route, score, price: entry?.price ?? 0,
364
- priceKnown: entry?.priceKnown ?? !!entry,
365
- latencyMs: run.latencyMs, ts: new Date().toISOString(), source,
366
- why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}; ` +
367
- `tools=${run.toolCalls}, toolErrors=${run.toolErrors}, changedFiles=${run.changedFiles}`,
368
- toolCalls: run.toolCalls, toolErrors: run.toolErrors, changedFiles: run.changedFiles,
369
- },
370
- costUsd: run.costUsd,
371
- sessionId: run.sessionId,
372
- };
373
- }
package/dist/policy.js DELETED
@@ -1,185 +0,0 @@
1
- /** Policy loading, validation, and the swap gate. */
2
- import { existsSync, readFileSync } from "node:fs";
3
- export const DEFAULT_POLICY = {
4
- version: 2,
5
- mode: "recommend",
6
- auto_apply: {
7
- enabled: false,
8
- min_score_gain: 0.1,
9
- max_price_ratio: 1.5,
10
- require_frontier: true,
11
- },
12
- budgets: { max_usd_per_trial: 0.25, max_trials_per_day: 20, max_usd_per_day: 5.0 },
13
- providers: { allow: ["*"], deny: [] },
14
- watch: { catalog_interval_min: 360, traces_interval_sec: 5, trial_interval_min: 30, models_per_task: 3 },
15
- fallback: { auto_update: true, min_score: 0.6, apply_on_error: true },
16
- judge_model: null,
17
- strategist_model: null,
18
- max_usd_per_m: 20.0,
19
- route_overrides: {},
20
- pi: { provider_extensions: [] },
21
- };
22
- function num(v, fallback) {
23
- return typeof v === "number" && Number.isFinite(v) ? v : fallback;
24
- }
25
- function finiteAtLeast(value, min, label) {
26
- if (typeof value !== "number" || !Number.isFinite(value) || value < min)
27
- throw new Error(`${label} must be a finite number >= ${min}`);
28
- return value;
29
- }
30
- function positiveInteger(value, label) {
31
- const result = finiteAtLeast(value, 1, label);
32
- if (!Number.isInteger(result))
33
- throw new Error(`${label} must be an integer`);
34
- return result;
35
- }
36
- function routeOverrides(value) {
37
- if (value === undefined)
38
- return {};
39
- if (!value || typeof value !== "object" || Array.isArray(value))
40
- throw new Error("route_overrides must be an object keyed by provider:modelId");
41
- const result = {};
42
- for (const [key, raw] of Object.entries(value)) {
43
- const separator = key.indexOf(":");
44
- if (separator <= 0 || separator === key.length - 1)
45
- throw new Error(`route override ${key} must use provider:modelId`);
46
- if (!raw || typeof raw !== "object" || Array.isArray(raw))
47
- throw new Error(`route override ${key} must be an object`);
48
- const input = raw.input;
49
- if (input !== undefined && (!Array.isArray(input) || input.length === 0 ||
50
- input.some((item) => item !== "text" && item !== "image")))
51
- throw new Error(`route override ${key}.input must contain only text or image`);
52
- const cost = raw.cost;
53
- let normalizedCost;
54
- if (cost !== undefined) {
55
- if (!cost || typeof cost !== "object" || Array.isArray(cost))
56
- throw new Error(`route override ${key}.cost must be an object`);
57
- const c = cost;
58
- normalizedCost = {
59
- input: finiteAtLeast(c.input, 0, `route override ${key}.cost.input`),
60
- output: finiteAtLeast(c.output, 0, `route override ${key}.cost.output`),
61
- ...(c.cacheRead === undefined ? {} : {
62
- cacheRead: finiteAtLeast(c.cacheRead, 0, `route override ${key}.cost.cacheRead`),
63
- }),
64
- ...(c.cacheWrite === undefined ? {} : {
65
- cacheWrite: finiteAtLeast(c.cacheWrite, 0, `route override ${key}.cost.cacheWrite`),
66
- }),
67
- };
68
- }
69
- const contextWindow = raw.context_window;
70
- const maxTokens = raw.max_tokens;
71
- result[key] = {
72
- ...(normalizedCost ? { cost: normalizedCost } : {}),
73
- ...(contextWindow === undefined ? {} : {
74
- context_window: positiveInteger(contextWindow, `route override ${key}.context_window`),
75
- }),
76
- ...(maxTokens === undefined ? {} : {
77
- max_tokens: positiveInteger(maxTokens, `route override ${key}.max_tokens`),
78
- }),
79
- ...(input === undefined ? {} : { input: [...new Set(input)] }),
80
- };
81
- }
82
- return result;
83
- }
84
- function providerExtensions(value) {
85
- if (value === undefined)
86
- return [];
87
- if (!Array.isArray(value) || value.some((item) => typeof item !== "string" || !item.trim()))
88
- throw new Error("pi.provider_extensions must be an array of non-empty file paths");
89
- return [...new Set(value.map((item) => String(item).trim()))];
90
- }
91
- /** Load a policy file, filling defaults for missing keys. Throws on invalid JSON. */
92
- export function loadPolicy(path) {
93
- if (!existsSync(path))
94
- return structuredClone(DEFAULT_POLICY);
95
- const raw = JSON.parse(readFileSync(path, "utf8"));
96
- const d = DEFAULT_POLICY;
97
- return {
98
- version: num(raw.version, d.version),
99
- mode: raw.mode === "auto" ? "auto" : "recommend",
100
- auto_apply: {
101
- enabled: Boolean(raw.auto_apply?.enabled ?? d.auto_apply.enabled),
102
- min_score_gain: num(raw.auto_apply?.min_score_gain, d.auto_apply.min_score_gain),
103
- max_price_ratio: num(raw.auto_apply?.max_price_ratio, d.auto_apply.max_price_ratio),
104
- require_frontier: Boolean(raw.auto_apply?.require_frontier ?? d.auto_apply.require_frontier),
105
- },
106
- budgets: {
107
- max_usd_per_trial: num(raw.budgets?.max_usd_per_trial, d.budgets.max_usd_per_trial),
108
- max_trials_per_day: num(raw.budgets?.max_trials_per_day, d.budgets.max_trials_per_day),
109
- max_usd_per_day: num(raw.budgets?.max_usd_per_day, d.budgets.max_usd_per_day),
110
- },
111
- providers: {
112
- allow: Array.isArray(raw.providers?.allow) ? raw.providers.allow.map(String) : d.providers.allow,
113
- deny: Array.isArray(raw.providers?.deny) ? raw.providers.deny.map(String) : d.providers.deny,
114
- },
115
- watch: {
116
- catalog_interval_min: num(raw.watch?.catalog_interval_min, d.watch.catalog_interval_min),
117
- traces_interval_sec: num(raw.watch?.traces_interval_sec, d.watch.traces_interval_sec),
118
- trial_interval_min: num(raw.watch?.trial_interval_min, d.watch.trial_interval_min),
119
- models_per_task: num(raw.watch?.models_per_task, d.watch.models_per_task),
120
- },
121
- fallback: {
122
- auto_update: Boolean(raw.fallback?.auto_update ?? d.fallback.auto_update),
123
- min_score: num(raw.fallback?.min_score, d.fallback.min_score),
124
- apply_on_error: Boolean(raw.fallback?.apply_on_error ?? d.fallback.apply_on_error),
125
- },
126
- judge_model: raw.judge_model ?? null,
127
- strategist_model: raw.strategist_model ?? null,
128
- max_usd_per_m: num(raw.max_usd_per_m, d.max_usd_per_m),
129
- route_overrides: routeOverrides(raw.route_overrides),
130
- pi: { provider_extensions: providerExtensions(raw.pi?.provider_extensions) },
131
- };
132
- }
133
- /** Match either a logical vendor/model id or its concrete route provider. */
134
- export function providerAllowed(policy, modelId, routeProvider) {
135
- const vendor = modelId.split("/")[0];
136
- const match = (pat) => pat === "*" || pat === modelId || pat === routeProvider ||
137
- (pat.endsWith("/*") && (pat.slice(0, -2) === vendor || pat.slice(0, -2) === routeProvider));
138
- if (policy.providers.deny.some(match))
139
- return false;
140
- return policy.providers.allow.some(match);
141
- }
142
- /**
143
- * The swap gate: decide whether a recommendation may be applied without
144
- * human approval. Always returns the reasons for the decision so the
145
- * recommendation can explain itself.
146
- */
147
- export function gate(policy, input) {
148
- const reasons = [];
149
- if (!input.providerOk) {
150
- reasons.push("provider not allowed by policy");
151
- return { autoApply: false, reasons };
152
- }
153
- if (input.baselineMeasured === false) {
154
- reasons.push("current model has no measured baseline");
155
- return { autoApply: false, reasons };
156
- }
157
- if (input.sessionBound === false) {
158
- reasons.push("recommendation is not bound to a pi session");
159
- return { autoApply: false, reasons };
160
- }
161
- if (input.priceKnown === false) {
162
- reasons.push("candidate or current-model price is unknown");
163
- return { autoApply: false, reasons };
164
- }
165
- if (policy.mode !== "auto" || !policy.auto_apply.enabled) {
166
- reasons.push(`policy mode is "${policy.mode}" (auto_apply ${policy.auto_apply.enabled ? "enabled" : "disabled"})`);
167
- return { autoApply: false, reasons };
168
- }
169
- let ok = true;
170
- if (input.scoreGain < policy.auto_apply.min_score_gain) {
171
- reasons.push(`score gain ${input.scoreGain.toFixed(3)} < min ${policy.auto_apply.min_score_gain}`);
172
- ok = false;
173
- }
174
- if (input.priceRatio > policy.auto_apply.max_price_ratio) {
175
- reasons.push(`price ratio ${input.priceRatio.toFixed(2)}x > max ${policy.auto_apply.max_price_ratio}x`);
176
- ok = false;
177
- }
178
- if (policy.auto_apply.require_frontier && !input.onFrontier) {
179
- reasons.push("candidate is not on the pareto frontier");
180
- ok = false;
181
- }
182
- if (ok)
183
- reasons.unshift("all auto-apply thresholds met");
184
- return { autoApply: ok, reasons };
185
- }
package/dist/providers.js DELETED
@@ -1 +0,0 @@
1
- export {};
package/dist/recommend.js DELETED
@@ -1,76 +0,0 @@
1
- /** Build policy-gated recommendations from a task's frontier + the harness's current model. */
2
- import { createHash } from "node:crypto";
3
- import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
4
- import { gate, providerAllowed } from "./policy.js";
5
- import { pointKey, routeLabel } from "./routes.js";
6
- export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null, currentRoute = currentModel ? points.find((p) => p.model === currentModel)?.route ?? null : null) {
7
- const scored = points.filter((p) => p.score > 0);
8
- if (scored.length === 0)
9
- return null;
10
- const best = pickBest(scored);
11
- if (!best)
12
- return null;
13
- if (currentModel && best.model === currentModel &&
14
- (!currentRoute || pointKey(best) === `${currentRoute.provider}:${currentRoute.modelId}`))
15
- return null;
16
- const current = currentModel ? scored.find((p) => currentRoute
17
- ? pointKey(p) === `${currentRoute.provider}:${currentRoute.modelId}`
18
- : p.model === currentModel) : undefined;
19
- const scoreGain = current ? best.score - current.score : 0;
20
- // pickBest may choose a cheaper or faster model at equal quality.
21
- if (current && scoreGain < 0)
22
- return null;
23
- const priceKnown = best.priceKnown !== false && (!current || current.priceKnown !== false);
24
- const priceRatio = priceKnown && current && current.price > 0 ? best.price / current.price : 1;
25
- const frontierModels = paretoFrontier(scored).map(pointKey);
26
- const onFrontier = frontierModels.includes(pointKey(best));
27
- const fb = pickFallback(scored, best, policy.fallback.min_score);
28
- const decision = gate(policy, {
29
- scoreGain,
30
- priceRatio,
31
- priceKnown,
32
- onFrontier,
33
- providerOk: providerAllowed(policy, best.model, best.route?.provider),
34
- baselineMeasured: !!current,
35
- sessionBound: !!sessionFile,
36
- });
37
- const bestLabel = best.route ? routeLabel(best.route) : best.model;
38
- const reason = `${bestLabel} scores ${best.score.toFixed(2)} vs ` +
39
- (current ? `${current.score.toFixed(2)} for ${current.model}` : "no baseline measured") +
40
- ` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), ` +
41
- (best.priceKnown === false ? "with unknown price" : `at $${best.price.toFixed(2)}/M`) +
42
- (current && priceKnown ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
43
- (best.why ? `. Judge: ${best.why}` : "");
44
- return {
45
- schemaVersion: 1,
46
- id: createHash("sha1")
47
- .update(`${taskKey}:${pointKey(best)}:${Date.now()}`)
48
- .digest("hex")
49
- .slice(0, 10),
50
- ts: new Date().toISOString(),
51
- taskKey,
52
- taskLabel,
53
- sessionFile,
54
- currentModel,
55
- currentRoute,
56
- recommended: { model: best.model, route: best.route, reason },
57
- fallback: fb
58
- ? {
59
- model: fb.model,
60
- route: fb.route,
61
- reason: `score ${fb.score.toFixed(2)} ` +
62
- (fb.priceKnown === false ? "with unknown price" : `at $${fb.price.toFixed(2)}/M`) +
63
- (fb.latencyMs ? `, ~${Math.round(fb.latencyMs)}ms` : ""),
64
- }
65
- : null,
66
- evidence: {
67
- scoreGain: Math.round(scoreGain * 1000) / 1000,
68
- priceRatio: Math.round(priceRatio * 100) / 100,
69
- priceKnown,
70
- onFrontier,
71
- trials: scored.length,
72
- },
73
- policy: decision,
74
- status: "pending",
75
- };
76
- }