openmerit 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -42
- package/dist/catalog.js +11 -1
- package/dist/cli.js +62 -8
- package/dist/daemon.js +111 -51
- package/dist/frontier.js +13 -6
- package/dist/harness.js +1 -0
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +159 -22
- package/dist/pi-trials.js +203 -74
- package/dist/policy.js +8 -3
- package/dist/providers.js +1 -0
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/store.js +2 -0
- package/dist/strategist.js +18 -14
- package/dist/task-input.js +27 -3
- package/dist/traces.js +6 -2
- package/dist/trials.js +38 -8
- package/extension/openmerit.ts +116 -27
- package/instructions/OPENMERIT.md +8 -5
- package/package.json +5 -2
- package/rules.md +39 -0
package/dist/pi-trials.js
CHANGED
|
@@ -1,12 +1,15 @@
|
|
|
1
1
|
/** Run a task through pi for each model, using pi's JSON event stream as the measurement source. */
|
|
2
2
|
import { spawn, spawnSync } from "node:child_process";
|
|
3
|
-
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
3
|
+
import { cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
4
4
|
import { tmpdir } from "node:os";
|
|
5
|
-
import { join } from "node:path";
|
|
5
|
+
import { dirname, join, relative } from "node:path";
|
|
6
|
+
import { createHash } from "node:crypto";
|
|
6
7
|
import { judge, judgeWithImages } from "./judge.js";
|
|
7
8
|
import { knownInvoiceScore } from "./invoice-eval.js";
|
|
8
9
|
import { paths, readJson } from "./store.js";
|
|
9
|
-
import { sessionImages, taskInputKey } from "./task-input.js";
|
|
10
|
+
import { sessionFiles, sessionImages, taskInputKey } from "./task-input.js";
|
|
11
|
+
import { legacyOpenRouterRoute, meritModelId, modelRoute, routeKey } from "./routes.js";
|
|
12
|
+
import { directChatClient } from "./llm.js";
|
|
10
13
|
function answerText(content) {
|
|
11
14
|
if (typeof content === "string")
|
|
12
15
|
return content;
|
|
@@ -15,6 +18,59 @@ function answerText(content) {
|
|
|
15
18
|
return content.filter((x) => !!x && typeof x === "object" && x.type === "text")
|
|
16
19
|
.map((x) => x.text ?? "").join("\n");
|
|
17
20
|
}
|
|
21
|
+
/** Pi-backed harness adapter. Future harnesses can implement the same seam. */
|
|
22
|
+
export class PiHarnessAdapter {
|
|
23
|
+
runTask(model, input) {
|
|
24
|
+
return executePiTask(model, input.task, input.cwd, input.images, input.files);
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
const defaultHarness = new PiHarnessAdapter();
|
|
28
|
+
const KNOWN_TRIAL_TOOLS = new Set([
|
|
29
|
+
"read", "grep", "find", "ls", "bash", "powershell", "edit", "write",
|
|
30
|
+
]);
|
|
31
|
+
/**
|
|
32
|
+
* Candidate runs have no tools by default. Users may explicitly opt into a
|
|
33
|
+
* list, but a copied cwd is not an OS sandbox for absolute paths or network.
|
|
34
|
+
*/
|
|
35
|
+
export function piTrialToolArgs(setting = process.env.OPENMERIT_PI_TRIAL_TOOLS) {
|
|
36
|
+
if (!setting?.trim())
|
|
37
|
+
return ["--no-tools"];
|
|
38
|
+
const requested = [...new Set(setting.split(",").map((tool) => tool.trim()).filter(Boolean))];
|
|
39
|
+
if (requested.length === 1 && requested[0] === "none")
|
|
40
|
+
return ["--no-tools"];
|
|
41
|
+
const unknown = requested.filter((tool) => !KNOWN_TRIAL_TOOLS.has(tool));
|
|
42
|
+
if (requested.length === 0 || unknown.length > 0) {
|
|
43
|
+
throw new Error(`invalid OPENMERIT_PI_TRIAL_TOOLS${unknown.length ? `: ${unknown.join(", ")}` : ""}`);
|
|
44
|
+
}
|
|
45
|
+
return ["--tools", requested.join(",")];
|
|
46
|
+
}
|
|
47
|
+
function fileSnapshot(root) {
|
|
48
|
+
const out = new Map();
|
|
49
|
+
function walk(dir) {
|
|
50
|
+
for (const name of readdirSync(dir)) {
|
|
51
|
+
if (name === ".git" || name === "node_modules" || name === ".openmerit")
|
|
52
|
+
continue;
|
|
53
|
+
const file = join(dir, name);
|
|
54
|
+
const rel = relative(root, file);
|
|
55
|
+
const st = statSync(file);
|
|
56
|
+
if (st.isDirectory())
|
|
57
|
+
walk(file);
|
|
58
|
+
else if (st.isFile() && st.size < 20_000_000) {
|
|
59
|
+
out.set(rel, createHash("sha1").update(readFileSync(file)).digest("hex"));
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
walk(root);
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
function isolatedWorkspace(cwd) {
|
|
67
|
+
const root = mkdtempSync(join(tmpdir(), "openmerit-pi-workspace-"));
|
|
68
|
+
cpSync(cwd, root, { recursive: true, filter: (src) => {
|
|
69
|
+
const rel = relative(cwd, src);
|
|
70
|
+
return !rel.split("/").some((part) => part === ".git" || part === "node_modules" || part === ".openmerit");
|
|
71
|
+
} });
|
|
72
|
+
return root;
|
|
73
|
+
}
|
|
18
74
|
/** Use model A's completed result from the active pi session when it matches this task. */
|
|
19
75
|
export function recordedActiveTask(task, model, images = [], sessionFile, sessionBytes) {
|
|
20
76
|
const state = sessionFile ? null : readJson(paths.harnessState(), {});
|
|
@@ -23,6 +79,8 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
|
|
|
23
79
|
return null;
|
|
24
80
|
let user = null;
|
|
25
81
|
const assistants = [];
|
|
82
|
+
let toolCalls = 0;
|
|
83
|
+
let toolErrors = 0;
|
|
26
84
|
for (const line of readFileSync(file).subarray(0, sessionBytes).toString("utf8").split("\n")) {
|
|
27
85
|
if (!line.trim())
|
|
28
86
|
continue;
|
|
@@ -38,15 +96,24 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
|
|
|
38
96
|
if (entry.message.role === "user") {
|
|
39
97
|
user = entry.message;
|
|
40
98
|
assistants.length = 0;
|
|
99
|
+
toolCalls = 0;
|
|
100
|
+
toolErrors = 0;
|
|
41
101
|
}
|
|
42
|
-
else if (user && entry.message.role === "assistant")
|
|
102
|
+
else if (user && entry.message.role === "assistant") {
|
|
43
103
|
assistants.push(entry.message);
|
|
104
|
+
if (Array.isArray(entry.message.content))
|
|
105
|
+
toolCalls += entry.message.content.filter((part) => part?.type === "toolCall").length;
|
|
106
|
+
}
|
|
107
|
+
else if (user && entry.message.role === "toolResult" &&
|
|
108
|
+
entry.message.isError)
|
|
109
|
+
toolErrors++;
|
|
44
110
|
}
|
|
45
111
|
const userImages = user ? sessionImages(user.content) : null;
|
|
46
|
-
|
|
47
|
-
|
|
112
|
+
const files = sessionFiles(task);
|
|
113
|
+
if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
|
|
114
|
+
taskInputKey(task, images, files))
|
|
48
115
|
return null;
|
|
49
|
-
const matching = assistants.filter((m) => (m.provider
|
|
116
|
+
const matching = assistants.filter((m) => meritModelId(m.provider ?? "openrouter", m.model ?? "") === model);
|
|
50
117
|
if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
|
|
51
118
|
return null;
|
|
52
119
|
const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
|
|
@@ -59,6 +126,9 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
|
|
|
59
126
|
costUsd: matching.reduce((n, m) => n + (m.usage?.cost?.total ?? 0), 0),
|
|
60
127
|
latencyMs: Math.max(0, last - first),
|
|
61
128
|
errors: 0,
|
|
129
|
+
toolCalls,
|
|
130
|
+
toolErrors,
|
|
131
|
+
changedFiles: 0,
|
|
62
132
|
};
|
|
63
133
|
}
|
|
64
134
|
/** Read the exact latest completed text task from the pi extension's session marker. */
|
|
@@ -91,31 +161,50 @@ export function settledActiveTask(snapshot) {
|
|
|
91
161
|
Array.isArray(entry.message.content) && entry.message.content.some((b) => b?.type === "toolCall"))
|
|
92
162
|
usedTools = true;
|
|
93
163
|
}
|
|
94
|
-
if (!userContent
|
|
164
|
+
if (!userContent)
|
|
95
165
|
return null;
|
|
96
166
|
const images = sessionImages(userContent);
|
|
97
167
|
if (!images)
|
|
98
168
|
return null;
|
|
99
169
|
const task = answerText(userContent);
|
|
100
|
-
|
|
170
|
+
const files = sessionFiles(task);
|
|
171
|
+
if (!task.trim() || taskInputKey(task, images, files) !== st.settledTaskKey)
|
|
101
172
|
return null;
|
|
102
173
|
const run = recordedActiveTask(task, st.currentModel, images, st.sessionFile, st.sessionBytes);
|
|
103
174
|
if (!run)
|
|
104
175
|
return null;
|
|
176
|
+
const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
|
|
177
|
+
const routes = st.routes?.length ? st.routes : [route];
|
|
105
178
|
return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
|
|
106
|
-
settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd()
|
|
179
|
+
settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
|
|
180
|
+
route, routes };
|
|
107
181
|
}
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
182
|
+
export function piExecutable() {
|
|
183
|
+
return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
|
|
184
|
+
}
|
|
185
|
+
/** Read provider/model routes from pi when no live extension snapshot is available. */
|
|
186
|
+
export function availablePiRoutes(snapshot) {
|
|
187
|
+
if (snapshot?.length)
|
|
188
|
+
return snapshot;
|
|
189
|
+
const result = spawnSync(piExecutable(), ["--offline", "--list-models"], { encoding: "utf8" });
|
|
111
190
|
if (result.error || result.status !== 0)
|
|
112
|
-
throw new Error("could not list
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
191
|
+
throw new Error("could not list models from pi");
|
|
192
|
+
const routes = [];
|
|
193
|
+
for (const line of result.stdout.split("\n").slice(1)) {
|
|
194
|
+
const [provider, id, , , , images] = line.trim().split(/\s+/);
|
|
195
|
+
if (!provider || !id || id.startsWith("~"))
|
|
196
|
+
continue;
|
|
197
|
+
routes.push(modelRoute(provider, id, { input: images === "yes" ? ["text", "image"] : ["text"] }));
|
|
198
|
+
}
|
|
199
|
+
return routes;
|
|
200
|
+
}
|
|
201
|
+
/** Backward-compatible logical-id view used by the standalone OpenRouter trial command. */
|
|
202
|
+
export function availablePiModels() {
|
|
203
|
+
return new Set(availablePiRoutes().map((route) => route.meritId));
|
|
116
204
|
}
|
|
117
205
|
/** Each invocation is a fresh pi run, so model B does not inherit model A's answer. */
|
|
118
|
-
export async function executePiTask(model, task, cwd = process.cwd(), images = []) {
|
|
206
|
+
export async function executePiTask(model, task, cwd = process.cwd(), images = [], files = []) {
|
|
207
|
+
const route = typeof model === "string" ? legacyOpenRouterRoute(model) : model;
|
|
119
208
|
const imageDir = images.length ? mkdtempSync(join(tmpdir(), "openmerit-pi-image-")) : null;
|
|
120
209
|
const suffix = {
|
|
121
210
|
"image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif",
|
|
@@ -127,85 +216,122 @@ export async function executePiTask(model, task, cwd = process.cwd(), images = [
|
|
|
127
216
|
writeFileSync(file, Buffer.from(image.data, "base64"));
|
|
128
217
|
return `@${file}`;
|
|
129
218
|
});
|
|
130
|
-
const
|
|
131
|
-
"--provider", "openrouter", "--model", model, "--mode", "json",
|
|
132
|
-
"--offline", "--no-extensions", "--no-tools", "--print", ...imageArgs, task,
|
|
133
|
-
], { cwd, stdio: ["ignore", "pipe", "pipe"] });
|
|
134
|
-
let buffer = "";
|
|
135
|
-
let stderr = "";
|
|
136
|
-
let answer = "";
|
|
137
|
-
let costUsd = 0;
|
|
138
|
-
let errors = 0;
|
|
139
|
-
let sessionId;
|
|
140
|
-
const started = Date.now();
|
|
141
|
-
function consume(line) {
|
|
142
|
-
if (!line.trim())
|
|
143
|
-
return;
|
|
144
|
-
let event;
|
|
145
|
-
try {
|
|
146
|
-
event = JSON.parse(line);
|
|
147
|
-
}
|
|
148
|
-
catch {
|
|
149
|
-
return;
|
|
150
|
-
}
|
|
151
|
-
if (event.type === "session" && event.id)
|
|
152
|
-
sessionId = event.id;
|
|
153
|
-
if (event.type !== "message_end" || event.message?.role !== "assistant")
|
|
154
|
-
return;
|
|
155
|
-
answer = answerText(event.message.content) || answer;
|
|
156
|
-
costUsd += event.message.usage?.cost?.total ?? 0;
|
|
157
|
-
if (event.message.stopReason === "error")
|
|
158
|
-
errors++;
|
|
159
|
-
}
|
|
160
|
-
child.stdout.on("data", (chunk) => {
|
|
161
|
-
buffer += chunk.toString("utf8");
|
|
162
|
-
let n;
|
|
163
|
-
while ((n = buffer.indexOf("\n")) >= 0) {
|
|
164
|
-
consume(buffer.slice(0, n));
|
|
165
|
-
buffer = buffer.slice(n + 1);
|
|
166
|
-
}
|
|
167
|
-
});
|
|
168
|
-
child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
|
|
169
|
-
let exitCode;
|
|
219
|
+
const workspace = isolatedWorkspace(cwd);
|
|
170
220
|
try {
|
|
171
|
-
|
|
221
|
+
const attachmentDir = join(workspace, ".openmerit-attachments");
|
|
222
|
+
mkdirSync(attachmentDir, { recursive: true });
|
|
223
|
+
const attachmentPaths = new Map();
|
|
224
|
+
const fileArgs = files.map((file, i) => {
|
|
225
|
+
const safeName = `${String(i).padStart(3, "0")}-${file.name.replace(/[^A-Za-z0-9._-]/g, "_")}`;
|
|
226
|
+
const target = join(attachmentDir, safeName);
|
|
227
|
+
cpSync(file.path, target);
|
|
228
|
+
attachmentPaths.set(file.path, target);
|
|
229
|
+
return `@${target}`;
|
|
230
|
+
});
|
|
231
|
+
const candidateTask = [...attachmentPaths.entries()].reduce((text, [source, target]) => text.split(source).join(target), task);
|
|
232
|
+
const before = fileSnapshot(workspace);
|
|
233
|
+
const traceFile = paths.trialTrace(`${routeKey(route)}:${task}:${Date.now()}`);
|
|
234
|
+
const traceLines = [];
|
|
235
|
+
const child = spawn(piExecutable(), [
|
|
236
|
+
"--provider", route.provider, "--model", route.modelId, "--mode", "json",
|
|
237
|
+
"--offline", "--no-extensions", "--approve", ...piTrialToolArgs(),
|
|
238
|
+
"--print", ...imageArgs, ...fileArgs, candidateTask,
|
|
239
|
+
], { cwd: workspace, stdio: ["ignore", "pipe", "pipe"] });
|
|
240
|
+
let buffer = "";
|
|
241
|
+
let stderr = "";
|
|
242
|
+
let answer = "";
|
|
243
|
+
let costUsd = 0;
|
|
244
|
+
let errors = 0;
|
|
245
|
+
let sessionId;
|
|
246
|
+
let toolCalls = 0;
|
|
247
|
+
let toolErrors = 0;
|
|
248
|
+
const started = Date.now();
|
|
249
|
+
function consume(line) {
|
|
250
|
+
if (!line.trim())
|
|
251
|
+
return;
|
|
252
|
+
traceLines.push(line);
|
|
253
|
+
let event;
|
|
254
|
+
try {
|
|
255
|
+
event = JSON.parse(line);
|
|
256
|
+
}
|
|
257
|
+
catch {
|
|
258
|
+
return;
|
|
259
|
+
}
|
|
260
|
+
if (event.type === "session" && event.id)
|
|
261
|
+
sessionId = event.id;
|
|
262
|
+
const content = event.message?.content;
|
|
263
|
+
if (event.message?.role === "assistant" && Array.isArray(content))
|
|
264
|
+
toolCalls += content.filter((part) => part?.type === "toolCall").length;
|
|
265
|
+
if (event.message?.role === "toolResult" && event.message.isError)
|
|
266
|
+
toolErrors++;
|
|
267
|
+
if (event.type !== "message_end" || event.message?.role !== "assistant")
|
|
268
|
+
return;
|
|
269
|
+
answer = answerText(event.message.content) || answer;
|
|
270
|
+
costUsd += event.message.usage?.cost?.total ?? 0;
|
|
271
|
+
if (event.message.stopReason === "error")
|
|
272
|
+
errors++;
|
|
273
|
+
}
|
|
274
|
+
child.stdout.on("data", (chunk) => {
|
|
275
|
+
buffer += chunk.toString("utf8");
|
|
276
|
+
let n;
|
|
277
|
+
while ((n = buffer.indexOf("\n")) >= 0) {
|
|
278
|
+
consume(buffer.slice(0, n));
|
|
279
|
+
buffer = buffer.slice(n + 1);
|
|
280
|
+
}
|
|
281
|
+
});
|
|
282
|
+
child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
|
|
283
|
+
const exitCode = await new Promise((resolve, reject) => {
|
|
172
284
|
child.on("error", reject);
|
|
173
285
|
child.on("close", (code) => resolve(code ?? 1));
|
|
174
286
|
});
|
|
287
|
+
consume(buffer);
|
|
288
|
+
mkdirSync(dirname(traceFile), { recursive: true });
|
|
289
|
+
writeFileSync(traceFile, traceLines.join("\n") + (traceLines.length ? "\n" : ""));
|
|
290
|
+
const after = fileSnapshot(workspace);
|
|
291
|
+
let changedFiles = 0;
|
|
292
|
+
for (const [file, hash] of after)
|
|
293
|
+
if (before.get(file) !== hash)
|
|
294
|
+
changedFiles++;
|
|
295
|
+
for (const file of before.keys())
|
|
296
|
+
if (!after.has(file))
|
|
297
|
+
changedFiles++;
|
|
298
|
+
if (exitCode !== 0 || !answer.trim()) {
|
|
299
|
+
throw new Error(`pi trial ${route.meritId} via ${route.provider} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
|
|
300
|
+
}
|
|
301
|
+
return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId,
|
|
302
|
+
toolCalls, toolErrors, changedFiles, traceFile };
|
|
175
303
|
}
|
|
176
304
|
finally {
|
|
177
305
|
if (imageDir)
|
|
178
306
|
rmSync(imageDir, { recursive: true, force: true });
|
|
307
|
+
rmSync(workspace, { recursive: true, force: true });
|
|
179
308
|
}
|
|
180
|
-
consume(buffer);
|
|
181
|
-
if (exitCode !== 0 || !answer.trim()) {
|
|
182
|
-
throw new Error(`pi trial ${model} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
|
|
183
|
-
}
|
|
184
|
-
return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId };
|
|
185
309
|
}
|
|
186
310
|
/** Quality uses the same OpenMerit judge; cost and latency come from pi itself. */
|
|
187
|
-
export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = []) {
|
|
311
|
+
export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = [], files = [], harness = defaultHarness, client = directChatClient(key)) {
|
|
188
312
|
try {
|
|
189
|
-
const run = await
|
|
190
|
-
return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images);
|
|
313
|
+
const run = await harness.runTask(entry?.route ?? model, { task, cwd, images, files });
|
|
314
|
+
return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images, client);
|
|
191
315
|
}
|
|
192
316
|
catch (e) {
|
|
193
317
|
const error = String(e);
|
|
194
318
|
return {
|
|
195
|
-
point: { model, score: 0, price: entry?.price ?? 0,
|
|
196
|
-
|
|
319
|
+
point: { schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
320
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
321
|
+
ts: new Date().toISOString(), source: "pi_trial", toolCalls: 0, toolErrors: 1,
|
|
322
|
+
changedFiles: 0, why: error.slice(0, 160) },
|
|
197
323
|
costUsd: 0, error,
|
|
198
324
|
};
|
|
199
325
|
}
|
|
200
326
|
}
|
|
201
|
-
export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = []) {
|
|
327
|
+
export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = [], client = directChatClient(key)) {
|
|
202
328
|
let score = 0;
|
|
203
329
|
let why = "judge unavailable";
|
|
204
330
|
try {
|
|
205
331
|
const pinned = images.length ? await knownInvoiceScore(task, images, run.answer) : null;
|
|
206
332
|
const j = pinned ?? (images.length
|
|
207
|
-
? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images)
|
|
208
|
-
: await judge(key, judgeModel, task, rubric, run.answer));
|
|
333
|
+
? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images, client)
|
|
334
|
+
: await judge(key, judgeModel, task, rubric, run.answer, client));
|
|
209
335
|
score = Math.max(0, Math.min(1, j.score));
|
|
210
336
|
why = j.why;
|
|
211
337
|
}
|
|
@@ -214,9 +340,12 @@ export async function scorePiRun(key, judgeModel, task, rubric, model, entry, ru
|
|
|
214
340
|
}
|
|
215
341
|
return {
|
|
216
342
|
point: {
|
|
217
|
-
model, score, price: entry?.price ?? 0,
|
|
343
|
+
schemaVersion: 1, model, route: entry?.route, score, price: entry?.price ?? 0,
|
|
344
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
218
345
|
latencyMs: run.latencyMs, ts: new Date().toISOString(), source,
|
|
219
|
-
why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}
|
|
346
|
+
why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}; ` +
|
|
347
|
+
`tools=${run.toolCalls}, toolErrors=${run.toolErrors}, changedFiles=${run.changedFiles}`,
|
|
348
|
+
toolCalls: run.toolCalls, toolErrors: run.toolErrors, changedFiles: run.changedFiles,
|
|
220
349
|
},
|
|
221
350
|
costUsd: run.costUsd,
|
|
222
351
|
sessionId: run.sessionId,
|
package/dist/policy.js
CHANGED
|
@@ -60,10 +60,11 @@ export function loadPolicy(path) {
|
|
|
60
60
|
max_usd_per_m: num(raw.max_usd_per_m, d.max_usd_per_m),
|
|
61
61
|
};
|
|
62
62
|
}
|
|
63
|
-
/**
|
|
64
|
-
export function providerAllowed(policy, modelId) {
|
|
63
|
+
/** Match either a logical vendor/model id or its concrete route provider. */
|
|
64
|
+
export function providerAllowed(policy, modelId, routeProvider) {
|
|
65
65
|
const vendor = modelId.split("/")[0];
|
|
66
|
-
const match = (pat) => pat === "*" || pat === modelId ||
|
|
66
|
+
const match = (pat) => pat === "*" || pat === modelId || pat === routeProvider ||
|
|
67
|
+
(pat.endsWith("/*") && (pat.slice(0, -2) === vendor || pat.slice(0, -2) === routeProvider));
|
|
67
68
|
if (policy.providers.deny.some(match))
|
|
68
69
|
return false;
|
|
69
70
|
return policy.providers.allow.some(match);
|
|
@@ -87,6 +88,10 @@ export function gate(policy, input) {
|
|
|
87
88
|
reasons.push("recommendation is not bound to a pi session");
|
|
88
89
|
return { autoApply: false, reasons };
|
|
89
90
|
}
|
|
91
|
+
if (input.priceKnown === false) {
|
|
92
|
+
reasons.push("candidate or current-model price is unknown");
|
|
93
|
+
return { autoApply: false, reasons };
|
|
94
|
+
}
|
|
90
95
|
if (policy.mode !== "auto" || !policy.auto_apply.enabled) {
|
|
91
96
|
reasons.push(`policy mode is "${policy.mode}" (auto_apply ${policy.auto_apply.enabled ? "enabled" : "disabled"})`);
|
|
92
97
|
return { autoApply: false, reasons };
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/recommend.js
CHANGED
|
@@ -2,40 +2,49 @@
|
|
|
2
2
|
import { createHash } from "node:crypto";
|
|
3
3
|
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
4
4
|
import { gate, providerAllowed } from "./policy.js";
|
|
5
|
-
|
|
5
|
+
import { pointKey, routeLabel } from "./routes.js";
|
|
6
|
+
export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null, currentRoute = currentModel ? points.find((p) => p.model === currentModel)?.route ?? null : null) {
|
|
6
7
|
const scored = points.filter((p) => p.score > 0);
|
|
7
8
|
if (scored.length === 0)
|
|
8
9
|
return null;
|
|
9
10
|
const best = pickBest(scored);
|
|
10
11
|
if (!best)
|
|
11
12
|
return null;
|
|
12
|
-
if (currentModel && best.model === currentModel
|
|
13
|
-
|
|
14
|
-
|
|
13
|
+
if (currentModel && best.model === currentModel &&
|
|
14
|
+
(!currentRoute || pointKey(best) === `${currentRoute.provider}:${currentRoute.modelId}`))
|
|
15
|
+
return null;
|
|
16
|
+
const current = currentModel ? scored.find((p) => currentRoute
|
|
17
|
+
? pointKey(p) === `${currentRoute.provider}:${currentRoute.modelId}`
|
|
18
|
+
: p.model === currentModel) : undefined;
|
|
15
19
|
const scoreGain = current ? best.score - current.score : 0;
|
|
16
20
|
// pickBest may choose a cheaper or faster model at equal quality.
|
|
17
21
|
if (current && scoreGain < 0)
|
|
18
22
|
return null;
|
|
19
|
-
const
|
|
20
|
-
const
|
|
21
|
-
const
|
|
23
|
+
const priceKnown = best.priceKnown !== false && (!current || current.priceKnown !== false);
|
|
24
|
+
const priceRatio = priceKnown && current && current.price > 0 ? best.price / current.price : 1;
|
|
25
|
+
const frontierModels = paretoFrontier(scored).map(pointKey);
|
|
26
|
+
const onFrontier = frontierModels.includes(pointKey(best));
|
|
22
27
|
const fb = pickFallback(scored, best, policy.fallback.min_score);
|
|
23
28
|
const decision = gate(policy, {
|
|
24
29
|
scoreGain,
|
|
25
30
|
priceRatio,
|
|
31
|
+
priceKnown,
|
|
26
32
|
onFrontier,
|
|
27
|
-
providerOk: providerAllowed(policy, best.model),
|
|
33
|
+
providerOk: providerAllowed(policy, best.model, best.route?.provider),
|
|
28
34
|
baselineMeasured: !!current,
|
|
29
35
|
sessionBound: !!sessionFile,
|
|
30
36
|
});
|
|
31
|
-
const
|
|
37
|
+
const bestLabel = best.route ? routeLabel(best.route) : best.model;
|
|
38
|
+
const reason = `${bestLabel} scores ${best.score.toFixed(2)} vs ` +
|
|
32
39
|
(current ? `${current.score.toFixed(2)} for ${current.model}` : "no baseline measured") +
|
|
33
|
-
` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}),
|
|
34
|
-
(
|
|
40
|
+
` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), ` +
|
|
41
|
+
(best.priceKnown === false ? "with unknown price" : `at $${best.price.toFixed(2)}/M`) +
|
|
42
|
+
(current && priceKnown ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
|
|
35
43
|
(best.why ? `. Judge: ${best.why}` : "");
|
|
36
44
|
return {
|
|
45
|
+
schemaVersion: 1,
|
|
37
46
|
id: createHash("sha1")
|
|
38
|
-
.update(`${taskKey}:${best
|
|
47
|
+
.update(`${taskKey}:${pointKey(best)}:${Date.now()}`)
|
|
39
48
|
.digest("hex")
|
|
40
49
|
.slice(0, 10),
|
|
41
50
|
ts: new Date().toISOString(),
|
|
@@ -43,17 +52,21 @@ export function buildRecommendation(taskKey, taskLabel, currentModel, points, po
|
|
|
43
52
|
taskLabel,
|
|
44
53
|
sessionFile,
|
|
45
54
|
currentModel,
|
|
46
|
-
|
|
55
|
+
currentRoute,
|
|
56
|
+
recommended: { model: best.model, route: best.route, reason },
|
|
47
57
|
fallback: fb
|
|
48
58
|
? {
|
|
49
59
|
model: fb.model,
|
|
50
|
-
|
|
60
|
+
route: fb.route,
|
|
61
|
+
reason: `score ${fb.score.toFixed(2)} ` +
|
|
62
|
+
(fb.priceKnown === false ? "with unknown price" : `at $${fb.price.toFixed(2)}/M`) +
|
|
51
63
|
(fb.latencyMs ? `, ~${Math.round(fb.latencyMs)}ms` : ""),
|
|
52
64
|
}
|
|
53
65
|
: null,
|
|
54
66
|
evidence: {
|
|
55
67
|
scoreGain: Math.round(scoreGain * 1000) / 1000,
|
|
56
68
|
priceRatio: Math.round(priceRatio * 100) / 100,
|
|
69
|
+
priceKnown,
|
|
57
70
|
onFrontier,
|
|
58
71
|
trials: scored.length,
|
|
59
72
|
},
|
package/dist/routes.js
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/** Provider-neutral model-route normalization. */
|
|
2
|
+
export function meritModelId(provider, modelId) {
|
|
3
|
+
if (provider === "openrouter")
|
|
4
|
+
return modelId;
|
|
5
|
+
return modelId.startsWith(`${provider}/`) ? modelId : `${provider}/${modelId}`;
|
|
6
|
+
}
|
|
7
|
+
export function modelRoute(provider, modelId, details = {}) {
|
|
8
|
+
return {
|
|
9
|
+
provider,
|
|
10
|
+
modelId,
|
|
11
|
+
meritId: meritModelId(provider, modelId),
|
|
12
|
+
input: details.input ?? ["text"],
|
|
13
|
+
...details,
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
export function legacyOpenRouterRoute(meritId) {
|
|
17
|
+
return modelRoute("openrouter", meritId);
|
|
18
|
+
}
|
|
19
|
+
export function routeKey(route) {
|
|
20
|
+
return `${route.provider}:${route.modelId}`;
|
|
21
|
+
}
|
|
22
|
+
export function pointKey(point) {
|
|
23
|
+
return point.route ? routeKey(point.route) : `openrouter:${point.model}`;
|
|
24
|
+
}
|
|
25
|
+
export function routeLabel(route) {
|
|
26
|
+
return route.provider === "openrouter" ? route.meritId : `${route.meritId} via ${route.provider}`;
|
|
27
|
+
}
|
|
28
|
+
export function routePrice(route) {
|
|
29
|
+
if (!route.cost || !Number.isFinite(route.cost.input) || !Number.isFinite(route.cost.output))
|
|
30
|
+
return { price: 0, known: false };
|
|
31
|
+
return { price: Math.round(((route.cost.input + route.cost.output) / 2 + Number.EPSILON) * 1e4) / 1e4,
|
|
32
|
+
known: true };
|
|
33
|
+
}
|
|
34
|
+
export function catalogEntryFromRoute(route) {
|
|
35
|
+
const priced = routePrice(route);
|
|
36
|
+
return {
|
|
37
|
+
id: route.meritId,
|
|
38
|
+
name: routeLabel(route),
|
|
39
|
+
ctx: route.contextWindow ?? 0,
|
|
40
|
+
pp: route.cost ? route.cost.input / 1e6 : 0,
|
|
41
|
+
pc: route.cost ? route.cost.output / 1e6 : 0,
|
|
42
|
+
price: priced.price,
|
|
43
|
+
priceKnown: priced.known,
|
|
44
|
+
inputModalities: route.input,
|
|
45
|
+
route,
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
export function routeCatalog(routes) {
|
|
49
|
+
const out = new Map();
|
|
50
|
+
for (const route of routes)
|
|
51
|
+
out.set(routeKey(route), catalogEntryFromRoute(route));
|
|
52
|
+
return out;
|
|
53
|
+
}
|
|
54
|
+
/** Preserve the route while enriching an OpenRouter entry with its live catalog metadata. */
|
|
55
|
+
export function enrichRouteEntry(entry, live) {
|
|
56
|
+
if (!live)
|
|
57
|
+
return entry;
|
|
58
|
+
return { ...live, route: entry.route, id: entry.id, name: entry.name, priceKnown: true };
|
|
59
|
+
}
|
package/dist/store.js
CHANGED
|
@@ -18,10 +18,12 @@ export const paths = {
|
|
|
18
18
|
benchmarksDigest: () => join(stateDir(), "benchmarks", "digest.json"),
|
|
19
19
|
traceCursor: () => join(stateDir(), "traces", "cursor.json"),
|
|
20
20
|
observations: () => join(stateDir(), "traces", "observations.jsonl"),
|
|
21
|
+
events: () => join(stateDir(), "events.jsonl"),
|
|
21
22
|
harnessState: () => join(stateDir(), "harness-state.json"),
|
|
22
23
|
ledger: () => join(stateDir(), "ledger.json"),
|
|
23
24
|
watchProcessed: () => join(stateDir(), "watch", "processed.json"),
|
|
24
25
|
sessionJob: (marker) => join(stateDir(), "watch", "jobs", sha1(marker) + ".json"),
|
|
26
|
+
trialTrace: (id) => join(stateDir(), "traces", "trials", `${sha1(id)}.jsonl`),
|
|
25
27
|
envFile: () => join(stateDir(), ".env"),
|
|
26
28
|
};
|
|
27
29
|
export function sha1(text) {
|
package/dist/strategist.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
/** Strategist: pick the next untried model to trial, benchmark-aware. */
|
|
2
|
-
import {
|
|
2
|
+
import { directChatClient } from "./llm.js";
|
|
3
3
|
import { parseObj } from "./judge.js";
|
|
4
|
+
import { routeKey, routeLabel } from "./routes.js";
|
|
4
5
|
export const STRAT_PREFS = [
|
|
5
6
|
"anthropic/claude-3.7-sonnet",
|
|
6
7
|
"anthropic/claude-3.5-sonnet",
|
|
7
8
|
"openai/gpt-4o",
|
|
8
9
|
];
|
|
9
|
-
const STRAT_PROMPT = `We are finding the best
|
|
10
|
+
const STRAT_PROMPT = `We are finding the best configured model route for a task via iterative trials.
|
|
10
11
|
TASK: {task}
|
|
11
12
|
RUBRIC: {rubric}
|
|
12
13
|
RELEVANT PUBLIC BENCHMARKS FOR THIS TASK: {benchmarks}
|
|
@@ -14,15 +15,16 @@ RESULTS SO FAR (model=score): {results}
|
|
|
14
15
|
Pick ONE untried model from the catalog below that is likely to do well on this
|
|
15
16
|
task, based on the benchmarks relevant to it. Balance quality against price so
|
|
16
17
|
a pareto frontier emerges.
|
|
17
|
-
Return ONLY json: {{"next_model": "<
|
|
18
|
-
|
|
18
|
+
Return ONLY json: {{"next_model": "<route id>", "why": "<one line>"}}
|
|
19
|
+
ROUTES (route id | logical model | blended $/1M tokens | ctx):
|
|
19
20
|
{catalog}`;
|
|
20
21
|
/** Pick the next model to trial, or null when the catalog is exhausted. */
|
|
21
|
-
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates) {
|
|
22
|
-
const
|
|
23
|
-
|
|
22
|
+
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
|
|
23
|
+
const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
|
|
24
|
+
const ok = (c) => (c.priceKnown === false || c.price <= maxPrice) &&
|
|
25
|
+
!tried.has(entryKey(c)) &&
|
|
24
26
|
c.ctx >= 4096 &&
|
|
25
|
-
!failedVendors.has(c.id.split("/")[0]);
|
|
27
|
+
!failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
|
|
26
28
|
// Benchmark-shortlisted models get priority in the listing.
|
|
27
29
|
const pool = [...cat.values()].filter(ok);
|
|
28
30
|
if (pool.length === 0)
|
|
@@ -31,15 +33,16 @@ export async function pickNext(key, stratModel, task, rubric, benchmarks, result
|
|
|
31
33
|
pool.sort((a, b) => {
|
|
32
34
|
const pa = priority.has(a.id) ? 0 : 1;
|
|
33
35
|
const pb = priority.has(b.id) ? 0 : 1;
|
|
34
|
-
return pa - pb || a.price - b.price;
|
|
36
|
+
return pa - pb || (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price;
|
|
35
37
|
});
|
|
36
38
|
const lines = pool
|
|
37
|
-
.map((c) => `${c
|
|
39
|
+
.map((c) => `${entryKey(c)} | ${c.route ? routeLabel(c.route) : c.id} | ` +
|
|
40
|
+
`${c.priceKnown === false ? "unknown" : c.price.toFixed(2)} | ${c.ctx}${priority.has(c.id) ? " | BENCHMARK" : ""}`)
|
|
38
41
|
.join("\n");
|
|
39
42
|
const res = results
|
|
40
43
|
.map((r) => `${r.model}=${r.score.toFixed(2)}`)
|
|
41
44
|
.join(" ") || "none yet";
|
|
42
|
-
const { content } = await chat(
|
|
45
|
+
const { content } = await client.chat(stratModel, STRAT_PROMPT.replace("{task}", task)
|
|
43
46
|
.replace("{rubric}", rubric)
|
|
44
47
|
.replace("{benchmarks}", benchmarks.join(", "))
|
|
45
48
|
.replace("{results}", res)
|
|
@@ -55,10 +58,11 @@ export async function pickNext(key, stratModel, task, rubric, benchmarks, result
|
|
|
55
58
|
/* fall through to cheap pick */
|
|
56
59
|
}
|
|
57
60
|
if (pick) {
|
|
58
|
-
const entry = cat.get(pick);
|
|
61
|
+
const entry = cat.get(pick) ?? [...cat.values()].find((candidate) => candidate.id === pick);
|
|
59
62
|
if (entry && ok(entry))
|
|
60
|
-
return { model:
|
|
63
|
+
return { model: entry.id, route: entry.route, why };
|
|
61
64
|
}
|
|
62
65
|
const cheapest = pool[0];
|
|
63
|
-
return { model: cheapest.id,
|
|
66
|
+
return { model: cheapest.id, route: cheapest.route,
|
|
67
|
+
why: `fallback pick: cheapest untried (strategist pick ${pick ?? "none"} unusable)` };
|
|
64
68
|
}
|