openmerit 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -46
- package/dist/cli.js +62 -8
- package/dist/daemon.js +104 -48
- package/dist/frontier.js +13 -6
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +102 -1
- package/dist/pi-trials.js +46 -23
- package/dist/policy.js +8 -3
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/store.js +1 -0
- package/dist/strategist.js +18 -14
- package/dist/traces.js +6 -2
- package/dist/trials.js +38 -8
- package/extension/openmerit.ts +99 -23
- package/instructions/OPENMERIT.md +2 -2
- package/package.json +5 -2
- package/rules.md +39 -0
package/dist/frontier.js
CHANGED
|
@@ -1,10 +1,16 @@
|
|
|
1
1
|
/** Pareto frontier over (score up, price down, latency down) + best-fit/fallback picks. */
|
|
2
|
+
import { pointKey } from "./routes.js";
|
|
3
|
+
function effectivePrice(point) {
|
|
4
|
+
return point.priceKnown === false ? Number.POSITIVE_INFINITY : point.price;
|
|
5
|
+
}
|
|
2
6
|
/** a dominates b if a is no worse on every objective and strictly better on one. */
|
|
3
7
|
export function dominates(a, b) {
|
|
4
8
|
const aLat = a.latencyMs ?? Number.POSITIVE_INFINITY;
|
|
5
9
|
const bLat = b.latencyMs ?? Number.POSITIVE_INFINITY;
|
|
6
|
-
const
|
|
7
|
-
const
|
|
10
|
+
const aPrice = effectivePrice(a);
|
|
11
|
+
const bPrice = effectivePrice(b);
|
|
12
|
+
const noWorse = a.score >= b.score && aPrice <= bPrice && aLat <= bLat;
|
|
13
|
+
const better = a.score > b.score || aPrice < bPrice || aLat < bLat;
|
|
8
14
|
return noWorse && better;
|
|
9
15
|
}
|
|
10
16
|
/**
|
|
@@ -14,11 +20,11 @@ export function dominates(a, b) {
|
|
|
14
20
|
*/
|
|
15
21
|
export function paretoFrontier(points) {
|
|
16
22
|
const fr = points.filter((p) => !points.some((q) => q !== p && dominates(q, p)));
|
|
17
|
-
return fr.sort((a, b) => a
|
|
23
|
+
return fr.sort((a, b) => effectivePrice(a) - effectivePrice(b) || b.score - a.score);
|
|
18
24
|
}
|
|
19
25
|
/** Simple 2-objective chain (score vs price only), cheapest first. */
|
|
20
26
|
export function scorePriceChain(points) {
|
|
21
|
-
const pts = [...points].sort((a, b) => a
|
|
27
|
+
const pts = [...points].sort((a, b) => effectivePrice(a) - effectivePrice(b) || b.score - a.score);
|
|
22
28
|
const chain = [];
|
|
23
29
|
let best = -1;
|
|
24
30
|
for (const p of pts) {
|
|
@@ -31,7 +37,8 @@ export function scorePriceChain(points) {
|
|
|
31
37
|
}
|
|
32
38
|
/** Highest score; ties broken by lower price then lower latency. */
|
|
33
39
|
export function pickBest(points) {
|
|
34
|
-
return [...points].sort((a, b) => b.score - a.score ||
|
|
40
|
+
return [...points].sort((a, b) => b.score - a.score || effectivePrice(a) - effectivePrice(b) ||
|
|
41
|
+
(a.latencyMs ?? 1e18) - (b.latencyMs ?? 1e18))[0];
|
|
35
42
|
}
|
|
36
43
|
/**
|
|
37
44
|
* Fallback for `best`: prefer the cheapest frontier point that still clears
|
|
@@ -39,7 +46,7 @@ export function pickBest(points) {
|
|
|
39
46
|
* `best` itself.
|
|
40
47
|
*/
|
|
41
48
|
export function pickFallback(points, best, minScore) {
|
|
42
|
-
const others = points.filter((p) => p
|
|
49
|
+
const others = points.filter((p) => pointKey(p) !== pointKey(best));
|
|
43
50
|
if (others.length === 0)
|
|
44
51
|
return undefined;
|
|
45
52
|
const fr = paretoFrontier(others).filter((p) => p.score >= minScore);
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/** Stable boundaries for trace inputs and telemetry outputs. */
|
|
2
|
+
import { randomUUID } from "node:crypto";
|
|
3
|
+
import { appendJsonl, paths } from "./store.js";
|
|
4
|
+
import { ingestNewTraces } from "./traces.js";
|
|
5
|
+
export class PiTraceObservationSource {
|
|
6
|
+
sessionsDir;
|
|
7
|
+
id = "pi-jsonl";
|
|
8
|
+
constructor(sessionsDir) {
|
|
9
|
+
this.sessionsDir = sessionsDir;
|
|
10
|
+
}
|
|
11
|
+
read() { return ingestNewTraces(this.sessionsDir); }
|
|
12
|
+
}
|
|
13
|
+
export class LocalJsonlEventSink {
|
|
14
|
+
id = "local-jsonl";
|
|
15
|
+
emit(event) { appendJsonl(paths.events(), event); }
|
|
16
|
+
}
|
|
17
|
+
export function meritEvent(type, payload, source = "openmerit") {
|
|
18
|
+
return { schemaVersion: 1, id: randomUUID(), ts: new Date().toISOString(), source, type, payload };
|
|
19
|
+
}
|
package/dist/judge.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/** Judge: score a model's answer against a rubric. */
|
|
2
|
-
import {
|
|
2
|
+
import { directChatClient } from "./llm.js";
|
|
3
3
|
export const JUDGE_PREFS = ["openai/gpt-4o-mini", "openai/gpt-4o"];
|
|
4
4
|
const JUDGE_PROMPT = `You are grading a model's answer.
|
|
5
5
|
TASK: {task}
|
|
@@ -13,8 +13,8 @@ export function parseObj(txt) {
|
|
|
13
13
|
throw new Error("no json object found");
|
|
14
14
|
return JSON.parse(m[0]);
|
|
15
15
|
}
|
|
16
|
-
export async function judge(key, judgeModel, task, rubric, answer) {
|
|
17
|
-
const { content } = await chat(
|
|
16
|
+
export async function judge(key, judgeModel, task, rubric, answer, client = directChatClient(key)) {
|
|
17
|
+
const { content } = await client.chat(judgeModel, JUDGE_PROMPT.replace("{task}", task)
|
|
18
18
|
.replace("{rubric}", rubric)
|
|
19
19
|
.replace("{answer}", answer.slice(0, 16000)), 1024, 0);
|
|
20
20
|
try {
|
|
@@ -26,13 +26,13 @@ export async function judge(key, judgeModel, task, rubric, answer) {
|
|
|
26
26
|
}
|
|
27
27
|
}
|
|
28
28
|
/** Grade OCR answers against the actual uploaded pixels, not answer text alone. */
|
|
29
|
-
export async function judgeWithImages(key, judgeModel, task, rubric, answer, images) {
|
|
29
|
+
export async function judgeWithImages(key, judgeModel, task, rubric, answer, images, client = directChatClient(key)) {
|
|
30
30
|
const prompt = `You are grading an answer to a visual task. Inspect the attached image carefully.\n` +
|
|
31
31
|
`TASK: ${task}\nRUBRIC: ${rubric}\nANSWER: ${answer.slice(0, 16000)}\n` +
|
|
32
32
|
`For invoice extraction, check every visible field and line item against the image. ` +
|
|
33
33
|
`Penalize missing, invented, or mistyped values and invalid JSON. ` +
|
|
34
34
|
`Return ONLY json: {"score": <0..1>, "why": "<one line>"}`;
|
|
35
|
-
const { content } = await chatWithImages(
|
|
35
|
+
const { content } = await client.chatWithImages(judgeModel, prompt, images, 1024);
|
|
36
36
|
try {
|
|
37
37
|
const j = parseObj(content);
|
|
38
38
|
return { score: Number(j.score ?? 0), why: String(j.why ?? "") };
|
package/dist/llm.js
CHANGED
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
/** Minimal provider clients for judge and strategist requests. */
|
|
2
|
-
import { existsSync, readFileSync } from "node:fs";
|
|
2
|
+
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
3
|
+
import { spawn } from "node:child_process";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import { join } from "node:path";
|
|
3
6
|
import { paths } from "./store.js";
|
|
7
|
+
import { legacyOpenRouterRoute } from "./routes.js";
|
|
4
8
|
const OR = "https://openrouter.ai/api/v1";
|
|
5
9
|
/** Load OPENROUTER_API_KEY from env or ~/.openmerit/.env. */
|
|
6
10
|
export function loadKey() {
|
|
@@ -100,3 +104,100 @@ export async function chat(key, model, prompt, maxTokens, temperature) {
|
|
|
100
104
|
export async function chatWithImages(key, model, prompt, images, maxTokens) {
|
|
101
105
|
return providerFor(model, key).chatWithImages(model, prompt, images, maxTokens);
|
|
102
106
|
}
|
|
107
|
+
export function directChatClient(key) {
|
|
108
|
+
const id = (model) => typeof model === "string" ? model : model.meritId;
|
|
109
|
+
return {
|
|
110
|
+
chat: (model, prompt, maxTokens, temperature) => chat(key, id(model), prompt, maxTokens, temperature),
|
|
111
|
+
chatWithImages: (model, prompt, images, maxTokens) => chatWithImages(key, id(model), prompt, images, maxTokens),
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
function textContent(content) {
|
|
115
|
+
if (typeof content === "string")
|
|
116
|
+
return content;
|
|
117
|
+
if (!Array.isArray(content))
|
|
118
|
+
return "";
|
|
119
|
+
return content.filter((part) => !!part && typeof part === "object" && part.type === "text")
|
|
120
|
+
.map((part) => part.text ?? "").join("\n");
|
|
121
|
+
}
|
|
122
|
+
/** Use pi's model registry and credential store without exposing provider secrets to OpenMerit. */
|
|
123
|
+
export class PiCliChatClient {
|
|
124
|
+
executable;
|
|
125
|
+
onSpend;
|
|
126
|
+
id = "pi-cli";
|
|
127
|
+
constructor(executable = process.env.OPENMERIT_PI_BIN?.trim() || "pi", onSpend) {
|
|
128
|
+
this.executable = executable;
|
|
129
|
+
this.onSpend = onSpend;
|
|
130
|
+
}
|
|
131
|
+
chat(model, prompt, _maxTokens, _temperature) {
|
|
132
|
+
return this.run(model, prompt, []);
|
|
133
|
+
}
|
|
134
|
+
chatWithImages(model, prompt, images, _maxTokens) {
|
|
135
|
+
return this.run(model, prompt, images);
|
|
136
|
+
}
|
|
137
|
+
async run(target, prompt, images) {
|
|
138
|
+
const route = typeof target === "string" ? legacyOpenRouterRoute(target) : target;
|
|
139
|
+
const dir = mkdtempSync(join(tmpdir(), "openmerit-pi-chat-"));
|
|
140
|
+
try {
|
|
141
|
+
const imageArgs = images.map((image, index) => {
|
|
142
|
+
const suffix = { "image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif" }[image.mimeType];
|
|
143
|
+
if (!suffix)
|
|
144
|
+
throw new Error(`unsupported pi image: ${image.mimeType}`);
|
|
145
|
+
const file = join(dir, `image-${index}.${suffix}`);
|
|
146
|
+
writeFileSync(file, Buffer.from(image.data, "base64"));
|
|
147
|
+
return `@${file}`;
|
|
148
|
+
});
|
|
149
|
+
const child = spawn(this.executable, [
|
|
150
|
+
"--provider", route.provider, "--model", route.modelId, "--mode", "json", "--offline",
|
|
151
|
+
"--no-extensions", "--no-tools", "--print", ...imageArgs, prompt,
|
|
152
|
+
], { cwd: dir, stdio: ["ignore", "pipe", "pipe"] });
|
|
153
|
+
let stdout = "";
|
|
154
|
+
let stderr = "";
|
|
155
|
+
child.stdout.on("data", (chunk) => { stdout += chunk.toString("utf8"); });
|
|
156
|
+
child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
|
|
157
|
+
const code = await new Promise((resolve, reject) => {
|
|
158
|
+
child.on("error", reject);
|
|
159
|
+
child.on("close", (value) => resolve(value ?? 1));
|
|
160
|
+
});
|
|
161
|
+
let answer = "";
|
|
162
|
+
let promptTokens = 0;
|
|
163
|
+
let completionTokens = 0;
|
|
164
|
+
let totalTokens = 0;
|
|
165
|
+
let costUsd = 0;
|
|
166
|
+
let failed = false;
|
|
167
|
+
for (const line of stdout.split("\n")) {
|
|
168
|
+
if (!line.trim())
|
|
169
|
+
continue;
|
|
170
|
+
let event;
|
|
171
|
+
try {
|
|
172
|
+
event = JSON.parse(line);
|
|
173
|
+
}
|
|
174
|
+
catch {
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
if (event.type !== "message_end" || event.message?.role !== "assistant")
|
|
178
|
+
continue;
|
|
179
|
+
answer = textContent(event.message.content) || answer;
|
|
180
|
+
promptTokens += event.message.usage?.input ?? 0;
|
|
181
|
+
completionTokens += event.message.usage?.output ?? 0;
|
|
182
|
+
totalTokens += event.message.usage?.totalTokens ?? 0;
|
|
183
|
+
costUsd += event.message.usage?.cost?.total ?? 0;
|
|
184
|
+
if (event.message.stopReason === "error")
|
|
185
|
+
failed = true;
|
|
186
|
+
}
|
|
187
|
+
if (code !== 0 || failed || !answer.trim()) {
|
|
188
|
+
throw new Error(`pi model call ${route.provider}/${route.modelId} failed: ` +
|
|
189
|
+
(stderr.trim().slice(0, 240) || `exit ${code}, empty answer`));
|
|
190
|
+
}
|
|
191
|
+
this.onSpend?.(costUsd);
|
|
192
|
+
return { content: answer, usage: {
|
|
193
|
+
prompt_tokens: promptTokens || undefined,
|
|
194
|
+
completion_tokens: completionTokens || undefined,
|
|
195
|
+
total_tokens: totalTokens || promptTokens + completionTokens || undefined,
|
|
196
|
+
costUsd,
|
|
197
|
+
} };
|
|
198
|
+
}
|
|
199
|
+
finally {
|
|
200
|
+
rmSync(dir, { recursive: true, force: true });
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
}
|
package/dist/pi-trials.js
CHANGED
|
@@ -8,6 +8,8 @@ import { judge, judgeWithImages } from "./judge.js";
|
|
|
8
8
|
import { knownInvoiceScore } from "./invoice-eval.js";
|
|
9
9
|
import { paths, readJson } from "./store.js";
|
|
10
10
|
import { sessionFiles, sessionImages, taskInputKey } from "./task-input.js";
|
|
11
|
+
import { legacyOpenRouterRoute, meritModelId, modelRoute, routeKey } from "./routes.js";
|
|
12
|
+
import { directChatClient } from "./llm.js";
|
|
11
13
|
function answerText(content) {
|
|
12
14
|
if (typeof content === "string")
|
|
13
15
|
return content;
|
|
@@ -107,10 +109,11 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
|
|
|
107
109
|
toolErrors++;
|
|
108
110
|
}
|
|
109
111
|
const userImages = user ? sessionImages(user.content) : null;
|
|
110
|
-
|
|
111
|
-
|
|
112
|
+
const files = sessionFiles(task);
|
|
113
|
+
if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
|
|
114
|
+
taskInputKey(task, images, files))
|
|
112
115
|
return null;
|
|
113
|
-
const matching = assistants.filter((m) => (m.provider
|
|
116
|
+
const matching = assistants.filter((m) => meritModelId(m.provider ?? "openrouter", m.model ?? "") === model);
|
|
114
117
|
if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
|
|
115
118
|
return null;
|
|
116
119
|
const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
|
|
@@ -170,20 +173,38 @@ export function settledActiveTask(snapshot) {
|
|
|
170
173
|
const run = recordedActiveTask(task, st.currentModel, images, st.sessionFile, st.sessionBytes);
|
|
171
174
|
if (!run)
|
|
172
175
|
return null;
|
|
176
|
+
const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
|
|
177
|
+
const routes = st.routes?.length ? st.routes : [route];
|
|
173
178
|
return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
|
|
174
|
-
settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files
|
|
179
|
+
settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
|
|
180
|
+
route, routes };
|
|
175
181
|
}
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
182
|
+
export function piExecutable() {
|
|
183
|
+
return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
|
|
184
|
+
}
|
|
185
|
+
/** Read provider/model routes from pi when no live extension snapshot is available. */
|
|
186
|
+
export function availablePiRoutes(snapshot) {
|
|
187
|
+
if (snapshot?.length)
|
|
188
|
+
return snapshot;
|
|
189
|
+
const result = spawnSync(piExecutable(), ["--offline", "--list-models"], { encoding: "utf8" });
|
|
179
190
|
if (result.error || result.status !== 0)
|
|
180
|
-
throw new Error("could not list
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
191
|
+
throw new Error("could not list models from pi");
|
|
192
|
+
const routes = [];
|
|
193
|
+
for (const line of result.stdout.split("\n").slice(1)) {
|
|
194
|
+
const [provider, id, , , , images] = line.trim().split(/\s+/);
|
|
195
|
+
if (!provider || !id || id.startsWith("~"))
|
|
196
|
+
continue;
|
|
197
|
+
routes.push(modelRoute(provider, id, { input: images === "yes" ? ["text", "image"] : ["text"] }));
|
|
198
|
+
}
|
|
199
|
+
return routes;
|
|
200
|
+
}
|
|
201
|
+
/** Backward-compatible logical-id view used by the standalone OpenRouter trial command. */
|
|
202
|
+
export function availablePiModels() {
|
|
203
|
+
return new Set(availablePiRoutes().map((route) => route.meritId));
|
|
184
204
|
}
|
|
185
205
|
/** Each invocation is a fresh pi run, so model B does not inherit model A's answer. */
|
|
186
206
|
export async function executePiTask(model, task, cwd = process.cwd(), images = [], files = []) {
|
|
207
|
+
const route = typeof model === "string" ? legacyOpenRouterRoute(model) : model;
|
|
187
208
|
const imageDir = images.length ? mkdtempSync(join(tmpdir(), "openmerit-pi-image-")) : null;
|
|
188
209
|
const suffix = {
|
|
189
210
|
"image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif",
|
|
@@ -209,10 +230,10 @@ export async function executePiTask(model, task, cwd = process.cwd(), images = [
|
|
|
209
230
|
});
|
|
210
231
|
const candidateTask = [...attachmentPaths.entries()].reduce((text, [source, target]) => text.split(source).join(target), task);
|
|
211
232
|
const before = fileSnapshot(workspace);
|
|
212
|
-
const traceFile = paths.trialTrace(`${
|
|
233
|
+
const traceFile = paths.trialTrace(`${routeKey(route)}:${task}:${Date.now()}`);
|
|
213
234
|
const traceLines = [];
|
|
214
|
-
const child = spawn(
|
|
215
|
-
"--provider",
|
|
235
|
+
const child = spawn(piExecutable(), [
|
|
236
|
+
"--provider", route.provider, "--model", route.modelId, "--mode", "json",
|
|
216
237
|
"--offline", "--no-extensions", "--approve", ...piTrialToolArgs(),
|
|
217
238
|
"--print", ...imageArgs, ...fileArgs, candidateTask,
|
|
218
239
|
], { cwd: workspace, stdio: ["ignore", "pipe", "pipe"] });
|
|
@@ -275,7 +296,7 @@ export async function executePiTask(model, task, cwd = process.cwd(), images = [
|
|
|
275
296
|
if (!after.has(file))
|
|
276
297
|
changedFiles++;
|
|
277
298
|
if (exitCode !== 0 || !answer.trim()) {
|
|
278
|
-
throw new Error(`pi trial ${
|
|
299
|
+
throw new Error(`pi trial ${route.meritId} via ${route.provider} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
|
|
279
300
|
}
|
|
280
301
|
return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId,
|
|
281
302
|
toolCalls, toolErrors, changedFiles, traceFile };
|
|
@@ -287,29 +308,30 @@ export async function executePiTask(model, task, cwd = process.cwd(), images = [
|
|
|
287
308
|
}
|
|
288
309
|
}
|
|
289
310
|
/** Quality uses the same OpenMerit judge; cost and latency come from pi itself. */
|
|
290
|
-
export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = [], files = [], harness = defaultHarness) {
|
|
311
|
+
export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = [], files = [], harness = defaultHarness, client = directChatClient(key)) {
|
|
291
312
|
try {
|
|
292
|
-
const run = await harness.runTask(model, { task, cwd, images, files });
|
|
293
|
-
return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images);
|
|
313
|
+
const run = await harness.runTask(entry?.route ?? model, { task, cwd, images, files });
|
|
314
|
+
return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images, client);
|
|
294
315
|
}
|
|
295
316
|
catch (e) {
|
|
296
317
|
const error = String(e);
|
|
297
318
|
return {
|
|
298
|
-
point: { model, score: 0, price: entry?.price ?? 0,
|
|
319
|
+
point: { schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
320
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
299
321
|
ts: new Date().toISOString(), source: "pi_trial", toolCalls: 0, toolErrors: 1,
|
|
300
322
|
changedFiles: 0, why: error.slice(0, 160) },
|
|
301
323
|
costUsd: 0, error,
|
|
302
324
|
};
|
|
303
325
|
}
|
|
304
326
|
}
|
|
305
|
-
export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = []) {
|
|
327
|
+
export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = [], client = directChatClient(key)) {
|
|
306
328
|
let score = 0;
|
|
307
329
|
let why = "judge unavailable";
|
|
308
330
|
try {
|
|
309
331
|
const pinned = images.length ? await knownInvoiceScore(task, images, run.answer) : null;
|
|
310
332
|
const j = pinned ?? (images.length
|
|
311
|
-
? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images)
|
|
312
|
-
: await judge(key, judgeModel, task, rubric, run.answer));
|
|
333
|
+
? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images, client)
|
|
334
|
+
: await judge(key, judgeModel, task, rubric, run.answer, client));
|
|
313
335
|
score = Math.max(0, Math.min(1, j.score));
|
|
314
336
|
why = j.why;
|
|
315
337
|
}
|
|
@@ -318,7 +340,8 @@ export async function scorePiRun(key, judgeModel, task, rubric, model, entry, ru
|
|
|
318
340
|
}
|
|
319
341
|
return {
|
|
320
342
|
point: {
|
|
321
|
-
model, score, price: entry?.price ?? 0,
|
|
343
|
+
schemaVersion: 1, model, route: entry?.route, score, price: entry?.price ?? 0,
|
|
344
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
322
345
|
latencyMs: run.latencyMs, ts: new Date().toISOString(), source,
|
|
323
346
|
why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}; ` +
|
|
324
347
|
`tools=${run.toolCalls}, toolErrors=${run.toolErrors}, changedFiles=${run.changedFiles}`,
|
package/dist/policy.js
CHANGED
|
@@ -60,10 +60,11 @@ export function loadPolicy(path) {
|
|
|
60
60
|
max_usd_per_m: num(raw.max_usd_per_m, d.max_usd_per_m),
|
|
61
61
|
};
|
|
62
62
|
}
|
|
63
|
-
/**
|
|
64
|
-
export function providerAllowed(policy, modelId) {
|
|
63
|
+
/** Match either a logical vendor/model id or its concrete route provider. */
|
|
64
|
+
export function providerAllowed(policy, modelId, routeProvider) {
|
|
65
65
|
const vendor = modelId.split("/")[0];
|
|
66
|
-
const match = (pat) => pat === "*" || pat === modelId ||
|
|
66
|
+
const match = (pat) => pat === "*" || pat === modelId || pat === routeProvider ||
|
|
67
|
+
(pat.endsWith("/*") && (pat.slice(0, -2) === vendor || pat.slice(0, -2) === routeProvider));
|
|
67
68
|
if (policy.providers.deny.some(match))
|
|
68
69
|
return false;
|
|
69
70
|
return policy.providers.allow.some(match);
|
|
@@ -87,6 +88,10 @@ export function gate(policy, input) {
|
|
|
87
88
|
reasons.push("recommendation is not bound to a pi session");
|
|
88
89
|
return { autoApply: false, reasons };
|
|
89
90
|
}
|
|
91
|
+
if (input.priceKnown === false) {
|
|
92
|
+
reasons.push("candidate or current-model price is unknown");
|
|
93
|
+
return { autoApply: false, reasons };
|
|
94
|
+
}
|
|
90
95
|
if (policy.mode !== "auto" || !policy.auto_apply.enabled) {
|
|
91
96
|
reasons.push(`policy mode is "${policy.mode}" (auto_apply ${policy.auto_apply.enabled ? "enabled" : "disabled"})`);
|
|
92
97
|
return { autoApply: false, reasons };
|
package/dist/recommend.js
CHANGED
|
@@ -2,40 +2,49 @@
|
|
|
2
2
|
import { createHash } from "node:crypto";
|
|
3
3
|
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
4
4
|
import { gate, providerAllowed } from "./policy.js";
|
|
5
|
-
|
|
5
|
+
import { pointKey, routeLabel } from "./routes.js";
|
|
6
|
+
export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null, currentRoute = currentModel ? points.find((p) => p.model === currentModel)?.route ?? null : null) {
|
|
6
7
|
const scored = points.filter((p) => p.score > 0);
|
|
7
8
|
if (scored.length === 0)
|
|
8
9
|
return null;
|
|
9
10
|
const best = pickBest(scored);
|
|
10
11
|
if (!best)
|
|
11
12
|
return null;
|
|
12
|
-
if (currentModel && best.model === currentModel
|
|
13
|
-
|
|
14
|
-
|
|
13
|
+
if (currentModel && best.model === currentModel &&
|
|
14
|
+
(!currentRoute || pointKey(best) === `${currentRoute.provider}:${currentRoute.modelId}`))
|
|
15
|
+
return null;
|
|
16
|
+
const current = currentModel ? scored.find((p) => currentRoute
|
|
17
|
+
? pointKey(p) === `${currentRoute.provider}:${currentRoute.modelId}`
|
|
18
|
+
: p.model === currentModel) : undefined;
|
|
15
19
|
const scoreGain = current ? best.score - current.score : 0;
|
|
16
20
|
// pickBest may choose a cheaper or faster model at equal quality.
|
|
17
21
|
if (current && scoreGain < 0)
|
|
18
22
|
return null;
|
|
19
|
-
const
|
|
20
|
-
const
|
|
21
|
-
const
|
|
23
|
+
const priceKnown = best.priceKnown !== false && (!current || current.priceKnown !== false);
|
|
24
|
+
const priceRatio = priceKnown && current && current.price > 0 ? best.price / current.price : 1;
|
|
25
|
+
const frontierModels = paretoFrontier(scored).map(pointKey);
|
|
26
|
+
const onFrontier = frontierModels.includes(pointKey(best));
|
|
22
27
|
const fb = pickFallback(scored, best, policy.fallback.min_score);
|
|
23
28
|
const decision = gate(policy, {
|
|
24
29
|
scoreGain,
|
|
25
30
|
priceRatio,
|
|
31
|
+
priceKnown,
|
|
26
32
|
onFrontier,
|
|
27
|
-
providerOk: providerAllowed(policy, best.model),
|
|
33
|
+
providerOk: providerAllowed(policy, best.model, best.route?.provider),
|
|
28
34
|
baselineMeasured: !!current,
|
|
29
35
|
sessionBound: !!sessionFile,
|
|
30
36
|
});
|
|
31
|
-
const
|
|
37
|
+
const bestLabel = best.route ? routeLabel(best.route) : best.model;
|
|
38
|
+
const reason = `${bestLabel} scores ${best.score.toFixed(2)} vs ` +
|
|
32
39
|
(current ? `${current.score.toFixed(2)} for ${current.model}` : "no baseline measured") +
|
|
33
|
-
` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}),
|
|
34
|
-
(
|
|
40
|
+
` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), ` +
|
|
41
|
+
(best.priceKnown === false ? "with unknown price" : `at $${best.price.toFixed(2)}/M`) +
|
|
42
|
+
(current && priceKnown ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
|
|
35
43
|
(best.why ? `. Judge: ${best.why}` : "");
|
|
36
44
|
return {
|
|
45
|
+
schemaVersion: 1,
|
|
37
46
|
id: createHash("sha1")
|
|
38
|
-
.update(`${taskKey}:${best
|
|
47
|
+
.update(`${taskKey}:${pointKey(best)}:${Date.now()}`)
|
|
39
48
|
.digest("hex")
|
|
40
49
|
.slice(0, 10),
|
|
41
50
|
ts: new Date().toISOString(),
|
|
@@ -43,17 +52,21 @@ export function buildRecommendation(taskKey, taskLabel, currentModel, points, po
|
|
|
43
52
|
taskLabel,
|
|
44
53
|
sessionFile,
|
|
45
54
|
currentModel,
|
|
46
|
-
|
|
55
|
+
currentRoute,
|
|
56
|
+
recommended: { model: best.model, route: best.route, reason },
|
|
47
57
|
fallback: fb
|
|
48
58
|
? {
|
|
49
59
|
model: fb.model,
|
|
50
|
-
|
|
60
|
+
route: fb.route,
|
|
61
|
+
reason: `score ${fb.score.toFixed(2)} ` +
|
|
62
|
+
(fb.priceKnown === false ? "with unknown price" : `at $${fb.price.toFixed(2)}/M`) +
|
|
51
63
|
(fb.latencyMs ? `, ~${Math.round(fb.latencyMs)}ms` : ""),
|
|
52
64
|
}
|
|
53
65
|
: null,
|
|
54
66
|
evidence: {
|
|
55
67
|
scoreGain: Math.round(scoreGain * 1000) / 1000,
|
|
56
68
|
priceRatio: Math.round(priceRatio * 100) / 100,
|
|
69
|
+
priceKnown,
|
|
57
70
|
onFrontier,
|
|
58
71
|
trials: scored.length,
|
|
59
72
|
},
|
package/dist/routes.js
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/** Provider-neutral model-route normalization. */
|
|
2
|
+
export function meritModelId(provider, modelId) {
|
|
3
|
+
if (provider === "openrouter")
|
|
4
|
+
return modelId;
|
|
5
|
+
return modelId.startsWith(`${provider}/`) ? modelId : `${provider}/${modelId}`;
|
|
6
|
+
}
|
|
7
|
+
export function modelRoute(provider, modelId, details = {}) {
|
|
8
|
+
return {
|
|
9
|
+
provider,
|
|
10
|
+
modelId,
|
|
11
|
+
meritId: meritModelId(provider, modelId),
|
|
12
|
+
input: details.input ?? ["text"],
|
|
13
|
+
...details,
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
export function legacyOpenRouterRoute(meritId) {
|
|
17
|
+
return modelRoute("openrouter", meritId);
|
|
18
|
+
}
|
|
19
|
+
export function routeKey(route) {
|
|
20
|
+
return `${route.provider}:${route.modelId}`;
|
|
21
|
+
}
|
|
22
|
+
export function pointKey(point) {
|
|
23
|
+
return point.route ? routeKey(point.route) : `openrouter:${point.model}`;
|
|
24
|
+
}
|
|
25
|
+
export function routeLabel(route) {
|
|
26
|
+
return route.provider === "openrouter" ? route.meritId : `${route.meritId} via ${route.provider}`;
|
|
27
|
+
}
|
|
28
|
+
export function routePrice(route) {
|
|
29
|
+
if (!route.cost || !Number.isFinite(route.cost.input) || !Number.isFinite(route.cost.output))
|
|
30
|
+
return { price: 0, known: false };
|
|
31
|
+
return { price: Math.round(((route.cost.input + route.cost.output) / 2 + Number.EPSILON) * 1e4) / 1e4,
|
|
32
|
+
known: true };
|
|
33
|
+
}
|
|
34
|
+
export function catalogEntryFromRoute(route) {
|
|
35
|
+
const priced = routePrice(route);
|
|
36
|
+
return {
|
|
37
|
+
id: route.meritId,
|
|
38
|
+
name: routeLabel(route),
|
|
39
|
+
ctx: route.contextWindow ?? 0,
|
|
40
|
+
pp: route.cost ? route.cost.input / 1e6 : 0,
|
|
41
|
+
pc: route.cost ? route.cost.output / 1e6 : 0,
|
|
42
|
+
price: priced.price,
|
|
43
|
+
priceKnown: priced.known,
|
|
44
|
+
inputModalities: route.input,
|
|
45
|
+
route,
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
export function routeCatalog(routes) {
|
|
49
|
+
const out = new Map();
|
|
50
|
+
for (const route of routes)
|
|
51
|
+
out.set(routeKey(route), catalogEntryFromRoute(route));
|
|
52
|
+
return out;
|
|
53
|
+
}
|
|
54
|
+
/** Preserve the route while enriching an OpenRouter entry with its live catalog metadata. */
|
|
55
|
+
export function enrichRouteEntry(entry, live) {
|
|
56
|
+
if (!live)
|
|
57
|
+
return entry;
|
|
58
|
+
return { ...live, route: entry.route, id: entry.id, name: entry.name, priceKnown: true };
|
|
59
|
+
}
|
package/dist/store.js
CHANGED
|
@@ -18,6 +18,7 @@ export const paths = {
|
|
|
18
18
|
benchmarksDigest: () => join(stateDir(), "benchmarks", "digest.json"),
|
|
19
19
|
traceCursor: () => join(stateDir(), "traces", "cursor.json"),
|
|
20
20
|
observations: () => join(stateDir(), "traces", "observations.jsonl"),
|
|
21
|
+
events: () => join(stateDir(), "events.jsonl"),
|
|
21
22
|
harnessState: () => join(stateDir(), "harness-state.json"),
|
|
22
23
|
ledger: () => join(stateDir(), "ledger.json"),
|
|
23
24
|
watchProcessed: () => join(stateDir(), "watch", "processed.json"),
|
package/dist/strategist.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
/** Strategist: pick the next untried model to trial, benchmark-aware. */
|
|
2
|
-
import {
|
|
2
|
+
import { directChatClient } from "./llm.js";
|
|
3
3
|
import { parseObj } from "./judge.js";
|
|
4
|
+
import { routeKey, routeLabel } from "./routes.js";
|
|
4
5
|
export const STRAT_PREFS = [
|
|
5
6
|
"anthropic/claude-3.7-sonnet",
|
|
6
7
|
"anthropic/claude-3.5-sonnet",
|
|
7
8
|
"openai/gpt-4o",
|
|
8
9
|
];
|
|
9
|
-
const STRAT_PROMPT = `We are finding the best
|
|
10
|
+
const STRAT_PROMPT = `We are finding the best configured model route for a task via iterative trials.
|
|
10
11
|
TASK: {task}
|
|
11
12
|
RUBRIC: {rubric}
|
|
12
13
|
RELEVANT PUBLIC BENCHMARKS FOR THIS TASK: {benchmarks}
|
|
@@ -14,15 +15,16 @@ RESULTS SO FAR (model=score): {results}
|
|
|
14
15
|
Pick ONE untried model from the catalog below that is likely to do well on this
|
|
15
16
|
task, based on the benchmarks relevant to it. Balance quality against price so
|
|
16
17
|
a pareto frontier emerges.
|
|
17
|
-
Return ONLY json: {{"next_model": "<
|
|
18
|
-
|
|
18
|
+
Return ONLY json: {{"next_model": "<route id>", "why": "<one line>"}}
|
|
19
|
+
ROUTES (route id | logical model | blended $/1M tokens | ctx):
|
|
19
20
|
{catalog}`;
|
|
20
21
|
/** Pick the next model to trial, or null when the catalog is exhausted. */
|
|
21
|
-
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates) {
|
|
22
|
-
const
|
|
23
|
-
|
|
22
|
+
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
|
|
23
|
+
const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
|
|
24
|
+
const ok = (c) => (c.priceKnown === false || c.price <= maxPrice) &&
|
|
25
|
+
!tried.has(entryKey(c)) &&
|
|
24
26
|
c.ctx >= 4096 &&
|
|
25
|
-
!failedVendors.has(c.id.split("/")[0]);
|
|
27
|
+
!failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
|
|
26
28
|
// Benchmark-shortlisted models get priority in the listing.
|
|
27
29
|
const pool = [...cat.values()].filter(ok);
|
|
28
30
|
if (pool.length === 0)
|
|
@@ -31,15 +33,16 @@ export async function pickNext(key, stratModel, task, rubric, benchmarks, result
|
|
|
31
33
|
pool.sort((a, b) => {
|
|
32
34
|
const pa = priority.has(a.id) ? 0 : 1;
|
|
33
35
|
const pb = priority.has(b.id) ? 0 : 1;
|
|
34
|
-
return pa - pb || a.price - b.price;
|
|
36
|
+
return pa - pb || (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price;
|
|
35
37
|
});
|
|
36
38
|
const lines = pool
|
|
37
|
-
.map((c) => `${c
|
|
39
|
+
.map((c) => `${entryKey(c)} | ${c.route ? routeLabel(c.route) : c.id} | ` +
|
|
40
|
+
`${c.priceKnown === false ? "unknown" : c.price.toFixed(2)} | ${c.ctx}${priority.has(c.id) ? " | BENCHMARK" : ""}`)
|
|
38
41
|
.join("\n");
|
|
39
42
|
const res = results
|
|
40
43
|
.map((r) => `${r.model}=${r.score.toFixed(2)}`)
|
|
41
44
|
.join(" ") || "none yet";
|
|
42
|
-
const { content } = await chat(
|
|
45
|
+
const { content } = await client.chat(stratModel, STRAT_PROMPT.replace("{task}", task)
|
|
43
46
|
.replace("{rubric}", rubric)
|
|
44
47
|
.replace("{benchmarks}", benchmarks.join(", "))
|
|
45
48
|
.replace("{results}", res)
|
|
@@ -55,10 +58,11 @@ export async function pickNext(key, stratModel, task, rubric, benchmarks, result
|
|
|
55
58
|
/* fall through to cheap pick */
|
|
56
59
|
}
|
|
57
60
|
if (pick) {
|
|
58
|
-
const entry = cat.get(pick);
|
|
61
|
+
const entry = cat.get(pick) ?? [...cat.values()].find((candidate) => candidate.id === pick);
|
|
59
62
|
if (entry && ok(entry))
|
|
60
|
-
return { model:
|
|
63
|
+
return { model: entry.id, route: entry.route, why };
|
|
61
64
|
}
|
|
62
65
|
const cheapest = pool[0];
|
|
63
|
-
return { model: cheapest.id,
|
|
66
|
+
return { model: cheapest.id, route: cheapest.route,
|
|
67
|
+
why: `fallback pick: cheapest untried (strategist pick ${pick ?? "none"} unusable)` };
|
|
64
68
|
}
|
package/dist/traces.js
CHANGED
|
@@ -7,6 +7,7 @@ import { readdirSync, readFileSync, existsSync, statSync } from "node:fs";
|
|
|
7
7
|
import { homedir } from "node:os";
|
|
8
8
|
import { join } from "node:path";
|
|
9
9
|
import { paths, readJson, taskKey, writeJson } from "./store.js";
|
|
10
|
+
import { meritModelId, modelRoute } from "./routes.js";
|
|
10
11
|
export function defaultSessionsDir() {
|
|
11
12
|
return join(homedir(), ".pi", "agent", "sessions");
|
|
12
13
|
}
|
|
@@ -53,6 +54,7 @@ export function observationsFromEntries(entries, sessionFile) {
|
|
|
53
54
|
continue;
|
|
54
55
|
}
|
|
55
56
|
current = {
|
|
57
|
+
schemaVersion: 1,
|
|
56
58
|
taskKey: taskKey(text),
|
|
57
59
|
taskLabel: text.replace(/\s+/g, " ").trim().slice(0, 120),
|
|
58
60
|
model: null,
|
|
@@ -68,8 +70,10 @@ export function observationsFromEntries(entries, sessionFile) {
|
|
|
68
70
|
if (!current)
|
|
69
71
|
continue;
|
|
70
72
|
if (m.role === "assistant") {
|
|
71
|
-
if (!current.model && m.provider && m.model)
|
|
72
|
-
current.model =
|
|
73
|
+
if (!current.model && m.provider && m.model) {
|
|
74
|
+
current.model = meritModelId(m.provider, m.model);
|
|
75
|
+
current.route = modelRoute(m.provider, m.model);
|
|
76
|
+
}
|
|
73
77
|
current.costUsd += m.usage?.cost?.total ?? 0;
|
|
74
78
|
current.tokens += m.usage?.totalTokens ?? 0;
|
|
75
79
|
if (m.stopReason === "error")
|