openmerit 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/pi-trials.js CHANGED
@@ -1,12 +1,15 @@
1
1
  /** Run a task through pi for each model, using pi's JSON event stream as the measurement source. */
2
2
  import { spawn, spawnSync } from "node:child_process";
3
- import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
3
+ import { cpSync, existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
5
- import { join } from "node:path";
5
+ import { dirname, join, relative } from "node:path";
6
+ import { createHash } from "node:crypto";
6
7
  import { judge, judgeWithImages } from "./judge.js";
7
8
  import { knownInvoiceScore } from "./invoice-eval.js";
8
9
  import { paths, readJson } from "./store.js";
9
- import { sessionImages, taskInputKey } from "./task-input.js";
10
+ import { sessionFiles, sessionImages, taskInputKey } from "./task-input.js";
11
+ import { legacyOpenRouterRoute, meritModelId, modelRoute, routeKey } from "./routes.js";
12
+ import { directChatClient } from "./llm.js";
10
13
  function answerText(content) {
11
14
  if (typeof content === "string")
12
15
  return content;
@@ -15,6 +18,59 @@ function answerText(content) {
15
18
  return content.filter((x) => !!x && typeof x === "object" && x.type === "text")
16
19
  .map((x) => x.text ?? "").join("\n");
17
20
  }
21
+ /** Pi-backed harness adapter. Future harnesses can implement the same seam. */
22
+ export class PiHarnessAdapter {
23
+ runTask(model, input) {
24
+ return executePiTask(model, input.task, input.cwd, input.images, input.files);
25
+ }
26
+ }
27
+ const defaultHarness = new PiHarnessAdapter();
28
+ const KNOWN_TRIAL_TOOLS = new Set([
29
+ "read", "grep", "find", "ls", "bash", "powershell", "edit", "write",
30
+ ]);
31
+ /**
32
+ * Candidate runs have no tools by default. Users may explicitly opt into a
33
+ * list, but a copied cwd is not an OS sandbox for absolute paths or network.
34
+ */
35
+ export function piTrialToolArgs(setting = process.env.OPENMERIT_PI_TRIAL_TOOLS) {
36
+ if (!setting?.trim())
37
+ return ["--no-tools"];
38
+ const requested = [...new Set(setting.split(",").map((tool) => tool.trim()).filter(Boolean))];
39
+ if (requested.length === 1 && requested[0] === "none")
40
+ return ["--no-tools"];
41
+ const unknown = requested.filter((tool) => !KNOWN_TRIAL_TOOLS.has(tool));
42
+ if (requested.length === 0 || unknown.length > 0) {
43
+ throw new Error(`invalid OPENMERIT_PI_TRIAL_TOOLS${unknown.length ? `: ${unknown.join(", ")}` : ""}`);
44
+ }
45
+ return ["--tools", requested.join(",")];
46
+ }
47
+ function fileSnapshot(root) {
48
+ const out = new Map();
49
+ function walk(dir) {
50
+ for (const name of readdirSync(dir)) {
51
+ if (name === ".git" || name === "node_modules" || name === ".openmerit")
52
+ continue;
53
+ const file = join(dir, name);
54
+ const rel = relative(root, file);
55
+ const st = statSync(file);
56
+ if (st.isDirectory())
57
+ walk(file);
58
+ else if (st.isFile() && st.size < 20_000_000) {
59
+ out.set(rel, createHash("sha1").update(readFileSync(file)).digest("hex"));
60
+ }
61
+ }
62
+ }
63
+ walk(root);
64
+ return out;
65
+ }
66
+ function isolatedWorkspace(cwd) {
67
+ const root = mkdtempSync(join(tmpdir(), "openmerit-pi-workspace-"));
68
+ cpSync(cwd, root, { recursive: true, filter: (src) => {
69
+ const rel = relative(cwd, src);
70
+ return !rel.split("/").some((part) => part === ".git" || part === "node_modules" || part === ".openmerit");
71
+ } });
72
+ return root;
73
+ }
18
74
  /** Use model A's completed result from the active pi session when it matches this task. */
19
75
  export function recordedActiveTask(task, model, images = [], sessionFile, sessionBytes) {
20
76
  const state = sessionFile ? null : readJson(paths.harnessState(), {});
@@ -23,6 +79,8 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
23
79
  return null;
24
80
  let user = null;
25
81
  const assistants = [];
82
+ let toolCalls = 0;
83
+ let toolErrors = 0;
26
84
  for (const line of readFileSync(file).subarray(0, sessionBytes).toString("utf8").split("\n")) {
27
85
  if (!line.trim())
28
86
  continue;
@@ -38,15 +96,24 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
38
96
  if (entry.message.role === "user") {
39
97
  user = entry.message;
40
98
  assistants.length = 0;
99
+ toolCalls = 0;
100
+ toolErrors = 0;
41
101
  }
42
- else if (user && entry.message.role === "assistant")
102
+ else if (user && entry.message.role === "assistant") {
43
103
  assistants.push(entry.message);
104
+ if (Array.isArray(entry.message.content))
105
+ toolCalls += entry.message.content.filter((part) => part?.type === "toolCall").length;
106
+ }
107
+ else if (user && entry.message.role === "toolResult" &&
108
+ entry.message.isError)
109
+ toolErrors++;
44
110
  }
45
111
  const userImages = user ? sessionImages(user.content) : null;
46
- if (!user || !userImages || taskInputKey(answerText(user.content), userImages) !==
47
- taskInputKey(task, images))
112
+ const files = sessionFiles(task);
113
+ if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
114
+ taskInputKey(task, images, files))
48
115
  return null;
49
- const matching = assistants.filter((m) => (m.provider === "openrouter" ? m.model : `${m.provider}/${m.model}`) === model);
116
+ const matching = assistants.filter((m) => meritModelId(m.provider ?? "openrouter", m.model ?? "") === model);
50
117
  if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
51
118
  return null;
52
119
  const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
@@ -59,6 +126,9 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
59
126
  costUsd: matching.reduce((n, m) => n + (m.usage?.cost?.total ?? 0), 0),
60
127
  latencyMs: Math.max(0, last - first),
61
128
  errors: 0,
129
+ toolCalls,
130
+ toolErrors,
131
+ changedFiles: 0,
62
132
  };
63
133
  }
64
134
  /** Read the exact latest completed text task from the pi extension's session marker. */
@@ -91,31 +161,50 @@ export function settledActiveTask(snapshot) {
91
161
  Array.isArray(entry.message.content) && entry.message.content.some((b) => b?.type === "toolCall"))
92
162
  usedTools = true;
93
163
  }
94
- if (!userContent || usedTools)
164
+ if (!userContent)
95
165
  return null;
96
166
  const images = sessionImages(userContent);
97
167
  if (!images)
98
168
  return null;
99
169
  const task = answerText(userContent);
100
- if (!task.trim() || taskInputKey(task, images) !== st.settledTaskKey)
170
+ const files = sessionFiles(task);
171
+ if (!task.trim() || taskInputKey(task, images, files) !== st.settledTaskKey)
101
172
  return null;
102
173
  const run = recordedActiveTask(task, st.currentModel, images, st.sessionFile, st.sessionBytes);
103
174
  if (!run)
104
175
  return null;
176
+ const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
177
+ const routes = st.routes?.length ? st.routes : [route];
105
178
  return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
106
- settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd() };
179
+ settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
180
+ route, routes };
107
181
  }
108
- /** Limit candidates to models this installed pi can actually select. */
109
- export function availablePiModels() {
110
- const result = spawnSync("pi", ["--offline", "--list-models", "openrouter"], { encoding: "utf8" });
182
+ export function piExecutable() {
183
+ return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
184
+ }
185
+ /** Read provider/model routes from pi when no live extension snapshot is available. */
186
+ export function availablePiRoutes(snapshot) {
187
+ if (snapshot?.length)
188
+ return snapshot;
189
+ const result = spawnSync(piExecutable(), ["--offline", "--list-models"], { encoding: "utf8" });
111
190
  if (result.error || result.status !== 0)
112
- throw new Error("could not list pi OpenRouter models");
113
- return new Set(result.stdout.split("\n").slice(1)
114
- .map((line) => line.trim().split(/\s+/)[1])
115
- .filter((id) => !!id && !id.startsWith("~")));
191
+ throw new Error("could not list models from pi");
192
+ const routes = [];
193
+ for (const line of result.stdout.split("\n").slice(1)) {
194
+ const [provider, id, , , , images] = line.trim().split(/\s+/);
195
+ if (!provider || !id || id.startsWith("~"))
196
+ continue;
197
+ routes.push(modelRoute(provider, id, { input: images === "yes" ? ["text", "image"] : ["text"] }));
198
+ }
199
+ return routes;
200
+ }
201
+ /** Backward-compatible logical-id view used by the standalone OpenRouter trial command. */
202
+ export function availablePiModels() {
203
+ return new Set(availablePiRoutes().map((route) => route.meritId));
116
204
  }
117
205
  /** Each invocation is a fresh pi run, so model B does not inherit model A's answer. */
118
- export async function executePiTask(model, task, cwd = process.cwd(), images = []) {
206
+ export async function executePiTask(model, task, cwd = process.cwd(), images = [], files = []) {
207
+ const route = typeof model === "string" ? legacyOpenRouterRoute(model) : model;
119
208
  const imageDir = images.length ? mkdtempSync(join(tmpdir(), "openmerit-pi-image-")) : null;
120
209
  const suffix = {
121
210
  "image/jpeg": "jpg", "image/png": "png", "image/webp": "webp", "image/gif": "gif",
@@ -127,85 +216,122 @@ export async function executePiTask(model, task, cwd = process.cwd(), images = [
127
216
  writeFileSync(file, Buffer.from(image.data, "base64"));
128
217
  return `@${file}`;
129
218
  });
130
- const child = spawn("pi", [
131
- "--provider", "openrouter", "--model", model, "--mode", "json",
132
- "--offline", "--no-extensions", "--no-tools", "--print", ...imageArgs, task,
133
- ], { cwd, stdio: ["ignore", "pipe", "pipe"] });
134
- let buffer = "";
135
- let stderr = "";
136
- let answer = "";
137
- let costUsd = 0;
138
- let errors = 0;
139
- let sessionId;
140
- const started = Date.now();
141
- function consume(line) {
142
- if (!line.trim())
143
- return;
144
- let event;
145
- try {
146
- event = JSON.parse(line);
147
- }
148
- catch {
149
- return;
150
- }
151
- if (event.type === "session" && event.id)
152
- sessionId = event.id;
153
- if (event.type !== "message_end" || event.message?.role !== "assistant")
154
- return;
155
- answer = answerText(event.message.content) || answer;
156
- costUsd += event.message.usage?.cost?.total ?? 0;
157
- if (event.message.stopReason === "error")
158
- errors++;
159
- }
160
- child.stdout.on("data", (chunk) => {
161
- buffer += chunk.toString("utf8");
162
- let n;
163
- while ((n = buffer.indexOf("\n")) >= 0) {
164
- consume(buffer.slice(0, n));
165
- buffer = buffer.slice(n + 1);
166
- }
167
- });
168
- child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
169
- let exitCode;
219
+ const workspace = isolatedWorkspace(cwd);
170
220
  try {
171
- exitCode = await new Promise((resolve, reject) => {
221
+ const attachmentDir = join(workspace, ".openmerit-attachments");
222
+ mkdirSync(attachmentDir, { recursive: true });
223
+ const attachmentPaths = new Map();
224
+ const fileArgs = files.map((file, i) => {
225
+ const safeName = `${String(i).padStart(3, "0")}-${file.name.replace(/[^A-Za-z0-9._-]/g, "_")}`;
226
+ const target = join(attachmentDir, safeName);
227
+ cpSync(file.path, target);
228
+ attachmentPaths.set(file.path, target);
229
+ return `@${target}`;
230
+ });
231
+ const candidateTask = [...attachmentPaths.entries()].reduce((text, [source, target]) => text.split(source).join(target), task);
232
+ const before = fileSnapshot(workspace);
233
+ const traceFile = paths.trialTrace(`${routeKey(route)}:${task}:${Date.now()}`);
234
+ const traceLines = [];
235
+ const child = spawn(piExecutable(), [
236
+ "--provider", route.provider, "--model", route.modelId, "--mode", "json",
237
+ "--offline", "--no-extensions", "--approve", ...piTrialToolArgs(),
238
+ "--print", ...imageArgs, ...fileArgs, candidateTask,
239
+ ], { cwd: workspace, stdio: ["ignore", "pipe", "pipe"] });
240
+ let buffer = "";
241
+ let stderr = "";
242
+ let answer = "";
243
+ let costUsd = 0;
244
+ let errors = 0;
245
+ let sessionId;
246
+ let toolCalls = 0;
247
+ let toolErrors = 0;
248
+ const started = Date.now();
249
+ function consume(line) {
250
+ if (!line.trim())
251
+ return;
252
+ traceLines.push(line);
253
+ let event;
254
+ try {
255
+ event = JSON.parse(line);
256
+ }
257
+ catch {
258
+ return;
259
+ }
260
+ if (event.type === "session" && event.id)
261
+ sessionId = event.id;
262
+ const content = event.message?.content;
263
+ if (event.message?.role === "assistant" && Array.isArray(content))
264
+ toolCalls += content.filter((part) => part?.type === "toolCall").length;
265
+ if (event.message?.role === "toolResult" && event.message.isError)
266
+ toolErrors++;
267
+ if (event.type !== "message_end" || event.message?.role !== "assistant")
268
+ return;
269
+ answer = answerText(event.message.content) || answer;
270
+ costUsd += event.message.usage?.cost?.total ?? 0;
271
+ if (event.message.stopReason === "error")
272
+ errors++;
273
+ }
274
+ child.stdout.on("data", (chunk) => {
275
+ buffer += chunk.toString("utf8");
276
+ let n;
277
+ while ((n = buffer.indexOf("\n")) >= 0) {
278
+ consume(buffer.slice(0, n));
279
+ buffer = buffer.slice(n + 1);
280
+ }
281
+ });
282
+ child.stderr.on("data", (chunk) => { stderr += chunk.toString("utf8"); });
283
+ const exitCode = await new Promise((resolve, reject) => {
172
284
  child.on("error", reject);
173
285
  child.on("close", (code) => resolve(code ?? 1));
174
286
  });
287
+ consume(buffer);
288
+ mkdirSync(dirname(traceFile), { recursive: true });
289
+ writeFileSync(traceFile, traceLines.join("\n") + (traceLines.length ? "\n" : ""));
290
+ const after = fileSnapshot(workspace);
291
+ let changedFiles = 0;
292
+ for (const [file, hash] of after)
293
+ if (before.get(file) !== hash)
294
+ changedFiles++;
295
+ for (const file of before.keys())
296
+ if (!after.has(file))
297
+ changedFiles++;
298
+ if (exitCode !== 0 || !answer.trim()) {
299
+ throw new Error(`pi trial ${route.meritId} via ${route.provider} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
300
+ }
301
+ return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId,
302
+ toolCalls, toolErrors, changedFiles, traceFile };
175
303
  }
176
304
  finally {
177
305
  if (imageDir)
178
306
  rmSync(imageDir, { recursive: true, force: true });
307
+ rmSync(workspace, { recursive: true, force: true });
179
308
  }
180
- consume(buffer);
181
- if (exitCode !== 0 || !answer.trim()) {
182
- throw new Error(`pi trial ${model} failed: ${stderr.trim().slice(0, 180) || `exit ${exitCode}, empty answer`}`);
183
- }
184
- return { answer, costUsd, latencyMs: Date.now() - started, errors, sessionId };
185
309
  }
186
310
  /** Quality uses the same OpenMerit judge; cost and latency come from pi itself. */
187
- export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = []) {
311
+ export async function runPiTrial(key, judgeModel, task, rubric, model, entry, cwd = process.cwd(), images = [], files = [], harness = defaultHarness, client = directChatClient(key)) {
188
312
  try {
189
- const run = await executePiTask(model, task, cwd, images);
190
- return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images);
313
+ const run = await harness.runTask(entry?.route ?? model, { task, cwd, images, files });
314
+ return await scorePiRun(key, judgeModel, task, rubric, model, entry, run, "pi_trial", images, client);
191
315
  }
192
316
  catch (e) {
193
317
  const error = String(e);
194
318
  return {
195
- point: { model, score: 0, price: entry?.price ?? 0,
196
- ts: new Date().toISOString(), source: "pi_trial", why: error.slice(0, 160) },
319
+ point: { schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
320
+ priceKnown: entry?.priceKnown ?? !!entry,
321
+ ts: new Date().toISOString(), source: "pi_trial", toolCalls: 0, toolErrors: 1,
322
+ changedFiles: 0, why: error.slice(0, 160) },
197
323
  costUsd: 0, error,
198
324
  };
199
325
  }
200
326
  }
201
- export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = []) {
327
+ export async function scorePiRun(key, judgeModel, task, rubric, model, entry, run, source, images = [], client = directChatClient(key)) {
202
328
  let score = 0;
203
329
  let why = "judge unavailable";
204
330
  try {
205
331
  const pinned = images.length ? await knownInvoiceScore(task, images, run.answer) : null;
206
332
  const j = pinned ?? (images.length
207
- ? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images)
208
- : await judge(key, judgeModel, task, rubric, run.answer));
333
+ ? await judgeWithImages(key, judgeModel, task, rubric, run.answer, images, client)
334
+ : await judge(key, judgeModel, task, rubric, run.answer, client));
209
335
  score = Math.max(0, Math.min(1, j.score));
210
336
  why = j.why;
211
337
  }
@@ -214,9 +340,12 @@ export async function scorePiRun(key, judgeModel, task, rubric, model, entry, ru
214
340
  }
215
341
  return {
216
342
  point: {
217
- model, score, price: entry?.price ?? 0,
343
+ schemaVersion: 1, model, route: entry?.route, score, price: entry?.price ?? 0,
344
+ priceKnown: entry?.priceKnown ?? !!entry,
218
345
  latencyMs: run.latencyMs, ts: new Date().toISOString(), source,
219
- why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}`,
346
+ why: `${why}${run.errors ? `; pi errors: ${run.errors}` : ""}; ` +
347
+ `tools=${run.toolCalls}, toolErrors=${run.toolErrors}, changedFiles=${run.changedFiles}`,
348
+ toolCalls: run.toolCalls, toolErrors: run.toolErrors, changedFiles: run.changedFiles,
220
349
  },
221
350
  costUsd: run.costUsd,
222
351
  sessionId: run.sessionId,
package/dist/policy.js CHANGED
@@ -60,10 +60,11 @@ export function loadPolicy(path) {
60
60
  max_usd_per_m: num(raw.max_usd_per_m, d.max_usd_per_m),
61
61
  };
62
62
  }
63
- /** Is a model id allowed by the provider allow/deny lists? ("*" matches all; "vendor/*" matches a vendor.) */
64
- export function providerAllowed(policy, modelId) {
63
+ /** Match either a logical vendor/model id or its concrete route provider. */
64
+ export function providerAllowed(policy, modelId, routeProvider) {
65
65
  const vendor = modelId.split("/")[0];
66
- const match = (pat) => pat === "*" || pat === modelId || (pat.endsWith("/*") && pat.slice(0, -2) === vendor);
66
+ const match = (pat) => pat === "*" || pat === modelId || pat === routeProvider ||
67
+ (pat.endsWith("/*") && (pat.slice(0, -2) === vendor || pat.slice(0, -2) === routeProvider));
67
68
  if (policy.providers.deny.some(match))
68
69
  return false;
69
70
  return policy.providers.allow.some(match);
@@ -87,6 +88,10 @@ export function gate(policy, input) {
87
88
  reasons.push("recommendation is not bound to a pi session");
88
89
  return { autoApply: false, reasons };
89
90
  }
91
+ if (input.priceKnown === false) {
92
+ reasons.push("candidate or current-model price is unknown");
93
+ return { autoApply: false, reasons };
94
+ }
90
95
  if (policy.mode !== "auto" || !policy.auto_apply.enabled) {
91
96
  reasons.push(`policy mode is "${policy.mode}" (auto_apply ${policy.auto_apply.enabled ? "enabled" : "disabled"})`);
92
97
  return { autoApply: false, reasons };
@@ -0,0 +1 @@
1
+ export {};
package/dist/recommend.js CHANGED
@@ -2,40 +2,49 @@
2
2
  import { createHash } from "node:crypto";
3
3
  import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
4
4
  import { gate, providerAllowed } from "./policy.js";
5
- export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null) {
5
+ import { pointKey, routeLabel } from "./routes.js";
6
+ export function buildRecommendation(taskKey, taskLabel, currentModel, points, policy, sessionFile = null, currentRoute = currentModel ? points.find((p) => p.model === currentModel)?.route ?? null : null) {
6
7
  const scored = points.filter((p) => p.score > 0);
7
8
  if (scored.length === 0)
8
9
  return null;
9
10
  const best = pickBest(scored);
10
11
  if (!best)
11
12
  return null;
12
- if (currentModel && best.model === currentModel)
13
- return null; // already on the best model
14
- const current = currentModel ? scored.find((p) => p.model === currentModel) : undefined;
13
+ if (currentModel && best.model === currentModel &&
14
+ (!currentRoute || pointKey(best) === `${currentRoute.provider}:${currentRoute.modelId}`))
15
+ return null;
16
+ const current = currentModel ? scored.find((p) => currentRoute
17
+ ? pointKey(p) === `${currentRoute.provider}:${currentRoute.modelId}`
18
+ : p.model === currentModel) : undefined;
15
19
  const scoreGain = current ? best.score - current.score : 0;
16
20
  // pickBest may choose a cheaper or faster model at equal quality.
17
21
  if (current && scoreGain < 0)
18
22
  return null;
19
- const priceRatio = current && current.price > 0 ? best.price / current.price : 1;
20
- const frontierModels = paretoFrontier(scored).map((p) => p.model);
21
- const onFrontier = frontierModels.includes(best.model);
23
+ const priceKnown = best.priceKnown !== false && (!current || current.priceKnown !== false);
24
+ const priceRatio = priceKnown && current && current.price > 0 ? best.price / current.price : 1;
25
+ const frontierModels = paretoFrontier(scored).map(pointKey);
26
+ const onFrontier = frontierModels.includes(pointKey(best));
22
27
  const fb = pickFallback(scored, best, policy.fallback.min_score);
23
28
  const decision = gate(policy, {
24
29
  scoreGain,
25
30
  priceRatio,
31
+ priceKnown,
26
32
  onFrontier,
27
- providerOk: providerAllowed(policy, best.model),
33
+ providerOk: providerAllowed(policy, best.model, best.route?.provider),
28
34
  baselineMeasured: !!current,
29
35
  sessionBound: !!sessionFile,
30
36
  });
31
- const reason = `${best.model} scores ${best.score.toFixed(2)} vs ` +
37
+ const bestLabel = best.route ? routeLabel(best.route) : best.model;
38
+ const reason = `${bestLabel} scores ${best.score.toFixed(2)} vs ` +
32
39
  (current ? `${current.score.toFixed(2)} for ${current.model}` : "no baseline measured") +
33
- ` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), at $${best.price.toFixed(2)}/M` +
34
- (current ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
40
+ ` (${scoreGain > 0 ? `quality gain ${scoreGain.toFixed(2)}` : "equal measured quality"}), ` +
41
+ (best.priceKnown === false ? "with unknown price" : `at $${best.price.toFixed(2)}/M`) +
42
+ (current && priceKnown ? ` (${priceRatio.toFixed(2)}x current price)` : "") +
35
43
  (best.why ? `. Judge: ${best.why}` : "");
36
44
  return {
45
+ schemaVersion: 1,
37
46
  id: createHash("sha1")
38
- .update(`${taskKey}:${best.model}:${Date.now()}`)
47
+ .update(`${taskKey}:${pointKey(best)}:${Date.now()}`)
39
48
  .digest("hex")
40
49
  .slice(0, 10),
41
50
  ts: new Date().toISOString(),
@@ -43,17 +52,21 @@ export function buildRecommendation(taskKey, taskLabel, currentModel, points, po
43
52
  taskLabel,
44
53
  sessionFile,
45
54
  currentModel,
46
- recommended: { model: best.model, reason },
55
+ currentRoute,
56
+ recommended: { model: best.model, route: best.route, reason },
47
57
  fallback: fb
48
58
  ? {
49
59
  model: fb.model,
50
- reason: `score ${fb.score.toFixed(2)} at $${fb.price.toFixed(2)}/M` +
60
+ route: fb.route,
61
+ reason: `score ${fb.score.toFixed(2)} ` +
62
+ (fb.priceKnown === false ? "with unknown price" : `at $${fb.price.toFixed(2)}/M`) +
51
63
  (fb.latencyMs ? `, ~${Math.round(fb.latencyMs)}ms` : ""),
52
64
  }
53
65
  : null,
54
66
  evidence: {
55
67
  scoreGain: Math.round(scoreGain * 1000) / 1000,
56
68
  priceRatio: Math.round(priceRatio * 100) / 100,
69
+ priceKnown,
57
70
  onFrontier,
58
71
  trials: scored.length,
59
72
  },
package/dist/routes.js ADDED
@@ -0,0 +1,59 @@
1
+ /** Provider-neutral model-route normalization. */
2
+ export function meritModelId(provider, modelId) {
3
+ if (provider === "openrouter")
4
+ return modelId;
5
+ return modelId.startsWith(`${provider}/`) ? modelId : `${provider}/${modelId}`;
6
+ }
7
+ export function modelRoute(provider, modelId, details = {}) {
8
+ return {
9
+ provider,
10
+ modelId,
11
+ meritId: meritModelId(provider, modelId),
12
+ input: details.input ?? ["text"],
13
+ ...details,
14
+ };
15
+ }
16
+ export function legacyOpenRouterRoute(meritId) {
17
+ return modelRoute("openrouter", meritId);
18
+ }
19
+ export function routeKey(route) {
20
+ return `${route.provider}:${route.modelId}`;
21
+ }
22
+ export function pointKey(point) {
23
+ return point.route ? routeKey(point.route) : `openrouter:${point.model}`;
24
+ }
25
+ export function routeLabel(route) {
26
+ return route.provider === "openrouter" ? route.meritId : `${route.meritId} via ${route.provider}`;
27
+ }
28
+ export function routePrice(route) {
29
+ if (!route.cost || !Number.isFinite(route.cost.input) || !Number.isFinite(route.cost.output))
30
+ return { price: 0, known: false };
31
+ return { price: Math.round(((route.cost.input + route.cost.output) / 2 + Number.EPSILON) * 1e4) / 1e4,
32
+ known: true };
33
+ }
34
+ export function catalogEntryFromRoute(route) {
35
+ const priced = routePrice(route);
36
+ return {
37
+ id: route.meritId,
38
+ name: routeLabel(route),
39
+ ctx: route.contextWindow ?? 0,
40
+ pp: route.cost ? route.cost.input / 1e6 : 0,
41
+ pc: route.cost ? route.cost.output / 1e6 : 0,
42
+ price: priced.price,
43
+ priceKnown: priced.known,
44
+ inputModalities: route.input,
45
+ route,
46
+ };
47
+ }
48
+ export function routeCatalog(routes) {
49
+ const out = new Map();
50
+ for (const route of routes)
51
+ out.set(routeKey(route), catalogEntryFromRoute(route));
52
+ return out;
53
+ }
54
+ /** Preserve the route while enriching an OpenRouter entry with its live catalog metadata. */
55
+ export function enrichRouteEntry(entry, live) {
56
+ if (!live)
57
+ return entry;
58
+ return { ...live, route: entry.route, id: entry.id, name: entry.name, priceKnown: true };
59
+ }
package/dist/store.js CHANGED
@@ -18,10 +18,12 @@ export const paths = {
18
18
  benchmarksDigest: () => join(stateDir(), "benchmarks", "digest.json"),
19
19
  traceCursor: () => join(stateDir(), "traces", "cursor.json"),
20
20
  observations: () => join(stateDir(), "traces", "observations.jsonl"),
21
+ events: () => join(stateDir(), "events.jsonl"),
21
22
  harnessState: () => join(stateDir(), "harness-state.json"),
22
23
  ledger: () => join(stateDir(), "ledger.json"),
23
24
  watchProcessed: () => join(stateDir(), "watch", "processed.json"),
24
25
  sessionJob: (marker) => join(stateDir(), "watch", "jobs", sha1(marker) + ".json"),
26
+ trialTrace: (id) => join(stateDir(), "traces", "trials", `${sha1(id)}.jsonl`),
25
27
  envFile: () => join(stateDir(), ".env"),
26
28
  };
27
29
  export function sha1(text) {
@@ -1,12 +1,13 @@
1
1
  /** Strategist: pick the next untried model to trial, benchmark-aware. */
2
- import { chat } from "./llm.js";
2
+ import { directChatClient } from "./llm.js";
3
3
  import { parseObj } from "./judge.js";
4
+ import { routeKey, routeLabel } from "./routes.js";
4
5
  export const STRAT_PREFS = [
5
6
  "anthropic/claude-3.7-sonnet",
6
7
  "anthropic/claude-3.5-sonnet",
7
8
  "openai/gpt-4o",
8
9
  ];
9
- const STRAT_PROMPT = `We are finding the best OpenRouter model for a task via iterative trials.
10
+ const STRAT_PROMPT = `We are finding the best configured model route for a task via iterative trials.
10
11
  TASK: {task}
11
12
  RUBRIC: {rubric}
12
13
  RELEVANT PUBLIC BENCHMARKS FOR THIS TASK: {benchmarks}
@@ -14,15 +15,16 @@ RESULTS SO FAR (model=score): {results}
14
15
  Pick ONE untried model from the catalog below that is likely to do well on this
15
16
  task, based on the benchmarks relevant to it. Balance quality against price so
16
17
  a pareto frontier emerges.
17
- Return ONLY json: {{"next_model": "<catalog id>", "why": "<one line>"}}
18
- CATALOG (id | blended $/1M tokens | ctx):
18
+ Return ONLY json: {{"next_model": "<route id>", "why": "<one line>"}}
19
+ ROUTES (route id | logical model | blended $/1M tokens | ctx):
19
20
  {catalog}`;
20
21
  /** Pick the next model to trial, or null when the catalog is exhausted. */
21
- export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates) {
22
- const ok = (c) => c.price <= maxPrice &&
23
- !tried.has(c.id) &&
22
+ export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
23
+ const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
24
+ const ok = (c) => (c.priceKnown === false || c.price <= maxPrice) &&
25
+ !tried.has(entryKey(c)) &&
24
26
  c.ctx >= 4096 &&
25
- !failedVendors.has(c.id.split("/")[0]);
27
+ !failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
26
28
  // Benchmark-shortlisted models get priority in the listing.
27
29
  const pool = [...cat.values()].filter(ok);
28
30
  if (pool.length === 0)
@@ -31,15 +33,16 @@ export async function pickNext(key, stratModel, task, rubric, benchmarks, result
31
33
  pool.sort((a, b) => {
32
34
  const pa = priority.has(a.id) ? 0 : 1;
33
35
  const pb = priority.has(b.id) ? 0 : 1;
34
- return pa - pb || a.price - b.price;
36
+ return pa - pb || (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price;
35
37
  });
36
38
  const lines = pool
37
- .map((c) => `${c.id} | ${c.price.toFixed(2)} | ${c.ctx}${priority.has(c.id) ? " | BENCHMARK" : ""}`)
39
+ .map((c) => `${entryKey(c)} | ${c.route ? routeLabel(c.route) : c.id} | ` +
40
+ `${c.priceKnown === false ? "unknown" : c.price.toFixed(2)} | ${c.ctx}${priority.has(c.id) ? " | BENCHMARK" : ""}`)
38
41
  .join("\n");
39
42
  const res = results
40
43
  .map((r) => `${r.model}=${r.score.toFixed(2)}`)
41
44
  .join(" ") || "none yet";
42
- const { content } = await chat(key, stratModel, STRAT_PROMPT.replace("{task}", task)
45
+ const { content } = await client.chat(stratModel, STRAT_PROMPT.replace("{task}", task)
43
46
  .replace("{rubric}", rubric)
44
47
  .replace("{benchmarks}", benchmarks.join(", "))
45
48
  .replace("{results}", res)
@@ -55,10 +58,11 @@ export async function pickNext(key, stratModel, task, rubric, benchmarks, result
55
58
  /* fall through to cheap pick */
56
59
  }
57
60
  if (pick) {
58
- const entry = cat.get(pick);
61
+ const entry = cat.get(pick) ?? [...cat.values()].find((candidate) => candidate.id === pick);
59
62
  if (entry && ok(entry))
60
- return { model: pick, why };
63
+ return { model: entry.id, route: entry.route, why };
61
64
  }
62
65
  const cheapest = pool[0];
63
- return { model: cheapest.id, why: `fallback pick: cheapest untried (strategist pick ${pick ?? "none"} unusable)` };
66
+ return { model: cheapest.id, route: cheapest.route,
67
+ why: `fallback pick: cheapest untried (strategist pick ${pick ?? "none"} unusable)` };
64
68
  }