openmerit 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,29 @@
1
1
  /** A pi text task may carry images embedded in its saved session message. */
2
2
  import { createHash } from "node:crypto";
3
+ import { existsSync, readFileSync, statSync } from "node:fs";
4
+ import { basename } from "node:path";
3
5
  import { taskKey } from "./store.js";
6
+ /** Recover files Pi expanded into a task message (for example @invoice.pdf). */
7
+ export function sessionFiles(text) {
8
+ const files = [];
9
+ const seen = new Set();
10
+ const re = /<file\s+name=["']([^"']+)["'][^>]*>/g;
11
+ for (const match of text.matchAll(re)) {
12
+ const file = match[1];
13
+ if (!file || seen.has(file) || !existsSync(file))
14
+ continue;
15
+ try {
16
+ const stat = statSync(file);
17
+ if (!stat.isFile() || stat.size > 100_000_000)
18
+ continue;
19
+ files.push({ path: file, name: basename(file), size: stat.size,
20
+ sha256: createHash("sha256").update(readFileSync(file)).digest("hex") });
21
+ seen.add(file);
22
+ }
23
+ catch { /* inaccessible attachment */ }
24
+ }
25
+ return files;
26
+ }
4
27
  const SUPPORTED = new Set(["image/jpeg", "image/png", "image/webp", "image/gif"]);
5
28
  export function sessionImages(content) {
6
29
  if (!Array.isArray(content))
@@ -20,11 +43,12 @@ export function sessionImages(content) {
20
43
  return images;
21
44
  }
22
45
  /** Text-only keys remain compatible; image bytes distinguish same-prompt documents. */
23
- export function taskInputKey(text, images) {
24
- if (images.length === 0)
46
+ export function taskInputKey(text, images, files = []) {
47
+ if (images.length === 0 && files.length === 0)
25
48
  return taskKey(text);
26
49
  const normalized = text.toLowerCase().replace(/\s+/g, " ").trim();
27
50
  const hashes = images.map((image) => `${image.mimeType}:` +
28
51
  createHash("sha256").update(Buffer.from(image.data, "base64")).digest("hex"));
29
- return taskKey(`${normalized}\nimages:${hashes.join(",")}`);
52
+ const fileHashes = files.map((file) => `${file.name}:${file.sha256}`).join(",");
53
+ return taskKey(`${normalized}\nimages:${hashes.join(",")}\nfiles:${fileHashes}`);
30
54
  }
package/dist/traces.js CHANGED
@@ -7,6 +7,7 @@ import { readdirSync, readFileSync, existsSync, statSync } from "node:fs";
7
7
  import { homedir } from "node:os";
8
8
  import { join } from "node:path";
9
9
  import { paths, readJson, taskKey, writeJson } from "./store.js";
10
+ import { meritModelId, modelRoute } from "./routes.js";
10
11
  export function defaultSessionsDir() {
11
12
  return join(homedir(), ".pi", "agent", "sessions");
12
13
  }
@@ -53,6 +54,7 @@ export function observationsFromEntries(entries, sessionFile) {
53
54
  continue;
54
55
  }
55
56
  current = {
57
+ schemaVersion: 1,
56
58
  taskKey: taskKey(text),
57
59
  taskLabel: text.replace(/\s+/g, " ").trim().slice(0, 120),
58
60
  model: null,
@@ -68,8 +70,10 @@ export function observationsFromEntries(entries, sessionFile) {
68
70
  if (!current)
69
71
  continue;
70
72
  if (m.role === "assistant") {
71
- if (!current.model && m.provider && m.model)
72
- current.model = `${m.provider}/${m.model}`;
73
+ if (!current.model && m.provider && m.model) {
74
+ current.model = meritModelId(m.provider, m.model);
75
+ current.route = modelRoute(m.provider, m.model);
76
+ }
73
77
  current.costUsd += m.usage?.cost?.total ?? 0;
74
78
  current.tokens += m.usage?.totalTokens ?? 0;
75
79
  if (m.stopReason === "error")
package/dist/trials.js CHANGED
@@ -1,5 +1,5 @@
1
1
  /** Shadow trials: run candidate models against observed/declared tasks on the background track. */
2
- import { chat } from "./llm.js";
2
+ import { directChatClient } from "./llm.js";
3
3
  import { judge, parseObj } from "./judge.js";
4
4
  import { paths, readJson, writeJson } from "./store.js";
5
5
  function today() {
@@ -15,6 +15,21 @@ export function budgetOk(policy) {
15
15
  return { ok: false, reason: `daily budget exhausted ($${policy.budgets.max_usd_per_day})` };
16
16
  return { ok: true };
17
17
  }
18
+ /** Conservative admission estimate for one candidate answer (rough input tokens + 4K output). */
19
+ export function trialBudgetOk(policy, entry, inputChars) {
20
+ if (entry.priceKnown === false)
21
+ return { ok: false, reason: "route price is unknown" };
22
+ const inputTokens = Math.ceil(inputChars / 4);
23
+ const outputTokens = Math.min(entry.route?.maxTokens ?? 4096, 4096);
24
+ const estimatedUsd = entry.pp * inputTokens + entry.pc * outputTokens;
25
+ if (estimatedUsd > policy.budgets.max_usd_per_trial)
26
+ return {
27
+ ok: false,
28
+ estimatedUsd,
29
+ reason: `estimated trial cost $${estimatedUsd.toFixed(4)} exceeds $${policy.budgets.max_usd_per_trial.toFixed(4)} limit`,
30
+ };
31
+ return { ok: true, estimatedUsd };
32
+ }
18
33
  export function recordTrialSpend(usd) {
19
34
  let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
20
35
  if (ledger.date !== today())
@@ -23,16 +38,26 @@ export function recordTrialSpend(usd) {
23
38
  ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
24
39
  writeJson(paths.ledger(), ledger);
25
40
  }
41
+ /** Record non-candidate merit-loop spend without consuming a trial slot. */
42
+ export function recordMeritSpend(usd) {
43
+ if (!(usd > 0))
44
+ return;
45
+ let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
46
+ if (ledger.date !== today())
47
+ ledger = { date: today(), trials: 0, usd: 0 };
48
+ ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
49
+ writeJson(paths.ledger(), ledger);
50
+ }
26
51
  const RUBRIC_PROMPT = `Write a compact grading rubric for answers to the task below.
27
52
  It must list the concrete criteria for a score of 1.0 and how to deduct.
28
53
  TASK: {task}
29
54
  Return ONLY json: {{"rubric": "<text>"}}`;
30
55
  /** Derive (and cache) a grading rubric for a task that came from traces. */
31
- export async function ensureRubric(key, judgeModel, task, cache) {
56
+ export async function ensureRubric(key, judgeModel, task, cache, client = directChatClient(key)) {
32
57
  const hit = cache.get(task);
33
58
  if (hit)
34
59
  return hit;
35
- const { content } = await chat(key, judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
60
+ const { content } = await client.chat(judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
36
61
  let rubric = "Score 1.0 for a fully correct, complete, usable answer; deduct for errors, omissions, or unusable output.";
37
62
  try {
38
63
  const obj = parseObj(content);
@@ -46,19 +71,20 @@ export async function ensureRubric(key, judgeModel, task, cache) {
46
71
  return rubric;
47
72
  }
48
73
  /** Run one shadow trial: generate with `model`, judge the output, return a frontier point + cost. */
49
- export async function runTrial(key, judgeModel, task, rubric, model, entry) {
74
+ export async function runTrial(key, judgeModel, task, rubric, model, entry, client = directChatClient(key)) {
50
75
  const started = Date.now();
51
76
  let gen;
52
77
  let usage = {};
53
78
  try {
54
- const r = await chat(key, model, task, 4096, 0.2);
79
+ const r = await client.chat(entry?.route ?? model, task, 4096, 0.2);
55
80
  gen = r.content;
56
81
  usage = r.usage;
57
82
  }
58
83
  catch (e) {
59
84
  return {
60
85
  point: {
61
- model, score: 0, price: entry?.price ?? 0,
86
+ schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
87
+ priceKnown: entry?.priceKnown ?? !!entry,
62
88
  ts: new Date().toISOString(), source: "shadow_trial", why: String(e).slice(0, 160),
63
89
  },
64
90
  costUsd: 0,
@@ -70,7 +96,8 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
70
96
  if (!gen.trim()) {
71
97
  return {
72
98
  point: {
73
- model, score: 0, price: entry?.price ?? 0, latencyMs,
99
+ schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
100
+ priceKnown: entry?.priceKnown ?? !!entry, latencyMs,
74
101
  ts: new Date().toISOString(), source: "shadow_trial", why: "empty response",
75
102
  },
76
103
  costUsd,
@@ -79,7 +106,7 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
79
106
  let score = 0;
80
107
  let why = "judge error";
81
108
  try {
82
- const j = await judge(key, judgeModel, task, rubric, gen);
109
+ const j = await judge(key, judgeModel, task, rubric, gen, client);
83
110
  score = j.score;
84
111
  why = j.why;
85
112
  }
@@ -88,9 +115,12 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
88
115
  }
89
116
  return {
90
117
  point: {
118
+ schemaVersion: 1,
91
119
  model,
120
+ route: entry?.route,
92
121
  score,
93
122
  price: entry?.price ?? 0,
123
+ priceKnown: entry?.priceKnown ?? !!entry,
94
124
  latencyMs,
95
125
  ts: new Date().toISOString(),
96
126
  source: "shadow_trial",
@@ -19,7 +19,7 @@
19
19
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
20
20
  import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
21
21
  import { homedir } from "node:os";
22
- import { join } from "node:path";
22
+ import { basename, join } from "node:path";
23
23
  import { createHash } from "node:crypto";
24
24
  import { spawn, type ChildProcessByStdio } from "node:child_process";
25
25
  import type { Readable } from "node:stream";
@@ -46,6 +46,9 @@ interface TrialProgress {
46
46
  latencyMs?: number;
47
47
  costUsd?: number;
48
48
  error?: string;
49
+ toolCalls?: number;
50
+ toolErrors?: number;
51
+ changedFiles?: number;
49
52
  }
50
53
 
51
54
  function parseTrialProgress(line: string): TrialProgress | null {
@@ -63,7 +66,9 @@ function parseTrialProgress(line: string): TrialProgress | null {
63
66
  function completedTrialText(trial: TrialProgress): string {
64
67
  return `${trial.model}: quality ${(trial.score ?? 0).toFixed(2)}, ` +
65
68
  `$${(trial.price ?? 0).toFixed(2)}/M, ${Math.round(trial.latencyMs ?? 0)}ms, ` +
66
- `run $${(trial.costUsd ?? 0).toFixed(4)}` + (trial.error ? ` (failed: ${trial.error})` : "");
69
+ `run $${(trial.costUsd ?? 0).toFixed(4)}, tools ${trial.toolCalls ?? 0}` +
70
+ (trial.toolErrors || trial.changedFiles ? `, tool errors ${trial.toolErrors ?? 0}, changed files ${trial.changedFiles ?? 0}` : "") +
71
+ (trial.error ? ` (failed: ${trial.error})` : "");
67
72
  }
68
73
 
69
74
  interface ExtensionPolicy {
@@ -110,19 +115,31 @@ function loadPolicy(): ExtensionPolicy {
110
115
  }
111
116
 
112
117
  interface Recommendation {
118
+ schemaVersion?: 1;
113
119
  id: string;
114
120
  ts: string;
115
121
  taskKey: string;
116
122
  taskLabel: string;
117
123
  sessionFile?: string | null;
118
124
  currentModel: string | null;
119
- recommended: { model: string; reason: string };
120
- fallback: { model: string; reason: string } | null;
121
- evidence: { scoreGain: number; priceRatio: number; onFrontier: boolean; trials: number };
125
+ currentRoute?: ModelRoute | null;
126
+ recommended: { model: string; route?: ModelRoute; reason: string };
127
+ fallback: { model: string; route?: ModelRoute; reason: string } | null;
128
+ evidence: { scoreGain: number; priceRatio: number; priceKnown?: boolean; onFrontier: boolean; trials: number };
122
129
  policy: { autoApply: boolean; reasons: string[] };
123
130
  status: "pending" | "applied" | "dismissed" | "expired";
124
131
  }
125
132
 
133
+ interface ModelRoute {
134
+ provider: string;
135
+ modelId: string;
136
+ meritId: string;
137
+ input: ("text" | "image")[];
138
+ cost?: { input: number; output: number; cacheRead?: number; cacheWrite?: number };
139
+ contextWindow?: number;
140
+ maxTokens?: number;
141
+ }
142
+
126
143
  interface RecordedTrial {
127
144
  taskKey?: string;
128
145
  sessionFile?: string;
@@ -171,10 +188,18 @@ function userTaskKey(content: unknown): string | null {
171
188
  const images = parts.filter((b) => b?.type === "image");
172
189
  if (parts.length !== parts.filter((b) => b?.type === "text" || b?.type === "image").length ||
173
190
  images.some((b) => typeof b.data !== "string" || typeof b.mimeType !== "string")) return null;
174
- if (images.length === 0) return taskKey(text);
191
+ const files: string[] = [];
192
+ const fileRe = /<file\s+name=["']([^"']+)["'][^>]*>/g;
193
+ for (const match of text.matchAll(fileRe)) {
194
+ const file = match[1];
195
+ if (!file || !existsSync(file) || files.includes(file)) continue;
196
+ try { files.push(`${basename(file)}:` + createHash("sha256").update(readFileSync(file)).digest("hex")); }
197
+ catch { /* inaccessible attachment */ }
198
+ }
199
+ if (images.length === 0 && files.length === 0) return taskKey(text);
175
200
  const hashes = images.map((b) => `${b.mimeType}:` +
176
201
  createHash("sha256").update(Buffer.from(b.data, "base64")).digest("hex"));
177
- return taskKey(`${text.toLowerCase().replace(/\s+/g, " ").trim()}\nimages:${hashes.join(",")}`);
202
+ return taskKey(`${text.toLowerCase().replace(/\s+/g, " ").trim()}\nimages:${hashes.join(",")}\nfiles:${files.join(",")}`);
178
203
  }
179
204
 
180
205
  function latestSessionTaskKey(ctx: ExtensionContext): string | null {
@@ -235,7 +260,11 @@ function suggestionText(rec: Recommendation, trials: RecordedTrial[]): string {
235
260
  }
236
261
 
237
262
  function canonicalModel(provider: string, id: string): string {
238
- return provider === "openrouter" ? id : `${provider}/${id}`;
263
+ return provider === "openrouter" || id.startsWith(`${provider}/`) ? id : `${provider}/${id}`;
264
+ }
265
+
266
+ function routeDisplay(route: ModelRoute | undefined, fallback: string): string {
267
+ return route && route.provider !== "openrouter" ? `${route.meritId} via ${route.provider}` : fallback;
239
268
  }
240
269
 
241
270
  function setStatus(rec: Recommendation, status: Recommendation["status"]): void {
@@ -243,11 +272,44 @@ function setStatus(rec: Recommendation, status: Recommendation["status"]): void
243
272
  }
244
273
 
245
274
  let fallbackModel: string | null = null;
275
+ let fallbackRoute: ModelRoute | null = null;
276
+
277
+ function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
278
+ cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
279
+ contextWindow?: number; maxTokens?: number }): ModelRoute {
280
+ return {
281
+ provider: model.provider,
282
+ modelId: model.id,
283
+ meritId: canonicalModel(model.provider, model.id),
284
+ input: model.input ?? ["text"],
285
+ cost: model.cost && typeof model.cost.input === "number" && typeof model.cost.output === "number"
286
+ ? { input: model.cost.input, output: model.cost.output,
287
+ cacheRead: model.cost.cacheRead, cacheWrite: model.cost.cacheWrite }
288
+ : undefined,
289
+ contextWindow: model.contextWindow,
290
+ maxTokens: model.maxTokens,
291
+ };
292
+ }
293
+
294
+ function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
295
+ const registry = ctx.modelRegistry as unknown as { getAvailable?: () => Array<Parameters<typeof routeFor>[0]> };
296
+ const scoped = (ctx.scopedModels ?? []) as readonly { model: Parameters<typeof routeFor>[0] }[];
297
+ const models = scoped.length ? scoped.map((item) => item.model) : registry.getAvailable?.() ?? [];
298
+ if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
299
+ models.unshift(ctx.model);
300
+ const byRoute = new Map<string, ModelRoute>();
301
+ for (const model of models) {
302
+ const route = routeFor(model);
303
+ byRoute.set(`${route.provider}:${route.modelId}`, route);
304
+ }
305
+ return [...byRoute.values()];
306
+ }
246
307
 
247
308
  function reportHarnessState(ctx: ExtensionContext, settled = false): void {
248
309
  try {
249
310
  mkdirSync(HOME, { recursive: true });
250
311
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
312
+ const currentRoute = ctx.model ? routeFor(ctx.model) : null;
251
313
  if (settled) settledTaskKey = latestSessionTaskKey(ctx);
252
314
  const prior = existsSync(STATE_FILE)
253
315
  ? JSON.parse(readFileSync(STATE_FILE, "utf8")) as { sessionFile?: string | null; settledAt?: string }
@@ -258,8 +320,12 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
258
320
  writeFileSync(
259
321
  STATE_FILE,
260
322
  JSON.stringify({
323
+ schemaVersion: 1,
261
324
  currentModel: model,
325
+ currentRoute,
326
+ routes: eligibleRoutes(ctx),
262
327
  fallbackModel,
328
+ fallbackRoute,
263
329
  sessionFile,
264
330
  settledTaskKey,
265
331
  settledAt,
@@ -275,8 +341,11 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
275
341
  /** Restore the persisted fallback so a fresh session keeps it even with no pending recs. */
276
342
  function restoreFallback(): void {
277
343
  try {
278
- const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as { fallbackModel?: string | null };
344
+ const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
345
+ fallbackModel?: string | null; fallbackRoute?: ModelRoute | null;
346
+ };
279
347
  if (st.fallbackModel) fallbackModel = st.fallbackModel;
348
+ if (st.fallbackRoute) fallbackRoute = st.fallbackRoute;
280
349
  } catch {
281
350
  /* no state yet */
282
351
  }
@@ -331,7 +400,15 @@ function catalogEntry(openrouterId: string): CatalogEntryLite | undefined {
331
400
  * OpenRouter catalog, so inject anything missing via registerProvider
332
401
  * (takes effect immediately, no /reload needed).
333
402
  */
334
- function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string) {
403
+ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string, route?: ModelRoute) {
404
+ const registry = ctx.modelRegistry as unknown as {
405
+ find?: (provider: string, id: string) => unknown | undefined;
406
+ };
407
+ if (route) {
408
+ const exact = registry.find?.(route.provider, route.modelId);
409
+ if (exact) return exact;
410
+ if (route.provider !== "openrouter") return undefined;
411
+ }
335
412
  const existing = resolveModel(ctx, openrouterId);
336
413
  if (existing) return existing;
337
414
 
@@ -359,16 +436,14 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
359
436
  })),
360
437
  });
361
438
 
362
- const reg = ctx.modelRegistry as unknown as {
363
- find?: (provider: string, id: string) => unknown | undefined;
364
- };
365
- return reg.find?.(INJECT_PROVIDER, openrouterId);
439
+ return registry.find?.(INJECT_PROVIDER, openrouterId);
366
440
  }
367
441
 
368
442
  export default function openmerit(pi: ExtensionAPI) {
369
443
  let pollTimer: ReturnType<typeof setInterval> | null = null;
370
444
  let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
371
- const queuedJobs: { sessionFile: string; settledAt: string; model: string; key: string; cwd: string; bytes: number }[] = [];
445
+ const queuedJobs: { sessionFile: string; settledAt: string; model: string; route: ModelRoute;
446
+ key: string; cwd: string; bytes: number }[] = [];
372
447
  const startedJobs = new Set<string>();
373
448
  let activeSession: string | null = null;
374
449
  let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
@@ -395,7 +470,7 @@ export default function openmerit(pi: ExtensionAPI) {
395
470
  return;
396
471
  }
397
472
  const child = spawn(process.execPath, [ENGINE_FILE, "session-trial", job.sessionFile,
398
- job.settledAt, job.model, job.key, job.cwd, String(job.bytes)], {
473
+ job.settledAt, job.model, job.key, job.cwd, String(job.bytes), job.route.provider, job.route.modelId], {
399
474
  cwd: job.cwd, env: process.env, stdio: ["ignore", "pipe", "pipe"],
400
475
  detached: process.platform !== "win32",
401
476
  });
@@ -453,7 +528,8 @@ export default function openmerit(pi: ExtensionAPI) {
453
528
  function queueSettledTask(ctx: ExtensionContext): void {
454
529
  const sessionFile = ctx.sessionManager.getSessionFile();
455
530
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
456
- if (!sessionFile || !model || !settledTaskKey || !existsSync(sessionFile)) return;
531
+ const route = ctx.model ? routeFor(ctx.model) : null;
532
+ if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
457
533
  // The alpha compares completed response tasks. The engine validates images,
458
534
  // tool use and the baseline answer before any provider calls.
459
535
  let completed = false;
@@ -477,7 +553,7 @@ export default function openmerit(pi: ExtensionAPI) {
477
553
  if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
478
554
  return;
479
555
  }
480
- queuedJobs.push({ sessionFile, settledAt, model, key: settledTaskKey, cwd: ctx.cwd, bytes });
556
+ queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
481
557
  startNextJob(ctx);
482
558
  }
483
559
 
@@ -489,11 +565,18 @@ export default function openmerit(pi: ExtensionAPI) {
489
565
 
490
566
  async function applyRecommendation(ctx: ExtensionContext, rec: Recommendation): Promise<boolean> {
491
567
  const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
492
- if (rec.currentModel && current !== rec.currentModel) {
568
+ const currentRoute = ctx.model ? routeFor(ctx.model) : null;
569
+ if (rec.currentRoute && currentRoute &&
570
+ (rec.currentRoute.provider !== currentRoute.provider || rec.currentRoute.modelId !== currentRoute.modelId)) {
571
+ ctx.ui.notify(`openmerit: recommendation targets ${rec.currentRoute.provider}/${rec.currentRoute.modelId}; ` +
572
+ `current route is ${currentRoute.provider}/${currentRoute.modelId}`, "warning");
573
+ return false;
574
+ }
575
+ if (!rec.currentRoute && rec.currentModel && current !== rec.currentModel) {
493
576
  ctx.ui.notify(`openmerit: recommendation was measured against ${rec.currentModel}; current model is ${current ?? "unknown"}`, "warning");
494
577
  return false;
495
578
  }
496
- const model = ensureModel(pi, ctx, rec.recommended.model);
579
+ const model = ensureModel(pi, ctx, rec.recommended.model, rec.recommended.route);
497
580
  if (!model) {
498
581
  ctx.ui.notify(
499
582
  `openmerit: ${rec.recommended.model} not found in pi's model registry (add the provider/model first)`,
@@ -504,8 +587,12 @@ export default function openmerit(pi: ExtensionAPI) {
504
587
  const ok = await pi.setModel(model as Parameters<typeof pi.setModel>[0]);
505
588
  if (ok) {
506
589
  setStatus(rec, "applied");
507
- if (rec.fallback) fallbackModel = rec.fallback.model;
508
- ctx.ui.notify(`openmerit: switched to ${rec.recommended.model}\n${rec.recommended.reason}`, "info");
590
+ if (rec.fallback) {
591
+ fallbackModel = rec.fallback.model;
592
+ fallbackRoute = rec.fallback.route ?? null;
593
+ }
594
+ ctx.ui.notify(`openmerit: switched to ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n` +
595
+ rec.recommended.reason, "info");
509
596
  reportHarnessState(ctx);
510
597
  } else {
511
598
  ctx.ui.notify(`openmerit: no auth for ${rec.recommended.model}`, "error");
@@ -537,7 +624,7 @@ export default function openmerit(pi: ExtensionAPI) {
537
624
  // startup. Notify (fire-and-forget) and let the user act via /openmerit.
538
625
  if (ctx.hasUI) {
539
626
  ctx.ui.notify(
540
- `openmerit recommends: ${rec.recommended.model}\n${rec.recommended.reason}\n` +
627
+ `openmerit recommends: ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n${rec.recommended.reason}\n` +
541
628
  `gate: ${rec.policy.reasons.join("; ")}\n(/openmerit apply to switch, /openmerit dismiss to ignore)`,
542
629
  "info",
543
630
  );
@@ -608,7 +695,7 @@ export default function openmerit(pi: ExtensionAPI) {
608
695
  const policy = loadPolicy();
609
696
  if (policy.fallback?.apply_on_error === false) return;
610
697
  const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "current model";
611
- const model = ensureModel(pi, ctx, fallbackModel);
698
+ const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
612
699
  if (!model) return;
613
700
 
614
701
  const auto = policy.mode === "auto";
@@ -660,7 +747,7 @@ export default function openmerit(pi: ExtensionAPI) {
660
747
  ctx.ui.notify("openmerit: no fallback set yet", "info");
661
748
  return;
662
749
  }
663
- const model = ensureModel(pi, ctx, fallbackModel);
750
+ const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
664
751
  if (model && (await pi.setModel(model as Parameters<typeof pi.setModel>[0]))) {
665
752
  ctx.ui.notify(`openmerit: switched to fallback ${fallbackModel}`, "info");
666
753
  reportHarnessState(ctx);
@@ -670,7 +757,8 @@ export default function openmerit(pi: ExtensionAPI) {
670
757
  return;
671
758
  }
672
759
 
673
- const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "unknown";
760
+ const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
761
+ const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
674
762
  const budget = trialBudget();
675
763
  const liveProgress = progress?.sessionFile === ctx.sessionManager.getSessionFile() ? progress : null;
676
764
  const comparison = trialJob ? liveProgress?.current
@@ -693,7 +781,8 @@ export default function openmerit(pi: ExtensionAPI) {
693
781
  for (const r of pending.slice(-5)) {
694
782
  lines.push(
695
783
  "",
696
- `-> ${r.recommended.model} (gain ${r.evidence.scoreGain}, ${r.evidence.priceRatio}x price, frontier=${r.evidence.onFrontier})`,
784
+ `-> ${routeDisplay(r.recommended.route, r.recommended.model)} (gain ${r.evidence.scoreGain}, ` +
785
+ `${r.evidence.priceKnown === false ? "price unknown" : `${r.evidence.priceRatio}x price`}, frontier=${r.evidence.onFrontier})`,
697
786
  ` ${r.recommended.reason}`,
698
787
  ` gate: ${r.policy.reasons.join("; ") || "n/a"}`,
699
788
  );
@@ -8,8 +8,9 @@ the best model for the task at hand, at the best price, with a vetted fallback.
8
8
 
9
9
  1. **Observes** this harness's session traces (models used, tokens, cost,
10
10
  latency, errors) without intercepting or slowing down your work.
11
- 2. **Evaluates** candidate models for each completed text or image task sequentially
12
- through pi and scores their outputs with a judge model.
11
+ 2. **Evaluates** candidate models for each completed task sequentially through
12
+ pi and scores their outputs with a judge model. Candidate tools are disabled
13
+ by default.
13
14
  3. **Maintains a pareto frontier** per task (quality vs. cost vs. latency) and
14
15
  an aggregate frontier across all of your tasks.
15
16
  4. **Watches for new model releases** (provider catalogs + public benchmarks)
@@ -43,9 +44,11 @@ the best model for the task at hand, at the best price, with a vetted fallback.
43
44
 
44
45
  ## Guarantees
45
46
 
46
- - Per-task comparisons send the task text, uploaded images (when present), and
47
- candidate answers through pi/OpenRouter for model runs and judging. Use
48
- non-sensitive examples while evaluating this alpha.
47
+ - Per-task comparisons send the task text, uploaded files or images (when
48
+ present), and candidate answers through the model routes configured in Pi
49
+ for model runs and judging. Candidate Pi runs default to no tools; any tool access is an
50
+ explicit user opt-in. Use non-sensitive examples while
51
+ evaluating this alpha.
49
52
  - The shipped policy is supervised. Swaps happen automatically only after the
50
53
  user opts in and the configured quality/cost guardrails pass; otherwise they
51
54
  remain recommendations for a human to approve.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmerit",
3
- "version": "0.1.0",
3
+ "version": "0.1.2",
4
4
  "description": "Find better models for each pi task by comparing quality, cost, and latency.",
5
5
  "type": "module",
6
6
  "keywords": [
@@ -17,10 +17,13 @@
17
17
  "extension/openmerit.ts",
18
18
  "instructions/",
19
19
  "examples/",
20
+ "rules.md",
20
21
  "benchmark/invoice_ocr/data/"
21
22
  ],
22
23
  "pi": {
23
- "extensions": ["./extension/openmerit.ts"]
24
+ "extensions": [
25
+ "./extension/openmerit.ts"
26
+ ]
24
27
  },
25
28
  "bin": {
26
29
  "openmerit": "dist/cli.js"
package/rules.md ADDED
@@ -0,0 +1,39 @@
1
+ # OpenMerit product and architecture rules
2
+
3
+ - **Optimize for useful model choices.** OpenMerit exists to find better task-specific tradeoffs between quality, cost, and latency—not to become a general-purpose agent or telemetry platform.
4
+
5
+ - **Act as a control plane, not a gateway.** Normal traffic stays between the harness and provider; OpenMerit observes evidence, runs explicit trials, and returns recommendations without intercepting ordinary work.
6
+
7
+ - **Make Pi plug-and-play first.** Pi and Pi-based harnesses are the first supported user experience because that is where early users already work; broader harness support can follow without weakening this path.
8
+
9
+ - **Keep the core harness-neutral.** Pi is the first `HarnessAdapter`, not a permanent assumption in scoring, recommendations, policy, or stored data, so other harnesses can be added without rewriting the merit loop.
10
+
11
+ - **Keep model providers interchangeable.** OpenRouter is a first-class provider and discovery source, not the foundation of the domain model; provider-specific behavior belongs behind a `ModelProviderAdapter`.
12
+
13
+ - **Require end-to-end provider neutrality.** Discovery, authentication, execution, scoring, and swapping must preserve the selected route; a feature is not provider-neutral if only its API client is abstracted.
14
+
15
+ - **Separate model identity from execution route.** Keep logical model IDs in stable `vendor/model` form, while recording the actual provider or route separately, so the same model can be compared through OpenRouter, a native provider, or a local provider.
16
+
17
+ - **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
18
+
19
+ - **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
20
+
21
+ - **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
22
+
23
+ - **Use provider-neutral core records.** Normalize integrations into stable concepts such as `TaskObservation`, `CandidateRun`, `TrialScore`, `Frontier`, `Recommendation`, and `MeritEvent`, so integrations do not leak their schemas into the decision engine.
24
+
25
+ - **Optimize per task before aggregating per agent.** Agent-level conclusions must be built from measured task evidence rather than assumed from global model rankings.
26
+
27
+ - **Version persisted events and evolve them additively.** Existing 0.1.x state must remain readable, and append-only recommendation history must stay intact; migrations should normalize old records rather than invalidate them.
28
+
29
+ - **Keep policy as the sole auto-swap authority.** Trials and strategists may recommend changes, but only the policy gate may approve automatic application, and every recommendation must retain its gate reasons.
30
+
31
+ - **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
32
+
33
+ - **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
34
+
35
+ - **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.
36
+
37
+ - **Keep integrations optional and the core lightweight.** New providers, harnesses, and observability services should not impose credentials, network calls, or heavy dependencies on users who do not enable them.
38
+
39
+ - **Keep 0.1.x focused.** The near-term bar is reliable, provider-neutral use for Pi tinkerers and solo hackers; additional harnesses and hosted observability integrations belong in later releases unless required to prove the boundaries work.