openmerit 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -46
- package/dist/cli.js +62 -8
- package/dist/daemon.js +104 -48
- package/dist/frontier.js +13 -6
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +102 -1
- package/dist/pi-trials.js +46 -23
- package/dist/policy.js +8 -3
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/store.js +1 -0
- package/dist/strategist.js +18 -14
- package/dist/traces.js +6 -2
- package/dist/trials.js +38 -8
- package/extension/openmerit.ts +99 -23
- package/instructions/OPENMERIT.md +2 -2
- package/package.json +5 -2
- package/rules.md +39 -0
package/dist/trials.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/** Shadow trials: run candidate models against observed/declared tasks on the background track. */
|
|
2
|
-
import {
|
|
2
|
+
import { directChatClient } from "./llm.js";
|
|
3
3
|
import { judge, parseObj } from "./judge.js";
|
|
4
4
|
import { paths, readJson, writeJson } from "./store.js";
|
|
5
5
|
function today() {
|
|
@@ -15,6 +15,21 @@ export function budgetOk(policy) {
|
|
|
15
15
|
return { ok: false, reason: `daily budget exhausted ($${policy.budgets.max_usd_per_day})` };
|
|
16
16
|
return { ok: true };
|
|
17
17
|
}
|
|
18
|
+
/** Conservative admission estimate for one candidate answer (rough input tokens + 4K output). */
|
|
19
|
+
export function trialBudgetOk(policy, entry, inputChars) {
|
|
20
|
+
if (entry.priceKnown === false)
|
|
21
|
+
return { ok: false, reason: "route price is unknown" };
|
|
22
|
+
const inputTokens = Math.ceil(inputChars / 4);
|
|
23
|
+
const outputTokens = Math.min(entry.route?.maxTokens ?? 4096, 4096);
|
|
24
|
+
const estimatedUsd = entry.pp * inputTokens + entry.pc * outputTokens;
|
|
25
|
+
if (estimatedUsd > policy.budgets.max_usd_per_trial)
|
|
26
|
+
return {
|
|
27
|
+
ok: false,
|
|
28
|
+
estimatedUsd,
|
|
29
|
+
reason: `estimated trial cost $${estimatedUsd.toFixed(4)} exceeds $${policy.budgets.max_usd_per_trial.toFixed(4)} limit`,
|
|
30
|
+
};
|
|
31
|
+
return { ok: true, estimatedUsd };
|
|
32
|
+
}
|
|
18
33
|
export function recordTrialSpend(usd) {
|
|
19
34
|
let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
20
35
|
if (ledger.date !== today())
|
|
@@ -23,16 +38,26 @@ export function recordTrialSpend(usd) {
|
|
|
23
38
|
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
24
39
|
writeJson(paths.ledger(), ledger);
|
|
25
40
|
}
|
|
41
|
+
/** Record non-candidate merit-loop spend without consuming a trial slot. */
|
|
42
|
+
export function recordMeritSpend(usd) {
|
|
43
|
+
if (!(usd > 0))
|
|
44
|
+
return;
|
|
45
|
+
let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
46
|
+
if (ledger.date !== today())
|
|
47
|
+
ledger = { date: today(), trials: 0, usd: 0 };
|
|
48
|
+
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
49
|
+
writeJson(paths.ledger(), ledger);
|
|
50
|
+
}
|
|
26
51
|
const RUBRIC_PROMPT = `Write a compact grading rubric for answers to the task below.
|
|
27
52
|
It must list the concrete criteria for a score of 1.0 and how to deduct.
|
|
28
53
|
TASK: {task}
|
|
29
54
|
Return ONLY json: {{"rubric": "<text>"}}`;
|
|
30
55
|
/** Derive (and cache) a grading rubric for a task that came from traces. */
|
|
31
|
-
export async function ensureRubric(key, judgeModel, task, cache) {
|
|
56
|
+
export async function ensureRubric(key, judgeModel, task, cache, client = directChatClient(key)) {
|
|
32
57
|
const hit = cache.get(task);
|
|
33
58
|
if (hit)
|
|
34
59
|
return hit;
|
|
35
|
-
const { content } = await chat(
|
|
60
|
+
const { content } = await client.chat(judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
|
|
36
61
|
let rubric = "Score 1.0 for a fully correct, complete, usable answer; deduct for errors, omissions, or unusable output.";
|
|
37
62
|
try {
|
|
38
63
|
const obj = parseObj(content);
|
|
@@ -46,19 +71,20 @@ export async function ensureRubric(key, judgeModel, task, cache) {
|
|
|
46
71
|
return rubric;
|
|
47
72
|
}
|
|
48
73
|
/** Run one shadow trial: generate with `model`, judge the output, return a frontier point + cost. */
|
|
49
|
-
export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
74
|
+
export async function runTrial(key, judgeModel, task, rubric, model, entry, client = directChatClient(key)) {
|
|
50
75
|
const started = Date.now();
|
|
51
76
|
let gen;
|
|
52
77
|
let usage = {};
|
|
53
78
|
try {
|
|
54
|
-
const r = await chat(
|
|
79
|
+
const r = await client.chat(entry?.route ?? model, task, 4096, 0.2);
|
|
55
80
|
gen = r.content;
|
|
56
81
|
usage = r.usage;
|
|
57
82
|
}
|
|
58
83
|
catch (e) {
|
|
59
84
|
return {
|
|
60
85
|
point: {
|
|
61
|
-
model, score: 0, price: entry?.price ?? 0,
|
|
86
|
+
schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
87
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
62
88
|
ts: new Date().toISOString(), source: "shadow_trial", why: String(e).slice(0, 160),
|
|
63
89
|
},
|
|
64
90
|
costUsd: 0,
|
|
@@ -70,7 +96,8 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
|
70
96
|
if (!gen.trim()) {
|
|
71
97
|
return {
|
|
72
98
|
point: {
|
|
73
|
-
model, score: 0, price: entry?.price ?? 0,
|
|
99
|
+
schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
100
|
+
priceKnown: entry?.priceKnown ?? !!entry, latencyMs,
|
|
74
101
|
ts: new Date().toISOString(), source: "shadow_trial", why: "empty response",
|
|
75
102
|
},
|
|
76
103
|
costUsd,
|
|
@@ -79,7 +106,7 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
|
79
106
|
let score = 0;
|
|
80
107
|
let why = "judge error";
|
|
81
108
|
try {
|
|
82
|
-
const j = await judge(key, judgeModel, task, rubric, gen);
|
|
109
|
+
const j = await judge(key, judgeModel, task, rubric, gen, client);
|
|
83
110
|
score = j.score;
|
|
84
111
|
why = j.why;
|
|
85
112
|
}
|
|
@@ -88,9 +115,12 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
|
88
115
|
}
|
|
89
116
|
return {
|
|
90
117
|
point: {
|
|
118
|
+
schemaVersion: 1,
|
|
91
119
|
model,
|
|
120
|
+
route: entry?.route,
|
|
92
121
|
score,
|
|
93
122
|
price: entry?.price ?? 0,
|
|
123
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
94
124
|
latencyMs,
|
|
95
125
|
ts: new Date().toISOString(),
|
|
96
126
|
source: "shadow_trial",
|
package/extension/openmerit.ts
CHANGED
|
@@ -115,19 +115,31 @@ function loadPolicy(): ExtensionPolicy {
|
|
|
115
115
|
}
|
|
116
116
|
|
|
117
117
|
interface Recommendation {
|
|
118
|
+
schemaVersion?: 1;
|
|
118
119
|
id: string;
|
|
119
120
|
ts: string;
|
|
120
121
|
taskKey: string;
|
|
121
122
|
taskLabel: string;
|
|
122
123
|
sessionFile?: string | null;
|
|
123
124
|
currentModel: string | null;
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
125
|
+
currentRoute?: ModelRoute | null;
|
|
126
|
+
recommended: { model: string; route?: ModelRoute; reason: string };
|
|
127
|
+
fallback: { model: string; route?: ModelRoute; reason: string } | null;
|
|
128
|
+
evidence: { scoreGain: number; priceRatio: number; priceKnown?: boolean; onFrontier: boolean; trials: number };
|
|
127
129
|
policy: { autoApply: boolean; reasons: string[] };
|
|
128
130
|
status: "pending" | "applied" | "dismissed" | "expired";
|
|
129
131
|
}
|
|
130
132
|
|
|
133
|
+
interface ModelRoute {
|
|
134
|
+
provider: string;
|
|
135
|
+
modelId: string;
|
|
136
|
+
meritId: string;
|
|
137
|
+
input: ("text" | "image")[];
|
|
138
|
+
cost?: { input: number; output: number; cacheRead?: number; cacheWrite?: number };
|
|
139
|
+
contextWindow?: number;
|
|
140
|
+
maxTokens?: number;
|
|
141
|
+
}
|
|
142
|
+
|
|
131
143
|
interface RecordedTrial {
|
|
132
144
|
taskKey?: string;
|
|
133
145
|
sessionFile?: string;
|
|
@@ -248,7 +260,11 @@ function suggestionText(rec: Recommendation, trials: RecordedTrial[]): string {
|
|
|
248
260
|
}
|
|
249
261
|
|
|
250
262
|
function canonicalModel(provider: string, id: string): string {
|
|
251
|
-
return provider === "openrouter" ? id : `${provider}/${id}`;
|
|
263
|
+
return provider === "openrouter" || id.startsWith(`${provider}/`) ? id : `${provider}/${id}`;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
function routeDisplay(route: ModelRoute | undefined, fallback: string): string {
|
|
267
|
+
return route && route.provider !== "openrouter" ? `${route.meritId} via ${route.provider}` : fallback;
|
|
252
268
|
}
|
|
253
269
|
|
|
254
270
|
function setStatus(rec: Recommendation, status: Recommendation["status"]): void {
|
|
@@ -256,11 +272,44 @@ function setStatus(rec: Recommendation, status: Recommendation["status"]): void
|
|
|
256
272
|
}
|
|
257
273
|
|
|
258
274
|
let fallbackModel: string | null = null;
|
|
275
|
+
let fallbackRoute: ModelRoute | null = null;
|
|
276
|
+
|
|
277
|
+
function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
|
|
278
|
+
cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
|
|
279
|
+
contextWindow?: number; maxTokens?: number }): ModelRoute {
|
|
280
|
+
return {
|
|
281
|
+
provider: model.provider,
|
|
282
|
+
modelId: model.id,
|
|
283
|
+
meritId: canonicalModel(model.provider, model.id),
|
|
284
|
+
input: model.input ?? ["text"],
|
|
285
|
+
cost: model.cost && typeof model.cost.input === "number" && typeof model.cost.output === "number"
|
|
286
|
+
? { input: model.cost.input, output: model.cost.output,
|
|
287
|
+
cacheRead: model.cost.cacheRead, cacheWrite: model.cost.cacheWrite }
|
|
288
|
+
: undefined,
|
|
289
|
+
contextWindow: model.contextWindow,
|
|
290
|
+
maxTokens: model.maxTokens,
|
|
291
|
+
};
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
|
|
295
|
+
const registry = ctx.modelRegistry as unknown as { getAvailable?: () => Array<Parameters<typeof routeFor>[0]> };
|
|
296
|
+
const scoped = (ctx.scopedModels ?? []) as readonly { model: Parameters<typeof routeFor>[0] }[];
|
|
297
|
+
const models = scoped.length ? scoped.map((item) => item.model) : registry.getAvailable?.() ?? [];
|
|
298
|
+
if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
|
|
299
|
+
models.unshift(ctx.model);
|
|
300
|
+
const byRoute = new Map<string, ModelRoute>();
|
|
301
|
+
for (const model of models) {
|
|
302
|
+
const route = routeFor(model);
|
|
303
|
+
byRoute.set(`${route.provider}:${route.modelId}`, route);
|
|
304
|
+
}
|
|
305
|
+
return [...byRoute.values()];
|
|
306
|
+
}
|
|
259
307
|
|
|
260
308
|
function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
261
309
|
try {
|
|
262
310
|
mkdirSync(HOME, { recursive: true });
|
|
263
311
|
const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
312
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : null;
|
|
264
313
|
if (settled) settledTaskKey = latestSessionTaskKey(ctx);
|
|
265
314
|
const prior = existsSync(STATE_FILE)
|
|
266
315
|
? JSON.parse(readFileSync(STATE_FILE, "utf8")) as { sessionFile?: string | null; settledAt?: string }
|
|
@@ -271,8 +320,12 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
271
320
|
writeFileSync(
|
|
272
321
|
STATE_FILE,
|
|
273
322
|
JSON.stringify({
|
|
323
|
+
schemaVersion: 1,
|
|
274
324
|
currentModel: model,
|
|
325
|
+
currentRoute,
|
|
326
|
+
routes: eligibleRoutes(ctx),
|
|
275
327
|
fallbackModel,
|
|
328
|
+
fallbackRoute,
|
|
276
329
|
sessionFile,
|
|
277
330
|
settledTaskKey,
|
|
278
331
|
settledAt,
|
|
@@ -288,8 +341,11 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
288
341
|
/** Restore the persisted fallback so a fresh session keeps it even with no pending recs. */
|
|
289
342
|
function restoreFallback(): void {
|
|
290
343
|
try {
|
|
291
|
-
const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
|
|
344
|
+
const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
|
|
345
|
+
fallbackModel?: string | null; fallbackRoute?: ModelRoute | null;
|
|
346
|
+
};
|
|
292
347
|
if (st.fallbackModel) fallbackModel = st.fallbackModel;
|
|
348
|
+
if (st.fallbackRoute) fallbackRoute = st.fallbackRoute;
|
|
293
349
|
} catch {
|
|
294
350
|
/* no state yet */
|
|
295
351
|
}
|
|
@@ -344,7 +400,15 @@ function catalogEntry(openrouterId: string): CatalogEntryLite | undefined {
|
|
|
344
400
|
* OpenRouter catalog, so inject anything missing via registerProvider
|
|
345
401
|
* (takes effect immediately, no /reload needed).
|
|
346
402
|
*/
|
|
347
|
-
function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string) {
|
|
403
|
+
function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string, route?: ModelRoute) {
|
|
404
|
+
const registry = ctx.modelRegistry as unknown as {
|
|
405
|
+
find?: (provider: string, id: string) => unknown | undefined;
|
|
406
|
+
};
|
|
407
|
+
if (route) {
|
|
408
|
+
const exact = registry.find?.(route.provider, route.modelId);
|
|
409
|
+
if (exact) return exact;
|
|
410
|
+
if (route.provider !== "openrouter") return undefined;
|
|
411
|
+
}
|
|
348
412
|
const existing = resolveModel(ctx, openrouterId);
|
|
349
413
|
if (existing) return existing;
|
|
350
414
|
|
|
@@ -372,16 +436,14 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
|
|
|
372
436
|
})),
|
|
373
437
|
});
|
|
374
438
|
|
|
375
|
-
|
|
376
|
-
find?: (provider: string, id: string) => unknown | undefined;
|
|
377
|
-
};
|
|
378
|
-
return reg.find?.(INJECT_PROVIDER, openrouterId);
|
|
439
|
+
return registry.find?.(INJECT_PROVIDER, openrouterId);
|
|
379
440
|
}
|
|
380
441
|
|
|
381
442
|
export default function openmerit(pi: ExtensionAPI) {
|
|
382
443
|
let pollTimer: ReturnType<typeof setInterval> | null = null;
|
|
383
444
|
let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
|
|
384
|
-
const queuedJobs: { sessionFile: string; settledAt: string; model: string;
|
|
445
|
+
const queuedJobs: { sessionFile: string; settledAt: string; model: string; route: ModelRoute;
|
|
446
|
+
key: string; cwd: string; bytes: number }[] = [];
|
|
385
447
|
const startedJobs = new Set<string>();
|
|
386
448
|
let activeSession: string | null = null;
|
|
387
449
|
let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
|
|
@@ -408,7 +470,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
408
470
|
return;
|
|
409
471
|
}
|
|
410
472
|
const child = spawn(process.execPath, [ENGINE_FILE, "session-trial", job.sessionFile,
|
|
411
|
-
job.settledAt, job.model, job.key, job.cwd, String(job.bytes)], {
|
|
473
|
+
job.settledAt, job.model, job.key, job.cwd, String(job.bytes), job.route.provider, job.route.modelId], {
|
|
412
474
|
cwd: job.cwd, env: process.env, stdio: ["ignore", "pipe", "pipe"],
|
|
413
475
|
detached: process.platform !== "win32",
|
|
414
476
|
});
|
|
@@ -466,7 +528,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
466
528
|
function queueSettledTask(ctx: ExtensionContext): void {
|
|
467
529
|
const sessionFile = ctx.sessionManager.getSessionFile();
|
|
468
530
|
const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
469
|
-
|
|
531
|
+
const route = ctx.model ? routeFor(ctx.model) : null;
|
|
532
|
+
if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
|
|
470
533
|
// The alpha compares completed response tasks. The engine validates images,
|
|
471
534
|
// tool use and the baseline answer before any provider calls.
|
|
472
535
|
let completed = false;
|
|
@@ -490,7 +553,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
490
553
|
if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
|
|
491
554
|
return;
|
|
492
555
|
}
|
|
493
|
-
queuedJobs.push({ sessionFile, settledAt, model, key: settledTaskKey, cwd: ctx.cwd, bytes });
|
|
556
|
+
queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
|
|
494
557
|
startNextJob(ctx);
|
|
495
558
|
}
|
|
496
559
|
|
|
@@ -502,11 +565,18 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
502
565
|
|
|
503
566
|
async function applyRecommendation(ctx: ExtensionContext, rec: Recommendation): Promise<boolean> {
|
|
504
567
|
const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
505
|
-
|
|
568
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : null;
|
|
569
|
+
if (rec.currentRoute && currentRoute &&
|
|
570
|
+
(rec.currentRoute.provider !== currentRoute.provider || rec.currentRoute.modelId !== currentRoute.modelId)) {
|
|
571
|
+
ctx.ui.notify(`openmerit: recommendation targets ${rec.currentRoute.provider}/${rec.currentRoute.modelId}; ` +
|
|
572
|
+
`current route is ${currentRoute.provider}/${currentRoute.modelId}`, "warning");
|
|
573
|
+
return false;
|
|
574
|
+
}
|
|
575
|
+
if (!rec.currentRoute && rec.currentModel && current !== rec.currentModel) {
|
|
506
576
|
ctx.ui.notify(`openmerit: recommendation was measured against ${rec.currentModel}; current model is ${current ?? "unknown"}`, "warning");
|
|
507
577
|
return false;
|
|
508
578
|
}
|
|
509
|
-
const model = ensureModel(pi, ctx, rec.recommended.model);
|
|
579
|
+
const model = ensureModel(pi, ctx, rec.recommended.model, rec.recommended.route);
|
|
510
580
|
if (!model) {
|
|
511
581
|
ctx.ui.notify(
|
|
512
582
|
`openmerit: ${rec.recommended.model} not found in pi's model registry (add the provider/model first)`,
|
|
@@ -517,8 +587,12 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
517
587
|
const ok = await pi.setModel(model as Parameters<typeof pi.setModel>[0]);
|
|
518
588
|
if (ok) {
|
|
519
589
|
setStatus(rec, "applied");
|
|
520
|
-
if (rec.fallback)
|
|
521
|
-
|
|
590
|
+
if (rec.fallback) {
|
|
591
|
+
fallbackModel = rec.fallback.model;
|
|
592
|
+
fallbackRoute = rec.fallback.route ?? null;
|
|
593
|
+
}
|
|
594
|
+
ctx.ui.notify(`openmerit: switched to ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n` +
|
|
595
|
+
rec.recommended.reason, "info");
|
|
522
596
|
reportHarnessState(ctx);
|
|
523
597
|
} else {
|
|
524
598
|
ctx.ui.notify(`openmerit: no auth for ${rec.recommended.model}`, "error");
|
|
@@ -550,7 +624,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
550
624
|
// startup. Notify (fire-and-forget) and let the user act via /openmerit.
|
|
551
625
|
if (ctx.hasUI) {
|
|
552
626
|
ctx.ui.notify(
|
|
553
|
-
`openmerit recommends: ${rec.recommended.model}\n${rec.recommended.reason}\n` +
|
|
627
|
+
`openmerit recommends: ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n${rec.recommended.reason}\n` +
|
|
554
628
|
`gate: ${rec.policy.reasons.join("; ")}\n(/openmerit apply to switch, /openmerit dismiss to ignore)`,
|
|
555
629
|
"info",
|
|
556
630
|
);
|
|
@@ -621,7 +695,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
621
695
|
const policy = loadPolicy();
|
|
622
696
|
if (policy.fallback?.apply_on_error === false) return;
|
|
623
697
|
const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "current model";
|
|
624
|
-
const model = ensureModel(pi, ctx, fallbackModel);
|
|
698
|
+
const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
|
|
625
699
|
if (!model) return;
|
|
626
700
|
|
|
627
701
|
const auto = policy.mode === "auto";
|
|
@@ -673,7 +747,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
673
747
|
ctx.ui.notify("openmerit: no fallback set yet", "info");
|
|
674
748
|
return;
|
|
675
749
|
}
|
|
676
|
-
const model = ensureModel(pi, ctx, fallbackModel);
|
|
750
|
+
const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
|
|
677
751
|
if (model && (await pi.setModel(model as Parameters<typeof pi.setModel>[0]))) {
|
|
678
752
|
ctx.ui.notify(`openmerit: switched to fallback ${fallbackModel}`, "info");
|
|
679
753
|
reportHarnessState(ctx);
|
|
@@ -683,7 +757,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
683
757
|
return;
|
|
684
758
|
}
|
|
685
759
|
|
|
686
|
-
const
|
|
760
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
|
|
761
|
+
const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
|
|
687
762
|
const budget = trialBudget();
|
|
688
763
|
const liveProgress = progress?.sessionFile === ctx.sessionManager.getSessionFile() ? progress : null;
|
|
689
764
|
const comparison = trialJob ? liveProgress?.current
|
|
@@ -706,7 +781,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
706
781
|
for (const r of pending.slice(-5)) {
|
|
707
782
|
lines.push(
|
|
708
783
|
"",
|
|
709
|
-
`-> ${r.recommended.model} (gain ${r.evidence.scoreGain},
|
|
784
|
+
`-> ${routeDisplay(r.recommended.route, r.recommended.model)} (gain ${r.evidence.scoreGain}, ` +
|
|
785
|
+
`${r.evidence.priceKnown === false ? "price unknown" : `${r.evidence.priceRatio}x price`}, frontier=${r.evidence.onFrontier})`,
|
|
710
786
|
` ${r.recommended.reason}`,
|
|
711
787
|
` gate: ${r.policy.reasons.join("; ") || "n/a"}`,
|
|
712
788
|
);
|
|
@@ -45,8 +45,8 @@ the best model for the task at hand, at the best price, with a vetted fallback.
|
|
|
45
45
|
## Guarantees
|
|
46
46
|
|
|
47
47
|
- Per-task comparisons send the task text, uploaded files or images (when
|
|
48
|
-
present), and candidate answers through
|
|
49
|
-
judging. Candidate Pi runs default to no tools; any tool access is an
|
|
48
|
+
present), and candidate answers through the model routes configured in Pi
|
|
49
|
+
for model runs and judging. Candidate Pi runs default to no tools; any tool access is an
|
|
50
50
|
explicit user opt-in. Use non-sensitive examples while
|
|
51
51
|
evaluating this alpha.
|
|
52
52
|
- The shipped policy is supervised. Swaps happen automatically only after the
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "openmerit",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.2",
|
|
4
4
|
"description": "Find better models for each pi task by comparing quality, cost, and latency.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": [
|
|
@@ -17,10 +17,13 @@
|
|
|
17
17
|
"extension/openmerit.ts",
|
|
18
18
|
"instructions/",
|
|
19
19
|
"examples/",
|
|
20
|
+
"rules.md",
|
|
20
21
|
"benchmark/invoice_ocr/data/"
|
|
21
22
|
],
|
|
22
23
|
"pi": {
|
|
23
|
-
"extensions": [
|
|
24
|
+
"extensions": [
|
|
25
|
+
"./extension/openmerit.ts"
|
|
26
|
+
]
|
|
24
27
|
},
|
|
25
28
|
"bin": {
|
|
26
29
|
"openmerit": "dist/cli.js"
|
package/rules.md
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# OpenMerit product and architecture rules
|
|
2
|
+
|
|
3
|
+
- **Optimize for useful model choices.** OpenMerit exists to find better task-specific tradeoffs between quality, cost, and latency—not to become a general-purpose agent or telemetry platform.
|
|
4
|
+
|
|
5
|
+
- **Act as a control plane, not a gateway.** Normal traffic stays between the harness and provider; OpenMerit observes evidence, runs explicit trials, and returns recommendations without intercepting ordinary work.
|
|
6
|
+
|
|
7
|
+
- **Make Pi plug-and-play first.** Pi and Pi-based harnesses are the first supported user experience because that is where early users already work; broader harness support can follow without weakening this path.
|
|
8
|
+
|
|
9
|
+
- **Keep the core harness-neutral.** Pi is the first `HarnessAdapter`, not a permanent assumption in scoring, recommendations, policy, or stored data, so other harnesses can be added without rewriting the merit loop.
|
|
10
|
+
|
|
11
|
+
- **Keep model providers interchangeable.** OpenRouter is a first-class provider and discovery source, not the foundation of the domain model; provider-specific behavior belongs behind a `ModelProviderAdapter`.
|
|
12
|
+
|
|
13
|
+
- **Require end-to-end provider neutrality.** Discovery, authentication, execution, scoring, and swapping must preserve the selected route; a feature is not provider-neutral if only its API client is abstracted.
|
|
14
|
+
|
|
15
|
+
- **Separate model identity from execution route.** Keep logical model IDs in stable `vendor/model` form, while recording the actual provider or route separately, so the same model can be compared through OpenRouter, a native provider, or a local provider.
|
|
16
|
+
|
|
17
|
+
- **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
|
|
18
|
+
|
|
19
|
+
- **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
|
|
20
|
+
|
|
21
|
+
- **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
|
|
22
|
+
|
|
23
|
+
- **Use provider-neutral core records.** Normalize integrations into stable concepts such as `TaskObservation`, `CandidateRun`, `TrialScore`, `Frontier`, `Recommendation`, and `MeritEvent`, so integrations do not leak their schemas into the decision engine.
|
|
24
|
+
|
|
25
|
+
- **Optimize per task before aggregating per agent.** Agent-level conclusions must be built from measured task evidence rather than assumed from global model rankings.
|
|
26
|
+
|
|
27
|
+
- **Version persisted events and evolve them additively.** Existing 0.1.x state must remain readable, and append-only recommendation history must stay intact; migrations should normalize old records rather than invalidate them.
|
|
28
|
+
|
|
29
|
+
- **Keep policy as the sole auto-swap authority.** Trials and strategists may recommend changes, but only the policy gate may approve automatic application, and every recommendation must retain its gate reasons.
|
|
30
|
+
|
|
31
|
+
- **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
|
|
32
|
+
|
|
33
|
+
- **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
|
|
34
|
+
|
|
35
|
+
- **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.
|
|
36
|
+
|
|
37
|
+
- **Keep integrations optional and the core lightweight.** New providers, harnesses, and observability services should not impose credentials, network calls, or heavy dependencies on users who do not enable them.
|
|
38
|
+
|
|
39
|
+
- **Keep 0.1.x focused.** The near-term bar is reliable, provider-neutral use for Pi tinkerers and solo hackers; additional harnesses and hosted observability integrations belong in later releases unless required to prove the boundaries work.
|