openmerit 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +89 -47
- package/dist/cli.js +48 -121
- package/dist/daemon.js +131 -52
- package/dist/diagnostics.js +194 -0
- package/dist/frontier.js +13 -6
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +102 -1
- package/dist/pi-trials.js +60 -24
- package/dist/policy.js +8 -3
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/standalone.js +220 -0
- package/dist/store.js +43 -10
- package/dist/strategist.js +18 -14
- package/dist/traces.js +6 -2
- package/dist/trials.js +50 -11
- package/examples/task.example.json +1 -0
- package/extension/openmerit.ts +134 -32
- package/instructions/OPENMERIT.md +2 -2
- package/package.json +5 -2
- package/rules.md +39 -0
package/extension/openmerit.ts
CHANGED
|
@@ -17,7 +17,9 @@
|
|
|
17
17
|
*/
|
|
18
18
|
|
|
19
19
|
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
20
|
-
import {
|
|
20
|
+
import {
|
|
21
|
+
appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync,
|
|
22
|
+
} from "node:fs";
|
|
21
23
|
import { homedir } from "node:os";
|
|
22
24
|
import { basename, join } from "node:path";
|
|
23
25
|
import { createHash } from "node:crypto";
|
|
@@ -84,7 +86,12 @@ function trialBudget(): { summary: string; reason: string | null } {
|
|
|
84
86
|
const today = new Date().toISOString().slice(0, 10);
|
|
85
87
|
let ledger: { date?: string; trials?: number; usd?: number } = {};
|
|
86
88
|
try { ledger = JSON.parse(readFileSync(LEDGER_FILE, "utf8")) as typeof ledger; }
|
|
87
|
-
catch {
|
|
89
|
+
catch {
|
|
90
|
+
if (existsSync(LEDGER_FILE)) return {
|
|
91
|
+
summary: "ledger unreadable",
|
|
92
|
+
reason: "trial ledger is malformed; run `openmerit doctor` before spending",
|
|
93
|
+
};
|
|
94
|
+
}
|
|
88
95
|
const trials = ledger.date === today ? ledger.trials ?? 0 : 0;
|
|
89
96
|
const usd = ledger.date === today ? ledger.usd ?? 0 : 0;
|
|
90
97
|
const reason = trials >= limit ? `daily trial cap reached (${limit})`
|
|
@@ -107,27 +114,41 @@ function latestSessionJob(sessionFile: string | undefined): string | null {
|
|
|
107
114
|
}
|
|
108
115
|
|
|
109
116
|
function loadPolicy(): ExtensionPolicy {
|
|
117
|
+
if (!existsSync(POLICY_FILE)) return {};
|
|
110
118
|
try {
|
|
111
119
|
return JSON.parse(readFileSync(POLICY_FILE, "utf8")) as ExtensionPolicy;
|
|
112
120
|
} catch {
|
|
113
|
-
return {}
|
|
121
|
+
return { mode: "recommend", fallback: { apply_on_error: false },
|
|
122
|
+
budgets: { max_trials_per_day: 0, max_usd_per_day: 0 } };
|
|
114
123
|
}
|
|
115
124
|
}
|
|
116
125
|
|
|
117
126
|
interface Recommendation {
|
|
127
|
+
schemaVersion?: 1;
|
|
118
128
|
id: string;
|
|
119
129
|
ts: string;
|
|
120
130
|
taskKey: string;
|
|
121
131
|
taskLabel: string;
|
|
122
132
|
sessionFile?: string | null;
|
|
123
133
|
currentModel: string | null;
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
134
|
+
currentRoute?: ModelRoute | null;
|
|
135
|
+
recommended: { model: string; route?: ModelRoute; reason: string };
|
|
136
|
+
fallback: { model: string; route?: ModelRoute; reason: string } | null;
|
|
137
|
+
evidence: { scoreGain: number; priceRatio: number; priceKnown?: boolean; onFrontier: boolean; trials: number };
|
|
127
138
|
policy: { autoApply: boolean; reasons: string[] };
|
|
128
139
|
status: "pending" | "applied" | "dismissed" | "expired";
|
|
129
140
|
}
|
|
130
141
|
|
|
142
|
+
interface ModelRoute {
|
|
143
|
+
provider: string;
|
|
144
|
+
modelId: string;
|
|
145
|
+
meritId: string;
|
|
146
|
+
input: ("text" | "image")[];
|
|
147
|
+
cost?: { input: number; output: number; cacheRead?: number; cacheWrite?: number };
|
|
148
|
+
contextWindow?: number;
|
|
149
|
+
maxTokens?: number;
|
|
150
|
+
}
|
|
151
|
+
|
|
131
152
|
interface RecordedTrial {
|
|
132
153
|
taskKey?: string;
|
|
133
154
|
sessionFile?: string;
|
|
@@ -161,7 +182,17 @@ function appendJsonl(file: string, obj: unknown): void {
|
|
|
161
182
|
function latestRecommendations(): Recommendation[] {
|
|
162
183
|
const byId = new Map<string, Recommendation>();
|
|
163
184
|
for (const r of readJsonl<Recommendation>(RECS_FILE)) byId.set(r.id, r);
|
|
164
|
-
|
|
185
|
+
let malformed = false;
|
|
186
|
+
if (existsSync(RECS_FILE)) {
|
|
187
|
+
for (const line of readFileSync(RECS_FILE, "utf8").split("\n")) {
|
|
188
|
+
if (!line.trim()) continue;
|
|
189
|
+
try { JSON.parse(line); } catch { malformed = true; break; }
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return [...byId.values()].map((rec) => malformed && rec.status === "pending" && rec.policy.autoApply
|
|
193
|
+
? { ...rec, policy: { autoApply: false,
|
|
194
|
+
reasons: [...rec.policy.reasons, "recommendation history contains a malformed line; manual review required"] } }
|
|
195
|
+
: rec);
|
|
165
196
|
}
|
|
166
197
|
|
|
167
198
|
function taskKey(text: string): string {
|
|
@@ -248,7 +279,11 @@ function suggestionText(rec: Recommendation, trials: RecordedTrial[]): string {
|
|
|
248
279
|
}
|
|
249
280
|
|
|
250
281
|
function canonicalModel(provider: string, id: string): string {
|
|
251
|
-
return provider === "openrouter" ? id : `${provider}/${id}`;
|
|
282
|
+
return provider === "openrouter" || id.startsWith(`${provider}/`) ? id : `${provider}/${id}`;
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
function routeDisplay(route: ModelRoute | undefined, fallback: string): string {
|
|
286
|
+
return route && route.provider !== "openrouter" ? `${route.meritId} via ${route.provider}` : fallback;
|
|
252
287
|
}
|
|
253
288
|
|
|
254
289
|
function setStatus(rec: Recommendation, status: Recommendation["status"]): void {
|
|
@@ -256,11 +291,54 @@ function setStatus(rec: Recommendation, status: Recommendation["status"]): void
|
|
|
256
291
|
}
|
|
257
292
|
|
|
258
293
|
let fallbackModel: string | null = null;
|
|
294
|
+
let fallbackRoute: ModelRoute | null = null;
|
|
295
|
+
|
|
296
|
+
function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
|
|
297
|
+
cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
|
|
298
|
+
contextWindow?: number; maxTokens?: number }): ModelRoute {
|
|
299
|
+
return {
|
|
300
|
+
provider: model.provider,
|
|
301
|
+
modelId: model.id,
|
|
302
|
+
meritId: canonicalModel(model.provider, model.id),
|
|
303
|
+
input: model.input ?? ["text"],
|
|
304
|
+
cost: model.cost && typeof model.cost.input === "number" && typeof model.cost.output === "number"
|
|
305
|
+
? { input: model.cost.input, output: model.cost.output,
|
|
306
|
+
cacheRead: model.cost.cacheRead, cacheWrite: model.cost.cacheWrite }
|
|
307
|
+
: undefined,
|
|
308
|
+
contextWindow: model.contextWindow,
|
|
309
|
+
maxTokens: model.maxTokens,
|
|
310
|
+
};
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
|
|
314
|
+
const registry = ctx.modelRegistry as unknown as { getAvailable?: () => Array<Parameters<typeof routeFor>[0]> };
|
|
315
|
+
const scoped = (ctx.scopedModels ?? []) as readonly { model: Parameters<typeof routeFor>[0] }[];
|
|
316
|
+
const models = scoped.length ? scoped.map((item) => item.model) : registry.getAvailable?.() ?? [];
|
|
317
|
+
if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
|
|
318
|
+
models.unshift(ctx.model);
|
|
319
|
+
const byRoute = new Map<string, ModelRoute>();
|
|
320
|
+
for (const model of models) {
|
|
321
|
+
const route = routeFor(model);
|
|
322
|
+
byRoute.set(`${route.provider}:${route.modelId}`, route);
|
|
323
|
+
}
|
|
324
|
+
return [...byRoute.values()];
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
function writeState(value: unknown): void {
|
|
328
|
+
const temp = `${STATE_FILE}.tmp-${process.pid}-${Date.now()}`;
|
|
329
|
+
try {
|
|
330
|
+
writeFileSync(temp, JSON.stringify(value) + "\n");
|
|
331
|
+
renameSync(temp, STATE_FILE);
|
|
332
|
+
} finally {
|
|
333
|
+
if (existsSync(temp)) rmSync(temp, { force: true });
|
|
334
|
+
}
|
|
335
|
+
}
|
|
259
336
|
|
|
260
337
|
function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
261
338
|
try {
|
|
262
339
|
mkdirSync(HOME, { recursive: true });
|
|
263
340
|
const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
341
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : null;
|
|
264
342
|
if (settled) settledTaskKey = latestSessionTaskKey(ctx);
|
|
265
343
|
const prior = existsSync(STATE_FILE)
|
|
266
344
|
? JSON.parse(readFileSync(STATE_FILE, "utf8")) as { sessionFile?: string | null; settledAt?: string }
|
|
@@ -268,18 +346,19 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
268
346
|
const sessionFile = ctx.sessionManager.getSessionFile() ?? null;
|
|
269
347
|
const settledAt = settled ? new Date().toISOString()
|
|
270
348
|
: prior.sessionFile === sessionFile ? prior.settledAt ?? null : null;
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
JSON.stringify({
|
|
349
|
+
writeState({
|
|
350
|
+
schemaVersion: 1,
|
|
274
351
|
currentModel: model,
|
|
352
|
+
currentRoute,
|
|
353
|
+
routes: eligibleRoutes(ctx),
|
|
275
354
|
fallbackModel,
|
|
355
|
+
fallbackRoute,
|
|
276
356
|
sessionFile,
|
|
277
357
|
settledTaskKey,
|
|
278
358
|
settledAt,
|
|
279
359
|
cwd: ctx.cwd,
|
|
280
360
|
updatedAt: new Date().toISOString(),
|
|
281
|
-
})
|
|
282
|
-
);
|
|
361
|
+
});
|
|
283
362
|
} catch {
|
|
284
363
|
/* never break the host session over reporting */
|
|
285
364
|
}
|
|
@@ -288,8 +367,11 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
288
367
|
/** Restore the persisted fallback so a fresh session keeps it even with no pending recs. */
|
|
289
368
|
function restoreFallback(): void {
|
|
290
369
|
try {
|
|
291
|
-
const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
|
|
370
|
+
const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
|
|
371
|
+
fallbackModel?: string | null; fallbackRoute?: ModelRoute | null;
|
|
372
|
+
};
|
|
292
373
|
if (st.fallbackModel) fallbackModel = st.fallbackModel;
|
|
374
|
+
if (st.fallbackRoute) fallbackRoute = st.fallbackRoute;
|
|
293
375
|
} catch {
|
|
294
376
|
/* no state yet */
|
|
295
377
|
}
|
|
@@ -344,7 +426,15 @@ function catalogEntry(openrouterId: string): CatalogEntryLite | undefined {
|
|
|
344
426
|
* OpenRouter catalog, so inject anything missing via registerProvider
|
|
345
427
|
* (takes effect immediately, no /reload needed).
|
|
346
428
|
*/
|
|
347
|
-
function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string) {
|
|
429
|
+
function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string, route?: ModelRoute) {
|
|
430
|
+
const registry = ctx.modelRegistry as unknown as {
|
|
431
|
+
find?: (provider: string, id: string) => unknown | undefined;
|
|
432
|
+
};
|
|
433
|
+
if (route) {
|
|
434
|
+
const exact = registry.find?.(route.provider, route.modelId);
|
|
435
|
+
if (exact) return exact;
|
|
436
|
+
if (route.provider !== "openrouter") return undefined;
|
|
437
|
+
}
|
|
348
438
|
const existing = resolveModel(ctx, openrouterId);
|
|
349
439
|
if (existing) return existing;
|
|
350
440
|
|
|
@@ -372,16 +462,14 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
|
|
|
372
462
|
})),
|
|
373
463
|
});
|
|
374
464
|
|
|
375
|
-
|
|
376
|
-
find?: (provider: string, id: string) => unknown | undefined;
|
|
377
|
-
};
|
|
378
|
-
return reg.find?.(INJECT_PROVIDER, openrouterId);
|
|
465
|
+
return registry.find?.(INJECT_PROVIDER, openrouterId);
|
|
379
466
|
}
|
|
380
467
|
|
|
381
468
|
export default function openmerit(pi: ExtensionAPI) {
|
|
382
469
|
let pollTimer: ReturnType<typeof setInterval> | null = null;
|
|
383
470
|
let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
|
|
384
|
-
const queuedJobs: { sessionFile: string; settledAt: string; model: string;
|
|
471
|
+
const queuedJobs: { sessionFile: string; settledAt: string; model: string; route: ModelRoute;
|
|
472
|
+
key: string; cwd: string; bytes: number }[] = [];
|
|
385
473
|
const startedJobs = new Set<string>();
|
|
386
474
|
let activeSession: string | null = null;
|
|
387
475
|
let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
|
|
@@ -408,7 +496,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
408
496
|
return;
|
|
409
497
|
}
|
|
410
498
|
const child = spawn(process.execPath, [ENGINE_FILE, "session-trial", job.sessionFile,
|
|
411
|
-
job.settledAt, job.model, job.key, job.cwd, String(job.bytes)], {
|
|
499
|
+
job.settledAt, job.model, job.key, job.cwd, String(job.bytes), job.route.provider, job.route.modelId], {
|
|
412
500
|
cwd: job.cwd, env: process.env, stdio: ["ignore", "pipe", "pipe"],
|
|
413
501
|
detached: process.platform !== "win32",
|
|
414
502
|
});
|
|
@@ -466,7 +554,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
466
554
|
function queueSettledTask(ctx: ExtensionContext): void {
|
|
467
555
|
const sessionFile = ctx.sessionManager.getSessionFile();
|
|
468
556
|
const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
469
|
-
|
|
557
|
+
const route = ctx.model ? routeFor(ctx.model) : null;
|
|
558
|
+
if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
|
|
470
559
|
// The alpha compares completed response tasks. The engine validates images,
|
|
471
560
|
// tool use and the baseline answer before any provider calls.
|
|
472
561
|
let completed = false;
|
|
@@ -490,7 +579,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
490
579
|
if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
|
|
491
580
|
return;
|
|
492
581
|
}
|
|
493
|
-
queuedJobs.push({ sessionFile, settledAt, model, key: settledTaskKey, cwd: ctx.cwd, bytes });
|
|
582
|
+
queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
|
|
494
583
|
startNextJob(ctx);
|
|
495
584
|
}
|
|
496
585
|
|
|
@@ -502,11 +591,18 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
502
591
|
|
|
503
592
|
async function applyRecommendation(ctx: ExtensionContext, rec: Recommendation): Promise<boolean> {
|
|
504
593
|
const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
505
|
-
|
|
594
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : null;
|
|
595
|
+
if (rec.currentRoute && currentRoute &&
|
|
596
|
+
(rec.currentRoute.provider !== currentRoute.provider || rec.currentRoute.modelId !== currentRoute.modelId)) {
|
|
597
|
+
ctx.ui.notify(`openmerit: recommendation targets ${rec.currentRoute.provider}/${rec.currentRoute.modelId}; ` +
|
|
598
|
+
`current route is ${currentRoute.provider}/${currentRoute.modelId}`, "warning");
|
|
599
|
+
return false;
|
|
600
|
+
}
|
|
601
|
+
if (!rec.currentRoute && rec.currentModel && current !== rec.currentModel) {
|
|
506
602
|
ctx.ui.notify(`openmerit: recommendation was measured against ${rec.currentModel}; current model is ${current ?? "unknown"}`, "warning");
|
|
507
603
|
return false;
|
|
508
604
|
}
|
|
509
|
-
const model = ensureModel(pi, ctx, rec.recommended.model);
|
|
605
|
+
const model = ensureModel(pi, ctx, rec.recommended.model, rec.recommended.route);
|
|
510
606
|
if (!model) {
|
|
511
607
|
ctx.ui.notify(
|
|
512
608
|
`openmerit: ${rec.recommended.model} not found in pi's model registry (add the provider/model first)`,
|
|
@@ -517,8 +613,12 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
517
613
|
const ok = await pi.setModel(model as Parameters<typeof pi.setModel>[0]);
|
|
518
614
|
if (ok) {
|
|
519
615
|
setStatus(rec, "applied");
|
|
520
|
-
if (rec.fallback)
|
|
521
|
-
|
|
616
|
+
if (rec.fallback) {
|
|
617
|
+
fallbackModel = rec.fallback.model;
|
|
618
|
+
fallbackRoute = rec.fallback.route ?? null;
|
|
619
|
+
}
|
|
620
|
+
ctx.ui.notify(`openmerit: switched to ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n` +
|
|
621
|
+
rec.recommended.reason, "info");
|
|
522
622
|
reportHarnessState(ctx);
|
|
523
623
|
} else {
|
|
524
624
|
ctx.ui.notify(`openmerit: no auth for ${rec.recommended.model}`, "error");
|
|
@@ -550,7 +650,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
550
650
|
// startup. Notify (fire-and-forget) and let the user act via /openmerit.
|
|
551
651
|
if (ctx.hasUI) {
|
|
552
652
|
ctx.ui.notify(
|
|
553
|
-
`openmerit recommends: ${rec.recommended.model}\n${rec.recommended.reason}\n` +
|
|
653
|
+
`openmerit recommends: ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n${rec.recommended.reason}\n` +
|
|
554
654
|
`gate: ${rec.policy.reasons.join("; ")}\n(/openmerit apply to switch, /openmerit dismiss to ignore)`,
|
|
555
655
|
"info",
|
|
556
656
|
);
|
|
@@ -621,7 +721,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
621
721
|
const policy = loadPolicy();
|
|
622
722
|
if (policy.fallback?.apply_on_error === false) return;
|
|
623
723
|
const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "current model";
|
|
624
|
-
const model = ensureModel(pi, ctx, fallbackModel);
|
|
724
|
+
const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
|
|
625
725
|
if (!model) return;
|
|
626
726
|
|
|
627
727
|
const auto = policy.mode === "auto";
|
|
@@ -673,7 +773,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
673
773
|
ctx.ui.notify("openmerit: no fallback set yet", "info");
|
|
674
774
|
return;
|
|
675
775
|
}
|
|
676
|
-
const model = ensureModel(pi, ctx, fallbackModel);
|
|
776
|
+
const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
|
|
677
777
|
if (model && (await pi.setModel(model as Parameters<typeof pi.setModel>[0]))) {
|
|
678
778
|
ctx.ui.notify(`openmerit: switched to fallback ${fallbackModel}`, "info");
|
|
679
779
|
reportHarnessState(ctx);
|
|
@@ -683,7 +783,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
683
783
|
return;
|
|
684
784
|
}
|
|
685
785
|
|
|
686
|
-
const
|
|
786
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
|
|
787
|
+
const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
|
|
687
788
|
const budget = trialBudget();
|
|
688
789
|
const liveProgress = progress?.sessionFile === ctx.sessionManager.getSessionFile() ? progress : null;
|
|
689
790
|
const comparison = trialJob ? liveProgress?.current
|
|
@@ -706,7 +807,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
706
807
|
for (const r of pending.slice(-5)) {
|
|
707
808
|
lines.push(
|
|
708
809
|
"",
|
|
709
|
-
`-> ${r.recommended.model} (gain ${r.evidence.scoreGain},
|
|
810
|
+
`-> ${routeDisplay(r.recommended.route, r.recommended.model)} (gain ${r.evidence.scoreGain}, ` +
|
|
811
|
+
`${r.evidence.priceKnown === false ? "price unknown" : `${r.evidence.priceRatio}x price`}, frontier=${r.evidence.onFrontier})`,
|
|
710
812
|
` ${r.recommended.reason}`,
|
|
711
813
|
` gate: ${r.policy.reasons.join("; ") || "n/a"}`,
|
|
712
814
|
);
|
|
@@ -45,8 +45,8 @@ the best model for the task at hand, at the best price, with a vetted fallback.
|
|
|
45
45
|
## Guarantees
|
|
46
46
|
|
|
47
47
|
- Per-task comparisons send the task text, uploaded files or images (when
|
|
48
|
-
present), and candidate answers through
|
|
49
|
-
judging. Candidate Pi runs default to no tools; any tool access is an
|
|
48
|
+
present), and candidate answers through the model routes configured in Pi
|
|
49
|
+
for model runs and judging. Candidate Pi runs default to no tools; any tool access is an
|
|
50
50
|
explicit user opt-in. Use non-sensitive examples while
|
|
51
51
|
evaluating this alpha.
|
|
52
52
|
- The shipped policy is supervised. Swaps happen automatically only after the
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "openmerit",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.3",
|
|
4
4
|
"description": "Find better models for each pi task by comparing quality, cost, and latency.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": [
|
|
@@ -17,10 +17,13 @@
|
|
|
17
17
|
"extension/openmerit.ts",
|
|
18
18
|
"instructions/",
|
|
19
19
|
"examples/",
|
|
20
|
+
"rules.md",
|
|
20
21
|
"benchmark/invoice_ocr/data/"
|
|
21
22
|
],
|
|
22
23
|
"pi": {
|
|
23
|
-
"extensions": [
|
|
24
|
+
"extensions": [
|
|
25
|
+
"./extension/openmerit.ts"
|
|
26
|
+
]
|
|
24
27
|
},
|
|
25
28
|
"bin": {
|
|
26
29
|
"openmerit": "dist/cli.js"
|
package/rules.md
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# OpenMerit product and architecture rules
|
|
2
|
+
|
|
3
|
+
- **Optimize for useful model choices.** OpenMerit exists to find better task-specific tradeoffs between quality, cost, and latency—not to become a general-purpose agent or telemetry platform.
|
|
4
|
+
|
|
5
|
+
- **Act as a control plane, not a gateway.** Normal traffic stays between the harness and provider; OpenMerit observes evidence, runs explicit trials, and returns recommendations without intercepting ordinary work.
|
|
6
|
+
|
|
7
|
+
- **Make Pi plug-and-play first.** Pi and Pi-based harnesses are the first supported user experience because that is where early users already work; broader harness support can follow without weakening this path.
|
|
8
|
+
|
|
9
|
+
- **Keep the core harness-neutral.** Pi is the first `HarnessAdapter`, not a permanent assumption in scoring, recommendations, policy, or stored data, so other harnesses can be added without rewriting the merit loop.
|
|
10
|
+
|
|
11
|
+
- **Keep model providers interchangeable.** OpenRouter is a first-class provider and discovery source, not the foundation of the domain model; provider-specific behavior belongs behind a `ModelProviderAdapter`.
|
|
12
|
+
|
|
13
|
+
- **Require end-to-end provider neutrality.** Discovery, authentication, execution, scoring, and swapping must preserve the selected route; a feature is not provider-neutral if only its API client is abstracted.
|
|
14
|
+
|
|
15
|
+
- **Separate model identity from execution route.** Keep logical model IDs in stable `vendor/model` form, while recording the actual provider or route separately, so the same model can be compared through OpenRouter, a native provider, or a local provider.
|
|
16
|
+
|
|
17
|
+
- **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
|
|
18
|
+
|
|
19
|
+
- **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
|
|
20
|
+
|
|
21
|
+
- **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
|
|
22
|
+
|
|
23
|
+
- **Use provider-neutral core records.** Normalize integrations into stable concepts such as `TaskObservation`, `CandidateRun`, `TrialScore`, `Frontier`, `Recommendation`, and `MeritEvent`, so integrations do not leak their schemas into the decision engine.
|
|
24
|
+
|
|
25
|
+
- **Optimize per task before aggregating per agent.** Agent-level conclusions must be built from measured task evidence rather than assumed from global model rankings.
|
|
26
|
+
|
|
27
|
+
- **Version persisted events and evolve them additively.** Existing 0.1.x state must remain readable, and append-only recommendation history must stay intact; migrations should normalize old records rather than invalidate them.
|
|
28
|
+
|
|
29
|
+
- **Keep policy as the sole auto-swap authority.** Trials and strategists may recommend changes, but only the policy gate may approve automatic application, and every recommendation must retain its gate reasons.
|
|
30
|
+
|
|
31
|
+
- **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
|
|
32
|
+
|
|
33
|
+
- **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
|
|
34
|
+
|
|
35
|
+
- **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.
|
|
36
|
+
|
|
37
|
+
- **Keep integrations optional and the core lightweight.** New providers, harnesses, and observability services should not impose credentials, network calls, or heavy dependencies on users who do not enable them.
|
|
38
|
+
|
|
39
|
+
- **Keep 0.1.x focused.** The near-term bar is reliable, provider-neutral use for Pi tinkerers and solo hackers; additional harnesses and hosted observability integrations belong in later releases unless required to prove the boundaries work.
|