openmerit 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -42
- package/dist/catalog.js +11 -1
- package/dist/cli.js +62 -8
- package/dist/daemon.js +111 -51
- package/dist/frontier.js +13 -6
- package/dist/harness.js +1 -0
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +159 -22
- package/dist/pi-trials.js +203 -74
- package/dist/policy.js +8 -3
- package/dist/providers.js +1 -0
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/store.js +2 -0
- package/dist/strategist.js +18 -14
- package/dist/task-input.js +27 -3
- package/dist/traces.js +6 -2
- package/dist/trials.js +38 -8
- package/extension/openmerit.ts +116 -27
- package/instructions/OPENMERIT.md +8 -5
- package/package.json +5 -2
- package/rules.md +39 -0
package/dist/task-input.js
CHANGED
|
@@ -1,6 +1,29 @@
|
|
|
1
1
|
/** A pi text task may carry images embedded in its saved session message. */
|
|
2
2
|
import { createHash } from "node:crypto";
|
|
3
|
+
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
4
|
+
import { basename } from "node:path";
|
|
3
5
|
import { taskKey } from "./store.js";
|
|
6
|
+
/** Recover files Pi expanded into a task message (for example @invoice.pdf). */
|
|
7
|
+
export function sessionFiles(text) {
|
|
8
|
+
const files = [];
|
|
9
|
+
const seen = new Set();
|
|
10
|
+
const re = /<file\s+name=["']([^"']+)["'][^>]*>/g;
|
|
11
|
+
for (const match of text.matchAll(re)) {
|
|
12
|
+
const file = match[1];
|
|
13
|
+
if (!file || seen.has(file) || !existsSync(file))
|
|
14
|
+
continue;
|
|
15
|
+
try {
|
|
16
|
+
const stat = statSync(file);
|
|
17
|
+
if (!stat.isFile() || stat.size > 100_000_000)
|
|
18
|
+
continue;
|
|
19
|
+
files.push({ path: file, name: basename(file), size: stat.size,
|
|
20
|
+
sha256: createHash("sha256").update(readFileSync(file)).digest("hex") });
|
|
21
|
+
seen.add(file);
|
|
22
|
+
}
|
|
23
|
+
catch { /* inaccessible attachment */ }
|
|
24
|
+
}
|
|
25
|
+
return files;
|
|
26
|
+
}
|
|
4
27
|
const SUPPORTED = new Set(["image/jpeg", "image/png", "image/webp", "image/gif"]);
|
|
5
28
|
export function sessionImages(content) {
|
|
6
29
|
if (!Array.isArray(content))
|
|
@@ -20,11 +43,12 @@ export function sessionImages(content) {
|
|
|
20
43
|
return images;
|
|
21
44
|
}
|
|
22
45
|
/** Text-only keys remain compatible; image bytes distinguish same-prompt documents. */
|
|
23
|
-
export function taskInputKey(text, images) {
|
|
24
|
-
if (images.length === 0)
|
|
46
|
+
export function taskInputKey(text, images, files = []) {
|
|
47
|
+
if (images.length === 0 && files.length === 0)
|
|
25
48
|
return taskKey(text);
|
|
26
49
|
const normalized = text.toLowerCase().replace(/\s+/g, " ").trim();
|
|
27
50
|
const hashes = images.map((image) => `${image.mimeType}:` +
|
|
28
51
|
createHash("sha256").update(Buffer.from(image.data, "base64")).digest("hex"));
|
|
29
|
-
|
|
52
|
+
const fileHashes = files.map((file) => `${file.name}:${file.sha256}`).join(",");
|
|
53
|
+
return taskKey(`${normalized}\nimages:${hashes.join(",")}\nfiles:${fileHashes}`);
|
|
30
54
|
}
|
package/dist/traces.js
CHANGED
|
@@ -7,6 +7,7 @@ import { readdirSync, readFileSync, existsSync, statSync } from "node:fs";
|
|
|
7
7
|
import { homedir } from "node:os";
|
|
8
8
|
import { join } from "node:path";
|
|
9
9
|
import { paths, readJson, taskKey, writeJson } from "./store.js";
|
|
10
|
+
import { meritModelId, modelRoute } from "./routes.js";
|
|
10
11
|
export function defaultSessionsDir() {
|
|
11
12
|
return join(homedir(), ".pi", "agent", "sessions");
|
|
12
13
|
}
|
|
@@ -53,6 +54,7 @@ export function observationsFromEntries(entries, sessionFile) {
|
|
|
53
54
|
continue;
|
|
54
55
|
}
|
|
55
56
|
current = {
|
|
57
|
+
schemaVersion: 1,
|
|
56
58
|
taskKey: taskKey(text),
|
|
57
59
|
taskLabel: text.replace(/\s+/g, " ").trim().slice(0, 120),
|
|
58
60
|
model: null,
|
|
@@ -68,8 +70,10 @@ export function observationsFromEntries(entries, sessionFile) {
|
|
|
68
70
|
if (!current)
|
|
69
71
|
continue;
|
|
70
72
|
if (m.role === "assistant") {
|
|
71
|
-
if (!current.model && m.provider && m.model)
|
|
72
|
-
current.model =
|
|
73
|
+
if (!current.model && m.provider && m.model) {
|
|
74
|
+
current.model = meritModelId(m.provider, m.model);
|
|
75
|
+
current.route = modelRoute(m.provider, m.model);
|
|
76
|
+
}
|
|
73
77
|
current.costUsd += m.usage?.cost?.total ?? 0;
|
|
74
78
|
current.tokens += m.usage?.totalTokens ?? 0;
|
|
75
79
|
if (m.stopReason === "error")
|
package/dist/trials.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/** Shadow trials: run candidate models against observed/declared tasks on the background track. */
|
|
2
|
-
import {
|
|
2
|
+
import { directChatClient } from "./llm.js";
|
|
3
3
|
import { judge, parseObj } from "./judge.js";
|
|
4
4
|
import { paths, readJson, writeJson } from "./store.js";
|
|
5
5
|
function today() {
|
|
@@ -15,6 +15,21 @@ export function budgetOk(policy) {
|
|
|
15
15
|
return { ok: false, reason: `daily budget exhausted ($${policy.budgets.max_usd_per_day})` };
|
|
16
16
|
return { ok: true };
|
|
17
17
|
}
|
|
18
|
+
/** Conservative admission estimate for one candidate answer (rough input tokens + 4K output). */
|
|
19
|
+
export function trialBudgetOk(policy, entry, inputChars) {
|
|
20
|
+
if (entry.priceKnown === false)
|
|
21
|
+
return { ok: false, reason: "route price is unknown" };
|
|
22
|
+
const inputTokens = Math.ceil(inputChars / 4);
|
|
23
|
+
const outputTokens = Math.min(entry.route?.maxTokens ?? 4096, 4096);
|
|
24
|
+
const estimatedUsd = entry.pp * inputTokens + entry.pc * outputTokens;
|
|
25
|
+
if (estimatedUsd > policy.budgets.max_usd_per_trial)
|
|
26
|
+
return {
|
|
27
|
+
ok: false,
|
|
28
|
+
estimatedUsd,
|
|
29
|
+
reason: `estimated trial cost $${estimatedUsd.toFixed(4)} exceeds $${policy.budgets.max_usd_per_trial.toFixed(4)} limit`,
|
|
30
|
+
};
|
|
31
|
+
return { ok: true, estimatedUsd };
|
|
32
|
+
}
|
|
18
33
|
export function recordTrialSpend(usd) {
|
|
19
34
|
let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
20
35
|
if (ledger.date !== today())
|
|
@@ -23,16 +38,26 @@ export function recordTrialSpend(usd) {
|
|
|
23
38
|
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
24
39
|
writeJson(paths.ledger(), ledger);
|
|
25
40
|
}
|
|
41
|
+
/** Record non-candidate merit-loop spend without consuming a trial slot. */
|
|
42
|
+
export function recordMeritSpend(usd) {
|
|
43
|
+
if (!(usd > 0))
|
|
44
|
+
return;
|
|
45
|
+
let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
46
|
+
if (ledger.date !== today())
|
|
47
|
+
ledger = { date: today(), trials: 0, usd: 0 };
|
|
48
|
+
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
49
|
+
writeJson(paths.ledger(), ledger);
|
|
50
|
+
}
|
|
26
51
|
const RUBRIC_PROMPT = `Write a compact grading rubric for answers to the task below.
|
|
27
52
|
It must list the concrete criteria for a score of 1.0 and how to deduct.
|
|
28
53
|
TASK: {task}
|
|
29
54
|
Return ONLY json: {{"rubric": "<text>"}}`;
|
|
30
55
|
/** Derive (and cache) a grading rubric for a task that came from traces. */
|
|
31
|
-
export async function ensureRubric(key, judgeModel, task, cache) {
|
|
56
|
+
export async function ensureRubric(key, judgeModel, task, cache, client = directChatClient(key)) {
|
|
32
57
|
const hit = cache.get(task);
|
|
33
58
|
if (hit)
|
|
34
59
|
return hit;
|
|
35
|
-
const { content } = await chat(
|
|
60
|
+
const { content } = await client.chat(judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
|
|
36
61
|
let rubric = "Score 1.0 for a fully correct, complete, usable answer; deduct for errors, omissions, or unusable output.";
|
|
37
62
|
try {
|
|
38
63
|
const obj = parseObj(content);
|
|
@@ -46,19 +71,20 @@ export async function ensureRubric(key, judgeModel, task, cache) {
|
|
|
46
71
|
return rubric;
|
|
47
72
|
}
|
|
48
73
|
/** Run one shadow trial: generate with `model`, judge the output, return a frontier point + cost. */
|
|
49
|
-
export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
74
|
+
export async function runTrial(key, judgeModel, task, rubric, model, entry, client = directChatClient(key)) {
|
|
50
75
|
const started = Date.now();
|
|
51
76
|
let gen;
|
|
52
77
|
let usage = {};
|
|
53
78
|
try {
|
|
54
|
-
const r = await chat(
|
|
79
|
+
const r = await client.chat(entry?.route ?? model, task, 4096, 0.2);
|
|
55
80
|
gen = r.content;
|
|
56
81
|
usage = r.usage;
|
|
57
82
|
}
|
|
58
83
|
catch (e) {
|
|
59
84
|
return {
|
|
60
85
|
point: {
|
|
61
|
-
model, score: 0, price: entry?.price ?? 0,
|
|
86
|
+
schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
87
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
62
88
|
ts: new Date().toISOString(), source: "shadow_trial", why: String(e).slice(0, 160),
|
|
63
89
|
},
|
|
64
90
|
costUsd: 0,
|
|
@@ -70,7 +96,8 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
|
70
96
|
if (!gen.trim()) {
|
|
71
97
|
return {
|
|
72
98
|
point: {
|
|
73
|
-
model, score: 0, price: entry?.price ?? 0,
|
|
99
|
+
schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
|
|
100
|
+
priceKnown: entry?.priceKnown ?? !!entry, latencyMs,
|
|
74
101
|
ts: new Date().toISOString(), source: "shadow_trial", why: "empty response",
|
|
75
102
|
},
|
|
76
103
|
costUsd,
|
|
@@ -79,7 +106,7 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
|
79
106
|
let score = 0;
|
|
80
107
|
let why = "judge error";
|
|
81
108
|
try {
|
|
82
|
-
const j = await judge(key, judgeModel, task, rubric, gen);
|
|
109
|
+
const j = await judge(key, judgeModel, task, rubric, gen, client);
|
|
83
110
|
score = j.score;
|
|
84
111
|
why = j.why;
|
|
85
112
|
}
|
|
@@ -88,9 +115,12 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
|
|
|
88
115
|
}
|
|
89
116
|
return {
|
|
90
117
|
point: {
|
|
118
|
+
schemaVersion: 1,
|
|
91
119
|
model,
|
|
120
|
+
route: entry?.route,
|
|
92
121
|
score,
|
|
93
122
|
price: entry?.price ?? 0,
|
|
123
|
+
priceKnown: entry?.priceKnown ?? !!entry,
|
|
94
124
|
latencyMs,
|
|
95
125
|
ts: new Date().toISOString(),
|
|
96
126
|
source: "shadow_trial",
|
package/extension/openmerit.ts
CHANGED
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
20
20
|
import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
|
|
21
21
|
import { homedir } from "node:os";
|
|
22
|
-
import { join } from "node:path";
|
|
22
|
+
import { basename, join } from "node:path";
|
|
23
23
|
import { createHash } from "node:crypto";
|
|
24
24
|
import { spawn, type ChildProcessByStdio } from "node:child_process";
|
|
25
25
|
import type { Readable } from "node:stream";
|
|
@@ -46,6 +46,9 @@ interface TrialProgress {
|
|
|
46
46
|
latencyMs?: number;
|
|
47
47
|
costUsd?: number;
|
|
48
48
|
error?: string;
|
|
49
|
+
toolCalls?: number;
|
|
50
|
+
toolErrors?: number;
|
|
51
|
+
changedFiles?: number;
|
|
49
52
|
}
|
|
50
53
|
|
|
51
54
|
function parseTrialProgress(line: string): TrialProgress | null {
|
|
@@ -63,7 +66,9 @@ function parseTrialProgress(line: string): TrialProgress | null {
|
|
|
63
66
|
function completedTrialText(trial: TrialProgress): string {
|
|
64
67
|
return `${trial.model}: quality ${(trial.score ?? 0).toFixed(2)}, ` +
|
|
65
68
|
`$${(trial.price ?? 0).toFixed(2)}/M, ${Math.round(trial.latencyMs ?? 0)}ms, ` +
|
|
66
|
-
`run $${(trial.costUsd ?? 0).toFixed(4)}
|
|
69
|
+
`run $${(trial.costUsd ?? 0).toFixed(4)}, tools ${trial.toolCalls ?? 0}` +
|
|
70
|
+
(trial.toolErrors || trial.changedFiles ? `, tool errors ${trial.toolErrors ?? 0}, changed files ${trial.changedFiles ?? 0}` : "") +
|
|
71
|
+
(trial.error ? ` (failed: ${trial.error})` : "");
|
|
67
72
|
}
|
|
68
73
|
|
|
69
74
|
interface ExtensionPolicy {
|
|
@@ -110,19 +115,31 @@ function loadPolicy(): ExtensionPolicy {
|
|
|
110
115
|
}
|
|
111
116
|
|
|
112
117
|
interface Recommendation {
|
|
118
|
+
schemaVersion?: 1;
|
|
113
119
|
id: string;
|
|
114
120
|
ts: string;
|
|
115
121
|
taskKey: string;
|
|
116
122
|
taskLabel: string;
|
|
117
123
|
sessionFile?: string | null;
|
|
118
124
|
currentModel: string | null;
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
125
|
+
currentRoute?: ModelRoute | null;
|
|
126
|
+
recommended: { model: string; route?: ModelRoute; reason: string };
|
|
127
|
+
fallback: { model: string; route?: ModelRoute; reason: string } | null;
|
|
128
|
+
evidence: { scoreGain: number; priceRatio: number; priceKnown?: boolean; onFrontier: boolean; trials: number };
|
|
122
129
|
policy: { autoApply: boolean; reasons: string[] };
|
|
123
130
|
status: "pending" | "applied" | "dismissed" | "expired";
|
|
124
131
|
}
|
|
125
132
|
|
|
133
|
+
interface ModelRoute {
|
|
134
|
+
provider: string;
|
|
135
|
+
modelId: string;
|
|
136
|
+
meritId: string;
|
|
137
|
+
input: ("text" | "image")[];
|
|
138
|
+
cost?: { input: number; output: number; cacheRead?: number; cacheWrite?: number };
|
|
139
|
+
contextWindow?: number;
|
|
140
|
+
maxTokens?: number;
|
|
141
|
+
}
|
|
142
|
+
|
|
126
143
|
interface RecordedTrial {
|
|
127
144
|
taskKey?: string;
|
|
128
145
|
sessionFile?: string;
|
|
@@ -171,10 +188,18 @@ function userTaskKey(content: unknown): string | null {
|
|
|
171
188
|
const images = parts.filter((b) => b?.type === "image");
|
|
172
189
|
if (parts.length !== parts.filter((b) => b?.type === "text" || b?.type === "image").length ||
|
|
173
190
|
images.some((b) => typeof b.data !== "string" || typeof b.mimeType !== "string")) return null;
|
|
174
|
-
|
|
191
|
+
const files: string[] = [];
|
|
192
|
+
const fileRe = /<file\s+name=["']([^"']+)["'][^>]*>/g;
|
|
193
|
+
for (const match of text.matchAll(fileRe)) {
|
|
194
|
+
const file = match[1];
|
|
195
|
+
if (!file || !existsSync(file) || files.includes(file)) continue;
|
|
196
|
+
try { files.push(`${basename(file)}:` + createHash("sha256").update(readFileSync(file)).digest("hex")); }
|
|
197
|
+
catch { /* inaccessible attachment */ }
|
|
198
|
+
}
|
|
199
|
+
if (images.length === 0 && files.length === 0) return taskKey(text);
|
|
175
200
|
const hashes = images.map((b) => `${b.mimeType}:` +
|
|
176
201
|
createHash("sha256").update(Buffer.from(b.data, "base64")).digest("hex"));
|
|
177
|
-
return taskKey(`${text.toLowerCase().replace(/\s+/g, " ").trim()}\nimages:${hashes.join(",")}`);
|
|
202
|
+
return taskKey(`${text.toLowerCase().replace(/\s+/g, " ").trim()}\nimages:${hashes.join(",")}\nfiles:${files.join(",")}`);
|
|
178
203
|
}
|
|
179
204
|
|
|
180
205
|
function latestSessionTaskKey(ctx: ExtensionContext): string | null {
|
|
@@ -235,7 +260,11 @@ function suggestionText(rec: Recommendation, trials: RecordedTrial[]): string {
|
|
|
235
260
|
}
|
|
236
261
|
|
|
237
262
|
function canonicalModel(provider: string, id: string): string {
|
|
238
|
-
return provider === "openrouter" ? id : `${provider}/${id}`;
|
|
263
|
+
return provider === "openrouter" || id.startsWith(`${provider}/`) ? id : `${provider}/${id}`;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
function routeDisplay(route: ModelRoute | undefined, fallback: string): string {
|
|
267
|
+
return route && route.provider !== "openrouter" ? `${route.meritId} via ${route.provider}` : fallback;
|
|
239
268
|
}
|
|
240
269
|
|
|
241
270
|
function setStatus(rec: Recommendation, status: Recommendation["status"]): void {
|
|
@@ -243,11 +272,44 @@ function setStatus(rec: Recommendation, status: Recommendation["status"]): void
|
|
|
243
272
|
}
|
|
244
273
|
|
|
245
274
|
let fallbackModel: string | null = null;
|
|
275
|
+
let fallbackRoute: ModelRoute | null = null;
|
|
276
|
+
|
|
277
|
+
function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
|
|
278
|
+
cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
|
|
279
|
+
contextWindow?: number; maxTokens?: number }): ModelRoute {
|
|
280
|
+
return {
|
|
281
|
+
provider: model.provider,
|
|
282
|
+
modelId: model.id,
|
|
283
|
+
meritId: canonicalModel(model.provider, model.id),
|
|
284
|
+
input: model.input ?? ["text"],
|
|
285
|
+
cost: model.cost && typeof model.cost.input === "number" && typeof model.cost.output === "number"
|
|
286
|
+
? { input: model.cost.input, output: model.cost.output,
|
|
287
|
+
cacheRead: model.cost.cacheRead, cacheWrite: model.cost.cacheWrite }
|
|
288
|
+
: undefined,
|
|
289
|
+
contextWindow: model.contextWindow,
|
|
290
|
+
maxTokens: model.maxTokens,
|
|
291
|
+
};
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
|
|
295
|
+
const registry = ctx.modelRegistry as unknown as { getAvailable?: () => Array<Parameters<typeof routeFor>[0]> };
|
|
296
|
+
const scoped = (ctx.scopedModels ?? []) as readonly { model: Parameters<typeof routeFor>[0] }[];
|
|
297
|
+
const models = scoped.length ? scoped.map((item) => item.model) : registry.getAvailable?.() ?? [];
|
|
298
|
+
if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
|
|
299
|
+
models.unshift(ctx.model);
|
|
300
|
+
const byRoute = new Map<string, ModelRoute>();
|
|
301
|
+
for (const model of models) {
|
|
302
|
+
const route = routeFor(model);
|
|
303
|
+
byRoute.set(`${route.provider}:${route.modelId}`, route);
|
|
304
|
+
}
|
|
305
|
+
return [...byRoute.values()];
|
|
306
|
+
}
|
|
246
307
|
|
|
247
308
|
function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
248
309
|
try {
|
|
249
310
|
mkdirSync(HOME, { recursive: true });
|
|
250
311
|
const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
312
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : null;
|
|
251
313
|
if (settled) settledTaskKey = latestSessionTaskKey(ctx);
|
|
252
314
|
const prior = existsSync(STATE_FILE)
|
|
253
315
|
? JSON.parse(readFileSync(STATE_FILE, "utf8")) as { sessionFile?: string | null; settledAt?: string }
|
|
@@ -258,8 +320,12 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
258
320
|
writeFileSync(
|
|
259
321
|
STATE_FILE,
|
|
260
322
|
JSON.stringify({
|
|
323
|
+
schemaVersion: 1,
|
|
261
324
|
currentModel: model,
|
|
325
|
+
currentRoute,
|
|
326
|
+
routes: eligibleRoutes(ctx),
|
|
262
327
|
fallbackModel,
|
|
328
|
+
fallbackRoute,
|
|
263
329
|
sessionFile,
|
|
264
330
|
settledTaskKey,
|
|
265
331
|
settledAt,
|
|
@@ -275,8 +341,11 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
275
341
|
/** Restore the persisted fallback so a fresh session keeps it even with no pending recs. */
|
|
276
342
|
function restoreFallback(): void {
|
|
277
343
|
try {
|
|
278
|
-
const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
|
|
344
|
+
const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
|
|
345
|
+
fallbackModel?: string | null; fallbackRoute?: ModelRoute | null;
|
|
346
|
+
};
|
|
279
347
|
if (st.fallbackModel) fallbackModel = st.fallbackModel;
|
|
348
|
+
if (st.fallbackRoute) fallbackRoute = st.fallbackRoute;
|
|
280
349
|
} catch {
|
|
281
350
|
/* no state yet */
|
|
282
351
|
}
|
|
@@ -331,7 +400,15 @@ function catalogEntry(openrouterId: string): CatalogEntryLite | undefined {
|
|
|
331
400
|
* OpenRouter catalog, so inject anything missing via registerProvider
|
|
332
401
|
* (takes effect immediately, no /reload needed).
|
|
333
402
|
*/
|
|
334
|
-
function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string) {
|
|
403
|
+
function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string, route?: ModelRoute) {
|
|
404
|
+
const registry = ctx.modelRegistry as unknown as {
|
|
405
|
+
find?: (provider: string, id: string) => unknown | undefined;
|
|
406
|
+
};
|
|
407
|
+
if (route) {
|
|
408
|
+
const exact = registry.find?.(route.provider, route.modelId);
|
|
409
|
+
if (exact) return exact;
|
|
410
|
+
if (route.provider !== "openrouter") return undefined;
|
|
411
|
+
}
|
|
335
412
|
const existing = resolveModel(ctx, openrouterId);
|
|
336
413
|
if (existing) return existing;
|
|
337
414
|
|
|
@@ -359,16 +436,14 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
|
|
|
359
436
|
})),
|
|
360
437
|
});
|
|
361
438
|
|
|
362
|
-
|
|
363
|
-
find?: (provider: string, id: string) => unknown | undefined;
|
|
364
|
-
};
|
|
365
|
-
return reg.find?.(INJECT_PROVIDER, openrouterId);
|
|
439
|
+
return registry.find?.(INJECT_PROVIDER, openrouterId);
|
|
366
440
|
}
|
|
367
441
|
|
|
368
442
|
export default function openmerit(pi: ExtensionAPI) {
|
|
369
443
|
let pollTimer: ReturnType<typeof setInterval> | null = null;
|
|
370
444
|
let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
|
|
371
|
-
const queuedJobs: { sessionFile: string; settledAt: string; model: string;
|
|
445
|
+
const queuedJobs: { sessionFile: string; settledAt: string; model: string; route: ModelRoute;
|
|
446
|
+
key: string; cwd: string; bytes: number }[] = [];
|
|
372
447
|
const startedJobs = new Set<string>();
|
|
373
448
|
let activeSession: string | null = null;
|
|
374
449
|
let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
|
|
@@ -395,7 +470,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
395
470
|
return;
|
|
396
471
|
}
|
|
397
472
|
const child = spawn(process.execPath, [ENGINE_FILE, "session-trial", job.sessionFile,
|
|
398
|
-
job.settledAt, job.model, job.key, job.cwd, String(job.bytes)], {
|
|
473
|
+
job.settledAt, job.model, job.key, job.cwd, String(job.bytes), job.route.provider, job.route.modelId], {
|
|
399
474
|
cwd: job.cwd, env: process.env, stdio: ["ignore", "pipe", "pipe"],
|
|
400
475
|
detached: process.platform !== "win32",
|
|
401
476
|
});
|
|
@@ -453,7 +528,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
453
528
|
function queueSettledTask(ctx: ExtensionContext): void {
|
|
454
529
|
const sessionFile = ctx.sessionManager.getSessionFile();
|
|
455
530
|
const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
456
|
-
|
|
531
|
+
const route = ctx.model ? routeFor(ctx.model) : null;
|
|
532
|
+
if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
|
|
457
533
|
// The alpha compares completed response tasks. The engine validates images,
|
|
458
534
|
// tool use and the baseline answer before any provider calls.
|
|
459
535
|
let completed = false;
|
|
@@ -477,7 +553,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
477
553
|
if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
|
|
478
554
|
return;
|
|
479
555
|
}
|
|
480
|
-
queuedJobs.push({ sessionFile, settledAt, model, key: settledTaskKey, cwd: ctx.cwd, bytes });
|
|
556
|
+
queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
|
|
481
557
|
startNextJob(ctx);
|
|
482
558
|
}
|
|
483
559
|
|
|
@@ -489,11 +565,18 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
489
565
|
|
|
490
566
|
async function applyRecommendation(ctx: ExtensionContext, rec: Recommendation): Promise<boolean> {
|
|
491
567
|
const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
|
|
492
|
-
|
|
568
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : null;
|
|
569
|
+
if (rec.currentRoute && currentRoute &&
|
|
570
|
+
(rec.currentRoute.provider !== currentRoute.provider || rec.currentRoute.modelId !== currentRoute.modelId)) {
|
|
571
|
+
ctx.ui.notify(`openmerit: recommendation targets ${rec.currentRoute.provider}/${rec.currentRoute.modelId}; ` +
|
|
572
|
+
`current route is ${currentRoute.provider}/${currentRoute.modelId}`, "warning");
|
|
573
|
+
return false;
|
|
574
|
+
}
|
|
575
|
+
if (!rec.currentRoute && rec.currentModel && current !== rec.currentModel) {
|
|
493
576
|
ctx.ui.notify(`openmerit: recommendation was measured against ${rec.currentModel}; current model is ${current ?? "unknown"}`, "warning");
|
|
494
577
|
return false;
|
|
495
578
|
}
|
|
496
|
-
const model = ensureModel(pi, ctx, rec.recommended.model);
|
|
579
|
+
const model = ensureModel(pi, ctx, rec.recommended.model, rec.recommended.route);
|
|
497
580
|
if (!model) {
|
|
498
581
|
ctx.ui.notify(
|
|
499
582
|
`openmerit: ${rec.recommended.model} not found in pi's model registry (add the provider/model first)`,
|
|
@@ -504,8 +587,12 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
504
587
|
const ok = await pi.setModel(model as Parameters<typeof pi.setModel>[0]);
|
|
505
588
|
if (ok) {
|
|
506
589
|
setStatus(rec, "applied");
|
|
507
|
-
if (rec.fallback)
|
|
508
|
-
|
|
590
|
+
if (rec.fallback) {
|
|
591
|
+
fallbackModel = rec.fallback.model;
|
|
592
|
+
fallbackRoute = rec.fallback.route ?? null;
|
|
593
|
+
}
|
|
594
|
+
ctx.ui.notify(`openmerit: switched to ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n` +
|
|
595
|
+
rec.recommended.reason, "info");
|
|
509
596
|
reportHarnessState(ctx);
|
|
510
597
|
} else {
|
|
511
598
|
ctx.ui.notify(`openmerit: no auth for ${rec.recommended.model}`, "error");
|
|
@@ -537,7 +624,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
537
624
|
// startup. Notify (fire-and-forget) and let the user act via /openmerit.
|
|
538
625
|
if (ctx.hasUI) {
|
|
539
626
|
ctx.ui.notify(
|
|
540
|
-
`openmerit recommends: ${rec.recommended.model}\n${rec.recommended.reason}\n` +
|
|
627
|
+
`openmerit recommends: ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n${rec.recommended.reason}\n` +
|
|
541
628
|
`gate: ${rec.policy.reasons.join("; ")}\n(/openmerit apply to switch, /openmerit dismiss to ignore)`,
|
|
542
629
|
"info",
|
|
543
630
|
);
|
|
@@ -608,7 +695,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
608
695
|
const policy = loadPolicy();
|
|
609
696
|
if (policy.fallback?.apply_on_error === false) return;
|
|
610
697
|
const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "current model";
|
|
611
|
-
const model = ensureModel(pi, ctx, fallbackModel);
|
|
698
|
+
const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
|
|
612
699
|
if (!model) return;
|
|
613
700
|
|
|
614
701
|
const auto = policy.mode === "auto";
|
|
@@ -660,7 +747,7 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
660
747
|
ctx.ui.notify("openmerit: no fallback set yet", "info");
|
|
661
748
|
return;
|
|
662
749
|
}
|
|
663
|
-
const model = ensureModel(pi, ctx, fallbackModel);
|
|
750
|
+
const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
|
|
664
751
|
if (model && (await pi.setModel(model as Parameters<typeof pi.setModel>[0]))) {
|
|
665
752
|
ctx.ui.notify(`openmerit: switched to fallback ${fallbackModel}`, "info");
|
|
666
753
|
reportHarnessState(ctx);
|
|
@@ -670,7 +757,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
670
757
|
return;
|
|
671
758
|
}
|
|
672
759
|
|
|
673
|
-
const
|
|
760
|
+
const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
|
|
761
|
+
const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
|
|
674
762
|
const budget = trialBudget();
|
|
675
763
|
const liveProgress = progress?.sessionFile === ctx.sessionManager.getSessionFile() ? progress : null;
|
|
676
764
|
const comparison = trialJob ? liveProgress?.current
|
|
@@ -693,7 +781,8 @@ export default function openmerit(pi: ExtensionAPI) {
|
|
|
693
781
|
for (const r of pending.slice(-5)) {
|
|
694
782
|
lines.push(
|
|
695
783
|
"",
|
|
696
|
-
`-> ${r.recommended.model} (gain ${r.evidence.scoreGain},
|
|
784
|
+
`-> ${routeDisplay(r.recommended.route, r.recommended.model)} (gain ${r.evidence.scoreGain}, ` +
|
|
785
|
+
`${r.evidence.priceKnown === false ? "price unknown" : `${r.evidence.priceRatio}x price`}, frontier=${r.evidence.onFrontier})`,
|
|
697
786
|
` ${r.recommended.reason}`,
|
|
698
787
|
` gate: ${r.policy.reasons.join("; ") || "n/a"}`,
|
|
699
788
|
);
|
|
@@ -8,8 +8,9 @@ the best model for the task at hand, at the best price, with a vetted fallback.
|
|
|
8
8
|
|
|
9
9
|
1. **Observes** this harness's session traces (models used, tokens, cost,
|
|
10
10
|
latency, errors) without intercepting or slowing down your work.
|
|
11
|
-
2. **Evaluates** candidate models for each completed
|
|
12
|
-
|
|
11
|
+
2. **Evaluates** candidate models for each completed task sequentially through
|
|
12
|
+
pi and scores their outputs with a judge model. Candidate tools are disabled
|
|
13
|
+
by default.
|
|
13
14
|
3. **Maintains a pareto frontier** per task (quality vs. cost vs. latency) and
|
|
14
15
|
an aggregate frontier across all of your tasks.
|
|
15
16
|
4. **Watches for new model releases** (provider catalogs + public benchmarks)
|
|
@@ -43,9 +44,11 @@ the best model for the task at hand, at the best price, with a vetted fallback.
|
|
|
43
44
|
|
|
44
45
|
## Guarantees
|
|
45
46
|
|
|
46
|
-
- Per-task comparisons send the task text, uploaded images (when
|
|
47
|
-
candidate answers through
|
|
48
|
-
|
|
47
|
+
- Per-task comparisons send the task text, uploaded files or images (when
|
|
48
|
+
present), and candidate answers through the model routes configured in Pi
|
|
49
|
+
for model runs and judging. Candidate Pi runs default to no tools; any tool access is an
|
|
50
|
+
explicit user opt-in. Use non-sensitive examples while
|
|
51
|
+
evaluating this alpha.
|
|
49
52
|
- The shipped policy is supervised. Swaps happen automatically only after the
|
|
50
53
|
user opts in and the configured quality/cost guardrails pass; otherwise they
|
|
51
54
|
remain recommendations for a human to approve.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "openmerit",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.2",
|
|
4
4
|
"description": "Find better models for each pi task by comparing quality, cost, and latency.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": [
|
|
@@ -17,10 +17,13 @@
|
|
|
17
17
|
"extension/openmerit.ts",
|
|
18
18
|
"instructions/",
|
|
19
19
|
"examples/",
|
|
20
|
+
"rules.md",
|
|
20
21
|
"benchmark/invoice_ocr/data/"
|
|
21
22
|
],
|
|
22
23
|
"pi": {
|
|
23
|
-
"extensions": [
|
|
24
|
+
"extensions": [
|
|
25
|
+
"./extension/openmerit.ts"
|
|
26
|
+
]
|
|
24
27
|
},
|
|
25
28
|
"bin": {
|
|
26
29
|
"openmerit": "dist/cli.js"
|
package/rules.md
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# OpenMerit product and architecture rules
|
|
2
|
+
|
|
3
|
+
- **Optimize for useful model choices.** OpenMerit exists to find better task-specific tradeoffs between quality, cost, and latency—not to become a general-purpose agent or telemetry platform.
|
|
4
|
+
|
|
5
|
+
- **Act as a control plane, not a gateway.** Normal traffic stays between the harness and provider; OpenMerit observes evidence, runs explicit trials, and returns recommendations without intercepting ordinary work.
|
|
6
|
+
|
|
7
|
+
- **Make Pi plug-and-play first.** Pi and Pi-based harnesses are the first supported user experience because that is where early users already work; broader harness support can follow without weakening this path.
|
|
8
|
+
|
|
9
|
+
- **Keep the core harness-neutral.** Pi is the first `HarnessAdapter`, not a permanent assumption in scoring, recommendations, policy, or stored data, so other harnesses can be added without rewriting the merit loop.
|
|
10
|
+
|
|
11
|
+
- **Keep model providers interchangeable.** OpenRouter is a first-class provider and discovery source, not the foundation of the domain model; provider-specific behavior belongs behind a `ModelProviderAdapter`.
|
|
12
|
+
|
|
13
|
+
- **Require end-to-end provider neutrality.** Discovery, authentication, execution, scoring, and swapping must preserve the selected route; a feature is not provider-neutral if only its API client is abstracted.
|
|
14
|
+
|
|
15
|
+
- **Separate model identity from execution route.** Keep logical model IDs in stable `vendor/model` form, while recording the actual provider or route separately, so the same model can be compared through OpenRouter, a native provider, or a local provider.
|
|
16
|
+
|
|
17
|
+
- **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
|
|
18
|
+
|
|
19
|
+
- **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
|
|
20
|
+
|
|
21
|
+
- **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
|
|
22
|
+
|
|
23
|
+
- **Use provider-neutral core records.** Normalize integrations into stable concepts such as `TaskObservation`, `CandidateRun`, `TrialScore`, `Frontier`, `Recommendation`, and `MeritEvent`, so integrations do not leak their schemas into the decision engine.
|
|
24
|
+
|
|
25
|
+
- **Optimize per task before aggregating per agent.** Agent-level conclusions must be built from measured task evidence rather than assumed from global model rankings.
|
|
26
|
+
|
|
27
|
+
- **Version persisted events and evolve them additively.** Existing 0.1.x state must remain readable, and append-only recommendation history must stay intact; migrations should normalize old records rather than invalidate them.
|
|
28
|
+
|
|
29
|
+
- **Keep policy as the sole auto-swap authority.** Trials and strategists may recommend changes, but only the policy gate may approve automatic application, and every recommendation must retain its gate reasons.
|
|
30
|
+
|
|
31
|
+
- **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
|
|
32
|
+
|
|
33
|
+
- **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
|
|
34
|
+
|
|
35
|
+
- **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.
|
|
36
|
+
|
|
37
|
+
- **Keep integrations optional and the core lightweight.** New providers, harnesses, and observability services should not impose credentials, network calls, or heavy dependencies on users who do not enable them.
|
|
38
|
+
|
|
39
|
+
- **Keep 0.1.x focused.** The near-term bar is reliable, provider-neutral use for Pi tinkerers and solo hackers; additional harnesses and hosted observability integrations belong in later releases unless required to prove the boundaries work.
|