openmerit 0.1.2 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -9
- package/dist/cli.js +26 -153
- package/dist/daemon.js +27 -4
- package/dist/diagnostics.js +194 -0
- package/dist/pi-trials.js +18 -5
- package/dist/standalone.js +220 -0
- package/dist/store.js +42 -10
- package/dist/strategist.js +1 -1
- package/dist/trials.js +13 -4
- package/examples/task.example.json +1 -0
- package/extension/openmerit.ts +35 -9
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -54,13 +54,14 @@ package into pi, then initialize its policy and state:
|
|
|
54
54
|
```bash
|
|
55
55
|
pi install npm:openmerit
|
|
56
56
|
npx --yes openmerit init
|
|
57
|
+
npx --yes openmerit verify
|
|
57
58
|
pi list
|
|
58
59
|
```
|
|
59
60
|
|
|
60
61
|
`pi install` makes the extension and its trial engine available to pi. The
|
|
61
62
|
one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
|
|
62
63
|
globally with `npm install --global openmerit` only if you also want persistent
|
|
63
|
-
shell access to `openmerit status`, `doctor`, `frontier`, or the optional watcher. Install
|
|
64
|
+
shell access to `openmerit status`, `doctor`, `verify`, `frontier`, or the optional watcher. Install
|
|
64
65
|
the extension from only one source—remove any older copied `openmerit.ts` first
|
|
65
66
|
so pi does not load it twice.
|
|
66
67
|
|
|
@@ -120,8 +121,9 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
120
121
|
3. Verify the extension appears in `pi list`. The instruction file
|
|
121
122
|
[`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
|
|
122
123
|
pi project's AGENTS.md for agent context, but the extension does
|
|
123
|
-
not require it. Run `npx openmerit
|
|
124
|
-
|
|
124
|
+
not require it. Run `npx openmerit verify` for a provider-free core self-test,
|
|
125
|
+
then `npx openmerit doctor` after starting Pi once to inspect the eligible
|
|
126
|
+
route snapshot and configuration without printing secrets.
|
|
125
127
|
|
|
126
128
|
4. Start pi in one terminal with any configured model. For example:
|
|
127
129
|
|
|
@@ -179,8 +181,13 @@ schema, so its published scores are not directly comparable to these pi runs.
|
|
|
179
181
|
|
|
180
182
|
### Inspect or troubleshoot a run
|
|
181
183
|
|
|
182
|
-
`npx openmerit doctor` checks Pi, policy, eligible routes
|
|
183
|
-
|
|
184
|
+
`npx openmerit doctor` checks Pi, policy, eligible routes and prices, saved
|
|
185
|
+
session state, duplicate package sources, append-only files, and optional
|
|
186
|
+
OpenRouter enrichment. Add `--json` for a sanitized diagnostic report suitable
|
|
187
|
+
for a bug report; it contains no credentials, prompts, or trace contents.
|
|
188
|
+
`npx openmerit verify` runs an offline self-test of atomic state, JSONL recovery,
|
|
189
|
+
route preservation, policy evidence, and neutral events without contacting a
|
|
190
|
+
provider. `npx openmerit status` shows the latest pi model and pending recommendations;
|
|
184
191
|
`npx openmerit frontier` shows measured quality, blended price, latency, and
|
|
185
192
|
the chosen frontier per task. `/openmerit` inside pi shows the current model,
|
|
186
193
|
fallback, the model currently being compared, completed models with quality,
|
|
@@ -203,10 +210,21 @@ Closing or switching the session cancels the active job.
|
|
|
203
210
|
Candidate runs use the same text and uploaded image bytes, but they do not
|
|
204
211
|
replay earlier answers or file changes. An exact task in another session
|
|
205
212
|
(including identical image bytes) appears as **advice**, not a pending swap;
|
|
206
|
-
the new session still gets its own comparison.
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
213
|
+
the new session still gets its own comparison.
|
|
214
|
+
|
|
215
|
+
The optional `trial` command runs controlled text-task comparisons through the
|
|
216
|
+
same exact Pi provider routes and credential store as the automatic session
|
|
217
|
+
path:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
npx openmerit trial examples/task.example.json --rounds 3
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
`initial_model` is the stable `vendor/model` identity. When Pi exposes that
|
|
224
|
+
model through more than one provider, set `initial_route` to
|
|
225
|
+
`provider:model-id` (for example `openai:gpt-4o-mini` or
|
|
226
|
+
`openrouter:openai/gpt-4o-mini`). Judge and strategist calls also run through
|
|
227
|
+
Pi. An OpenRouter key only adds optional catalog and public-benchmark metadata.
|
|
210
228
|
|
|
211
229
|
## Alpha boundaries
|
|
212
230
|
|
|
@@ -251,6 +269,10 @@ session path is provider-neutral.
|
|
|
251
269
|
configuration). A provider registered only at runtime by another extension
|
|
252
270
|
is visible in the route snapshot but cannot yet be executed by the isolated
|
|
253
271
|
subprocess.
|
|
272
|
+
- JSON state snapshots are replaced atomically. Append-only readers skip and
|
|
273
|
+
report malformed or interrupted lines while retaining later valid records.
|
|
274
|
+
A job owned by a crashed process is reclaimable instead of remaining stuck
|
|
275
|
+
in `running`; completed jobs remain final.
|
|
254
276
|
|
|
255
277
|
## State layout (`~/.openmerit/`)
|
|
256
278
|
|
package/dist/cli.js
CHANGED
|
@@ -1,36 +1,18 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor. */
|
|
3
|
-
import { copyFileSync, existsSync, mkdirSync
|
|
2
|
+
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor / verify. */
|
|
3
|
+
import { copyFileSync, existsSync, mkdirSync } from "node:fs";
|
|
4
4
|
import { dirname, join } from "node:path";
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
6
6
|
import { loadPolicy } from "./policy.js";
|
|
7
7
|
import { loadKey } from "./llm.js";
|
|
8
|
-
import { fetchCatalog } from "./catalog.js";
|
|
9
8
|
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
10
|
-
import {
|
|
11
|
-
import { appendJsonl, paths, readJson, readJsonl, taskKey } from "./store.js";
|
|
9
|
+
import { paths, readJson, readJsonl } from "./store.js";
|
|
12
10
|
import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
|
|
13
|
-
import {
|
|
14
|
-
import { availablePiModels, availablePiRoutes, piExecutable, recordedActiveTask, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
15
|
-
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
16
|
-
import { JUDGE_PREFS } from "./judge.js";
|
|
17
|
-
import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
|
|
11
|
+
import { settledActiveTask } from "./pi-trials.js";
|
|
18
12
|
import { modelRoute, routeLabel } from "./routes.js";
|
|
19
|
-
import {
|
|
13
|
+
import { runStandaloneTrial } from "./standalone.js";
|
|
14
|
+
import { collectDoctorReport, renderDiagnosticReport, runOfflineVerify } from "./diagnostics.js";
|
|
20
15
|
const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
21
|
-
function resolvePref(cat, prefs, label) {
|
|
22
|
-
for (const p of prefs)
|
|
23
|
-
if (p && cat.has(p))
|
|
24
|
-
return p;
|
|
25
|
-
for (const p of prefs) {
|
|
26
|
-
if (!p)
|
|
27
|
-
continue;
|
|
28
|
-
for (const id of [...cat.keys()].sort())
|
|
29
|
-
if (id.includes(p))
|
|
30
|
-
return id;
|
|
31
|
-
}
|
|
32
|
-
throw new Error(`could not resolve ${label} model`);
|
|
33
|
-
}
|
|
34
16
|
function cmdInit() {
|
|
35
17
|
const policyPath = paths.policy();
|
|
36
18
|
mkdirSync(dirname(policyPath), { recursive: true });
|
|
@@ -55,129 +37,16 @@ function optionalOpenRouterKey() {
|
|
|
55
37
|
return null;
|
|
56
38
|
}
|
|
57
39
|
}
|
|
58
|
-
function
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
console.log(`pi executable : ${executable}`);
|
|
63
|
-
console.log(`pi detected : ${version.status === 0 ? `yes (${version.stdout.trim()})` : "no"}`);
|
|
64
|
-
try {
|
|
65
|
-
const policy = loadPolicy(paths.policy());
|
|
66
|
-
console.log(`policy : valid (v${policy.version}, ${policy.mode})`);
|
|
67
|
-
}
|
|
68
|
-
catch (error) {
|
|
69
|
-
console.log(`policy : invalid (${error.message})`);
|
|
70
|
-
}
|
|
71
|
-
let routes = state.routes ?? [];
|
|
72
|
-
const hasLiveSnapshot = routes.length > 0;
|
|
73
|
-
if (routes.length === 0 && version.status === 0) {
|
|
74
|
-
try {
|
|
75
|
-
routes = availablePiRoutes();
|
|
76
|
-
}
|
|
77
|
-
catch { /* reported below */ }
|
|
78
|
-
}
|
|
79
|
-
console.log(`eligible routes: ${routes.length}`);
|
|
80
|
-
for (const route of routes.slice(0, 8))
|
|
81
|
-
console.log(` - ${routeLabel(route)}${route.cost ? "" : " (price unknown)"}`);
|
|
82
|
-
if (routes.length > 8)
|
|
83
|
-
console.log(` … ${routes.length - 8} more`);
|
|
84
|
-
console.log(`active route : ${state.currentRoute ? routeLabel(state.currentRoute) : "not reported; start pi once"}`);
|
|
85
|
-
const pricedRoutes = routes.filter((route) => !!route.cost).length;
|
|
86
|
-
console.log(`alternates : ${hasLiveSnapshot
|
|
87
|
-
? pricedRoutes >= 2 ? "ready" : "need at least two priced eligible pi routes"
|
|
88
|
-
: "start Pi once to capture route prices and capabilities"}`);
|
|
89
|
-
console.log(`OpenRouter enrichment: ${optionalOpenRouterKey() ? "available" : "not configured (optional)"}`);
|
|
40
|
+
function printReport(report, json) {
|
|
41
|
+
console.log(json ? JSON.stringify(report, null, 2) : renderDiagnosticReport(report));
|
|
42
|
+
if (!report.ok)
|
|
43
|
+
process.exitCode = 1;
|
|
90
44
|
}
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
const cat = await fetchCatalog(key);
|
|
97
|
-
const piModels = availablePiModels();
|
|
98
|
-
for (const id of [...cat.keys()])
|
|
99
|
-
if (!piModels.has(id))
|
|
100
|
-
cat.delete(id);
|
|
101
|
-
if (!cat.has(cfg.initial_model))
|
|
102
|
-
throw new Error(`initial_model ${cfg.initial_model} is unavailable in pi's OpenRouter registry`);
|
|
103
|
-
console.log(`catalog: ${cat.size} models selectable in pi`);
|
|
104
|
-
const judge = resolvePref(cat, [cfg.judge_model ?? policy.judge_model, ...JUDGE_PREFS], "judge");
|
|
105
|
-
const strat = resolvePref(cat, [cfg.strategist_model ?? policy.strategist_model, ...STRAT_PREFS], "strategist");
|
|
106
|
-
const { category, benchmarks } = relevantBenchmarks(cfg.task);
|
|
107
|
-
let benchmarkCandidates = [];
|
|
108
|
-
try {
|
|
109
|
-
benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, cat);
|
|
110
|
-
}
|
|
111
|
-
catch (e) {
|
|
112
|
-
console.log(`OpenRouter benchmark shortlist unavailable: ${e.message}`);
|
|
113
|
-
}
|
|
114
|
-
console.log(`judge: ${judge} | strategist: ${strat} | benchmarks: ${benchmarks.join(", ")}`);
|
|
115
|
-
console.log(`OpenRouter benchmark candidates: ${benchmarkCandidates.slice(0, 5).join(", ") || "none"}`);
|
|
116
|
-
const tKey = taskKey(cfg.task);
|
|
117
|
-
const maxPrice = cfg.max_usd_per_m ?? policy.max_usd_per_m;
|
|
118
|
-
const tried = new Set();
|
|
119
|
-
const results = [];
|
|
120
|
-
const failedVendors = new Set();
|
|
121
|
-
for (let i = 0; i < rounds; i++) {
|
|
122
|
-
const budget = budgetOk(policy);
|
|
123
|
-
if (!budget.ok) {
|
|
124
|
-
console.log(`budget: ${budget.reason}; stopping`);
|
|
125
|
-
break;
|
|
126
|
-
}
|
|
127
|
-
let model;
|
|
128
|
-
let why;
|
|
129
|
-
if (i === 0) {
|
|
130
|
-
model = cfg.initial_model;
|
|
131
|
-
why = "initial model";
|
|
132
|
-
}
|
|
133
|
-
else {
|
|
134
|
-
const pick = await pickNext(key, strat, cfg.task, cfg.eval, benchmarks, results, cat, tried, maxPrice, failedVendors, benchmarkCandidates);
|
|
135
|
-
if (!pick)
|
|
136
|
-
break;
|
|
137
|
-
model = pick.model;
|
|
138
|
-
why = pick.why;
|
|
139
|
-
}
|
|
140
|
-
if (tried.has(model))
|
|
141
|
-
break;
|
|
142
|
-
console.log(`[round ${i + 1}/${rounds}] ${model} (${why}) ...`);
|
|
143
|
-
const observed = i === 0 ? recordedActiveTask(cfg.task, model) : null;
|
|
144
|
-
if (i === 0)
|
|
145
|
-
console.log(observed ? " using model A's active pi session trace" : " active trace unavailable; running model A in a fresh pi session");
|
|
146
|
-
const { point, costUsd, error, sessionId } = observed
|
|
147
|
-
? await scorePiRun(key, judge, cfg.task, cfg.eval, model, cat.get(model), observed, "trace")
|
|
148
|
-
: await runPiTrial(key, judge, cfg.task, cfg.eval, model, cat.get(model));
|
|
149
|
-
tried.add(model);
|
|
150
|
-
results.push(point);
|
|
151
|
-
recordTrialSpend(costUsd);
|
|
152
|
-
appendJsonl(paths.trials(), { ...point, taskKey: tKey });
|
|
153
|
-
if (sessionId)
|
|
154
|
-
console.log(` pi trace session: ${sessionId}`);
|
|
155
|
-
if (error) {
|
|
156
|
-
failedVendors.add(model.split("/")[0]);
|
|
157
|
-
console.log(` FAILED: ${error.slice(0, 160)}`);
|
|
158
|
-
}
|
|
159
|
-
else {
|
|
160
|
-
console.log(` score=${point.score.toFixed(2)} $/M=${point.price} est_cost=$${costUsd.toFixed(4)} ${(point.why ?? "").slice(0, 110)}`);
|
|
161
|
-
}
|
|
162
|
-
}
|
|
163
|
-
const pts = results.filter((p) => p.score > 0);
|
|
164
|
-
if (pts.length === 0) {
|
|
165
|
-
console.log("no trials completed");
|
|
166
|
-
process.exit(1);
|
|
167
|
-
}
|
|
168
|
-
const chain = paretoFrontier(pts);
|
|
169
|
-
const best = pickBest(pts);
|
|
170
|
-
const fb = pickFallback(pts, best, policy.fallback.min_score);
|
|
171
|
-
console.log("\n=== pareto frontier (nondominated: score up, price down, latency down) ===");
|
|
172
|
-
for (const p of chain)
|
|
173
|
-
console.log(` ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M`);
|
|
174
|
-
console.log(`\nBEST FIT : ${best.model} (score ${best.score.toFixed(2)}, $${best.price.toFixed(2)}/M)`);
|
|
175
|
-
console.log(fb ? `FALLBACK : ${fb.model} (score ${fb.score.toFixed(2)}, $${fb.price.toFixed(2)}/M)` : "FALLBACK : n/a");
|
|
176
|
-
const rec = buildRecommendation(tKey, cfg.task.slice(0, 120), cfg.initial_model, pts, policy);
|
|
177
|
-
if (rec) {
|
|
178
|
-
appendJsonl(paths.recommendations(), rec);
|
|
179
|
-
console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${rec.policy.autoApply})`);
|
|
180
|
-
}
|
|
45
|
+
function cmdDoctor(json) {
|
|
46
|
+
printReport(collectDoctorReport(), json);
|
|
47
|
+
}
|
|
48
|
+
function cmdVerify(json) {
|
|
49
|
+
printReport(runOfflineVerify(), json);
|
|
181
50
|
}
|
|
182
51
|
function loadFrontiers() {
|
|
183
52
|
const trials = readJsonl(paths.trials());
|
|
@@ -243,12 +112,12 @@ function cmdStatus() {
|
|
|
243
112
|
}
|
|
244
113
|
async function main() {
|
|
245
114
|
const [cmd, ...args] = process.argv.slice(2);
|
|
246
|
-
const policy = loadPolicy(paths.policy());
|
|
247
115
|
switch (cmd) {
|
|
248
116
|
case "init":
|
|
249
117
|
cmdInit();
|
|
250
118
|
break;
|
|
251
119
|
case "session-trial": {
|
|
120
|
+
const policy = loadPolicy(paths.policy());
|
|
252
121
|
const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
|
|
253
122
|
if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
|
|
254
123
|
throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
|
|
@@ -269,29 +138,32 @@ async function main() {
|
|
|
269
138
|
}
|
|
270
139
|
case "watch":
|
|
271
140
|
if (args.includes("--once"))
|
|
272
|
-
await tickOnce(policy);
|
|
141
|
+
await tickOnce(loadPolicy(paths.policy()));
|
|
273
142
|
else
|
|
274
|
-
await runDaemon(policy);
|
|
143
|
+
await runDaemon(loadPolicy(paths.policy()));
|
|
275
144
|
break;
|
|
276
145
|
case "trial": {
|
|
277
146
|
const file = args[0];
|
|
278
147
|
if (!file)
|
|
279
148
|
throw new Error("usage: openmerit trial <task.json> [--rounds N]");
|
|
280
149
|
const rIdx = args.indexOf("--rounds");
|
|
281
|
-
await
|
|
150
|
+
await runStandaloneTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
|
|
282
151
|
break;
|
|
283
152
|
}
|
|
284
153
|
case "frontier":
|
|
285
154
|
cmdFrontier(args[0]);
|
|
286
155
|
break;
|
|
287
156
|
case "recommend":
|
|
288
|
-
recommendTick(policy);
|
|
157
|
+
recommendTick(loadPolicy(paths.policy()));
|
|
289
158
|
break;
|
|
290
159
|
case "status":
|
|
291
160
|
cmdStatus();
|
|
292
161
|
break;
|
|
293
162
|
case "doctor":
|
|
294
|
-
cmdDoctor();
|
|
163
|
+
cmdDoctor(args.includes("--json"));
|
|
164
|
+
break;
|
|
165
|
+
case "verify":
|
|
166
|
+
cmdVerify(args.includes("--json"));
|
|
295
167
|
break;
|
|
296
168
|
default:
|
|
297
169
|
console.log(`openmerit — external model-merit harness
|
|
@@ -304,7 +176,8 @@ usage: openmerit <command>
|
|
|
304
176
|
frontier [taskKey] print pareto frontier(s)
|
|
305
177
|
recommend emit recommendations now
|
|
306
178
|
status harness model, queued models, pending recommendations
|
|
307
|
-
doctor
|
|
179
|
+
doctor [--json] diagnose pi, policy, routes, state, and installation
|
|
180
|
+
verify [--json] run an offline, provider-free core self-test`);
|
|
308
181
|
if (cmd && cmd !== "help" && cmd !== "--help")
|
|
309
182
|
process.exitCode = 1;
|
|
310
183
|
}
|
package/dist/daemon.js
CHANGED
|
@@ -42,6 +42,28 @@ function resolveRoutePref(cat, prefs, label, requireImages = false) {
|
|
|
42
42
|
return fallback;
|
|
43
43
|
throw new Error(`could not resolve ${label} route from Pi's eligible models`);
|
|
44
44
|
}
|
|
45
|
+
function processIsAlive(pid) {
|
|
46
|
+
if (!Number.isInteger(pid) || pid <= 0)
|
|
47
|
+
return false;
|
|
48
|
+
try {
|
|
49
|
+
process.kill(pid, 0);
|
|
50
|
+
return true;
|
|
51
|
+
}
|
|
52
|
+
catch (error) {
|
|
53
|
+
return error.code === "EPERM";
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
/** Completed jobs stay final; live owners retain their claim; abandoned claims can retry. */
|
|
57
|
+
export function taskJobIsClaimed(prior, marker, now = Date.now()) {
|
|
58
|
+
if (!prior || prior.marker !== marker || prior.status === "failed")
|
|
59
|
+
return false;
|
|
60
|
+
if (prior.status !== "running")
|
|
61
|
+
return true;
|
|
62
|
+
if (prior.ownerPid !== undefined)
|
|
63
|
+
return processIsAlive(prior.ownerPid);
|
|
64
|
+
const updated = Date.parse(prior.updatedAt);
|
|
65
|
+
return Number.isFinite(updated) && now - updated < 15 * 60_000;
|
|
66
|
+
}
|
|
45
67
|
function emitTrialProgress(progress) {
|
|
46
68
|
console.log(`[openmerit/progress] ${JSON.stringify(progress)}`);
|
|
47
69
|
}
|
|
@@ -158,12 +180,13 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
158
180
|
: `${task.sessionFile}:${task.sessionBytes ?? task.settledAt}`;
|
|
159
181
|
const processedPath = settledTask === undefined ? paths.watchProcessed() : paths.sessionJob(marker);
|
|
160
182
|
const prior = readJson(processedPath, null);
|
|
161
|
-
if (prior
|
|
183
|
+
if (taskJobIsClaimed(prior, marker))
|
|
162
184
|
return null;
|
|
163
185
|
if (!budgetOk(policy).ok)
|
|
164
186
|
return null;
|
|
165
187
|
// Claim the settled turn before provider calls so the next poll cannot duplicate it.
|
|
166
|
-
writeJson(processedPath, { marker, status: "running",
|
|
188
|
+
writeJson(processedPath, { marker, status: "running", ownerPid: process.pid,
|
|
189
|
+
updatedAt: new Date().toISOString() });
|
|
167
190
|
try {
|
|
168
191
|
let cat = routeCatalog(task.routes);
|
|
169
192
|
if (!cat.has(routeKey(task.route)))
|
|
@@ -258,12 +281,12 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
258
281
|
else
|
|
259
282
|
console.log(`[openmerit] task ${tKey}: active model remains best`);
|
|
260
283
|
writeJson(processedPath, { marker, status: rec ? "complete" : "no_swap",
|
|
261
|
-
updatedAt: new Date().toISOString() });
|
|
284
|
+
ownerPid: process.pid, updatedAt: new Date().toISOString() });
|
|
262
285
|
return rec;
|
|
263
286
|
}
|
|
264
287
|
catch (error) {
|
|
265
288
|
writeJson(processedPath, { marker, status: "failed",
|
|
266
|
-
updatedAt: new Date().toISOString() });
|
|
289
|
+
ownerPid: process.pid, updatedAt: new Date().toISOString() });
|
|
267
290
|
throw error;
|
|
268
291
|
}
|
|
269
292
|
}
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
/** Read-only diagnostics plus a provider-free self-test for first-run support. */
|
|
2
|
+
import { spawnSync } from "node:child_process";
|
|
3
|
+
import { appendFileSync, existsSync, mkdtempSync, readFileSync, rmSync } from "node:fs";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import { basename, join } from "node:path";
|
|
6
|
+
import { LocalJsonlEventSink, meritEvent } from "./integrations.js";
|
|
7
|
+
import { piExecutable, availablePiRoutes } from "./pi-trials.js";
|
|
8
|
+
import { DEFAULT_POLICY, loadPolicy, providerAllowed } from "./policy.js";
|
|
9
|
+
import { buildRecommendation } from "./recommend.js";
|
|
10
|
+
import { modelRoute, routeCatalog, routeKey, routeLabel } from "./routes.js";
|
|
11
|
+
import { appendJsonl, paths, readJsonReport, readJsonlReport, writeJson, } from "./store.js";
|
|
12
|
+
function check(id, status, message, details) {
|
|
13
|
+
return { id, status, message, ...(details?.length ? { details } : {}) };
|
|
14
|
+
}
|
|
15
|
+
function packageSources(output) {
|
|
16
|
+
return [...new Set(output.split("\n").filter((line) => /^ {2}\S/.test(line)).map((line) => line.trim())
|
|
17
|
+
.filter((line) => /openmerit/i.test(line) && /^(npm:|git:|https?:|\.?\.?\/|\/)/.test(line)))];
|
|
18
|
+
}
|
|
19
|
+
function safePackageSource(source) {
|
|
20
|
+
if (source.startsWith("npm:/"))
|
|
21
|
+
return `npm:local/${basename(source)}`;
|
|
22
|
+
if (source.startsWith("/"))
|
|
23
|
+
return `local:${basename(source)}`;
|
|
24
|
+
return source;
|
|
25
|
+
}
|
|
26
|
+
/** No credentials, prompts, or trace contents are included in this report. */
|
|
27
|
+
export function collectDoctorReport() {
|
|
28
|
+
const checks = [];
|
|
29
|
+
const executable = piExecutable();
|
|
30
|
+
const versionResult = spawnSync(executable, ["--version"], { encoding: "utf8" });
|
|
31
|
+
const piVersion = versionResult.status === 0 ? versionResult.stdout.trim() : null;
|
|
32
|
+
checks.push(piVersion
|
|
33
|
+
? check("pi", "pass", `Pi detected (${piVersion})`)
|
|
34
|
+
: check("pi", "fail", "Pi is not executable"));
|
|
35
|
+
try {
|
|
36
|
+
const policy = loadPolicy(paths.policy());
|
|
37
|
+
checks.push(check("policy", "pass", `Policy is valid (v${policy.version}, ${policy.mode})`));
|
|
38
|
+
}
|
|
39
|
+
catch (error) {
|
|
40
|
+
checks.push(check("policy", "fail", `Policy is invalid: ${error.message}`));
|
|
41
|
+
}
|
|
42
|
+
const stateRead = readJsonReport(paths.harnessState(), {});
|
|
43
|
+
checks.push(!stateRead.valid
|
|
44
|
+
? check("harness-state", "fail", "harness-state.json is malformed; start Pi to replace it atomically")
|
|
45
|
+
: stateRead.exists
|
|
46
|
+
? check("harness-state", "pass", "Pi extension state snapshot is readable")
|
|
47
|
+
: check("harness-state", "warn", "No Pi extension state yet; start Pi once with OpenMerit installed"));
|
|
48
|
+
let routes = stateRead.value.routes ?? [];
|
|
49
|
+
const liveSnapshot = routes.length > 0;
|
|
50
|
+
if (!routes.length && piVersion) {
|
|
51
|
+
try {
|
|
52
|
+
routes = availablePiRoutes();
|
|
53
|
+
}
|
|
54
|
+
catch { /* the check below explains it */ }
|
|
55
|
+
}
|
|
56
|
+
let policy = DEFAULT_POLICY;
|
|
57
|
+
try {
|
|
58
|
+
policy = loadPolicy(paths.policy());
|
|
59
|
+
}
|
|
60
|
+
catch { /* already reported */ }
|
|
61
|
+
routes = routes.filter((route) => providerAllowed(policy, route.meritId, route.provider));
|
|
62
|
+
const priced = routes.filter((route) => !!route.cost);
|
|
63
|
+
checks.push(routes.length >= 2
|
|
64
|
+
? check("routes", "pass", `${routes.length} Pi model routes are visible`, routes.slice(0, 8).map((route) => `${routeLabel(route)}${route.cost ? "" : " (price unknown)"}`))
|
|
65
|
+
: check("routes", "fail", `Only ${routes.length} policy-eligible Pi route${routes.length === 1 ? " is" : "s are"} visible; at least two are required`));
|
|
66
|
+
checks.push(liveSnapshot
|
|
67
|
+
? priced.length >= 2
|
|
68
|
+
? check("pricing", "pass", `${priced.length} eligible routes expose pricing`)
|
|
69
|
+
: check("pricing", "warn", "Fewer than two eligible routes expose pricing; unknown-price routes cannot auto-apply")
|
|
70
|
+
: check("pricing", "warn", "Route prices are unavailable until the extension reports a live Pi snapshot"));
|
|
71
|
+
const session = stateRead.value.sessionFile;
|
|
72
|
+
checks.push(!session
|
|
73
|
+
? check("session", "warn", "No active saved Pi session is reported yet")
|
|
74
|
+
: existsSync(session)
|
|
75
|
+
? check("session", "pass", "The reported Pi session file is readable")
|
|
76
|
+
: check("session", "warn", "The reported Pi session file no longer exists"));
|
|
77
|
+
if (piVersion) {
|
|
78
|
+
const listed = spawnSync(executable, ["list"], { encoding: "utf8" });
|
|
79
|
+
const sources = listed.status === 0 ? packageSources(listed.stdout) : [];
|
|
80
|
+
const safeSources = sources.map(safePackageSource);
|
|
81
|
+
checks.push(sources.length > 1
|
|
82
|
+
? check("installation", "fail", "OpenMerit is installed from multiple Pi package sources", safeSources)
|
|
83
|
+
: sources.length === 1
|
|
84
|
+
? check("installation", "pass", `One Pi package source is installed`, safeSources)
|
|
85
|
+
: check("installation", "warn", "OpenMerit is not listed as an installed Pi package; use `pi install npm:openmerit`"));
|
|
86
|
+
}
|
|
87
|
+
const jsonlFiles = [
|
|
88
|
+
["recommendations.jsonl", paths.recommendations()],
|
|
89
|
+
["trials.jsonl", paths.trials()],
|
|
90
|
+
["traces/observations.jsonl", paths.observations()],
|
|
91
|
+
["events.jsonl", paths.events()],
|
|
92
|
+
];
|
|
93
|
+
const malformed = jsonlFiles.flatMap(([label, file]) => {
|
|
94
|
+
const invalid = readJsonlReport(file).invalidLines;
|
|
95
|
+
return invalid.length ? [`${label}: lines ${invalid.join(", ")}`] : [];
|
|
96
|
+
});
|
|
97
|
+
checks.push(malformed.length
|
|
98
|
+
? check("jsonl", "warn", "Malformed JSONL lines were skipped; valid later records remain readable", malformed)
|
|
99
|
+
: check("jsonl", "pass", "Append-only JSONL state is readable"));
|
|
100
|
+
const ledger = readJsonReport(paths.ledger(), { date: "", trials: 0, usd: 0 });
|
|
101
|
+
checks.push(ledger.valid
|
|
102
|
+
? check("ledger", "pass", "Trial-spend ledger is readable")
|
|
103
|
+
: check("ledger", "fail", "Trial-spend ledger is malformed; spending remains disabled until repaired"));
|
|
104
|
+
let openRouter = false;
|
|
105
|
+
try {
|
|
106
|
+
const env = process.env.OPENROUTER_API_KEY;
|
|
107
|
+
if (env?.trim())
|
|
108
|
+
openRouter = true;
|
|
109
|
+
else if (existsSync(paths.envFile()))
|
|
110
|
+
openRouter = /^\s*OPENROUTER_API_KEY\s*=\s*\S+/m
|
|
111
|
+
.test(readFileSync(paths.envFile(), "utf8"));
|
|
112
|
+
}
|
|
113
|
+
catch { /* optional */ }
|
|
114
|
+
checks.push(check("openrouter-enrichment", "pass", openRouter ? "Optional OpenRouter enrichment is configured" : "Optional OpenRouter enrichment is not configured"));
|
|
115
|
+
const activeRoute = stateRead.value.currentRoute ? routeLabel(stateRead.value.currentRoute) : null;
|
|
116
|
+
return {
|
|
117
|
+
schemaVersion: 1,
|
|
118
|
+
ok: !checks.some((item) => item.status === "fail"),
|
|
119
|
+
checks,
|
|
120
|
+
summary: { piVersion, routeCount: routes.length, pricedRouteCount: priced.length, activeRoute },
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
export function renderDiagnosticReport(report) {
|
|
124
|
+
const icon = { pass: "PASS", warn: "WARN", fail: "FAIL" };
|
|
125
|
+
return report.checks.flatMap((item) => [
|
|
126
|
+
`${icon[item.status].padEnd(4)} ${item.id.padEnd(22)} ${item.message}`,
|
|
127
|
+
...(item.details ?? []).map((detail) => ` - ${detail}`),
|
|
128
|
+
]).join("\n");
|
|
129
|
+
}
|
|
130
|
+
/** Exercise core state, routing, recommendation, and event paths without provider calls. */
|
|
131
|
+
export function runOfflineVerify() {
|
|
132
|
+
const previousHome = process.env.OPENMERIT_HOME;
|
|
133
|
+
const root = mkdtempSync(join(tmpdir(), "openmerit-verify-"));
|
|
134
|
+
const checks = [];
|
|
135
|
+
try {
|
|
136
|
+
process.env.OPENMERIT_HOME = root;
|
|
137
|
+
const active = modelRoute("native", "model-a", {
|
|
138
|
+
input: ["text"], cost: { input: 1, output: 2 }, contextWindow: 8192, maxTokens: 4096,
|
|
139
|
+
});
|
|
140
|
+
const candidate = modelRoute("local", "model-b", {
|
|
141
|
+
input: ["text"], cost: { input: 0.2, output: 0.4 }, contextWindow: 8192, maxTokens: 4096,
|
|
142
|
+
});
|
|
143
|
+
writeJson(paths.harnessState(), { schemaVersion: 1, currentRoute: active, routes: [active, candidate] });
|
|
144
|
+
const state = readJsonReport(paths.harnessState(), {});
|
|
145
|
+
checks.push(state.valid && state.value.routes?.length === 2
|
|
146
|
+
? check("atomic-json", "pass", "Atomic JSON state round-trip passed")
|
|
147
|
+
: check("atomic-json", "fail", "Atomic JSON state round-trip failed"));
|
|
148
|
+
const entries = routeCatalog([active, candidate]);
|
|
149
|
+
const point = (route, score) => {
|
|
150
|
+
const entry = entries.get(routeKey(route));
|
|
151
|
+
return { schemaVersion: 1, model: route.meritId, route, score, price: entry.price,
|
|
152
|
+
priceKnown: entry.priceKnown, ts: new Date().toISOString(), source: "pi_trial" };
|
|
153
|
+
};
|
|
154
|
+
const recommendation = buildRecommendation("verify-task", "offline verification", active.meritId, [point(active, 0.6), point(candidate, 0.9)], DEFAULT_POLICY, "/verify/session.jsonl", active);
|
|
155
|
+
checks.push(recommendation?.recommended.route?.provider === "local" &&
|
|
156
|
+
recommendation.policy.reasons.length > 0
|
|
157
|
+
? check("provider-neutral", "pass", "Exact provider routes and policy reasons survive recommendation")
|
|
158
|
+
: check("provider-neutral", "fail", "Provider route or policy evidence was lost"));
|
|
159
|
+
if (recommendation)
|
|
160
|
+
appendJsonl(paths.recommendations(), recommendation);
|
|
161
|
+
appendJsonl(paths.recommendations(), { schemaVersion: 0, id: "legacy", status: "dismissed" });
|
|
162
|
+
appendFileSync(paths.recommendations(), "{interrupted");
|
|
163
|
+
const history = readJsonlReport(paths.recommendations());
|
|
164
|
+
checks.push(history.records.length === 2 && history.invalidLines.length === 1
|
|
165
|
+
? check("jsonl-recovery", "pass", "Valid records survive legacy and interrupted JSONL lines")
|
|
166
|
+
: check("jsonl-recovery", "fail", "JSONL recovery did not preserve valid history"));
|
|
167
|
+
const sink = new LocalJsonlEventSink();
|
|
168
|
+
if (recommendation)
|
|
169
|
+
sink.emit(meritEvent("recommendation.created", recommendation));
|
|
170
|
+
const events = readJsonlReport(paths.events());
|
|
171
|
+
checks.push(events.records.length === 1
|
|
172
|
+
? check("events", "pass", "Versioned neutral event sink round-trip passed")
|
|
173
|
+
: check("events", "fail", "Neutral event sink round-trip failed"));
|
|
174
|
+
checks.push(loadPolicy(paths.policy()).mode === "recommend"
|
|
175
|
+
? check("compatibility", "pass", "Missing 0.1.x policy fields normalize to safe defaults")
|
|
176
|
+
: check("compatibility", "fail", "Policy defaults are unsafe"));
|
|
177
|
+
}
|
|
178
|
+
catch (error) {
|
|
179
|
+
checks.push(check("verify", "fail", error.message));
|
|
180
|
+
}
|
|
181
|
+
finally {
|
|
182
|
+
if (previousHome === undefined)
|
|
183
|
+
delete process.env.OPENMERIT_HOME;
|
|
184
|
+
else
|
|
185
|
+
process.env.OPENMERIT_HOME = previousHome;
|
|
186
|
+
rmSync(root, { recursive: true, force: true });
|
|
187
|
+
}
|
|
188
|
+
return {
|
|
189
|
+
schemaVersion: 1,
|
|
190
|
+
ok: !checks.some((item) => item.status === "fail"),
|
|
191
|
+
checks,
|
|
192
|
+
summary: { piVersion: null, routeCount: 2, pricedRouteCount: 2, activeRoute: "native/model-a via native" },
|
|
193
|
+
};
|
|
194
|
+
}
|
package/dist/pi-trials.js
CHANGED
|
@@ -113,7 +113,9 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
|
|
|
113
113
|
if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
|
|
114
114
|
taskInputKey(task, images, files))
|
|
115
115
|
return null;
|
|
116
|
-
const matching = assistants.filter((m) =>
|
|
116
|
+
const matching = assistants.filter((m) => typeof model === "string"
|
|
117
|
+
? meritModelId(m.provider ?? "openrouter", m.model ?? "") === model
|
|
118
|
+
: m.provider === model.provider && m.model === model.modelId);
|
|
117
119
|
if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
|
|
118
120
|
return null;
|
|
119
121
|
const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
|
|
@@ -170,10 +172,10 @@ export function settledActiveTask(snapshot) {
|
|
|
170
172
|
const files = sessionFiles(task);
|
|
171
173
|
if (!task.trim() || taskInputKey(task, images, files) !== st.settledTaskKey)
|
|
172
174
|
return null;
|
|
173
|
-
const
|
|
175
|
+
const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
|
|
176
|
+
const run = recordedActiveTask(task, route, images, st.sessionFile, st.sessionBytes);
|
|
174
177
|
if (!run)
|
|
175
178
|
return null;
|
|
176
|
-
const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
|
|
177
179
|
const routes = st.routes?.length ? st.routes : [route];
|
|
178
180
|
return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
|
|
179
181
|
settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
|
|
@@ -182,6 +184,13 @@ export function settledActiveTask(snapshot) {
|
|
|
182
184
|
export function piExecutable() {
|
|
183
185
|
return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
|
|
184
186
|
}
|
|
187
|
+
function compactNumber(value) {
|
|
188
|
+
const match = value.match(/^(\d+(?:\.\d+)?)([KMG])?$/i);
|
|
189
|
+
if (!match)
|
|
190
|
+
return undefined;
|
|
191
|
+
const scale = { K: 1_000, M: 1_000_000, G: 1_000_000_000 }[(match[2] ?? "").toUpperCase()] ?? 1;
|
|
192
|
+
return Math.round(Number(match[1]) * scale);
|
|
193
|
+
}
|
|
185
194
|
/** Read provider/model routes from pi when no live extension snapshot is available. */
|
|
186
195
|
export function availablePiRoutes(snapshot) {
|
|
187
196
|
if (snapshot?.length)
|
|
@@ -191,10 +200,14 @@ export function availablePiRoutes(snapshot) {
|
|
|
191
200
|
throw new Error("could not list models from pi");
|
|
192
201
|
const routes = [];
|
|
193
202
|
for (const line of result.stdout.split("\n").slice(1)) {
|
|
194
|
-
const [provider, id, , , , images] = line.trim().split(/\s+/);
|
|
203
|
+
const [provider, id, context, maxOut, , images] = line.trim().split(/\s+/);
|
|
195
204
|
if (!provider || !id || id.startsWith("~"))
|
|
196
205
|
continue;
|
|
197
|
-
routes.push(modelRoute(provider, id, {
|
|
206
|
+
routes.push(modelRoute(provider, id, {
|
|
207
|
+
input: images === "yes" ? ["text", "image"] : ["text"],
|
|
208
|
+
contextWindow: compactNumber(context),
|
|
209
|
+
maxTokens: compactNumber(maxOut),
|
|
210
|
+
}));
|
|
198
211
|
}
|
|
199
212
|
return routes;
|
|
200
213
|
}
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
/** Provider-neutral standalone task trials executed through Pi routes. */
|
|
2
|
+
import { readFileSync } from "node:fs";
|
|
3
|
+
import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
|
|
4
|
+
import { fetchCatalog, saveSnapshot } from "./catalog.js";
|
|
5
|
+
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
6
|
+
import { JUDGE_PREFS } from "./judge.js";
|
|
7
|
+
import { loadKey, PiCliChatClient } from "./llm.js";
|
|
8
|
+
import { availablePiRoutes, recordedActiveTask, runPiTrial, scorePiRun } from "./pi-trials.js";
|
|
9
|
+
import { loadPolicy, providerAllowed } from "./policy.js";
|
|
10
|
+
import { buildRecommendation } from "./recommend.js";
|
|
11
|
+
import { enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
|
|
12
|
+
import { appendJsonl, paths, readJson, taskKey } from "./store.js";
|
|
13
|
+
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
14
|
+
import { budgetOk, recordMeritSpend, recordTrialSpend, trialBudgetOk } from "./trials.js";
|
|
15
|
+
function optionalOpenRouterKey() {
|
|
16
|
+
try {
|
|
17
|
+
return loadKey();
|
|
18
|
+
}
|
|
19
|
+
catch {
|
|
20
|
+
return null;
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
function configuredRoutes() {
|
|
24
|
+
const state = readJson(paths.harnessState(), {});
|
|
25
|
+
let routes = state.routes?.length ? state.routes : availablePiRoutes();
|
|
26
|
+
if (state.currentRoute && !routes.some((route) => routeKey(route) === routeKey(state.currentRoute)))
|
|
27
|
+
routes = [state.currentRoute, ...routes];
|
|
28
|
+
const unique = new Map(routes.map((route) => [routeKey(route), route]));
|
|
29
|
+
return { routes: [...unique.values()], currentRoute: state.currentRoute ?? null };
|
|
30
|
+
}
|
|
31
|
+
function selectorKey(selector) {
|
|
32
|
+
if (!selector)
|
|
33
|
+
return null;
|
|
34
|
+
if (typeof selector === "string")
|
|
35
|
+
return selector;
|
|
36
|
+
const modelId = selector.modelId ?? selector.model_id;
|
|
37
|
+
return selector.provider && modelId ? `${selector.provider}:${modelId}` : null;
|
|
38
|
+
}
|
|
39
|
+
function matchingEntries(cat, selector) {
|
|
40
|
+
const exact = cat.get(selector);
|
|
41
|
+
if (exact)
|
|
42
|
+
return [exact];
|
|
43
|
+
return [...cat.values()].filter((entry) => entry.id === selector || entry.route?.meritId === selector ||
|
|
44
|
+
(entry.route && `${entry.route.provider}/${entry.route.modelId}` === selector));
|
|
45
|
+
}
|
|
46
|
+
function resolveInitial(cat, cfg, currentRoute) {
|
|
47
|
+
const exactKey = selectorKey(cfg.initial_route);
|
|
48
|
+
if (cfg.initial_route && !exactKey)
|
|
49
|
+
throw new Error("initial_route must include provider and modelId");
|
|
50
|
+
if (exactKey) {
|
|
51
|
+
const exact = cat.get(exactKey);
|
|
52
|
+
if (!exact)
|
|
53
|
+
throw new Error(`initial route ${exactKey} is not eligible in Pi`);
|
|
54
|
+
if (cfg.initial_model !== exact.id && cfg.initial_model !== exactKey)
|
|
55
|
+
throw new Error(`initial_route ${exactKey} resolves to ${exact.id}, not initial_model ${cfg.initial_model}`);
|
|
56
|
+
return exact;
|
|
57
|
+
}
|
|
58
|
+
const matches = matchingEntries(cat, cfg.initial_model);
|
|
59
|
+
if (matches.length === 0)
|
|
60
|
+
throw new Error(`initial_model ${cfg.initial_model} is not eligible in Pi`);
|
|
61
|
+
if (matches.length === 1)
|
|
62
|
+
return matches[0];
|
|
63
|
+
if (currentRoute) {
|
|
64
|
+
const current = matches.find((entry) => entry.route && routeKey(entry.route) === routeKey(currentRoute));
|
|
65
|
+
if (current)
|
|
66
|
+
return current;
|
|
67
|
+
}
|
|
68
|
+
throw new Error(`initial_model ${cfg.initial_model} has multiple Pi routes; set initial_route to one of: ` +
|
|
69
|
+
matches.map((entry) => routeKey(entry.route)).join(", "));
|
|
70
|
+
}
|
|
71
|
+
function resolveMeritRoute(cat, prefs, label) {
|
|
72
|
+
for (const pref of prefs) {
|
|
73
|
+
if (!pref)
|
|
74
|
+
continue;
|
|
75
|
+
const matches = matchingEntries(cat, pref);
|
|
76
|
+
if (matches.length)
|
|
77
|
+
return matches.sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
|
|
78
|
+
}
|
|
79
|
+
for (const pref of prefs) {
|
|
80
|
+
if (!pref)
|
|
81
|
+
continue;
|
|
82
|
+
const fuzzy = [...cat.values()].find((entry) => entry.id.includes(pref));
|
|
83
|
+
if (fuzzy)
|
|
84
|
+
return fuzzy;
|
|
85
|
+
}
|
|
86
|
+
const fallback = [...cat.values()].sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
|
|
87
|
+
if (fallback)
|
|
88
|
+
return fallback;
|
|
89
|
+
throw new Error(`could not resolve ${label} route from Pi's eligible models`);
|
|
90
|
+
}
|
|
91
|
+
async function buildCatalog(routes, key) {
|
|
92
|
+
let cat = routeCatalog(routes);
|
|
93
|
+
if (!key)
|
|
94
|
+
return cat;
|
|
95
|
+
try {
|
|
96
|
+
const live = await fetchCatalog(key);
|
|
97
|
+
saveSnapshot(live);
|
|
98
|
+
cat = new Map([...cat].map(([id, entry]) => [id,
|
|
99
|
+
entry.route?.provider === "openrouter" ? enrichRouteEntry(entry, live.get(entry.id)) : entry]));
|
|
100
|
+
}
|
|
101
|
+
catch (error) {
|
|
102
|
+
console.log(`OpenRouter enrichment unavailable: ${error.message}`);
|
|
103
|
+
}
|
|
104
|
+
return cat;
|
|
105
|
+
}
|
|
106
|
+
export async function runStandaloneTrial(taskFile, rounds) {
|
|
107
|
+
const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
|
|
108
|
+
if (!cfg.task?.trim() || !cfg.eval?.trim() || !cfg.initial_model?.trim())
|
|
109
|
+
throw new Error("task JSON requires task, eval, and initial_model");
|
|
110
|
+
if (!Number.isInteger(rounds) || rounds < 1)
|
|
111
|
+
throw new Error("--rounds must be a positive integer");
|
|
112
|
+
const policy = loadPolicy(paths.policy());
|
|
113
|
+
const key = optionalOpenRouterKey();
|
|
114
|
+
const { routes, currentRoute } = configuredRoutes();
|
|
115
|
+
let cat = await buildCatalog(routes, key);
|
|
116
|
+
for (const [id, entry] of [...cat]) {
|
|
117
|
+
if (!entry.route || !providerAllowed(policy, entry.id, entry.route.provider) ||
|
|
118
|
+
(entry.priceKnown !== false && entry.price > (cfg.max_usd_per_m ?? policy.max_usd_per_m)))
|
|
119
|
+
cat.delete(id);
|
|
120
|
+
}
|
|
121
|
+
if (cat.size === 0)
|
|
122
|
+
throw new Error("Pi exposes no model routes allowed by the current policy");
|
|
123
|
+
const initial = resolveInitial(cat, cfg, currentRoute);
|
|
124
|
+
const judge = resolveMeritRoute(cat, [cfg.judge_model, policy.judge_model, ...JUDGE_PREFS, initial.id], "judge");
|
|
125
|
+
const strategist = resolveMeritRoute(cat, [cfg.strategist_model, policy.strategist_model, ...STRAT_PREFS, initial.id], "strategist");
|
|
126
|
+
const client = new PiCliChatClient(undefined, recordMeritSpend);
|
|
127
|
+
const { category, benchmarks } = relevantBenchmarks(cfg.task);
|
|
128
|
+
let benchmarkCandidates = [];
|
|
129
|
+
if (key) {
|
|
130
|
+
try {
|
|
131
|
+
benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, new Map([...cat.values()].map((entry) => [entry.id, entry])));
|
|
132
|
+
}
|
|
133
|
+
catch (error) {
|
|
134
|
+
console.log(`OpenRouter benchmark shortlist unavailable: ${error.message}`);
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
console.log(`routes: ${cat.size} eligible in Pi${key ? " (OpenRouter enrichment enabled)" : ""}`);
|
|
138
|
+
console.log(`initial: ${routeLabel(initial.route)} | judge: ${routeLabel(judge.route)} | ` +
|
|
139
|
+
`strategist: ${routeLabel(strategist.route)} | benchmarks: ${benchmarks.join(", ")}`);
|
|
140
|
+
const tKey = taskKey(cfg.task);
|
|
141
|
+
const tried = new Set();
|
|
142
|
+
const results = [];
|
|
143
|
+
const failedProviders = new Set();
|
|
144
|
+
for (let i = 0; i < rounds; i++) {
|
|
145
|
+
const daily = budgetOk(policy);
|
|
146
|
+
if (!daily.ok) {
|
|
147
|
+
console.log(`budget: ${daily.reason}; stopping`);
|
|
148
|
+
break;
|
|
149
|
+
}
|
|
150
|
+
let entry;
|
|
151
|
+
let why;
|
|
152
|
+
if (i === 0) {
|
|
153
|
+
entry = initial;
|
|
154
|
+
why = "initial route";
|
|
155
|
+
}
|
|
156
|
+
else {
|
|
157
|
+
const pick = await pickNext(key ?? "", strategist.route, cfg.task, cfg.eval, benchmarks, results, cat, tried, cfg.max_usd_per_m ?? policy.max_usd_per_m, failedProviders, benchmarkCandidates, client);
|
|
158
|
+
if (!pick)
|
|
159
|
+
break;
|
|
160
|
+
entry = pick.route ? cat.get(routeKey(pick.route)) : cat.get(pick.model);
|
|
161
|
+
why = pick.why;
|
|
162
|
+
}
|
|
163
|
+
const route = entry.route;
|
|
164
|
+
const keyForRoute = routeKey(route);
|
|
165
|
+
if (tried.has(keyForRoute))
|
|
166
|
+
break;
|
|
167
|
+
if (i > 0) {
|
|
168
|
+
const admission = trialBudgetOk(policy, entry, cfg.task.length);
|
|
169
|
+
if (!admission.ok) {
|
|
170
|
+
tried.add(keyForRoute);
|
|
171
|
+
console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} skipped: ${admission.reason}`);
|
|
172
|
+
continue;
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} (${why}) ...`);
|
|
176
|
+
const observed = i === 0 ? recordedActiveTask(cfg.task, route) : null;
|
|
177
|
+
if (i === 0)
|
|
178
|
+
console.log(observed
|
|
179
|
+
? " using the exact provider route from Pi's active session trace"
|
|
180
|
+
: " active trace unavailable; running the initial route in a fresh Pi session");
|
|
181
|
+
const outcome = observed
|
|
182
|
+
? await scorePiRun(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, observed, "trace", [], client)
|
|
183
|
+
: await runPiTrial(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, cfg.cwd ?? process.cwd(), [], [], undefined, client);
|
|
184
|
+
tried.add(keyForRoute);
|
|
185
|
+
results.push(outcome.point);
|
|
186
|
+
// Reusing the active Pi answer is observation, not a new candidate call.
|
|
187
|
+
if (!observed)
|
|
188
|
+
recordTrialSpend(outcome.costUsd);
|
|
189
|
+
appendJsonl(paths.trials(), { ...outcome.point, taskKey: tKey });
|
|
190
|
+
if (outcome.sessionId)
|
|
191
|
+
console.log(` pi trace session: ${outcome.sessionId}`);
|
|
192
|
+
if (outcome.error) {
|
|
193
|
+
failedProviders.add(route.provider);
|
|
194
|
+
console.log(` FAILED: ${outcome.error.slice(0, 160)}`);
|
|
195
|
+
}
|
|
196
|
+
else {
|
|
197
|
+
const price = outcome.point.priceKnown === false ? "price unknown" : `$${outcome.point.price}/M`;
|
|
198
|
+
console.log(` score=${outcome.point.score.toFixed(2)} ${price} ` +
|
|
199
|
+
`run=$${outcome.costUsd.toFixed(4)} ${(outcome.point.why ?? "").slice(0, 100)}`);
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
const scored = results.filter((point) => point.score > 0);
|
|
203
|
+
if (scored.length === 0)
|
|
204
|
+
throw new Error("no trials completed");
|
|
205
|
+
const frontier = paretoFrontier(scored);
|
|
206
|
+
const best = pickBest(scored);
|
|
207
|
+
const fallback = pickFallback(scored, best, policy.fallback.min_score);
|
|
208
|
+
console.log("\n=== pareto frontier (quality up, price and latency down) ===");
|
|
209
|
+
for (const point of frontier)
|
|
210
|
+
console.log(` ${(point.route ? routeLabel(point.route) : point.model).padEnd(52)} ` +
|
|
211
|
+
`score=${point.score.toFixed(2)} ${point.priceKnown === false ? "price unknown" : `$${point.price.toFixed(2)}/M`}`);
|
|
212
|
+
console.log(`\nBEST FIT : ${best.route ? routeLabel(best.route) : best.model}`);
|
|
213
|
+
console.log(fallback ? `FALLBACK : ${fallback.route ? routeLabel(fallback.route) : fallback.model}` : "FALLBACK : n/a");
|
|
214
|
+
const recommendation = buildRecommendation(tKey, cfg.task.slice(0, 120), initial.id, results, policy, null, initial.route);
|
|
215
|
+
if (recommendation) {
|
|
216
|
+
appendJsonl(paths.recommendations(), recommendation);
|
|
217
|
+
console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${recommendation.policy.autoApply})`);
|
|
218
|
+
}
|
|
219
|
+
return { points: results, recommendation };
|
|
220
|
+
}
|
package/dist/store.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/** State directory layout + JSONL persistence. All state lives under OPENMERIT_HOME (default ~/.openmerit). */
|
|
2
2
|
import { createHash } from "node:crypto";
|
|
3
|
-
import { appendFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
3
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync, } from "node:fs";
|
|
4
4
|
import { homedir } from "node:os";
|
|
5
5
|
import { join } from "node:path";
|
|
6
6
|
export function stateDir() {
|
|
@@ -38,20 +38,52 @@ export function appendJsonl(file, obj) {
|
|
|
38
38
|
mkdirSync(join(file, ".."), { recursive: true });
|
|
39
39
|
appendFileSync(file, JSON.stringify(obj) + "\n");
|
|
40
40
|
}
|
|
41
|
+
/**
|
|
42
|
+
* Read append-only state without letting one interrupted or malformed line
|
|
43
|
+
* hide the remaining history. Diagnostics can surface invalid line numbers.
|
|
44
|
+
*/
|
|
45
|
+
export function readJsonlReport(file) {
|
|
46
|
+
if (!existsSync(file))
|
|
47
|
+
return { records: [], invalidLines: [] };
|
|
48
|
+
const records = [];
|
|
49
|
+
const invalidLines = [];
|
|
50
|
+
for (const [index, line] of readFileSync(file, "utf8").split("\n").entries()) {
|
|
51
|
+
if (!line.trim())
|
|
52
|
+
continue;
|
|
53
|
+
try {
|
|
54
|
+
records.push(JSON.parse(line));
|
|
55
|
+
}
|
|
56
|
+
catch {
|
|
57
|
+
invalidLines.push(index + 1);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
return { records, invalidLines };
|
|
61
|
+
}
|
|
41
62
|
export function readJsonl(file) {
|
|
63
|
+
return readJsonlReport(file).records;
|
|
64
|
+
}
|
|
65
|
+
export function readJsonReport(file, fallback) {
|
|
42
66
|
if (!existsSync(file))
|
|
43
|
-
return
|
|
44
|
-
|
|
45
|
-
.
|
|
46
|
-
|
|
47
|
-
|
|
67
|
+
return { value: fallback, exists: false, valid: true };
|
|
68
|
+
try {
|
|
69
|
+
return { value: JSON.parse(readFileSync(file, "utf8")), exists: true, valid: true };
|
|
70
|
+
}
|
|
71
|
+
catch (error) {
|
|
72
|
+
return { value: fallback, exists: true, valid: false, error: error.message };
|
|
73
|
+
}
|
|
48
74
|
}
|
|
49
75
|
export function readJson(file, fallback) {
|
|
50
|
-
|
|
51
|
-
return fallback;
|
|
52
|
-
return JSON.parse(readFileSync(file, "utf8"));
|
|
76
|
+
return readJsonReport(file, fallback).value;
|
|
53
77
|
}
|
|
54
78
|
export function writeJson(file, obj) {
|
|
55
79
|
mkdirSync(join(file, ".."), { recursive: true });
|
|
56
|
-
|
|
80
|
+
const temp = `${file}.tmp-${process.pid}-${Date.now()}`;
|
|
81
|
+
try {
|
|
82
|
+
writeFileSync(temp, JSON.stringify(obj, null, 2) + "\n");
|
|
83
|
+
renameSync(temp, file);
|
|
84
|
+
}
|
|
85
|
+
finally {
|
|
86
|
+
if (existsSync(temp))
|
|
87
|
+
rmSync(temp, { force: true });
|
|
88
|
+
}
|
|
57
89
|
}
|
package/dist/strategist.js
CHANGED
|
@@ -21,7 +21,7 @@ ROUTES (route id | logical model | blended $/1M tokens | ctx):
|
|
|
21
21
|
/** Pick the next model to trial, or null when the catalog is exhausted. */
|
|
22
22
|
export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
|
|
23
23
|
const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
|
|
24
|
-
const ok = (c) =>
|
|
24
|
+
const ok = (c) => c.priceKnown !== false && c.price <= maxPrice &&
|
|
25
25
|
!tried.has(entryKey(c)) &&
|
|
26
26
|
c.ctx >= 4096 &&
|
|
27
27
|
!failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
|
package/dist/trials.js
CHANGED
|
@@ -1,12 +1,15 @@
|
|
|
1
1
|
/** Shadow trials: run candidate models against observed/declared tasks on the background track. */
|
|
2
2
|
import { directChatClient } from "./llm.js";
|
|
3
3
|
import { judge, parseObj } from "./judge.js";
|
|
4
|
-
import { paths,
|
|
4
|
+
import { paths, readJsonReport, writeJson } from "./store.js";
|
|
5
5
|
function today() {
|
|
6
6
|
return new Date().toISOString().slice(0, 10);
|
|
7
7
|
}
|
|
8
8
|
export function budgetOk(policy) {
|
|
9
|
-
const
|
|
9
|
+
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
10
|
+
if (!report.valid)
|
|
11
|
+
return { ok: false, reason: "trial ledger is malformed; run `openmerit doctor`" };
|
|
12
|
+
const ledger = report.value;
|
|
10
13
|
if (ledger.date !== today())
|
|
11
14
|
return { ok: true }; // new day resets
|
|
12
15
|
if (ledger.trials >= policy.budgets.max_trials_per_day)
|
|
@@ -31,7 +34,10 @@ export function trialBudgetOk(policy, entry, inputChars) {
|
|
|
31
34
|
return { ok: true, estimatedUsd };
|
|
32
35
|
}
|
|
33
36
|
export function recordTrialSpend(usd) {
|
|
34
|
-
|
|
37
|
+
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
38
|
+
if (!report.valid)
|
|
39
|
+
throw new Error("trial ledger is malformed; refusing to reset spend");
|
|
40
|
+
let ledger = report.value;
|
|
35
41
|
if (ledger.date !== today())
|
|
36
42
|
ledger = { date: today(), trials: 0, usd: 0 };
|
|
37
43
|
ledger.trials += 1;
|
|
@@ -42,7 +48,10 @@ export function recordTrialSpend(usd) {
|
|
|
42
48
|
export function recordMeritSpend(usd) {
|
|
43
49
|
if (!(usd > 0))
|
|
44
50
|
return;
|
|
45
|
-
|
|
51
|
+
const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
|
|
52
|
+
if (!report.valid)
|
|
53
|
+
throw new Error("trial ledger is malformed; refusing to reset spend");
|
|
54
|
+
let ledger = report.value;
|
|
46
55
|
if (ledger.date !== today())
|
|
47
56
|
ledger = { date: today(), trials: 0, usd: 0 };
|
|
48
57
|
ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
|
|
@@ -2,5 +2,6 @@
|
|
|
2
2
|
"task": "Write a Python function `def word_ladder(begin, end, words)` that returns the SHORTEST transformation sequence from `begin` to `end`, where consecutive words differ by exactly one character and every intermediate word must be in `words`. Return [] if no path exists. If begin == end return [begin]. All words are the same length, lowercase a-z. Output only runnable code, no explanation.",
|
|
3
3
|
"eval": "Score 1.0 requires ALL of: (a) algorithm guaranteed to find a shortest path (BFS or equivalent, NOT DFS/greedy); (b) returned path starts with begin and ends with end with all intermediates in words; (c) returns [] when impossible; (d) begin==end returns [begin]; (e) begin need not be in words; (f) syntactically valid, runnable Python, no external imports, function named word_ladder; (g) output contains only code. Deduct ~0.2 per missing criterion; non-shortest-path algorithms score at most 0.5.",
|
|
4
4
|
"initial_model": "openai/gpt-4o-mini",
|
|
5
|
+
"initial_route": "openai:gpt-4o-mini",
|
|
5
6
|
"max_usd_per_m": 20
|
|
6
7
|
}
|
package/extension/openmerit.ts
CHANGED
|
@@ -17,7 +17,9 @@
|
|
|
17
17
|
*/
|
|
18
18
|
|
|
19
19
|
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
20
|
-
import {
|
|
20
|
+
import {
|
|
21
|
+
appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync,
|
|
22
|
+
} from "node:fs";
|
|
21
23
|
import { homedir } from "node:os";
|
|
22
24
|
import { basename, join } from "node:path";
|
|
23
25
|
import { createHash } from "node:crypto";
|
|
@@ -84,7 +86,12 @@ function trialBudget(): { summary: string; reason: string | null } {
|
|
|
84
86
|
const today = new Date().toISOString().slice(0, 10);
|
|
85
87
|
let ledger: { date?: string; trials?: number; usd?: number } = {};
|
|
86
88
|
try { ledger = JSON.parse(readFileSync(LEDGER_FILE, "utf8")) as typeof ledger; }
|
|
87
|
-
catch {
|
|
89
|
+
catch {
|
|
90
|
+
if (existsSync(LEDGER_FILE)) return {
|
|
91
|
+
summary: "ledger unreadable",
|
|
92
|
+
reason: "trial ledger is malformed; run `openmerit doctor` before spending",
|
|
93
|
+
};
|
|
94
|
+
}
|
|
88
95
|
const trials = ledger.date === today ? ledger.trials ?? 0 : 0;
|
|
89
96
|
const usd = ledger.date === today ? ledger.usd ?? 0 : 0;
|
|
90
97
|
const reason = trials >= limit ? `daily trial cap reached (${limit})`
|
|
@@ -107,10 +114,12 @@ function latestSessionJob(sessionFile: string | undefined): string | null {
|
|
|
107
114
|
}
|
|
108
115
|
|
|
109
116
|
function loadPolicy(): ExtensionPolicy {
|
|
117
|
+
if (!existsSync(POLICY_FILE)) return {};
|
|
110
118
|
try {
|
|
111
119
|
return JSON.parse(readFileSync(POLICY_FILE, "utf8")) as ExtensionPolicy;
|
|
112
120
|
} catch {
|
|
113
|
-
return {}
|
|
121
|
+
return { mode: "recommend", fallback: { apply_on_error: false },
|
|
122
|
+
budgets: { max_trials_per_day: 0, max_usd_per_day: 0 } };
|
|
114
123
|
}
|
|
115
124
|
}
|
|
116
125
|
|
|
@@ -173,7 +182,17 @@ function appendJsonl(file: string, obj: unknown): void {
|
|
|
173
182
|
function latestRecommendations(): Recommendation[] {
|
|
174
183
|
const byId = new Map<string, Recommendation>();
|
|
175
184
|
for (const r of readJsonl<Recommendation>(RECS_FILE)) byId.set(r.id, r);
|
|
176
|
-
|
|
185
|
+
let malformed = false;
|
|
186
|
+
if (existsSync(RECS_FILE)) {
|
|
187
|
+
for (const line of readFileSync(RECS_FILE, "utf8").split("\n")) {
|
|
188
|
+
if (!line.trim()) continue;
|
|
189
|
+
try { JSON.parse(line); } catch { malformed = true; break; }
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return [...byId.values()].map((rec) => malformed && rec.status === "pending" && rec.policy.autoApply
|
|
193
|
+
? { ...rec, policy: { autoApply: false,
|
|
194
|
+
reasons: [...rec.policy.reasons, "recommendation history contains a malformed line; manual review required"] } }
|
|
195
|
+
: rec);
|
|
177
196
|
}
|
|
178
197
|
|
|
179
198
|
function taskKey(text: string): string {
|
|
@@ -305,6 +324,16 @@ function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
|
|
|
305
324
|
return [...byRoute.values()];
|
|
306
325
|
}
|
|
307
326
|
|
|
327
|
+
function writeState(value: unknown): void {
|
|
328
|
+
const temp = `${STATE_FILE}.tmp-${process.pid}-${Date.now()}`;
|
|
329
|
+
try {
|
|
330
|
+
writeFileSync(temp, JSON.stringify(value) + "\n");
|
|
331
|
+
renameSync(temp, STATE_FILE);
|
|
332
|
+
} finally {
|
|
333
|
+
if (existsSync(temp)) rmSync(temp, { force: true });
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
|
|
308
337
|
function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
309
338
|
try {
|
|
310
339
|
mkdirSync(HOME, { recursive: true });
|
|
@@ -317,9 +346,7 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
317
346
|
const sessionFile = ctx.sessionManager.getSessionFile() ?? null;
|
|
318
347
|
const settledAt = settled ? new Date().toISOString()
|
|
319
348
|
: prior.sessionFile === sessionFile ? prior.settledAt ?? null : null;
|
|
320
|
-
|
|
321
|
-
STATE_FILE,
|
|
322
|
-
JSON.stringify({
|
|
349
|
+
writeState({
|
|
323
350
|
schemaVersion: 1,
|
|
324
351
|
currentModel: model,
|
|
325
352
|
currentRoute,
|
|
@@ -331,8 +358,7 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
|
|
|
331
358
|
settledAt,
|
|
332
359
|
cwd: ctx.cwd,
|
|
333
360
|
updatedAt: new Date().toISOString(),
|
|
334
|
-
})
|
|
335
|
-
);
|
|
361
|
+
});
|
|
336
362
|
} catch {
|
|
337
363
|
/* never break the host session over reporting */
|
|
338
364
|
}
|