openmerit 0.1.2 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -54,13 +54,14 @@ package into pi, then initialize its policy and state:
54
54
  ```bash
55
55
  pi install npm:openmerit
56
56
  npx --yes openmerit init
57
+ npx --yes openmerit verify
57
58
  pi list
58
59
  ```
59
60
 
60
61
  `pi install` makes the extension and its trial engine available to pi. The
61
62
  one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
62
63
  globally with `npm install --global openmerit` only if you also want persistent
63
- shell access to `openmerit status`, `doctor`, `frontier`, or the optional watcher. Install
64
+ shell access to `openmerit status`, `doctor`, `verify`, `frontier`, or the optional watcher. Install
64
65
  the extension from only one source—remove any older copied `openmerit.ts` first
65
66
  so pi does not load it twice.
66
67
 
@@ -120,8 +121,9 @@ Keep that checkout available because pi loads a local package from its path.
120
121
  3. Verify the extension appears in `pi list`. The instruction file
121
122
  [`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
122
123
  pi project's AGENTS.md for agent context, but the extension does
123
- not require it. Run `npx openmerit doctor` after starting Pi once to inspect
124
- the eligible route snapshot and configuration without printing secrets.
124
+ not require it. Run `npx openmerit verify` for a provider-free core self-test,
125
+ then `npx openmerit doctor` after starting Pi once to inspect the eligible
126
+ route snapshot and configuration without printing secrets.
125
127
 
126
128
  4. Start pi in one terminal with any configured model. For example:
127
129
 
@@ -179,8 +181,13 @@ schema, so its published scores are not directly comparable to these pi runs.
179
181
 
180
182
  ### Inspect or troubleshoot a run
181
183
 
182
- `npx openmerit doctor` checks Pi, policy, eligible routes, and optional
183
- OpenRouter enrichment. `npx openmerit status` shows the latest pi model and pending recommendations;
184
+ `npx openmerit doctor` checks Pi, policy, eligible routes and prices, saved
185
+ session state, duplicate package sources, append-only files, and optional
186
+ OpenRouter enrichment. Add `--json` for a sanitized diagnostic report suitable
187
+ for a bug report; it contains no credentials, prompts, or trace contents.
188
+ `npx openmerit verify` runs an offline self-test of atomic state, JSONL recovery,
189
+ route preservation, policy evidence, and neutral events without contacting a
190
+ provider. `npx openmerit status` shows the latest pi model and pending recommendations;
184
191
  `npx openmerit frontier` shows measured quality, blended price, latency, and
185
192
  the chosen frontier per task. `/openmerit` inside pi shows the current model,
186
193
  fallback, the model currently being compared, completed models with quality,
@@ -203,10 +210,21 @@ Closing or switching the session cancels the active job.
203
210
  Candidate runs use the same text and uploaded image bytes, but they do not
204
211
  replay earlier answers or file changes. An exact task in another session
205
212
  (including identical image bytes) appears as **advice**, not a pending swap;
206
- the new session still gets its own comparison. The optional `trial` command
207
- remains for controlled text-task runs from a task JSON file; that legacy
208
- standalone command is still OpenRouter-specific in 0.1.2. The automatic Pi
209
- session path is provider-neutral.
213
+ the new session still gets its own comparison.
214
+
215
+ The optional `trial` command runs controlled text-task comparisons through the
216
+ same exact Pi provider routes and credential store as the automatic session
217
+ path:
218
+
219
+ ```bash
220
+ npx openmerit trial examples/task.example.json --rounds 3
221
+ ```
222
+
223
+ `initial_model` is the stable `vendor/model` identity. When Pi exposes that
224
+ model through more than one provider, set `initial_route` to
225
+ `provider:model-id` (for example `openai:gpt-4o-mini` or
226
+ `openrouter:openai/gpt-4o-mini`). Judge and strategist calls also run through
227
+ Pi. An OpenRouter key only adds optional catalog and public-benchmark metadata.
210
228
 
211
229
  ## Alpha boundaries
212
230
 
@@ -251,6 +269,10 @@ session path is provider-neutral.
251
269
  configuration). A provider registered only at runtime by another extension
252
270
  is visible in the route snapshot but cannot yet be executed by the isolated
253
271
  subprocess.
272
+ - JSON state snapshots are replaced atomically. Append-only readers skip and
273
+ report malformed or interrupted lines while retaining later valid records.
274
+ A job owned by a crashed process is reclaimable instead of remaining stuck
275
+ in `running`; completed jobs remain final.
254
276
 
255
277
  ## State layout (`~/.openmerit/`)
256
278
 
package/dist/cli.js CHANGED
@@ -1,36 +1,18 @@
1
1
  #!/usr/bin/env node
2
- /** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor. */
3
- import { copyFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
2
+ /** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor / verify. */
3
+ import { copyFileSync, existsSync, mkdirSync } from "node:fs";
4
4
  import { dirname, join } from "node:path";
5
5
  import { fileURLToPath } from "node:url";
6
6
  import { loadPolicy } from "./policy.js";
7
7
  import { loadKey } from "./llm.js";
8
- import { fetchCatalog } from "./catalog.js";
9
8
  import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
10
- import { buildRecommendation } from "./recommend.js";
11
- import { appendJsonl, paths, readJson, readJsonl, taskKey } from "./store.js";
9
+ import { paths, readJson, readJsonl } from "./store.js";
12
10
  import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
13
- import { budgetOk, recordTrialSpend } from "./trials.js";
14
- import { availablePiModels, availablePiRoutes, piExecutable, recordedActiveTask, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
15
- import { pickNext, STRAT_PREFS } from "./strategist.js";
16
- import { JUDGE_PREFS } from "./judge.js";
17
- import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
11
+ import { settledActiveTask } from "./pi-trials.js";
18
12
  import { modelRoute, routeLabel } from "./routes.js";
19
- import { spawnSync } from "node:child_process";
13
+ import { runStandaloneTrial } from "./standalone.js";
14
+ import { collectDoctorReport, renderDiagnosticReport, runOfflineVerify } from "./diagnostics.js";
20
15
  const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
21
- function resolvePref(cat, prefs, label) {
22
- for (const p of prefs)
23
- if (p && cat.has(p))
24
- return p;
25
- for (const p of prefs) {
26
- if (!p)
27
- continue;
28
- for (const id of [...cat.keys()].sort())
29
- if (id.includes(p))
30
- return id;
31
- }
32
- throw new Error(`could not resolve ${label} model`);
33
- }
34
16
  function cmdInit() {
35
17
  const policyPath = paths.policy();
36
18
  mkdirSync(dirname(policyPath), { recursive: true });
@@ -55,129 +37,16 @@ function optionalOpenRouterKey() {
55
37
  return null;
56
38
  }
57
39
  }
58
- function cmdDoctor() {
59
- const executable = piExecutable();
60
- const version = spawnSync(executable, ["--version"], { encoding: "utf8" });
61
- const state = readJson(paths.harnessState(), {});
62
- console.log(`pi executable : ${executable}`);
63
- console.log(`pi detected : ${version.status === 0 ? `yes (${version.stdout.trim()})` : "no"}`);
64
- try {
65
- const policy = loadPolicy(paths.policy());
66
- console.log(`policy : valid (v${policy.version}, ${policy.mode})`);
67
- }
68
- catch (error) {
69
- console.log(`policy : invalid (${error.message})`);
70
- }
71
- let routes = state.routes ?? [];
72
- const hasLiveSnapshot = routes.length > 0;
73
- if (routes.length === 0 && version.status === 0) {
74
- try {
75
- routes = availablePiRoutes();
76
- }
77
- catch { /* reported below */ }
78
- }
79
- console.log(`eligible routes: ${routes.length}`);
80
- for (const route of routes.slice(0, 8))
81
- console.log(` - ${routeLabel(route)}${route.cost ? "" : " (price unknown)"}`);
82
- if (routes.length > 8)
83
- console.log(` … ${routes.length - 8} more`);
84
- console.log(`active route : ${state.currentRoute ? routeLabel(state.currentRoute) : "not reported; start pi once"}`);
85
- const pricedRoutes = routes.filter((route) => !!route.cost).length;
86
- console.log(`alternates : ${hasLiveSnapshot
87
- ? pricedRoutes >= 2 ? "ready" : "need at least two priced eligible pi routes"
88
- : "start Pi once to capture route prices and capabilities"}`);
89
- console.log(`OpenRouter enrichment: ${optionalOpenRouterKey() ? "available" : "not configured (optional)"}`);
40
+ function printReport(report, json) {
41
+ console.log(json ? JSON.stringify(report, null, 2) : renderDiagnosticReport(report));
42
+ if (!report.ok)
43
+ process.exitCode = 1;
90
44
  }
91
- async function cmdTrial(taskFile, rounds) {
92
- const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
93
- const policy = loadPolicy(paths.policy());
94
- const key = loadKey();
95
- console.log("fetching OpenRouter catalog...");
96
- const cat = await fetchCatalog(key);
97
- const piModels = availablePiModels();
98
- for (const id of [...cat.keys()])
99
- if (!piModels.has(id))
100
- cat.delete(id);
101
- if (!cat.has(cfg.initial_model))
102
- throw new Error(`initial_model ${cfg.initial_model} is unavailable in pi's OpenRouter registry`);
103
- console.log(`catalog: ${cat.size} models selectable in pi`);
104
- const judge = resolvePref(cat, [cfg.judge_model ?? policy.judge_model, ...JUDGE_PREFS], "judge");
105
- const strat = resolvePref(cat, [cfg.strategist_model ?? policy.strategist_model, ...STRAT_PREFS], "strategist");
106
- const { category, benchmarks } = relevantBenchmarks(cfg.task);
107
- let benchmarkCandidates = [];
108
- try {
109
- benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, cat);
110
- }
111
- catch (e) {
112
- console.log(`OpenRouter benchmark shortlist unavailable: ${e.message}`);
113
- }
114
- console.log(`judge: ${judge} | strategist: ${strat} | benchmarks: ${benchmarks.join(", ")}`);
115
- console.log(`OpenRouter benchmark candidates: ${benchmarkCandidates.slice(0, 5).join(", ") || "none"}`);
116
- const tKey = taskKey(cfg.task);
117
- const maxPrice = cfg.max_usd_per_m ?? policy.max_usd_per_m;
118
- const tried = new Set();
119
- const results = [];
120
- const failedVendors = new Set();
121
- for (let i = 0; i < rounds; i++) {
122
- const budget = budgetOk(policy);
123
- if (!budget.ok) {
124
- console.log(`budget: ${budget.reason}; stopping`);
125
- break;
126
- }
127
- let model;
128
- let why;
129
- if (i === 0) {
130
- model = cfg.initial_model;
131
- why = "initial model";
132
- }
133
- else {
134
- const pick = await pickNext(key, strat, cfg.task, cfg.eval, benchmarks, results, cat, tried, maxPrice, failedVendors, benchmarkCandidates);
135
- if (!pick)
136
- break;
137
- model = pick.model;
138
- why = pick.why;
139
- }
140
- if (tried.has(model))
141
- break;
142
- console.log(`[round ${i + 1}/${rounds}] ${model} (${why}) ...`);
143
- const observed = i === 0 ? recordedActiveTask(cfg.task, model) : null;
144
- if (i === 0)
145
- console.log(observed ? " using model A's active pi session trace" : " active trace unavailable; running model A in a fresh pi session");
146
- const { point, costUsd, error, sessionId } = observed
147
- ? await scorePiRun(key, judge, cfg.task, cfg.eval, model, cat.get(model), observed, "trace")
148
- : await runPiTrial(key, judge, cfg.task, cfg.eval, model, cat.get(model));
149
- tried.add(model);
150
- results.push(point);
151
- recordTrialSpend(costUsd);
152
- appendJsonl(paths.trials(), { ...point, taskKey: tKey });
153
- if (sessionId)
154
- console.log(` pi trace session: ${sessionId}`);
155
- if (error) {
156
- failedVendors.add(model.split("/")[0]);
157
- console.log(` FAILED: ${error.slice(0, 160)}`);
158
- }
159
- else {
160
- console.log(` score=${point.score.toFixed(2)} $/M=${point.price} est_cost=$${costUsd.toFixed(4)} ${(point.why ?? "").slice(0, 110)}`);
161
- }
162
- }
163
- const pts = results.filter((p) => p.score > 0);
164
- if (pts.length === 0) {
165
- console.log("no trials completed");
166
- process.exit(1);
167
- }
168
- const chain = paretoFrontier(pts);
169
- const best = pickBest(pts);
170
- const fb = pickFallback(pts, best, policy.fallback.min_score);
171
- console.log("\n=== pareto frontier (nondominated: score up, price down, latency down) ===");
172
- for (const p of chain)
173
- console.log(` ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M`);
174
- console.log(`\nBEST FIT : ${best.model} (score ${best.score.toFixed(2)}, $${best.price.toFixed(2)}/M)`);
175
- console.log(fb ? `FALLBACK : ${fb.model} (score ${fb.score.toFixed(2)}, $${fb.price.toFixed(2)}/M)` : "FALLBACK : n/a");
176
- const rec = buildRecommendation(tKey, cfg.task.slice(0, 120), cfg.initial_model, pts, policy);
177
- if (rec) {
178
- appendJsonl(paths.recommendations(), rec);
179
- console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${rec.policy.autoApply})`);
180
- }
45
+ function cmdDoctor(json) {
46
+ printReport(collectDoctorReport(), json);
47
+ }
48
+ function cmdVerify(json) {
49
+ printReport(runOfflineVerify(), json);
181
50
  }
182
51
  function loadFrontiers() {
183
52
  const trials = readJsonl(paths.trials());
@@ -243,12 +112,12 @@ function cmdStatus() {
243
112
  }
244
113
  async function main() {
245
114
  const [cmd, ...args] = process.argv.slice(2);
246
- const policy = loadPolicy(paths.policy());
247
115
  switch (cmd) {
248
116
  case "init":
249
117
  cmdInit();
250
118
  break;
251
119
  case "session-trial": {
120
+ const policy = loadPolicy(paths.policy());
252
121
  const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
253
122
  if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
254
123
  throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
@@ -269,29 +138,32 @@ async function main() {
269
138
  }
270
139
  case "watch":
271
140
  if (args.includes("--once"))
272
- await tickOnce(policy);
141
+ await tickOnce(loadPolicy(paths.policy()));
273
142
  else
274
- await runDaemon(policy);
143
+ await runDaemon(loadPolicy(paths.policy()));
275
144
  break;
276
145
  case "trial": {
277
146
  const file = args[0];
278
147
  if (!file)
279
148
  throw new Error("usage: openmerit trial <task.json> [--rounds N]");
280
149
  const rIdx = args.indexOf("--rounds");
281
- await cmdTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
150
+ await runStandaloneTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
282
151
  break;
283
152
  }
284
153
  case "frontier":
285
154
  cmdFrontier(args[0]);
286
155
  break;
287
156
  case "recommend":
288
- recommendTick(policy);
157
+ recommendTick(loadPolicy(paths.policy()));
289
158
  break;
290
159
  case "status":
291
160
  cmdStatus();
292
161
  break;
293
162
  case "doctor":
294
- cmdDoctor();
163
+ cmdDoctor(args.includes("--json"));
164
+ break;
165
+ case "verify":
166
+ cmdVerify(args.includes("--json"));
295
167
  break;
296
168
  default:
297
169
  console.log(`openmerit — external model-merit harness
@@ -304,7 +176,8 @@ usage: openmerit <command>
304
176
  frontier [taskKey] print pareto frontier(s)
305
177
  recommend emit recommendations now
306
178
  status harness model, queued models, pending recommendations
307
- doctor verify pi, policy, routes, and optional enrichment`);
179
+ doctor [--json] diagnose pi, policy, routes, state, and installation
180
+ verify [--json] run an offline, provider-free core self-test`);
308
181
  if (cmd && cmd !== "help" && cmd !== "--help")
309
182
  process.exitCode = 1;
310
183
  }
package/dist/daemon.js CHANGED
@@ -42,6 +42,28 @@ function resolveRoutePref(cat, prefs, label, requireImages = false) {
42
42
  return fallback;
43
43
  throw new Error(`could not resolve ${label} route from Pi's eligible models`);
44
44
  }
45
+ function processIsAlive(pid) {
46
+ if (!Number.isInteger(pid) || pid <= 0)
47
+ return false;
48
+ try {
49
+ process.kill(pid, 0);
50
+ return true;
51
+ }
52
+ catch (error) {
53
+ return error.code === "EPERM";
54
+ }
55
+ }
56
+ /** Completed jobs stay final; live owners retain their claim; abandoned claims can retry. */
57
+ export function taskJobIsClaimed(prior, marker, now = Date.now()) {
58
+ if (!prior || prior.marker !== marker || prior.status === "failed")
59
+ return false;
60
+ if (prior.status !== "running")
61
+ return true;
62
+ if (prior.ownerPid !== undefined)
63
+ return processIsAlive(prior.ownerPid);
64
+ const updated = Date.parse(prior.updatedAt);
65
+ return Number.isFinite(updated) && now - updated < 15 * 60_000;
66
+ }
45
67
  function emitTrialProgress(progress) {
46
68
  console.log(`[openmerit/progress] ${JSON.stringify(progress)}`);
47
69
  }
@@ -158,12 +180,13 @@ export async function autoTaskTick(key, policy, settledTask) {
158
180
  : `${task.sessionFile}:${task.sessionBytes ?? task.settledAt}`;
159
181
  const processedPath = settledTask === undefined ? paths.watchProcessed() : paths.sessionJob(marker);
160
182
  const prior = readJson(processedPath, null);
161
- if (prior?.marker === marker && prior.status !== "failed")
183
+ if (taskJobIsClaimed(prior, marker))
162
184
  return null;
163
185
  if (!budgetOk(policy).ok)
164
186
  return null;
165
187
  // Claim the settled turn before provider calls so the next poll cannot duplicate it.
166
- writeJson(processedPath, { marker, status: "running", updatedAt: new Date().toISOString() });
188
+ writeJson(processedPath, { marker, status: "running", ownerPid: process.pid,
189
+ updatedAt: new Date().toISOString() });
167
190
  try {
168
191
  let cat = routeCatalog(task.routes);
169
192
  if (!cat.has(routeKey(task.route)))
@@ -258,12 +281,12 @@ export async function autoTaskTick(key, policy, settledTask) {
258
281
  else
259
282
  console.log(`[openmerit] task ${tKey}: active model remains best`);
260
283
  writeJson(processedPath, { marker, status: rec ? "complete" : "no_swap",
261
- updatedAt: new Date().toISOString() });
284
+ ownerPid: process.pid, updatedAt: new Date().toISOString() });
262
285
  return rec;
263
286
  }
264
287
  catch (error) {
265
288
  writeJson(processedPath, { marker, status: "failed",
266
- updatedAt: new Date().toISOString() });
289
+ ownerPid: process.pid, updatedAt: new Date().toISOString() });
267
290
  throw error;
268
291
  }
269
292
  }
@@ -0,0 +1,194 @@
1
+ /** Read-only diagnostics plus a provider-free self-test for first-run support. */
2
+ import { spawnSync } from "node:child_process";
3
+ import { appendFileSync, existsSync, mkdtempSync, readFileSync, rmSync } from "node:fs";
4
+ import { tmpdir } from "node:os";
5
+ import { basename, join } from "node:path";
6
+ import { LocalJsonlEventSink, meritEvent } from "./integrations.js";
7
+ import { piExecutable, availablePiRoutes } from "./pi-trials.js";
8
+ import { DEFAULT_POLICY, loadPolicy, providerAllowed } from "./policy.js";
9
+ import { buildRecommendation } from "./recommend.js";
10
+ import { modelRoute, routeCatalog, routeKey, routeLabel } from "./routes.js";
11
+ import { appendJsonl, paths, readJsonReport, readJsonlReport, writeJson, } from "./store.js";
12
+ function check(id, status, message, details) {
13
+ return { id, status, message, ...(details?.length ? { details } : {}) };
14
+ }
15
+ function packageSources(output) {
16
+ return [...new Set(output.split("\n").filter((line) => /^ {2}\S/.test(line)).map((line) => line.trim())
17
+ .filter((line) => /openmerit/i.test(line) && /^(npm:|git:|https?:|\.?\.?\/|\/)/.test(line)))];
18
+ }
19
+ function safePackageSource(source) {
20
+ if (source.startsWith("npm:/"))
21
+ return `npm:local/${basename(source)}`;
22
+ if (source.startsWith("/"))
23
+ return `local:${basename(source)}`;
24
+ return source;
25
+ }
26
+ /** No credentials, prompts, or trace contents are included in this report. */
27
+ export function collectDoctorReport() {
28
+ const checks = [];
29
+ const executable = piExecutable();
30
+ const versionResult = spawnSync(executable, ["--version"], { encoding: "utf8" });
31
+ const piVersion = versionResult.status === 0 ? versionResult.stdout.trim() : null;
32
+ checks.push(piVersion
33
+ ? check("pi", "pass", `Pi detected (${piVersion})`)
34
+ : check("pi", "fail", "Pi is not executable"));
35
+ try {
36
+ const policy = loadPolicy(paths.policy());
37
+ checks.push(check("policy", "pass", `Policy is valid (v${policy.version}, ${policy.mode})`));
38
+ }
39
+ catch (error) {
40
+ checks.push(check("policy", "fail", `Policy is invalid: ${error.message}`));
41
+ }
42
+ const stateRead = readJsonReport(paths.harnessState(), {});
43
+ checks.push(!stateRead.valid
44
+ ? check("harness-state", "fail", "harness-state.json is malformed; start Pi to replace it atomically")
45
+ : stateRead.exists
46
+ ? check("harness-state", "pass", "Pi extension state snapshot is readable")
47
+ : check("harness-state", "warn", "No Pi extension state yet; start Pi once with OpenMerit installed"));
48
+ let routes = stateRead.value.routes ?? [];
49
+ const liveSnapshot = routes.length > 0;
50
+ if (!routes.length && piVersion) {
51
+ try {
52
+ routes = availablePiRoutes();
53
+ }
54
+ catch { /* the check below explains it */ }
55
+ }
56
+ let policy = DEFAULT_POLICY;
57
+ try {
58
+ policy = loadPolicy(paths.policy());
59
+ }
60
+ catch { /* already reported */ }
61
+ routes = routes.filter((route) => providerAllowed(policy, route.meritId, route.provider));
62
+ const priced = routes.filter((route) => !!route.cost);
63
+ checks.push(routes.length >= 2
64
+ ? check("routes", "pass", `${routes.length} Pi model routes are visible`, routes.slice(0, 8).map((route) => `${routeLabel(route)}${route.cost ? "" : " (price unknown)"}`))
65
+ : check("routes", "fail", `Only ${routes.length} policy-eligible Pi route${routes.length === 1 ? " is" : "s are"} visible; at least two are required`));
66
+ checks.push(liveSnapshot
67
+ ? priced.length >= 2
68
+ ? check("pricing", "pass", `${priced.length} eligible routes expose pricing`)
69
+ : check("pricing", "warn", "Fewer than two eligible routes expose pricing; unknown-price routes cannot auto-apply")
70
+ : check("pricing", "warn", "Route prices are unavailable until the extension reports a live Pi snapshot"));
71
+ const session = stateRead.value.sessionFile;
72
+ checks.push(!session
73
+ ? check("session", "warn", "No active saved Pi session is reported yet")
74
+ : existsSync(session)
75
+ ? check("session", "pass", "The reported Pi session file is readable")
76
+ : check("session", "warn", "The reported Pi session file no longer exists"));
77
+ if (piVersion) {
78
+ const listed = spawnSync(executable, ["list"], { encoding: "utf8" });
79
+ const sources = listed.status === 0 ? packageSources(listed.stdout) : [];
80
+ const safeSources = sources.map(safePackageSource);
81
+ checks.push(sources.length > 1
82
+ ? check("installation", "fail", "OpenMerit is installed from multiple Pi package sources", safeSources)
83
+ : sources.length === 1
84
+ ? check("installation", "pass", `One Pi package source is installed`, safeSources)
85
+ : check("installation", "warn", "OpenMerit is not listed as an installed Pi package; use `pi install npm:openmerit`"));
86
+ }
87
+ const jsonlFiles = [
88
+ ["recommendations.jsonl", paths.recommendations()],
89
+ ["trials.jsonl", paths.trials()],
90
+ ["traces/observations.jsonl", paths.observations()],
91
+ ["events.jsonl", paths.events()],
92
+ ];
93
+ const malformed = jsonlFiles.flatMap(([label, file]) => {
94
+ const invalid = readJsonlReport(file).invalidLines;
95
+ return invalid.length ? [`${label}: lines ${invalid.join(", ")}`] : [];
96
+ });
97
+ checks.push(malformed.length
98
+ ? check("jsonl", "warn", "Malformed JSONL lines were skipped; valid later records remain readable", malformed)
99
+ : check("jsonl", "pass", "Append-only JSONL state is readable"));
100
+ const ledger = readJsonReport(paths.ledger(), { date: "", trials: 0, usd: 0 });
101
+ checks.push(ledger.valid
102
+ ? check("ledger", "pass", "Trial-spend ledger is readable")
103
+ : check("ledger", "fail", "Trial-spend ledger is malformed; spending remains disabled until repaired"));
104
+ let openRouter = false;
105
+ try {
106
+ const env = process.env.OPENROUTER_API_KEY;
107
+ if (env?.trim())
108
+ openRouter = true;
109
+ else if (existsSync(paths.envFile()))
110
+ openRouter = /^\s*OPENROUTER_API_KEY\s*=\s*\S+/m
111
+ .test(readFileSync(paths.envFile(), "utf8"));
112
+ }
113
+ catch { /* optional */ }
114
+ checks.push(check("openrouter-enrichment", "pass", openRouter ? "Optional OpenRouter enrichment is configured" : "Optional OpenRouter enrichment is not configured"));
115
+ const activeRoute = stateRead.value.currentRoute ? routeLabel(stateRead.value.currentRoute) : null;
116
+ return {
117
+ schemaVersion: 1,
118
+ ok: !checks.some((item) => item.status === "fail"),
119
+ checks,
120
+ summary: { piVersion, routeCount: routes.length, pricedRouteCount: priced.length, activeRoute },
121
+ };
122
+ }
123
+ export function renderDiagnosticReport(report) {
124
+ const icon = { pass: "PASS", warn: "WARN", fail: "FAIL" };
125
+ return report.checks.flatMap((item) => [
126
+ `${icon[item.status].padEnd(4)} ${item.id.padEnd(22)} ${item.message}`,
127
+ ...(item.details ?? []).map((detail) => ` - ${detail}`),
128
+ ]).join("\n");
129
+ }
130
+ /** Exercise core state, routing, recommendation, and event paths without provider calls. */
131
+ export function runOfflineVerify() {
132
+ const previousHome = process.env.OPENMERIT_HOME;
133
+ const root = mkdtempSync(join(tmpdir(), "openmerit-verify-"));
134
+ const checks = [];
135
+ try {
136
+ process.env.OPENMERIT_HOME = root;
137
+ const active = modelRoute("native", "model-a", {
138
+ input: ["text"], cost: { input: 1, output: 2 }, contextWindow: 8192, maxTokens: 4096,
139
+ });
140
+ const candidate = modelRoute("local", "model-b", {
141
+ input: ["text"], cost: { input: 0.2, output: 0.4 }, contextWindow: 8192, maxTokens: 4096,
142
+ });
143
+ writeJson(paths.harnessState(), { schemaVersion: 1, currentRoute: active, routes: [active, candidate] });
144
+ const state = readJsonReport(paths.harnessState(), {});
145
+ checks.push(state.valid && state.value.routes?.length === 2
146
+ ? check("atomic-json", "pass", "Atomic JSON state round-trip passed")
147
+ : check("atomic-json", "fail", "Atomic JSON state round-trip failed"));
148
+ const entries = routeCatalog([active, candidate]);
149
+ const point = (route, score) => {
150
+ const entry = entries.get(routeKey(route));
151
+ return { schemaVersion: 1, model: route.meritId, route, score, price: entry.price,
152
+ priceKnown: entry.priceKnown, ts: new Date().toISOString(), source: "pi_trial" };
153
+ };
154
+ const recommendation = buildRecommendation("verify-task", "offline verification", active.meritId, [point(active, 0.6), point(candidate, 0.9)], DEFAULT_POLICY, "/verify/session.jsonl", active);
155
+ checks.push(recommendation?.recommended.route?.provider === "local" &&
156
+ recommendation.policy.reasons.length > 0
157
+ ? check("provider-neutral", "pass", "Exact provider routes and policy reasons survive recommendation")
158
+ : check("provider-neutral", "fail", "Provider route or policy evidence was lost"));
159
+ if (recommendation)
160
+ appendJsonl(paths.recommendations(), recommendation);
161
+ appendJsonl(paths.recommendations(), { schemaVersion: 0, id: "legacy", status: "dismissed" });
162
+ appendFileSync(paths.recommendations(), "{interrupted");
163
+ const history = readJsonlReport(paths.recommendations());
164
+ checks.push(history.records.length === 2 && history.invalidLines.length === 1
165
+ ? check("jsonl-recovery", "pass", "Valid records survive legacy and interrupted JSONL lines")
166
+ : check("jsonl-recovery", "fail", "JSONL recovery did not preserve valid history"));
167
+ const sink = new LocalJsonlEventSink();
168
+ if (recommendation)
169
+ sink.emit(meritEvent("recommendation.created", recommendation));
170
+ const events = readJsonlReport(paths.events());
171
+ checks.push(events.records.length === 1
172
+ ? check("events", "pass", "Versioned neutral event sink round-trip passed")
173
+ : check("events", "fail", "Neutral event sink round-trip failed"));
174
+ checks.push(loadPolicy(paths.policy()).mode === "recommend"
175
+ ? check("compatibility", "pass", "Missing 0.1.x policy fields normalize to safe defaults")
176
+ : check("compatibility", "fail", "Policy defaults are unsafe"));
177
+ }
178
+ catch (error) {
179
+ checks.push(check("verify", "fail", error.message));
180
+ }
181
+ finally {
182
+ if (previousHome === undefined)
183
+ delete process.env.OPENMERIT_HOME;
184
+ else
185
+ process.env.OPENMERIT_HOME = previousHome;
186
+ rmSync(root, { recursive: true, force: true });
187
+ }
188
+ return {
189
+ schemaVersion: 1,
190
+ ok: !checks.some((item) => item.status === "fail"),
191
+ checks,
192
+ summary: { piVersion: null, routeCount: 2, pricedRouteCount: 2, activeRoute: "native/model-a via native" },
193
+ };
194
+ }
package/dist/pi-trials.js CHANGED
@@ -113,7 +113,9 @@ export function recordedActiveTask(task, model, images = [], sessionFile, sessio
113
113
  if (!user || !userImages || taskInputKey(answerText(user.content), userImages, sessionFiles(answerText(user.content))) !==
114
114
  taskInputKey(task, images, files))
115
115
  return null;
116
- const matching = assistants.filter((m) => meritModelId(m.provider ?? "openrouter", m.model ?? "") === model);
116
+ const matching = assistants.filter((m) => typeof model === "string"
117
+ ? meritModelId(m.provider ?? "openrouter", m.model ?? "") === model
118
+ : m.provider === model.provider && m.model === model.modelId);
117
119
  if (matching.length === 0 || matching.some((m) => m.stopReason === "error"))
118
120
  return null;
119
121
  const answer = matching.map((m) => answerText(m.content)).filter(Boolean).at(-1) ?? "";
@@ -170,10 +172,10 @@ export function settledActiveTask(snapshot) {
170
172
  const files = sessionFiles(task);
171
173
  if (!task.trim() || taskInputKey(task, images, files) !== st.settledTaskKey)
172
174
  return null;
173
- const run = recordedActiveTask(task, st.currentModel, images, st.sessionFile, st.sessionBytes);
175
+ const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
176
+ const run = recordedActiveTask(task, route, images, st.sessionFile, st.sessionBytes);
174
177
  if (!run)
175
178
  return null;
176
- const route = st.currentRoute ?? legacyOpenRouterRoute(st.currentModel);
177
179
  const routes = st.routes?.length ? st.routes : [route];
178
180
  return { task, images, model: st.currentModel, run, sessionFile: st.sessionFile,
179
181
  settledAt: st.settledAt, sessionBytes: st.sessionBytes, cwd: st.cwd ?? process.cwd(), usedTools, files,
@@ -182,6 +184,13 @@ export function settledActiveTask(snapshot) {
182
184
  export function piExecutable() {
183
185
  return process.env.OPENMERIT_PI_BIN?.trim() || "pi";
184
186
  }
187
+ function compactNumber(value) {
188
+ const match = value.match(/^(\d+(?:\.\d+)?)([KMG])?$/i);
189
+ if (!match)
190
+ return undefined;
191
+ const scale = { K: 1_000, M: 1_000_000, G: 1_000_000_000 }[(match[2] ?? "").toUpperCase()] ?? 1;
192
+ return Math.round(Number(match[1]) * scale);
193
+ }
185
194
  /** Read provider/model routes from pi when no live extension snapshot is available. */
186
195
  export function availablePiRoutes(snapshot) {
187
196
  if (snapshot?.length)
@@ -191,10 +200,14 @@ export function availablePiRoutes(snapshot) {
191
200
  throw new Error("could not list models from pi");
192
201
  const routes = [];
193
202
  for (const line of result.stdout.split("\n").slice(1)) {
194
- const [provider, id, , , , images] = line.trim().split(/\s+/);
203
+ const [provider, id, context, maxOut, , images] = line.trim().split(/\s+/);
195
204
  if (!provider || !id || id.startsWith("~"))
196
205
  continue;
197
- routes.push(modelRoute(provider, id, { input: images === "yes" ? ["text", "image"] : ["text"] }));
206
+ routes.push(modelRoute(provider, id, {
207
+ input: images === "yes" ? ["text", "image"] : ["text"],
208
+ contextWindow: compactNumber(context),
209
+ maxTokens: compactNumber(maxOut),
210
+ }));
198
211
  }
199
212
  return routes;
200
213
  }
@@ -0,0 +1,220 @@
1
+ /** Provider-neutral standalone task trials executed through Pi routes. */
2
+ import { readFileSync } from "node:fs";
3
+ import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
4
+ import { fetchCatalog, saveSnapshot } from "./catalog.js";
5
+ import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
6
+ import { JUDGE_PREFS } from "./judge.js";
7
+ import { loadKey, PiCliChatClient } from "./llm.js";
8
+ import { availablePiRoutes, recordedActiveTask, runPiTrial, scorePiRun } from "./pi-trials.js";
9
+ import { loadPolicy, providerAllowed } from "./policy.js";
10
+ import { buildRecommendation } from "./recommend.js";
11
+ import { enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
12
+ import { appendJsonl, paths, readJson, taskKey } from "./store.js";
13
+ import { pickNext, STRAT_PREFS } from "./strategist.js";
14
+ import { budgetOk, recordMeritSpend, recordTrialSpend, trialBudgetOk } from "./trials.js";
15
+ function optionalOpenRouterKey() {
16
+ try {
17
+ return loadKey();
18
+ }
19
+ catch {
20
+ return null;
21
+ }
22
+ }
23
+ function configuredRoutes() {
24
+ const state = readJson(paths.harnessState(), {});
25
+ let routes = state.routes?.length ? state.routes : availablePiRoutes();
26
+ if (state.currentRoute && !routes.some((route) => routeKey(route) === routeKey(state.currentRoute)))
27
+ routes = [state.currentRoute, ...routes];
28
+ const unique = new Map(routes.map((route) => [routeKey(route), route]));
29
+ return { routes: [...unique.values()], currentRoute: state.currentRoute ?? null };
30
+ }
31
+ function selectorKey(selector) {
32
+ if (!selector)
33
+ return null;
34
+ if (typeof selector === "string")
35
+ return selector;
36
+ const modelId = selector.modelId ?? selector.model_id;
37
+ return selector.provider && modelId ? `${selector.provider}:${modelId}` : null;
38
+ }
39
+ function matchingEntries(cat, selector) {
40
+ const exact = cat.get(selector);
41
+ if (exact)
42
+ return [exact];
43
+ return [...cat.values()].filter((entry) => entry.id === selector || entry.route?.meritId === selector ||
44
+ (entry.route && `${entry.route.provider}/${entry.route.modelId}` === selector));
45
+ }
46
+ function resolveInitial(cat, cfg, currentRoute) {
47
+ const exactKey = selectorKey(cfg.initial_route);
48
+ if (cfg.initial_route && !exactKey)
49
+ throw new Error("initial_route must include provider and modelId");
50
+ if (exactKey) {
51
+ const exact = cat.get(exactKey);
52
+ if (!exact)
53
+ throw new Error(`initial route ${exactKey} is not eligible in Pi`);
54
+ if (cfg.initial_model !== exact.id && cfg.initial_model !== exactKey)
55
+ throw new Error(`initial_route ${exactKey} resolves to ${exact.id}, not initial_model ${cfg.initial_model}`);
56
+ return exact;
57
+ }
58
+ const matches = matchingEntries(cat, cfg.initial_model);
59
+ if (matches.length === 0)
60
+ throw new Error(`initial_model ${cfg.initial_model} is not eligible in Pi`);
61
+ if (matches.length === 1)
62
+ return matches[0];
63
+ if (currentRoute) {
64
+ const current = matches.find((entry) => entry.route && routeKey(entry.route) === routeKey(currentRoute));
65
+ if (current)
66
+ return current;
67
+ }
68
+ throw new Error(`initial_model ${cfg.initial_model} has multiple Pi routes; set initial_route to one of: ` +
69
+ matches.map((entry) => routeKey(entry.route)).join(", "));
70
+ }
71
+ function resolveMeritRoute(cat, prefs, label) {
72
+ for (const pref of prefs) {
73
+ if (!pref)
74
+ continue;
75
+ const matches = matchingEntries(cat, pref);
76
+ if (matches.length)
77
+ return matches.sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
78
+ }
79
+ for (const pref of prefs) {
80
+ if (!pref)
81
+ continue;
82
+ const fuzzy = [...cat.values()].find((entry) => entry.id.includes(pref));
83
+ if (fuzzy)
84
+ return fuzzy;
85
+ }
86
+ const fallback = [...cat.values()].sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
87
+ if (fallback)
88
+ return fallback;
89
+ throw new Error(`could not resolve ${label} route from Pi's eligible models`);
90
+ }
91
+ async function buildCatalog(routes, key) {
92
+ let cat = routeCatalog(routes);
93
+ if (!key)
94
+ return cat;
95
+ try {
96
+ const live = await fetchCatalog(key);
97
+ saveSnapshot(live);
98
+ cat = new Map([...cat].map(([id, entry]) => [id,
99
+ entry.route?.provider === "openrouter" ? enrichRouteEntry(entry, live.get(entry.id)) : entry]));
100
+ }
101
+ catch (error) {
102
+ console.log(`OpenRouter enrichment unavailable: ${error.message}`);
103
+ }
104
+ return cat;
105
+ }
106
+ export async function runStandaloneTrial(taskFile, rounds) {
107
+ const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
108
+ if (!cfg.task?.trim() || !cfg.eval?.trim() || !cfg.initial_model?.trim())
109
+ throw new Error("task JSON requires task, eval, and initial_model");
110
+ if (!Number.isInteger(rounds) || rounds < 1)
111
+ throw new Error("--rounds must be a positive integer");
112
+ const policy = loadPolicy(paths.policy());
113
+ const key = optionalOpenRouterKey();
114
+ const { routes, currentRoute } = configuredRoutes();
115
+ let cat = await buildCatalog(routes, key);
116
+ for (const [id, entry] of [...cat]) {
117
+ if (!entry.route || !providerAllowed(policy, entry.id, entry.route.provider) ||
118
+ (entry.priceKnown !== false && entry.price > (cfg.max_usd_per_m ?? policy.max_usd_per_m)))
119
+ cat.delete(id);
120
+ }
121
+ if (cat.size === 0)
122
+ throw new Error("Pi exposes no model routes allowed by the current policy");
123
+ const initial = resolveInitial(cat, cfg, currentRoute);
124
+ const judge = resolveMeritRoute(cat, [cfg.judge_model, policy.judge_model, ...JUDGE_PREFS, initial.id], "judge");
125
+ const strategist = resolveMeritRoute(cat, [cfg.strategist_model, policy.strategist_model, ...STRAT_PREFS, initial.id], "strategist");
126
+ const client = new PiCliChatClient(undefined, recordMeritSpend);
127
+ const { category, benchmarks } = relevantBenchmarks(cfg.task);
128
+ let benchmarkCandidates = [];
129
+ if (key) {
130
+ try {
131
+ benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, new Map([...cat.values()].map((entry) => [entry.id, entry])));
132
+ }
133
+ catch (error) {
134
+ console.log(`OpenRouter benchmark shortlist unavailable: ${error.message}`);
135
+ }
136
+ }
137
+ console.log(`routes: ${cat.size} eligible in Pi${key ? " (OpenRouter enrichment enabled)" : ""}`);
138
+ console.log(`initial: ${routeLabel(initial.route)} | judge: ${routeLabel(judge.route)} | ` +
139
+ `strategist: ${routeLabel(strategist.route)} | benchmarks: ${benchmarks.join(", ")}`);
140
+ const tKey = taskKey(cfg.task);
141
+ const tried = new Set();
142
+ const results = [];
143
+ const failedProviders = new Set();
144
+ for (let i = 0; i < rounds; i++) {
145
+ const daily = budgetOk(policy);
146
+ if (!daily.ok) {
147
+ console.log(`budget: ${daily.reason}; stopping`);
148
+ break;
149
+ }
150
+ let entry;
151
+ let why;
152
+ if (i === 0) {
153
+ entry = initial;
154
+ why = "initial route";
155
+ }
156
+ else {
157
+ const pick = await pickNext(key ?? "", strategist.route, cfg.task, cfg.eval, benchmarks, results, cat, tried, cfg.max_usd_per_m ?? policy.max_usd_per_m, failedProviders, benchmarkCandidates, client);
158
+ if (!pick)
159
+ break;
160
+ entry = pick.route ? cat.get(routeKey(pick.route)) : cat.get(pick.model);
161
+ why = pick.why;
162
+ }
163
+ const route = entry.route;
164
+ const keyForRoute = routeKey(route);
165
+ if (tried.has(keyForRoute))
166
+ break;
167
+ if (i > 0) {
168
+ const admission = trialBudgetOk(policy, entry, cfg.task.length);
169
+ if (!admission.ok) {
170
+ tried.add(keyForRoute);
171
+ console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} skipped: ${admission.reason}`);
172
+ continue;
173
+ }
174
+ }
175
+ console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} (${why}) ...`);
176
+ const observed = i === 0 ? recordedActiveTask(cfg.task, route) : null;
177
+ if (i === 0)
178
+ console.log(observed
179
+ ? " using the exact provider route from Pi's active session trace"
180
+ : " active trace unavailable; running the initial route in a fresh Pi session");
181
+ const outcome = observed
182
+ ? await scorePiRun(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, observed, "trace", [], client)
183
+ : await runPiTrial(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, cfg.cwd ?? process.cwd(), [], [], undefined, client);
184
+ tried.add(keyForRoute);
185
+ results.push(outcome.point);
186
+ // Reusing the active Pi answer is observation, not a new candidate call.
187
+ if (!observed)
188
+ recordTrialSpend(outcome.costUsd);
189
+ appendJsonl(paths.trials(), { ...outcome.point, taskKey: tKey });
190
+ if (outcome.sessionId)
191
+ console.log(` pi trace session: ${outcome.sessionId}`);
192
+ if (outcome.error) {
193
+ failedProviders.add(route.provider);
194
+ console.log(` FAILED: ${outcome.error.slice(0, 160)}`);
195
+ }
196
+ else {
197
+ const price = outcome.point.priceKnown === false ? "price unknown" : `$${outcome.point.price}/M`;
198
+ console.log(` score=${outcome.point.score.toFixed(2)} ${price} ` +
199
+ `run=$${outcome.costUsd.toFixed(4)} ${(outcome.point.why ?? "").slice(0, 100)}`);
200
+ }
201
+ }
202
+ const scored = results.filter((point) => point.score > 0);
203
+ if (scored.length === 0)
204
+ throw new Error("no trials completed");
205
+ const frontier = paretoFrontier(scored);
206
+ const best = pickBest(scored);
207
+ const fallback = pickFallback(scored, best, policy.fallback.min_score);
208
+ console.log("\n=== pareto frontier (quality up, price and latency down) ===");
209
+ for (const point of frontier)
210
+ console.log(` ${(point.route ? routeLabel(point.route) : point.model).padEnd(52)} ` +
211
+ `score=${point.score.toFixed(2)} ${point.priceKnown === false ? "price unknown" : `$${point.price.toFixed(2)}/M`}`);
212
+ console.log(`\nBEST FIT : ${best.route ? routeLabel(best.route) : best.model}`);
213
+ console.log(fallback ? `FALLBACK : ${fallback.route ? routeLabel(fallback.route) : fallback.model}` : "FALLBACK : n/a");
214
+ const recommendation = buildRecommendation(tKey, cfg.task.slice(0, 120), initial.id, results, policy, null, initial.route);
215
+ if (recommendation) {
216
+ appendJsonl(paths.recommendations(), recommendation);
217
+ console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${recommendation.policy.autoApply})`);
218
+ }
219
+ return { points: results, recommendation };
220
+ }
package/dist/store.js CHANGED
@@ -1,6 +1,6 @@
1
1
  /** State directory layout + JSONL persistence. All state lives under OPENMERIT_HOME (default ~/.openmerit). */
2
2
  import { createHash } from "node:crypto";
3
- import { appendFileSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
3
+ import { appendFileSync, existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync, } from "node:fs";
4
4
  import { homedir } from "node:os";
5
5
  import { join } from "node:path";
6
6
  export function stateDir() {
@@ -38,20 +38,52 @@ export function appendJsonl(file, obj) {
38
38
  mkdirSync(join(file, ".."), { recursive: true });
39
39
  appendFileSync(file, JSON.stringify(obj) + "\n");
40
40
  }
41
+ /**
42
+ * Read append-only state without letting one interrupted or malformed line
43
+ * hide the remaining history. Diagnostics can surface invalid line numbers.
44
+ */
45
+ export function readJsonlReport(file) {
46
+ if (!existsSync(file))
47
+ return { records: [], invalidLines: [] };
48
+ const records = [];
49
+ const invalidLines = [];
50
+ for (const [index, line] of readFileSync(file, "utf8").split("\n").entries()) {
51
+ if (!line.trim())
52
+ continue;
53
+ try {
54
+ records.push(JSON.parse(line));
55
+ }
56
+ catch {
57
+ invalidLines.push(index + 1);
58
+ }
59
+ }
60
+ return { records, invalidLines };
61
+ }
41
62
  export function readJsonl(file) {
63
+ return readJsonlReport(file).records;
64
+ }
65
+ export function readJsonReport(file, fallback) {
42
66
  if (!existsSync(file))
43
- return [];
44
- return readFileSync(file, "utf8")
45
- .split("\n")
46
- .filter((l) => l.trim())
47
- .map((l) => JSON.parse(l));
67
+ return { value: fallback, exists: false, valid: true };
68
+ try {
69
+ return { value: JSON.parse(readFileSync(file, "utf8")), exists: true, valid: true };
70
+ }
71
+ catch (error) {
72
+ return { value: fallback, exists: true, valid: false, error: error.message };
73
+ }
48
74
  }
49
75
  export function readJson(file, fallback) {
50
- if (!existsSync(file))
51
- return fallback;
52
- return JSON.parse(readFileSync(file, "utf8"));
76
+ return readJsonReport(file, fallback).value;
53
77
  }
54
78
  export function writeJson(file, obj) {
55
79
  mkdirSync(join(file, ".."), { recursive: true });
56
- writeFileSync(file, JSON.stringify(obj, null, 2) + "\n");
80
+ const temp = `${file}.tmp-${process.pid}-${Date.now()}`;
81
+ try {
82
+ writeFileSync(temp, JSON.stringify(obj, null, 2) + "\n");
83
+ renameSync(temp, file);
84
+ }
85
+ finally {
86
+ if (existsSync(temp))
87
+ rmSync(temp, { force: true });
88
+ }
57
89
  }
@@ -21,7 +21,7 @@ ROUTES (route id | logical model | blended $/1M tokens | ctx):
21
21
  /** Pick the next model to trial, or null when the catalog is exhausted. */
22
22
  export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
23
23
  const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
24
- const ok = (c) => (c.priceKnown === false || c.price <= maxPrice) &&
24
+ const ok = (c) => c.priceKnown !== false && c.price <= maxPrice &&
25
25
  !tried.has(entryKey(c)) &&
26
26
  c.ctx >= 4096 &&
27
27
  !failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
package/dist/trials.js CHANGED
@@ -1,12 +1,15 @@
1
1
  /** Shadow trials: run candidate models against observed/declared tasks on the background track. */
2
2
  import { directChatClient } from "./llm.js";
3
3
  import { judge, parseObj } from "./judge.js";
4
- import { paths, readJson, writeJson } from "./store.js";
4
+ import { paths, readJsonReport, writeJson } from "./store.js";
5
5
  function today() {
6
6
  return new Date().toISOString().slice(0, 10);
7
7
  }
8
8
  export function budgetOk(policy) {
9
- const ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
9
+ const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
10
+ if (!report.valid)
11
+ return { ok: false, reason: "trial ledger is malformed; run `openmerit doctor`" };
12
+ const ledger = report.value;
10
13
  if (ledger.date !== today())
11
14
  return { ok: true }; // new day resets
12
15
  if (ledger.trials >= policy.budgets.max_trials_per_day)
@@ -31,7 +34,10 @@ export function trialBudgetOk(policy, entry, inputChars) {
31
34
  return { ok: true, estimatedUsd };
32
35
  }
33
36
  export function recordTrialSpend(usd) {
34
- let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
37
+ const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
38
+ if (!report.valid)
39
+ throw new Error("trial ledger is malformed; refusing to reset spend");
40
+ let ledger = report.value;
35
41
  if (ledger.date !== today())
36
42
  ledger = { date: today(), trials: 0, usd: 0 };
37
43
  ledger.trials += 1;
@@ -42,7 +48,10 @@ export function recordTrialSpend(usd) {
42
48
  export function recordMeritSpend(usd) {
43
49
  if (!(usd > 0))
44
50
  return;
45
- let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
51
+ const report = readJsonReport(paths.ledger(), { date: today(), trials: 0, usd: 0 });
52
+ if (!report.valid)
53
+ throw new Error("trial ledger is malformed; refusing to reset spend");
54
+ let ledger = report.value;
46
55
  if (ledger.date !== today())
47
56
  ledger = { date: today(), trials: 0, usd: 0 };
48
57
  ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
@@ -2,5 +2,6 @@
2
2
  "task": "Write a Python function `def word_ladder(begin, end, words)` that returns the SHORTEST transformation sequence from `begin` to `end`, where consecutive words differ by exactly one character and every intermediate word must be in `words`. Return [] if no path exists. If begin == end return [begin]. All words are the same length, lowercase a-z. Output only runnable code, no explanation.",
3
3
  "eval": "Score 1.0 requires ALL of: (a) algorithm guaranteed to find a shortest path (BFS or equivalent, NOT DFS/greedy); (b) returned path starts with begin and ends with end with all intermediates in words; (c) returns [] when impossible; (d) begin==end returns [begin]; (e) begin need not be in words; (f) syntactically valid, runnable Python, no external imports, function named word_ladder; (g) output contains only code. Deduct ~0.2 per missing criterion; non-shortest-path algorithms score at most 0.5.",
4
4
  "initial_model": "openai/gpt-4o-mini",
5
+ "initial_route": "openai:gpt-4o-mini",
5
6
  "max_usd_per_m": 20
6
7
  }
@@ -17,7 +17,9 @@
17
17
  */
18
18
 
19
19
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
20
- import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
20
+ import {
21
+ appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync,
22
+ } from "node:fs";
21
23
  import { homedir } from "node:os";
22
24
  import { basename, join } from "node:path";
23
25
  import { createHash } from "node:crypto";
@@ -84,7 +86,12 @@ function trialBudget(): { summary: string; reason: string | null } {
84
86
  const today = new Date().toISOString().slice(0, 10);
85
87
  let ledger: { date?: string; trials?: number; usd?: number } = {};
86
88
  try { ledger = JSON.parse(readFileSync(LEDGER_FILE, "utf8")) as typeof ledger; }
87
- catch { /* no trial spend yet */ }
89
+ catch {
90
+ if (existsSync(LEDGER_FILE)) return {
91
+ summary: "ledger unreadable",
92
+ reason: "trial ledger is malformed; run `openmerit doctor` before spending",
93
+ };
94
+ }
88
95
  const trials = ledger.date === today ? ledger.trials ?? 0 : 0;
89
96
  const usd = ledger.date === today ? ledger.usd ?? 0 : 0;
90
97
  const reason = trials >= limit ? `daily trial cap reached (${limit})`
@@ -107,10 +114,12 @@ function latestSessionJob(sessionFile: string | undefined): string | null {
107
114
  }
108
115
 
109
116
  function loadPolicy(): ExtensionPolicy {
117
+ if (!existsSync(POLICY_FILE)) return {};
110
118
  try {
111
119
  return JSON.parse(readFileSync(POLICY_FILE, "utf8")) as ExtensionPolicy;
112
120
  } catch {
113
- return {};
121
+ return { mode: "recommend", fallback: { apply_on_error: false },
122
+ budgets: { max_trials_per_day: 0, max_usd_per_day: 0 } };
114
123
  }
115
124
  }
116
125
 
@@ -173,7 +182,17 @@ function appendJsonl(file: string, obj: unknown): void {
173
182
  function latestRecommendations(): Recommendation[] {
174
183
  const byId = new Map<string, Recommendation>();
175
184
  for (const r of readJsonl<Recommendation>(RECS_FILE)) byId.set(r.id, r);
176
- return [...byId.values()];
185
+ let malformed = false;
186
+ if (existsSync(RECS_FILE)) {
187
+ for (const line of readFileSync(RECS_FILE, "utf8").split("\n")) {
188
+ if (!line.trim()) continue;
189
+ try { JSON.parse(line); } catch { malformed = true; break; }
190
+ }
191
+ }
192
+ return [...byId.values()].map((rec) => malformed && rec.status === "pending" && rec.policy.autoApply
193
+ ? { ...rec, policy: { autoApply: false,
194
+ reasons: [...rec.policy.reasons, "recommendation history contains a malformed line; manual review required"] } }
195
+ : rec);
177
196
  }
178
197
 
179
198
  function taskKey(text: string): string {
@@ -305,6 +324,16 @@ function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
305
324
  return [...byRoute.values()];
306
325
  }
307
326
 
327
+ function writeState(value: unknown): void {
328
+ const temp = `${STATE_FILE}.tmp-${process.pid}-${Date.now()}`;
329
+ try {
330
+ writeFileSync(temp, JSON.stringify(value) + "\n");
331
+ renameSync(temp, STATE_FILE);
332
+ } finally {
333
+ if (existsSync(temp)) rmSync(temp, { force: true });
334
+ }
335
+ }
336
+
308
337
  function reportHarnessState(ctx: ExtensionContext, settled = false): void {
309
338
  try {
310
339
  mkdirSync(HOME, { recursive: true });
@@ -317,9 +346,7 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
317
346
  const sessionFile = ctx.sessionManager.getSessionFile() ?? null;
318
347
  const settledAt = settled ? new Date().toISOString()
319
348
  : prior.sessionFile === sessionFile ? prior.settledAt ?? null : null;
320
- writeFileSync(
321
- STATE_FILE,
322
- JSON.stringify({
349
+ writeState({
323
350
  schemaVersion: 1,
324
351
  currentModel: model,
325
352
  currentRoute,
@@ -331,8 +358,7 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
331
358
  settledAt,
332
359
  cwd: ctx.cwd,
333
360
  updatedAt: new Date().toISOString(),
334
- }) + "\n",
335
- );
361
+ });
336
362
  } catch {
337
363
  /* never break the host session over reporting */
338
364
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmerit",
3
- "version": "0.1.2",
3
+ "version": "0.1.3",
4
4
  "description": "Find better models for each pi task by comparing quality, cost, and latency.",
5
5
  "type": "module",
6
6
  "keywords": [