openmerit 0.1.2 → 0.1.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -18
- package/dist/cli.js +26 -153
- package/dist/daemon.js +56 -14
- package/dist/diagnostics.js +227 -0
- package/dist/llm.js +6 -2
- package/dist/pi-config.js +46 -0
- package/dist/pi-trials.js +30 -10
- package/dist/policy.js +71 -1
- package/dist/routes.js +15 -0
- package/dist/standalone.js +224 -0
- package/dist/store.js +42 -10
- package/dist/strategist.js +1 -1
- package/dist/trials.js +13 -4
- package/examples/task.example.json +1 -0
- package/extension/openmerit.ts +182 -29
- package/instructions/OPENMERIT.md +9 -0
- package/instructions/openmerit.policy.json +6 -2
- package/package.json +1 -1
- package/rules.md +4 -0
package/README.md
CHANGED
|
@@ -54,13 +54,14 @@ package into pi, then initialize its policy and state:
|
|
|
54
54
|
```bash
|
|
55
55
|
pi install npm:openmerit
|
|
56
56
|
npx --yes openmerit init
|
|
57
|
+
npx --yes openmerit verify
|
|
57
58
|
pi list
|
|
58
59
|
```
|
|
59
60
|
|
|
60
61
|
`pi install` makes the extension and its trial engine available to pi. The
|
|
61
62
|
one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
|
|
62
63
|
globally with `npm install --global openmerit` only if you also want persistent
|
|
63
|
-
shell access to `openmerit status`, `doctor`, `frontier`, or the optional watcher. Install
|
|
64
|
+
shell access to `openmerit status`, `doctor`, `verify`, `frontier`, or the optional watcher. Install
|
|
64
65
|
the extension from only one source—remove any older copied `openmerit.ts` first
|
|
65
66
|
so pi does not load it twice.
|
|
66
67
|
|
|
@@ -120,8 +121,9 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
120
121
|
3. Verify the extension appears in `pi list`. The instruction file
|
|
121
122
|
[`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
|
|
122
123
|
pi project's AGENTS.md for agent context, but the extension does
|
|
123
|
-
not require it. Run `npx openmerit
|
|
124
|
-
|
|
124
|
+
not require it. Run `npx openmerit verify` for a provider-free core self-test,
|
|
125
|
+
then `npx openmerit doctor` after starting Pi once to inspect the eligible
|
|
126
|
+
route snapshot and configuration without printing secrets.
|
|
125
127
|
|
|
126
128
|
4. Start pi in one terminal with any configured model. For example:
|
|
127
129
|
|
|
@@ -140,6 +142,12 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
140
142
|
switch. Then send a second message to see which model actually handles it.
|
|
141
143
|
You do not need a watcher terminal or the standalone `trial` command.
|
|
142
144
|
|
|
145
|
+
Use `/openmerit pause` to stop the active comparison and suppress automatic
|
|
146
|
+
comparisons, `/openmerit resume` to enable them for future completed tasks,
|
|
147
|
+
and `/openmerit compare` to explicitly compare the latest completed task
|
|
148
|
+
even while automatic comparisons are paused. `/openmerit doctor` runs the
|
|
149
|
+
sanitized setup checks without leaving Pi.
|
|
150
|
+
|
|
143
151
|
Candidate comparisons have no tools by default, even when the observed task
|
|
144
152
|
used tools. See **Alpha boundaries** before explicitly enabling candidate
|
|
145
153
|
tools.
|
|
@@ -148,6 +156,57 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
148
156
|
`"mode": "auto"` and `"auto_apply.enabled": true`, review the score-gain
|
|
149
157
|
and price-ratio thresholds, and run the next task.
|
|
150
158
|
|
|
159
|
+
### Configure custom or local routes
|
|
160
|
+
|
|
161
|
+
Pi normally supplies each route's price, context window, output limit, and
|
|
162
|
+
modalities. Some custom and local providers omit that metadata. OpenMerit does
|
|
163
|
+
not guess that a local model is free: add an override keyed by the exact
|
|
164
|
+
`provider:modelId` route in `~/.openmerit/policy.json` instead:
|
|
165
|
+
|
|
166
|
+
```json
|
|
167
|
+
{
|
|
168
|
+
"route_overrides": {
|
|
169
|
+
"ollama:qwen3:8b": {
|
|
170
|
+
"cost": { "input": 0, "output": 0 },
|
|
171
|
+
"context_window": 32768,
|
|
172
|
+
"max_tokens": 4096,
|
|
173
|
+
"input": ["text"]
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
All fields are optional, but both `cost.input` and `cost.output` are required
|
|
180
|
+
when declaring cost. Prices use Pi's dollars-per-million-token units. The
|
|
181
|
+
override applies only to that exact route; it does not change the stable
|
|
182
|
+
`vendor/model` identity or another provider's route to the same model.
|
|
183
|
+
|
|
184
|
+
If the provider itself is registered at runtime by a Pi extension, explicitly
|
|
185
|
+
allow that provider-registration file in the same policy. Paths must be
|
|
186
|
+
absolute, existing files:
|
|
187
|
+
|
|
188
|
+
```json
|
|
189
|
+
{
|
|
190
|
+
"pi": {
|
|
191
|
+
"provider_extensions": [
|
|
192
|
+
"/absolute/path/to/ollama-provider.ts"
|
|
193
|
+
]
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
OpenMerit still launches subprocesses with extension discovery disabled, then
|
|
199
|
+
loads only these explicit files. Do not add OpenMerit's own extension. An
|
|
200
|
+
allowlisted extension executes code in every candidate, judge, and strategist
|
|
201
|
+
Pi subprocess, so list only provider extensions you trust. Run
|
|
202
|
+
`npx openmerit doctor` or `/openmerit doctor` to validate the files and confirm
|
|
203
|
+
that route overrides match Pi's visible routes.
|
|
204
|
+
|
|
205
|
+
Existing `0.1.x` policy files remain valid: missing `route_overrides` and
|
|
206
|
+
`pi.provider_extensions` fields normalize to empty safe defaults. `openmerit init`
|
|
207
|
+
continues to preserve an existing policy, so add these fields manually
|
|
208
|
+
only when you need them.
|
|
209
|
+
|
|
151
210
|
### Try an invoice-to-JSON task
|
|
152
211
|
|
|
153
212
|
Use the same one-terminal setup. Start pi with a vision-capable model, for
|
|
@@ -179,14 +238,21 @@ schema, so its published scores are not directly comparable to these pi runs.
|
|
|
179
238
|
|
|
180
239
|
### Inspect or troubleshoot a run
|
|
181
240
|
|
|
182
|
-
`npx openmerit doctor` checks Pi, policy, eligible routes
|
|
183
|
-
|
|
241
|
+
`npx openmerit doctor` checks Pi, policy, eligible routes and prices, saved
|
|
242
|
+
session state, duplicate package sources, append-only files, and optional
|
|
243
|
+
OpenRouter enrichment. Add `--json` for a sanitized diagnostic report suitable
|
|
244
|
+
for a bug report; it contains no credentials, prompts, or trace contents.
|
|
245
|
+
`npx openmerit verify` runs an offline self-test of atomic state, JSONL recovery,
|
|
246
|
+
route preservation, policy evidence, and neutral events without contacting a
|
|
247
|
+
provider. `npx openmerit status` shows the latest pi model and pending recommendations;
|
|
184
248
|
`npx openmerit frontier` shows measured quality, blended price, latency, and
|
|
185
249
|
the chosen frontier per task. `/openmerit` inside pi shows the current model,
|
|
186
250
|
fallback, the model currently being compared, completed models with quality,
|
|
187
251
|
cost, and latency, trial budget, pending recommendations, and any exact-task result
|
|
188
252
|
measured in another session. When the gate declines an automatic swap, it
|
|
189
|
-
prints reasons and `/openmerit apply` remains available.
|
|
253
|
+
prints reasons and `/openmerit apply` remains available. It also lists every
|
|
254
|
+
route skipped during the current comparison with the relevant policy, pricing,
|
|
255
|
+
modality, or per-trial budget reason.
|
|
190
256
|
|
|
191
257
|
The daily trial-count and dollar limits come from `~/.openmerit/policy.json`.
|
|
192
258
|
If a limit is reached, the extension reports why it skipped the comparison;
|
|
@@ -203,10 +269,21 @@ Closing or switching the session cancels the active job.
|
|
|
203
269
|
Candidate runs use the same text and uploaded image bytes, but they do not
|
|
204
270
|
replay earlier answers or file changes. An exact task in another session
|
|
205
271
|
(including identical image bytes) appears as **advice**, not a pending swap;
|
|
206
|
-
the new session still gets its own comparison.
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
272
|
+
the new session still gets its own comparison.
|
|
273
|
+
|
|
274
|
+
The optional `trial` command runs controlled text-task comparisons through the
|
|
275
|
+
same exact Pi provider routes and credential store as the automatic session
|
|
276
|
+
path:
|
|
277
|
+
|
|
278
|
+
```bash
|
|
279
|
+
npx openmerit trial examples/task.example.json --rounds 3
|
|
280
|
+
```
|
|
281
|
+
|
|
282
|
+
`initial_model` is the stable `vendor/model` identity. When Pi exposes that
|
|
283
|
+
model through more than one provider, set `initial_route` to
|
|
284
|
+
`provider:model-id` (for example `openai:gpt-4o-mini` or
|
|
285
|
+
`openrouter:openai/gpt-4o-mini`). Judge and strategist calls also run through
|
|
286
|
+
Pi. An OpenRouter key only adds optional catalog and public-benchmark metadata.
|
|
210
287
|
|
|
211
288
|
## Alpha boundaries
|
|
212
289
|
|
|
@@ -237,7 +314,9 @@ session path is provider-neutral.
|
|
|
237
314
|
for automatic session comparisons plus the daily candidate count.
|
|
238
315
|
`max_usd_per_trial` is a conservative admission estimate based on known Pi
|
|
239
316
|
prices and a 4K answer; it is not a provider-side hard cap. Routes without
|
|
240
|
-
known pricing are excluded
|
|
317
|
+
known pricing are excluded as candidates and cannot auto-apply. Use an exact
|
|
318
|
+
`route_overrides` entry for a custom/local route whose price is known; zero
|
|
319
|
+
cost must be stated explicitly.
|
|
241
320
|
- Each comparison has one observed baseline plus a small candidate slate and
|
|
242
321
|
one quality score per answer. Treat recommendations as experimental evidence,
|
|
243
322
|
not a universal model ranking.
|
|
@@ -246,18 +325,21 @@ session path is provider-neutral.
|
|
|
246
325
|
enrichment source. `HarnessAdapter`, `ModelProviderAdapter`,
|
|
247
326
|
`ObservationSource`, and `EventSink` remain separate integration boundaries;
|
|
248
327
|
Pi and local JSONL are the implementations shipped in this release.
|
|
249
|
-
- Candidate subprocesses can use Pi built-ins
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
328
|
+
- Candidate subprocesses can use Pi built-ins, configured custom/local routes,
|
|
329
|
+
and providers registered by explicitly allowlisted extension files. Normal
|
|
330
|
+
extension discovery remains disabled, and OpenMerit refuses to load its own
|
|
331
|
+
extension recursively.
|
|
332
|
+
- JSON state snapshots are replaced atomically. Append-only readers skip and
|
|
333
|
+
report malformed or interrupted lines while retaining later valid records.
|
|
334
|
+
A job owned by a crashed process is reclaimable instead of remaining stuck
|
|
335
|
+
in `running`; completed jobs remain final.
|
|
254
336
|
|
|
255
337
|
## State layout (`~/.openmerit/`)
|
|
256
338
|
|
|
257
339
|
| file | contents |
|
|
258
340
|
|---|---|
|
|
259
|
-
| `policy.json` |
|
|
260
|
-
| `harness-state.json` | current/fallback routes, Pi's eligible route snapshot, latest session and settled task |
|
|
341
|
+
| `policy.json` | gate thresholds, budgets, intervals, exact-route metadata overrides, and allowlisted Pi provider extensions |
|
|
342
|
+
| `harness-state.json` | current/fallback routes, Pi's eligible route snapshot, pause state, latest session and settled task |
|
|
261
343
|
| `recommendations.jsonl` | append-only session-bound recommendations, routes, evidence, gate reasons, and status updates |
|
|
262
344
|
| `trials.jsonl` | every model trial point, including its provider route when known |
|
|
263
345
|
| `traces/observations.jsonl` | task observations extracted from session traces |
|
package/dist/cli.js
CHANGED
|
@@ -1,36 +1,18 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor. */
|
|
3
|
-
import { copyFileSync, existsSync, mkdirSync
|
|
2
|
+
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor / verify. */
|
|
3
|
+
import { copyFileSync, existsSync, mkdirSync } from "node:fs";
|
|
4
4
|
import { dirname, join } from "node:path";
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
6
6
|
import { loadPolicy } from "./policy.js";
|
|
7
7
|
import { loadKey } from "./llm.js";
|
|
8
|
-
import { fetchCatalog } from "./catalog.js";
|
|
9
8
|
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
10
|
-
import {
|
|
11
|
-
import { appendJsonl, paths, readJson, readJsonl, taskKey } from "./store.js";
|
|
9
|
+
import { paths, readJson, readJsonl } from "./store.js";
|
|
12
10
|
import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
|
|
13
|
-
import {
|
|
14
|
-
import { availablePiModels, availablePiRoutes, piExecutable, recordedActiveTask, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
15
|
-
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
16
|
-
import { JUDGE_PREFS } from "./judge.js";
|
|
17
|
-
import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
|
|
11
|
+
import { settledActiveTask } from "./pi-trials.js";
|
|
18
12
|
import { modelRoute, routeLabel } from "./routes.js";
|
|
19
|
-
import {
|
|
13
|
+
import { runStandaloneTrial } from "./standalone.js";
|
|
14
|
+
import { collectDoctorReport, renderDiagnosticReport, runOfflineVerify } from "./diagnostics.js";
|
|
20
15
|
const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
21
|
-
function resolvePref(cat, prefs, label) {
|
|
22
|
-
for (const p of prefs)
|
|
23
|
-
if (p && cat.has(p))
|
|
24
|
-
return p;
|
|
25
|
-
for (const p of prefs) {
|
|
26
|
-
if (!p)
|
|
27
|
-
continue;
|
|
28
|
-
for (const id of [...cat.keys()].sort())
|
|
29
|
-
if (id.includes(p))
|
|
30
|
-
return id;
|
|
31
|
-
}
|
|
32
|
-
throw new Error(`could not resolve ${label} model`);
|
|
33
|
-
}
|
|
34
16
|
function cmdInit() {
|
|
35
17
|
const policyPath = paths.policy();
|
|
36
18
|
mkdirSync(dirname(policyPath), { recursive: true });
|
|
@@ -55,129 +37,16 @@ function optionalOpenRouterKey() {
|
|
|
55
37
|
return null;
|
|
56
38
|
}
|
|
57
39
|
}
|
|
58
|
-
function
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
console.log(`pi executable : ${executable}`);
|
|
63
|
-
console.log(`pi detected : ${version.status === 0 ? `yes (${version.stdout.trim()})` : "no"}`);
|
|
64
|
-
try {
|
|
65
|
-
const policy = loadPolicy(paths.policy());
|
|
66
|
-
console.log(`policy : valid (v${policy.version}, ${policy.mode})`);
|
|
67
|
-
}
|
|
68
|
-
catch (error) {
|
|
69
|
-
console.log(`policy : invalid (${error.message})`);
|
|
70
|
-
}
|
|
71
|
-
let routes = state.routes ?? [];
|
|
72
|
-
const hasLiveSnapshot = routes.length > 0;
|
|
73
|
-
if (routes.length === 0 && version.status === 0) {
|
|
74
|
-
try {
|
|
75
|
-
routes = availablePiRoutes();
|
|
76
|
-
}
|
|
77
|
-
catch { /* reported below */ }
|
|
78
|
-
}
|
|
79
|
-
console.log(`eligible routes: ${routes.length}`);
|
|
80
|
-
for (const route of routes.slice(0, 8))
|
|
81
|
-
console.log(` - ${routeLabel(route)}${route.cost ? "" : " (price unknown)"}`);
|
|
82
|
-
if (routes.length > 8)
|
|
83
|
-
console.log(` … ${routes.length - 8} more`);
|
|
84
|
-
console.log(`active route : ${state.currentRoute ? routeLabel(state.currentRoute) : "not reported; start pi once"}`);
|
|
85
|
-
const pricedRoutes = routes.filter((route) => !!route.cost).length;
|
|
86
|
-
console.log(`alternates : ${hasLiveSnapshot
|
|
87
|
-
? pricedRoutes >= 2 ? "ready" : "need at least two priced eligible pi routes"
|
|
88
|
-
: "start Pi once to capture route prices and capabilities"}`);
|
|
89
|
-
console.log(`OpenRouter enrichment: ${optionalOpenRouterKey() ? "available" : "not configured (optional)"}`);
|
|
40
|
+
function printReport(report, json) {
|
|
41
|
+
console.log(json ? JSON.stringify(report, null, 2) : renderDiagnosticReport(report));
|
|
42
|
+
if (!report.ok)
|
|
43
|
+
process.exitCode = 1;
|
|
90
44
|
}
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
const cat = await fetchCatalog(key);
|
|
97
|
-
const piModels = availablePiModels();
|
|
98
|
-
for (const id of [...cat.keys()])
|
|
99
|
-
if (!piModels.has(id))
|
|
100
|
-
cat.delete(id);
|
|
101
|
-
if (!cat.has(cfg.initial_model))
|
|
102
|
-
throw new Error(`initial_model ${cfg.initial_model} is unavailable in pi's OpenRouter registry`);
|
|
103
|
-
console.log(`catalog: ${cat.size} models selectable in pi`);
|
|
104
|
-
const judge = resolvePref(cat, [cfg.judge_model ?? policy.judge_model, ...JUDGE_PREFS], "judge");
|
|
105
|
-
const strat = resolvePref(cat, [cfg.strategist_model ?? policy.strategist_model, ...STRAT_PREFS], "strategist");
|
|
106
|
-
const { category, benchmarks } = relevantBenchmarks(cfg.task);
|
|
107
|
-
let benchmarkCandidates = [];
|
|
108
|
-
try {
|
|
109
|
-
benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, cat);
|
|
110
|
-
}
|
|
111
|
-
catch (e) {
|
|
112
|
-
console.log(`OpenRouter benchmark shortlist unavailable: ${e.message}`);
|
|
113
|
-
}
|
|
114
|
-
console.log(`judge: ${judge} | strategist: ${strat} | benchmarks: ${benchmarks.join(", ")}`);
|
|
115
|
-
console.log(`OpenRouter benchmark candidates: ${benchmarkCandidates.slice(0, 5).join(", ") || "none"}`);
|
|
116
|
-
const tKey = taskKey(cfg.task);
|
|
117
|
-
const maxPrice = cfg.max_usd_per_m ?? policy.max_usd_per_m;
|
|
118
|
-
const tried = new Set();
|
|
119
|
-
const results = [];
|
|
120
|
-
const failedVendors = new Set();
|
|
121
|
-
for (let i = 0; i < rounds; i++) {
|
|
122
|
-
const budget = budgetOk(policy);
|
|
123
|
-
if (!budget.ok) {
|
|
124
|
-
console.log(`budget: ${budget.reason}; stopping`);
|
|
125
|
-
break;
|
|
126
|
-
}
|
|
127
|
-
let model;
|
|
128
|
-
let why;
|
|
129
|
-
if (i === 0) {
|
|
130
|
-
model = cfg.initial_model;
|
|
131
|
-
why = "initial model";
|
|
132
|
-
}
|
|
133
|
-
else {
|
|
134
|
-
const pick = await pickNext(key, strat, cfg.task, cfg.eval, benchmarks, results, cat, tried, maxPrice, failedVendors, benchmarkCandidates);
|
|
135
|
-
if (!pick)
|
|
136
|
-
break;
|
|
137
|
-
model = pick.model;
|
|
138
|
-
why = pick.why;
|
|
139
|
-
}
|
|
140
|
-
if (tried.has(model))
|
|
141
|
-
break;
|
|
142
|
-
console.log(`[round ${i + 1}/${rounds}] ${model} (${why}) ...`);
|
|
143
|
-
const observed = i === 0 ? recordedActiveTask(cfg.task, model) : null;
|
|
144
|
-
if (i === 0)
|
|
145
|
-
console.log(observed ? " using model A's active pi session trace" : " active trace unavailable; running model A in a fresh pi session");
|
|
146
|
-
const { point, costUsd, error, sessionId } = observed
|
|
147
|
-
? await scorePiRun(key, judge, cfg.task, cfg.eval, model, cat.get(model), observed, "trace")
|
|
148
|
-
: await runPiTrial(key, judge, cfg.task, cfg.eval, model, cat.get(model));
|
|
149
|
-
tried.add(model);
|
|
150
|
-
results.push(point);
|
|
151
|
-
recordTrialSpend(costUsd);
|
|
152
|
-
appendJsonl(paths.trials(), { ...point, taskKey: tKey });
|
|
153
|
-
if (sessionId)
|
|
154
|
-
console.log(` pi trace session: ${sessionId}`);
|
|
155
|
-
if (error) {
|
|
156
|
-
failedVendors.add(model.split("/")[0]);
|
|
157
|
-
console.log(` FAILED: ${error.slice(0, 160)}`);
|
|
158
|
-
}
|
|
159
|
-
else {
|
|
160
|
-
console.log(` score=${point.score.toFixed(2)} $/M=${point.price} est_cost=$${costUsd.toFixed(4)} ${(point.why ?? "").slice(0, 110)}`);
|
|
161
|
-
}
|
|
162
|
-
}
|
|
163
|
-
const pts = results.filter((p) => p.score > 0);
|
|
164
|
-
if (pts.length === 0) {
|
|
165
|
-
console.log("no trials completed");
|
|
166
|
-
process.exit(1);
|
|
167
|
-
}
|
|
168
|
-
const chain = paretoFrontier(pts);
|
|
169
|
-
const best = pickBest(pts);
|
|
170
|
-
const fb = pickFallback(pts, best, policy.fallback.min_score);
|
|
171
|
-
console.log("\n=== pareto frontier (nondominated: score up, price down, latency down) ===");
|
|
172
|
-
for (const p of chain)
|
|
173
|
-
console.log(` ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M`);
|
|
174
|
-
console.log(`\nBEST FIT : ${best.model} (score ${best.score.toFixed(2)}, $${best.price.toFixed(2)}/M)`);
|
|
175
|
-
console.log(fb ? `FALLBACK : ${fb.model} (score ${fb.score.toFixed(2)}, $${fb.price.toFixed(2)}/M)` : "FALLBACK : n/a");
|
|
176
|
-
const rec = buildRecommendation(tKey, cfg.task.slice(0, 120), cfg.initial_model, pts, policy);
|
|
177
|
-
if (rec) {
|
|
178
|
-
appendJsonl(paths.recommendations(), rec);
|
|
179
|
-
console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${rec.policy.autoApply})`);
|
|
180
|
-
}
|
|
45
|
+
function cmdDoctor(json) {
|
|
46
|
+
printReport(collectDoctorReport(), json);
|
|
47
|
+
}
|
|
48
|
+
function cmdVerify(json) {
|
|
49
|
+
printReport(runOfflineVerify(), json);
|
|
181
50
|
}
|
|
182
51
|
function loadFrontiers() {
|
|
183
52
|
const trials = readJsonl(paths.trials());
|
|
@@ -243,12 +112,12 @@ function cmdStatus() {
|
|
|
243
112
|
}
|
|
244
113
|
async function main() {
|
|
245
114
|
const [cmd, ...args] = process.argv.slice(2);
|
|
246
|
-
const policy = loadPolicy(paths.policy());
|
|
247
115
|
switch (cmd) {
|
|
248
116
|
case "init":
|
|
249
117
|
cmdInit();
|
|
250
118
|
break;
|
|
251
119
|
case "session-trial": {
|
|
120
|
+
const policy = loadPolicy(paths.policy());
|
|
252
121
|
const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
|
|
253
122
|
if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
|
|
254
123
|
throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
|
|
@@ -269,29 +138,32 @@ async function main() {
|
|
|
269
138
|
}
|
|
270
139
|
case "watch":
|
|
271
140
|
if (args.includes("--once"))
|
|
272
|
-
await tickOnce(policy);
|
|
141
|
+
await tickOnce(loadPolicy(paths.policy()));
|
|
273
142
|
else
|
|
274
|
-
await runDaemon(policy);
|
|
143
|
+
await runDaemon(loadPolicy(paths.policy()));
|
|
275
144
|
break;
|
|
276
145
|
case "trial": {
|
|
277
146
|
const file = args[0];
|
|
278
147
|
if (!file)
|
|
279
148
|
throw new Error("usage: openmerit trial <task.json> [--rounds N]");
|
|
280
149
|
const rIdx = args.indexOf("--rounds");
|
|
281
|
-
await
|
|
150
|
+
await runStandaloneTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
|
|
282
151
|
break;
|
|
283
152
|
}
|
|
284
153
|
case "frontier":
|
|
285
154
|
cmdFrontier(args[0]);
|
|
286
155
|
break;
|
|
287
156
|
case "recommend":
|
|
288
|
-
recommendTick(policy);
|
|
157
|
+
recommendTick(loadPolicy(paths.policy()));
|
|
289
158
|
break;
|
|
290
159
|
case "status":
|
|
291
160
|
cmdStatus();
|
|
292
161
|
break;
|
|
293
162
|
case "doctor":
|
|
294
|
-
cmdDoctor();
|
|
163
|
+
cmdDoctor(args.includes("--json"));
|
|
164
|
+
break;
|
|
165
|
+
case "verify":
|
|
166
|
+
cmdVerify(args.includes("--json"));
|
|
295
167
|
break;
|
|
296
168
|
default:
|
|
297
169
|
console.log(`openmerit — external model-merit harness
|
|
@@ -304,7 +176,8 @@ usage: openmerit <command>
|
|
|
304
176
|
frontier [taskKey] print pareto frontier(s)
|
|
305
177
|
recommend emit recommendations now
|
|
306
178
|
status harness model, queued models, pending recommendations
|
|
307
|
-
doctor
|
|
179
|
+
doctor [--json] diagnose pi, policy, routes, state, and installation
|
|
180
|
+
verify [--json] run an offline, provider-free core self-test`);
|
|
308
181
|
if (cmd && cmd !== "help" && cmd !== "--help")
|
|
309
182
|
process.exitCode = 1;
|
|
310
183
|
}
|
package/dist/daemon.js
CHANGED
|
@@ -11,10 +11,10 @@ import { buildRecommendation } from "./recommend.js";
|
|
|
11
11
|
import { appendJsonl, paths, readJson, readJsonl, writeJson, } from "./store.js";
|
|
12
12
|
import { budgetOk, ensureRubric, recordMeritSpend, recordTrialSpend, runTrial, trialBudgetOk } from "./trials.js";
|
|
13
13
|
import { JUDGE_PREFS } from "./judge.js";
|
|
14
|
-
import { runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
14
|
+
import { PiHarnessAdapter, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
15
15
|
import { taskInputKey } from "./task-input.js";
|
|
16
16
|
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
17
|
-
import { enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
|
|
17
|
+
import { applyRouteOverrides, enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
|
|
18
18
|
import { PiCliChatClient } from "./llm.js";
|
|
19
19
|
import { LocalJsonlEventSink, PiTraceObservationSource, meritEvent } from "./integrations.js";
|
|
20
20
|
function resolveRoutePref(cat, prefs, label, requireImages = false) {
|
|
@@ -42,6 +42,28 @@ function resolveRoutePref(cat, prefs, label, requireImages = false) {
|
|
|
42
42
|
return fallback;
|
|
43
43
|
throw new Error(`could not resolve ${label} route from Pi's eligible models`);
|
|
44
44
|
}
|
|
45
|
+
function processIsAlive(pid) {
|
|
46
|
+
if (!Number.isInteger(pid) || pid <= 0)
|
|
47
|
+
return false;
|
|
48
|
+
try {
|
|
49
|
+
process.kill(pid, 0);
|
|
50
|
+
return true;
|
|
51
|
+
}
|
|
52
|
+
catch (error) {
|
|
53
|
+
return error.code === "EPERM";
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
/** Completed jobs stay final; live owners retain their claim; abandoned claims can retry. */
|
|
57
|
+
export function taskJobIsClaimed(prior, marker, now = Date.now()) {
|
|
58
|
+
if (!prior || prior.marker !== marker || prior.status === "failed")
|
|
59
|
+
return false;
|
|
60
|
+
if (prior.status !== "running")
|
|
61
|
+
return true;
|
|
62
|
+
if (prior.ownerPid !== undefined)
|
|
63
|
+
return processIsAlive(prior.ownerPid);
|
|
64
|
+
const updated = Date.parse(prior.updatedAt);
|
|
65
|
+
return Number.isFinite(updated) && now - updated < 15 * 60_000;
|
|
66
|
+
}
|
|
45
67
|
function emitTrialProgress(progress) {
|
|
46
68
|
console.log(`[openmerit/progress] ${JSON.stringify(progress)}`);
|
|
47
69
|
}
|
|
@@ -150,20 +172,25 @@ export async function trialTick(key, policy) {
|
|
|
150
172
|
/** Compare the one completed active pi task, sequentially, then target its session. */
|
|
151
173
|
export async function autoTaskTick(key, policy, settledTask) {
|
|
152
174
|
const sink = new LocalJsonlEventSink();
|
|
153
|
-
|
|
175
|
+
let task = settledTask === undefined ? settledActiveTask() : settledTask;
|
|
154
176
|
if (!task)
|
|
155
177
|
return null;
|
|
178
|
+
const routes = applyRouteOverrides(task.routes, policy.route_overrides);
|
|
179
|
+
const currentRoute = applyRouteOverrides([task.route], policy.route_overrides)[0];
|
|
180
|
+
task = { ...task, routes, route: currentRoute };
|
|
181
|
+
const tKey = taskInputKey(task.task, task.images, task.files);
|
|
156
182
|
const marker = settledTask === undefined
|
|
157
183
|
? `${task.sessionFile}:${task.settledAt}`
|
|
158
184
|
: `${task.sessionFile}:${task.sessionBytes ?? task.settledAt}`;
|
|
159
185
|
const processedPath = settledTask === undefined ? paths.watchProcessed() : paths.sessionJob(marker);
|
|
160
186
|
const prior = readJson(processedPath, null);
|
|
161
|
-
if (prior
|
|
187
|
+
if (taskJobIsClaimed(prior, marker))
|
|
162
188
|
return null;
|
|
163
189
|
if (!budgetOk(policy).ok)
|
|
164
190
|
return null;
|
|
165
191
|
// Claim the settled turn before provider calls so the next poll cannot duplicate it.
|
|
166
|
-
writeJson(processedPath, { marker, status: "running",
|
|
192
|
+
writeJson(processedPath, { marker, status: "running", ownerPid: process.pid,
|
|
193
|
+
updatedAt: new Date().toISOString() });
|
|
167
194
|
try {
|
|
168
195
|
let cat = routeCatalog(task.routes);
|
|
169
196
|
if (!cat.has(routeKey(task.route)))
|
|
@@ -179,13 +206,28 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
179
206
|
console.log(`[openmerit] OpenRouter enrichment unavailable: ${error.message}`);
|
|
180
207
|
}
|
|
181
208
|
}
|
|
209
|
+
const excluded = [];
|
|
182
210
|
for (const [id, entry] of [...cat]) {
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
(
|
|
186
|
-
|
|
211
|
+
const reasons = [];
|
|
212
|
+
if (entry.priceKnown !== false && entry.price > policy.max_usd_per_m)
|
|
213
|
+
reasons.push(`price $${entry.price.toFixed(2)}/M exceeds $${policy.max_usd_per_m.toFixed(2)}/M policy limit`);
|
|
214
|
+
if (!providerAllowed(policy, entry.id, entry.route?.provider))
|
|
215
|
+
reasons.push("provider is denied by policy");
|
|
216
|
+
if (task.images.length > 0 && !entry.inputModalities?.includes("image"))
|
|
217
|
+
reasons.push("route does not advertise image input");
|
|
218
|
+
if (id !== routeKey(task.route)) {
|
|
219
|
+
const admission = trialBudgetOk(policy, entry, task.task.length);
|
|
220
|
+
if (!admission.ok && admission.reason)
|
|
221
|
+
reasons.push(admission.reason);
|
|
222
|
+
}
|
|
223
|
+
if (reasons.length) {
|
|
187
224
|
cat.delete(id);
|
|
225
|
+
excluded.push({ model: entry.route ? routeLabel(entry.route) : entry.id, reasons: [...new Set(reasons)] });
|
|
226
|
+
}
|
|
188
227
|
}
|
|
228
|
+
if (settledTask !== undefined)
|
|
229
|
+
emitTrialProgress({ phase: "selection", taskKey: tKey,
|
|
230
|
+
eligible: [...cat.values()].map((entry) => entry.route ? routeLabel(entry.route) : entry.id), excluded });
|
|
189
231
|
const activeEntry = cat.get(routeKey(task.route));
|
|
190
232
|
if (!activeEntry)
|
|
191
233
|
throw new Error(`active route ${task.route.provider}/${task.route.modelId} is excluded by policy or task capabilities`);
|
|
@@ -193,7 +235,8 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
193
235
|
throw new Error("Pi exposes no eligible alternate model route for this task");
|
|
194
236
|
const judge = resolveRoutePref(cat, [policy.judge_model, ...JUDGE_PREFS, task.model], "judge", task.images.length > 0);
|
|
195
237
|
const strategist = resolveRoutePref(cat, [policy.strategist_model, ...STRAT_PREFS, task.model], "strategist");
|
|
196
|
-
const client = new PiCliChatClient(undefined, recordMeritSpend);
|
|
238
|
+
const client = new PiCliChatClient(undefined, recordMeritSpend, policy.pi.provider_extensions);
|
|
239
|
+
const harness = new PiHarnessAdapter(policy.pi.provider_extensions);
|
|
197
240
|
const { category, benchmarks } = relevantBenchmarks(task.task);
|
|
198
241
|
let benchmarkCandidates = [];
|
|
199
242
|
try {
|
|
@@ -204,7 +247,6 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
204
247
|
console.log(`[openmerit] benchmark shortlist unavailable: ${e.message}`);
|
|
205
248
|
}
|
|
206
249
|
const rubric = await ensureRubric(key ?? "", judge.route, task.task, new Map(), client);
|
|
207
|
-
const tKey = taskInputKey(task.task, task.images, task.files);
|
|
208
250
|
const points = [];
|
|
209
251
|
const tried = new Set();
|
|
210
252
|
const failedVendors = new Set();
|
|
@@ -233,7 +275,7 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
233
275
|
emitTrialProgress({ phase: "start", taskKey: tKey,
|
|
234
276
|
model: pick.route ? routeLabel(pick.route) : pick.model, index: i + 1, total });
|
|
235
277
|
const entry = pick.route ? cat.get(routeKey(pick.route)) : cat.get(pick.model);
|
|
236
|
-
const { point, costUsd, error } = await runPiTrial(key ?? "", judge.route, task.task, rubric, pick.model, entry, task.cwd, task.images, task.files,
|
|
278
|
+
const { point, costUsd, error } = await runPiTrial(key ?? "", judge.route, task.task, rubric, pick.model, entry, task.cwd, task.images, task.files, harness, client);
|
|
237
279
|
tried.add(pick.route ? routeKey(pick.route) : pick.model);
|
|
238
280
|
points.push(point);
|
|
239
281
|
recordTrialSpend(costUsd);
|
|
@@ -258,12 +300,12 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
258
300
|
else
|
|
259
301
|
console.log(`[openmerit] task ${tKey}: active model remains best`);
|
|
260
302
|
writeJson(processedPath, { marker, status: rec ? "complete" : "no_swap",
|
|
261
|
-
updatedAt: new Date().toISOString() });
|
|
303
|
+
ownerPid: process.pid, updatedAt: new Date().toISOString() });
|
|
262
304
|
return rec;
|
|
263
305
|
}
|
|
264
306
|
catch (error) {
|
|
265
307
|
writeJson(processedPath, { marker, status: "failed",
|
|
266
|
-
updatedAt: new Date().toISOString() });
|
|
308
|
+
ownerPid: process.pid, updatedAt: new Date().toISOString() });
|
|
267
309
|
throw error;
|
|
268
310
|
}
|
|
269
311
|
}
|