openmerit 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +89 -47
- package/dist/cli.js +48 -121
- package/dist/daemon.js +131 -52
- package/dist/diagnostics.js +194 -0
- package/dist/frontier.js +13 -6
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +102 -1
- package/dist/pi-trials.js +60 -24
- package/dist/policy.js +8 -3
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/standalone.js +220 -0
- package/dist/store.js +43 -10
- package/dist/strategist.js +18 -14
- package/dist/traces.js +6 -2
- package/dist/trials.js +50 -11
- package/examples/task.example.json +1 -0
- package/extension/openmerit.ts +134 -32
- package/instructions/OPENMERIT.md +2 -2
- package/package.json +5 -2
- package/rules.md +39 -0
package/README.md
CHANGED
|
@@ -32,10 +32,10 @@ pi task A → saved pi trace → extension trial job → pi trials B, C → scor
|
|
|
32
32
|
([`src/frontier.ts`](src/frontier.ts), 3 objectives: quality ↑, price ↓,
|
|
33
33
|
latency ↓). The aggregate across all of an agent's tasks is the union of
|
|
34
34
|
its task frontiers (`openmerit frontier`).
|
|
35
|
-
4. **Model discovery** — the
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
35
|
+
4. **Model discovery** — the active Pi session's scoped models (or Pi's
|
|
36
|
+
authenticated available-model registry when unscoped) are the candidate
|
|
37
|
+
pool. OpenRouter is an optional route and enrichment source: its catalog can
|
|
38
|
+
be snapshotted and its benchmark signals can improve the shortlist
|
|
39
39
|
([`src/catalog.ts`](src/catalog.ts) + [`src/daemon.ts`](src/daemon.ts)).
|
|
40
40
|
5. **Informing the main harness** — recommendations land in
|
|
41
41
|
`~/.openmerit/recommendations.jsonl`. The pi extension
|
|
@@ -47,19 +47,21 @@ pi task A → saved pi trace → extension trial job → pi trials B, C → scor
|
|
|
47
47
|
|
|
48
48
|
## Install from npm
|
|
49
49
|
|
|
50
|
-
You need Node 22.18+, pi 0.85.1+, and
|
|
51
|
-
|
|
50
|
+
You need Node 22.18+, pi 0.85.1+, and at least two eligible model routes
|
|
51
|
+
authenticated in Pi. OpenRouter is supported but not required. Install the
|
|
52
|
+
package into pi, then initialize its policy and state:
|
|
52
53
|
|
|
53
54
|
```bash
|
|
54
55
|
pi install npm:openmerit
|
|
55
56
|
npx --yes openmerit init
|
|
57
|
+
npx --yes openmerit verify
|
|
56
58
|
pi list
|
|
57
59
|
```
|
|
58
60
|
|
|
59
61
|
`pi install` makes the extension and its trial engine available to pi. The
|
|
60
62
|
one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
|
|
61
63
|
globally with `npm install --global openmerit` only if you also want persistent
|
|
62
|
-
shell access to `openmerit status`, `frontier`, or the optional watcher. Install
|
|
64
|
+
shell access to `openmerit status`, `doctor`, `verify`, `frontier`, or the optional watcher. Install
|
|
63
65
|
the extension from only one source—remove any older copied `openmerit.ts` first
|
|
64
66
|
so pi does not load it twice.
|
|
65
67
|
|
|
@@ -85,40 +87,45 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
85
87
|
policy**. The shipped policy uses `"mode": "recommend"` and does not change
|
|
86
88
|
the active model without approval.
|
|
87
89
|
|
|
88
|
-
2.
|
|
89
|
-
|
|
90
|
-
|
|
90
|
+
2. Authenticate the providers you want to compare in Pi, using `/login` or
|
|
91
|
+
Pi's normal environment/model configuration. OpenMerit takes its eligible
|
|
92
|
+
model routes, capabilities, prices, and credentials from Pi; judge and
|
|
93
|
+
strategist calls use the same routes and do not require duplicate keys.
|
|
94
|
+
|
|
95
|
+
For example, verify one provider without printing its credential:
|
|
91
96
|
|
|
92
97
|
```bash
|
|
93
|
-
|
|
94
|
-
chmod 600 ~/.openmerit/.env
|
|
98
|
+
pi auth check --provider openai --json
|
|
95
99
|
```
|
|
96
100
|
|
|
97
|
-
|
|
98
|
-
`/login openrouter` inside pi, or export `OPENROUTER_API_KEY` in the shell
|
|
99
|
-
that starts pi. Verify pi's side without printing the key:
|
|
101
|
+
Add an OpenRouter route the same way if you want its routed catalog:
|
|
100
102
|
|
|
101
103
|
```bash
|
|
102
104
|
pi auth check --provider openrouter --json
|
|
103
105
|
```
|
|
104
106
|
|
|
105
|
-
|
|
106
|
-
|
|
107
|
+
An `OPENROUTER_API_KEY` in the process environment or
|
|
108
|
+
`~/.openmerit/.env` additionally enables OpenRouter catalog and public-
|
|
109
|
+
benchmark enrichment. It is optional for native-provider comparisons. If
|
|
110
|
+
using the file, protect it:
|
|
107
111
|
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
112
|
+
```bash
|
|
113
|
+
nano ~/.openmerit/.env
|
|
114
|
+
chmod 600 ~/.openmerit/.env
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Pi does not read OpenMerit's `.env` file for its ordinary sessions, so an
|
|
118
|
+
OpenRouter route still needs Pi authentication. Older `model_search/.env`
|
|
119
|
+
files are not read.
|
|
115
120
|
|
|
116
121
|
3. Verify the extension appears in `pi list`. The instruction file
|
|
117
122
|
[`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
|
|
118
123
|
pi project's AGENTS.md for agent context, but the extension does
|
|
119
|
-
not require it.
|
|
124
|
+
not require it. Run `npx openmerit verify` for a provider-free core self-test,
|
|
125
|
+
then `npx openmerit doctor` after starting Pi once to inspect the eligible
|
|
126
|
+
route snapshot and configuration without printing secrets.
|
|
120
127
|
|
|
121
|
-
4. Start pi in one terminal with
|
|
128
|
+
4. Start pi in one terminal with any configured model. For example:
|
|
122
129
|
|
|
123
130
|
```bash
|
|
124
131
|
pi --provider openrouter --model openai/gpt-4o-mini
|
|
@@ -127,7 +134,8 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
127
134
|
For a first text task, ask: “Give the shortest valid word ladder from cat
|
|
128
135
|
to dog. Each step changes one letter and must be a common English word.
|
|
129
136
|
Return only the path.” Wait for A to finish and leave the pi session open.
|
|
130
|
-
The extension automatically starts B and C **sequentially through
|
|
137
|
+
The extension automatically starts B and C **sequentially through their
|
|
138
|
+
selected Pi routes**,
|
|
131
139
|
reports each score in Pi, and writes a recommendation for this exact
|
|
132
140
|
session. With the shipped supervised policy, use
|
|
133
141
|
`/openmerit` inside pi to inspect the evidence and `/openmerit apply` to
|
|
@@ -173,7 +181,13 @@ schema, so its published scores are not directly comparable to these pi runs.
|
|
|
173
181
|
|
|
174
182
|
### Inspect or troubleshoot a run
|
|
175
183
|
|
|
176
|
-
`npx openmerit
|
|
184
|
+
`npx openmerit doctor` checks Pi, policy, eligible routes and prices, saved
|
|
185
|
+
session state, duplicate package sources, append-only files, and optional
|
|
186
|
+
OpenRouter enrichment. Add `--json` for a sanitized diagnostic report suitable
|
|
187
|
+
for a bug report; it contains no credentials, prompts, or trace contents.
|
|
188
|
+
`npx openmerit verify` runs an offline self-test of atomic state, JSONL recovery,
|
|
189
|
+
route preservation, policy evidence, and neutral events without contacting a
|
|
190
|
+
provider. `npx openmerit status` shows the latest pi model and pending recommendations;
|
|
177
191
|
`npx openmerit frontier` shows measured quality, blended price, latency, and
|
|
178
192
|
the chosen frontier per task. `/openmerit` inside pi shows the current model,
|
|
179
193
|
fallback, the model currently being compared, completed models with quality,
|
|
@@ -189,23 +203,37 @@ candidate-spend threshold.
|
|
|
189
203
|
|
|
190
204
|
If no comparison starts after Pi settles, confirm that pi loaded the extension
|
|
191
205
|
(`pi list`), the task finished, and the pi session is saved (do not use
|
|
192
|
-
`--no-session`). Image candidates must advertise
|
|
193
|
-
|
|
206
|
+
`--no-session`). Image candidates must advertise image input in Pi's model
|
|
207
|
+
registry. The extension queues
|
|
194
208
|
completed tasks from its current session and runs one comparison at a time.
|
|
195
209
|
Closing or switching the session cancels the active job.
|
|
196
210
|
Candidate runs use the same text and uploaded image bytes, but they do not
|
|
197
211
|
replay earlier answers or file changes. An exact task in another session
|
|
198
212
|
(including identical image bytes) appears as **advice**, not a pending swap;
|
|
199
|
-
the new session still gets its own comparison.
|
|
200
|
-
|
|
213
|
+
the new session still gets its own comparison.
|
|
214
|
+
|
|
215
|
+
The optional `trial` command runs controlled text-task comparisons through the
|
|
216
|
+
same exact Pi provider routes and credential store as the automatic session
|
|
217
|
+
path:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
npx openmerit trial examples/task.example.json --rounds 3
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
`initial_model` is the stable `vendor/model` identity. When Pi exposes that
|
|
224
|
+
model through more than one provider, set `initial_route` to
|
|
225
|
+
`provider:model-id` (for example `openai:gpt-4o-mini` or
|
|
226
|
+
`openrouter:openai/gpt-4o-mini`). Judge and strategist calls also run through
|
|
227
|
+
Pi. An OpenRouter key only adds optional catalog and public-benchmark metadata.
|
|
201
228
|
|
|
202
229
|
## Alpha boundaries
|
|
203
230
|
|
|
204
|
-
- With the extension installed and
|
|
205
|
-
settled task can start comparison calls automatically. Candidate,
|
|
206
|
-
strategist requests send the task text, attached files or images,
|
|
207
|
-
candidate output to
|
|
208
|
-
tasks and conservative account limits while evaluating
|
|
231
|
+
- With the extension installed and at least two eligible Pi routes, each
|
|
232
|
+
supported settled task can start comparison calls automatically. Candidate,
|
|
233
|
+
judge, and strategist requests send the task text, attached files or images,
|
|
234
|
+
and candidate output to the configured providers and can incur charges. Use
|
|
235
|
+
non-sensitive test tasks and conservative account limits while evaluating
|
|
236
|
+
this alpha.
|
|
209
237
|
- The extension queues tasks from its current pi session and compares one at a
|
|
210
238
|
time. Candidate runs use isolated temporary copies of the original working
|
|
211
239
|
directory and have no Pi tools by default. Their temporary changes are
|
|
@@ -223,25 +251,39 @@ remains for controlled text-task runs from a task JSON file.
|
|
|
223
251
|
candidate sandbox and passed back to Pi as `@` file inputs. This covers PDFs,
|
|
224
252
|
CSVs, spreadsheets, and other files that Pi can open; OpenMerit does not
|
|
225
253
|
implement a separate parser for them.
|
|
226
|
-
- `ledger.json` counts reported candidate
|
|
227
|
-
|
|
228
|
-
|
|
254
|
+
- `ledger.json` counts reported candidate, rubric, judge, and strategist spend
|
|
255
|
+
for automatic session comparisons plus the daily candidate count.
|
|
256
|
+
`max_usd_per_trial` is a conservative admission estimate based on known Pi
|
|
257
|
+
prices and a 4K answer; it is not a provider-side hard cap. Routes without
|
|
258
|
+
known pricing are excluded from automatic comparisons and cannot auto-apply.
|
|
229
259
|
- Each comparison has one observed baseline plus a small candidate slate and
|
|
230
260
|
one quality score per answer. Treat recommendations as experimental evidence,
|
|
231
261
|
not a universal model ranking.
|
|
232
|
-
- Candidate execution and
|
|
233
|
-
|
|
234
|
-
|
|
262
|
+
- Candidate execution, judging, and strategy use the provider/model routes
|
|
263
|
+
exposed by Pi. OpenRouter remains an optional route plus catalog/benchmark
|
|
264
|
+
enrichment source. `HarnessAdapter`, `ModelProviderAdapter`,
|
|
265
|
+
`ObservationSource`, and `EventSink` remain separate integration boundaries;
|
|
266
|
+
Pi and local JSONL are the implementations shipped in this release.
|
|
267
|
+
- Candidate subprocesses can use Pi built-ins and custom/local routes available
|
|
268
|
+
without loading extensions (for example routes from Pi's model
|
|
269
|
+
configuration). A provider registered only at runtime by another extension
|
|
270
|
+
is visible in the route snapshot but cannot yet be executed by the isolated
|
|
271
|
+
subprocess.
|
|
272
|
+
- JSON state snapshots are replaced atomically. Append-only readers skip and
|
|
273
|
+
report malformed or interrupted lines while retaining later valid records.
|
|
274
|
+
A job owned by a crashed process is reclaimable instead of remaining stuck
|
|
275
|
+
in `running`; completed jobs remain final.
|
|
235
276
|
|
|
236
277
|
## State layout (`~/.openmerit/`)
|
|
237
278
|
|
|
238
279
|
| file | contents |
|
|
239
280
|
|---|---|
|
|
240
281
|
| `policy.json` | the policy file (gate thresholds, budgets, intervals) |
|
|
241
|
-
| `harness-state.json` |
|
|
242
|
-
| `recommendations.jsonl` |
|
|
243
|
-
| `trials.jsonl` | every model trial point
|
|
244
|
-
| `observations.jsonl` | task observations extracted from session traces |
|
|
282
|
+
| `harness-state.json` | current/fallback routes, Pi's eligible route snapshot, latest session and settled task |
|
|
283
|
+
| `recommendations.jsonl` | append-only session-bound recommendations, routes, evidence, gate reasons, and status updates |
|
|
284
|
+
| `trials.jsonl` | every model trial point, including its provider route when known |
|
|
285
|
+
| `traces/observations.jsonl` | task observations extracted from session traces |
|
|
286
|
+
| `events.jsonl` | versioned provider-neutral observation, trial, and recommendation events for future sinks |
|
|
245
287
|
| `traces/trials/*.jsonl` | raw Pi JSON event streams for candidate trials |
|
|
246
288
|
| `catalog/snapshot.json` + `candidates.json` | catalog snapshot (including input modalities) + new-model queue |
|
|
247
289
|
| `benchmarks/digest.json` | public-benchmark scores per model (seed + refresh) |
|
package/dist/cli.js
CHANGED
|
@@ -1,34 +1,18 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
/** openmerit CLI: init / watch / trial / frontier / recommend / status. */
|
|
3
|
-
import { copyFileSync, existsSync, mkdirSync
|
|
2
|
+
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor / verify. */
|
|
3
|
+
import { copyFileSync, existsSync, mkdirSync } from "node:fs";
|
|
4
4
|
import { dirname, join } from "node:path";
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
6
6
|
import { loadPolicy } from "./policy.js";
|
|
7
7
|
import { loadKey } from "./llm.js";
|
|
8
|
-
import { fetchCatalog } from "./catalog.js";
|
|
9
8
|
import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
|
|
10
|
-
import {
|
|
11
|
-
import { appendJsonl, paths, readJson, readJsonl, taskKey } from "./store.js";
|
|
9
|
+
import { paths, readJson, readJsonl } from "./store.js";
|
|
12
10
|
import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
|
|
13
|
-
import {
|
|
14
|
-
import {
|
|
15
|
-
import {
|
|
16
|
-
import {
|
|
17
|
-
import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
|
|
11
|
+
import { settledActiveTask } from "./pi-trials.js";
|
|
12
|
+
import { modelRoute, routeLabel } from "./routes.js";
|
|
13
|
+
import { runStandaloneTrial } from "./standalone.js";
|
|
14
|
+
import { collectDoctorReport, renderDiagnosticReport, runOfflineVerify } from "./diagnostics.js";
|
|
18
15
|
const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
19
|
-
function resolvePref(cat, prefs, label) {
|
|
20
|
-
for (const p of prefs)
|
|
21
|
-
if (p && cat.has(p))
|
|
22
|
-
return p;
|
|
23
|
-
for (const p of prefs) {
|
|
24
|
-
if (!p)
|
|
25
|
-
continue;
|
|
26
|
-
for (const id of [...cat.keys()].sort())
|
|
27
|
-
if (id.includes(p))
|
|
28
|
-
return id;
|
|
29
|
-
}
|
|
30
|
-
throw new Error(`could not resolve ${label} model`);
|
|
31
|
-
}
|
|
32
16
|
function cmdInit() {
|
|
33
17
|
const policyPath = paths.policy();
|
|
34
18
|
mkdirSync(dirname(policyPath), { recursive: true });
|
|
@@ -43,99 +27,27 @@ function cmdInit() {
|
|
|
43
27
|
console.log(`ext. -> ${join(ROOT, "extension", "openmerit.ts")}`);
|
|
44
28
|
console.log(`pi npm -> pi install npm:openmerit`);
|
|
45
29
|
console.log(`pi local-> pi install "${ROOT}"`);
|
|
46
|
-
console.log("\nNext:
|
|
30
|
+
console.log("\nNext: authenticate at least two models in pi, install one extension source, then start pi. OpenRouter is optional enrichment; candidate tools are disabled by default.");
|
|
47
31
|
}
|
|
48
|
-
|
|
49
|
-
const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
|
|
50
|
-
const policy = loadPolicy(paths.policy());
|
|
51
|
-
const key = loadKey();
|
|
52
|
-
console.log("fetching OpenRouter catalog...");
|
|
53
|
-
const cat = await fetchCatalog(key);
|
|
54
|
-
const piModels = availablePiModels();
|
|
55
|
-
for (const id of [...cat.keys()])
|
|
56
|
-
if (!piModels.has(id))
|
|
57
|
-
cat.delete(id);
|
|
58
|
-
if (!cat.has(cfg.initial_model))
|
|
59
|
-
throw new Error(`initial_model ${cfg.initial_model} is unavailable in pi's OpenRouter registry`);
|
|
60
|
-
console.log(`catalog: ${cat.size} models selectable in pi`);
|
|
61
|
-
const judge = resolvePref(cat, [cfg.judge_model ?? policy.judge_model, ...JUDGE_PREFS], "judge");
|
|
62
|
-
const strat = resolvePref(cat, [cfg.strategist_model ?? policy.strategist_model, ...STRAT_PREFS], "strategist");
|
|
63
|
-
const { category, benchmarks } = relevantBenchmarks(cfg.task);
|
|
64
|
-
let benchmarkCandidates = [];
|
|
32
|
+
function optionalOpenRouterKey() {
|
|
65
33
|
try {
|
|
66
|
-
|
|
67
|
-
}
|
|
68
|
-
catch (e) {
|
|
69
|
-
console.log(`OpenRouter benchmark shortlist unavailable: ${e.message}`);
|
|
70
|
-
}
|
|
71
|
-
console.log(`judge: ${judge} | strategist: ${strat} | benchmarks: ${benchmarks.join(", ")}`);
|
|
72
|
-
console.log(`OpenRouter benchmark candidates: ${benchmarkCandidates.slice(0, 5).join(", ") || "none"}`);
|
|
73
|
-
const tKey = taskKey(cfg.task);
|
|
74
|
-
const maxPrice = cfg.max_usd_per_m ?? policy.max_usd_per_m;
|
|
75
|
-
const tried = new Set();
|
|
76
|
-
const results = [];
|
|
77
|
-
const failedVendors = new Set();
|
|
78
|
-
for (let i = 0; i < rounds; i++) {
|
|
79
|
-
const budget = budgetOk(policy);
|
|
80
|
-
if (!budget.ok) {
|
|
81
|
-
console.log(`budget: ${budget.reason}; stopping`);
|
|
82
|
-
break;
|
|
83
|
-
}
|
|
84
|
-
let model;
|
|
85
|
-
let why;
|
|
86
|
-
if (i === 0) {
|
|
87
|
-
model = cfg.initial_model;
|
|
88
|
-
why = "initial model";
|
|
89
|
-
}
|
|
90
|
-
else {
|
|
91
|
-
const pick = await pickNext(key, strat, cfg.task, cfg.eval, benchmarks, results, cat, tried, maxPrice, failedVendors, benchmarkCandidates);
|
|
92
|
-
if (!pick)
|
|
93
|
-
break;
|
|
94
|
-
model = pick.model;
|
|
95
|
-
why = pick.why;
|
|
96
|
-
}
|
|
97
|
-
if (tried.has(model))
|
|
98
|
-
break;
|
|
99
|
-
console.log(`[round ${i + 1}/${rounds}] ${model} (${why}) ...`);
|
|
100
|
-
const observed = i === 0 ? recordedActiveTask(cfg.task, model) : null;
|
|
101
|
-
if (i === 0)
|
|
102
|
-
console.log(observed ? " using model A's active pi session trace" : " active trace unavailable; running model A in a fresh pi session");
|
|
103
|
-
const { point, costUsd, error, sessionId } = observed
|
|
104
|
-
? await scorePiRun(key, judge, cfg.task, cfg.eval, model, cat.get(model), observed, "trace")
|
|
105
|
-
: await runPiTrial(key, judge, cfg.task, cfg.eval, model, cat.get(model));
|
|
106
|
-
tried.add(model);
|
|
107
|
-
results.push(point);
|
|
108
|
-
recordTrialSpend(costUsd);
|
|
109
|
-
appendJsonl(paths.trials(), { ...point, taskKey: tKey });
|
|
110
|
-
if (sessionId)
|
|
111
|
-
console.log(` pi trace session: ${sessionId}`);
|
|
112
|
-
if (error) {
|
|
113
|
-
failedVendors.add(model.split("/")[0]);
|
|
114
|
-
console.log(` FAILED: ${error.slice(0, 160)}`);
|
|
115
|
-
}
|
|
116
|
-
else {
|
|
117
|
-
console.log(` score=${point.score.toFixed(2)} $/M=${point.price} est_cost=$${costUsd.toFixed(4)} ${(point.why ?? "").slice(0, 110)}`);
|
|
118
|
-
}
|
|
34
|
+
return loadKey();
|
|
119
35
|
}
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
console.log("no trials completed");
|
|
123
|
-
process.exit(1);
|
|
124
|
-
}
|
|
125
|
-
const chain = paretoFrontier(pts);
|
|
126
|
-
const best = pickBest(pts);
|
|
127
|
-
const fb = pickFallback(pts, best, policy.fallback.min_score);
|
|
128
|
-
console.log("\n=== pareto frontier (nondominated: score up, price down, latency down) ===");
|
|
129
|
-
for (const p of chain)
|
|
130
|
-
console.log(` ${p.model.padEnd(44)} score=${p.score.toFixed(2)} $${p.price.toFixed(2)}/M`);
|
|
131
|
-
console.log(`\nBEST FIT : ${best.model} (score ${best.score.toFixed(2)}, $${best.price.toFixed(2)}/M)`);
|
|
132
|
-
console.log(fb ? `FALLBACK : ${fb.model} (score ${fb.score.toFixed(2)}, $${fb.price.toFixed(2)}/M)` : "FALLBACK : n/a");
|
|
133
|
-
const rec = buildRecommendation(tKey, cfg.task.slice(0, 120), cfg.initial_model, pts, policy);
|
|
134
|
-
if (rec) {
|
|
135
|
-
appendJsonl(paths.recommendations(), rec);
|
|
136
|
-
console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${rec.policy.autoApply})`);
|
|
36
|
+
catch {
|
|
37
|
+
return null;
|
|
137
38
|
}
|
|
138
39
|
}
|
|
40
|
+
function printReport(report, json) {
|
|
41
|
+
console.log(json ? JSON.stringify(report, null, 2) : renderDiagnosticReport(report));
|
|
42
|
+
if (!report.ok)
|
|
43
|
+
process.exitCode = 1;
|
|
44
|
+
}
|
|
45
|
+
function cmdDoctor(json) {
|
|
46
|
+
printReport(collectDoctorReport(), json);
|
|
47
|
+
}
|
|
48
|
+
function cmdVerify(json) {
|
|
49
|
+
printReport(runOfflineVerify(), json);
|
|
50
|
+
}
|
|
139
51
|
function loadFrontiers() {
|
|
140
52
|
const trials = readJsonl(paths.trials());
|
|
141
53
|
const observations = readJsonl(paths.observations());
|
|
@@ -189,7 +101,8 @@ function cmdStatus() {
|
|
|
189
101
|
latestById.set(r.id, r);
|
|
190
102
|
const recs = [...latestById.values()].filter((r) => r.status === "pending");
|
|
191
103
|
const frontiers = loadFrontiers();
|
|
192
|
-
console.log(`harness model : ${harness.
|
|
104
|
+
console.log(`harness model : ${harness.currentRoute ? routeLabel(harness.currentRoute) : harness.currentModel ?? "unknown"} ` +
|
|
105
|
+
`(as of ${harness.updatedAt ?? "n/a"})`);
|
|
193
106
|
console.log(`tasks tracked : ${frontiers.length}`);
|
|
194
107
|
console.log(`new models queued for trial: ${queue.newModels.length}${queue.newModels.length ? " — " + queue.newModels.slice(0, 5).map((m) => m.id).join(", ") + (queue.newModels.length > 5 ? "…" : "") : ""}`);
|
|
195
108
|
console.log(`pending recommendations: ${recs.length}`);
|
|
@@ -199,47 +112,59 @@ function cmdStatus() {
|
|
|
199
112
|
}
|
|
200
113
|
async function main() {
|
|
201
114
|
const [cmd, ...args] = process.argv.slice(2);
|
|
202
|
-
const policy = loadPolicy(paths.policy());
|
|
203
115
|
switch (cmd) {
|
|
204
116
|
case "init":
|
|
205
117
|
cmdInit();
|
|
206
118
|
break;
|
|
207
119
|
case "session-trial": {
|
|
208
|
-
const
|
|
120
|
+
const policy = loadPolicy(paths.policy());
|
|
121
|
+
const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
|
|
209
122
|
if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
|
|
210
123
|
throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
|
|
211
124
|
const sessionBytes = Number(bytesText);
|
|
212
125
|
if (!Number.isSafeInteger(sessionBytes) || sessionBytes <= 0)
|
|
213
126
|
throw new Error("invalid session byte limit");
|
|
214
|
-
const
|
|
127
|
+
const state = readJson(paths.harnessState(), {});
|
|
128
|
+
const currentRoute = provider && providerModelId
|
|
129
|
+
? state.routes?.find((route) => route.provider === provider && route.modelId === providerModelId) ??
|
|
130
|
+
modelRoute(provider, providerModelId)
|
|
131
|
+
: state.currentRoute;
|
|
132
|
+
const task = settledActiveTask({ sessionFile, settledAt, currentModel, currentRoute,
|
|
133
|
+
routes: state.routes, settledTaskKey, cwd, sessionBytes });
|
|
215
134
|
if (!task)
|
|
216
135
|
throw new Error("the specified pi session has no completed, supported task matching this marker");
|
|
217
|
-
await autoTaskTick(
|
|
136
|
+
await autoTaskTick(optionalOpenRouterKey(), policy, task);
|
|
218
137
|
break;
|
|
219
138
|
}
|
|
220
139
|
case "watch":
|
|
221
140
|
if (args.includes("--once"))
|
|
222
|
-
await tickOnce(policy);
|
|
141
|
+
await tickOnce(loadPolicy(paths.policy()));
|
|
223
142
|
else
|
|
224
|
-
await runDaemon(policy);
|
|
143
|
+
await runDaemon(loadPolicy(paths.policy()));
|
|
225
144
|
break;
|
|
226
145
|
case "trial": {
|
|
227
146
|
const file = args[0];
|
|
228
147
|
if (!file)
|
|
229
148
|
throw new Error("usage: openmerit trial <task.json> [--rounds N]");
|
|
230
149
|
const rIdx = args.indexOf("--rounds");
|
|
231
|
-
await
|
|
150
|
+
await runStandaloneTrial(file, rIdx >= 0 ? Number(args[rIdx + 1]) : 5);
|
|
232
151
|
break;
|
|
233
152
|
}
|
|
234
153
|
case "frontier":
|
|
235
154
|
cmdFrontier(args[0]);
|
|
236
155
|
break;
|
|
237
156
|
case "recommend":
|
|
238
|
-
recommendTick(policy);
|
|
157
|
+
recommendTick(loadPolicy(paths.policy()));
|
|
239
158
|
break;
|
|
240
159
|
case "status":
|
|
241
160
|
cmdStatus();
|
|
242
161
|
break;
|
|
162
|
+
case "doctor":
|
|
163
|
+
cmdDoctor(args.includes("--json"));
|
|
164
|
+
break;
|
|
165
|
+
case "verify":
|
|
166
|
+
cmdVerify(args.includes("--json"));
|
|
167
|
+
break;
|
|
243
168
|
default:
|
|
244
169
|
console.log(`openmerit — external model-merit harness
|
|
245
170
|
|
|
@@ -250,7 +175,9 @@ usage: openmerit <command>
|
|
|
250
175
|
trial <task.json> iterative model search on one task spec [--rounds N]
|
|
251
176
|
frontier [taskKey] print pareto frontier(s)
|
|
252
177
|
recommend emit recommendations now
|
|
253
|
-
status harness model, queued models, pending recommendations
|
|
178
|
+
status harness model, queued models, pending recommendations
|
|
179
|
+
doctor [--json] diagnose pi, policy, routes, state, and installation
|
|
180
|
+
verify [--json] run an offline, provider-free core self-test`);
|
|
254
181
|
if (cmd && cmd !== "help" && cmd !== "--help")
|
|
255
182
|
process.exitCode = 1;
|
|
256
183
|
}
|