openmerit 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -46
- package/dist/cli.js +62 -8
- package/dist/daemon.js +104 -48
- package/dist/frontier.js +13 -6
- package/dist/integrations.js +19 -0
- package/dist/judge.js +5 -5
- package/dist/llm.js +102 -1
- package/dist/pi-trials.js +46 -23
- package/dist/policy.js +8 -3
- package/dist/recommend.js +27 -14
- package/dist/routes.js +59 -0
- package/dist/store.js +1 -0
- package/dist/strategist.js +18 -14
- package/dist/traces.js +6 -2
- package/dist/trials.js +38 -8
- package/extension/openmerit.ts +99 -23
- package/instructions/OPENMERIT.md +2 -2
- package/package.json +5 -2
- package/rules.md +39 -0
package/README.md
CHANGED
|
@@ -32,10 +32,10 @@ pi task A → saved pi trace → extension trial job → pi trials B, C → scor
|
|
|
32
32
|
([`src/frontier.ts`](src/frontier.ts), 3 objectives: quality ↑, price ↓,
|
|
33
33
|
latency ↓). The aggregate across all of an agent's tasks is the union of
|
|
34
34
|
its task frontiers (`openmerit frontier`).
|
|
35
|
-
4. **Model discovery** — the
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
35
|
+
4. **Model discovery** — the active Pi session's scoped models (or Pi's
|
|
36
|
+
authenticated available-model registry when unscoped) are the candidate
|
|
37
|
+
pool. OpenRouter is an optional route and enrichment source: its catalog can
|
|
38
|
+
be snapshotted and its benchmark signals can improve the shortlist
|
|
39
39
|
([`src/catalog.ts`](src/catalog.ts) + [`src/daemon.ts`](src/daemon.ts)).
|
|
40
40
|
5. **Informing the main harness** — recommendations land in
|
|
41
41
|
`~/.openmerit/recommendations.jsonl`. The pi extension
|
|
@@ -47,8 +47,9 @@ pi task A → saved pi trace → extension trial job → pi trials B, C → scor
|
|
|
47
47
|
|
|
48
48
|
## Install from npm
|
|
49
49
|
|
|
50
|
-
You need Node 22.18+, pi 0.85.1+, and
|
|
51
|
-
|
|
50
|
+
You need Node 22.18+, pi 0.85.1+, and at least two eligible model routes
|
|
51
|
+
authenticated in Pi. OpenRouter is supported but not required. Install the
|
|
52
|
+
package into pi, then initialize its policy and state:
|
|
52
53
|
|
|
53
54
|
```bash
|
|
54
55
|
pi install npm:openmerit
|
|
@@ -59,7 +60,7 @@ pi list
|
|
|
59
60
|
`pi install` makes the extension and its trial engine available to pi. The
|
|
60
61
|
one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
|
|
61
62
|
globally with `npm install --global openmerit` only if you also want persistent
|
|
62
|
-
shell access to `openmerit status`, `frontier`, or the optional watcher. Install
|
|
63
|
+
shell access to `openmerit status`, `doctor`, `frontier`, or the optional watcher. Install
|
|
63
64
|
the extension from only one source—remove any older copied `openmerit.ts` first
|
|
64
65
|
so pi does not load it twice.
|
|
65
66
|
|
|
@@ -85,40 +86,44 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
85
86
|
policy**. The shipped policy uses `"mode": "recommend"` and does not change
|
|
86
87
|
the active model without approval.
|
|
87
88
|
|
|
88
|
-
2.
|
|
89
|
-
|
|
90
|
-
|
|
89
|
+
2. Authenticate the providers you want to compare in Pi, using `/login` or
|
|
90
|
+
Pi's normal environment/model configuration. OpenMerit takes its eligible
|
|
91
|
+
model routes, capabilities, prices, and credentials from Pi; judge and
|
|
92
|
+
strategist calls use the same routes and do not require duplicate keys.
|
|
93
|
+
|
|
94
|
+
For example, verify one provider without printing its credential:
|
|
91
95
|
|
|
92
96
|
```bash
|
|
93
|
-
|
|
94
|
-
chmod 600 ~/.openmerit/.env
|
|
97
|
+
pi auth check --provider openai --json
|
|
95
98
|
```
|
|
96
99
|
|
|
97
|
-
|
|
98
|
-
`/login openrouter` inside pi, or export `OPENROUTER_API_KEY` in the shell
|
|
99
|
-
that starts pi. Verify pi's side without printing the key:
|
|
100
|
+
Add an OpenRouter route the same way if you want its routed catalog:
|
|
100
101
|
|
|
101
102
|
```bash
|
|
102
103
|
pi auth check --provider openrouter --json
|
|
103
104
|
```
|
|
104
105
|
|
|
105
|
-
|
|
106
|
-
|
|
106
|
+
An `OPENROUTER_API_KEY` in the process environment or
|
|
107
|
+
`~/.openmerit/.env` additionally enables OpenRouter catalog and public-
|
|
108
|
+
benchmark enrichment. It is optional for native-provider comparisons. If
|
|
109
|
+
using the file, protect it:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
nano ~/.openmerit/.env
|
|
113
|
+
chmod 600 ~/.openmerit/.env
|
|
114
|
+
```
|
|
107
115
|
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
`OPENMERIT_DIRECT_PROVIDER=google` plus the corresponding provider key;
|
|
112
|
-
OpenRouter model IDs remain on OpenRouter by default. The harness and
|
|
113
|
-
provider interfaces are extension seams for future integrations, not a
|
|
114
|
-
claim that other harnesses already work.
|
|
116
|
+
Pi does not read OpenMerit's `.env` file for its ordinary sessions, so an
|
|
117
|
+
OpenRouter route still needs Pi authentication. Older `model_search/.env`
|
|
118
|
+
files are not read.
|
|
115
119
|
|
|
116
120
|
3. Verify the extension appears in `pi list`. The instruction file
|
|
117
121
|
[`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
|
|
118
122
|
pi project's AGENTS.md for agent context, but the extension does
|
|
119
|
-
not require it.
|
|
123
|
+
not require it. Run `npx openmerit doctor` after starting Pi once to inspect
|
|
124
|
+
the eligible route snapshot and configuration without printing secrets.
|
|
120
125
|
|
|
121
|
-
4. Start pi in one terminal with
|
|
126
|
+
4. Start pi in one terminal with any configured model. For example:
|
|
122
127
|
|
|
123
128
|
```bash
|
|
124
129
|
pi --provider openrouter --model openai/gpt-4o-mini
|
|
@@ -127,7 +132,8 @@ Keep that checkout available because pi loads a local package from its path.
|
|
|
127
132
|
For a first text task, ask: “Give the shortest valid word ladder from cat
|
|
128
133
|
to dog. Each step changes one letter and must be a common English word.
|
|
129
134
|
Return only the path.” Wait for A to finish and leave the pi session open.
|
|
130
|
-
The extension automatically starts B and C **sequentially through
|
|
135
|
+
The extension automatically starts B and C **sequentially through their
|
|
136
|
+
selected Pi routes**,
|
|
131
137
|
reports each score in Pi, and writes a recommendation for this exact
|
|
132
138
|
session. With the shipped supervised policy, use
|
|
133
139
|
`/openmerit` inside pi to inspect the evidence and `/openmerit apply` to
|
|
@@ -173,7 +179,8 @@ schema, so its published scores are not directly comparable to these pi runs.
|
|
|
173
179
|
|
|
174
180
|
### Inspect or troubleshoot a run
|
|
175
181
|
|
|
176
|
-
`npx openmerit
|
|
182
|
+
`npx openmerit doctor` checks Pi, policy, eligible routes, and optional
|
|
183
|
+
OpenRouter enrichment. `npx openmerit status` shows the latest pi model and pending recommendations;
|
|
177
184
|
`npx openmerit frontier` shows measured quality, blended price, latency, and
|
|
178
185
|
the chosen frontier per task. `/openmerit` inside pi shows the current model,
|
|
179
186
|
fallback, the model currently being compared, completed models with quality,
|
|
@@ -189,23 +196,26 @@ candidate-spend threshold.
|
|
|
189
196
|
|
|
190
197
|
If no comparison starts after Pi settles, confirm that pi loaded the extension
|
|
191
198
|
(`pi list`), the task finished, and the pi session is saved (do not use
|
|
192
|
-
`--no-session`). Image candidates must advertise
|
|
193
|
-
|
|
199
|
+
`--no-session`). Image candidates must advertise image input in Pi's model
|
|
200
|
+
registry. The extension queues
|
|
194
201
|
completed tasks from its current session and runs one comparison at a time.
|
|
195
202
|
Closing or switching the session cancels the active job.
|
|
196
203
|
Candidate runs use the same text and uploaded image bytes, but they do not
|
|
197
204
|
replay earlier answers or file changes. An exact task in another session
|
|
198
205
|
(including identical image bytes) appears as **advice**, not a pending swap;
|
|
199
206
|
the new session still gets its own comparison. The optional `trial` command
|
|
200
|
-
remains for controlled text-task runs from a task JSON file
|
|
207
|
+
remains for controlled text-task runs from a task JSON file; that legacy
|
|
208
|
+
standalone command is still OpenRouter-specific in 0.1.2. The automatic Pi
|
|
209
|
+
session path is provider-neutral.
|
|
201
210
|
|
|
202
211
|
## Alpha boundaries
|
|
203
212
|
|
|
204
|
-
- With the extension installed and
|
|
205
|
-
settled task can start comparison calls automatically. Candidate,
|
|
206
|
-
strategist requests send the task text, attached files or images,
|
|
207
|
-
candidate output to
|
|
208
|
-
tasks and conservative account limits while evaluating
|
|
213
|
+
- With the extension installed and at least two eligible Pi routes, each
|
|
214
|
+
supported settled task can start comparison calls automatically. Candidate,
|
|
215
|
+
judge, and strategist requests send the task text, attached files or images,
|
|
216
|
+
and candidate output to the configured providers and can incur charges. Use
|
|
217
|
+
non-sensitive test tasks and conservative account limits while evaluating
|
|
218
|
+
this alpha.
|
|
209
219
|
- The extension queues tasks from its current pi session and compares one at a
|
|
210
220
|
time. Candidate runs use isolated temporary copies of the original working
|
|
211
221
|
directory and have no Pi tools by default. Their temporary changes are
|
|
@@ -223,25 +233,35 @@ remains for controlled text-task runs from a task JSON file.
|
|
|
223
233
|
candidate sandbox and passed back to Pi as `@` file inputs. This covers PDFs,
|
|
224
234
|
CSVs, spreadsheets, and other files that Pi can open; OpenMerit does not
|
|
225
235
|
implement a separate parser for them.
|
|
226
|
-
- `ledger.json` counts reported candidate
|
|
227
|
-
|
|
228
|
-
|
|
236
|
+
- `ledger.json` counts reported candidate, rubric, judge, and strategist spend
|
|
237
|
+
for automatic session comparisons plus the daily candidate count.
|
|
238
|
+
`max_usd_per_trial` is a conservative admission estimate based on known Pi
|
|
239
|
+
prices and a 4K answer; it is not a provider-side hard cap. Routes without
|
|
240
|
+
known pricing are excluded from automatic comparisons and cannot auto-apply.
|
|
229
241
|
- Each comparison has one observed baseline plus a small candidate slate and
|
|
230
242
|
one quality score per answer. Treat recommendations as experimental evidence,
|
|
231
243
|
not a universal model ranking.
|
|
232
|
-
- Candidate execution and
|
|
233
|
-
|
|
234
|
-
|
|
244
|
+
- Candidate execution, judging, and strategy use the provider/model routes
|
|
245
|
+
exposed by Pi. OpenRouter remains an optional route plus catalog/benchmark
|
|
246
|
+
enrichment source. `HarnessAdapter`, `ModelProviderAdapter`,
|
|
247
|
+
`ObservationSource`, and `EventSink` remain separate integration boundaries;
|
|
248
|
+
Pi and local JSONL are the implementations shipped in this release.
|
|
249
|
+
- Candidate subprocesses can use Pi built-ins and custom/local routes available
|
|
250
|
+
without loading extensions (for example routes from Pi's model
|
|
251
|
+
configuration). A provider registered only at runtime by another extension
|
|
252
|
+
is visible in the route snapshot but cannot yet be executed by the isolated
|
|
253
|
+
subprocess.
|
|
235
254
|
|
|
236
255
|
## State layout (`~/.openmerit/`)
|
|
237
256
|
|
|
238
257
|
| file | contents |
|
|
239
258
|
|---|---|
|
|
240
259
|
| `policy.json` | the policy file (gate thresholds, budgets, intervals) |
|
|
241
|
-
| `harness-state.json` |
|
|
242
|
-
| `recommendations.jsonl` |
|
|
243
|
-
| `trials.jsonl` | every model trial point
|
|
244
|
-
| `observations.jsonl` | task observations extracted from session traces |
|
|
260
|
+
| `harness-state.json` | current/fallback routes, Pi's eligible route snapshot, latest session and settled task |
|
|
261
|
+
| `recommendations.jsonl` | append-only session-bound recommendations, routes, evidence, gate reasons, and status updates |
|
|
262
|
+
| `trials.jsonl` | every model trial point, including its provider route when known |
|
|
263
|
+
| `traces/observations.jsonl` | task observations extracted from session traces |
|
|
264
|
+
| `events.jsonl` | versioned provider-neutral observation, trial, and recommendation events for future sinks |
|
|
245
265
|
| `traces/trials/*.jsonl` | raw Pi JSON event streams for candidate trials |
|
|
246
266
|
| `catalog/snapshot.json` + `candidates.json` | catalog snapshot (including input modalities) + new-model queue |
|
|
247
267
|
| `benchmarks/digest.json` | public-benchmark scores per model (seed + refresh) |
|
package/dist/cli.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
/** openmerit CLI: init / watch / trial / frontier / recommend / status. */
|
|
2
|
+
/** openmerit CLI: init / watch / trial / frontier / recommend / status / doctor. */
|
|
3
3
|
import { copyFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
4
4
|
import { dirname, join } from "node:path";
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
@@ -11,10 +11,12 @@ import { buildRecommendation } from "./recommend.js";
|
|
|
11
11
|
import { appendJsonl, paths, readJson, readJsonl, taskKey } from "./store.js";
|
|
12
12
|
import { autoTaskTick, recommendTick, runDaemon, tickOnce } from "./daemon.js";
|
|
13
13
|
import { budgetOk, recordTrialSpend } from "./trials.js";
|
|
14
|
-
import { availablePiModels, recordedActiveTask, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
14
|
+
import { availablePiModels, availablePiRoutes, piExecutable, recordedActiveTask, runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
15
15
|
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
16
16
|
import { JUDGE_PREFS } from "./judge.js";
|
|
17
17
|
import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
|
|
18
|
+
import { modelRoute, routeLabel } from "./routes.js";
|
|
19
|
+
import { spawnSync } from "node:child_process";
|
|
18
20
|
const ROOT = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
19
21
|
function resolvePref(cat, prefs, label) {
|
|
20
22
|
for (const p of prefs)
|
|
@@ -43,7 +45,48 @@ function cmdInit() {
|
|
|
43
45
|
console.log(`ext. -> ${join(ROOT, "extension", "openmerit.ts")}`);
|
|
44
46
|
console.log(`pi npm -> pi install npm:openmerit`);
|
|
45
47
|
console.log(`pi local-> pi install "${ROOT}"`);
|
|
46
|
-
console.log("\nNext:
|
|
48
|
+
console.log("\nNext: authenticate at least two models in pi, install one extension source, then start pi. OpenRouter is optional enrichment; candidate tools are disabled by default.");
|
|
49
|
+
}
|
|
50
|
+
function optionalOpenRouterKey() {
|
|
51
|
+
try {
|
|
52
|
+
return loadKey();
|
|
53
|
+
}
|
|
54
|
+
catch {
|
|
55
|
+
return null;
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
function cmdDoctor() {
|
|
59
|
+
const executable = piExecutable();
|
|
60
|
+
const version = spawnSync(executable, ["--version"], { encoding: "utf8" });
|
|
61
|
+
const state = readJson(paths.harnessState(), {});
|
|
62
|
+
console.log(`pi executable : ${executable}`);
|
|
63
|
+
console.log(`pi detected : ${version.status === 0 ? `yes (${version.stdout.trim()})` : "no"}`);
|
|
64
|
+
try {
|
|
65
|
+
const policy = loadPolicy(paths.policy());
|
|
66
|
+
console.log(`policy : valid (v${policy.version}, ${policy.mode})`);
|
|
67
|
+
}
|
|
68
|
+
catch (error) {
|
|
69
|
+
console.log(`policy : invalid (${error.message})`);
|
|
70
|
+
}
|
|
71
|
+
let routes = state.routes ?? [];
|
|
72
|
+
const hasLiveSnapshot = routes.length > 0;
|
|
73
|
+
if (routes.length === 0 && version.status === 0) {
|
|
74
|
+
try {
|
|
75
|
+
routes = availablePiRoutes();
|
|
76
|
+
}
|
|
77
|
+
catch { /* reported below */ }
|
|
78
|
+
}
|
|
79
|
+
console.log(`eligible routes: ${routes.length}`);
|
|
80
|
+
for (const route of routes.slice(0, 8))
|
|
81
|
+
console.log(` - ${routeLabel(route)}${route.cost ? "" : " (price unknown)"}`);
|
|
82
|
+
if (routes.length > 8)
|
|
83
|
+
console.log(` … ${routes.length - 8} more`);
|
|
84
|
+
console.log(`active route : ${state.currentRoute ? routeLabel(state.currentRoute) : "not reported; start pi once"}`);
|
|
85
|
+
const pricedRoutes = routes.filter((route) => !!route.cost).length;
|
|
86
|
+
console.log(`alternates : ${hasLiveSnapshot
|
|
87
|
+
? pricedRoutes >= 2 ? "ready" : "need at least two priced eligible pi routes"
|
|
88
|
+
: "start Pi once to capture route prices and capabilities"}`);
|
|
89
|
+
console.log(`OpenRouter enrichment: ${optionalOpenRouterKey() ? "available" : "not configured (optional)"}`);
|
|
47
90
|
}
|
|
48
91
|
async function cmdTrial(taskFile, rounds) {
|
|
49
92
|
const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
|
|
@@ -189,7 +232,8 @@ function cmdStatus() {
|
|
|
189
232
|
latestById.set(r.id, r);
|
|
190
233
|
const recs = [...latestById.values()].filter((r) => r.status === "pending");
|
|
191
234
|
const frontiers = loadFrontiers();
|
|
192
|
-
console.log(`harness model : ${harness.
|
|
235
|
+
console.log(`harness model : ${harness.currentRoute ? routeLabel(harness.currentRoute) : harness.currentModel ?? "unknown"} ` +
|
|
236
|
+
`(as of ${harness.updatedAt ?? "n/a"})`);
|
|
193
237
|
console.log(`tasks tracked : ${frontiers.length}`);
|
|
194
238
|
console.log(`new models queued for trial: ${queue.newModels.length}${queue.newModels.length ? " — " + queue.newModels.slice(0, 5).map((m) => m.id).join(", ") + (queue.newModels.length > 5 ? "…" : "") : ""}`);
|
|
195
239
|
console.log(`pending recommendations: ${recs.length}`);
|
|
@@ -205,16 +249,22 @@ async function main() {
|
|
|
205
249
|
cmdInit();
|
|
206
250
|
break;
|
|
207
251
|
case "session-trial": {
|
|
208
|
-
const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText] = args;
|
|
252
|
+
const [sessionFile, settledAt, currentModel, settledTaskKey, cwd, bytesText, provider, providerModelId] = args;
|
|
209
253
|
if (!sessionFile || !settledAt || !currentModel || !settledTaskKey || !cwd || !bytesText)
|
|
210
254
|
throw new Error("usage: openmerit session-trial <session-file> <settled-at> <model> <task-key> <cwd> <session-bytes>");
|
|
211
255
|
const sessionBytes = Number(bytesText);
|
|
212
256
|
if (!Number.isSafeInteger(sessionBytes) || sessionBytes <= 0)
|
|
213
257
|
throw new Error("invalid session byte limit");
|
|
214
|
-
const
|
|
258
|
+
const state = readJson(paths.harnessState(), {});
|
|
259
|
+
const currentRoute = provider && providerModelId
|
|
260
|
+
? state.routes?.find((route) => route.provider === provider && route.modelId === providerModelId) ??
|
|
261
|
+
modelRoute(provider, providerModelId)
|
|
262
|
+
: state.currentRoute;
|
|
263
|
+
const task = settledActiveTask({ sessionFile, settledAt, currentModel, currentRoute,
|
|
264
|
+
routes: state.routes, settledTaskKey, cwd, sessionBytes });
|
|
215
265
|
if (!task)
|
|
216
266
|
throw new Error("the specified pi session has no completed, supported task matching this marker");
|
|
217
|
-
await autoTaskTick(
|
|
267
|
+
await autoTaskTick(optionalOpenRouterKey(), policy, task);
|
|
218
268
|
break;
|
|
219
269
|
}
|
|
220
270
|
case "watch":
|
|
@@ -240,6 +290,9 @@ async function main() {
|
|
|
240
290
|
case "status":
|
|
241
291
|
cmdStatus();
|
|
242
292
|
break;
|
|
293
|
+
case "doctor":
|
|
294
|
+
cmdDoctor();
|
|
295
|
+
break;
|
|
243
296
|
default:
|
|
244
297
|
console.log(`openmerit — external model-merit harness
|
|
245
298
|
|
|
@@ -250,7 +303,8 @@ usage: openmerit <command>
|
|
|
250
303
|
trial <task.json> iterative model search on one task spec [--rounds N]
|
|
251
304
|
frontier [taskKey] print pareto frontier(s)
|
|
252
305
|
recommend emit recommendations now
|
|
253
|
-
status harness model, queued models, pending recommendations
|
|
306
|
+
status harness model, queued models, pending recommendations
|
|
307
|
+
doctor verify pi, policy, routes, and optional enrichment`);
|
|
254
308
|
if (cmd && cmd !== "help" && cmd !== "--help")
|
|
255
309
|
process.exitCode = 1;
|
|
256
310
|
}
|
package/dist/daemon.js
CHANGED
|
@@ -9,15 +9,46 @@ import { loadKey } from "./llm.js";
|
|
|
9
9
|
import { providerAllowed } from "./policy.js";
|
|
10
10
|
import { buildRecommendation } from "./recommend.js";
|
|
11
11
|
import { appendJsonl, paths, readJson, readJsonl, writeJson, } from "./store.js";
|
|
12
|
-
import {
|
|
13
|
-
import { budgetOk, ensureRubric, recordTrialSpend, runTrial } from "./trials.js";
|
|
12
|
+
import { budgetOk, ensureRubric, recordMeritSpend, recordTrialSpend, runTrial, trialBudgetOk } from "./trials.js";
|
|
14
13
|
import { JUDGE_PREFS } from "./judge.js";
|
|
15
|
-
import {
|
|
14
|
+
import { runPiTrial, scorePiRun, settledActiveTask } from "./pi-trials.js";
|
|
16
15
|
import { taskInputKey } from "./task-input.js";
|
|
17
16
|
import { pickNext, STRAT_PREFS } from "./strategist.js";
|
|
17
|
+
import { enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
|
|
18
|
+
import { PiCliChatClient } from "./llm.js";
|
|
19
|
+
import { LocalJsonlEventSink, PiTraceObservationSource, meritEvent } from "./integrations.js";
|
|
20
|
+
function resolveRoutePref(cat, prefs, label, requireImages = false) {
|
|
21
|
+
const available = [...cat.values()].filter((entry) => !!entry.route && entry.priceKnown !== false &&
|
|
22
|
+
(!requireImages || entry.inputModalities?.includes("image")));
|
|
23
|
+
for (const pref of prefs) {
|
|
24
|
+
if (!pref)
|
|
25
|
+
continue;
|
|
26
|
+
const routed = cat.get(pref);
|
|
27
|
+
if (routed && available.includes(routed))
|
|
28
|
+
return routed;
|
|
29
|
+
const exact = available.find((entry) => entry.id === pref);
|
|
30
|
+
if (exact)
|
|
31
|
+
return exact;
|
|
32
|
+
}
|
|
33
|
+
for (const pref of prefs) {
|
|
34
|
+
if (!pref)
|
|
35
|
+
continue;
|
|
36
|
+
const fuzzy = available.find((entry) => entry.id.includes(pref));
|
|
37
|
+
if (fuzzy)
|
|
38
|
+
return fuzzy;
|
|
39
|
+
}
|
|
40
|
+
const fallback = available.sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
|
|
41
|
+
if (fallback)
|
|
42
|
+
return fallback;
|
|
43
|
+
throw new Error(`could not resolve ${label} route from Pi's eligible models`);
|
|
44
|
+
}
|
|
18
45
|
function emitTrialProgress(progress) {
|
|
19
46
|
console.log(`[openmerit/progress] ${JSON.stringify(progress)}`);
|
|
20
47
|
}
|
|
48
|
+
function persistTrial(point, sink) {
|
|
49
|
+
appendJsonl(paths.trials(), point);
|
|
50
|
+
sink.emit(meritEvent("trial.completed", point));
|
|
51
|
+
}
|
|
21
52
|
function resolvePref(cat, prefs, label) {
|
|
22
53
|
for (const p of prefs) {
|
|
23
54
|
if (p && cat.has(p))
|
|
@@ -34,10 +65,12 @@ function resolvePref(cat, prefs, label) {
|
|
|
34
65
|
throw new Error(`could not resolve ${label} model from preferences`);
|
|
35
66
|
}
|
|
36
67
|
/** Ingest new session-trace bytes into the observation log. */
|
|
37
|
-
export function tracesTick() {
|
|
38
|
-
const fresh =
|
|
39
|
-
for (const o of fresh)
|
|
68
|
+
export function tracesTick(source = new PiTraceObservationSource(), sink = new LocalJsonlEventSink()) {
|
|
69
|
+
const fresh = source.read();
|
|
70
|
+
for (const o of fresh) {
|
|
40
71
|
appendJsonl(paths.observations(), o);
|
|
72
|
+
sink.emit(meritEvent("observation.recorded", o, source.id));
|
|
73
|
+
}
|
|
41
74
|
if (fresh.length > 0)
|
|
42
75
|
console.log(`[openmerit] traces: ${fresh.length} new observation(s)`);
|
|
43
76
|
return fresh;
|
|
@@ -68,6 +101,7 @@ export async function catalogTick(key, policy) {
|
|
|
68
101
|
* thin). Budget-capped per policy.
|
|
69
102
|
*/
|
|
70
103
|
export async function trialTick(key, policy) {
|
|
104
|
+
const sink = new LocalJsonlEventSink();
|
|
71
105
|
const budget = budgetOk(policy);
|
|
72
106
|
if (!budget.ok) {
|
|
73
107
|
console.log(`[openmerit] trials paused: ${budget.reason}`);
|
|
@@ -106,7 +140,7 @@ export async function trialTick(key, policy) {
|
|
|
106
140
|
const rubric = await ensureRubric(key, judge, obs.taskLabel, rubricCache);
|
|
107
141
|
const { point, costUsd, error } = await runTrial(key, judge, obs.taskLabel, rubric, candidate.id, cat.get(candidate.id));
|
|
108
142
|
recordTrialSpend(costUsd);
|
|
109
|
-
|
|
143
|
+
persistTrial({ ...point, taskKey: tKey }, sink);
|
|
110
144
|
if (error)
|
|
111
145
|
console.log(`[openmerit] trial failed: ${error.slice(0, 120)}`);
|
|
112
146
|
else
|
|
@@ -115,6 +149,7 @@ export async function trialTick(key, policy) {
|
|
|
115
149
|
}
|
|
116
150
|
/** Compare the one completed active pi task, sequentially, then target its session. */
|
|
117
151
|
export async function autoTaskTick(key, policy, settledTask) {
|
|
152
|
+
const sink = new LocalJsonlEventSink();
|
|
118
153
|
const task = settledTask === undefined ? settledActiveTask() : settledTask;
|
|
119
154
|
if (!task)
|
|
120
155
|
return null;
|
|
@@ -130,74 +165,94 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
130
165
|
// Claim the settled turn before provider calls so the next poll cannot duplicate it.
|
|
131
166
|
writeJson(processedPath, { marker, status: "running", updatedAt: new Date().toISOString() });
|
|
132
167
|
try {
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
168
|
+
let cat = routeCatalog(task.routes);
|
|
169
|
+
if (!cat.has(routeKey(task.route)))
|
|
170
|
+
cat.set(routeKey(task.route), routeCatalog([task.route]).values().next().value);
|
|
171
|
+
if (key) {
|
|
172
|
+
try {
|
|
173
|
+
const live = await fetchCatalog(key);
|
|
174
|
+
saveSnapshot(live);
|
|
175
|
+
cat = new Map([...cat].map(([id, entry]) => [id,
|
|
176
|
+
entry.route?.provider === "openrouter" ? enrichRouteEntry(entry, live.get(entry.id)) : entry]));
|
|
177
|
+
}
|
|
178
|
+
catch (error) {
|
|
179
|
+
console.log(`[openmerit] OpenRouter enrichment unavailable: ${error.message}`);
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
for (const [id, entry] of [...cat]) {
|
|
183
|
+
if ((entry.priceKnown !== false && entry.price > policy.max_usd_per_m) ||
|
|
184
|
+
!providerAllowed(policy, entry.id, entry.route?.provider) ||
|
|
185
|
+
(task.images.length > 0 && !entry.inputModalities?.includes("image")) ||
|
|
186
|
+
(id !== routeKey(task.route) && !trialBudgetOk(policy, entry, task.task.length).ok))
|
|
139
187
|
cat.delete(id);
|
|
140
188
|
}
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
189
|
+
const activeEntry = cat.get(routeKey(task.route));
|
|
190
|
+
if (!activeEntry)
|
|
191
|
+
throw new Error(`active route ${task.route.provider}/${task.route.modelId} is excluded by policy or task capabilities`);
|
|
192
|
+
if (cat.size < 2)
|
|
193
|
+
throw new Error("Pi exposes no eligible alternate model route for this task");
|
|
194
|
+
const judge = resolveRoutePref(cat, [policy.judge_model, ...JUDGE_PREFS, task.model], "judge", task.images.length > 0);
|
|
195
|
+
const strategist = resolveRoutePref(cat, [policy.strategist_model, ...STRAT_PREFS, task.model], "strategist");
|
|
196
|
+
const client = new PiCliChatClient(undefined, recordMeritSpend);
|
|
145
197
|
const { category, benchmarks } = relevantBenchmarks(task.task);
|
|
146
198
|
let benchmarkCandidates = [];
|
|
147
199
|
try {
|
|
148
|
-
|
|
200
|
+
if (key)
|
|
201
|
+
benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, new Map([...cat.values()].map((entry) => [entry.id, entry])));
|
|
149
202
|
}
|
|
150
203
|
catch (e) {
|
|
151
204
|
console.log(`[openmerit] benchmark shortlist unavailable: ${e.message}`);
|
|
152
205
|
}
|
|
153
|
-
const rubric = await ensureRubric(key, judge, task.task, new Map());
|
|
154
|
-
const tKey = taskInputKey(task.task, task.images);
|
|
206
|
+
const rubric = await ensureRubric(key ?? "", judge.route, task.task, new Map(), client);
|
|
207
|
+
const tKey = taskInputKey(task.task, task.images, task.files);
|
|
155
208
|
const points = [];
|
|
156
209
|
const tried = new Set();
|
|
157
210
|
const failedVendors = new Set();
|
|
158
211
|
const total = Math.max(1, Math.floor(policy.watch.models_per_task));
|
|
159
212
|
if (settledTask !== undefined)
|
|
160
213
|
emitTrialProgress({ phase: "start", taskKey: tKey,
|
|
161
|
-
model: task.
|
|
162
|
-
const baseline = await scorePiRun(key, judge, task.task, rubric, task.model,
|
|
214
|
+
model: routeLabel(task.route), index: 1, total });
|
|
215
|
+
const baseline = await scorePiRun(key ?? "", judge.route, task.task, rubric, task.model, activeEntry, task.run, "trace", task.images, client);
|
|
163
216
|
points.push(baseline.point);
|
|
164
|
-
tried.add(task.
|
|
165
|
-
|
|
217
|
+
tried.add(routeKey(task.route));
|
|
218
|
+
persistTrial({ ...baseline.point, taskKey: tKey, sessionFile: task.sessionFile }, sink);
|
|
166
219
|
console.log(`[openmerit] task ${tKey}: A=${task.model} score=${baseline.point.score.toFixed(2)} from pi trace`);
|
|
167
220
|
if (settledTask !== undefined)
|
|
168
221
|
emitTrialProgress({ phase: "complete", taskKey: tKey,
|
|
169
|
-
model: task.
|
|
222
|
+
model: routeLabel(task.route), index: 1, total, score: baseline.point.score,
|
|
170
223
|
price: baseline.point.price, latencyMs: baseline.point.latencyMs,
|
|
171
224
|
costUsd: baseline.costUsd, toolCalls: baseline.point.toolCalls,
|
|
172
225
|
toolErrors: baseline.point.toolErrors, changedFiles: baseline.point.changedFiles });
|
|
173
226
|
for (let i = 1; i < total; i++) {
|
|
174
227
|
if (!budgetOk(policy).ok)
|
|
175
228
|
break;
|
|
176
|
-
const pick = await pickNext(key, strategist, task.task, rubric, benchmarks, points, cat, tried, policy.max_usd_per_m, failedVendors, benchmarkCandidates);
|
|
229
|
+
const pick = await pickNext(key ?? "", strategist.route, task.task, rubric, benchmarks, points, cat, tried, policy.max_usd_per_m, failedVendors, benchmarkCandidates, client);
|
|
177
230
|
if (!pick)
|
|
178
231
|
break;
|
|
179
232
|
if (settledTask !== undefined)
|
|
180
233
|
emitTrialProgress({ phase: "start", taskKey: tKey,
|
|
181
|
-
model: pick.model, index: i + 1, total });
|
|
182
|
-
const
|
|
183
|
-
|
|
234
|
+
model: pick.route ? routeLabel(pick.route) : pick.model, index: i + 1, total });
|
|
235
|
+
const entry = pick.route ? cat.get(routeKey(pick.route)) : cat.get(pick.model);
|
|
236
|
+
const { point, costUsd, error } = await runPiTrial(key ?? "", judge.route, task.task, rubric, pick.model, entry, task.cwd, task.images, task.files, undefined, client);
|
|
237
|
+
tried.add(pick.route ? routeKey(pick.route) : pick.model);
|
|
184
238
|
points.push(point);
|
|
185
239
|
recordTrialSpend(costUsd);
|
|
186
|
-
|
|
240
|
+
persistTrial({ ...point, taskKey: tKey, sessionFile: task.sessionFile }, sink);
|
|
187
241
|
if (error)
|
|
188
|
-
failedVendors.add(pick.model.split("/")[0]);
|
|
242
|
+
failedVendors.add(pick.route?.provider ?? pick.model.split("/")[0]);
|
|
189
243
|
console.log(`[openmerit] task ${tKey}: ${pick.model} score=${point.score.toFixed(2)} cost=$${costUsd.toFixed(4)} ` +
|
|
190
244
|
`tools=${point.toolCalls ?? 0} toolErrors=${point.toolErrors ?? 0} changedFiles=${point.changedFiles ?? 0}` +
|
|
191
245
|
`${error ? ` error=${error.slice(0, 80)}` : ""}`);
|
|
192
246
|
if (settledTask !== undefined)
|
|
193
247
|
emitTrialProgress({ phase: "complete", taskKey: tKey,
|
|
194
|
-
model: pick.model, index: i + 1, total, score: point.score, price: point.price,
|
|
248
|
+
model: pick.route ? routeLabel(pick.route) : pick.model, index: i + 1, total, score: point.score, price: point.price,
|
|
195
249
|
latencyMs: point.latencyMs, costUsd, error: error?.slice(0, 120),
|
|
196
250
|
toolCalls: point.toolCalls, toolErrors: point.toolErrors, changedFiles: point.changedFiles });
|
|
197
251
|
}
|
|
198
|
-
const rec = buildRecommendation(tKey, task.task.slice(0, 120), task.model, points, policy, task.sessionFile);
|
|
252
|
+
const rec = buildRecommendation(tKey, task.task.slice(0, 120), task.model, points, policy, task.sessionFile, task.route);
|
|
199
253
|
if (rec) {
|
|
200
254
|
appendJsonl(paths.recommendations(), rec);
|
|
255
|
+
sink.emit(meritEvent("recommendation.created", rec));
|
|
201
256
|
console.log(`[openmerit] task ${tKey}: selected ${rec.recommended.model}; auto=${rec.policy.autoApply}`);
|
|
202
257
|
}
|
|
203
258
|
else
|
|
@@ -214,8 +269,10 @@ export async function autoTaskTick(key, policy, settledTask) {
|
|
|
214
269
|
}
|
|
215
270
|
/** Rebuild frontiers from trials and emit recommendations for the harness's current model. */
|
|
216
271
|
export function recommendTick(policy) {
|
|
272
|
+
const sink = new LocalJsonlEventSink();
|
|
217
273
|
const harness = readJson(paths.harnessState(), {});
|
|
218
274
|
const currentModel = harness.currentModel ?? null;
|
|
275
|
+
const currentRoute = harness.currentRoute ?? null;
|
|
219
276
|
const trials = readJsonl(paths.trials());
|
|
220
277
|
const observations = readJsonl(paths.observations());
|
|
221
278
|
const labelByTask = new Map();
|
|
@@ -237,16 +294,17 @@ export function recommendTick(policy) {
|
|
|
237
294
|
latestById.set(r.id, r);
|
|
238
295
|
const existing = new Set([...latestById.values()]
|
|
239
296
|
.filter((r) => r.status === "pending" || r.status === "dismissed")
|
|
240
|
-
.map((r) => `${r.taskKey}:${r.recommended.model}`));
|
|
297
|
+
.map((r) => `${r.taskKey}:${r.recommended.route ? routeKey(r.recommended.route) : `openrouter:${r.recommended.model}`}`));
|
|
241
298
|
const emitted = [];
|
|
242
299
|
for (const [tKey, points] of byTask) {
|
|
243
|
-
const rec = buildRecommendation(tKey, labelByTask.get(tKey) ?? tKey, currentModel, points, policy);
|
|
300
|
+
const rec = buildRecommendation(tKey, labelByTask.get(tKey) ?? tKey, currentModel, points, policy, null, currentRoute);
|
|
244
301
|
if (!rec)
|
|
245
302
|
continue;
|
|
246
|
-
const dedupeKey = `${rec.taskKey}:${rec.recommended.model}`;
|
|
303
|
+
const dedupeKey = `${rec.taskKey}:${rec.recommended.route ? routeKey(rec.recommended.route) : `openrouter:${rec.recommended.model}`}`;
|
|
247
304
|
if (existing.has(dedupeKey))
|
|
248
305
|
continue;
|
|
249
306
|
appendJsonl(paths.recommendations(), rec);
|
|
307
|
+
sink.emit(meritEvent("recommendation.created", rec));
|
|
250
308
|
emitted.push(rec);
|
|
251
309
|
console.log(`[openmerit] recommendation: ${rec.recommended.model} <- ${currentModel ?? "unknown"} ` +
|
|
252
310
|
`(gain ${rec.evidence.scoreGain}, auto=${rec.policy.autoApply})`);
|
|
@@ -261,12 +319,12 @@ export async function tickOnce(policy) {
|
|
|
261
319
|
key = loadKey();
|
|
262
320
|
}
|
|
263
321
|
catch (e) {
|
|
264
|
-
console.log(`[openmerit] ${e.message};
|
|
322
|
+
console.log(`[openmerit] ${e.message}; OpenRouter enrichment disabled`);
|
|
265
323
|
}
|
|
266
324
|
if (key) {
|
|
267
325
|
await catalogTick(key, policy);
|
|
268
|
-
await autoTaskTick(key, policy);
|
|
269
326
|
}
|
|
327
|
+
await autoTaskTick(key, policy);
|
|
270
328
|
}
|
|
271
329
|
/** Run forever, honoring policy watch intervals. */
|
|
272
330
|
export async function runDaemon(policy) {
|
|
@@ -291,17 +349,15 @@ export async function runDaemon(policy) {
|
|
|
291
349
|
catch {
|
|
292
350
|
/* no key; skip provider ticks */
|
|
293
351
|
}
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
lastCatalog = now;
|
|
299
|
-
}
|
|
300
|
-
await autoTaskTick(key, policy);
|
|
301
|
-
}
|
|
302
|
-
catch (e) {
|
|
303
|
-
console.log(`[openmerit] provider tick failed: ${e.message}`);
|
|
352
|
+
try {
|
|
353
|
+
if (key && wantCatalog) {
|
|
354
|
+
await catalogTick(key, policy);
|
|
355
|
+
lastCatalog = now;
|
|
304
356
|
}
|
|
357
|
+
await autoTaskTick(key, policy);
|
|
358
|
+
}
|
|
359
|
+
catch (e) {
|
|
360
|
+
console.log(`[openmerit] provider tick failed: ${e.message}`);
|
|
305
361
|
}
|
|
306
362
|
}
|
|
307
363
|
await new Promise((r) => setTimeout(r, tracesMs));
|