openmerit 0.1.1 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/trials.js CHANGED
@@ -1,5 +1,5 @@
1
1
  /** Shadow trials: run candidate models against observed/declared tasks on the background track. */
2
- import { chat } from "./llm.js";
2
+ import { directChatClient } from "./llm.js";
3
3
  import { judge, parseObj } from "./judge.js";
4
4
  import { paths, readJson, writeJson } from "./store.js";
5
5
  function today() {
@@ -15,6 +15,21 @@ export function budgetOk(policy) {
15
15
  return { ok: false, reason: `daily budget exhausted ($${policy.budgets.max_usd_per_day})` };
16
16
  return { ok: true };
17
17
  }
18
+ /** Conservative admission estimate for one candidate answer (rough input tokens + 4K output). */
19
+ export function trialBudgetOk(policy, entry, inputChars) {
20
+ if (entry.priceKnown === false)
21
+ return { ok: false, reason: "route price is unknown" };
22
+ const inputTokens = Math.ceil(inputChars / 4);
23
+ const outputTokens = Math.min(entry.route?.maxTokens ?? 4096, 4096);
24
+ const estimatedUsd = entry.pp * inputTokens + entry.pc * outputTokens;
25
+ if (estimatedUsd > policy.budgets.max_usd_per_trial)
26
+ return {
27
+ ok: false,
28
+ estimatedUsd,
29
+ reason: `estimated trial cost $${estimatedUsd.toFixed(4)} exceeds $${policy.budgets.max_usd_per_trial.toFixed(4)} limit`,
30
+ };
31
+ return { ok: true, estimatedUsd };
32
+ }
18
33
  export function recordTrialSpend(usd) {
19
34
  let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
20
35
  if (ledger.date !== today())
@@ -23,16 +38,26 @@ export function recordTrialSpend(usd) {
23
38
  ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
24
39
  writeJson(paths.ledger(), ledger);
25
40
  }
41
+ /** Record non-candidate merit-loop spend without consuming a trial slot. */
42
+ export function recordMeritSpend(usd) {
43
+ if (!(usd > 0))
44
+ return;
45
+ let ledger = readJson(paths.ledger(), { date: today(), trials: 0, usd: 0 });
46
+ if (ledger.date !== today())
47
+ ledger = { date: today(), trials: 0, usd: 0 };
48
+ ledger.usd = Math.round((ledger.usd + usd) * 1e6) / 1e6;
49
+ writeJson(paths.ledger(), ledger);
50
+ }
26
51
  const RUBRIC_PROMPT = `Write a compact grading rubric for answers to the task below.
27
52
  It must list the concrete criteria for a score of 1.0 and how to deduct.
28
53
  TASK: {task}
29
54
  Return ONLY json: {{"rubric": "<text>"}}`;
30
55
  /** Derive (and cache) a grading rubric for a task that came from traces. */
31
- export async function ensureRubric(key, judgeModel, task, cache) {
56
+ export async function ensureRubric(key, judgeModel, task, cache, client = directChatClient(key)) {
32
57
  const hit = cache.get(task);
33
58
  if (hit)
34
59
  return hit;
35
- const { content } = await chat(key, judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
60
+ const { content } = await client.chat(judgeModel, RUBRIC_PROMPT.replace("{task}", task.slice(0, 8000)), 1024, 0);
36
61
  let rubric = "Score 1.0 for a fully correct, complete, usable answer; deduct for errors, omissions, or unusable output.";
37
62
  try {
38
63
  const obj = parseObj(content);
@@ -46,19 +71,20 @@ export async function ensureRubric(key, judgeModel, task, cache) {
46
71
  return rubric;
47
72
  }
48
73
  /** Run one shadow trial: generate with `model`, judge the output, return a frontier point + cost. */
49
- export async function runTrial(key, judgeModel, task, rubric, model, entry) {
74
+ export async function runTrial(key, judgeModel, task, rubric, model, entry, client = directChatClient(key)) {
50
75
  const started = Date.now();
51
76
  let gen;
52
77
  let usage = {};
53
78
  try {
54
- const r = await chat(key, model, task, 4096, 0.2);
79
+ const r = await client.chat(entry?.route ?? model, task, 4096, 0.2);
55
80
  gen = r.content;
56
81
  usage = r.usage;
57
82
  }
58
83
  catch (e) {
59
84
  return {
60
85
  point: {
61
- model, score: 0, price: entry?.price ?? 0,
86
+ schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
87
+ priceKnown: entry?.priceKnown ?? !!entry,
62
88
  ts: new Date().toISOString(), source: "shadow_trial", why: String(e).slice(0, 160),
63
89
  },
64
90
  costUsd: 0,
@@ -70,7 +96,8 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
70
96
  if (!gen.trim()) {
71
97
  return {
72
98
  point: {
73
- model, score: 0, price: entry?.price ?? 0, latencyMs,
99
+ schemaVersion: 1, model, route: entry?.route, score: 0, price: entry?.price ?? 0,
100
+ priceKnown: entry?.priceKnown ?? !!entry, latencyMs,
74
101
  ts: new Date().toISOString(), source: "shadow_trial", why: "empty response",
75
102
  },
76
103
  costUsd,
@@ -79,7 +106,7 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
79
106
  let score = 0;
80
107
  let why = "judge error";
81
108
  try {
82
- const j = await judge(key, judgeModel, task, rubric, gen);
109
+ const j = await judge(key, judgeModel, task, rubric, gen, client);
83
110
  score = j.score;
84
111
  why = j.why;
85
112
  }
@@ -88,9 +115,12 @@ export async function runTrial(key, judgeModel, task, rubric, model, entry) {
88
115
  }
89
116
  return {
90
117
  point: {
118
+ schemaVersion: 1,
91
119
  model,
120
+ route: entry?.route,
92
121
  score,
93
122
  price: entry?.price ?? 0,
123
+ priceKnown: entry?.priceKnown ?? !!entry,
94
124
  latencyMs,
95
125
  ts: new Date().toISOString(),
96
126
  source: "shadow_trial",
@@ -115,19 +115,31 @@ function loadPolicy(): ExtensionPolicy {
115
115
  }
116
116
 
117
117
  interface Recommendation {
118
+ schemaVersion?: 1;
118
119
  id: string;
119
120
  ts: string;
120
121
  taskKey: string;
121
122
  taskLabel: string;
122
123
  sessionFile?: string | null;
123
124
  currentModel: string | null;
124
- recommended: { model: string; reason: string };
125
- fallback: { model: string; reason: string } | null;
126
- evidence: { scoreGain: number; priceRatio: number; onFrontier: boolean; trials: number };
125
+ currentRoute?: ModelRoute | null;
126
+ recommended: { model: string; route?: ModelRoute; reason: string };
127
+ fallback: { model: string; route?: ModelRoute; reason: string } | null;
128
+ evidence: { scoreGain: number; priceRatio: number; priceKnown?: boolean; onFrontier: boolean; trials: number };
127
129
  policy: { autoApply: boolean; reasons: string[] };
128
130
  status: "pending" | "applied" | "dismissed" | "expired";
129
131
  }
130
132
 
133
+ interface ModelRoute {
134
+ provider: string;
135
+ modelId: string;
136
+ meritId: string;
137
+ input: ("text" | "image")[];
138
+ cost?: { input: number; output: number; cacheRead?: number; cacheWrite?: number };
139
+ contextWindow?: number;
140
+ maxTokens?: number;
141
+ }
142
+
131
143
  interface RecordedTrial {
132
144
  taskKey?: string;
133
145
  sessionFile?: string;
@@ -248,7 +260,11 @@ function suggestionText(rec: Recommendation, trials: RecordedTrial[]): string {
248
260
  }
249
261
 
250
262
  function canonicalModel(provider: string, id: string): string {
251
- return provider === "openrouter" ? id : `${provider}/${id}`;
263
+ return provider === "openrouter" || id.startsWith(`${provider}/`) ? id : `${provider}/${id}`;
264
+ }
265
+
266
+ function routeDisplay(route: ModelRoute | undefined, fallback: string): string {
267
+ return route && route.provider !== "openrouter" ? `${route.meritId} via ${route.provider}` : fallback;
252
268
  }
253
269
 
254
270
  function setStatus(rec: Recommendation, status: Recommendation["status"]): void {
@@ -256,11 +272,44 @@ function setStatus(rec: Recommendation, status: Recommendation["status"]): void
256
272
  }
257
273
 
258
274
  let fallbackModel: string | null = null;
275
+ let fallbackRoute: ModelRoute | null = null;
276
+
277
+ function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
278
+ cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
279
+ contextWindow?: number; maxTokens?: number }): ModelRoute {
280
+ return {
281
+ provider: model.provider,
282
+ modelId: model.id,
283
+ meritId: canonicalModel(model.provider, model.id),
284
+ input: model.input ?? ["text"],
285
+ cost: model.cost && typeof model.cost.input === "number" && typeof model.cost.output === "number"
286
+ ? { input: model.cost.input, output: model.cost.output,
287
+ cacheRead: model.cost.cacheRead, cacheWrite: model.cost.cacheWrite }
288
+ : undefined,
289
+ contextWindow: model.contextWindow,
290
+ maxTokens: model.maxTokens,
291
+ };
292
+ }
293
+
294
+ function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
295
+ const registry = ctx.modelRegistry as unknown as { getAvailable?: () => Array<Parameters<typeof routeFor>[0]> };
296
+ const scoped = (ctx.scopedModels ?? []) as readonly { model: Parameters<typeof routeFor>[0] }[];
297
+ const models = scoped.length ? scoped.map((item) => item.model) : registry.getAvailable?.() ?? [];
298
+ if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
299
+ models.unshift(ctx.model);
300
+ const byRoute = new Map<string, ModelRoute>();
301
+ for (const model of models) {
302
+ const route = routeFor(model);
303
+ byRoute.set(`${route.provider}:${route.modelId}`, route);
304
+ }
305
+ return [...byRoute.values()];
306
+ }
259
307
 
260
308
  function reportHarnessState(ctx: ExtensionContext, settled = false): void {
261
309
  try {
262
310
  mkdirSync(HOME, { recursive: true });
263
311
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
312
+ const currentRoute = ctx.model ? routeFor(ctx.model) : null;
264
313
  if (settled) settledTaskKey = latestSessionTaskKey(ctx);
265
314
  const prior = existsSync(STATE_FILE)
266
315
  ? JSON.parse(readFileSync(STATE_FILE, "utf8")) as { sessionFile?: string | null; settledAt?: string }
@@ -271,8 +320,12 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
271
320
  writeFileSync(
272
321
  STATE_FILE,
273
322
  JSON.stringify({
323
+ schemaVersion: 1,
274
324
  currentModel: model,
325
+ currentRoute,
326
+ routes: eligibleRoutes(ctx),
275
327
  fallbackModel,
328
+ fallbackRoute,
276
329
  sessionFile,
277
330
  settledTaskKey,
278
331
  settledAt,
@@ -288,8 +341,11 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
288
341
  /** Restore the persisted fallback so a fresh session keeps it even with no pending recs. */
289
342
  function restoreFallback(): void {
290
343
  try {
291
- const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as { fallbackModel?: string | null };
344
+ const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
345
+ fallbackModel?: string | null; fallbackRoute?: ModelRoute | null;
346
+ };
292
347
  if (st.fallbackModel) fallbackModel = st.fallbackModel;
348
+ if (st.fallbackRoute) fallbackRoute = st.fallbackRoute;
293
349
  } catch {
294
350
  /* no state yet */
295
351
  }
@@ -344,7 +400,15 @@ function catalogEntry(openrouterId: string): CatalogEntryLite | undefined {
344
400
  * OpenRouter catalog, so inject anything missing via registerProvider
345
401
  * (takes effect immediately, no /reload needed).
346
402
  */
347
- function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string) {
403
+ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string, route?: ModelRoute) {
404
+ const registry = ctx.modelRegistry as unknown as {
405
+ find?: (provider: string, id: string) => unknown | undefined;
406
+ };
407
+ if (route) {
408
+ const exact = registry.find?.(route.provider, route.modelId);
409
+ if (exact) return exact;
410
+ if (route.provider !== "openrouter") return undefined;
411
+ }
348
412
  const existing = resolveModel(ctx, openrouterId);
349
413
  if (existing) return existing;
350
414
 
@@ -372,16 +436,14 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
372
436
  })),
373
437
  });
374
438
 
375
- const reg = ctx.modelRegistry as unknown as {
376
- find?: (provider: string, id: string) => unknown | undefined;
377
- };
378
- return reg.find?.(INJECT_PROVIDER, openrouterId);
439
+ return registry.find?.(INJECT_PROVIDER, openrouterId);
379
440
  }
380
441
 
381
442
  export default function openmerit(pi: ExtensionAPI) {
382
443
  let pollTimer: ReturnType<typeof setInterval> | null = null;
383
444
  let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
384
- const queuedJobs: { sessionFile: string; settledAt: string; model: string; key: string; cwd: string; bytes: number }[] = [];
445
+ const queuedJobs: { sessionFile: string; settledAt: string; model: string; route: ModelRoute;
446
+ key: string; cwd: string; bytes: number }[] = [];
385
447
  const startedJobs = new Set<string>();
386
448
  let activeSession: string | null = null;
387
449
  let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
@@ -408,7 +470,7 @@ export default function openmerit(pi: ExtensionAPI) {
408
470
  return;
409
471
  }
410
472
  const child = spawn(process.execPath, [ENGINE_FILE, "session-trial", job.sessionFile,
411
- job.settledAt, job.model, job.key, job.cwd, String(job.bytes)], {
473
+ job.settledAt, job.model, job.key, job.cwd, String(job.bytes), job.route.provider, job.route.modelId], {
412
474
  cwd: job.cwd, env: process.env, stdio: ["ignore", "pipe", "pipe"],
413
475
  detached: process.platform !== "win32",
414
476
  });
@@ -466,7 +528,8 @@ export default function openmerit(pi: ExtensionAPI) {
466
528
  function queueSettledTask(ctx: ExtensionContext): void {
467
529
  const sessionFile = ctx.sessionManager.getSessionFile();
468
530
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
469
- if (!sessionFile || !model || !settledTaskKey || !existsSync(sessionFile)) return;
531
+ const route = ctx.model ? routeFor(ctx.model) : null;
532
+ if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
470
533
  // The alpha compares completed response tasks. The engine validates images,
471
534
  // tool use and the baseline answer before any provider calls.
472
535
  let completed = false;
@@ -490,7 +553,7 @@ export default function openmerit(pi: ExtensionAPI) {
490
553
  if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
491
554
  return;
492
555
  }
493
- queuedJobs.push({ sessionFile, settledAt, model, key: settledTaskKey, cwd: ctx.cwd, bytes });
556
+ queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
494
557
  startNextJob(ctx);
495
558
  }
496
559
 
@@ -502,11 +565,18 @@ export default function openmerit(pi: ExtensionAPI) {
502
565
 
503
566
  async function applyRecommendation(ctx: ExtensionContext, rec: Recommendation): Promise<boolean> {
504
567
  const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
505
- if (rec.currentModel && current !== rec.currentModel) {
568
+ const currentRoute = ctx.model ? routeFor(ctx.model) : null;
569
+ if (rec.currentRoute && currentRoute &&
570
+ (rec.currentRoute.provider !== currentRoute.provider || rec.currentRoute.modelId !== currentRoute.modelId)) {
571
+ ctx.ui.notify(`openmerit: recommendation targets ${rec.currentRoute.provider}/${rec.currentRoute.modelId}; ` +
572
+ `current route is ${currentRoute.provider}/${currentRoute.modelId}`, "warning");
573
+ return false;
574
+ }
575
+ if (!rec.currentRoute && rec.currentModel && current !== rec.currentModel) {
506
576
  ctx.ui.notify(`openmerit: recommendation was measured against ${rec.currentModel}; current model is ${current ?? "unknown"}`, "warning");
507
577
  return false;
508
578
  }
509
- const model = ensureModel(pi, ctx, rec.recommended.model);
579
+ const model = ensureModel(pi, ctx, rec.recommended.model, rec.recommended.route);
510
580
  if (!model) {
511
581
  ctx.ui.notify(
512
582
  `openmerit: ${rec.recommended.model} not found in pi's model registry (add the provider/model first)`,
@@ -517,8 +587,12 @@ export default function openmerit(pi: ExtensionAPI) {
517
587
  const ok = await pi.setModel(model as Parameters<typeof pi.setModel>[0]);
518
588
  if (ok) {
519
589
  setStatus(rec, "applied");
520
- if (rec.fallback) fallbackModel = rec.fallback.model;
521
- ctx.ui.notify(`openmerit: switched to ${rec.recommended.model}\n${rec.recommended.reason}`, "info");
590
+ if (rec.fallback) {
591
+ fallbackModel = rec.fallback.model;
592
+ fallbackRoute = rec.fallback.route ?? null;
593
+ }
594
+ ctx.ui.notify(`openmerit: switched to ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n` +
595
+ rec.recommended.reason, "info");
522
596
  reportHarnessState(ctx);
523
597
  } else {
524
598
  ctx.ui.notify(`openmerit: no auth for ${rec.recommended.model}`, "error");
@@ -550,7 +624,7 @@ export default function openmerit(pi: ExtensionAPI) {
550
624
  // startup. Notify (fire-and-forget) and let the user act via /openmerit.
551
625
  if (ctx.hasUI) {
552
626
  ctx.ui.notify(
553
- `openmerit recommends: ${rec.recommended.model}\n${rec.recommended.reason}\n` +
627
+ `openmerit recommends: ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n${rec.recommended.reason}\n` +
554
628
  `gate: ${rec.policy.reasons.join("; ")}\n(/openmerit apply to switch, /openmerit dismiss to ignore)`,
555
629
  "info",
556
630
  );
@@ -621,7 +695,7 @@ export default function openmerit(pi: ExtensionAPI) {
621
695
  const policy = loadPolicy();
622
696
  if (policy.fallback?.apply_on_error === false) return;
623
697
  const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "current model";
624
- const model = ensureModel(pi, ctx, fallbackModel);
698
+ const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
625
699
  if (!model) return;
626
700
 
627
701
  const auto = policy.mode === "auto";
@@ -673,7 +747,7 @@ export default function openmerit(pi: ExtensionAPI) {
673
747
  ctx.ui.notify("openmerit: no fallback set yet", "info");
674
748
  return;
675
749
  }
676
- const model = ensureModel(pi, ctx, fallbackModel);
750
+ const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
677
751
  if (model && (await pi.setModel(model as Parameters<typeof pi.setModel>[0]))) {
678
752
  ctx.ui.notify(`openmerit: switched to fallback ${fallbackModel}`, "info");
679
753
  reportHarnessState(ctx);
@@ -683,7 +757,8 @@ export default function openmerit(pi: ExtensionAPI) {
683
757
  return;
684
758
  }
685
759
 
686
- const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "unknown";
760
+ const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
761
+ const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
687
762
  const budget = trialBudget();
688
763
  const liveProgress = progress?.sessionFile === ctx.sessionManager.getSessionFile() ? progress : null;
689
764
  const comparison = trialJob ? liveProgress?.current
@@ -706,7 +781,8 @@ export default function openmerit(pi: ExtensionAPI) {
706
781
  for (const r of pending.slice(-5)) {
707
782
  lines.push(
708
783
  "",
709
- `-> ${r.recommended.model} (gain ${r.evidence.scoreGain}, ${r.evidence.priceRatio}x price, frontier=${r.evidence.onFrontier})`,
784
+ `-> ${routeDisplay(r.recommended.route, r.recommended.model)} (gain ${r.evidence.scoreGain}, ` +
785
+ `${r.evidence.priceKnown === false ? "price unknown" : `${r.evidence.priceRatio}x price`}, frontier=${r.evidence.onFrontier})`,
710
786
  ` ${r.recommended.reason}`,
711
787
  ` gate: ${r.policy.reasons.join("; ") || "n/a"}`,
712
788
  );
@@ -45,8 +45,8 @@ the best model for the task at hand, at the best price, with a vetted fallback.
45
45
  ## Guarantees
46
46
 
47
47
  - Per-task comparisons send the task text, uploaded files or images (when
48
- present), and candidate answers through pi/OpenRouter for model runs and
49
- judging. Candidate Pi runs default to no tools; any tool access is an
48
+ present), and candidate answers through the model routes configured in Pi
49
+ for model runs and judging. Candidate Pi runs default to no tools; any tool access is an
50
50
  explicit user opt-in. Use non-sensitive examples while
51
51
  evaluating this alpha.
52
52
  - The shipped policy is supervised. Swaps happen automatically only after the
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmerit",
3
- "version": "0.1.1",
3
+ "version": "0.1.2",
4
4
  "description": "Find better models for each pi task by comparing quality, cost, and latency.",
5
5
  "type": "module",
6
6
  "keywords": [
@@ -17,10 +17,13 @@
17
17
  "extension/openmerit.ts",
18
18
  "instructions/",
19
19
  "examples/",
20
+ "rules.md",
20
21
  "benchmark/invoice_ocr/data/"
21
22
  ],
22
23
  "pi": {
23
- "extensions": ["./extension/openmerit.ts"]
24
+ "extensions": [
25
+ "./extension/openmerit.ts"
26
+ ]
24
27
  },
25
28
  "bin": {
26
29
  "openmerit": "dist/cli.js"
package/rules.md ADDED
@@ -0,0 +1,39 @@
1
+ # OpenMerit product and architecture rules
2
+
3
+ - **Optimize for useful model choices.** OpenMerit exists to find better task-specific tradeoffs between quality, cost, and latency—not to become a general-purpose agent or telemetry platform.
4
+
5
+ - **Act as a control plane, not a gateway.** Normal traffic stays between the harness and provider; OpenMerit observes evidence, runs explicit trials, and returns recommendations without intercepting ordinary work.
6
+
7
+ - **Make Pi plug-and-play first.** Pi and Pi-based harnesses are the first supported user experience because that is where early users already work; broader harness support can follow without weakening this path.
8
+
9
+ - **Keep the core harness-neutral.** Pi is the first `HarnessAdapter`, not a permanent assumption in scoring, recommendations, policy, or stored data, so other harnesses can be added without rewriting the merit loop.
10
+
11
+ - **Keep model providers interchangeable.** OpenRouter is a first-class provider and discovery source, not the foundation of the domain model; provider-specific behavior belongs behind a `ModelProviderAdapter`.
12
+
13
+ - **Require end-to-end provider neutrality.** Discovery, authentication, execution, scoring, and swapping must preserve the selected route; a feature is not provider-neutral if only its API client is abstracted.
14
+
15
+ - **Separate model identity from execution route.** Keep logical model IDs in stable `vendor/model` form, while recording the actual provider or route separately, so the same model can be compared through OpenRouter, a native provider, or a local provider.
16
+
17
+ - **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
18
+
19
+ - **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
20
+
21
+ - **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
22
+
23
+ - **Use provider-neutral core records.** Normalize integrations into stable concepts such as `TaskObservation`, `CandidateRun`, `TrialScore`, `Frontier`, `Recommendation`, and `MeritEvent`, so integrations do not leak their schemas into the decision engine.
24
+
25
+ - **Optimize per task before aggregating per agent.** Agent-level conclusions must be built from measured task evidence rather than assumed from global model rankings.
26
+
27
+ - **Version persisted events and evolve them additively.** Existing 0.1.x state must remain readable, and append-only recommendation history must stay intact; migrations should normalize old records rather than invalidate them.
28
+
29
+ - **Keep policy as the sole auto-swap authority.** Trials and strategists may recommend changes, but only the policy gate may approve automatic application, and every recommendation must retain its gate reasons.
30
+
31
+ - **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
32
+
33
+ - **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
34
+
35
+ - **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.
36
+
37
+ - **Keep integrations optional and the core lightweight.** New providers, harnesses, and observability services should not impose credentials, network calls, or heavy dependencies on users who do not enable them.
38
+
39
+ - **Keep 0.1.x focused.** The near-term bar is reliable, provider-neutral use for Pi tinkerers and solo hackers; additional harnesses and hosted observability integrations belong in later releases unless required to prove the boundaries work.