openmerit 0.1.2 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,7 +17,9 @@
17
17
  */
18
18
 
19
19
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
20
- import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
20
+ import {
21
+ appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync,
22
+ } from "node:fs";
21
23
  import { homedir } from "node:os";
22
24
  import { basename, join } from "node:path";
23
25
  import { createHash } from "node:crypto";
@@ -35,7 +37,7 @@ const JOB_DIR = join(HOME, "watch", "jobs");
35
37
  const ENGINE_FILE = join(fileURLToPath(new URL("..", import.meta.url)), "dist", "cli.js");
36
38
  const PROGRESS_PREFIX = "[openmerit/progress] ";
37
39
 
38
- interface TrialProgress {
40
+ interface TrialRunProgress {
39
41
  phase: "start" | "complete";
40
42
  taskKey: string;
41
43
  model: string;
@@ -51,19 +53,35 @@ interface TrialProgress {
51
53
  changedFiles?: number;
52
54
  }
53
55
 
56
+ interface TrialSelectionProgress {
57
+ phase: "selection";
58
+ taskKey: string;
59
+ eligible: string[];
60
+ excluded: { model: string; reasons: string[] }[];
61
+ }
62
+
63
+ type TrialProgress = TrialRunProgress | TrialSelectionProgress;
64
+
54
65
  function parseTrialProgress(line: string): TrialProgress | null {
55
66
  if (!line.startsWith(PROGRESS_PREFIX)) return null;
56
67
  try {
57
68
  const value = JSON.parse(line.slice(PROGRESS_PREFIX.length)) as TrialProgress;
58
- if ((value.phase !== "start" && value.phase !== "complete") ||
59
- typeof value.taskKey !== "string" || typeof value.model !== "string" ||
69
+ if (typeof value.taskKey !== "string") return null;
70
+ if (value.phase === "selection") {
71
+ if (!Array.isArray(value.eligible) || !Array.isArray(value.excluded) ||
72
+ value.eligible.some((item) => typeof item !== "string") ||
73
+ value.excluded.some((item) => typeof item?.model !== "string" || !Array.isArray(item.reasons) ||
74
+ item.reasons.some((reason) => typeof reason !== "string"))) return null;
75
+ return value;
76
+ }
77
+ if ((value.phase !== "start" && value.phase !== "complete") || typeof value.model !== "string" ||
60
78
  !Number.isInteger(value.index) || !Number.isInteger(value.total) ||
61
79
  value.index < 1 || value.total < value.index) return null;
62
80
  return value;
63
81
  } catch { return null; }
64
82
  }
65
83
 
66
- function completedTrialText(trial: TrialProgress): string {
84
+ function completedTrialText(trial: TrialRunProgress): string {
67
85
  return `${trial.model}: quality ${(trial.score ?? 0).toFixed(2)}, ` +
68
86
  `$${(trial.price ?? 0).toFixed(2)}/M, ${Math.round(trial.latencyMs ?? 0)}ms, ` +
69
87
  `run $${(trial.costUsd ?? 0).toFixed(4)}, tools ${trial.toolCalls ?? 0}` +
@@ -75,6 +93,12 @@ interface ExtensionPolicy {
75
93
  mode?: string;
76
94
  fallback?: { apply_on_error?: boolean };
77
95
  budgets?: { max_trials_per_day?: number; max_usd_per_day?: number };
96
+ route_overrides?: Record<string, {
97
+ cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
98
+ context_window?: number;
99
+ max_tokens?: number;
100
+ input?: ("text" | "image")[];
101
+ }>;
78
102
  }
79
103
 
80
104
  function trialBudget(): { summary: string; reason: string | null } {
@@ -84,7 +108,12 @@ function trialBudget(): { summary: string; reason: string | null } {
84
108
  const today = new Date().toISOString().slice(0, 10);
85
109
  let ledger: { date?: string; trials?: number; usd?: number } = {};
86
110
  try { ledger = JSON.parse(readFileSync(LEDGER_FILE, "utf8")) as typeof ledger; }
87
- catch { /* no trial spend yet */ }
111
+ catch {
112
+ if (existsSync(LEDGER_FILE)) return {
113
+ summary: "ledger unreadable",
114
+ reason: "trial ledger is malformed; run `openmerit doctor` before spending",
115
+ };
116
+ }
88
117
  const trials = ledger.date === today ? ledger.trials ?? 0 : 0;
89
118
  const usd = ledger.date === today ? ledger.usd ?? 0 : 0;
90
119
  const reason = trials >= limit ? `daily trial cap reached (${limit})`
@@ -107,10 +136,12 @@ function latestSessionJob(sessionFile: string | undefined): string | null {
107
136
  }
108
137
 
109
138
  function loadPolicy(): ExtensionPolicy {
139
+ if (!existsSync(POLICY_FILE)) return {};
110
140
  try {
111
141
  return JSON.parse(readFileSync(POLICY_FILE, "utf8")) as ExtensionPolicy;
112
142
  } catch {
113
- return {};
143
+ return { mode: "recommend", fallback: { apply_on_error: false },
144
+ budgets: { max_trials_per_day: 0, max_usd_per_day: 0 } };
114
145
  }
115
146
  }
116
147
 
@@ -173,7 +204,17 @@ function appendJsonl(file: string, obj: unknown): void {
173
204
  function latestRecommendations(): Recommendation[] {
174
205
  const byId = new Map<string, Recommendation>();
175
206
  for (const r of readJsonl<Recommendation>(RECS_FILE)) byId.set(r.id, r);
176
- return [...byId.values()];
207
+ let malformed = false;
208
+ if (existsSync(RECS_FILE)) {
209
+ for (const line of readFileSync(RECS_FILE, "utf8").split("\n")) {
210
+ if (!line.trim()) continue;
211
+ try { JSON.parse(line); } catch { malformed = true; break; }
212
+ }
213
+ }
214
+ return [...byId.values()].map((rec) => malformed && rec.status === "pending" && rec.policy.autoApply
215
+ ? { ...rec, policy: { autoApply: false,
216
+ reasons: [...rec.policy.reasons, "recommendation history contains a malformed line; manual review required"] } }
217
+ : rec);
177
218
  }
178
219
 
179
220
  function taskKey(text: string): string {
@@ -218,6 +259,7 @@ function latestSessionTaskKey(ctx: ExtensionContext): string | null {
218
259
  }
219
260
 
220
261
  let settledTaskKey: string | null = null;
262
+ let comparisonsPaused = false;
221
263
 
222
264
  function pendingRecommendations(ctx?: ExtensionContext): Recommendation[] {
223
265
  const pending = latestRecommendations().filter((r) => r.status === "pending");
@@ -276,8 +318,8 @@ let fallbackRoute: ModelRoute | null = null;
276
318
 
277
319
  function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
278
320
  cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
279
- contextWindow?: number; maxTokens?: number }): ModelRoute {
280
- return {
321
+ contextWindow?: number; maxTokens?: number }, overrides = loadPolicy().route_overrides ?? {}): ModelRoute {
322
+ const route: ModelRoute = {
281
323
  provider: model.provider,
282
324
  modelId: model.id,
283
325
  meritId: canonicalModel(model.provider, model.id),
@@ -289,6 +331,33 @@ function routeFor(model: { provider: string; id: string; input?: ("text" | "imag
289
331
  contextWindow: model.contextWindow,
290
332
  maxTokens: model.maxTokens,
291
333
  };
334
+ const override = overrides[`${route.provider}:${route.modelId}`];
335
+ if (!override) return route;
336
+ const finite = (value: unknown, min: number) =>
337
+ typeof value === "number" && Number.isFinite(value) && value >= min ? value : undefined;
338
+ const integer = (value: unknown) => {
339
+ const normalized = finite(value, 1);
340
+ return normalized !== undefined && Number.isInteger(normalized) ? normalized : undefined;
341
+ };
342
+ const input = Array.isArray(override.input) && override.input.length > 0 &&
343
+ override.input.every((item) => item === "text" || item === "image")
344
+ ? [...new Set(override.input)] : undefined;
345
+ const costInput = finite(override.cost?.input, 0);
346
+ const costOutput = finite(override.cost?.output, 0);
347
+ return {
348
+ ...route,
349
+ ...(costInput === undefined || costOutput === undefined ? {} : { cost: {
350
+ input: costInput,
351
+ output: costOutput,
352
+ cacheRead: finite(override.cost?.cacheRead, 0),
353
+ cacheWrite: finite(override.cost?.cacheWrite, 0),
354
+ }}),
355
+ ...(integer(override.context_window) === undefined ? {} : {
356
+ contextWindow: integer(override.context_window),
357
+ }),
358
+ ...(integer(override.max_tokens) === undefined ? {} : { maxTokens: integer(override.max_tokens) }),
359
+ ...(input ? { input } : {}),
360
+ };
292
361
  }
293
362
 
294
363
  function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
@@ -298,13 +367,24 @@ function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
298
367
  if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
299
368
  models.unshift(ctx.model);
300
369
  const byRoute = new Map<string, ModelRoute>();
370
+ const overrides = loadPolicy().route_overrides ?? {};
301
371
  for (const model of models) {
302
- const route = routeFor(model);
372
+ const route = routeFor(model, overrides);
303
373
  byRoute.set(`${route.provider}:${route.modelId}`, route);
304
374
  }
305
375
  return [...byRoute.values()];
306
376
  }
307
377
 
378
+ function writeState(value: unknown): void {
379
+ const temp = `${STATE_FILE}.tmp-${process.pid}-${Date.now()}`;
380
+ try {
381
+ writeFileSync(temp, JSON.stringify(value) + "\n");
382
+ renameSync(temp, STATE_FILE);
383
+ } finally {
384
+ if (existsSync(temp)) rmSync(temp, { force: true });
385
+ }
386
+ }
387
+
308
388
  function reportHarnessState(ctx: ExtensionContext, settled = false): void {
309
389
  try {
310
390
  mkdirSync(HOME, { recursive: true });
@@ -317,10 +397,8 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
317
397
  const sessionFile = ctx.sessionManager.getSessionFile() ?? null;
318
398
  const settledAt = settled ? new Date().toISOString()
319
399
  : prior.sessionFile === sessionFile ? prior.settledAt ?? null : null;
320
- writeFileSync(
321
- STATE_FILE,
322
- JSON.stringify({
323
- schemaVersion: 1,
400
+ writeState({
401
+ schemaVersion: 2,
324
402
  currentModel: model,
325
403
  currentRoute,
326
404
  routes: eligibleRoutes(ctx),
@@ -329,10 +407,10 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
329
407
  sessionFile,
330
408
  settledTaskKey,
331
409
  settledAt,
410
+ comparisonsPaused,
332
411
  cwd: ctx.cwd,
333
412
  updatedAt: new Date().toISOString(),
334
- }) + "\n",
335
- );
413
+ });
336
414
  } catch {
337
415
  /* never break the host session over reporting */
338
416
  }
@@ -439,6 +517,34 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
439
517
  return registry.find?.(INJECT_PROVIDER, openrouterId);
440
518
  }
441
519
 
520
+ async function doctorSummary(): Promise<{ text: string; ok: boolean }> {
521
+ if (!existsSync(ENGINE_FILE)) return {
522
+ text: "trial engine missing; reinstall OpenMerit or run npm ci in the checkout",
523
+ ok: false,
524
+ };
525
+ const child = spawn(process.execPath, [ENGINE_FILE, "doctor", "--json"], {
526
+ env: process.env, stdio: ["ignore", "pipe", "pipe"],
527
+ });
528
+ let stdout = "";
529
+ let stderr = "";
530
+ child.stdout.on("data", (chunk: Buffer) => { stdout += chunk.toString(); });
531
+ child.stderr.on("data", (chunk: Buffer) => { stderr += chunk.toString(); });
532
+ const code = await new Promise<number>((resolve, reject) => {
533
+ child.on("error", reject);
534
+ child.on("close", (value) => resolve(value ?? 1));
535
+ });
536
+ try {
537
+ const report = JSON.parse(stdout) as { ok?: boolean; checks?: {
538
+ id?: string; status?: string; message?: string;
539
+ }[] };
540
+ const lines = (report.checks ?? []).map((item) =>
541
+ `${(item.status ?? "?").toUpperCase().padEnd(4)} ${item.id ?? "check"}: ${item.message ?? ""}`);
542
+ return { text: lines.join("\n") || "doctor returned no checks", ok: report.ok === true };
543
+ } catch {
544
+ return { text: stderr.trim().slice(0, 500) || stdout.trim().slice(0, 500) || `doctor exited ${code}`, ok: false };
545
+ }
546
+ }
547
+
442
548
  export default function openmerit(pi: ExtensionAPI) {
443
549
  let pollTimer: ReturnType<typeof setInterval> | null = null;
444
550
  let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
@@ -446,8 +552,8 @@ export default function openmerit(pi: ExtensionAPI) {
446
552
  key: string; cwd: string; bytes: number }[] = [];
447
553
  const startedJobs = new Set<string>();
448
554
  let activeSession: string | null = null;
449
- let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
450
- completed: TrialProgress[] } | null = null;
555
+ let progress: { sessionFile: string; taskKey: string; current: TrialRunProgress | null;
556
+ completed: TrialRunProgress[]; eligible: string[]; excluded: TrialSelectionProgress["excluded"] } | null = null;
451
557
 
452
558
  function stopTrialJob(): void {
453
559
  if (!trialJob) return;
@@ -475,14 +581,20 @@ export default function openmerit(pi: ExtensionAPI) {
475
581
  detached: process.platform !== "win32",
476
582
  });
477
583
  trialJob = child;
478
- progress = { sessionFile: job.sessionFile, taskKey: job.key, current: null, completed: [] };
584
+ progress = { sessionFile: job.sessionFile, taskKey: job.key, current: null, completed: [],
585
+ eligible: [], excluded: [] };
479
586
  if (ctx.hasUI) ctx.ui.notify(`openmerit: comparing models for task ${job.key} (A=${job.model})`, "info");
480
587
  let stdout = "";
481
588
  let stderr = "";
482
589
  function onLine(line: string): void {
483
590
  const trial = parseTrialProgress(line);
484
591
  if (trial && job.sessionFile === activeSession && progress?.taskKey === trial.taskKey) {
485
- if (trial.phase === "start") {
592
+ if (trial.phase === "selection") {
593
+ progress.eligible = trial.eligible;
594
+ progress.excluded = trial.excluded;
595
+ if (ctx.hasUI) ctx.ui.notify(`openmerit: ${trial.eligible.length} routes eligible; ` +
596
+ `${trial.excluded.length} skipped (/openmerit for reasons)`, "info");
597
+ } else if (trial.phase === "start") {
486
598
  progress.current = trial;
487
599
  if (ctx.hasUI) ctx.ui.notify(
488
600
  `openmerit: trial ${trial.index}/${trial.total} comparing ${trial.model}` +
@@ -525,11 +637,13 @@ export default function openmerit(pi: ExtensionAPI) {
525
637
  });
526
638
  }
527
639
 
528
- function queueSettledTask(ctx: ExtensionContext): void {
640
+ function queueSettledTask(ctx: ExtensionContext, force = false): string {
641
+ if (comparisonsPaused && !force) return "comparisons are paused";
529
642
  const sessionFile = ctx.sessionManager.getSessionFile();
530
643
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
531
644
  const route = ctx.model ? routeFor(ctx.model) : null;
532
- if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
645
+ if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile))
646
+ return "no saved completed task is available";
533
647
  // The alpha compares completed response tasks. The engine validates images,
534
648
  // tool use and the baseline answer before any provider calls.
535
649
  let completed = false;
@@ -542,19 +656,20 @@ export default function openmerit(pi: ExtensionAPI) {
542
656
  userTaskKeyFromAnswer(entry.message.content)) completed = true;
543
657
  } catch { /* partial session line */ }
544
658
  }
545
- if (!completed) return;
659
+ if (!completed) return "the latest task has not completed successfully";
546
660
  const settledAt = new Date().toISOString();
547
661
  const bytes = readFileSync(sessionFile).length;
548
662
  const marker = `${sessionFile}:${settledTaskKey}:${model}:${bytes}`;
549
- if (startedJobs.has(marker)) return;
663
+ if (startedJobs.has(marker)) return "this task is already queued or was compared in this Pi session";
550
664
  startedJobs.add(marker);
551
665
  const budget = trialBudget();
552
666
  if (budget.reason) {
553
667
  if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
554
- return;
668
+ return budget.reason;
555
669
  }
556
670
  queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
557
671
  startNextJob(ctx);
672
+ return "comparison queued";
558
673
  }
559
674
 
560
675
  function userTaskKeyFromAnswer(content: unknown): boolean {
@@ -641,11 +756,12 @@ export default function openmerit(pi: ExtensionAPI) {
641
756
  restoreFallback();
642
757
  try {
643
758
  const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
644
- sessionFile?: string | null; settledTaskKey?: string | null;
759
+ sessionFile?: string | null; settledTaskKey?: string | null; comparisonsPaused?: boolean;
645
760
  };
761
+ comparisonsPaused = st.comparisonsPaused === true;
646
762
  settledTaskKey = st.sessionFile === ctx.sessionManager.getSessionFile()
647
763
  ? st.settledTaskKey ?? null : null;
648
- } catch { settledTaskKey = null; }
764
+ } catch { settledTaskKey = null; comparisonsPaused = false; }
649
765
  reportHarnessState(ctx);
650
766
  await surface(ctx);
651
767
  if (pollTimer) clearInterval(pollTimer);
@@ -722,7 +838,7 @@ export default function openmerit(pi: ExtensionAPI) {
722
838
  });
723
839
 
724
840
  pi.registerCommand("openmerit", {
725
- description: "OpenMerit: status, pending recommendations, apply/dismiss",
841
+ description: "OpenMerit: status, compare, pause/resume, doctor, apply/dismiss",
726
842
  handler: async (args, ctx) => {
727
843
  const pending = pendingRecommendations(ctx);
728
844
  const prior = priorExactTaskSuggestion(ctx);
@@ -756,6 +872,37 @@ export default function openmerit(pi: ExtensionAPI) {
756
872
  }
757
873
  return;
758
874
  }
875
+ if (sub === "pause") {
876
+ comparisonsPaused = true;
877
+ queuedJobs.length = 0;
878
+ stopTrialJob();
879
+ startedJobs.clear();
880
+ progress = null;
881
+ reportHarnessState(ctx);
882
+ ctx.ui.notify("openmerit: automatic comparisons paused; `/openmerit compare` remains available", "info");
883
+ return;
884
+ }
885
+ if (sub === "resume") {
886
+ comparisonsPaused = false;
887
+ reportHarnessState(ctx);
888
+ ctx.ui.notify("openmerit: automatic comparisons resumed for future completed tasks", "info");
889
+ return;
890
+ }
891
+ if (sub === "compare") {
892
+ settledTaskKey = latestSessionTaskKey(ctx);
893
+ const result = queueSettledTask(ctx, true);
894
+ if (result !== "comparison queued") ctx.ui.notify(`openmerit: ${result}`, "warning");
895
+ return;
896
+ }
897
+ if (sub === "doctor") {
898
+ try {
899
+ const result = await doctorSummary();
900
+ ctx.ui.notify(`openmerit doctor:\n${result.text}`, result.ok ? "info" : "warning");
901
+ } catch (error) {
902
+ ctx.ui.notify(`openmerit doctor failed: ${(error as Error).message}`, "error");
903
+ }
904
+ return;
905
+ }
759
906
 
760
907
  const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
761
908
  const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
@@ -765,6 +912,7 @@ export default function openmerit(pi: ExtensionAPI) {
765
912
  ? `running — ${liveProgress.current.model} (trial ${liveProgress.current.index}/${liveProgress.current.total})`
766
913
  : "running — selecting next model"
767
914
  : queuedJobs.length ? `queued (${queuedJobs.length})`
915
+ : comparisonsPaused ? "paused"
768
916
  : latestSessionJob(ctx.sessionManager.getSessionFile() ?? undefined) ?? "not started";
769
917
  const lines = [
770
918
  `current model : ${current}`,
@@ -777,6 +925,11 @@ export default function openmerit(pi: ExtensionAPI) {
777
925
  lines.push("", "models tested:");
778
926
  for (const trial of liveProgress.completed) lines.push(` ${completedTrialText(trial)}`);
779
927
  }
928
+ if (liveProgress?.excluded.length) {
929
+ lines.push("", "routes skipped:");
930
+ for (const route of liveProgress.excluded)
931
+ lines.push(` ${route.model}: ${route.reasons.join("; ")}`);
932
+ }
780
933
  if (prior) lines.push("", suggestionText(prior.recommendation, prior.trials));
781
934
  for (const r of pending.slice(-5)) {
782
935
  lines.push(
@@ -39,6 +39,12 @@ the best model for the task at hand, at the best price, with a vetted fallback.
39
39
  recommendations.
40
40
  - `/openmerit apply` / `/openmerit dismiss` — act on a pending recommendation
41
41
  when automatic application is declined by the policy gate.
42
+ - `/openmerit pause` / `/openmerit resume` — stop or restart automatic task
43
+ comparisons without changing the saved policy.
44
+ - `/openmerit compare` — explicitly compare the latest completed task, even
45
+ while automatic comparisons are paused.
46
+ - `/openmerit doctor` — inspect sanitized installation, route, pricing, state,
47
+ and provider-extension diagnostics inside Pi.
42
48
  - Policy changes (auto-apply thresholds, budgets, provider allow-lists) are
43
49
  made by the user in `~/.openmerit/policy.json`, not by you.
44
50
 
@@ -52,3 +58,6 @@ the best model for the task at hand, at the best price, with a vetted fallback.
52
58
  - The shipped policy is supervised. Swaps happen automatically only after the
53
59
  user opts in and the configured quality/cost guardrails pass; otherwise they
54
60
  remain recommendations for a human to approve.
61
+ - Isolated subprocesses disable extension discovery. Provider-registration
62
+ extensions run only when the user lists their absolute file paths in
63
+ `policy.json`; OpenMerit refuses to load itself recursively.
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": 1,
2
+ "version": 2,
3
3
  "mode": "recommend",
4
4
  "auto_apply": {
5
5
  "enabled": false,
@@ -29,5 +29,9 @@
29
29
  },
30
30
  "judge_model": null,
31
31
  "strategist_model": null,
32
- "max_usd_per_m": 20.0
32
+ "max_usd_per_m": 20.0,
33
+ "route_overrides": {},
34
+ "pi": {
35
+ "provider_extensions": []
36
+ }
33
37
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmerit",
3
- "version": "0.1.2",
3
+ "version": "0.1.4",
4
4
  "description": "Find better models for each pi task by comparing quality, cost, and latency.",
5
5
  "type": "module",
6
6
  "keywords": [
package/rules.md CHANGED
@@ -16,6 +16,8 @@
16
16
 
17
17
  - **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
18
18
 
19
+ - **Require explicit metadata instead of price guesses.** Missing price, context, output, or modality data for a custom route may be supplied by an exact-route override; never assume a local route is free or copy metadata across providers.
20
+
19
21
  - **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
20
22
 
21
23
  - **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
@@ -30,6 +32,8 @@
30
32
 
31
33
  - **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
32
34
 
35
+ - **Allowlist subprocess extensions.** Keep Pi extension discovery disabled in isolated trials and load only explicitly named provider-registration files; never recursively load OpenMerit or inherit unrelated harness extensions.
36
+
33
37
  - **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
34
38
 
35
39
  - **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.