openmerit 0.1.1 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,7 +17,9 @@
17
17
  */
18
18
 
19
19
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
20
- import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
20
+ import {
21
+ appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, writeFileSync,
22
+ } from "node:fs";
21
23
  import { homedir } from "node:os";
22
24
  import { basename, join } from "node:path";
23
25
  import { createHash } from "node:crypto";
@@ -84,7 +86,12 @@ function trialBudget(): { summary: string; reason: string | null } {
84
86
  const today = new Date().toISOString().slice(0, 10);
85
87
  let ledger: { date?: string; trials?: number; usd?: number } = {};
86
88
  try { ledger = JSON.parse(readFileSync(LEDGER_FILE, "utf8")) as typeof ledger; }
87
- catch { /* no trial spend yet */ }
89
+ catch {
90
+ if (existsSync(LEDGER_FILE)) return {
91
+ summary: "ledger unreadable",
92
+ reason: "trial ledger is malformed; run `openmerit doctor` before spending",
93
+ };
94
+ }
88
95
  const trials = ledger.date === today ? ledger.trials ?? 0 : 0;
89
96
  const usd = ledger.date === today ? ledger.usd ?? 0 : 0;
90
97
  const reason = trials >= limit ? `daily trial cap reached (${limit})`
@@ -107,27 +114,41 @@ function latestSessionJob(sessionFile: string | undefined): string | null {
107
114
  }
108
115
 
109
116
  function loadPolicy(): ExtensionPolicy {
117
+ if (!existsSync(POLICY_FILE)) return {};
110
118
  try {
111
119
  return JSON.parse(readFileSync(POLICY_FILE, "utf8")) as ExtensionPolicy;
112
120
  } catch {
113
- return {};
121
+ return { mode: "recommend", fallback: { apply_on_error: false },
122
+ budgets: { max_trials_per_day: 0, max_usd_per_day: 0 } };
114
123
  }
115
124
  }
116
125
 
117
126
  interface Recommendation {
127
+ schemaVersion?: 1;
118
128
  id: string;
119
129
  ts: string;
120
130
  taskKey: string;
121
131
  taskLabel: string;
122
132
  sessionFile?: string | null;
123
133
  currentModel: string | null;
124
- recommended: { model: string; reason: string };
125
- fallback: { model: string; reason: string } | null;
126
- evidence: { scoreGain: number; priceRatio: number; onFrontier: boolean; trials: number };
134
+ currentRoute?: ModelRoute | null;
135
+ recommended: { model: string; route?: ModelRoute; reason: string };
136
+ fallback: { model: string; route?: ModelRoute; reason: string } | null;
137
+ evidence: { scoreGain: number; priceRatio: number; priceKnown?: boolean; onFrontier: boolean; trials: number };
127
138
  policy: { autoApply: boolean; reasons: string[] };
128
139
  status: "pending" | "applied" | "dismissed" | "expired";
129
140
  }
130
141
 
142
+ interface ModelRoute {
143
+ provider: string;
144
+ modelId: string;
145
+ meritId: string;
146
+ input: ("text" | "image")[];
147
+ cost?: { input: number; output: number; cacheRead?: number; cacheWrite?: number };
148
+ contextWindow?: number;
149
+ maxTokens?: number;
150
+ }
151
+
131
152
  interface RecordedTrial {
132
153
  taskKey?: string;
133
154
  sessionFile?: string;
@@ -161,7 +182,17 @@ function appendJsonl(file: string, obj: unknown): void {
161
182
  function latestRecommendations(): Recommendation[] {
162
183
  const byId = new Map<string, Recommendation>();
163
184
  for (const r of readJsonl<Recommendation>(RECS_FILE)) byId.set(r.id, r);
164
- return [...byId.values()];
185
+ let malformed = false;
186
+ if (existsSync(RECS_FILE)) {
187
+ for (const line of readFileSync(RECS_FILE, "utf8").split("\n")) {
188
+ if (!line.trim()) continue;
189
+ try { JSON.parse(line); } catch { malformed = true; break; }
190
+ }
191
+ }
192
+ return [...byId.values()].map((rec) => malformed && rec.status === "pending" && rec.policy.autoApply
193
+ ? { ...rec, policy: { autoApply: false,
194
+ reasons: [...rec.policy.reasons, "recommendation history contains a malformed line; manual review required"] } }
195
+ : rec);
165
196
  }
166
197
 
167
198
  function taskKey(text: string): string {
@@ -248,7 +279,11 @@ function suggestionText(rec: Recommendation, trials: RecordedTrial[]): string {
248
279
  }
249
280
 
250
281
  function canonicalModel(provider: string, id: string): string {
251
- return provider === "openrouter" ? id : `${provider}/${id}`;
282
+ return provider === "openrouter" || id.startsWith(`${provider}/`) ? id : `${provider}/${id}`;
283
+ }
284
+
285
+ function routeDisplay(route: ModelRoute | undefined, fallback: string): string {
286
+ return route && route.provider !== "openrouter" ? `${route.meritId} via ${route.provider}` : fallback;
252
287
  }
253
288
 
254
289
  function setStatus(rec: Recommendation, status: Recommendation["status"]): void {
@@ -256,11 +291,54 @@ function setStatus(rec: Recommendation, status: Recommendation["status"]): void
256
291
  }
257
292
 
258
293
  let fallbackModel: string | null = null;
294
+ let fallbackRoute: ModelRoute | null = null;
295
+
296
+ function routeFor(model: { provider: string; id: string; input?: ("text" | "image")[];
297
+ cost?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number };
298
+ contextWindow?: number; maxTokens?: number }): ModelRoute {
299
+ return {
300
+ provider: model.provider,
301
+ modelId: model.id,
302
+ meritId: canonicalModel(model.provider, model.id),
303
+ input: model.input ?? ["text"],
304
+ cost: model.cost && typeof model.cost.input === "number" && typeof model.cost.output === "number"
305
+ ? { input: model.cost.input, output: model.cost.output,
306
+ cacheRead: model.cost.cacheRead, cacheWrite: model.cost.cacheWrite }
307
+ : undefined,
308
+ contextWindow: model.contextWindow,
309
+ maxTokens: model.maxTokens,
310
+ };
311
+ }
312
+
313
+ function eligibleRoutes(ctx: ExtensionContext): ModelRoute[] {
314
+ const registry = ctx.modelRegistry as unknown as { getAvailable?: () => Array<Parameters<typeof routeFor>[0]> };
315
+ const scoped = (ctx.scopedModels ?? []) as readonly { model: Parameters<typeof routeFor>[0] }[];
316
+ const models = scoped.length ? scoped.map((item) => item.model) : registry.getAvailable?.() ?? [];
317
+ if (ctx.model && !models.some((model) => model.provider === ctx.model!.provider && model.id === ctx.model!.id))
318
+ models.unshift(ctx.model);
319
+ const byRoute = new Map<string, ModelRoute>();
320
+ for (const model of models) {
321
+ const route = routeFor(model);
322
+ byRoute.set(`${route.provider}:${route.modelId}`, route);
323
+ }
324
+ return [...byRoute.values()];
325
+ }
326
+
327
+ function writeState(value: unknown): void {
328
+ const temp = `${STATE_FILE}.tmp-${process.pid}-${Date.now()}`;
329
+ try {
330
+ writeFileSync(temp, JSON.stringify(value) + "\n");
331
+ renameSync(temp, STATE_FILE);
332
+ } finally {
333
+ if (existsSync(temp)) rmSync(temp, { force: true });
334
+ }
335
+ }
259
336
 
260
337
  function reportHarnessState(ctx: ExtensionContext, settled = false): void {
261
338
  try {
262
339
  mkdirSync(HOME, { recursive: true });
263
340
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
341
+ const currentRoute = ctx.model ? routeFor(ctx.model) : null;
264
342
  if (settled) settledTaskKey = latestSessionTaskKey(ctx);
265
343
  const prior = existsSync(STATE_FILE)
266
344
  ? JSON.parse(readFileSync(STATE_FILE, "utf8")) as { sessionFile?: string | null; settledAt?: string }
@@ -268,18 +346,19 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
268
346
  const sessionFile = ctx.sessionManager.getSessionFile() ?? null;
269
347
  const settledAt = settled ? new Date().toISOString()
270
348
  : prior.sessionFile === sessionFile ? prior.settledAt ?? null : null;
271
- writeFileSync(
272
- STATE_FILE,
273
- JSON.stringify({
349
+ writeState({
350
+ schemaVersion: 1,
274
351
  currentModel: model,
352
+ currentRoute,
353
+ routes: eligibleRoutes(ctx),
275
354
  fallbackModel,
355
+ fallbackRoute,
276
356
  sessionFile,
277
357
  settledTaskKey,
278
358
  settledAt,
279
359
  cwd: ctx.cwd,
280
360
  updatedAt: new Date().toISOString(),
281
- }) + "\n",
282
- );
361
+ });
283
362
  } catch {
284
363
  /* never break the host session over reporting */
285
364
  }
@@ -288,8 +367,11 @@ function reportHarnessState(ctx: ExtensionContext, settled = false): void {
288
367
  /** Restore the persisted fallback so a fresh session keeps it even with no pending recs. */
289
368
  function restoreFallback(): void {
290
369
  try {
291
- const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as { fallbackModel?: string | null };
370
+ const st = JSON.parse(readFileSync(STATE_FILE, "utf8")) as {
371
+ fallbackModel?: string | null; fallbackRoute?: ModelRoute | null;
372
+ };
292
373
  if (st.fallbackModel) fallbackModel = st.fallbackModel;
374
+ if (st.fallbackRoute) fallbackRoute = st.fallbackRoute;
293
375
  } catch {
294
376
  /* no state yet */
295
377
  }
@@ -344,7 +426,15 @@ function catalogEntry(openrouterId: string): CatalogEntryLite | undefined {
344
426
  * OpenRouter catalog, so inject anything missing via registerProvider
345
427
  * (takes effect immediately, no /reload needed).
346
428
  */
347
- function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string) {
429
+ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: string, route?: ModelRoute) {
430
+ const registry = ctx.modelRegistry as unknown as {
431
+ find?: (provider: string, id: string) => unknown | undefined;
432
+ };
433
+ if (route) {
434
+ const exact = registry.find?.(route.provider, route.modelId);
435
+ if (exact) return exact;
436
+ if (route.provider !== "openrouter") return undefined;
437
+ }
348
438
  const existing = resolveModel(ctx, openrouterId);
349
439
  if (existing) return existing;
350
440
 
@@ -372,16 +462,14 @@ function ensureModel(pi: ExtensionAPI, ctx: ExtensionContext, openrouterId: stri
372
462
  })),
373
463
  });
374
464
 
375
- const reg = ctx.modelRegistry as unknown as {
376
- find?: (provider: string, id: string) => unknown | undefined;
377
- };
378
- return reg.find?.(INJECT_PROVIDER, openrouterId);
465
+ return registry.find?.(INJECT_PROVIDER, openrouterId);
379
466
  }
380
467
 
381
468
  export default function openmerit(pi: ExtensionAPI) {
382
469
  let pollTimer: ReturnType<typeof setInterval> | null = null;
383
470
  let trialJob: ChildProcessByStdio<null, Readable, Readable> | null = null;
384
- const queuedJobs: { sessionFile: string; settledAt: string; model: string; key: string; cwd: string; bytes: number }[] = [];
471
+ const queuedJobs: { sessionFile: string; settledAt: string; model: string; route: ModelRoute;
472
+ key: string; cwd: string; bytes: number }[] = [];
385
473
  const startedJobs = new Set<string>();
386
474
  let activeSession: string | null = null;
387
475
  let progress: { sessionFile: string; taskKey: string; current: TrialProgress | null;
@@ -408,7 +496,7 @@ export default function openmerit(pi: ExtensionAPI) {
408
496
  return;
409
497
  }
410
498
  const child = spawn(process.execPath, [ENGINE_FILE, "session-trial", job.sessionFile,
411
- job.settledAt, job.model, job.key, job.cwd, String(job.bytes)], {
499
+ job.settledAt, job.model, job.key, job.cwd, String(job.bytes), job.route.provider, job.route.modelId], {
412
500
  cwd: job.cwd, env: process.env, stdio: ["ignore", "pipe", "pipe"],
413
501
  detached: process.platform !== "win32",
414
502
  });
@@ -466,7 +554,8 @@ export default function openmerit(pi: ExtensionAPI) {
466
554
  function queueSettledTask(ctx: ExtensionContext): void {
467
555
  const sessionFile = ctx.sessionManager.getSessionFile();
468
556
  const model = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
469
- if (!sessionFile || !model || !settledTaskKey || !existsSync(sessionFile)) return;
557
+ const route = ctx.model ? routeFor(ctx.model) : null;
558
+ if (!sessionFile || !model || !route || !settledTaskKey || !existsSync(sessionFile)) return;
470
559
  // The alpha compares completed response tasks. The engine validates images,
471
560
  // tool use and the baseline answer before any provider calls.
472
561
  let completed = false;
@@ -490,7 +579,7 @@ export default function openmerit(pi: ExtensionAPI) {
490
579
  if (ctx.hasUI) ctx.ui.notify(`openmerit: comparison skipped: ${budget.reason}. ${budget.summary}`, "warning");
491
580
  return;
492
581
  }
493
- queuedJobs.push({ sessionFile, settledAt, model, key: settledTaskKey, cwd: ctx.cwd, bytes });
582
+ queuedJobs.push({ sessionFile, settledAt, model, route, key: settledTaskKey, cwd: ctx.cwd, bytes });
494
583
  startNextJob(ctx);
495
584
  }
496
585
 
@@ -502,11 +591,18 @@ export default function openmerit(pi: ExtensionAPI) {
502
591
 
503
592
  async function applyRecommendation(ctx: ExtensionContext, rec: Recommendation): Promise<boolean> {
504
593
  const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : null;
505
- if (rec.currentModel && current !== rec.currentModel) {
594
+ const currentRoute = ctx.model ? routeFor(ctx.model) : null;
595
+ if (rec.currentRoute && currentRoute &&
596
+ (rec.currentRoute.provider !== currentRoute.provider || rec.currentRoute.modelId !== currentRoute.modelId)) {
597
+ ctx.ui.notify(`openmerit: recommendation targets ${rec.currentRoute.provider}/${rec.currentRoute.modelId}; ` +
598
+ `current route is ${currentRoute.provider}/${currentRoute.modelId}`, "warning");
599
+ return false;
600
+ }
601
+ if (!rec.currentRoute && rec.currentModel && current !== rec.currentModel) {
506
602
  ctx.ui.notify(`openmerit: recommendation was measured against ${rec.currentModel}; current model is ${current ?? "unknown"}`, "warning");
507
603
  return false;
508
604
  }
509
- const model = ensureModel(pi, ctx, rec.recommended.model);
605
+ const model = ensureModel(pi, ctx, rec.recommended.model, rec.recommended.route);
510
606
  if (!model) {
511
607
  ctx.ui.notify(
512
608
  `openmerit: ${rec.recommended.model} not found in pi's model registry (add the provider/model first)`,
@@ -517,8 +613,12 @@ export default function openmerit(pi: ExtensionAPI) {
517
613
  const ok = await pi.setModel(model as Parameters<typeof pi.setModel>[0]);
518
614
  if (ok) {
519
615
  setStatus(rec, "applied");
520
- if (rec.fallback) fallbackModel = rec.fallback.model;
521
- ctx.ui.notify(`openmerit: switched to ${rec.recommended.model}\n${rec.recommended.reason}`, "info");
616
+ if (rec.fallback) {
617
+ fallbackModel = rec.fallback.model;
618
+ fallbackRoute = rec.fallback.route ?? null;
619
+ }
620
+ ctx.ui.notify(`openmerit: switched to ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n` +
621
+ rec.recommended.reason, "info");
522
622
  reportHarnessState(ctx);
523
623
  } else {
524
624
  ctx.ui.notify(`openmerit: no auth for ${rec.recommended.model}`, "error");
@@ -550,7 +650,7 @@ export default function openmerit(pi: ExtensionAPI) {
550
650
  // startup. Notify (fire-and-forget) and let the user act via /openmerit.
551
651
  if (ctx.hasUI) {
552
652
  ctx.ui.notify(
553
- `openmerit recommends: ${rec.recommended.model}\n${rec.recommended.reason}\n` +
653
+ `openmerit recommends: ${routeDisplay(rec.recommended.route, rec.recommended.model)}\n${rec.recommended.reason}\n` +
554
654
  `gate: ${rec.policy.reasons.join("; ")}\n(/openmerit apply to switch, /openmerit dismiss to ignore)`,
555
655
  "info",
556
656
  );
@@ -621,7 +721,7 @@ export default function openmerit(pi: ExtensionAPI) {
621
721
  const policy = loadPolicy();
622
722
  if (policy.fallback?.apply_on_error === false) return;
623
723
  const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "current model";
624
- const model = ensureModel(pi, ctx, fallbackModel);
724
+ const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
625
725
  if (!model) return;
626
726
 
627
727
  const auto = policy.mode === "auto";
@@ -673,7 +773,7 @@ export default function openmerit(pi: ExtensionAPI) {
673
773
  ctx.ui.notify("openmerit: no fallback set yet", "info");
674
774
  return;
675
775
  }
676
- const model = ensureModel(pi, ctx, fallbackModel);
776
+ const model = ensureModel(pi, ctx, fallbackModel, fallbackRoute ?? undefined);
677
777
  if (model && (await pi.setModel(model as Parameters<typeof pi.setModel>[0]))) {
678
778
  ctx.ui.notify(`openmerit: switched to fallback ${fallbackModel}`, "info");
679
779
  reportHarnessState(ctx);
@@ -683,7 +783,8 @@ export default function openmerit(pi: ExtensionAPI) {
683
783
  return;
684
784
  }
685
785
 
686
- const current = ctx.model ? canonicalModel(ctx.model.provider, ctx.model.id) : "unknown";
786
+ const currentRoute = ctx.model ? routeFor(ctx.model) : undefined;
787
+ const current = currentRoute ? routeDisplay(currentRoute, currentRoute.meritId) : "unknown";
687
788
  const budget = trialBudget();
688
789
  const liveProgress = progress?.sessionFile === ctx.sessionManager.getSessionFile() ? progress : null;
689
790
  const comparison = trialJob ? liveProgress?.current
@@ -706,7 +807,8 @@ export default function openmerit(pi: ExtensionAPI) {
706
807
  for (const r of pending.slice(-5)) {
707
808
  lines.push(
708
809
  "",
709
- `-> ${r.recommended.model} (gain ${r.evidence.scoreGain}, ${r.evidence.priceRatio}x price, frontier=${r.evidence.onFrontier})`,
810
+ `-> ${routeDisplay(r.recommended.route, r.recommended.model)} (gain ${r.evidence.scoreGain}, ` +
811
+ `${r.evidence.priceKnown === false ? "price unknown" : `${r.evidence.priceRatio}x price`}, frontier=${r.evidence.onFrontier})`,
710
812
  ` ${r.recommended.reason}`,
711
813
  ` gate: ${r.policy.reasons.join("; ") || "n/a"}`,
712
814
  );
@@ -45,8 +45,8 @@ the best model for the task at hand, at the best price, with a vetted fallback.
45
45
  ## Guarantees
46
46
 
47
47
  - Per-task comparisons send the task text, uploaded files or images (when
48
- present), and candidate answers through pi/OpenRouter for model runs and
49
- judging. Candidate Pi runs default to no tools; any tool access is an
48
+ present), and candidate answers through the model routes configured in Pi
49
+ for model runs and judging. Candidate Pi runs default to no tools; any tool access is an
50
50
  explicit user opt-in. Use non-sensitive examples while
51
51
  evaluating this alpha.
52
52
  - The shipped policy is supervised. Swaps happen automatically only after the
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openmerit",
3
- "version": "0.1.1",
3
+ "version": "0.1.3",
4
4
  "description": "Find better models for each pi task by comparing quality, cost, and latency.",
5
5
  "type": "module",
6
6
  "keywords": [
@@ -17,10 +17,13 @@
17
17
  "extension/openmerit.ts",
18
18
  "instructions/",
19
19
  "examples/",
20
+ "rules.md",
20
21
  "benchmark/invoice_ocr/data/"
21
22
  ],
22
23
  "pi": {
23
- "extensions": ["./extension/openmerit.ts"]
24
+ "extensions": [
25
+ "./extension/openmerit.ts"
26
+ ]
24
27
  },
25
28
  "bin": {
26
29
  "openmerit": "dist/cli.js"
package/rules.md ADDED
@@ -0,0 +1,39 @@
1
+ # OpenMerit product and architecture rules
2
+
3
+ - **Optimize for useful model choices.** OpenMerit exists to find better task-specific tradeoffs between quality, cost, and latency—not to become a general-purpose agent or telemetry platform.
4
+
5
+ - **Act as a control plane, not a gateway.** Normal traffic stays between the harness and provider; OpenMerit observes evidence, runs explicit trials, and returns recommendations without intercepting ordinary work.
6
+
7
+ - **Make Pi plug-and-play first.** Pi and Pi-based harnesses are the first supported user experience because that is where early users already work; broader harness support can follow without weakening this path.
8
+
9
+ - **Keep the core harness-neutral.** Pi is the first `HarnessAdapter`, not a permanent assumption in scoring, recommendations, policy, or stored data, so other harnesses can be added without rewriting the merit loop.
10
+
11
+ - **Keep model providers interchangeable.** OpenRouter is a first-class provider and discovery source, not the foundation of the domain model; provider-specific behavior belongs behind a `ModelProviderAdapter`.
12
+
13
+ - **Require end-to-end provider neutrality.** Discovery, authentication, execution, scoring, and swapping must preserve the selected route; a feature is not provider-neutral if only its API client is abstracted.
14
+
15
+ - **Separate model identity from execution route.** Keep logical model IDs in stable `vendor/model` form, while recording the actual provider or route separately, so the same model can be compared through OpenRouter, a native provider, or a local provider.
16
+
17
+ - **Respect the harness's eligible model pool.** Prefer models already available and configured in the active harness; external catalogs may enrich or expand discovery but must not silently override harness scope or credentials.
18
+
19
+ - **Treat public benchmarks as priors, not proof.** Benchmarks help shortlist candidates, but merit comes from trials on the user's actual task.
20
+
21
+ - **Keep four integration boundaries distinct.** Model execution (`ModelProviderAdapter`), agent execution (`HarnessAdapter`), incoming traces (`ObservationSource`), and outgoing telemetry (`EventSink`) solve different problems and must not be coupled.
22
+
23
+ - **Use provider-neutral core records.** Normalize integrations into stable concepts such as `TaskObservation`, `CandidateRun`, `TrialScore`, `Frontier`, `Recommendation`, and `MeritEvent`, so integrations do not leak their schemas into the decision engine.
24
+
25
+ - **Optimize per task before aggregating per agent.** Agent-level conclusions must be built from measured task evidence rather than assumed from global model rankings.
26
+
27
+ - **Version persisted events and evolve them additively.** Existing 0.1.x state must remain readable, and append-only recommendation history must stay intact; migrations should normalize old records rather than invalidate them.
28
+
29
+ - **Keep policy as the sole auto-swap authority.** Trials and strategists may recommend changes, but only the policy gate may approve automatic application, and every recommendation must retain its gate reasons.
30
+
31
+ - **Default to safe, explicit trials.** Candidate tool access stays off unless deliberately allowed, trials remain isolated, and temporary workspaces are cleaned up because trying a model must not expose or damage a user's project by surprise.
32
+
33
+ - **Keep the shadow track isolated, not necessarily concurrent.** Candidate trials may run sequentially to respect cost, rate, and safety limits while remaining separate from the user's live task.
34
+
35
+ - **Treat local JSONL as the default integration, not a lock-in.** Local traces and events should work without an external service; systems such as Langfuse can later plug in as observation sources or event sinks.
36
+
37
+ - **Keep integrations optional and the core lightweight.** New providers, harnesses, and observability services should not impose credentials, network calls, or heavy dependencies on users who do not enable them.
38
+
39
+ - **Keep 0.1.x focused.** The near-term bar is reliable, provider-neutral use for Pi tinkerers and solo hackers; additional harnesses and hosted observability integrations belong in later releases unless required to prove the boundaries work.