auto-model-router 0.4.2 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,14 +7,14 @@
7
7
  },
8
8
  "metadata": {
9
9
  "description": "auto-model-router: a local cost/complexity-aware model router for Oh My Pi, backed by OpenRouter",
10
- "version": "0.4.2",
10
+ "version": "0.4.3",
11
11
  "pluginRoot": "."
12
12
  },
13
13
  "plugins": [
14
14
  {
15
15
  "name": "auto-model-router",
16
16
  "description": "Local cost/complexity-aware model router for Oh My Pi, backed by OpenRouter. Runs in-process, routes per turn by price and task complexity, with budget caps, mid-stream escalation, and cache-aware hysteresis.",
17
- "version": "0.4.2",
17
+ "version": "0.4.3",
18
18
  "author": {
19
19
  "name": "drewappling",
20
20
  "email": "drewappling@gmail.com"
package/README.md CHANGED
@@ -497,6 +497,8 @@ escalation signal, error. Three views aggregate it, all from the same
497
497
  unreachable.
498
498
  - `auto-model-router report --days 7 [--harness <id>] [--json]` on the terminal.
499
499
  - `GET /v1/router/report?days=7&harness=<id>` for dashboards.
500
+ - `GET /v1/router/summary?harness=<id>` — the daily summary as JSON (`auto=1`
501
+ applies the once-a-day gate and returns `due: false` when nothing is due).
500
502
 
501
503
  What it shows, for the window:
502
504
 
@@ -510,6 +512,18 @@ What it shows, for the window:
510
512
  | by day | UTC calendar days: dispatches, spend, cache hit |
511
513
  | same traffic on one model | the window's tokens priced on each `report.baselines` model at list price with the window's cache hit rate, and what share the router saved against it |
512
514
 
515
+ **Soft-failure spikes.** `/health` (`softFailures.spikes`), `/router status`
516
+ and the daily summary list any model whose failure rate over the last hour —
517
+ probe rejections such as `empty_completion` or `repeat_tool_call` that
518
+ OpenRouter counts as success, plus attributable transport errors — is at
519
+ least 25%, at least twice its own rate over the preceding 7 days, and covers
520
+ at least 5 dispatches with 3 failures. This is visibility only: two weeks of
521
+ ledger data showed soft failures do not cluster tightly enough for a breaker
522
+ to save money (after a burst, the next 15 minutes ran 84–1,577 successes per
523
+ 13–50 failures), and OpenRouter's provider failover plus the router's own
524
+ escalation already cover the retry. Use a spike as the cue to `/router pin`
525
+ or deny a model for the session.
526
+
513
527
  Spend follows the ledger's rule — the provider's reported cost when it gave
514
528
  one, else the usage-priced figure the router computed, else the forecast.
515
529
  Speed uses only clean streamed rows (TTFT recorded, no error); tokens/s is
@@ -534,7 +548,8 @@ Status); the subcommands go straight there:
534
548
  | --- | --- |
535
549
  | `/router config` | Section picker over **every** config key: Server, OpenRouter, Ollama Cloud, Benchmarks, Tiers, Tasks, Filters, Classifier, Escalation, Hysteresis, Exploration, Cache, Compaction, Context (agentdox), Budget, Ledger, Logging, Profiles. Only `ollama.prices` and `ollama.twins` (maps) stay YAML-only. |
536
550
  | `/router report` | Usage analytics in a fullscreen hub styled like `/models`: pick a view in the sidebar, set the window (24h / 7d / 30d / 90d) and the harness scope there too. `/router report 30d --all` presets them. See [Usage reports](#usage-reports). |
537
- | `/router status` | The router's `/health`: key sources, catalog size and age, Ollama availability, plan usage and cost bias, agentdox bridge. |
551
+ | `/router summary [--all]` | The last 24 hours in a few lines: spend against the day before, turns and conversations, cache hit, escalations, errors, model switches with tier moves, top models, savings against the first `report.baselines` model, digests and subagent spend, soft-failure spikes, and the Ollama meter with its runway. Posted automatically once a day at session start when `report.dailySummary` is on (the router keeps a per-harness marker, so several omp windows show it once between them, and a day with no turns and no spikes is skipped). |
552
+ | `/router status` | The router's `/health`: key sources, catalog size and age, Ollama availability, plan usage and cost bias, soft-failure spikes (below), agentdox bridge. |
538
553
  | `/router why` | Explain this session's last routed turn: model and provider, tier, classification source and confidence, cost, cache hit, latency, the full decision trail and classifier reasons, any feedback already given. |
539
554
  | `/router good` / `/router bad [note]` | Judge that turn. Recorded against the model that served it (`POST /v1/router/feedback`), shown per model in the report's `feedback` column, and the label the de-escalation work needs. `/router feedback good\|bad` is the same. |
540
555
  | `/router pin <model\|off>` | Route this session to one model until cleared (admitted past price, quality and trust filters; tool support and context window still apply). Escalations and failovers after the first attempt still run. |
@@ -668,6 +683,7 @@ Each task (`coding`, `vision`, `documentation`, `data`, `chat`) is a
668
683
  | `includeFree` | `false` | Include free models (rate-limited hard; usually excluded). |
669
684
  | `requireToolSupport` | `true` | Only models that support tool calls. |
670
685
  | `feedbackWeight` | `0` | How much a `/router good\|bad` verdict weighs in a model's trust rate: a bad verdict counts as this many failures, a good one as this many successes. `0` records verdicts without acting on them. |
686
+ | `feedbackByTask` | `false` | Count a verdict only when routing the same task type as the judged turn (coding, vision, documentation, data, chat), so a model that codes well but explains badly keeps its coding trust. Verdicts on turns with no recorded task count everywhere. |
671
687
  | `minTrust` | `0.7` | Minimum success rate; models below this (after `minTrustSamples`) are demoted. |
672
688
  | `minTrustSamples` | `12` | Attempts before trust is enforced. |
673
689
  | `trustScopedByHarness` | `false` | `true` = each harness reads only its own trust rows. |
@@ -684,7 +700,7 @@ Each task (`coding`, `vision`, `documentation`, `data`, `chat`) is a
684
700
  | Key | Default | Meaning |
685
701
  | --- | --- | --- |
686
702
  | `ambiguityThreshold` | `0.6` | Below this heuristic confidence, the adjudicator model decides the tier. |
687
- | `learnedModelPath` | unset | A model written by `bun tools/train-classifier.ts` (logistic regression over the ledger's recorded features, label = the turn escalated). When set, every decision records `learned: p(escalate)=…`. Advisory only: it never moves a tier until replay shows it should. |
703
+ | `learnedModelPath` | unset | A model written by `bun tools/train-classifier.ts` (logistic regression over the ledger's recorded features; label = the turn escalated, or with `--label feedback` the turn was judged bad via `/router bad`). When set, every decision records `learned: p(escalate)=…` or `learned: p(bad)=…`. Advisory only: it never moves a tier until replay shows it should. |
688
704
  | `model` | `qwen/qwen3.7-flash` | Adjudicator model slug. |
689
705
  | `maxCostFraction` | `0.02` | Adjudicator cost cap as a fraction of the turn's budget. |
690
706
  | `maxCostUsd` | `0.002` | Absolute adjudicator cost cap, USD. |
@@ -748,6 +764,8 @@ shrunk results stay shrunk (rewriting them would break the prompt cache).
748
764
  | `keepHeadBytes` / `keepTailBytes` | `512` / `512` | Bytes kept around the elision breadcrumb. |
749
765
  | `elideSupersededReads` | `true` | Stub an older result when a newer call to the same resource supersedes it. |
750
766
  | `collapseDuplicateResults` | `true` | Collapse byte-identical repeated results to a single copy. |
767
+ | `digestToolResults` | `false` | Summarising compaction: when the plan gains an edit, a cheap model (`digest.tier`/`digest.model`, under `digest.maxCostUsd` and `digest.timeoutMs`) digests the tool result instead of it being cut to head+tail or a stub. The digest is stored on the edit, so the dispatched bytes stay identical on later turns and the cache holds. Applies when the turn routed at or above `digest.fromTier`; works without `digest.enabled`. |
768
+ | `digestMaxPerTurn` | `2` | Digests per turn at most (largest results first); the rest of a plan's new edits stay plain until a later turn. |
751
769
 
752
770
  ### `cache` — prompt-cache breakpoints
753
771
 
@@ -809,6 +827,7 @@ is a ledger row (`requestedModel` `digest`) and the report totals them.
809
827
  | Key | Default | Meaning |
810
828
  | --- | --- | --- |
811
829
  | `baselines` | `anthropic/claude-opus-5`, `anthropic/claude-sonnet-5` | Models the report prices the window's traffic on as a single-model counterfactual. Unknown slugs are skipped. |
830
+ | `dailySummary` | `true` | Post the daily summary (below) into the transcript at the first interactive omp session start of each day. Hot-reloads. |
812
831
 
813
832
  ### `ledger` — cost measurement
814
833
 
@@ -6,6 +6,7 @@
6
6
  */
7
7
 
8
8
  import type { UsageReport } from "../src/cost/report.ts";
9
+ import type { DailySummary } from "../src/cost/summary.ts";
9
10
 
10
11
  export interface ReportRequest {
11
12
  windowDays: number;
@@ -61,6 +62,44 @@ export async function fetchReport(
61
62
  return (await res.json()) as UsageReport;
62
63
  }
63
64
 
65
+ /** GETs the daily summary; `auto` asks the router whether one is due today. Throws on any failure. */
66
+ export async function fetchSummary(
67
+ baseUrl: string,
68
+ harnessId: string,
69
+ auto: boolean,
70
+ headers: Record<string, string>,
71
+ fetchImpl: FetchLike = fetch,
72
+ timeoutMs = 5_000,
73
+ ): Promise<{ due: boolean; reason?: string; summary: DailySummary | null }> {
74
+ const params = new URLSearchParams();
75
+ if (harnessId !== "") params.set("harness", harnessId);
76
+ if (auto) params.set("auto", "1");
77
+ const q = params.toString();
78
+ const res = await fetchImpl(`${baseUrl}/v1/router/summary${q === "" ? "" : `?${q}`}`, { headers, signal: AbortSignal.timeout(timeoutMs) });
79
+ if (!res.ok) throw new Error(`router returned ${res.status}`);
80
+ return (await res.json()) as { due: boolean; reason?: string; summary: DailySummary | null };
81
+ }
82
+
83
+ /** One spiking model as `/health` reports it. */
84
+ export interface SoftFailureSpikeView {
85
+ slug?: string;
86
+ recentDispatches?: number;
87
+ recentFailures?: number;
88
+ recentRate?: number;
89
+ baselineDispatches?: number;
90
+ baselineRate?: number;
91
+ }
92
+
93
+ /** "slug: 40% of 10 failed in the last 1h (7d baseline 8% of 120)" — one line per spiking model. */
94
+ export function renderSoftFailureSpikes(spikes: readonly SoftFailureSpikeView[] | undefined | null, recentMs = 3_600_000, baselineDays = 7): string[] {
95
+ if (spikes === undefined || spikes === null || spikes.length === 0) return [];
96
+ const window = recentMs >= 3_600_000 ? `${(recentMs / 3_600_000).toFixed(recentMs % 3_600_000 === 0 ? 0 : 1)}h` : `${Math.round(recentMs / 60_000)}m`;
97
+ return spikes.map(
98
+ (s) =>
99
+ `${s.slug ?? "?"}: ${((s.recentRate ?? 0) * 100).toFixed(0)}% of ${s.recentDispatches ?? 0} failed in the last ${window} (${baselineDays}d baseline ${((s.baselineRate ?? 0) * 100).toFixed(0)}% of ${s.baselineDispatches ?? 0})`,
100
+ );
101
+ }
102
+
64
103
  /** The subset of `/health` the status view renders. */
65
104
  export interface HealthSnapshot {
66
105
  status?: string;
@@ -80,6 +119,7 @@ export interface HealthSnapshot {
80
119
  runway?: { dailyBurnUsd?: number; creditsLeftUsd?: number; days?: number | null } | null;
81
120
  costBias?: { configured?: number; effective?: number; biasUntilUsage?: number };
82
121
  } | null;
122
+ softFailures?: { recentMs?: number; baselineDays?: number; spikes?: SoftFailureSpikeView[] } | null;
83
123
  catalog?: {
84
124
  models?: number;
85
125
  ageMs?: number;
@@ -119,6 +159,12 @@ export function renderStatus(baseUrl: string, h: HealthSnapshot, nowMs = Date.no
119
159
  const rwText = rw !== undefined && rw !== null ? ` · burn $${(rw.dailyBurnUsd ?? 0).toFixed(2)}/day · ${rw.days === null || rw.days === undefined ? "credits left: unknown burn" : `~${Math.round(rw.days)} days of credits left`}` : "";
120
160
  if (o.meter !== undefined && o.meter !== null) out.push(`ollama billing: ${calText}${rwText}`);
121
161
  }
162
+ const sf = h.softFailures;
163
+ if (sf !== undefined && sf !== null) {
164
+ const lines = renderSoftFailureSpikes(sf.spikes, sf.recentMs, sf.baselineDays);
165
+ if (lines.length === 0) out.push("soft failures: no model spiking in the last hour");
166
+ else out.push(`soft failures SPIKING (${lines.length}):`, ...lines.map((l) => ` ${l}`));
167
+ }
122
168
  const a = h.agentdox;
123
169
  out.push(a === undefined || a === null ? "agentdox: off" : `agentdox: ${a.url ?? "?"} scope ${a.defaultScope ?? "?"}${a.recordTurns === true ? " · recording turns" : ""}`);
124
170
  return out.join("\n");
@@ -50,6 +50,7 @@ import { routerConfigPath, writeRouterConfig } from "../src/cli/config-cmd.ts";
50
50
  import { loadConfig } from "../src/config/load.ts";
51
51
  import type { RouterConfig } from "../src/config/types.ts";
52
52
  import { buildUsageReport, renderUsageReport, type UsageReport } from "../src/cost/report.ts";
53
+ import { buildDailySummary, renderDailySummary } from "../src/cost/summary.ts";
53
54
  import { openDb } from "../src/util/sqlite.ts";
54
55
 
55
56
  import type { ExtensionAPI, ExtensionContext } from "@oh-my-pi/pi-coding-agent";
@@ -61,6 +62,7 @@ import { ReportHub } from "./report-hub.ts";
61
62
  import {
62
63
  describeOverride,
63
64
  fetchReport,
65
+ fetchSummary,
64
66
  parseOverrideArgs,
65
67
  parseReportArgs,
66
68
  renderStatus,
@@ -77,12 +79,32 @@ const HARNESS_ID = process.env.OMP_HARNESS_ID ?? "";
77
79
 
78
80
  /** Custom message type for report/status output in the transcript. */
79
81
  const MESSAGE_TYPE = "auto-model-router";
82
+ /** The embedded router boots on session_start too; the auto summary waits for it this long. */
83
+ const SUMMARY_TRIES = 8;
84
+ const SUMMARY_RETRY_MS = 1_500;
80
85
 
81
86
  export default function (pi: ExtensionAPI): void {
82
87
  pi.setLabel("auto-model-router");
83
88
 
89
+ // Once a day, the first interactive session posts yesterday's summary. The
90
+ // router decides whether one is due (report.dailySummary, a per-harness
91
+ // marker), so several windows on one router show it once between them.
92
+ pi.on("session_start", async (_event, ctx) => {
93
+ if (!ctx.hasUI) return;
94
+ for (let i = 0; i < SUMMARY_TRIES; i++) {
95
+ try {
96
+ const r = await fetchSummary(routerBaseUrl(), HARNESS_ID, true, routerAuthHeaders(), fetch, 2_000);
97
+ if (r.due && r.summary !== null) post(pi, renderDailySummary(r.summary));
98
+ return;
99
+ } catch {
100
+ // Router still starting (embed) or absent: try again shortly, then give up quietly.
101
+ await new Promise((resolve) => setTimeout(resolve, SUMMARY_RETRY_MS));
102
+ }
103
+ }
104
+ });
105
+
84
106
  pi.registerCommand("router", {
85
- description: "auto-model-router: configure, usage report, status",
107
+ description: "auto-model-router: configure, usage report, status, daily summary",
86
108
  handler: async (args, ctx) => {
87
109
  const [verb = "", ...rest] = args.trim().split(/\s+/).filter((t) => t !== "");
88
110
  const tail = rest.join(" ");
@@ -95,6 +117,9 @@ export default function (pi: ExtensionAPI): void {
95
117
  case "status":
96
118
  case "health":
97
119
  return status(pi, ctx);
120
+ case "summary":
121
+ case "daily":
122
+ return summary(pi, ctx, tail);
98
123
  case "why":
99
124
  case "explain":
100
125
  return why(pi, ctx);
@@ -116,20 +141,22 @@ export default function (pi: ExtensionAPI): void {
116
141
  case "":
117
142
  break;
118
143
  default:
119
- ctx.ui.notify(`unknown /router subcommand "${verb}" (config | report | status | why | good | bad | pin | tier)`, "warn");
144
+ ctx.ui.notify(`unknown /router subcommand "${verb}" (config | report | summary | status | why | good | bad | pin | tier)`, "warn");
120
145
  return;
121
146
  }
122
147
 
123
148
  const chosen = await ctx.ui.select("auto-model-router", [
124
149
  { label: "Configure", description: "edit any router setting" },
125
150
  { label: "Report", description: "usage analytics; window and scope adjustable inside" },
126
- { label: "Status", description: "keys, catalog, Ollama, agentdox" },
151
+ { label: "Summary", description: "the last 24h in a few lines" },
152
+ { label: "Status", description: "keys, catalog, Ollama, agentdox, soft-failure spikes" },
127
153
  { label: "Why", description: "explain this session's last routed turn" },
128
154
  { label: "Override", description: "pin a model or force a tier for this session" },
129
155
  ]);
130
156
  if (chosen === undefined) return;
131
157
  if (chosen === "Configure") return configure(ctx);
132
158
  if (chosen === "Status") return status(pi, ctx);
159
+ if (chosen === "Summary") return summary(pi, ctx, "");
133
160
  if (chosen === "Report") return report(pi, ctx, "");
134
161
  if (chosen === "Why") return why(pi, ctx);
135
162
  if (chosen === "Override") return override(ctx, "tier", "");
@@ -299,6 +326,31 @@ async function override(ctx: ExtensionContext, verb: "pin" | "tier", text: strin
299
326
  }
300
327
  }
301
328
 
329
+ /** `/router summary [--all]`: the last 24h, from the router or (router down) the ledger directly. */
330
+ async function summary(pi: ExtensionAPI, ctx: ExtensionContext, argText: string): Promise<void> {
331
+ const harnessId = parseReportArgs(argText, HARNESS_ID).harnessId;
332
+ try {
333
+ const r = await fetchSummary(routerBaseUrl(), harnessId, false, routerAuthHeaders());
334
+ if (r.summary !== null) {
335
+ post(pi, renderDailySummary(r.summary));
336
+ return;
337
+ }
338
+ } catch {
339
+ // Fall through to the ledger.
340
+ }
341
+ const cfg = loadConfig();
342
+ if (!existsSync(cfg.ledger.path)) {
343
+ ctx.ui.notify(`router unreachable at ${routerBaseUrl()} and no ledger at ${cfg.ledger.path}`, "error");
344
+ return;
345
+ }
346
+ const db = openDb(cfg.ledger.path);
347
+ try {
348
+ post(pi, `${renderDailySummary(buildDailySummary(db, { harnessId }))}\n(router unreachable: read from the ledger; spikes and the Ollama meter need the router)`);
349
+ } finally {
350
+ db.close();
351
+ }
352
+ }
353
+
302
354
  async function status(pi: ExtensionAPI, ctx: ExtensionContext): Promise<void> {
303
355
  try {
304
356
  post(pi, await loadStatus());
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "auto-model-router",
3
- "version": "0.4.2",
3
+ "version": "0.4.3",
4
4
  "private": false,
5
5
  "description": "Local cost/complexity-aware model router for Oh My Pi, backed by OpenRouter",
6
6
  "type": "module",
@@ -184,6 +184,7 @@ export const WIZARD_SECTIONS: readonly SectionSpec[] = [
184
184
  { path: "filters.requireToolSupport", label: "Require tool support", kind: "boolean" },
185
185
  { path: "filters.minTrust", label: "Min trust", kind: "number", min: 0, max: 1 },
186
186
  { path: "filters.feedbackWeight", label: "Feedback weight in trust", kind: "number", min: 0, hint: "0=record only; a bad verdict = this many failures" },
187
+ { path: "filters.feedbackByTask", label: "Scope verdicts to the task type", kind: "boolean" },
187
188
  { path: "filters.minTrustSamples", label: "Min trust samples", kind: "number", min: 0 },
188
189
  { path: "filters.trustScopedByHarness", label: "Scope trust per harness", kind: "boolean" },
189
190
  { path: "filters.trustWindowDays", label: "Trust window", kind: "number", min: 0, hint: "days, 0=all time" },
@@ -280,6 +281,8 @@ export const WIZARD_SECTIONS: readonly SectionSpec[] = [
280
281
  { path: "compaction.keepTailBytes", label: "Keep tail bytes", kind: "number", min: 0 },
281
282
  { path: "compaction.elideSupersededReads", label: "Elide superseded reads", kind: "boolean" },
282
283
  { path: "compaction.collapseDuplicateResults", label: "Collapse duplicate results", kind: "boolean" },
284
+ { path: "compaction.digestToolResults", label: "Digest compacted results with a cheap model", kind: "boolean" },
285
+ { path: "compaction.digestMaxPerTurn", label: "Digests per turn at most", kind: "number", min: 0 },
283
286
  ],
284
287
  },
285
288
  {
@@ -327,7 +330,10 @@ export const WIZARD_SECTIONS: readonly SectionSpec[] = [
327
330
  },
328
331
  {
329
332
  title: "Report",
330
- fields: [{ path: "report.baselines", label: "Counterfactual baseline models", kind: "stringArray", hint: "comma-separated slugs" }],
333
+ fields: [
334
+ { path: "report.baselines", label: "Counterfactual baseline models", kind: "stringArray", hint: "comma-separated slugs" },
335
+ { path: "report.dailySummary", label: "Daily summary at session start", kind: "boolean" },
336
+ ],
331
337
  },
332
338
  {
333
339
  title: "Ledger",
@@ -103,6 +103,7 @@ export const DEFAULT_CONFIG: RouterConfig = {
103
103
  minTrust: 0.7,
104
104
  // Verdicts are recorded and reported first; weigh them once there are some.
105
105
  feedbackWeight: 0,
106
+ feedbackByTask: false,
106
107
  minTrustSamples: 12,
107
108
  // Shared trust by default: more samples, demotion guard stays effective
108
109
  // even with a tiny guardrail-narrowed catalog.
@@ -291,6 +292,10 @@ export const DEFAULT_CONFIG: RouterConfig = {
291
292
  keepTailBytes: 512,
292
293
  elideSupersededReads: true,
293
294
  collapseDuplicateResults: true,
295
+ // Off: a synchronous cheap-model call before dispatch, only worth it where
296
+ // stale tool output is the prompt and the turn is on a dear model.
297
+ digestToolResults: false,
298
+ digestMaxPerTurn: 2,
294
299
  },
295
300
  digest: {
296
301
  // Off until an operator turns it on: it changes what the model reads.
@@ -308,6 +313,8 @@ export const DEFAULT_CONFIG: RouterConfig = {
308
313
  report: {
309
314
  // The frontier pair most omp users would otherwise run on.
310
315
  baselines: ["anthropic/claude-opus-5", "anthropic/claude-sonnet-5"],
316
+ // One transcript message per day, at the first interactive session start.
317
+ dailySummary: true,
311
318
  },
312
319
  budget: {
313
320
  // No caps by default; at a configured ceiling, downgrade rather than fail.
@@ -90,6 +90,7 @@ const filters = z.strictObject({
90
90
  requireToolSupport: z.boolean().optional(),
91
91
  minTrust: z.number().min(0).max(1).optional(),
92
92
  feedbackWeight: z.number().nonnegative().optional(),
93
+ feedbackByTask: z.boolean().optional(),
93
94
  minTrustSamples: z.number().int().nonnegative().optional(),
94
95
  trustScopedByHarness: z.boolean().optional(),
95
96
  trustWindowDays: z.number().nonnegative().optional(),
@@ -205,6 +206,8 @@ const compaction = z.strictObject({
205
206
  keepTailBytes: z.number().int().nonnegative().optional(),
206
207
  elideSupersededReads: z.boolean().optional(),
207
208
  collapseDuplicateResults: z.boolean().optional(),
209
+ digestToolResults: z.boolean().optional(),
210
+ digestMaxPerTurn: z.number().int().nonnegative().optional(),
208
211
  });
209
212
 
210
213
  const budget = z.strictObject({
@@ -272,7 +275,7 @@ export const configInputSchema = z.strictObject({
272
275
  compaction: compaction.optional(),
273
276
  budget: budget.optional(),
274
277
  profiles: z.array(profile).optional(),
275
- report: z.strictObject({ baselines: z.array(z.string()).optional() }).optional(),
278
+ report: z.strictObject({ baselines: z.array(z.string()).optional(), dailySummary: z.boolean().optional() }).optional(),
276
279
  digest: z
277
280
  .strictObject({
278
281
  enabled: z.boolean().optional(),
@@ -232,6 +232,15 @@ export interface FilterConfig {
232
232
  * once a week of verdicts is in the report.
233
233
  */
234
234
  feedbackWeight: number;
235
+ /**
236
+ * Count a verdict toward a model's trust only when routing the same task
237
+ * type the judged turn was (the ledger's `task`: coding, vision,
238
+ * documentation, data, chat). A model that writes good code but bad prose
239
+ * then keeps its coding trust. Verdicts on turns with no recorded task
240
+ * count for every task. Off by default: verdicts are scarce, and pooling
241
+ * them converges sooner.
242
+ */
243
+ feedbackByTask: boolean;
235
244
  /** Attempts required before `minTrust` is enforced against a model. */
236
245
  minTrustSamples: number;
237
246
  /**
@@ -589,6 +598,13 @@ export interface ReportConfig {
589
598
  * are skipped.
590
599
  */
591
600
  baselines: string[];
601
+ /**
602
+ * Post a one-screen summary of the last 24 hours (spend, top models, cache
603
+ * hit, escalations, soft-failure spikes, Ollama meter) into the transcript
604
+ * at the first omp session start of each day. `/router summary` shows it
605
+ * on demand regardless.
606
+ */
607
+ dailySummary: boolean;
592
608
  }
593
609
 
594
610
  export interface BudgetConfig {
@@ -746,6 +762,17 @@ export interface CompactionConfig {
746
762
  keepTailBytes: number;
747
763
  /** Elide an older tool result when a newer call to the same resource supersedes it. */
748
764
  elideSupersededReads: boolean;
765
+ /**
766
+ * Summarising compaction: when the plan gains an edit, a cheap model
767
+ * (`digest.tier` / `digest.model`, under `digest.maxCostUsd` and
768
+ * `digest.timeoutMs`) digests the tool result instead of it being cut to
769
+ * head+tail or a stub. The digest is stored on the edit, so the bytes sent
770
+ * stay identical on later turns. Applies when the turn routed at or above
771
+ * `digest.fromTier`; does not need `digest.enabled`.
772
+ */
773
+ digestToolResults: boolean;
774
+ /** Digests per turn at most; the rest of a plan's new edits stay plain until a later turn. */
775
+ digestMaxPerTurn: number;
749
776
  /** Collapse byte-identical repeated tool results to a single copy. */
750
777
  collapseDuplicateResults: boolean;
751
778
  }
@@ -27,6 +27,7 @@ import type {
27
27
  ModelCacheReliability,
28
28
  ModelLatency,
29
29
  ModelTrust,
30
+ SoftFailureSpike,
30
31
  UsageCounts,
31
32
  } from "./types.ts";
32
33
 
@@ -35,6 +36,20 @@ const MIN_CALIBRATION_SAMPLES = 20;
35
36
  /** Calibration samples outside this bytes-per-token band are provider accounting quirks, not tokenizer facts. */
36
37
  const MIN_SANE_BYTES_PER_TOKEN = 1.5;
37
38
  const MAX_SANE_BYTES_PER_TOKEN = 8;
39
+ /**
40
+ * Soft-failure spike detection (visibility only). A model is spiking when, over
41
+ * the recent window, it has at least SPIKE_MIN_DISPATCHES dispatches, at least
42
+ * SPIKE_MIN_FAILURES of them failed, its failure rate is at least
43
+ * SPIKE_MIN_RATE, and that rate is at least SPIKE_RATIO × its own baseline
44
+ * rate over the preceding window (a model with no baseline failures spikes on
45
+ * the absolute floor alone).
46
+ */
47
+ const SPIKE_RECENT_MS = 60 * 60_000;
48
+ const SPIKE_BASELINE_MS = 7 * 24 * 60 * 60_000;
49
+ const SPIKE_MIN_DISPATCHES = 5;
50
+ const SPIKE_MIN_FAILURES = 3;
51
+ const SPIKE_MIN_RATE = 0.25;
52
+ const SPIKE_RATIO = 2;
38
53
  /** Escalated attempts needed before their measured cost is trusted. */
39
54
  const MIN_ESCALATION_SAMPLES = 10;
40
55
  /** The escalation-cost aggregate scans a window of rows; memoised for this long. */
@@ -334,11 +349,23 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
334
349
  `SELECT ${FEEDBACK_SELECT} FROM feedback f JOIN ledger l ON l.id = f.ledger_id WHERE f.slug = ? AND l.harness_id = ? AND f.created_at_ms > ?`,
335
350
  );
336
351
  const allFeedbackStmt = db.query(`SELECT f.slug, ${FEEDBACK_SELECT} FROM feedback f WHERE f.created_at_ms > ? GROUP BY f.slug`);
337
- const feedbackFor = (slug: string, harnessId: string | undefined, cutoff: number): FeedbackRow | null => {
352
+ // Task-scoped variants (filters.feedbackByTask): the judged turn's task
353
+ // must match, or be unrecorded (older rows, or a turn that never classified).
354
+ const feedbackTaskStmt = db.query(
355
+ `SELECT ${FEEDBACK_SELECT} FROM feedback f JOIN ledger l ON l.id = f.ledger_id WHERE f.slug = ? AND (l.task = ? OR l.task IS NULL) AND f.created_at_ms > ?`,
356
+ );
357
+ const feedbackHarnessTaskStmt = db.query(
358
+ `SELECT ${FEEDBACK_SELECT} FROM feedback f JOIN ledger l ON l.id = f.ledger_id WHERE f.slug = ? AND l.harness_id = ? AND (l.task = ? OR l.task IS NULL) AND f.created_at_ms > ?`,
359
+ );
360
+ const feedbackFor = (slug: string, harnessId: string | undefined, cutoff: number, task?: string): FeedbackRow | null => {
338
361
  if (cfg.filters.feedbackWeight <= 0) return null;
339
- return harnessId !== undefined && harnessId !== ""
340
- ? (feedbackHarnessStmt.get(slug, harnessId, cutoff) as FeedbackRow | null)
341
- : (feedbackStmt.get(slug, cutoff) as FeedbackRow | null);
362
+ const byHarness = harnessId !== undefined && harnessId !== "";
363
+ if (cfg.filters.feedbackByTask && task !== undefined && task !== "") {
364
+ return byHarness
365
+ ? (feedbackHarnessTaskStmt.get(slug, harnessId, task, cutoff) as FeedbackRow | null)
366
+ : (feedbackTaskStmt.get(slug, task, cutoff) as FeedbackRow | null);
367
+ }
368
+ return byHarness ? (feedbackHarnessStmt.get(slug, harnessId, cutoff) as FeedbackRow | null) : (feedbackStmt.get(slug, cutoff) as FeedbackRow | null);
342
369
  };
343
370
  const latencyStmt = db.query(
344
371
  `SELECT ${LATENCY_SELECT} FROM (SELECT * FROM ledger WHERE slug = ? ORDER BY created_at_ms DESC LIMIT ${LATENCY_WINDOW_ROWS})`,
@@ -363,6 +390,19 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
363
390
  FROM ledger WHERE attempt > 0 AND error IS NULL AND created_at_ms >= ?`,
364
391
  );
365
392
  let escalationMemo: { atMs: number; windowDays: number; value: EscalationCost | null } | null = null;
393
+ // Per-model failure counts over two adjacent windows: [recentStart, now] and
394
+ // [baselineStart, recentStart). Wasted rows (the failed attempt a retry
395
+ // replaced) stay in: they ARE the soft failures being counted. Digest side
396
+ // calls are excluded: they are not the session's turns.
397
+ const softFailureStmt = db.query(
398
+ `SELECT COALESCE(served_slug, slug) AS slug,
399
+ SUM(CASE WHEN created_at_ms >= $recentStart THEN 1 ELSE 0 END) AS recent_n,
400
+ SUM(CASE WHEN created_at_ms >= $recentStart AND (escalation_signal IS NOT NULL OR (${ATTRIBUTABLE_ERROR})) THEN 1 ELSE 0 END) AS recent_f,
401
+ SUM(CASE WHEN created_at_ms < $recentStart THEN 1 ELSE 0 END) AS base_n,
402
+ SUM(CASE WHEN created_at_ms < $recentStart AND (escalation_signal IS NOT NULL OR (${ATTRIBUTABLE_ERROR})) THEN 1 ELSE 0 END) AS base_f
403
+ FROM ledger WHERE created_at_ms >= $baselineStart AND created_at_ms <= $now AND requested_model <> 'digest'
404
+ GROUP BY COALESCE(served_slug, slug)`,
405
+ );
366
406
  const cacheMetaStmt = db.query("SELECT fetched_at_ms FROM catalog_cache WHERE id = 1");
367
407
  const cachePayloadStmt = db.query("SELECT payload FROM catalog_cache WHERE id = 1");
368
408
 
@@ -466,7 +506,7 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
466
506
  return computeBlendedRate(db, cfg, windowDays);
467
507
  },
468
508
 
469
- trust(slug: string, harnessId?: string): ModelTrust | null {
509
+ trust(slug: string, harnessId?: string, task?: string): ModelTrust | null {
470
510
  // Read the window at CALL time, not at construction: hot reload mutates
471
511
  // the shared config object in place, so a pinned value would ignore an
472
512
  // edit until restart. 0 => cutoff 0 => every row qualifies.
@@ -476,7 +516,7 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
476
516
  ? (trustHarnessStmt.get(slug, harnessId, cutoff) as TrustRow | null)
477
517
  : (trustStmt.get(slug, cutoff) as TrustRow | null);
478
518
  if (row === null || row.attempts === 0) return null;
479
- return toTrust(slug, row, feedbackFor(slug, harnessId, cutoff), cfg.filters.feedbackWeight);
519
+ return toTrust(slug, row, feedbackFor(slug, harnessId, cutoff, task), cfg.filters.feedbackWeight);
480
520
  },
481
521
 
482
522
  allTrust(): ModelTrust[] {
@@ -497,7 +537,7 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
497
537
  if (row === null) return null;
498
538
  return toLatency(slug, row);
499
539
  },
500
- signals(slugs: readonly string[], harnessId?: string): Map<string, LedgerSignals> {
540
+ signals(slugs: readonly string[], harnessId?: string, task?: string): Map<string, LedgerSignals> {
501
541
  const cutoff = cfg.filters.trustWindowDays > 0 ? Date.now() - cfg.filters.trustWindowDays * DAY_MS : 0;
502
542
  const hasHarness = harnessId !== undefined && harnessId !== "";
503
543
  const out = new Map<string, LedgerSignals>();
@@ -509,7 +549,7 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
509
549
  ? (latencyHarnessStmt.get(slug, harnessId) as LatencyRow | null)
510
550
  : (latencyStmt.get(slug) as LatencyRow | null);
511
551
  out.set(slug, {
512
- trust: trustRow === null || trustRow.attempts === 0 ? null : toTrust(slug, trustRow, feedbackFor(slug, harnessId, cutoff), cfg.filters.feedbackWeight),
552
+ trust: trustRow === null || trustRow.attempts === 0 ? null : toTrust(slug, trustRow, feedbackFor(slug, harnessId, cutoff, task), cfg.filters.feedbackWeight),
513
553
  latency: latencyRow === null ? null : toLatency(slug, latencyRow),
514
554
  });
515
555
  }
@@ -552,6 +592,33 @@ export function createLedger(db: Database, cfg: RouterConfig): Ledger {
552
592
  const rows = recentStmt.all(limit) as LedgerRow[];
553
593
  return rows.map(toEntry);
554
594
  },
595
+ softFailureSpikes(nowMs = Date.now(), recentMs = SPIKE_RECENT_MS, baselineMs = SPIKE_BASELINE_MS): SoftFailureSpike[] {
596
+ const rows = softFailureStmt.all({ $now: nowMs, $recentStart: nowMs - recentMs, $baselineStart: nowMs - recentMs - baselineMs }) as {
597
+ slug: string;
598
+ recent_n: number;
599
+ recent_f: number;
600
+ base_n: number;
601
+ base_f: number;
602
+ }[];
603
+ const spikes: SoftFailureSpike[] = [];
604
+ for (const r of rows) {
605
+ if (r.recent_n < SPIKE_MIN_DISPATCHES || r.recent_f < SPIKE_MIN_FAILURES) continue;
606
+ const recentRate = r.recent_f / r.recent_n;
607
+ const baselineRate = r.base_n > 0 ? r.base_f / r.base_n : 0;
608
+ if (recentRate < SPIKE_MIN_RATE || recentRate < SPIKE_RATIO * baselineRate) continue;
609
+ spikes.push({
610
+ slug: r.slug,
611
+ recentDispatches: r.recent_n,
612
+ recentFailures: r.recent_f,
613
+ recentRate,
614
+ baselineDispatches: r.base_n,
615
+ baselineFailures: r.base_f,
616
+ baselineRate,
617
+ });
618
+ }
619
+ spikes.sort((a, b) => b.recentRate - a.recentRate || b.recentFailures - a.recentFailures);
620
+ return spikes;
621
+ },
555
622
  providerSpendSince(slugPrefix: string, sinceMs: number): number {
556
623
  const row = providerSpendStmt.get(sinceMs, `${slugPrefix}%`) as { total: number } | null;
557
624
  return row?.total ?? 0;
@@ -185,14 +185,19 @@ function toRow(r: RawRow, windowSpend: number): ReportRow {
185
185
  */
186
186
  export function buildUsageReport(
187
187
  db: Database,
188
- opts: { windowDays: number; harnessId?: string; nowMs?: number; baselines?: readonly BaselinePrice[] },
188
+ opts: { windowDays: number; harnessId?: string; nowMs?: number; baselines?: readonly BaselinePrice[]; /** Exclusive upper bound; default open-ended. */ untilMs?: number },
189
189
  ): UsageReport {
190
190
  const nowMs = opts.nowMs ?? Date.now();
191
191
  const windowDays = Math.max(1, opts.windowDays);
192
192
  const sinceMs = nowMs - windowDays * 86_400_000;
193
193
  const harnessId = opts.harnessId ?? "";
194
- const where = harnessId === "" ? "created_at_ms >= $since" : "created_at_ms >= $since AND harness_id = $harness";
195
- const bind = harnessId === "" ? { $since: sinceMs } : { $since: sinceMs, $harness: harnessId };
194
+ const untilMs = opts.untilMs;
195
+ const where = [
196
+ "created_at_ms >= $since",
197
+ ...(untilMs === undefined ? [] : ["created_at_ms < $until"]),
198
+ ...(harnessId === "" ? [] : ["harness_id = $harness"]),
199
+ ].join(" AND ");
200
+ const bind = { $since: sinceMs, ...(untilMs === undefined ? {} : { $until: untilMs }), ...(harnessId === "" ? {} : { $harness: harnessId }) };
196
201
 
197
202
  const t = db
198
203
  .query(