@coreplane/switchboard 1.248.0 → 1.249.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/assets/config/config.example.yaml +44 -18
  2. package/dist/assets/deploy/cloudflare/preflight.mjs +21 -19
  3. package/dist/assets/deploy/cloudflare-memory/worker.ts +31 -0
  4. package/dist/assets/deploy/cloudflare-resident/drain.ts +109 -0
  5. package/dist/assets/deploy/cloudflare-resident/refresh.ts +5 -3
  6. package/dist/assets/deploy/cloudflare-resident/threadErr.ts +32 -3
  7. package/dist/assets/deploy/cloudflare-resident/worker.ts +235 -13
  8. package/dist/assets/deploy/cloudflare-sandbox/worker.ts +56 -9
  9. package/dist/assets/deploy/secrets.manifest.json +6 -0
  10. package/dist/assets/package-lock.json +3 -3
  11. package/dist/assets/package.json +1 -1
  12. package/dist/assets/source.json +3 -3
  13. package/dist/assets/src/agents/registry.ts +4 -4
  14. package/dist/assets/src/core/budgets.ts +24 -0
  15. package/dist/assets/src/core/coordinator/contract.ts +4 -3
  16. package/dist/assets/src/core/coordinator/driver.ts +40 -7
  17. package/dist/assets/src/core/modelCard.ts +348 -0
  18. package/dist/assets/src/core/modelPricing.ts +14 -5
  19. package/dist/assets/src/core/modelRegistry.ts +51 -0
  20. package/dist/assets/src/core/provider.ts +103 -0
  21. package/dist/assets/src/core/refusal.ts +181 -0
  22. package/dist/assets/src/core/runEvents.ts +43 -0
  23. package/dist/assets/src/core/ship/coordinator.ts +80 -12
  24. package/dist/assets/src/core/ship/handoff.ts +54 -19
  25. package/dist/assets/src/core/trace/workerTrace.ts +9 -3
  26. package/dist/assets/src/core/types.ts +327 -0
  27. package/dist/assets/src/deploy/liveGate.ts +35 -0
  28. package/dist/assets/src/deploy/restart.ts +12 -11
  29. package/dist/assets/src/execution/residentRefresh.ts +26 -1
  30. package/dist/assets/src/execution/sandboxErrors.ts +122 -4
  31. package/dist/assets/web/dist/.vite/manifest.json +30 -30
  32. package/dist/assets/web/dist/assets/DeliveryPage-DF4aQypG.js +1 -0
  33. package/dist/assets/web/dist/assets/{HomePage-DYxC0izY.js → HomePage-BpQRky8B.js} +1 -1
  34. package/dist/assets/web/dist/assets/{ResidentDetailPage-DLIpWYOc.js → ResidentDetailPage-BIUXyz6K.js} +1 -1
  35. package/dist/assets/web/dist/assets/{ResidentsIndexPage-6LipuDjR.js → ResidentsIndexPage-BZymgSAb.js} +1 -1
  36. package/dist/assets/web/dist/assets/{RunFoldRow-V-iSy64e.js → RunFoldRow-3m4CPRI4.js} +1 -1
  37. package/dist/assets/web/dist/assets/{RunRoutePage-DUalB1u2.js → RunRoutePage-bgkjkA0p.js} +3 -3
  38. package/dist/assets/web/dist/assets/{RunsIndexPage-B9Ba1KdD.js → RunsIndexPage-8S944AzB.js} +1 -1
  39. package/dist/assets/web/dist/assets/{ScheduledPage-KdjLtD_7.js → ScheduledPage-8bBtG9y3.js} +1 -1
  40. package/dist/assets/web/dist/assets/{SettingsPage-IT5l_NaL.js → SettingsPage-DQeNvfaV.js} +1 -1
  41. package/dist/assets/web/dist/assets/{StatusDot-DBHAl4Il.js → StatusDot-BPE5syBa.js} +1 -1
  42. package/dist/assets/web/dist/assets/{Tooltip-_LEjptLV.js → Tooltip-DkoeZfTs.js} +1 -1
  43. package/dist/assets/web/dist/assets/UnitRoutePage-BUzw--Ii.js +1 -0
  44. package/dist/assets/web/dist/assets/{dist-BcYPGOBL.js → dist-D11y9ZJ4.js} +1 -1
  45. package/dist/assets/web/dist/assets/{main-DUfSE0dj.js → main-B6LcgNM6.js} +2 -2
  46. package/dist/cli.js +2475 -1021
  47. package/package.json +1 -1
  48. package/dist/assets/web/dist/assets/DeliveryPage-NP4g6bQd.js +0 -1
  49. package/dist/assets/web/dist/assets/UnitRoutePage-DPBsvGPR.js +0 -1
@@ -81,6 +81,12 @@
81
81
  "workers": ["resident"],
82
82
  "note": "Read-only /residents + debug routes; what the resident deploy preflight needs (CI holds this one). Self-minted."
83
83
  },
84
+ {
85
+ "name": "RESIDENT_DRAIN_TOKEN",
86
+ "workers": ["resident"],
87
+ "optional": true,
88
+ "note": "Drain-only bearer: POST /drain and /undrain, nothing else (docs/reference/specs/resident-repos.md item 69). What the release deploy holds (CI) so the resident step can close the fleet to new runs and land once the runs in flight end, without the admin bearer. Self-minted; optional — unset, only admin can drain and the deploy waits without one."
89
+ },
84
90
  {
85
91
  "name": "R2_ACCESS_KEY_ID",
86
92
  "workers": ["resident", "sandbox"],
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.248.0",
3
+ "version": "1.249.1",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "switchboard",
9
- "version": "1.248.0",
9
+ "version": "1.249.1",
10
10
  "license": "Apache-2.0",
11
11
  "workspaces": [
12
12
  "web",
@@ -20445,7 +20445,7 @@
20445
20445
  },
20446
20446
  "packages/switchboard": {
20447
20447
  "name": "@coreplane/switchboard",
20448
- "version": "1.248.0",
20448
+ "version": "1.249.1",
20449
20449
  "license": "Apache-2.0",
20450
20450
  "dependencies": {
20451
20451
  "@earendil-works/pi-ai": "0.85.1",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.248.0",
3
+ "version": "1.249.1",
4
4
  "private": true,
5
5
  "description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
6
6
  "license": "Apache-2.0",
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "1.248.0",
3
- "commit": "e2ba9e4c1b52ac15c841be8ad086b017d4a14ced",
4
- "builtAt": "2026-09-18T04:34:16.576Z"
2
+ "version": "1.249.1",
3
+ "commit": "7fa16da36339e9c0b04e3f9c037002efcf5975f4",
4
+ "builtAt": "2026-09-18T08:08:43.148Z"
5
5
  }
@@ -188,7 +188,7 @@ const UNIT_CONTRACT = `UNIT CONTRACT: when your first user turn carries a \`${CO
188
188
  // it to the unit's board issue without a person writing it there. Both coding
189
189
  // prompts carry this verbatim, right after the contract paragraph, so the
190
190
  // sandbox and resident children read the same rule.
191
- const UNIT_HANDOFF = `UNIT HANDOFF: when your first user turn carries a \`${CONTRACT_HEADING}\` block, call the submit_handoff tool once, after submit_pr_description and before your final message, with the typed handoff — deviations: where you departed from the unit as written (from, to, why); followUps: what you found and did not do, and where it belongs (what, where); unproven: which of the unit's test scenarios or criteria you could not prove, and why (criterion, why). Switchboard records it on the run and posts it to the unit's board issue, where a person decides each row's disposition; you never edit the plan's ledger yourself. An empty handoff is submitted as three empty lists, never skipped — a missing handoff reads as an unfinished run, not as nothing to say. Without a \`${CONTRACT_HEADING}\` block, do not call it.`;
191
+ const UNIT_HANDOFF = `UNIT HANDOFF: when your first user turn carries a \`${CONTRACT_HEADING}\` block, call the submit_handoff tool once, after submit_pr_description and before your final message, with the typed handoff — deviations: where you departed from the unit as written (from, to, why); followUps: what you found and did not do, and where it belongs (what, where); unproven: which of the unit's test scenarios or criteria you could not prove, and why (criterion, why); landed (optional): what of the unit was already on the base when you began, and the pull request or commit that carries it (what, where) — when the whole unit is already there, push nothing of your own and open no pull request; the handoff's landed rows end the unit done. Switchboard records it on the run and posts it to the unit's board issue, where a person decides each row's disposition; you never edit the plan's ledger yourself. An empty handoff is submitted as three empty lists, never skipped — a missing handoff reads as an unfinished run, not as nothing to say. Without a \`${CONTRACT_HEADING}\` block, do not call it.`;
192
192
 
193
193
  // What both execution images carry beyond git and the package managers
194
194
  // (docs/reference/specs/execution.md item 10), said in one sentence by every
@@ -544,7 +544,7 @@ Your tools work without a workspace: the GitHub tools — \`github_repos\` (the
544
544
 
545
545
  ${statusCardRule('"Read the issue and its thread", "Post the comment"')} A one-step answer needs no checklist; post one when the request has steps the person would wait on.
546
546
 
547
- You cannot run commands, clone repositories, edit code, or review pull requests, and you cannot search the web. Other Switchboard agents can: for a code change or a pull request tell the user to re-send with \`agent:ship\` (it makes the change, opens the PR and loops review); for a PR review, \`agent:review\`; for a web-research question, \`agent:research\` (e.g. "\`agent:ship in acme/api: fix the failing login test\`", "\`agent:research compare X and Y\`"). Delete an issue only when the user explicitly asked to delete it (closing is an update).`;
547
+ You cannot run commands, clone repositories, edit code, or review pull requests, and you cannot search the web. Other Switchboard agents can, and a plain message reaches them by itself: for a code change or a pull request, say what you found and that the change is not yours to make, and that asking for it in plain words in a new message — "in acme/api: fix the failing login test" — starts the agent that makes the change, opens the PR and loops review; the same for a PR review ("review <PR URL>") and a web-research question ("compare X and Y on the web"). Never hand back a command or an \`agent:…\` line for the person to type: describe the ask in their words. Delete an issue only when the user explicitly asked to delete it (closing is an update).`;
548
548
 
549
549
  // The explore agent (docs/reference/specs/agent-explore.md): a long, read-only
550
550
  // investigation — "run our CI locally and validate the claims", "how long does
@@ -567,9 +567,9 @@ THE DELIVERABLE IS A CLAIM TABLE. Turn the request into the claims it makes or a
567
567
 
568
568
  TIME. Your budget is up to two hours — less when a boundary or the request's \`budget:\` directive clipped it, which the runtime-config block above says — and the wrap-up warning tells you when to stop starting new checks. A single command is capped at ${BASH_TIMEOUT_MAX_MS / 60_000} minutes (pass the bash tool's \`timeoutMs\`, up to ${BASH_TIMEOUT_MAX_MS} ms, for a long one). A job that needs longer — a full suite, a build, a pipeline run — is started detached and polled across tool calls: \`setsid -f sh -c '<command> > /tmp/job.log 2>&1; echo $? > /tmp/job.exit'\`, then \`tail -n 40 /tmp/job.log\` and \`cat /tmp/job.exit\` on later calls (a plain background job dies with the command that started it; a \`setsid -f\` job outlives it). Batch commands into few tool calls; never explore file by file.
569
569
 
570
- READ-ONLY: NEVER open a pull request, and never commit or push — no branch, no \`gh pr create\`, no PR or issue write of any kind. You hold a read credential and your job is to find out, not to change. If the investigation shows a change is needed, say exactly what and where in your write-up and point the user at \`agent:ship\` (it makes the change, opens the PR and loops review).
570
+ READ-ONLY: NEVER open a pull request, and never commit or push — no branch, no \`gh pr create\`, no PR or issue write of any kind. You hold a read credential and your job is to find out, not to change. If the investigation shows a change is needed, say exactly what and where in your write-up, and that asking for it in plain words in a new message ("in <owner/name>: <the change>") starts the agent that makes the change, opens the PR and loops review — never hand back a command or an \`agent:…\` line to type.
571
571
 
572
- You cannot attach or post files: your whole answer is text. Never say a file is attached or below — name its path in the workspace and describe it (what it shows, its size) instead; a person who needs the file itself asks \`agent:coding\`, which can attach.
572
+ You cannot attach or post files: your whole answer is text. Never say a file is attached or below — name its path in the workspace and describe it (what it shows, its size) instead; a person who needs the file itself asks for it to be attached in a new message, which reaches a preset that can.
573
573
 
574
574
  ${statusCardRule('"Clone and install", "Time the full suite"')}
575
575
 
@@ -18,6 +18,10 @@
18
18
 
19
19
  export const MINUTE_MS = 60_000;
20
20
 
21
+ /** Minutes → milliseconds, for a duration a request names in minutes (a drain's
22
+ * length): the multiplication lives here so no other file holds it. */
23
+ export const minutesToMs = (minutes: number): number => minutes * MINUTE_MS;
24
+
21
25
  /** One calendar day in milliseconds: the unit the daily cost and delivery ranges
22
26
  * step by (`dayOf`, the day count of a range). A day is not a lease, but it is
23
27
  * a duration, and every duration is read from this table rather than written
@@ -61,6 +65,26 @@ export const ATTACH_REQUEST_MIN_MS = 30_000;
61
65
  * that opened the question stands. */
62
66
  export const HARNESS_PROBE_WAIT_MS = 5 * MINUTE_MS;
63
67
 
68
+ /** The resident fleet drain (docs/decisions/0059; docs/reference/specs/resident-repos.md
69
+ * item 69; docs/reference/specs/release-and-deploy.md item 31). A deploy closes
70
+ * the fleet to new runs and waits for the runs in flight to end: `deployWaitMaxMs`
71
+ * is that wait, past a coding child's whole lease (its ask plus its write-up),
72
+ * the longest a run in flight can outlive the drain's start; the drain itself
73
+ * lasts the wait plus `marginMinutes`, and the registry caps any drain at
74
+ * `maxMinutes` so one nobody lifted is an hour and a half, not a day. A run
75
+ * asked during a drain waits at its attach one `pollMs` at a time under its
76
+ * own lease less `leaseReserveMs` (what the attach and the work after it
77
+ * need), `waitMaxMs` with no lease to clip it. */
78
+ export const DRAIN = {
79
+ pollMs: 30_000,
80
+ waitMaxMs: 60 * MINUTE_MS,
81
+ leaseReserveMs: 10 * MINUTE_MS,
82
+ deployWaitMaxMs: 60 * MINUTE_MS,
83
+ marginMinutes: 5,
84
+ maxMinutes: 90,
85
+ defaultMinutes: 60,
86
+ } as const;
87
+
64
88
  /** The presets that run the tool loop, and the one pipeline preset. */
65
89
  export const LOOP_PRESETS = ["general", "coding", "review", "research", "explore", "conductor"] as const;
66
90
  export type LoopPreset = (typeof LOOP_PRESETS)[number];
@@ -233,9 +233,10 @@ export interface CoordinatorUnit {
233
233
  /** The unit's thread, once opened; a generated plan's is the requesting thread from the start. */
234
234
  threadKey?: string;
235
235
  sourceUrl?: string;
236
- /** The unit's review thread, opened once beside the unit's thread: every
237
- * review round runs there (record 0034), so the review child's worktree is
238
- * readonly and its own and no round wipes the coding thread's. */
236
+ /** Retired (record 0055): a bot before it opened a review thread beside the
237
+ * unit's and ran every review round there. A row that carries one keeps its
238
+ * review rounds there; a new row never gets one — every child of the unit
239
+ * runs in the unit's thread. */
239
240
  reviewThread?: { threadKey: string; sourceUrl?: string };
240
241
  /** The unit's board issue in the repository, when one titled by the unit id exists — the handoff's destination. */
241
242
  issue?: number;
@@ -307,16 +307,27 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
307
307
  };
308
308
  }
309
309
 
310
+ /** The check runs a pr-check answer carries at the head (agent-ship item 9), shape-checked. */
311
+ function isCommitChecks(v: unknown): v is { total: number; pending: string[]; failed: string[] } {
312
+ if (typeof v !== "object" || v === null) return false;
313
+ const c = v as Record<string, unknown>;
314
+ const names = (x: unknown) => Array.isArray(x) && x.every((n) => typeof n === "string");
315
+ return typeof c.total === "number" && names(c.pending) && names(c.failed);
316
+ }
317
+
310
318
  function prCheckReturn(step: string, a: BotAnswer): StepReturn {
311
319
  const { ok, state, prNumber, url, headSha, sha, mergedAt, at } = a.body;
312
320
  if (ok === true && state === "none") {
313
- const { unrecovered } = a.body;
321
+ const { unrecovered, aheadOfBase } = a.body;
314
322
  return {
315
323
  type: "pr-check",
316
324
  step,
317
325
  pr: {
318
326
  state: "none",
319
327
  ...(unrecovered === "no_commits" || unrecovered === "no_base" ? { unrecovered } : {}),
328
+ // The branch's commits over the base, when the bot could read them
329
+ // (agent-ship item 12): zero is the `already_landed` ending's fact.
330
+ ...(typeof aheadOfBase === "number" ? { aheadOfBase } : {}),
320
331
  },
321
332
  at,
322
333
  };
@@ -331,6 +342,7 @@ function prCheckReturn(step: string, a: BotAnswer): StepReturn {
331
342
  url,
332
343
  ...(typeof headSha === "string" ? { headSha } : {}),
333
344
  ...(typeof a.body.autoMergeEnabled === "boolean" ? { autoMergeEnabled: a.body.autoMergeEnabled } : {}),
345
+ ...(isCommitChecks(a.body.checks) ? { checks: a.body.checks } : {}),
334
346
  },
335
347
  at,
336
348
  };
@@ -561,13 +573,20 @@ async function runUnit(
561
573
  `${prefix}/end/pr-facts`,
562
574
  answerOf(
563
575
  "pr-check",
564
- await step.do(`${prefix}/end/pr-facts`, STEP_CONFIG, () => call(bot, "pr-check", tag)),
576
+ await step.do(`${prefix}/end/pr-facts`, STEP_CONFIG, () =>
577
+ call(bot, "pr-check", { ...tag, checks: true }),
578
+ ),
565
579
  ),
566
580
  );
567
581
  if (check.type === "pr-check" && check.pr.state === "merged")
568
582
  endFacts = { merged: { sha: check.pr.sha, mergedAt: check.pr.mergedAt } };
569
- else if (check.type === "pr-check" && check.pr.state === "open" && check.pr.autoMergeEnabled !== undefined)
570
- endFacts = { autoMergeEnabled: check.pr.autoMergeEnabled };
583
+ else if (check.type === "pr-check" && check.pr.state === "open")
584
+ endFacts = {
585
+ ...(check.pr.autoMergeEnabled !== undefined ? { autoMergeEnabled: check.pr.autoMergeEnabled } : {}),
586
+ // The checks at the approved head (record 0055): the report's
587
+ // headline is a claim about them, never "merge-ready" over a red one.
588
+ ...(check.pr.checks !== undefined ? { checks: check.pr.checks } : {}),
589
+ };
571
590
  } catch {
572
591
  // the report simply omits the fact
573
592
  }
@@ -650,7 +669,9 @@ async function walk(step: StepRunner, bot: CoordinatorBot, instanceId: string):
650
669
  });
651
670
  }
652
671
  endings[next] = ending.kind;
653
- cursor = settleUnit(graph, cursor, next, ending.kind === "merged" ? "done" : "failed");
672
+ // A unit is done for its dependents when the base carries its scope: the
673
+ // runner's merge, or a scope that had already landed before the attempt.
674
+ cursor = settleUnit(graph, cursor, next, isSettledDone(ending.kind) ? "done" : "failed");
654
675
  }
655
676
  // Blocked units, in the plan's order: each told its own ending, so the rows
656
677
  // and the summary say why it never ran. Every blocked unit's ending is known
@@ -669,15 +690,27 @@ async function walk(step: StepRunner, bot: CoordinatorBot, instanceId: string):
669
690
  await step.do(`${id}/end`, STEP_CONFIG, () => call(bot, "unit-end", body));
670
691
  }
671
692
  if (!cursorFinished(cursor)) throw new Error(`the plan's cursor did not finish: ${JSON.stringify(cursor.status)}`);
672
- const settled = (kind: string) => kind === "merged" || kind === "merge_ready";
673
693
  return {
674
694
  instance: instanceId,
675
695
  ...(plan.planId !== undefined ? { planId: plan.planId } : {}),
676
696
  units: endings,
677
- outcome: cursor.order.every((id) => settled(endings[id] ?? "")) ? "completed" : "failed",
697
+ outcome: cursor.order.every((id) => isSettledOutcome(endings[id] ?? "")) ? "completed" : "failed",
678
698
  };
679
699
  }
680
700
 
701
+ /** The endings whose unit's scope is on the base, so its dependents run on a
702
+ * base that carries it (agent-ship item 12): the runner's merge, a merge found
703
+ * already made, or a scope that had landed before the attempt. */
704
+ function isSettledDone(kind: string): boolean {
705
+ return kind === "merged" || kind === "already_landed";
706
+ }
707
+
708
+ /** The endings a plan closes ✅ over: every `isSettledDone` one, plus
709
+ * merge-ready — the work stands and a person's merge is the only gate left. */
710
+ function isSettledOutcome(kind: string): boolean {
711
+ return isSettledDone(kind) || kind === "merge_ready";
712
+ }
713
+
681
714
  /** The Workflow's body. The finish is asked on every path — as `failed`, best
682
715
  * effort, when the walk threw — and the cause is rethrown, so the instance's
683
716
  * own status says what happened and the parent's record exists either way. */
@@ -0,0 +1,348 @@
1
+ // The model card (record 0052): the resolved description of one
2
+ // `<block>/<model>` for one run — the wire, the vendor, the effort levels, the
3
+ // output-cap field, the window, the input kinds, the cache rule and the price
4
+ // source — layered operator over registry over wire defaults, each field with
5
+ // the layer that named it. `decideControls` then answers, before the first
6
+ // call, whether each control is native, degraded with a reason, or refused.
7
+ //
8
+ // Node-free on purpose: the resolver reads a `CardRegistry` seam, never pi's
9
+ // files, so the memory Worker's build can carry this leaf. The production
10
+ // catalog lives in ./modelRegistry.ts.
11
+
12
+ import { EFFORT_LEVELS, type Effort } from "../effort.js";
13
+ import { vendorOf, wireOf, type ProviderConfig, type Wire } from "./provider.js";
14
+ import type { RegistryCard } from "./modelRegistry.js";
15
+
16
+ /** Which layer named a field: the operator's block, the registry card, or the
17
+ * wire's structural default (the one place a vendor's name may appear). */
18
+ export type Provenance = "operator" | "registry" | "wire";
19
+
20
+ /** One tier's wire word and how the card arrived at it: `named` when the
21
+ * layer names this tier itself (its own wire word, whatever the spelling —
22
+ * `LOW`, `default` — or pi's rule for an unlisted low/medium/high), false
23
+ * when the word is a lower tier's, reached by pi's fallback rule. */
24
+ export interface LevelWord {
25
+ word: string;
26
+ named: boolean;
27
+ }
28
+
29
+ /** The card's levels: each of our tiers → the wire's word with its standing,
30
+ * or `refused`; or `unknown` as a whole (no layer names this model's levels). */
31
+ export type LevelMap = Record<Effort, LevelWord | "refused"> | "unknown";
32
+ export type InputSupport = boolean | "unknown";
33
+ export type CacheRule = "automatic" | "markers" | "none" | "unknown";
34
+
35
+ export interface CardPrice {
36
+ input: number;
37
+ output: number;
38
+ cacheRead: number;
39
+ cacheWrite: number;
40
+ }
41
+
42
+ export interface ModelCard {
43
+ ref: string;
44
+ block: string;
45
+ model: string;
46
+ vendor: string;
47
+ wire: Wire;
48
+ levels: LevelMap;
49
+ /** The body field the output cap is spelled with on this model's wire. */
50
+ capField: string;
51
+ window: number;
52
+ inputs: { image: InputSupport; document: InputSupport };
53
+ cache: CacheRule;
54
+ /** The rate card when a layer names one; the meter applies it in a later slice. */
55
+ price?: CardPrice;
56
+ provenance: Record<"levels" | "capField" | "window" | "inputs" | "cache" | "price", Provenance>;
57
+ }
58
+
59
+ /** The catalog seam `resolveModelCard` reads — pi's registry behind
60
+ * ./modelRegistry.ts in production, a table in tests. */
61
+ export interface CardRegistry {
62
+ card(catalog: string | undefined, wire: Wire, model: string): RegistryCard | undefined;
63
+ }
64
+
65
+ /** The output-cap field each wire spells the cap with, unvouched. */
66
+ const WIRE_CAP_FIELD: Readonly<Record<Wire, string>> = {
67
+ "anthropic-messages": "max_tokens",
68
+ "openai-chat": "max_completion_tokens",
69
+ "openai-responses": "max_output_tokens",
70
+ };
71
+
72
+ /** pi's own default for a `models.json` card without a window: compacting
73
+ * early beats overflowing a window no layer can vouch for. */
74
+ export const UNKNOWN_WINDOW = 128_000;
75
+
76
+ /** The vendor-keyed cache table, the wire defaults' one vendor-name table
77
+ * (record 0052): the vendors whose caches are automatic, and Anthropic, whose
78
+ * cache needs explicit markers. A vendor not here is `unknown`. */
79
+ const VENDOR_CACHE: Readonly<Record<string, CacheRule>> = {
80
+ anthropic: "markers",
81
+ openai: "automatic",
82
+ deepseek: "automatic",
83
+ groq: "automatic",
84
+ moonshot: "automatic",
85
+ moonshotai: "automatic",
86
+ grok: "automatic",
87
+ xai: "automatic",
88
+ gemini: "automatic",
89
+ google: "automatic",
90
+ };
91
+
92
+ /** The highest tier below `tier` the map names (pi's fallback rule), or
93
+ * undefined when it names none. */
94
+ function highestNamedBelow(map: Record<string, string | null> | undefined, tier: Effort): string | undefined {
95
+ const at = EFFORT_LEVELS.indexOf(tier);
96
+ for (let i = at - 1; i >= 0; i--) {
97
+ const word = map?.[EFFORT_LEVELS[i]!];
98
+ if (typeof word === "string") return word;
99
+ }
100
+ return undefined;
101
+ }
102
+
103
+ /** A registry or operator level map read by pi's rule: `null` refuses a tier;
104
+ * `low`, `medium`, `high` are native unless null; `xhigh` and `max` are
105
+ * native only when named, else the highest named tier below them (refused
106
+ * when none is named). `reasoning: false` refuses every tier. */
107
+ function levelMapOf(map: Record<string, string | null> | undefined, reasoning: boolean | undefined): LevelMap {
108
+ const out = {} as Record<Effort, LevelWord | "refused">;
109
+ for (const tier of EFFORT_LEVELS) {
110
+ if (reasoning === false) {
111
+ out[tier] = "refused";
112
+ continue;
113
+ }
114
+ const word = map?.[tier];
115
+ if (word === null) out[tier] = "refused";
116
+ else if (typeof word === "string") out[tier] = { word, named: true };
117
+ else if (tier === "low" || tier === "medium" || tier === "high") out[tier] = { word: tier, named: true };
118
+ else {
119
+ const below = highestNamedBelow(map, tier);
120
+ out[tier] = below === undefined ? "refused" : { word: below, named: false };
121
+ }
122
+ }
123
+ return out;
124
+ }
125
+
126
+ function priceOf(
127
+ raw: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number } | undefined,
128
+ ): CardPrice | undefined {
129
+ if (!raw) return undefined;
130
+ const { input, output, cacheRead, cacheWrite } = raw;
131
+ if (input === undefined || output === undefined || cacheRead === undefined || cacheWrite === undefined)
132
+ return undefined;
133
+ return { input, output, cacheRead, cacheWrite };
134
+ }
135
+
136
+ /**
137
+ * One run's card: the operator's `models.<id>` override over the registry card
138
+ * the block's `catalog` names over the wire's structural defaults, each field
139
+ * carrying the layer that named it (record 0052). The block must exist —
140
+ * the dispatcher checks it first — and a ref the block does not cover still
141
+ * resolves, from the wire defaults, saying so.
142
+ */
143
+ export function resolveModelCard(
144
+ ref: string,
145
+ blocks: Readonly<Record<string, ProviderConfig>>,
146
+ registry: CardRegistry,
147
+ ): ModelCard {
148
+ const vendor = vendorOf(ref, blocks);
149
+ const block = blocks[vendor.block];
150
+ const wire = block ? wireOf(block) : "openai-chat";
151
+ const override = block?.models?.[vendor.model];
152
+ const card: RegistryCard | undefined = registry.card(block?.catalog ?? vendor.block, wire, vendor.model);
153
+
154
+ const levels: LevelMap =
155
+ override?.levels !== undefined
156
+ ? levelMapOf(override.levels, undefined)
157
+ : card?.thinkingLevelMap !== undefined || card?.reasoning !== undefined
158
+ ? levelMapOf(card?.thinkingLevelMap, card?.reasoning)
159
+ : "unknown";
160
+ const levelsProvenance: Provenance =
161
+ override?.levels !== undefined ? "operator" : levels === "unknown" ? "wire" : "registry";
162
+
163
+ const capField = override?.capField ?? (card?.compat?.maxTokensField as string | undefined) ?? WIRE_CAP_FIELD[wire];
164
+ const capProvenance: Provenance =
165
+ override?.capField !== undefined ? "operator" : card?.compat?.maxTokensField !== undefined ? "registry" : "wire";
166
+
167
+ const window = override?.window ?? card?.contextWindow ?? UNKNOWN_WINDOW;
168
+ const windowProvenance: Provenance =
169
+ override?.window !== undefined ? "operator" : card?.contextWindow !== undefined ? "registry" : "wire";
170
+
171
+ const image: InputSupport =
172
+ override?.inputs?.image ?? (card?.input !== undefined ? card.input.includes("image") : "unknown");
173
+ const document: InputSupport = override?.inputs?.document ?? "unknown";
174
+ const inputsProvenance: Provenance =
175
+ override?.inputs !== undefined ? "operator" : card?.input !== undefined ? "registry" : "wire";
176
+
177
+ const cache = override?.cache ?? VENDOR_CACHE[vendor.vendor] ?? "unknown";
178
+ const cacheProvenance: Provenance = override?.cache !== undefined ? "operator" : "wire";
179
+
180
+ const price = priceOf(override?.price) ?? priceOf(card?.cost);
181
+ const priceProvenance: Provenance =
182
+ priceOf(override?.price) !== undefined ? "operator" : price !== undefined ? "registry" : "wire";
183
+
184
+ return {
185
+ ref,
186
+ block: vendor.block,
187
+ model: vendor.model,
188
+ vendor: vendor.vendor,
189
+ wire,
190
+ levels,
191
+ capField,
192
+ window,
193
+ inputs: { image, document },
194
+ cache,
195
+ ...(price ? { price } : {}),
196
+ provenance: {
197
+ levels: levelsProvenance,
198
+ capField: capProvenance,
199
+ window: windowProvenance,
200
+ inputs: inputsProvenance,
201
+ cache: cacheProvenance,
202
+ price: priceProvenance,
203
+ },
204
+ };
205
+ }
206
+
207
+ /** What the request asks of the model's controls, for `decideControls`. */
208
+ export interface AskedControls {
209
+ effort?: Effort;
210
+ images?: number;
211
+ documents?: number;
212
+ }
213
+
214
+ export type ControlName = "effort" | "cap" | "inputs" | "window" | "cache";
215
+
216
+ /** One control's decision before the first call (record 0052): `native`
217
+ * (the wire's own word goes out, vouched for by a layer), `degraded` (the
218
+ * card cannot vouch — a fallback with `applied` differing from `asked`, or an
219
+ * unvouched send with `vouched: false`), or `refused` (the run never starts). */
220
+ export interface ControlDecision {
221
+ control: ControlName;
222
+ outcome: "native" | "degraded" | "refused";
223
+ asked?: string;
224
+ applied?: string;
225
+ vouched: boolean;
226
+ why: string;
227
+ }
228
+
229
+ /** Every control decided against the card (record 0052). A control the request does
230
+ * not ask about is not decided at all — no note, no refusal. The output cap
231
+ * and the window are always decided: every run sends one and compacts
232
+ * somewhere. */
233
+ export function decideControls(card: ModelCard, asked: AskedControls): ControlDecision[] {
234
+ const decisions: ControlDecision[] = [];
235
+
236
+ if (asked.effort !== undefined) {
237
+ const tier = asked.effort;
238
+ if (card.levels === "unknown") {
239
+ decisions.push({
240
+ control: "effort",
241
+ outcome: "degraded",
242
+ asked: tier,
243
+ applied: tier,
244
+ vouched: false,
245
+ why: `no layer names ${card.model}'s levels; the word goes out unvouched`,
246
+ });
247
+ } else {
248
+ const level = card.levels[tier];
249
+ if (level === "refused") {
250
+ decisions.push({
251
+ control: "effort",
252
+ outcome: "refused",
253
+ asked: tier,
254
+ vouched: false,
255
+ why: `${card.model} does not take effort "${tier}"`,
256
+ });
257
+ } else if (level.named) {
258
+ decisions.push({
259
+ control: "effort",
260
+ outcome: "native",
261
+ asked: tier,
262
+ applied: level.word,
263
+ vouched: true,
264
+ why: "",
265
+ });
266
+ } else {
267
+ decisions.push({
268
+ control: "effort",
269
+ outcome: "degraded",
270
+ asked: tier,
271
+ applied: level.word,
272
+ vouched: true,
273
+ why: `${card.model} does not take effort "${tier}"; the highest named tier below it is "${level.word}"`,
274
+ });
275
+ }
276
+ }
277
+ }
278
+
279
+ const capVouched = card.provenance.capField !== "wire";
280
+ decisions.push({
281
+ control: "cap",
282
+ outcome: capVouched ? "native" : "degraded",
283
+ applied: card.capField,
284
+ vouched: capVouched,
285
+ why: capVouched ? "" : `no layer names the cap field; the cap goes out as ${card.capField} unvouched`,
286
+ });
287
+
288
+ const images = asked.images ?? 0;
289
+ const documents = asked.documents ?? 0;
290
+ if (images > 0) {
291
+ if (card.inputs.image === true)
292
+ decisions.push({
293
+ control: "inputs",
294
+ outcome: "native",
295
+ asked: "image",
296
+ applied: "image",
297
+ vouched: true,
298
+ why: "",
299
+ });
300
+ else if (card.inputs.image === false)
301
+ decisions.push({
302
+ control: "inputs",
303
+ outcome: "refused",
304
+ asked: "image",
305
+ vouched: false,
306
+ why: `${card.model} takes no images`,
307
+ });
308
+ else
309
+ decisions.push({
310
+ control: "inputs",
311
+ outcome: "degraded",
312
+ asked: "image",
313
+ applied: "image",
314
+ vouched: false,
315
+ why: `no layer names ${card.model}'s image support; the image goes out unvouched`,
316
+ });
317
+ }
318
+ if (documents > 0) {
319
+ decisions.push({
320
+ control: "inputs",
321
+ outcome: "degraded",
322
+ asked: "document",
323
+ applied: "text",
324
+ vouched: false,
325
+ why: "documents reach a provider as a text stub until the harness carries files",
326
+ });
327
+ }
328
+
329
+ const windowVouched = card.provenance.window !== "wire";
330
+ decisions.push({
331
+ control: "window",
332
+ outcome: windowVouched ? "native" : "degraded",
333
+ applied: String(card.window),
334
+ vouched: windowVouched,
335
+ why: windowVouched ? "" : `no layer names the window; compacting at ${card.window}`,
336
+ });
337
+
338
+ const cacheNative = card.cache !== "unknown";
339
+ decisions.push({
340
+ control: "cache",
341
+ outcome: cacheNative ? "native" : "degraded",
342
+ applied: card.cache,
343
+ vouched: cacheNative,
344
+ why: cacheNative ? "" : `no layer names ${card.model}'s cache rule`,
345
+ });
346
+
347
+ return decisions;
348
+ }
@@ -1,3 +1,4 @@
1
+ import { parseModelRef } from "./provider.js";
1
2
  import type { ModelUsage, RunUsage } from "./runUsage.js";
2
3
 
3
4
  // The price of a model's tokens (docs/reference/specs/costs.md): one table
@@ -110,9 +111,14 @@ export function parseModelPrices(raw: unknown): ModelPriceTable {
110
111
  throw new Error("costs.prices must be a mapping of <provider>/<model> → rates");
111
112
  const out: Record<string, ModelPrice> = {};
112
113
  for (const [ref, value] of Object.entries(raw as Record<string, unknown>)) {
113
- const slash = ref.indexOf("/");
114
- if (slash <= 0 || slash === ref.length - 1)
115
- throw new Error(`costs.prices.${ref} must be keyed <provider>/<model>, the ref a run's spans name`);
114
+ let shaped: boolean;
115
+ try {
116
+ const parsed = parseModelRef(ref);
117
+ shaped = parsed.provider !== "" && parsed.model !== "";
118
+ } catch {
119
+ shaped = false;
120
+ }
121
+ if (!shaped) throw new Error(`costs.prices.${ref} must be keyed <provider>/<model>, the ref a run's spans name`);
116
122
  if (typeof value !== "object" || value === null || Array.isArray(value))
117
123
  throw new Error(
118
124
  `costs.prices.${ref} must be a mapping of { input, output, cacheRead, cacheWrite } in USD per million tokens`,
@@ -160,8 +166,11 @@ export interface PricedModelUsage extends ModelUsage {
160
166
  usd: number | null;
161
167
  }
162
168
 
163
- /** `anthropic/claude-fable-5` → `claude-fable-5`: the spans name the provider, the price table the model. */
164
- export const modelIdOf = (ref: string): string => (ref.includes("/") ? ref.slice(ref.indexOf("/") + 1) : ref);
169
+ /** `anthropic/claude-fable-5` → `claude-fable-5`: the spans name the provider, the price
170
+ * table the model. The list fallback drops the provider prefix alone (costs.md item 4b),
171
+ * so an aggregator's vendor-prefixed ref (`openrouter/anthropic/claude-…`) stays a list
172
+ * miss — reported unpriced, never silently billed at another provider's rate. */
173
+ export const modelIdOf = (ref: string): string => (ref.includes("/") ? parseModelRef(ref).model : ref);
165
174
 
166
175
  /** A run's dollars as every surface prints them (costs.md item 4c): cents from a
167
176
  * dollar up (`$1.24`), a tenth of a cent below that (`$0.038`), and `<$0.001`