@cohortapp/agent-sdk 2.9.0 → 2.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -106,6 +106,10 @@ export function yamlScalar(s) {
106
106
  * unless ev already carries an explicit inbox `id`.
107
107
  * - service ev.channel (the inbox dir + classifier "service").
108
108
  * - channel friendly label (ev.channel_label) else channel_id else channel.
109
+ * DISPLAY ONLY — `dm/casey`, `task/<title>`, `doc/<name>`.
110
+ * Never an address. The send path (scripts/daemon/deliver)
111
+ * used to fall back to it when `channel_id` was blank, which
112
+ * is how a board reply came to be posted to a task's title.
109
113
  * - channel_id native chat id (ev.source.chatId || ev.channel_id).
110
114
  * - sender ev.from.name || ev.from.id.
111
115
  * - sender_privilege ev.sender_privilege (resolved upstream) || "unknown".
@@ -101,6 +101,25 @@ export const BANNED_OPENERS = Object.freeze([
101
101
  */
102
102
  const COHORT_ID = /^(?:[a-z][a-z0-9]{14,}|[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})$/;
103
103
 
104
+ /**
105
+ * A Cohort recipient that is an ENTITY THREAD rather than a room:
106
+ * `task:<id>` / `file:<id>` / `decision:<id>`, minted by
107
+ * `scripts/daemon/deliver.mjs#cohortReplyRoute` for the surfaces that have no
108
+ * channel (a board item's comments, a doc's comments, a decision's thread).
109
+ *
110
+ * These are as internal as a channel id — more so, since hq resolves them
111
+ * org-scoped — but the colon fails {@link COHORT_ID}, so without this every
112
+ * board reply classified EXTERNAL: `failPolicy` flipped to fail-CLOSED for the
113
+ * whole board lane, and every `recipient_class: external` information-barrier
114
+ * rule started applying to internal board chatter. The prefix set is CLOSED and
115
+ * the id half is still shape-checked, so nothing that is not one of our own
116
+ * entity handles is laundered into "internal" by adding a colon.
117
+ *
118
+ * `file:` also carries the opaque `fileKey` of a chat attachment, which is not
119
+ * cuid-shaped, so the id half only has to be a non-empty token with no `@`.
120
+ */
121
+ const COHORT_ENTITY_REF = /^(?:task|file|decision):[^\s@:][^\s@]*$/;
122
+
104
123
  /** One-shot guard so the org-recipient warning is not emitted per message. */
105
124
  let _warnedExternalOrgRecipient = false;
106
125
 
@@ -146,7 +165,10 @@ export function classifyRecipient(channel, recipient, internalDomains = []) {
146
165
  // anything else on this channel (an email address handed to the org mailer)
147
166
  // still falls through to the domain test below, so a genuinely external
148
167
  // address is NOT laundered into "internal" by arriving on this channel.
149
- if (COHORT_ID.test(r)) return "internal";
168
+ // A room (a channel/member id) or an entity thread (`task:<id>` …) — both
169
+ // are rows inside the org the agent belongs to. See COHORT_ENTITY_REF for
170
+ // what the second form cost while it fell through to "external".
171
+ if (COHORT_ID.test(r) || COHORT_ENTITY_REF.test(r)) return "internal";
150
172
  break;
151
173
  case "slack":
152
174
  case "telegram":
@@ -570,6 +570,30 @@ describe("classifyRecipient", () => {
570
570
  );
571
571
  });
572
572
 
573
+ // REGRESSION — the SAME defect one shape further on. `deliver.mjs` answers a
574
+ // roomless surface on its entity thread and addresses it `task:<id>` /
575
+ // `file:<id>` / `decision:<id>`. The colon fails COHORT_ID, so every board,
576
+ // doc and decision reply classified EXTERNAL — fail-CLOSED policy faults for
577
+ // the whole board lane, and `recipient_class: external` barrier rules applied
578
+ // to internal board chatter. The module printed its own warning about it on
579
+ // every process.
580
+ it("classifies Cohort ENTITY threads as internal — a board reply is not an outsider", () => {
581
+ assert.equal(classifyRecipient("cohort", "task:cmrlw2l3600e9hh8n10hprdrd"), "internal");
582
+ assert.equal(classifyRecipient("cohort", "file:cmsspqi9500dmo401gg51ntz7"), "internal");
583
+ assert.equal(classifyRecipient("cohort", "decision:cmdec0000000000000000000"), "internal");
584
+ // The chat-attachment fileKey is opaque, not cuid-shaped.
585
+ assert.equal(classifyRecipient("cohort", "file:fk_abc123"), "internal");
586
+ assert.equal(classifyRecipient("cohort-org", "task:cmrlw2l3600e9hh8n10hprdrd"), "internal");
587
+ });
588
+
589
+ it("the entity-ref prefix set is CLOSED — it is not a colon wildcard", () => {
590
+ // Otherwise "internal" becomes a thing anyone can claim by adding a colon.
591
+ assert.equal(classifyRecipient("cohort", "mailto:someone@external.com"), "external");
592
+ assert.equal(classifyRecipient("cohort", "approval:aprv_journey_msopegu4"), "external");
593
+ assert.equal(classifyRecipient("cohort", "task:"), "external");
594
+ assert.equal(classifyRecipient("cohort", "task:someone@external.com"), "external");
595
+ });
596
+
573
597
  it("does NOT launder a genuinely external address arriving on the org channel", () => {
574
598
  // The whole point of scoping the case to id-shaped recipients: an email or
575
599
  // a phone number handed to an org surface must still be external, or the
@@ -160,16 +160,64 @@ function buildLadder(band, pct, spent, cap) {
160
160
  const TIER_ORDER = ["frontier", "default", "fast", "cheap"];
161
161
  const TIER_RANK = Object.freeze({ frontier: 0, default: 1, fast: 2, cheap: 3 });
162
162
 
163
+ // Providers whose whole line is cheap-tier by construction — the ladder's bottom
164
+ // rung. Matched on the ref's PROVIDER segment ("deepseek/deepseek-v4-flash"),
165
+ // which is a far stronger signal than any substring of the model name.
166
+ //
167
+ // STRUCK 2026-08-13: this block used to justify the `xai` entry by saying that
168
+ // without it "degradeChain must ... fail open and treat it as un-degradable".
169
+ // That inverted the design. degradeChain's filter KEEPS a null tier on purpose
170
+ // (`tier == null || TIER_RANK[tier] >= TIER_RANK.fast`) — failing open is the
171
+ // intended safety property, pinned by the "an unrecognised ref stays null so
172
+ // degradeChain fails OPEN" test, not a hazard to be designed away. An unknown
173
+ // xAI row was already kept, and is still kept.
174
+ //
175
+ // The honest reason `xai` is listed: inferTier is the layer's public answer to
176
+ // "how expensive is this ref", exported via `_internals` and read by the tests
177
+ // and by anything that later wants to RANK rather than merely filter. Getting a
178
+ // $0.20/MTok row filed on the $1/$5 Haiku rung is wrong as an answer even while
179
+ // degradeChain cannot act on the difference. Refresh alongside the catalog.
180
+ const CHEAP_PROVIDERS = new Set(["deepseek", "moonshot", "qwen", "xai"]);
181
+
182
+ // Model-name fragments that mark a cheap-tier row when the ref carries no
183
+ // provider segment (a hand-written chain member like "kimi-k2.6").
184
+ const CHEAP_HINTS = ["cheap", "deepseek", "kimi", "glm", "qwen", "grok"];
185
+
163
186
  // Heuristic tier inference from a literal ref when no alias tag is present. Keeps
164
187
  // degradeChain useful for chains built from raw "provider/model" refs.
188
+ //
189
+ // ORDER MATTERS, and it changed (2026-08-13): the cheap check now runs BEFORE
190
+ // the fast check. It used to run after, so every cheap-provider row whose model
191
+ // name happened to end in "flash" — deepseek/deepseek-v4-flash ($0.14/$0.28, the
192
+ // cheapest row in the bundled catalog) and qwen/qwen3.5-flash ($0.10/$0.40) —
193
+ // was classified as the Haiku-class "fast" rung ($1/$5), a ~7x mis-read of the
194
+ // ladder's own bottom rung.
195
+ //
196
+ // SCOPE OF THE FIX — STRUCK 2026-08-13: the original note here claimed the bug
197
+ // "did mis-rank the 'keep the single cheapest member' safety net". It could not
198
+ // have. That branch runs only when `kept` is EMPTY, i.e. when every non-sentinel
199
+ // member classified frontier/default — and a member classified fast OR cheap is
200
+ // never dropped, so it can never be one of the members the safety net sorts.
201
+ // Verified by exhausting degradeChain over cheap/fast/frontier chains at bands
202
+ // 100/125/150 under both orderings: byte-identical output.
203
+ //
204
+ // So this reorder changes NO routing today. It is a fix to the CLASSIFIER's
205
+ // answer, not to a live routing defect: degradeChain only asks "is this member
206
+ // frontier/default?", collapsing fast, cheap and null into one keep-bucket. The
207
+ // distinction is kept correct because inferTier is exported via `_internals` as
208
+ // this layer's tier oracle, and the next caller to RANK on it (rather than
209
+ // filter) would silently inherit the 7x error. Do not re-order it back on the
210
+ // grounds that "no test on behaviour fails" — that is the point being recorded.
165
211
  function inferTier(member) {
166
212
  const alias = typeof member === "object" && member ? member.alias : null;
167
213
  if (alias && TIER_RANK[alias] != null) return alias;
168
214
  const ref = String((member && member.ref) || member || "").toLowerCase();
169
215
  if (ref.includes("opus")) return "frontier";
170
216
  if (ref.includes("sonnet")) return "default";
217
+ const provider = ref.includes("/") ? ref.split("/")[0] : null;
218
+ if (provider && CHEAP_PROVIDERS.has(provider)) return "cheap";
219
+ if (CHEAP_HINTS.some((h) => ref.includes(h))) return "cheap";
171
220
  if (ref.includes("haiku") || ref.includes("flash") || ref.includes("mini")) return "fast";
172
- if (ref.includes("cheap") || ref.includes("deepseek") || ref.includes("kimi") || ref.includes("glm") || ref.includes("qwen")) return "cheap";
173
221
  return null;
174
222
  }
175
223
 
@@ -489,6 +537,10 @@ function emitPinOverridden(sessionKey, live, reason, deps) {
489
537
  // grep, summarisation) to a Haiku-Explore subagent so the main loop's cache
490
538
  // survives. The shape mirrors Claude Code's `--agents` JSON (name → { model,
491
539
  // description }).
540
+ // The fast tier was re-verified against the vendor docs on 2026-08-13 and did
541
+ // NOT move when opus/sonnet went to -5: claude-haiku-4-5 is still current, so
542
+ // this id is deliberately unchanged. It rides `claude --agents`, so the bare
543
+ // (undated) alias is intentional — the CLI resolves it to the current build.
492
544
  const HAIKU_EXPLORE = Object.freeze({
493
545
  "Haiku-Explore": {
494
546
  model: "claude-haiku-4-5",
@@ -178,6 +178,82 @@ test("degradeChain: de-dupes refs that collapse onto the same model", () => {
178
178
  assert.equal(out.length, 1);
179
179
  });
180
180
 
181
+ // ---------------------------------------------------------------------------
182
+ // inferTier — the untagged-ref heuristic behind degradeChain
183
+ // ---------------------------------------------------------------------------
184
+
185
+ const { inferTier } = _internals;
186
+
187
+ test("inferTier: a cheap PROVIDER beats a 'flash'-shaped model name", () => {
188
+ // The regression this locks: the fast check (haiku|flash|mini) used to run
189
+ // BEFORE the cheap check, so the two cheapest rows in the bundled catalog were
190
+ // filed on the Haiku rung — deepseek-v4-flash is $0.14/$0.28 against Haiku's
191
+ // $1/$5, a ~7x mis-read of the ladder's own bottom rung.
192
+ assert.equal(inferTier({ ref: "deepseek/deepseek-v4-flash" }), "cheap");
193
+ assert.equal(inferTier({ ref: "qwen/qwen3.5-flash" }), "cheap");
194
+ // …while a genuinely fast-tier Anthropic row is untouched.
195
+ assert.equal(inferTier({ ref: "anthropic/claude-haiku-4-5" }), "fast");
196
+ assert.equal(inferTier({ ref: "anthropic/claude-opus-5" }), "frontier");
197
+ assert.equal(inferTier({ ref: "anthropic/claude-sonnet-5" }), "default");
198
+ });
199
+
200
+ test("inferTier: every cheap provider the economics layer NAMES is classified", () => {
201
+ // economics.mjs has always claimed deepseek/kimi/glm/qwen as the cheap tier;
202
+ // xai/grok is added so the classifier is ready the moment a Grok row lands
203
+ // (an agent can add one today via catalog_overrides — see model-router.test.mjs).
204
+ for (const ref of [
205
+ "deepseek/deepseek-v4-flash",
206
+ "moonshot/kimi-k2.6",
207
+ "qwen/qwen3-coder-plus",
208
+ "xai/grok-4-fast",
209
+ "zhipu/glm-5",
210
+ "kimi-k2.6", // bare, no provider segment
211
+ "grok-4-fast", // bare, no provider segment
212
+ ]) {
213
+ assert.equal(inferTier({ ref }), "cheap", `${ref} should classify as cheap-tier`);
214
+ }
215
+ });
216
+
217
+ test("inferTier: a cheap provider's WHOLE line is cheap, not just its famous model", () => {
218
+ // The model-name hints ("kimi", "grok", "glm") would carry the refs above on
219
+ // their own, which makes them a trap: the next model each of these providers
220
+ // ships will not be called "grok-4" or "kimi-k2". The provider segment is the
221
+ // durable signal, so these refs — none of which contain a hint word — are the
222
+ // ones that actually hold CHEAP_PROVIDERS in place.
223
+ assert.equal(inferTier({ ref: "xai/sonic-2" }), "cheap");
224
+ assert.equal(inferTier({ ref: "moonshot/k3-preview" }), "cheap");
225
+ assert.equal(inferTier({ ref: "deepseek/v5" }), "cheap");
226
+ });
227
+
228
+ test("inferTier: an unrecognised ref stays null so degradeChain fails OPEN", () => {
229
+ // A model we cannot classify must be KEPT, never dropped: a budget heuristic
230
+ // that silently deletes candidates it doesn't recognise strands the caller
231
+ // with an empty chain. Asserted against a MIXED chain on purpose — with an
232
+ // unknown row alone, the "keep the single cheapest" safety net returns it
233
+ // either way, so a fail-closed filter would slip through unnoticed.
234
+ assert.equal(inferTier({ ref: "acme/some-new-model" }), null);
235
+ const out = degradeChain(
236
+ [{ ref: "anthropic/claude-opus-5" }, { ref: "acme/some-new-model" }],
237
+ 150,
238
+ {}
239
+ );
240
+ assert.deepEqual(
241
+ out.map((m) => m.ref),
242
+ ["acme/some-new-model"],
243
+ "frontier dropped, the unclassifiable row survives"
244
+ );
245
+ });
246
+
247
+ test("degradeChain: a cheap third-party row survives band 150 alongside fast rows", () => {
248
+ const chain = [
249
+ { ref: "anthropic/claude-opus-5" },
250
+ { ref: "anthropic/claude-haiku-4-5" },
251
+ { ref: "xai/grok-4-fast" },
252
+ ];
253
+ const out = degradeChain(chain, 150, {}).map((m) => m.ref);
254
+ assert.deepEqual(out, ["anthropic/claude-haiku-4-5", "xai/grok-4-fast"]);
255
+ });
256
+
181
257
  // ---------------------------------------------------------------------------
182
258
  // batchLaneFor — classify deferrable vs realtime
183
259
  // ---------------------------------------------------------------------------
@@ -831,10 +831,25 @@ function forceAnthropicDecision(catalog, req, { now, opts, audit, tried = [], re
831
831
  if (!row) {
832
832
  // No catalog at all — degrade to a synthetic chosen so the caller still has
833
833
  // a usable model flag (stock `claude` defaults to its account model).
834
+ // Ids current as of 2026-08-13; see pickAnthropicRow's provenance block.
835
+ // No succession fallback is needed here precisely BECAUSE there is no
836
+ // catalog to miss: this id goes straight onto `claude --model`, so it must
837
+ // be the id we actually want the CLI to run.
838
+ // REACHABILITY (2026-08-13): pickAnthropicRow returns null only when the
839
+ // catalog holds no available anthropic row at all, which on a real install
840
+ // means the bundled YAML failed to load. In production the 4-x rows always
841
+ // win before this point, so do NOT cite this branch as proof the SDK emits
842
+ // -5 ids — the only caller that reaches it today is the empty-catalog test.
843
+ // STRUCK (2026-08-13): `ref` was hardcoded to the sonnet ref even when `id`
844
+ // resolved to opus, so a critical/thinking request produced the pair
845
+ // {model: opus, ref: sonnet} — and `ref` is what feeds cacheAffinityKey and
846
+ // chain[0].ref, i.e. the pin and the audit trail both named the wrong model.
847
+ const syntheticId =
848
+ req.needs_thinking || req.priority === "critical" ? "claude-opus-5" : "claude-sonnet-5";
834
849
  const synthetic = {
835
850
  provider: "anthropic",
836
- id: req.needs_thinking || req.priority === "critical" ? "claude-opus-4-8" : "claude-sonnet-4-6",
837
- ref: "anthropic/claude-sonnet-4-6",
851
+ id: syntheticId,
852
+ ref: `anthropic/${syntheticId}`,
838
853
  status: "available",
839
854
  auth_env: null,
840
855
  endpoints: {},
@@ -951,13 +966,51 @@ function noRouteDecision(req, { now, opts, audit, tried, explainSkips, lane }) {
951
966
  * Pick the best available Anthropic row from the catalog for a request:
952
967
  * frontier (opus) for thinking/critical, else the default workhorse (sonnet),
953
968
  * else any available anthropic row.
969
+ *
970
+ * ⚠ READ FIRST: THE -5 REFS BELOW ARE INERT ON THIS TREE. lookupModel misses
971
+ * every one of them, because lib/model-router/catalog/anthropic.yaml still
972
+ * carries only 4-x rows, so the succession tail is what actually answers and
973
+ * `spawnArgs.modelFlag` (= row.id, see modelFlagFor above) puts
974
+ * `claude-sonnet-4-6` / `claude-opus-4-8` on `claude --model` for every request
975
+ * that reaches this safety net — which, with no config/model-routing.yaml in the
976
+ * repo, is every request on the no-v2-config path. Adding the two -5 rows to
977
+ * that YAML is the step that makes this list mean what it says; it has NOT been
978
+ * done. Do not read the ordering below as evidence the SDK runs -5 models.
979
+ *
980
+ * MODEL IDS — PROVENANCE AND WHY THE LEGACY REFS ARE STILL HERE (2026-08-13):
981
+ * the current Anthropic ids are claude-opus-5 / claude-sonnet-5 /
982
+ * claude-haiku-4-5-20251001 (verified against the vendor docs on 2026-08-13;
983
+ * the fast tier did not move). They lead each list. The claude-*-4-x refs
984
+ * BELOW them are not leftovers — they are the succession fallback, and they
985
+ * must stay until lib/model-router/catalog/anthropic.yaml carries -5 rows:
986
+ * lookupModel() is an EXACT byRef hit with no `replaced_by` chasing, so a ref
987
+ * the catalog does not have simply returns null. Listing only the -5 refs
988
+ * today would silently drop this safety net through to "any available
989
+ * anthropic row" — an unordered `models.find()` that would hand critical work
990
+ * whatever row happens to sit first in the file. That silent downgrade is the
991
+ * exact failure this ordered list exists to prevent.
992
+ *
993
+ * REFRESH: this list is the LAST-RESORT floor (kill switch / no config /
994
+ * nothing survived the gates), not the routing table. The routing table is the
995
+ * catalog + the config `aliases:`; refresh those first (see the provenance
996
+ * block in lib/model-router.mjs). When -5 rows land in the bundled catalog,
997
+ * DELETE the 4-x entries here — do not let this list accumulate a model
998
+ * museum, or the "best available" ordering stops meaning anything.
954
999
  */
955
1000
  function pickAnthropicRow(catalog, req) {
956
1001
  if (!catalog) return null;
957
1002
  const wantFrontier = req.needs_thinking || req.priority === "critical" || req.agent_role === "regulatory";
958
1003
  const order = wantFrontier
959
- ? ["anthropic/claude-opus-4-8", "anthropic/claude-sonnet-4-6", "anthropic/claude-haiku-4-5"]
960
- : ["anthropic/claude-sonnet-4-6", "anthropic/claude-haiku-4-5", "anthropic/claude-opus-4-8"];
1004
+ ? [
1005
+ "anthropic/claude-opus-5", "anthropic/claude-opus-4-8",
1006
+ "anthropic/claude-sonnet-5", "anthropic/claude-sonnet-4-6",
1007
+ "anthropic/claude-haiku-4-5",
1008
+ ]
1009
+ : [
1010
+ "anthropic/claude-sonnet-5", "anthropic/claude-sonnet-4-6",
1011
+ "anthropic/claude-haiku-4-5",
1012
+ "anthropic/claude-opus-5", "anthropic/claude-opus-4-8",
1013
+ ];
961
1014
  for (const ref of order) {
962
1015
  const row = lookupModel(catalog, ref);
963
1016
  if (row && row.status === "available") {
@@ -8,15 +8,22 @@
8
8
  * existing Claude-CLI-on-Max-subscription behaviour is preserved verbatim.
9
9
  *
10
10
  * Why this exists:
11
- * - Anthropic Sonnet 4.6 / Opus 4.7 are still the quality ceiling for
11
+ * - The Anthropic frontier/workhorse pair is still the quality ceiling for
12
12
  * agent loops, but background work (classification, log triage,
13
- * scheduled summaries) does NOT need them.
13
+ * scheduled summaries) does NOT need it. (STRUCK 2026-08-13: this line
14
+ * used to name "Sonnet 4.6 / Opus 4.7" as the ceiling. Naming versions in
15
+ * a rationale paragraph is how a docblock rots — the CURRENT pair is
16
+ * recorded in exactly one place now, ANTHROPIC_DEFAULT.models below, with
17
+ * its provenance and refresh path.)
14
18
  * - Moonshot exposes a native `/anthropic` Messages endpoint, so
15
19
  * Kimi K2.6 is a literal drop-in for Claude Code — no router,
16
20
  * no translation, no tool-use degradation.
17
- * - Qwen3-Coder via OpenRouter is one router-hop away and significantly
18
- * cheaper; usable for agents that don't need extended-thinking or
19
- * parallel tool calls.
21
+ * - Qwen is one hop away and significantly cheaper; usable for agents that
22
+ * don't need extended-thinking or parallel tool calls. (A v1 config can
23
+ * reach it through any OpenAI-compat gateway, which is what the v1 tests
24
+ * model; the v2 bundled catalog reaches it on DashScope and marks it
25
+ * harness.session:false — Qwen is a direct/batch lane model there, not a
26
+ * session host.)
20
27
  *
21
28
  * What this is NOT:
22
29
  * - Not a wire-format translator. We rely on the backend speaking
@@ -90,6 +97,67 @@ try {
90
97
  // Anthropic-only fallback so existing agents keep working unchanged.
91
98
  // ---------------------------------------------------------------------------
92
99
 
100
+ /**
101
+ * MODEL IDS — PROVENANCE AND REFRESH PATH (last refreshed 2026-08-13)
102
+ * ────────────────────────────────────────────────────────────────────
103
+ * WHERE THIS LIST CAME FROM: the Anthropic model list was verified against the
104
+ * vendor docs on 2026-08-13 (context7 → Anthropic model reference). Current at
105
+ * that date: claude-opus-5 (frontier, 1M ctx, 128k out, $5/$25 per MTok),
106
+ * claude-sonnet-5 (workhorse), claude-haiku-4-5-20251001 (fast/classifier —
107
+ * still current, deliberately NOT bumped). Two more exist and are deliberately
108
+ * absent from this map: claude-fable-5 ($10/$50, long-running agents — a cost
109
+ * decision, not a default) and claude-mythos-5 (restricted to Project
110
+ * Glasswing — must never be offered as a general option).
111
+ *
112
+ * HOW THE NEXT PERSON REFRESHES IT — in order of preference, and note that the
113
+ * first two need NO code change and NO release, which is the whole point:
114
+ *
115
+ * 1. v2 (schema_version: 2) agents: the CATALOG is the source of truth, not
116
+ * this file. Add/replace rows in lib/model-router/catalog/<provider>.yaml
117
+ * (ships with the SDK) or, without touching the SDK at all, patch them
118
+ * per-agent via `catalog_overrides:` in config/model-routing.yaml — that
119
+ * layer outranks the bundled rows (see catalog.mjs's authority merge) and
120
+ * can even introduce a provider the SDK does not bundle. Then repoint the
121
+ * `aliases:` (frontier/default/fast/cheap) at the new refs.
122
+ * 2. v1 agents: declare `backends.anthropic.models` in
123
+ * config/model-routing.yaml. normaliseConfig only injects the map below
124
+ * when `backends.anthropic` is ABSENT, so a config-declared map wins
125
+ * outright — no PR, no build.
126
+ * 3. Editing the map below is the LAST resort: it is the floor for an agent
127
+ * with no routing config at all.
128
+ *
129
+ * WHY THE STALE IDS HERE WERE SURVIVABLE (and so went unnoticed for a release):
130
+ * for `transport: "anthropic-cli"` — the default, keychain-OAuth path — THIS
131
+ * FILE'S modelFlagFor() sends the version-agnostic shorthand
132
+ * "opus"/"sonnet"/"haiku" and lets Anthropic resolve the current version, so a
133
+ * stale id in the map below did not reach the wire. It DID reach
134
+ * `resolved.model`, which is what the cost ledger, telemetry and session records
135
+ * store; a stale id there mis-prices and mis-attributes work. That is the damage
136
+ * this refresh repairs ON THE v1 PATH.
137
+ *
138
+ * ── DO NOT GENERALISE THAT TO THE ROUTER. THE v2 LANE IS STILL STALE. ────────
139
+ * NARROWED 2026-08-13, because the sentence above reads as a router-wide
140
+ * reassurance and is false as one. Two different functions are called
141
+ * `modelFlagFor`:
142
+ * - this file's (v1) — maps tier → "opus"/"sonnet"/"haiku" shorthand, and
143
+ * only for the four known tiers; an unrecognised `request.tier` falls
144
+ * through to `return resolved.model`, the literal id.
145
+ * - resolve.mjs's (v2, line ~468) — `return row.id`, ALWAYS the literal
146
+ * catalog id. spawn.mjs then does `args.push("--model", String(modelFlag))`.
147
+ * So on the v2 lane the catalog id IS the wire value, and the SHIPPED catalog
148
+ * (lib/model-router/catalog/anthropic.yaml) still carries only 4-x rows.
149
+ * Measured on this tree, with no routing config present — the default state,
150
+ * since the repo ships no config/model-routing.yaml at all:
151
+ * ordinary session → claude --model claude-sonnet-4-6
152
+ * critical/thinking → claude --model claude-opus-4-8
153
+ * The -5 ids added to resolve.mjs's pickAnthropicRow preference list are
154
+ * therefore INERT in production: lookupModel misses them against the bundled
155
+ * catalog and the ordered succession tail answers with the 4-x row. They become
156
+ * live the moment anthropic.yaml gains `claude-opus-5` / `claude-sonnet-5` rows,
157
+ * which is the one remaining step of this refresh and is NOT done. Until then,
158
+ * "the agent-sdk uses the latest Anthropic models" is true of the v1 floor only.
159
+ * See the guard test "the SHIPPED catalog is what decides the wire value".
160
+ */
93
161
  const ANTHROPIC_DEFAULT = Object.freeze({
94
162
  transport: "anthropic-cli", // special: do nothing (current behaviour)
95
163
  base_url: "https://api.anthropic.com",
@@ -102,11 +170,30 @@ const ANTHROPIC_DEFAULT = Object.freeze({
102
170
  "parallel_tools",
103
171
  "long_context_1m",
104
172
  ],
105
- pricing: { input_per_m: 3.0, output_per_m: 15.0 }, // Sonnet 4.6 ballpark
173
+ // $/MTok. STRUCK "Sonnet 4.6 ballpark" (2026-08-13): these figures are
174
+ // LINE-INHERITED from Sonnet 4.6's list price and are NOT vendor-confirmed for
175
+ // Sonnet 5 — only Opus 5 ($5/$25) was confirmed in the 2026-08-13 sweep. The
176
+ // v2 lane does not use this at all (it prices from the catalog row's `cost`
177
+ // block, with its own cost_provenance stamp).
178
+ //
179
+ // NARROWED 2026-08-13: an earlier version of this note opened "$/MTok for the
180
+ // `default` tier", which implies a per-tier price. THERE IS NO PER-TIER PRICE
181
+ // ON THE v1 LANE. `resolveBackend` returns `pricing: backend.pricing` — this
182
+ // one block — for every tier, and `estimateCost` multiplies by it regardless
183
+ // of which model was chosen. So premium work is billed at the WORKHORSE rate:
184
+ // with premium now claude-opus-5 at a confirmed $5/$25, every v1 opus estimate
185
+ // is low by ~1.7x on both input and output, and always was (Opus 4.7 was
186
+ // dearer than $3/$15 too). Refreshing the ids did not touch this and did not
187
+ // cause it. Fixing it means per-tier pricing in the backend shape — a schema
188
+ // change with config-compat consequences, deliberately not smuggled into an
189
+ // id refresh. Until then, treat v1 `estimateCost` as a floor, not a bill.
190
+ pricing: { input_per_m: 3.0, output_per_m: 15.0 },
106
191
  models: {
107
192
  classifier: "claude-haiku-4-5-20251001",
108
- default: "claude-sonnet-4-6",
109
- premium: "claude-opus-4-7",
193
+ default: "claude-sonnet-5",
194
+ premium: "claude-opus-5",
195
+ // Current as of 2026-08-13 — the fast tier did NOT move when opus/sonnet
196
+ // did. Do not "helpfully" bump this to a -5 id that does not exist.
110
197
  fast: "claude-haiku-4-5-20251001",
111
198
  },
112
199
  });