@cohortapp/agent-sdk 2.9.1 → 2.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/channels/inbox-item.mjs +4 -0
- package/lib/comms/send-gate.mjs +23 -1
- package/lib/comms/send-gate.test.mjs +24 -0
- package/lib/model-router/economics.mjs +53 -1
- package/lib/model-router/economics.test.mjs +76 -0
- package/lib/model-router/resolve.mjs +57 -4
- package/lib/model-router.mjs +95 -8
- package/lib/model-router.test.mjs +305 -5
- package/lib/org/inbound/project.mjs +9 -5
- package/lib/org/messaging.mjs +6 -1
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +78 -3
- package/lib/org/protocol.test.mjs +14 -2
- package/package.json +1 -1
- package/scripts/daemon/assurance.mjs +12 -1
- package/scripts/daemon/deliver.mjs +457 -33
- package/scripts/daemon/deliver.test.mjs +564 -0
- package/scripts/daemon/responder-history.test.mjs +18 -2
- package/scripts/daemon/responder.mjs +7 -1
|
@@ -106,6 +106,10 @@ export function yamlScalar(s) {
|
|
|
106
106
|
* unless ev already carries an explicit inbox `id`.
|
|
107
107
|
* - service ev.channel (the inbox dir + classifier "service").
|
|
108
108
|
* - channel friendly label (ev.channel_label) else channel_id else channel.
|
|
109
|
+
* DISPLAY ONLY — `dm/casey`, `task/<title>`, `doc/<name>`.
|
|
110
|
+
* Never an address. The send path (scripts/daemon/deliver)
|
|
111
|
+
* used to fall back to it when `channel_id` was blank, which
|
|
112
|
+
* is how a board reply came to be posted to a task's title.
|
|
109
113
|
* - channel_id native chat id (ev.source.chatId || ev.channel_id).
|
|
110
114
|
* - sender ev.from.name || ev.from.id.
|
|
111
115
|
* - sender_privilege ev.sender_privilege (resolved upstream) || "unknown".
|
package/lib/comms/send-gate.mjs
CHANGED
|
@@ -101,6 +101,25 @@ export const BANNED_OPENERS = Object.freeze([
|
|
|
101
101
|
*/
|
|
102
102
|
const COHORT_ID = /^(?:[a-z][a-z0-9]{14,}|[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})$/;
|
|
103
103
|
|
|
104
|
+
/**
|
|
105
|
+
* A Cohort recipient that is an ENTITY THREAD rather than a room:
|
|
106
|
+
* `task:<id>` / `file:<id>` / `decision:<id>`, minted by
|
|
107
|
+
* `scripts/daemon/deliver.mjs#cohortReplyRoute` for the surfaces that have no
|
|
108
|
+
* channel (a board item's comments, a doc's comments, a decision's thread).
|
|
109
|
+
*
|
|
110
|
+
* These are as internal as a channel id — more so, since hq resolves them
|
|
111
|
+
* org-scoped — but the colon fails {@link COHORT_ID}, so without this every
|
|
112
|
+
* board reply classified EXTERNAL: `failPolicy` flipped to fail-CLOSED for the
|
|
113
|
+
* whole board lane, and every `recipient_class: external` information-barrier
|
|
114
|
+
* rule started applying to internal board chatter. The prefix set is CLOSED and
|
|
115
|
+
* the id half is still shape-checked, so nothing that is not one of our own
|
|
116
|
+
* entity handles is laundered into "internal" by adding a colon.
|
|
117
|
+
*
|
|
118
|
+
* `file:` also carries the opaque `fileKey` of a chat attachment, which is not
|
|
119
|
+
* cuid-shaped, so the id half only has to be a non-empty token with no `@`.
|
|
120
|
+
*/
|
|
121
|
+
const COHORT_ENTITY_REF = /^(?:task|file|decision):[^\s@:][^\s@]*$/;
|
|
122
|
+
|
|
104
123
|
/** One-shot guard so the org-recipient warning is not emitted per message. */
|
|
105
124
|
let _warnedExternalOrgRecipient = false;
|
|
106
125
|
|
|
@@ -146,7 +165,10 @@ export function classifyRecipient(channel, recipient, internalDomains = []) {
|
|
|
146
165
|
// anything else on this channel (an email address handed to the org mailer)
|
|
147
166
|
// still falls through to the domain test below, so a genuinely external
|
|
148
167
|
// address is NOT laundered into "internal" by arriving on this channel.
|
|
149
|
-
|
|
168
|
+
// A room (a channel/member id) or an entity thread (`task:<id>` …) — both
|
|
169
|
+
// are rows inside the org the agent belongs to. See COHORT_ENTITY_REF for
|
|
170
|
+
// what the second form cost while it fell through to "external".
|
|
171
|
+
if (COHORT_ID.test(r) || COHORT_ENTITY_REF.test(r)) return "internal";
|
|
150
172
|
break;
|
|
151
173
|
case "slack":
|
|
152
174
|
case "telegram":
|
|
@@ -570,6 +570,30 @@ describe("classifyRecipient", () => {
|
|
|
570
570
|
);
|
|
571
571
|
});
|
|
572
572
|
|
|
573
|
+
// REGRESSION — the SAME defect one shape further on. `deliver.mjs` answers a
|
|
574
|
+
// roomless surface on its entity thread and addresses it `task:<id>` /
|
|
575
|
+
// `file:<id>` / `decision:<id>`. The colon fails COHORT_ID, so every board,
|
|
576
|
+
// doc and decision reply classified EXTERNAL — fail-CLOSED policy faults for
|
|
577
|
+
// the whole board lane, and `recipient_class: external` barrier rules applied
|
|
578
|
+
// to internal board chatter. The module printed its own warning about it on
|
|
579
|
+
// every process.
|
|
580
|
+
it("classifies Cohort ENTITY threads as internal — a board reply is not an outsider", () => {
|
|
581
|
+
assert.equal(classifyRecipient("cohort", "task:cmrlw2l3600e9hh8n10hprdrd"), "internal");
|
|
582
|
+
assert.equal(classifyRecipient("cohort", "file:cmsspqi9500dmo401gg51ntz7"), "internal");
|
|
583
|
+
assert.equal(classifyRecipient("cohort", "decision:cmdec0000000000000000000"), "internal");
|
|
584
|
+
// The chat-attachment fileKey is opaque, not cuid-shaped.
|
|
585
|
+
assert.equal(classifyRecipient("cohort", "file:fk_abc123"), "internal");
|
|
586
|
+
assert.equal(classifyRecipient("cohort-org", "task:cmrlw2l3600e9hh8n10hprdrd"), "internal");
|
|
587
|
+
});
|
|
588
|
+
|
|
589
|
+
it("the entity-ref prefix set is CLOSED — it is not a colon wildcard", () => {
|
|
590
|
+
// Otherwise "internal" becomes a thing anyone can claim by adding a colon.
|
|
591
|
+
assert.equal(classifyRecipient("cohort", "mailto:someone@external.com"), "external");
|
|
592
|
+
assert.equal(classifyRecipient("cohort", "approval:aprv_journey_msopegu4"), "external");
|
|
593
|
+
assert.equal(classifyRecipient("cohort", "task:"), "external");
|
|
594
|
+
assert.equal(classifyRecipient("cohort", "task:someone@external.com"), "external");
|
|
595
|
+
});
|
|
596
|
+
|
|
573
597
|
it("does NOT launder a genuinely external address arriving on the org channel", () => {
|
|
574
598
|
// The whole point of scoping the case to id-shaped recipients: an email or
|
|
575
599
|
// a phone number handed to an org surface must still be external, or the
|
|
@@ -160,16 +160,64 @@ function buildLadder(band, pct, spent, cap) {
|
|
|
160
160
|
const TIER_ORDER = ["frontier", "default", "fast", "cheap"];
|
|
161
161
|
const TIER_RANK = Object.freeze({ frontier: 0, default: 1, fast: 2, cheap: 3 });
|
|
162
162
|
|
|
163
|
+
// Providers whose whole line is cheap-tier by construction — the ladder's bottom
|
|
164
|
+
// rung. Matched on the ref's PROVIDER segment ("deepseek/deepseek-v4-flash"),
|
|
165
|
+
// which is a far stronger signal than any substring of the model name.
|
|
166
|
+
//
|
|
167
|
+
// STRUCK 2026-08-13: this block used to justify the `xai` entry by saying that
|
|
168
|
+
// without it "degradeChain must ... fail open and treat it as un-degradable".
|
|
169
|
+
// That inverted the design. degradeChain's filter KEEPS a null tier on purpose
|
|
170
|
+
// (`tier == null || TIER_RANK[tier] >= TIER_RANK.fast`) — failing open is the
|
|
171
|
+
// intended safety property, pinned by the "an unrecognised ref stays null so
|
|
172
|
+
// degradeChain fails OPEN" test, not a hazard to be designed away. An unknown
|
|
173
|
+
// xAI row was already kept, and is still kept.
|
|
174
|
+
//
|
|
175
|
+
// The honest reason `xai` is listed: inferTier is the layer's public answer to
|
|
176
|
+
// "how expensive is this ref", exported via `_internals` and read by the tests
|
|
177
|
+
// and by anything that later wants to RANK rather than merely filter. Getting a
|
|
178
|
+
// $0.20/MTok row filed on the $1/$5 Haiku rung is wrong as an answer even while
|
|
179
|
+
// degradeChain cannot act on the difference. Refresh alongside the catalog.
|
|
180
|
+
const CHEAP_PROVIDERS = new Set(["deepseek", "moonshot", "qwen", "xai"]);
|
|
181
|
+
|
|
182
|
+
// Model-name fragments that mark a cheap-tier row when the ref carries no
|
|
183
|
+
// provider segment (a hand-written chain member like "kimi-k2.6").
|
|
184
|
+
const CHEAP_HINTS = ["cheap", "deepseek", "kimi", "glm", "qwen", "grok"];
|
|
185
|
+
|
|
163
186
|
// Heuristic tier inference from a literal ref when no alias tag is present. Keeps
|
|
164
187
|
// degradeChain useful for chains built from raw "provider/model" refs.
|
|
188
|
+
//
|
|
189
|
+
// ORDER MATTERS, and it changed (2026-08-13): the cheap check now runs BEFORE
|
|
190
|
+
// the fast check. It used to run after, so every cheap-provider row whose model
|
|
191
|
+
// name happened to end in "flash" — deepseek/deepseek-v4-flash ($0.14/$0.28, the
|
|
192
|
+
// cheapest row in the bundled catalog) and qwen/qwen3.5-flash ($0.10/$0.40) —
|
|
193
|
+
// was classified as the Haiku-class "fast" rung ($1/$5), a ~7x mis-read of the
|
|
194
|
+
// ladder's own bottom rung.
|
|
195
|
+
//
|
|
196
|
+
// SCOPE OF THE FIX — STRUCK 2026-08-13: the original note here claimed the bug
|
|
197
|
+
// "did mis-rank the 'keep the single cheapest member' safety net". It could not
|
|
198
|
+
// have. That branch runs only when `kept` is EMPTY, i.e. when every non-sentinel
|
|
199
|
+
// member classified frontier/default — and a member classified fast OR cheap is
|
|
200
|
+
// never dropped, so it can never be one of the members the safety net sorts.
|
|
201
|
+
// Verified by exhausting degradeChain over cheap/fast/frontier chains at bands
|
|
202
|
+
// 100/125/150 under both orderings: byte-identical output.
|
|
203
|
+
//
|
|
204
|
+
// So this reorder changes NO routing today. It is a fix to the CLASSIFIER's
|
|
205
|
+
// answer, not to a live routing defect: degradeChain only asks "is this member
|
|
206
|
+
// frontier/default?", collapsing fast, cheap and null into one keep-bucket. The
|
|
207
|
+
// distinction is kept correct because inferTier is exported via `_internals` as
|
|
208
|
+
// this layer's tier oracle, and the next caller to RANK on it (rather than
|
|
209
|
+
// filter) would silently inherit the 7x error. Do not re-order it back on the
|
|
210
|
+
// grounds that "no test on behaviour fails" — that is the point being recorded.
|
|
165
211
|
function inferTier(member) {
|
|
166
212
|
const alias = typeof member === "object" && member ? member.alias : null;
|
|
167
213
|
if (alias && TIER_RANK[alias] != null) return alias;
|
|
168
214
|
const ref = String((member && member.ref) || member || "").toLowerCase();
|
|
169
215
|
if (ref.includes("opus")) return "frontier";
|
|
170
216
|
if (ref.includes("sonnet")) return "default";
|
|
217
|
+
const provider = ref.includes("/") ? ref.split("/")[0] : null;
|
|
218
|
+
if (provider && CHEAP_PROVIDERS.has(provider)) return "cheap";
|
|
219
|
+
if (CHEAP_HINTS.some((h) => ref.includes(h))) return "cheap";
|
|
171
220
|
if (ref.includes("haiku") || ref.includes("flash") || ref.includes("mini")) return "fast";
|
|
172
|
-
if (ref.includes("cheap") || ref.includes("deepseek") || ref.includes("kimi") || ref.includes("glm") || ref.includes("qwen")) return "cheap";
|
|
173
221
|
return null;
|
|
174
222
|
}
|
|
175
223
|
|
|
@@ -489,6 +537,10 @@ function emitPinOverridden(sessionKey, live, reason, deps) {
|
|
|
489
537
|
// grep, summarisation) to a Haiku-Explore subagent so the main loop's cache
|
|
490
538
|
// survives. The shape mirrors Claude Code's `--agents` JSON (name → { model,
|
|
491
539
|
// description }).
|
|
540
|
+
// The fast tier was re-verified against the vendor docs on 2026-08-13 and did
|
|
541
|
+
// NOT move when opus/sonnet went to -5: claude-haiku-4-5 is still current, so
|
|
542
|
+
// this id is deliberately unchanged. It rides `claude --agents`, so the bare
|
|
543
|
+
// (undated) alias is intentional — the CLI resolves it to the current build.
|
|
492
544
|
const HAIKU_EXPLORE = Object.freeze({
|
|
493
545
|
"Haiku-Explore": {
|
|
494
546
|
model: "claude-haiku-4-5",
|
|
@@ -178,6 +178,82 @@ test("degradeChain: de-dupes refs that collapse onto the same model", () => {
|
|
|
178
178
|
assert.equal(out.length, 1);
|
|
179
179
|
});
|
|
180
180
|
|
|
181
|
+
// ---------------------------------------------------------------------------
|
|
182
|
+
// inferTier — the untagged-ref heuristic behind degradeChain
|
|
183
|
+
// ---------------------------------------------------------------------------
|
|
184
|
+
|
|
185
|
+
const { inferTier } = _internals;
|
|
186
|
+
|
|
187
|
+
test("inferTier: a cheap PROVIDER beats a 'flash'-shaped model name", () => {
|
|
188
|
+
// The regression this locks: the fast check (haiku|flash|mini) used to run
|
|
189
|
+
// BEFORE the cheap check, so the two cheapest rows in the bundled catalog were
|
|
190
|
+
// filed on the Haiku rung — deepseek-v4-flash is $0.14/$0.28 against Haiku's
|
|
191
|
+
// $1/$5, a ~7x mis-read of the ladder's own bottom rung.
|
|
192
|
+
assert.equal(inferTier({ ref: "deepseek/deepseek-v4-flash" }), "cheap");
|
|
193
|
+
assert.equal(inferTier({ ref: "qwen/qwen3.5-flash" }), "cheap");
|
|
194
|
+
// …while a genuinely fast-tier Anthropic row is untouched.
|
|
195
|
+
assert.equal(inferTier({ ref: "anthropic/claude-haiku-4-5" }), "fast");
|
|
196
|
+
assert.equal(inferTier({ ref: "anthropic/claude-opus-5" }), "frontier");
|
|
197
|
+
assert.equal(inferTier({ ref: "anthropic/claude-sonnet-5" }), "default");
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
test("inferTier: every cheap provider the economics layer NAMES is classified", () => {
|
|
201
|
+
// economics.mjs has always claimed deepseek/kimi/glm/qwen as the cheap tier;
|
|
202
|
+
// xai/grok is added so the classifier is ready the moment a Grok row lands
|
|
203
|
+
// (an agent can add one today via catalog_overrides — see model-router.test.mjs).
|
|
204
|
+
for (const ref of [
|
|
205
|
+
"deepseek/deepseek-v4-flash",
|
|
206
|
+
"moonshot/kimi-k2.6",
|
|
207
|
+
"qwen/qwen3-coder-plus",
|
|
208
|
+
"xai/grok-4-fast",
|
|
209
|
+
"zhipu/glm-5",
|
|
210
|
+
"kimi-k2.6", // bare, no provider segment
|
|
211
|
+
"grok-4-fast", // bare, no provider segment
|
|
212
|
+
]) {
|
|
213
|
+
assert.equal(inferTier({ ref }), "cheap", `${ref} should classify as cheap-tier`);
|
|
214
|
+
}
|
|
215
|
+
});
|
|
216
|
+
|
|
217
|
+
test("inferTier: a cheap provider's WHOLE line is cheap, not just its famous model", () => {
|
|
218
|
+
// The model-name hints ("kimi", "grok", "glm") would carry the refs above on
|
|
219
|
+
// their own, which makes them a trap: the next model each of these providers
|
|
220
|
+
// ships will not be called "grok-4" or "kimi-k2". The provider segment is the
|
|
221
|
+
// durable signal, so these refs — none of which contain a hint word — are the
|
|
222
|
+
// ones that actually hold CHEAP_PROVIDERS in place.
|
|
223
|
+
assert.equal(inferTier({ ref: "xai/sonic-2" }), "cheap");
|
|
224
|
+
assert.equal(inferTier({ ref: "moonshot/k3-preview" }), "cheap");
|
|
225
|
+
assert.equal(inferTier({ ref: "deepseek/v5" }), "cheap");
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
test("inferTier: an unrecognised ref stays null so degradeChain fails OPEN", () => {
|
|
229
|
+
// A model we cannot classify must be KEPT, never dropped: a budget heuristic
|
|
230
|
+
// that silently deletes candidates it doesn't recognise strands the caller
|
|
231
|
+
// with an empty chain. Asserted against a MIXED chain on purpose — with an
|
|
232
|
+
// unknown row alone, the "keep the single cheapest" safety net returns it
|
|
233
|
+
// either way, so a fail-closed filter would slip through unnoticed.
|
|
234
|
+
assert.equal(inferTier({ ref: "acme/some-new-model" }), null);
|
|
235
|
+
const out = degradeChain(
|
|
236
|
+
[{ ref: "anthropic/claude-opus-5" }, { ref: "acme/some-new-model" }],
|
|
237
|
+
150,
|
|
238
|
+
{}
|
|
239
|
+
);
|
|
240
|
+
assert.deepEqual(
|
|
241
|
+
out.map((m) => m.ref),
|
|
242
|
+
["acme/some-new-model"],
|
|
243
|
+
"frontier dropped, the unclassifiable row survives"
|
|
244
|
+
);
|
|
245
|
+
});
|
|
246
|
+
|
|
247
|
+
test("degradeChain: a cheap third-party row survives band 150 alongside fast rows", () => {
|
|
248
|
+
const chain = [
|
|
249
|
+
{ ref: "anthropic/claude-opus-5" },
|
|
250
|
+
{ ref: "anthropic/claude-haiku-4-5" },
|
|
251
|
+
{ ref: "xai/grok-4-fast" },
|
|
252
|
+
];
|
|
253
|
+
const out = degradeChain(chain, 150, {}).map((m) => m.ref);
|
|
254
|
+
assert.deepEqual(out, ["anthropic/claude-haiku-4-5", "xai/grok-4-fast"]);
|
|
255
|
+
});
|
|
256
|
+
|
|
181
257
|
// ---------------------------------------------------------------------------
|
|
182
258
|
// batchLaneFor — classify deferrable vs realtime
|
|
183
259
|
// ---------------------------------------------------------------------------
|
|
@@ -831,10 +831,25 @@ function forceAnthropicDecision(catalog, req, { now, opts, audit, tried = [], re
|
|
|
831
831
|
if (!row) {
|
|
832
832
|
// No catalog at all — degrade to a synthetic chosen so the caller still has
|
|
833
833
|
// a usable model flag (stock `claude` defaults to its account model).
|
|
834
|
+
// Ids current as of 2026-08-13; see pickAnthropicRow's provenance block.
|
|
835
|
+
// No succession fallback is needed here precisely BECAUSE there is no
|
|
836
|
+
// catalog to miss: this id goes straight onto `claude --model`, so it must
|
|
837
|
+
// be the id we actually want the CLI to run.
|
|
838
|
+
// REACHABILITY (2026-08-13): pickAnthropicRow returns null only when the
|
|
839
|
+
// catalog holds no available anthropic row at all, which on a real install
|
|
840
|
+
// means the bundled YAML failed to load. In production the 4-x rows always
|
|
841
|
+
// win before this point, so do NOT cite this branch as proof the SDK emits
|
|
842
|
+
// -5 ids — the only caller that reaches it today is the empty-catalog test.
|
|
843
|
+
// STRUCK (2026-08-13): `ref` was hardcoded to the sonnet ref even when `id`
|
|
844
|
+
// resolved to opus, so a critical/thinking request produced the pair
|
|
845
|
+
// {model: opus, ref: sonnet} — and `ref` is what feeds cacheAffinityKey and
|
|
846
|
+
// chain[0].ref, i.e. the pin and the audit trail both named the wrong model.
|
|
847
|
+
const syntheticId =
|
|
848
|
+
req.needs_thinking || req.priority === "critical" ? "claude-opus-5" : "claude-sonnet-5";
|
|
834
849
|
const synthetic = {
|
|
835
850
|
provider: "anthropic",
|
|
836
|
-
id:
|
|
837
|
-
ref:
|
|
851
|
+
id: syntheticId,
|
|
852
|
+
ref: `anthropic/${syntheticId}`,
|
|
838
853
|
status: "available",
|
|
839
854
|
auth_env: null,
|
|
840
855
|
endpoints: {},
|
|
@@ -951,13 +966,51 @@ function noRouteDecision(req, { now, opts, audit, tried, explainSkips, lane }) {
|
|
|
951
966
|
* Pick the best available Anthropic row from the catalog for a request:
|
|
952
967
|
* frontier (opus) for thinking/critical, else the default workhorse (sonnet),
|
|
953
968
|
* else any available anthropic row.
|
|
969
|
+
*
|
|
970
|
+
* ⚠ READ FIRST: THE -5 REFS BELOW ARE INERT ON THIS TREE. lookupModel misses
|
|
971
|
+
* every one of them, because lib/model-router/catalog/anthropic.yaml still
|
|
972
|
+
* carries only 4-x rows, so the succession tail is what actually answers and
|
|
973
|
+
* `spawnArgs.modelFlag` (= row.id, see modelFlagFor above) puts
|
|
974
|
+
* `claude-sonnet-4-6` / `claude-opus-4-8` on `claude --model` for every request
|
|
975
|
+
* that reaches this safety net — which, with no config/model-routing.yaml in the
|
|
976
|
+
* repo, is every request on the no-v2-config path. Adding the two -5 rows to
|
|
977
|
+
* that YAML is the step that makes this list mean what it says; it has NOT been
|
|
978
|
+
* done. Do not read the ordering below as evidence the SDK runs -5 models.
|
|
979
|
+
*
|
|
980
|
+
* MODEL IDS — PROVENANCE AND WHY THE LEGACY REFS ARE STILL HERE (2026-08-13):
|
|
981
|
+
* the current Anthropic ids are claude-opus-5 / claude-sonnet-5 /
|
|
982
|
+
* claude-haiku-4-5-20251001 (verified against the vendor docs on 2026-08-13;
|
|
983
|
+
* the fast tier did not move). They lead each list. The claude-*-4-x refs
|
|
984
|
+
* BELOW them are not leftovers — they are the succession fallback, and they
|
|
985
|
+
* must stay until lib/model-router/catalog/anthropic.yaml carries -5 rows:
|
|
986
|
+
* lookupModel() is an EXACT byRef hit with no `replaced_by` chasing, so a ref
|
|
987
|
+
* the catalog does not have simply returns null. Listing only the -5 refs
|
|
988
|
+
* today would silently drop this safety net through to "any available
|
|
989
|
+
* anthropic row" — an unordered `models.find()` that would hand critical work
|
|
990
|
+
* whatever row happens to sit first in the file. That silent downgrade is the
|
|
991
|
+
* exact failure this ordered list exists to prevent.
|
|
992
|
+
*
|
|
993
|
+
* REFRESH: this list is the LAST-RESORT floor (kill switch / no config /
|
|
994
|
+
* nothing survived the gates), not the routing table. The routing table is the
|
|
995
|
+
* catalog + the config `aliases:`; refresh those first (see the provenance
|
|
996
|
+
* block in lib/model-router.mjs). When -5 rows land in the bundled catalog,
|
|
997
|
+
* DELETE the 4-x entries here — do not let this list accumulate a model
|
|
998
|
+
* museum, or the "best available" ordering stops meaning anything.
|
|
954
999
|
*/
|
|
955
1000
|
function pickAnthropicRow(catalog, req) {
|
|
956
1001
|
if (!catalog) return null;
|
|
957
1002
|
const wantFrontier = req.needs_thinking || req.priority === "critical" || req.agent_role === "regulatory";
|
|
958
1003
|
const order = wantFrontier
|
|
959
|
-
? [
|
|
960
|
-
|
|
1004
|
+
? [
|
|
1005
|
+
"anthropic/claude-opus-5", "anthropic/claude-opus-4-8",
|
|
1006
|
+
"anthropic/claude-sonnet-5", "anthropic/claude-sonnet-4-6",
|
|
1007
|
+
"anthropic/claude-haiku-4-5",
|
|
1008
|
+
]
|
|
1009
|
+
: [
|
|
1010
|
+
"anthropic/claude-sonnet-5", "anthropic/claude-sonnet-4-6",
|
|
1011
|
+
"anthropic/claude-haiku-4-5",
|
|
1012
|
+
"anthropic/claude-opus-5", "anthropic/claude-opus-4-8",
|
|
1013
|
+
];
|
|
961
1014
|
for (const ref of order) {
|
|
962
1015
|
const row = lookupModel(catalog, ref);
|
|
963
1016
|
if (row && row.status === "available") {
|
package/lib/model-router.mjs
CHANGED
|
@@ -8,15 +8,22 @@
|
|
|
8
8
|
* existing Claude-CLI-on-Max-subscription behaviour is preserved verbatim.
|
|
9
9
|
*
|
|
10
10
|
* Why this exists:
|
|
11
|
-
* - Anthropic
|
|
11
|
+
* - The Anthropic frontier/workhorse pair is still the quality ceiling for
|
|
12
12
|
* agent loops, but background work (classification, log triage,
|
|
13
|
-
* scheduled summaries) does NOT need
|
|
13
|
+
* scheduled summaries) does NOT need it. (STRUCK 2026-08-13: this line
|
|
14
|
+
* used to name "Sonnet 4.6 / Opus 4.7" as the ceiling. Naming versions in
|
|
15
|
+
* a rationale paragraph is how a docblock rots — the CURRENT pair is
|
|
16
|
+
* recorded in exactly one place now, ANTHROPIC_DEFAULT.models below, with
|
|
17
|
+
* its provenance and refresh path.)
|
|
14
18
|
* - Moonshot exposes a native `/anthropic` Messages endpoint, so
|
|
15
19
|
* Kimi K2.6 is a literal drop-in for Claude Code — no router,
|
|
16
20
|
* no translation, no tool-use degradation.
|
|
17
|
-
* -
|
|
18
|
-
*
|
|
19
|
-
*
|
|
21
|
+
* - Qwen is one hop away and significantly cheaper; usable for agents that
|
|
22
|
+
* don't need extended-thinking or parallel tool calls. (A v1 config can
|
|
23
|
+
* reach it through any OpenAI-compat gateway, which is what the v1 tests
|
|
24
|
+
* model; the v2 bundled catalog reaches it on DashScope and marks it
|
|
25
|
+
* harness.session:false — Qwen is a direct/batch lane model there, not a
|
|
26
|
+
* session host.)
|
|
20
27
|
*
|
|
21
28
|
* What this is NOT:
|
|
22
29
|
* - Not a wire-format translator. We rely on the backend speaking
|
|
@@ -90,6 +97,67 @@ try {
|
|
|
90
97
|
// Anthropic-only fallback so existing agents keep working unchanged.
|
|
91
98
|
// ---------------------------------------------------------------------------
|
|
92
99
|
|
|
100
|
+
/**
|
|
101
|
+
* MODEL IDS — PROVENANCE AND REFRESH PATH (last refreshed 2026-08-13)
|
|
102
|
+
* ────────────────────────────────────────────────────────────────────
|
|
103
|
+
* WHERE THIS LIST CAME FROM: the Anthropic model list was verified against the
|
|
104
|
+
* vendor docs on 2026-08-13 (context7 → Anthropic model reference). Current at
|
|
105
|
+
* that date: claude-opus-5 (frontier, 1M ctx, 128k out, $5/$25 per MTok),
|
|
106
|
+
* claude-sonnet-5 (workhorse), claude-haiku-4-5-20251001 (fast/classifier —
|
|
107
|
+
* still current, deliberately NOT bumped). Two more exist and are deliberately
|
|
108
|
+
* absent from this map: claude-fable-5 ($10/$50, long-running agents — a cost
|
|
109
|
+
* decision, not a default) and claude-mythos-5 (restricted to Project
|
|
110
|
+
* Glasswing — must never be offered as a general option).
|
|
111
|
+
*
|
|
112
|
+
* HOW THE NEXT PERSON REFRESHES IT — in order of preference, and note that the
|
|
113
|
+
* first two need NO code change and NO release, which is the whole point:
|
|
114
|
+
*
|
|
115
|
+
* 1. v2 (schema_version: 2) agents: the CATALOG is the source of truth, not
|
|
116
|
+
* this file. Add/replace rows in lib/model-router/catalog/<provider>.yaml
|
|
117
|
+
* (ships with the SDK) or, without touching the SDK at all, patch them
|
|
118
|
+
* per-agent via `catalog_overrides:` in config/model-routing.yaml — that
|
|
119
|
+
* layer outranks the bundled rows (see catalog.mjs's authority merge) and
|
|
120
|
+
* can even introduce a provider the SDK does not bundle. Then repoint the
|
|
121
|
+
* `aliases:` (frontier/default/fast/cheap) at the new refs.
|
|
122
|
+
* 2. v1 agents: declare `backends.anthropic.models` in
|
|
123
|
+
* config/model-routing.yaml. normaliseConfig only injects the map below
|
|
124
|
+
* when `backends.anthropic` is ABSENT, so a config-declared map wins
|
|
125
|
+
* outright — no PR, no build.
|
|
126
|
+
* 3. Editing the map below is the LAST resort: it is the floor for an agent
|
|
127
|
+
* with no routing config at all.
|
|
128
|
+
*
|
|
129
|
+
* WHY THE STALE IDS HERE WERE SURVIVABLE (and so went unnoticed for a release):
|
|
130
|
+
* for `transport: "anthropic-cli"` — the default, keychain-OAuth path — THIS
|
|
131
|
+
* FILE'S modelFlagFor() sends the version-agnostic shorthand
|
|
132
|
+
* "opus"/"sonnet"/"haiku" and lets Anthropic resolve the current version, so a
|
|
133
|
+
* stale id in the map below did not reach the wire. It DID reach
|
|
134
|
+
* `resolved.model`, which is what the cost ledger, telemetry and session records
|
|
135
|
+
* store; a stale id there mis-prices and mis-attributes work. That is the damage
|
|
136
|
+
* this refresh repairs ON THE v1 PATH.
|
|
137
|
+
*
|
|
138
|
+
* ── DO NOT GENERALISE THAT TO THE ROUTER. THE v2 LANE IS STILL STALE. ────────
|
|
139
|
+
* NARROWED 2026-08-13, because the sentence above reads as a router-wide
|
|
140
|
+
* reassurance and is false as one. Two different functions are called
|
|
141
|
+
* `modelFlagFor`:
|
|
142
|
+
* - this file's (v1) — maps tier → "opus"/"sonnet"/"haiku" shorthand, and
|
|
143
|
+
* only for the four known tiers; an unrecognised `request.tier` falls
|
|
144
|
+
* through to `return resolved.model`, the literal id.
|
|
145
|
+
* - resolve.mjs's (v2, line ~468) — `return row.id`, ALWAYS the literal
|
|
146
|
+
* catalog id. spawn.mjs then does `args.push("--model", String(modelFlag))`.
|
|
147
|
+
* So on the v2 lane the catalog id IS the wire value, and the SHIPPED catalog
|
|
148
|
+
* (lib/model-router/catalog/anthropic.yaml) still carries only 4-x rows.
|
|
149
|
+
* Measured on this tree, with no routing config present — the default state,
|
|
150
|
+
* since the repo ships no config/model-routing.yaml at all:
|
|
151
|
+
* ordinary session → claude --model claude-sonnet-4-6
|
|
152
|
+
* critical/thinking → claude --model claude-opus-4-8
|
|
153
|
+
* The -5 ids added to resolve.mjs's pickAnthropicRow preference list are
|
|
154
|
+
* therefore INERT in production: lookupModel misses them against the bundled
|
|
155
|
+
* catalog and the ordered succession tail answers with the 4-x row. They become
|
|
156
|
+
* live the moment anthropic.yaml gains `claude-opus-5` / `claude-sonnet-5` rows,
|
|
157
|
+
* which is the one remaining step of this refresh and is NOT done. Until then,
|
|
158
|
+
* "the agent-sdk uses the latest Anthropic models" is true of the v1 floor only.
|
|
159
|
+
* See the guard test "the SHIPPED catalog is what decides the wire value".
|
|
160
|
+
*/
|
|
93
161
|
const ANTHROPIC_DEFAULT = Object.freeze({
|
|
94
162
|
transport: "anthropic-cli", // special: do nothing (current behaviour)
|
|
95
163
|
base_url: "https://api.anthropic.com",
|
|
@@ -102,11 +170,30 @@ const ANTHROPIC_DEFAULT = Object.freeze({
|
|
|
102
170
|
"parallel_tools",
|
|
103
171
|
"long_context_1m",
|
|
104
172
|
],
|
|
105
|
-
|
|
173
|
+
// $/MTok. STRUCK "Sonnet 4.6 ballpark" (2026-08-13): these figures are
|
|
174
|
+
// LINE-INHERITED from Sonnet 4.6's list price and are NOT vendor-confirmed for
|
|
175
|
+
// Sonnet 5 — only Opus 5 ($5/$25) was confirmed in the 2026-08-13 sweep. The
|
|
176
|
+
// v2 lane does not use this at all (it prices from the catalog row's `cost`
|
|
177
|
+
// block, with its own cost_provenance stamp).
|
|
178
|
+
//
|
|
179
|
+
// NARROWED 2026-08-13: an earlier version of this note opened "$/MTok for the
|
|
180
|
+
// `default` tier", which implies a per-tier price. THERE IS NO PER-TIER PRICE
|
|
181
|
+
// ON THE v1 LANE. `resolveBackend` returns `pricing: backend.pricing` — this
|
|
182
|
+
// one block — for every tier, and `estimateCost` multiplies by it regardless
|
|
183
|
+
// of which model was chosen. So premium work is billed at the WORKHORSE rate:
|
|
184
|
+
// with premium now claude-opus-5 at a confirmed $5/$25, every v1 opus estimate
|
|
185
|
+
// is low by ~1.7x on both input and output, and always was (Opus 4.7 was
|
|
186
|
+
// dearer than $3/$15 too). Refreshing the ids did not touch this and did not
|
|
187
|
+
// cause it. Fixing it means per-tier pricing in the backend shape — a schema
|
|
188
|
+
// change with config-compat consequences, deliberately not smuggled into an
|
|
189
|
+
// id refresh. Until then, treat v1 `estimateCost` as a floor, not a bill.
|
|
190
|
+
pricing: { input_per_m: 3.0, output_per_m: 15.0 },
|
|
106
191
|
models: {
|
|
107
192
|
classifier: "claude-haiku-4-5-20251001",
|
|
108
|
-
default: "claude-sonnet-
|
|
109
|
-
premium: "claude-opus-
|
|
193
|
+
default: "claude-sonnet-5",
|
|
194
|
+
premium: "claude-opus-5",
|
|
195
|
+
// Current as of 2026-08-13 — the fast tier did NOT move when opus/sonnet
|
|
196
|
+
// did. Do not "helpfully" bump this to a -5 id that does not exist.
|
|
110
197
|
fast: "claude-haiku-4-5-20251001",
|
|
111
198
|
},
|
|
112
199
|
});
|