auto-model-router 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +50 -0
- package/package.json +1 -1
- package/src/catalog/composite.ts +16 -1
- package/src/catalog/static-catalog.ts +89 -0
- package/src/catalog/types.ts +2 -2
- package/src/config/apply.ts +5 -1
- package/src/config/defaults.ts +3 -0
- package/src/config/load.ts +3 -0
- package/src/config/schema.ts +38 -0
- package/src/config/types.ts +55 -0
- package/src/config/upstreams.ts +31 -0
- package/src/cost/report.ts +23 -2
- package/src/cost/views.ts +2 -1
- package/src/lib.ts +2 -1
- package/src/server/http.ts +12 -2
- package/src/server/providers.ts +34 -1
- package/src/upstream/anthropic.ts +470 -0
- package/src/upstream/compat.ts +267 -0
- package/src/upstream/multi.ts +23 -9
- package/test/config-wizard.test.ts +1 -1
- package/test/failover.test.ts +1 -0
- package/test/turn.test.ts +1 -0
- package/test/upstreams.test.ts +378 -0
|
@@ -0,0 +1,378 @@
|
|
|
1
|
+
import { describe, expect, test } from "bun:test";
|
|
2
|
+
|
|
3
|
+
import { createCompositeCatalog } from "../src/catalog/composite.ts";
|
|
4
|
+
import type { OllamaCatalogSource } from "../src/catalog/ollama-catalog.ts";
|
|
5
|
+
import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
6
|
+
import { buildUpstreamModels, createStaticCatalogSource } from "../src/catalog/static-catalog.ts";
|
|
7
|
+
import type { CatalogModel, CatalogSnapshot, CatalogSource } from "../src/catalog/types.ts";
|
|
8
|
+
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
9
|
+
import { configInputSchema, RESERVED_UPSTREAM_IDS } from "../src/config/schema.ts";
|
|
10
|
+
import type { RouterConfig, UpstreamEntry } from "../src/config/types.ts";
|
|
11
|
+
import { completeUpstreamEntry } from "../src/config/upstreams.ts";
|
|
12
|
+
import { applyConfigPatch } from "../src/config/apply.ts";
|
|
13
|
+
import { providerOfSlug, setKnownUpstreamIds } from "../src/cost/report.ts";
|
|
14
|
+
import { NO_USAGE } from "../src/upstream/ollama-usage.ts";
|
|
15
|
+
import { ANTHROPIC_VERSION, classifyAnthropicStatus, createAnthropicClient, createAnthropicTranslator, readSseFrames, toAnthropicBody } from "../src/upstream/anthropic.ts";
|
|
16
|
+
import { classifyCompatStatus, compatEndpoint, createCompatClient, toCompatBody } from "../src/upstream/compat.ts";
|
|
17
|
+
import { createMultiUpstream, namedUpstreamOf } from "../src/upstream/multi.ts";
|
|
18
|
+
import type { Dispatch, DispatchOptions, UpstreamClient } from "../src/upstream/types.ts";
|
|
19
|
+
|
|
20
|
+
const NL = String.fromCharCode(10);
|
|
21
|
+
|
|
22
|
+
function entry(over: Partial<UpstreamEntry> & { id: string; kind: UpstreamEntry["kind"] }): UpstreamEntry {
|
|
23
|
+
return completeUpstreamEntry({ baseUrl: "https://api.example/v1", apiKey: "sk-x", models: [{ id: "m1", input: 1, output: 4 }], ...over });
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function cfgWith(upstreams: UpstreamEntry[], logLevel: RouterConfig["logLevel"] = "silent"): RouterConfig {
|
|
27
|
+
return { ...structuredClone(DEFAULT_CONFIG), upstreams, logLevel };
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** An OpenRouter-shaped record, for twins. */
|
|
31
|
+
function orRaw(id: string, coding: number, image = false): Record<string, unknown> {
|
|
32
|
+
return {
|
|
33
|
+
id,
|
|
34
|
+
canonical_slug: id,
|
|
35
|
+
name: id,
|
|
36
|
+
context_length: 128_000,
|
|
37
|
+
top_provider: { max_completion_tokens: 16_000 },
|
|
38
|
+
pricing: { prompt: "0.0000025", completion: "0.00001" },
|
|
39
|
+
supported_parameters: ["tools", "reasoning", "tool_choice"],
|
|
40
|
+
architecture: { input_modalities: image ? ["text", "image"] : ["text"], tokenizer: "GPT" },
|
|
41
|
+
benchmarks: { artificial_analysis: { coding_index: coding, intelligence_index: coding - 10, agentic_index: coding - 20 } },
|
|
42
|
+
created: 1_700_000_000,
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function sse(frames: string[]): Response {
|
|
47
|
+
return new Response(new ReadableStream({ start: (c) => { for (const f of frames) c.enqueue(new TextEncoder().encode(`${f}${NL}${NL}`)); c.close(); } }), { status: 200, headers: { "content-type": "text/event-stream" } });
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
describe("upstreams config", () => {
|
|
51
|
+
test("an entry is accepted with defaults filled; a reserved or malformed id is refused", () => {
|
|
52
|
+
const ok = configInputSchema.safeParse({ upstreams: [{ id: "openai-direct", kind: "openai", baseUrl: "https://api.openai.com/v1", apiKey: "sk", models: [{ id: "gpt-4o", input: 2.5, output: 10 }] }] });
|
|
53
|
+
expect(ok.success).toBe(true);
|
|
54
|
+
const full = completeUpstreamEntry({ id: "vllm", kind: "openai", baseUrl: "http://vllm:8000/v1", models: [] });
|
|
55
|
+
expect(full).toMatchObject({ enabled: true, apiKey: "", apiVersion: "2024-10-21", headers: {}, timeoutMs: 600_000, rateLimitCooldownMs: 60_000, quotaCooldownMs: 900_000 });
|
|
56
|
+
for (const id of ["openai", "anthropic", "ollama", "openrouter"]) {
|
|
57
|
+
expect(RESERVED_UPSTREAM_IDS).toContain(id);
|
|
58
|
+
expect(configInputSchema.safeParse({ upstreams: [{ id, kind: "openai", baseUrl: "https://x", models: [] }] }).success).toBe(false);
|
|
59
|
+
}
|
|
60
|
+
expect(configInputSchema.safeParse({ upstreams: [{ id: "Bad Id", kind: "openai", baseUrl: "https://x", models: [] }] }).success).toBe(false);
|
|
61
|
+
expect(configInputSchema.safeParse({ upstreams: [{ id: "x", kind: "bedrock", baseUrl: "https://x", models: [] }] }).success).toBe(false);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test("a live patch replaces the list and completes sparse entries", () => {
|
|
65
|
+
const cfg = cfgWith([]);
|
|
66
|
+
const changed = applyConfigPatch(cfg, { upstreams: [{ id: "azure-eu", kind: "azure", baseUrl: "https://r.openai.azure.com", apiKey: "k", models: [{ id: "gpt-4o-deploy", input: 2.5, output: 10 }] }] } as never);
|
|
67
|
+
expect(changed).toEqual(["upstreams"]);
|
|
68
|
+
expect(cfg.upstreams[0]).toMatchObject({ id: "azure-eu", enabled: true, apiVersion: "2024-10-21", headers: {}, timeoutMs: 600_000 });
|
|
69
|
+
});
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
describe("static catalog", () => {
|
|
73
|
+
const twins = [normalizeCatalogModel(orRaw("openai/gpt-4o", 70, true))!, normalizeCatalogModel(orRaw("anthropic/claude-sonnet-4", 80))!];
|
|
74
|
+
|
|
75
|
+
test("models are priced per token, namespaced by the entry id, and borrow the twin's scores and modalities", () => {
|
|
76
|
+
const e = entry({ id: "openai-direct", kind: "openai", models: [{ id: "gpt-4o", input: 2.5, output: 10, cachedInput: 1.25 }, { id: "custom-ft", input: 3, output: 12, contextLength: 32_000, quality: { coding: 55 }, supportsTools: false }] });
|
|
77
|
+
const models = buildUpstreamModels(e, twins);
|
|
78
|
+
expect(models.map((m) => m.slug)).toEqual(["openai-direct/gpt-4o", "openai-direct/custom-ft"]);
|
|
79
|
+
const gpt = models[0]!;
|
|
80
|
+
expect(gpt.provider).toBe("openai-direct");
|
|
81
|
+
expect(gpt.price).toEqual({ prompt: 2.5e-6, completion: 1e-5, cacheRead: 1.25e-6 });
|
|
82
|
+
expect(gpt.quality).toEqual({ coding: 70, intelligence: 60, agentic: 50 }); // the twin openai/gpt-4o
|
|
83
|
+
expect(gpt.inputModalities).toEqual(["text", "image"]);
|
|
84
|
+
expect(gpt.contextLength).toBe(128_000);
|
|
85
|
+
expect(gpt.maxCompletionTokens).toBe(16_000);
|
|
86
|
+
expect(gpt.tokenizer).toBe("GPT");
|
|
87
|
+
const ft = models[1]!;
|
|
88
|
+
expect(ft.quality).toEqual({ coding: 55 });
|
|
89
|
+
expect(ft.contextLength).toBe(32_000);
|
|
90
|
+
expect(ft.supportsTools).toBe(false);
|
|
91
|
+
expect(ft.inputModalities).toEqual(["text"]);
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
test("an explicit twin wins over the name match; an anthropic entry defaults to the Claude tokenizer", () => {
|
|
95
|
+
const e = entry({ id: "anthropic-direct", kind: "anthropic", models: [{ id: "claude-sonnet-4-20250514", input: 3, output: 15, twin: "anthropic/claude-sonnet-4", cacheWrite: 3.75, cachedInput: 0.3 }, { id: "claude-unknown", input: 1, output: 5 }] });
|
|
96
|
+
const [sonnet, unknown] = buildUpstreamModels(e, twins);
|
|
97
|
+
expect(sonnet!.quality.coding).toBe(80);
|
|
98
|
+
expect(sonnet!.price.cacheWrite).toBe(3.75e-6);
|
|
99
|
+
expect(unknown!.quality).toEqual({});
|
|
100
|
+
expect(unknown!.tokenizer).toBe("Claude");
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
test("the source rebuilds only when the entries or the OpenRouter models change", () => {
|
|
104
|
+
const cfg = cfgWith([entry({ id: "vllm", kind: "openai" })]);
|
|
105
|
+
const src = createStaticCatalogSource(cfg);
|
|
106
|
+
const first = src.get(twins);
|
|
107
|
+
expect(src.get(twins)).toBe(first);
|
|
108
|
+
applyConfigPatch(cfg, { upstreams: [{ id: "vllm", kind: "openai", baseUrl: "http://vllm:8000/v1", models: [{ id: "llama", input: 0, output: 0 }] }] } as never);
|
|
109
|
+
const second = src.get(twins);
|
|
110
|
+
expect(second).not.toBe(first);
|
|
111
|
+
expect(second[0]!.isFree).toBe(true);
|
|
112
|
+
});
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
describe("dispatch by slug prefix", () => {
|
|
116
|
+
const fake = (name: string, seen: string[]): UpstreamClient => ({
|
|
117
|
+
dispatch: async (o: DispatchOptions): Promise<Dispatch> => {
|
|
118
|
+
seen.push(`${name}:${String(o.body.model)}`);
|
|
119
|
+
return { chunks: (async function* () {})(), generationId: async () => null };
|
|
120
|
+
},
|
|
121
|
+
complete: async (b) => {
|
|
122
|
+
seen.push(`${name}:complete:${String(b.model)}`);
|
|
123
|
+
return { text: "", costUsd: null };
|
|
124
|
+
},
|
|
125
|
+
fetchModels: async () => [],
|
|
126
|
+
fetchModelsForUser: async () => [],
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
test("ollama/, a named id, and everything else", async () => {
|
|
130
|
+
const seen: string[] = [];
|
|
131
|
+
const named = new Map([["azure-eu", fake("azure", seen)]]);
|
|
132
|
+
const multi = createMultiUpstream(fake("openrouter", seen), fake("ollama", seen), (id) => named.get(id), () => named.keys());
|
|
133
|
+
const signal = new AbortController().signal;
|
|
134
|
+
await multi.dispatch({ body: { model: "ollama/glm" }, sessionId: "s", signal });
|
|
135
|
+
await multi.dispatch({ body: { model: "azure-eu/gpt-4o" }, sessionId: "s", signal });
|
|
136
|
+
await multi.dispatch({ body: { model: "openai/gpt-4o" }, sessionId: "s", signal });
|
|
137
|
+
await multi.complete({ model: "azure-eu/gpt-4o" }, signal);
|
|
138
|
+
expect(seen).toEqual(["ollama:ollama/glm", "azure:azure-eu/gpt-4o", "openrouter:openai/gpt-4o", "azure:complete:azure-eu/gpt-4o"]);
|
|
139
|
+
expect(namedUpstreamOf("openai/gpt-4o", ["openai-direct"])).toBeNull();
|
|
140
|
+
expect(namedUpstreamOf("openai-direct/gpt-4o", ["openai-direct"])).toBe("openai-direct");
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
test("the ledger names the provider of a slug from the known ids", () => {
|
|
144
|
+
setKnownUpstreamIds(["azure-eu", "bad id"]);
|
|
145
|
+
expect(providerOfSlug("azure-eu/gpt-4o")).toBe("azure-eu");
|
|
146
|
+
expect(providerOfSlug("openai/gpt-4o")).toBe("openrouter");
|
|
147
|
+
expect(providerOfSlug("ollama/glm")).toBe("ollama");
|
|
148
|
+
expect(providerOfSlug("bad id/x")).toBe("openrouter");
|
|
149
|
+
setKnownUpstreamIds([]);
|
|
150
|
+
});
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
describe("the OpenAI-compatible client", () => {
|
|
154
|
+
test("the body loses the router's OpenRouter dialect: prefix, cascade, session, cache markers; reasoning becomes reasoning_effort", () => {
|
|
155
|
+
const out = toCompatBody("vllm", { model: "vllm/llama", models: ["vllm/llama", "vllm/other"], session_id: "s", stream: true, reasoning: { effort: "xhigh" }, messages: [{ role: "system", content: [{ type: "text", text: "sys", cache_control: { type: "ephemeral" } }] }, { role: "user", content: "hi" }] });
|
|
156
|
+
expect(out.model).toBe("llama");
|
|
157
|
+
expect(out.models).toBeUndefined();
|
|
158
|
+
expect(out.session_id).toBeUndefined();
|
|
159
|
+
expect(out.reasoning).toBeUndefined();
|
|
160
|
+
expect(out.reasoning_effort).toBe("high");
|
|
161
|
+
expect(out.stream_options).toEqual({ include_usage: true });
|
|
162
|
+
expect((out.messages as { content: Record<string, unknown>[] }[])[0]!.content[0]).toEqual({ type: "text", text: "sys" });
|
|
163
|
+
expect(toCompatBody("vllm", { model: "vllm/llama", reasoning: { enabled: false } }).reasoning_effort).toBeUndefined();
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
test("endpoints: OpenAI bears a token; Azure names the deployment in the path and keys with api-key", () => {
|
|
167
|
+
const oa = compatEndpoint(entry({ id: "openai-direct", kind: "openai", baseUrl: "https://api.openai.com/v1/", headers: { "x-org": "o" } }), "gpt-4o");
|
|
168
|
+
expect(oa.url).toBe("https://api.openai.com/v1/chat/completions");
|
|
169
|
+
expect(oa.headers).toMatchObject({ authorization: "Bearer sk-x", "x-org": "o" });
|
|
170
|
+
const az = compatEndpoint(entry({ id: "azure-eu", kind: "azure", baseUrl: "https://r.openai.azure.com", apiVersion: "2024-10-21" }), "gpt-4o-deploy");
|
|
171
|
+
expect(az.url).toBe("https://r.openai.azure.com/openai/deployments/gpt-4o-deploy/chat/completions?api-version=2024-10-21");
|
|
172
|
+
expect(az.headers["api-key"]).toBe("sk-x");
|
|
173
|
+
expect(az.headers.authorization).toBeUndefined();
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
test("statuses: OpenAI's insufficient_quota 429 is the account, a plain 429 the moment; 400 context is final", () => {
|
|
177
|
+
expect(classifyCompatStatus("x", 429, { error: { code: "insufficient_quota", message: "You exceeded your current quota" } })).toMatchObject({ kind: "quota", retryable: true });
|
|
178
|
+
expect(classifyCompatStatus("x", 429, { error: { message: "Rate limit reached" } })).toMatchObject({ kind: "rate_limit", retryable: true });
|
|
179
|
+
expect(classifyCompatStatus("x", 400, { error: { message: "This model's maximum context length is 8192 tokens" } })).toMatchObject({ kind: "context_length", retryable: false });
|
|
180
|
+
expect(classifyCompatStatus("x", 401, {})).toMatchObject({ kind: "auth", retryable: false });
|
|
181
|
+
expect(classifyCompatStatus("x", 503, {})).toMatchObject({ kind: "upstream_error", retryable: true });
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
test("dispatch: the served model is re-prefixed, the key travels, and a quota answer opens the breaker", async () => {
|
|
185
|
+
let sent: Record<string, unknown> | null = null;
|
|
186
|
+
let url = "";
|
|
187
|
+
let auth: string | null = null;
|
|
188
|
+
const fetchImpl = async (u: string, init?: RequestInit): Promise<Response> => {
|
|
189
|
+
url = u;
|
|
190
|
+
sent = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
191
|
+
auth = (init?.headers as Record<string, string>).authorization ?? null;
|
|
192
|
+
return sse([
|
|
193
|
+
`data: ${JSON.stringify({ id: "g1", model: "gpt-4o-2024-08-06", choices: [{ index: 0, delta: { content: "hi" }, finish_reason: null }] })}`,
|
|
194
|
+
`data: ${JSON.stringify({ id: "g1", model: "gpt-4o-2024-08-06", choices: [{ index: 0, delta: {}, finish_reason: "stop" }], usage: { prompt_tokens: 10, completion_tokens: 2, prompt_tokens_details: { cached_tokens: 4 } } })}`,
|
|
195
|
+
"data: [DONE]",
|
|
196
|
+
]);
|
|
197
|
+
};
|
|
198
|
+
const cfg = cfgWith([entry({ id: "openai-direct", kind: "openai", baseUrl: "https://api.openai.com/v1", models: [{ id: "gpt-4o", input: 2.5, output: 10 }] })]);
|
|
199
|
+
const client = createCompatClient(cfg, "openai-direct", fetchImpl);
|
|
200
|
+
const d = await client.dispatch({ body: { model: "openai-direct/gpt-4o", messages: [], stream: true }, sessionId: "s", signal: new AbortController().signal });
|
|
201
|
+
const chunks = [];
|
|
202
|
+
for await (const c of d.chunks) chunks.push(c);
|
|
203
|
+
expect(url).toBe("https://api.openai.com/v1/chat/completions");
|
|
204
|
+
expect(sent!.model).toBe("gpt-4o");
|
|
205
|
+
expect(auth as string | null).toBe("Bearer sk-x");
|
|
206
|
+
const start = chunks[0]!.events.find((e) => e.type === "start");
|
|
207
|
+
expect(start && start.type === "start" ? start.servedSlug : null).toBe("openai-direct/gpt-4o-2024-08-06");
|
|
208
|
+
expect(chunks[0]!.raw.model).toBe("openai-direct/gpt-4o-2024-08-06");
|
|
209
|
+
expect(await d.generationId()).toBe("g1");
|
|
210
|
+
const usage = chunks[1]!.events.find((e) => e.type === "usage");
|
|
211
|
+
expect(usage && usage.type === "usage" ? usage.usage.cachedTokens : null).toBe(4);
|
|
212
|
+
|
|
213
|
+
// Out of quota: retryable elsewhere, and this upstream is hidden for its quota cooldown.
|
|
214
|
+
const broke = createCompatClient(cfg, "openai-direct", async () => Response.json({ error: { code: "insufficient_quota", message: "quota" } }, { status: 429 }));
|
|
215
|
+
let caught: unknown = null;
|
|
216
|
+
try {
|
|
217
|
+
await broke.dispatch({ body: { model: "openai-direct/gpt-4o", messages: [] }, sessionId: "s", signal: new AbortController().signal });
|
|
218
|
+
} catch (err) {
|
|
219
|
+
caught = err;
|
|
220
|
+
}
|
|
221
|
+
expect(caught).toMatchObject({ kind: "quota", retryable: true });
|
|
222
|
+
expect(broke.available()).toBe(false);
|
|
223
|
+
expect(broke.lastTrip()?.kind).toBe("quota");
|
|
224
|
+
// The live key applies to the next call without a new client.
|
|
225
|
+
applyConfigPatch(cfg, { upstreams: [{ id: "openai-direct", kind: "openai", baseUrl: "https://api.openai.com/v1", apiKey: "sk-new", models: [{ id: "gpt-4o", input: 2.5, output: 10 }] }] } as never);
|
|
226
|
+
await client.dispatch({ body: { model: "openai-direct/gpt-4o", messages: [], stream: true }, sessionId: "s", signal: new AbortController().signal });
|
|
227
|
+
expect(auth as string | null).toBe("Bearer sk-new");
|
|
228
|
+
});
|
|
229
|
+
});
|
|
230
|
+
|
|
231
|
+
describe("the Anthropic client", () => {
|
|
232
|
+
test("the request: system blocks keep cache markers, turns alternate, tools and results map, thinking follows the effort", () => {
|
|
233
|
+
const body = {
|
|
234
|
+
model: "anthropic-direct/claude-sonnet-4",
|
|
235
|
+
stream: true,
|
|
236
|
+
max_tokens: 1000,
|
|
237
|
+
temperature: 0.2,
|
|
238
|
+
reasoning: { effort: "medium" },
|
|
239
|
+
tool_choice: "required",
|
|
240
|
+
tools: [{ type: "function", function: { name: "read", description: "read a file", parameters: { type: "object", properties: { path: { type: "string" } } } } }],
|
|
241
|
+
messages: [
|
|
242
|
+
{ role: "system", content: [{ type: "text", text: "be brief", cache_control: { type: "ephemeral" } }] },
|
|
243
|
+
{ role: "user", content: [{ type: "text", text: "look" }, { type: "image_url", image_url: { url: "data:image/png;base64,AAAA" } }] },
|
|
244
|
+
{ role: "assistant", content: "", tool_calls: [{ id: "call_1", type: "function", function: { name: "read", arguments: "{\"path\":\"a.ts\"}" } }] },
|
|
245
|
+
{ role: "tool", tool_call_id: "call_1", content: "contents of a" },
|
|
246
|
+
{ role: "tool", tool_call_id: "call_2", content: [{ type: "text", text: "second" }] },
|
|
247
|
+
{ role: "user", content: "and now?" },
|
|
248
|
+
],
|
|
249
|
+
};
|
|
250
|
+
const out = toAnthropicBody(body, { modelId: "claude-sonnet-4", supportsReasoning: true, maxCompletionTokens: 64_000 });
|
|
251
|
+
expect(out.model).toBe("claude-sonnet-4");
|
|
252
|
+
expect(out.system).toEqual([{ type: "text", text: "be brief", cache_control: { type: "ephemeral" } }]);
|
|
253
|
+
const msgs = out.messages as { role: string; content: Record<string, unknown>[] }[];
|
|
254
|
+
expect(msgs.map((m) => m.role)).toEqual(["user", "assistant", "user"]);
|
|
255
|
+
expect(msgs[0]!.content[1]).toEqual({ type: "image", source: { type: "base64", media_type: "image/png", data: "AAAA" } });
|
|
256
|
+
expect(msgs[1]!.content).toEqual([{ type: "tool_use", id: "call_1", name: "read", input: { path: "a.ts" } }]);
|
|
257
|
+
// Two tool results and the next user turn fold into one user message.
|
|
258
|
+
expect(msgs[2]!.content).toEqual([{ type: "tool_result", tool_use_id: "call_1", content: "contents of a" }, { type: "tool_result", tool_use_id: "call_2", content: "second" }, { type: "text", text: "and now?" }]);
|
|
259
|
+
expect(out.tools).toEqual([{ name: "read", description: "read a file", input_schema: { type: "object", properties: { path: { type: "string" } } } }]);
|
|
260
|
+
expect(out.tool_choice).toEqual({ type: "any", disable_parallel_tool_use: false });
|
|
261
|
+
// Thinking: the budget for medium, max_tokens raised above it, sampling knobs dropped.
|
|
262
|
+
expect(out.thinking).toEqual({ type: "enabled", budget_tokens: 8192 });
|
|
263
|
+
expect(out.max_tokens).toBe(8192 + 1024);
|
|
264
|
+
expect(out.temperature).toBeUndefined();
|
|
265
|
+
// Without reasoning support the effort is ignored and the caller's cap stands, bounded by the model's.
|
|
266
|
+
const plain = toAnthropicBody({ ...body, max_tokens: 100_000, reasoning: undefined }, { modelId: "m", supportsReasoning: false, maxCompletionTokens: 8192 });
|
|
267
|
+
expect(plain.thinking).toBeUndefined();
|
|
268
|
+
expect(plain.max_tokens).toBe(8192);
|
|
269
|
+
expect(plain.temperature).toBe(0.2);
|
|
270
|
+
});
|
|
271
|
+
|
|
272
|
+
test("the stream: message_start opens, text and tool blocks become chunks, message_delta closes with usage in the OpenAI convention", () => {
|
|
273
|
+
const t = createAnthropicTranslator("anthropic-direct/claude-sonnet-4");
|
|
274
|
+
const push = (event: string, data: Record<string, unknown>) => t.push({ event, data: JSON.stringify(data) });
|
|
275
|
+
const start = push("message_start", { type: "message_start", message: { id: "msg_1", model: "claude-sonnet-4-20250514", usage: { input_tokens: 100, cache_read_input_tokens: 40, cache_creation_input_tokens: 10 } } })!;
|
|
276
|
+
expect(start.events).toEqual([{ type: "start", servedSlug: "anthropic-direct/claude-sonnet-4", generationId: "msg_1" }]);
|
|
277
|
+
expect(start.raw).toMatchObject({ id: "msg_1", model: "anthropic-direct/claude-sonnet-4", object: "chat.completion.chunk" });
|
|
278
|
+
expect(push("ping", { type: "ping" })).toBeNull();
|
|
279
|
+
expect(push("content_block_start", { type: "content_block_start", index: 0, content_block: { type: "text", text: "" } })).toBeNull();
|
|
280
|
+
const text = push("content_block_delta", { type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "Hello" } })!;
|
|
281
|
+
expect(text.events).toEqual([{ type: "text", delta: "Hello" }]);
|
|
282
|
+
expect((text.raw.choices as { delta: { content: string } }[])[0]!.delta.content).toBe("Hello");
|
|
283
|
+
const think = push("content_block_delta", { type: "content_block_delta", index: 0, delta: { type: "thinking_delta", thinking: "hmm" } })!;
|
|
284
|
+
expect(think.events).toEqual([{ type: "reasoning", delta: "hmm" }]);
|
|
285
|
+
const toolStart = push("content_block_start", { type: "content_block_start", index: 1, content_block: { type: "tool_use", id: "toolu_1", name: "read", input: {} } })!;
|
|
286
|
+
expect(toolStart.events).toEqual([{ type: "tool_call", index: 0, id: "toolu_1", name: "read" }]);
|
|
287
|
+
expect((toolStart.raw.choices as { delta: { tool_calls: unknown[] } }[])[0]!.delta.tool_calls).toEqual([{ index: 0, id: "toolu_1", type: "function", function: { name: "read", arguments: "" } }]);
|
|
288
|
+
const args = push("content_block_delta", { type: "content_block_delta", index: 1, delta: { type: "input_json_delta", partial_json: "{\"path\":" } })!;
|
|
289
|
+
expect(args.events).toEqual([{ type: "tool_call", index: 0, argsDelta: "{\"path\":" }]);
|
|
290
|
+
// A second tool block gets the next tool index whatever its block index.
|
|
291
|
+
const tool2 = push("content_block_start", { type: "content_block_start", index: 3, content_block: { type: "tool_use", id: "toolu_2", name: "write", input: {} } })!;
|
|
292
|
+
expect(tool2.events[0]).toMatchObject({ type: "tool_call", index: 1, id: "toolu_2" });
|
|
293
|
+
const end = push("message_delta", { type: "message_delta", delta: { stop_reason: "tool_use" }, usage: { output_tokens: 25 } })!;
|
|
294
|
+
expect(end.events).toEqual([
|
|
295
|
+
{ type: "finish", reason: "tool_calls" },
|
|
296
|
+
{ type: "usage", usage: { promptTokens: 150, cachedTokens: 40, cacheWriteTokens: 10, completionTokens: 25, reasoningTokens: 0, images: 0 }, reportedCostUsd: null },
|
|
297
|
+
]);
|
|
298
|
+
expect(end.raw.usage).toEqual({ prompt_tokens: 150, completion_tokens: 25, total_tokens: 175, prompt_tokens_details: { cached_tokens: 40, cache_write_tokens: 10 } });
|
|
299
|
+
expect((end.raw.choices as { finish_reason: string }[])[0]!.finish_reason).toBe("tool_calls");
|
|
300
|
+
expect(push("message_stop", { type: "message_stop" })).toBeNull();
|
|
301
|
+
expect(() => push("error", { type: "error", error: { type: "overloaded_error", message: "Overloaded" } })).toThrow("Overloaded");
|
|
302
|
+
});
|
|
303
|
+
|
|
304
|
+
test("SSE frames keep their event names, across split chunks and CRLF", async () => {
|
|
305
|
+
const bytes = new TextEncoder().encode("event: message_start\r\ndata: {\"a\":1}\r\n\r\n: keepalive\r\nevent: ping\r\ndata: {}\r\n\r\ndata: {\"tail\":true}");
|
|
306
|
+
const stream = new ReadableStream({ start: (c) => { c.enqueue(bytes.slice(0, 20)); c.enqueue(bytes.slice(20)); c.close(); } });
|
|
307
|
+
const frames = [];
|
|
308
|
+
for await (const f of readSseFrames(stream)) frames.push(f);
|
|
309
|
+
expect(frames).toEqual([{ event: "message_start", data: "{\"a\":1}" }, { event: "ping", data: "{}" }, { event: "", data: "{\"tail\":true}" }]);
|
|
310
|
+
});
|
|
311
|
+
|
|
312
|
+
test("dispatch: the Messages request goes out with the version and key, and the chunks come back router-shaped", async () => {
|
|
313
|
+
let url = "";
|
|
314
|
+
let headers: Record<string, string> = {};
|
|
315
|
+
let sent: Record<string, unknown> | null = null;
|
|
316
|
+
const fetchImpl = async (u: string, init?: RequestInit): Promise<Response> => {
|
|
317
|
+
url = u;
|
|
318
|
+
headers = init?.headers as Record<string, string>;
|
|
319
|
+
sent = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
320
|
+
return sse([
|
|
321
|
+
`event: message_start${NL}data: ${JSON.stringify({ type: "message_start", message: { id: "msg_9", model: "claude-sonnet-4-20250514", usage: { input_tokens: 5 } } })}`,
|
|
322
|
+
`event: content_block_delta${NL}data: ${JSON.stringify({ type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "pong" } })}`,
|
|
323
|
+
`event: message_delta${NL}data: ${JSON.stringify({ type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 1 } })}`,
|
|
324
|
+
`event: message_stop${NL}data: {"type":"message_stop"}`,
|
|
325
|
+
]);
|
|
326
|
+
};
|
|
327
|
+
const cfg = cfgWith([entry({ id: "anthropic-direct", kind: "anthropic", baseUrl: "https://api.anthropic.com", apiKey: "sk-ant", models: [{ id: "claude-sonnet-4", input: 3, output: 15, supportsReasoning: true, maxCompletionTokens: 64_000 }] })]);
|
|
328
|
+
const client = createAnthropicClient(cfg, "anthropic-direct", fetchImpl);
|
|
329
|
+
const d = await client.dispatch({ body: { model: "anthropic-direct/claude-sonnet-4", messages: [{ role: "user", content: "ping" }], max_tokens: 50 }, sessionId: "s", signal: new AbortController().signal });
|
|
330
|
+
const chunks = [];
|
|
331
|
+
for await (const c of d.chunks) chunks.push(c);
|
|
332
|
+
expect(url).toBe("https://api.anthropic.com/v1/messages");
|
|
333
|
+
expect(headers).toMatchObject({ "x-api-key": "sk-ant", "anthropic-version": ANTHROPIC_VERSION });
|
|
334
|
+
expect(sent).toMatchObject({ model: "claude-sonnet-4", stream: true, max_tokens: 50, messages: [{ role: "user", content: [{ type: "text", text: "ping" }] }] });
|
|
335
|
+
expect(await d.generationId()).toBe("msg_9");
|
|
336
|
+
const types = chunks.flatMap((c) => c.events.map((e) => e.type));
|
|
337
|
+
expect(types).toEqual(["start", "text", "finish", "usage"]);
|
|
338
|
+
expect(chunks[0]!.raw.model).toBe("anthropic-direct/claude-sonnet-4");
|
|
339
|
+
// A 529 is a moment: retryable, and the breaker opens for the rate-limit cooldown.
|
|
340
|
+
const overloaded = createAnthropicClient(cfg, "anthropic-direct", async () => Response.json({ error: { type: "overloaded_error", message: "Overloaded" } }, { status: 529 }));
|
|
341
|
+
let caught: unknown = null;
|
|
342
|
+
try {
|
|
343
|
+
await overloaded.dispatch({ body: { model: "anthropic-direct/claude-sonnet-4", messages: [] }, sessionId: "s", signal: new AbortController().signal });
|
|
344
|
+
} catch (err) {
|
|
345
|
+
caught = err;
|
|
346
|
+
}
|
|
347
|
+
expect(caught).toMatchObject({ kind: "upstream_error", retryable: true });
|
|
348
|
+
expect(overloaded.available()).toBe(false);
|
|
349
|
+
expect(classifyAnthropicStatus("a", 400, { error: { message: "prompt is too long: 250000 tokens" } })).toMatchObject({ kind: "context_length", retryable: false });
|
|
350
|
+
expect(classifyAnthropicStatus("a", 401, {})).toMatchObject({ kind: "auth", retryable: false });
|
|
351
|
+
});
|
|
352
|
+
});
|
|
353
|
+
|
|
354
|
+
describe("the composite catalog with named upstreams", () => {
|
|
355
|
+
const twins = [normalizeCatalogModel(orRaw("openai/gpt-4o", 70))!];
|
|
356
|
+
const base: CatalogSnapshot = { models: twins, fetchedAtMs: 1 };
|
|
357
|
+
const openrouter: CatalogSource = { get: async () => base, refresh: async () => base, peek: () => base, find: (s) => twins.find((m) => m.slug === s) };
|
|
358
|
+
const none: CatalogModel[] = [];
|
|
359
|
+
const ollama: OllamaCatalogSource = { get: async () => none, peek: () => none, invalidate: () => {} };
|
|
360
|
+
const always = { available: () => true, cooldownUntilMs: () => null, lastTrip: () => null };
|
|
361
|
+
|
|
362
|
+
test("named models join the snapshot, keep its identity while nothing changes, and vanish while their breaker is open", async () => {
|
|
363
|
+
const cfg = cfgWith([entry({ id: "vllm", kind: "openai", models: [{ id: "llama", input: 0.1, output: 0.4 }] })]);
|
|
364
|
+
const staticSrc = createStaticCatalogSource(cfg);
|
|
365
|
+
let serving = true;
|
|
366
|
+
const cat = createCompositeCatalog(openrouter, ollama, always, { costBias: 1, biasUntilUsage: 1, usage: NO_USAGE, named: { models: (b) => staticSrc.get(b), serving: () => serving } });
|
|
367
|
+
const snap = await cat.get();
|
|
368
|
+
expect(snap.models.map((m: CatalogModel) => m.slug)).toEqual(["openai/gpt-4o", "vllm/llama"]);
|
|
369
|
+
expect(await cat.get()).toBe(snap);
|
|
370
|
+
expect(cat.find("vllm/llama")?.provider).toBe("vllm");
|
|
371
|
+
serving = false;
|
|
372
|
+
const hidden = await cat.get();
|
|
373
|
+
expect(hidden).not.toBe(snap);
|
|
374
|
+
expect(hidden.models.map((m: CatalogModel) => m.slug)).toEqual(["openai/gpt-4o"]);
|
|
375
|
+
serving = true;
|
|
376
|
+
expect((await cat.get()).models).toHaveLength(2);
|
|
377
|
+
});
|
|
378
|
+
});
|