@latimer-woods-tech/llm 0.5.0 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -1,12 +1,143 @@
1
1
  // src/index.ts
2
2
  import {
3
- InternalError,
3
+ InternalError as InternalError3,
4
4
  RateLimitError,
5
- ValidationError,
5
+ ValidationError as ValidationError2,
6
6
  toErrorResponse
7
7
  } from "@latimer-woods-tech/errors";
8
8
 
9
+ // src/gcp-token.ts
10
+ import { InternalError, ValidationError } from "@latimer-woods-tech/errors";
11
+ var EXPIRY_SKEW_MS = 3e5;
12
+ var tokenCache = /* @__PURE__ */ new Map();
13
+ function parseServiceAccountKey(gcpSaKey) {
14
+ const text = gcpSaKey.trimStart().startsWith("{") ? gcpSaKey : atob(gcpSaKey);
15
+ const key = JSON.parse(text);
16
+ if (!key.client_email || !key.private_key || !key.token_uri) {
17
+ throw new ValidationError("GCP_SA_KEY is missing client_email, private_key, or token_uri");
18
+ }
19
+ return key;
20
+ }
21
+ function base64UrlEncode(bytes) {
22
+ let binary = "";
23
+ for (let i = 0; i < bytes.length; i++) binary += String.fromCharCode(bytes[i]);
24
+ return btoa(binary).replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/g, "");
25
+ }
26
+ async function importPrivateKey(pem) {
27
+ const body = pem.split("\n").filter((line) => !line.startsWith("-----")).join("");
28
+ const binary = atob(body);
29
+ const bytes = new Uint8Array(binary.length);
30
+ for (let i = 0; i < binary.length; i++) bytes[i] = binary.charCodeAt(i);
31
+ return crypto.subtle.importKey(
32
+ "pkcs8",
33
+ bytes.buffer,
34
+ { name: "RSASSA-PKCS1-v1_5", hash: "SHA-256" },
35
+ false,
36
+ ["sign"]
37
+ );
38
+ }
39
+ async function createAssertion(key, nowSeconds) {
40
+ const encoder = new TextEncoder();
41
+ const header = base64UrlEncode(encoder.encode(JSON.stringify({ alg: "RS256", typ: "JWT" })));
42
+ const payload = base64UrlEncode(
43
+ encoder.encode(
44
+ JSON.stringify({
45
+ iss: key.client_email,
46
+ scope: "https://www.googleapis.com/auth/cloud-platform",
47
+ aud: key.token_uri,
48
+ exp: nowSeconds + 3600,
49
+ iat: nowSeconds
50
+ })
51
+ )
52
+ );
53
+ const signingInput = `${header}.${payload}`;
54
+ const cryptoKey = await importPrivateKey(key.private_key);
55
+ const signature = await crypto.subtle.sign(
56
+ "RSASSA-PKCS1-v1_5",
57
+ cryptoKey,
58
+ encoder.encode(signingInput).buffer
59
+ );
60
+ return `${signingInput}.${base64UrlEncode(new Uint8Array(signature))}`;
61
+ }
62
+ async function mintGcpAccessToken(gcpSaKey, fetchImpl) {
63
+ const key = parseServiceAccountKey(gcpSaKey);
64
+ const now = Date.now();
65
+ const cached = tokenCache.get(key.client_email);
66
+ if (cached && cached.expiresAt - EXPIRY_SKEW_MS > now) return cached.token;
67
+ const assertion = await createAssertion(key, Math.floor(now / 1e3));
68
+ const response = await fetchImpl(key.token_uri, {
69
+ method: "POST",
70
+ headers: { "content-type": "application/x-www-form-urlencoded" },
71
+ body: new URLSearchParams({
72
+ grant_type: "urn:ietf:params:oauth:grant-type:jwt-bearer",
73
+ assertion
74
+ }).toString()
75
+ }).catch((cause) => {
76
+ throw new InternalError(`GCP token exchange failed: ${String(cause)}`);
77
+ });
78
+ if (!response.ok) {
79
+ throw new InternalError(`GCP token exchange returned ${response.status}`);
80
+ }
81
+ const data = await response.json();
82
+ if (!data.access_token) {
83
+ throw new InternalError("GCP token exchange returned no access_token");
84
+ }
85
+ const ttlMs = (data.expires_in > 0 ? data.expires_in : 3600) * 1e3;
86
+ tokenCache.set(key.client_email, { token: data.access_token, expiresAt: now + ttlMs });
87
+ return data.access_token;
88
+ }
89
+ var METADATA_BASE = "http://metadata.google.internal/computeMetadata/v1";
90
+ var METADATA_HEADER = { "Metadata-Flavor": "Google" };
91
+ var METADATA_TIMEOUT_MS = 2e3;
92
+ var adcCache;
93
+ async function fetchAdcAccessToken(fetchImpl) {
94
+ const now = Date.now();
95
+ if (adcCache && adcCache.expiresAt - EXPIRY_SKEW_MS > now) {
96
+ return { token: adcCache.token, project: adcCache.project };
97
+ }
98
+ const controller = new AbortController();
99
+ const timer = setTimeout(() => controller.abort(), METADATA_TIMEOUT_MS);
100
+ try {
101
+ const tokenRes = await fetchImpl(
102
+ `${METADATA_BASE}/instance/service-accounts/default/token`,
103
+ { headers: METADATA_HEADER, signal: controller.signal }
104
+ );
105
+ if (!tokenRes.ok) {
106
+ throw new InternalError(`GCP metadata token endpoint returned ${tokenRes.status}`);
107
+ }
108
+ const data = await tokenRes.json();
109
+ if (!data.access_token) {
110
+ throw new InternalError("GCP metadata token endpoint returned no access_token");
111
+ }
112
+ let project;
113
+ try {
114
+ const projRes = await fetchImpl(`${METADATA_BASE}/project/project-id`, {
115
+ headers: METADATA_HEADER,
116
+ signal: controller.signal
117
+ });
118
+ if (projRes.ok) project = (await projRes.text()).trim() || void 0;
119
+ } catch {
120
+ }
121
+ const ttlMs = (data.expires_in > 0 ? data.expires_in : 3600) * 1e3;
122
+ adcCache = { token: data.access_token, project, expiresAt: now + ttlMs };
123
+ return { token: data.access_token, project };
124
+ } catch (cause) {
125
+ if (cause instanceof InternalError) throw cause;
126
+ throw new InternalError(`GCP metadata credential unavailable: ${String(cause)}`);
127
+ } finally {
128
+ clearTimeout(timer);
129
+ }
130
+ }
131
+ function clearGcpTokenCache() {
132
+ tokenCache.clear();
133
+ adcCache = void 0;
134
+ }
135
+ function serviceAccountProjectId(gcpSaKey) {
136
+ return parseServiceAccountKey(gcpSaKey).project_id;
137
+ }
138
+
9
139
  // src/embed.ts
140
+ import { InternalError as InternalError2 } from "@latimer-woods-tech/errors";
10
141
  var DEFAULT_EMBEDDING_MODEL = "@cf/baai/bge-base-en-v1.5";
11
142
  async function embed(ai, input, opts) {
12
143
  const model = opts?.model ?? DEFAULT_EMBEDDING_MODEL;
@@ -22,6 +153,41 @@ async function embed(ai, input, opts) {
22
153
  dims: vectors[0].length
23
154
  };
24
155
  }
156
+ var LOCAL_EMBEDDING_MODEL = "nomic-embed-text";
157
+ async function embedLocal(env, input) {
158
+ const texts = Array.isArray(input) ? input : [input];
159
+ const res = await fetch(
160
+ `${env.AI_GATEWAY_BASE_URL}/custom-local-gpu/v1/embeddings`,
161
+ {
162
+ method: "POST",
163
+ headers: {
164
+ "Content-Type": "application/json",
165
+ // CF Access service-token headers only sent when present (bearer-only rails still work).
166
+ ...env.GPU_LLM_API_TOKEN ? { Authorization: `Bearer ${env.GPU_LLM_API_TOKEN}` } : {},
167
+ ...env.GPU_LLM_ACCESS_CLIENT_ID ? { "CF-Access-Client-Id": env.GPU_LLM_ACCESS_CLIENT_ID } : {},
168
+ ...env.GPU_LLM_ACCESS_CLIENT_SECRET ? { "CF-Access-Client-Secret": env.GPU_LLM_ACCESS_CLIENT_SECRET } : {}
169
+ },
170
+ body: JSON.stringify({ model: LOCAL_EMBEDDING_MODEL, input: texts })
171
+ }
172
+ );
173
+ if (!res.ok) {
174
+ throw new InternalError2(
175
+ `embedLocal(): rail ${res.status}: ${(await res.text()).slice(0, 160)}`
176
+ );
177
+ }
178
+ const json = await res.json();
179
+ const vectors = (json.data ?? []).map((d) => d.embedding);
180
+ if (vectors.length === 0) {
181
+ throw new InternalError2(
182
+ `embedLocal(): rail returned no vectors for model ${LOCAL_EMBEDDING_MODEL}`
183
+ );
184
+ }
185
+ return {
186
+ vectors,
187
+ model: LOCAL_EMBEDDING_MODEL,
188
+ dims: vectors[0].length
189
+ };
190
+ }
25
191
 
26
192
  // src/index.ts
27
193
  function contentToText(content) {
@@ -35,25 +201,60 @@ function systemText(opts, messages) {
35
201
  }
36
202
  var MODELS = {
37
203
  anthropic: {
38
- fast: "claude-haiku-4-20250514",
39
- balanced: "claude-sonnet-4-6",
40
- smart: "claude-opus-4-7"
204
+ // `claude-haiku-4-20250514` was never a model Anthropic served — it 404s
205
+ // (verified live against /v1/messages). Date suffixes are never appended to
206
+ // an alias; the alias is `claude-haiku-4-5`, which resolves server-side to
207
+ // `claude-haiku-4-5-20251001`. This is the `fast` tier's FALLBACK, so the
208
+ // 404 stayed invisible for as long as the Grok primary held.
209
+ fast: "claude-haiku-4-5",
210
+ // Current-generation Sonnet/Opus (2026-08 refresh). The previous pins —
211
+ // `claude-sonnet-4-6` / `claude-opus-4-7` — were a generation behind, and
212
+ // the opus-4-7 leg was HARD-BROKEN: Claude 4.7+ rejects `temperature`
213
+ // (400 "`temperature` is deprecated for this model", verified live
214
+ // 2026-08-14) and buildAnthropicRequest always sent one, so every `smart`
215
+ // call 400'd on its Anthropic primary and silently served Gemini Flash.
216
+ // Sonnet 5 / Opus 5 cost the same list price or less than the models they
217
+ // replace ($3/$15, $5/$25 per 1M). The sampling/thinking request-shape
218
+ // differences these models introduce are handled in buildAnthropicRequest.
219
+ balanced: "claude-sonnet-5",
220
+ smart: "claude-opus-5"
41
221
  },
42
222
  gemini: {
43
- smart: "gemini-2.5-pro"
223
+ // `gemini-2.5-flash`, not `-pro`: the leg's job here is a fast, reliable,
224
+ // JSON-returning fallback. Gemini 2.5 *Pro* mandates a thinking budget of
225
+ // ≥128 tokens that is drawn from `maxOutputTokens` and CANNOT be disabled
226
+ // (thinkingBudget=0 is rejected). On the render-runner's large judge /
227
+ // generation prompts — and any low-`maxTokens` call (headline uses 40) —
228
+ // the thinking phase exhausts the whole budget and Vertex returns 200 with
229
+ // an empty candidate (finishReason MAX_TOKENS, no text), which the router
230
+ // treats as a failed leg. Flash supports `thinkingBudget: 0` (set in
231
+ // buildGeminiRequest), so text is always emitted. Both are Vertex-served.
232
+ smart: "gemini-2.5-flash"
44
233
  },
45
234
  groq: {
46
- verifier: "llama-4-maverick"
235
+ // `llama-4-maverick` was NOT a model Groq serves — every `verifier` call
236
+ // 404'd (the tier has no fallback, so it was hard-broken), and it was also
237
+ // the `workbench` fallback. Groq's llama-4 offering is `scout`, which is
238
+ // blocked at our org level; `llama-3.3-70b-versatile` is served and
239
+ // unblocked (verified against /v1/models and a live completion).
240
+ verifier: "llama-3.3-70b-versatile"
47
241
  },
48
242
  grok: {
49
243
  fast: "grok-4.3"
50
244
  },
51
245
  deepseek: {
52
246
  workbench: "deepseek-chat"
247
+ },
248
+ local: {
249
+ // Self-hosted qwen3:8b on the dedicated fast GPU, reached through the
250
+ // `custom-local-fast-gpu` AI Gateway provider's native Ollama chat route.
251
+ fast: "qwen3:8b",
252
+ workbench: "qwen3.6:27b"
53
253
  }
54
254
  };
55
255
  var DEFAULT_MAX_TOKENS = 1024;
56
256
  var DEFAULT_TEMPERATURE = 0.7;
257
+ var DEFAULT_VERTEX_LOCATION = "us-central1";
57
258
  var DEFAULT_LONG_CONTEXT_THRESHOLD = 15e4;
58
259
  var BACKOFF_BASE_MS = 500;
59
260
  var BACKOFF_CAP_MS = 8e3;
@@ -88,19 +289,33 @@ async function recordOrgCostUsage(kv, todayKey, monthKey, costUsd, opts) {
88
289
  }
89
290
  }
90
291
  var MODEL_PRICE_PER_1M = {
91
- // Anthropic Haiku 4
92
- "claude-haiku-4-20250514": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
93
- "claude-haiku-4-5-20251001": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
292
+ // Anthropic Haiku 4.5 — `claude-haiku-4-5` is the routed alias; the dated id
293
+ // is what the API echoes back in `response.model`, so both must price.
294
+ // (Rates corrected 2026-08: the previous $0.80/$4.00 rows UNDERSTATED the
295
+ // actual $1/$5 list price.)
296
+ "claude-haiku-4-5": { input: 1, output: 5, cacheRead: 0.1, cacheWrite: 1.25 },
297
+ "claude-haiku-4-5-20251001": { input: 1, output: 5, cacheRead: 0.1, cacheWrite: 1.25 },
94
298
  // Anthropic Sonnet 4
95
299
  "claude-sonnet-4-20250514": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
96
300
  "claude-sonnet-4-6": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
97
- // Anthropic Opus 4
301
+ // Anthropic Sonnet 5 — the routed `balanced` primary. List $3/$15; intro
302
+ // pricing ($2/$10) runs through 2026-08-31 — priced at list here, the
303
+ // conservative upper bound (table convention).
304
+ "claude-sonnet-5": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
305
+ // Anthropic Opus 4 (dated id) — genuinely the $15/$75 era.
98
306
  "claude-opus-4-20250514": { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
99
- "claude-opus-4-7": { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
100
- // Gemini 2.5 Pro
307
+ // Opus 4.7 was NEVER $15/$75 — it launched at $5/$25. The old row copied the
308
+ // Opus 4 rate and overstated every smart-tier ledger entry 3x.
309
+ "claude-opus-4-7": { input: 5, output: 25, cacheRead: 0.5, cacheWrite: 6.25 },
310
+ // Anthropic Opus 5 — the routed `smart` primary. Same $5/$25 as Opus 4.8/4.7.
311
+ "claude-opus-5": { input: 5, output: 25, cacheRead: 0.5, cacheWrite: 6.25 },
312
+ // Gemini 2.5 Flash — the routed `smart`/long-context model (JSON fallback leg).
313
+ "gemini-2.5-flash": { input: 0.3, output: 2.5, cacheRead: 0.075, cacheWrite: 0.3 },
314
+ // Gemini 2.5 Pro — retained for historical ledger rows (was the routed model).
101
315
  "gemini-2.5-pro": { input: 1.25, output: 10, cacheRead: 0.31, cacheWrite: 4.5 },
102
- // Groq Llama 4 Maverick
103
- "llama-4-maverick": { input: 0.5, output: 0.77, cacheRead: 0.05, cacheWrite: 0.5 },
316
+ // Groq Llama 3.3 70B Versatile (`verifier` tier). Groq has no prompt caching,
317
+ // so cache rates are 0.00 — same convention as grok-4.3 below.
318
+ "llama-3.3-70b-versatile": { input: 0.59, output: 0.79, cacheRead: 0, cacheWrite: 0 },
104
319
  // Grok 4.3
105
320
  "grok-4.3": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
106
321
  // DeepSeek API pricing as of 2026-05: cache-write conservatively uses cache-miss input pricing.
@@ -108,7 +323,10 @@ var MODEL_PRICE_PER_1M = {
108
323
  "deepseek-reasoner": { input: 0.55, output: 2.19, cacheRead: 0.14, cacheWrite: 0.55 },
109
324
  // Deprecated aliases retained for historical ledger rows.
110
325
  "grok-4-fast": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
111
- "grok-3-mini-latest": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 }
326
+ "grok-3-mini-latest": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
327
+ // Self-hosted qwen3 on the GPU box — zero marginal cost (electricity aside).
328
+ "qwen3:8b": { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
329
+ "qwen3.6:27b": { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }
112
330
  };
113
331
  var PRICE_FALLBACK = MODEL_PRICE_PER_1M["claude-opus-4-7"];
114
332
  function estimateCostUsd(tokens, model) {
@@ -150,15 +368,22 @@ function computeBackoffMs(attempt) {
150
368
  const jitter = Math.floor(Math.random() * BACKOFF_JITTER_MAX_MS);
151
369
  return Math.min(BACKOFF_BASE_MS * Math.pow(2, attempt) + jitter, BACKOFF_CAP_MS);
152
370
  }
371
+ var ANTHROPIC_SAMPLING_REMOVED = /^claude-(opus-4-[78]|opus-5|sonnet-5|fable-5|mythos-5)/;
372
+ var ANTHROPIC_THINKING_DEFAULT_ON = /^claude-(opus-5|sonnet-5)$/;
153
373
  function buildAnthropicRequest(model, messages, opts, env, streaming = false) {
154
374
  const sys = systemText(opts, messages);
155
375
  const filtered = messages.filter((m) => m.role !== "system");
156
376
  const body = {
157
377
  model,
158
378
  max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
159
- temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
160
379
  messages: filtered.map((m) => ({ role: m.role, content: m.content }))
161
380
  };
381
+ if (!ANTHROPIC_SAMPLING_REMOVED.test(model)) {
382
+ body.temperature = opts.temperature ?? DEFAULT_TEMPERATURE;
383
+ }
384
+ if (ANTHROPIC_THINKING_DEFAULT_ON.test(model)) {
385
+ body.thinking = { type: "disabled" };
386
+ }
162
387
  if (streaming) {
163
388
  body.stream = true;
164
389
  }
@@ -186,7 +411,18 @@ function buildAnthropicRequest(model, messages, opts, env, streaming = false) {
186
411
  body: JSON.stringify(body)
187
412
  };
188
413
  }
189
- function buildGeminiRequest(model, messages, opts, env) {
414
+ async function resolveVertexAuth(env, fetchImpl) {
415
+ if (env.GCP_SA_KEY) {
416
+ try {
417
+ return { token: await mintGcpAccessToken(env.GCP_SA_KEY, fetchImpl) };
418
+ } catch (error) {
419
+ if (!env.VERTEX_ACCESS_TOKEN) throw error;
420
+ }
421
+ }
422
+ if (env.VERTEX_ACCESS_TOKEN) return { token: env.VERTEX_ACCESS_TOKEN };
423
+ return fetchAdcAccessToken(fetchImpl);
424
+ }
425
+ function buildGeminiRequest(model, messages, opts, env, accessToken, adcProject) {
190
426
  const sys = systemText(opts, messages);
191
427
  const contents = messages.filter((m) => m.role !== "system").map((m) => ({
192
428
  role: m.role === "assistant" ? "model" : "user",
@@ -196,18 +432,26 @@ function buildGeminiRequest(model, messages, opts, env) {
196
432
  contents,
197
433
  generationConfig: {
198
434
  maxOutputTokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
199
- temperature: opts.temperature ?? DEFAULT_TEMPERATURE
435
+ temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
436
+ // Disable "thinking": Gemini 2.5 draws thinking tokens from
437
+ // maxOutputTokens, so on a large prompt (or a small token budget) the
438
+ // model can spend the entire budget thinking and return an empty
439
+ // candidate (finishReason MAX_TOKENS). This leg wants deterministic text
440
+ // out, so thinking is turned off (supported by gemini-2.5-flash).
441
+ thinkingConfig: { thinkingBudget: 0 }
200
442
  }
201
443
  };
202
444
  if (sys) {
203
445
  body.systemInstruction = { parts: [{ text: sys }] };
204
446
  }
205
- const path = `v1/projects/${env.VERTEX_PROJECT}/locations/${env.VERTEX_LOCATION}/publishers/google/models/${model}:generateContent`;
447
+ const project = env.VERTEX_PROJECT || (env.GCP_SA_KEY ? serviceAccountProjectId(env.GCP_SA_KEY) : adcProject ?? "");
448
+ const location = env.VERTEX_LOCATION || DEFAULT_VERTEX_LOCATION;
449
+ const path = `v1/projects/${project}/locations/${location}/publishers/google/models/${model}:generateContent`;
206
450
  return {
207
451
  url: `${env.AI_GATEWAY_BASE_URL}/google-vertex-ai/${path}`,
208
452
  headers: {
209
453
  "content-type": "application/json",
210
- authorization: `Bearer ${env.VERTEX_ACCESS_TOKEN}`
454
+ authorization: `Bearer ${accessToken}`
211
455
  },
212
456
  body: JSON.stringify(body)
213
457
  };
@@ -241,6 +485,53 @@ function toOpenAiMessages(messages, sys) {
241
485
  }
242
486
  return out;
243
487
  }
488
+ function toOllamaMessages(messages, sys) {
489
+ const toolNames = /* @__PURE__ */ new Map();
490
+ for (const message of messages) {
491
+ if (typeof message.content === "string") continue;
492
+ for (const block of message.content) {
493
+ if (block.type === "tool_use") toolNames.set(block.id, block.name);
494
+ }
495
+ }
496
+ const out = [];
497
+ if (sys) out.push({ role: "system", content: sys });
498
+ for (const message of messages) {
499
+ if (message.role === "system") continue;
500
+ if (typeof message.content === "string") {
501
+ out.push({ role: message.role, content: message.content });
502
+ continue;
503
+ }
504
+ let text = "";
505
+ const toolCalls = [];
506
+ const results = [];
507
+ for (const block of message.content) {
508
+ if (block.type === "text") text += block.text;
509
+ else if (block.type === "tool_use") {
510
+ toolCalls.push({
511
+ type: "function",
512
+ function: { index: toolCalls.length, name: block.name, arguments: block.input }
513
+ });
514
+ } else if (block.type === "tool_result") {
515
+ results.push({ toolUseId: block.tool_use_id, content: block.content });
516
+ }
517
+ }
518
+ if (results.length > 0) {
519
+ for (const result of results) {
520
+ out.push({
521
+ role: "tool",
522
+ tool_name: toolNames.get(result.toolUseId) ?? result.toolUseId,
523
+ content: result.content
524
+ });
525
+ }
526
+ if (text) out.push({ role: "user", content: text });
527
+ } else if (toolCalls.length > 0) {
528
+ out.push({ role: "assistant", content: text, tool_calls: toolCalls });
529
+ } else {
530
+ out.push({ role: message.role, content: text });
531
+ }
532
+ }
533
+ return out;
534
+ }
244
535
  function openAiTools(opts) {
245
536
  if (!opts.tools || opts.tools.length === 0) return void 0;
246
537
  return opts.tools.map((t) => ({
@@ -274,7 +565,7 @@ function buildGroqRequest(model, messages, opts, env) {
274
565
  }
275
566
  function buildGrokRequest(model, messages, opts, env) {
276
567
  if (!env.GROK_API_KEY) {
277
- throw new ValidationError("GROK_API_KEY required for grok-* model override");
568
+ throw new ValidationError2("GROK_API_KEY required for grok-* model override");
278
569
  }
279
570
  const sys = systemText(opts, messages);
280
571
  const body = {
@@ -303,7 +594,7 @@ function buildGrokRequest(model, messages, opts, env) {
303
594
  }
304
595
  function buildDeepSeekRequest(model, messages, opts, env) {
305
596
  if (!env.DEEPSEEK_API_KEY) {
306
- throw new ValidationError("DEEPSEEK_API_KEY required for workbench tier or deepseek-* model override");
597
+ throw new ValidationError2("DEEPSEEK_API_KEY required for workbench tier or deepseek-* model override");
307
598
  }
308
599
  const sys = systemText(opts, messages);
309
600
  const body = {
@@ -327,6 +618,51 @@ function buildDeepSeekRequest(model, messages, opts, env) {
327
618
  body: JSON.stringify(body)
328
619
  };
329
620
  }
621
+ function buildLocalRequest(model, messages, opts, env) {
622
+ if (!env.GPU_LLM_API_TOKEN) {
623
+ throw new ValidationError2("GPU_LLM_API_TOKEN required for the local provider");
624
+ }
625
+ const sys = systemText(opts, messages);
626
+ const fastMode = model === MODELS.local.fast;
627
+ const headers = {
628
+ "content-type": "application/json",
629
+ authorization: `Bearer ${env.GPU_LLM_API_TOKEN}`
630
+ };
631
+ if (env.GPU_LLM_ACCESS_CLIENT_ID) headers["CF-Access-Client-Id"] = env.GPU_LLM_ACCESS_CLIENT_ID;
632
+ if (env.GPU_LLM_ACCESS_CLIENT_SECRET) headers["CF-Access-Client-Secret"] = env.GPU_LLM_ACCESS_CLIENT_SECRET;
633
+ const tools = openAiTools(opts);
634
+ if (fastMode) {
635
+ if (typeof opts.toolChoice === "object") {
636
+ throw new ValidationError2("named tool choice is not supported by the native Qwen 8B route");
637
+ }
638
+ return {
639
+ url: `${env.AI_GATEWAY_BASE_URL}/custom-local-fast-gpu/api/chat`,
640
+ headers,
641
+ body: JSON.stringify({
642
+ model,
643
+ stream: false,
644
+ think: false,
645
+ messages: toOllamaMessages(messages, sys),
646
+ options: {
647
+ num_predict: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
648
+ temperature: opts.temperature ?? DEFAULT_TEMPERATURE
649
+ },
650
+ ...tools && opts.toolChoice !== "none" ? { tools } : {}
651
+ })
652
+ };
653
+ }
654
+ return {
655
+ url: `${env.AI_GATEWAY_BASE_URL}/custom-local-gpu/v1/chat/completions`,
656
+ headers,
657
+ body: JSON.stringify({
658
+ model,
659
+ max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
660
+ temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
661
+ messages: toOpenAiMessages(messages, sys),
662
+ ...tools ? { tools, tool_choice: openAiToolChoice(opts.toolChoice ?? "auto") } : {}
663
+ })
664
+ };
665
+ }
330
666
  function normalizeAnthropicStop(reason) {
331
667
  switch (reason) {
332
668
  case "end_turn":
@@ -411,6 +747,29 @@ function parseOpenAi(json) {
411
747
  stopReason: normalizeOpenAiStop(choice?.finish_reason)
412
748
  };
413
749
  }
750
+ function parseOllamaChat(json) {
751
+ const response = json;
752
+ const toolCalls = (response.message?.tool_calls ?? []).filter((call) => typeof call.function?.name === "string").map((call, index) => ({
753
+ id: `call_${index}`,
754
+ name: call.function.name,
755
+ arguments: typeof call.function?.arguments === "string" ? parseToolArgs(call.function.arguments) : call.function?.arguments ?? {}
756
+ }));
757
+ let stopReason;
758
+ if (toolCalls.length > 0) stopReason = "tool_use";
759
+ else if (response.done_reason === "length") stopReason = "max_tokens";
760
+ else if (response.done) stopReason = "end";
761
+ else if (response.done_reason) stopReason = "other";
762
+ return {
763
+ // Deliberately exclude message.thinking: callers receive only the bounded
764
+ // classification answer even if a non-conforming origin returns a trace.
765
+ content: response.message?.content ?? "",
766
+ input: response.prompt_eval_count ?? 0,
767
+ output: response.eval_count ?? 0,
768
+ model: response.model,
769
+ toolCalls: toolCalls.length > 0 ? toolCalls : void 0,
770
+ stopReason
771
+ };
772
+ }
414
773
  async function callWithBackoff(provider, request, fetchImpl, signal, logger, nowFn) {
415
774
  function exhaustAndThrow(err) {
416
775
  markProviderCoolingDown(provider, nowFn ?? Date.now);
@@ -474,7 +833,7 @@ async function callWithBackoff(provider, request, fetchImpl, signal, logger, now
474
833
  function isProviderError(err) {
475
834
  return typeof err === "object" && err !== null && typeof err.status === "number" && typeof err.message === "string" && typeof err.provider === "string";
476
835
  }
477
- var TOOL_CAPABLE_PROVIDERS = /* @__PURE__ */ new Set(["anthropic", "grok", "deepseek"]);
836
+ var TOOL_CAPABLE_PROVIDERS = /* @__PURE__ */ new Set(["anthropic", "grok", "deepseek", "local"]);
478
837
  function plan(tier, opts, tokenEstimate) {
479
838
  if (opts.model) {
480
839
  const m = opts.model;
@@ -482,6 +841,7 @@ function plan(tier, opts, tokenEstimate) {
482
841
  if (m.startsWith("gemini")) return { primary: { provider: "gemini", model: m } };
483
842
  if (m.startsWith("grok")) return { primary: { provider: "grok", model: m } };
484
843
  if (m.startsWith("deepseek")) return { primary: { provider: "deepseek", model: m } };
844
+ if (m.startsWith("qwen")) return { primary: { provider: "local", model: m } };
485
845
  return { primary: { provider: "groq", model: m } };
486
846
  }
487
847
  const longContext = tokenEstimate >= (opts.longContextThreshold ?? DEFAULT_LONG_CONTEXT_THRESHOLD);
@@ -531,9 +891,11 @@ async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
531
891
  case "anthropic":
532
892
  req = buildAnthropicRequest(leg.model, messages, opts, env);
533
893
  break;
534
- case "gemini":
535
- req = buildGeminiRequest(leg.model, messages, opts, env);
894
+ case "gemini": {
895
+ const vertexAuth = await resolveVertexAuth(env, fetchImpl);
896
+ req = buildGeminiRequest(leg.model, messages, opts, env, vertexAuth.token, vertexAuth.project);
536
897
  break;
898
+ }
537
899
  case "groq":
538
900
  req = buildGroqRequest(leg.model, messages, opts, env);
539
901
  break;
@@ -543,6 +905,9 @@ async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
543
905
  case "deepseek":
544
906
  req = buildDeepSeekRequest(leg.model, messages, opts, env);
545
907
  break;
908
+ case "local":
909
+ req = buildLocalRequest(leg.model, messages, opts, env);
910
+ break;
546
911
  }
547
912
  const aigMetadata = buildAigMetadata(opts);
548
913
  if (aigMetadata) req.headers["cf-aig-metadata"] = aigMetadata;
@@ -565,14 +930,20 @@ async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
565
930
  return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
566
931
  case "deepseek":
567
932
  return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
933
+ case "local":
934
+ return {
935
+ parsed: leg.model === MODELS.local.fast ? parseOllamaChat(json) : parseOpenAi(json),
936
+ gatewayRequestId,
937
+ attempts
938
+ };
568
939
  }
569
940
  }
570
941
  async function complete(messages, env, opts = {}, deps = {}) {
571
942
  if (messages.length === 0) {
572
- throw new ValidationError("messages must not be empty");
943
+ throw new ValidationError2("messages must not be empty");
573
944
  }
574
945
  if (!env.AI_GATEWAY_BASE_URL) {
575
- throw new ValidationError("AI_GATEWAY_BASE_URL is required in 0.3.0");
946
+ throw new ValidationError2("AI_GATEWAY_BASE_URL is required in 0.3.0");
576
947
  }
577
948
  const fetchImpl = deps.fetch ?? fetch;
578
949
  const now = deps.now ?? (() => Date.now());
@@ -581,7 +952,13 @@ async function complete(messages, env, opts = {}, deps = {}) {
581
952
  const tier = opts.tier ?? "balanced";
582
953
  const system = systemText(opts, messages);
583
954
  const tokenEstimate = estimateTokens(messages, system);
584
- const route = plan(tier, opts, tokenEstimate);
955
+ let route = plan(tier, opts, tokenEstimate);
956
+ if (env.LLM_LOCAL_FIRST && tier === "fast" && !opts.model && env.GPU_LLM_API_TOKEN) {
957
+ route = { primary: { provider: "local", model: MODELS.local.fast }, fallback: route.primary };
958
+ }
959
+ if (env.LLM_LOCAL_WORKBENCH && tier === "workbench" && !opts.model && env.GPU_LLM_API_TOKEN) {
960
+ route = { primary: { provider: "local", model: MODELS.local.workbench }, fallback: route.primary };
961
+ }
585
962
  const kv = env.LLM_COST_KV;
586
963
  const todayKey = `llm:daily-cost:${isoDate(now())}`;
587
964
  const monthKey = `llm:monthly-cost:${isoMonth(now())}`;
@@ -616,7 +993,7 @@ async function complete(messages, env, opts = {}, deps = {}) {
616
993
  if (opts.tools && opts.tools.length > 0) {
617
994
  routeLegs = routeLegs.filter((l) => TOOL_CAPABLE_PROVIDERS.has(l.provider));
618
995
  if (routeLegs.length === 0) {
619
- throw new ValidationError(
996
+ throw new ValidationError2(
620
997
  `tool-calling requires a tool-capable provider (${[...TOOL_CAPABLE_PROVIDERS].join(", ")}); tier '${tier}' has none \u2014 use tier fast/balanced/smart or a claude-* model override`
621
998
  );
622
999
  }
@@ -629,7 +1006,7 @@ async function complete(messages, env, opts = {}, deps = {}) {
629
1006
  }
630
1007
  if (opts.signal?.aborted) {
631
1008
  return toErrorResponse(
632
- new InternalError("llm call aborted", { provider: leg.provider, model: leg.model })
1009
+ new InternalError3("llm call aborted", { provider: leg.provider, model: leg.model })
633
1010
  );
634
1011
  }
635
1012
  try {
@@ -704,7 +1081,7 @@ async function complete(messages, env, opts = {}, deps = {}) {
704
1081
  } catch (e) {
705
1082
  if (e instanceof DOMException && e.name === "AbortError") {
706
1083
  return toErrorResponse(
707
- new InternalError("llm call aborted", { provider: leg.provider, model: leg.model })
1084
+ new InternalError3("llm call aborted", { provider: leg.provider, model: leg.model })
708
1085
  );
709
1086
  }
710
1087
  if (isProviderError(e)) {
@@ -721,15 +1098,15 @@ async function complete(messages, env, opts = {}, deps = {}) {
721
1098
  }
722
1099
  }
723
1100
  return toErrorResponse(
724
- new InternalError("LLM_ALL_PROVIDERS_FAILED", { attempts: attemptLog, tier, tokenEstimate })
1101
+ new InternalError3("LLM_ALL_PROVIDERS_FAILED", { attempts: attemptLog, tier, tokenEstimate })
725
1102
  );
726
1103
  }
727
1104
  async function* completionStream(messages, env, opts = {}) {
728
1105
  if (messages.length === 0) {
729
- throw new ValidationError("messages must not be empty");
1106
+ throw new ValidationError2("messages must not be empty");
730
1107
  }
731
1108
  if (!env.AI_GATEWAY_BASE_URL) {
732
- throw new ValidationError("AI_GATEWAY_BASE_URL is required in 0.3.0");
1109
+ throw new ValidationError2("AI_GATEWAY_BASE_URL is required in 0.3.0");
733
1110
  }
734
1111
  const deps = opts.deps ?? {};
735
1112
  const fetchImpl = deps.fetch ?? fetch;
@@ -744,7 +1121,7 @@ async function* completionStream(messages, env, opts = {}) {
744
1121
  if (streamLeg.provider !== "anthropic") {
745
1122
  const result = await complete(messages, env, opts, deps);
746
1123
  if (result.error !== null || result.data === null) {
747
- throw new InternalError("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
1124
+ throw new InternalError3("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
748
1125
  }
749
1126
  yield result.data.content;
750
1127
  return result.data;
@@ -753,7 +1130,7 @@ async function* completionStream(messages, env, opts = {}) {
753
1130
  logger?.warn?.("llm.provider.coolingDown", { provider: streamLeg.provider });
754
1131
  const result = await complete(messages, env, opts, deps);
755
1132
  if (result.error !== null || result.data === null) {
756
- throw new InternalError("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
1133
+ throw new InternalError3("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
757
1134
  }
758
1135
  yield result.data.content;
759
1136
  return result.data;
@@ -773,12 +1150,12 @@ async function* completionStream(messages, env, opts = {}) {
773
1150
  });
774
1151
  } catch (e) {
775
1152
  if (e instanceof DOMException && e.name === "AbortError") {
776
- throw new InternalError("llm call aborted", {
1153
+ throw new InternalError3("llm call aborted", {
777
1154
  provider: streamLeg.provider,
778
1155
  model: streamLeg.model
779
1156
  });
780
1157
  }
781
- throw new InternalError("llm stream fetch failed", {
1158
+ throw new InternalError3("llm stream fetch failed", {
782
1159
  message: e instanceof Error ? e.message : String(e)
783
1160
  });
784
1161
  }
@@ -790,7 +1167,7 @@ async function* completionStream(messages, env, opts = {}) {
790
1167
  }
791
1168
  const result = await complete(messages, env, opts, deps);
792
1169
  if (result.error !== null || result.data === null) {
793
- throw new InternalError("LLM_ALL_PROVIDERS_FAILED", {
1170
+ throw new InternalError3("LLM_ALL_PROVIDERS_FAILED", {
794
1171
  streamError: `${streamLeg.provider} ${String(response.status)}: ${text.slice(0, 300)}`,
795
1172
  error: result.error
796
1173
  });
@@ -799,7 +1176,7 @@ async function* completionStream(messages, env, opts = {}) {
799
1176
  return result.data;
800
1177
  }
801
1178
  if (!response.body) {
802
- throw new InternalError("llm stream response body is null", {
1179
+ throw new InternalError3("llm stream response body is null", {
803
1180
  provider: streamLeg.provider
804
1181
  });
805
1182
  }
@@ -917,15 +1294,20 @@ function assertGrounding(response, sources) {
917
1294
  export {
918
1295
  BASE_BACKOFF_MS,
919
1296
  DEFAULT_EMBEDDING_MODEL,
1297
+ LOCAL_EMBEDDING_MODEL,
920
1298
  MODELS,
921
1299
  MODEL_PRICE_PER_1M,
922
1300
  PROVIDER_COOLDOWN_MS,
923
1301
  assertGrounding,
1302
+ clearGcpTokenCache,
924
1303
  clearProviderCooldown,
925
1304
  complete,
926
1305
  completionStream,
927
1306
  embed,
1307
+ embedLocal,
928
1308
  isProviderCoolingDown,
929
- markProviderCoolingDown
1309
+ markProviderCoolingDown,
1310
+ mintGcpAccessToken,
1311
+ serviceAccountProjectId
930
1312
  };
931
1313
  //# sourceMappingURL=index.mjs.map