@latimer-woods-tech/llm 0.4.4 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -1,10 +1,195 @@
1
1
  // src/index.ts
2
2
  import {
3
- InternalError,
3
+ InternalError as InternalError3,
4
4
  RateLimitError,
5
- ValidationError,
5
+ ValidationError as ValidationError2,
6
6
  toErrorResponse
7
7
  } from "@latimer-woods-tech/errors";
8
+
9
+ // src/gcp-token.ts
10
+ import { InternalError, ValidationError } from "@latimer-woods-tech/errors";
11
+ var EXPIRY_SKEW_MS = 3e5;
12
+ var tokenCache = /* @__PURE__ */ new Map();
13
+ function parseServiceAccountKey(gcpSaKey) {
14
+ const text = gcpSaKey.trimStart().startsWith("{") ? gcpSaKey : atob(gcpSaKey);
15
+ const key = JSON.parse(text);
16
+ if (!key.client_email || !key.private_key || !key.token_uri) {
17
+ throw new ValidationError("GCP_SA_KEY is missing client_email, private_key, or token_uri");
18
+ }
19
+ return key;
20
+ }
21
+ function base64UrlEncode(bytes) {
22
+ let binary = "";
23
+ for (let i = 0; i < bytes.length; i++) binary += String.fromCharCode(bytes[i]);
24
+ return btoa(binary).replace(/\+/g, "-").replace(/\//g, "_").replace(/=+$/g, "");
25
+ }
26
+ async function importPrivateKey(pem) {
27
+ const body = pem.split("\n").filter((line) => !line.startsWith("-----")).join("");
28
+ const binary = atob(body);
29
+ const bytes = new Uint8Array(binary.length);
30
+ for (let i = 0; i < binary.length; i++) bytes[i] = binary.charCodeAt(i);
31
+ return crypto.subtle.importKey(
32
+ "pkcs8",
33
+ bytes.buffer,
34
+ { name: "RSASSA-PKCS1-v1_5", hash: "SHA-256" },
35
+ false,
36
+ ["sign"]
37
+ );
38
+ }
39
+ async function createAssertion(key, nowSeconds) {
40
+ const encoder = new TextEncoder();
41
+ const header = base64UrlEncode(encoder.encode(JSON.stringify({ alg: "RS256", typ: "JWT" })));
42
+ const payload = base64UrlEncode(
43
+ encoder.encode(
44
+ JSON.stringify({
45
+ iss: key.client_email,
46
+ scope: "https://www.googleapis.com/auth/cloud-platform",
47
+ aud: key.token_uri,
48
+ exp: nowSeconds + 3600,
49
+ iat: nowSeconds
50
+ })
51
+ )
52
+ );
53
+ const signingInput = `${header}.${payload}`;
54
+ const cryptoKey = await importPrivateKey(key.private_key);
55
+ const signature = await crypto.subtle.sign(
56
+ "RSASSA-PKCS1-v1_5",
57
+ cryptoKey,
58
+ encoder.encode(signingInput).buffer
59
+ );
60
+ return `${signingInput}.${base64UrlEncode(new Uint8Array(signature))}`;
61
+ }
62
+ async function mintGcpAccessToken(gcpSaKey, fetchImpl) {
63
+ const key = parseServiceAccountKey(gcpSaKey);
64
+ const now = Date.now();
65
+ const cached = tokenCache.get(key.client_email);
66
+ if (cached && cached.expiresAt - EXPIRY_SKEW_MS > now) return cached.token;
67
+ const assertion = await createAssertion(key, Math.floor(now / 1e3));
68
+ const response = await fetchImpl(key.token_uri, {
69
+ method: "POST",
70
+ headers: { "content-type": "application/x-www-form-urlencoded" },
71
+ body: new URLSearchParams({
72
+ grant_type: "urn:ietf:params:oauth:grant-type:jwt-bearer",
73
+ assertion
74
+ }).toString()
75
+ }).catch((cause) => {
76
+ throw new InternalError(`GCP token exchange failed: ${String(cause)}`);
77
+ });
78
+ if (!response.ok) {
79
+ throw new InternalError(`GCP token exchange returned ${response.status}`);
80
+ }
81
+ const data = await response.json();
82
+ if (!data.access_token) {
83
+ throw new InternalError("GCP token exchange returned no access_token");
84
+ }
85
+ const ttlMs = (data.expires_in > 0 ? data.expires_in : 3600) * 1e3;
86
+ tokenCache.set(key.client_email, { token: data.access_token, expiresAt: now + ttlMs });
87
+ return data.access_token;
88
+ }
89
+ var METADATA_BASE = "http://metadata.google.internal/computeMetadata/v1";
90
+ var METADATA_HEADER = { "Metadata-Flavor": "Google" };
91
+ var METADATA_TIMEOUT_MS = 2e3;
92
+ var adcCache;
93
+ async function fetchAdcAccessToken(fetchImpl) {
94
+ const now = Date.now();
95
+ if (adcCache && adcCache.expiresAt - EXPIRY_SKEW_MS > now) {
96
+ return { token: adcCache.token, project: adcCache.project };
97
+ }
98
+ const controller = new AbortController();
99
+ const timer = setTimeout(() => controller.abort(), METADATA_TIMEOUT_MS);
100
+ try {
101
+ const tokenRes = await fetchImpl(
102
+ `${METADATA_BASE}/instance/service-accounts/default/token`,
103
+ { headers: METADATA_HEADER, signal: controller.signal }
104
+ );
105
+ if (!tokenRes.ok) {
106
+ throw new InternalError(`GCP metadata token endpoint returned ${tokenRes.status}`);
107
+ }
108
+ const data = await tokenRes.json();
109
+ if (!data.access_token) {
110
+ throw new InternalError("GCP metadata token endpoint returned no access_token");
111
+ }
112
+ let project;
113
+ try {
114
+ const projRes = await fetchImpl(`${METADATA_BASE}/project/project-id`, {
115
+ headers: METADATA_HEADER,
116
+ signal: controller.signal
117
+ });
118
+ if (projRes.ok) project = (await projRes.text()).trim() || void 0;
119
+ } catch {
120
+ }
121
+ const ttlMs = (data.expires_in > 0 ? data.expires_in : 3600) * 1e3;
122
+ adcCache = { token: data.access_token, project, expiresAt: now + ttlMs };
123
+ return { token: data.access_token, project };
124
+ } catch (cause) {
125
+ if (cause instanceof InternalError) throw cause;
126
+ throw new InternalError(`GCP metadata credential unavailable: ${String(cause)}`);
127
+ } finally {
128
+ clearTimeout(timer);
129
+ }
130
+ }
131
+ function clearGcpTokenCache() {
132
+ tokenCache.clear();
133
+ adcCache = void 0;
134
+ }
135
+ function serviceAccountProjectId(gcpSaKey) {
136
+ return parseServiceAccountKey(gcpSaKey).project_id;
137
+ }
138
+
139
+ // src/embed.ts
140
+ import { InternalError as InternalError2 } from "@latimer-woods-tech/errors";
141
+ var DEFAULT_EMBEDDING_MODEL = "@cf/baai/bge-base-en-v1.5";
142
+ async function embed(ai, input, opts) {
143
+ const model = opts?.model ?? DEFAULT_EMBEDDING_MODEL;
144
+ const texts = Array.isArray(input) ? input : [input];
145
+ const result = await ai.run(model, { text: texts });
146
+ const vectors = result.data;
147
+ if (!vectors || vectors.length === 0) {
148
+ throw new Error(`embed(): Workers AI returned no vectors for model ${model}`);
149
+ }
150
+ return {
151
+ vectors,
152
+ model,
153
+ dims: vectors[0].length
154
+ };
155
+ }
156
+ var LOCAL_EMBEDDING_MODEL = "nomic-embed-text";
157
+ async function embedLocal(env, input) {
158
+ const texts = Array.isArray(input) ? input : [input];
159
+ const res = await fetch(
160
+ `${env.AI_GATEWAY_BASE_URL}/custom-local-gpu/v1/embeddings`,
161
+ {
162
+ method: "POST",
163
+ headers: {
164
+ "Content-Type": "application/json",
165
+ // CF Access service-token headers only sent when present (bearer-only rails still work).
166
+ ...env.GPU_LLM_API_TOKEN ? { Authorization: `Bearer ${env.GPU_LLM_API_TOKEN}` } : {},
167
+ ...env.GPU_LLM_ACCESS_CLIENT_ID ? { "CF-Access-Client-Id": env.GPU_LLM_ACCESS_CLIENT_ID } : {},
168
+ ...env.GPU_LLM_ACCESS_CLIENT_SECRET ? { "CF-Access-Client-Secret": env.GPU_LLM_ACCESS_CLIENT_SECRET } : {}
169
+ },
170
+ body: JSON.stringify({ model: LOCAL_EMBEDDING_MODEL, input: texts })
171
+ }
172
+ );
173
+ if (!res.ok) {
174
+ throw new InternalError2(
175
+ `embedLocal(): rail ${res.status}: ${(await res.text()).slice(0, 160)}`
176
+ );
177
+ }
178
+ const json = await res.json();
179
+ const vectors = (json.data ?? []).map((d) => d.embedding);
180
+ if (vectors.length === 0) {
181
+ throw new InternalError2(
182
+ `embedLocal(): rail returned no vectors for model ${LOCAL_EMBEDDING_MODEL}`
183
+ );
184
+ }
185
+ return {
186
+ vectors,
187
+ model: LOCAL_EMBEDDING_MODEL,
188
+ dims: vectors[0].length
189
+ };
190
+ }
191
+
192
+ // src/index.ts
8
193
  function contentToText(content) {
9
194
  if (typeof content === "string") return content;
10
195
  return content.map((b) => b.type === "text" ? b.text : b.type === "tool_result" ? b.content : "").join("");
@@ -16,25 +201,51 @@ function systemText(opts, messages) {
16
201
  }
17
202
  var MODELS = {
18
203
  anthropic: {
19
- fast: "claude-haiku-4-20250514",
204
+ // `claude-haiku-4-20250514` was never a model Anthropic served — it 404s
205
+ // (verified live against /v1/messages). Date suffixes are never appended to
206
+ // an alias; the alias is `claude-haiku-4-5`, which resolves server-side to
207
+ // `claude-haiku-4-5-20251001`. This is the `fast` tier's FALLBACK, so the
208
+ // 404 stayed invisible for as long as the Grok primary held.
209
+ fast: "claude-haiku-4-5",
20
210
  balanced: "claude-sonnet-4-6",
21
211
  smart: "claude-opus-4-7"
22
212
  },
23
213
  gemini: {
24
- smart: "gemini-2.5-pro"
214
+ // `gemini-2.5-flash`, not `-pro`: the leg's job here is a fast, reliable,
215
+ // JSON-returning fallback. Gemini 2.5 *Pro* mandates a thinking budget of
216
+ // ≥128 tokens that is drawn from `maxOutputTokens` and CANNOT be disabled
217
+ // (thinkingBudget=0 is rejected). On the render-runner's large judge /
218
+ // generation prompts — and any low-`maxTokens` call (headline uses 40) —
219
+ // the thinking phase exhausts the whole budget and Vertex returns 200 with
220
+ // an empty candidate (finishReason MAX_TOKENS, no text), which the router
221
+ // treats as a failed leg. Flash supports `thinkingBudget: 0` (set in
222
+ // buildGeminiRequest), so text is always emitted. Both are Vertex-served.
223
+ smart: "gemini-2.5-flash"
25
224
  },
26
225
  groq: {
27
- verifier: "llama-4-maverick"
226
+ // `llama-4-maverick` was NOT a model Groq serves — every `verifier` call
227
+ // 404'd (the tier has no fallback, so it was hard-broken), and it was also
228
+ // the `workbench` fallback. Groq's llama-4 offering is `scout`, which is
229
+ // blocked at our org level; `llama-3.3-70b-versatile` is served and
230
+ // unblocked (verified against /v1/models and a live completion).
231
+ verifier: "llama-3.3-70b-versatile"
28
232
  },
29
233
  grok: {
30
234
  fast: "grok-4.3"
31
235
  },
32
236
  deepseek: {
33
237
  workbench: "deepseek-chat"
238
+ },
239
+ local: {
240
+ // Self-hosted qwen3:8b on the GPU box, reached via the `custom-local-gpu`
241
+ // AI Gateway provider. The `fast` tier's optional zero-cost primary.
242
+ fast: "qwen3:8b",
243
+ workbench: "qwen3.6:27b"
34
244
  }
35
245
  };
36
246
  var DEFAULT_MAX_TOKENS = 1024;
37
247
  var DEFAULT_TEMPERATURE = 0.7;
248
+ var DEFAULT_VERTEX_LOCATION = "us-central1";
38
249
  var DEFAULT_LONG_CONTEXT_THRESHOLD = 15e4;
39
250
  var BACKOFF_BASE_MS = 500;
40
251
  var BACKOFF_CAP_MS = 8e3;
@@ -69,8 +280,9 @@ async function recordOrgCostUsage(kv, todayKey, monthKey, costUsd, opts) {
69
280
  }
70
281
  }
71
282
  var MODEL_PRICE_PER_1M = {
72
- // Anthropic Haiku 4
73
- "claude-haiku-4-20250514": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
283
+ // Anthropic Haiku 4.5 — `claude-haiku-4-5` is the routed alias; the dated id
284
+ // is what the API echoes back in `response.model`, so both must price.
285
+ "claude-haiku-4-5": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
74
286
  "claude-haiku-4-5-20251001": { input: 0.8, output: 4, cacheRead: 0.08, cacheWrite: 1 },
75
287
  // Anthropic Sonnet 4
76
288
  "claude-sonnet-4-20250514": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 },
@@ -78,10 +290,13 @@ var MODEL_PRICE_PER_1M = {
78
290
  // Anthropic Opus 4
79
291
  "claude-opus-4-20250514": { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
80
292
  "claude-opus-4-7": { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 },
81
- // Gemini 2.5 Pro
293
+ // Gemini 2.5 Flash — the routed `smart`/long-context model (JSON fallback leg).
294
+ "gemini-2.5-flash": { input: 0.3, output: 2.5, cacheRead: 0.075, cacheWrite: 0.3 },
295
+ // Gemini 2.5 Pro — retained for historical ledger rows (was the routed model).
82
296
  "gemini-2.5-pro": { input: 1.25, output: 10, cacheRead: 0.31, cacheWrite: 4.5 },
83
- // Groq Llama 4 Maverick
84
- "llama-4-maverick": { input: 0.5, output: 0.77, cacheRead: 0.05, cacheWrite: 0.5 },
297
+ // Groq Llama 3.3 70B Versatile (`verifier` tier). Groq has no prompt caching,
298
+ // so cache rates are 0.00 — same convention as grok-4.3 below.
299
+ "llama-3.3-70b-versatile": { input: 0.59, output: 0.79, cacheRead: 0, cacheWrite: 0 },
85
300
  // Grok 4.3
86
301
  "grok-4.3": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
87
302
  // DeepSeek API pricing as of 2026-05: cache-write conservatively uses cache-miss input pricing.
@@ -89,7 +304,10 @@ var MODEL_PRICE_PER_1M = {
89
304
  "deepseek-reasoner": { input: 0.55, output: 2.19, cacheRead: 0.14, cacheWrite: 0.55 },
90
305
  // Deprecated aliases retained for historical ledger rows.
91
306
  "grok-4-fast": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
92
- "grok-3-mini-latest": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 }
307
+ "grok-3-mini-latest": { input: 1.25, output: 2.5, cacheRead: 0, cacheWrite: 0 },
308
+ // Self-hosted qwen3 on the GPU box — zero marginal cost (electricity aside).
309
+ "qwen3:8b": { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
310
+ "qwen3.6:27b": { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }
93
311
  };
94
312
  var PRICE_FALLBACK = MODEL_PRICE_PER_1M["claude-opus-4-7"];
95
313
  function estimateCostUsd(tokens, model) {
@@ -167,7 +385,18 @@ function buildAnthropicRequest(model, messages, opts, env, streaming = false) {
167
385
  body: JSON.stringify(body)
168
386
  };
169
387
  }
170
- function buildGeminiRequest(model, messages, opts, env) {
388
+ async function resolveVertexAuth(env, fetchImpl) {
389
+ if (env.GCP_SA_KEY) {
390
+ try {
391
+ return { token: await mintGcpAccessToken(env.GCP_SA_KEY, fetchImpl) };
392
+ } catch (error) {
393
+ if (!env.VERTEX_ACCESS_TOKEN) throw error;
394
+ }
395
+ }
396
+ if (env.VERTEX_ACCESS_TOKEN) return { token: env.VERTEX_ACCESS_TOKEN };
397
+ return fetchAdcAccessToken(fetchImpl);
398
+ }
399
+ function buildGeminiRequest(model, messages, opts, env, accessToken, adcProject) {
171
400
  const sys = systemText(opts, messages);
172
401
  const contents = messages.filter((m) => m.role !== "system").map((m) => ({
173
402
  role: m.role === "assistant" ? "model" : "user",
@@ -177,18 +406,26 @@ function buildGeminiRequest(model, messages, opts, env) {
177
406
  contents,
178
407
  generationConfig: {
179
408
  maxOutputTokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
180
- temperature: opts.temperature ?? DEFAULT_TEMPERATURE
409
+ temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
410
+ // Disable "thinking": Gemini 2.5 draws thinking tokens from
411
+ // maxOutputTokens, so on a large prompt (or a small token budget) the
412
+ // model can spend the entire budget thinking and return an empty
413
+ // candidate (finishReason MAX_TOKENS). This leg wants deterministic text
414
+ // out, so thinking is turned off (supported by gemini-2.5-flash).
415
+ thinkingConfig: { thinkingBudget: 0 }
181
416
  }
182
417
  };
183
418
  if (sys) {
184
419
  body.systemInstruction = { parts: [{ text: sys }] };
185
420
  }
186
- const path = `v1/projects/${env.VERTEX_PROJECT}/locations/${env.VERTEX_LOCATION}/publishers/google/models/${model}:generateContent`;
421
+ const project = env.VERTEX_PROJECT || (env.GCP_SA_KEY ? serviceAccountProjectId(env.GCP_SA_KEY) : adcProject ?? "");
422
+ const location = env.VERTEX_LOCATION || DEFAULT_VERTEX_LOCATION;
423
+ const path = `v1/projects/${project}/locations/${location}/publishers/google/models/${model}:generateContent`;
187
424
  return {
188
425
  url: `${env.AI_GATEWAY_BASE_URL}/google-vertex-ai/${path}`,
189
426
  headers: {
190
427
  "content-type": "application/json",
191
- authorization: `Bearer ${env.VERTEX_ACCESS_TOKEN}`
428
+ authorization: `Bearer ${accessToken}`
192
429
  },
193
430
  body: JSON.stringify(body)
194
431
  };
@@ -255,7 +492,7 @@ function buildGroqRequest(model, messages, opts, env) {
255
492
  }
256
493
  function buildGrokRequest(model, messages, opts, env) {
257
494
  if (!env.GROK_API_KEY) {
258
- throw new ValidationError("GROK_API_KEY required for grok-* model override");
495
+ throw new ValidationError2("GROK_API_KEY required for grok-* model override");
259
496
  }
260
497
  const sys = systemText(opts, messages);
261
498
  const body = {
@@ -284,7 +521,7 @@ function buildGrokRequest(model, messages, opts, env) {
284
521
  }
285
522
  function buildDeepSeekRequest(model, messages, opts, env) {
286
523
  if (!env.DEEPSEEK_API_KEY) {
287
- throw new ValidationError("DEEPSEEK_API_KEY required for workbench tier or deepseek-* model override");
524
+ throw new ValidationError2("DEEPSEEK_API_KEY required for workbench tier or deepseek-* model override");
288
525
  }
289
526
  const sys = systemText(opts, messages);
290
527
  const body = {
@@ -308,6 +545,32 @@ function buildDeepSeekRequest(model, messages, opts, env) {
308
545
  body: JSON.stringify(body)
309
546
  };
310
547
  }
548
+ function buildLocalRequest(model, messages, opts, env) {
549
+ if (!env.GPU_LLM_API_TOKEN) {
550
+ throw new ValidationError2("GPU_LLM_API_TOKEN required for the local provider");
551
+ }
552
+ const sys = systemText(opts, messages);
553
+ const fastMode = model === MODELS.local.fast;
554
+ const augmented = fastMode ? sys ? `${sys} /no_think` : "/no_think" : sys;
555
+ const headers = {
556
+ "content-type": "application/json",
557
+ authorization: `Bearer ${env.GPU_LLM_API_TOKEN}`
558
+ };
559
+ if (env.GPU_LLM_ACCESS_CLIENT_ID) headers["CF-Access-Client-Id"] = env.GPU_LLM_ACCESS_CLIENT_ID;
560
+ if (env.GPU_LLM_ACCESS_CLIENT_SECRET) headers["CF-Access-Client-Secret"] = env.GPU_LLM_ACCESS_CLIENT_SECRET;
561
+ const tools = openAiTools(opts);
562
+ return {
563
+ url: `${env.AI_GATEWAY_BASE_URL}/custom-local-gpu/v1/chat/completions`,
564
+ headers,
565
+ body: JSON.stringify({
566
+ model,
567
+ max_tokens: opts.maxTokens ?? DEFAULT_MAX_TOKENS,
568
+ temperature: opts.temperature ?? DEFAULT_TEMPERATURE,
569
+ messages: toOpenAiMessages(messages, augmented),
570
+ ...tools ? { tools, tool_choice: openAiToolChoice(opts.toolChoice ?? "auto") } : {}
571
+ })
572
+ };
573
+ }
311
574
  function normalizeAnthropicStop(reason) {
312
575
  switch (reason) {
313
576
  case "end_turn":
@@ -455,7 +718,7 @@ async function callWithBackoff(provider, request, fetchImpl, signal, logger, now
455
718
  function isProviderError(err) {
456
719
  return typeof err === "object" && err !== null && typeof err.status === "number" && typeof err.message === "string" && typeof err.provider === "string";
457
720
  }
458
- var TOOL_CAPABLE_PROVIDERS = /* @__PURE__ */ new Set(["anthropic", "grok", "deepseek"]);
721
+ var TOOL_CAPABLE_PROVIDERS = /* @__PURE__ */ new Set(["anthropic", "grok", "deepseek", "local"]);
459
722
  function plan(tier, opts, tokenEstimate) {
460
723
  if (opts.model) {
461
724
  const m = opts.model;
@@ -463,6 +726,7 @@ function plan(tier, opts, tokenEstimate) {
463
726
  if (m.startsWith("gemini")) return { primary: { provider: "gemini", model: m } };
464
727
  if (m.startsWith("grok")) return { primary: { provider: "grok", model: m } };
465
728
  if (m.startsWith("deepseek")) return { primary: { provider: "deepseek", model: m } };
729
+ if (m.startsWith("qwen")) return { primary: { provider: "local", model: m } };
466
730
  return { primary: { provider: "groq", model: m } };
467
731
  }
468
732
  const longContext = tokenEstimate >= (opts.longContextThreshold ?? DEFAULT_LONG_CONTEXT_THRESHOLD);
@@ -512,9 +776,11 @@ async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
512
776
  case "anthropic":
513
777
  req = buildAnthropicRequest(leg.model, messages, opts, env);
514
778
  break;
515
- case "gemini":
516
- req = buildGeminiRequest(leg.model, messages, opts, env);
779
+ case "gemini": {
780
+ const vertexAuth = await resolveVertexAuth(env, fetchImpl);
781
+ req = buildGeminiRequest(leg.model, messages, opts, env, vertexAuth.token, vertexAuth.project);
517
782
  break;
783
+ }
518
784
  case "groq":
519
785
  req = buildGroqRequest(leg.model, messages, opts, env);
520
786
  break;
@@ -524,6 +790,9 @@ async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
524
790
  case "deepseek":
525
791
  req = buildDeepSeekRequest(leg.model, messages, opts, env);
526
792
  break;
793
+ case "local":
794
+ req = buildLocalRequest(leg.model, messages, opts, env);
795
+ break;
527
796
  }
528
797
  const aigMetadata = buildAigMetadata(opts);
529
798
  if (aigMetadata) req.headers["cf-aig-metadata"] = aigMetadata;
@@ -546,14 +815,16 @@ async function callOne(leg, messages, opts, env, fetchImpl, logger, nowFn) {
546
815
  return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
547
816
  case "deepseek":
548
817
  return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
818
+ case "local":
819
+ return { parsed: parseOpenAi(json), gatewayRequestId, attempts };
549
820
  }
550
821
  }
551
822
  async function complete(messages, env, opts = {}, deps = {}) {
552
823
  if (messages.length === 0) {
553
- throw new ValidationError("messages must not be empty");
824
+ throw new ValidationError2("messages must not be empty");
554
825
  }
555
826
  if (!env.AI_GATEWAY_BASE_URL) {
556
- throw new ValidationError("AI_GATEWAY_BASE_URL is required in 0.3.0");
827
+ throw new ValidationError2("AI_GATEWAY_BASE_URL is required in 0.3.0");
557
828
  }
558
829
  const fetchImpl = deps.fetch ?? fetch;
559
830
  const now = deps.now ?? (() => Date.now());
@@ -562,7 +833,13 @@ async function complete(messages, env, opts = {}, deps = {}) {
562
833
  const tier = opts.tier ?? "balanced";
563
834
  const system = systemText(opts, messages);
564
835
  const tokenEstimate = estimateTokens(messages, system);
565
- const route = plan(tier, opts, tokenEstimate);
836
+ let route = plan(tier, opts, tokenEstimate);
837
+ if (env.LLM_LOCAL_FIRST && tier === "fast" && !opts.model && env.GPU_LLM_API_TOKEN) {
838
+ route = { primary: { provider: "local", model: MODELS.local.fast }, fallback: route.primary };
839
+ }
840
+ if (env.LLM_LOCAL_WORKBENCH && tier === "workbench" && !opts.model && env.GPU_LLM_API_TOKEN) {
841
+ route = { primary: { provider: "local", model: MODELS.local.workbench }, fallback: route.primary };
842
+ }
566
843
  const kv = env.LLM_COST_KV;
567
844
  const todayKey = `llm:daily-cost:${isoDate(now())}`;
568
845
  const monthKey = `llm:monthly-cost:${isoMonth(now())}`;
@@ -597,7 +874,7 @@ async function complete(messages, env, opts = {}, deps = {}) {
597
874
  if (opts.tools && opts.tools.length > 0) {
598
875
  routeLegs = routeLegs.filter((l) => TOOL_CAPABLE_PROVIDERS.has(l.provider));
599
876
  if (routeLegs.length === 0) {
600
- throw new ValidationError(
877
+ throw new ValidationError2(
601
878
  `tool-calling requires a tool-capable provider (${[...TOOL_CAPABLE_PROVIDERS].join(", ")}); tier '${tier}' has none \u2014 use tier fast/balanced/smart or a claude-* model override`
602
879
  );
603
880
  }
@@ -610,7 +887,7 @@ async function complete(messages, env, opts = {}, deps = {}) {
610
887
  }
611
888
  if (opts.signal?.aborted) {
612
889
  return toErrorResponse(
613
- new InternalError("llm call aborted", { provider: leg.provider, model: leg.model })
890
+ new InternalError3("llm call aborted", { provider: leg.provider, model: leg.model })
614
891
  );
615
892
  }
616
893
  try {
@@ -685,7 +962,7 @@ async function complete(messages, env, opts = {}, deps = {}) {
685
962
  } catch (e) {
686
963
  if (e instanceof DOMException && e.name === "AbortError") {
687
964
  return toErrorResponse(
688
- new InternalError("llm call aborted", { provider: leg.provider, model: leg.model })
965
+ new InternalError3("llm call aborted", { provider: leg.provider, model: leg.model })
689
966
  );
690
967
  }
691
968
  if (isProviderError(e)) {
@@ -702,15 +979,15 @@ async function complete(messages, env, opts = {}, deps = {}) {
702
979
  }
703
980
  }
704
981
  return toErrorResponse(
705
- new InternalError("LLM_ALL_PROVIDERS_FAILED", { attempts: attemptLog, tier, tokenEstimate })
982
+ new InternalError3("LLM_ALL_PROVIDERS_FAILED", { attempts: attemptLog, tier, tokenEstimate })
706
983
  );
707
984
  }
708
985
  async function* completionStream(messages, env, opts = {}) {
709
986
  if (messages.length === 0) {
710
- throw new ValidationError("messages must not be empty");
987
+ throw new ValidationError2("messages must not be empty");
711
988
  }
712
989
  if (!env.AI_GATEWAY_BASE_URL) {
713
- throw new ValidationError("AI_GATEWAY_BASE_URL is required in 0.3.0");
990
+ throw new ValidationError2("AI_GATEWAY_BASE_URL is required in 0.3.0");
714
991
  }
715
992
  const deps = opts.deps ?? {};
716
993
  const fetchImpl = deps.fetch ?? fetch;
@@ -725,7 +1002,7 @@ async function* completionStream(messages, env, opts = {}) {
725
1002
  if (streamLeg.provider !== "anthropic") {
726
1003
  const result = await complete(messages, env, opts, deps);
727
1004
  if (result.error !== null || result.data === null) {
728
- throw new InternalError("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
1005
+ throw new InternalError3("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
729
1006
  }
730
1007
  yield result.data.content;
731
1008
  return result.data;
@@ -734,7 +1011,7 @@ async function* completionStream(messages, env, opts = {}) {
734
1011
  logger?.warn?.("llm.provider.coolingDown", { provider: streamLeg.provider });
735
1012
  const result = await complete(messages, env, opts, deps);
736
1013
  if (result.error !== null || result.data === null) {
737
- throw new InternalError("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
1014
+ throw new InternalError3("LLM_ALL_PROVIDERS_FAILED", { error: result.error });
738
1015
  }
739
1016
  yield result.data.content;
740
1017
  return result.data;
@@ -754,12 +1031,12 @@ async function* completionStream(messages, env, opts = {}) {
754
1031
  });
755
1032
  } catch (e) {
756
1033
  if (e instanceof DOMException && e.name === "AbortError") {
757
- throw new InternalError("llm call aborted", {
1034
+ throw new InternalError3("llm call aborted", {
758
1035
  provider: streamLeg.provider,
759
1036
  model: streamLeg.model
760
1037
  });
761
1038
  }
762
- throw new InternalError("llm stream fetch failed", {
1039
+ throw new InternalError3("llm stream fetch failed", {
763
1040
  message: e instanceof Error ? e.message : String(e)
764
1041
  });
765
1042
  }
@@ -771,7 +1048,7 @@ async function* completionStream(messages, env, opts = {}) {
771
1048
  }
772
1049
  const result = await complete(messages, env, opts, deps);
773
1050
  if (result.error !== null || result.data === null) {
774
- throw new InternalError("LLM_ALL_PROVIDERS_FAILED", {
1051
+ throw new InternalError3("LLM_ALL_PROVIDERS_FAILED", {
775
1052
  streamError: `${streamLeg.provider} ${String(response.status)}: ${text.slice(0, 300)}`,
776
1053
  error: result.error
777
1054
  });
@@ -780,7 +1057,7 @@ async function* completionStream(messages, env, opts = {}) {
780
1057
  return result.data;
781
1058
  }
782
1059
  if (!response.body) {
783
- throw new InternalError("llm stream response body is null", {
1060
+ throw new InternalError3("llm stream response body is null", {
784
1061
  provider: streamLeg.provider
785
1062
  });
786
1063
  }
@@ -897,14 +1174,21 @@ function assertGrounding(response, sources) {
897
1174
  }
898
1175
  export {
899
1176
  BASE_BACKOFF_MS,
1177
+ DEFAULT_EMBEDDING_MODEL,
1178
+ LOCAL_EMBEDDING_MODEL,
900
1179
  MODELS,
901
1180
  MODEL_PRICE_PER_1M,
902
1181
  PROVIDER_COOLDOWN_MS,
903
1182
  assertGrounding,
1183
+ clearGcpTokenCache,
904
1184
  clearProviderCooldown,
905
1185
  complete,
906
1186
  completionStream,
1187
+ embed,
1188
+ embedLocal,
907
1189
  isProviderCoolingDown,
908
- markProviderCoolingDown
1190
+ markProviderCoolingDown,
1191
+ mintGcpAccessToken,
1192
+ serviceAccountProjectId
909
1193
  };
910
1194
  //# sourceMappingURL=index.mjs.map