@latimer-woods-tech/llm 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,16 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.6.0 — 2026-08-03
4
+
5
+ ### Added — local Qwen workbench and tool protocol
6
+
7
+ - `LLM_LOCAL_WORKBENCH` routes workbench tasks to `qwen3.6:27b` first, with
8
+ DeepSeek retained as an independent cloud fallback.
9
+ - The local OpenAI-compatible provider now forwards normalized tool schemas,
10
+ tool choice, assistant tool calls, and tool results.
11
+ - Explicit `qwen*` model overrides route locally, and both local models are
12
+ priced at zero marginal token cost in the canonical ledger table.
13
+
3
14
  ## 0.4.4 — 2026-06-03
4
15
 
5
16
  ### Added — streaming tool-calls (Phase 1c; Anthropic)
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Adrian Perry / Latimer Woods Technology
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/dist/index.d.mts CHANGED
@@ -5,13 +5,17 @@ import { Logger } from '@latimer-woods-tech/logger';
5
5
  * Workers AI embedding API.
6
6
  *
7
7
  * Pins to bge-base-en-v1.5 (768-dim cosine) — the platform standard.
8
+ * @remarks Local-rail errors use @latimer-woods-tech/errors (observability ratchet).
8
9
  * model_version is returned with every result so callers can record it
9
10
  * for provenance-aware re-embed when the model is swapped.
10
11
  *
11
12
  * Contract: inject the Workers AI binding (env.AI) — never import a vendor SDK.
12
13
  */
14
+ /** The platform-standard embedding model (768-dim cosine) used when no override is given. */
13
15
  declare const DEFAULT_EMBEDDING_MODEL = "@cf/baai/bge-base-en-v1.5";
16
+ /** Supported embedding model identifiers — widen deliberately, since dims are part of the contract. */
14
17
  type EmbeddingModel = '@cf/baai/bge-base-en-v1.5';
18
+ /** One embedding call's output: the vectors plus the provenance needed to re-embed later. */
15
19
  interface EmbedResult {
16
20
  vectors: number[][];
17
21
  model: string;
@@ -38,6 +42,54 @@ interface AiBinding {
38
42
  declare function embed(ai: AiBinding, input: string | string[], opts?: {
39
43
  model?: EmbeddingModel;
40
44
  }): Promise<EmbedResult>;
45
+ /**
46
+ * Self-hosted GPU embedding model (nomic-embed-text, 768-dim), reached through
47
+ * the same CF AI Gateway custom provider as the local chat provider.
48
+ *
49
+ * ⚠️ VECTOR-SPACE WARNING: nomic-embed-text and bge-base-en-v1.5 are BOTH 768-dim
50
+ * but occupy DIFFERENT embedding spaces. Their vectors are NOT comparable. A store
51
+ * built with one model must be QUERIED — and, when migrating, RE-EMBEDDED — with
52
+ * the SAME model. Never write vectors from both models into one store.
53
+ */
54
+ declare const LOCAL_EMBEDDING_MODEL = "nomic-embed-text";
55
+ /** Env subset needed to reach the local embedding rail (shares GPU_LLM_* with the chat provider). */
56
+ interface LocalEmbedEnv {
57
+ AI_GATEWAY_BASE_URL: string;
58
+ GPU_LLM_API_TOKEN?: string;
59
+ GPU_LLM_ACCESS_CLIENT_ID?: string;
60
+ GPU_LLM_ACCESS_CLIENT_SECRET?: string;
61
+ }
62
+ /**
63
+ * Embeds text on the self-hosted GPU rail (nomic-embed-text) via the AI Gateway
64
+ * custom provider — the $0 alternative to Workers AI {@link embed}.
65
+ *
66
+ * Unlike the local CHAT provider, this has NO cross-provider fallback ON PURPOSE:
67
+ * falling back to Workers AI (bge) would silently write an incompatible-space
68
+ * vector into a nomic store and corrupt retrieval. If the rail is unavailable
69
+ * this THROWS and the caller decides (retry local, or defer the write) — it must
70
+ * NOT substitute a different model.
71
+ *
72
+ * @param env - Gateway base URL + GPU_LLM_* auth (same secrets as the chat provider).
73
+ * @param input - Single string or array of strings to embed.
74
+ * @returns Embedding vectors, model `nomic-embed-text`, and dimension count.
75
+ * @throws On any transport/HTTP/empty-response error (no silent fallback).
76
+ */
77
+ declare function embedLocal(env: LocalEmbedEnv, input: string | string[]): Promise<EmbedResult>;
78
+
79
+ /**
80
+ * Exchange a service-account key for a cloud-platform access token, reusing a
81
+ * cached token until it is within {@link EXPIRY_SKEW_MS} of expiring.
82
+ *
83
+ * @param gcpSaKey - Service-account JSON key, base64-encoded or raw.
84
+ * @param fetchImpl - Fetch implementation (injectable for tests).
85
+ * @returns A bearer token valid for Vertex AI.
86
+ * @throws ValidationError when the key is malformed; InternalError when the exchange fails.
87
+ */
88
+ declare function mintGcpAccessToken(gcpSaKey: string, fetchImpl: typeof fetch): Promise<string>;
89
+ /** Reset the per-isolate token cache. Tests only. */
90
+ declare function clearGcpTokenCache(): void;
91
+ /** The project the service account belongs to — the Vertex project by default. */
92
+ declare function serviceAccountProjectId(gcpSaKey: string): string;
41
93
 
42
94
  /**
43
95
  * A tool the model may call. `parameters` is a JSON Schema object describing
@@ -94,7 +146,7 @@ interface LLMMessage {
94
146
  * Quality tier selected by the caller. Routing is workload-split:
95
147
  * - `fast` → Grok 4.3 with Anthropic Haiku fallback (routine drafts/small jobs)
96
148
  * - `balanced` → Anthropic Sonnet (default)
97
- * - `smart` → Anthropic Opus OR Gemini 2.5 Pro if input is long-context (>150k tokens estimated)
149
+ * - `smart` → Anthropic Opus OR Gemini 2.5 Flash if input is long-context (>150k tokens estimated)
98
150
  * - `verifier` → Groq Llama (cheap second opinion; only used from verifier code path)
99
151
  * - `workbench` → DeepSeek Chat with Groq fallback (boring, reviewable, non-sensitive batch work)
100
152
  */
@@ -173,7 +225,7 @@ interface LLMOptions {
173
225
  /**
174
226
  * Provider that produced an LLM response.
175
227
  */
176
- type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek';
228
+ type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek' | 'local';
177
229
  /**
178
230
  * Result returned by a successful completion.
179
231
  */
@@ -220,13 +272,47 @@ interface LLMEnv {
220
272
  /** Optional — only required when caller passes `{ model: 'grok-*' }` override. */
221
273
  GROK_API_KEY?: string;
222
274
  /**
223
- * Google Cloud short-lived access token with `aiplatform.endpoints.predict`.
224
- * Callers mint this via the JWT-bearer flow (service account → token exchange);
225
- * see `docs/runbooks/rotate-gcp-sa.md`. Token must be valid for ≥ 5 minutes.
275
+ * Cost optimization: when true, `fast`-tier calls route to the self-hosted GPU
276
+ * (qwen3 via the `custom-local-gpu` AI Gateway provider) FIRST, with the normal
277
+ * fast route (Grok→Haiku) as automatic fallback on any error. Off by default →
278
+ * fully dormant. Requires {@link LLMEnv.GPU_LLM_API_TOKEN}.
279
+ */
280
+ LLM_LOCAL_FIRST?: boolean;
281
+ /**
282
+ * Agentic-work opt-in: route the `workbench` tier to local Qwen 27B first,
283
+ * with DeepSeek as the independent cloud fallback.
226
284
  */
227
- VERTEX_ACCESS_TOKEN: string;
228
- VERTEX_PROJECT: string;
229
- VERTEX_LOCATION: string;
285
+ LLM_LOCAL_WORKBENCH?: boolean;
286
+ /** Bearer for the local GPU proxy (`GPU_LLM_API_TOKEN` in Secret Manager). Required to enable local. */
287
+ GPU_LLM_API_TOKEN?: string;
288
+ /** Cloudflare Access service-token id for the local GPU app; forwarded by the gateway to the origin. */
289
+ GPU_LLM_ACCESS_CLIENT_ID?: string;
290
+ /** Cloudflare Access service-token secret for the local GPU app. */
291
+ GPU_LLM_ACCESS_CLIENT_SECRET?: string;
292
+ /**
293
+ * GCP service-account JSON key (base64-encoded or raw) with `roles/aiplatform.user`.
294
+ * The credential for the `gemini` leg in environments WITHOUT a metadata server
295
+ * (Cloudflare Workers): the package exchanges it for a fresh access token per
296
+ * isolate and caches it (see `./gcp-token.ts`).
297
+ *
298
+ * Optional. When neither this nor {@link LLMEnv.VERTEX_ACCESS_TOKEN} is set, the
299
+ * leg falls back to Application Default Credentials — the GCP metadata server's
300
+ * ambient service-account token (keyless; the default on Cloud Run / GCE). Off
301
+ * GCP with no static credential, the metadata probe fails and the chain falls
302
+ * through to its next leg.
303
+ */
304
+ GCP_SA_KEY?: string;
305
+ /**
306
+ * Pre-minted Google Cloud access token with `aiplatform.endpoints.predict`.
307
+ * Explicit override, used only when {@link LLMEnv.GCP_SA_KEY} is absent and
308
+ * before ADC is attempted. Valid for ~1 hour from minting, so it cannot be a
309
+ * durable Worker secret.
310
+ */
311
+ VERTEX_ACCESS_TOKEN?: string;
312
+ /** Vertex project. Defaults to the `GCP_SA_KEY`'s own `project_id`. */
313
+ VERTEX_PROJECT?: string;
314
+ /** Vertex location. Defaults to `us-central1`. */
315
+ VERTEX_LOCATION?: string;
230
316
  /**
231
317
  * Optional KV store for org-level daily/monthly cost tracking and enforcement.
232
318
  * When provided alongside {@link LLMOptions.dailyCapUsd} or {@link LLMOptions.monthlyCapUsd},
@@ -289,15 +375,15 @@ interface LLMDeps {
289
375
  }
290
376
  declare const MODELS: {
291
377
  readonly anthropic: {
292
- readonly fast: "claude-haiku-4-20250514";
378
+ readonly fast: "claude-haiku-4-5";
293
379
  readonly balanced: "claude-sonnet-4-6";
294
380
  readonly smart: "claude-opus-4-7";
295
381
  };
296
382
  readonly gemini: {
297
- readonly smart: "gemini-2.5-pro";
383
+ readonly smart: "gemini-2.5-flash";
298
384
  };
299
385
  readonly groq: {
300
- readonly verifier: "llama-4-maverick";
386
+ readonly verifier: "llama-3.3-70b-versatile";
301
387
  };
302
388
  readonly grok: {
303
389
  readonly fast: "grok-4.3";
@@ -305,6 +391,10 @@ declare const MODELS: {
305
391
  readonly deepseek: {
306
392
  readonly workbench: "deepseek-chat";
307
393
  };
394
+ readonly local: {
395
+ readonly fast: "qwen3:8b";
396
+ readonly workbench: "qwen3.6:27b";
397
+ };
308
398
  };
309
399
  /** Cooldown duration in ms after a provider exhausts all retries. */
310
400
  declare const PROVIDER_COOLDOWN_MS = 30000;
@@ -343,8 +433,8 @@ declare const BASE_BACKOFF_MS = 250;
343
433
  *
344
434
  * Routing summary (0.3.0):
345
435
  * - `fast` → Grok 4.3; Anthropic Haiku fallback when Grok is unavailable
346
- * - `balanced` → Anthropic Sonnet; Gemini 2.5 Pro if `longContextThreshold` exceeded
347
- * - `smart` → Anthropic Opus; Gemini 2.5 Pro if long-context
436
+ * - `balanced` → Anthropic Sonnet; Gemini 2.5 Flash if `longContextThreshold` exceeded
437
+ * - `smart` → Anthropic Opus; Gemini 2.5 Flash if long-context
348
438
  * - `verifier` → Groq Llama 3.3 70B (no fallback — verifier is inherently cheap/best-effort)
349
439
  * - `workbench` → DeepSeek Chat; Groq fallback for boring/reviewable internal batch jobs
350
440
  *
@@ -414,4 +504,4 @@ declare function completionStream(messages: LLMMessage[], env: LLMEnv, opts?: LL
414
504
  */
415
505
  declare function assertGrounding(response: string, sources: string[]): boolean;
416
506
 
417
- export { type AiBinding, BASE_BACKOFF_MS, type CostKvStore, DEFAULT_EMBEDDING_MODEL, type EmbedResult, type EmbeddingModel, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearProviderCooldown, complete, completionStream, embed, isProviderCoolingDown, markProviderCoolingDown };
507
+ export { type AiBinding, BASE_BACKOFF_MS, type CostKvStore, DEFAULT_EMBEDDING_MODEL, type EmbedResult, type EmbeddingModel, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, LOCAL_EMBEDDING_MODEL, type LocalEmbedEnv, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearGcpTokenCache, clearProviderCooldown, complete, completionStream, embed, embedLocal, isProviderCoolingDown, markProviderCoolingDown, mintGcpAccessToken, serviceAccountProjectId };