@latimer-woods-tech/llm 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/LICENSE +21 -0
- package/dist/index.d.mts +104 -14
- package/dist/index.mjs +299 -36
- package/dist/index.mjs.map +1 -1
- package/package.json +2 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,16 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.6.0 — 2026-08-03
|
|
4
|
+
|
|
5
|
+
### Added — local Qwen workbench and tool protocol
|
|
6
|
+
|
|
7
|
+
- `LLM_LOCAL_WORKBENCH` routes workbench tasks to `qwen3.6:27b` first, with
|
|
8
|
+
DeepSeek retained as an independent cloud fallback.
|
|
9
|
+
- The local OpenAI-compatible provider now forwards normalized tool schemas,
|
|
10
|
+
tool choice, assistant tool calls, and tool results.
|
|
11
|
+
- Explicit `qwen*` model overrides route locally, and both local models are
|
|
12
|
+
priced at zero marginal token cost in the canonical ledger table.
|
|
13
|
+
|
|
3
14
|
## 0.4.4 — 2026-06-03
|
|
4
15
|
|
|
5
16
|
### Added — streaming tool-calls (Phase 1c; Anthropic)
|
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Adrian Perry / Latimer Woods Technology
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/dist/index.d.mts
CHANGED
|
@@ -5,13 +5,17 @@ import { Logger } from '@latimer-woods-tech/logger';
|
|
|
5
5
|
* Workers AI embedding API.
|
|
6
6
|
*
|
|
7
7
|
* Pins to bge-base-en-v1.5 (768-dim cosine) — the platform standard.
|
|
8
|
+
* @remarks Local-rail errors use @latimer-woods-tech/errors (observability ratchet).
|
|
8
9
|
* model_version is returned with every result so callers can record it
|
|
9
10
|
* for provenance-aware re-embed when the model is swapped.
|
|
10
11
|
*
|
|
11
12
|
* Contract: inject the Workers AI binding (env.AI) — never import a vendor SDK.
|
|
12
13
|
*/
|
|
14
|
+
/** The platform-standard embedding model (768-dim cosine) used when no override is given. */
|
|
13
15
|
declare const DEFAULT_EMBEDDING_MODEL = "@cf/baai/bge-base-en-v1.5";
|
|
16
|
+
/** Supported embedding model identifiers — widen deliberately, since dims are part of the contract. */
|
|
14
17
|
type EmbeddingModel = '@cf/baai/bge-base-en-v1.5';
|
|
18
|
+
/** One embedding call's output: the vectors plus the provenance needed to re-embed later. */
|
|
15
19
|
interface EmbedResult {
|
|
16
20
|
vectors: number[][];
|
|
17
21
|
model: string;
|
|
@@ -38,6 +42,54 @@ interface AiBinding {
|
|
|
38
42
|
declare function embed(ai: AiBinding, input: string | string[], opts?: {
|
|
39
43
|
model?: EmbeddingModel;
|
|
40
44
|
}): Promise<EmbedResult>;
|
|
45
|
+
/**
|
|
46
|
+
* Self-hosted GPU embedding model (nomic-embed-text, 768-dim), reached through
|
|
47
|
+
* the same CF AI Gateway custom provider as the local chat provider.
|
|
48
|
+
*
|
|
49
|
+
* ⚠️ VECTOR-SPACE WARNING: nomic-embed-text and bge-base-en-v1.5 are BOTH 768-dim
|
|
50
|
+
* but occupy DIFFERENT embedding spaces. Their vectors are NOT comparable. A store
|
|
51
|
+
* built with one model must be QUERIED — and, when migrating, RE-EMBEDDED — with
|
|
52
|
+
* the SAME model. Never write vectors from both models into one store.
|
|
53
|
+
*/
|
|
54
|
+
declare const LOCAL_EMBEDDING_MODEL = "nomic-embed-text";
|
|
55
|
+
/** Env subset needed to reach the local embedding rail (shares GPU_LLM_* with the chat provider). */
|
|
56
|
+
interface LocalEmbedEnv {
|
|
57
|
+
AI_GATEWAY_BASE_URL: string;
|
|
58
|
+
GPU_LLM_API_TOKEN?: string;
|
|
59
|
+
GPU_LLM_ACCESS_CLIENT_ID?: string;
|
|
60
|
+
GPU_LLM_ACCESS_CLIENT_SECRET?: string;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Embeds text on the self-hosted GPU rail (nomic-embed-text) via the AI Gateway
|
|
64
|
+
* custom provider — the $0 alternative to Workers AI {@link embed}.
|
|
65
|
+
*
|
|
66
|
+
* Unlike the local CHAT provider, this has NO cross-provider fallback ON PURPOSE:
|
|
67
|
+
* falling back to Workers AI (bge) would silently write an incompatible-space
|
|
68
|
+
* vector into a nomic store and corrupt retrieval. If the rail is unavailable
|
|
69
|
+
* this THROWS and the caller decides (retry local, or defer the write) — it must
|
|
70
|
+
* NOT substitute a different model.
|
|
71
|
+
*
|
|
72
|
+
* @param env - Gateway base URL + GPU_LLM_* auth (same secrets as the chat provider).
|
|
73
|
+
* @param input - Single string or array of strings to embed.
|
|
74
|
+
* @returns Embedding vectors, model `nomic-embed-text`, and dimension count.
|
|
75
|
+
* @throws On any transport/HTTP/empty-response error (no silent fallback).
|
|
76
|
+
*/
|
|
77
|
+
declare function embedLocal(env: LocalEmbedEnv, input: string | string[]): Promise<EmbedResult>;
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Exchange a service-account key for a cloud-platform access token, reusing a
|
|
81
|
+
* cached token until it is within {@link EXPIRY_SKEW_MS} of expiring.
|
|
82
|
+
*
|
|
83
|
+
* @param gcpSaKey - Service-account JSON key, base64-encoded or raw.
|
|
84
|
+
* @param fetchImpl - Fetch implementation (injectable for tests).
|
|
85
|
+
* @returns A bearer token valid for Vertex AI.
|
|
86
|
+
* @throws ValidationError when the key is malformed; InternalError when the exchange fails.
|
|
87
|
+
*/
|
|
88
|
+
declare function mintGcpAccessToken(gcpSaKey: string, fetchImpl: typeof fetch): Promise<string>;
|
|
89
|
+
/** Reset the per-isolate token cache. Tests only. */
|
|
90
|
+
declare function clearGcpTokenCache(): void;
|
|
91
|
+
/** The project the service account belongs to — the Vertex project by default. */
|
|
92
|
+
declare function serviceAccountProjectId(gcpSaKey: string): string;
|
|
41
93
|
|
|
42
94
|
/**
|
|
43
95
|
* A tool the model may call. `parameters` is a JSON Schema object describing
|
|
@@ -94,7 +146,7 @@ interface LLMMessage {
|
|
|
94
146
|
* Quality tier selected by the caller. Routing is workload-split:
|
|
95
147
|
* - `fast` → Grok 4.3 with Anthropic Haiku fallback (routine drafts/small jobs)
|
|
96
148
|
* - `balanced` → Anthropic Sonnet (default)
|
|
97
|
-
* - `smart` → Anthropic Opus OR Gemini 2.5
|
|
149
|
+
* - `smart` → Anthropic Opus OR Gemini 2.5 Flash if input is long-context (>150k tokens estimated)
|
|
98
150
|
* - `verifier` → Groq Llama (cheap second opinion; only used from verifier code path)
|
|
99
151
|
* - `workbench` → DeepSeek Chat with Groq fallback (boring, reviewable, non-sensitive batch work)
|
|
100
152
|
*/
|
|
@@ -173,7 +225,7 @@ interface LLMOptions {
|
|
|
173
225
|
/**
|
|
174
226
|
* Provider that produced an LLM response.
|
|
175
227
|
*/
|
|
176
|
-
type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek';
|
|
228
|
+
type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek' | 'local';
|
|
177
229
|
/**
|
|
178
230
|
* Result returned by a successful completion.
|
|
179
231
|
*/
|
|
@@ -220,13 +272,47 @@ interface LLMEnv {
|
|
|
220
272
|
/** Optional — only required when caller passes `{ model: 'grok-*' }` override. */
|
|
221
273
|
GROK_API_KEY?: string;
|
|
222
274
|
/**
|
|
223
|
-
*
|
|
224
|
-
*
|
|
225
|
-
*
|
|
275
|
+
* Cost optimization: when true, `fast`-tier calls route to the self-hosted GPU
|
|
276
|
+
* (qwen3 via the `custom-local-gpu` AI Gateway provider) FIRST, with the normal
|
|
277
|
+
* fast route (Grok→Haiku) as automatic fallback on any error. Off by default →
|
|
278
|
+
* fully dormant. Requires {@link LLMEnv.GPU_LLM_API_TOKEN}.
|
|
279
|
+
*/
|
|
280
|
+
LLM_LOCAL_FIRST?: boolean;
|
|
281
|
+
/**
|
|
282
|
+
* Agentic-work opt-in: route the `workbench` tier to local Qwen 27B first,
|
|
283
|
+
* with DeepSeek as the independent cloud fallback.
|
|
226
284
|
*/
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
285
|
+
LLM_LOCAL_WORKBENCH?: boolean;
|
|
286
|
+
/** Bearer for the local GPU proxy (`GPU_LLM_API_TOKEN` in Secret Manager). Required to enable local. */
|
|
287
|
+
GPU_LLM_API_TOKEN?: string;
|
|
288
|
+
/** Cloudflare Access service-token id for the local GPU app; forwarded by the gateway to the origin. */
|
|
289
|
+
GPU_LLM_ACCESS_CLIENT_ID?: string;
|
|
290
|
+
/** Cloudflare Access service-token secret for the local GPU app. */
|
|
291
|
+
GPU_LLM_ACCESS_CLIENT_SECRET?: string;
|
|
292
|
+
/**
|
|
293
|
+
* GCP service-account JSON key (base64-encoded or raw) with `roles/aiplatform.user`.
|
|
294
|
+
* The credential for the `gemini` leg in environments WITHOUT a metadata server
|
|
295
|
+
* (Cloudflare Workers): the package exchanges it for a fresh access token per
|
|
296
|
+
* isolate and caches it (see `./gcp-token.ts`).
|
|
297
|
+
*
|
|
298
|
+
* Optional. When neither this nor {@link LLMEnv.VERTEX_ACCESS_TOKEN} is set, the
|
|
299
|
+
* leg falls back to Application Default Credentials — the GCP metadata server's
|
|
300
|
+
* ambient service-account token (keyless; the default on Cloud Run / GCE). Off
|
|
301
|
+
* GCP with no static credential, the metadata probe fails and the chain falls
|
|
302
|
+
* through to its next leg.
|
|
303
|
+
*/
|
|
304
|
+
GCP_SA_KEY?: string;
|
|
305
|
+
/**
|
|
306
|
+
* Pre-minted Google Cloud access token with `aiplatform.endpoints.predict`.
|
|
307
|
+
* Explicit override, used only when {@link LLMEnv.GCP_SA_KEY} is absent and
|
|
308
|
+
* before ADC is attempted. Valid for ~1 hour from minting, so it cannot be a
|
|
309
|
+
* durable Worker secret.
|
|
310
|
+
*/
|
|
311
|
+
VERTEX_ACCESS_TOKEN?: string;
|
|
312
|
+
/** Vertex project. Defaults to the `GCP_SA_KEY`'s own `project_id`. */
|
|
313
|
+
VERTEX_PROJECT?: string;
|
|
314
|
+
/** Vertex location. Defaults to `us-central1`. */
|
|
315
|
+
VERTEX_LOCATION?: string;
|
|
230
316
|
/**
|
|
231
317
|
* Optional KV store for org-level daily/monthly cost tracking and enforcement.
|
|
232
318
|
* When provided alongside {@link LLMOptions.dailyCapUsd} or {@link LLMOptions.monthlyCapUsd},
|
|
@@ -289,15 +375,15 @@ interface LLMDeps {
|
|
|
289
375
|
}
|
|
290
376
|
declare const MODELS: {
|
|
291
377
|
readonly anthropic: {
|
|
292
|
-
readonly fast: "claude-haiku-4-
|
|
378
|
+
readonly fast: "claude-haiku-4-5";
|
|
293
379
|
readonly balanced: "claude-sonnet-4-6";
|
|
294
380
|
readonly smart: "claude-opus-4-7";
|
|
295
381
|
};
|
|
296
382
|
readonly gemini: {
|
|
297
|
-
readonly smart: "gemini-2.5-
|
|
383
|
+
readonly smart: "gemini-2.5-flash";
|
|
298
384
|
};
|
|
299
385
|
readonly groq: {
|
|
300
|
-
readonly verifier: "llama-
|
|
386
|
+
readonly verifier: "llama-3.3-70b-versatile";
|
|
301
387
|
};
|
|
302
388
|
readonly grok: {
|
|
303
389
|
readonly fast: "grok-4.3";
|
|
@@ -305,6 +391,10 @@ declare const MODELS: {
|
|
|
305
391
|
readonly deepseek: {
|
|
306
392
|
readonly workbench: "deepseek-chat";
|
|
307
393
|
};
|
|
394
|
+
readonly local: {
|
|
395
|
+
readonly fast: "qwen3:8b";
|
|
396
|
+
readonly workbench: "qwen3.6:27b";
|
|
397
|
+
};
|
|
308
398
|
};
|
|
309
399
|
/** Cooldown duration in ms after a provider exhausts all retries. */
|
|
310
400
|
declare const PROVIDER_COOLDOWN_MS = 30000;
|
|
@@ -343,8 +433,8 @@ declare const BASE_BACKOFF_MS = 250;
|
|
|
343
433
|
*
|
|
344
434
|
* Routing summary (0.3.0):
|
|
345
435
|
* - `fast` → Grok 4.3; Anthropic Haiku fallback when Grok is unavailable
|
|
346
|
-
* - `balanced` → Anthropic Sonnet; Gemini 2.5
|
|
347
|
-
* - `smart` → Anthropic Opus; Gemini 2.5
|
|
436
|
+
* - `balanced` → Anthropic Sonnet; Gemini 2.5 Flash if `longContextThreshold` exceeded
|
|
437
|
+
* - `smart` → Anthropic Opus; Gemini 2.5 Flash if long-context
|
|
348
438
|
* - `verifier` → Groq Llama 3.3 70B (no fallback — verifier is inherently cheap/best-effort)
|
|
349
439
|
* - `workbench` → DeepSeek Chat; Groq fallback for boring/reviewable internal batch jobs
|
|
350
440
|
*
|
|
@@ -414,4 +504,4 @@ declare function completionStream(messages: LLMMessage[], env: LLMEnv, opts?: LL
|
|
|
414
504
|
*/
|
|
415
505
|
declare function assertGrounding(response: string, sources: string[]): boolean;
|
|
416
506
|
|
|
417
|
-
export { type AiBinding, BASE_BACKOFF_MS, type CostKvStore, DEFAULT_EMBEDDING_MODEL, type EmbedResult, type EmbeddingModel, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearProviderCooldown, complete, completionStream, embed, isProviderCoolingDown, markProviderCoolingDown };
|
|
507
|
+
export { type AiBinding, BASE_BACKOFF_MS, type CostKvStore, DEFAULT_EMBEDDING_MODEL, type EmbedResult, type EmbeddingModel, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, LOCAL_EMBEDDING_MODEL, type LocalEmbedEnv, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearGcpTokenCache, clearProviderCooldown, complete, completionStream, embed, embedLocal, isProviderCoolingDown, markProviderCoolingDown, mintGcpAccessToken, serviceAccountProjectId };
|