@latimer-woods-tech/llm 0.4.4 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/LICENSE +21 -0
- package/dist/index.d.mts +142 -14
- package/dist/index.mjs +320 -36
- package/dist/index.mjs.map +1 -1
- package/package.json +2 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,16 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.6.0 — 2026-08-03
|
|
4
|
+
|
|
5
|
+
### Added — local Qwen workbench and tool protocol
|
|
6
|
+
|
|
7
|
+
- `LLM_LOCAL_WORKBENCH` routes workbench tasks to `qwen3.6:27b` first, with
|
|
8
|
+
DeepSeek retained as an independent cloud fallback.
|
|
9
|
+
- The local OpenAI-compatible provider now forwards normalized tool schemas,
|
|
10
|
+
tool choice, assistant tool calls, and tool results.
|
|
11
|
+
- Explicit `qwen*` model overrides route locally, and both local models are
|
|
12
|
+
priced at zero marginal token cost in the canonical ledger table.
|
|
13
|
+
|
|
3
14
|
## 0.4.4 — 2026-06-03
|
|
4
15
|
|
|
5
16
|
### Added — streaming tool-calls (Phase 1c; Anthropic)
|
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Adrian Perry / Latimer Woods Technology
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/dist/index.d.mts
CHANGED
|
@@ -1,6 +1,96 @@
|
|
|
1
1
|
import { FactoryResponse } from '@latimer-woods-tech/errors';
|
|
2
2
|
import { Logger } from '@latimer-woods-tech/logger';
|
|
3
3
|
|
|
4
|
+
/**
|
|
5
|
+
* Workers AI embedding API.
|
|
6
|
+
*
|
|
7
|
+
* Pins to bge-base-en-v1.5 (768-dim cosine) — the platform standard.
|
|
8
|
+
* @remarks Local-rail errors use @latimer-woods-tech/errors (observability ratchet).
|
|
9
|
+
* model_version is returned with every result so callers can record it
|
|
10
|
+
* for provenance-aware re-embed when the model is swapped.
|
|
11
|
+
*
|
|
12
|
+
* Contract: inject the Workers AI binding (env.AI) — never import a vendor SDK.
|
|
13
|
+
*/
|
|
14
|
+
/** The platform-standard embedding model (768-dim cosine) used when no override is given. */
|
|
15
|
+
declare const DEFAULT_EMBEDDING_MODEL = "@cf/baai/bge-base-en-v1.5";
|
|
16
|
+
/** Supported embedding model identifiers — widen deliberately, since dims are part of the contract. */
|
|
17
|
+
type EmbeddingModel = '@cf/baai/bge-base-en-v1.5';
|
|
18
|
+
/** One embedding call's output: the vectors plus the provenance needed to re-embed later. */
|
|
19
|
+
interface EmbedResult {
|
|
20
|
+
vectors: number[][];
|
|
21
|
+
model: string;
|
|
22
|
+
dims: number;
|
|
23
|
+
}
|
|
24
|
+
/** Minimal Workers AI binding shape needed for embeddings. */
|
|
25
|
+
interface AiBinding {
|
|
26
|
+
run(model: string, inputs: {
|
|
27
|
+
text: string | string[];
|
|
28
|
+
}): Promise<{
|
|
29
|
+
data: number[][];
|
|
30
|
+
}>;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Embeds one or more text strings using the Workers AI binding.
|
|
34
|
+
*
|
|
35
|
+
* @param ai - Workers AI binding (`env.AI`) — injected, not imported.
|
|
36
|
+
* @param input - Single string or array of strings to embed.
|
|
37
|
+
* @param opts - Optional model override (must be a supported EmbeddingModel).
|
|
38
|
+
* @returns Embedding vectors, model name, and dimension count.
|
|
39
|
+
*
|
|
40
|
+
* @throws When the AI binding call fails (let the caller decide whether to catch).
|
|
41
|
+
*/
|
|
42
|
+
declare function embed(ai: AiBinding, input: string | string[], opts?: {
|
|
43
|
+
model?: EmbeddingModel;
|
|
44
|
+
}): Promise<EmbedResult>;
|
|
45
|
+
/**
|
|
46
|
+
* Self-hosted GPU embedding model (nomic-embed-text, 768-dim), reached through
|
|
47
|
+
* the same CF AI Gateway custom provider as the local chat provider.
|
|
48
|
+
*
|
|
49
|
+
* ⚠️ VECTOR-SPACE WARNING: nomic-embed-text and bge-base-en-v1.5 are BOTH 768-dim
|
|
50
|
+
* but occupy DIFFERENT embedding spaces. Their vectors are NOT comparable. A store
|
|
51
|
+
* built with one model must be QUERIED — and, when migrating, RE-EMBEDDED — with
|
|
52
|
+
* the SAME model. Never write vectors from both models into one store.
|
|
53
|
+
*/
|
|
54
|
+
declare const LOCAL_EMBEDDING_MODEL = "nomic-embed-text";
|
|
55
|
+
/** Env subset needed to reach the local embedding rail (shares GPU_LLM_* with the chat provider). */
|
|
56
|
+
interface LocalEmbedEnv {
|
|
57
|
+
AI_GATEWAY_BASE_URL: string;
|
|
58
|
+
GPU_LLM_API_TOKEN?: string;
|
|
59
|
+
GPU_LLM_ACCESS_CLIENT_ID?: string;
|
|
60
|
+
GPU_LLM_ACCESS_CLIENT_SECRET?: string;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Embeds text on the self-hosted GPU rail (nomic-embed-text) via the AI Gateway
|
|
64
|
+
* custom provider — the $0 alternative to Workers AI {@link embed}.
|
|
65
|
+
*
|
|
66
|
+
* Unlike the local CHAT provider, this has NO cross-provider fallback ON PURPOSE:
|
|
67
|
+
* falling back to Workers AI (bge) would silently write an incompatible-space
|
|
68
|
+
* vector into a nomic store and corrupt retrieval. If the rail is unavailable
|
|
69
|
+
* this THROWS and the caller decides (retry local, or defer the write) — it must
|
|
70
|
+
* NOT substitute a different model.
|
|
71
|
+
*
|
|
72
|
+
* @param env - Gateway base URL + GPU_LLM_* auth (same secrets as the chat provider).
|
|
73
|
+
* @param input - Single string or array of strings to embed.
|
|
74
|
+
* @returns Embedding vectors, model `nomic-embed-text`, and dimension count.
|
|
75
|
+
* @throws On any transport/HTTP/empty-response error (no silent fallback).
|
|
76
|
+
*/
|
|
77
|
+
declare function embedLocal(env: LocalEmbedEnv, input: string | string[]): Promise<EmbedResult>;
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Exchange a service-account key for a cloud-platform access token, reusing a
|
|
81
|
+
* cached token until it is within {@link EXPIRY_SKEW_MS} of expiring.
|
|
82
|
+
*
|
|
83
|
+
* @param gcpSaKey - Service-account JSON key, base64-encoded or raw.
|
|
84
|
+
* @param fetchImpl - Fetch implementation (injectable for tests).
|
|
85
|
+
* @returns A bearer token valid for Vertex AI.
|
|
86
|
+
* @throws ValidationError when the key is malformed; InternalError when the exchange fails.
|
|
87
|
+
*/
|
|
88
|
+
declare function mintGcpAccessToken(gcpSaKey: string, fetchImpl: typeof fetch): Promise<string>;
|
|
89
|
+
/** Reset the per-isolate token cache. Tests only. */
|
|
90
|
+
declare function clearGcpTokenCache(): void;
|
|
91
|
+
/** The project the service account belongs to — the Vertex project by default. */
|
|
92
|
+
declare function serviceAccountProjectId(gcpSaKey: string): string;
|
|
93
|
+
|
|
4
94
|
/**
|
|
5
95
|
* A tool the model may call. `parameters` is a JSON Schema object describing
|
|
6
96
|
* the tool's input. Provider-agnostic; normalized per provider at request time.
|
|
@@ -56,7 +146,7 @@ interface LLMMessage {
|
|
|
56
146
|
* Quality tier selected by the caller. Routing is workload-split:
|
|
57
147
|
* - `fast` → Grok 4.3 with Anthropic Haiku fallback (routine drafts/small jobs)
|
|
58
148
|
* - `balanced` → Anthropic Sonnet (default)
|
|
59
|
-
* - `smart` → Anthropic Opus OR Gemini 2.5
|
|
149
|
+
* - `smart` → Anthropic Opus OR Gemini 2.5 Flash if input is long-context (>150k tokens estimated)
|
|
60
150
|
* - `verifier` → Groq Llama (cheap second opinion; only used from verifier code path)
|
|
61
151
|
* - `workbench` → DeepSeek Chat with Groq fallback (boring, reviewable, non-sensitive batch work)
|
|
62
152
|
*/
|
|
@@ -135,7 +225,7 @@ interface LLMOptions {
|
|
|
135
225
|
/**
|
|
136
226
|
* Provider that produced an LLM response.
|
|
137
227
|
*/
|
|
138
|
-
type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek';
|
|
228
|
+
type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek' | 'local';
|
|
139
229
|
/**
|
|
140
230
|
* Result returned by a successful completion.
|
|
141
231
|
*/
|
|
@@ -182,13 +272,47 @@ interface LLMEnv {
|
|
|
182
272
|
/** Optional — only required when caller passes `{ model: 'grok-*' }` override. */
|
|
183
273
|
GROK_API_KEY?: string;
|
|
184
274
|
/**
|
|
185
|
-
*
|
|
186
|
-
*
|
|
187
|
-
*
|
|
275
|
+
* Cost optimization: when true, `fast`-tier calls route to the self-hosted GPU
|
|
276
|
+
* (qwen3 via the `custom-local-gpu` AI Gateway provider) FIRST, with the normal
|
|
277
|
+
* fast route (Grok→Haiku) as automatic fallback on any error. Off by default →
|
|
278
|
+
* fully dormant. Requires {@link LLMEnv.GPU_LLM_API_TOKEN}.
|
|
279
|
+
*/
|
|
280
|
+
LLM_LOCAL_FIRST?: boolean;
|
|
281
|
+
/**
|
|
282
|
+
* Agentic-work opt-in: route the `workbench` tier to local Qwen 27B first,
|
|
283
|
+
* with DeepSeek as the independent cloud fallback.
|
|
284
|
+
*/
|
|
285
|
+
LLM_LOCAL_WORKBENCH?: boolean;
|
|
286
|
+
/** Bearer for the local GPU proxy (`GPU_LLM_API_TOKEN` in Secret Manager). Required to enable local. */
|
|
287
|
+
GPU_LLM_API_TOKEN?: string;
|
|
288
|
+
/** Cloudflare Access service-token id for the local GPU app; forwarded by the gateway to the origin. */
|
|
289
|
+
GPU_LLM_ACCESS_CLIENT_ID?: string;
|
|
290
|
+
/** Cloudflare Access service-token secret for the local GPU app. */
|
|
291
|
+
GPU_LLM_ACCESS_CLIENT_SECRET?: string;
|
|
292
|
+
/**
|
|
293
|
+
* GCP service-account JSON key (base64-encoded or raw) with `roles/aiplatform.user`.
|
|
294
|
+
* The credential for the `gemini` leg in environments WITHOUT a metadata server
|
|
295
|
+
* (Cloudflare Workers): the package exchanges it for a fresh access token per
|
|
296
|
+
* isolate and caches it (see `./gcp-token.ts`).
|
|
297
|
+
*
|
|
298
|
+
* Optional. When neither this nor {@link LLMEnv.VERTEX_ACCESS_TOKEN} is set, the
|
|
299
|
+
* leg falls back to Application Default Credentials — the GCP metadata server's
|
|
300
|
+
* ambient service-account token (keyless; the default on Cloud Run / GCE). Off
|
|
301
|
+
* GCP with no static credential, the metadata probe fails and the chain falls
|
|
302
|
+
* through to its next leg.
|
|
303
|
+
*/
|
|
304
|
+
GCP_SA_KEY?: string;
|
|
305
|
+
/**
|
|
306
|
+
* Pre-minted Google Cloud access token with `aiplatform.endpoints.predict`.
|
|
307
|
+
* Explicit override, used only when {@link LLMEnv.GCP_SA_KEY} is absent and
|
|
308
|
+
* before ADC is attempted. Valid for ~1 hour from minting, so it cannot be a
|
|
309
|
+
* durable Worker secret.
|
|
188
310
|
*/
|
|
189
|
-
VERTEX_ACCESS_TOKEN
|
|
190
|
-
|
|
191
|
-
|
|
311
|
+
VERTEX_ACCESS_TOKEN?: string;
|
|
312
|
+
/** Vertex project. Defaults to the `GCP_SA_KEY`'s own `project_id`. */
|
|
313
|
+
VERTEX_PROJECT?: string;
|
|
314
|
+
/** Vertex location. Defaults to `us-central1`. */
|
|
315
|
+
VERTEX_LOCATION?: string;
|
|
192
316
|
/**
|
|
193
317
|
* Optional KV store for org-level daily/monthly cost tracking and enforcement.
|
|
194
318
|
* When provided alongside {@link LLMOptions.dailyCapUsd} or {@link LLMOptions.monthlyCapUsd},
|
|
@@ -251,15 +375,15 @@ interface LLMDeps {
|
|
|
251
375
|
}
|
|
252
376
|
declare const MODELS: {
|
|
253
377
|
readonly anthropic: {
|
|
254
|
-
readonly fast: "claude-haiku-4-
|
|
378
|
+
readonly fast: "claude-haiku-4-5";
|
|
255
379
|
readonly balanced: "claude-sonnet-4-6";
|
|
256
380
|
readonly smart: "claude-opus-4-7";
|
|
257
381
|
};
|
|
258
382
|
readonly gemini: {
|
|
259
|
-
readonly smart: "gemini-2.5-
|
|
383
|
+
readonly smart: "gemini-2.5-flash";
|
|
260
384
|
};
|
|
261
385
|
readonly groq: {
|
|
262
|
-
readonly verifier: "llama-
|
|
386
|
+
readonly verifier: "llama-3.3-70b-versatile";
|
|
263
387
|
};
|
|
264
388
|
readonly grok: {
|
|
265
389
|
readonly fast: "grok-4.3";
|
|
@@ -267,6 +391,10 @@ declare const MODELS: {
|
|
|
267
391
|
readonly deepseek: {
|
|
268
392
|
readonly workbench: "deepseek-chat";
|
|
269
393
|
};
|
|
394
|
+
readonly local: {
|
|
395
|
+
readonly fast: "qwen3:8b";
|
|
396
|
+
readonly workbench: "qwen3.6:27b";
|
|
397
|
+
};
|
|
270
398
|
};
|
|
271
399
|
/** Cooldown duration in ms after a provider exhausts all retries. */
|
|
272
400
|
declare const PROVIDER_COOLDOWN_MS = 30000;
|
|
@@ -305,8 +433,8 @@ declare const BASE_BACKOFF_MS = 250;
|
|
|
305
433
|
*
|
|
306
434
|
* Routing summary (0.3.0):
|
|
307
435
|
* - `fast` → Grok 4.3; Anthropic Haiku fallback when Grok is unavailable
|
|
308
|
-
* - `balanced` → Anthropic Sonnet; Gemini 2.5
|
|
309
|
-
* - `smart` → Anthropic Opus; Gemini 2.5
|
|
436
|
+
* - `balanced` → Anthropic Sonnet; Gemini 2.5 Flash if `longContextThreshold` exceeded
|
|
437
|
+
* - `smart` → Anthropic Opus; Gemini 2.5 Flash if long-context
|
|
310
438
|
* - `verifier` → Groq Llama 3.3 70B (no fallback — verifier is inherently cheap/best-effort)
|
|
311
439
|
* - `workbench` → DeepSeek Chat; Groq fallback for boring/reviewable internal batch jobs
|
|
312
440
|
*
|
|
@@ -376,4 +504,4 @@ declare function completionStream(messages: LLMMessage[], env: LLMEnv, opts?: LL
|
|
|
376
504
|
*/
|
|
377
505
|
declare function assertGrounding(response: string, sources: string[]): boolean;
|
|
378
506
|
|
|
379
|
-
export { BASE_BACKOFF_MS, type CostKvStore, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearProviderCooldown, complete, completionStream, isProviderCoolingDown, markProviderCoolingDown };
|
|
507
|
+
export { type AiBinding, BASE_BACKOFF_MS, type CostKvStore, DEFAULT_EMBEDDING_MODEL, type EmbedResult, type EmbeddingModel, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, LOCAL_EMBEDDING_MODEL, type LocalEmbedEnv, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearGcpTokenCache, clearProviderCooldown, complete, completionStream, embed, embedLocal, isProviderCoolingDown, markProviderCoolingDown, mintGcpAccessToken, serviceAccountProjectId };
|