@latimer-woods-tech/llm 0.4.2 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,28 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.4.3 — 2026-06-03
4
+
5
+ ### Added — tool-calling for OpenAI-style providers (Phase 1b: Grok, DeepSeek)
6
+
7
+ - **Grok and DeepSeek now support tool-calling.** Requests carry OpenAI-format
8
+ `tools` + `tool_choice`; responses parse `tool_calls` + `finish_reason` into
9
+ the same normalized `LLMResult.toolCalls` / `stopReason` as Anthropic.
10
+ - Anthropic-shaped tool blocks are converted to OpenAI wire format:
11
+ `tool_use` → an assistant message with `tool_calls`; `tool_result` → a
12
+ standalone `tool` message keyed by `tool_call_id`.
13
+ - **`TOOL_CAPABLE_PROVIDERS`** widened to `anthropic, grok, deepseek`, so the
14
+ `fast` (Grok→Haiku) and `workbench` (DeepSeek) tiers support tool loops while
15
+ still failing closed for `verifier` (Groq Llama).
16
+ - Malformed tool-call argument JSON is tolerated (billed as `{}`, never throws).
17
+
18
+ ### Not yet
19
+
20
+ - **Gemini** tool-calling is deferred to a focused follow-up — its
21
+ `tool_use_id`↔function-name correlation and schema constraints need dedicated
22
+ handling. Gemini stays out of `TOOL_CAPABLE_PROVIDERS` until then.
23
+
24
+ ---
25
+
3
26
  ## 0.4.2 — 2026-06-03
4
27
 
5
28
  ### Added — tool-calling (Agent Runtime Phase 1a; Anthropic)
@@ -0,0 +1,379 @@
1
+ import { FactoryResponse } from '@latimer-woods-tech/errors';
2
+ import { Logger } from '@latimer-woods-tech/logger';
3
+
4
+ /**
5
+ * A tool the model may call. `parameters` is a JSON Schema object describing
6
+ * the tool's input. Provider-agnostic; normalized per provider at request time.
7
+ */
8
+ interface LLMTool {
9
+ name: string;
10
+ description?: string;
11
+ /** JSON Schema for the tool's input arguments. */
12
+ parameters: Record<string, unknown>;
13
+ }
14
+ /**
15
+ * A tool invocation requested by the model, normalized across providers.
16
+ */
17
+ interface LLMToolCall {
18
+ /** Provider-assigned call id; echo it back in the matching tool_result. */
19
+ id: string;
20
+ name: string;
21
+ /** Parsed argument object the model passed to the tool. */
22
+ arguments: Record<string, unknown>;
23
+ }
24
+ /**
25
+ * Structured content block for tool-calling conversations. The field shapes
26
+ * mirror the Anthropic Messages wire format so they pass through unchanged.
27
+ */
28
+ type LLMContentBlock = {
29
+ type: 'text';
30
+ text: string;
31
+ } | {
32
+ type: 'tool_use';
33
+ id: string;
34
+ name: string;
35
+ input: Record<string, unknown>;
36
+ } | {
37
+ type: 'tool_result';
38
+ tool_use_id: string;
39
+ content: string;
40
+ is_error?: boolean;
41
+ };
42
+ /**
43
+ * Single chat message exchanged with an LLM provider.
44
+ *
45
+ * `content` is a plain string in the common case. For tool-calling
46
+ * conversations it may be an array of {@link LLMContentBlock}s (e.g. an
47
+ * assistant turn carrying `tool_use` blocks, or a user turn carrying
48
+ * `tool_result` blocks). Providers that don't support tool-calling receive
49
+ * the text projection of the content (see `contentToText`).
50
+ */
51
+ interface LLMMessage {
52
+ role: 'user' | 'assistant' | 'system';
53
+ content: string | LLMContentBlock[];
54
+ }
55
+ /**
56
+ * Quality tier selected by the caller. Routing is workload-split:
57
+ * - `fast` → Grok 4.3 with Anthropic Haiku fallback (routine drafts/small jobs)
58
+ * - `balanced` → Anthropic Sonnet (default)
59
+ * - `smart` → Anthropic Opus OR Gemini 2.5 Pro if input is long-context (>150k tokens estimated)
60
+ * - `verifier` → Groq Llama (cheap second opinion; only used from verifier code path)
61
+ * - `workbench` → DeepSeek Chat with Groq fallback (boring, reviewable, non-sensitive batch work)
62
+ */
63
+ type LLMTier = 'fast' | 'balanced' | 'smart' | 'verifier' | 'workbench';
64
+ /**
65
+ * Options that influence LLM completion behaviour.
66
+ */
67
+ interface LLMOptions {
68
+ /** Quality tier; see {@link LLMTier}. Defaults to `balanced`. */
69
+ tier?: LLMTier;
70
+ /** Explicit model override. Takes precedence over tier. */
71
+ model?: string;
72
+ maxTokens?: number;
73
+ temperature?: number;
74
+ system?: string;
75
+ /** Token budget above which we force long-context routing (Gemini). */
76
+ longContextThreshold?: number;
77
+ /** Per-call cancellation signal. Aborts the in-flight provider request. */
78
+ signal?: AbortSignal;
79
+ /** Optional run identifier stamped on ledger rows + logs. */
80
+ runId?: string;
81
+ /** Optional project identifier stamped on ledger rows + logs. */
82
+ project?: string;
83
+ /** Optional actor identifier (supervisor / worker / human). */
84
+ actor?: string;
85
+ /** Optional workload label used in logs and cost-policy call sites. */
86
+ workload?: string;
87
+ /** Grok reasoning effort. Defaults to `none` for cost-controlled fast/draft calls. */
88
+ reasoningEffort?: 'none' | 'low' | 'medium' | 'high';
89
+ /** Anthropic prompt-cache control. Defaults to `true` for `system` prompts ≥ 1024 tokens. */
90
+ promptCache?: boolean;
91
+ /**
92
+ * Maximum estimated cost in USD for this completion.
93
+ * This cap is enforced after the provider returns because it uses actual
94
+ * response token counts to compute the final cost.
95
+ * If the post-call estimated cost exceeds this cap, `complete` returns a
96
+ * {@link RateLimitError} with code `LLM_COST_CAP_EXCEEDED` and
97
+ * `completionStream` throws the same error.
98
+ * Pricing is based on {@link MODEL_PRICE_PER_1M}; unknown models default to
99
+ * Opus rates (conservative upper bound).
100
+ */
101
+ maxCostUsd?: number;
102
+ /**
103
+ * Org-level daily cost cap in USD. Requires `env.LLM_COST_KV` to be set.
104
+ * When today's cumulative spend read from KV is >= this value, `complete`
105
+ * returns a {@link RateLimitError} with code `LLM_DAILY_CAP_EXCEEDED`
106
+ * without making any provider call. After a successful call the daily
107
+ * accumulator is updated in KV (TTL: 48 h).
108
+ */
109
+ dailyCapUsd?: number;
110
+ /**
111
+ * Org-level monthly cost cap in USD. Requires `env.LLM_COST_KV` to be set.
112
+ * Same enforcement pattern as {@link dailyCapUsd} but keyed by YYYY-MM.
113
+ * KV TTL: 40 days.
114
+ */
115
+ monthlyCapUsd?: number;
116
+ /**
117
+ * Metering context. When supplied and `deps.onRecord` is set, a {@link LLMRecordRow}
118
+ * is emitted after every successful completion. Errors are swallowed.
119
+ */
120
+ ledger?: LLMRecordContext;
121
+ /**
122
+ * Tools the model may call. When present, routing **fails closed** to
123
+ * tool-capable providers — failover never falls back to a provider that
124
+ * can't honour the tool schema. See {@link LLMResult.toolCalls}.
125
+ */
126
+ tools?: LLMTool[];
127
+ /**
128
+ * Tool-selection policy. `'auto'` (default when `tools` is set) lets the
129
+ * model decide; `'none'` forbids tool use; `{ name }` forces a specific tool.
130
+ */
131
+ toolChoice?: 'auto' | 'none' | {
132
+ name: string;
133
+ };
134
+ }
135
+ /**
136
+ * Provider that produced an LLM response.
137
+ */
138
+ type LLMProvider = 'anthropic' | 'gemini' | 'groq' | 'grok' | 'deepseek';
139
+ /**
140
+ * Result returned by a successful completion.
141
+ */
142
+ interface LLMResult {
143
+ content: string;
144
+ provider: LLMProvider;
145
+ model: string;
146
+ tier: LLMTier;
147
+ tokens: {
148
+ input: number;
149
+ output: number;
150
+ cacheRead?: number;
151
+ cacheWrite?: number;
152
+ };
153
+ latency: number;
154
+ /** Number of attempts before success (1 = primary succeeded). */
155
+ attempts: number;
156
+ /** Monotonic request id from AI Gateway, if present in headers. */
157
+ gatewayRequestId?: string;
158
+ /**
159
+ * Why generation stopped, normalized across providers. `'tool_use'` means
160
+ * the model is requesting one or more tool calls (see {@link toolCalls}).
161
+ */
162
+ stopReason?: 'end' | 'tool_use' | 'max_tokens' | 'other';
163
+ /**
164
+ * Tool calls the model requested, normalized across providers. Present
165
+ * (non-empty) when `stopReason === 'tool_use'`.
166
+ */
167
+ toolCalls?: LLMToolCall[];
168
+ }
169
+ /**
170
+ * Environment bindings required by {@link complete}.
171
+ *
172
+ * `AI_GATEWAY_BASE_URL` is REQUIRED in 0.3.0. All provider calls flow through the
173
+ * Cloudflare AI Gateway for unified logging, rate limiting, and cost telemetry.
174
+ * In test/dev the caller may pass a custom fetch impl that short-circuits this.
175
+ */
176
+ interface LLMEnv {
177
+ AI_GATEWAY_BASE_URL: string;
178
+ ANTHROPIC_API_KEY: string;
179
+ GROQ_API_KEY: string;
180
+ /** Optional — only required for `{ tier: 'workbench' }` or `deepseek-*` model overrides. */
181
+ DEEPSEEK_API_KEY?: string;
182
+ /** Optional — only required when caller passes `{ model: 'grok-*' }` override. */
183
+ GROK_API_KEY?: string;
184
+ /**
185
+ * Google Cloud short-lived access token with `aiplatform.endpoints.predict`.
186
+ * Callers mint this via the JWT-bearer flow (service account → token exchange);
187
+ * see `docs/runbooks/rotate-gcp-sa.md`. Token must be valid for ≥ 5 minutes.
188
+ */
189
+ VERTEX_ACCESS_TOKEN: string;
190
+ VERTEX_PROJECT: string;
191
+ VERTEX_LOCATION: string;
192
+ /**
193
+ * Optional KV store for org-level daily/monthly cost tracking and enforcement.
194
+ * When provided alongside {@link LLMOptions.dailyCapUsd} or {@link LLMOptions.monthlyCapUsd},
195
+ * `complete` will block calls that would exceed the declared cap.
196
+ * Any KV-like store satisfying `get`/`put` works (e.g. Cloudflare KV, in-memory stub).
197
+ */
198
+ LLM_COST_KV?: CostKvStore;
199
+ }
200
+ /**
201
+ * Minimal KV store interface for org-level LLM cost tracking.
202
+ * Cloudflare KV satisfies this. An in-memory stub is sufficient for tests.
203
+ */
204
+ interface CostKvStore {
205
+ get(key: string): Promise<string | null>;
206
+ put(key: string, value: string, options?: {
207
+ expirationTtl?: number;
208
+ }): Promise<void>;
209
+ }
210
+ /**
211
+ * Caller-supplied context stamped on every metering row.
212
+ * Mirrors the `LLMRecordContext` in `@latimer-woods-tech/llm-meter`; kept inline
213
+ * to avoid a circular dependency (llm-meter imports llm).
214
+ */
215
+ interface LLMRecordContext {
216
+ project: string;
217
+ actor: string;
218
+ runId?: string;
219
+ workload?: string;
220
+ tenantId?: string;
221
+ }
222
+ /**
223
+ * Row shape passed to the optional {@link LLMDeps.onRecord} callback.
224
+ * Callers can wire this directly to `recordCall` from `@latimer-woods-tech/llm-meter`.
225
+ */
226
+ interface LLMRecordRow extends LLMRecordContext {
227
+ model: string;
228
+ provider: LLMProvider;
229
+ tier: LLMTier;
230
+ inputTokens: number;
231
+ outputTokens: number;
232
+ cacheReadTokens: number;
233
+ cacheWriteTokens: number;
234
+ latencyMs: number;
235
+ costUsd: number;
236
+ yyyyMm: string;
237
+ }
238
+ /**
239
+ * Optional dependencies for {@link complete}.
240
+ */
241
+ interface LLMDeps {
242
+ fetch?: typeof fetch;
243
+ logger?: Logger;
244
+ now?: () => number;
245
+ /**
246
+ * Optional metering callback. Called after every successful completion.
247
+ * Errors are swallowed so metering never blocks the caller.
248
+ * Wire to `recordCall` from `@latimer-woods-tech/llm-meter`.
249
+ */
250
+ onRecord?: (row: LLMRecordRow) => Promise<void>;
251
+ }
252
+ declare const MODELS: {
253
+ readonly anthropic: {
254
+ readonly fast: "claude-haiku-4-20250514";
255
+ readonly balanced: "claude-sonnet-4-6";
256
+ readonly smart: "claude-opus-4-7";
257
+ };
258
+ readonly gemini: {
259
+ readonly smart: "gemini-2.5-pro";
260
+ };
261
+ readonly groq: {
262
+ readonly verifier: "llama-4-maverick";
263
+ };
264
+ readonly grok: {
265
+ readonly fast: "grok-4.3";
266
+ };
267
+ readonly deepseek: {
268
+ readonly workbench: "deepseek-chat";
269
+ };
270
+ };
271
+ /** Cooldown duration in ms after a provider exhausts all retries. */
272
+ declare const PROVIDER_COOLDOWN_MS = 30000;
273
+ /**
274
+ * Returns `true` if the provider is currently in its cooldown window.
275
+ * Uses the injected `now` function (or `Date.now`) for testability.
276
+ */
277
+ declare function isProviderCoolingDown(provider: LLMProvider, now?: () => number): boolean;
278
+ /**
279
+ * USD cost per 1 million tokens for each model.
280
+ * Source: Anthropic / Google / xAI pricing pages as of 2026-05.
281
+ * Keep these model names in sync with the default routing constants in
282
+ * {@link MODELS}; unknown models fall back to Opus rates (conservative upper bound).
283
+ *
284
+ * CANONICAL pricing source for the platform. `@latimer-woods-tech/llm-meter`
285
+ * derives its cents-denominated rates from this table and a drift-guard test
286
+ * there fails CI if they diverge — make all rate changes here.
287
+ */
288
+ declare const MODEL_PRICE_PER_1M: Record<string, {
289
+ input: number;
290
+ output: number;
291
+ cacheRead: number;
292
+ cacheWrite: number;
293
+ }>;
294
+ /**
295
+ * Marks a provider as cooling down for {@link PROVIDER_COOLDOWN_MS} milliseconds.
296
+ */
297
+ declare function markProviderCoolingDown(provider: LLMProvider, now?: () => number): void;
298
+ /**
299
+ * Clears the cooldown state for a provider after a successful call.
300
+ */
301
+ declare function clearProviderCooldown(provider: LLMProvider): void;
302
+ declare const BASE_BACKOFF_MS = 250;
303
+ /**
304
+ * Run a completion through the routing plan for the requested tier.
305
+ *
306
+ * Routing summary (0.3.0):
307
+ * - `fast` → Grok 4.3; Anthropic Haiku fallback when Grok is unavailable
308
+ * - `balanced` → Anthropic Sonnet; Gemini 2.5 Pro if `longContextThreshold` exceeded
309
+ * - `smart` → Anthropic Opus; Gemini 2.5 Pro if long-context
310
+ * - `verifier` → Groq Llama 3.3 70B (no fallback — verifier is inherently cheap/best-effort)
311
+ * - `workbench` → DeepSeek Chat; Groq fallback for boring/reviewable internal batch jobs
312
+ *
313
+ * All provider traffic flows through Cloudflare AI Gateway at `AI_GATEWAY_BASE_URL`.
314
+ *
315
+ * Per-provider reliability guarantees (0.4.0):
316
+ * - Exponential backoff with jitter on 429 / 5xx (base 500ms, cap 8s, up to 2 retries).
317
+ * - Provider cooldown: after exhausting retries the provider is marked cooling down
318
+ * for 30 seconds; subsequent calls skip it and go straight to the fallback leg.
319
+ *
320
+ * @param messages - Ordered chat history.
321
+ * @param env - API key + gateway bindings.
322
+ * @param opts - Optional tier/model/parameters override.
323
+ * @param deps - Optional fetch/logger/clock injection (for testing).
324
+ * @returns A {@link FactoryResponse} carrying either an {@link LLMResult} or
325
+ * an error (`LLM_ALL_PROVIDERS_FAILED`, `LLM_RATE_LIMITED`, or `INTERNAL_ERROR`).
326
+ */
327
+ declare function complete(messages: LLMMessage[], env: LLMEnv, opts?: LLMOptions, deps?: LLMDeps): Promise<FactoryResponse<LLMResult>>;
328
+ /**
329
+ * Streams a completion from the primary Anthropic provider, yielding text chunks
330
+ * as they arrive. Falls back to the non-streaming {@link complete} function when
331
+ * the provider does not support streaming (i.e. a non-Anthropic primary is selected).
332
+ *
333
+ * The generator's **return value** (accessible via `gen.return()` or by consuming
334
+ * the full iteration) is an {@link LLMResult} with the same shape as {@link complete}.
335
+ *
336
+ * Usage pattern:
337
+ * ```ts
338
+ * const gen = completionStream(messages, env, opts);
339
+ * for await (const chunk of gen) {
340
+ * // stream chunk to client
341
+ * }
342
+ * const result = (await gen.return(undefined)).value; // LLMResult
343
+ * ```
344
+ *
345
+ * @param messages - Ordered chat history.
346
+ * @param env - API key + gateway bindings.
347
+ * @param opts - Optional tier/model/parameters override. Accepts `deps` as nested field.
348
+ * @returns An async generator that yields `string` chunks and returns an {@link LLMResult}.
349
+ */
350
+ declare function completionStream(messages: LLMMessage[], env: LLMEnv, opts?: LLMOptions & {
351
+ deps?: LLMDeps;
352
+ }): AsyncGenerator<string, LLMResult, unknown>;
353
+ /**
354
+ * Returns `true` if `response` contains at least one verbatim phrase of at
355
+ * least 5 consecutive whitespace-delimited tokens that also appears in one of
356
+ * the `sources` strings.
357
+ *
358
+ * Returns `true` unconditionally when `sources` is empty (no grounding
359
+ * documents means grounding cannot be violated).
360
+ *
361
+ * This is a lightweight guard for RAG pipelines — it detects obvious
362
+ * hallucinations where the model generates content not present in any
363
+ * retrieved source. It is NOT a semantic similarity check.
364
+ *
365
+ * @param response - The LLM-generated text to inspect.
366
+ * @param sources - Retrieved source documents to check against.
367
+ * @returns `true` if the response is grounded, `false` if hallucination detected.
368
+ *
369
+ * @example
370
+ * ```ts
371
+ * const grounded = assertGrounding(llmAnswer, retrievedDocs);
372
+ * if (!grounded) {
373
+ * // flag or re-rank the response
374
+ * }
375
+ * ```
376
+ */
377
+ declare function assertGrounding(response: string, sources: string[]): boolean;
378
+
379
+ export { BASE_BACKOFF_MS, type CostKvStore, type LLMContentBlock, type LLMDeps, type LLMEnv, type LLMMessage, type LLMOptions, type LLMProvider, type LLMRecordContext, type LLMRecordRow, type LLMResult, type LLMTier, type LLMTool, type LLMToolCall, MODELS, MODEL_PRICE_PER_1M, PROVIDER_COOLDOWN_MS, assertGrounding, clearProviderCooldown, complete, completionStream, isProviderCoolingDown, markProviderCoolingDown };