vern-llm 1.7.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -110,8 +110,79 @@ interface JsonSchemaSpec {
110
110
  }
111
111
 
112
112
  //#endregion
113
- //#region src/types/usage.d.ts
113
+ //#region src/types/tools.d.ts
114
114
  //# sourceMappingURL=schema.d.ts.map
115
+ /**
116
+ * Describes a capability the model may request, not the capability
117
+ * itself. VernLLM transports this to the provider and parses what comes
118
+ * back; it never executes anything.
119
+ */
120
+ interface ToolDefinition {
121
+ name: string;
122
+ description: string;
123
+ /** JSON Schema for the tool's input. */
124
+ parameters: Record<string, unknown>;
125
+ /**
126
+ * Optional client-side validator run on the parsed `arguments` before
127
+ * they're handed back to the caller, mirroring the `schema: SchemaLike<T>`
128
+ * pattern already used for response validation (see `types/schema.ts`).
129
+ * Reuses that zero-dependency, `safeParse`-compatible shape instead of
130
+ * requiring a JSON Schema validator (e.g. ajv) as a new dependency.
131
+ * Failed validation throws `LLMError('validation')`. If omitted, VernLLM
132
+ * parses arguments as JSON but does not validate them further.
133
+ */
134
+ argumentsSchema?: SchemaLike<unknown>;
135
+ }
136
+ /** A single tool invocation requested by the model. */
137
+ interface ToolCall {
138
+ id: string;
139
+ name: string;
140
+ /** Parsed JSON arguments (and validated, if `argumentsSchema` was set). */
141
+ arguments: unknown;
142
+ }
143
+ /** The application's result of executing a `ToolCall`, sent back to the model. */
144
+ interface ToolResult {
145
+ toolCallId: string;
146
+ content: unknown;
147
+ /**
148
+ * Signals a failed tool execution back to the model (matches Anthropic's
149
+ * native `is_error` on tool_result blocks). Only `fromAnthropic` honors
150
+ * this today, Gemini and Bedrock have no equivalent wire concept, so
151
+ * other adapters ignore it silently.
152
+ */
153
+ isError?: boolean;
154
+ }
155
+ /** `call()` result when `tools` was set and the model produced a normal answer. */
156
+ interface ContentResult<T> {
157
+ type: 'content';
158
+ content: T;
159
+ }
160
+ /** `call()` result when `tools` was set and the model requested one or more tools. */
161
+ interface ToolCallResult {
162
+ type: 'tool_calls';
163
+ toolCalls: ToolCall[];
164
+ /** Any text the model produced alongside the tool request, if present. */
165
+ content?: string;
166
+ }
167
+ type CallWithToolsResult<T> = ContentResult<T> | ToolCallResult;
168
+ /**
169
+ * Runtime-safe check for whether a `call()` result is a `tool_calls`
170
+ * result. Prefer this over relying on TypeScript's static narrowing
171
+ * whenever `params` passed to `call()` wasn't a literal with `tools`
172
+ * inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
173
+ * that case TS may have typed the result as plain `T` even though it's
174
+ * actually a `CallWithToolsResult<T>` at runtime, and this check works
175
+ * either way.
176
+ */
177
+ declare function isToolCallResult(result: unknown): result is ToolCallResult;
178
+ /** What the model should do about tools on a given call. */
179
+ type ToolChoice = 'auto' | 'none' | 'required' | {
180
+ name: string;
181
+ };
182
+
183
+ //#endregion
184
+ //#region src/types/usage.d.ts
185
+ //# sourceMappingURL=tools.d.ts.map
115
186
  type ReserveUsage = (params: {
116
187
  coalesced: boolean;
117
188
  signal?: AbortSignal;
@@ -144,15 +215,39 @@ interface TokenUsage {
144
215
  model: string;
145
216
  }
146
217
  type OnUsage = (usage: TokenUsage) => void;
218
+ /**
219
+ * Called when a provider response arrives but VernLLM's own post-processing
220
+ * then fails, after usage data was already present in that response. Covers
221
+ * any error thrown after usage extraction, not just parse/validation, since
222
+ * everything in that path only runs once a response, and real spend, has
223
+ * already arrived. Fires once per failed attempt with extractable usage,
224
+ * never for transport failures, where no response means no honest number
225
+ * to report.
226
+ */
227
+ type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
147
228
 
148
229
  //#endregion
149
230
  //#region src/types/call.d.ts
150
231
  //# sourceMappingURL=usage.d.ts.map
151
- /** A single prior turn in a multi-turn conversation, passed via `history`. */
152
- interface ConversationTurn {
153
- role: 'user' | 'assistant';
232
+ /**
233
+ * A single prior turn in a multi-turn conversation, passed via `history`.
234
+ *
235
+ * Supports normal user/assistant messages and tool continuations: an assistant
236
+ * turn may include `toolCalls`, and a tool turn carries the matching
237
+ * `toolResults`. A tool turn must immediately follow an assistant tool call
238
+ * turn, and every requested tool call must have a result.
239
+ */
240
+ type ConversationTurn = {
241
+ role: 'user';
154
242
  content: string;
155
- }
243
+ } | {
244
+ role: 'assistant';
245
+ content?: string;
246
+ toolCalls?: ToolCall[];
247
+ } | {
248
+ role: 'tool';
249
+ toolResults: ToolResult[];
250
+ };
156
251
  /** A plain text segment of a multimodal `userContent` array. */
157
252
  interface TextBlock {
158
253
  type: 'text';
@@ -180,15 +275,28 @@ interface CallParams<T = unknown> extends UsageHooks {
180
275
  /** Current user message, as text or multimodal content blocks. */
181
276
  userContent: string | ContentBlock[];
182
277
  /**
183
- * Previous conversation turns. Must alternate user/assistant and end with
184
- * an assistant turn; invalid history throws LLMError('validation').
278
+ * Previous conversation turns. Must alternate roles; tool turns must follow
279
+ * assistant tool calls. Invalid history throws LLMError('validation').
185
280
  */
186
281
  history?: ConversationTurn[];
187
- temperature?: number;
282
+ /**
283
+ * Generation temperature. Default 0.2, not the provider's own default.
284
+ * Pass `null` to omit `temperature` from the request entirely, so the
285
+ * provider applies its own default instead.
286
+ */
287
+ temperature?: number | null;
188
288
  jsonMode?: boolean;
189
289
  maxTokens?: number;
190
290
  requestId?: string;
191
291
  signal?: AbortSignal;
292
+ /**
293
+ * Per-call override for the instance's `chunkIdleTimeoutMs` (max gap
294
+ * between stream chunks once opened). Only applies when `stream: true`.
295
+ * Useful for routes using reasoning-heavy models with documented long
296
+ * silent gaps mid-stream. Pass 0 to disable the idle timeout for this
297
+ * call.
298
+ */
299
+ chunkIdleTimeoutMs?: number;
192
300
  /** Overrides the instance model for this call. */
193
301
  model?: string;
194
302
  /** Reasoning effort for supported reasoning models. */
@@ -202,17 +310,233 @@ interface CallParams<T = unknown> extends UsageHooks {
202
310
  * Implies jsonMode: true.
203
311
  */
204
312
  schema?: SchemaLike<T>;
313
+ /**
314
+ * Tools the model may call. When set, `call()` always returns a
315
+ * `CallWithToolsResult<T>` discriminated union instead of `T` directly
316
+ * (see `CallWithToolsResult`), a breaking-change point: omitting `tools`
317
+ * keeps `call()`'s old `Promise<T>` behavior exactly.
318
+ *
319
+ * Mutually exclusive with `jsonSchema`/`schema`: on Anthropic and
320
+ * Bedrock, `jsonSchema` is implemented internally as a forced single-tool
321
+ * call, which would collide with real tools. Setting both throws
322
+ * `LLMError('validation')`.
323
+ */
324
+ tools?: ToolDefinition[];
325
+ /** Defaults to `'auto'` when `tools` is set. */
326
+ toolChoice?: ToolChoice;
327
+ /**
328
+ * Streams the response incrementally instead of resolving once. Default:
329
+ * false. Requires a client/adapter that implements `createStream`.
330
+ * Retry/timeout/circuit-breaker guarantees apply only to opening the
331
+ * stream (through the first chunk); a failure after that point rejects
332
+ * `finalResult` directly and is not retried, since a mid-stream failure
333
+ * isn't connection-time evidence for the circuit breaker, the attempt
334
+ * already counted as a success once the first chunk arrived. Once the
335
+ * stream opens successfully, `finalResult` still resolves to the same
336
+ * validated `T`/`CallWithToolsResult<T>` shape `call()` would have
337
+ * returned for the same params with `stream` omitted. See
338
+ * `StreamCallResult`.
339
+ */
340
+ stream?: boolean;
205
341
  }
206
- interface CachedCallParams<T> extends UsageHooks {
342
+ /**
343
+ * A `CallParams` variant where tool calling is explicitly enabled.
344
+ *
345
+ * Requiring `tools` to be present allows TypeScript to select the
346
+ * tool-aware `call()` overload and return `CallWithToolsResult<T>` instead
347
+ * of the normal `T` response type.
348
+ */
349
+ type ToolEnabledCallParams<T> = CallParams<T> & {
350
+ tools: NonNullable<CallParams<T>['tools']>;
351
+ };
352
+ /** Shared cache-configuration fields, minus the internal `fn` primitive. */
353
+ interface CachedCallInput extends UsageHooks {
207
354
  cacheKey: string;
208
355
  ttl: number;
209
- fn: () => Promise<T>;
210
356
  signal?: AbortSignal;
211
357
  }
358
+ /**
359
+ * Parameters for a cached LLM call without tool calling.
360
+ *
361
+ * Combines the cache configuration with the `CallParams` passed to
362
+ * `VernLLM.call()`. The cached value is the normal LLM response type `T`.
363
+ */
364
+ type CachedCallParams<T> = CachedCallInput & {
365
+ call: CallParams<T>;
366
+ };
367
+ /**
368
+ * Parameters for a cached LLM call with tool calling enabled.
369
+ *
370
+ * The cached value includes the full `CallWithToolsResult<T>`, meaning
371
+ * tool requests and normal content responses are cached exactly as returned
372
+ * by the model.
373
+ */
374
+ type CachedToolCallParams<T> = CachedCallInput & {
375
+ call: ToolEnabledCallParams<T>;
376
+ };
212
377
 
213
378
  //#endregion
214
- //#region src/types/client.d.ts
379
+ //#region src/types/stream.d.ts
215
380
  //# sourceMappingURL=call.d.ts.map
381
+ /** One incremental unit of a streaming response, as delivered to the caller. */
382
+ type StreamChunk = {
383
+ type: 'text-delta';
384
+ delta: string;
385
+ } | {
386
+ type: 'tool_call_delta';
387
+ index: number;
388
+ id?: string;
389
+ name?: string;
390
+ argsDelta?: string;
391
+ /**
392
+ * True when `argsDelta` is the whole set of arguments, not a
393
+ * fragment. Set for Gemini (its API returns function-call args
394
+ * whole in one chunk) and for cache/replay chunks, which are
395
+ * one-shot too. Omitted or `false` for a genuine fragment from
396
+ * providers that do stream incrementally (OpenAI-compatible,
397
+ * Anthropic, Bedrock).
398
+ */
399
+ complete?: boolean;
400
+ } | {
401
+ type: 'usage';
402
+ usage: TokenUsage;
403
+ };
404
+ /**
405
+ * What `call()` returns when `stream: true`. `chunks` is for live rendering;
406
+ * `finalResult` resolves to the same validated `T`/`CallWithToolsResult<T>`
407
+ * shape `call()` would have returned had `stream` been omitted, once the
408
+ * stream completes successfully.
409
+ *
410
+ * `chunks` is single-use and supports only one consumer: iterating it more
411
+ * than once, or from more than one place concurrently, shares the same
412
+ * underlying buffered stream rather than replaying or forking it, which can
413
+ * split chunks unpredictably between consumers. Stopping iteration early
414
+ * (e.g. `break`ing out of a `for await`) does not cancel or otherwise
415
+ * signal the underlying stream, the background pump keeps running to
416
+ * completion regardless, buffering any chunks emitted after that point, so
417
+ * `finalResult` still settles normally even if `chunks` is abandoned or
418
+ * never read at all.
419
+ *
420
+ * Unread chunks are buffered internally for the duration of one stream,
421
+ * this is what lets a caller start iterating `chunks` after the stream has
422
+ * already progressed (or finished) and still see everything. That backlog
423
+ * is capped: an unusually large stream whose `chunks` is never read at all
424
+ * has its oldest buffered chunks dropped once the backlog grows past
425
+ * roughly twice a fixed internal limit, trimmed back down to that limit in
426
+ * one batch rather than one-at-a-time, bounding both peak memory and the
427
+ * eviction work itself for that pathological case instead of the array
428
+ * growing (or being trimmed) proportional to the whole stream's output.
429
+ * Ordinary consumption, even started somewhat late, stays far under the
430
+ * limit and is unaffected.
431
+ */
432
+ interface StreamCallResult<R> {
433
+ chunks: AsyncIterable<StreamChunk>;
434
+ finalResult: Promise<R>;
435
+ }
436
+ /**
437
+ * A `CallParams` variant where streaming is explicitly enabled.
438
+ *
439
+ * Requiring `stream: true` to be statically present allows TypeScript to
440
+ * select the streaming `call()` overload and return `StreamCallResult<...>`
441
+ * instead of the normal, single-shot response type.
442
+ */
443
+ type StreamEnabledCallParams<T> = CallParams<T> & {
444
+ stream: true;
445
+ };
446
+ /**
447
+ * The adapter-facing, pre-normalization shape a `createStream` client
448
+ * implementation emits, analogous to how `WireMessage`/`WireToolCall`
449
+ * already sit between `CallParams` and each provider's own wire format.
450
+ */
451
+ type WireStreamChunk = {
452
+ type: 'text-delta';
453
+ delta: string;
454
+ } | {
455
+ type: 'tool_call_delta';
456
+ index: number;
457
+ id?: string;
458
+ name?: string;
459
+ argumentsDelta?: string;
460
+ /** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
461
+ complete?: boolean;
462
+ } | {
463
+ type: 'usage';
464
+ usage: {
465
+ prompt_tokens?: number;
466
+ completion_tokens?: number;
467
+ total_tokens?: number;
468
+ };
469
+ } | {
470
+ /**
471
+ * A provider keep-alive signal with no content of its own (e.g.
472
+ * Anthropic's `ping` events, an SSE comment-line heartbeat).
473
+ * Adapters yield this so the stream loop resets its idle timeout.
474
+ * Never surfaced to callers as a `StreamChunk`.
475
+ */
476
+ type: 'ping';
477
+ };
478
+ /**
479
+ * Parameters for a cached, streaming LLM call without tool calling.
480
+ *
481
+ * The cached value is `T`, same as `CachedCallParams<T>`, but a miss
482
+ * relays live `chunks` to the caller while the result is being generated,
483
+ * and a hit synthesizes a one-shot `chunks` replay from the cached value
484
+ * (see `VernLLM.cachedCall`'s docs for exactly what that replay looks
485
+ * like).
486
+ */
487
+ type CachedStreamCallParams<T> = CachedCallInput & {
488
+ call: StreamEnabledCallParams<T>;
489
+ };
490
+ /**
491
+ * Parameters for a cached, streaming LLM call with tool calling enabled.
492
+ *
493
+ * The cached value is the full `CallWithToolsResult<T>`, same as
494
+ * `CachedToolCallParams<T>`, with the same live-chunks-on-miss,
495
+ * replayed-chunks-on-hit behavior as `CachedStreamCallParams<T>`.
496
+ */
497
+ type CachedStreamToolCallParams<T> = CachedCallInput & {
498
+ call: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>;
499
+ };
500
+
501
+ //#endregion
502
+ //#region src/types/client.d.ts
503
+ //# sourceMappingURL=stream.d.ts.map
504
+ /** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
505
+ interface WireToolCall {
506
+ id: string;
507
+ type: 'function';
508
+ function: {
509
+ name: string;
510
+ /** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
511
+ arguments: string;
512
+ };
513
+ }
514
+ /** One entry of `LLMClient`'s `messages` array, named so callers building it can annotate against it. */
515
+ type WireMessage = {
516
+ role: 'system';
517
+ content: string;
518
+ } | {
519
+ role: 'user';
520
+ content: string | ContentBlock[];
521
+ } | {
522
+ role: 'assistant';
523
+ /** Optional: an assistant turn that only requested tools has no text. */
524
+ content?: string;
525
+ tool_calls?: WireToolCall[];
526
+ } | {
527
+ role: 'tool';
528
+ tool_call_id: string;
529
+ content: string;
530
+ /** Only honored by `fromAnthropic` today (maps to `tool_result.is_error`); other adapters ignore it. */
531
+ is_error?: boolean;
532
+ };
533
+ /** The OpenAI-shaped wire `tool_choice`. */
534
+ type WireToolChoice = 'auto' | 'none' | 'required' | {
535
+ type: 'function';
536
+ function: {
537
+ name: string;
538
+ };
539
+ };
216
540
  /**
217
541
  * Minimal shape compatible with the OpenAI SDKs chat.completions.create,
218
542
  * so consumers can pass an OpenAI client directly
@@ -226,7 +550,7 @@ interface LLMClient {
226
550
  completions: {
227
551
  create(params: {
228
552
  model: string;
229
- temperature: number;
553
+ temperature?: number;
230
554
  max_tokens: number;
231
555
  response_format?: {
232
556
  type: 'json_object';
@@ -241,19 +565,29 @@ interface LLMClient {
241
565
  };
242
566
  /** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
243
567
  reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
244
- messages: Array<{
245
- role: 'system' | 'assistant';
246
- content: string;
247
- } | {
248
- role: 'user';
249
- content: string | ContentBlock[];
568
+ /** Tools the model may call, OpenAI's `function`-wrapped shape. */
569
+ tools?: Array<{
570
+ type: 'function';
571
+ function: {
572
+ name: string;
573
+ description: string;
574
+ parameters: Record<string, unknown>;
575
+ };
250
576
  }>;
577
+ tool_choice?: WireToolChoice;
578
+ /**
579
+ * Wire-format messages. Breaking change for custom adapters:
580
+ * implementations must handle tool messages and assistant tool_calls.
581
+ * Exhaustive switches over only system/user/assistant roles may no longer compile.
582
+ */
583
+ messages: WireMessage[];
251
584
  }, options: {
252
585
  signal: AbortSignal;
253
586
  }): Promise<{
254
587
  choices?: Array<{
255
588
  message?: {
256
589
  content?: string | null;
590
+ tool_calls?: WireToolCall[];
257
591
  };
258
592
  }>;
259
593
  usage?: {
@@ -262,6 +596,15 @@ interface LLMClient {
262
596
  total_tokens?: number;
263
597
  };
264
598
  }>;
599
+ /**
600
+ * Optional. Required only for `stream: true` calls. Adapters/clients
601
+ * that don't implement this make `stream: true` throw a clear
602
+ * `LLMError('validation')` rather than a confusing runtime failure.
603
+ * Takes the same request shape as `create`, minus the response type.
604
+ */
605
+ createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
606
+ signal: AbortSignal;
607
+ }): AsyncIterable<WireStreamChunk>;
265
608
  };
266
609
  };
267
610
  }
@@ -337,10 +680,26 @@ interface VernLLMOptions {
337
680
  maxRetries?: number;
338
681
  /** Per-attempt timeout in ms. Default 25000 */
339
682
  timeoutMs?: number;
683
+ /**
684
+ * For `stream: true` calls: max gap allowed between chunks once the
685
+ * stream has opened, in ms. Resets on every chunk, including keep-alive
686
+ * pings. `timeoutMs` only covers opening the stream and its first
687
+ * chunk; this covers every gap after that. Also counts as a
688
+ * circuit-breaker failure, unlike other mid-stream errors, since a
689
+ * provider that streams one chunk then stalls should still trip it.
690
+ * Default 30000. Pass 0 or negative to disable.
691
+ */
692
+ chunkIdleTimeoutMs?: number;
340
693
  /** Base delay for exponential backoff in ms. Default 500 */
341
694
  baseDelayMs?: number;
342
695
  /** Default max_tokens for calls that don't override it. Default 1000 */
343
696
  defaultMaxTokens?: number;
697
+ /**
698
+ * Default temperature for calls that don't override it. Default 0.2, not
699
+ * the provider's own default. Pass `null` to omit `temperature` from the
700
+ * request entirely, so the provider applies its own default instead.
701
+ */
702
+ defaultTemperature?: number | null;
344
703
  /** Enables debug logging of raw model output (logs up to 800 chars of each
345
704
  * response). Off by default */
346
705
  debug?: boolean;
@@ -352,6 +711,18 @@ interface VernLLMOptions {
352
711
  parseJson?: (content: string) => unknown;
353
712
  /** Called after every successful call with token usage, if the provider reports it */
354
713
  onUsage?: OnUsage;
714
+ /**
715
+ * Called when a provider response arrives but VernLLM's own post-processing
716
+ * then fails, after usage data was already present in that response.
717
+ * Separate from `onUsage`, which only fires on full success.
718
+ *
719
+ * For non-streaming calls, never fires for transport failures (timeout,
720
+ * network error, non-retryable status), since no response means no usage
721
+ * to report. For streaming calls, this is not guaranteed: a stream can
722
+ * deliver a usage chunk and then fail later (e.g. an idle timeout waiting
723
+ * for the final close), in which case this does fire.
724
+ */
725
+ onUsageFailure?: OnUsageFailure;
355
726
  /** Injectable logger. Defaults to a console-based logger gated by `debug` */
356
727
  logger?: Logger;
357
728
  /**
@@ -366,31 +737,34 @@ interface VernLLMOptions {
366
737
  //#region src/vernLLM.d.ts
367
738
  //# sourceMappingURL=options.d.ts.map
368
739
  /**
369
- * A resilient layer around an LLM chat completions client, this is VernLLM!
740
+ * A resilient layer around an LLM chat completions client. This is VernLLM!
370
741
  *
371
- * Adds retry with backoff/jitter, per-attempt timeouts, an optional circuit breaker,
372
- * JSON parsing with optional schema validation, usage tracking, and an
373
- * optional response cache, all configurable, all opt-in beyond sensible
374
- * defaults.
742
+ * Adds retry with backoff and jitter, per-attempt timeouts, an optional
743
+ * circuit breaker, JSON parsing with optional schema validation, usage
744
+ * tracking, and an optional response cache. All configurable, all opt-in
745
+ * beyond sensible defaults.
375
746
  */
376
747
  declare class VernLLM {
377
748
  private readonly client;
378
749
  private readonly model;
379
750
  private readonly maxRetries;
380
751
  private readonly timeoutMs;
752
+ private readonly chunkIdleTimeoutMs;
381
753
  private readonly baseDelayMs;
382
754
  private readonly defaultMaxTokens;
755
+ private readonly defaultTemperature;
383
756
  private readonly cache;
384
757
  private readonly nonRetryableStatus;
385
758
  private readonly inFlight;
386
759
  private readonly parseJson;
387
760
  private readonly onUsage?;
761
+ private readonly onUsageFailure?;
388
762
  private readonly logger;
389
763
  private readonly breaker?;
390
764
  /**
391
- * @param options - Client, model, and tunables. Notable defaults:
392
- * `maxRetries` 1, `timeoutMs` 25000, `baseDelayMs` 500 (exponential backoff
393
- * base), `defaultMaxTokens` 1000, `cache` an in-memory adapter,
765
+ * @param options Client, model, and tunables. Defaults: `maxRetries` 1,
766
+ * `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
767
+ * `defaultTemperature` 0.2, `cache` an in-memory adapter,
394
768
  * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
395
769
  */
396
770
  constructor(options: VernLLMOptions);
@@ -399,23 +773,103 @@ declare class VernLLM {
399
773
  /**
400
774
  * Makes a single logical LLM call, retrying on failure per the configured
401
775
  * policy. Fails fast if the breaker is open or the signal is already
402
- * aborted. On exhausting retries, records a breaker failure and rejects
403
- * with a normalized LLMError.
776
+ * aborted. Rejects with a normalized LLMError on exhausted retries.
777
+ *
778
+ * When `tools` is set, returns a `CallWithToolsResult<T>` instead of `T`:
779
+ * `{ type: 'content', content }` or `{ type: 'tool_calls', toolCalls,
780
+ * content? }`. VernLLM never executes tools; run them yourself and
781
+ * continue via `history` (see `ConversationTurn`). Mutually exclusive
782
+ * with `jsonSchema`/`schema`.
783
+ *
784
+ * TypeScript only picks the tools-aware overload when `tools` is
785
+ * statically present on `params`. If set conditionally on a plain
786
+ * `CallParams<T>`, use `isToolCallResult()` to check the shape at
787
+ * runtime instead. See the Tool Calling docs for details.
404
788
  *
405
- * @param params - System/user content plus per-call overrides (model,
406
- * temperature, jsonMode, schema, signal, etc). See `CallParams`.
407
- * @returns The parsed (and optionally schema-validated) response, or the
408
- * raw string content when `jsonMode` is false and no `jsonSchema` is set.
789
+ * The same static-vs-dynamic caveat applies to `stream`: TypeScript only
790
+ * selects the streaming overload (returning `StreamCallResult<...>`) when
791
+ * `stream: true` is statically present on `params`. A `stream` value set
792
+ * conditionally on a plain `CallParams<T>` still resolves to `Promise<T>`
793
+ * (or `Promise<CallWithToolsResult<T>>`) at the type level even though
794
+ * the actual runtime result is the `{ chunks, finalResult }` streaming
795
+ * shape whenever `stream` evaluates to `true`, callers doing this should
796
+ * narrow/cast accordingly rather than relying on the static return type.
797
+ *
798
+ * @param params System/user content plus per-call overrides. See `CallParams`.
799
+ * @returns Without `tools` or `stream`: the parsed response, or raw
800
+ * string if `jsonMode` is false. With `tools`: a `CallWithToolsResult<T>`.
801
+ * With `stream: true` (statically): a `{ chunks, finalResult }`
802
+ * `StreamCallResult`, `finalResult` resolving to whichever of the above
803
+ * shapes applies once the stream completes. See `StreamCallResult`.
409
804
  */
805
+ call<T = unknown>(params: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
806
+ call<T = unknown>(params: StreamEnabledCallParams<T>): Promise<StreamCallResult<T>>;
807
+ call<T = unknown>(params: ToolEnabledCallParams<T>): Promise<CallWithToolsResult<T>>;
410
808
  call<T = unknown>(params: CallParams<T>): Promise<T>;
411
- /** Runs `fn`, retrying with backoff according to `shouldRetry`. */
412
- private retryWithBackoff;
413
809
  /**
414
- * Performs a single attempt: builds the request, dispatches it with a
415
- * timeout, and shapes the response. Throws on an empty response so the
810
+ * Performs a single attempt: builds the request (translating `tools` to
811
+ * wire shape when present), dispatches it with a timeout, and shapes the
812
+ * response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
813
+ * set. Throws on an empty response (no text and no tool_calls) so the
416
814
  * retry loop treats it like any other transient failure.
417
815
  */
418
816
  private executeCall;
817
+ /**
818
+ * Shapes a fully-arrived response (content and/or tool_calls, already
819
+ * extracted from the provider's payload) into `T` or a
820
+ * `CallWithToolsResult<T>`. Reused by the streaming path once it has
821
+ * buffered the full text/tool-call deltas, so there's no separate
822
+ * parsing/validation logic for streaming.
823
+ *
824
+ * Normalizes and reports usage failure on error itself, so every caller
825
+ * gets identical error handling without duplicating it.
826
+ */
827
+ private finalizeResponse;
828
+ /**
829
+ * Opens a stream for a single attempt: builds the request exactly like
830
+ * `executeCall`, then requires `createStream` on the client (a clear
831
+ * `validation` error if the adapter doesn't support it). The timeout
832
+ * wraps stream construction and the first `.next()` together, not just
833
+ * construction: calling an `async function*` returns an iterator
834
+ * synchronously without running its body until `.next()` is first
835
+ * invoked, so timing only construction would time an operation that's
836
+ * always instant, not the actual connection. Both are folded into a
837
+ * single `withTimeout` so the same abort signal reaches whatever the
838
+ * adapter's `createStream` uses internally for its first network
839
+ * round-trip.
840
+ *
841
+ * Circuit-breaker success is recorded once the stream fully completes,
842
+ * not on the first chunk arriving, so a connection that opens but then
843
+ * dies mid-stream isn't masked as a success (see `buildStreamResult`).
844
+ */
845
+ private executeStreamCall;
846
+ /**
847
+ * The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
848
+ * an async generator that yields translated `StreamChunk`s to the caller
849
+ * live, as they arrive, with no per-chunk timeout and no bound on total
850
+ * duration, and accumulates text/tool-call deltas internally so that
851
+ * `finalizeResponse` can produce `finalResult` once the stream completes.
852
+ *
853
+ * Two separate try/catches: the iteration loop's catch handles errors
854
+ * the transport itself throws, which aren't normalized yet, so that
855
+ * happens here along with the one `reportUsageFailure` call for them.
856
+ * The second catch, around `finalizeResponse`, does not re-normalize or
857
+ * re-report since `finalizeResponse` already does both internally.
858
+ * Circuit-breaker success is only recorded once the stream fully
859
+ * completes, not when the first chunk arrives, so a connection that
860
+ * opens and then dies mid-way still counts as a failure below instead
861
+ * of masking it.
862
+ */
863
+ private buildStreamResult;
864
+ /**
865
+ * Checks every `ToolCall` against the `tools` that were offered, catching
866
+ * a hallucinated tool name early instead of letting it reach the
867
+ * application's dispatch table. Then runs each tool's `argumentsSchema`,
868
+ * if present, throwing `LLMError('validation')` on failure.
869
+ */
870
+ private validateToolCallArguments;
871
+ /** Runs `fn`, retrying with backoff according to `shouldRetry`. */
872
+ private retryWithBackoff;
419
873
  /**
420
874
  * Validates `history` alternates user/assistant turns, since providers
421
875
  * like Anthropic/Gemini reject or mishandle consecutive same-role turns.
@@ -423,6 +877,16 @@ declare class VernLLM {
423
877
  private validateHistory;
424
878
  /** Applies per-call defaults and shapes params into the client's request object. */
425
879
  private buildRequestPayload;
880
+ /** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
881
+ private buildWireToolChoice;
882
+ /**
883
+ * Expands one `ConversationTurn` into one or more wire messages. Plain
884
+ * user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
885
+ * to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
886
+ * turn expands into one wire `tool` message per `toolResult`, since
887
+ * OpenAI-shaped wire format wants one message per tool_call_id.
888
+ */
889
+ private turnToWireMessages;
426
890
  /**
427
891
  * Chooses the response format: a provider-native `jsonSchema` takes
428
892
  * priority when supplied (constrains generation directly), otherwise
@@ -430,8 +894,23 @@ declare class VernLLM {
430
894
  * requested, or no format at all for plain text responses.
431
895
  */
432
896
  private buildResponseFormat;
433
- /** Reports token usage to `onUsage`, swallowing and logging any error it throws. */
434
- private recordUsage;
897
+ /**
898
+ * Pulls `TokenUsage` out of a raw response, if the provider reported it.
899
+ * Extraction doesn't depend on what happens to the response afterward, so
900
+ * a malformed body can still yield usage if the provider's usage block
901
+ * itself came through intact.
902
+ */
903
+ private extractUsage;
904
+ /** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
905
+ private reportUsage;
906
+ /**
907
+ * Reports token usage spent on an attempt that then failed, so it isn't
908
+ * dropped alongside the error. Covers any error thrown after usage
909
+ * extraction, since all of them happen only after a response (real
910
+ * spend) already arrived. Swallows and logs any error `onUsageFailure`
911
+ * itself throws.
912
+ */
913
+ private reportUsageFailure;
435
914
  /** Parses response content as JSON and validates it against `schema` when supplied. */
436
915
  private parseAndValidate;
437
916
  /**
@@ -448,38 +927,97 @@ declare class VernLLM {
448
927
  * supports deletion. Cache invalidation is the caller's responsibility;
449
928
  * only the application knows when cached data is stale.
450
929
  *
451
- * @param key - The raw cache key (resolved through the adapter's
930
+ * @param key The raw cache key (resolved through the adapter's
452
931
  * `resolveKey`, if any, before deletion).
453
932
  */
454
933
  deleteCache(key: string): Promise<void>;
455
934
  /**
456
- * Cache wrapper around caller-supplied logic. Concurrent misses for the
457
- * same `cacheKey` share a single in-flight call, avoiding cache stampedes.
935
+ * Internal cache primitive around caller-supplied logic. Concurrent misses
936
+ * for the same `cacheKey` share a single in-flight call, avoiding cache
937
+ * stampedes.
938
+ *
939
+ * Not part of the public API. Backs the public `cachedCall()`, which
940
+ * always composes this with `call()` so cached results get the same
941
+ * retry/timeout/circuit-breaker guarantees as any other LLM call.
458
942
  *
459
- * @param params - `cacheKey`, `ttl`, `fn` (the work to run on a cache
943
+ * @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
460
944
  * miss, typically `() => this.call(...)`), and optional
461
- * `reserveUsage`/`refundUsage`/`signal`. See `CachedCallParams`.
945
+ * `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
462
946
  * @returns The cached value on a hit, or the result of `fn()` on a miss.
463
947
  */
464
- cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
948
+ private runCached;
465
949
  /** Starts the shared fn() call for a cache miss and tracks it in the in-flight map until it settles. */
466
950
  private registerTrigger;
467
951
  /** Runs `fn` and writes its result to the cache. */
468
952
  private runAndCache;
953
+ /**
954
+ * Streaming counterpart to `runCached`. Three cases:
955
+ *
956
+ * - Hit: no live generation to relay. Returns immediately with
957
+ * `finalResult` resolved to the cached value and a one-shot `chunks`
958
+ * replay built from it, so `for await (const c of chunks)` call sites
959
+ * work identically on a hit or a miss. No usage hooks fire, since
960
+ * nothing was actually spent.
961
+ * - Miss, nothing else in flight for this key: delegates to
962
+ * `registerStreamTrigger`, which opens the stream and relays its
963
+ * `chunks` live.
964
+ * - Miss, but another call for the same key is already in flight: this
965
+ * call has no live chunks of its own to relay, so it's treated like a
966
+ * delayed hit. `finalResult` shares the trigger's in-flight promise
967
+ * (the same `this.inFlight` map non-streaming `runCached` uses, so
968
+ * streaming and non-streaming `cachedCall`s for the same key coalesce
969
+ * against each other too), and `chunks` is a one-shot replay built
970
+ * once that promise resolves.
971
+ */
972
+ private runCachedStream;
973
+ /**
974
+ * Opens the shared stream for a cache miss and tracks its settled value
975
+ * in `this.inFlight` until it resolves or rejects. Writes to the cache
976
+ * on success only, matching `runAndCache`.
977
+ *
978
+ * Registers the in-flight promise synchronously, before anything async
979
+ * runs, so a concurrent `cachedCall` for the same key always sees it in
980
+ * time to join instead of triggering its own stream. Settlement is
981
+ * wired onto the whole `withReservedUsageForStream` call rather than a
982
+ * line inside its callback, so any failure point (reserving usage,
983
+ * opening the stream, or the stream itself) reliably settles the
984
+ * in-flight entry instead of leaving it stuck.
985
+ */
986
+ private registerStreamTrigger;
469
987
  /** Logs a failed refundUsage attempt via the configured logger. */
470
988
  private logRefundError;
471
989
  /**
472
- * Convenience wrapper composing `call` + `cachedCall`, so cached LLM calls
990
+ * Cache wrapper composing `call` + caching, so cached LLM calls
473
991
  * automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
474
- * `refundUsage` are read from the top-level params only.
992
+ * `refundUsage` are read from the top-level params only. Concurrent misses
993
+ * for the same `cacheKey` share a single in-flight call, avoiding cache
994
+ * stampedes. Supports `stream: true` and `tools` in any combination.
995
+ *
996
+ * When `call.tools` is set, this caches the whole `CallWithToolsResult`,
997
+ * including `tool_calls` results, not just final answers. Whether
998
+ * that's appropriate depends on the tool: caching "the model decided to
999
+ * call get_weather" is usually fine to reuse briefly, but caching a
1000
+ * decision made under permissions or account state that can change
1001
+ * between calls is not. Use a short `ttl` or a separate `cacheKey` for
1002
+ * such tools if this distinction matters.
1003
+ *
1004
+ * There is no public way to cache an arbitrary non-LLM function through
1005
+ * `VernLLM`. This method always composes with `call()`. For
1006
+ * general-purpose caching unrelated to an LLM call, use a dedicated
1007
+ * caching library at the application level instead.
475
1008
  *
476
- * @param params - `cachedCall` params (`cacheKey`, `ttl`, etc, minus `fn`)
477
- * plus `call`, the `CallParams` to pass through to `this.call(...)`.
1009
+ * @param params `cacheKey`, `ttl`, and optional
1010
+ * `reserveUsage`/`refundUsage`/`signal`, plus `call`, the `CallParams`
1011
+ * (optionally with `tools` and/or `stream`) to pass through to
1012
+ * `this.call(...)`. The top-level `signal` governs the cached operation
1013
+ * and its usage hooks only; to also abort the underlying provider
1014
+ * request, set `signal` inside `call`.
478
1015
  * @returns The cached value on a hit, or the freshly-called result on a miss.
479
1016
  */
480
- cachedLLMCall<T>(params: Omit<CachedCallParams<T>, 'fn'> & {
481
- call: CallParams<T>;
482
- }): Promise<T>;
1017
+ cachedCall<T>(params: CachedStreamToolCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
1018
+ cachedCall<T>(params: CachedStreamCallParams<T>): Promise<StreamCallResult<T>>;
1019
+ cachedCall<T>(params: CachedToolCallParams<T>): Promise<CallWithToolsResult<T>>;
1020
+ cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
483
1021
  /**
484
1022
  * @returns The current circuit breaker state (`'closed' | 'open' |
485
1023
  * 'half-open'`), or undefined if no circuit breaker was configured.
@@ -488,8 +1026,64 @@ declare class VernLLM {
488
1026
  }
489
1027
 
490
1028
  //#endregion
491
- //#region src/adapters/anthropic.d.ts
1029
+ //#region src/internal/sse.d.ts
492
1030
  //# sourceMappingURL=vernLLM.d.ts.map
1031
+ /**
1032
+ * Parses a Server-Sent-Events byte/text stream into the JSON payload of
1033
+ * each `data:` frame, in arrival order. Generic over transport: works with
1034
+ * anything that hands back progressively-arriving `Uint8Array` or `string`
1035
+ * chunks via async iteration: native `fetch`'s `response.body` (wrapped
1036
+ * to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
1037
+ * Node `Readable` (already async-iterable, no wrapping needed), etc, so
1038
+ * this framing layer doesn't care which transport produced the bytes.
1039
+ *
1040
+ * Follows the SSE spec's frame-delimiting rules closely enough for LLM
1041
+ * streaming responses: frames are separated by a blank line, each frame
1042
+ * may carry one or more `data:` lines (joined with `\n` per spec when
1043
+ * there's more than one), `:`-prefixed lines are comments and ignored, and
1044
+ * other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
1045
+ * only needs the payload. A frame whose data is exactly `[DONE]` (the
1046
+ * sentinel several providers, notably OpenAI, send to mark stream end)
1047
+ * ends iteration without yielding it.
1048
+ *
1049
+ * Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
1050
+ * to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
1051
+ * alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
1052
+ * stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
1053
+ * lines.
1054
+ *
1055
+ * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
1056
+ * how malformed JSON is handled elsewhere in VernLLM.
1057
+ */
1058
+ declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
1059
+ /**
1060
+ * Sentinel yielded by `parseSseStream` for a comment-only frame (no
1061
+ * `data:` payload), the mechanism providers use for SSE keep-alive
1062
+ * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
1063
+ * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
1064
+ */
1065
+ declare const SSE_PING: unique symbol;
1066
+
1067
+ //#endregion
1068
+ //#region src/internal/imageFormat.d.ts
1069
+ //# sourceMappingURL=sse.d.ts.map
1070
+ /**
1071
+ * MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
1072
+ * the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
1073
+ * Converse all natively support, so a `ContentBlock[]` that validates for
1074
+ * one provider validates for all of them.
1075
+ */
1076
+ declare const SUPPORTED_IMAGE_MIME_TYPES: readonly ["image/png", "image/jpeg", "image/gif", "image/webp"];
1077
+ type SupportedImageMimeType = (typeof SUPPORTED_IMAGE_MIME_TYPES)[number];
1078
+
1079
+ //#endregion
1080
+ //#region src/adapters/anthropic.d.ts
1081
+ /**
1082
+ * Validates an `ImageBlock.mimeType` against the shared supported set.
1083
+ * Throws a non-retryable `LLMError('validation')`, since an unsupported
1084
+ * mimeType is a permanent failure, retrying the same input can't fix it,
1085
+ * the same way a schema-validation or JSON-parse failure isn't retried.
1086
+ */
493
1087
  /** Anthropic's native per-block content shape for a message. */
494
1088
  type AnthropicContentBlock = {
495
1089
  type: 'text';
@@ -498,9 +1092,19 @@ type AnthropicContentBlock = {
498
1092
  type: 'image';
499
1093
  source: {
500
1094
  type: 'base64';
501
- media_type: string;
1095
+ media_type: SupportedImageMimeType;
502
1096
  data: string;
503
1097
  };
1098
+ } | {
1099
+ type: 'tool_use';
1100
+ id: string;
1101
+ name: string;
1102
+ input: unknown;
1103
+ } | {
1104
+ type: 'tool_result';
1105
+ tool_use_id: string;
1106
+ content: string;
1107
+ is_error?: boolean;
504
1108
  };
505
1109
  /** Minimal structural type for the Anthropic SDK's `messages.create` */
506
1110
  interface AnthropicClient {
@@ -517,10 +1121,19 @@ interface AnthropicClient {
517
1121
  tools?: Array<{
518
1122
  name: string;
519
1123
  description?: string;
520
- input_schema: Record<string, unknown>;
1124
+ input_schema: {
1125
+ type: 'object';
1126
+ [key: string]: unknown;
1127
+ };
521
1128
  strict?: boolean;
522
1129
  }>;
523
1130
  tool_choice?: {
1131
+ type: 'auto';
1132
+ } | {
1133
+ type: 'any';
1134
+ } | {
1135
+ type: 'none';
1136
+ } | {
524
1137
  type: 'tool';
525
1138
  name: string;
526
1139
  };
@@ -530,6 +1143,7 @@ interface AnthropicClient {
530
1143
  content: Array<{
531
1144
  type: string;
532
1145
  text?: string;
1146
+ id?: string;
533
1147
  name?: string;
534
1148
  input?: unknown;
535
1149
  }>;
@@ -566,13 +1180,30 @@ type GeminiPart = {
566
1180
  mimeType: string;
567
1181
  data: string;
568
1182
  };
1183
+ } | {
1184
+ functionCall: {
1185
+ name: string;
1186
+ args: unknown;
1187
+ };
1188
+ } | {
1189
+ functionResponse: {
1190
+ name: string;
1191
+ response: unknown;
1192
+ };
569
1193
  };
570
1194
  /**
571
- * Minimal structural type for VernLLM's two-argument wrapper around Gemini
572
- * `generateContent`. The wrapper exposes a request shape aligned with the
573
- * adapter interface, with top-level `systemInstruction` and
574
- * `generationConfig` fields, while transport options (such as `AbortSignal`)
575
- * are passed separately as the second argument.
1195
+ * Structural type matching the real `@google/genai` SDK's `ai.models`
1196
+ * object: `generateContent`/`generateContentStream` both take a single
1197
+ * `{ model, contents, config }` argument (config carries
1198
+ * `systemInstruction`, `tools`, `toolConfig`, generation settings, and
1199
+ * `abortSignal` all together), matching the real SDK closely enough that
1200
+ * `fromGemini(ai.models)` works directly, e.g:
1201
+ *
1202
+ * ```ts
1203
+ * import { GoogleGenAI } from '@google/genai';
1204
+ * const ai = new GoogleGenAI({ apiKey: '...' });
1205
+ * const llm = new VernLLM({ client: fromGemini(ai.models), model: 'gemini-2.5-flash' });
1206
+ * ```
576
1207
  */
577
1208
  interface GeminiClient {
578
1209
  generateContent(params: {
@@ -581,24 +1212,40 @@ interface GeminiClient {
581
1212
  role: 'user' | 'model';
582
1213
  parts: GeminiPart[];
583
1214
  }>;
584
- systemInstruction?: {
585
- parts: Array<{
586
- text: string;
587
- }>;
588
- };
589
- generationConfig?: {
1215
+ config?: {
1216
+ systemInstruction?: {
1217
+ parts: Array<{
1218
+ text: string;
1219
+ }>;
1220
+ };
590
1221
  temperature?: number;
591
1222
  maxOutputTokens?: number;
592
1223
  responseMimeType?: string;
593
1224
  responseSchema?: Record<string, unknown>;
1225
+ tools?: Array<{
1226
+ functionDeclarations: Array<{
1227
+ name: string;
1228
+ description?: string;
1229
+ parameters: Record<string, unknown>;
1230
+ }>;
1231
+ }>;
1232
+ toolConfig?: {
1233
+ functionCallingConfig: {
1234
+ mode: 'AUTO' | 'ANY' | 'NONE';
1235
+ allowedFunctionNames?: string[];
1236
+ };
1237
+ };
1238
+ abortSignal?: AbortSignal;
594
1239
  };
595
- }, options: {
596
- signal: AbortSignal;
597
1240
  }): Promise<{
598
1241
  candidates?: Array<{
599
1242
  content?: {
600
1243
  parts?: Array<{
601
1244
  text?: string;
1245
+ functionCall?: {
1246
+ name: string;
1247
+ args: unknown;
1248
+ };
602
1249
  }>;
603
1250
  };
604
1251
  }>;
@@ -608,6 +1255,32 @@ interface GeminiClient {
608
1255
  totalTokenCount?: number;
609
1256
  };
610
1257
  }>;
1258
+ /**
1259
+ * Optional. Required only for `stream: true` calls. Takes the same
1260
+ * request shape as `generateContent`. Matching the real SDK's own
1261
+ * `generateContentStream`, this resolves to an `AsyncIterable` (rather
1262
+ * than returning one synchronously) of partial responses, each chunk
1263
+ * holding the same `candidates[].content.parts[]` structure as
1264
+ * `generateContent`'s response, just incremental.
1265
+ */
1266
+ generateContentStream?(params: Parameters<GeminiClient['generateContent']>[0]): Promise<AsyncIterable<{
1267
+ candidates?: Array<{
1268
+ content?: {
1269
+ parts?: Array<{
1270
+ text?: string;
1271
+ functionCall?: {
1272
+ name: string;
1273
+ args: unknown;
1274
+ };
1275
+ }>;
1276
+ };
1277
+ }>;
1278
+ usageMetadata?: {
1279
+ promptTokenCount?: number;
1280
+ candidatesTokenCount?: number;
1281
+ totalTokenCount?: number;
1282
+ };
1283
+ }>>;
611
1284
  }
612
1285
  /**
613
1286
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
@@ -619,6 +1292,25 @@ interface GeminiClient {
619
1292
  * `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
620
1293
  * models use a token budget, not an effort tier, so it's dropped, same as
621
1294
  * Anthropic.
1295
+ *
1296
+ * `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
1297
+ * `tool_choice` maps to `toolConfig.functionCallingConfig`. `jsonSchema`
1298
+ * and `tools` are mutually exclusive by the time a call reaches here
1299
+ * (enforced in vernLLM.ts), so `responseSchema` and `tools` never
1300
+ * both apply.
1301
+ *
1302
+ * `createStream` calls `generateContentStream` (optional on `GeminiClient`
1303
+ *, required only if the caller sets `stream: true`) and translates each
1304
+ * partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
1305
+ * Gemini's own function-calling API doesn't stream tool-call arguments
1306
+ * incrementally: a `functionCall` part always arrives whole in one chunk,
1307
+ * so each one is emitted as a single, complete `tool_call_delta` (a
1308
+ * one-shot "delta" containing the full arguments) rather than accumulated
1309
+ * fragments, that's a real difference in the underlying API, not
1310
+ * something this adapter can smooth over. `usageMetadata` is (per Gemini's
1311
+ * own behavior) only reliably present on the last chunk, so the `usage`
1312
+ * `WireStreamChunk` is emitted once, after the stream completes, from
1313
+ * whichever chunk's `usageMetadata` was seen last.
622
1314
  */
623
1315
  declare function fromGemini(geminiClient: GeminiClient): LLMClient;
624
1316
 
@@ -636,6 +1328,20 @@ type BedrockContentBlock = {
636
1328
  bytes: Uint8Array;
637
1329
  };
638
1330
  };
1331
+ } | {
1332
+ toolUse: {
1333
+ toolUseId: string;
1334
+ name: string;
1335
+ input: unknown;
1336
+ };
1337
+ } | {
1338
+ toolResult: {
1339
+ toolUseId: string;
1340
+ content: Array<{
1341
+ text: string;
1342
+ }>;
1343
+ status?: 'success' | 'error';
1344
+ };
639
1345
  };
640
1346
  /**
641
1347
  * Minimal structural type matching AWS Bedrock's Converse API. This is
@@ -682,6 +1388,10 @@ interface BedrockConverseClient {
682
1388
  tool: {
683
1389
  name: string;
684
1390
  };
1391
+ } | {
1392
+ auto: Record<string, never>;
1393
+ } | {
1394
+ any: Record<string, never>;
685
1395
  };
686
1396
  };
687
1397
  }, options: {
@@ -692,6 +1402,7 @@ interface BedrockConverseClient {
692
1402
  content?: Array<{
693
1403
  text?: string;
694
1404
  toolUse?: {
1405
+ toolUseId?: string;
695
1406
  name?: string;
696
1407
  input?: unknown;
697
1408
  };
@@ -704,7 +1415,89 @@ interface BedrockConverseClient {
704
1415
  totalTokens?: number;
705
1416
  };
706
1417
  }>;
1418
+ /**
1419
+ * Optional. Required only for `stream: true` calls. Takes the same
1420
+ * request shape `converse` does, returning `{ stream }`, matching
1421
+ * `ConverseStreamCommand`'s real AWS SDK v3 output shape, an
1422
+ * `AsyncIterable` of incremental events under a `stream` property,
1423
+ * rather than the whole response being the iterable directly.
1424
+ */
1425
+ converseStream?(params: Parameters<BedrockConverseClient['converse']>[0], options: {
1426
+ signal: AbortSignal;
1427
+ }): Promise<{
1428
+ stream: AsyncIterable<BedrockConverseStreamEvent>;
1429
+ }>;
707
1430
  }
1431
+ /**
1432
+ * One event of a Bedrock `ConverseStreamCommand` response's `stream`.
1433
+ * Content blocks (text or toolUse) are identified by `contentBlockIndex`,
1434
+ * Converse's own convention for correlating start/delta/stop events across
1435
+ * possibly-interleaved blocks, mirrored directly by VernLLM's
1436
+ * `tool_call_delta.index`.
1437
+ */
1438
+ type BedrockConverseStreamEvent = {
1439
+ messageStart: {
1440
+ role: 'assistant';
1441
+ };
1442
+ } | {
1443
+ contentBlockStart: {
1444
+ contentBlockIndex: number;
1445
+ start?: {
1446
+ toolUse?: {
1447
+ toolUseId?: string;
1448
+ name?: string;
1449
+ };
1450
+ };
1451
+ };
1452
+ } | {
1453
+ contentBlockDelta: {
1454
+ contentBlockIndex: number;
1455
+ delta?: {
1456
+ text?: string;
1457
+ } | {
1458
+ toolUse?: {
1459
+ input?: string;
1460
+ };
1461
+ };
1462
+ };
1463
+ } | {
1464
+ contentBlockStop: {
1465
+ contentBlockIndex: number;
1466
+ };
1467
+ } | {
1468
+ messageStop: {
1469
+ stopReason?: string;
1470
+ };
1471
+ } | {
1472
+ metadata: {
1473
+ usage?: {
1474
+ inputTokens?: number;
1475
+ outputTokens?: number;
1476
+ totalTokens?: number;
1477
+ };
1478
+ };
1479
+ } | {
1480
+ internalServerException: {
1481
+ message?: string;
1482
+ };
1483
+ } | {
1484
+ modelStreamErrorException: {
1485
+ message?: string;
1486
+ originalStatusCode?: number;
1487
+ };
1488
+ } | {
1489
+ validationException: {
1490
+ message?: string;
1491
+ };
1492
+ } | {
1493
+ throttlingException: {
1494
+ message?: string;
1495
+ };
1496
+ } | {
1497
+ serviceUnavailableException: {
1498
+ message?: string;
1499
+ };
1500
+ };
708
1501
  /**
709
1502
  * Optional configuration for `fromBedrock`.
710
1503
  */
@@ -743,6 +1536,22 @@ interface BedrockAdapterOptions {
743
1536
  * `response_format: json_object` (no schema to build a tool from) and
744
1537
  * `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
745
1538
  * instruction and are dropped respectively.
1539
+ *
1540
+ * `tools` maps to Converse's native `toolConfig`/`toolUse`/`toolResult`;
1541
+ * `tool_choice` maps to `toolConfig.toolChoice`. Mutually exclusive with
1542
+ * `jsonSchema` by the time a call reaches here (enforced in vernLLM.ts).
1543
+ *
1544
+ * `createStream` calls `converseStream` (optional on `BedrockConverseClient`
1545
+ *, required only if the caller sets `stream: true`) and translates its
1546
+ * `contentBlockStart`/`contentBlockDelta`/`metadata` events into
1547
+ * `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
1548
+ * same as `fromAnthropic`'s block-index tracking (Converse's streaming
1549
+ * shape is structurally close to Anthropic's own, both being tool-use-aware
1550
+ * content-block streams), including the same `json-tool` unwrapping: a
1551
+ * `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
1552
+ * `text-delta`, not `tool_call_delta`, so the accumulated result lands in
1553
+ * `finalizeResponse`'s `content` path exactly like the non-streaming
1554
+ * `create` branch above unwraps it.
746
1555
  */
747
1556
  declare function fromBedrock(bedrockClient: BedrockConverseClient, options?: BedrockAdapterOptions): LLMClient;
748
1557
 
@@ -772,6 +1581,22 @@ type RequestLike = (url: string, init: {
772
1581
  body?: string;
773
1582
  signal?: AbortSignal;
774
1583
  }) => Promise<ResponseLike>;
1584
+ /**
1585
+ * A streaming-capable request function. Unlike `RequestLike`, which returns
1586
+ * a fully-buffered `ResponseLike`, this resolves to an `AsyncIterable` of
1587
+ * progressively-arriving chunks, the common ground across transports:
1588
+ * native `fetch`'s `response.body` (wrapped to be iterable; see
1589
+ * `webStreamToAsyncIterable` below), axios's Node `Readable` in
1590
+ * `responseType: 'stream'` mode (already async-iterable, no wrapping
1591
+ * needed), `node-fetch`, `undici`, etc, all satisfy this with little or no
1592
+ * glue code. Defaults to native `fetch`.
1593
+ */
1594
+ type StreamRequestLike = (url: string, init: {
1595
+ method: string;
1596
+ headers: Record<string, string>;
1597
+ body?: string;
1598
+ signal?: AbortSignal;
1599
+ }) => Promise<AsyncIterable<Uint8Array | string>>;
775
1600
  interface FetchAdapterConfig {
776
1601
  /** Endpoint URL, or a function of the request in case it depends on model/params */
777
1602
  url: string | ((params: ChatRequest) => string);
@@ -788,17 +1613,65 @@ interface FetchAdapterConfig {
788
1613
  /** Maps VernLLMs internal chat-completion request into the providers raw request body */
789
1614
  mapRequest: (params: ChatRequest) => unknown;
790
1615
  /**
791
- * Maps the providers raw JSON response into `{ content, usage? }`
792
- * `content` is the assistants text (JSON string when JSON mode was requested)
1616
+ * Maps the providers raw JSON response into `{ content, usage?, toolCalls? }`
1617
+ * `content` is the assistants text (JSON string when JSON mode was requested).
1618
+ * `content` may be empty/omitted when the model responded with only tool
1619
+ * calls and no text.
1620
+ *
1621
+ * `toolCalls`, when the model requested one or more tools, is the list of
1622
+ * calls as flat `{ id, name, arguments }` entries (matching this config's
1623
+ * own `toolCalls?: Array<{ id: string; name: string; arguments: string }>`
1624
+ * return type below), each entry's `arguments` already JSON-*encoded* as a
1625
+ * string (not the parsed object), mirroring the wire format every
1626
+ * OpenAI-compatible provider uses. `fromFetch` itself converts these into
1627
+ * `WireToolCall`'s `type`/`function`-wrapped shape before returning them
1628
+ * from `create`. VernLLM parses (and validates, if `argumentsSchema` was
1629
+ * set) the arguments string internally, mapResponse doesn't need to do
1630
+ * that itself.
793
1631
  */
794
1632
  mapResponse: (json: unknown) => {
795
- content: string;
1633
+ content?: string;
796
1634
  usage?: {
797
1635
  promptTokens?: number;
798
1636
  completionTokens?: number;
799
1637
  totalTokens?: number;
800
1638
  };
1639
+ toolCalls?: Array<{
1640
+ id: string;
1641
+ name: string;
1642
+ arguments: string;
1643
+ }>;
801
1644
  };
1645
+ /**
1646
+ * Optional. Required only for `stream: true` calls. The function used to
1647
+ * open a streaming HTTP request. Takes the same request shape as
1648
+ * `request`, but resolves to an `AsyncIterable` of progressively-arriving
1649
+ * `Uint8Array` or `string` chunks instead of a buffered `ResponseLike`.
1650
+ * Defaults to native `fetch`.
1651
+ */
1652
+ requestStream?: StreamRequestLike;
1653
+ /**
1654
+ * Optional. How the raw stream bytes are split into individual event
1655
+ * payloads. Defaults to Server-Sent Events framing (`data: ...` blocks
1656
+ * separated by a blank line, `[DONE]` sentinel honored, see
1657
+ * `parseSseStream`), which covers the large majority of LLM providers'
1658
+ * streaming HTTP endpoints. Override this for a provider that frames its
1659
+ * stream differently, e.g. newline-delimited JSON (NDJSON) with no SSE
1660
+ * envelope.
1661
+ */
1662
+ parseStreamFrames?: (chunks: AsyncIterable<Uint8Array | string>) => AsyncIterable<unknown>;
1663
+ /**
1664
+ * Optional. Required only for `stream: true` calls. Maps one parsed
1665
+ * stream event (already extracted from its frame by `parseStreamFrames`)
1666
+ * into zero, one, or more `WireStreamChunk`s, mirrors `mapResponse`'s
1667
+ * role for the non-streaming path, just per-event instead of once for
1668
+ * the whole body. Return `undefined` to skip an event that carries
1669
+ * nothing VernLLM needs (e.g. a provider's keep-alive ping). Configs
1670
+ * that don't implement this make `stream: true` throw a clear
1671
+ * `LLMError('validation')` rather than a confusing runtime failure or a
1672
+ * silently empty stream.
1673
+ */
1674
+ mapStreamEvent?: (event: unknown) => WireStreamChunk | WireStreamChunk[] | undefined;
802
1675
  }
803
1676
  /**
804
1677
  * A fetch-based escape hatch for providers with no SDK, or where pulling one
@@ -810,6 +1683,34 @@ interface FetchAdapterConfig {
810
1683
  * Non-2xx responses throw an error with `.status` set to the HTTP status
811
1684
  * code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
812
1685
  * 401/403) applies here too
1686
+ *
1687
+ * Tool calling works the same way as every other adapter: `mapRequest`
1688
+ * receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
1689
+ * translate them into whatever shape the provider's wire format expects
1690
+ * (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
1691
+ * field). On the way back, `mapResponse` may return a `toolCalls` array
1692
+ * (id/name/JSON-encoded-arguments-string per call) alongside or instead of
1693
+ * `content`; VernLLM parses and (if `argumentsSchema` was set) validates
1694
+ * those arguments the same way it does for every other adapter. For
1695
+ * `stream: true`, tool-call deltas go through the existing
1696
+ * `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
1697
+ * no separate config is needed for streaming vs non-streaming tool calls.
1698
+ *
1699
+
1700
+ * `createStream` requires `mapStreamEvent` (there's no non-streaming
1701
+ * response to fall back on, unlike the other three optional streaming
1702
+ * seams). It opens the request via `requestStream` (defaults to native
1703
+ * `fetch`), splits the raw bytes into individual events via
1704
+ * `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
1705
+ * and translates each event into `WireStreamChunk`(s) via
1706
+ * `mapStreamEvent`. Both seams are overridable per-config for providers
1707
+ * that don't fit the SSE-over-fetch default. If a custom `request`
1708
+ * transport is configured, `requestStream` must be configured too,
1709
+ * `requestStream` never silently falls back to `request` (see
1710
+ * `createStream`'s own comment for why), so a `stream: true` call with
1711
+ * `request` set but no `requestStream` throws a clear
1712
+ * `LLMError('validation')` instead of quietly using unrelated native
1713
+ * `fetch`.
813
1714
  */
814
1715
  declare function fromFetch(config: FetchAdapterConfig): LLMClient;
815
1716
 
@@ -834,11 +1735,46 @@ declare function fromFetch(config: FetchAdapterConfig): LLMClient;
834
1735
  * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
835
1736
  * the actual compatibility contract is the JSON each provider sends and
836
1737
  * receives over the wire, not the SDKs TS types.
1738
+ *
1739
+ * `createStream` is implemented by calling the same underlying
1740
+ * `chat.completions.create` with `stream: true` (and, for providers that
1741
+ * support it, `stream_options: { include_usage: true }`, so a final usage
1742
+ * block arrives), the OpenAI SDK, and every OpenAI-compatible client
1743
+ * modeled on it, returns an `AsyncIterable` of SSE chunks instead of a
1744
+ * single completion object when `stream: true` is set. Each chunk is
1745
+ * translated into `WireStreamChunk`(s) via `toWireStreamChunks`.
1746
+ *
1747
+ * Note on long-running reasoning models: this adapter consumes the
1748
+ * underlying SDK's already-parsed stream rather than raw SSE bytes, so
1749
+ * unlike `fromFetch`/`fromAnthropic` it cannot see comment-only keep-alive
1750
+ * ping frames. Combined with `chunkIdleTimeoutMs`'s 30 second default and
1751
+ * `reasoningEffort` (documented to have long silent gaps for o-series and
1752
+ * similar models), a long-running reasoning call on this adapter can trip
1753
+ * the idle timeout even though the provider is still working. Raise or
1754
+ * disable `chunkIdleTimeoutMs` per call for those routes, see `CallParams`.
837
1755
  */
838
- declare function fromOpenAICompatible(client: unknown): LLMClient;
1756
+ interface OpenAICompatibleAdapterOptions {
1757
+ /**
1758
+ * Whether the provider supports `stream_options.include_usage`. Not
1759
+ * every "OpenAI-compatible" provider is guaranteed to, so this defaults
1760
+ * to `true` (matching OpenAI, Groq, Mistral, and most others observed)
1761
+ * and should be set to `false` for a provider verified not to support
1762
+ * it. When `false`, `stream_options` is omitted entirely and no usage
1763
+ * block will arrive on the stream; callers relying on streamed `usage`
1764
+ * with such a provider won't get one.
1765
+ */
1766
+ supportsStreamUsage?: boolean;
1767
+ }
1768
+ declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
839
1769
  /** Groqs SDK matches the OpenAI wire format */
840
1770
  declare const fromGroq: typeof fromOpenAICompatible;
841
- /** Mistrals `chat.completions`-shaped client (or their OpenAI-compat endpoint) */
1771
+ /**
1772
+ * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
1773
+ * endpoint). Mistral supports `stream_options.include_usage` (added after
1774
+ * an earlier period where it returned a 422 for unrecognized fields, per
1775
+ * Mistral's changelog and streaming docs), so this is a plain alias like
1776
+ * the others, `supportsStreamUsage` defaults to `true`.
1777
+ */
842
1778
  declare const fromMistral: typeof fromOpenAICompatible;
843
1779
  /** DeepSeeks API is OpenAI-compatible */
844
1780
  declare const fromDeepSeek: typeof fromOpenAICompatible;
@@ -929,5 +1865,5 @@ declare const from01AI: typeof fromOpenAICompatible;
929
1865
  //#endregion
930
1866
  //# sourceMappingURL=openaiCompatible.d.ts.map
931
1867
 
932
- export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CallParams, CircuitBreaker, CircuitBreakerOptions, ConsoleLogger, ContentBlock, ConversationTurn, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorType, Logger, NormalizedCacheAdapter, OnUsage, RefundUsage, ReserveUsage, SchemaLike, TextBlock, TieredCacheAdapter, TokenUsage, VernLLM, VernLLMOptions, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError };
1868
+ export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedStreamCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallParams, CallWithToolsResult, CircuitBreaker, CircuitBreakerOptions, ConsoleLogger, ContentBlock, ContentResult, ConversationTurn, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorType, Logger, NormalizedCacheAdapter, OnUsage, RefundUsage, ReserveUsage, SSE_PING, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, TextBlock, TieredCacheAdapter, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolResult, VernLLM, VernLLMOptions, WireMessage, WireStreamChunk, WireToolCall, WireToolChoice, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError, isToolCallResult, parseSseStream };
933
1869
  //# sourceMappingURL=index.d.cts.map