vern-llm 1.7.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,14 +1,109 @@
1
+ //#region src/circuitBreaker.d.ts
2
+ interface CircuitBreakerOptions {
3
+ /** Consecutive failures before the circuit opens, default 5 */
4
+ threshold?: number;
5
+ /** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
6
+ cooldownMs?: number;
7
+ /**
8
+ * Called after every real state change, never for a no-op transition
9
+ * (e.g. open to open). `model` is the resolved model of whichever call
10
+ * triggered this specific transition (the `model` passed to whichever
11
+ * of `assertClosed`/`recordSuccess`/`recordFailure` caused it).
12
+ *
13
+ * With `isolateByModel` off (the default), this is a label only: the
14
+ * breaker still counts failures across every model together, so a
15
+ * threshold crossing can be the sum of several different models'
16
+ * failures even though only the triggering call's `model` is reported
17
+ * here. With `isolateByModel` on, it's exact: each model has its own
18
+ * counter, so the transition really was caused solely by that model.
19
+ */
20
+ onStateChange?: (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string) => void;
21
+ /**
22
+ * Track a separate circuit per resolved model instead of one shared
23
+ * circuit for the whole instance. A failure on one model then never
24
+ * opens another model's circuit, at the cost of slower detection for
25
+ * an outage spread across many distinct models (each model's counter
26
+ * must independently cross `threshold`). Default false: one shared
27
+ * circuit, matching every version before this option existed.
28
+ *
29
+ * A call that omits `model` (only possible calling `CircuitBreaker`
30
+ * directly, `VernLLM` always passes one) falls into one shared bucket
31
+ * alongside every other call that also omits it.
32
+ */
33
+ isolateByModel?: boolean;
34
+ }
35
+ type CircuitState = 'closed' | 'open' | 'half-open';
36
+ /**
37
+ * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures across
38
+ * calls. Once the threshold is hit, short-circuits new calls with an
39
+ * LLMError('circuit_open') instead of hitting the provider, until the
40
+ * cooldown elapses and a single trial call is allowed through
41
+ */
42
+ declare class CircuitBreaker {
43
+ private readonly threshold;
44
+ private readonly cooldownMs;
45
+ private readonly onStateChange?;
46
+ private readonly isolateByModel;
47
+ private readonly sharedBucket;
48
+ private readonly bucketsByModel;
49
+ constructor(options?: CircuitBreakerOptions);
50
+ /** Returns the bucket for a model if one already exists, without allocating. */
51
+ private lookupBucket;
52
+ /** Creates and stores a bucket for a model when the first mutation needs one. */
53
+ private ensureBucketFor;
54
+ /** Every state mutation routes through here, so `onStateChange` fires exactly once per real change. */
55
+ private transition;
56
+ /**
57
+ * Throws if the circuit is open and the cooldown hasn't elapsed, or if
58
+ * the circuit is half-open and a trial call is already in flight.
59
+ * Otherwise, if the circuit just became eligible for a trial (cooldown
60
+ * elapsed, or half-open with no trial currently running), this call
61
+ * becomes that trial
62
+ */
63
+ assertClosed(model?: string): void;
64
+ recordSuccess(model?: string): void;
65
+ recordFailure(model?: string): void;
66
+ /**
67
+ * With `isolateByModel` off (the default), `model` is ignored and the
68
+ * one shared circuit's state is returned, unchanged from every version
69
+ * before this option existed. With `isolateByModel` on, returns that
70
+ * model's own state, `'closed'` for a model never seen yet, same as a
71
+ * fresh breaker.
72
+ */
73
+ getState(model?: string): CircuitState;
74
+ } //#endregion
1
75
  //#region src/types/errors.d.ts
76
+
77
+ //# sourceMappingURL=circuitBreaker.d.ts.map
2
78
  type LLMErrorType = 'timeout' | 'api' | 'parse' | 'validation' | 'circuit_open' | 'quota_exceeded' | 'unknown' | 'aborted';
79
+ /**
80
+ * Machine readable discriminator within a `type`, for cases where `type`
81
+ * alone is too coarse to act on. Optional and additive: errors thrown
82
+ * before a given code existed simply omit it.
83
+ */
84
+ type LLMErrorCode = 'unknown_tool' | 'duplicate_tool_call_id' | 'local_rate_limit' | 'provider_rate_limited' | 'fallback_exhausted';
85
+ /** One tool call's contract failure, used to report every bad call in a response at once. */
86
+ interface ToolIssue {
87
+ name: string;
88
+ toolCallId: string;
89
+ code: LLMErrorCode;
90
+ detail?: unknown;
91
+ }
3
92
  declare class LLMError extends Error {
4
93
  type: LLMErrorType;
5
94
  status?: number | undefined;
6
95
  issues?: unknown | undefined;
7
96
  cause?: unknown | undefined;
8
97
  retryAfterMs?: number | undefined;
9
- constructor(message: string, type: LLMErrorType, status?: number | undefined, issues?: unknown | undefined, cause?: unknown | undefined, retryAfterMs?: number | undefined);
98
+ /** Stable discriminator within `type`. Absent on errors predating it. */
99
+ code?: LLMErrorCode | undefined;
100
+ constructor(message: string, type: LLMErrorType, status?: number | undefined, issues?: unknown | undefined, cause?: unknown | undefined, retryAfterMs?: number | undefined, /** Stable discriminator within `type`. Absent on errors predating it. */
101
+ code?: LLMErrorCode | undefined);
102
+ /** Every tool contract failure in one response, when there is more than one. */
103
+ toolIssues?: ToolIssue[];
10
104
  }
11
105
  declare function isLLMError(err: unknown): err is LLMError;
106
+
12
107
  //#endregion
13
108
  //#region src/types/cache.d.ts
14
109
  //# sourceMappingURL=errors.d.ts.map
@@ -77,8 +172,206 @@ declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
77
172
  }
78
173
 
79
174
  //#endregion
80
- //#region src/types/schema.d.ts
175
+ //#region src/rateLimit.d.ts
81
176
  //# sourceMappingURL=cache.d.ts.map
177
+ /** The request shape sent to `LLMClient['chat']['completions']['create']`, used for token estimation. */
178
+ type WireRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
179
+ /** Which configured bucket is currently blocking a call. */
180
+ type RateLimitReason = 'concurrency' | 'rpm' | 'tpm';
181
+ interface RateLimitOptions {
182
+ /** Max requests per minute. Omit for unlimited. */
183
+ requestsPerMinute?: number;
184
+ /**
185
+ * Max tokens per minute. Enforced against a pre-flight estimate, then
186
+ * reconciled against reported usage once the call completes. Omit for
187
+ * unlimited.
188
+ */
189
+ tokensPerMinute?: number;
190
+ /** Max requests in flight at once. Default 0, meaning unlimited. */
191
+ maxConcurrent?: number;
192
+ /**
193
+ * Max time a call may sit queued waiting for capacity, in ms. Exceeding
194
+ * it throws rather than hanging forever. Default 30000. Pass 0 to wait
195
+ * indefinitely.
196
+ */
197
+ maxQueueMs?: number;
198
+ /** Max queued calls before new ones reject immediately instead of queueing. Default 0, unbounded. */
199
+ maxQueueSize?: number;
200
+ /**
201
+ * Pre-flight token estimate for `tokensPerMinute`. Defaults to a
202
+ * chars/4 heuristic over message content plus `max_tokens`.
203
+ */
204
+ estimateTokens?: (request: WireRequest) => number;
205
+ }
206
+ interface RateLimitAcquireResult {
207
+ /**
208
+ * Releases the concurrency slot this attempt held and reconciles the
209
+ * token bucket against real usage, when `actualTokens` is supplied.
210
+ * Idempotent: only the first call does anything. Must run in a
211
+ * `finally` block so a slot is never leaked on a failed attempt.
212
+ */
213
+ release: (actualTokens?: number) => void;
214
+ /** How long this attempt waited in queue before capacity was available. */
215
+ waitedMs: number;
216
+ /** Which bucket was blocking this attempt just before it cleared, if any wait happened. */
217
+ reason?: RateLimitReason;
218
+ }
219
+ /** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
220
+ declare function defaultEstimateTokens(request: WireRequest): number;
221
+ /**
222
+ * Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
223
+ * concurrency) behind one FIFO queue, so a large call isn't starved by a
224
+ * stream of small ones. Any bucket omitted from `options` has infinite
225
+ * capacity and never blocks.
226
+ */
227
+ declare class RateLimiter {
228
+ private readonly requests?;
229
+ private readonly tokens?;
230
+ private readonly concurrency?;
231
+ private readonly maxQueueMs;
232
+ private readonly maxQueueSize;
233
+ private readonly estimateTokensFn;
234
+ private readonly queue;
235
+ /**
236
+ * A single scheduled re-check for the head of the queue when it's
237
+ * blocked on a bucket that refills on its own clock (rpm/tpm), so a
238
+ * queue that nobody calls `acquire`/`release` on again isn't stuck
239
+ * forever waiting for an external trigger to re-drain it. Not needed
240
+ * for a concurrency block, which only clears via `release`.
241
+ */
242
+ private wakeTimer?;
243
+ constructor(options: RateLimitOptions);
244
+ /** Pre-flight token estimate for a request, per the configured (or default) heuristic. */
245
+ estimate(request: WireRequest): number;
246
+ /**
247
+ * Waits for capacity in every configured bucket, then takes from each.
248
+ * The returned `release` gives the concurrency slot back and reconciles
249
+ * the token bucket against real usage; it must run in a `finally` block.
250
+ */
251
+ acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
252
+ private queueFullError;
253
+ private enqueue;
254
+ /**
255
+ * Checks and takes from every configured bucket as one atomic unit: if
256
+ * any bucket lacks capacity, whatever was already taken from the
257
+ * earlier ones in this attempt is rolled back before reporting which
258
+ * bucket blocked.
259
+ */
260
+ private tryAcquireBuckets;
261
+ /** Drains the queue head first. Stops at the first waiter that still can't proceed, so no one is starved out of turn. */
262
+ private drain;
263
+ /**
264
+ * Schedules a one-shot re-check of the queue for whenever the bucket
265
+ * that's currently blocking the head waiter should next have enough
266
+ * capacity. A no-op for a concurrency block (only `release` can clear
267
+ * that) or while a wake is already pending.
268
+ */
269
+ private scheduleWake;
270
+ /**
271
+ * Builds the one-shot release closure for an acquired slot. Only the
272
+ * concurrency bucket is given back on release; the requests-per-minute
273
+ * bucket is a real spend that only recovers via its own refill, and the
274
+ * tokens bucket is reconciled against `actualTokens` rather than fully
275
+ * refunded, since real tokens really were spent.
276
+ */
277
+ private makeRelease;
278
+ }
279
+
280
+ //#endregion
281
+ //#region src/types/fallback.d.ts
282
+ //# sourceMappingURL=rateLimit.d.ts.map
283
+ /**
284
+ * One provider to try after the primary (or after an earlier fallback
285
+ * target) fails. Order is the policy: VernLLM never reorders, scores, or
286
+ * selects a target, it only walks the list as given.
287
+ *
288
+ * Most per-target overrides fall back to the parent `VernLLM` instance's
289
+ * own option when omitted, so a target only needs to specify what's
290
+ * actually different about it (a different client/model is the common
291
+ * case). `circuitBreaker` and `rateLimit` are the exception: they are
292
+ * never inherited from the parent, since a breaker or limiter tuned for
293
+ * the primary provider's limits is rarely right for a fallback's. Leave
294
+ * them unset on a target to run it without one, even if the parent has
295
+ * one configured.
296
+ */
297
+ interface FallbackTarget {
298
+ client: LLMClient;
299
+ model: string;
300
+ /** Label for events, errors, and `TokenUsage.provider`. Default `` `fallback[${index}]` ``. */
301
+ name?: string;
302
+ maxRetries?: number;
303
+ timeoutMs?: number;
304
+ chunkIdleTimeoutMs?: number;
305
+ baseDelayMs?: number;
306
+ defaultMaxTokens?: number;
307
+ defaultTemperature?: number | null;
308
+ nonRetryableStatus?: number[];
309
+ /** This target's own circuit breaker, independent of every other target's. Not inherited from the parent's `circuitBreaker`. */
310
+ circuitBreaker?: boolean | CircuitBreakerOptions;
311
+ /** This target's own rate limiter, independent of every other target's. Not inherited from the parent's `rateLimit`. */
312
+ rateLimit?: RateLimitOptions;
313
+ }
314
+ /**
315
+ * Written into `CallParams['meta']` once `call()` resolves, so a caller
316
+ * who wants provider identity on the same line as the result doesn't need
317
+ * to read it back out of `onUsage`.
318
+ */
319
+ interface CallMeta {
320
+ provider: string;
321
+ model: string;
322
+ /** `-1` if the primary target answered, otherwise the index into `fallback`. */
323
+ fallbackIndex: number;
324
+ usedFallback: boolean;
325
+ /** Attempts made against the target that ultimately answered, including the successful one. */
326
+ attempts: number;
327
+ }
328
+ /** One target's circuit state, as returned by `VernLLM.getCircuitStates()`. */
329
+ interface TargetCircuitState {
330
+ provider: string;
331
+ /** Position in the chain: `0` for the primary, `1`+ for fallback targets. */
332
+ index: number;
333
+ isFallback: boolean;
334
+ /** `undefined` if that target has no circuit breaker configured. */
335
+ state: CircuitState | undefined;
336
+ }
337
+ /** One target's failure, recorded on the way to either the next target or `FallbackExhaustedError`. */
338
+ interface FallbackAttempt {
339
+ /** `-1` for the primary target. */
340
+ index: number;
341
+ provider: string;
342
+ model: string;
343
+ error: LLMError;
344
+ }
345
+ /**
346
+ * Decides what happens after a target's own retries are exhausted or
347
+ * abandoned early. Called once per failed target. `'retry'` is not a
348
+ * valid return here: retrying already happened inside the target, this
349
+ * only decides whether to move on to the next one or stop.
350
+ */
351
+ type FallbackOn = (error: LLMError, context: {
352
+ isLastTarget: boolean;
353
+ }) => 'next' | 'stop';
354
+ /**
355
+ * The default `fallbackOn` policy. Exported so a caller can wrap rather
356
+ * than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
357
+ */
358
+ declare const defaultFallbackOn: FallbackOn;
359
+ /**
360
+ * Thrown when the chain gives up, whether because the last target failed
361
+ * or `fallbackOn` chose to stop early. Carries each attempt in order so
362
+ * an outage across providers stays debuggable without reproducing it.
363
+ * Extends `LLMError` so `isLLMError` and any `instanceof LLMError` check
364
+ * still passes, inheriting the last failure's `type` so existing
365
+ * type-based handling keeps working on a fallback-exhausted error too.
366
+ */
367
+ declare class FallbackExhaustedError extends LLMError {
368
+ readonly attempts: FallbackAttempt[];
369
+ constructor(attempts: FallbackAttempt[]);
370
+ }
371
+
372
+ //#endregion
373
+ //#region src/types/schema.d.ts
374
+ //# sourceMappingURL=fallback.d.ts.map
82
375
  /**
83
376
  * Minimal structural type for a Zod-like schema, so this package doesnt need
84
377
  * a hard dependency on a specific Zod major version. Any object exposing
@@ -110,8 +403,79 @@ interface JsonSchemaSpec {
110
403
  }
111
404
 
112
405
  //#endregion
113
- //#region src/types/usage.d.ts
406
+ //#region src/types/tools.d.ts
114
407
  //# sourceMappingURL=schema.d.ts.map
408
+ /**
409
+ * Describes a capability the model may request, not the capability
410
+ * itself. VernLLM transports this to the provider and parses what comes
411
+ * back; it never executes anything.
412
+ */
413
+ interface ToolDefinition {
414
+ name: string;
415
+ description: string;
416
+ /** JSON Schema for the tool's input. */
417
+ parameters: Record<string, unknown>;
418
+ /**
419
+ * Optional client-side validator run on the parsed `arguments` before
420
+ * they're handed back to the caller, mirroring the `schema: SchemaLike<T>`
421
+ * pattern already used for response validation (see `types/schema.ts`).
422
+ * Reuses that zero-dependency, `safeParse`-compatible shape instead of
423
+ * requiring a JSON Schema validator (e.g. ajv) as a new dependency.
424
+ * Failed validation throws `LLMError('validation')`. If omitted, VernLLM
425
+ * parses arguments as JSON but does not validate them further.
426
+ */
427
+ argumentsSchema?: SchemaLike<unknown>;
428
+ }
429
+ /** A single tool invocation requested by the model. */
430
+ interface ToolCall {
431
+ id: string;
432
+ name: string;
433
+ /** Parsed JSON arguments (and validated, if `argumentsSchema` was set). */
434
+ arguments: unknown;
435
+ }
436
+ /** The application's result of executing a `ToolCall`, sent back to the model. */
437
+ interface ToolResult {
438
+ toolCallId: string;
439
+ content: unknown;
440
+ /**
441
+ * Signals a failed tool execution back to the model (matches Anthropic's
442
+ * native `is_error` on tool_result blocks). Only `fromAnthropic` honors
443
+ * this today, Gemini and Bedrock have no equivalent wire concept, so
444
+ * other adapters ignore it silently.
445
+ */
446
+ isError?: boolean;
447
+ }
448
+ /** `call()` result when `tools` was set and the model produced a normal answer. */
449
+ interface ContentResult<T> {
450
+ type: 'content';
451
+ content: T;
452
+ }
453
+ /** `call()` result when `tools` was set and the model requested one or more tools. */
454
+ interface ToolCallResult {
455
+ type: 'tool_calls';
456
+ toolCalls: ToolCall[];
457
+ /** Any text the model produced alongside the tool request, if present. */
458
+ content?: string;
459
+ }
460
+ type CallWithToolsResult<T> = ContentResult<T> | ToolCallResult;
461
+ /**
462
+ * Runtime-safe check for whether a `call()` result is a `tool_calls`
463
+ * result. Prefer this over relying on TypeScript's static narrowing
464
+ * whenever `params` passed to `call()` wasn't a literal with `tools`
465
+ * inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
466
+ * that case TS may have typed the result as plain `T` even though it's
467
+ * actually a `CallWithToolsResult<T>` at runtime, and this check works
468
+ * either way.
469
+ */
470
+ declare function isToolCallResult(result: unknown): result is ToolCallResult;
471
+ /** What the model should do about tools on a given call. */
472
+ type ToolChoice = 'auto' | 'none' | 'required' | {
473
+ name: string;
474
+ };
475
+
476
+ //#endregion
477
+ //#region src/types/usage.d.ts
478
+ //# sourceMappingURL=tools.d.ts.map
115
479
  type ReserveUsage = (params: {
116
480
  coalesced: boolean;
117
481
  signal?: AbortSignal;
@@ -142,17 +506,55 @@ interface TokenUsage {
142
506
  totalTokens: number;
143
507
  requestId: string;
144
508
  model: string;
509
+ /**
510
+ * The provider target that produced this usage. See `VernLLMOptions['name']`,
511
+ * default `'primary'`. Optional so consumers constructing a `TokenUsage`
512
+ * themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
513
+ * populates it. Absent means the same as `'primary'` if you need a value.
514
+ */
515
+ provider?: string;
516
+ /**
517
+ * Whether this usage came from a fallback target rather than the
518
+ * primary. Optional for the same reason `provider` is: `VernLLM`
519
+ * always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
520
+ * isn't forced to.
521
+ */
522
+ usedFallback?: boolean;
145
523
  }
146
524
  type OnUsage = (usage: TokenUsage) => void;
525
+ /**
526
+ * Called when a provider response arrives but VernLLM's own post-processing
527
+ * then fails, after usage data was already present in that response. Covers
528
+ * any error thrown after usage extraction, not just parse/validation, since
529
+ * everything in that path only runs once a response, and real spend, has
530
+ * already arrived. Fires once per failed attempt with extractable usage,
531
+ * never for transport failures, where no response means no honest number
532
+ * to report.
533
+ */
534
+ type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
147
535
 
148
536
  //#endregion
149
537
  //#region src/types/call.d.ts
150
538
  //# sourceMappingURL=usage.d.ts.map
151
- /** A single prior turn in a multi-turn conversation, passed via `history`. */
152
- interface ConversationTurn {
153
- role: 'user' | 'assistant';
539
+ /**
540
+ * A single prior turn in a multi-turn conversation, passed via `history`.
541
+ *
542
+ * Supports normal user/assistant messages and tool continuations: an assistant
543
+ * turn may include `toolCalls`, and a tool turn carries the matching
544
+ * `toolResults`. A tool turn must immediately follow an assistant tool call
545
+ * turn, and every requested tool call must have a result.
546
+ */
547
+ type ConversationTurn = {
548
+ role: 'user';
154
549
  content: string;
155
- }
550
+ } | {
551
+ role: 'assistant';
552
+ content?: string;
553
+ toolCalls?: ToolCall[];
554
+ } | {
555
+ role: 'tool';
556
+ toolResults: ToolResult[];
557
+ };
156
558
  /** A plain text segment of a multimodal `userContent` array. */
157
559
  interface TextBlock {
158
560
  type: 'text';
@@ -180,15 +582,28 @@ interface CallParams<T = unknown> extends UsageHooks {
180
582
  /** Current user message, as text or multimodal content blocks. */
181
583
  userContent: string | ContentBlock[];
182
584
  /**
183
- * Previous conversation turns. Must alternate user/assistant and end with
184
- * an assistant turn; invalid history throws LLMError('validation').
585
+ * Previous conversation turns. Must alternate roles; tool turns must follow
586
+ * assistant tool calls. Invalid history throws LLMError('validation').
185
587
  */
186
588
  history?: ConversationTurn[];
187
- temperature?: number;
589
+ /**
590
+ * Generation temperature. Default 0.2, not the provider's own default.
591
+ * Pass `null` to omit `temperature` from the request entirely, so the
592
+ * provider applies its own default instead.
593
+ */
594
+ temperature?: number | null;
188
595
  jsonMode?: boolean;
189
596
  maxTokens?: number;
190
597
  requestId?: string;
191
598
  signal?: AbortSignal;
599
+ /**
600
+ * Per-call override for the instance's `chunkIdleTimeoutMs` (max gap
601
+ * between stream chunks once opened). Only applies when `stream: true`.
602
+ * Useful for routes using reasoning-heavy models with documented long
603
+ * silent gaps mid-stream. Pass 0 to disable the idle timeout for this
604
+ * call.
605
+ */
606
+ chunkIdleTimeoutMs?: number;
192
607
  /** Overrides the instance model for this call. */
193
608
  model?: string;
194
609
  /** Reasoning effort for supported reasoning models. */
@@ -202,17 +617,253 @@ interface CallParams<T = unknown> extends UsageHooks {
202
617
  * Implies jsonMode: true.
203
618
  */
204
619
  schema?: SchemaLike<T>;
620
+ /**
621
+ * Tools the model may call. When set, `call()` always returns a
622
+ * `CallWithToolsResult<T>` discriminated union instead of `T` directly
623
+ * (see `CallWithToolsResult`), a breaking-change point: omitting `tools`
624
+ * keeps `call()`'s old `Promise<T>` behavior exactly.
625
+ *
626
+ * Can be combined with `jsonSchema` on Gemini and OpenAI-compatible
627
+ * clients unconditionally (neither ever restricted the combination:
628
+ * Gemini builds `responseSchema`/`tools` as independent fields, OpenAI-
629
+ * compatible clients pass both straight through). On Anthropic and
630
+ * Bedrock, combining the two is opt-in per call site, via each
631
+ * adapter's `nativeStructuredOutputModels` option: models not covered
632
+ * by it still throw `LLMError('validation')`, since `jsonSchema` falls
633
+ * back to a forced single-tool call there, which would collide with
634
+ * real tools. See `fromAnthropic`/`fromBedrock`.
635
+ *
636
+ * `schema` (client-side validation, distinct from `jsonSchema`) was
637
+ * never restricted from combining with `tools` on any provider.
638
+ */
639
+ tools?: ToolDefinition[];
640
+ /** Defaults to `'auto'` when `tools` is set. */
641
+ toolChoice?: ToolChoice;
642
+ /**
643
+ * Streams the response incrementally instead of resolving once. Default:
644
+ * false. Requires a client/adapter that implements `createStream`.
645
+ * Retry/timeout/circuit-breaker guarantees apply only to opening the
646
+ * stream (through the first chunk); a failure after that point rejects
647
+ * `finalResult` directly and is not retried, since a mid-stream failure
648
+ * isn't connection-time evidence for the circuit breaker, the attempt
649
+ * already counted as a success once the first chunk arrived. Once the
650
+ * stream opens successfully, `finalResult` still resolves to the same
651
+ * validated `T`/`CallWithToolsResult<T>` shape `call()` would have
652
+ * returned for the same params with `stream` omitted. See
653
+ * `StreamCallResult`.
654
+ */
655
+ stream?: boolean;
656
+ /**
657
+ * Optional out-parameter for provider identity. Pass `{}` (or any object
658
+ * with a mutable `current` property) and `call()` writes a `CallMeta`
659
+ * into `meta.current` before returning, alongside whatever `onUsage`
660
+ * already reports. Ignored for `stream: true`, since `call()` returns
661
+ * before the outcome (and so the target that answered) is known; read
662
+ * `TokenUsage.provider`/`usedFallback` from `onUsage` for streaming
663
+ * calls instead.
664
+ */
665
+ meta?: {
666
+ current?: CallMeta;
667
+ };
205
668
  }
206
- interface CachedCallParams<T> extends UsageHooks {
669
+ /**
670
+ * A `CallParams` variant where tool calling is explicitly enabled.
671
+ *
672
+ * Requiring `tools` to be present allows TypeScript to select the
673
+ * tool-aware `call()` overload and return `CallWithToolsResult<T>` instead
674
+ * of the normal `T` response type.
675
+ */
676
+ type ToolEnabledCallParams<T> = CallParams<T> & {
677
+ tools: NonNullable<CallParams<T>['tools']>;
678
+ };
679
+ /** Shared cache-configuration fields, minus the internal `fn` primitive. */
680
+ interface CachedCallInput extends UsageHooks {
207
681
  cacheKey: string;
208
682
  ttl: number;
209
- fn: () => Promise<T>;
210
683
  signal?: AbortSignal;
211
684
  }
685
+ /**
686
+ * Parameters for a cached LLM call without tool calling.
687
+ *
688
+ * Combines the cache configuration with the `CallParams` passed to
689
+ * `VernLLM.call()`. The cached value is the normal LLM response type `T`.
690
+ */
691
+ type CachedCallParams<T> = CachedCallInput & {
692
+ call: CallParams<T>;
693
+ };
694
+ /**
695
+ * Parameters for a cached LLM call with tool calling enabled.
696
+ *
697
+ * The cached value includes the full `CallWithToolsResult<T>`, meaning
698
+ * tool requests and normal content responses are cached exactly as returned
699
+ * by the model.
700
+ */
701
+ type CachedToolCallParams<T> = CachedCallInput & {
702
+ call: ToolEnabledCallParams<T>;
703
+ };
212
704
 
213
705
  //#endregion
214
- //#region src/types/client.d.ts
706
+ //#region src/types/stream.d.ts
215
707
  //# sourceMappingURL=call.d.ts.map
708
+ /** One incremental unit of a streaming response, as delivered to the caller. */
709
+ type StreamChunk = {
710
+ type: 'text-delta';
711
+ delta: string;
712
+ } | {
713
+ type: 'tool_call_delta';
714
+ index: number;
715
+ id?: string;
716
+ name?: string;
717
+ argsDelta?: string;
718
+ /**
719
+ * True when `argsDelta` is the whole set of arguments, not a
720
+ * fragment. Set for Gemini (its API returns function-call args
721
+ * whole in one chunk) and for cache/replay chunks, which are
722
+ * one-shot too. Omitted or `false` for a genuine fragment from
723
+ * providers that do stream incrementally (OpenAI-compatible,
724
+ * Anthropic, Bedrock).
725
+ */
726
+ complete?: boolean;
727
+ } | {
728
+ type: 'usage';
729
+ usage: TokenUsage;
730
+ };
731
+ /**
732
+ * What `call()` returns when `stream: true`. `chunks` is for live rendering;
733
+ * `finalResult` resolves to the same validated `T`/`CallWithToolsResult<T>`
734
+ * shape `call()` would have returned had `stream` been omitted, once the
735
+ * stream completes successfully.
736
+ *
737
+ * `chunks` is single-use and supports only one consumer: iterating it more
738
+ * than once, or from more than one place concurrently, shares the same
739
+ * underlying buffered stream rather than replaying or forking it, which can
740
+ * split chunks unpredictably between consumers. Stopping iteration early
741
+ * (e.g. `break`ing out of a `for await`) does not cancel or otherwise
742
+ * signal the underlying stream, the background pump keeps running to
743
+ * completion regardless, buffering any chunks emitted after that point, so
744
+ * `finalResult` still settles normally even if `chunks` is abandoned or
745
+ * never read at all.
746
+ *
747
+ * Unread chunks are buffered internally for the duration of one stream,
748
+ * this is what lets a caller start iterating `chunks` after the stream has
749
+ * already progressed (or finished) and still see everything. That backlog
750
+ * is capped: an unusually large stream whose `chunks` is never read at all
751
+ * has its oldest buffered chunks dropped once the backlog grows past
752
+ * roughly twice a fixed internal limit, trimmed back down to that limit in
753
+ * one batch rather than one-at-a-time, bounding both peak memory and the
754
+ * eviction work itself for that pathological case instead of the array
755
+ * growing (or being trimmed) proportional to the whole stream's output.
756
+ * Ordinary consumption, even started somewhat late, stays far under the
757
+ * limit and is unaffected.
758
+ */
759
+ interface StreamCallResult<R> {
760
+ chunks: AsyncIterable<StreamChunk>;
761
+ finalResult: Promise<R>;
762
+ }
763
+ /**
764
+ * A `CallParams` variant where streaming is explicitly enabled.
765
+ *
766
+ * Requiring `stream: true` to be statically present allows TypeScript to
767
+ * select the streaming `call()` overload and return `StreamCallResult<...>`
768
+ * instead of the normal, single-shot response type.
769
+ */
770
+ type StreamEnabledCallParams<T> = CallParams<T> & {
771
+ stream: true;
772
+ };
773
+ /**
774
+ * The adapter-facing, pre-normalization shape a `createStream` client
775
+ * implementation emits, analogous to how `WireMessage`/`WireToolCall`
776
+ * already sit between `CallParams` and each provider's own wire format.
777
+ */
778
+ type WireStreamChunk = {
779
+ type: 'text-delta';
780
+ delta: string;
781
+ } | {
782
+ type: 'tool_call_delta';
783
+ index: number;
784
+ id?: string;
785
+ name?: string;
786
+ argumentsDelta?: string;
787
+ /** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
788
+ complete?: boolean;
789
+ } | {
790
+ type: 'usage';
791
+ usage: {
792
+ prompt_tokens?: number;
793
+ completion_tokens?: number;
794
+ total_tokens?: number;
795
+ };
796
+ } | {
797
+ /**
798
+ * A provider keep-alive signal with no content of its own (e.g.
799
+ * Anthropic's `ping` events, an SSE comment-line heartbeat).
800
+ * Adapters yield this so the stream loop resets its idle timeout.
801
+ * Never surfaced to callers as a `StreamChunk`.
802
+ */
803
+ type: 'ping';
804
+ };
805
+ /**
806
+ * Parameters for a cached, streaming LLM call without tool calling.
807
+ *
808
+ * The cached value is `T`, same as `CachedCallParams<T>`, but a miss
809
+ * relays live `chunks` to the caller while the result is being generated,
810
+ * and a hit synthesizes a one-shot `chunks` replay from the cached value
811
+ * (see `VernLLM.cachedCall`'s docs for exactly what that replay looks
812
+ * like).
813
+ */
814
+ type CachedStreamCallParams<T> = CachedCallInput & {
815
+ call: StreamEnabledCallParams<T>;
816
+ };
817
+ /**
818
+ * Parameters for a cached, streaming LLM call with tool calling enabled.
819
+ *
820
+ * The cached value is the full `CallWithToolsResult<T>`, same as
821
+ * `CachedToolCallParams<T>`, with the same live-chunks-on-miss,
822
+ * replayed-chunks-on-hit behavior as `CachedStreamCallParams<T>`.
823
+ */
824
+ type CachedStreamToolCallParams<T> = CachedCallInput & {
825
+ call: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>;
826
+ };
827
+
828
+ //#endregion
829
+ //#region src/types/client.d.ts
830
+ //# sourceMappingURL=stream.d.ts.map
831
+ /** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
832
+ interface WireToolCall {
833
+ id: string;
834
+ type: 'function';
835
+ function: {
836
+ name: string;
837
+ /** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
838
+ arguments: string;
839
+ };
840
+ }
841
+ /** One entry of `LLMClient`'s `messages` array, named so callers building it can annotate against it. */
842
+ type WireMessage = {
843
+ role: 'system';
844
+ content: string;
845
+ } | {
846
+ role: 'user';
847
+ content: string | ContentBlock[];
848
+ } | {
849
+ role: 'assistant';
850
+ /** Optional: an assistant turn that only requested tools has no text. */
851
+ content?: string;
852
+ tool_calls?: WireToolCall[];
853
+ } | {
854
+ role: 'tool';
855
+ tool_call_id: string;
856
+ content: string;
857
+ /** Only honored by `fromAnthropic` today (maps to `tool_result.is_error`); other adapters ignore it. */
858
+ is_error?: boolean;
859
+ };
860
+ /** The OpenAI-shaped wire `tool_choice`. */
861
+ type WireToolChoice = 'auto' | 'none' | 'required' | {
862
+ type: 'function';
863
+ function: {
864
+ name: string;
865
+ };
866
+ };
216
867
  /**
217
868
  * Minimal shape compatible with the OpenAI SDKs chat.completions.create,
218
869
  * so consumers can pass an OpenAI client directly
@@ -226,7 +877,7 @@ interface LLMClient {
226
877
  completions: {
227
878
  create(params: {
228
879
  model: string;
229
- temperature: number;
880
+ temperature?: number;
230
881
  max_tokens: number;
231
882
  response_format?: {
232
883
  type: 'json_object';
@@ -241,19 +892,29 @@ interface LLMClient {
241
892
  };
242
893
  /** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
243
894
  reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
244
- messages: Array<{
245
- role: 'system' | 'assistant';
246
- content: string;
247
- } | {
248
- role: 'user';
249
- content: string | ContentBlock[];
895
+ /** Tools the model may call, OpenAI's `function`-wrapped shape. */
896
+ tools?: Array<{
897
+ type: 'function';
898
+ function: {
899
+ name: string;
900
+ description: string;
901
+ parameters: Record<string, unknown>;
902
+ };
250
903
  }>;
904
+ tool_choice?: WireToolChoice;
905
+ /**
906
+ * Wire-format messages. Breaking change for custom adapters:
907
+ * implementations must handle tool messages and assistant tool_calls.
908
+ * Exhaustive switches over only system/user/assistant roles may no longer compile.
909
+ */
910
+ messages: WireMessage[];
251
911
  }, options: {
252
912
  signal: AbortSignal;
253
913
  }): Promise<{
254
914
  choices?: Array<{
255
915
  message?: {
256
916
  content?: string | null;
917
+ tool_calls?: WireToolCall[];
257
918
  };
258
919
  }>;
259
920
  usage?: {
@@ -262,54 +923,22 @@ interface LLMClient {
262
923
  total_tokens?: number;
263
924
  };
264
925
  }>;
926
+ /**
927
+ * Optional. Required only for `stream: true` calls. Adapters/clients
928
+ * that don't implement this make `stream: true` throw a clear
929
+ * `LLMError('validation')` rather than a confusing runtime failure.
930
+ * Takes the same request shape as `create`, minus the response type.
931
+ */
932
+ createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
933
+ signal: AbortSignal;
934
+ }): AsyncIterable<WireStreamChunk>;
265
935
  };
266
936
  };
267
937
  }
268
938
 
269
- //#endregion
270
- //#region src/circuitBreaker.d.ts
271
- //# sourceMappingURL=client.d.ts.map
272
- interface CircuitBreakerOptions {
273
- /** Consecutive failures before the circuit opens, default 5 */
274
- threshold?: number;
275
- /** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
276
- cooldownMs?: number;
277
- }
278
- type CircuitState = 'closed' | 'open' | 'half-open';
279
- /**
280
- * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures across
281
- * calls. Once the threshold is hit, short-circuits new calls with an
282
- * LLMError('circuit_open') instead of hitting the provider, until the
283
- * cooldown elapses and a single trial call is allowed through
284
- */
285
- declare class CircuitBreaker {
286
- private state;
287
- private consecutiveFailures;
288
- private openedAt;
289
- private threshold;
290
- private cooldownMs;
291
- /**
292
- * True while a single half-open trial call is in flight. Guards against
293
- * multiple concurrent callers all treating themselves as "the" trial once
294
- * the cooldown elapses
295
- */
296
- private trialInFlight;
297
- constructor(options?: CircuitBreakerOptions);
298
- /**
299
- * Throws if the circuit is open and the cooldown hasn't elapsed, or if
300
- * the circuit is half-open and a trial call is already in flight.
301
- * Otherwise, if the circuit just became eligible for a trial (cooldown
302
- * elapsed, or half-open with no trial currently running), this call
303
- * becomes that trial
304
- */
305
- assertClosed(): void;
306
- recordSuccess(): void;
307
- recordFailure(): void;
308
- getState(): CircuitState;
309
- }
310
-
311
939
  //#endregion
312
940
  //#region src/logger.d.ts
941
+ //# sourceMappingURL=client.d.ts.map
313
942
  interface Logger {
314
943
  debug(message: string): void;
315
944
  warn(message: string): void;
@@ -328,22 +957,125 @@ declare class ConsoleLogger implements Logger {
328
957
  }
329
958
 
330
959
  //#endregion
331
- //#region src/types/options.d.ts
960
+ //#region src/types/events.d.ts
332
961
  //# sourceMappingURL=logger.d.ts.map
962
+ /**
963
+ * Reports what happened during a call. Fire and forget, mirroring
964
+ * `onUsage`: the return value is never read and a throwing handler cannot
965
+ * change what the call does, only what gets reported about it.
966
+ */
967
+ type VernLLMEvent = {
968
+ kind: 'retry';
969
+ requestId: string;
970
+ provider: string;
971
+ /** The model actually resolved for this call (honors a per-call `model` override). */
972
+ model: string;
973
+ /** The 1-based retry ordinal (the 1st retry is `1`, not the overall attempt count). */
974
+ attempt: number;
975
+ maxRetries: number;
976
+ delayMs: number;
977
+ retryAfterHonored: boolean;
978
+ error: LLMError;
979
+ } | {
980
+ kind: 'circuit_state';
981
+ provider: string;
982
+ /**
983
+ * The model of the call that triggered this specific transition
984
+ * (whatever was passed to the `assertClosed`/`recordSuccess`/
985
+ * `recordFailure` call that caused it), not a property of the
986
+ * circuit itself: the breaker still counts failures across every
987
+ * model together, so a threshold crossing can be the sum of
988
+ * several different models' failures even though only the
989
+ * triggering call's `model` is reported here.
990
+ */
991
+ model: string;
992
+ from: CircuitState;
993
+ to: CircuitState;
994
+ consecutiveFailures: number;
995
+ } | {
996
+ kind: 'fallback';
997
+ requestId: string;
998
+ /** Provider name of the target that just failed. */
999
+ from: string;
1000
+ /** Provider name of the target about to be tried next. */
1001
+ to: string;
1002
+ /** `-1` for the primary target, otherwise the index into `fallback`. */
1003
+ fromIndex: number;
1004
+ toIndex: number;
1005
+ /** The normalized error that caused `from` to be abandoned. */
1006
+ error: LLMError;
1007
+ /** Time spent on `from`, including its own retries, before giving up. */
1008
+ elapsedMs: number;
1009
+ } | {
1010
+ kind: 'rate_limited';
1011
+ requestId: string;
1012
+ provider: string;
1013
+ /** The model actually resolved for this call (honors a per-call `model` override). */
1014
+ model: string;
1015
+ /** How long this attempt sat queued for capacity before it was let through. */
1016
+ waitedMs: number;
1017
+ /** Which configured bucket was blocking this attempt just before it cleared. */
1018
+ reason: 'concurrency' | 'rpm' | 'tpm';
1019
+ };
1020
+ type OnEvent = (event: VernLLMEvent) => void;
1021
+
1022
+ //#endregion
1023
+ //#region src/types/options.d.ts
1024
+ //# sourceMappingURL=events.d.ts.map
333
1025
  interface VernLLMOptions {
334
1026
  client: LLMClient;
335
1027
  model: string;
1028
+ /**
1029
+ * Label for this provider in usage (`TokenUsage.provider`) and events.
1030
+ * Default `'primary'`.
1031
+ */
1032
+ name?: string;
336
1033
  /** Max retries after the first attempt. Default 1 (2 attempts total) */
337
1034
  maxRetries?: number;
338
1035
  /** Per-attempt timeout in ms. Default 25000 */
339
1036
  timeoutMs?: number;
1037
+ /**
1038
+ * For `stream: true` calls: max gap allowed between chunks once the
1039
+ * stream has opened, in ms. Resets on every chunk, including keep-alive
1040
+ * pings. `timeoutMs` only covers opening the stream and its first
1041
+ * chunk; this covers every gap after that. Also counts as a
1042
+ * circuit-breaker failure, unlike other mid-stream errors, since a
1043
+ * provider that streams one chunk then stalls should still trip it.
1044
+ * Default 30000. Pass 0 or negative to disable.
1045
+ */
1046
+ chunkIdleTimeoutMs?: number;
340
1047
  /** Base delay for exponential backoff in ms. Default 500 */
341
1048
  baseDelayMs?: number;
342
1049
  /** Default max_tokens for calls that don't override it. Default 1000 */
343
1050
  defaultMaxTokens?: number;
344
- /** Enables debug logging of raw model output (logs up to 800 chars of each
345
- * response). Off by default */
1051
+ /**
1052
+ * Default temperature for calls that don't override it. Default 0.2, not
1053
+ * the provider's own default. Pass `null` to omit `temperature` from the
1054
+ * request entirely, so the provider applies its own default instead.
1055
+ */
1056
+ defaultTemperature?: number | null;
1057
+ /**
1058
+ * Enables debug logging of raw model output (logs up to 800 chars of each
1059
+ * response) and provider errors. Off by default. Only controls the
1060
+ * default `ConsoleLogger`: when a custom `logger` is supplied instead,
1061
+ * that logger's own `debug()` implementation decides whether messages
1062
+ * are emitted, and this option has no effect on it.
1063
+ */
346
1064
  debug?: boolean;
1065
+ /**
1066
+ * Applied before every internal `logger.debug()` call: the raw output
1067
+ * logged on success, and the provider error logged on a failed call or
1068
+ * a failed stream open. This is the one piece of logging an app can't
1069
+ * intercept itself, since it's a direct call into `logger.debug`
1070
+ * rather than something routed through `onEvent`/`onUsage`; anything
1071
+ * caught elsewhere (events, `LLMError.cause`) already passes through
1072
+ * the app's own callback and can be redacted there instead. Runs
1073
+ * before `logger.debug()` regardless of whether that call ends up
1074
+ * emitting anything, so with a custom `logger`, `redact` still applies
1075
+ * even without `debug: true`; see `debug` for why. Default: identity
1076
+ * (no redaction).
1077
+ */
1078
+ redact?: (text: string) => string;
347
1079
  /** Cache adapter for cachedCall. Defaults to an in-memory adapter */
348
1080
  cache?: CacheAdapter;
349
1081
  /** HTTP status codes that should fail fast without retrying. Default [400, 401, 403, 404, 422] */
@@ -352,6 +1084,18 @@ interface VernLLMOptions {
352
1084
  parseJson?: (content: string) => unknown;
353
1085
  /** Called after every successful call with token usage, if the provider reports it */
354
1086
  onUsage?: OnUsage;
1087
+ /**
1088
+ * Called when a provider response arrives but VernLLM's own post-processing
1089
+ * then fails, after usage data was already present in that response.
1090
+ * Separate from `onUsage`, which only fires on full success.
1091
+ *
1092
+ * For non-streaming calls, never fires for transport failures (timeout,
1093
+ * network error, non-retryable status), since no response means no usage
1094
+ * to report. For streaming calls, this is not guaranteed: a stream can
1095
+ * deliver a usage chunk and then fail later (e.g. an idle timeout waiting
1096
+ * for the final close), in which case this does fire.
1097
+ */
1098
+ onUsageFailure?: OnUsageFailure;
355
1099
  /** Injectable logger. Defaults to a console-based logger gated by `debug` */
356
1100
  logger?: Logger;
357
1101
  /**
@@ -360,136 +1104,289 @@ interface VernLLMOptions {
360
1104
  * Pass `true` for defaults, or an options object to tune threshold/cooldown
361
1105
  */
362
1106
  circuitBreaker?: boolean | CircuitBreakerOptions;
1107
+ /**
1108
+ * Reports retries and circuit-breaker state transitions as they happen.
1109
+ * Fire and forget: a throwing handler is caught and logged, and its
1110
+ * return value is never read, so it cannot influence the call.
1111
+ */
1112
+ onEvent?: OnEvent;
1113
+ /**
1114
+ * Client-side rate limiting. Queues calls locally to stay under the
1115
+ * configured requests/tokens-per-minute or concurrency caps, instead of
1116
+ * letting the provider reject them. Independent of the `Retry-After`
1117
+ * handling already applied to a provider 429: this avoids tripping the
1118
+ * limit in the first place. Omit for unlimited (the default).
1119
+ */
1120
+ rateLimit?: RateLimitOptions;
1121
+ /**
1122
+ * Ordered targets tried after the primary, in order, once it (and its
1123
+ * own retries) is exhausted or abandoned. Order is the policy: VernLLM
1124
+ * never reorders, scores, or selects between targets. Each target keeps
1125
+ * its own retry state, circuit breaker, and rate limiter, independent
1126
+ * of every other target's. A single `FallbackTarget` is equivalent to
1127
+ * `[target]`.
1128
+ */
1129
+ fallback?: FallbackTarget | FallbackTarget[];
1130
+ /**
1131
+ * Decides what happens after a target fails: `'next'` to move on to
1132
+ * the following target (or throw, if it was the last one), `'stop'` to
1133
+ * give up immediately without trying any remaining targets. Called
1134
+ * once per failed target, after that target's own retries are
1135
+ * exhausted or abandoned early, so `'retry'` is never a valid return
1136
+ * here. Defaults to `defaultFallbackOn`, which stops on
1137
+ * parse/validation/aborted/quota errors and on tool-contract failures
1138
+ * (the model ignoring the request, not the provider being unhealthy),
1139
+ * and moves on for everything else.
1140
+ */
1141
+ fallbackOn?: FallbackOn;
363
1142
  }
364
1143
 
365
1144
  //#endregion
366
1145
  //#region src/vernLLM.d.ts
367
1146
  //# sourceMappingURL=options.d.ts.map
368
1147
  /**
369
- * A resilient layer around an LLM chat completions client, this is VernLLM!
1148
+ * A resilient layer around an LLM chat completions client. This is VernLLM!
370
1149
  *
371
- * Adds retry with backoff/jitter, per-attempt timeouts, an optional circuit breaker,
372
- * JSON parsing with optional schema validation, usage tracking, and an
373
- * optional response cache, all configurable, all opt-in beyond sensible
374
- * defaults.
1150
+ * Adds retry with backoff and jitter, per-attempt timeouts, an optional
1151
+ * circuit breaker, JSON parsing with optional schema validation, usage
1152
+ * tracking, and an optional response cache. All configurable, all opt-in
1153
+ * beyond sensible defaults.
375
1154
  */
376
1155
  declare class VernLLM {
377
- private readonly client;
378
- private readonly model;
379
- private readonly maxRetries;
380
- private readonly timeoutMs;
381
- private readonly baseDelayMs;
382
- private readonly defaultMaxTokens;
383
- private readonly cache;
384
- private readonly nonRetryableStatus;
385
- private readonly inFlight;
386
- private readonly parseJson;
387
- private readonly onUsage?;
388
1156
  private readonly logger;
389
- private readonly breaker?;
390
1157
  /**
391
- * @param options - Client, model, and tunables. Notable defaults:
392
- * `maxRetries` 1, `timeoutMs` 25000, `baseDelayMs` 500 (exponential backoff
393
- * base), `defaultMaxTokens` 1000, `cache` an in-memory adapter,
394
- * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
1158
+ * One `CallExecutor` per provider target: index 0 is the primary,
1159
+ * everything after it is a `fallback` target, in the order declared.
1160
+ * Each owns its own request building, retry/timeout, circuit breaker,
1161
+ * and rate limiter. `call()` walks this array in `runFallbackChain`,
1162
+ * moving to the next entry only when `fallbackOn` says to.
395
1163
  */
396
- constructor(options: VernLLMOptions);
397
- /** Resolves a cache key through the adapter when it supports normalization. */
398
- private resolveCacheKey;
1164
+ private readonly executors;
1165
+ /** Decides whether a failed target is followed by the next one or the chain stops. See `VernLLMOptions['fallbackOn']`. */
1166
+ private readonly fallbackOn;
1167
+ /** Reports a `'fallback'` event when the chain moves to the next target. Shared `onEvent` plumbing, same as every executor's. */
1168
+ private readonly reportEvent;
399
1169
  /**
400
- * Makes a single logical LLM call, retrying on failure per the configured
401
- * policy. Fails fast if the breaker is open or the signal is already
402
- * aborted. On exhausting retries, records a breaker failure and rejects
403
- * with a normalized LLMError.
404
- *
405
- * @param params - System/user content plus per-call overrides (model,
406
- * temperature, jsonMode, schema, signal, etc). See `CallParams`.
407
- * @returns The parsed (and optionally schema-validated) response, or the
408
- * raw string content when `jsonMode` is false and no `jsonSchema` is set.
1170
+ * Owns cache key resolution, cache reads/writes, and in-flight
1171
+ * coalescing for `cachedCall()`. Independent of `executor`: it only
1172
+ * ever calls back into `this.call()` as an opaque function.
409
1173
  */
410
- call<T = unknown>(params: CallParams<T>): Promise<T>;
411
- /** Runs `fn`, retrying with backoff according to `shouldRetry`. */
412
- private retryWithBackoff;
1174
+ private readonly cacheOrchestrator;
413
1175
  /**
414
- * Performs a single attempt: builds the request, dispatches it with a
415
- * timeout, and shapes the response. Throws on an empty response so the
416
- * retry loop treats it like any other transient failure.
1176
+ * @param options Client, model, and tunables. Defaults: `maxRetries` 1,
1177
+ * `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
1178
+ * `defaultTemperature` 0.2, `cache` an in-memory adapter,
1179
+ * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
417
1180
  */
418
- private executeCall;
1181
+ constructor(options: VernLLMOptions);
1182
+ /** Logs a failed refundUsage attempt via the configured logger. */
1183
+ private logRefundError;
419
1184
  /**
420
- * Validates `history` alternates user/assistant turns, since providers
421
- * like Anthropic/Gemini reject or mishandle consecutive same-role turns.
1185
+ * Walks `this.executors` in order, running `attempt` against each until
1186
+ * one succeeds or every target has failed. `run` on a lone target
1187
+ * (no `fallback` configured) throws exactly what it throws today: the
1188
+ * loop's single iteration path is unchanged from pre-fallback behavior.
1189
+ *
1190
+ * For streaming, `attempt` is `executor.runStream`, whose own retries
1191
+ * only cover *opening* the stream (see `CallExecutor.runStream`). A
1192
+ * mid-stream failure surfaces through `finalResult` after this function
1193
+ * has already returned, so it's never seen here and never falls over,
1194
+ * per the streaming limitation: splicing a second model's output into a
1195
+ * response the consumer has already partially rendered would corrupt
1196
+ * it.
422
1197
  */
423
- private validateHistory;
424
- /** Applies per-call defaults and shapes params into the client's request object. */
425
- private buildRequestPayload;
1198
+ private runFallbackChain;
426
1199
  /**
427
- * Chooses the response format: a provider-native `jsonSchema` takes
428
- * priority when supplied (constrains generation directly), otherwise
429
- * falls back to the looser `json_object` mode when JSON output is
430
- * requested, or no format at all for plain text responses.
1200
+ * Makes a single logical LLM call, retrying on failure per the configured
1201
+ * policy. Fails fast if the breaker is open or the signal is already
1202
+ * aborted. Rejects with a normalized LLMError on exhausted retries.
1203
+ *
1204
+ * When `tools` is set, returns a `CallWithToolsResult<T>` instead of `T`:
1205
+ * `{ type: 'content', content }` or `{ type: 'tool_calls', toolCalls,
1206
+ * content? }`. VernLLM never executes tools; run them yourself and
1207
+ * continue via `history` (see `ConversationTurn`). Mutually exclusive
1208
+ * with `jsonSchema`/`schema`.
1209
+ *
1210
+ * TypeScript only picks the tools-aware overload when `tools` is
1211
+ * statically present on `params`. If set conditionally on a plain
1212
+ * `CallParams<T>`, use `isToolCallResult()` to check the shape at
1213
+ * runtime instead. See the Tool Calling docs for details.
1214
+ *
1215
+ * The same static-vs-dynamic caveat applies to `stream`: TypeScript only
1216
+ * selects the streaming overload (returning `StreamCallResult<...>`) when
1217
+ * `stream: true` is statically present on `params`. A `stream` value set
1218
+ * conditionally on a plain `CallParams<T>` still resolves to `Promise<T>`
1219
+ * (or `Promise<CallWithToolsResult<T>>`) at the type level even though
1220
+ * the actual runtime result is the `{ chunks, finalResult }` streaming
1221
+ * shape whenever `stream` evaluates to `true`, callers doing this should
1222
+ * narrow/cast accordingly rather than relying on the static return type.
1223
+ *
1224
+ * @param params System/user content plus per-call overrides. See `CallParams`.
1225
+ * @returns Without `tools` or `stream`: the parsed response, or raw
1226
+ * string if `jsonMode` is false. With `tools`: a `CallWithToolsResult<T>`.
1227
+ * With `stream: true` (statically): a `{ chunks, finalResult }`
1228
+ * `StreamCallResult`, `finalResult` resolving to whichever of the above
1229
+ * shapes applies once the stream completes. See `StreamCallResult`.
431
1230
  */
432
- private buildResponseFormat;
433
- /** Reports token usage to `onUsage`, swallowing and logging any error it throws. */
434
- private recordUsage;
435
- /** Parses response content as JSON and validates it against `schema` when supplied. */
436
- private parseAndValidate;
1231
+ call<T = unknown>(params: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
1232
+ call<T = unknown>(params: StreamEnabledCallParams<T>): Promise<StreamCallResult<T>>;
1233
+ call<T = unknown>(params: ToolEnabledCallParams<T>): Promise<CallWithToolsResult<T>>;
1234
+ call<T = unknown>(params: CallParams<T>): Promise<T>;
437
1235
  /**
438
- * Waits out the backoff delay for a retry attempt, honoring a
439
- * Retry-After header on the failed attempt's error when present.
440
- * Both Retry-After and plain exponential backoff are capped at the same
441
- * max delay (see `DEFAULT_MAX_DELAY_MS` in `vernLLM.utils.ts`).
1236
+ * Thin delegator kept private on `VernLLM` (rather than only existing on
1237
+ * `CacheOrchestrator`) since it's the one caching primitive exercised
1238
+ * directly by white-box tests, independent of the public `cachedCall()`
1239
+ * surface.
442
1240
  */
443
- private recoverDelay;
444
- /** Decides whether a failed attempt is worth retrying. */
445
- private shouldRetry;
1241
+ private runCached;
446
1242
  /**
447
1243
  * Removes a cached response by key when the configured cache adapter
448
1244
  * supports deletion. Cache invalidation is the caller's responsibility;
449
1245
  * only the application knows when cached data is stale.
450
1246
  *
451
- * @param key - The raw cache key (resolved through the adapter's
1247
+ * @param key The raw cache key (resolved through the adapter's
452
1248
  * `resolveKey`, if any, before deletion).
453
1249
  */
454
1250
  deleteCache(key: string): Promise<void>;
455
1251
  /**
456
- * Cache wrapper around caller-supplied logic. Concurrent misses for the
457
- * same `cacheKey` share a single in-flight call, avoiding cache stampedes.
458
- *
459
- * @param params - `cacheKey`, `ttl`, `fn` (the work to run on a cache
460
- * miss, typically `() => this.call(...)`), and optional
461
- * `reserveUsage`/`refundUsage`/`signal`. See `CachedCallParams`.
462
- * @returns The cached value on a hit, or the result of `fn()` on a miss.
463
- */
464
- cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
465
- /** Starts the shared fn() call for a cache miss and tracks it in the in-flight map until it settles. */
466
- private registerTrigger;
467
- /** Runs `fn` and writes its result to the cache. */
468
- private runAndCache;
469
- /** Logs a failed refundUsage attempt via the configured logger. */
470
- private logRefundError;
471
- /**
472
- * Convenience wrapper composing `call` + `cachedCall`, so cached LLM calls
1252
+ * Cache wrapper composing `call` + caching, so cached LLM calls
473
1253
  * automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
474
- * `refundUsage` are read from the top-level params only.
1254
+ * `refundUsage` are read from the top-level params only. Concurrent misses
1255
+ * for the same `cacheKey` share a single in-flight call, avoiding cache
1256
+ * stampedes. Supports `stream: true` and `tools` in any combination.
475
1257
  *
476
- * @param params - `cachedCall` params (`cacheKey`, `ttl`, etc, minus `fn`)
477
- * plus `call`, the `CallParams` to pass through to `this.call(...)`.
1258
+ * When `call.tools` is set, this caches the whole `CallWithToolsResult`,
1259
+ * including `tool_calls` results, not just final answers. Whether
1260
+ * that's appropriate depends on the tool: caching "the model decided to
1261
+ * call get_weather" is usually fine to reuse briefly, but caching a
1262
+ * decision made under permissions or account state that can change
1263
+ * between calls is not. Use a short `ttl` or a separate `cacheKey` for
1264
+ * such tools if this distinction matters.
1265
+ *
1266
+ * There is no public way to cache an arbitrary non-LLM function through
1267
+ * `VernLLM`. This method always composes with `call()`. For
1268
+ * general-purpose caching unrelated to an LLM call, use a dedicated
1269
+ * caching library at the application level instead.
1270
+ *
1271
+ * @param params `cacheKey`, `ttl`, and optional
1272
+ * `reserveUsage`/`refundUsage`/`signal`, plus `call`, the `CallParams`
1273
+ * (optionally with `tools` and/or `stream`) to pass through to
1274
+ * `this.call(...)`. The top-level `signal` governs the cached operation
1275
+ * and its usage hooks only; to also abort the underlying provider
1276
+ * request, set `signal` inside `call`.
478
1277
  * @returns The cached value on a hit, or the freshly-called result on a miss.
479
1278
  */
480
- cachedLLMCall<T>(params: Omit<CachedCallParams<T>, 'fn'> & {
481
- call: CallParams<T>;
482
- }): Promise<T>;
1279
+ cachedCall<T>(params: CachedStreamToolCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
1280
+ cachedCall<T>(params: CachedStreamCallParams<T>): Promise<StreamCallResult<T>>;
1281
+ cachedCall<T>(params: CachedToolCallParams<T>): Promise<CallWithToolsResult<T>>;
1282
+ cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
483
1283
  /**
1284
+ * @param model With `circuitBreaker.isolateByModel` on, returns that
1285
+ * model's own circuit state instead of the shared one. Ignored
1286
+ * otherwise. Omit for the shared circuit (the default) or, under
1287
+ * isolation, the state of calls that didn't resolve a model.
484
1288
  * @returns The current circuit breaker state (`'closed' | 'open' |
485
1289
  * 'half-open'`), or undefined if no circuit breaker was configured.
486
1290
  */
487
- getCircuitState(): ("closed" | "open" | "half-open") | undefined;
1291
+ getCircuitState(model?: string): CircuitState | undefined;
1292
+ /**
1293
+ * @param model With `circuitBreaker.isolateByModel` on, returns each
1294
+ * target's circuit state for that model instead of its shared state.
1295
+ * Ignored otherwise. Omit for the shared circuit (the default) or, under
1296
+ * isolation, the state of calls that didn't resolve a model.
1297
+ * @returns The current circuit state for every target in declaration
1298
+ * order, including the primary and all fallback targets. Each entry
1299
+ * includes the target's provider name, chain index, whether it is a
1300
+ * fallback, and its circuit state, or undefined if that target has no
1301
+ * circuit breaker configured.
1302
+ */
1303
+ getCircuitStates(model?: string): TargetCircuitState[];
488
1304
  }
489
1305
 
490
1306
  //#endregion
491
- //#region src/adapters/anthropic.d.ts
1307
+ //#region src/adapters/internal/sse.d.ts
492
1308
  //# sourceMappingURL=vernLLM.d.ts.map
1309
+ /**
1310
+ * Parses a Server-Sent-Events byte/text stream into the JSON payload of
1311
+ * each `data:` frame, in arrival order. Generic over transport: works with
1312
+ * anything that hands back progressively-arriving `Uint8Array` or `string`
1313
+ * chunks via async iteration: native `fetch`'s `response.body` (wrapped
1314
+ * to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
1315
+ * Node `Readable` (already async-iterable, no wrapping needed), etc, so
1316
+ * this framing layer doesn't care which transport produced the bytes.
1317
+ *
1318
+ * Follows the SSE spec's frame-delimiting rules closely enough for LLM
1319
+ * streaming responses: frames are separated by a blank line, each frame
1320
+ * may carry one or more `data:` lines (joined with `\n` per spec when
1321
+ * there's more than one), `:`-prefixed lines are comments and ignored, and
1322
+ * other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
1323
+ * only needs the payload. A frame whose data is exactly `[DONE]` (the
1324
+ * sentinel several providers, notably OpenAI, send to mark stream end)
1325
+ * ends iteration without yielding it.
1326
+ *
1327
+ * Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
1328
+ * to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
1329
+ * alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
1330
+ * stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
1331
+ * lines.
1332
+ *
1333
+ * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
1334
+ * how malformed JSON is handled elsewhere in VernLLM.
1335
+ */
1336
+ declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
1337
+ /**
1338
+ * Sentinel yielded by `parseSseStream` for a comment-only frame (no
1339
+ * `data:` payload), the mechanism providers use for SSE keep-alive
1340
+ * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
1341
+ * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
1342
+ */
1343
+ declare const SSE_PING: unique symbol;
1344
+
1345
+ //#endregion
1346
+ //#region src/adapters/internal/imageFormat.d.ts
1347
+ //# sourceMappingURL=sse.d.ts.map
1348
+ /**
1349
+ * MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
1350
+ * the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
1351
+ * Converse all natively support, so a `ContentBlock[]` that validates for
1352
+ * one provider validates for all of them.
1353
+ */
1354
+ declare const SUPPORTED_IMAGE_MIME_TYPES: readonly ["image/png", "image/jpeg", "image/gif", "image/webp"];
1355
+ type SupportedImageMimeType = (typeof SUPPORTED_IMAGE_MIME_TYPES)[number];
1356
+
1357
+ //#endregion
1358
+ //#region src/adapters/internal/nativeStructuredOutput.d.ts
1359
+ /**
1360
+ * Validates an `ImageBlock.mimeType` against the shared supported set.
1361
+ * Throws a non-retryable `LLMError('validation')`, since an unsupported
1362
+ * mimeType is a permanent failure, retrying the same input can't fix it,
1363
+ * the same way a schema-validation or JSON-parse failure isn't retried.
1364
+ */
1365
+
1366
+ /**
1367
+ * A static allow-list or predicate naming which models support native,
1368
+ * schema-constrained output as its own request field — Anthropic's
1369
+ * `output_config.format`, Bedrock's `outputConfig.textFormat` — separate
1370
+ * from `tools`/`tool_choice`, so it can be combined with real,
1371
+ * caller-supplied `tools` in the same request.
1372
+ *
1373
+ * There is no built-in default list here. Which models support this is
1374
+ * Anthropic's and Bedrock's call to make, not this package's, and it
1375
+ * changes over time; hardcoding a guessed list would risk silently
1376
+ * routing a request onto a field a given model doesn't actually support,
1377
+ * trading a clear `LLMError('validation')` for a confusing error from the
1378
+ * provider instead. So this is opt-in: pass the model IDs you've verified
1379
+ * against the provider's own docs (or a predicate). Left unset, no model
1380
+ * is treated as native-capable, `jsonSchema` keeps using the older
1381
+ * forced-single-tool-call emulation, and `tools` + `jsonSchema` together
1382
+ * is rejected, exactly this package's behavior before native support was
1383
+ * added.
1384
+ */
1385
+ type ModelCapabilityOverride = string[] | ((model: string) => boolean);
1386
+
1387
+ //#endregion
1388
+ //#region src/adapters/anthropic.d.ts
1389
+ /** Resolves whether `model` is covered by a caller-supplied allow-list/predicate. */
493
1390
  /** Anthropic's native per-block content shape for a message. */
494
1391
  type AnthropicContentBlock = {
495
1392
  type: 'text';
@@ -498,9 +1395,19 @@ type AnthropicContentBlock = {
498
1395
  type: 'image';
499
1396
  source: {
500
1397
  type: 'base64';
501
- media_type: string;
1398
+ media_type: SupportedImageMimeType;
502
1399
  data: string;
503
1400
  };
1401
+ } | {
1402
+ type: 'tool_use';
1403
+ id: string;
1404
+ name: string;
1405
+ input: unknown;
1406
+ } | {
1407
+ type: 'tool_result';
1408
+ tool_use_id: string;
1409
+ content: string;
1410
+ is_error?: boolean;
504
1411
  };
505
1412
  /** Minimal structural type for the Anthropic SDK's `messages.create` */
506
1413
  interface AnthropicClient {
@@ -517,19 +1424,51 @@ interface AnthropicClient {
517
1424
  tools?: Array<{
518
1425
  name: string;
519
1426
  description?: string;
520
- input_schema: Record<string, unknown>;
1427
+ input_schema: {
1428
+ type: 'object';
1429
+ [key: string]: unknown;
1430
+ };
521
1431
  strict?: boolean;
522
1432
  }>;
523
1433
  tool_choice?: {
1434
+ type: 'auto';
1435
+ } | {
1436
+ type: 'any';
1437
+ } | {
1438
+ type: 'none';
1439
+ } | {
524
1440
  type: 'tool';
525
1441
  name: string;
526
1442
  };
1443
+ /**
1444
+ * Native, schema-constrained output: a separate request field from
1445
+ * `tools`/`tool_choice`, so it can be sent alongside real tool
1446
+ * calls. Only built by this adapter for models covered by
1447
+ * `nativeStructuredOutputModels` (opt-in, see
1448
+ * `AnthropicAdapterOptions`); other models keep getting
1449
+ * `jsonSchema` emulated as a forced single tool call, the
1450
+ * pre-existing behavior.
1451
+ *
1452
+ * Matches the real Anthropic API's `output_config.format` shape
1453
+ * exactly: just `type` and `schema`, no `name`/`description`/
1454
+ * `strict`. Those three exist on VernLLM's own `jsonSchema` API
1455
+ * (and are still forwarded on the legacy forced-tool-call path,
1456
+ * where they're real `Tool` fields), but the native structured-
1457
+ * output endpoint has no equivalent for any of them.
1458
+ */
1459
+ output_config?: {
1460
+ format: {
1461
+ type: 'json_schema';
1462
+ schema: Record<string, unknown>;
1463
+ };
1464
+ };
527
1465
  }, options: {
528
1466
  signal: AbortSignal;
529
1467
  }): Promise<{
530
1468
  content: Array<{
531
1469
  type: string;
532
1470
  text?: string;
1471
+ id?: string;
533
1472
  name?: string;
534
1473
  input?: unknown;
535
1474
  }>;
@@ -540,21 +1479,50 @@ interface AnthropicClient {
540
1479
  }>;
541
1480
  };
542
1481
  }
1482
+ /** Optional configuration for `fromAnthropic`. */
1483
+ interface AnthropicAdapterOptions {
1484
+ /**
1485
+ * Which models support native, schema-constrained output
1486
+ * (`output_config.format`), independent of `tools`/`tool_choice`, so it
1487
+ * can be combined with real `tools` in one request. Pass a static list
1488
+ * of model IDs (verified against Anthropic's own docs) or a predicate.
1489
+ *
1490
+ * There is no built-in default here (see `supportsNativeStructuredOutput`
1491
+ * for why). Left unset, every model uses the older forced-single-tool-
1492
+ * call emulation, and `tools` + `jsonSchema` together is rejected,
1493
+ * exactly this adapter's behavior before native support was added.
1494
+ */
1495
+ nativeStructuredOutputModels?: ModelCapabilityOverride;
1496
+ }
543
1497
  /**
544
1498
  * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
545
1499
  * interface VernLLM uses for OpenAI/Groq.
546
1500
  *
547
- * `response_format: json_schema` is mapped to Anthropic's forced tool-use:
548
- * a single tool is defined with `input_schema` set to the caller's schema,
549
- * `description` forwarded when provided, and `strict` forwarded when set.
550
- * `tool_choice` forces the model to call it. Provider-constrained schema
551
- * matching applies only when `strict: true` is forwarded and supported.
1501
+ * `response_format: json_schema`, on a model covered by
1502
+ * `options.nativeStructuredOutputModels`, is sent as `output_config.format`,
1503
+ * its own request field, independent of `tools`/`tool_choice`, so it can be
1504
+ * combined with real, caller-supplied `tools` in the same request. Only
1505
+ * `type` and `schema` are sent on this path, the real Anthropic API's
1506
+ * `output_config.format` has no `name`/`description`/`strict` fields.
1507
+ *
1508
+ * On any other model (the default, since `nativeStructuredOutputModels` is
1509
+ * opt-in), `response_format: json_schema` is mapped to Anthropic's forced
1510
+ * tool-use instead: a single tool is defined with `input_schema` set to
1511
+ * the caller's schema, `description` forwarded when provided, and `strict`
1512
+ * forwarded when set, and `tool_choice` forces the model to call it. This
1513
+ * legacy path cannot be combined with real `tools` (both would need the
1514
+ * same `tools`/`tool_choice` field), and a call that tries throws
1515
+ * `LLMError('validation')` before reaching the API. Provider-constrained
1516
+ * schema matching applies only when `strict: true` is forwarded and
1517
+ * supported.
552
1518
  *
553
1519
  * `response_format: json_object` (no schema to build a tool from) falls
554
1520
  * back to a system-prompt instruction, since there's nothing to constrain
555
- * generation against.
1521
+ * generation against. Unlike `jsonSchema`, this combines with real `tools`
1522
+ * freely on every model: it's a prompt nudge, not a request field, so
1523
+ * there's nothing for it to collide with.
556
1524
  */
557
- declare function fromAnthropic(anthropicClient: AnthropicClient): LLMClient;
1525
+ declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
558
1526
 
559
1527
  //#endregion
560
1528
  //#region src/adapters/gemini.d.ts
@@ -566,13 +1534,30 @@ type GeminiPart = {
566
1534
  mimeType: string;
567
1535
  data: string;
568
1536
  };
1537
+ } | {
1538
+ functionCall: {
1539
+ name: string;
1540
+ args: unknown;
1541
+ };
1542
+ } | {
1543
+ functionResponse: {
1544
+ name: string;
1545
+ response: unknown;
1546
+ };
569
1547
  };
570
1548
  /**
571
- * Minimal structural type for VernLLM's two-argument wrapper around Gemini
572
- * `generateContent`. The wrapper exposes a request shape aligned with the
573
- * adapter interface, with top-level `systemInstruction` and
574
- * `generationConfig` fields, while transport options (such as `AbortSignal`)
575
- * are passed separately as the second argument.
1549
+ * Structural type matching the real `@google/genai` SDK's `ai.models`
1550
+ * object: `generateContent`/`generateContentStream` both take a single
1551
+ * `{ model, contents, config }` argument (config carries
1552
+ * `systemInstruction`, `tools`, `toolConfig`, generation settings, and
1553
+ * `abortSignal` all together), matching the real SDK closely enough that
1554
+ * `fromGemini(ai.models)` works directly, e.g:
1555
+ *
1556
+ * ```ts
1557
+ * import { GoogleGenAI } from '@google/genai';
1558
+ * const ai = new GoogleGenAI({ apiKey: '...' });
1559
+ * const llm = new VernLLM({ client: fromGemini(ai.models), model: 'gemini-2.5-flash' });
1560
+ * ```
576
1561
  */
577
1562
  interface GeminiClient {
578
1563
  generateContent(params: {
@@ -581,24 +1566,40 @@ interface GeminiClient {
581
1566
  role: 'user' | 'model';
582
1567
  parts: GeminiPart[];
583
1568
  }>;
584
- systemInstruction?: {
585
- parts: Array<{
586
- text: string;
587
- }>;
588
- };
589
- generationConfig?: {
1569
+ config?: {
1570
+ systemInstruction?: {
1571
+ parts: Array<{
1572
+ text: string;
1573
+ }>;
1574
+ };
590
1575
  temperature?: number;
591
1576
  maxOutputTokens?: number;
592
1577
  responseMimeType?: string;
593
1578
  responseSchema?: Record<string, unknown>;
1579
+ tools?: Array<{
1580
+ functionDeclarations: Array<{
1581
+ name: string;
1582
+ description?: string;
1583
+ parameters: Record<string, unknown>;
1584
+ }>;
1585
+ }>;
1586
+ toolConfig?: {
1587
+ functionCallingConfig: {
1588
+ mode: 'AUTO' | 'ANY' | 'NONE';
1589
+ allowedFunctionNames?: string[];
1590
+ };
1591
+ };
1592
+ abortSignal?: AbortSignal;
594
1593
  };
595
- }, options: {
596
- signal: AbortSignal;
597
1594
  }): Promise<{
598
1595
  candidates?: Array<{
599
1596
  content?: {
600
1597
  parts?: Array<{
601
1598
  text?: string;
1599
+ functionCall?: {
1600
+ name: string;
1601
+ args: unknown;
1602
+ };
602
1603
  }>;
603
1604
  };
604
1605
  }>;
@@ -608,6 +1609,32 @@ interface GeminiClient {
608
1609
  totalTokenCount?: number;
609
1610
  };
610
1611
  }>;
1612
+ /**
1613
+ * Optional. Required only for `stream: true` calls. Takes the same
1614
+ * request shape as `generateContent`. Matching the real SDK's own
1615
+ * `generateContentStream`, this resolves to an `AsyncIterable` (rather
1616
+ * than returning one synchronously) of partial responses, each chunk
1617
+ * holding the same `candidates[].content.parts[]` structure as
1618
+ * `generateContent`'s response, just incremental.
1619
+ */
1620
+ generateContentStream?(params: Parameters<GeminiClient['generateContent']>[0]): Promise<AsyncIterable<{
1621
+ candidates?: Array<{
1622
+ content?: {
1623
+ parts?: Array<{
1624
+ text?: string;
1625
+ functionCall?: {
1626
+ name: string;
1627
+ args: unknown;
1628
+ };
1629
+ }>;
1630
+ };
1631
+ }>;
1632
+ usageMetadata?: {
1633
+ promptTokenCount?: number;
1634
+ candidatesTokenCount?: number;
1635
+ totalTokenCount?: number;
1636
+ };
1637
+ }>>;
611
1638
  }
612
1639
  /**
613
1640
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
@@ -619,6 +1646,25 @@ interface GeminiClient {
619
1646
  * `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
620
1647
  * models use a token budget, not an effort tier, so it's dropped, same as
621
1648
  * Anthropic.
1649
+ *
1650
+ * `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
1651
+ * `tool_choice` maps to `toolConfig.functionCallingConfig`. Gemini accepts
1652
+ * `responseSchema` and `tools` in the same request natively, so both are
1653
+ * set independently here and no special-casing is needed for the
1654
+ * combination, unlike `fromAnthropic`/`fromBedrock`.
1655
+ *
1656
+ * `createStream` calls `generateContentStream` (optional on `GeminiClient`
1657
+ *, required only if the caller sets `stream: true`) and translates each
1658
+ * partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
1659
+ * Gemini's own function-calling API doesn't stream tool-call arguments
1660
+ * incrementally: a `functionCall` part always arrives whole in one chunk,
1661
+ * so each one is emitted as a single, complete `tool_call_delta` (a
1662
+ * one-shot "delta" containing the full arguments) rather than accumulated
1663
+ * fragments, that's a real difference in the underlying API, not
1664
+ * something this adapter can smooth over. `usageMetadata` is (per Gemini's
1665
+ * own behavior) only reliably present on the last chunk, so the `usage`
1666
+ * `WireStreamChunk` is emitted once, after the stream completes, from
1667
+ * whichever chunk's `usageMetadata` was seen last.
622
1668
  */
623
1669
  declare function fromGemini(geminiClient: GeminiClient): LLMClient;
624
1670
 
@@ -636,6 +1682,20 @@ type BedrockContentBlock = {
636
1682
  bytes: Uint8Array;
637
1683
  };
638
1684
  };
1685
+ } | {
1686
+ toolUse: {
1687
+ toolUseId: string;
1688
+ name: string;
1689
+ input: unknown;
1690
+ };
1691
+ } | {
1692
+ toolResult: {
1693
+ toolUseId: string;
1694
+ content: Array<{
1695
+ text: string;
1696
+ }>;
1697
+ status?: 'success' | 'error';
1698
+ };
639
1699
  };
640
1700
  /**
641
1701
  * Minimal structural type matching AWS Bedrock's Converse API. This is
@@ -682,6 +1742,38 @@ interface BedrockConverseClient {
682
1742
  tool: {
683
1743
  name: string;
684
1744
  };
1745
+ } | {
1746
+ auto: Record<string, never>;
1747
+ } | {
1748
+ any: Record<string, never>;
1749
+ };
1750
+ };
1751
+ /**
1752
+ * Native, schema-constrained output: a separate request field from
1753
+ * `toolConfig`, so it can be sent alongside real tool calls. Only
1754
+ * built by this adapter for models covered by
1755
+ * `nativeStructuredOutputModels` (opt-in, see
1756
+ * `BedrockAdapterOptions`); other models keep getting `jsonSchema`
1757
+ * emulated as a forced single tool call via `toolConfig`, the
1758
+ * pre-existing behavior.
1759
+ *
1760
+ * Matches the real Bedrock Converse API's `outputConfig.textFormat`
1761
+ * shape exactly: the schema itself is nested one level deeper, under
1762
+ * `structure.jsonSchema`, not flat on `textFormat`, and `schema` is
1763
+ * a JSON-encoded *string*, not a parsed object, unlike every other
1764
+ * schema field this adapter builds (`toolSpec.inputSchema.json`
1765
+ * included). There is no `strict` field here, unlike `toolSpec`.
1766
+ */
1767
+ outputConfig?: {
1768
+ textFormat: {
1769
+ type: 'json_schema';
1770
+ structure: {
1771
+ jsonSchema: {
1772
+ schema: string;
1773
+ name?: string;
1774
+ description?: string;
1775
+ };
1776
+ };
685
1777
  };
686
1778
  };
687
1779
  }, options: {
@@ -692,6 +1784,7 @@ interface BedrockConverseClient {
692
1784
  content?: Array<{
693
1785
  text?: string;
694
1786
  toolUse?: {
1787
+ toolUseId?: string;
695
1788
  name?: string;
696
1789
  input?: unknown;
697
1790
  };
@@ -704,24 +1797,122 @@ interface BedrockConverseClient {
704
1797
  totalTokens?: number;
705
1798
  };
706
1799
  }>;
1800
+ /**
1801
+ * Optional. Required only for `stream: true` calls. Takes the same
1802
+ * request shape `converse` does, returning `{ stream }`, matching
1803
+ * `ConverseStreamCommand`'s real AWS SDK v3 output shape, an
1804
+ * `AsyncIterable` of incremental events under a `stream` property,
1805
+ * rather than the whole response being the iterable directly.
1806
+ */
1807
+ converseStream?(params: Parameters<BedrockConverseClient['converse']>[0], options: {
1808
+ signal: AbortSignal;
1809
+ }): Promise<{
1810
+ stream: AsyncIterable<BedrockConverseStreamEvent>;
1811
+ }>;
707
1812
  }
1813
+ /**
1814
+ * One event of a Bedrock `ConverseStreamCommand` response's `stream`.
1815
+ * Content blocks (text or toolUse) are identified by `contentBlockIndex`,
1816
+ * Converse's own convention for correlating start/delta/stop events across
1817
+ * possibly-interleaved blocks, mirrored directly by VernLLM's
1818
+ * `tool_call_delta.index`.
1819
+ */
1820
+ type BedrockConverseStreamEvent = {
1821
+ messageStart: {
1822
+ role: 'assistant';
1823
+ };
1824
+ } | {
1825
+ contentBlockStart: {
1826
+ contentBlockIndex: number;
1827
+ start?: {
1828
+ toolUse?: {
1829
+ toolUseId?: string;
1830
+ name?: string;
1831
+ };
1832
+ };
1833
+ };
1834
+ } | {
1835
+ contentBlockDelta: {
1836
+ contentBlockIndex: number;
1837
+ delta?: {
1838
+ text?: string;
1839
+ } | {
1840
+ toolUse?: {
1841
+ input?: string;
1842
+ };
1843
+ };
1844
+ };
1845
+ } | {
1846
+ contentBlockStop: {
1847
+ contentBlockIndex: number;
1848
+ };
1849
+ } | {
1850
+ messageStop: {
1851
+ stopReason?: string;
1852
+ };
1853
+ } | {
1854
+ metadata: {
1855
+ usage?: {
1856
+ inputTokens?: number;
1857
+ outputTokens?: number;
1858
+ totalTokens?: number;
1859
+ };
1860
+ };
1861
+ } | {
1862
+ internalServerException: {
1863
+ message?: string;
1864
+ };
1865
+ } | {
1866
+ modelStreamErrorException: {
1867
+ message?: string;
1868
+ originalStatusCode?: number;
1869
+ };
1870
+ } | {
1871
+ validationException: {
1872
+ message?: string;
1873
+ };
1874
+ } | {
1875
+ throttlingException: {
1876
+ message?: string;
1877
+ };
1878
+ } | {
1879
+ serviceUnavailableException: {
1880
+ message?: string;
1881
+ };
1882
+ };
708
1883
  /**
709
1884
  * Optional configuration for `fromBedrock`.
710
1885
  */
711
1886
  interface BedrockAdapterOptions {
712
1887
  /**
713
- * Optional preflight check for tool-use support, needed for `jsonSchema`
714
- * structured output. VernLLM never guesses capability from a failed
715
- * call's error message (AWS's error text isn't a documented, stable
716
- * contract), so this is opt-in: pass either a static list of tool-use
717
- * -capable model IDs, or a predicate function, and VernLLM will reject
718
- * unsupported models with a clear `LLMError('validation')` *before*
719
- * dispatching the request, instead of on the wire.
1888
+ * Optional preflight check for tool-use support, needed whenever a
1889
+ * `jsonSchema` call ends up sending Converse `toolConfig` — either the
1890
+ * legacy forced-single-tool-call emulation, or real `tools` sent
1891
+ * alongside native structured output (`outputConfig`). VernLLM never
1892
+ * guesses capability from a failed call's error message (AWS's error
1893
+ * text isn't a documented, stable contract), so this is opt-in: pass
1894
+ * either a static list of tool-use-capable model IDs, or a predicate
1895
+ * function, and VernLLM will reject unsupported models with a clear
1896
+ * `LLMError('validation')` *before* dispatching the request, instead of
1897
+ * on the wire.
720
1898
  *
721
1899
  * Left unset (default), no preflight check runs, and a `jsonSchema` call
722
1900
  * to an unsupported model surfaces Bedrock's raw `converse` error as-is.
723
1901
  */
724
1902
  toolUseSupportedModels?: string[] | ((modelId: string) => boolean);
1903
+ /**
1904
+ * Which models support native, schema-constrained output
1905
+ * (`outputConfig.textFormat`), independent of `toolConfig`, so it can be
1906
+ * combined with real `tools` in one request. Pass a static list of
1907
+ * model IDs (verified against Bedrock's own docs) or a predicate.
1908
+ *
1909
+ * There is no built-in default here (see `supportsNativeStructuredOutput`
1910
+ * for why). Left unset, every model uses the older forced-single-tool-
1911
+ * call emulation via `toolConfig`, and `tools` + `jsonSchema` together is
1912
+ * rejected, exactly this adapter's behavior before native support was
1913
+ * added.
1914
+ */
1915
+ nativeStructuredOutputModels?: ModelCapabilityOverride;
725
1916
  }
726
1917
  /**
727
1918
  * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
@@ -731,18 +1922,48 @@ interface BedrockAdapterOptions {
731
1922
  * regardless of which underlying model `modelId` points at, as long as
732
1923
  * that model supports Converse (most current-generation ones do)
733
1924
  *
734
- * `response_format: json_schema` is mapped to Converse's `toolConfig`: a
735
- * single tool is defined from the schema, description, and strictness settings,
736
- * and `toolChoice` forces the model to call it. Provider-constrained schema
737
- * matching applies only when `strict: true` is forwarded and supported.
738
- * Native tool support varies by model family; pass
739
- * `toolUseSupportedModels` to preflight-check it (see
1925
+ * `response_format: json_schema`, on a model covered by
1926
+ * `options.nativeStructuredOutputModels` (opt-in, unset by default), is
1927
+ * sent as `outputConfig.textFormat`, its own request field, independent of
1928
+ * `toolConfig`, so it can be combined with real, caller-supplied `tools`
1929
+ * in the same request. Matches the real Converse API's shape exactly: the
1930
+ * schema is nested under `structure.jsonSchema` and JSON-encoded as a
1931
+ * string, not the parsed object `toolConfig`'s tool schemas use, and there
1932
+ * is no `strict` field on this path.
1933
+ *
1934
+ * On any other model (the default), `response_format: json_schema` is
1935
+ * mapped to Converse's `toolConfig` instead: a single tool is defined from
1936
+ * the schema, description, and strictness settings, and `toolChoice`
1937
+ * forces the model to call it. This legacy path cannot be combined with
1938
+ * real `tools` (both would need the same `toolConfig`), and a call that
1939
+ * tries throws `LLMError('validation')` before reaching the API.
1940
+ * Provider-constrained schema matching applies only when `strict: true` is
1941
+ * forwarded and supported. Native tool support varies by model family;
1942
+ * pass `toolUseSupportedModels` to preflight-check it (see
740
1943
  * `BedrockAdapterOptions`), otherwise a `jsonSchema` call to an
741
1944
  * unsupported model surfaces Bedrock's raw error unchanged.
742
1945
  *
743
1946
  * `response_format: json_object` (no schema to build a tool from) and
744
1947
  * `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
745
- * instruction and are dropped respectively.
1948
+ * instruction and are dropped respectively. Unlike `jsonSchema`,
1949
+ * `json_object` combines with real `tools` freely on every model: it's a
1950
+ * prompt nudge, not a request field, so there's nothing for it to collide
1951
+ * with.
1952
+ *
1953
+ * `tools` alone maps to Converse's native `toolConfig`/`toolUse`/
1954
+ * `toolResult`; `tool_choice` maps to `toolConfig.toolChoice`.
1955
+ *
1956
+ * `createStream` calls `converseStream` (optional on `BedrockConverseClient`
1957
+ *, required only if the caller sets `stream: true`) and translates its
1958
+ * `contentBlockStart`/`contentBlockDelta`/`metadata` events into
1959
+ * `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
1960
+ * same as `fromAnthropic`'s block-index tracking (Converse's streaming
1961
+ * shape is structurally close to Anthropic's own, both being tool-use-aware
1962
+ * content-block streams), including the same `json-tool` unwrapping: a
1963
+ * `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
1964
+ * `text-delta`, not `tool_call_delta`, so the accumulated result lands in
1965
+ * `finalizeResponse`'s `content` path exactly like the non-streaming
1966
+ * `create` branch above unwraps it.
746
1967
  */
747
1968
  declare function fromBedrock(bedrockClient: BedrockConverseClient, options?: BedrockAdapterOptions): LLMClient;
748
1969
 
@@ -772,6 +1993,22 @@ type RequestLike = (url: string, init: {
772
1993
  body?: string;
773
1994
  signal?: AbortSignal;
774
1995
  }) => Promise<ResponseLike>;
1996
+ /**
1997
+ * A streaming-capable request function. Unlike `RequestLike`, which returns
1998
+ * a fully-buffered `ResponseLike`, this resolves to an `AsyncIterable` of
1999
+ * progressively-arriving chunks, the common ground across transports:
2000
+ * native `fetch`'s `response.body` (wrapped to be iterable; see
2001
+ * `webStreamToAsyncIterable` below), axios's Node `Readable` in
2002
+ * `responseType: 'stream'` mode (already async-iterable, no wrapping
2003
+ * needed), `node-fetch`, `undici`, etc, all satisfy this with little or no
2004
+ * glue code. Defaults to native `fetch`.
2005
+ */
2006
+ type StreamRequestLike = (url: string, init: {
2007
+ method: string;
2008
+ headers: Record<string, string>;
2009
+ body?: string;
2010
+ signal?: AbortSignal;
2011
+ }) => Promise<AsyncIterable<Uint8Array | string>>;
775
2012
  interface FetchAdapterConfig {
776
2013
  /** Endpoint URL, or a function of the request in case it depends on model/params */
777
2014
  url: string | ((params: ChatRequest) => string);
@@ -788,17 +2025,65 @@ interface FetchAdapterConfig {
788
2025
  /** Maps VernLLMs internal chat-completion request into the providers raw request body */
789
2026
  mapRequest: (params: ChatRequest) => unknown;
790
2027
  /**
791
- * Maps the providers raw JSON response into `{ content, usage? }`
792
- * `content` is the assistants text (JSON string when JSON mode was requested)
2028
+ * Maps the providers raw JSON response into `{ content, usage?, toolCalls? }`
2029
+ * `content` is the assistants text (JSON string when JSON mode was requested).
2030
+ * `content` may be empty/omitted when the model responded with only tool
2031
+ * calls and no text.
2032
+ *
2033
+ * `toolCalls`, when the model requested one or more tools, is the list of
2034
+ * calls as flat `{ id, name, arguments }` entries (matching this config's
2035
+ * own `toolCalls?: Array<{ id: string; name: string; arguments: string }>`
2036
+ * return type below), each entry's `arguments` already JSON-*encoded* as a
2037
+ * string (not the parsed object), mirroring the wire format every
2038
+ * OpenAI-compatible provider uses. `fromFetch` itself converts these into
2039
+ * `WireToolCall`'s `type`/`function`-wrapped shape before returning them
2040
+ * from `create`. VernLLM parses (and validates, if `argumentsSchema` was
2041
+ * set) the arguments string internally, mapResponse doesn't need to do
2042
+ * that itself.
793
2043
  */
794
2044
  mapResponse: (json: unknown) => {
795
- content: string;
2045
+ content?: string;
796
2046
  usage?: {
797
2047
  promptTokens?: number;
798
2048
  completionTokens?: number;
799
2049
  totalTokens?: number;
800
2050
  };
2051
+ toolCalls?: Array<{
2052
+ id: string;
2053
+ name: string;
2054
+ arguments: string;
2055
+ }>;
801
2056
  };
2057
+ /**
2058
+ * Optional. Required only for `stream: true` calls. The function used to
2059
+ * open a streaming HTTP request. Takes the same request shape as
2060
+ * `request`, but resolves to an `AsyncIterable` of progressively-arriving
2061
+ * `Uint8Array` or `string` chunks instead of a buffered `ResponseLike`.
2062
+ * Defaults to native `fetch`.
2063
+ */
2064
+ requestStream?: StreamRequestLike;
2065
+ /**
2066
+ * Optional. How the raw stream bytes are split into individual event
2067
+ * payloads. Defaults to Server-Sent Events framing (`data: ...` blocks
2068
+ * separated by a blank line, `[DONE]` sentinel honored, see
2069
+ * `parseSseStream`), which covers the large majority of LLM providers'
2070
+ * streaming HTTP endpoints. Override this for a provider that frames its
2071
+ * stream differently, e.g. newline-delimited JSON (NDJSON) with no SSE
2072
+ * envelope.
2073
+ */
2074
+ parseStreamFrames?: (chunks: AsyncIterable<Uint8Array | string>) => AsyncIterable<unknown>;
2075
+ /**
2076
+ * Optional. Required only for `stream: true` calls. Maps one parsed
2077
+ * stream event (already extracted from its frame by `parseStreamFrames`)
2078
+ * into zero, one, or more `WireStreamChunk`s, mirrors `mapResponse`'s
2079
+ * role for the non-streaming path, just per-event instead of once for
2080
+ * the whole body. Return `undefined` to skip an event that carries
2081
+ * nothing VernLLM needs (e.g. a provider's keep-alive ping). Configs
2082
+ * that don't implement this make `stream: true` throw a clear
2083
+ * `LLMError('validation')` rather than a confusing runtime failure or a
2084
+ * silently empty stream.
2085
+ */
2086
+ mapStreamEvent?: (event: unknown) => WireStreamChunk | WireStreamChunk[] | undefined;
802
2087
  }
803
2088
  /**
804
2089
  * A fetch-based escape hatch for providers with no SDK, or where pulling one
@@ -810,6 +2095,34 @@ interface FetchAdapterConfig {
810
2095
  * Non-2xx responses throw an error with `.status` set to the HTTP status
811
2096
  * code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
812
2097
  * 401/403) applies here too
2098
+ *
2099
+ * Tool calling works the same way as every other adapter: `mapRequest`
2100
+ * receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
2101
+ * translate them into whatever shape the provider's wire format expects
2102
+ * (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
2103
+ * field). On the way back, `mapResponse` may return a `toolCalls` array
2104
+ * (id/name/JSON-encoded-arguments-string per call) alongside or instead of
2105
+ * `content`; VernLLM parses and (if `argumentsSchema` was set) validates
2106
+ * those arguments the same way it does for every other adapter. For
2107
+ * `stream: true`, tool-call deltas go through the existing
2108
+ * `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
2109
+ * no separate config is needed for streaming vs non-streaming tool calls.
2110
+ *
2111
+
2112
+ * `createStream` requires `mapStreamEvent` (there's no non-streaming
2113
+ * response to fall back on, unlike the other three optional streaming
2114
+ * seams). It opens the request via `requestStream` (defaults to native
2115
+ * `fetch`), splits the raw bytes into individual events via
2116
+ * `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
2117
+ * and translates each event into `WireStreamChunk`(s) via
2118
+ * `mapStreamEvent`. Both seams are overridable per-config for providers
2119
+ * that don't fit the SSE-over-fetch default. If a custom `request`
2120
+ * transport is configured, `requestStream` must be configured too,
2121
+ * `requestStream` never silently falls back to `request` (see
2122
+ * `createStream`'s own comment for why), so a `stream: true` call with
2123
+ * `request` set but no `requestStream` throws a clear
2124
+ * `LLMError('validation')` instead of quietly using unrelated native
2125
+ * `fetch`.
813
2126
  */
814
2127
  declare function fromFetch(config: FetchAdapterConfig): LLMClient;
815
2128
 
@@ -834,11 +2147,46 @@ declare function fromFetch(config: FetchAdapterConfig): LLMClient;
834
2147
  * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
835
2148
  * the actual compatibility contract is the JSON each provider sends and
836
2149
  * receives over the wire, not the SDKs TS types.
2150
+ *
2151
+ * `createStream` is implemented by calling the same underlying
2152
+ * `chat.completions.create` with `stream: true` (and, for providers that
2153
+ * support it, `stream_options: { include_usage: true }`, so a final usage
2154
+ * block arrives), the OpenAI SDK, and every OpenAI-compatible client
2155
+ * modeled on it, returns an `AsyncIterable` of SSE chunks instead of a
2156
+ * single completion object when `stream: true` is set. Each chunk is
2157
+ * translated into `WireStreamChunk`(s) via `toWireStreamChunks`.
2158
+ *
2159
+ * Note on long-running reasoning models: this adapter consumes the
2160
+ * underlying SDK's already-parsed stream rather than raw SSE bytes, so
2161
+ * unlike `fromFetch`/`fromAnthropic` it cannot see comment-only keep-alive
2162
+ * ping frames. Combined with `chunkIdleTimeoutMs`'s 30 second default and
2163
+ * `reasoningEffort` (documented to have long silent gaps for o-series and
2164
+ * similar models), a long-running reasoning call on this adapter can trip
2165
+ * the idle timeout even though the provider is still working. Raise or
2166
+ * disable `chunkIdleTimeoutMs` per call for those routes, see `CallParams`.
837
2167
  */
838
- declare function fromOpenAICompatible(client: unknown): LLMClient;
2168
+ interface OpenAICompatibleAdapterOptions {
2169
+ /**
2170
+ * Whether the provider supports `stream_options.include_usage`. Not
2171
+ * every "OpenAI-compatible" provider is guaranteed to, so this defaults
2172
+ * to `true` (matching OpenAI, Groq, Mistral, and most others observed)
2173
+ * and should be set to `false` for a provider verified not to support
2174
+ * it. When `false`, `stream_options` is omitted entirely and no usage
2175
+ * block will arrive on the stream; callers relying on streamed `usage`
2176
+ * with such a provider won't get one.
2177
+ */
2178
+ supportsStreamUsage?: boolean;
2179
+ }
2180
+ declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
839
2181
  /** Groqs SDK matches the OpenAI wire format */
840
2182
  declare const fromGroq: typeof fromOpenAICompatible;
841
- /** Mistrals `chat.completions`-shaped client (or their OpenAI-compat endpoint) */
2183
+ /**
2184
+ * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
2185
+ * endpoint). Mistral supports `stream_options.include_usage` (added after
2186
+ * an earlier period where it returned a 422 for unrecognized fields, per
2187
+ * Mistral's changelog and streaming docs), so this is a plain alias like
2188
+ * the others, `supportsStreamUsage` defaults to `true`.
2189
+ */
842
2190
  declare const fromMistral: typeof fromOpenAICompatible;
843
2191
  /** DeepSeeks API is OpenAI-compatible */
844
2192
  declare const fromDeepSeek: typeof fromOpenAICompatible;
@@ -929,5 +2277,5 @@ declare const from01AI: typeof fromOpenAICompatible;
929
2277
  //#endregion
930
2278
  //# sourceMappingURL=openaiCompatible.d.ts.map
931
2279
 
932
- export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CallParams, CircuitBreaker, CircuitBreakerOptions, ConsoleLogger, ContentBlock, ConversationTurn, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorType, Logger, NormalizedCacheAdapter, OnUsage, RefundUsage, ReserveUsage, SchemaLike, TextBlock, TieredCacheAdapter, TokenUsage, VernLLM, VernLLMOptions, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError };
2280
+ export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedStreamCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallMeta, CallParams, CallWithToolsResult, CircuitBreaker, CircuitBreakerOptions, CircuitState, ConsoleLogger, ContentBlock, ContentResult, ConversationTurn, FallbackAttempt, FallbackExhaustedError, FallbackOn, FallbackTarget, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorCode, LLMErrorType, Logger, NormalizedCacheAdapter, OnEvent, OnUsage, RateLimitAcquireResult, RateLimitOptions, RateLimitReason, RateLimiter, RefundUsage, ReserveUsage, SSE_PING, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, TargetCircuitState, TextBlock, TieredCacheAdapter, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolIssue, ToolResult, VernLLM, VernLLMEvent, VernLLMOptions, WireMessage, WireRequest, WireStreamChunk, WireToolCall, WireToolChoice, defaultEstimateTokens, defaultFallbackOn, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError, isToolCallResult, parseSseStream };
933
2281
  //# sourceMappingURL=index.d.cts.map