vern-llm 2.9.1 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,2233 +1,99 @@
1
- //#region src/types/errors.d.ts
2
- type LLMErrorType = 'timeout' | 'api' | 'network' | 'parse' | 'validation' | 'invalid_params' | 'rate_limited' | 'quota_exceeded' | 'circuit_open' | 'fallback_exhausted' | 'aborted' | 'unknown';
3
- /**
4
- * Machine readable discriminator within a `type`, for cases where `type`
5
- * alone is too coarse to act on. Optional and additive: errors thrown
6
- * before a given code existed simply omit it. Not owned by a single type;
7
- * e.g. `authentication`/`authorization` apply the same way regardless of
8
- * which type wraps them.
9
- */
10
- type LLMErrorCode = 'unknown_tool' | 'duplicate_tool_call_id' | 'tool_choice_none_violated' | 'unexpected_tool_calls' | 'unsupported_capability' | 'duplicate_tool_names' | 'unknown_tool_choice' | 'duplicate_tool_result_ids' | 'unknown_tool_result_ids' | 'missing_tool_results' | 'middleware_threw' | 'rate_limit_queue_full' | 'rate_limit_queue_timeout' | 'rate_limit_capacity_exceeded' | 'provider_rate_limited' | 'retry_budget_exhausted' | 'request_timeout' | 'idle_timeout' | 'middleware_timeout' | 'deadline_exceeded' | 'authentication' | 'authorization' | 'not_found' | 'payload_too_large' | 'server_error' | 'empty_response' | 'connection_failed' | 'circuit_cooling_down' | 'circuit_trial_in_flight' | 'fallback_exhausted' | 'tool_arguments_parse_failed' | 'stream_frame_invalid' | 'soft_failure_detected';
11
- /** One tool call's contract failure, used to report every bad call in a response at once. */
12
- interface ToolIssue {
13
- name: string;
14
- toolCallId: string;
15
- code: LLMErrorCode;
16
- detail?: unknown;
17
- }
18
- /**
19
- * The specific values behind a `duplicate_tool_names` failure: the
20
- * offending call's `tools` array had more than one entry sharing a name.
21
- */
22
- interface DuplicateToolNamesIssue {
23
- names: string[];
24
- }
25
- /**
26
- * The specific values behind an `unknown_tool_choice` failure: `toolChoice`
27
- * named a tool that wasn't in the call's own `tools` array.
28
- */
29
- interface UnknownToolChoiceIssue {
30
- requested: string;
31
- available: string[];
32
- }
33
- /**
34
- * The specific values behind a `duplicate_tool_result_ids` /
35
- * `unknown_tool_result_ids` / `missing_tool_results` failure: which
36
- * `history` turn was affected, and which `toolCallId`s were the problem.
37
- */
38
- interface HistoryToolResultIssue {
39
- historyIndex: number;
40
- ids: string[];
41
- }
42
- /**
43
- * The specific values behind an `unsupported_capability` failure: which
44
- * capability the current adapter/client/model doesn't support.
45
- */
46
- interface UnsupportedCapabilityIssue {
47
- capability: string;
48
- }
49
- /**
50
- * Maps each `LLMErrorCode` that carries structured `issues` to that
51
- * payload's exact shape. Not every code appears here: most `invalid_params`
52
- * failures are a single deterministic fact the `message` already states in
53
- * full, so adding a typed `issues` entry for them would only duplicate the
54
- * message into a field, the same near-duplicate-code problem `code` itself
55
- * avoids. Codes that repeat here are exactly the ones whose `message`
56
- * already string-joins a list a caller might want to consume directly
57
- * rather than re-parse out of prose, or that otherwise want a place to
58
- * report the exact captured values of a failure.
59
- *
60
- * Deliberately not a mapped type over the whole `LLMErrorCode` union: a
61
- * schema-validation failure's `issues` (the caller's own Zod-compatible
62
- * validator's error object) has no code and no shape VernLLM could know in
63
- * advance, so it stays untyped on `LLMError.issues` itself rather than
64
- * forcing every code into this table.
65
- */
66
- interface LLMErrorIssuesByCode {
67
- unknown_tool: ToolIssue[];
68
- duplicate_tool_call_id: ToolIssue[];
69
- duplicate_tool_names: DuplicateToolNamesIssue;
70
- unknown_tool_choice: UnknownToolChoiceIssue;
71
- duplicate_tool_result_ids: HistoryToolResultIssue;
72
- unknown_tool_result_ids: HistoryToolResultIssue;
73
- missing_tool_results: HistoryToolResultIssue;
74
- unsupported_capability: UnsupportedCapabilityIssue;
75
- }
76
- /**
77
- * Point-in-time copy of an `LLMError`'s fields, produced by
78
- * `LLMError.toSnapshot()`. This is what `RetryAttempt.error` holds
79
- * instead of a live `LLMError`.
80
- *
81
- * A past attempt only needs to be describable (message, type, code,
82
- * whether it was retryable), never thrown again. So it skips `Error`'s
83
- * behavior, `instanceof` identity, and any live getter. Using the full
84
- * `LLMError` class here would also make the type self referential
85
- * through its own `attempts` field.
86
- *
87
- * Has no `cause`. `cause` is `unknown` and never validated by VernLLM,
88
- * and it is meant to be read directly on the live error you just
89
- * caught, not carried indefinitely inside history. `type`, `code`,
90
- * `status`, and `issues` are the structured fields a snapshot carries
91
- * instead.
92
- *
93
- * `attempts` is still present, since a recorded attempt can itself be
94
- * the terminal failure of an inner retry loop with its own history (see
95
- * `FallbackAttempt`). That's a tree of past data, not a cycle.
96
- */
97
- interface LLMErrorSnapshot {
98
- message: string;
99
- type: LLMErrorType;
100
- status?: number;
101
- issues?: unknown;
102
- retryAfterMs?: number;
103
- code?: LLMErrorCode;
104
- /** Computed once, at snapshot time, since a snapshot has no live getter. */
105
- retryable: boolean;
106
- /** This attempt's own prior attempts, if it was itself the terminal failure of a retry loop. */
107
- attempts?: RetryAttempt[];
108
- }
109
- /**
110
- * Point-in-time copy of the request an attempt sent, produced by
111
- * `toRequestSnapshot()`. This is what `RetryAttempt.request` holds.
112
- * Mirrors `LLMErrorSnapshot`: plain data, never thrown or dispatched
113
- * again, safe to serialize and store.
114
- */
115
- interface LLMRequestSnapshot {
116
- /** Provider id this attempt targeted, e.g. "openai". */
117
- provider: string;
118
- /** Model id this attempt targeted. */
119
- model: string;
120
- /** The payload as actually sent for this attempt, after any transform/repair. Passed through `safeBody`. */
121
- body: unknown;
122
- /** Non sensitive request headers. Auth headers are stripped before the snapshot is built, never included. */
123
- headers?: Record<string, string>;
124
- /** Wall clock time the attempt started, ms since epoch. */
125
- startedAt: number;
126
- }
127
- /**
128
- * One failed attempt on the way to a terminal error: which attempt index
129
- * it was, and a snapshot of the error it failed with. The base shape
130
- * every richer attempt record (e.g. `FallbackAttempt`) extends, rather
131
- * than duplicates.
132
- */
133
- interface RetryAttempt {
134
- index: number;
135
- error: LLMErrorSnapshot;
136
- /** What was sent for this attempt. Optional: absent for attempts predating this field. */
137
- request?: LLMRequestSnapshot;
138
- }
139
- /** Optional fields for constructing an {@link LLMError}. `message` and `type` stay positional since every throw site sets both. */
140
- interface LLMErrorOptions {
141
- status?: number;
142
- issues?: unknown;
143
- cause?: unknown;
144
- retryAfterMs?: number;
145
- /** Stable discriminator within `type`. Absent on errors predating it. */
146
- code?: LLMErrorCode;
147
- /** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
148
- attempts?: RetryAttempt[];
149
- }
150
- export declare class LLMError extends Error {
151
- type: LLMErrorType;
152
- status?: number;
153
- issues?: unknown;
154
- cause?: unknown;
155
- retryAfterMs?: number;
156
- /** Stable discriminator within `type`. Absent on errors predating it. */
157
- code?: LLMErrorCode;
158
- /** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
159
- attempts?: RetryAttempt[];
160
- constructor(message: string, type: LLMErrorType, options?: LLMErrorOptions);
161
- /**
162
- * Computed purely from `type`/`code`, independent of any specific call's
163
- * `nonRetryableStatus` list. False for `parse`/`validation`/
164
- * `invalid_params`/`aborted` types (the caller's own input, the model's
165
- * own response, or intentional cancellation, none of which are the
166
- * provider being unhealthy), the tool contract codes, the local
167
- * rate limit codes, and the middleware timeout code.
168
- * Subclasses (see `FallbackExhaustedError`) may override this when `type`
169
- * alone carries no retry signal.
170
- */
171
- get retryable(): boolean;
172
- /**
173
- * Whether this failure should count toward the circuit breaker's
174
- * failure threshold. Not the same question as `retryable`:
175
- * `quota_exceeded` is retryable but says nothing about provider
176
- * health, so it's excluded here even though `retryable` is true for
177
- * it. Always false whenever `retryable` is false.
178
- */
179
- get countsTowardBreaker(): boolean;
180
- /**
181
- * Copies this error's fields into an {@link LLMErrorSnapshot}, for
182
- * recording as a `RetryAttempt`/`FallbackAttempt`. `retryable` is
183
- * captured here since a snapshot has no getter of its own. `cause` is
184
- * not copied, see `LLMErrorSnapshot`'s own doc. `issues` and every
185
- * nested `attempts` entry's own `issues` go through `safeAttempts`,
186
- * since a schema validation failure's `issues` is a caller supplied
187
- * value, not controlled by VernLLM, and `attempts` is itself a public
188
- * constructor option a caller can hand build.
189
- */
190
- toSnapshot(): LLMErrorSnapshot;
191
- /**
192
- * Controls what `JSON.stringify(err)` produces. Omits `cause` for the
193
- * same reason `toSnapshot()` does: `cause` is `unknown` and never
194
- * validated by VernLLM, and some SDK errors carry circular structures
195
- * `JSON.stringify` cannot serialize at all. Read `err.cause` directly
196
- * instead. `issues`, including every nested `attempts` entry's own
197
- * `issues`, goes through `safeAttempts` for the same reason: a schema
198
- * validation failure's `issues` is caller supplied and not guaranteed
199
- * circular free. Also includes `message` and `retryable`, which a
200
- * plain property walk would otherwise miss: `message` is
201
- * non-enumerable on `Error`, and `retryable` is a getter, not an own
202
- * property.
203
- */
204
- toJSON(): Record<string, unknown>;
205
- }
206
- export declare function isLLMError(err: unknown): err is LLMError;
207
- /**
208
- * Narrows `err.issues` to the exact shape {@link LLMErrorIssuesByCode} maps
209
- * `code` to, for any code listed there. `code` stays the only discriminator
210
- * VernLLM uses; this just gives that existing check a typed return instead
211
- * of requiring a manual cast of `issues`:
212
- *
213
- * ```ts
214
- * if (isLLMError(err) && hasIssues(err, 'duplicate_tool_names')) {
215
- * console.log(err.issues.names); // string[], no cast needed
216
- * }
217
- * ```
218
- */
219
- export declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
220
- code: C;
221
- issues: LLMErrorIssuesByCode[C];
222
- };
223
- //#endregion
224
- //#region src/types/cache.d.ts
225
- interface CacheAdapter<T = unknown> {
226
- get(key: string): Promise<{
227
- hit: boolean;
228
- value: T | null;
229
- }>;
230
- set(key: string, value: T, ttl: number): Promise<void>;
231
- delete?(key: string): Promise<void>;
232
- resolveKey?(key: string): Promise<string>;
233
- }
234
- /**
235
- * Which entry `InMemoryCacheAdapter` evicts once `maxSize` is exceeded.
236
- * `'fifo'` (default) drops the oldest inserted entry. `'lru'` drops the
237
- * least recently read or written entry.
238
- */
239
- type EvictionOption = 'fifo' | 'lru';
240
- /**
241
- * Trivial default so the package works out of the box with no external deps.
242
- * Not shared across processes, swap in Redis/Upstash/etc for production.
243
- */
244
- export declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
245
- private readonly maxSize;
246
- private store;
247
- private readonly eviction;
248
- constructor(maxSize?: number, eviction?: EvictionOption);
249
- get(key: string): Promise<{
250
- hit: boolean;
251
- value: T | null;
252
- }>;
253
- set(key: string, value: T, ttl: number): Promise<void>;
254
- delete(key: string): Promise<void>;
255
- private cleanupExpiredEntries;
256
- private enforceSizeLimit;
257
- }
258
- /**
259
- * Normalizes keys before caching to avoid duplicate entries from formatting differences.
260
- */
261
- export declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
262
- private readonly inner;
263
- constructor(inner?: CacheAdapter<T>);
264
- private normalize;
265
- resolveKey(key: string): Promise<string>;
266
- get(key: string): Promise<{
267
- hit: boolean;
268
- value: T | null;
269
- }>;
270
- set(key: string, value: T, ttl: number): Promise<void>;
271
- delete(key: string): Promise<void>;
272
- }
273
- /**
274
- * Two-tier cache with fast local L1 and shared L2.
275
- * L2 hits are promoted back to L1.
276
- */
277
- export declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
278
- private readonly l1;
279
- private readonly l2;
280
- private readonly l1Ttl?;
281
- constructor(l1: CacheAdapter<T>, l2: CacheAdapter<T>, l1Ttl?: number | undefined);
282
- /**
283
- * Forwards to L1's `resolveKey` if it has one, otherwise L2's. L1 is
284
- * preferred since `get()` checks L1 first, so its notion of "the same
285
- * key" is the one that determines whether a lookup can skip L2 entirely.
286
- */
287
- resolveKey(key: string): Promise<string>;
288
- get(key: string): Promise<{
289
- hit: boolean;
290
- value: T | null;
291
- }>;
292
- set(key: string, value: T, ttl: number): Promise<void>;
293
- delete(key: string): Promise<void>;
294
- }
295
- //#endregion
296
- //#region src/logger.d.ts
297
- interface Logger {
298
- debug(message: string): void;
299
- warn(message: string): void;
300
- error(message: string, meta?: Record<string, unknown>): void;
301
- }
302
- /**
303
- * Default logger. `debug` is gated by the `debug` option on VernLLM
304
- * warn/error always fire since they indicate real problems (retries, cache failures)
305
- */
306
- export declare class ConsoleLogger implements Logger {
307
- private debugEnabled;
308
- constructor(debugEnabled: boolean);
309
- debug(message: string): void;
310
- warn(message: string): void;
311
- error(message: string, meta?: Record<string, unknown>): void;
312
- }
313
- //#endregion
314
- //#region src/types/usage.d.ts
315
- type ReserveUsage = (params: {
316
- coalesced: boolean;
317
- signal?: AbortSignal;
318
- }) => Promise<void>;
319
- type RefundUsage = (params: {
320
- coalesced: boolean;
321
- signal?: AbortSignal;
322
- }) => Promise<void>;
323
- /**
324
- * The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
325
- * and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
326
- * has one definition instead of being redeclared at each use site.
327
- */
328
- interface UsageHooks {
329
- /**
330
- * Reserves usage before the request. Failures become
331
- * LLMError('quota_exceeded').
332
- */
333
- reserveUsage?: ReserveUsage;
334
- /**
335
- * Refunds usage after a failed call if reservation succeeded.
336
- */
337
- refundUsage?: RefundUsage;
338
- }
339
- interface TokenUsage {
340
- promptTokens: number;
341
- completionTokens: number;
342
- totalTokens: number;
343
- /**
344
- * Tokens spent on internal reasoning, a subset of `completionTokens`,
345
- * never added on top of it. Undefined when the provider's response
346
- * doesn't report a separate reasoning figure, e.g. Bedrock Converse
347
- * without an explicit `additionalModelResponseFieldPaths` request.
348
- */
349
- reasoningTokens?: number;
350
- requestId: string;
351
- model: string;
352
- /**
353
- * The provider target that produced this usage. See `VernLLMOptions['name']`,
354
- * default `'primary'`. Optional so consumers constructing a `TokenUsage`
355
- * themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
356
- * populates it. Absent means the same as `'primary'` if you need a value.
357
- */
358
- provider?: string;
359
- /**
360
- * Whether this usage came from a fallback target rather than the
361
- * primary. Optional for the same reason `provider` is: `VernLLM`
362
- * always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
363
- * isn't forced to.
364
- */
365
- usedFallback?: boolean;
366
- }
367
- type OnUsage = (usage: TokenUsage) => void;
368
- /**
369
- * Called when a provider response arrives but VernLLM's own post-processing
370
- * then fails, after usage data was already present in that response. Covers
371
- * any error thrown after usage extraction, not just parse/validation, since
372
- * everything in that path only runs once a response, and real spend, has
373
- * already arrived. Fires once per failed attempt with extractable usage,
374
- * never for transport failures, where no response means no honest number
375
- * to report.
376
- */
377
- type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
378
- //#endregion
379
- //#region src/types/events.d.ts
380
- /**
381
- * Reports what happened during a call. Fire and forget, mirroring
382
- * `onUsage`: the return value is never read and a throwing handler cannot
383
- * change what the call does, only what gets reported about it.
384
- */
385
- type VernLLMEvent = {
386
- kind: 'retry';
387
- requestId: string;
388
- provider: string;
389
- /** The model actually resolved for this call (honors a per-call `model` override). */
390
- model: string;
391
- /** The 1-based retry ordinal (the 1st retry is `1`, not the overall attempt count). */
392
- attempt: number;
393
- maxRetries: number;
394
- delayMs: number;
395
- retryAfterHonored: boolean;
396
- error: LLMError;
397
- } | {
398
- kind: 'circuit_state';
399
- provider: string;
400
- /**
401
- * The model of the call that triggered this specific transition
402
- * (whatever was passed to the `assertClosed`/`recordSuccess`/
403
- * `recordFailure` call that caused it), not a property of the
404
- * circuit itself: the breaker still counts failures across every
405
- * model together, so a threshold crossing can be the sum of
406
- * several different models' failures even though only the
407
- * triggering call's `model` is reported here.
408
- */
409
- model: string;
410
- from: CircuitState;
411
- to: CircuitState;
412
- consecutiveFailures: number;
413
- } | {
414
- kind: 'fallback';
415
- requestId: string;
416
- /** Provider name of the target that just failed. */
417
- from: string;
418
- /** Provider name of the target about to be tried next. */
419
- to: string;
420
- /** `-1` for the primary target, otherwise the index into `fallback`. */
421
- fromIndex: number;
422
- toIndex: number;
423
- /** The normalized error that caused `from` to be abandoned. */
424
- error: LLMError;
425
- /** Time spent on `from`, including its own retries, before giving up. */
426
- elapsedMs: number;
427
- } | {
428
- kind: 'rate_limited';
429
- requestId: string;
430
- provider: string;
431
- /** The model actually resolved for this call (honors a per-call `model` override). */
432
- model: string;
433
- /** How long this attempt sat queued for capacity before it was let through. */
434
- waitedMs: number;
435
- /** Which configured bucket was blocking this attempt just before it cleared. */
436
- reason: 'concurrency' | 'rpm' | 'tpm';
437
- } | {
438
- kind: 'middleware';
439
- requestId: string;
440
- /** This middleware's `name`, or its array position if unnamed. */
441
- middleware: string;
442
- hook: 'transform' | 'wrap_short_circuit' | 'enabled_skip';
443
- /** For `hook: 'transform'` only: which top-level fields the merged patch touched. */
444
- patchedFields?: string[];
445
- } | {
446
- /**
447
- * Reported once a call fully succeeds. Same data `VernLLMOptions.onUsage`
448
- * receives; that option is sugar over this event, not a second
449
- * reporting path, see `makeEventReporter`.
450
- */
451
- kind: 'usage';
452
- requestId: string;
453
- usage: TokenUsage;
454
- } | {
455
- /**
456
- * A provider response arrived, carrying real usage, and VernLLM's own
457
- * post-processing then failed. Fires once per failed attempt with
458
- * extractable usage, matching `VernLLMOptions.onUsageFailure`'s own
459
- * granularity, which this event is sugar over, not a second path.
460
- */
461
- kind: 'usage_failure';
462
- requestId: string;
463
- usage: TokenUsage;
464
- error: LLMError;
465
- };
466
- type OnEvent = (event: VernLLMEvent) => void;
467
- //#endregion
468
- //#region src/types/middleware.d.ts
469
- /** Capabilities of the target a middleware hook is currently looking at. */
470
- interface MiddlewareCapabilities {
471
- /**
472
- * Whether this target honors `response_format: { type: 'json_object' }`
473
- * as a real constraint. Mirrors `LLMClient.supportsJsonObjectMode`.
474
- * `false` for `fromAnthropic` and `fromBedrock`.
475
- */
476
- supportsJsonObjectMode: boolean;
477
- }
478
- /**
479
- * Not exported. Distinguishes `MiddlewareStateKey<T>` from
480
- * `MiddlewareRef` and from a plain `{ debugName }` object literal at
481
- * the type level, even though all three have the identical runtime
482
- * shape. Without this, `MiddlewareStateKey<T>`/`MiddlewareRef` are
483
- * structurally just `{ debugName: string }`, so TypeScript would treat
484
- * a state key as a valid middleware ref (or vice versa), and would let
485
- * anyone hand-write `{ debugName: 'auth' }` in place of a real
486
- * `createMiddlewareRef` result. Neither is possible once this brand is
487
- * required: only `createStateKey`, which alone has access to this
488
- * symbol, can produce a value satisfying `MiddlewareStateKey<T>`.
489
- */
490
- declare const stateKeyBrand: unique symbol;
491
- /**
492
- * A typed reference to one slot in `ctx.state`. Create one with
493
- * `createStateKey`, export it, and import the same reference wherever
494
- * another middleware needs to read or write the same value. There's no
495
- * string key anywhere in this path, so a typo becomes a missing import
496
- * or an undefined variable, a compile error, instead of a silently
497
- * created new property.
498
- */
499
- interface MiddlewareStateKey<T> {
500
- readonly debugName: string;
501
- readonly [stateKeyBrand]: true;
502
- /**
503
- * Never set at runtime; exists purely so `T` is actually used
504
- * somewhere in this interface's shape (a phantom type), which is what
505
- * lets `MiddlewareStateBag.get`/`set` infer the right type for a given
506
- * key instead of two `MiddlewareStateKey<string>` and
507
- * `MiddlewareStateKey<number>` keys being structurally identical.
508
- */
509
- readonly __phantom?: T;
510
- }
511
- /** Creates a new, distinct `MiddlewareStateKey`. `debugName` is used only in log lines and the `'middleware'` event; it never affects equality. */
512
- export declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
513
- /** Not exported. See `stateKeyBrand`; same reasoning, distinct symbol, so the two token types can't be cross-assigned either. */
514
- declare const middlewareRefBrand: unique symbol;
515
- /**
516
- * A typed reference to one middleware's identity, for `runsAfter`/
517
- * `runsBefore` to target. Purely an ordering concern: unlike `name`,
518
- * `ref` is never used as a display label anywhere (`name` still covers
519
- * that), only as a `runsAfter`/`runsBefore` match target. Create one
520
- * with `createMiddlewareRef`, export it from the package that owns the
521
- * middleware, and have any dependent import the same reference instead
522
- * of typing a matching `name` string. Same reasoning as
523
- * `MiddlewareStateKey`: a typo becomes a missing import, a compile
524
- * error, instead of a silently unresolved (or worse, silently
525
- * colliding) string.
526
- */
527
- interface MiddlewareRef {
528
- readonly debugName: string;
529
- readonly [middlewareRefBrand]: true;
530
- }
531
- /** Creates a new, distinct `MiddlewareRef`. `debugName` is used only in error messages when a reference doesn't resolve; it never affects equality, so two refs with the same `debugName` never collide. */
532
- export declare function createMiddlewareRef(debugName: string): MiddlewareRef;
533
- /**
534
- * A `runsAfter`/`runsBefore` entry that escalates an unresolved
535
- * reference from a warning to a construction-time throw. Wrap a
536
- * `MiddlewareRef` with `requireRef` when the dependency isn't optional:
537
- * a bare `MiddlewareRef` in `runsAfter`/`runsBefore` means "order
538
- * relative to this if it's registered," which is the right default for
539
- * a dependency a third party may reasonably not have installed. A
540
- * `RequiredMiddlewareRef` means "this middleware must not run without
541
- * that dependency having already run". The app should fail to start
542
- * rather than run with a silently-missing ordering guarantee.
543
- */
544
- interface RequiredMiddlewareRef {
545
- readonly ref: MiddlewareRef;
546
- }
547
- /** Wraps `ref` so `runsAfter`/`runsBefore` throws at `VernLLM` construction time if it doesn't resolve, instead of warning and continuing. */
548
- export declare function requireRef(ref: MiddlewareRef): RequiredMiddlewareRef;
549
- /**
550
- * Typed, per-logical-call storage two middleware can deliberately share a
551
- * value through (a span ID one sets, another reads). Backed by a plain
552
- * `Map` internally, created once per logical call and never read or
553
- * written by VernLLM itself.
554
- */
555
- interface MiddlewareStateBag {
556
- get<T>(key: MiddlewareStateKey<T>): T | undefined;
557
- set<T>(key: MiddlewareStateKey<T>, value: T): void;
558
- }
559
- /** A plain, `Map`-backed `MiddlewareStateBag`. */
560
- export declare function createMiddlewareStateBag(): MiddlewareStateBag;
561
- /** Fields every `MiddlewareContext` variant carries, regardless of `stage`. */
562
- interface MiddlewareContextBase {
563
- requestId: string;
564
- /** Capabilities of the target this stage's identity fields describe. */
565
- capabilities: MiddlewareCapabilities;
566
- signal?: AbortSignal;
567
- /** Shared, collision-proof state for two middleware to deliberately coordinate through. See `MiddlewareStateBag`. */
568
- state: MiddlewareStateBag;
569
- /** Simple, string-keyed scratch space, pre-namespaced to this one middleware so two middleware can never collide here even by accident. */
570
- own: Record<string, unknown>;
571
- /**
572
- * Every registered middleware's resolved label, in `transformOrder`,
573
- * frozen. Lets a middleware make an informed call, like skipping a
574
- * duplicate action when it detects another known middleware by name
575
- * already handles it, without needing to know anything else about
576
- * that middleware's own configuration.
577
- */
578
- registeredMiddlewareNames: readonly string[];
579
- }
580
- /**
581
- * The `ctx` `transform` receives, and every attempt-scoped event context
582
- * (`'retry'`, `'fallback'`, `'circuit_state'`, `'middleware'`). Built once
583
- * a specific target has actually been selected for this attempt, so every
584
- * field describes the real target, not a placeholder.
585
- */
586
- interface AttemptContext extends MiddlewareContextBase {
587
- stage: 'attempt';
588
- /** The target this attempt is actually dispatched to. */
589
- requestedProvider: string;
590
- requestedModel: string;
591
- isFallbackAttempt: boolean;
592
- /**
593
- * The real, current attempt number for this dispatch.
594
- *
595
- * Exception: on a `'circuit_state'` event triggered by a pre-dispatch
596
- * check (`assertClosed`, before any attempt has been made), this is
597
- * `1` regardless of which attempt is about to run, since no attempt
598
- * exists yet to report. Every other `'circuit_state'` event, and every
599
- * other attempt-scoped event, reports the real attempt number.
600
- */
601
- attempt: number;
602
- }
603
- /**
604
- * The `ctx` `wrap` receives before `next()` resolves (and `onError`'s own
605
- * `ctx`, built the same way under the hood). Built once, before any
606
- * fallback target is chosen, so it only ever describes the primary
607
- * target. There is no real "requested" target yet, and no attempt count,
608
- * fallback flag, or per-attempt capability to report. Read `next()`'s
609
- * resolved `CallResult.meta` once you need to know what actually
610
- * happened.
611
- */
612
- interface PreDispatchContext extends MiddlewareContextBase {
613
- stage: 'pre-dispatch';
614
- /** The primary target only, not necessarily who ends up answering. */
615
- primaryProvider: string;
616
- primaryModel: string;
617
- }
618
- /**
619
- * `enabled` and `onEvent` are called from both stages (gating/observing
620
- * `transform` as well as `wrap`), so they receive this union and must
621
- * narrow on `ctx.stage` before reading stage-specific fields.
622
- * `transform` and `wrap` themselves receive the single variant that's
623
- * always accurate for them (`AttemptContext`/`PreDispatchContext`
624
- * respectively). See `VernLLMMiddleware`.
625
- */
626
- type MiddlewareContext = AttemptContext | PreDispatchContext;
627
- /** The `response_format` shape `RequestBuilder` can put on the wire. */
628
- type WireResponseFormat = {
629
- type: 'json_object';
630
- } | {
631
- type: 'json_schema';
632
- json_schema: {
633
- name: string;
634
- schema: Record<string, unknown>;
635
- strict?: boolean;
636
- description?: string;
637
- };
638
- };
639
- /** A tool as it appears on the wire, OpenAI's `function`-wrapped shape. */
640
- interface WireTool {
641
- type: 'function';
642
- function: {
643
- name: string;
644
- description: string;
645
- parameters: Record<string, unknown>;
646
- };
647
- }
648
- /**
649
- * The wire-shaped request `RequestBuilder.build()` produces for one call
650
- * attempt, before dispatch. Read only inside `transform`; return a patch
651
- * of the fields you want to change instead of the whole object.
652
- */
653
- interface WireCallRequest {
654
- model: string;
655
- temperature?: number;
656
- max_tokens: number;
657
- response_format?: WireResponseFormat;
658
- reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
659
- budget_tokens?: number;
660
- tools?: WireTool[];
661
- tool_choice?: WireToolChoice;
662
- messages: WireMessage[];
663
- }
664
- /**
665
- * What `transform` returns: a patch merged onto the request that
666
- * `RequestBuilder.build()` (plus every earlier middleware's own patch)
667
- * already produced, not a replacement for it. `model` and
668
- * `response_format` can't be expressed here at all, since everything
669
- * downstream that attributes a call to a target keys off the values
670
- * `RequestBuilder` already resolved for those two fields, not off
671
- * whatever ends up on the wire request. `messages`/`tools` are joined by
672
- * a separate `add*` field, appended rather than replaced, so two
673
- * independently written middleware can each add to the list without one
674
- * silently clobbering what the other already added.
675
- */
676
- interface WireCallRequestPatch {
677
- temperature?: number;
678
- max_tokens?: number;
679
- reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
680
- budget_tokens?: number;
681
- tool_choice?: WireToolChoice;
682
- /** Replaces the whole message list. Prefer `addMessages` unless a full replace is genuinely the intent. */
683
- messages?: WireMessage[];
684
- /** Appended after whatever earlier middleware already added. Never clobbers a prior addition. */
685
- addMessages?: WireMessage[];
686
- /** Replaces the whole tool list. Prefer `addTools`, same reasoning as `messages`/`addMessages`. */
687
- tools?: WireTool[];
688
- /** Appended after whatever earlier middleware already added. Never clobbers a prior addition. */
689
- addTools?: WireTool[];
690
- }
691
- /**
692
- * The settled outcome of one logical call, passed to `wrap`'s `next()`.
693
- * `meta` is populated once a target has actually answered, for both
694
- * streaming and non-streaming calls (`undefined` only on a cache hit,
695
- * where nothing was actually spent).
696
- */
697
- interface CallResult<T = unknown> {
698
- value: T;
699
- meta?: CallMeta;
700
- }
701
- /**
702
- * One entry in `VernLLMOptions.middleware`. All four hooks are optional;
703
- * an entry that sets none of them is inert. See the middleware docs for
704
- * how `transform`, `wrap`, `onEvent`, and `enabled` compose across
705
- * several entries.
706
- */
707
- interface VernLLMMiddleware {
708
- /** Used in log lines and the `'middleware'` event. Defaults to this entry's array position when omitted. */
709
- name?: string;
710
- /**
711
- * This entry's own identity, purely for another middleware's
712
- * `runsAfter`/`runsBefore` to target. Create with `createMiddlewareRef`,
713
- * export it, and have a dependent import the same reference. Optional:
714
- * only needed if something else must be able to depend on this
715
- * specific entry. Unrelated to `name`: `ref` is never shown in logs,
716
- * `name` is never matched against for ordering.
717
- */
718
- ref?: MiddlewareRef;
719
- /** Sort key for composition order, ascending, ties broken by array order. See the middleware docs for what "lower runs first" means for `wrap`. */
720
- priority?: number;
721
- /**
722
- * Other middleware this entry must run after, breaking ties
723
- * `priority` alone can't express. Matched by `ref` identity, so a
724
- * typo or a stale copy simply fails to resolve instead of silently
725
- * matching the wrong entry. A bare `MiddlewareRef` that doesn't
726
- * resolve is dropped, not an error, since a third party may
727
- * reasonably reference a well known middleware that isn't installed
728
- * everywhere; wrap it with `requireRef` to make that same target
729
- * mandatory instead, throwing at `VernLLM` construction time if it's
730
- * missing. A cycle across `runsAfter`/`runsBefore` always throws,
731
- * regardless of whether any individual entry is required.
732
- */
733
- runsAfter?: (MiddlewareRef | RequiredMiddlewareRef)[];
734
- /**
735
- * Other middleware this entry must run before. See `runsAfter`; a
736
- * bare reference is dropped if unresolved, a `requireRef`-wrapped one
737
- * throws.
738
- */
739
- runsBefore?: (MiddlewareRef | RequiredMiddlewareRef)[];
740
- /**
741
- * Pins this entry's slot in `wrap` nesting only, independent of
742
- * `priority`/`runsAfter`/`runsBefore`, which still govern
743
- * `transform`/`onEvent` order. `'outermost'` sees the net
744
- * `CallResult` of every retry, fallback, and other middleware's
745
- * `wrap`; `'innermost'` sits closest to the real dispatch. A numeric
746
- * value behaves like `priority`, but only for `wrap` nesting.
747
- */
748
- position?: 'outermost' | 'innermost' | number;
749
- /**
750
- * Boolean for a static on/off switch, or a predicate evaluated per
751
- * call. A throwing, rejecting, or timed-out predicate is logged and
752
- * treated as `false` for that call.
753
- */
754
- enabled?: boolean | ((ctx: MiddlewareContext) => boolean | Promise<boolean>);
755
- /** Per-middleware override of the instance-level `middlewareTimeoutMs`, applied to this entry's `transform` and function `enabled`. `<= 0` means unbounded (no timer at all). */
756
- timeoutMs?: number;
757
- /** Transforms the outgoing wire request for one attempt. Runs once per attempt, including retries. `ctx` is always accurate to the real target for this attempt. */
758
- transform?: (request: Readonly<WireCallRequest>, ctx: AttemptContext) => WireCallRequestPatch | Promise<WireCallRequestPatch>;
759
- /**
760
- * Wraps one whole logical call, exactly once, regardless of how many
761
- * retries or fallback targets ran underneath it. `ctx` is built once,
762
- * before any fallback target is chosen, so it only describes the
763
- * primary target. There is no `requestedProvider`/`isFallbackAttempt`/
764
- * `attempt` to read here. Read `next()`'s resolved `CallResult.meta`
765
- * for what actually happened.
766
- */
767
- wrap?: (request: Readonly<WireCallRequest>, next: () => Promise<CallResult>, ctx: PreDispatchContext) => Promise<CallResult>;
768
- /** Observes the same events reported on `VernLLMOptions.onEvent`, filtered by this middleware's own `enabled`. Called from both stages; narrow on `ctx.stage` before reading stage-specific fields. */
769
- onEvent?: (event: VernLLMEvent, ctx: MiddlewareContext) => void;
770
- }
771
- //#endregion
772
- //#region src/circuitBreaker.d.ts
773
- /** The call this mutation happened as part of, forwarded to `onStateChange` untouched. */
774
- interface CircuitBreakerCallContext {
775
- requestId: string;
776
- state: MiddlewareStateBag;
777
- signal?: AbortSignal;
778
- /** Omitted for calls before any attempt exists, like `assertClosed`'s pre-dispatch check. */
779
- attempt?: number;
780
- }
781
- /**
782
- * Fires after every real state change, never a no-op transition. `model`
783
- * is the resolved model of whichever call triggered it. With
784
- * `isolateByModel` off, failures are still counted across every model.
785
- * Shared by `CircuitBreakerOptions` and `CircuitBreakerAdapter`, so a
786
- * custom adapter reports state changes the same way the built in
787
- * `CircuitBreaker` does.
788
- */
789
- type CircuitBreakerStateChangeHandler = (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
790
- interface CircuitBreakerOptions {
791
- /** Consecutive failures before the circuit opens, default 5 */
792
- threshold?: number;
793
- /** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
794
- cooldownMs?: number;
795
- onStateChange?: CircuitBreakerStateChangeHandler;
796
- /**
797
- * Track a separate circuit per resolved model instead of one shared
798
- * circuit. Default false. A call that omits `model` falls into one
799
- * shared bucket alongside every other call that also omits it.
800
- */
801
- isolateByModel?: boolean;
802
- /** Trial calls allowed through per half-open cycle. Default 1, clamped to at least 1. */
803
- halfOpenProbes?: number;
804
- /** Fraction of `halfOpenProbes` that must succeed to close the circuit. Default 1, clamped to `[0, 1]`. */
805
- halfOpenSuccessRatio?: number;
806
- /**
807
- * Grows `cooldownMs` on each repeat open instead of a fixed wait.
808
- * `{ multiplier, maxMs }` covers exponential growth; a `CooldownBackoff`
809
- * function covers anything else. Omitted means `cooldownMs` stays fixed.
810
- */
811
- cooldownBackoff?: ExponentialBackoffOptions | CooldownBackoff;
812
- /**
813
- * Decides when a bucket's failures should open the circuit.
814
- * `{ kind: 'consecutive', threshold }` (the default) opens after that
815
- * many failures in a row. `{ kind: 'rolling', windowMs, minCalls,
816
- * failureRatio }` opens once at least `minCalls` calls have landed in
817
- * the trailing `windowMs` and the failure ratio reaches `failureRatio`.
818
- * `minCalls` must be a non-negative integer; `failureRatio` must be
819
- * finite and within `[0, 1]`. Both are validated at construction,
820
- * thrown as `RangeError`. A `TrippingPolicy` covers anything else, one
821
- * instance shared across every model automatically under
822
- * `isolateByModel`, since it tracks its own state per key rather than
823
- * owning one flat counter.
824
- */
825
- tripping?: TrippingOption;
826
- }
827
- /** Computes the cooldown for a bucket's `reopenCount`-th repeat open. */
828
- type CooldownBackoff = (reopenCount: number, baseCooldownMs: number) => number;
829
- interface ExponentialBackoffOptions {
830
- /** Growth factor applied per repeat open, e.g. 2 doubles each time. */
831
- multiplier: number;
832
- /** Upper bound on the computed cooldown, in ms. Default `Infinity`. */
833
- maxMs?: number;
834
- }
835
- /**
836
- * Decides when a bucket's failures should open the circuit. Keyed by
837
- * `key` (a resolved model, or the shared bucket's key when
838
- * `isolateByModel` is off) rather than holding one flat counter, so a
839
- * single `TrippingPolicy` instance is always safe to share across every
840
- * bucket: `CircuitBreaker` never needs to clone or construct a fresh one
841
- * per model, `isolateByModel` isolation falls out of `key` alone.
842
- */
843
- interface TrippingPolicy {
844
- onSuccess(key: string): void;
845
- /** Returns true if this failure should open the circuit for `key`. */
846
- onFailure(key: string): boolean;
847
- reset(key: string): void;
848
- /**
849
- * Called when `key`'s bucket is discarded (closed and idle, under
850
- * `isolateByModel`), so a keyed policy can release that key's state.
851
- * Optional: omit if there's nothing to release.
852
- */
853
- forget?(key: string): void;
854
- }
855
- export declare class ConsecutiveTripping implements TrippingPolicy {
856
- private readonly threshold;
857
- private failuresByKey;
858
- constructor(threshold: number);
859
- onSuccess(key: string): void;
860
- onFailure(key: string): boolean;
861
- reset(key: string): void;
862
- forget(key: string): void;
863
- }
864
- export declare class RollingTripping implements TrippingPolicy {
865
- private readonly windowMs;
866
- private readonly minCalls;
867
- private readonly failureRatio;
868
- private ratiosByKey;
869
- constructor(windowMs: number, minCalls: number, failureRatio: number);
870
- private ratioFor;
871
- onSuccess(key: string): void;
872
- onFailure(key: string): boolean;
873
- reset(key: string): void;
874
- forget(key: string): void;
875
- }
876
- /** Not exported. Internal shorthand union for `CircuitBreakerOptions.tripping`. */
877
- type TrippingOption = {
878
- kind: 'consecutive';
879
- threshold: number;
880
- } | {
881
- kind: 'rolling';
882
- windowMs: number;
883
- minCalls: number;
884
- failureRatio: number;
885
- } | TrippingPolicy;
886
- type CircuitState = 'closed' | 'open' | 'half-open';
887
- /**
888
- * What VernLLM's dispatch layer needs from a breaker. `CircuitBreaker`
889
- * implements this; a caller wanting cross process coordination can hand
890
- * over their own instance instead.
891
- *
892
- * `assertClosed`, `recordSuccess`, `recordFailure`, and `onStateChange`
893
- * are required, mirroring `RateLimiterAdapter`'s four required methods.
894
- * `onStateChange` is required so `circuit_state` events can't go
895
- * silently missing; a no-op `() => {}` is fine if you don't care.
896
- *
897
- * `getState`, `getFailureBreakdown`, `isolateByModel`, `open`, and
898
- * `close` are optional. Omitting one makes the matching call a no-op
899
- * or return `undefined`/`false`, same as no breaker configured.
900
- * `open`/`close` are optional since they let VernLLM force a
901
- * transition, control a distributed adapter may not want to grant.
902
- */
903
- interface CircuitBreakerAdapter {
904
- /** Throws when the circuit is open (or half open with no trial slot free) for `model`. */
905
- assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
906
- recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
907
- /** `code`, when present, is the failing call's `LLMErrorCode`. */
908
- recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
909
- getState?(model?: string): CircuitState;
910
- /** Failure counts by `LLMErrorCode` for `model`'s bucket, `'unknown'` for one that carried no code. */
911
- getFailureBreakdown?(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
912
- /** Whether this adapter tracks failures per model, mirroring `CircuitBreakerOptions.isolateByModel`. Read by `warnIfModelUnsupported`'s diagnostic warning and by `VernLLM.getCircuitStates()`'s public output; omit if the notion doesn't apply to your adapter, `false` is assumed. */
913
- isolateByModel?: boolean;
914
- /** Manually opens the circuit, as if enough consecutive failures had just happened. Optional: an adapter that doesn't want external callers forcing a transition can omit it. */
915
- open?(model?: string, context?: CircuitBreakerCallContext): void;
916
- /** Manually closes the circuit, without requiring a real success first. Same opt-in reasoning as `open`. */
917
- close?(model?: string, context?: CircuitBreakerCallContext): void;
918
- /** Gives back a half-open trial slot when a call ends without `recordSuccess` or `recordFailure`. Idempotent, and a no-op for a call that holds no slot. */
919
- releaseTrial?(model?: string, context?: CircuitBreakerCallContext): void;
920
- /** Awaited right before `assertClosed` to refresh local state. Never blocks or fails a call: a rejection or `prepareTimeoutMs` is logged and the call carries on. */
921
- prepare?(model?: string, context?: CircuitBreakerCallContext): Promise<void>;
922
- /** How long to wait for `prepare`, in ms. Default 1000. */
923
- prepareTimeoutMs?: number;
924
- /** Live counterpart of `getState`, read by `VernLLM.readCircuitStates()`. */
925
- readState?(model?: string): Promise<CircuitState>;
926
- /** Receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
927
- setLogger?(logger: Logger): void;
928
- /**
929
- * Called after every real state change, never a no-op transition. VernLLM
930
- * wraps it the same way it wraps the built in `CircuitBreaker`'s
931
- * `onStateChange`: every call still reports a `circuit_state` event
932
- * first, then this hook is chained after that, wrapped so a throw here
933
- * can't break the call that triggered it.
934
- */
935
- onStateChange: CircuitBreakerStateChangeHandler;
936
- }
937
- /**
938
- * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures
939
- * across calls. Once the threshold is hit, short-circuits new calls with
940
- * LLMError('circuit_open') until the cooldown elapses and a trial succeeds.
941
- */
942
- export declare class CircuitBreaker implements CircuitBreakerAdapter {
943
- private readonly cooldownMs;
944
- /** Satisfies `CircuitBreakerAdapter.onStateChange`, required there. Defaults to a no-op when `options.onStateChange` is omitted. */
945
- readonly onStateChange: CircuitBreakerStateChangeHandler;
946
- /** Whether this breaker tracks failures per model instead of one shared circuit. */
947
- readonly isolateByModel: boolean;
948
- private readonly halfOpenProbes;
949
- private readonly halfOpenSuccessRatio;
950
- private readonly cooldownBackoff?;
951
- /** One instance, keyed per model internally. See `TrippingPolicy`. */
952
- private readonly tripping;
953
- private readonly sharedBucket;
954
- private readonly bucketsByModel;
955
- constructor(options?: CircuitBreakerOptions);
956
- /**
957
- * Throws if the circuit is open and the cooldown hasn't elapsed, or if
958
- * half-open with every trial slot claimed. Otherwise claims a trial slot.
959
- */
960
- assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
961
- recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
962
- /** `code`, when present, is the failing `LLMError`'s `code`. Missing attributes to `'unknown'`. */
963
- recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
964
- /**
965
- * Gives back the half-open trial slot `context`'s call claimed, when
966
- * that call ended without recording an outcome. No-op without a
967
- * `context`, when the bucket isn't half-open, or when the call's permit
968
- * is stale or already spent (an outcome was recorded), so calling it
969
- * defensively on every failure path is safe.
970
- */
971
- releaseTrial(model?: string, context?: CircuitBreakerCallContext): void;
972
- /** With `isolateByModel` off, `model` is ignored and the shared circuit's state is returned. */
973
- getState(model?: string): CircuitState;
974
- /** Failure counts by `LLMErrorCode` for `model`'s bucket. Returned as a plain object copy. */
975
- getFailureBreakdown(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
976
- /** Manually opens the circuit, as if `threshold` consecutive failures had just happened. */
977
- open(model?: string, context?: CircuitBreakerCallContext): void;
978
- /** Manually closes the circuit and resets its failure count, without requiring a real success first. */
979
- close(model?: string, context?: CircuitBreakerCallContext): void;
980
- /**
981
- * Opens `bucket`: stamps `openedAt`/`cooldownMsForOpen` and transitions
982
- * to `open`. Shared by `recordFailure`'s trip, `settleTrialIfComplete`'s
983
- * reopen, and the manual `open()`, all of which reach this with
984
- * `bucket.trial` already `null`.
985
- */
986
- private openBucket;
987
- /** Computes and clamps the cooldown for `bucket`'s current `reopenCount`. Called once, on open. */
988
- private computeCooldown;
989
- /** Returns the bucket for a model if one already exists, without allocating. */
990
- private lookupBucket;
991
- /**
992
- * The key `tripping` is called with. Real per-model isolation under
993
- * `isolateByModel`, matching `ensureBucketFor`/`lookupBucket`'s own
994
- * per-model key. Otherwise one fixed shared key regardless of what
995
- * `model` was passed, matching `sharedBucket` being the one and only
996
- * bucket in that mode: `model` is never allowed to split tripping state
997
- * when `isolateByModel` is off, the same way it never splits which
998
- * bucket a call lands in.
999
- */
1000
- private trippingKeyFor;
1001
- /** Creates and stores a bucket for a model when the first mutation needs one. */
1002
- private ensureBucketFor;
1003
- /** Drops an idle model's bucket and lets `tripping` release that key's state too. */
1004
- private forgetModel;
1005
- /** Every state mutation routes through here, so `onStateChange` fires exactly once per real change. */
1006
- private transition;
1007
- /** Once every admitted trial has reported in, closes or reopens based on `halfOpenSuccessRatio`. */
1008
- private settleTrialIfComplete;
1009
- }
1010
- //#endregion
1011
- //#region src/internal/retryBudget.d.ts
1012
- /**
1013
- * Tunables for a `RetryBudget`. `windowMs`/`minCalls` behave the same as
1014
- * `RollingTripping`'s (see `circuitBreaker.ts`): `minCalls` gates the
1015
- * check so a cold start with too little traffic to judge doesn't trip.
1016
- * `retryRatio` is the max fraction of calls in the window allowed to be
1017
- * retries before the budget stops allowing more. `minCalls` must be a
1018
- * non-negative integer; `retryRatio` must be finite and within `[0, 1]`.
1019
- * Both are validated at construction, thrown as `RangeError`.
1020
- */
1021
- interface RetryBudgetOptions {
1022
- windowMs: number;
1023
- minCalls: number;
1024
- retryRatio: number;
1025
- }
1026
- /**
1027
- * Caps how much of a target's recent traffic is allowed to be retries,
1028
- * independent of the circuit breaker. The breaker asks whether the
1029
- * provider is healthy; this asks whether retrying is still worth the
1030
- * capacity it costs, regardless of provider health. Reuses `RollingRatio`,
1031
- * the same primitive `RollingTripping` is built on, rather than a second
1032
- * hand rolled window.
1033
- */
1034
- export declare class RetryBudget {
1035
- private readonly options;
1036
- private readonly ratio;
1037
- constructor(options: RetryBudgetOptions);
1038
- /**
1039
- * Throws `LLMError('retry_budget_exhausted')` once at least `minCalls`
1040
- * calls have landed in the trailing `windowMs` and the retry ratio
1041
- * among them has reached `retryRatio`. A no-op otherwise.
1042
- */
1043
- assertAvailable(): void;
1044
- /** Records one attempt. `isRetry` is false for a call's first attempt, true for every attempt after it. */
1045
- recordAttempt(isRetry: boolean): void;
1046
- /** Current traffic and retry ratio in the trailing window. */
1047
- getSnapshot(): {
1048
- attempts: number;
1049
- retryRatio: number;
1050
- };
1051
- }
1052
- //#endregion
1053
- //#region src/internal/utils/rate-limit/rateLimitHint.utils.d.ts
1054
- /** A normalized read of a provider's rate limit headers. */
1055
- interface ProviderRateLimitHint {
1056
- remainingRequests?: number;
1057
- limitRequests?: number;
1058
- resetAfterMs?: number;
1059
- }
1060
- //#endregion
1061
- //#region src/rateLimit.d.ts
1062
- /** The request shape sent to `LLMClient['chat']['completions']['create']`, used for token estimation. */
1063
- type WireRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
1064
- /** Which configured bucket is currently blocking a call. */
1065
- type RateLimitReason = 'concurrency' | 'rpm' | 'tpm';
1066
- interface RateLimitOptions {
1067
- /** Max requests per minute. Omit for unlimited. */
1068
- requestsPerMinute?: number;
1069
- /**
1070
- * Max tokens per minute. Enforced against a pre-flight estimate, then
1071
- * reconciled against reported usage once the call completes. Omit for
1072
- * unlimited.
1073
- */
1074
- tokensPerMinute?: number;
1075
- /** Max requests in flight at once. Default 0, meaning unlimited. */
1076
- maxConcurrent?: number;
1077
- /**
1078
- * Max time a call may sit queued waiting for capacity, in ms. Exceeding
1079
- * it throws rather than hanging forever. Default 30000. Pass 0 to wait
1080
- * indefinitely.
1081
- */
1082
- maxQueueMs?: number;
1083
- /** Max queued calls before new ones reject immediately instead of queueing. Default 0, unbounded. */
1084
- maxQueueSize?: number;
1085
- /**
1086
- * Pre-flight token estimate for `tokensPerMinute`. Defaults to a
1087
- * chars/4 heuristic over message content plus `max_tokens`.
1088
- */
1089
- estimateTokens?: (request: WireRequest) => number;
1090
- /**
1091
- * Scales the pre-flight estimate down before it's reserved against
1092
- * `tokensPerMinute`, since most calls don't use their full `max_tokens`
1093
- * budget. Applied after `estimateTokens`, as rate-limiter bookkeeping
1094
- * only; never changes the `max_tokens` sent to the provider.
1095
- * `release`'s `actualTokens` still reconciles against real usage
1096
- * afterward. Default `1` (today's behavior, no scaling). Must be a
1097
- * finite number greater than `0`; values above `1` are clamped to `1`.
1098
- */
1099
- estimateFraction?: number;
1100
- /**
1101
- * AIMD against the `requestsPerMinute` bucket. Omit for a fixed
1102
- * ceiling, today's behavior. Requires `requestsPerMinute`.
1103
- */
1104
- aimd?: AimdOptions;
1105
- }
1106
- interface AimdOptions {
1107
- /** Added to the requests-per-minute ceiling on every clean release. */
1108
- increaseBy: number;
1109
- /** Multiplied against the ceiling on a rate-limit signal. Must be greater than `0` and at most `1`; clamped otherwise. */
1110
- decreaseFactor: number;
1111
- /** Floor the ceiling never shrinks below. */
1112
- minCapacity: number;
1113
- /** Ceiling the bucket never grows above. */
1114
- maxCapacity: number;
1115
- /**
1116
- * Shrink proactively once a provider hint reports `remainingRequests`
1117
- * at or below this, before a real 429 happens. Default 0, meaning
1118
- * off.
1119
- */
1120
- proactiveFloor?: number;
1121
- }
1122
- interface RateLimitState {
1123
- /** Requests still available this window, or `undefined` if `requestsPerMinute` isn't configured. */
1124
- requestsRemaining?: number;
1125
- /** Tokens still available this window, or `undefined` if `tokensPerMinute` isn't configured. */
1126
- tokensRemaining?: number;
1127
- /** Concurrency slots currently in use, or `undefined` if `maxConcurrent` isn't configured. */
1128
- concurrentInFlight?: number;
1129
- }
1130
- interface RateLimitAcquireResult {
1131
- /**
1132
- * Releases the concurrency slot this attempt held and reconciles the
1133
- * token bucket against real usage, when `actualTokens` is supplied.
1134
- * Idempotent: only the first call does anything. Must run in a
1135
- * `finally` block so a slot is never leaked on a failed attempt.
1136
- */
1137
- release: (actualTokens?: number, success?: boolean) => void;
1138
- /** How long this attempt waited in queue before capacity was available. */
1139
- waitedMs: number;
1140
- /** Which bucket was blocking this attempt just before it cleared, if any wait happened. */
1141
- reason?: RateLimitReason;
1142
- }
1143
- /** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
1144
- export declare function defaultEstimateTokens(request: WireRequest): number;
1145
- /**
1146
- * What VernLLM's dispatch layer needs from a limiter. `RateLimiter`
1147
- * implements this; a caller wanting cross-process coordination can hand
1148
- * over their own instance instead, see `buildRateLimit`. Every method is
1149
- * required, `RateLimiter` itself already no-ops the AIMD methods when
1150
- * `aimd` isn't configured, so a custom limiter follows the same pattern.
1151
- */
1152
- interface RateLimiterAdapter {
1153
- estimate(request: WireRequest): number;
1154
- acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
1155
- signalRateLimit(): void;
1156
- reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
1157
- /** Optional: current bucket levels, for introspection. Omit if the adapter has no state worth reporting. */
1158
- getState?(): RateLimitState;
1159
- /** Optional: live bucket levels for `VernLLM.readRateLimitState()`, the async counterpart of `getState`. */
1160
- readState?(): Promise<RateLimitState>;
1161
- /** Optional: receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
1162
- setLogger?(logger: Logger): void;
1163
- }
1164
- /**
1165
- * Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
1166
- * concurrency) behind one FIFO queue, so a large call isn't starved by a
1167
- * stream of small ones. Any bucket omitted from `options` has infinite
1168
- * capacity and never blocks.
1169
- */
1170
- export declare class RateLimiter implements RateLimiterAdapter {
1171
- private readonly requests?;
1172
- private readonly tokens?;
1173
- private readonly concurrency?;
1174
- /** Buckets in acquire precedence order (concurrency, rpm, tpm), omitted ones filtered out. Built once so order can't drift between `tryAcquireBuckets` and `scheduleWake`. */
1175
- private readonly buckets;
1176
- private readonly maxQueueMs;
1177
- private readonly maxQueueSize;
1178
- private readonly estimateTokensFn;
1179
- private readonly estimateFraction;
1180
- private readonly aimd?;
1181
- private readonly queue;
1182
- /**
1183
- * A single scheduled re-check for the head of the queue when it's
1184
- * blocked on a bucket that refills on its own clock (rpm/tpm), so a
1185
- * queue that nobody calls `acquire`/`release` on again isn't stuck
1186
- * forever waiting for an external trigger to re-drain it. Not needed
1187
- * for a concurrency block, which only clears via `release`.
1188
- */
1189
- private wakeTimer?;
1190
- constructor(options: RateLimitOptions);
1191
- /**
1192
- * Pre-flight token estimate for a request, per the configured (or
1193
- * default) heuristic, scaled by `estimateFraction`. This is the sole
1194
- * value reserved against `tokensPerMinute` and later reconciled in
1195
- * `release`; the provider-facing `max_tokens` on the request itself is
1196
- * never touched.
1197
- */
1198
- estimate(request: WireRequest): number;
1199
- /**
1200
- * Waits for capacity in every configured bucket, then takes from each.
1201
- * The returned `release` gives the concurrency slot back and reconciles
1202
- * the token bucket against real usage; it must run in a `finally` block.
1203
- */
1204
- acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
1205
- private queueFullError;
1206
- private enqueue;
1207
- /** Takes from every configured bucket as one atomic unit, in `this.buckets`' order. Rolls back whatever was already taken if any bucket lacks capacity. */
1208
- private tryAcquireBuckets;
1209
- /** Drains the queue head first. Stops at the first waiter that still can't proceed, so no one is starved out of turn. */
1210
- private drain;
1211
- /**
1212
- * Schedules a one-shot re-check of the queue for whenever the bucket
1213
- * that's currently blocking the head waiter should next have enough
1214
- * capacity. A no-op for a concurrency block (only `release` can clear
1215
- * that) or while a wake is already pending.
1216
- */
1217
- private scheduleWake;
1218
- /**
1219
- * Builds the one-shot release closure for an acquired slot. Only the
1220
- * concurrency bucket is given back on release; the requests-per-minute
1221
- * bucket is a real spend that only recovers via its own refill, and the
1222
- * tokens bucket is reconciled against `actualTokens` rather than fully
1223
- * refunded, since real tokens really were spent.
1224
- *
1225
- * `success` defaults to `false`: the AIMD ceiling only grows when the
1226
- * caller explicitly confirms a successful attempt. A failed or
1227
- * rate-limited attempt still releases its slot (so nothing leaks), but
1228
- * must not also grow the ceiling right back up after
1229
- * `signalRateLimit()` just shrank it.
1230
- */
1231
- private makeRelease;
1232
- /** Shared guard and resize call behind both AIMD halves below; only the arithmetic differs. */
1233
- private resizeRequestsCeiling;
1234
- /** AIMD's additive-increase half: grows the ceiling by `aimd.increaseBy` on a clean release. No-op without `aimd`/`requestsPerMinute`. */
1235
- private growOnSuccess;
1236
- /**
1237
- * AIMD's multiplicative-decrease half. Called on a real 429, and,
1238
- * where an adapter can produce a hint, proactively via
1239
- * `reactToRateLimitHint`. Never throws or blocks a call itself, only
1240
- * adjusts the ceiling as a side effect.
1241
- */
1242
- signalRateLimit(): void;
1243
- /**
1244
- * AIMD's proactive entry point: shrinks via `signalRateLimit()` if
1245
- * `hint.remainingRequests` is at or below `aimd.proactiveFloor`.
1246
- */
1247
- reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
1248
- /**
1249
- * Current bucket levels, read live rather than cached. `concurrency`
1250
- * tracks free slots internally, so `concurrentInFlight` is reported as
1251
- * `capacity - available`, the inverse of what the bucket itself holds.
1252
- */
1253
- getState(): RateLimitState;
1254
- }
1255
- //#endregion
1256
- //#region src/internal/utils/rate-limit/rateLimitAdapter.utils.d.ts
1257
- /** Not exported. Internal shorthand only, so this union isn't duplicated between the public option fields and `buildRateLimit`'s own signature. */
1258
- type RateLimitOption = RateLimitOptions | RateLimiterAdapter;
1259
- //#endregion
1260
- //#region src/types/fallback.d.ts
1261
- /**
1262
- * One provider to try after the primary (or after an earlier fallback
1263
- * target) fails. Order is the policy: VernLLM never reorders, scores, or
1264
- * selects a target, it only walks the list as given.
1265
- *
1266
- * Most per-target overrides fall back to the parent `VernLLM` instance's
1267
- * own option when omitted, so a target only needs to specify what's
1268
- * actually different about it (a different client/model is the common
1269
- * case). `circuitBreaker`, `rateLimit`, and `retryBudget` are the
1270
- * exception: they are never inherited from the parent, since a breaker,
1271
- * limiter, or budget tuned for the primary provider's limits is rarely
1272
- * right for a fallback's. Leave them unset on a target to run it without
1273
- * one, even if the parent has one configured.
1274
- */
1275
- interface FallbackTarget {
1276
- client: LLMClient;
1277
- model: string;
1278
- /** Label for events, errors, and `TokenUsage.provider`. Default `` `fallback[${index}]` ``. */
1279
- name?: string;
1280
- maxRetries?: number;
1281
- timeoutMs?: number;
1282
- chunkIdleTimeoutMs?: number;
1283
- baseDelayMs?: number;
1284
- defaultMaxTokens?: number;
1285
- defaultTemperature?: number | null;
1286
- defaultReasoningEffort?: 'minimal' | 'low' | 'medium' | 'high';
1287
- defaultBudgetTokens?: number;
1288
- nonRetryableStatus?: number[];
1289
- /** This target's own circuit breaker, independent of every other target's. Not inherited from the parent's `circuitBreaker`. */
1290
- circuitBreaker?: boolean | CircuitBreakerOptions;
1291
- /** This target's own rate limiter, independent of every other target's. Not inherited from the parent's `rateLimit`. */
1292
- rateLimit?: RateLimitOption;
1293
- /** This target's own retry budget, independent of every other target's. Not inherited from the parent's `retryBudget`. */
1294
- retryBudget?: RetryBudgetOptions;
1295
- /**
1296
- * Reclassifies an otherwise-successful result from this target as a
1297
- * failure. Falls back to the parent `VernLLM` instance's own
1298
- * `detectSoftFailure` when omitted, same as most other per-target
1299
- * options (unlike `circuitBreaker`/`rateLimit`, which never inherit).
1300
- */
1301
- detectSoftFailure?: DetectSoftFailure;
1302
- }
1303
- /**
1304
- * Written into `CallParams['meta']` once `call()` resolves, so a caller
1305
- * who wants provider identity on the same line as the result doesn't need
1306
- * to read it back out of `onUsage`.
1307
- */
1308
- interface CallMeta {
1309
- provider: string;
1310
- model: string;
1311
- /** `-1` if the primary target answered, otherwise the index into `fallback`. */
1312
- fallbackIndex: number;
1313
- usedFallback: boolean;
1314
- /** Attempts made against the target that ultimately answered, including the successful one. */
1315
- attempts: number;
1316
- }
1317
- /** One target's circuit state, as returned by `VernLLM.getCircuitStates()`. */
1318
- interface TargetCircuitState {
1319
- provider: string;
1320
- /** Position in the chain: `0` for the primary, `1`+ for fallback targets. */
1321
- index: number;
1322
- isFallback: boolean;
1323
- /** Whether this target tracks failures per model. `false` means `model` on `getCircuitStates` had no effect on this entry. */
1324
- isolateByModel: boolean;
1325
- /** `undefined` if that target has no circuit breaker configured. */
1326
- state: CircuitState | undefined;
1327
- }
1328
- /** Which target/model `VernLLM.getCircuitState`, `openCircuit`, and `closeCircuit` act on. */
1329
- interface CircuitTarget {
1330
- /** Which target to act on. `0` is the primary, `1`+ are fallbacks. Defaults to `0`. */
1331
- index?: number;
1332
- /** Which model bucket to act on, if the resolved target isolates by model. */
1333
- model?: string;
1334
- }
1335
- /**
1336
- * One target's failure, recorded on the way to either the next target or
1337
- * `FallbackExhaustedError`. Extends `RetryAttempt`: `index` is `-1` for
1338
- * the primary target here (rather than a plain retry count), and
1339
- * `provider`/`model` identify which target failed.
1340
- */
1341
- interface FallbackAttempt extends RetryAttempt {
1342
- provider: string;
1343
- model: string;
1344
- }
1345
- /**
1346
- * Decides what happens after a target's own retries are exhausted or
1347
- * abandoned early. Called once per failed target. `'retry'` is not a
1348
- * valid return here: retrying already happened inside the target, this
1349
- * only decides whether to move on to the next one or stop.
1350
- */
1351
- type FallbackOn = (error: LLMError, context: {
1352
- isLastTarget: boolean;
1353
- }) => 'next' | 'stop';
1354
- /**
1355
- * The default `fallbackOn` policy. Exported so a caller can wrap rather
1356
- * than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
1357
- */
1358
- export declare const defaultFallbackOn: FallbackOn;
1359
- /**
1360
- * Thrown when the chain gives up, whether because the last target failed
1361
- * or `fallbackOn` chose to stop early. Carries each attempt in order so
1362
- * an outage across providers stays debuggable without reproducing it.
1363
- * Extends `LLMError` so `isLLMError` and any `instanceof LLMError` check
1364
- * still passes, inheriting the last failure's `type`/`status`/`retryAfterMs`
1365
- * so existing type-based handling, including reading `retryAfterMs` on an
1366
- * `'api'`-typed error, keeps working on a fallback-exhausted error too.
1367
- */
1368
- export declare class FallbackExhaustedError extends LLMError {
1369
- readonly attempts: FallbackAttempt[];
1370
- constructor(attempts: FallbackAttempt[]);
1371
- /**
1372
- * `type: 'fallback_exhausted'` by itself says nothing about whether
1373
- * retrying could help; the reason the last target failed does. Defers to
1374
- * that attempt's own `retryable` instead of anything about this class's
1375
- * own type.
1376
- */
1377
- get retryable(): boolean;
1378
- }
1379
- /** Narrows `err` to {@link FallbackExhaustedError}, for direct access to its `attempts` (`provider`/`model` per failed target) without a manual `instanceof` check. */
1380
- export declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
1381
- /**
1382
- * Creates an empty ref box to pass as `CallParams['meta']`, so a caller can
1383
- * read the `CallMeta` written by `call()` on the same line as the result
1384
- * instead of pre-declaring a `{ current?: CallMeta }` by hand.
1385
- *
1386
- * @example
1387
- * const meta = metaRef();
1388
- * const result = await vern.call({ userContent: '...', meta });
1389
- * meta.current?.provider;
1390
- */
1391
- export declare function metaRef(): {
1392
- current?: CallMeta;
1393
- };
1394
- //#endregion
1395
- //#region src/types/schema.d.ts
1396
- /**
1397
- * Minimal structural type for a Zod-like schema, so this package doesnt need
1398
- * a hard dependency on a specific Zod major version. Any object exposing
1399
- * `safeParse` (Zod v3/v4, and most Zod-compatible validators) should satisfy this
1400
- */
1401
- interface SchemaLike<T> {
1402
- safeParse(data: unknown): {
1403
- success: true;
1404
- data: T;
1405
- } | {
1406
- success: false;
1407
- error: unknown;
1408
- };
1409
- }
1410
- /**
1411
- * A provider-native JSON Schema for structured outputs (OpenAI/Groq
1412
- * `response_format: { type: 'json_schema' }`) This is the wire-format
1413
- * schema the model is constrained to generate against, distinct from
1414
- * `schema`, which is a client-side Zod validator run on the parsed result
1415
- * You can use one, both, or neither; using both gets you provider-level
1416
- * constraint plus client-side type inference/validation as a safety net
1417
- */
1418
- interface JsonSchemaSpec {
1419
- name: string;
1420
- schema: Record<string, unknown>;
1421
- /** Enforces the schema strictly (OpenAI-specific), default true when supported */
1422
- strict?: boolean;
1423
- description?: string;
1424
- }
1425
- //#endregion
1426
- //#region src/types/tools.d.ts
1427
- /**
1428
- * Describes a capability the model may request, not the capability
1429
- * itself. VernLLM transports this to the provider and parses what comes
1430
- * back; it never executes anything.
1431
- */
1432
- interface ToolDefinition<Name extends string = string, Args = unknown> {
1433
- name: Name;
1434
- description: string;
1435
- /** JSON Schema for the tool's input. */
1436
- parameters: Record<string, unknown>;
1437
- /**
1438
- * Optional client-side validator run on the parsed `arguments` before
1439
- * they're handed back to the caller, mirroring the `schema: SchemaLike<T>`
1440
- * pattern already used for response validation (see `types/schema.ts`).
1441
- * Reuses that zero-dependency, `safeParse`-compatible shape instead of
1442
- * requiring a JSON Schema validator (e.g. ajv) as a new dependency.
1443
- * Failed validation throws `LLMError('validation')`. If omitted, VernLLM
1444
- * parses arguments as JSON but does not validate them further.
1445
- *
1446
- * When set, `Args` (and therefore `Name`) flow into the `ToolCall`s
1447
- * returned by `call()`/`cachedCall()`, provided the tool was declared
1448
- * with `defineTool()` or otherwise has a literal `name`; see
1449
- * `defineTool()` below for why a plain object literal often doesn't.
1450
- */
1451
- argumentsSchema?: SchemaLike<Args>;
1452
- }
1453
- /**
1454
- * Preserves a tool definition's literal `name` (and its `argumentsSchema`'s
1455
- * inferred `Args`) so it can discriminate a `ToolCall` union later.
1456
- *
1457
- * A plain object literal like `{ name: 'get_weather', ... }` widens `name`
1458
- * to `string` unless annotated `as const`, which silently defeats
1459
- * `ToolCall` narrowing the moment a second tool is added to the same
1460
- * `tools: [...]` array (single-tool arrays still narrow fine even without
1461
- * this, since there's nothing to discriminate against but that stops
1462
- * being true as soon as a second tool shows up). Wrapping the same object
1463
- * in `defineTool()` preserves the literal `name` type without requiring
1464
- * `as const` at every call site.
1465
- */
1466
- export declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
1467
- /** Maps a single `ToolDefinition` to its matching `ToolCall` shape. */
1468
- type ToolCallFor<T> = T extends ToolDefinition<infer N, infer A> ? {
1469
- id: string;
1470
- name: N;
1471
- arguments: A;
1472
- } : never;
1473
- /**
1474
- * A single tool invocation requested by the model.
1475
- *
1476
- * When `Tools` is a literal tuple (e.g. inferred from `tools: [getWeather,
1477
- * cancelOrder]` at a `call()`/`cachedCall()` site), this is a discriminated
1478
- * union keyed by `name`. Checking `call.name === 'get_weather'` narrows
1479
- * `call.arguments` to that tool's `Args` with no cast needed. Without a
1480
- * literal `Tools` (the default), this collapses back to today's
1481
- * `{ id: string; name: string; arguments: unknown }`.
1482
- */
1483
- type ToolCall<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ToolCallFor<Tools[number]>;
1484
- /** The application's result of executing a `ToolCall`, sent back to the model. */
1485
- interface ToolResult {
1486
- toolCallId: string;
1487
- content: unknown;
1488
- /**
1489
- * Signals a failed tool execution back to the model (matches Anthropic's
1490
- * native `is_error` on tool_result blocks). Only `fromAnthropic` honors
1491
- * this today, Gemini and Bedrock have no equivalent wire concept, so
1492
- * other adapters ignore it silently.
1493
- */
1494
- isError?: boolean;
1495
- }
1496
- /** `call()` result when `tools` was set and the model produced a normal answer. */
1497
- interface ContentResult<T> {
1498
- type: 'content';
1499
- content: T;
1500
- }
1501
- /** `call()` result when `tools` was set and the model requested one or more tools. */
1502
- interface ToolCallResult<Tools extends readonly ToolDefinition[] = ToolDefinition[]> {
1503
- type: 'tool_calls';
1504
- toolCalls: ToolCall<Tools>[];
1505
- /** Any text the model produced alongside the tool request, if present. */
1506
- content?: string;
1507
- }
1508
- type CallWithToolsResult<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ContentResult<T> | ToolCallResult<Tools>;
1509
- /** Recovers `Tools` from a `result` already typed `ContentResult<T> | ToolCallResult<Tools>`. Falls back to `ToolDefinition[]`. */
1510
- type ExtractTools<R> = Extract<R, ToolCallResult<ToolDefinition[]>> extends ToolCallResult<infer Tools> ? Tools : ToolDefinition[];
1511
- /** Explicit `Tools` type argument if given, otherwise inferred via `ExtractTools`. `never` marks "unset". */
1512
- type ResolvedTools<Tools, R> = [Tools] extends [never] ? ExtractTools<R> : Tools;
1513
- /**
1514
- * Runtime check for whether a `call()` result is a `tool_calls` result. Use
1515
- * this instead of trusting static narrowing whenever `tools` was set
1516
- * conditionally, see `ConditionalToolCallParams`.
1517
- *
1518
- * ```ts
1519
- * const result = await llm.call({ userContent: '...', tools: someCondition ? [myTool] : undefined });
1520
- * if (isToolCallResult(result)) {
1521
- * // result.toolCalls[number].arguments typed per tool, inferred automatically
1522
- * }
1523
- * ```
1524
- *
1525
- * Pass `Tools` explicitly to override inference, e.g. `isToolCallResult<typeof tools>(result)`.
1526
- */
1527
- export declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
1528
- /** What the model should do about tools on a given call. */
1529
- type ToolChoice = 'auto' | 'none' | 'required' | {
1530
- name: string;
1531
- };
1532
- //#endregion
1533
- //#region src/types/call.d.ts
1534
- /**
1535
- * Any valid JSON value: a primitive, `null`, or a JSON array/object made
1536
- * of the same. This is what `call()` returns when `jsonMode: true`.
1537
- */
1538
- type JsonValue = string | number | boolean | null | JsonValue[] | {
1539
- [key: string]: JsonValue;
1540
- };
1541
- /**
1542
- * Content for an `assistant` turn in `history`. Accepts a string or a
1543
- * parsed `JsonValue`, so a prior `jsonMode: true` response can be pushed
1544
- * straight back into history. Request construction stringifies non-string
1545
- * content before it's sent to the provider.
1546
- */
1547
- type AssistantContent = string | JsonValue;
1548
- /**
1549
- * A single prior turn in a multi-turn conversation, passed via `history`.
1550
- *
1551
- * Supports normal user/assistant messages and tool continuations: an assistant
1552
- * turn may include `toolCalls`, and a tool turn carries the matching
1553
- * `toolResults`. A tool turn must immediately follow an assistant tool call
1554
- * turn, and every requested tool call must have a result.
1555
- */
1556
- type ConversationTurn = {
1557
- role: 'user';
1558
- content: string;
1559
- } | {
1560
- role: 'assistant';
1561
- content?: AssistantContent;
1562
- toolCalls?: ToolCall[];
1563
- } | {
1564
- role: 'tool';
1565
- toolResults: ToolResult[];
1566
- };
1567
- /** A plain text segment of a multimodal `userContent` array. */
1568
- interface TextBlock {
1569
- type: 'text';
1570
- text: string;
1571
- }
1572
- /**
1573
- * An inline image segment of a multimodal `userContent` array.
1574
- *
1575
- * `data` is the raw base64-encoded image bytes, with no `data:` URL prefix
1576
- * (adapters that need a data URL, e.g. OpenAI-compatible `image_url`, build
1577
- * it themselves from `mimeType` + `data`; adapters that need raw bytes, e.g.
1578
- * Bedrock, decode the base64 themselves).
1579
- */
1580
- interface ImageBlock {
1581
- type: 'image';
1582
- /** Base64-encoded image bytes, no `data:` prefix */
1583
- data: string;
1584
- /** e.g. 'image/png', 'image/jpeg', 'image/webp', 'image/gif' */
1585
- mimeType: string;
1586
- }
1587
- /** A single segment of multimodal `userContent`. */
1588
- type ContentBlock = TextBlock | ImageBlock;
1589
- /**
1590
- * Every field of a call request except the `reserveUsage`/`refundUsage`
1591
- * hooks from `UsageHooks`. `CallParams` is this plus `UsageHooks`; the
1592
- * `Cached*` param types below are call sites that want the request shape
1593
- * without those two hooks (usage is metered once, at the `cachedCall`
1594
- * level, not per-request), and use this directly instead of re-deriving
1595
- * it with `Omit<CallParams<T>, 'reserveUsage' | 'refundUsage'>` each time.
1596
- */
1597
- interface LLMRequestShape<T = unknown, Tools extends readonly ToolDefinition[] = ToolDefinition[]> {
1598
- systemPrompt?: string;
1599
- /** Current user message, as text or multimodal content blocks. */
1600
- userContent: string | ContentBlock[];
1601
- /**
1602
- * Previous conversation turns. Must alternate roles; tool turns must follow
1603
- * assistant tool calls. Invalid history throws LLMError('invalid_params').
1604
- */
1605
- history?: ConversationTurn[];
1606
- /**
1607
- * Generation temperature. Default 0.2, not the provider's own default.
1608
- * Pass `null` to omit `temperature` from the request entirely, so the
1609
- * provider applies its own default instead.
1610
- */
1611
- temperature?: number | null;
1612
- jsonMode?: boolean;
1613
- maxTokens?: number;
1614
- requestId?: string;
1615
- signal?: AbortSignal;
1616
- /**
1617
- * Total time budget in ms for this whole call, across every retry and
1618
- * every fallback target. Unlike timeoutMs, which resets on each attempt,
1619
- * this is a single clock starting when call is invoked. The call is
1620
- * aborted once this elapses, even mid retry or mid fallback, the same
1621
- * way an aborted signal is today. Omit for no overall deadline, only
1622
- * the existing per attempt timeoutMs applies.
1623
- *
1624
- * Only bounds getting to a final result: choosing a target, retrying,
1625
- * and opening a stream. It does not extend to the time spent reading a
1626
- * stream after it has opened. Use chunkIdleTimeoutMs for gaps between
1627
- * chunks once a stream is open.
1628
- */
1629
- deadlineMs?: number;
1630
- /**
1631
- * Per-call override for the instance's `chunkIdleTimeoutMs` (max gap
1632
- * between stream chunks once opened). Only applies when `stream: true`.
1633
- * Useful for routes using reasoning-heavy models with documented long
1634
- * silent gaps mid-stream. Pass 0 to disable the idle timeout for this
1635
- * call.
1636
- */
1637
- chunkIdleTimeoutMs?: number;
1638
- /** Overrides the instance model for this call. */
1639
- model?: string;
1640
- /**
1641
- * Reasoning effort for supported reasoning models. Pass `null` to
1642
- * explicitly skip an instance-level `defaultReasoningEffort` for this
1643
- * one call (e.g. a call using a forced `toolChoice`, which Anthropic
1644
- * rejects alongside any reasoning at all), the same way `temperature:
1645
- * null` opts a call out of `defaultTemperature`. Omitting the field
1646
- * entirely (`undefined`) defers to the instance default instead.
1647
- */
1648
- reasoningEffort?: 'minimal' | 'low' | 'medium' | 'high' | null;
1649
- /**
1650
- * Token budget for internal reasoning, for models with a native numeric
1651
- * budget (Anthropic's `budget_tokens`, Gemini's `thinkingBudget`). On a
1652
- * provider that only understands `reasoningEffort` tiers (OpenAI-
1653
- * compatible), this is converted to the nearest tier instead of sent as
1654
- * a raw number. When both `budgetTokens` and `reasoningEffort` are set,
1655
- * each adapter prefers whichever field it natively understands and
1656
- * ignores the other. See the reasoning budget docs for the conversion
1657
- * table used in each direction. Pass `null` to explicitly skip an
1658
- * instance-level `defaultBudgetTokens` for this one call, mirroring
1659
- * `reasoningEffort: null` above; omitting the field entirely defers to
1660
- * the instance default.
1661
- */
1662
- budgetTokens?: number | null;
1663
- /**
1664
- * Provider-native JSON Schema output constraint. Implies jsonMode: true.
1665
- */
1666
- jsonSchema?: JsonSchemaSpec;
1667
- /**
1668
- * Validates parsed JSON output. Failure throws LLMError('validation').
1669
- * Implies jsonMode: true.
1670
- */
1671
- schema?: SchemaLike<T>;
1672
- /**
1673
- * Tools the model may call. When set, `call()` returns a
1674
- * `CallWithToolsResult<T>` union instead of `T` directly. Combining with
1675
- * `jsonSchema` is provider-dependent; see the Tool Calling docs.
1676
- *
1677
- * Passed as a literal array (or via `defineTool()`-wrapped entries, see
1678
- * `types/tools.ts`), this also drives the `Tools` type parameter, which
1679
- * narrows `CallWithToolsResult`'s `toolCalls[number].arguments` per tool.
1680
- */
1681
- tools?: Tools;
1682
- /** Defaults to `'auto'` when `tools` is set. */
1683
- toolChoice?: ToolChoice;
1684
- /**
1685
- * Streams the response incrementally instead of resolving once. Default:
1686
- * false. Requires a client/adapter that implements `createStream`.
1687
- * Retry/timeout/circuit-breaker guarantees apply only to opening the
1688
- * stream (through the first chunk); a failure after that point rejects
1689
- * `finalResult` directly and is not retried, since a mid-stream failure
1690
- * isn't connection-time evidence for the circuit breaker, the attempt
1691
- * already counted as a success once the first chunk arrived. Once the
1692
- * stream opens successfully, `finalResult` still resolves to the same
1693
- * validated `T`/`CallWithToolsResult<T>` shape `call()` would have
1694
- * returned for the same params with `stream` omitted. See
1695
- * `StreamCallResult`.
1696
- */
1697
- stream?: boolean;
1698
- /**
1699
- * Optional out-parameter for provider identity. Pass `{}` (or any object
1700
- * with a mutable `current` property) and `call()` writes a `CallMeta`
1701
- * into `meta.current` before returning, alongside whatever `onUsage`
1702
- * already reports. This includes `stream: true`: the target is chosen
1703
- * once the stream opens, which is also the point `call()` itself
1704
- * returns `{ chunks, finalResult }`, so `meta.current` is already set
1705
- * by then. `TokenUsage.provider`/`usedFallback` from `onUsage` reports
1706
- * the same information asynchronously, for both streaming and
1707
- * non-streaming calls.
1708
- *
1709
- * `meta.current` is only written once execution actually reaches and
1710
- * selects a provider target. A `wrap` middleware that short-circuits
1711
- * without calling `next()` never reaches that point, so `meta.current`
1712
- * is left untouched; if the same holder object is reused across calls,
1713
- * it can still hold a prior call's target.
1714
- */
1715
- meta?: {
1716
- current?: CallMeta;
1717
- };
1718
- }
1719
- interface CallParams<T = unknown, Tools extends readonly ToolDefinition[] = ToolDefinition[]> extends LLMRequestShape<T, Tools>, UsageHooks {}
1720
- /**
1721
- * A `CallParams` variant where tool calling is explicitly enabled.
1722
- *
1723
- * Requiring `tools` to be present allows TypeScript to select the
1724
- * tool-aware `call()` overload and return `CallWithToolsResult<T>` instead
1725
- * of the normal `T` response type.
1726
- */
1727
- type ToolEnabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1728
- tools: NonNullable<CallParams<T, Tools>['tools']>;
1729
- };
1730
- /**
1731
- * A `CallParams` variant for tools set conditionally, e.g. `tools:
1732
- * someCondition ? [myTool] : undefined`. Selects the `call()` overload
1733
- * returning the honest union `T | CallWithToolsResult<T, Tools>` instead of
1734
- * falling through to plain `T` (which is what happened before this type
1735
- * existed, since `ToolDefinition[] | undefined` matched neither
1736
- * `ToolEnabledCallParams` nor `ToolsDisabledCallParams`). Forces an
1737
- * `isToolCallResult()` check before treating the result as plain
1738
- * content. Omitting `tools` entirely still resolves to plain `T`, since
1739
- * tools genuinely cannot have run there.
1740
- *
1741
- * `Tools` still can't reliably infer a literal tuple here the way
1742
- * `ToolEnabledCallParams` does for an inline array (a ternary/variable
1743
- * expression doesn't carry the same `const`-literal preservation), so
1744
- * getting typed `arguments` out of a conditional-tools result also needs
1745
- * an explicit `Tools` type argument on `isToolCallResult<Tools>()` when
1746
- * narrowing, see its docs.
1747
- */
1748
- type ConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1749
- tools: Tools | undefined;
1750
- };
1751
- /** Conditional tool-call parameters whose non-tool result is plain text. */
1752
- type ConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ConditionalToolCallParams<string, Tools> & {
1753
- jsonMode: false;
1754
- };
1755
- /**
1756
- * A `CallParams` variant where tools are offered but the model is barred
1757
- * from calling one. `toolChoice: 'none'` guarantees the response can never
1758
- * be a `tool_calls` result, so `call()` can narrow straight to
1759
- * `ContentResult<T>` instead of the full `CallWithToolsResult<T>` union.
1760
- * A call site that already knows it forced `'none'` no longer needs a
1761
- * runtime `isToolCallResult` check, or to remember that `String(result)`
1762
- * on the wrapper object silently produces `"[object Object]"` instead of
1763
- * throwing. The type itself rules that shape out.
1764
- */
1765
- type ToolsDisabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1766
- tools: NonNullable<CallParams<T, Tools>['tools']>;
1767
- toolChoice: 'none';
1768
- };
1769
- /**
1770
- * `CallParams` with `jsonMode: false`. Selects the `call()` overload
1771
- * that returns a plain `string`. `jsonSchema` is typed `never` here: a
1772
- * truthy `jsonSchema` forces JSON parsing at runtime regardless of
1773
- * `jsonMode` (see `RequestBuilder.build()`), so `jsonMode: false` +
1774
- * `jsonSchema` together would otherwise still match this overload and
1775
- * falsely promise a `string`.
1776
- */
1777
- type JsonModeDisabledCallParams = Omit<CallParams<unknown>, 'jsonSchema'> & {
1778
- jsonMode: false;
1779
- jsonSchema?: never;
1780
- };
1781
- /**
1782
- * `CallParams` with `jsonMode: true` and no `schema`. Selects the
1783
- * `call()` overload that returns a `JsonValue`.
1784
- *
1785
- * `schema` is explicitly typed `never` here, not just omitted: `CallParams<JsonValue>['schema']`
1786
- * would be `SchemaLike<JsonValue> | undefined`, and a schema whose inferred result type is
1787
- * itself structurally assignable to `JsonValue` (e.g. a schema for `string[]` or
1788
- * `Record<string, string>`) would still satisfy that shape, incorrectly selecting this
1789
- * overload over the schema-aware generic one and widening the result to `JsonValue`. Forcing
1790
- * `schema?: never` makes any call that sets `schema` fail this overload's structural check
1791
- * regardless of the schema's result type, so it always falls through to the generic
1792
- * `CallParams<T>` overload and infers `T` from the schema instead.
1793
- */
1794
- type JsonModeEnabledCallParams = Omit<CallParams<JsonValue>, 'schema'> & {
1795
- jsonMode: true;
1796
- schema?: never;
1797
- };
1798
- /** Shared cache-configuration fields, minus the internal `fn` primitive. */
1799
- interface CachedCallInput extends UsageHooks {
1800
- cacheKey: string;
1801
- ttl: number;
1802
- signal?: AbortSignal;
1
+ import { $ as SchemaLike, $t as createStateKey, A as ConditionalToolCallParams, At as ConsecutiveTripping, B as ThinkingBlock, Bt as MiddlewareRef, C as CachedConditionalToolCallParams, Cn as hasIssues, Ct as RetryBudgetOptions, D as CallContext, Dt as CircuitBreakerOptions, E as CachedToolCallParams, Et as CircuitBreakerCallContext, F as JsonModeDisabledCallParams, Ft as AttemptContext, G as ToolCall, Gt as RequiredMiddlewareRef, H as ToolsDisabledCallParams, Ht as MiddlewareStateEntry, I as JsonModeEnabledCallParams, It as CallResult, J as ToolDefinition, Jt as WireCallRequestPatch, K as ToolCallResult, Kt as VernLLMMiddleware, L as JsonValue, Lt as MiddlewareCapabilities, M as ConversationTurn, Mt as ExponentialBackoffOptions, N as DetectSoftFailure, Nt as RollingTripping, O as CallParams, Ot as CircuitBreakerStateChangeHandler, P as ImageBlock, Pt as TrippingPolicy, Q as JsonSchemaSpec, Qt as createMiddlewareStateBag, R as LLMRequestShape, Rt as MiddlewareContext, S as CachedConditionalStringToolCallParams, Sn as UnsupportedCapabilityIssue, St as RetryBudget, T as CachedJsonModeEnabledCallParams, Tt as CircuitBreakerAdapter, U as CallWithToolsResult, Ut as MiddlewareStateKey, V as ToolEnabledCallParams, Vt as MiddlewareStateBag, W as ContentResult, Wt as PreDispatchContext, X as defineTool, Xt as WireTool, Y as ToolResult, Yt as WireResponseFormat, Z as isToolCallResult, Zt as createMiddlewareRef, _ as StreamJsonModeEnabledCallParams, _n as LLMErrorType, _t as RateLimiter, a as WireToolChoice, an as OnUsageFailure, at as FallbackOnContext, b as AssistantContent, bn as ToolIssue, c as CachedStreamConditionalToolCallParams, cn as TokenUsage, ct as TargetInfo, d as CachedStreamToolCallParams, dn as DuplicateToolNamesIssue, dt as metaRef, en as requireRef, et as CallMeta, f as StreamCallResult, fn as HistoryToolResultIssue, ft as RateLimitOption, g as StreamJsonModeDisabledCallParams, gn as LLMErrorSnapshot, gt as RateLimitState, h as StreamEnabledCallParams, hn as LLMErrorIssuesByCode, ht as RateLimitReason, i as WireToolCall, in as OnUsage, it as FallbackOn, j as ContentBlock, jt as CooldownBackoff, k as ConditionalStringToolCallParams, kt as CircuitState, l as CachedStreamJsonModeDisabledCallParams, ln as ConsoleLogger, lt as defaultFallbackOn, m as StreamConditionalStringToolCallParams, mn as LLMErrorCode, mt as RateLimitOptions, n as LLMClient, nn as OnEvent, nt as FallbackAttempt, o as CachedStreamCallParams, on as RefundUsage, ot as FallbackTarget, p as StreamChunk, pn as LLMError, pt as RateLimitAcquireResult, q as ToolChoice, qt as WireCallRequest, r as WireMessage, rn as VernLLMEvent, rt as FallbackExhaustedError, s as CachedStreamConditionalStringToolCallParams, sn as ReserveUsage, st as TargetCircuitState, t as AdapterInfo, tn as stateEntry, tt as CircuitTarget, u as CachedStreamJsonModeEnabledCallParams, un as Logger, ut as isFallbackExhaustedError, v as WireStreamChunk, vn as LLMRequestSnapshot, vt as RateLimiterAdapter, w as CachedJsonModeDisabledCallParams, wn as isLLMError, wt as CircuitBreaker, x as CachedCallParams, xn as UnknownToolChoiceIssue, xt as CircuitBreakerOption, y as isStreamResult, yn as RetryAttempt, yt as WireRequest, z as TextBlock, zt as MiddlewareContextBase } from "./client-HWxkwVvj.cjs";
2
+ //#region src/types/cache.d.ts
3
+ interface CacheAdapter<T = unknown> {
4
+ get(key: string): Promise<{
5
+ hit: boolean;
6
+ value: T | null;
7
+ }>;
8
+ set(key: string, value: T, ttl: number): Promise<void>;
9
+ delete?(key: string): Promise<void>;
10
+ resolveKey?(key: string): Promise<string>;
1803
11
  }
1804
12
  /**
1805
- * Parameters for a cached LLM call without tool calling: cache config
1806
- * plus the `CallParams` passed to `call()`. `reserveUsage`/`refundUsage`
1807
- * belong at the top level (`CachedCallInput`), not nested in `call`; see
1808
- * the caching docs for why.
1809
- */
1810
- type CachedCallParams<T> = CachedCallInput & {
1811
- call: LLMRequestShape<T>;
1812
- };
1813
- /**
1814
- * Parameters for a cached LLM call with tool calling enabled.
1815
- *
1816
- * The cached value includes the full `CallWithToolsResult<T>`, meaning
1817
- * tool requests and normal content responses are cached exactly as returned
1818
- * by the model.
1819
- *
1820
- * See `CachedCallParams` for why `reserveUsage`/`refundUsage` are omitted
1821
- * from `call`'s type here too.
1822
- */
1823
- type CachedToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
1824
- call: LLMRequestShape<T, Tools> & {
1825
- tools: NonNullable<LLMRequestShape<T, Tools>['tools']>;
1826
- };
1827
- };
1828
- /**
1829
- * Parameters for a cached LLM call with `call.tools` set conditionally.
1830
- * Selects the `cachedCall()` overload that returns the honest union
1831
- * `T | CallWithToolsResult<T>` instead of narrowing to plain `T`. See
1832
- * `ConditionalToolCallParams` for why this overload exists.
1833
- */
1834
- type CachedConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
1835
- call: LLMRequestShape<T, Tools> & {
1836
- tools: Tools | undefined;
1837
- };
1838
- };
1839
- /** Cached conditional tool-call parameters whose non-tool result is plain text. */
1840
- type CachedConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedConditionalToolCallParams<string, Tools> & {
1841
- call: {
1842
- jsonMode: false;
1843
- };
1844
- };
1845
- /**
1846
- * Parameters for a cached LLM call with `jsonMode: false`. Selects the
1847
- * `cachedCall()` overload that returns a plain `string`.
13
+ * Which entry `InMemoryCacheAdapter` evicts once `maxSize` is exceeded.
14
+ * `'fifo'` (default) drops the oldest inserted entry. `'lru'` drops the
15
+ * least recently read or written entry.
1848
16
  */
1849
- type CachedJsonModeDisabledCallParams = CachedCallInput & {
1850
- call: Omit<LLMRequestShape<unknown>, 'jsonSchema'> & {
1851
- jsonMode: false;
1852
- jsonSchema?: never;
1853
- };
1854
- };
17
+ type EvictionOption = 'fifo' | 'lru';
1855
18
  /**
1856
- * Parameters for a cached LLM call with `jsonMode: true` and no `schema`.
1857
- * Selects the `cachedCall()` overload that returns a `JsonValue`.
19
+ * The default in-process cache. Not shared across processes; use a shared backend in production.
20
+ * Values are copied on `set` and `get`, so mutating a result never changes the next hit. A value
21
+ * that can't be copied faithfully, or a missing, NaN, zero or negative `ttl`, is not stored and
22
+ * drops any existing entry.
1858
23
  */
1859
- type CachedJsonModeEnabledCallParams = CachedCallInput & {
1860
- call: Omit<LLMRequestShape<JsonValue>, 'schema'> & {
1861
- jsonMode: true;
1862
- schema?: never;
1863
- };
1864
- };
1865
- /** Context handed to `DetectSoftFailure` alongside the response it's inspecting. */
1866
- interface SoftFailureMeta {
1867
- requestId: string;
1868
- model: string;
1869
- providerName: string;
1870
- isFallback: boolean;
1871
- /** 1-based, matching `CallMeta.attempts`. */
1872
- attempt: number;
1873
- /**
1874
- * Token usage for this attempt, if the provider reported it on this
1875
- * response. `undefined` when the provider omitted usage, not when
1876
- * usage was zero, so a cost check should treat a missing value as
1877
- * unknown rather than as free.
1878
- */
1879
- usage?: TokenUsage;
24
+ export declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
25
+ private readonly maxSize;
26
+ private store;
27
+ private readonly eviction;
28
+ constructor(maxSize?: number, eviction?: EvictionOption);
29
+ get(key: string): Promise<{
30
+ hit: boolean;
31
+ value: T | null;
32
+ }>;
33
+ set(key: string, value: T, ttl: number): Promise<void>;
34
+ delete(key: string): Promise<void>;
35
+ private cleanupExpiredEntries;
36
+ private enforceSizeLimit;
1880
37
  }
1881
38
  /**
1882
- * Inspects an otherwise-successful result and optionally reclassifies it
1883
- * as a failure. Returning `undefined` leaves the result as a success;
1884
- * returning an `LLMErrorCode` fails the attempt with that code, feeding
1885
- * the same retry and circuit-breaker paths a thrown error would. A
1886
- * result that parses fine but is empty, truncated, or a low-confidence
1887
- * refusal is otherwise invisible to both.
1888
- */
1889
- type DetectSoftFailure<T = unknown> = (result: T | CallWithToolsResult<T>, meta: SoftFailureMeta) => LLMErrorCode | undefined;
1890
- //#endregion
1891
- //#region src/types/stream.d.ts
1892
- /** One incremental unit of a streaming response, as delivered to the caller. */
1893
- type StreamChunk = {
1894
- type: 'text-delta';
1895
- delta: string;
1896
- } | {
1897
- type: 'tool_call_delta';
1898
- index: number;
1899
- id?: string;
1900
- name?: string;
1901
- argsDelta?: string;
1902
- /**
1903
- * True when `argsDelta` is the whole set of arguments, not a
1904
- * fragment. Set for Gemini (its API returns function-call args
1905
- * whole in one chunk) and for cache/replay chunks, which are
1906
- * one-shot too. Omitted or `false` for a genuine fragment from
1907
- * providers that do stream incrementally (OpenAI-compatible,
1908
- * Anthropic, Bedrock).
1909
- */
1910
- complete?: boolean;
1911
- } | {
1912
- type: 'usage';
1913
- usage: TokenUsage;
1914
- };
1915
- /**
1916
- * What `call()` returns when `stream: true`. `finalResult` resolves to
1917
- * the same shape `call()` would have returned with `stream` omitted.
1918
- * `chunks` is single-use and buffered; see the streaming docs for the
1919
- * full consumption/backpressure semantics.
39
+ * Normalizes keys to avoid duplicates from formatting that can't change a prompt's meaning: Unicode
40
+ * composition, line endings and outer whitespace. Case, punctuation and inner whitespace are kept,
41
+ * since `"2+2"` and `"2-2"` must never share an answer.
1920
42
  */
1921
- interface StreamCallResult<R> {
1922
- chunks: AsyncIterable<StreamChunk>;
1923
- finalResult: Promise<R>;
43
+ export declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
44
+ private readonly inner;
45
+ constructor(inner?: CacheAdapter<T>);
46
+ private normalize;
47
+ resolveKey(key: string): Promise<string>;
48
+ get(key: string): Promise<{
49
+ hit: boolean;
50
+ value: T | null;
51
+ }>;
52
+ set(key: string, value: T, ttl: number): Promise<void>;
53
+ delete(key: string): Promise<void>;
1924
54
  }
1925
55
  /**
1926
- * A `CallParams` variant where streaming is explicitly enabled.
1927
- *
1928
- * Requiring `stream: true` to be statically present allows TypeScript to
1929
- * select the streaming `call()` overload and return `StreamCallResult<...>`
1930
- * instead of the normal, single-shot response type.
1931
- */
1932
- type StreamEnabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1933
- stream: true;
1934
- };
1935
- /** Streaming conditional tool-call parameters whose non-tool result is text. */
1936
- type StreamConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = StreamEnabledCallParams<string, Tools> & ConditionalStringToolCallParams<Tools>;
1937
- /** Recovers `T` from a `result` already typed `T | StreamCallResult<T>`. Falls back to `unknown`. */
1938
- type ExtractStreamValue<R> = Extract<R, StreamCallResult<unknown>> extends StreamCallResult<infer V> ? V : unknown;
1939
- /**
1940
- * Runtime check for whether a `call()` result is a `StreamCallResult`
1941
- * (`{ chunks, finalResult }`) rather than the resolved value directly.
1942
- * Useful when `stream` was computed conditionally and cast/narrowed
1943
- * manually, since TypeScript's `call()` overloads only pick the streaming
1944
- * shape for a literal `stream: true` at the call site.
1945
- *
1946
- * ```ts
1947
- * const params = someCondition ? { userContent: '...', stream: true } : { userContent: '...' };
1948
- * const result = await llm.call(params as CallParams<string> | (CallParams<string> & { stream: true }));
1949
- * if (isStreamResult(result)) {
1950
- * for await (const chunk of result.chunks) { ... }
1951
- * }
1952
- * ```
1953
- */
1954
- export declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
1955
- /**
1956
- * `StreamEnabledCallParams` with `jsonMode: false`. Selects the streaming
1957
- * `call()` overload whose `finalResult` resolves to a plain `string`.
1958
- * `jsonSchema` is typed `never` for the same reason as
1959
- * `JsonModeDisabledCallParams`.
56
+ * Two tier cache: fast local L1, shared L2, with L2 hits promoted to L1. An L1 entry never outlives
57
+ * an L2 entry this adapter wrote. A promoted entry another process wrote keeps `l1Ttl`, since L2
58
+ * doesn't report remaining ttl.
1960
59
  */
1961
- type StreamJsonModeDisabledCallParams = Omit<StreamEnabledCallParams<unknown>, 'jsonSchema'> & {
1962
- jsonMode: false;
1963
- jsonSchema?: never;
1964
- };
1965
- /**
1966
- * `StreamEnabledCallParams` with `jsonMode: true` and no `schema`. Selects
1967
- * the streaming `call()` overload whose `finalResult` resolves to a
1968
- * `JsonValue`.
1969
- *
1970
- * `schema` is explicitly `never` here for the same reason as
1971
- * `JsonModeEnabledCallParams`: a schema whose result type is itself
1972
- * structurally assignable to `JsonValue` would otherwise still satisfy this
1973
- * overload's shape and incorrectly widen the result to `JsonValue` instead
1974
- * of the schema's real type.
1975
- */
1976
- type StreamJsonModeEnabledCallParams = Omit<StreamEnabledCallParams<JsonValue>, 'schema'> & {
1977
- jsonMode: true;
1978
- schema?: never;
1979
- };
1980
- /**
1981
- * The adapter-facing, pre-normalization shape a `createStream` client
1982
- * implementation emits, analogous to how `WireMessage`/`WireToolCall`
1983
- * already sit between `CallParams` and each provider's own wire format.
1984
- */
1985
- type WireStreamChunk = {
1986
- type: 'text-delta';
1987
- delta: string;
1988
- } | {
1989
- type: 'tool_call_delta';
1990
- index: number;
1991
- id?: string;
1992
- name?: string;
1993
- argumentsDelta?: string;
1994
- /** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
1995
- complete?: boolean;
1996
- } | {
1997
- type: 'usage';
1998
- usage: {
1999
- prompt_tokens?: number;
2000
- completion_tokens?: number;
2001
- total_tokens?: number;
2002
- completion_tokens_details?: {
2003
- reasoning_tokens?: number;
2004
- };
2005
- };
2006
- } | {
60
+ export declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
61
+ private readonly l1;
62
+ private readonly l2;
63
+ private readonly l1Ttl?;
64
+ /** L2 expiry (epoch ms) per key this adapter wrote, oldest write first. */
65
+ private readonly l2ExpiresAt;
2007
66
  /**
2008
- * A provider keep-alive signal with no content of its own (e.g.
2009
- * Anthropic's `ping` events, an SSE comment-line heartbeat).
2010
- * Adapters yield this so the stream loop resets its idle timeout.
2011
- * Never surfaced to callers as a `StreamChunk`.
67
+ * Expiries as a min heap, so a write only visits expired records. Stale heap entries are skipped
68
+ * unless the map still holds that exact expiry.
2012
69
  */
2013
- type: 'ping';
2014
- } | {
70
+ private expiryHeap;
71
+ constructor(l1: CacheAdapter<T>, l2: CacheAdapter<T>, l1Ttl?: number | undefined);
2015
72
  /**
2016
- * AIMD's proactive rate-limit hint, read off the stream's
2017
- * response headers (where the adapter/SDK can get at them) and
2018
- * yielded once, as early as possible. Mirrors `attachRateLimitHint`
2019
- * for the non-streaming path, just carried as a chunk instead of a
2020
- * hidden property on a response object, since a stream has no
2021
- * single response value to attach one to. Never surfaced to
2022
- * callers as a `StreamChunk`.
73
+ * Forwards to L1's `resolveKey` if it has one, otherwise L2's. L1 is
74
+ * preferred since `get()` checks L1 first, so its notion of "the same
75
+ * key" is the one that determines whether a lookup can skip L2 entirely.
2023
76
  */
2024
- type: 'rate_limit_hint';
2025
- hint: ProviderRateLimitHint;
2026
- };
2027
- /**
2028
- * Parameters for a cached, streaming LLM call without tool calling.
2029
- *
2030
- * The cached value is `T`, same as `CachedCallParams<T>`, but a miss
2031
- * relays live `chunks` to the caller while the result is being generated,
2032
- * and a hit synthesizes a one-shot `chunks` replay from the cached value
2033
- * (see `VernLLM.cachedCall`'s docs for exactly what that replay looks
2034
- * like).
2035
- *
2036
- * `reserveUsage`/`refundUsage` are omitted from `call`'s type; see
2037
- * `CachedCallParams` for why they belong at the top level here too.
2038
- */
2039
- type CachedStreamCallParams<T> = CachedCallInput & {
2040
- call: LLMRequestShape<T> & {
2041
- stream: true;
2042
- };
2043
- };
2044
- /**
2045
- * Parameters for a cached, streaming LLM call with tool calling enabled.
2046
- *
2047
- * The cached value is the full `CallWithToolsResult<T>`, same as
2048
- * `CachedToolCallParams<T>`, with the same live-chunks-on-miss,
2049
- * replayed-chunks-on-hit behavior as `CachedStreamCallParams<T>`.
2050
- */
2051
- type CachedStreamToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
2052
- call: LLMRequestShape<T, Tools> & {
2053
- stream: true;
2054
- tools: NonNullable<LLMRequestShape<T, Tools>['tools']>;
2055
- };
2056
- };
2057
- /**
2058
- * Parameters for a cached, streaming LLM call with `call.tools` set
2059
- * conditionally. Selects the `cachedCall()` overload whose `finalResult`
2060
- * (on a miss) or cached value (on a hit) is the honest union
2061
- * `T | CallWithToolsResult<T>` instead of narrowing to plain `T`. See
2062
- * `ConditionalToolCallParams` for why this overload exists.
2063
- */
2064
- type CachedStreamConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
2065
- call: LLMRequestShape<T, Tools> & {
2066
- stream: true;
2067
- tools: Tools | undefined;
2068
- };
2069
- };
2070
- /** Cached streaming conditional tool-call parameters whose non-tool result is text. */
2071
- type CachedStreamConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedStreamConditionalToolCallParams<string, Tools> & {
2072
- call: {
2073
- jsonMode: false;
2074
- };
2075
- };
2076
- /**
2077
- * Parameters for a cached, streaming LLM call with `jsonMode: false`.
2078
- * Selects the `cachedCall()` overload whose `finalResult` (on a miss) or
2079
- * cached value (on a hit) is a plain `string`.
2080
- */
2081
- type CachedStreamJsonModeDisabledCallParams = CachedCallInput & {
2082
- call: Omit<LLMRequestShape<unknown>, 'jsonSchema'> & {
2083
- stream: true;
2084
- jsonMode: false;
2085
- jsonSchema?: never;
2086
- };
2087
- };
2088
- /**
2089
- * Parameters for a cached, streaming LLM call with `jsonMode: true` and no
2090
- * `schema`. Selects the `cachedCall()` overload whose `finalResult` (on a
2091
- * miss) or cached value (on a hit) is a `JsonValue`.
2092
- */
2093
- type CachedStreamJsonModeEnabledCallParams = CachedCallInput & {
2094
- call: Omit<LLMRequestShape<JsonValue>, 'schema'> & {
2095
- stream: true;
2096
- jsonMode: true;
2097
- schema?: never;
2098
- };
2099
- };
2100
- //#endregion
2101
- //#region src/types/client.d.ts
2102
- /** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
2103
- interface WireToolCall {
2104
- id: string;
2105
- type: 'function';
2106
- function: {
2107
- name: string;
2108
- /** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
2109
- arguments: string;
2110
- };
77
+ resolveKey(key: string): Promise<string>;
78
+ get(key: string): Promise<{
79
+ hit: boolean;
80
+ value: T | null;
81
+ }>;
82
+ set(key: string, value: T, ttl: number): Promise<void>;
83
+ delete(key: string): Promise<void>;
84
+ /** L1 ttl for a promoted entry, capped at the seconds L2 has left when that is known. */
85
+ private promotionTtl;
86
+ private trackExpiry;
87
+ /** Drops every tracked record that has expired, visiting only those. */
88
+ private pruneExpired;
2111
89
  }
2112
- /** One entry of `LLMClient`'s `messages` array, named so callers building it can annotate against it. */
2113
- type WireMessage = {
2114
- role: 'system';
2115
- content: string;
2116
- } | {
2117
- role: 'user';
2118
- content: string | ContentBlock[];
2119
- } | {
2120
- role: 'assistant';
2121
- /** Optional: an assistant turn that only requested tools has no text. */
2122
- content?: string;
2123
- tool_calls?: WireToolCall[];
2124
- } | {
2125
- role: 'tool';
2126
- tool_call_id: string;
2127
- content: string;
2128
- /** Honored by `fromAnthropic` (maps to `tool_result.is_error`) and `fromBedrock` (maps to `toolResult.status`); other adapters ignore it. */
2129
- is_error?: boolean;
2130
- };
2131
- /** The OpenAI-shaped wire `tool_choice`. */
2132
- type WireToolChoice = 'auto' | 'none' | 'required' | {
2133
- type: 'function';
2134
- function: {
2135
- name: string;
2136
- };
2137
- };
90
+ //#endregion
91
+ //#region src/internal/utils/rate-limit/tokenEstimate.utils.d.ts
2138
92
  /**
2139
- * Minimal shape similar to the OpenAI SDK's chat.completions.create API,
2140
- * `response_format.json_schema` and `reasoning_effort` are optional on the wire
2141
- * providers that don't support them will just ignore fields they don't recognize,
2142
- * but not every SDKs TS types accept them, hence this being a structural type
2143
- * rather than importing the SDKs own params type
93
+ * Default `estimateTokens`: chars/4 over every message's text, plus
94
+ * `IMAGE_TOKEN_ESTIMATE` per image, plus the requested `max_tokens`.
2144
95
  */
2145
- interface LLMClient {
2146
- /**
2147
- * Whether this client supports OpenAI's `response_format: { type:
2148
- * 'json_object' }` as a real, API-level constraint. Defaults to `true`
2149
- * when omitted (every OpenAI-compatible client and `fromGemini` map it to
2150
- * a real field). `fromAnthropic` and `fromBedrock` set this to `false`:
2151
- * neither provider has a field that mechanically guarantees JSON output
2152
- * for this mode, so `RequestBuilder` downgrades a *default* (unset)
2153
- * `jsonMode` to plain text for these clients instead of requesting
2154
- * `json_object` and getting an unenforced, provider-side no-op back. An
2155
- * *explicit* `jsonMode: true` still throws for such clients, since that's
2156
- * a caller deliberately asking for a guarantee the client can't provide.
2157
- */
2158
- supportsJsonObjectMode?: boolean;
2159
- chat: {
2160
- completions: {
2161
- create(params: {
2162
- model: string;
2163
- temperature?: number;
2164
- max_tokens: number;
2165
- response_format?: {
2166
- type: 'json_object';
2167
- } | {
2168
- type: 'json_schema';
2169
- json_schema: {
2170
- name: string;
2171
- schema: Record<string, unknown>;
2172
- strict?: boolean;
2173
- description?: string;
2174
- };
2175
- };
2176
- /** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
2177
- reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
2178
- /**
2179
- * Numeric reasoning token budget, for providers with a native
2180
- * budget field (Anthropic, Gemini). Ignored by clients that only
2181
- * understand `reasoning_effort` tiers, use that field instead for
2182
- * those.
2183
- */
2184
- budget_tokens?: number;
2185
- /** Tools the model may call, OpenAI's `function`-wrapped shape. */
2186
- tools?: Array<{
2187
- type: 'function';
2188
- function: {
2189
- name: string;
2190
- description: string;
2191
- parameters: Record<string, unknown>;
2192
- };
2193
- }>;
2194
- tool_choice?: WireToolChoice;
2195
- /**
2196
- * Wire-format messages. Breaking change for custom adapters:
2197
- * implementations must handle tool messages and assistant tool_calls.
2198
- * Exhaustive switches over only system/user/assistant roles may no longer compile.
2199
- */
2200
- messages: WireMessage[];
2201
- }, options: {
2202
- signal: AbortSignal;
2203
- }): Promise<{
2204
- choices?: Array<{
2205
- message?: {
2206
- content?: string | null;
2207
- tool_calls?: WireToolCall[];
2208
- };
2209
- }>;
2210
- usage?: {
2211
- prompt_tokens?: number;
2212
- completion_tokens?: number;
2213
- total_tokens?: number;
2214
- completion_tokens_details?: {
2215
- reasoning_tokens?: number;
2216
- };
2217
- };
2218
- }>;
2219
- /**
2220
- * Optional. Required only for `stream: true` calls. Adapters/clients
2221
- * that don't implement this make `stream: true` throw a clear
2222
- * `LLMError('validation')` rather than a confusing runtime failure.
2223
- * Takes the same request shape as `create`, minus the response type.
2224
- */
2225
- createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
2226
- signal: AbortSignal;
2227
- }): AsyncIterable<WireStreamChunk>;
2228
- };
2229
- };
2230
- }
96
+ export declare function defaultEstimateTokens(request: WireRequest): number;
2231
97
  //#endregion
2232
98
  //#region src/internal/utils/cache/cacheAdapter.utils.d.ts
2233
99
  /**
@@ -2241,16 +107,6 @@ type CacheOption = {
2241
107
  eviction?: EvictionOption;
2242
108
  } | CacheAdapter;
2243
109
  //#endregion
2244
- //#region src/internal/utils/circuit-breaker/circuitBreakerAdapter.utils.d.ts
2245
- /**
2246
- * Not re-exported from the package root, imported directly from this
2247
- * internal module by `VernLLMOptions.circuitBreaker`'s own type (see
2248
- * options.ts) so that union isn't duplicated between the public option
2249
- * field and `buildCircuitBreaker`'s own signature below, same pattern
2250
- * `CacheOption` and `RateLimitOption` already use for their own options.
2251
- */
2252
- type CircuitBreakerOption = boolean | CircuitBreakerOptions | CircuitBreakerAdapter;
2253
- //#endregion
2254
110
  //#region src/types/options.d.ts
2255
111
  interface VernLLMOptions {
2256
112
  client: LLMClient;
@@ -2265,17 +121,24 @@ interface VernLLMOptions {
2265
121
  /** Per-attempt timeout in ms. Default 25000 */
2266
122
  timeoutMs?: number;
2267
123
  /**
2268
- * For `stream: true` calls: max gap allowed between chunks once the
2269
- * stream has opened, in ms. Resets on every chunk, including keep-alive
2270
- * pings. `timeoutMs` only covers opening the stream and its first
2271
- * chunk; this covers every gap after that. Also counts as a
2272
- * circuit-breaker failure, unlike other mid-stream errors, since a
2273
- * provider that streams one chunk then stalls should still trip it.
2274
- * Default 30000. Pass 0 or negative to disable.
124
+ * Max gap between stream chunks once open, in ms. Resets on every chunk, pings included. Unlike
125
+ * other mid-stream errors it counts toward the breaker, so a provider that stalls after one chunk
126
+ * still trips it. Default 30000; 0 or negative disables.
2275
127
  */
2276
128
  chunkIdleTimeoutMs?: number;
129
+ /**
130
+ * How long a `chunks` reader may stop pulling on a full buffer before it is detached. Its next
131
+ * pull rejects with code `reader_stall_timeout`, while the stream finishes so `finalResult`
132
+ * settles and its slot is freed. Off by default.
133
+ */
134
+ readerStallTimeoutMs?: number;
2277
135
  /** Base delay for exponential backoff in ms. Default 500 */
2278
136
  baseDelayMs?: number;
137
+ /**
138
+ * Longest `Retry-After` wait honored, in ms. Also caps `LLMError.retryAfterMs`. `0` retries at
139
+ * once; `Infinity` removes the cap. Default 10000. Negative or NaN throws.
140
+ */
141
+ maxRetryAfterMs?: number;
2279
142
  /** Default max_tokens for calls that don't override it. Default 1000 */
2280
143
  defaultMaxTokens?: number;
2281
144
  /**
@@ -2284,76 +147,42 @@ interface VernLLMOptions {
2284
147
  * request entirely, so the provider applies its own default instead.
2285
148
  */
2286
149
  defaultTemperature?: number | null;
2287
- /**
2288
- * Default reasoning effort for calls that don't override it. Not sent
2289
- * when omitted, same as leaving `reasoningEffort` unset on a call. See
2290
- * `budgetTokens`/`reasoningEffort` on `CallParams` for how the two
2291
- * relate and how each adapter converts between them.
2292
- */
150
+ /** Default reasoning effort for calls that don't set one. Not sent when omitted. */
2293
151
  defaultReasoningEffort?: 'minimal' | 'low' | 'medium' | 'high';
2294
- /**
2295
- * Default reasoning token budget for calls that don't override it. Not
2296
- * sent when omitted. If both this and `defaultReasoningEffort` are set,
2297
- * each adapter still prefers whichever field it natively understands,
2298
- * same as at the per-call level.
2299
- */
152
+ /** Default reasoning token budget for calls that don't set one. Not sent when omitted. */
2300
153
  defaultBudgetTokens?: number;
2301
154
  /**
2302
- * Enables debug logging of raw model output (logs up to 800 chars of each
2303
- * response) and provider errors. Off by default. Only controls the
2304
- * default `ConsoleLogger`: when a custom `logger` is supplied instead,
2305
- * that logger's own `debug()` implementation decides whether messages
2306
- * are emitted, and this option has no effect on it.
155
+ * Logs raw model output (up to 800 chars) and provider errors. Only affects the default
156
+ * `ConsoleLogger`; a custom `logger` decides for itself.
2307
157
  */
2308
158
  debug?: boolean;
2309
159
  /**
2310
- * Applied before every internal `logger.debug()` call: the raw output
2311
- * logged on success, and the provider error logged on a failed call or
2312
- * a failed stream open. This is the one piece of logging an app can't
2313
- * intercept itself, since it's a direct call into `logger.debug`
2314
- * rather than something routed through `onEvent`/`onUsage`; anything
2315
- * caught elsewhere (events, `LLMError.cause`) already passes through
2316
- * the app's own callback and can be redacted there instead. Runs
2317
- * before `logger.debug()` regardless of whether that call ends up
2318
- * emitting anything, so with a custom `logger`, `redact` still applies
2319
- * even without `debug: true`; see `debug` for why. Default: identity
2320
- * (no redaction).
160
+ * Applied to model output and provider errors before VernLLM's own `logger.debug()` calls, the
161
+ * one log path an app can't intercept. Runs even without `debug: true`, since a custom logger may
162
+ * emit debug anyway. Default: no redaction.
2321
163
  */
2322
164
  redact?: (text: string) => string;
2323
165
  /**
2324
- * Cache for cachedCall. `{ maxSize, eviction }` configures the
2325
- * built-in in-memory adapter (`eviction` default `'fifo'`). Pass a
2326
- * `CacheAdapter` directly for a real backend. Default: in-memory,
2327
- * maxSize 1000, fifo.
166
+ * Cache for `cachedCall`. `{ maxSize, eviction }` configures the built in adapter; pass a
167
+ * `CacheAdapter` for a real backend. Default in memory, 1000 entries, fifo.
2328
168
  */
2329
169
  cache?: CacheOption;
2330
170
  /**
2331
- * Reclassifies an otherwise-successful result as a failure, e.g. a
2332
- * response that parsed fine but came back empty or truncated. Runs
2333
- * once per attempt, right after a response is validated. Returning
2334
- * `undefined` leaves the result untouched; returning an
2335
- * `LLMErrorCode` fails that attempt with it, feeding the same retry
2336
- * and circuit-breaker paths a thrown error would. A throwing hook is
2337
- * caught, logged, and treated as no soft failure, so a broken hook
2338
- * degrades safely instead of failing every call.
171
+ * Reclassifies an otherwise successful result as a failure. Runs once per attempt after
172
+ * validation; returning an `LLMErrorCode` fails the attempt through the normal retry and breaker
173
+ * paths. A throwing hook is logged and ignored.
2339
174
  */
2340
175
  detectSoftFailure?: DetectSoftFailure;
2341
- /** HTTP status codes that should fail fast without retrying. Default [400, 401, 403, 404, 422] */
176
+ /** HTTP status codes that should fail fast without retrying. Default [400, 401, 402, 403, 404, 413, 422] */
2342
177
  nonRetryableStatus?: number[];
2343
178
  /** Custom JSON parser. Must return undefined/null on failure. Default: JSON.parse wrapped in try/catch */
2344
179
  parseJson?: (content: string) => unknown;
2345
180
  /** Called after every successful call with token usage, if the provider reports it */
2346
181
  onUsage?: OnUsage;
2347
182
  /**
2348
- * Called when a provider response arrives but VernLLM's own post-processing
2349
- * then fails, after usage data was already present in that response.
2350
- * Separate from `onUsage`, which only fires on full success.
2351
- *
2352
- * For non-streaming calls, never fires for transport failures (timeout,
2353
- * network error, non-retryable status), since no response means no usage
2354
- * to report. For streaming calls, this is not guaranteed: a stream can
2355
- * deliver a usage chunk and then fail later (e.g. an idle timeout waiting
2356
- * for the final close), in which case this does fire.
183
+ * Fires when a response carried usage but post-processing then failed. `onUsage` only fires on
184
+ * success. A non-streaming transport failure has no usage to report; a stream can deliver usage
185
+ * and fail later, which does fire this.
2357
186
  */
2358
187
  onUsageFailure?: OnUsageFailure;
2359
188
  /**
@@ -2362,11 +191,8 @@ interface VernLLMOptions {
2362
191
  */
2363
192
  logger?: Logger | 'silent';
2364
193
  /**
2365
- * Enables a circuit breaker that short-circuits calls after repeated
2366
- * consecutive failures, instead of continuing to hammer a down provider
2367
- * Pass `true` for defaults, or an options object to tune threshold/cooldown.
2368
- * Pass a `CircuitBreakerAdapter` instead for cross-process coordination,
2369
- * the same pattern `cache` and `rateLimit` already support.
194
+ * Short-circuits calls after repeated failures. `true` for defaults, options to tune, or a
195
+ * `CircuitBreakerAdapter` for cross-process state.
2370
196
  */
2371
197
  circuitBreaker?: CircuitBreakerOption;
2372
198
  /**
@@ -2376,175 +202,95 @@ interface VernLLMOptions {
2376
202
  */
2377
203
  onEvent?: OnEvent;
2378
204
  /**
2379
- * Client-side rate limiting. Queues calls locally to stay under the
2380
- * configured requests/tokens-per-minute or concurrency caps, instead of
2381
- * letting the provider reject them. Independent of the `Retry-After`
2382
- * handling already applied to a provider 429: this avoids tripping the
2383
- * limit in the first place. Omit for unlimited (the default).
2384
- *
2385
- * A plain config object builds an in-process limiter. Pass a
2386
- * `RateLimiterAdapter` instead for cross-process coordination.
205
+ * Client side rate limiting: queues calls to stay under request, token or concurrency caps rather
206
+ * than letting the provider reject them. A config object builds an in-process limiter; pass a
207
+ * `RateLimiterAdapter` for cross-process. Omit for unlimited.
2387
208
  */
2388
209
  rateLimit?: RateLimitOption;
2389
210
  /**
2390
- * Caps how much of this target's recent traffic is allowed to be
2391
- * retries, independent of `circuitBreaker`. Once at least `minCalls`
2392
- * calls have landed in the trailing `windowMs` and the retry ratio
2393
- * among them reaches `retryRatio`, further retries against this target
2394
- * throw `LLMError('retry_budget_exhausted')` instead of retrying,
2395
- * protecting the target's real capacity even while its breaker is
2396
- * still closed. Omit for no budget (the default). Never inherited by
2397
- * `fallback` targets, same as `circuitBreaker`/`rateLimit`.
211
+ * Caps the share of this target's recent traffic that may be retries. Once `minCalls` calls land
212
+ * in `windowMs` and the retry ratio reaches `retryRatio`, retries throw `retry_budget_exhausted`,
213
+ * even while the breaker is closed. Not inherited by fallback targets.
2398
214
  */
2399
215
  retryBudget?: RetryBudgetOptions;
2400
216
  /**
2401
- * Ordered targets tried after the primary, in order, once it (and its
2402
- * own retries) is exhausted or abandoned. Order is the policy: VernLLM
2403
- * never reorders, scores, or selects between targets. Each target keeps
2404
- * its own retry state, circuit breaker, and rate limiter, independent
2405
- * of every other target's. A single `FallbackTarget` is equivalent to
2406
- * `[target]`.
217
+ * Targets tried in order after the primary is exhausted. VernLLM never reorders them. Each keeps
218
+ * its own retries, breaker and limiter. A single target equals `[target]`.
2407
219
  */
2408
220
  fallback?: FallbackTarget | FallbackTarget[];
2409
221
  /**
2410
- * Decides what happens after a target fails: `'next'` to move on to
2411
- * the following target (or throw, if it was the last one), `'stop'` to
2412
- * give up immediately without trying any remaining targets. Called
2413
- * once per failed target, after that target's own retries are
2414
- * exhausted or abandoned early, so `'retry'` is never a valid return
2415
- * here. Defaults to `defaultFallbackOn`, which stops on
2416
- * parse/validation/aborted/quota errors and on tool-contract failures
2417
- * (the model ignoring the request, not the provider being unhealthy),
2418
- * and moves on for everything else.
222
+ * Whether a failed target moves on (`'next'`) or ends the call (`'stop'`). Called once per failed
223
+ * target after its own retries. Defaults to `defaultFallbackOn`; see its docs for what it stops
224
+ * on.
2419
225
  */
2420
226
  fallbackOn?: FallbackOn;
2421
227
  /**
2422
- * Transforms outgoing requests and/or wraps whole logical calls,
2423
- * without touching retry, circuit breaker, or fallback internals.
2424
- * Defaults to an empty array. See `VernLLMMiddleware` for the four
2425
- * available hooks (`transform`, `wrap`, `onEvent`, `enabled`).
228
+ * Request transforms and call wrappers that leave retry, breaker and fallback internals alone.
229
+ * See `VernLLMMiddleware` for the hooks.
2426
230
  */
2427
231
  middleware?: VernLLMMiddleware[];
2428
232
  /**
2429
- * Bounds `transform` and a function `enabled`, the same way every
2430
- * other blocking operation in the package is already bounded.
2431
- * Overridable per middleware via that entry's own `timeoutMs`.
2432
- * `<= 0` means unbounded (no timer at all). Default 5000.
233
+ * Bounds `transform` and a function `enabled`. Overridable per middleware with `timeoutMs`. `<=
234
+ * 0` means unbounded. Default 5000.
2433
235
  */
2434
236
  middlewareTimeoutMs?: number;
2435
237
  }
2436
238
  //#endregion
2437
239
  //#region src/types/createMiddleware.d.ts
2438
240
  /**
2439
- * `VernLLMMiddleware` plus `onError`, a convenience for the common "I
2440
- * only care about failures" case. Everything else is passed through to
2441
- * the resulting `VernLLMMiddleware` unchanged; setting `wrap` directly
2442
- * alongside `onError` is an error, since `onError` builds its own `wrap`
2443
- * under the hood, and building it around a `wrap` you also supplied
2444
- * would silently drop one of the two.
241
+ * `VernLLMMiddleware` plus `onError`, for middleware that only cares about failures. `wrap` can't
242
+ * be set alongside it, since `onError` builds its own.
2445
243
  */
2446
244
  type CreateMiddlewareOptions = Omit<VernLLMMiddleware, 'wrap'> & {
2447
245
  wrap?: undefined;
2448
246
  /**
2449
- * Called with this call's terminal error, if it fails: the same error
2450
- * `wrap`'s own `next()` would reject with. Never called on success,
2451
- * and never called for a failure some *other* middleware's `wrap`
2452
- * already swallowed by short-circuiting with its own `CallResult`.
2453
- * The original error is always rethrown afterward, `onError` only
2454
- * observes it, exactly like `onUsage`/`onEvent` elsewhere: a throwing
2455
- * `onError` is discarded (not logged, this helper has no `Logger` of
2456
- * its own to log through) and otherwise has no effect on the call.
2457
- * `ctx` is `wrap`'s own pre-dispatch context (`onError` builds a `wrap`
2458
- * under the hood), so it only describes the primary target.
247
+ * Called with the call's terminal error. Not called on success or when another `wrap` swallowed
248
+ * the failure. Only observes: the error is always rethrown and a throwing `onError` is discarded.
249
+ * `ctx` is `wrap`'s pre-dispatch context.
2459
250
  */
2460
251
  onError?: (error: LLMError, ctx: PreDispatchContext) => void | Promise<void>;
2461
252
  };
2462
253
  /**
2463
- * Builds a `VernLLMMiddleware` entry. Plain pass-through when `onError`
2464
- * is omitted; when it's set, wraps it in a `wrap` that calls `next()`,
2465
- * reports `onError` on a rejection, and always rethrows the original
2466
- * error afterward, so `onError` never changes what the call itself
2467
- * returns or throws, only what gets observed about it.
254
+ * Builds a middleware entry. With `onError`, adds a `wrap` that reports rejections and always
255
+ * rethrows, so the call's outcome never changes.
2468
256
  */
2469
257
  export declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMiddleware;
2470
258
  //#endregion
2471
259
  //#region src/vernLLM.d.ts
2472
260
  /**
2473
- * A LLM call framework for resilience, observability and control. This is VernLLM!
2474
- *
2475
- * Adds retry with backoff and jitter, per-attempt timeouts, an optional
2476
- * circuit breaker, JSON parsing with optional schema validation, usage
2477
- * tracking, and an optional response cache. All configurable, all opt-in
2478
- * beyond sensible defaults.
261
+ * The LLM call framework: retries, timeouts, circuit breaking, fallback, rate limiting, caching and
262
+ * middleware around any provider adapter.
2479
263
  */
2480
264
  export declare class VernLLM {
2481
265
  private readonly logger;
2482
- /**
2483
- * One `CallExecutor` per provider target: index 0 is the primary,
2484
- * everything after it is a `fallback` target, in the order declared.
2485
- * Walked by `runFallbackChain`, moving to the next entry only when
2486
- * `fallbackOn` says to.
2487
- */
266
+ /** One per target: index 0 is the primary, then each fallback in declared order. */
2488
267
  private readonly executors;
2489
- /** Decides whether a failed target is followed by the next one or the chain stops. See `VernLLMOptions['fallbackOn']`. */
2490
- private readonly fallbackOn;
2491
- /** Reports a `'fallback'` event when the chain moves to the next target. Shared `onEvent` plumbing, same as every executor's. */
2492
- private readonly reportEvent;
2493
- /** Owns cache reads/writes and in-flight coalescing for `cachedCall()`. Only calls back into `this.call()` as an opaque function. */
268
+ /** Owns cache reads, writes and in-flight coalescing for `cachedCall()`. */
2494
269
  private readonly cacheOrchestrator;
2495
- /**
2496
- * See `VernLLMOptions.middleware`. Every resolved view of composition
2497
- * order, built once here at construction time by
2498
- * `buildMiddlewarePipeline`. Nothing downstream computes order
2499
- * itself; each consumer reads `transformOrder`, `wrapOrder`, or
2500
- * `names`, whichever it actually needs.
2501
- */
2502
- private readonly pipeline;
2503
- /** See `VernLLMOptions.middlewareTimeoutMs`. Bounds `transform` and a function `enabled`; `wrap` itself is never bounded by this. */
2504
- private readonly middlewareTimeoutMs;
2505
- /**
2506
- * Maps `cachedCall()`'s inner `this.call(...)` params to its own
2507
- * `middlewareState`, so that call's own `runOperation` skips wrapping
2508
- * again and reuses the same state bag `wrap` just ran with (so a
2509
- * value `wrap` sets is visible to `transform`, same as a direct
2510
- * call). Keyed by object identity, not `requestId`, since two
2511
- * concurrent `cachedCall()`s can share an explicit `requestId`.
270
+ /** Every target in declared order, the pool `targets` selects from. */
271
+ private readonly declaredTargets;
272
+ /** Everything but `targets`, which each call sets for its own order. */
273
+ private readonly logicalCallDependencies;
274
+ private readonly runOperationDependencies;
275
+ /** What each call's scope needs to deliver a `ctx.emit` event. */
276
+ private readonly callScopeDependencies;
277
+ /**
278
+ * Marks `cachedCall()`'s inner call params with its state bag, so the inner
279
+ * `runOperation` skips `wrap` and reuses that bag. Keyed by object identity,
280
+ * since concurrent `cachedCall()`s can share an explicit `requestId`.
2512
281
  */
2513
282
  private readonly cachedCallInnerParams;
2514
- /**
2515
- * Shares one `CallMeta` holder across every `cachedCall()` in flight
2516
- * for the same resolved cache key, so a joining invocation (never
2517
- * calls `call()` itself) reports the trigger's real metadata instead
2518
- * of `undefined`. A true cache hit never creates an entry, so it
2519
- * still reports no metadata correctly.
2520
- */
283
+ /** Shared `CallMeta` holders per resolved cache key. See `claimMetaHolder`. */
2521
284
  private readonly cachedCallMeta;
2522
- /**
2523
- * @param options Client, model, and tunables. Defaults: `maxRetries` 1,
2524
- * `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
2525
- * `defaultTemperature` 0.2, `cache` an in-memory adapter,
2526
- * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
2527
- */
285
+ /** @param options Client, model and tunables. See `VernLLMOptions` for each default. */
2528
286
  constructor(options: VernLLMOptions);
2529
- /** Logs a failed refundUsage attempt via the configured logger. */
2530
- private logRefundError;
2531
- /** Everything `executeLogicalCall`/`executeLogicalStreamCall` (in `logicalCall.ts`) need from this instance, gathered once so `call()` doesn't rebuild it per invocation. */
2532
- private get logicalCallDependencies();
2533
- /** Everything `runOperation` (in `runOperation.ts`) needs from this instance, gathered once so `call()`/`cachedCall()` don't rebuild it per invocation. */
2534
- private get runOperationDependencies();
2535
287
  /**
2536
- * Makes a single logical LLM call, retrying on failure per the configured
2537
- * policy. Fails fast if the breaker is open or the signal is already
2538
- * aborted. Rejects with a normalized `LLMError` on exhausted retries.
2539
- *
2540
- * Supports `tools`, `stream`, and JSON mode/schema, in any combination.
2541
- * See the Tool Calling and Streaming docs for return-shape details and
2542
- * the TypeScript overloads that select between them.
288
+ * Makes one logical call, with retries, fallback and the breaker applied. Rejects with a
289
+ * normalized `LLMError`. See the Tool Calling and Streaming docs for the return shapes.
2543
290
  *
2544
- * @param params System/user content plus per-call overrides. See `CallParams`.
2545
- * @returns The parsed response (or raw string if `jsonMode` is false), a
2546
- * `CallWithToolsResult<T>` when `tools` is set, or a `{ chunks,
2547
- * finalResult }` `StreamCallResult` when `stream: true`. See `StreamCallResult`.
291
+ * @param params Content plus per call overrides. See `CallParams`.
292
+ * @returns The parsed response, a `CallWithToolsResult<T>` with `tools`, or a `StreamCallResult`
293
+ * with `stream: true`.
2548
294
  */
2549
295
  call<T = unknown, const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: StreamEnabledCallParams<T, Tools> & ToolsDisabledCallParams<T, Tools>): Promise<StreamCallResult<ContentResult<T>>>;
2550
296
  call<T = unknown, const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: StreamEnabledCallParams<T, Tools> & ToolEnabledCallParams<T, Tools>): Promise<StreamCallResult<CallWithToolsResult<T, Tools>>>;
@@ -2561,37 +307,26 @@ export declare class VernLLM {
2561
307
  call(params: JsonModeEnabledCallParams): Promise<JsonValue>;
2562
308
  call<T = unknown>(params: CallParams<T>): Promise<T>;
2563
309
  /**
2564
- * Thin delegator kept private on `VernLLM` (rather than only existing on
2565
- * `CacheOrchestrator`) since it's the one caching primitive exercised
2566
- * directly by white-box tests, independent of the public `cachedCall()`
2567
- * surface.
310
+ * The order a call starts with: `requested` by name, or every target as declared. Throws before
311
+ * any timer or hook exists, so an invalid order never starts a call.
2568
312
  */
313
+ private resolveTargets;
314
+ /** Kept on `VernLLM` since tests drive the caching core directly through it. */
2569
315
  private runCached;
2570
316
  /**
2571
- * Removes a cached response by key when the configured cache adapter
2572
- * supports deletion. Cache invalidation is the caller's responsibility;
2573
- * only the application knows when cached data is stale.
317
+ * Removes a cached response when the adapter supports deletion. Invalidation is up to the app.
2574
318
  *
2575
- * @param key The raw cache key (resolved through the adapter's
2576
- * `resolveKey`, if any, before deletion).
319
+ * @param key The raw cache key, resolved through the adapter's `resolveKey` first.
2577
320
  */
2578
321
  deleteCache(key: string): Promise<void>;
2579
322
  /**
2580
- * Cache wrapper composing `call` + caching, so cached LLM calls
2581
- * automatically get retry/timeout/circuit-breaker behavior. Concurrent
2582
- * misses for the same `cacheKey` share a single in-flight call, avoiding
2583
- * cache stampedes. Supports `stream: true` and `tools` in any combination.
323
+ * `call()` with caching. Concurrent misses for one `cacheKey` share a single in-flight call.
324
+ * Works with `stream` and `tools`; with tools the whole result is cached, tool call decisions
325
+ * included. A caller's own abort or deadline only ends its wait; the shared request is aborted
326
+ * once every caller has left.
2584
327
  *
2585
- * When `call.tools` is set, this caches the whole result including any
2586
- * `tool_calls` decision, not just final answers; use a short `ttl` or a
2587
- * separate `cacheKey` if a tool's result shouldn't be reused across calls.
2588
- *
2589
- * @param params `cacheKey`, `ttl`, optional
2590
- * `reserveUsage`/`refundUsage`/`signal`, plus `call`, the `CallParams`
2591
- * to pass through to `this.call(...)`. The top-level `signal` governs
2592
- * the cached operation and its usage hooks only; to also abort the
2593
- * underlying provider request, set `signal` inside `call`.
2594
- * @returns The cached value on a hit, or the freshly-called result on a miss.
328
+ * @param params Cache settings plus `call`, the `CallParams` for the underlying call.
329
+ * @returns The cached value on a hit, or the fresh result on a miss.
2595
330
  */
2596
331
  cachedCall<T, const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: CachedStreamToolCallParams<T, Tools>): Promise<StreamCallResult<CallWithToolsResult<T, Tools>>>;
2597
332
  cachedCall<const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: CachedStreamConditionalStringToolCallParams<Tools>): Promise<StreamCallResult<string | CallWithToolsResult<string, Tools>>>;
@@ -2609,9 +344,7 @@ export declare class VernLLM {
2609
344
  * @param target.index Which target to read. Defaults to the primary.
2610
345
  * @param target.model Which model bucket to read, if the target isolates by model.
2611
346
  * @returns The breaker state, or `undefined` if that target has no breaker.
2612
- * @throws {RangeError} If `target.index` names no target. Lets a real
2613
- * target with no breaker (`undefined`) stay distinguishable from a
2614
- * target that doesn't exist.
347
+ * @throws {RangeError} If `target.index` names no target.
2615
348
  */
2616
349
  getCircuitState(target?: CircuitTarget): CircuitState | undefined;
2617
350
  /**
@@ -2624,10 +357,8 @@ export declare class VernLLM {
2624
357
  getFailureBreakdown(target?: CircuitTarget): Partial<Record<LLMErrorCode | 'unknown', number>> | undefined;
2625
358
  /**
2626
359
  * @param target.index Which target to read. Defaults to the primary.
2627
- * @returns This target's current retry traffic/ratio in the trailing
2628
- * window, or `undefined` if that target has no retry budget
2629
- * configured. A budget is target-scoped, not model-scoped, so unlike
2630
- * `getFailureBreakdown` there's no `target.model` to pass.
360
+ * @returns The retry traffic and ratio in the trailing window, or `undefined` without a budget.
361
+ * Budgets are per target, not per model.
2631
362
  * @throws {RangeError} If `target.index` names no target.
2632
363
  */
2633
364
  getRetryBudgetState(target?: Pick<CircuitTarget, 'index'>): {
@@ -2636,10 +367,8 @@ export declare class VernLLM {
2636
367
  } | undefined;
2637
368
  /**
2638
369
  * @param target.index Which target to read. Defaults to the primary.
2639
- * @returns This target's current rate limit levels, or `undefined` if
2640
- * that target has no limiter configured. A limiter is target-scoped,
2641
- * not model-scoped, so unlike `getFailureBreakdown` there's no
2642
- * `target.model` to pass.
370
+ * @returns Current rate limit levels, or `undefined` without a limiter. Limiters are per target,
371
+ * not per model.
2643
372
  * @throws {RangeError} If `target.index` names no target.
2644
373
  */
2645
374
  getRateLimitState(target?: Pick<CircuitTarget, 'index'>): RateLimitState | undefined;
@@ -2686,1233 +415,15 @@ export declare class VernLLM {
2686
415
  //#endregion
2687
416
  //#region src/paramsHelpers.d.ts
2688
417
  /**
2689
- * Identity function preserving `params`'s own precise type, unlike a `:
2690
- * CallParams<T>` annotation, which would widen `tools` away and break the
2691
- * `ConditionalToolCallParams<T>` overload for `tools: someCondition ?
2692
- * [tool] : undefined`. Use it when you need `call()` params in a named,
2693
- * reusable variable; skip it when you can pass the object inline.
2694
- *
2695
- * ```ts
2696
- * const params = defineCallParams({
2697
- * userContent: 'What is the weather?',
2698
- * tools: someCondition ? [weatherTool] : undefined,
2699
- * });
2700
- * const result = await llm.call(params);
2701
- * // result: unknown | CallWithToolsResult<unknown>, same as inline
2702
- * ```
2703
- *
2704
- * `T` isn't a parameter here; pin it via `llm.call<T>(params)` as usual.
2705
- * `defineCachedCallParams` is the `cachedCall()` counterpart.
418
+ * Keeps `params`' precise type for a reusable variable. A `CallParams<T>` annotation would widen
419
+ * `tools` and lose the conditional tools overload. Pin `T` through `llm.call<T>(params)`.
2706
420
  */
2707
421
  export declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
2708
422
  /**
2709
- * The `cachedCall()` counterpart to `defineCallParams`: preserves the
2710
- * whole `{ cacheKey, ttl, call }` object, `call.tools` included, in one
2711
- * named variable.
2712
- *
2713
- * ```ts
2714
- * const params = defineCachedCallParams({
2715
- * cacheKey: 'weather-ny',
2716
- * ttl: 60,
2717
- * call: { userContent: 'What is the weather?', tools: someCondition ? [weatherTool] : undefined },
2718
- * });
2719
- * const result = await llm.cachedCall(params);
2720
- * ```
2721
- */
2722
- export declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
2723
- //#endregion
2724
- //#region src/adapters/internal/sse.d.ts
2725
- /**
2726
- * Parses a Server-Sent-Events byte/text stream into the JSON payload of
2727
- * each `data:` frame, in arrival order. Generic over transport: works with
2728
- * anything that hands back progressively-arriving `Uint8Array` or `string`
2729
- * chunks via async iteration: native `fetch`'s `response.body` (wrapped
2730
- * to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
2731
- * Node `Readable` (already async-iterable, no wrapping needed), etc, so
2732
- * this framing layer doesn't care which transport produced the bytes.
2733
- *
2734
- * Follows the SSE spec's frame-delimiting rules closely enough for LLM
2735
- * streaming responses: frames are separated by a blank line, each frame
2736
- * may carry one or more `data:` lines (joined with `\n` per spec when
2737
- * there's more than one), `:`-prefixed lines are comments and ignored, and
2738
- * other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
2739
- * only needs the payload. A frame whose data is exactly `[DONE]` (the
2740
- * sentinel several providers, notably OpenAI, send to mark stream end)
2741
- * ends iteration without yielding it.
2742
- *
2743
- * Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
2744
- * to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
2745
- * alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
2746
- * stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
2747
- * lines.
2748
- *
2749
- * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
2750
- * how malformed JSON is handled elsewhere in VernLLM.
2751
- */
2752
- export declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
2753
- /**
2754
- * Sentinel yielded by `parseSseStream` for a comment-only frame (no
2755
- * `data:` payload), the mechanism providers use for SSE keep-alive
2756
- * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
2757
- * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
2758
- */
2759
- export declare const SSE_PING: unique symbol;
2760
- //#endregion
2761
- //#region src/adapters/internal/imageFormat.d.ts
2762
- /**
2763
- * MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
2764
- * the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
2765
- * Converse all natively support, so a `ContentBlock[]` that validates for
2766
- * one provider validates for all of them.
2767
- */
2768
- declare const SUPPORTED_IMAGE_MIME_TYPES: readonly ["image/png", "image/jpeg", "image/gif", "image/webp"];
2769
- type SupportedImageMimeType = (typeof SUPPORTED_IMAGE_MIME_TYPES)[number];
2770
- //#endregion
2771
- //#region src/adapters/internal/nativeStructuredOutput.d.ts
2772
- /**
2773
- * A static allow-list or predicate naming which models support native,
2774
- * schema-constrained output as its own request field. Anthropic's
2775
- * `output_config.format` and Bedrock's `outputConfig.textFormat` separate
2776
- * from `tools`/`tool_choice`, so it can be combined with real,
2777
- * caller-supplied `tools` in the same request.
2778
- *
2779
- * There is no built-in default list here. Which models support this is
2780
- * Anthropic's and Bedrock's call to make, not this package's, and it
2781
- * changes over time; hardcoding a guessed list would risk silently
2782
- * routing a request onto a field a given model doesn't actually support,
2783
- * trading a clear `LLMError('invalid_params')` with
2784
- * `code: 'unsupported_capability'` for a confusing error from the provider
2785
- * instead. So this is opt-in: pass the model IDs you've verified against the
2786
- * provider's own docs (or a predicate). Left unset, no model is treated as
2787
- * native-capable, `jsonSchema` keeps using the older forced-single-tool-call
2788
- * emulation, and combining it with `tools` throws the coded capability error,
2789
- * exactly this package's behavior before native support was added.
2790
- */
2791
- type ModelCapabilityOverride = string[] | ((model: string) => boolean);
2792
- //#endregion
2793
- //#region src/adapters/internal/reasoningBudget.utils.d.ts
2794
- /**
2795
- * Shared conversion between the two reasoning controls VernLLM exposes:
2796
- * `reasoningEffort` (a tier string, OpenAI's native shape) and
2797
- * `budgetTokens` (a raw integer, Anthropic's and Gemini's native shape).
2798
- *
2799
- * Every adapter prefers its own native field when the caller set it, and
2800
- * only calls into this table when the caller set the other one instead.
2801
- * The numbers here are a guess, not a provider guarantee, callers who
2802
- * need a precise budget on a specific model should set `budgetTokens`
2803
- * directly rather than relying on this table's `reasoningEffort` mapping.
2804
- *
2805
- * The table itself is overridable per adapter instance, via
2806
- * `reasoningEffortTokens` on each `from*` adapter's options (see
2807
- * `AnthropicAdapterOptions`, `GeminiAdapterOptions`,
2808
- * `OpenAICompatibleAdapterOptions`, `BedrockAdapterOptions`), for callers
2809
- * who want `reasoningEffort` tiers to map onto different token counts
2810
- * than the defaults below, e.g. a model whose useful reasoning range
2811
- * doesn't match these numbers.
2812
- */
2813
- type EffortTokenTable = Record<'minimal' | 'low' | 'medium' | 'high', number>;
2814
- //#endregion
2815
- //#region src/adapters/anthropic.d.ts
2816
- /** Anthropic's native per-block content shape for a message. */
2817
- type AnthropicContentBlock = {
2818
- type: 'text';
2819
- text: string;
2820
- } | {
2821
- type: 'image';
2822
- source: {
2823
- type: 'base64';
2824
- media_type: SupportedImageMimeType;
2825
- data: string;
2826
- };
2827
- } | {
2828
- type: 'tool_use';
2829
- id: string;
2830
- name: string;
2831
- input: unknown;
2832
- } | {
2833
- type: 'tool_result';
2834
- tool_use_id: string;
2835
- content: string;
2836
- is_error?: boolean;
2837
- };
2838
- /** Minimal structural type for the Anthropic SDK's `messages.create` */
2839
- interface AnthropicClient {
2840
- messages: {
2841
- create(params: {
2842
- model: string;
2843
- max_tokens: number;
2844
- temperature?: number;
2845
- system?: string;
2846
- messages: Array<{
2847
- role: 'user' | 'assistant';
2848
- content: string | AnthropicContentBlock[];
2849
- }>;
2850
- tools?: Array<{
2851
- name: string;
2852
- description?: string;
2853
- input_schema: {
2854
- type: 'object';
2855
- [key: string]: unknown;
2856
- };
2857
- strict?: boolean;
2858
- }>;
2859
- tool_choice?: {
2860
- type: 'auto';
2861
- } | {
2862
- type: 'any';
2863
- } | {
2864
- type: 'none';
2865
- } | {
2866
- type: 'tool';
2867
- name: string;
2868
- };
2869
- /**
2870
- * Native, schema-constrained output: a separate request field from
2871
- * `tools`/`tool_choice`, so it can be sent alongside real tool
2872
- * calls. Only built by this adapter for models covered by
2873
- * `nativeStructuredOutputModels` (opt-in, see
2874
- * `AnthropicAdapterOptions`); other models keep getting
2875
- * `jsonSchema` emulated as a forced single tool call, the
2876
- * pre-existing behavior.
2877
- *
2878
- * Matches the real Anthropic API's `output_config.format` shape
2879
- * exactly: just `type` and `schema`, no `name`/`description`/
2880
- * `strict`. Those three exist on VernLLM's own `jsonSchema` API
2881
- * (and are still forwarded on the legacy forced-tool-call path,
2882
- * where they're real `Tool` fields), but the native structured-
2883
- * output endpoint has no equivalent for any of them.
2884
- */
2885
- output_config?: {
2886
- format?: {
2887
- type: 'json_schema';
2888
- schema: Record<string, unknown>;
2889
- };
2890
- /**
2891
- * Effort control for adaptive thinking, on models where manual
2892
- * `budget_tokens` thinking is no longer accepted (see
2893
- * `supportsManualThinkingBudget` in
2894
- * `adapters/internal/reasoningBudget.utils.ts`). Sibling to
2895
- * `format`, either or both may be present independently.
2896
- */
2897
- effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
2898
- };
2899
- /**
2900
- * Native reasoning control. `{ type: 'enabled', budget_tokens }`
2901
- * is built directly from `CallParams.budgetTokens`, or converted
2902
- * from `reasoningEffort`, on models that still accept a manual
2903
- * token budget. `{ type: 'adaptive' }` is sent instead, paired
2904
- * with `output_config.effort`, on models that only support
2905
- * adaptive thinking. See
2906
- * `adapters/internal/reasoningBudget.utils.ts`.
2907
- */
2908
- thinking?: {
2909
- type: 'enabled';
2910
- budget_tokens: number;
2911
- } | {
2912
- type: 'adaptive';
2913
- };
2914
- }, options: {
2915
- signal: AbortSignal;
2916
- }): Promise<{
2917
- content: Array<{
2918
- type: string;
2919
- text?: string;
2920
- id?: string;
2921
- name?: string;
2922
- input?: unknown;
2923
- }>;
2924
- usage?: {
2925
- input_tokens?: number;
2926
- output_tokens?: number;
2927
- output_tokens_details?: {
2928
- thinking_tokens?: number;
2929
- } | null;
2930
- };
2931
- }>;
2932
- };
2933
- }
2934
- /** Optional configuration for `fromAnthropic`. */
2935
- interface AnthropicAdapterOptions {
2936
- /**
2937
- * Which models support native, schema-constrained output
2938
- * (`output_config.format`), independent of `tools`/`tool_choice`, so it
2939
- * can be combined with real `tools` in one request. Pass a static list
2940
- * of model IDs (verified against Anthropic's own docs) or a predicate.
2941
- *
2942
- * There is no built-in default here (see `supportsNativeStructuredOutput`
2943
- * for why). Left unset, every model uses the older forced-single-tool-
2944
- * call emulation, and `tools` + `jsonSchema` together is rejected,
2945
- * exactly this adapter's behavior before native support was added.
2946
- */
2947
- nativeStructuredOutputModels?: ModelCapabilityOverride;
2948
- /**
2949
- * Overrides the token count `reasoningEffort` tiers map onto when the
2950
- * caller sets `reasoningEffort` but not `budgetTokens` (Claude has no
2951
- * tier concept of its own, see `adapters/internal/reasoningBudget.utils.ts`).
2952
- * Only the tiers listed are changed; any omitted tier keeps the
2953
- * built-in default. Has no effect when `budgetTokens` is set directly.
2954
- */
2955
- reasoningEffortTokens?: Partial<EffortTokenTable>;
2956
- /**
2957
- * Marks additional models as adaptive-only, on top of this package's
2958
- * own built-in rule (Claude Opus 4.7 and later, every Claude 5 tier
2959
- * model, see `isAdaptiveOnlyModel` in
2960
- * `adapters/internal/reasoningBudget.utils.ts`). Additive, not a
2961
- * replacement: it can correct a false negative (a newer model this
2962
- * package doesn't know about yet), it can't un-mark a model the
2963
- * built-in rule already caught. Pass a static list of model IDs or a
2964
- * predicate.
2965
- */
2966
- adaptiveOnlyModels?: ModelCapabilityOverride;
2967
- /**
2968
- * Whether the client's `messages.create` supports `.withResponse()`
2969
- * (needed for AIMD's proactive path). Default `false`, since
2970
- * `AnthropicClient` is structural and a test fake or thin wrapper
2971
- * won't implement it.
2972
- */
2973
- supportsWithResponse?: boolean;
2974
- }
2975
- /**
2976
- * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
2977
- * interface VernLLM uses for OpenAI/Groq.
2978
- *
2979
- * `response_format: json_schema`, on a model covered by
2980
- * `options.nativeStructuredOutputModels`, is sent as `output_config.format`,
2981
- * its own request field, independent of `tools`/`tool_choice`, so it can be
2982
- * combined with real, caller-supplied `tools` in the same request. Only
2983
- * `type` and `schema` are sent on this path, the real Anthropic API's
2984
- * `output_config.format` has no `name`/`description`/`strict` fields.
2985
- *
2986
- * On any other model (the default, since `nativeStructuredOutputModels` is
2987
- * opt-in), `response_format: json_schema` is mapped to Anthropic's forced
2988
- * tool-use instead: a single tool is defined with `input_schema` set to
2989
- * the caller's schema, `description` forwarded when provided, and `strict`
2990
- * forwarded when set, and `tool_choice` forces the model to call it. This
2991
- * legacy path cannot be combined with real `tools` (both would need the
2992
- * same `tools`/`tool_choice` field), and a call that tries throws
2993
- * `LLMError('invalid_params')` with `code: 'unsupported_capability'` and
2994
- * `issues: { capability: 'tools_with_json_schema' }` before reaching the API. Provider-constrained
2995
- * schema matching applies only when `strict: true` is forwarded and
2996
- * supported.
2997
- *
2998
- * `response_format: json_object` throws `LLMError('validation')`. Anthropic
2999
- * has no API-level field that mechanically guarantees JSON output the way
3000
- * OpenAI's `json_object` mode does; the only way to emulate it was a
3001
- * system-prompt instruction with no actual enforcement behind it, a
3002
- * guarantee this adapter no longer pretends to make. Use `jsonSchema`
3003
- * instead, which maps to a real constraint either way (native
3004
- * `output_config.format` or a forced tool call).
3005
- */
3006
- export declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
3007
- //#endregion
3008
- //#region src/adapters/gemini.d.ts
3009
- /**
3010
- * Gemini's native per-part content shape for a `contents` entry.
3011
- * `functionCall.args` and `functionResponse.response` are typed as
3012
- * `Record<string, unknown>` (not `unknown`) to match the real SDK's
3013
- * `FunctionCall.args` / `FunctionResponse.response`, see the doc comment
3014
- * on {@link GeminiClient}.
3015
- */
3016
- type GeminiPart = {
3017
- text: string;
3018
- } | {
3019
- inlineData: {
3020
- mimeType: string;
3021
- data: string;
3022
- };
3023
- } | {
3024
- functionCall: {
3025
- id?: string;
3026
- name: string;
3027
- args: Record<string, unknown>;
3028
- };
3029
- } | {
3030
- functionResponse: {
3031
- id?: string;
3032
- name: string;
3033
- response: Record<string, unknown>;
3034
- };
3035
- };
3036
- /**
3037
- * Structural type matching the real `@google/genai` SDK, in either shape
3038
- * it's commonly held in: the callable model methods directly (`ai.models`),
3039
- * or the complete top-level client (`ai`, via the optional `models` field
3040
- * below). Both work with `fromGemini` directly, with no cast:
3041
- *
3042
- * ```ts
3043
- * import { GoogleGenAI } from '@google/genai';
3044
- * const ai = new GoogleGenAI({ apiKey: '...' });
3045
- * const llm = new VernLLM({ client: fromGemini(ai), model: 'gemini-2.5-flash' });
3046
- * ```
3047
- *
3048
- * `generateContent` is optional so a `{ models: ... }`-shaped value is
3049
- * still a structural `GeminiClient`; `fromGemini` resolves `models` at
3050
- * runtime and throws if nothing callable results.
3051
- *
3052
- * Every field is shaped to be structurally assignable from the real SDK's
3053
- * generated types without importing them, so provider SDKs stay optional:
3054
- * `model` is required (the real SDK requires it), `functionCall.args` /
3055
- * `functionResponse.response` are `Record<string, unknown>` (matching the
3056
- * real SDK, not `unknown`), `toolConfig...mode` is `any` (TypeScript never
3057
- * treats a string-literal union as assignable to the real SDK's string
3058
- * enum), and response-side `functionCall.name` is optional (matching the
3059
- * real SDK).
3060
- */
3061
- interface GeminiClient {
3062
- /** Present when this is the whole top-level SDK client, not `ai.models`. `fromGemini` unwraps it at runtime. */
3063
- models?: GeminiClient;
3064
- generateContent?(params: {
3065
- model: string;
3066
- contents: Array<{
3067
- role: 'user' | 'model';
3068
- parts: GeminiPart[];
3069
- }>;
3070
- config?: {
3071
- systemInstruction?: {
3072
- parts: Array<{
3073
- text: string;
3074
- }>;
3075
- };
3076
- temperature?: number;
3077
- maxOutputTokens?: number;
3078
- responseMimeType?: string;
3079
- responseSchema?: Record<string, unknown>;
3080
- tools?: Array<{
3081
- functionDeclarations: Array<{
3082
- name: string;
3083
- description?: string;
3084
- parameters: Record<string, unknown>;
3085
- }>;
3086
- }>;
3087
- toolConfig?: {
3088
- functionCallingConfig: {
3089
- mode: any;
3090
- allowedFunctionNames?: string[];
3091
- };
3092
- };
3093
- /**
3094
- * Native reasoning control. `thinkingBudget` is built from
3095
- * `CallParams.budgetTokens` directly when set (0 disables thinking,
3096
- * -1 requests automatic budgeting, both passed through unchanged),
3097
- * or converted from `reasoningEffort`, on Gemini 2.5 and earlier
3098
- * models. `thinkingLevel` is used instead on Gemini 3 and later,
3099
- * which use a level-based control rather than a numeric budget.
3100
- * `any`, same reason as `toolConfig...mode` above, see class doc
3101
- * comment. See `usesGeminiThinkingLevel` in
3102
- * `adapters/internal/reasoningBudget.utils.ts`.
3103
- */
3104
- thinkingConfig?: {
3105
- thinkingBudget?: number;
3106
- thinkingLevel?: any;
3107
- };
3108
- abortSignal?: AbortSignal;
3109
- };
3110
- }): Promise<{
3111
- candidates?: Array<{
3112
- content?: {
3113
- parts?: Array<{
3114
- text?: string;
3115
- functionCall?: {
3116
- id?: string;
3117
- name?: string;
3118
- args?: unknown;
3119
- };
3120
- }>;
3121
- };
3122
- }>;
3123
- usageMetadata?: {
3124
- promptTokenCount?: number;
3125
- candidatesTokenCount?: number;
3126
- totalTokenCount?: number;
3127
- thoughtsTokenCount?: number;
3128
- };
3129
- }>;
3130
- /**
3131
- * Optional. Required only for `stream: true` calls. Takes the same
3132
- * request shape as `generateContent`. Matching the real SDK's own
3133
- * `generateContentStream`, this resolves to an `AsyncIterable` (rather
3134
- * than returning one synchronously) of partial responses, each chunk
3135
- * holding the same `candidates[].content.parts[]` structure as
3136
- * `generateContent`'s response, just incremental.
3137
- */
3138
- generateContentStream?(params: Parameters<NonNullable<GeminiClient['generateContent']>>[0]): Promise<AsyncIterable<{
3139
- candidates?: Array<{
3140
- content?: {
3141
- parts?: Array<{
3142
- text?: string;
3143
- functionCall?: {
3144
- id?: string;
3145
- name?: string;
3146
- args?: unknown;
3147
- };
3148
- }>;
3149
- };
3150
- }>;
3151
- usageMetadata?: {
3152
- promptTokenCount?: number;
3153
- candidatesTokenCount?: number;
3154
- totalTokenCount?: number;
3155
- thoughtsTokenCount?: number;
3156
- };
3157
- }>>;
3158
- }
3159
- /**
3160
- * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
3161
- * uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
3162
- * axis: a `contents` array instead of `messages`, a separate
3163
- * `systemInstruction` field instead of a `system` role message,
3164
- * `generationConfig` instead of top-level `temperature`/`max_tokens`, and
3165
- * native JSON Schema support via `responseMimeType: 'application/json'` +
3166
- * `responseSchema`. `reasoning_effort` has no native Gemini equivalent, so
3167
- * it's converted to a `thinkingConfig.thinkingBudget` token count; `budget_tokens`
3168
- * maps to `thinkingBudget` directly, Gemini's native reasoning control. See
3169
- * `adapters/internal/reasoningBudget.utils.ts`.
3170
- *
3171
- * `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
3172
- * `tool_choice` maps to `toolConfig.functionCallingConfig`. Gemini accepts
3173
- * `responseSchema` and `tools` in the same request natively, so both are
3174
- * set independently here and no special-casing is needed for the
3175
- * combination, unlike `fromAnthropic`/`fromBedrock`.
3176
- *
3177
- * `createStream` calls `generateContentStream` (optional on `GeminiClient`
3178
- *, required only if the caller sets `stream: true`) and translates each
3179
- * partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
3180
- * Gemini's own function-calling API doesn't stream tool-call arguments
3181
- * incrementally: a `functionCall` part always arrives whole in one chunk,
3182
- * so each one is emitted as a single, complete `tool_call_delta` (a
3183
- * one-shot "delta" containing the full arguments) rather than accumulated
3184
- * fragments, that's a real difference in the underlying API, not
3185
- * something this adapter can smooth over. `usageMetadata` is (per Gemini's
3186
- * own behavior) only reliably present on the last chunk, so the `usage`
3187
- * `WireStreamChunk` is emitted once, after the stream completes, from
3188
- * whichever chunk's `usageMetadata` was seen last.
3189
- *
3190
- * Accepts a `GeminiClient` in either shape it structurally covers: the
3191
- * callable model methods directly (`ai.models`), or the complete
3192
- * top-level client (`ai`), unwrapping `.models` internally when present.
3193
- * Both work with no cast: `fromGemini(ai.models)` and `fromGemini(ai)`.
3194
- * Throws `LLMError('invalid_params')` up front if nothing callable
3195
- * results.
3196
- */
3197
- interface GeminiAdapterOptions {
3198
- /**
3199
- * Overrides the token count `reasoningEffort` tiers map onto when the
3200
- * caller sets `reasoningEffort` but not `budgetTokens` (Gemini has no
3201
- * tier string of its own, see `adapters/internal/reasoningBudget.utils.ts`).
3202
- * Only the tiers listed are changed; any omitted tier keeps the
3203
- * built-in default. Has no effect when `budgetTokens` is set directly.
3204
- */
3205
- reasoningEffortTokens?: Partial<EffortTokenTable>;
3206
- /**
3207
- * Marks additional models as using `thinkingLevel` instead of
3208
- * `thinkingBudget`, on top of this package's own built-in rule (every
3209
- * Gemini 3 series model and later, see `usesGeminiThinkingLevel` in
3210
- * `adapters/internal/reasoningBudget.utils.ts`). Additive, not a
3211
- * replacement: it can correct a false negative (a newer model this
3212
- * package doesn't know about yet), it can't un-mark a model the
3213
- * built-in rule already caught. Pass a static list of model IDs or a
3214
- * predicate.
3215
- */
3216
- thinkingLevelModels?: ModelCapabilityOverride;
3217
- }
3218
- export declare function fromGemini(client: GeminiClient, options?: GeminiAdapterOptions): LLMClient;
3219
- //#endregion
3220
- //#region src/adapters/bedrock.d.ts
3221
- /** Bedrock Converse's supported inline image formats. */
3222
- type BedrockImageFormat = 'png' | 'jpeg' | 'gif' | 'webp';
3223
- /** Bedrock Converse's native per-block content shape for a message. */
3224
- type BedrockContentBlock = {
3225
- text: string;
3226
- } | {
3227
- image: {
3228
- format: BedrockImageFormat;
3229
- source: {
3230
- bytes: Uint8Array;
3231
- };
3232
- };
3233
- } | {
3234
- toolUse: {
3235
- toolUseId: string;
3236
- name: string;
3237
- input: unknown;
3238
- };
3239
- } | {
3240
- toolResult: {
3241
- toolUseId: string;
3242
- content: Array<{
3243
- text: string;
3244
- }>;
3245
- status?: 'success' | 'error';
3246
- };
3247
- };
3248
- /**
3249
- * Minimal structural type matching AWS Bedrock's Converse API. This is
3250
- * intentionally NOT `BedrockRuntimeClient` itself, the AWS SDK v3 client
3251
- * exposes `.send(command)`, not a direct `.converse()` method, and pulling
3252
- * in `@aws-sdk/client-bedrock-runtime` as a dependency just for its types
3253
- * isn't worth it for a structural adapter. Wrap your client, e.g:
3254
- *
3255
- * ```ts
3256
- * import { BedrockRuntimeClient, ConverseCommand } from '@aws-sdk/client-bedrock-runtime';
3257
- * const client = new BedrockRuntimeClient({ region: 'us-east-1' });
3258
- * const converseClient = {
3259
- * converse: (params, options) =>
3260
- * client.send(new ConverseCommand(params), { abortSignal: options.signal }),
3261
- * };
3262
- * ```
3263
- */
3264
- interface BedrockConverseClient {
3265
- converse(params: {
3266
- modelId: string;
3267
- messages: Array<{
3268
- role: 'user' | 'assistant';
3269
- content: BedrockContentBlock[];
3270
- }>;
3271
- system?: Array<{
3272
- text: string;
3273
- }>;
3274
- inferenceConfig?: {
3275
- temperature?: number;
3276
- maxTokens?: number;
3277
- };
3278
- toolConfig?: {
3279
- tools: Array<{
3280
- toolSpec: {
3281
- name: string;
3282
- description?: string;
3283
- inputSchema: {
3284
- json: Record<string, unknown>;
3285
- };
3286
- strict?: boolean;
3287
- };
3288
- }>;
3289
- toolChoice?: {
3290
- tool: {
3291
- name: string;
3292
- };
3293
- } | {
3294
- auto: Record<string, never>;
3295
- } | {
3296
- any: Record<string, never>;
3297
- };
3298
- };
3299
- /**
3300
- * Native, schema-constrained output: a separate request field from
3301
- * `toolConfig`, so it can be sent alongside real tool calls. Only
3302
- * built by this adapter for models covered by
3303
- * `nativeStructuredOutputModels` (opt-in, see
3304
- * `BedrockAdapterOptions`); other models keep getting `jsonSchema`
3305
- * emulated as a forced single tool call via `toolConfig`, the
3306
- * pre-existing behavior.
3307
- *
3308
- * Matches the real Bedrock Converse API's `outputConfig.textFormat`
3309
- * shape exactly: the schema itself is nested one level deeper, under
3310
- * `structure.jsonSchema`, not flat on `textFormat`, and `schema` is
3311
- * a JSON-encoded *string*, not a parsed object, unlike every other
3312
- * schema field this adapter builds (`toolSpec.inputSchema.json`
3313
- * included). There is no `strict` field here, unlike `toolSpec`.
3314
- */
3315
- outputConfig?: {
3316
- textFormat?: {
3317
- type: 'json_schema';
3318
- structure: {
3319
- jsonSchema: {
3320
- schema: string;
3321
- name?: string;
3322
- description?: string;
3323
- };
3324
- };
3325
- };
3326
- /**
3327
- * Effort control for adaptive thinking, on Claude models where
3328
- * manual `budget_tokens` thinking is no longer accepted (see
3329
- * `supportsManualThinkingBudget` in
3330
- * `adapters/internal/reasoningBudget.utils.ts`). Sibling to
3331
- * `textFormat`, either or both may be present independently.
3332
- */
3333
- effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
3334
- };
3335
- /**
3336
- * Model-specific passthrough. Converse has no reasoning-budget field
3337
- * of its own, so a token budget for a Claude model on Bedrock is
3338
- * forwarded here under Anthropic's own key, `{ thinking: { type:
3339
- * 'enabled', budget_tokens } }`. Non-Claude models get nothing here,
3340
- * there is no equivalent field to reach for.
3341
- */
3342
- additionalModelRequestFields?: Record<string, unknown>;
3343
- }, options: {
3344
- signal: AbortSignal;
3345
- }): Promise<{
3346
- output?: {
3347
- message?: {
3348
- content?: Array<{
3349
- text?: string;
3350
- toolUse?: {
3351
- toolUseId?: string;
3352
- name?: string;
3353
- input?: unknown;
3354
- };
3355
- }>;
3356
- };
3357
- };
3358
- usage?: {
3359
- inputTokens?: number;
3360
- outputTokens?: number;
3361
- totalTokens?: number;
3362
- };
3363
- }>;
3364
- /**
3365
- * Optional. Required only for `stream: true` calls. Takes the same
3366
- * request shape `converse` does, returning `{ stream }`, matching
3367
- * `ConverseStreamCommand`'s real AWS SDK v3 output shape, an
3368
- * `AsyncIterable` of incremental events under a `stream` property,
3369
- * rather than the whole response being the iterable directly.
3370
- */
3371
- converseStream?(params: Parameters<BedrockConverseClient['converse']>[0], options: {
3372
- signal: AbortSignal;
3373
- }): Promise<{
3374
- stream: AsyncIterable<BedrockConverseStreamEvent>;
3375
- }>;
3376
- }
3377
- /**
3378
- * One event of a Bedrock `ConverseStreamCommand` response's `stream`.
3379
- * Content blocks (text or toolUse) are identified by `contentBlockIndex`,
3380
- * Converse's own convention for correlating start/delta/stop events across
3381
- * possibly-interleaved blocks, mirrored directly by VernLLM's
3382
- * `tool_call_delta.index`.
3383
- */
3384
- type BedrockConverseStreamEvent = {
3385
- messageStart: {
3386
- role: 'assistant';
3387
- };
3388
- } | {
3389
- contentBlockStart: {
3390
- contentBlockIndex: number;
3391
- start?: {
3392
- toolUse?: {
3393
- toolUseId?: string;
3394
- name?: string;
3395
- };
3396
- };
3397
- };
3398
- } | {
3399
- contentBlockDelta: {
3400
- contentBlockIndex: number;
3401
- delta?: {
3402
- text?: string;
3403
- } | {
3404
- toolUse?: {
3405
- input?: string;
3406
- };
3407
- };
3408
- };
3409
- } | {
3410
- contentBlockStop: {
3411
- contentBlockIndex: number;
3412
- };
3413
- } | {
3414
- messageStop: {
3415
- stopReason?: string;
3416
- };
3417
- } | {
3418
- metadata: {
3419
- usage?: {
3420
- inputTokens?: number;
3421
- outputTokens?: number;
3422
- totalTokens?: number;
3423
- };
3424
- };
3425
- } | {
3426
- internalServerException: {
3427
- message?: string;
3428
- };
3429
- } | {
3430
- modelStreamErrorException: {
3431
- message?: string;
3432
- originalStatusCode?: number;
3433
- };
3434
- } | {
3435
- validationException: {
3436
- message?: string;
3437
- };
3438
- } | {
3439
- throttlingException: {
3440
- message?: string;
3441
- };
3442
- } | {
3443
- serviceUnavailableException: {
3444
- message?: string;
3445
- };
3446
- };
3447
- /**
3448
- * Optional configuration for `fromBedrock`.
3449
- */
3450
- interface BedrockAdapterOptions {
3451
- /**
3452
- * Optional preflight check for tool-use support, needed whenever a
3453
- * `jsonSchema` call ends up sending Converse `toolConfig`, either the
3454
- * legacy forced-single-tool-call emulation, or real `tools` sent
3455
- * alongside native structured output (`outputConfig`). VernLLM never
3456
- * guesses capability from a failed call's error message (AWS's error
3457
- * text isn't a documented, stable contract), so this is opt-in: pass
3458
- * either a static list of tool-use-capable model IDs, or a predicate
3459
- * function, and VernLLM will reject unsupported models with a clear
3460
- * `LLMError('validation')` *before* dispatching the request, instead of
3461
- * on the wire.
3462
- *
3463
- * Left unset (default), no preflight check runs, and a `jsonSchema` call
3464
- * to an unsupported model surfaces Bedrock's raw `converse` error as-is.
3465
- */
3466
- toolUseSupportedModels?: string[] | ((modelId: string) => boolean);
3467
- /**
3468
- * Which models support native, schema-constrained output
3469
- * (`outputConfig.textFormat`), independent of `toolConfig`, so it can be
3470
- * combined with real `tools` in one request. Pass a static list of
3471
- * model IDs (verified against Bedrock's own docs) or a predicate.
3472
- *
3473
- * There is no built-in default here (see `supportsNativeStructuredOutput`
3474
- * for why). Left unset, every model uses the older forced-single-tool-
3475
- * call emulation via `toolConfig`, and `tools` + `jsonSchema` together is
3476
- * rejected, exactly this adapter's behavior before native support was
3477
- * added.
3478
- */
3479
- nativeStructuredOutputModels?: ModelCapabilityOverride;
3480
- /**
3481
- * Overrides the token count `reasoningEffort` tiers map onto when the
3482
- * caller sets `reasoningEffort` but not `budgetTokens` (Converse has no
3483
- * tier string of its own, see `adapters/internal/reasoningBudget.utils.ts`).
3484
- * Only the tiers listed are changed; any omitted tier keeps the
3485
- * built-in default. Has no effect when `budgetTokens` is set directly,
3486
- * or when the target model isn't a Claude model.
3487
- */
3488
- reasoningEffortTokens?: Partial<EffortTokenTable>;
3489
- /**
3490
- * Marks additional models as adaptive-only, on top of this package's
3491
- * own built-in rule (Claude Opus 4.7 and later, every Claude 5 tier
3492
- * model, see `isAdaptiveOnlyModel` in
3493
- * `adapters/internal/reasoningBudget.utils.ts`). Additive, not a
3494
- * replacement: it can correct a false negative (a newer model this
3495
- * package doesn't know about yet), it can't un-mark a model the
3496
- * built-in rule already caught. Pass a static list of model IDs or a
3497
- * predicate.
3498
- */
3499
- adaptiveOnlyModels?: ModelCapabilityOverride;
3500
- }
3501
- /**
3502
- * Minimal structural shape of an AWS SDK v3 client that exposes `.send()`,
3503
- * matching `BedrockRuntimeClient` (and its abort-signal-aware call
3504
- * convention). Avoids importing `@aws-sdk/client-bedrock-runtime` for the
423
+ * `cachedCall()` counterpart to `defineCallParams`, keeping the whole `{ cacheKey, ttl, call }`
3505
424
  * type.
3506
425
  */
3507
- interface AwsSendClient {
3508
- send(command: unknown, options?: {
3509
- abortSignal?: AbortSignal;
3510
- }): Promise<unknown>;
3511
- }
3512
- /**
3513
- * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
3514
- * interface VernLLM uses for OpenAI/Groq. The Converse API is unified
3515
- * across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
3516
- * so unlike raw per-model Bedrock invocation, this one adapter works
3517
- * regardless of which underlying model `modelId` points at, as long as
3518
- * that model supports Converse (most current-generation ones do)
3519
- *
3520
- * `bedrockClient` accepts either a hand-written `BedrockConverseClient`
3521
- * (a `.converse()`/`.converseStream()` wrapper you provide) or a real AWS
3522
- * SDK v3 client (anything with `.send()`, matching `BedrockRuntimeClient`)
3523
- * directly, detected structurally. Passing a raw AWS client skips the
3524
- * hand-written wrapper entirely, internally doing what it would
3525
- * (`send(new ConverseCommand(...))`, `send(new
3526
- * ConverseStreamCommand(...))`). See `wrapAwsSendClient` for how that path
3527
- * is implemented, including why `@aws-sdk/client-bedrock-runtime` stays
3528
- * out of this package's dependencies either way.
3529
- *
3530
- * `response_format: json_schema`, on a model covered by
3531
- * `options.nativeStructuredOutputModels` (opt-in, unset by default), is
3532
- * sent as `outputConfig.textFormat`, its own request field, independent of
3533
- * `toolConfig`, so it can be combined with real, caller-supplied `tools`
3534
- * in the same request. Matches the real Converse API's shape exactly: the
3535
- * schema is nested under `structure.jsonSchema` and JSON-encoded as a
3536
- * string, not the parsed object `toolConfig`'s tool schemas use, and there
3537
- * is no `strict` field on this path.
3538
- *
3539
- * On any other model (the default), `response_format: json_schema` is
3540
- * mapped to Converse's `toolConfig` instead: a single tool is defined from
3541
- * the schema, description, and strictness settings, and `toolChoice`
3542
- * forces the model to call it. This legacy path cannot be combined with
3543
- * real `tools` (both would need the same `toolConfig`), and a call that
3544
- * tries throws `LLMError('invalid_params')` with `code: 'unsupported_capability'`
3545
- * and `issues: { capability: 'tools_with_json_schema' }` before reaching the API.
3546
- * Provider-constrained schema matching applies only when `strict: true` is
3547
- * forwarded and supported. Native tool support varies by model family;
3548
- * pass `toolUseSupportedModels` to preflight-check it (see
3549
- * `BedrockAdapterOptions`), otherwise a `jsonSchema` call to an
3550
- * unsupported model surfaces Bedrock's raw error unchanged.
3551
- *
3552
- * `response_format: json_object` throws `LLMError('validation')`: Converse
3553
- * has no field that mechanically guarantees JSON output, and the only way
3554
- * to emulate it was an unenforced system-prompt instruction, a guarantee
3555
- * this adapter no longer pretends to make. Use `jsonSchema` instead.
3556
- * `reasoning_effort` (no Converse equivalent) is converted to a token
3557
- * budget and forwarded via `additionalModelRequestFields` for Claude
3558
- * models only; `budget_tokens` is forwarded the same way directly. Both
3559
- * are silently dropped for non-Claude models, which have no equivalent
3560
- * field to reach for. See `adapters/internal/reasoningBudget.utils.ts`.
3561
- *
3562
- * `tools` alone maps to Converse's native `toolConfig`/`toolUse`/
3563
- * `toolResult`; `tool_choice` maps to `toolConfig.toolChoice`.
3564
- *
3565
- * `createStream` calls `converseStream` (optional on `BedrockConverseClient`
3566
- *, required only if the caller sets `stream: true`) and translates its
3567
- * `contentBlockStart`/`contentBlockDelta`/`metadata` events into
3568
- * `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
3569
- * same as `fromAnthropic`'s block-index tracking (Converse's streaming
3570
- * shape is structurally close to Anthropic's own, both being tool-use-aware
3571
- * content-block streams), including the same `json-tool` unwrapping: a
3572
- * `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
3573
- * `text-delta`, not `tool_call_delta`, so the accumulated result lands in
3574
- * `finalizeResponse`'s `content` path exactly like the non-streaming
3575
- * `create` branch above unwraps it.
3576
- */
3577
- export declare function fromBedrock(bedrockClient: BedrockConverseClient | AwsSendClient, options?: BedrockAdapterOptions): LLMClient;
3578
- //#endregion
3579
- //#region src/adapters/fetch.d.ts
3580
- /** The chat-completion-shaped request VernLLM builds internally */
3581
- type ChatRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
3582
- /**
3583
- * The minimal shape the fetch adapter needs from a response object.
3584
- * Native `fetch`'s `Response` satisfies this, but so do wrappers around
3585
- * `axios`, `node-fetch`, `undici`, etc, which makes `request` swappable
3586
- * without forcing consumers to polyfill the full `Response` interface
3587
- */
3588
- interface ResponseLike {
3589
- ok: boolean;
3590
- status: number;
3591
- headers: {
3592
- get(name: string): string | null;
3593
- };
3594
- text(): Promise<string>;
3595
- json(): Promise<unknown>;
3596
- }
3597
- /** A fetch-compatible request function; defaults to native `fetch` */
3598
- type RequestLike = (url: string, init: {
3599
- method: string;
3600
- headers: Record<string, string>;
3601
- body?: string;
3602
- signal?: AbortSignal;
3603
- }) => Promise<ResponseLike>;
3604
- /**
3605
- * A streaming-capable request function. Unlike `RequestLike`, which returns
3606
- * a fully-buffered `ResponseLike`, this resolves to an `AsyncIterable` of
3607
- * progressively-arriving chunks, the common ground across transports:
3608
- * native `fetch`'s `response.body` (wrapped to be iterable; see
3609
- * `webStreamToAsyncIterable` below), axios's Node `Readable` in
3610
- * `responseType: 'stream'` mode (already async-iterable, no wrapping
3611
- * needed), `node-fetch`, `undici`, etc, all satisfy this with little or no
3612
- * glue code. Defaults to native `fetch`.
3613
- */
3614
- type StreamRequestLike = (url: string, init: {
3615
- method: string;
3616
- headers: Record<string, string>;
3617
- body?: string;
3618
- signal?: AbortSignal;
3619
- }) => Promise<AsyncIterable<Uint8Array | string>>;
3620
- interface FetchAdapterConfig {
3621
- /** Endpoint URL, or a function of the request in case it depends on model/params */
3622
- url: string | ((params: ChatRequest) => string);
3623
- /** Static headers, or a function (sync or async) for things like refreshed auth tokens */
3624
- headers?: Record<string, string> | (() => Record<string, string> | Promise<Record<string, string>>);
3625
- /** HTTP method. Default 'POST' */
3626
- method?: string;
3627
- /**
3628
- * The function used to make the HTTP request. Defaults to native `fetch`.
3629
- * Swap in `axios`, `node-fetch`, or any other transport, as long as it
3630
- * resolves to a `ResponseLike` object
3631
- */
3632
- request?: RequestLike;
3633
- /** Maps VernLLMs internal chat-completion request into the providers raw request body */
3634
- mapRequest: (params: ChatRequest) => unknown;
3635
- /**
3636
- * Maps the providers raw JSON response into `{ content, usage?, toolCalls? }`
3637
- * `content` is the assistants text (JSON string when JSON mode was requested).
3638
- * `content` may be empty/omitted when the model responded with only tool
3639
- * calls and no text.
3640
- *
3641
- * `toolCalls`, when the model requested one or more tools, is the list of
3642
- * calls as flat `{ id, name, arguments }` entries (matching this config's
3643
- * own `toolCalls?: Array<{ id: string; name: string; arguments: string }>`
3644
- * return type below), each entry's `arguments` already JSON-*encoded* as a
3645
- * string (not the parsed object), mirroring the wire format every
3646
- * OpenAI-compatible provider uses. `fromFetch` itself converts these into
3647
- * `WireToolCall`'s `type`/`function`-wrapped shape before returning them
3648
- * from `create`. VernLLM parses (and validates, if `argumentsSchema` was
3649
- * set) the arguments string internally, mapResponse doesn't need to do
3650
- * that itself.
3651
- */
3652
- mapResponse: (json: unknown) => {
3653
- content?: string;
3654
- usage?: {
3655
- promptTokens?: number;
3656
- completionTokens?: number;
3657
- totalTokens?: number;
3658
- };
3659
- toolCalls?: Array<{
3660
- id: string;
3661
- name: string;
3662
- arguments: string;
3663
- }>;
3664
- };
3665
- /**
3666
- * Optional. Required only for `stream: true` calls. The function used to
3667
- * open a streaming HTTP request. Takes the same request shape as
3668
- * `request`, but resolves to an `AsyncIterable` of progressively-arriving
3669
- * `Uint8Array` or `string` chunks instead of a buffered `ResponseLike`.
3670
- * Defaults to native `fetch`.
3671
- */
3672
- requestStream?: StreamRequestLike;
3673
- /**
3674
- * Optional. How the raw stream bytes are split into individual event
3675
- * payloads. Defaults to Server-Sent Events framing (`data: ...` blocks
3676
- * separated by a blank line, `[DONE]` sentinel honored, see
3677
- * `parseSseStream`), which covers the large majority of LLM providers'
3678
- * streaming HTTP endpoints. Override this for a provider that frames its
3679
- * stream differently, e.g. newline-delimited JSON (NDJSON) with no SSE
3680
- * envelope.
3681
- */
3682
- parseStreamFrames?: (chunks: AsyncIterable<Uint8Array | string>) => AsyncIterable<unknown>;
3683
- /**
3684
- * Optional. Required only for `stream: true` calls. Maps one parsed
3685
- * stream event (already extracted from its frame by `parseStreamFrames`)
3686
- * into zero, one, or more `WireStreamChunk`s, mirrors `mapResponse`'s
3687
- * role for the non-streaming path, just per-event instead of once for
3688
- * the whole body. Return `undefined` to skip an event that carries
3689
- * nothing VernLLM needs (e.g. a provider's keep-alive ping). Configs
3690
- * that don't implement this make `stream: true` throw a clear
3691
- * `LLMError('validation')` rather than a confusing runtime failure or a
3692
- * silently empty stream.
3693
- */
3694
- mapStreamEvent?: (event: unknown) => WireStreamChunk | WireStreamChunk[] | undefined;
3695
- /**
3696
- * Optional. How to read AIMD's proactive rate limit hint off a
3697
- * successful response. Defaults to OpenAI's header set.
3698
- */
3699
- parseRateLimitHint?: (headers: ResponseLike['headers']) => ProviderRateLimitHint;
3700
- }
3701
- /**
3702
- * A fetch-based escape hatch for providers with no SDK, or where pulling one
3703
- * in isnt worth it. You supply the URL, headers, and two small mapping
3704
- * functions; this handles the HTTP call and slots the result into the same
3705
- * `LLMClient` shape every other adapter produces, so retries, timeouts,
3706
- * the circuit breaker, and JSON/schema handling all still work unmodified
3707
- *
3708
- * Non-2xx responses throw an error with `.status` set to the HTTP status
3709
- * code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
3710
- * 401/403) applies here too
3711
- *
3712
- * Tool calling works the same way as every other adapter: `mapRequest`
3713
- * receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
3714
- * translate them into whatever shape the provider's wire format expects
3715
- * (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
3716
- * field). On the way back, `mapResponse` may return a `toolCalls` array
3717
- * (id/name/JSON-encoded-arguments-string per call) alongside or instead of
3718
- * `content`; VernLLM parses and (if `argumentsSchema` was set) validates
3719
- * those arguments the same way it does for every other adapter. For
3720
- * `stream: true`, tool-call deltas go through the existing
3721
- * `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
3722
- * no separate config is needed for streaming vs non-streaming tool calls.
3723
- *
3724
- * `createStream` requires `mapStreamEvent` (there's no non-streaming
3725
- * response to fall back on, unlike the other three optional streaming
3726
- * seams). It opens the request via `requestStream` (defaults to native
3727
- * `fetch`), splits the raw bytes into individual events via
3728
- * `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
3729
- * and translates each event into `WireStreamChunk`(s) via
3730
- * `mapStreamEvent`. Both seams are overridable per-config for providers
3731
- * that don't fit the SSE-over-fetch default. If a custom `request`
3732
- * transport is configured, `requestStream` must be configured too,
3733
- * `requestStream` never silently falls back to `request` (see
3734
- * `createStream`'s own comment for why), so a `stream: true` call with
3735
- * `request` set but no `requestStream` throws a clear
3736
- * `LLMError('validation')` instead of quietly using unrelated native
3737
- * `fetch`.
3738
- */
3739
- export declare function fromFetch(config: FetchAdapterConfig): LLMClient;
3740
- //#endregion
3741
- //#region src/adapters/openaiCompatible.d.ts
3742
- /**
3743
- * Adapter for any SDK/client whose `chat.completions.create` already
3744
- * matches the OpenAI wire format: this covers most hosted inference
3745
- * providers, since "OpenAI-compatible" is a de facto standard for chat
3746
- * completion APIs. Almost everything passes straight through untouched,
3747
- * this exists purely so call sites read clearly (`fromMistral(client)` vs
3748
- * handing a Mistral client to something typed for OpenAI) and so a real
3749
- * transformation could be added later, per-provider, without a breaking
3750
- * change.
3751
- *
3752
- * The one thing that isn't a pure passthrough: a `ContentBlock[]`
3753
- * `userContent` is translated into OpenAI's native `image_url` content-part
3754
- * shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
3755
- * rather than a copy of any one provider's wire format.
3756
- *
3757
- * Not every SDKs own TypeScript types line up exactly with `LLMClient`
3758
- * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
3759
- * the actual compatibility contract is the JSON each provider sends and
3760
- * receives over the wire, not the SDKs TS types.
3761
- *
3762
- * `createStream` is implemented by calling the same underlying
3763
- * `chat.completions.create` with `stream: true` (and, for providers that
3764
- * support it, `stream_options: { include_usage: true }`, so a final usage
3765
- * block arrives), the OpenAI SDK, and every OpenAI-compatible client
3766
- * modeled on it, returns an `AsyncIterable` of SSE chunks instead of a
3767
- * single completion object when `stream: true` is set. Each chunk is
3768
- * translated into `WireStreamChunk`(s) via `toWireStreamChunks`.
3769
- *
3770
- * Note on long-running reasoning models: this adapter consumes the
3771
- * underlying SDK's already-parsed stream rather than raw SSE bytes, so
3772
- * unlike `fromFetch`/`fromAnthropic` it cannot see comment-only keep-alive
3773
- * ping frames. Combined with `chunkIdleTimeoutMs`'s 30 second default and
3774
- * `reasoningEffort` (documented to have long silent gaps for o-series and
3775
- * similar models), a long-running reasoning call on this adapter can trip
3776
- * the idle timeout even though the provider is still working. Raise or
3777
- * disable `chunkIdleTimeoutMs` per call for those routes, see `CallParams`.
3778
- */
3779
- interface OpenAICompatibleAdapterOptions {
3780
- /**
3781
- * Whether the provider supports `stream_options.include_usage`. Not
3782
- * every "OpenAI-compatible" provider is guaranteed to, so this defaults
3783
- * to `true` (matching OpenAI, Groq, Mistral, and most others observed)
3784
- * and should be set to `false` for a provider verified not to support
3785
- * it. When `false`, `stream_options` is omitted entirely and no usage
3786
- * block will arrive on the stream; callers relying on streamed `usage`
3787
- * with such a provider won't get one.
3788
- */
3789
- supportsStreamUsage?: boolean;
3790
- /**
3791
- * Overrides the token count `budgetTokens` buckets into when the caller
3792
- * sets `budgetTokens` but not `reasoningEffort` (OpenAI-compatible
3793
- * clients have no numeric budget field of their own, see
3794
- * `adapters/internal/reasoningBudget.utils.ts`). Only the tiers listed
3795
- * are changed; any omitted tier keeps the built-in default. Has no
3796
- * effect when `reasoningEffort` is set directly.
3797
- */
3798
- reasoningEffortTokens?: Partial<EffortTokenTable>;
3799
- /**
3800
- * Whether the client's request builder supports `.withResponse()`
3801
- * (needed for AIMD's proactive path). Default `false`, since not
3802
- * every "OpenAI-compatible" client is confirmed to support it.
3803
- */
3804
- supportsWithResponse?: boolean;
3805
- }
3806
- export declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
3807
- /**
3808
- * Named alias for the OpenAI SDK itself. A raw `new OpenAI(...)` instance
3809
- * structurally matches most of `LLMClient`, but newer `openai` SDK major
3810
- * versions have widened `ChatCompletionContentPart` (e.g. adding a `file`
3811
- * variant) in ways that no longer structurally satisfy VernLLM's
3812
- * provider-agnostic `ContentBlock[]` on `userContent`, so passing the SDK
3813
- * instance directly can fail to typecheck depending on the installed
3814
- * `openai` version. Wrapping with `fromOpenAI()` sidesteps that by
3815
- * translating through `unknown` at the boundary, and also picks up
3816
- * multimodal image translation and `createStream` wiring that a raw
3817
- * client doesn't have. See Migration Notes for details.
3818
- *
3819
- * `supportsWithResponse` defaults to `false` here too:
3820
- * `client` is `unknown`, so there's no way to verify it's really the
3821
- * official `openai` package's client versus a fake or a test double.
3822
- * Pass `supportsWithResponse: true` once you've confirmed it.
3823
- */
3824
- export declare const fromOpenAI: typeof fromOpenAICompatible;
3825
- /** Groqs SDK matches the OpenAI wire format */
3826
- export declare const fromGroq: typeof fromOpenAICompatible;
3827
- /**
3828
- * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
3829
- * endpoint). Mistral supports `stream_options.include_usage` (added after
3830
- * an earlier period where it returned a 422 for unrecognized fields, per
3831
- * Mistral's changelog and streaming docs), so this is a plain alias like
3832
- * the others, `supportsStreamUsage` defaults to `true`.
3833
- */
3834
- export declare const fromMistral: typeof fromOpenAICompatible;
3835
- /** DeepSeeks API is OpenAI-compatible */
3836
- export declare const fromDeepSeek: typeof fromOpenAICompatible;
3837
- /** Cerebras inference API is OpenAI-compatible */
3838
- export declare const fromCerebras: typeof fromOpenAICompatible;
3839
- /** Together AIs API is OpenAI-compatible */
3840
- export declare const fromTogether: typeof fromOpenAICompatible;
3841
- /** Fireworks AIs API is OpenAI-compatible */
3842
- export declare const fromFireworks: typeof fromOpenAICompatible;
3843
- /**
3844
- * Ollama exposes an OpenAI-compatible endpoint at `/v1/chat/completions`
3845
- * (as opposed to its native `/api/chat` format, which differs). Point an
3846
- * OpenAI SDK instances `baseURL` at your Ollama server and pass it here:
3847
- * this does not talk to Ollamas native API directly.
3848
- */
3849
- export declare const fromOllama: typeof fromOpenAICompatible;
3850
- /** OpenRouter's API is OpenAI-compatible */
3851
- export declare const fromOpenRouter: typeof fromOpenAICompatible;
3852
- /** Perplexity's API is OpenAI-compatible */
3853
- export declare const fromPerplexity: typeof fromOpenAICompatible;
3854
- /** DeepInfra's API is OpenAI-compatible */
3855
- export declare const fromDeepInfra: typeof fromOpenAICompatible;
3856
- /** Novita's API is OpenAI-compatible */
3857
- export declare const fromNovita: typeof fromOpenAICompatible;
3858
- /** Hyperbolic's API is OpenAI-compatible */
3859
- export declare const fromHyperbolic: typeof fromOpenAICompatible;
3860
- /** Moonshot's (Kimi) API is OpenAI-compatible */
3861
- export declare const fromMoonshot: typeof fromOpenAICompatible;
3862
- /** Zhipu's (GLM) API is OpenAI-compatible */
3863
- export declare const fromZhipu: typeof fromOpenAICompatible;
3864
- /**
3865
- * LM Studio exposes an OpenAI-compatible endpoint at `/v1/chat/completions`.
3866
- * Point an OpenAI SDK instance's `baseURL` at your local LM Studio server.
3867
- */
3868
- export declare const fromLMStudio: typeof fromOpenAICompatible;
3869
- /**
3870
- * vLLM's OpenAI-compatible server mode exposes `/v1/chat/completions`.
3871
- * Point an OpenAI SDK instance's `baseURL` at your vLLM server.
3872
- */
3873
- export declare const fromVLLM: typeof fromOpenAICompatible;
3874
- /** xAI's Grok API is OpenAI-compatible */
3875
- export declare const fromXAI: typeof fromOpenAICompatible;
3876
- /** NVIDIA NIM's hosted and self-hosted endpoints are OpenAI-compatible */
3877
- export declare const fromNvidiaNIM: typeof fromOpenAICompatible;
3878
- /** Vercel AI Gateway is OpenAI-compatible */
3879
- export declare const fromVercelAIGateway: typeof fromOpenAICompatible;
3880
- /** Cloudflare Workers AI exposes an OpenAI-compatible endpoint */
3881
- export declare const fromCloudflareWorkersAI: typeof fromOpenAICompatible;
3882
- /** Nebius AI Studio is OpenAI-compatible */
3883
- export declare const fromNebius: typeof fromOpenAICompatible;
3884
- /** SambaNova Cloud's API is OpenAI-compatible */
3885
- export declare const fromSambaNova: typeof fromOpenAICompatible;
3886
- /** Baseten's model hosting exposes an OpenAI-compatible endpoint */
3887
- export declare const fromBaseten: typeof fromOpenAICompatible;
3888
- /** Featherless AI's API is OpenAI-compatible */
3889
- export declare const fromFeatherless: typeof fromOpenAICompatible;
3890
- /** Friendli AI's serving endpoint is OpenAI-compatible */
3891
- export declare const fromFriendli: typeof fromOpenAICompatible;
3892
- /** SiliconFlow's API is OpenAI-compatible */
3893
- export declare const fromSiliconFlow: typeof fromOpenAICompatible;
3894
- /** Parasail's inference API is OpenAI-compatible */
3895
- export declare const fromParasail: typeof fromOpenAICompatible;
3896
- /** StepFun's API is OpenAI-compatible */
3897
- export declare const fromStepFun: typeof fromOpenAICompatible;
3898
- /** MiniMax's API is OpenAI-compatible */
3899
- export declare const fromMiniMax: typeof fromOpenAICompatible;
3900
- /** Lambda Labs' Inference API is OpenAI-compatible */
3901
- export declare const fromLambdaLabs: typeof fromOpenAICompatible;
3902
- /** Snowflake Cortex's LLM endpoint is OpenAI-compatible */
3903
- export declare const fromSnowflakeCortex: typeof fromOpenAICompatible;
3904
- /** Anyscale Endpoints' API is OpenAI-compatible */
3905
- export declare const fromAnyscale: typeof fromOpenAICompatible;
3906
- /** Lepton AI's inference API is OpenAI-compatible */
3907
- export declare const fromLepton: typeof fromOpenAICompatible;
3908
- /** Inference.net's API is OpenAI-compatible */
3909
- export declare const fromInferenceNet: typeof fromOpenAICompatible;
3910
- /** Infermatic's API is OpenAI-compatible */
3911
- export declare const fromInfermatic: typeof fromOpenAICompatible;
3912
- /** AtlasCloud's inference API is OpenAI-compatible */
3913
- export declare const fromAtlasCloud: typeof fromOpenAICompatible;
3914
- /** 01.AI's (Yi models) API is OpenAI-compatible */
3915
- export declare const from01AI: typeof fromOpenAICompatible;
426
+ export declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
3916
427
  //#endregion
3917
- export type { AnthropicClient, AssistantContent, AttemptContext, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedConditionalToolCallParams, CachedJsonModeDisabledCallParams, CachedJsonModeEnabledCallParams, CachedStreamCallParams, CachedStreamConditionalToolCallParams, CachedStreamJsonModeDisabledCallParams, CachedStreamJsonModeEnabledCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallMeta, CallParams, CallResult, CallWithToolsResult, CircuitBreakerAdapter, CircuitBreakerCallContext, CircuitBreakerOptions, CircuitBreakerStateChangeHandler, CircuitState, CircuitTarget, ConditionalToolCallParams, ContentBlock, ContentResult, ConversationTurn, CooldownBackoff, CreateMiddlewareOptions, DuplicateToolNamesIssue, EvictionOption, ExponentialBackoffOptions, FallbackAttempt, FallbackOn, FallbackTarget, FetchAdapterConfig, GeminiClient, HistoryToolResultIssue, ImageBlock, JsonModeDisabledCallParams, JsonModeEnabledCallParams, JsonSchemaSpec, JsonValue, LLMClient, LLMErrorCode, LLMErrorIssuesByCode, LLMErrorSnapshot, LLMErrorType, LLMRequestShape, LLMRequestSnapshot, Logger, MiddlewareCapabilities, MiddlewareContext, MiddlewareContextBase, MiddlewareRef, MiddlewareStateBag, MiddlewareStateKey, OnEvent, OnUsage, PreDispatchContext, RateLimitAcquireResult, RateLimitOptions, RateLimitReason, RateLimitState, RateLimiterAdapter, RefundUsage, RequiredMiddlewareRef, ReserveUsage, RetryAttempt, RetryBudgetOptions, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, StreamJsonModeDisabledCallParams, StreamJsonModeEnabledCallParams, TargetCircuitState, TextBlock, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolIssue, ToolResult, ToolsDisabledCallParams, TrippingPolicy, UnknownToolChoiceIssue, UnsupportedCapabilityIssue, VernLLMEvent, VernLLMMiddleware, VernLLMOptions, WireCallRequest, WireCallRequestPatch, WireMessage, WireRequest, WireResponseFormat, WireStreamChunk, WireTool, WireToolCall, WireToolChoice };
428
+ export { type AdapterInfo, type AssistantContent, type AttemptContext, type CacheAdapter, type CachedCallParams, type CachedConditionalToolCallParams, type CachedJsonModeDisabledCallParams, type CachedJsonModeEnabledCallParams, type CachedStreamCallParams, type CachedStreamConditionalToolCallParams, type CachedStreamJsonModeDisabledCallParams, type CachedStreamJsonModeEnabledCallParams, type CachedStreamToolCallParams, type CachedToolCallParams, type CallContext, type CallMeta, type CallParams, type CallResult, type CallWithToolsResult, CircuitBreaker, type CircuitBreakerAdapter, type CircuitBreakerCallContext, type CircuitBreakerOptions, type CircuitBreakerStateChangeHandler, type CircuitState, type CircuitTarget, type ConditionalToolCallParams, ConsecutiveTripping, ConsoleLogger, type ContentBlock, type ContentResult, type ConversationTurn, type CooldownBackoff, type CreateMiddlewareOptions, type DuplicateToolNamesIssue, type EvictionOption, type ExponentialBackoffOptions, type FallbackAttempt, FallbackExhaustedError, type FallbackOn, type FallbackOnContext, type FallbackTarget, type HistoryToolResultIssue, type ImageBlock, type JsonModeDisabledCallParams, type JsonModeEnabledCallParams, type JsonSchemaSpec, type JsonValue, type LLMClient, LLMError, type LLMErrorCode, type LLMErrorIssuesByCode, type LLMErrorSnapshot, type LLMErrorType, type LLMRequestShape, type LLMRequestSnapshot, type Logger, type MiddlewareCapabilities, type MiddlewareContext, type MiddlewareContextBase, type MiddlewareRef, type MiddlewareStateBag, type MiddlewareStateEntry, type MiddlewareStateKey, type OnEvent, type OnUsage, type PreDispatchContext, type RateLimitAcquireResult, type RateLimitOptions, type RateLimitReason, type RateLimitState, RateLimiter, type RateLimiterAdapter, type RefundUsage, type RequiredMiddlewareRef, type ReserveUsage, type RetryAttempt, RetryBudget, type RetryBudgetOptions, RollingTripping, type SchemaLike, type StreamCallResult, type StreamChunk, type StreamEnabledCallParams, type StreamJsonModeDisabledCallParams, type StreamJsonModeEnabledCallParams, type TargetCircuitState, type TargetInfo, type TextBlock, type ThinkingBlock, type TokenUsage, type ToolCall, type ToolCallResult, type ToolChoice, type ToolDefinition, type ToolEnabledCallParams, type ToolIssue, type ToolResult, type ToolsDisabledCallParams, type TrippingPolicy, type UnknownToolChoiceIssue, type UnsupportedCapabilityIssue, type VernLLMEvent, type VernLLMMiddleware, type VernLLMOptions, type WireCallRequest, type WireCallRequestPatch, type WireMessage, type WireRequest, type WireResponseFormat, type WireStreamChunk, type WireTool, type WireToolCall, type WireToolChoice, createMiddlewareRef, createMiddlewareStateBag, createStateKey, defaultFallbackOn, defineTool, hasIssues, isFallbackExhaustedError, isLLMError, isStreamResult, isToolCallResult, metaRef, requireRef, stateEntry };
3918
429
  //# sourceMappingURL=index.d.cts.map