vern-llm 2.9.1 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1871 @@
1
+ //#region src/types/errors.d.ts
2
+ type LLMErrorType = 'timeout' | 'api' | 'network' | 'parse' | 'validation' | 'invalid_params' | 'rate_limited' | 'quota_exceeded' | 'circuit_open' | 'fallback_exhausted' | 'aborted' | 'unknown';
3
+ /**
4
+ * Machine readable detail within a `type`, for when `type` alone is too coarse to act on. Optional:
5
+ * errors from before a code existed omit it.
6
+ */
7
+ type LLMErrorCode = 'unknown_tool' | 'duplicate_tool_call_id' | 'tool_choice_none_violated' | 'unexpected_tool_calls' | 'unsupported_capability' | 'duplicate_tool_names' | 'unknown_tool_choice' | 'duplicate_tool_result_ids' | 'unknown_tool_result_ids' | 'missing_tool_results' | 'invalid_context' | 'unknown_target' | 'no_eligible_targets' | 'middleware_threw' | 'rate_limit_queue_full' | 'rate_limit_queue_timeout' | 'rate_limit_capacity_exceeded' | 'provider_rate_limited' | 'retry_budget_exhausted' | 'request_timeout' | 'idle_timeout' | 'reader_stall_timeout' | 'middleware_timeout' | 'deadline_exceeded' | 'authentication' | 'authorization' | 'not_found' | 'payload_too_large' | 'server_error' | 'empty_response' | 'connection_failed' | 'circuit_cooling_down' | 'circuit_trial_in_flight' | 'fallback_exhausted' | 'tool_arguments_parse_failed' | 'stream_frame_invalid' | 'response_truncated' | 'soft_failure_detected';
8
+ /** One tool call's contract failure, used to report every bad call in a response at once. */
9
+ interface ToolIssue {
10
+ name: string;
11
+ toolCallId: string;
12
+ code: LLMErrorCode;
13
+ detail?: unknown;
14
+ }
15
+ /**
16
+ * The specific values behind a `duplicate_tool_names` failure: the
17
+ * offending call's `tools` array had more than one entry sharing a name.
18
+ */
19
+ interface DuplicateToolNamesIssue {
20
+ names: string[];
21
+ }
22
+ /**
23
+ * The specific values behind an `unknown_tool_choice` failure: `toolChoice`
24
+ * named a tool that wasn't in the call's own `tools` array.
25
+ */
26
+ interface UnknownToolChoiceIssue {
27
+ requested: string;
28
+ available: string[];
29
+ }
30
+ /**
31
+ * The specific values behind a `duplicate_tool_result_ids` /
32
+ * `unknown_tool_result_ids` / `missing_tool_results` failure: which
33
+ * `history` turn was affected, and which `toolCallId`s were the problem.
34
+ */
35
+ interface HistoryToolResultIssue {
36
+ historyIndex: number;
37
+ ids: string[];
38
+ }
39
+ /**
40
+ * The specific values behind an `unsupported_capability` failure: which
41
+ * capability the current adapter/client/model doesn't support.
42
+ */
43
+ interface UnsupportedCapabilityIssue {
44
+ capability: string;
45
+ }
46
+ /**
47
+ * The exact `issues` shape for each code that carries one. Codes whose message already says
48
+ * everything have no entry. Schema validation `issues` stay untyped, since they come from the
49
+ * caller's own validator.
50
+ */
51
+ interface LLMErrorIssuesByCode {
52
+ unknown_tool: ToolIssue[];
53
+ duplicate_tool_call_id: ToolIssue[];
54
+ duplicate_tool_names: DuplicateToolNamesIssue;
55
+ unknown_tool_choice: UnknownToolChoiceIssue;
56
+ duplicate_tool_result_ids: HistoryToolResultIssue;
57
+ unknown_tool_result_ids: HistoryToolResultIssue;
58
+ missing_tool_results: HistoryToolResultIssue;
59
+ unsupported_capability: UnsupportedCapabilityIssue;
60
+ }
61
+ /**
62
+ * Plain data copy of an `LLMError`, as held by `RetryAttempt.error`. Never thrown again, so no
63
+ * `cause` and no live getters. Nested `attempts` form a tree, not a cycle.
64
+ */
65
+ interface LLMErrorSnapshot {
66
+ message: string;
67
+ type: LLMErrorType;
68
+ status?: number;
69
+ issues?: unknown;
70
+ retryAfterMs?: number;
71
+ code?: LLMErrorCode;
72
+ /** Computed once, at snapshot time, since a snapshot has no live getter. */
73
+ retryable: boolean;
74
+ /** This attempt's own prior attempts, if it was itself the terminal failure of a retry loop. */
75
+ attempts?: RetryAttempt[];
76
+ }
77
+ /**
78
+ * Plain data copy of the request an attempt sent, as held by `RetryAttempt.request`. Safe to
79
+ * serialize and store.
80
+ */
81
+ interface LLMRequestSnapshot {
82
+ /** Provider id this attempt targeted, e.g. "openai". */
83
+ provider: string;
84
+ /** Model id this attempt targeted. */
85
+ model: string;
86
+ /** The payload as actually sent for this attempt, after any transform/repair. Passed through `safeBody`. */
87
+ body: unknown;
88
+ /** Non sensitive request headers. Auth headers are stripped before the snapshot is built, never included. */
89
+ headers?: Record<string, string>;
90
+ /** Wall clock time the attempt started, ms since epoch. */
91
+ startedAt: number;
92
+ }
93
+ /** One failed attempt: its index and a snapshot of its error. `FallbackAttempt` extends it. */
94
+ interface RetryAttempt {
95
+ index: number;
96
+ error: LLMErrorSnapshot;
97
+ /** What was sent for this attempt. Optional: absent for attempts predating this field. */
98
+ request?: LLMRequestSnapshot;
99
+ }
100
+ /** Optional fields for constructing an {@link LLMError}. `message` and `type` stay positional since every throw site sets both. */
101
+ interface LLMErrorOptions {
102
+ status?: number;
103
+ issues?: unknown;
104
+ cause?: unknown;
105
+ retryAfterMs?: number;
106
+ /** Stable discriminator within `type`. Absent on errors predating it. */
107
+ code?: LLMErrorCode;
108
+ /** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
109
+ attempts?: RetryAttempt[];
110
+ }
111
+ declare class LLMError extends Error {
112
+ type: LLMErrorType;
113
+ status?: number;
114
+ issues?: unknown;
115
+ cause?: unknown;
116
+ retryAfterMs?: number;
117
+ /** Stable discriminator within `type`. Absent on errors predating it. */
118
+ code?: LLMErrorCode;
119
+ /** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
120
+ attempts?: RetryAttempt[];
121
+ constructor(message: string, type: LLMErrorType, options?: LLMErrorOptions);
122
+ /**
123
+ * Whether retrying could help, from `type` and `code` alone. See Error Handling for the full
124
+ * list. `response_truncated` is retryable despite its `parse` type.
125
+ */
126
+ get retryable(): boolean;
127
+ /**
128
+ * Whether this failure counts toward the breaker. Stricter than `retryable`: `quota_exceeded` and
129
+ * any 4xx other than 408, 425 and 429 are excluded.
130
+ */
131
+ get countsTowardBreaker(): boolean;
132
+ /**
133
+ * Copies the fields into an `LLMErrorSnapshot`. `cause` is left out, and `issues` are made
134
+ * serialization safe.
135
+ */
136
+ toSnapshot(): LLMErrorSnapshot;
137
+ /**
138
+ * Serializes `message` and `retryable` too, which a plain property walk misses. `cause` is left
139
+ * out since SDK errors can be circular; read `err.cause` directly.
140
+ */
141
+ toJSON(): Record<string, unknown>;
142
+ }
143
+ declare function isLLMError(err: unknown): err is LLMError;
144
+ /**
145
+ * Narrows `err.issues` to the shape `LLMErrorIssuesByCode` maps `code` to, so no cast is needed.
146
+ */
147
+ declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
148
+ code: C;
149
+ issues: LLMErrorIssuesByCode[C];
150
+ };
151
+ //#endregion
152
+ //#region src/logger.d.ts
153
+ interface Logger {
154
+ debug(message: string): void;
155
+ warn(message: string): void;
156
+ error(message: string, meta?: Record<string, unknown>): void;
157
+ }
158
+ /**
159
+ * Default logger. `debug` is gated by the `debug` option on VernLLM
160
+ * warn/error always fire since they indicate real problems (retries, cache failures)
161
+ */
162
+ declare class ConsoleLogger implements Logger {
163
+ private debugEnabled;
164
+ constructor(debugEnabled: boolean);
165
+ debug(message: string): void;
166
+ warn(message: string): void;
167
+ error(message: string, meta?: Record<string, unknown>): void;
168
+ }
169
+ //#endregion
170
+ //#region src/types/usage.d.ts
171
+ type ReserveUsage = (params: {
172
+ coalesced: boolean;
173
+ signal?: AbortSignal;
174
+ }) => Promise<void>;
175
+ type RefundUsage = (params: {
176
+ coalesced: boolean;
177
+ signal?: AbortSignal;
178
+ }) => Promise<void>;
179
+ /**
180
+ * The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
181
+ * and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
182
+ * has one definition instead of being redeclared at each use site.
183
+ */
184
+ interface UsageHooks {
185
+ /**
186
+ * Reserves usage before the request. Failures become
187
+ * LLMError('quota_exceeded').
188
+ */
189
+ reserveUsage?: ReserveUsage;
190
+ /**
191
+ * Refunds usage after a failed call if reservation succeeded.
192
+ */
193
+ refundUsage?: RefundUsage;
194
+ }
195
+ interface TokenUsage {
196
+ /**
197
+ * Every input token the provider processed: uncached input plus cache reads plus cache writes.
198
+ */
199
+ promptTokens: number;
200
+ completionTokens: number;
201
+ totalTokens: number;
202
+ /**
203
+ * Reasoning tokens, a subset of `completionTokens`, not extra. `undefined` when the provider
204
+ * doesn't report them separately.
205
+ */
206
+ reasoningTokens?: number;
207
+ /** Cache reads, a subset of `promptTokens`. `undefined` when not reported. */
208
+ cacheReadTokens?: number;
209
+ /** Cache writes, a subset of `promptTokens`. `undefined` when not reported. */
210
+ cacheWriteTokens?: number;
211
+ /** `cacheWriteTokens` by the provider's TTL label. `undefined` when not split. */
212
+ cacheWriteTokensByTtl?: Readonly<Record<string, number>>;
213
+ requestId: string;
214
+ model: string;
215
+ /**
216
+ * The target that produced this usage, `'primary'` by default. Always set by VernLLM; optional
217
+ * for hand built values.
218
+ */
219
+ provider?: string;
220
+ /**
221
+ * Whether a fallback target produced this usage. Always set by VernLLM; optional for hand built
222
+ * values.
223
+ */
224
+ usedFallback?: boolean;
225
+ /** The call's `context`. */
226
+ context?: CallContext;
227
+ }
228
+ type OnUsage = (usage: TokenUsage) => void;
229
+ /**
230
+ * Called when a response carried usage but post-processing failed. Once per such attempt; never for
231
+ * transport failures, which have no usage to report.
232
+ */
233
+ type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
234
+ //#endregion
235
+ //#region src/types/events.d.ts
236
+ /**
237
+ * Reports what happened during a call. Fire and forget, mirroring
238
+ * `onUsage`: the return value is never read and a throwing handler cannot
239
+ * change what the call does, only what gets reported about it.
240
+ */
241
+ type VernLLMEvent = {
242
+ kind: 'retry';
243
+ requestId: string;
244
+ provider: string;
245
+ /** The model actually resolved for this call (honors a per-call `model` override). */
246
+ model: string;
247
+ /** The 1-based retry ordinal (the 1st retry is `1`, not the overall attempt count). */
248
+ attempt: number;
249
+ maxRetries: number;
250
+ delayMs: number;
251
+ retryAfterHonored: boolean;
252
+ error: LLMError;
253
+ context?: CallContext;
254
+ } | {
255
+ kind: 'circuit_state';
256
+ provider: string;
257
+ /**
258
+ * The model of the call that triggered this transition. Without `isolateByModel` the count
259
+ * spans every model, so the threshold may have been reached by several.
260
+ */
261
+ model: string;
262
+ from: CircuitState;
263
+ to: CircuitState;
264
+ consecutiveFailures: number;
265
+ context?: CallContext;
266
+ } | {
267
+ kind: 'fallback';
268
+ requestId: string;
269
+ /** Provider name of the target that just failed. */
270
+ from: string;
271
+ /** Provider name of the target about to be tried next. */
272
+ to: string;
273
+ /** `-1` for the primary target, otherwise the index into `fallback`. */
274
+ fromIndex: number;
275
+ toIndex: number;
276
+ /** The normalized error that caused `from` to be abandoned. */
277
+ error: LLMError;
278
+ /** Time spent on `from`, including its own retries, before giving up. */
279
+ elapsedMs: number;
280
+ context?: CallContext;
281
+ } | {
282
+ kind: 'rate_limited';
283
+ requestId: string;
284
+ provider: string;
285
+ /** The model actually resolved for this call (honors a per-call `model` override). */
286
+ model: string;
287
+ /** How long this attempt sat queued for capacity before it was let through. */
288
+ waitedMs: number;
289
+ /** Which configured bucket was blocking this attempt just before it cleared. */
290
+ reason: 'concurrency' | 'rpm' | 'tpm';
291
+ context?: CallContext;
292
+ } | {
293
+ kind: 'middleware';
294
+ requestId: string;
295
+ /** This middleware's `name`, or its array position if unnamed. */
296
+ middleware: string;
297
+ hook: 'transform' | 'wrap_short_circuit' | 'enabled_skip';
298
+ /** For `hook: 'transform'` only: which top-level fields the merged patch touched. */
299
+ patchedFields?: string[];
300
+ context?: CallContext;
301
+ } | {
302
+ /**
303
+ * Reported once a call fully succeeds. Same data `VernLLMOptions.onUsage`
304
+ * receives; that option is sugar over this event, not a second
305
+ * reporting path, see `makeEventReporter`.
306
+ */
307
+ kind: 'usage';
308
+ requestId: string;
309
+ usage: TokenUsage;
310
+ context?: CallContext;
311
+ } | {
312
+ /**
313
+ * A response carried usage and post-processing then failed. Once per such attempt;
314
+ * `onUsageFailure` is driven from this event.
315
+ */
316
+ kind: 'usage_failure';
317
+ requestId: string;
318
+ usage: TokenUsage;
319
+ error: LLMError;
320
+ context?: CallContext;
321
+ } | {
322
+ /** Reported by a middleware through `ctx.emit`. */
323
+ kind: 'custom';
324
+ requestId: string;
325
+ /** Namespaced by convention, e.g. `router.decision`. */
326
+ name: string;
327
+ /** Label of the emitting middleware. */
328
+ source: string;
329
+ data?: JsonValue;
330
+ context?: CallContext;
331
+ };
332
+ type OnEvent = (event: VernLLMEvent) => void;
333
+ //#endregion
334
+ //#region src/types/middleware.d.ts
335
+ /** Capabilities of the target a middleware hook is currently looking at. */
336
+ interface MiddlewareCapabilities {
337
+ /**
338
+ * Whether this target honors `response_format: { type: 'json_object' }`
339
+ * as a real constraint. Mirrors `LLMClient.supportsJsonObjectMode`.
340
+ * `false` for `fromAnthropic` and `fromBedrock`.
341
+ */
342
+ supportsJsonObjectMode: boolean;
343
+ }
344
+ /**
345
+ * Brands `MiddlewareStateKey` so a ref, or a hand written `{ debugName }`, can't pass as a state
346
+ * key. Only `createStateKey` can produce one.
347
+ */
348
+ declare const stateKeyBrand: unique symbol;
349
+ /**
350
+ * A typed reference to one slot in `ctx.state`. Create it with `createStateKey` and import it
351
+ * wherever the value is shared, so a typo is a compile error instead of a new property.
352
+ */
353
+ interface MiddlewareStateKey<T> {
354
+ readonly debugName: string;
355
+ readonly [stateKeyBrand]: true;
356
+ /**
357
+ * Never set. Makes keys of different `T` distinct types, so `get` and `set` infer the value type.
358
+ */
359
+ readonly __phantom?: T;
360
+ }
361
+ /** Creates a new, distinct `MiddlewareStateKey`. `debugName` is used only in log lines and the `'middleware'` event; it never affects equality. */
362
+ declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
363
+ /** Not exported. See `stateKeyBrand`; same reasoning, distinct symbol, so the two token types can't be cross-assigned either. */
364
+ declare const middlewareRefBrand: unique symbol;
365
+ /**
366
+ * A typed reference to a middleware, used only as a `runsAfter` or `runsBefore` target, never as a
367
+ * label. Create it with `createMiddlewareRef` and export it, so a typo is a compile error.
368
+ */
369
+ interface MiddlewareRef {
370
+ readonly debugName: string;
371
+ readonly [middlewareRefBrand]: true;
372
+ }
373
+ /** Creates a new, distinct `MiddlewareRef`. `debugName` is used only in error messages when a reference doesn't resolve; it never affects equality, so two refs with the same `debugName` never collide. */
374
+ declare function createMiddlewareRef(debugName: string): MiddlewareRef;
375
+ /**
376
+ * A `runsAfter` or `runsBefore` entry, made with `requireRef`, that throws at construction when the
377
+ * target isn't registered. A bare ref is optional and only warns.
378
+ */
379
+ interface RequiredMiddlewareRef {
380
+ readonly ref: MiddlewareRef;
381
+ }
382
+ /** Wraps `ref` so `runsAfter`/`runsBefore` throws at `VernLLM` construction time if it doesn't resolve, instead of warning and continuing. */
383
+ declare function requireRef(ref: MiddlewareRef): RequiredMiddlewareRef;
384
+ /**
385
+ * Typed per call storage that middleware share values through. VernLLM never reads or writes it.
386
+ */
387
+ interface MiddlewareStateBag {
388
+ get<T>(key: MiddlewareStateKey<T>): T | undefined;
389
+ set<T>(key: MiddlewareStateKey<T>, value: T): void;
390
+ }
391
+ /**
392
+ * One `[key, value]` pair for a call's `state`. A tuple can't tie each value to its own key's type,
393
+ * so `stateEntry` carries that check. Raw pairs are not type checked.
394
+ */
395
+ type MiddlewareStateEntry = readonly [MiddlewareStateKey<unknown>, unknown];
396
+ /** Builds a type checked `state` entry: `value` must match the key's type. */
397
+ declare function stateEntry<T>(key: MiddlewareStateKey<T>, value: T): MiddlewareStateEntry;
398
+ /** A plain, `Map`-backed `MiddlewareStateBag`, optionally seeded with `entries`. Later entries win. */
399
+ declare function createMiddlewareStateBag(entries?: readonly MiddlewareStateEntry[]): MiddlewareStateBag;
400
+ /** Fields every `MiddlewareContext` variant carries, regardless of `stage`. */
401
+ interface MiddlewareContextBase {
402
+ requestId: string;
403
+ /** Capabilities of the target this stage's identity fields describe. */
404
+ capabilities: MiddlewareCapabilities;
405
+ signal?: AbortSignal;
406
+ /** Shared, collision-proof state for two middleware to deliberately coordinate through. See `MiddlewareStateBag`. */
407
+ state: MiddlewareStateBag;
408
+ /** Simple, string-keyed scratch space, pre-namespaced to this one middleware so two middleware can never collide here even by accident. */
409
+ own: Record<string, unknown>;
410
+ /**
411
+ * Every registered middleware's label, in `transform` order, e.g. to skip work another known
412
+ * middleware already does.
413
+ */
414
+ registeredMiddlewareNames: readonly string[];
415
+ /**
416
+ * The labels from `registeredMiddlewareNames` whose entry defines a
417
+ * `transform`, in the same order, frozen. Lets a middleware tell
418
+ * whether any entry after it can still change the request.
419
+ */
420
+ transformMiddlewareNames: readonly string[];
421
+ /** Reports a `custom` event to `onEvent` and every middleware. Never throws. */
422
+ emit(name: string, data?: JsonValue): void;
423
+ /** The call's `context`, frozen. `undefined` when none was given. */
424
+ context: CallContext | undefined;
425
+ }
426
+ /**
427
+ * The `ctx` for `transform` and every attempt scoped event. Built once a target is selected, so
428
+ * every field describes the real target.
429
+ */
430
+ interface AttemptContext extends MiddlewareContextBase {
431
+ stage: 'attempt';
432
+ /** The target this attempt is actually dispatched to. */
433
+ requestedProvider: string;
434
+ /** The adapter behind this target. `{ name: 'custom' }` when its client doesn't identify one. */
435
+ adapter: AdapterInfo;
436
+ requestedModel: string;
437
+ isFallbackAttempt: boolean;
438
+ /**
439
+ * The current attempt number. A `'circuit_state'` event from the check before any attempt reports
440
+ * 1.
441
+ */
442
+ attempt: number;
443
+ }
444
+ /**
445
+ * The `ctx` for `wrap` and `onError`. Built before any target is chosen, so it only describes the
446
+ * primary. Read `next()`'s `CallResult.meta` for what happened.
447
+ */
448
+ interface PreDispatchContext extends MiddlewareContextBase {
449
+ stage: 'pre-dispatch';
450
+ /** The primary target only, not necessarily who ends up answering. */
451
+ primaryProvider: string;
452
+ /** The adapter behind the primary target. `{ name: 'custom' }` when its client doesn't identify one. */
453
+ primaryAdapter: AdapterInfo;
454
+ primaryModel: string;
455
+ /** The order so far: what this `wrap` may try, after every outer `wrap` narrowed it. */
456
+ targets: readonly TargetInfo[];
457
+ }
458
+ /**
459
+ * What `enabled` and `onEvent` receive, since they run in both stages. Narrow on `ctx.stage` before
460
+ * reading stage specific fields.
461
+ */
462
+ type MiddlewareContext = AttemptContext | PreDispatchContext;
463
+ /** The `response_format` shape `RequestBuilder` can put on the wire. */
464
+ type WireResponseFormat = {
465
+ type: 'json_object';
466
+ } | {
467
+ type: 'json_schema';
468
+ json_schema: {
469
+ name: string;
470
+ schema: Record<string, unknown>;
471
+ strict?: boolean;
472
+ description?: string;
473
+ };
474
+ };
475
+ /** A tool as it appears on the wire, OpenAI's `function`-wrapped shape. */
476
+ interface WireTool {
477
+ type: 'function';
478
+ function: {
479
+ name: string;
480
+ description: string;
481
+ parameters: Record<string, unknown>;
482
+ };
483
+ }
484
+ /**
485
+ * The wire-shaped request `RequestBuilder.build()` produces for one call
486
+ * attempt, before dispatch. Read only inside `transform`; return a patch
487
+ * of the fields you want to change instead of the whole object.
488
+ */
489
+ interface WireCallRequest {
490
+ model: string;
491
+ temperature?: number;
492
+ max_tokens: number;
493
+ response_format?: WireResponseFormat;
494
+ reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
495
+ budget_tokens?: number;
496
+ tools?: WireTool[];
497
+ tool_choice?: WireToolChoice;
498
+ messages: WireMessage[];
499
+ }
500
+ /**
501
+ * A patch `transform` returns, merged onto the built request. `model` and `response_format` can't
502
+ * be patched, since targets are attributed by the values already resolved. `add*` fields append, so
503
+ * two middleware can add without overwriting each other.
504
+ */
505
+ interface WireCallRequestPatch {
506
+ temperature?: number;
507
+ max_tokens?: number;
508
+ reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
509
+ budget_tokens?: number;
510
+ tool_choice?: WireToolChoice;
511
+ /** Replaces the whole message list. Prefer `addMessages` unless a full replace is genuinely the intent. */
512
+ messages?: WireMessage[];
513
+ /** Appended after whatever earlier middleware already added. Never clobbers a prior addition. */
514
+ addMessages?: WireMessage[];
515
+ /** Replaces the whole tool list. Prefer `addTools`, same reasoning as `messages`/`addMessages`. */
516
+ tools?: WireTool[];
517
+ /** Appended after whatever earlier middleware already added. Never clobbers a prior addition. */
518
+ addTools?: WireTool[];
519
+ }
520
+ /**
521
+ * The settled outcome of one logical call, from `wrap`'s `next()`. `meta` is `undefined` only on a
522
+ * cache hit.
523
+ */
524
+ interface CallResult<T = unknown> {
525
+ value: T;
526
+ meta?: CallMeta;
527
+ }
528
+ /**
529
+ * One entry in `VernLLMOptions.middleware`. Every hook is optional. See the middleware docs for how
530
+ * hooks compose.
531
+ */
532
+ interface VernLLMMiddleware {
533
+ /** Used in log lines and the `'middleware'` event. Defaults to this entry's array position when omitted. */
534
+ name?: string;
535
+ /**
536
+ * This entry's identity for other entries' `runsAfter` and `runsBefore`. Unrelated to `name`,
537
+ * which is only a label.
538
+ */
539
+ ref?: MiddlewareRef;
540
+ /** Sort key for composition order, ascending, ties broken by array order. See the middleware docs for what "lower runs first" means for `wrap`. */
541
+ priority?: number;
542
+ /**
543
+ * Middleware this entry runs after. An unresolved bare ref is dropped; wrap it with `requireRef`
544
+ * to throw at construction instead. A cycle always throws.
545
+ */
546
+ runsAfter?: (MiddlewareRef | RequiredMiddlewareRef)[];
547
+ /**
548
+ * Other middleware this entry must run before. See `runsAfter`; a
549
+ * bare reference is dropped if unresolved, a `requireRef`-wrapped one
550
+ * throws.
551
+ */
552
+ runsBefore?: (MiddlewareRef | RequiredMiddlewareRef)[];
553
+ /**
554
+ * Pins this entry in `wrap` nesting only. `'outermost'` sees the net result of every retry and
555
+ * fallback; `'innermost'` sits next to dispatch. A number works like `priority` for `wrap` only.
556
+ */
557
+ position?: 'outermost' | 'innermost' | number;
558
+ /**
559
+ * Boolean for a static on/off switch, or a predicate evaluated per
560
+ * call. A throwing, rejecting, or timed-out predicate is logged and
561
+ * treated as `false` for that call.
562
+ */
563
+ enabled?: boolean | ((ctx: MiddlewareContext) => boolean | Promise<boolean>);
564
+ /** Per-middleware override of the instance-level `middlewareTimeoutMs`, applied to this entry's `transform` and function `enabled`. `<= 0` means unbounded (no timer at all). */
565
+ timeoutMs?: number;
566
+ /** Transforms the outgoing wire request for one attempt. Runs once per attempt, including retries. `ctx` is always accurate to the real target for this attempt. */
567
+ transform?: (request: Readonly<WireCallRequest>, ctx: AttemptContext) => WireCallRequestPatch | Promise<WireCallRequestPatch>;
568
+ /**
569
+ * Wraps one whole logical call once, however many retries or targets ran. `ctx` describes the
570
+ * primary and the order so far (`ctx.targets`); read `next()`'s `CallResult.meta` for what
571
+ * happened. `next({ targets })` reorders or drops targets by name, never adds one.
572
+ */
573
+ wrap?: (request: Readonly<WireCallRequest>, next: (options?: {
574
+ targets?: readonly string[];
575
+ }) => Promise<CallResult>, ctx: PreDispatchContext) => Promise<CallResult>;
576
+ /**
577
+ * Wraps one attempt's provider request, after the limiter and every `transform`. `request` is
578
+ * exactly what the adapter gets. Runs per attempt, nested in `wrap` order.
579
+ *
580
+ * `next()` resolves when the response arrives, or at a stream's first content chunk; pings don't
581
+ * count. It rejects with the attempt's `LLMError`. A hook can observe but not change the outcome:
582
+ * returning or throwing without calling `next()` fails the attempt with code `middleware_threw`,
583
+ * and a throw after it is only logged.
584
+ */
585
+ dispatch?: (request: Readonly<WireCallRequest>, next: () => Promise<void>, ctx: AttemptContext) => Promise<void>;
586
+ /** Observes the same events reported on `VernLLMOptions.onEvent`, filtered by this middleware's own `enabled`. Called from both stages; narrow on `ctx.stage` before reading stage-specific fields. */
587
+ onEvent?: (event: VernLLMEvent, ctx: MiddlewareContext) => void;
588
+ }
589
+ //#endregion
590
+ //#region src/circuitBreaker.d.ts
591
+ /** The call this mutation happened as part of, forwarded to `onStateChange` untouched. */
592
+ interface CircuitBreakerCallContext {
593
+ requestId: string;
594
+ state: MiddlewareStateBag;
595
+ signal?: AbortSignal;
596
+ /** Omitted for calls before any attempt exists, like `assertClosed`'s pre-dispatch check. */
597
+ attempt?: number;
598
+ }
599
+ /**
600
+ * Fires after every real state change. `model` is the triggering call's resolved model. Shared by
601
+ * the options and the adapter interface, so custom adapters report changes the same way.
602
+ */
603
+ type CircuitBreakerStateChangeHandler = (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
604
+ interface CircuitBreakerOptions {
605
+ /** Consecutive failures before the circuit opens, default 5 */
606
+ threshold?: number;
607
+ /** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
608
+ cooldownMs?: number;
609
+ onStateChange?: CircuitBreakerStateChangeHandler;
610
+ /**
611
+ * Track a separate circuit per resolved model instead of one shared
612
+ * circuit. Default false. A call that omits `model` falls into one
613
+ * shared bucket alongside every other call that also omits it.
614
+ */
615
+ isolateByModel?: boolean;
616
+ /** Trial calls allowed through per half-open cycle. Default 1, clamped to at least 1. */
617
+ halfOpenProbes?: number;
618
+ /** Fraction of `halfOpenProbes` that must succeed to close the circuit. Default 1, clamped to `[0, 1]`. */
619
+ halfOpenSuccessRatio?: number;
620
+ /**
621
+ * Grows `cooldownMs` on each repeat open instead of a fixed wait.
622
+ * `{ multiplier, maxMs }` covers exponential growth; a `CooldownBackoff`
623
+ * function covers anything else. Omitted means `cooldownMs` stays fixed.
624
+ */
625
+ cooldownBackoff?: ExponentialBackoffOptions | CooldownBackoff;
626
+ /**
627
+ * When failures open the circuit. `{ kind: 'consecutive', threshold }` (default) opens after that
628
+ * many in a row; `{ kind: 'rolling', windowMs, minCalls, failureRatio }` opens once `minCalls`
629
+ * calls in the window reach `failureRatio`. Invalid values throw `RangeError` at construction. A
630
+ * custom `TrippingPolicy` is shared across models, keyed per model.
631
+ */
632
+ tripping?: TrippingOption;
633
+ }
634
+ /** Computes the cooldown for a bucket's `reopenCount`-th repeat open. */
635
+ type CooldownBackoff = (reopenCount: number, baseCooldownMs: number) => number;
636
+ interface ExponentialBackoffOptions {
637
+ /** Growth factor applied per repeat open, e.g. 2 doubles each time. */
638
+ multiplier: number;
639
+ /** Upper bound on the computed cooldown, in ms. Default `Infinity`. */
640
+ maxMs?: number;
641
+ }
642
+ /**
643
+ * Decides when failures open the circuit. Keyed by model (or one shared key), so one instance
644
+ * serves every bucket.
645
+ */
646
+ interface TrippingPolicy {
647
+ onSuccess(key: string): void;
648
+ /** Returns true if this failure should open the circuit for `key`. */
649
+ onFailure(key: string): boolean;
650
+ reset(key: string): void;
651
+ /**
652
+ * Called when `key`'s circuit is reset closed, so a keyed policy can drop that key's state. Not
653
+ * called on an ordinary success, whose state a rolling window still needs.
654
+ */
655
+ forget?(key: string): void;
656
+ }
657
+ declare class ConsecutiveTripping implements TrippingPolicy {
658
+ private readonly threshold;
659
+ private failuresByKey;
660
+ constructor(threshold: number);
661
+ onSuccess(key: string): void;
662
+ onFailure(key: string): boolean;
663
+ reset(key: string): void;
664
+ forget(key: string): void;
665
+ }
666
+ declare class RollingTripping implements TrippingPolicy {
667
+ private readonly windowMs;
668
+ private readonly minCalls;
669
+ private readonly failureRatio;
670
+ private ratiosByKey;
671
+ constructor(windowMs: number, minCalls: number, failureRatio: number);
672
+ private ratioFor;
673
+ onSuccess(key: string): void;
674
+ onFailure(key: string): boolean;
675
+ reset(key: string): void;
676
+ forget(key: string): void;
677
+ }
678
+ /** Not exported. Internal shorthand union for `CircuitBreakerOptions.tripping`. */
679
+ type TrippingOption = {
680
+ kind: 'consecutive';
681
+ threshold: number;
682
+ } | {
683
+ kind: 'rolling';
684
+ windowMs: number;
685
+ minCalls: number;
686
+ failureRatio: number;
687
+ } | TrippingPolicy;
688
+ type CircuitState = 'closed' | 'open' | 'half-open';
689
+ /**
690
+ * What VernLLM needs from a breaker. Pass your own for cross-process coordination. `onStateChange`
691
+ * is required so `circuit_state` events can't go missing; `() => {}` is fine. Omitting an optional
692
+ * member makes that call a no-op, as with no breaker. `open` and `close` are optional, since a
693
+ * distributed adapter may not allow forced transitions.
694
+ */
695
+ interface CircuitBreakerAdapter {
696
+ /** Throws when the circuit is open (or half open with no trial slot free) for `model`. */
697
+ assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
698
+ recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
699
+ /** `code`, when present, is the failing call's `LLMErrorCode`. */
700
+ recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
701
+ getState?(model?: string): CircuitState;
702
+ /** Failure counts by `LLMErrorCode` for `model`'s bucket, `'unknown'` for one that carried no code. */
703
+ getFailureBreakdown?(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
704
+ /** Whether this adapter tracks failures per model, mirroring `CircuitBreakerOptions.isolateByModel`. Read by `warnIfModelUnsupported`'s diagnostic warning and by `VernLLM.getCircuitStates()`'s public output; omit if the notion doesn't apply to your adapter, `false` is assumed. */
705
+ isolateByModel?: boolean;
706
+ /** Manually opens the circuit, as if enough consecutive failures had just happened. Optional: an adapter that doesn't want external callers forcing a transition can omit it. */
707
+ open?(model?: string, context?: CircuitBreakerCallContext): void;
708
+ /** Manually closes the circuit, without requiring a real success first. Same opt-in reasoning as `open`. */
709
+ close?(model?: string, context?: CircuitBreakerCallContext): void;
710
+ /** Gives back a half-open trial slot when a call ends without `recordSuccess` or `recordFailure`. Idempotent, and a no-op for a call that holds no slot. */
711
+ releaseTrial?(model?: string, context?: CircuitBreakerCallContext): void;
712
+ /** Awaited right before `assertClosed` to refresh local state. Never blocks or fails a call: a rejection or `prepareTimeoutMs` is logged and the call carries on. */
713
+ prepare?(model?: string, context?: CircuitBreakerCallContext): Promise<void>;
714
+ /** How long to wait for `prepare`, in ms. Default 1000. */
715
+ prepareTimeoutMs?: number;
716
+ /** Live counterpart of `getState`, read by `VernLLM.readCircuitStates()`. */
717
+ readState?(model?: string): Promise<CircuitState>;
718
+ /** Receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
719
+ setLogger?(logger: Logger): void;
720
+ /**
721
+ * Called after every real state change, after VernLLM reports its `circuit_state` event. A throw
722
+ * here is caught and logged.
723
+ */
724
+ onStateChange: CircuitBreakerStateChangeHandler;
725
+ }
726
+ /**
727
+ * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures
728
+ * across calls. Once the threshold is hit, short-circuits new calls with
729
+ * LLMError('circuit_open') until the cooldown elapses and a trial succeeds.
730
+ */
731
+ declare class CircuitBreaker implements CircuitBreakerAdapter {
732
+ private readonly cooldownMs;
733
+ /** Satisfies `CircuitBreakerAdapter.onStateChange`, required there. Defaults to a no-op when `options.onStateChange` is omitted. */
734
+ readonly onStateChange: CircuitBreakerStateChangeHandler;
735
+ /** Whether this breaker tracks failures per model instead of one shared circuit. */
736
+ readonly isolateByModel: boolean;
737
+ private readonly halfOpenProbes;
738
+ private readonly halfOpenSuccessRatio;
739
+ private readonly cooldownBackoff?;
740
+ /** One instance, keyed per model internally. See `TrippingPolicy`. */
741
+ private readonly tripping;
742
+ private readonly sharedBucket;
743
+ private readonly bucketsByModel;
744
+ /**
745
+ * Breaker-wide rather than per bucket, since an evicted bucket is
746
+ * recreated with no history and a per bucket generation would restart.
747
+ */
748
+ private generation;
749
+ /** Each call's generation at admission, keyed by its `state`, so its outcome can be told apart from a later generation's. */
750
+ private readonly admittedInGeneration;
751
+ constructor(options?: CircuitBreakerOptions);
752
+ /**
753
+ * Throws if the circuit is open and the cooldown hasn't elapsed, or if
754
+ * half-open with every trial slot claimed. Otherwise claims a trial slot.
755
+ */
756
+ assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
757
+ recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
758
+ /** `code`, when present, is the failing `LLMError`'s `code`. Missing attributes to `'unknown'`. */
759
+ recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
760
+ /**
761
+ * Gives back the trial slot `context`'s call claimed when it ended without an outcome. Safe to
762
+ * call on any failure path: a no-op without a live permit.
763
+ */
764
+ releaseTrial(model?: string, context?: CircuitBreakerCallContext): void;
765
+ /**
766
+ * The current state. Ignores `model` unless isolated by model. An open circuit past its cooldown
767
+ * reports `'half-open'`, though the real transition waits for the next call.
768
+ */
769
+ getState(model?: string): CircuitState;
770
+ /** Failure counts by `LLMErrorCode` for `model`'s bucket. Returned as a plain object copy. */
771
+ getFailureBreakdown(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
772
+ /** Manually opens the circuit, as if `threshold` consecutive failures had just happened. */
773
+ open(model?: string, context?: CircuitBreakerCallContext): void;
774
+ /** Manually closes the circuit and resets its failure count, without requiring a real success first. */
775
+ close(model?: string, context?: CircuitBreakerCallContext): void;
776
+ /**
777
+ * Stamps the open time and cooldown, then transitions to open. Callers have already cleared
778
+ * `trial`.
779
+ */
780
+ private openBucket;
781
+ /** Computes and clamps the cooldown for `bucket`'s current `reopenCount`. Called once, on open. */
782
+ private computeCooldown;
783
+ private markAdmitted;
784
+ /**
785
+ * True if this call was admitted before `bucket` last opened, so its outcome belongs to an ended
786
+ * generation. A call without context always counts.
787
+ */
788
+ private isFromEarlierGeneration;
789
+ /** Returns the bucket for a model if one already exists, without allocating. */
790
+ private lookupBucket;
791
+ /**
792
+ * The key for `tripping`: per model when isolated by model, otherwise one shared key, matching
793
+ * which bucket a call lands in.
794
+ */
795
+ private trippingKeyFor;
796
+ /** Creates and stores a bucket for a model when the first mutation needs one. */
797
+ private ensureBucketFor;
798
+ /** Drops an idle model's bucket, keeping its tripping state. */
799
+ private dropBucket;
800
+ /** Drops a reset model's bucket and lets `tripping` release that key's state too. */
801
+ private forgetModel;
802
+ /** Every state mutation routes through here, so `onStateChange` fires exactly once per real change. */
803
+ private transition;
804
+ /** Once every admitted trial has reported in, closes or reopens based on `halfOpenSuccessRatio`. */
805
+ private settleTrialIfComplete;
806
+ }
807
+ //#endregion
808
+ //#region src/internal/retryBudget.d.ts
809
+ /**
810
+ * `windowMs` and `minCalls` work as in `RollingTripping`: `minCalls` keeps a cold start from
811
+ * tripping. `retryRatio` is the largest share of calls in the window that may be retries. Invalid
812
+ * values throw `RangeError` at construction.
813
+ */
814
+ interface RetryBudgetOptions {
815
+ windowMs: number;
816
+ minCalls: number;
817
+ retryRatio: number;
818
+ }
819
+ /**
820
+ * Caps the share of a target's recent traffic that may be retries. The breaker asks whether the
821
+ * provider is healthy; this asks whether retrying is still worth the capacity it costs.
822
+ */
823
+ declare class RetryBudget {
824
+ private readonly options;
825
+ private readonly ratio;
826
+ constructor(options: RetryBudgetOptions);
827
+ /**
828
+ * Throws `LLMError('retry_budget_exhausted')` once at least `minCalls`
829
+ * calls have landed in the trailing `windowMs` and the retry ratio
830
+ * among them has reached `retryRatio`. A no-op otherwise.
831
+ */
832
+ assertAvailable(): void;
833
+ /** Records one attempt. `isRetry` is false for a call's first attempt, true for every attempt after it. */
834
+ recordAttempt(isRetry: boolean): void;
835
+ /** Current traffic and retry ratio in the trailing window. */
836
+ getSnapshot(): {
837
+ attempts: number;
838
+ retryRatio: number;
839
+ };
840
+ }
841
+ //#endregion
842
+ //#region src/internal/utils/circuit-breaker/circuitBreakerAdapter.utils.d.ts
843
+ /** The `circuitBreaker` option union, shared by `VernLLMOptions` and `buildCircuitBreaker`. */
844
+ type CircuitBreakerOption = boolean | CircuitBreakerOptions | CircuitBreakerAdapter;
845
+ //#endregion
846
+ //#region src/internal/utils/rate-limit/rateLimitHint.utils.d.ts
847
+ /** A normalized read of a provider's rate limit headers. */
848
+ interface ProviderRateLimitHint {
849
+ remainingRequests?: number;
850
+ limitRequests?: number;
851
+ resetAfterMs?: number;
852
+ }
853
+ //#endregion
854
+ //#region src/rateLimit.d.ts
855
+ /** The request shape sent to `LLMClient['chat']['completions']['create']`, used for token estimation. */
856
+ type WireRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
857
+ /** Which configured bucket is currently blocking a call. */
858
+ type RateLimitReason = 'concurrency' | 'rpm' | 'tpm';
859
+ interface RateLimitOptions {
860
+ /** Max requests per minute. Omit or pass 0 for unlimited. Otherwise a finite number of at least 1. */
861
+ requestsPerMinute?: number;
862
+ /**
863
+ * Max tokens per minute. Enforced against a pre-flight estimate, then
864
+ * reconciled against reported usage once the call completes. Omit or
865
+ * pass 0 for unlimited. Otherwise a finite number of at least 1.
866
+ */
867
+ tokensPerMinute?: number;
868
+ /** Max requests in flight at once, as a non-negative integer. Default 0, meaning unlimited. */
869
+ maxConcurrent?: number;
870
+ /**
871
+ * Max time a call may sit queued waiting for capacity, in ms. Exceeding
872
+ * it throws rather than hanging forever. Default 30000. Pass 0 to wait
873
+ * indefinitely. At most 2147483647, the longest timer delay.
874
+ */
875
+ maxQueueMs?: number;
876
+ /** Max queued calls before new ones reject immediately instead of queueing, as a non-negative integer. Default 0, unbounded. */
877
+ maxQueueSize?: number;
878
+ /**
879
+ * Pre-flight token estimate for `tokensPerMinute`. Defaults to a
880
+ * chars/4 heuristic over message text, a flat amount per image, plus
881
+ * `max_tokens`.
882
+ */
883
+ estimateTokens?: (request: WireRequest) => number;
884
+ /**
885
+ * Scales the token estimate reserved against `tokensPerMinute`, since most calls use less than
886
+ * `max_tokens`. Limiter bookkeeping only; `release` reconciles against real usage. Default 1,
887
+ * must be above 0, clamped to 1.
888
+ */
889
+ estimateFraction?: number;
890
+ /**
891
+ * AIMD against the `requestsPerMinute` bucket. Omit for a fixed
892
+ * ceiling, today's behavior. Requires `requestsPerMinute`; throws
893
+ * without it.
894
+ */
895
+ aimd?: AimdOptions;
896
+ }
897
+ interface AimdOptions {
898
+ /** Added to the requests-per-minute ceiling on every clean release. */
899
+ increaseBy: number;
900
+ /** Multiplied against the ceiling on a rate-limit signal. Must be greater than `0` and at most `1`; clamped otherwise. */
901
+ decreaseFactor: number;
902
+ /** Floor the ceiling never shrinks below. */
903
+ minCapacity: number;
904
+ /** Ceiling the bucket never grows above. */
905
+ maxCapacity: number;
906
+ /**
907
+ * Shrink proactively once a provider hint reports `remainingRequests`
908
+ * at or below this, before a real 429 happens. Default 0, meaning
909
+ * off.
910
+ */
911
+ proactiveFloor?: number;
912
+ }
913
+ interface RateLimitState {
914
+ /** Requests still available this window, or `undefined` if `requestsPerMinute` isn't configured. */
915
+ requestsRemaining?: number;
916
+ /** Tokens still available this window, or `undefined` if `tokensPerMinute` isn't configured. */
917
+ tokensRemaining?: number;
918
+ /** Concurrency slots currently in use, or `undefined` if `maxConcurrent` isn't configured. */
919
+ concurrentInFlight?: number;
920
+ }
921
+ interface RateLimitAcquireResult {
922
+ /**
923
+ * Frees the concurrency slot and reconciles tokens when `actualTokens` is given. Only the first
924
+ * call counts. Call it in a `finally` so a failed attempt never leaks a slot.
925
+ */
926
+ release: (actualTokens?: number, success?: boolean) => void;
927
+ /** How long this attempt waited in queue before capacity was available. */
928
+ waitedMs: number;
929
+ /** Which bucket was blocking this attempt just before it cleared, if any wait happened. */
930
+ reason?: RateLimitReason;
931
+ }
932
+ /**
933
+ * What VernLLM needs from a limiter. Pass your own for cross-process coordination. Every method is
934
+ * required; no-op the AIMD ones when unused, as `RateLimiter` does.
935
+ */
936
+ interface RateLimiterAdapter {
937
+ estimate(request: WireRequest): number;
938
+ acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
939
+ signalRateLimit(): void;
940
+ reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
941
+ /** Optional: current bucket levels, for introspection. Omit if the adapter has no state worth reporting. */
942
+ getState?(): RateLimitState;
943
+ /** Optional: live bucket levels for `VernLLM.readRateLimitState()`, the async counterpart of `getState`. */
944
+ readState?(): Promise<RateLimitState>;
945
+ /** Optional: receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
946
+ setLogger?(logger: Logger): void;
947
+ }
948
+ /**
949
+ * Per target limiter: up to three buckets (requests, tokens, concurrency) behind one FIFO queue, so
950
+ * a large call isn't starved by small ones. An omitted bucket never blocks.
951
+ */
952
+ declare class RateLimiter implements RateLimiterAdapter {
953
+ private readonly requests?;
954
+ private readonly tokens?;
955
+ private readonly concurrency?;
956
+ /** Buckets in acquire precedence order (concurrency, rpm, tpm), omitted ones filtered out. Built once so order can't drift between `tryAcquireBuckets` and `scheduleWake`. */
957
+ private readonly buckets;
958
+ private readonly maxQueueMs;
959
+ private readonly maxQueueSize;
960
+ private readonly estimateTokensFn;
961
+ private readonly estimateFraction;
962
+ private readonly aimd?;
963
+ private readonly queue;
964
+ /**
965
+ * A scheduled recheck of the queue head when it waits on a bucket that refills by time, so the
966
+ * queue doesn't wait for an unrelated acquire or release. A concurrency block only clears on
967
+ * release.
968
+ */
969
+ private wakeTimer?;
970
+ /** When AIMD last shrank the ceiling, see `AIMD_SHRINK_WINDOW_MS`. */
971
+ private lastShrinkAt?;
972
+ constructor(options: RateLimitOptions);
973
+ /**
974
+ * The token estimate reserved against `tokensPerMinute`, scaled by `estimateFraction`. The
975
+ * request's `max_tokens` is never changed.
976
+ */
977
+ estimate(request: WireRequest): number;
978
+ /**
979
+ * Waits for capacity in every configured bucket, then takes from each.
980
+ * The returned `release` gives the concurrency slot back and reconciles
981
+ * the token bucket against real usage; it must run in a `finally` block.
982
+ */
983
+ acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
984
+ private queueFullError;
985
+ private enqueue;
986
+ /** Takes from every configured bucket as one atomic unit, in `this.buckets`' order. Rolls back whatever was already taken if any bucket lacks capacity. */
987
+ private tryAcquireBuckets;
988
+ /** Drains the queue head first. Stops at the first waiter that still can't proceed, so no one is starved out of turn. */
989
+ private drain;
990
+ /**
991
+ * Schedules one recheck for when the bucket blocking the head should have capacity. No-op for a
992
+ * concurrency block or while a wake is pending.
993
+ */
994
+ private scheduleWake;
995
+ /**
996
+ * The one-shot release for an acquired slot. Concurrency is given back, requests only recover by
997
+ * refill, and tokens are reconciled against real usage. AIMD grows only when `success` is true,
998
+ * so a failed attempt can't undo a shrink.
999
+ */
1000
+ private makeRelease;
1001
+ /** Shared guard and resize call behind both AIMD halves below; only the arithmetic differs. */
1002
+ private resizeRequestsCeiling;
1003
+ /** AIMD's additive-increase half: grows the ceiling by `aimd.increaseBy` on a clean release. No-op without `aimd`/`requestsPerMinute`. */
1004
+ private growOnSuccess;
1005
+ /**
1006
+ * AIMD's decrease, on a real 429 or a low remaining hint. Only adjusts the ceiling, never throws.
1007
+ * Shrinks at most once per `AIMD_SHRINK_WINDOW_MS`.
1008
+ */
1009
+ signalRateLimit(): void;
1010
+ /**
1011
+ * AIMD's proactive entry point: shrinks via `signalRateLimit()` if
1012
+ * `hint.remainingRequests` is at or below `aimd.proactiveFloor`.
1013
+ */
1014
+ reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
1015
+ /**
1016
+ * Current bucket levels, read live rather than cached. `concurrency`
1017
+ * tracks free slots internally, so `concurrentInFlight` is reported as
1018
+ * `capacity - available`, the inverse of what the bucket itself holds.
1019
+ */
1020
+ getState(): RateLimitState;
1021
+ }
1022
+ //#endregion
1023
+ //#region src/internal/utils/rate-limit/rateLimitAdapter.utils.d.ts
1024
+ /** Not exported. Internal shorthand only, so this union isn't duplicated between the public option fields and `buildRateLimit`'s own signature. */
1025
+ type RateLimitOption = RateLimitOptions | RateLimiterAdapter;
1026
+ //#endregion
1027
+ //#region src/types/fallback.d.ts
1028
+ /**
1029
+ * A provider tried after the primary fails, in the order given. Omitted overrides fall back to the
1030
+ * instance's options, except `circuitBreaker`, `rateLimit` and `retryBudget`, which are never
1031
+ * inherited since limits tuned for the primary rarely fit a fallback.
1032
+ */
1033
+ interface FallbackTarget {
1034
+ client: LLMClient;
1035
+ model: string;
1036
+ /** Label for events, errors, and `TokenUsage.provider`. Default `` `fallback[${index}]` ``. */
1037
+ name?: string;
1038
+ maxRetries?: number;
1039
+ timeoutMs?: number;
1040
+ chunkIdleTimeoutMs?: number;
1041
+ readerStallTimeoutMs?: number;
1042
+ baseDelayMs?: number;
1043
+ maxRetryAfterMs?: number;
1044
+ defaultMaxTokens?: number;
1045
+ defaultTemperature?: number | null;
1046
+ defaultReasoningEffort?: 'minimal' | 'low' | 'medium' | 'high';
1047
+ defaultBudgetTokens?: number;
1048
+ nonRetryableStatus?: number[];
1049
+ /** This target's own breaker, or a shared `CircuitBreakerAdapter`. Not inherited from the parent's `circuitBreaker`. */
1050
+ circuitBreaker?: CircuitBreakerOption;
1051
+ /** This target's own rate limiter, independent of every other target's. Not inherited from the parent's `rateLimit`. */
1052
+ rateLimit?: RateLimitOption;
1053
+ /** This target's own retry budget, independent of every other target's. Not inherited from the parent's `retryBudget`. */
1054
+ retryBudget?: RetryBudgetOptions;
1055
+ /**
1056
+ * Reclassifies a successful result from this target as a failure. Inherits the instance's
1057
+ * `detectSoftFailure` when omitted.
1058
+ */
1059
+ detectSoftFailure?: DetectSoftFailure;
1060
+ }
1061
+ /**
1062
+ * Written into `CallParams['meta']` once `call()` resolves, so a caller
1063
+ * who wants provider identity on the same line as the result doesn't need
1064
+ * to read it back out of `onUsage`.
1065
+ */
1066
+ interface CallMeta {
1067
+ provider: string;
1068
+ model: string;
1069
+ /** `-1` if the primary target answered, otherwise the index into `fallback`. */
1070
+ fallbackIndex: number;
1071
+ usedFallback: boolean;
1072
+ /** Attempts made against the target that ultimately answered, including the successful one. */
1073
+ attempts: number;
1074
+ /** Position of the answering target in the order tried, 0 based. */
1075
+ position: number;
1076
+ }
1077
+ /** One configured target, as a middleware sees it in `ctx.targets`. */
1078
+ interface TargetInfo {
1079
+ /** The name `targets` selects by. */
1080
+ name: string;
1081
+ /** Declared position: 0 is the primary. */
1082
+ index: number;
1083
+ model: string;
1084
+ adapter: AdapterInfo;
1085
+ }
1086
+ /** One target's circuit state, as returned by `VernLLM.getCircuitStates()`. */
1087
+ interface TargetCircuitState {
1088
+ provider: string;
1089
+ /** Position in the chain: `0` for the primary, `1`+ for fallback targets. */
1090
+ index: number;
1091
+ isFallback: boolean;
1092
+ /** Whether this target tracks failures per model. `false` means `model` on `getCircuitStates` had no effect on this entry. */
1093
+ isolateByModel: boolean;
1094
+ /** `undefined` if that target has no circuit breaker configured. */
1095
+ state: CircuitState | undefined;
1096
+ }
1097
+ /** Which target/model `VernLLM.getCircuitState`, `openCircuit`, and `closeCircuit` act on. */
1098
+ interface CircuitTarget {
1099
+ /** Which target to act on. `0` is the primary, `1`+ are fallbacks. Defaults to `0`. */
1100
+ index?: number;
1101
+ /** Which model bucket to act on, if the resolved target isolates by model. */
1102
+ model?: string;
1103
+ }
1104
+ /**
1105
+ * One target's failure. `index` is `-1` for the primary; `provider` and `model` name the target.
1106
+ */
1107
+ interface FallbackAttempt extends RetryAttempt {
1108
+ provider: string;
1109
+ model: string;
1110
+ }
1111
+ /** What `fallbackOn` is told about the failed target and what would come after it. */
1112
+ interface FallbackOnContext {
1113
+ /** Whether no target is left in the order tried. The chain stops after the last one whatever `fallbackOn` returns. */
1114
+ isLastTarget: boolean;
1115
+ /** The target that just failed. `model` is what it ran, so the per call `model` shows on the primary only. */
1116
+ failed: TargetInfo;
1117
+ /** The target that would be tried next. `undefined` on the last target. */
1118
+ next?: TargetInfo;
1119
+ }
1120
+ /**
1121
+ * Whether to try the next target (`'next'`) or give up (`'stop'`) after a target's own retries.
1122
+ * Called once per failed target.
1123
+ */
1124
+ type FallbackOn = (error: LLMError, context: FallbackOnContext) => 'next' | 'stop';
1125
+ /**
1126
+ * The default `fallbackOn` policy. Exported so a caller can wrap rather
1127
+ * than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
1128
+ */
1129
+ declare const defaultFallbackOn: (error: LLMError, context: Pick<FallbackOnContext, 'isLastTarget'>) => 'next' | 'stop';
1130
+ /**
1131
+ * Thrown when the chain gives up, carrying every attempt in order. Extends `LLMError` and takes the
1132
+ * last failure's `type`, `status` and `retryAfterMs`, so existing handling keeps working.
1133
+ */
1134
+ declare class FallbackExhaustedError extends LLMError {
1135
+ readonly attempts: FallbackAttempt[];
1136
+ constructor(attempts: FallbackAttempt[]);
1137
+ /**
1138
+ * Defers to the last attempt's `retryable`, since `fallback_exhausted` alone says nothing about
1139
+ * it.
1140
+ */
1141
+ get retryable(): boolean;
1142
+ }
1143
+ /** Narrows `err` to {@link FallbackExhaustedError}, for direct access to its `attempts` (`provider`/`model` per failed target) without a manual `instanceof` check. */
1144
+ declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
1145
+ /**
1146
+ * An empty holder for `CallParams['meta']`, so `call()`'s `CallMeta` can be read without declaring
1147
+ * one by hand.
1148
+ *
1149
+ * @example
1150
+ * const meta = metaRef();
1151
+ * const result = await vern.call({ userContent: '...', meta });
1152
+ * meta.current?.provider;
1153
+ */
1154
+ declare function metaRef(): {
1155
+ current?: CallMeta;
1156
+ };
1157
+ //#endregion
1158
+ //#region src/types/schema.d.ts
1159
+ /**
1160
+ * Minimal structural type for a Zod-like schema, so this package doesnt need
1161
+ * a hard dependency on a specific Zod major version. Any object exposing
1162
+ * `safeParse` (Zod v3/v4, and most Zod-compatible validators) should satisfy this
1163
+ */
1164
+ interface SchemaLike<T> {
1165
+ safeParse(data: unknown): {
1166
+ success: true;
1167
+ data: T;
1168
+ } | {
1169
+ success: false;
1170
+ error: unknown;
1171
+ };
1172
+ }
1173
+ /**
1174
+ * A provider-native JSON Schema for structured outputs (OpenAI/Groq
1175
+ * `response_format: { type: 'json_schema' }`) This is the wire-format
1176
+ * schema the model is constrained to generate against, distinct from
1177
+ * `schema`, which is a client-side Zod validator run on the parsed result
1178
+ * You can use one, both, or neither; using both gets you provider-level
1179
+ * constraint plus client-side type inference/validation as a safety net
1180
+ */
1181
+ interface JsonSchemaSpec {
1182
+ name: string;
1183
+ schema: Record<string, unknown>;
1184
+ /** Enforces the schema strictly (OpenAI-specific), default true when supported */
1185
+ strict?: boolean;
1186
+ description?: string;
1187
+ }
1188
+ //#endregion
1189
+ //#region src/types/tools.d.ts
1190
+ /**
1191
+ * Describes a capability the model may request, not the capability
1192
+ * itself. VernLLM transports this to the provider and parses what comes
1193
+ * back; it never executes anything.
1194
+ */
1195
+ interface ToolDefinition<Name extends string = string, Args = unknown> {
1196
+ name: Name;
1197
+ description: string;
1198
+ /** JSON Schema for the tool's input. */
1199
+ parameters: Record<string, unknown>;
1200
+ /**
1201
+ * Validates the parsed `arguments`, using the same `safeParse` shape as `schema`. Failure throws
1202
+ * `LLMError('validation')`; on success `ToolCall.arguments` is the schema's output, defaults and
1203
+ * transforms applied. Without it, arguments are parsed as JSON only. Types flow into `ToolCall`
1204
+ * when the tool's `name` is literal, see `defineTool()`.
1205
+ */
1206
+ argumentsSchema?: SchemaLike<Args>;
1207
+ }
1208
+ /**
1209
+ * Keeps a tool's literal `name` and inferred `Args`, so `ToolCall` narrows by name. A plain object
1210
+ * literal widens `name` to `string` without `as const`, which breaks narrowing once a second tool
1211
+ * is added.
1212
+ */
1213
+ declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
1214
+ /** Maps a single `ToolDefinition` to its matching `ToolCall` shape. */
1215
+ type ToolCallFor<T> = T extends ToolDefinition<infer N, infer A> ? {
1216
+ id: string;
1217
+ name: N;
1218
+ arguments: A;
1219
+ } : never;
1220
+ /**
1221
+ * One tool call from the model. With a literal `Tools` tuple this is a union keyed by `name`, so
1222
+ * checking `name` narrows `arguments`. Otherwise `arguments` is `unknown`.
1223
+ */
1224
+ type ToolCall<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ToolCallFor<Tools[number]>;
1225
+ /** The application's result of executing a `ToolCall`, sent back to the model. */
1226
+ interface ToolResult {
1227
+ toolCallId: string;
1228
+ content: unknown;
1229
+ /**
1230
+ * Marks a failed tool execution. Sent as Anthropic's `is_error` and Bedrock's error status;
1231
+ * OpenAI-compatible adapters prefix the content with `Error: `; Gemini has no equivalent and
1232
+ * ignores it.
1233
+ */
1234
+ isError?: boolean;
1235
+ }
1236
+ /** `call()` result when `tools` was set and the model produced a normal answer. */
1237
+ interface ContentResult<T> {
1238
+ type: 'content';
1239
+ content: T;
1240
+ }
1241
+ /** `call()` result when `tools` was set and the model requested one or more tools. */
1242
+ interface ToolCallResult<Tools extends readonly ToolDefinition[] = ToolDefinition[]> {
1243
+ type: 'tool_calls';
1244
+ toolCalls: ToolCall<Tools>[];
1245
+ /** Any text the model produced alongside the tool request, if present. */
1246
+ content?: string;
1247
+ /**
1248
+ * Claude's reasoning blocks before this tool request, when thinking is
1249
+ * on (`fromAnthropic` and `fromBedrock`). Put them on the assistant
1250
+ * history turn with `toolCalls` so the tool loop can continue.
1251
+ */
1252
+ thinking?: ThinkingBlock[];
1253
+ }
1254
+ type CallWithToolsResult<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ContentResult<T> | ToolCallResult<Tools>;
1255
+ /** Recovers `Tools` from a `result` already typed `ContentResult<T> | ToolCallResult<Tools>`. Falls back to `ToolDefinition[]`. */
1256
+ type ExtractTools<R> = Extract<R, ToolCallResult<ToolDefinition[]>> extends ToolCallResult<infer Tools> ? Tools : ToolDefinition[];
1257
+ /** Explicit `Tools` type argument if given, otherwise inferred via `ExtractTools`. `never` marks "unset". */
1258
+ type ResolvedTools<Tools, R> = [Tools] extends [never] ? ExtractTools<R> : Tools;
1259
+ /**
1260
+ * Whether a `call()` result is a tool calls result. Use it whenever `tools` was set conditionally.
1261
+ * `toolCalls[number].arguments` is typed per tool; pass `Tools` explicitly to override inference,
1262
+ * e.g. `isToolCallResult<typeof tools>(result)`.
1263
+ */
1264
+ declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
1265
+ /** What the model should do about tools on a given call. */
1266
+ type ToolChoice = 'auto' | 'none' | 'required' | {
1267
+ name: string;
1268
+ };
1269
+ //#endregion
1270
+ //#region src/types/call.d.ts
1271
+ /**
1272
+ * Any valid JSON value: a primitive, `null`, or a JSON array/object made
1273
+ * of the same. This is what `call()` returns when `jsonMode: true`.
1274
+ */
1275
+ type JsonValue = string | number | boolean | null | JsonValue[] | {
1276
+ [key: string]: JsonValue;
1277
+ };
1278
+ /**
1279
+ * Plain JSON describing the request, e.g. tenant or routing constraints. Never sent to the
1280
+ * provider.
1281
+ */
1282
+ type CallContext = {
1283
+ readonly [key: string]: JsonValue;
1284
+ };
1285
+ /**
1286
+ * Content for an `assistant` turn in `history`. A parsed `JsonValue` from a `jsonMode` response can
1287
+ * go straight back in; it is stringified before sending.
1288
+ */
1289
+ type AssistantContent = string | JsonValue;
1290
+ /**
1291
+ * A reasoning block Claude produced before a tool call. Pass it back untouched on the assistant
1292
+ * turn that requested the tools: Claude with thinking on rejects the next call without it, and the
1293
+ * `signature` covers the text.
1294
+ */
1295
+ type ThinkingBlock = {
1296
+ type: 'thinking';
1297
+ thinking: string;
1298
+ signature: string;
1299
+ } | {
1300
+ type: 'redacted_thinking';
1301
+ data: string;
1302
+ };
1303
+ /**
1304
+ * One prior turn in `history`. A tool turn must directly follow the assistant turn that called the
1305
+ * tools, with a result for every call.
1306
+ */
1307
+ type ConversationTurn = {
1308
+ role: 'user';
1309
+ content: string;
1310
+ } | {
1311
+ role: 'assistant';
1312
+ content?: AssistantContent;
1313
+ toolCalls?: ToolCall[];
1314
+ /**
1315
+ * Reasoning blocks from this turn, as `ToolCallResult.thinking`
1316
+ * returned them. Sent ahead of the text and tool calls by
1317
+ * `fromAnthropic` and `fromBedrock`; other adapters drop them.
1318
+ */
1319
+ thinking?: ThinkingBlock[];
1320
+ } | {
1321
+ role: 'tool';
1322
+ toolResults: ToolResult[];
1323
+ };
1324
+ /** A plain text segment of a multimodal `userContent` array. */
1325
+ interface TextBlock {
1326
+ type: 'text';
1327
+ text: string;
1328
+ }
1329
+ /**
1330
+ * An inline image in a multimodal `userContent` array. `data` is raw base64 with no `data:` prefix;
1331
+ * each adapter converts it as its provider needs.
1332
+ */
1333
+ interface ImageBlock {
1334
+ type: 'image';
1335
+ /** Base64-encoded image bytes, no `data:` prefix */
1336
+ data: string;
1337
+ /** e.g. 'image/png', 'image/jpeg', 'image/webp', 'image/gif' */
1338
+ mimeType: string;
1339
+ }
1340
+ /** A single segment of multimodal `userContent`. */
1341
+ type ContentBlock = TextBlock | ImageBlock;
1342
+ /**
1343
+ * Every request field except the `UsageHooks`. The `Cached*` params use it, since `cachedCall`
1344
+ * meters usage once at the top level.
1345
+ */
1346
+ interface LLMRequestShape<T = unknown, Tools extends readonly ToolDefinition[] = ToolDefinition[]> {
1347
+ systemPrompt?: string;
1348
+ /** Current user message, as text or multimodal content blocks. */
1349
+ userContent: string | ContentBlock[];
1350
+ /**
1351
+ * Previous conversation turns. Must alternate roles; tool turns must follow
1352
+ * assistant tool calls. Invalid history throws LLMError('invalid_params').
1353
+ */
1354
+ history?: ConversationTurn[];
1355
+ /**
1356
+ * Generation temperature. Default 0.2, not the provider's own default.
1357
+ * Pass `null` to omit `temperature` from the request entirely, so the
1358
+ * provider applies its own default instead.
1359
+ */
1360
+ temperature?: number | null;
1361
+ jsonMode?: boolean;
1362
+ maxTokens?: number;
1363
+ requestId?: string;
1364
+ signal?: AbortSignal;
1365
+ /**
1366
+ * Total ms budget for the whole call, across retries and fallback targets, unlike the per attempt
1367
+ * `timeoutMs`. Covers getting to a result and opening a stream, not reading one; use
1368
+ * `chunkIdleTimeoutMs` for that. Omit or pass Infinity for none.
1369
+ */
1370
+ deadlineMs?: number;
1371
+ /**
1372
+ * Per call override for the max gap between stream chunks. Only applies with `stream: true`. Pass
1373
+ * 0 to disable.
1374
+ */
1375
+ chunkIdleTimeoutMs?: number;
1376
+ /**
1377
+ * Overrides the instance model for this call. Applies to the primary
1378
+ * target only, fallback targets always run their own configured model.
1379
+ */
1380
+ model?: string;
1381
+ /**
1382
+ * Reasoning effort for supported models. `null` skips `defaultReasoningEffort` for this call, as
1383
+ * `temperature: null` does; `undefined` uses the default.
1384
+ */
1385
+ reasoningEffort?: 'minimal' | 'low' | 'medium' | 'high' | null;
1386
+ /**
1387
+ * Reasoning token budget, for providers with a numeric budget. Providers with only effort tiers
1388
+ * get the nearest tier. With both set, each adapter uses the one it understands. `null` skips
1389
+ * `defaultBudgetTokens`.
1390
+ */
1391
+ budgetTokens?: number | null;
1392
+ /**
1393
+ * Provider-native JSON Schema output constraint. Implies jsonMode: true.
1394
+ */
1395
+ jsonSchema?: JsonSchemaSpec;
1396
+ /**
1397
+ * Validates parsed JSON output. Failure throws LLMError('validation').
1398
+ * Implies jsonMode: true.
1399
+ */
1400
+ schema?: SchemaLike<T>;
1401
+ /**
1402
+ * Tools the model may call. Makes `call()` return `CallWithToolsResult<T>`. A literal array also
1403
+ * types each tool's `arguments`.
1404
+ */
1405
+ tools?: Tools;
1406
+ /** Defaults to `'auto'` when `tools` is set. */
1407
+ toolChoice?: ToolChoice;
1408
+ /**
1409
+ * Streams the response. Requires an adapter with `createStream`. Retries and fallback cover the
1410
+ * stream until its first content; later failures reject `finalResult`. `finalResult` resolves to
1411
+ * the same shape a non-streaming call returns. See `StreamCallResult`.
1412
+ */
1413
+ stream?: boolean;
1414
+ /**
1415
+ * Out parameter written with the answering target's `CallMeta`, set before `call()` returns,
1416
+ * streams included. Left untouched when a `wrap` middleware short-circuits, so a reused holder
1417
+ * may keep an older value.
1418
+ */
1419
+ meta?: {
1420
+ current?: CallMeta;
1421
+ };
1422
+ /** See `CallContext`. Read by middleware, events and usage. */
1423
+ context?: CallContext;
1424
+ /** Values placed in `ctx.state` before any middleware runs. Build entries with `stateEntry`. */
1425
+ state?: readonly MiddlewareStateEntry[];
1426
+ /** Target names to try, in order. Default: every target, as declared. */
1427
+ targets?: readonly string[];
1428
+ }
1429
+ interface CallParams<T = unknown, Tools extends readonly ToolDefinition[] = ToolDefinition[]> extends LLMRequestShape<T, Tools>, UsageHooks {}
1430
+ /** `CallParams` with `tools` set, selecting the overload that returns `CallWithToolsResult<T>`. */
1431
+ type ToolEnabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1432
+ tools: NonNullable<CallParams<T, Tools>['tools']>;
1433
+ };
1434
+ /**
1435
+ * `CallParams` for conditionally set tools, e.g. `tools: flag ? [tool] : undefined`. Returns `T |
1436
+ * CallWithToolsResult<T, Tools>`, forcing an `isToolCallResult()` check. Typed `arguments` need an
1437
+ * explicit `Tools` argument on that check, since a ternary loses the literal tuple.
1438
+ */
1439
+ type ConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1440
+ tools: Tools | undefined;
1441
+ };
1442
+ /** Conditional tool-call parameters whose non-tool result is plain text. */
1443
+ type ConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ConditionalToolCallParams<string, Tools> & {
1444
+ jsonMode: false;
1445
+ };
1446
+ /**
1447
+ * `CallParams` with `toolChoice: 'none'`. The model can't call a tool, so `call()` returns
1448
+ * `ContentResult<T>` and no `isToolCallResult` check is needed.
1449
+ */
1450
+ type ToolsDisabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1451
+ tools: NonNullable<CallParams<T, Tools>['tools']>;
1452
+ toolChoice: 'none';
1453
+ };
1454
+ /**
1455
+ * `CallParams` with `jsonMode: false`, returning a `string`. `jsonSchema` is `never`, since a
1456
+ * schema forces JSON parsing regardless of `jsonMode`.
1457
+ */
1458
+ type JsonModeDisabledCallParams = Omit<CallParams<unknown>, 'jsonSchema'> & {
1459
+ jsonMode: false;
1460
+ jsonSchema?: never;
1461
+ };
1462
+ /**
1463
+ * `CallParams` with `jsonMode: true` and no `schema`, returning `JsonValue`. `schema` is `never`
1464
+ * rather than omitted, so a schema whose output happens to fit `JsonValue` still picks the schema
1465
+ * aware overload.
1466
+ */
1467
+ type JsonModeEnabledCallParams = Omit<CallParams<JsonValue>, 'schema'> & {
1468
+ jsonMode: true;
1469
+ schema?: never;
1470
+ };
1471
+ /** Shared cache-configuration fields, minus the internal `fn` primitive. */
1472
+ interface CachedCallInput extends UsageHooks {
1473
+ cacheKey: string;
1474
+ ttl: number;
1475
+ signal?: AbortSignal;
1476
+ }
1477
+ /**
1478
+ * A cached call without tools: cache config plus the `call` params. `reserveUsage` and
1479
+ * `refundUsage` go at the top level, not inside `call`.
1480
+ */
1481
+ type CachedCallParams<T> = CachedCallInput & {
1482
+ call: LLMRequestShape<T>;
1483
+ };
1484
+ /** A cached call with tools. Tool requests and content responses are cached exactly as returned. */
1485
+ type CachedToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
1486
+ call: LLMRequestShape<T, Tools> & {
1487
+ tools: NonNullable<LLMRequestShape<T, Tools>['tools']>;
1488
+ };
1489
+ };
1490
+ /**
1491
+ * A cached call with `call.tools` set conditionally. Returns `T | CallWithToolsResult<T>`, see
1492
+ * `ConditionalToolCallParams`.
1493
+ */
1494
+ type CachedConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
1495
+ call: LLMRequestShape<T, Tools> & {
1496
+ tools: Tools | undefined;
1497
+ };
1498
+ };
1499
+ /** Cached conditional tool-call parameters whose non-tool result is plain text. */
1500
+ type CachedConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedConditionalToolCallParams<string, Tools> & {
1501
+ call: {
1502
+ jsonMode: false;
1503
+ };
1504
+ };
1505
+ /**
1506
+ * Parameters for a cached LLM call with `jsonMode: false`. Selects the
1507
+ * `cachedCall()` overload that returns a plain `string`.
1508
+ */
1509
+ type CachedJsonModeDisabledCallParams = CachedCallInput & {
1510
+ call: Omit<LLMRequestShape<unknown>, 'jsonSchema'> & {
1511
+ jsonMode: false;
1512
+ jsonSchema?: never;
1513
+ };
1514
+ };
1515
+ /**
1516
+ * Parameters for a cached LLM call with `jsonMode: true` and no `schema`.
1517
+ * Selects the `cachedCall()` overload that returns a `JsonValue`.
1518
+ */
1519
+ type CachedJsonModeEnabledCallParams = CachedCallInput & {
1520
+ call: Omit<LLMRequestShape<JsonValue>, 'schema'> & {
1521
+ jsonMode: true;
1522
+ schema?: never;
1523
+ };
1524
+ };
1525
+ /** Context handed to `DetectSoftFailure` alongside the response it's inspecting. */
1526
+ interface SoftFailureMeta {
1527
+ requestId: string;
1528
+ model: string;
1529
+ providerName: string;
1530
+ isFallback: boolean;
1531
+ /** 1-based, matching `CallMeta.attempts`. */
1532
+ attempt: number;
1533
+ /**
1534
+ * Usage for this attempt, if reported. `undefined` means unknown, not zero, so treat it as such
1535
+ * in cost checks.
1536
+ */
1537
+ usage?: TokenUsage;
1538
+ }
1539
+ /**
1540
+ * Reclassifies an otherwise successful result. Return an `LLMErrorCode` to fail the attempt through
1541
+ * the normal retry and breaker paths, or `undefined` to keep it. Catches empty, truncated or
1542
+ * refusal answers that parse fine.
1543
+ */
1544
+ type DetectSoftFailure<T = unknown> = (result: T | CallWithToolsResult<T>, meta: SoftFailureMeta) => LLMErrorCode | undefined;
1545
+ //#endregion
1546
+ //#region src/types/stream.d.ts
1547
+ /** One incremental unit of a streaming response, as delivered to the caller. */
1548
+ type StreamChunk = {
1549
+ type: 'text-delta';
1550
+ delta: string;
1551
+ } | {
1552
+ type: 'tool_call_delta';
1553
+ index: number;
1554
+ id?: string;
1555
+ name?: string;
1556
+ argsDelta?: string;
1557
+ /**
1558
+ * True when `argsDelta` holds the whole arguments rather than a fragment, as with Gemini and
1559
+ * cache replays.
1560
+ */
1561
+ complete?: boolean;
1562
+ } | {
1563
+ type: 'usage';
1564
+ usage: TokenUsage;
1565
+ };
1566
+ /**
1567
+ * What `call()` returns with `stream: true`. `finalResult` resolves to the same shape a
1568
+ * non-streaming call returns. `chunks` is single use; see the streaming docs.
1569
+ */
1570
+ interface StreamCallResult<R> {
1571
+ chunks: AsyncIterable<StreamChunk>;
1572
+ finalResult: Promise<R>;
1573
+ }
1574
+ /** `CallParams` with `stream: true`, selecting the overload that returns `StreamCallResult`. */
1575
+ type StreamEnabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
1576
+ stream: true;
1577
+ };
1578
+ /** Streaming conditional tool-call parameters whose non-tool result is text. */
1579
+ type StreamConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = StreamEnabledCallParams<string, Tools> & ConditionalStringToolCallParams<Tools>;
1580
+ /** Recovers `T` from a `result` already typed `T | StreamCallResult<T>`. Falls back to `unknown`. */
1581
+ type ExtractStreamValue<R> = Extract<R, StreamCallResult<unknown>> extends StreamCallResult<infer V> ? V : unknown;
1582
+ /**
1583
+ * Whether a `call()` result is a `StreamCallResult`. Useful when `stream` was set conditionally,
1584
+ * since the overloads only pick the streaming shape for a literal `stream: true`.
1585
+ */
1586
+ declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
1587
+ /**
1588
+ * Streaming `jsonMode: false`, with `finalResult` as a `string`. `jsonSchema` is `never`, as in
1589
+ * `JsonModeDisabledCallParams`.
1590
+ */
1591
+ type StreamJsonModeDisabledCallParams = Omit<StreamEnabledCallParams<unknown>, 'jsonSchema'> & {
1592
+ jsonMode: false;
1593
+ jsonSchema?: never;
1594
+ };
1595
+ /**
1596
+ * Streaming `jsonMode: true` without `schema`, with `finalResult` as `JsonValue`. `schema` is
1597
+ * `never`, as in `JsonModeEnabledCallParams`.
1598
+ */
1599
+ type StreamJsonModeEnabledCallParams = Omit<StreamEnabledCallParams<JsonValue>, 'schema'> & {
1600
+ jsonMode: true;
1601
+ schema?: never;
1602
+ };
1603
+ /**
1604
+ * The adapter-facing, pre-normalization shape a `createStream` client
1605
+ * implementation emits, analogous to how `WireMessage`/`WireToolCall`
1606
+ * already sit between `CallParams` and each provider's own wire format.
1607
+ */
1608
+ type WireStreamChunk = {
1609
+ type: 'text-delta';
1610
+ delta: string;
1611
+ } | {
1612
+ type: 'tool_call_delta';
1613
+ index: number;
1614
+ id?: string;
1615
+ name?: string;
1616
+ argumentsDelta?: string;
1617
+ /** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
1618
+ complete?: boolean;
1619
+ } | {
1620
+ type: 'usage';
1621
+ usage: {
1622
+ prompt_tokens?: number;
1623
+ completion_tokens?: number;
1624
+ total_tokens?: number;
1625
+ completion_tokens_details?: {
1626
+ reasoning_tokens?: number;
1627
+ };
1628
+ /** Cache split of `prompt_tokens`. Follows OpenAI and OpenRouter naming. */
1629
+ prompt_tokens_details?: {
1630
+ cached_tokens?: number;
1631
+ cache_write_tokens?: number;
1632
+ /** Writes by TTL label, e.g. `{ '5m': 1200, '1h': 800 }`. */
1633
+ cache_write_tokens_by_ttl?: Record<string, number>;
1634
+ };
1635
+ };
1636
+ } | {
1637
+ /** A provider keep-alive with no content. Resets the idle timeout; never reaches callers. */
1638
+ type: 'ping';
1639
+ } | {
1640
+ /**
1641
+ * A rate limit hint from the stream's response headers, yielded as early as possible for
1642
+ * AIMD. The streaming counterpart of `attachRateLimitHint`. Never reaches callers.
1643
+ */
1644
+ type: 'rate_limit_hint';
1645
+ hint: ProviderRateLimitHint;
1646
+ } | {
1647
+ /**
1648
+ * One complete reasoning block, yielded once it has fully arrived.
1649
+ * Collected onto `ToolCallResult.thinking`, never surfaced to
1650
+ * callers as a `StreamChunk`.
1651
+ */
1652
+ type: 'thinking_block';
1653
+ block: ThinkingBlock;
1654
+ };
1655
+ /**
1656
+ * A cached streaming call without tools. A miss relays live chunks; a hit replays the cached value
1657
+ * as chunks. `reserveUsage` and `refundUsage` go at the top level.
1658
+ */
1659
+ type CachedStreamCallParams<T> = CachedCallInput & {
1660
+ call: LLMRequestShape<T> & {
1661
+ stream: true;
1662
+ };
1663
+ };
1664
+ /** A cached streaming call with tools, caching the full `CallWithToolsResult<T>`. */
1665
+ type CachedStreamToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
1666
+ call: LLMRequestShape<T, Tools> & {
1667
+ stream: true;
1668
+ tools: NonNullable<LLMRequestShape<T, Tools>['tools']>;
1669
+ };
1670
+ };
1671
+ /**
1672
+ * A cached streaming call with `call.tools` set conditionally, resolving to `T |
1673
+ * CallWithToolsResult<T>`.
1674
+ */
1675
+ type CachedStreamConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
1676
+ call: LLMRequestShape<T, Tools> & {
1677
+ stream: true;
1678
+ tools: Tools | undefined;
1679
+ };
1680
+ };
1681
+ /** Cached streaming conditional tool-call parameters whose non-tool result is text. */
1682
+ type CachedStreamConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedStreamConditionalToolCallParams<string, Tools> & {
1683
+ call: {
1684
+ jsonMode: false;
1685
+ };
1686
+ };
1687
+ /**
1688
+ * Parameters for a cached, streaming LLM call with `jsonMode: false`.
1689
+ * Selects the `cachedCall()` overload whose `finalResult` (on a miss) or
1690
+ * cached value (on a hit) is a plain `string`.
1691
+ */
1692
+ type CachedStreamJsonModeDisabledCallParams = CachedCallInput & {
1693
+ call: Omit<LLMRequestShape<unknown>, 'jsonSchema'> & {
1694
+ stream: true;
1695
+ jsonMode: false;
1696
+ jsonSchema?: never;
1697
+ };
1698
+ };
1699
+ /**
1700
+ * Parameters for a cached, streaming LLM call with `jsonMode: true` and no
1701
+ * `schema`. Selects the `cachedCall()` overload whose `finalResult` (on a
1702
+ * miss) or cached value (on a hit) is a `JsonValue`.
1703
+ */
1704
+ type CachedStreamJsonModeEnabledCallParams = CachedCallInput & {
1705
+ call: Omit<LLMRequestShape<JsonValue>, 'schema'> & {
1706
+ stream: true;
1707
+ jsonMode: true;
1708
+ schema?: never;
1709
+ };
1710
+ };
1711
+ //#endregion
1712
+ //#region src/types/client.d.ts
1713
+ /** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
1714
+ interface WireToolCall {
1715
+ id: string;
1716
+ type: 'function';
1717
+ function: {
1718
+ name: string;
1719
+ /** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
1720
+ arguments: string;
1721
+ };
1722
+ }
1723
+ /** One entry of `LLMClient`'s `messages` array, named so callers building it can annotate against it. */
1724
+ type WireMessage = {
1725
+ role: 'system';
1726
+ content: string;
1727
+ } | {
1728
+ role: 'user';
1729
+ content: string | ContentBlock[];
1730
+ } | {
1731
+ role: 'assistant';
1732
+ /** Optional: an assistant turn that only requested tools has no text. */
1733
+ content?: string;
1734
+ tool_calls?: WireToolCall[];
1735
+ /** Reasoning blocks to send back ahead of the text and tool calls. Adapters without the concept drop it. */
1736
+ thinking?: ThinkingBlock[];
1737
+ } | {
1738
+ role: 'tool';
1739
+ tool_call_id: string;
1740
+ content: string;
1741
+ /** Honored by `fromAnthropic` (maps to `tool_result.is_error`) and `fromBedrock` (maps to `toolResult.status`); other adapters ignore it. */
1742
+ is_error?: boolean;
1743
+ };
1744
+ /** The OpenAI-shaped wire `tool_choice`. */
1745
+ type WireToolChoice = 'auto' | 'none' | 'required' | {
1746
+ type: 'function';
1747
+ function: {
1748
+ name: string;
1749
+ };
1750
+ };
1751
+ /**
1752
+ * Which adapter built an `LLMClient`, and which provider it talks to.
1753
+ * Read by middleware through `AttemptContext.adapter`, so telemetry can
1754
+ * name the provider without guessing from the model id.
1755
+ */
1756
+ interface AdapterInfo {
1757
+ /** Adapter family, e.g. `'anthropic'`, `'openai-compatible'`, or `'custom'` for a client that sets none. */
1758
+ name: string;
1759
+ /**
1760
+ * The provider this client talks to, in the OpenTelemetry
1761
+ * `gen_ai.provider.name` vocabulary (`'openai'`, `'anthropic'`,
1762
+ * `'aws.bedrock'`, ...). Only set when the adapter knows it for certain.
1763
+ */
1764
+ provider?: string;
1765
+ }
1766
+ /**
1767
+ * The client shape adapters implement, modeled on OpenAI's `chat.completions.create`. Structural
1768
+ * rather than an SDK's own types, since not every SDK's types accept every field.
1769
+ */
1770
+ interface LLMClient {
1771
+ /**
1772
+ * Whether `response_format: 'json_object'` is really enforced. Defaults to `true`. `false` makes
1773
+ * a default `jsonMode` fall back to plain text instead of getting an unenforced no-op, while an
1774
+ * explicit `jsonMode: true` throws.
1775
+ */
1776
+ supportsJsonObjectMode?: boolean;
1777
+ /** Whether cache reads count toward this provider's token rate limit. Default `true`. */
1778
+ cacheReadsCountTowardRateLimit?: boolean;
1779
+ /** Which adapter built this client. Every built in adapter sets it; a hand written client may leave it out. */
1780
+ adapter?: AdapterInfo;
1781
+ /**
1782
+ * Hands the client the `VernLLM` instance's logger, once per target at
1783
+ * construction, for adapter log lines. A client shared by several
1784
+ * instances keeps the last one it was given.
1785
+ */
1786
+ setLogger?(logger: Logger): void;
1787
+ chat: {
1788
+ completions: {
1789
+ create(params: {
1790
+ model: string;
1791
+ temperature?: number;
1792
+ max_tokens: number;
1793
+ response_format?: {
1794
+ type: 'json_object';
1795
+ } | {
1796
+ type: 'json_schema';
1797
+ json_schema: {
1798
+ name: string;
1799
+ schema: Record<string, unknown>;
1800
+ strict?: boolean;
1801
+ description?: string;
1802
+ };
1803
+ };
1804
+ /** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
1805
+ reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
1806
+ /**
1807
+ * Numeric reasoning budget, for providers with one. Clients with only effort tiers use
1808
+ * `reasoning_effort` instead.
1809
+ */
1810
+ budget_tokens?: number;
1811
+ /** Tools the model may call, OpenAI's `function`-wrapped shape. */
1812
+ tools?: Array<{
1813
+ type: 'function';
1814
+ function: {
1815
+ name: string;
1816
+ description: string;
1817
+ parameters: Record<string, unknown>;
1818
+ };
1819
+ }>;
1820
+ tool_choice?: WireToolChoice;
1821
+ /**
1822
+ * Wire-format messages. Breaking change for custom adapters:
1823
+ * implementations must handle tool messages and assistant tool_calls.
1824
+ * Exhaustive switches over only system/user/assistant roles may no longer compile.
1825
+ */
1826
+ messages: WireMessage[];
1827
+ }, options: {
1828
+ signal: AbortSignal;
1829
+ }): Promise<{
1830
+ choices?: Array<{
1831
+ message?: {
1832
+ content?: string | null;
1833
+ tool_calls?: WireToolCall[];
1834
+ /** Reasoning blocks the model produced, in order. Only Claude adapters report them. */
1835
+ thinking?: ThinkingBlock[];
1836
+ };
1837
+ /**
1838
+ * Why generation stopped, in OpenAI's terms. Only `'length'` is read: output that then
1839
+ * fails to parse becomes the retryable `response_truncated`. Optional.
1840
+ */
1841
+ finish_reason?: string | null;
1842
+ }>;
1843
+ usage?: {
1844
+ prompt_tokens?: number;
1845
+ completion_tokens?: number;
1846
+ total_tokens?: number;
1847
+ completion_tokens_details?: {
1848
+ reasoning_tokens?: number;
1849
+ };
1850
+ /** Cache split of `prompt_tokens`. Follows OpenAI and OpenRouter naming. */
1851
+ prompt_tokens_details?: {
1852
+ cached_tokens?: number;
1853
+ cache_write_tokens?: number;
1854
+ /** Writes by TTL label, e.g. `{ '5m': 1200, '1h': 800 }`. */
1855
+ cache_write_tokens_by_ttl?: Record<string, number>;
1856
+ };
1857
+ };
1858
+ }>;
1859
+ /**
1860
+ * Required only for `stream: true`, which throws a clear error without it. Takes the same
1861
+ * request as `create`.
1862
+ */
1863
+ createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
1864
+ signal: AbortSignal;
1865
+ }): AsyncIterable<WireStreamChunk>;
1866
+ };
1867
+ };
1868
+ }
1869
+ //#endregion
1870
+ export { SchemaLike as $, createStateKey as $t, ConditionalToolCallParams as A, ConsecutiveTripping as At, ThinkingBlock as B, MiddlewareRef as Bt, CachedConditionalToolCallParams as C, hasIssues as Cn, RetryBudgetOptions as Ct, CallContext as D, CircuitBreakerOptions as Dt, CachedToolCallParams as E, CircuitBreakerCallContext as Et, JsonModeDisabledCallParams as F, AttemptContext as Ft, ToolCall as G, RequiredMiddlewareRef as Gt, ToolsDisabledCallParams as H, MiddlewareStateEntry as Ht, JsonModeEnabledCallParams as I, CallResult as It, ToolDefinition as J, WireCallRequestPatch as Jt, ToolCallResult as K, VernLLMMiddleware as Kt, JsonValue as L, MiddlewareCapabilities as Lt, ConversationTurn as M, ExponentialBackoffOptions as Mt, DetectSoftFailure as N, RollingTripping as Nt, CallParams as O, CircuitBreakerStateChangeHandler as Ot, ImageBlock as P, TrippingPolicy as Pt, JsonSchemaSpec as Q, createMiddlewareStateBag as Qt, LLMRequestShape as R, MiddlewareContext as Rt, CachedConditionalStringToolCallParams as S, UnsupportedCapabilityIssue as Sn, RetryBudget as St, CachedJsonModeEnabledCallParams as T, CircuitBreakerAdapter as Tt, CallWithToolsResult as U, MiddlewareStateKey as Ut, ToolEnabledCallParams as V, MiddlewareStateBag as Vt, ContentResult as W, PreDispatchContext as Wt, defineTool as X, WireTool as Xt, ToolResult as Y, WireResponseFormat as Yt, isToolCallResult as Z, createMiddlewareRef as Zt, StreamJsonModeEnabledCallParams as _, LLMErrorType as _n, RateLimiter as _t, WireToolChoice as a, OnUsageFailure as an, FallbackOnContext as at, AssistantContent as b, ToolIssue as bn, ProviderRateLimitHint as bt, CachedStreamConditionalToolCallParams as c, TokenUsage as cn, TargetInfo as ct, CachedStreamToolCallParams as d, DuplicateToolNamesIssue as dn, metaRef as dt, requireRef as en, CallMeta as et, StreamCallResult as f, HistoryToolResultIssue as fn, RateLimitOption as ft, StreamJsonModeDisabledCallParams as g, LLMErrorSnapshot as gn, RateLimitState as gt, StreamEnabledCallParams as h, LLMErrorIssuesByCode as hn, RateLimitReason as ht, WireToolCall as i, OnUsage as in, FallbackOn as it, ContentBlock as j, CooldownBackoff as jt, ConditionalStringToolCallParams as k, CircuitState as kt, CachedStreamJsonModeDisabledCallParams as l, ConsoleLogger as ln, defaultFallbackOn as lt, StreamConditionalStringToolCallParams as m, LLMErrorCode as mn, RateLimitOptions as mt, LLMClient as n, OnEvent as nn, FallbackAttempt as nt, CachedStreamCallParams as o, RefundUsage as on, FallbackTarget as ot, StreamChunk as p, LLMError as pn, RateLimitAcquireResult as pt, ToolChoice as q, WireCallRequest as qt, WireMessage as r, VernLLMEvent as rn, FallbackExhaustedError as rt, CachedStreamConditionalStringToolCallParams as s, ReserveUsage as sn, TargetCircuitState as st, AdapterInfo as t, stateEntry as tn, CircuitTarget as tt, CachedStreamJsonModeEnabledCallParams as u, Logger as un, isFallbackExhaustedError as ut, WireStreamChunk as v, LLMRequestSnapshot as vn, RateLimiterAdapter as vt, CachedJsonModeDisabledCallParams as w, isLLMError as wn, CircuitBreaker as wt, CachedCallParams as x, UnknownToolChoiceIssue as xn, CircuitBreakerOption as xt, isStreamResult as y, RetryAttempt as yn, WireRequest as yt, TextBlock as z, MiddlewareContextBase as zt };
1871
+ //# sourceMappingURL=client-HWxkwVvj.d.cts.map