vern-llm 2.7.0 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -147,7 +147,7 @@ interface LLMErrorOptions {
147
147
  /** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
148
148
  attempts?: RetryAttempt[];
149
149
  }
150
- declare class LLMError extends Error {
150
+ export declare class LLMError extends Error {
151
151
  type: LLMErrorType;
152
152
  status?: number;
153
153
  issues?: unknown;
@@ -203,7 +203,7 @@ declare class LLMError extends Error {
203
203
  */
204
204
  toJSON(): Record<string, unknown>;
205
205
  }
206
- declare function isLLMError(err: unknown): err is LLMError;
206
+ export declare function isLLMError(err: unknown): err is LLMError;
207
207
  /**
208
208
  * Narrows `err.issues` to the exact shape {@link LLMErrorIssuesByCode} maps
209
209
  * `code` to, for any code listed there. `code` stays the only discriminator
@@ -216,7 +216,7 @@ declare function isLLMError(err: unknown): err is LLMError;
216
216
  * }
217
217
  * ```
218
218
  */
219
- declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
219
+ export declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
220
220
  code: C;
221
221
  issues: LLMErrorIssuesByCode[C];
222
222
  };
@@ -241,7 +241,7 @@ type EvictionOption = 'fifo' | 'lru';
241
241
  * Trivial default so the package works out of the box with no external deps.
242
242
  * Not shared across processes, swap in Redis/Upstash/etc for production.
243
243
  */
244
- declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
244
+ export declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
245
245
  private readonly maxSize;
246
246
  private store;
247
247
  private readonly eviction;
@@ -258,7 +258,7 @@ declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
258
258
  /**
259
259
  * Normalizes keys before caching to avoid duplicate entries from formatting differences.
260
260
  */
261
- declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
261
+ export declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
262
262
  private readonly inner;
263
263
  constructor(inner?: CacheAdapter<T>);
264
264
  private normalize;
@@ -274,7 +274,7 @@ declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
274
274
  * Two-tier cache with fast local L1 and shared L2.
275
275
  * L2 hits are promoted back to L1.
276
276
  */
277
- declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
277
+ export declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
278
278
  private readonly l1;
279
279
  private readonly l2;
280
280
  private readonly l1Ttl?;
@@ -293,6 +293,89 @@ declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
293
293
  delete(key: string): Promise<void>;
294
294
  }
295
295
  //#endregion
296
+ //#region src/logger.d.ts
297
+ interface Logger {
298
+ debug(message: string): void;
299
+ warn(message: string): void;
300
+ error(message: string, meta?: Record<string, unknown>): void;
301
+ }
302
+ /**
303
+ * Default logger. `debug` is gated by the `debug` option on VernLLM
304
+ * warn/error always fire since they indicate real problems (retries, cache failures)
305
+ */
306
+ export declare class ConsoleLogger implements Logger {
307
+ private debugEnabled;
308
+ constructor(debugEnabled: boolean);
309
+ debug(message: string): void;
310
+ warn(message: string): void;
311
+ error(message: string, meta?: Record<string, unknown>): void;
312
+ }
313
+ //#endregion
314
+ //#region src/types/usage.d.ts
315
+ type ReserveUsage = (params: {
316
+ coalesced: boolean;
317
+ signal?: AbortSignal;
318
+ }) => Promise<void>;
319
+ type RefundUsage = (params: {
320
+ coalesced: boolean;
321
+ signal?: AbortSignal;
322
+ }) => Promise<void>;
323
+ /**
324
+ * The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
325
+ * and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
326
+ * has one definition instead of being redeclared at each use site.
327
+ */
328
+ interface UsageHooks {
329
+ /**
330
+ * Reserves usage before the request. Failures become
331
+ * LLMError('quota_exceeded').
332
+ */
333
+ reserveUsage?: ReserveUsage;
334
+ /**
335
+ * Refunds usage after a failed call if reservation succeeded.
336
+ */
337
+ refundUsage?: RefundUsage;
338
+ }
339
+ interface TokenUsage {
340
+ promptTokens: number;
341
+ completionTokens: number;
342
+ totalTokens: number;
343
+ /**
344
+ * Tokens spent on internal reasoning, a subset of `completionTokens`,
345
+ * never added on top of it. Undefined when the provider's response
346
+ * doesn't report a separate reasoning figure, e.g. Bedrock Converse
347
+ * without an explicit `additionalModelResponseFieldPaths` request.
348
+ */
349
+ reasoningTokens?: number;
350
+ requestId: string;
351
+ model: string;
352
+ /**
353
+ * The provider target that produced this usage. See `VernLLMOptions['name']`,
354
+ * default `'primary'`. Optional so consumers constructing a `TokenUsage`
355
+ * themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
356
+ * populates it. Absent means the same as `'primary'` if you need a value.
357
+ */
358
+ provider?: string;
359
+ /**
360
+ * Whether this usage came from a fallback target rather than the
361
+ * primary. Optional for the same reason `provider` is: `VernLLM`
362
+ * always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
363
+ * isn't forced to.
364
+ */
365
+ usedFallback?: boolean;
366
+ }
367
+ type OnUsage = (usage: TokenUsage) => void;
368
+ /**
369
+ * Called when a provider response arrives but VernLLM's own post-processing
370
+ * then fails, after usage data was already present in that response. Covers
371
+ * any error thrown after usage extraction, not just parse/validation, since
372
+ * everything in that path only runs once a response, and real spend, has
373
+ * already arrived. Fires once per failed attempt with extractable usage,
374
+ * never for transport failures, where no response means no honest number
375
+ * to report.
376
+ */
377
+ type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
378
+ //#endregion
296
379
  //#region src/types/events.d.ts
297
380
  /**
298
381
  * Reports what happened during a call. Fire and forget, mirroring
@@ -359,6 +442,26 @@ type VernLLMEvent = {
359
442
  hook: 'transform' | 'wrap_short_circuit' | 'enabled_skip';
360
443
  /** For `hook: 'transform'` only: which top-level fields the merged patch touched. */
361
444
  patchedFields?: string[];
445
+ } | {
446
+ /**
447
+ * Reported once a call fully succeeds. Same data `VernLLMOptions.onUsage`
448
+ * receives; that option is sugar over this event, not a second
449
+ * reporting path, see `makeEventReporter`.
450
+ */
451
+ kind: 'usage';
452
+ requestId: string;
453
+ usage: TokenUsage;
454
+ } | {
455
+ /**
456
+ * A provider response arrived, carrying real usage, and VernLLM's own
457
+ * post-processing then failed. Fires once per failed attempt with
458
+ * extractable usage, matching `VernLLMOptions.onUsageFailure`'s own
459
+ * granularity, which this event is sugar over, not a second path.
460
+ */
461
+ kind: 'usage_failure';
462
+ requestId: string;
463
+ usage: TokenUsage;
464
+ error: LLMError;
362
465
  };
363
466
  type OnEvent = (event: VernLLMEvent) => void;
364
467
  //#endregion
@@ -372,6 +475,19 @@ interface MiddlewareCapabilities {
372
475
  */
373
476
  supportsJsonObjectMode: boolean;
374
477
  }
478
+ /**
479
+ * Not exported. Distinguishes `MiddlewareStateKey<T>` from
480
+ * `MiddlewareRef` and from a plain `{ debugName }` object literal at
481
+ * the type level, even though all three have the identical runtime
482
+ * shape. Without this, `MiddlewareStateKey<T>`/`MiddlewareRef` are
483
+ * structurally just `{ debugName: string }`, so TypeScript would treat
484
+ * a state key as a valid middleware ref (or vice versa), and would let
485
+ * anyone hand-write `{ debugName: 'auth' }` in place of a real
486
+ * `createMiddlewareRef` result. Neither is possible once this brand is
487
+ * required: only `createStateKey`, which alone has access to this
488
+ * symbol, can produce a value satisfying `MiddlewareStateKey<T>`.
489
+ */
490
+ declare const stateKeyBrand: unique symbol;
375
491
  /**
376
492
  * A typed reference to one slot in `ctx.state`. Create one with
377
493
  * `createStateKey`, export it, and import the same reference wherever
@@ -382,6 +498,7 @@ interface MiddlewareCapabilities {
382
498
  */
383
499
  interface MiddlewareStateKey<T> {
384
500
  readonly debugName: string;
501
+ readonly [stateKeyBrand]: true;
385
502
  /**
386
503
  * Never set at runtime; exists purely so `T` is actually used
387
504
  * somewhere in this interface's shape (a phantom type), which is what
@@ -392,7 +509,43 @@ interface MiddlewareStateKey<T> {
392
509
  readonly __phantom?: T;
393
510
  }
394
511
  /** Creates a new, distinct `MiddlewareStateKey`. `debugName` is used only in log lines and the `'middleware'` event; it never affects equality. */
395
- declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
512
+ export declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
513
+ /** Not exported. See `stateKeyBrand`; same reasoning, distinct symbol, so the two token types can't be cross-assigned either. */
514
+ declare const middlewareRefBrand: unique symbol;
515
+ /**
516
+ * A typed reference to one middleware's identity, for `runsAfter`/
517
+ * `runsBefore` to target. Purely an ordering concern: unlike `name`,
518
+ * `ref` is never used as a display label anywhere (`name` still covers
519
+ * that), only as a `runsAfter`/`runsBefore` match target. Create one
520
+ * with `createMiddlewareRef`, export it from the package that owns the
521
+ * middleware, and have any dependent import the same reference instead
522
+ * of typing a matching `name` string. Same reasoning as
523
+ * `MiddlewareStateKey`: a typo becomes a missing import, a compile
524
+ * error, instead of a silently unresolved (or worse, silently
525
+ * colliding) string.
526
+ */
527
+ interface MiddlewareRef {
528
+ readonly debugName: string;
529
+ readonly [middlewareRefBrand]: true;
530
+ }
531
+ /** Creates a new, distinct `MiddlewareRef`. `debugName` is used only in error messages when a reference doesn't resolve; it never affects equality, so two refs with the same `debugName` never collide. */
532
+ export declare function createMiddlewareRef(debugName: string): MiddlewareRef;
533
+ /**
534
+ * A `runsAfter`/`runsBefore` entry that escalates an unresolved
535
+ * reference from a warning to a construction-time throw. Wrap a
536
+ * `MiddlewareRef` with `requireRef` when the dependency isn't optional:
537
+ * a bare `MiddlewareRef` in `runsAfter`/`runsBefore` means "order
538
+ * relative to this if it's registered," which is the right default for
539
+ * a dependency a third party may reasonably not have installed. A
540
+ * `RequiredMiddlewareRef` means "this middleware must not run without
541
+ * that dependency having already run". The app should fail to start
542
+ * rather than run with a silently-missing ordering guarantee.
543
+ */
544
+ interface RequiredMiddlewareRef {
545
+ readonly ref: MiddlewareRef;
546
+ }
547
+ /** Wraps `ref` so `runsAfter`/`runsBefore` throws at `VernLLM` construction time if it doesn't resolve, instead of warning and continuing. */
548
+ export declare function requireRef(ref: MiddlewareRef): RequiredMiddlewareRef;
396
549
  /**
397
550
  * Typed, per-logical-call storage two middleware can deliberately share a
398
551
  * value through (a span ID one sets, another reads). Backed by a plain
@@ -404,7 +557,7 @@ interface MiddlewareStateBag {
404
557
  set<T>(key: MiddlewareStateKey<T>, value: T): void;
405
558
  }
406
559
  /** A plain, `Map`-backed `MiddlewareStateBag`. */
407
- declare function createMiddlewareStateBag(): MiddlewareStateBag;
560
+ export declare function createMiddlewareStateBag(): MiddlewareStateBag;
408
561
  /** Fields every `MiddlewareContext` variant carries, regardless of `stage`. */
409
562
  interface MiddlewareContextBase {
410
563
  requestId: string;
@@ -554,22 +707,36 @@ interface CallResult<T = unknown> {
554
707
  interface VernLLMMiddleware {
555
708
  /** Used in log lines and the `'middleware'` event. Defaults to this entry's array position when omitted. */
556
709
  name?: string;
710
+ /**
711
+ * This entry's own identity, purely for another middleware's
712
+ * `runsAfter`/`runsBefore` to target. Create with `createMiddlewareRef`,
713
+ * export it, and have a dependent import the same reference. Optional:
714
+ * only needed if something else must be able to depend on this
715
+ * specific entry. Unrelated to `name`: `ref` is never shown in logs,
716
+ * `name` is never matched against for ordering.
717
+ */
718
+ ref?: MiddlewareRef;
557
719
  /** Sort key for composition order, ascending, ties broken by array order. See the middleware docs for what "lower runs first" means for `wrap`. */
558
720
  priority?: number;
559
721
  /**
560
- * Names of other middleware this entry must run after, breaking ties
561
- * `priority` alone can't express. A referenced name absent from the
562
- * registered set is dropped, not an error, since a third party may
563
- * reasonably reference a well known name that isn't installed
564
- * everywhere. A cycle across `runsAfter`/`runsBefore` throws at
565
- * `VernLLM` construction time.
722
+ * Other middleware this entry must run after, breaking ties
723
+ * `priority` alone can't express. Matched by `ref` identity, so a
724
+ * typo or a stale copy simply fails to resolve instead of silently
725
+ * matching the wrong entry. A bare `MiddlewareRef` that doesn't
726
+ * resolve is dropped, not an error, since a third party may
727
+ * reasonably reference a well known middleware that isn't installed
728
+ * everywhere; wrap it with `requireRef` to make that same target
729
+ * mandatory instead, throwing at `VernLLM` construction time if it's
730
+ * missing. A cycle across `runsAfter`/`runsBefore` always throws,
731
+ * regardless of whether any individual entry is required.
566
732
  */
567
- runsAfter?: string[];
733
+ runsAfter?: (MiddlewareRef | RequiredMiddlewareRef)[];
568
734
  /**
569
- * Names of other middleware this entry must run before. See
570
- * `runsAfter`; an absent reference is dropped, not an error.
735
+ * Other middleware this entry must run before. See `runsAfter`; a
736
+ * bare reference is dropped if unresolved, a `requireRef`-wrapped one
737
+ * throws.
571
738
  */
572
- runsBefore?: string[];
739
+ runsBefore?: (MiddlewareRef | RequiredMiddlewareRef)[];
573
740
  /**
574
741
  * Pins this entry's slot in `wrap` nesting only, independent of
575
742
  * `priority`/`runsAfter`/`runsBefore`, which still govern
@@ -611,17 +778,21 @@ interface CircuitBreakerCallContext {
611
778
  /** Omitted for calls before any attempt exists, like `assertClosed`'s pre-dispatch check. */
612
779
  attempt?: number;
613
780
  }
781
+ /**
782
+ * Fires after every real state change, never a no-op transition. `model`
783
+ * is the resolved model of whichever call triggered it. With
784
+ * `isolateByModel` off, failures are still counted across every model.
785
+ * Shared by `CircuitBreakerOptions` and `CircuitBreakerAdapter`, so a
786
+ * custom adapter reports state changes the same way the built in
787
+ * `CircuitBreaker` does.
788
+ */
789
+ type CircuitBreakerStateChangeHandler = (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
614
790
  interface CircuitBreakerOptions {
615
791
  /** Consecutive failures before the circuit opens, default 5 */
616
792
  threshold?: number;
617
793
  /** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
618
794
  cooldownMs?: number;
619
- /**
620
- * Fires after every real state change, never a no-op transition. `model`
621
- * is the resolved model of whichever call triggered it. With
622
- * `isolateByModel` off, failures are still counted across every model.
623
- */
624
- onStateChange?: (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
795
+ onStateChange?: CircuitBreakerStateChangeHandler;
625
796
  /**
626
797
  * Track a separate circuit per resolved model instead of one shared
627
798
  * circuit. Default false. A call that omits `model` falls into one
@@ -681,7 +852,7 @@ interface TrippingPolicy {
681
852
  */
682
853
  forget?(key: string): void;
683
854
  }
684
- declare class ConsecutiveTripping implements TrippingPolicy {
855
+ export declare class ConsecutiveTripping implements TrippingPolicy {
685
856
  private readonly threshold;
686
857
  private failuresByKey;
687
858
  constructor(threshold: number);
@@ -690,7 +861,7 @@ declare class ConsecutiveTripping implements TrippingPolicy {
690
861
  reset(key: string): void;
691
862
  forget(key: string): void;
692
863
  }
693
- declare class RollingTripping implements TrippingPolicy {
864
+ export declare class RollingTripping implements TrippingPolicy {
694
865
  private readonly windowMs;
695
866
  private readonly minCalls;
696
867
  private readonly failureRatio;
@@ -713,14 +884,65 @@ type TrippingOption = {
713
884
  failureRatio: number;
714
885
  } | TrippingPolicy;
715
886
  type CircuitState = 'closed' | 'open' | 'half-open';
887
+ /**
888
+ * What VernLLM's dispatch layer needs from a breaker. `CircuitBreaker`
889
+ * implements this; a caller wanting cross process coordination can hand
890
+ * over their own instance instead.
891
+ *
892
+ * `assertClosed`, `recordSuccess`, `recordFailure`, and `onStateChange`
893
+ * are required, mirroring `RateLimiterAdapter`'s four required methods.
894
+ * `onStateChange` is required so `circuit_state` events can't go
895
+ * silently missing; a no-op `() => {}` is fine if you don't care.
896
+ *
897
+ * `getState`, `getFailureBreakdown`, `isolateByModel`, `open`, and
898
+ * `close` are optional. Omitting one makes the matching call a no-op
899
+ * or return `undefined`/`false`, same as no breaker configured.
900
+ * `open`/`close` are optional since they let VernLLM force a
901
+ * transition, control a distributed adapter may not want to grant.
902
+ */
903
+ interface CircuitBreakerAdapter {
904
+ /** Throws when the circuit is open (or half open with no trial slot free) for `model`. */
905
+ assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
906
+ recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
907
+ /** `code`, when present, is the failing call's `LLMErrorCode`. */
908
+ recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
909
+ getState?(model?: string): CircuitState;
910
+ /** Failure counts by `LLMErrorCode` for `model`'s bucket, `'unknown'` for one that carried no code. */
911
+ getFailureBreakdown?(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
912
+ /** Whether this adapter tracks failures per model, mirroring `CircuitBreakerOptions.isolateByModel`. Read by `warnIfModelUnsupported`'s diagnostic warning and by `VernLLM.getCircuitStates()`'s public output; omit if the notion doesn't apply to your adapter, `false` is assumed. */
913
+ isolateByModel?: boolean;
914
+ /** Manually opens the circuit, as if enough consecutive failures had just happened. Optional: an adapter that doesn't want external callers forcing a transition can omit it. */
915
+ open?(model?: string, context?: CircuitBreakerCallContext): void;
916
+ /** Manually closes the circuit, without requiring a real success first. Same opt-in reasoning as `open`. */
917
+ close?(model?: string, context?: CircuitBreakerCallContext): void;
918
+ /** Gives back a half-open trial slot when a call ends without `recordSuccess` or `recordFailure`. Idempotent, and a no-op for a call that holds no slot. */
919
+ releaseTrial?(model?: string, context?: CircuitBreakerCallContext): void;
920
+ /** Awaited right before `assertClosed` to refresh local state. Never blocks or fails a call: a rejection or `prepareTimeoutMs` is logged and the call carries on. */
921
+ prepare?(model?: string, context?: CircuitBreakerCallContext): Promise<void>;
922
+ /** How long to wait for `prepare`, in ms. Default 1000. */
923
+ prepareTimeoutMs?: number;
924
+ /** Live counterpart of `getState`, read by `VernLLM.readCircuitStates()`. */
925
+ readState?(model?: string): Promise<CircuitState>;
926
+ /** Receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
927
+ setLogger?(logger: Logger): void;
928
+ /**
929
+ * Called after every real state change, never a no-op transition. VernLLM
930
+ * wraps it the same way it wraps the built in `CircuitBreaker`'s
931
+ * `onStateChange`: every call still reports a `circuit_state` event
932
+ * first, then this hook is chained after that, wrapped so a throw here
933
+ * can't break the call that triggered it.
934
+ */
935
+ onStateChange: CircuitBreakerStateChangeHandler;
936
+ }
716
937
  /**
717
938
  * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures
718
939
  * across calls. Once the threshold is hit, short-circuits new calls with
719
940
  * LLMError('circuit_open') until the cooldown elapses and a trial succeeds.
720
941
  */
721
- declare class CircuitBreaker {
942
+ export declare class CircuitBreaker implements CircuitBreakerAdapter {
722
943
  private readonly cooldownMs;
723
- private readonly onStateChange?;
944
+ /** Satisfies `CircuitBreakerAdapter.onStateChange`, required there. Defaults to a no-op when `options.onStateChange` is omitted. */
945
+ readonly onStateChange: CircuitBreakerStateChangeHandler;
724
946
  /** Whether this breaker tracks failures per model instead of one shared circuit. */
725
947
  readonly isolateByModel: boolean;
726
948
  private readonly halfOpenProbes;
@@ -739,6 +961,14 @@ declare class CircuitBreaker {
739
961
  recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
740
962
  /** `code`, when present, is the failing `LLMError`'s `code`. Missing attributes to `'unknown'`. */
741
963
  recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
964
+ /**
965
+ * Gives back the half-open trial slot `context`'s call claimed, when
966
+ * that call ended without recording an outcome. No-op without a
967
+ * `context`, when the bucket isn't half-open, or when the call's permit
968
+ * is stale or already spent (an outcome was recorded), so calling it
969
+ * defensively on every failure path is safe.
970
+ */
971
+ releaseTrial(model?: string, context?: CircuitBreakerCallContext): void;
742
972
  /** With `isolateByModel` off, `model` is ignored and the shared circuit's state is returned. */
743
973
  getState(model?: string): CircuitState;
744
974
  /** Failure counts by `LLMErrorCode` for `model`'s bucket. Returned as a plain object copy. */
@@ -801,7 +1031,7 @@ interface RetryBudgetOptions {
801
1031
  * the same primitive `RollingTripping` is built on, rather than a second
802
1032
  * hand rolled window.
803
1033
  */
804
- declare class RetryBudget {
1034
+ export declare class RetryBudget {
805
1035
  private readonly options;
806
1036
  private readonly ratio;
807
1037
  constructor(options: RetryBudgetOptions);
@@ -820,7 +1050,7 @@ declare class RetryBudget {
820
1050
  };
821
1051
  }
822
1052
  //#endregion
823
- //#region src/internal/utils/rateLimitHint.utils.d.ts
1053
+ //#region src/internal/utils/rate-limit/rateLimitHint.utils.d.ts
824
1054
  /** A normalized read of a provider's rate limit headers. */
825
1055
  interface ProviderRateLimitHint {
826
1056
  remainingRequests?: number;
@@ -911,7 +1141,7 @@ interface RateLimitAcquireResult {
911
1141
  reason?: RateLimitReason;
912
1142
  }
913
1143
  /** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
914
- declare function defaultEstimateTokens(request: WireRequest): number;
1144
+ export declare function defaultEstimateTokens(request: WireRequest): number;
915
1145
  /**
916
1146
  * What VernLLM's dispatch layer needs from a limiter. `RateLimiter`
917
1147
  * implements this; a caller wanting cross-process coordination can hand
@@ -926,6 +1156,10 @@ interface RateLimiterAdapter {
926
1156
  reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
927
1157
  /** Optional: current bucket levels, for introspection. Omit if the adapter has no state worth reporting. */
928
1158
  getState?(): RateLimitState;
1159
+ /** Optional: live bucket levels for `VernLLM.readRateLimitState()`, the async counterpart of `getState`. */
1160
+ readState?(): Promise<RateLimitState>;
1161
+ /** Optional: receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
1162
+ setLogger?(logger: Logger): void;
929
1163
  }
930
1164
  /**
931
1165
  * Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
@@ -933,7 +1167,7 @@ interface RateLimiterAdapter {
933
1167
  * stream of small ones. Any bucket omitted from `options` has infinite
934
1168
  * capacity and never blocks.
935
1169
  */
936
- declare class RateLimiter implements RateLimiterAdapter {
1170
+ export declare class RateLimiter implements RateLimiterAdapter {
937
1171
  private readonly requests?;
938
1172
  private readonly tokens?;
939
1173
  private readonly concurrency?;
@@ -1019,7 +1253,7 @@ declare class RateLimiter implements RateLimiterAdapter {
1019
1253
  getState(): RateLimitState;
1020
1254
  }
1021
1255
  //#endregion
1022
- //#region src/internal/utils/rateLimitAdapter.utils.d.ts
1256
+ //#region src/internal/utils/rate-limit/rateLimitAdapter.utils.d.ts
1023
1257
  /** Not exported. Internal shorthand only, so this union isn't duplicated between the public option fields and `buildRateLimit`'s own signature. */
1024
1258
  type RateLimitOption = RateLimitOptions | RateLimiterAdapter;
1025
1259
  //#endregion
@@ -1121,7 +1355,7 @@ type FallbackOn = (error: LLMError, context: {
1121
1355
  * The default `fallbackOn` policy. Exported so a caller can wrap rather
1122
1356
  * than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
1123
1357
  */
1124
- declare const defaultFallbackOn: FallbackOn;
1358
+ export declare const defaultFallbackOn: FallbackOn;
1125
1359
  /**
1126
1360
  * Thrown when the chain gives up, whether because the last target failed
1127
1361
  * or `fallbackOn` chose to stop early. Carries each attempt in order so
@@ -1131,7 +1365,7 @@ declare const defaultFallbackOn: FallbackOn;
1131
1365
  * so existing type-based handling, including reading `retryAfterMs` on an
1132
1366
  * `'api'`-typed error, keeps working on a fallback-exhausted error too.
1133
1367
  */
1134
- declare class FallbackExhaustedError extends LLMError {
1368
+ export declare class FallbackExhaustedError extends LLMError {
1135
1369
  readonly attempts: FallbackAttempt[];
1136
1370
  constructor(attempts: FallbackAttempt[]);
1137
1371
  /**
@@ -1143,7 +1377,7 @@ declare class FallbackExhaustedError extends LLMError {
1143
1377
  get retryable(): boolean;
1144
1378
  }
1145
1379
  /** Narrows `err` to {@link FallbackExhaustedError}, for direct access to its `attempts` (`provider`/`model` per failed target) without a manual `instanceof` check. */
1146
- declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
1380
+ export declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
1147
1381
  /**
1148
1382
  * Creates an empty ref box to pass as `CallParams['meta']`, so a caller can
1149
1383
  * read the `CallMeta` written by `call()` on the same line as the result
@@ -1154,7 +1388,7 @@ declare function isFallbackExhaustedError(err: unknown): err is FallbackExhauste
1154
1388
  * const result = await vern.call({ userContent: '...', meta });
1155
1389
  * meta.current?.provider;
1156
1390
  */
1157
- declare function metaRef(): {
1391
+ export declare function metaRef(): {
1158
1392
  current?: CallMeta;
1159
1393
  };
1160
1394
  //#endregion
@@ -1229,7 +1463,7 @@ interface ToolDefinition<Name extends string = string, Args = unknown> {
1229
1463
  * in `defineTool()` preserves the literal `name` type without requiring
1230
1464
  * `as const` at every call site.
1231
1465
  */
1232
- declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
1466
+ export declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
1233
1467
  /** Maps a single `ToolDefinition` to its matching `ToolCall` shape. */
1234
1468
  type ToolCallFor<T> = T extends ToolDefinition<infer N, infer A> ? {
1235
1469
  id: string;
@@ -1290,77 +1524,12 @@ type ResolvedTools<Tools, R> = [Tools] extends [never] ? ExtractTools<R> : Tools
1290
1524
  *
1291
1525
  * Pass `Tools` explicitly to override inference, e.g. `isToolCallResult<typeof tools>(result)`.
1292
1526
  */
1293
- declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
1527
+ export declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
1294
1528
  /** What the model should do about tools on a given call. */
1295
1529
  type ToolChoice = 'auto' | 'none' | 'required' | {
1296
1530
  name: string;
1297
1531
  };
1298
1532
  //#endregion
1299
- //#region src/types/usage.d.ts
1300
- type ReserveUsage = (params: {
1301
- coalesced: boolean;
1302
- signal?: AbortSignal;
1303
- }) => Promise<void>;
1304
- type RefundUsage = (params: {
1305
- coalesced: boolean;
1306
- signal?: AbortSignal;
1307
- }) => Promise<void>;
1308
- /**
1309
- * The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
1310
- * and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
1311
- * has one definition instead of being redeclared at each use site.
1312
- */
1313
- interface UsageHooks {
1314
- /**
1315
- * Reserves usage before the request. Failures become
1316
- * LLMError('quota_exceeded').
1317
- */
1318
- reserveUsage?: ReserveUsage;
1319
- /**
1320
- * Refunds usage after a failed call if reservation succeeded.
1321
- */
1322
- refundUsage?: RefundUsage;
1323
- }
1324
- interface TokenUsage {
1325
- promptTokens: number;
1326
- completionTokens: number;
1327
- totalTokens: number;
1328
- /**
1329
- * Tokens spent on internal reasoning, a subset of `completionTokens`,
1330
- * never added on top of it. Undefined when the provider's response
1331
- * doesn't report a separate reasoning figure, e.g. Bedrock Converse
1332
- * without an explicit `additionalModelResponseFieldPaths` request.
1333
- */
1334
- reasoningTokens?: number;
1335
- requestId: string;
1336
- model: string;
1337
- /**
1338
- * The provider target that produced this usage. See `VernLLMOptions['name']`,
1339
- * default `'primary'`. Optional so consumers constructing a `TokenUsage`
1340
- * themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
1341
- * populates it. Absent means the same as `'primary'` if you need a value.
1342
- */
1343
- provider?: string;
1344
- /**
1345
- * Whether this usage came from a fallback target rather than the
1346
- * primary. Optional for the same reason `provider` is: `VernLLM`
1347
- * always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
1348
- * isn't forced to.
1349
- */
1350
- usedFallback?: boolean;
1351
- }
1352
- type OnUsage = (usage: TokenUsage) => void;
1353
- /**
1354
- * Called when a provider response arrives but VernLLM's own post-processing
1355
- * then fails, after usage data was already present in that response. Covers
1356
- * any error thrown after usage extraction, not just parse/validation, since
1357
- * everything in that path only runs once a response, and real spend, has
1358
- * already arrived. Fires once per failed attempt with extractable usage,
1359
- * never for transport failures, where no response means no honest number
1360
- * to report.
1361
- */
1362
- type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
1363
- //#endregion
1364
1533
  //#region src/types/call.d.ts
1365
1534
  /**
1366
1535
  * Any valid JSON value: a primitive, `null`, or a JSON array/object made
@@ -1701,6 +1870,13 @@ interface SoftFailureMeta {
1701
1870
  isFallback: boolean;
1702
1871
  /** 1-based, matching `CallMeta.attempts`. */
1703
1872
  attempt: number;
1873
+ /**
1874
+ * Token usage for this attempt, if the provider reported it on this
1875
+ * response. `undefined` when the provider omitted usage, not when
1876
+ * usage was zero, so a cost check should treat a missing value as
1877
+ * unknown rather than as free.
1878
+ */
1879
+ usage?: TokenUsage;
1704
1880
  }
1705
1881
  /**
1706
1882
  * Inspects an otherwise-successful result and optionally reclassifies it
@@ -1775,7 +1951,7 @@ type ExtractStreamValue<R> = Extract<R, StreamCallResult<unknown>> extends Strea
1775
1951
  * }
1776
1952
  * ```
1777
1953
  */
1778
- declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
1954
+ export declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
1779
1955
  /**
1780
1956
  * `StreamEnabledCallParams` with `jsonMode: false`. Selects the streaming
1781
1957
  * `call()` overload whose `finalResult` resolves to a plain `string`.
@@ -2053,7 +2229,7 @@ interface LLMClient {
2053
2229
  };
2054
2230
  }
2055
2231
  //#endregion
2056
- //#region src/internal/utils/cacheAdapter.utils.d.ts
2232
+ //#region src/internal/utils/cache/cacheAdapter.utils.d.ts
2057
2233
  /**
2058
2234
  * Not exported. Internal shorthand for `VernLLMOptions.cache`, so the
2059
2235
  * union isn't duplicated between that field and `buildCache`'s own
@@ -2065,23 +2241,15 @@ type CacheOption = {
2065
2241
  eviction?: EvictionOption;
2066
2242
  } | CacheAdapter;
2067
2243
  //#endregion
2068
- //#region src/logger.d.ts
2069
- interface Logger {
2070
- debug(message: string): void;
2071
- warn(message: string): void;
2072
- error(message: string, meta?: Record<string, unknown>): void;
2073
- }
2244
+ //#region src/internal/utils/circuit-breaker/circuitBreakerAdapter.utils.d.ts
2074
2245
  /**
2075
- * Default logger. `debug` is gated by the `debug` option on VernLLM
2076
- * warn/error always fire since they indicate real problems (retries, cache failures)
2246
+ * Not re-exported from the package root, imported directly from this
2247
+ * internal module by `VernLLMOptions.circuitBreaker`'s own type (see
2248
+ * options.ts) so that union isn't duplicated between the public option
2249
+ * field and `buildCircuitBreaker`'s own signature below, same pattern
2250
+ * `CacheOption` and `RateLimitOption` already use for their own options.
2077
2251
  */
2078
- declare class ConsoleLogger implements Logger {
2079
- private debugEnabled;
2080
- constructor(debugEnabled: boolean);
2081
- debug(message: string): void;
2082
- warn(message: string): void;
2083
- error(message: string, meta?: Record<string, unknown>): void;
2084
- }
2252
+ type CircuitBreakerOption = boolean | CircuitBreakerOptions | CircuitBreakerAdapter;
2085
2253
  //#endregion
2086
2254
  //#region src/types/options.d.ts
2087
2255
  interface VernLLMOptions {
@@ -2196,9 +2364,11 @@ interface VernLLMOptions {
2196
2364
  /**
2197
2365
  * Enables a circuit breaker that short-circuits calls after repeated
2198
2366
  * consecutive failures, instead of continuing to hammer a down provider
2199
- * Pass `true` for defaults, or an options object to tune threshold/cooldown
2367
+ * Pass `true` for defaults, or an options object to tune threshold/cooldown.
2368
+ * Pass a `CircuitBreakerAdapter` instead for cross-process coordination,
2369
+ * the same pattern `cache` and `rateLimit` already support.
2200
2370
  */
2201
- circuitBreaker?: boolean | CircuitBreakerOptions;
2371
+ circuitBreaker?: CircuitBreakerOption;
2202
2372
  /**
2203
2373
  * Reports retries and circuit-breaker state transitions as they happen.
2204
2374
  * Fire and forget: a throwing handler is caught and logged, and its
@@ -2296,7 +2466,7 @@ type CreateMiddlewareOptions = Omit<VernLLMMiddleware, 'wrap'> & {
2296
2466
  * error afterward, so `onError` never changes what the call itself
2297
2467
  * returns or throws, only what gets observed about it.
2298
2468
  */
2299
- declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMiddleware;
2469
+ export declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMiddleware;
2300
2470
  //#endregion
2301
2471
  //#region src/vernLLM.d.ts
2302
2472
  /**
@@ -2307,7 +2477,7 @@ declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMidd
2307
2477
  * tracking, and an optional response cache. All configurable, all opt-in
2308
2478
  * beyond sensible defaults.
2309
2479
  */
2310
- declare class VernLLM {
2480
+ export declare class VernLLM {
2311
2481
  private readonly logger;
2312
2482
  /**
2313
2483
  * One `CallExecutor` per provider target: index 0 is the primary,
@@ -2478,6 +2648,22 @@ declare class VernLLM {
2478
2648
  * @returns Every target's state, in chain order.
2479
2649
  */
2480
2650
  getCircuitStates(model?: string): TargetCircuitState[];
2651
+ /**
2652
+ * The live counterpart of `getRateLimitState`, asking the limiter for its current levels.
2653
+ *
2654
+ * @param target.index Which target to read. Defaults to the primary.
2655
+ * @returns This target's live rate limit levels, or `undefined` if that target has no limiter
2656
+ * configured.
2657
+ * @throws {RangeError} If `target.index` names no target.
2658
+ */
2659
+ readRateLimitState(target?: Pick<CircuitTarget, 'index'>): Promise<RateLimitState | undefined>;
2660
+ /**
2661
+ * The live counterpart of `getCircuitStates`, asking each breaker for its current state.
2662
+ *
2663
+ * @param model Which model bucket to read, for targets that isolate by model.
2664
+ * @returns Every target's state, in chain order.
2665
+ */
2666
+ readCircuitStates(model?: string): Promise<TargetCircuitState[]>;
2481
2667
  /**
2482
2668
  * Manually opens a target's breaker, e.g. to pull a provider out of
2483
2669
  * rotation ahead of known maintenance instead of waiting for it to fail.
@@ -2518,7 +2704,7 @@ declare class VernLLM {
2518
2704
  * `T` isn't a parameter here; pin it via `llm.call<T>(params)` as usual.
2519
2705
  * `defineCachedCallParams` is the `cachedCall()` counterpart.
2520
2706
  */
2521
- declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
2707
+ export declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
2522
2708
  /**
2523
2709
  * The `cachedCall()` counterpart to `defineCallParams`: preserves the
2524
2710
  * whole `{ cacheKey, ttl, call }` object, `call.tools` included, in one
@@ -2533,7 +2719,7 @@ declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
2533
2719
  * const result = await llm.cachedCall(params);
2534
2720
  * ```
2535
2721
  */
2536
- declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
2722
+ export declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
2537
2723
  //#endregion
2538
2724
  //#region src/adapters/internal/sse.d.ts
2539
2725
  /**
@@ -2563,14 +2749,14 @@ declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(par
2563
2749
  * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
2564
2750
  * how malformed JSON is handled elsewhere in VernLLM.
2565
2751
  */
2566
- declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
2752
+ export declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
2567
2753
  /**
2568
2754
  * Sentinel yielded by `parseSseStream` for a comment-only frame (no
2569
2755
  * `data:` payload), the mechanism providers use for SSE keep-alive
2570
2756
  * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
2571
2757
  * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
2572
2758
  */
2573
- declare const SSE_PING: unique symbol;
2759
+ export declare const SSE_PING: unique symbol;
2574
2760
  //#endregion
2575
2761
  //#region src/adapters/internal/imageFormat.d.ts
2576
2762
  /**
@@ -2817,7 +3003,7 @@ interface AnthropicAdapterOptions {
2817
3003
  * instead, which maps to a real constraint either way (native
2818
3004
  * `output_config.format` or a forced tool call).
2819
3005
  */
2820
- declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
3006
+ export declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
2821
3007
  //#endregion
2822
3008
  //#region src/adapters/gemini.d.ts
2823
3009
  /**
@@ -3029,7 +3215,7 @@ interface GeminiAdapterOptions {
3029
3215
  */
3030
3216
  thinkingLevelModels?: ModelCapabilityOverride;
3031
3217
  }
3032
- declare function fromGemini(client: GeminiClient, options?: GeminiAdapterOptions): LLMClient;
3218
+ export declare function fromGemini(client: GeminiClient, options?: GeminiAdapterOptions): LLMClient;
3033
3219
  //#endregion
3034
3220
  //#region src/adapters/bedrock.d.ts
3035
3221
  /** Bedrock Converse's supported inline image formats. */
@@ -3388,7 +3574,7 @@ interface AwsSendClient {
3388
3574
  * `finalizeResponse`'s `content` path exactly like the non-streaming
3389
3575
  * `create` branch above unwraps it.
3390
3576
  */
3391
- declare function fromBedrock(bedrockClient: BedrockConverseClient | AwsSendClient, options?: BedrockAdapterOptions): LLMClient;
3577
+ export declare function fromBedrock(bedrockClient: BedrockConverseClient | AwsSendClient, options?: BedrockAdapterOptions): LLMClient;
3392
3578
  //#endregion
3393
3579
  //#region src/adapters/fetch.d.ts
3394
3580
  /** The chat-completion-shaped request VernLLM builds internally */
@@ -3550,7 +3736,7 @@ interface FetchAdapterConfig {
3550
3736
  * `LLMError('validation')` instead of quietly using unrelated native
3551
3737
  * `fetch`.
3552
3738
  */
3553
- declare function fromFetch(config: FetchAdapterConfig): LLMClient;
3739
+ export declare function fromFetch(config: FetchAdapterConfig): LLMClient;
3554
3740
  //#endregion
3555
3741
  //#region src/adapters/openaiCompatible.d.ts
3556
3742
  /**
@@ -3617,7 +3803,7 @@ interface OpenAICompatibleAdapterOptions {
3617
3803
  */
3618
3804
  supportsWithResponse?: boolean;
3619
3805
  }
3620
- declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
3806
+ export declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
3621
3807
  /**
3622
3808
  * Named alias for the OpenAI SDK itself. A raw `new OpenAI(...)` instance
3623
3809
  * structurally matches most of `LLMClient`, but newer `openai` SDK major
@@ -3635,9 +3821,9 @@ declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibl
3635
3821
  * official `openai` package's client versus a fake or a test double.
3636
3822
  * Pass `supportsWithResponse: true` once you've confirmed it.
3637
3823
  */
3638
- declare const fromOpenAI: typeof fromOpenAICompatible;
3824
+ export declare const fromOpenAI: typeof fromOpenAICompatible;
3639
3825
  /** Groqs SDK matches the OpenAI wire format */
3640
- declare const fromGroq: typeof fromOpenAICompatible;
3826
+ export declare const fromGroq: typeof fromOpenAICompatible;
3641
3827
  /**
3642
3828
  * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
3643
3829
  * endpoint). Mistral supports `stream_options.include_usage` (added after
@@ -3645,88 +3831,88 @@ declare const fromGroq: typeof fromOpenAICompatible;
3645
3831
  * Mistral's changelog and streaming docs), so this is a plain alias like
3646
3832
  * the others, `supportsStreamUsage` defaults to `true`.
3647
3833
  */
3648
- declare const fromMistral: typeof fromOpenAICompatible;
3834
+ export declare const fromMistral: typeof fromOpenAICompatible;
3649
3835
  /** DeepSeeks API is OpenAI-compatible */
3650
- declare const fromDeepSeek: typeof fromOpenAICompatible;
3836
+ export declare const fromDeepSeek: typeof fromOpenAICompatible;
3651
3837
  /** Cerebras inference API is OpenAI-compatible */
3652
- declare const fromCerebras: typeof fromOpenAICompatible;
3838
+ export declare const fromCerebras: typeof fromOpenAICompatible;
3653
3839
  /** Together AIs API is OpenAI-compatible */
3654
- declare const fromTogether: typeof fromOpenAICompatible;
3840
+ export declare const fromTogether: typeof fromOpenAICompatible;
3655
3841
  /** Fireworks AIs API is OpenAI-compatible */
3656
- declare const fromFireworks: typeof fromOpenAICompatible;
3842
+ export declare const fromFireworks: typeof fromOpenAICompatible;
3657
3843
  /**
3658
3844
  * Ollama exposes an OpenAI-compatible endpoint at `/v1/chat/completions`
3659
3845
  * (as opposed to its native `/api/chat` format, which differs). Point an
3660
3846
  * OpenAI SDK instances `baseURL` at your Ollama server and pass it here:
3661
3847
  * this does not talk to Ollamas native API directly.
3662
3848
  */
3663
- declare const fromOllama: typeof fromOpenAICompatible;
3849
+ export declare const fromOllama: typeof fromOpenAICompatible;
3664
3850
  /** OpenRouter's API is OpenAI-compatible */
3665
- declare const fromOpenRouter: typeof fromOpenAICompatible;
3851
+ export declare const fromOpenRouter: typeof fromOpenAICompatible;
3666
3852
  /** Perplexity's API is OpenAI-compatible */
3667
- declare const fromPerplexity: typeof fromOpenAICompatible;
3853
+ export declare const fromPerplexity: typeof fromOpenAICompatible;
3668
3854
  /** DeepInfra's API is OpenAI-compatible */
3669
- declare const fromDeepInfra: typeof fromOpenAICompatible;
3855
+ export declare const fromDeepInfra: typeof fromOpenAICompatible;
3670
3856
  /** Novita's API is OpenAI-compatible */
3671
- declare const fromNovita: typeof fromOpenAICompatible;
3857
+ export declare const fromNovita: typeof fromOpenAICompatible;
3672
3858
  /** Hyperbolic's API is OpenAI-compatible */
3673
- declare const fromHyperbolic: typeof fromOpenAICompatible;
3859
+ export declare const fromHyperbolic: typeof fromOpenAICompatible;
3674
3860
  /** Moonshot's (Kimi) API is OpenAI-compatible */
3675
- declare const fromMoonshot: typeof fromOpenAICompatible;
3861
+ export declare const fromMoonshot: typeof fromOpenAICompatible;
3676
3862
  /** Zhipu's (GLM) API is OpenAI-compatible */
3677
- declare const fromZhipu: typeof fromOpenAICompatible;
3863
+ export declare const fromZhipu: typeof fromOpenAICompatible;
3678
3864
  /**
3679
3865
  * LM Studio exposes an OpenAI-compatible endpoint at `/v1/chat/completions`.
3680
3866
  * Point an OpenAI SDK instance's `baseURL` at your local LM Studio server.
3681
3867
  */
3682
- declare const fromLMStudio: typeof fromOpenAICompatible;
3868
+ export declare const fromLMStudio: typeof fromOpenAICompatible;
3683
3869
  /**
3684
3870
  * vLLM's OpenAI-compatible server mode exposes `/v1/chat/completions`.
3685
3871
  * Point an OpenAI SDK instance's `baseURL` at your vLLM server.
3686
3872
  */
3687
- declare const fromVLLM: typeof fromOpenAICompatible;
3873
+ export declare const fromVLLM: typeof fromOpenAICompatible;
3688
3874
  /** xAI's Grok API is OpenAI-compatible */
3689
- declare const fromXAI: typeof fromOpenAICompatible;
3875
+ export declare const fromXAI: typeof fromOpenAICompatible;
3690
3876
  /** NVIDIA NIM's hosted and self-hosted endpoints are OpenAI-compatible */
3691
- declare const fromNvidiaNIM: typeof fromOpenAICompatible;
3877
+ export declare const fromNvidiaNIM: typeof fromOpenAICompatible;
3692
3878
  /** Vercel AI Gateway is OpenAI-compatible */
3693
- declare const fromVercelAIGateway: typeof fromOpenAICompatible;
3879
+ export declare const fromVercelAIGateway: typeof fromOpenAICompatible;
3694
3880
  /** Cloudflare Workers AI exposes an OpenAI-compatible endpoint */
3695
- declare const fromCloudflareWorkersAI: typeof fromOpenAICompatible;
3881
+ export declare const fromCloudflareWorkersAI: typeof fromOpenAICompatible;
3696
3882
  /** Nebius AI Studio is OpenAI-compatible */
3697
- declare const fromNebius: typeof fromOpenAICompatible;
3883
+ export declare const fromNebius: typeof fromOpenAICompatible;
3698
3884
  /** SambaNova Cloud's API is OpenAI-compatible */
3699
- declare const fromSambaNova: typeof fromOpenAICompatible;
3885
+ export declare const fromSambaNova: typeof fromOpenAICompatible;
3700
3886
  /** Baseten's model hosting exposes an OpenAI-compatible endpoint */
3701
- declare const fromBaseten: typeof fromOpenAICompatible;
3887
+ export declare const fromBaseten: typeof fromOpenAICompatible;
3702
3888
  /** Featherless AI's API is OpenAI-compatible */
3703
- declare const fromFeatherless: typeof fromOpenAICompatible;
3889
+ export declare const fromFeatherless: typeof fromOpenAICompatible;
3704
3890
  /** Friendli AI's serving endpoint is OpenAI-compatible */
3705
- declare const fromFriendli: typeof fromOpenAICompatible;
3891
+ export declare const fromFriendli: typeof fromOpenAICompatible;
3706
3892
  /** SiliconFlow's API is OpenAI-compatible */
3707
- declare const fromSiliconFlow: typeof fromOpenAICompatible;
3893
+ export declare const fromSiliconFlow: typeof fromOpenAICompatible;
3708
3894
  /** Parasail's inference API is OpenAI-compatible */
3709
- declare const fromParasail: typeof fromOpenAICompatible;
3895
+ export declare const fromParasail: typeof fromOpenAICompatible;
3710
3896
  /** StepFun's API is OpenAI-compatible */
3711
- declare const fromStepFun: typeof fromOpenAICompatible;
3897
+ export declare const fromStepFun: typeof fromOpenAICompatible;
3712
3898
  /** MiniMax's API is OpenAI-compatible */
3713
- declare const fromMiniMax: typeof fromOpenAICompatible;
3899
+ export declare const fromMiniMax: typeof fromOpenAICompatible;
3714
3900
  /** Lambda Labs' Inference API is OpenAI-compatible */
3715
- declare const fromLambdaLabs: typeof fromOpenAICompatible;
3901
+ export declare const fromLambdaLabs: typeof fromOpenAICompatible;
3716
3902
  /** Snowflake Cortex's LLM endpoint is OpenAI-compatible */
3717
- declare const fromSnowflakeCortex: typeof fromOpenAICompatible;
3903
+ export declare const fromSnowflakeCortex: typeof fromOpenAICompatible;
3718
3904
  /** Anyscale Endpoints' API is OpenAI-compatible */
3719
- declare const fromAnyscale: typeof fromOpenAICompatible;
3905
+ export declare const fromAnyscale: typeof fromOpenAICompatible;
3720
3906
  /** Lepton AI's inference API is OpenAI-compatible */
3721
- declare const fromLepton: typeof fromOpenAICompatible;
3907
+ export declare const fromLepton: typeof fromOpenAICompatible;
3722
3908
  /** Inference.net's API is OpenAI-compatible */
3723
- declare const fromInferenceNet: typeof fromOpenAICompatible;
3909
+ export declare const fromInferenceNet: typeof fromOpenAICompatible;
3724
3910
  /** Infermatic's API is OpenAI-compatible */
3725
- declare const fromInfermatic: typeof fromOpenAICompatible;
3911
+ export declare const fromInfermatic: typeof fromOpenAICompatible;
3726
3912
  /** AtlasCloud's inference API is OpenAI-compatible */
3727
- declare const fromAtlasCloud: typeof fromOpenAICompatible;
3913
+ export declare const fromAtlasCloud: typeof fromOpenAICompatible;
3728
3914
  /** 01.AI's (Yi models) API is OpenAI-compatible */
3729
- declare const from01AI: typeof fromOpenAICompatible;
3915
+ export declare const from01AI: typeof fromOpenAICompatible;
3730
3916
  //#endregion
3731
- export { type AnthropicClient, type AssistantContent, type AttemptContext, type BedrockConverseClient, type CacheAdapter, type CachedCallParams, type CachedConditionalToolCallParams, type CachedJsonModeDisabledCallParams, type CachedJsonModeEnabledCallParams, type CachedStreamCallParams, type CachedStreamConditionalToolCallParams, type CachedStreamJsonModeDisabledCallParams, type CachedStreamJsonModeEnabledCallParams, type CachedStreamToolCallParams, type CachedToolCallParams, type CallMeta, type CallParams, type CallResult, type CallWithToolsResult, CircuitBreaker, type CircuitBreakerOptions, type CircuitState, type CircuitTarget, type ConditionalToolCallParams, ConsecutiveTripping, ConsoleLogger, type ContentBlock, type ContentResult, type ConversationTurn, type CooldownBackoff, type CreateMiddlewareOptions, type DuplicateToolNamesIssue, type EvictionOption, type ExponentialBackoffOptions, type FallbackAttempt, FallbackExhaustedError, type FallbackOn, type FallbackTarget, type FetchAdapterConfig, type GeminiClient, type HistoryToolResultIssue, type ImageBlock, InMemoryCacheAdapter, type JsonModeDisabledCallParams, type JsonModeEnabledCallParams, type JsonSchemaSpec, type JsonValue, type LLMClient, LLMError, type LLMErrorCode, type LLMErrorIssuesByCode, type LLMErrorSnapshot, type LLMErrorType, type LLMRequestShape, type LLMRequestSnapshot, type Logger, type MiddlewareCapabilities, type MiddlewareContext, type MiddlewareContextBase, type MiddlewareStateBag, type MiddlewareStateKey, NormalizedCacheAdapter, type OnEvent, type OnUsage, type PreDispatchContext, type RateLimitAcquireResult, type RateLimitOptions, type RateLimitReason, type RateLimitState, RateLimiter, type RateLimiterAdapter, type RefundUsage, type ReserveUsage, type RetryAttempt, RetryBudget, type RetryBudgetOptions, RollingTripping, SSE_PING, type SchemaLike, type StreamCallResult, type StreamChunk, type StreamEnabledCallParams, type StreamJsonModeDisabledCallParams, type StreamJsonModeEnabledCallParams, type TargetCircuitState, type TextBlock, TieredCacheAdapter, type TokenUsage, type ToolCall, type ToolCallResult, type ToolChoice, type ToolDefinition, type ToolEnabledCallParams, type ToolIssue, type ToolResult, type ToolsDisabledCallParams, type TrippingPolicy, type UnknownToolChoiceIssue, type UnsupportedCapabilityIssue, VernLLM, type VernLLMEvent, type VernLLMMiddleware, type VernLLMOptions, type WireCallRequest, type WireCallRequestPatch, type WireMessage, type WireRequest, type WireResponseFormat, type WireStreamChunk, type WireTool, type WireToolCall, type WireToolChoice, createMiddleware, createMiddlewareStateBag, createStateKey, defaultEstimateTokens, defaultFallbackOn, defineCachedCallParams, defineCallParams, defineTool, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAI, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, hasIssues, isFallbackExhaustedError, isLLMError, isStreamResult, isToolCallResult, metaRef, parseSseStream };
3917
+ export type { AnthropicClient, AssistantContent, AttemptContext, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedConditionalToolCallParams, CachedJsonModeDisabledCallParams, CachedJsonModeEnabledCallParams, CachedStreamCallParams, CachedStreamConditionalToolCallParams, CachedStreamJsonModeDisabledCallParams, CachedStreamJsonModeEnabledCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallMeta, CallParams, CallResult, CallWithToolsResult, CircuitBreakerAdapter, CircuitBreakerCallContext, CircuitBreakerOptions, CircuitBreakerStateChangeHandler, CircuitState, CircuitTarget, ConditionalToolCallParams, ContentBlock, ContentResult, ConversationTurn, CooldownBackoff, CreateMiddlewareOptions, DuplicateToolNamesIssue, EvictionOption, ExponentialBackoffOptions, FallbackAttempt, FallbackOn, FallbackTarget, FetchAdapterConfig, GeminiClient, HistoryToolResultIssue, ImageBlock, JsonModeDisabledCallParams, JsonModeEnabledCallParams, JsonSchemaSpec, JsonValue, LLMClient, LLMErrorCode, LLMErrorIssuesByCode, LLMErrorSnapshot, LLMErrorType, LLMRequestShape, LLMRequestSnapshot, Logger, MiddlewareCapabilities, MiddlewareContext, MiddlewareContextBase, MiddlewareRef, MiddlewareStateBag, MiddlewareStateKey, OnEvent, OnUsage, PreDispatchContext, RateLimitAcquireResult, RateLimitOptions, RateLimitReason, RateLimitState, RateLimiterAdapter, RefundUsage, RequiredMiddlewareRef, ReserveUsage, RetryAttempt, RetryBudgetOptions, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, StreamJsonModeDisabledCallParams, StreamJsonModeEnabledCallParams, TargetCircuitState, TextBlock, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolIssue, ToolResult, ToolsDisabledCallParams, TrippingPolicy, UnknownToolChoiceIssue, UnsupportedCapabilityIssue, VernLLMEvent, VernLLMMiddleware, VernLLMOptions, WireCallRequest, WireCallRequestPatch, WireMessage, WireRequest, WireResponseFormat, WireStreamChunk, WireTool, WireToolCall, WireToolChoice };
3732
3918
  //# sourceMappingURL=index.d.cts.map