vern-llm 2.7.0 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -8
- package/dist/index.cjs +6 -6
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +359 -173
- package/dist/index.d.cts.map +1 -1
- package/dist/index.d.mts +359 -173
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +6 -6
- package/dist/index.mjs.map +1 -1
- package/package.json +14 -15
package/dist/index.d.mts
CHANGED
|
@@ -147,7 +147,7 @@ interface LLMErrorOptions {
|
|
|
147
147
|
/** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
|
|
148
148
|
attempts?: RetryAttempt[];
|
|
149
149
|
}
|
|
150
|
-
declare class LLMError extends Error {
|
|
150
|
+
export declare class LLMError extends Error {
|
|
151
151
|
type: LLMErrorType;
|
|
152
152
|
status?: number;
|
|
153
153
|
issues?: unknown;
|
|
@@ -203,7 +203,7 @@ declare class LLMError extends Error {
|
|
|
203
203
|
*/
|
|
204
204
|
toJSON(): Record<string, unknown>;
|
|
205
205
|
}
|
|
206
|
-
declare function isLLMError(err: unknown): err is LLMError;
|
|
206
|
+
export declare function isLLMError(err: unknown): err is LLMError;
|
|
207
207
|
/**
|
|
208
208
|
* Narrows `err.issues` to the exact shape {@link LLMErrorIssuesByCode} maps
|
|
209
209
|
* `code` to, for any code listed there. `code` stays the only discriminator
|
|
@@ -216,7 +216,7 @@ declare function isLLMError(err: unknown): err is LLMError;
|
|
|
216
216
|
* }
|
|
217
217
|
* ```
|
|
218
218
|
*/
|
|
219
|
-
declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
|
|
219
|
+
export declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
|
|
220
220
|
code: C;
|
|
221
221
|
issues: LLMErrorIssuesByCode[C];
|
|
222
222
|
};
|
|
@@ -241,7 +241,7 @@ type EvictionOption = 'fifo' | 'lru';
|
|
|
241
241
|
* Trivial default so the package works out of the box with no external deps.
|
|
242
242
|
* Not shared across processes, swap in Redis/Upstash/etc for production.
|
|
243
243
|
*/
|
|
244
|
-
declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
244
|
+
export declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
245
245
|
private readonly maxSize;
|
|
246
246
|
private store;
|
|
247
247
|
private readonly eviction;
|
|
@@ -258,7 +258,7 @@ declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
|
258
258
|
/**
|
|
259
259
|
* Normalizes keys before caching to avoid duplicate entries from formatting differences.
|
|
260
260
|
*/
|
|
261
|
-
declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
261
|
+
export declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
262
262
|
private readonly inner;
|
|
263
263
|
constructor(inner?: CacheAdapter<T>);
|
|
264
264
|
private normalize;
|
|
@@ -274,7 +274,7 @@ declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
|
274
274
|
* Two-tier cache with fast local L1 and shared L2.
|
|
275
275
|
* L2 hits are promoted back to L1.
|
|
276
276
|
*/
|
|
277
|
-
declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
277
|
+
export declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
278
278
|
private readonly l1;
|
|
279
279
|
private readonly l2;
|
|
280
280
|
private readonly l1Ttl?;
|
|
@@ -293,6 +293,89 @@ declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
|
293
293
|
delete(key: string): Promise<void>;
|
|
294
294
|
}
|
|
295
295
|
//#endregion
|
|
296
|
+
//#region src/logger.d.ts
|
|
297
|
+
interface Logger {
|
|
298
|
+
debug(message: string): void;
|
|
299
|
+
warn(message: string): void;
|
|
300
|
+
error(message: string, meta?: Record<string, unknown>): void;
|
|
301
|
+
}
|
|
302
|
+
/**
|
|
303
|
+
* Default logger. `debug` is gated by the `debug` option on VernLLM
|
|
304
|
+
* warn/error always fire since they indicate real problems (retries, cache failures)
|
|
305
|
+
*/
|
|
306
|
+
export declare class ConsoleLogger implements Logger {
|
|
307
|
+
private debugEnabled;
|
|
308
|
+
constructor(debugEnabled: boolean);
|
|
309
|
+
debug(message: string): void;
|
|
310
|
+
warn(message: string): void;
|
|
311
|
+
error(message: string, meta?: Record<string, unknown>): void;
|
|
312
|
+
}
|
|
313
|
+
//#endregion
|
|
314
|
+
//#region src/types/usage.d.ts
|
|
315
|
+
type ReserveUsage = (params: {
|
|
316
|
+
coalesced: boolean;
|
|
317
|
+
signal?: AbortSignal;
|
|
318
|
+
}) => Promise<void>;
|
|
319
|
+
type RefundUsage = (params: {
|
|
320
|
+
coalesced: boolean;
|
|
321
|
+
signal?: AbortSignal;
|
|
322
|
+
}) => Promise<void>;
|
|
323
|
+
/**
|
|
324
|
+
* The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
|
|
325
|
+
* and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
|
|
326
|
+
* has one definition instead of being redeclared at each use site.
|
|
327
|
+
*/
|
|
328
|
+
interface UsageHooks {
|
|
329
|
+
/**
|
|
330
|
+
* Reserves usage before the request. Failures become
|
|
331
|
+
* LLMError('quota_exceeded').
|
|
332
|
+
*/
|
|
333
|
+
reserveUsage?: ReserveUsage;
|
|
334
|
+
/**
|
|
335
|
+
* Refunds usage after a failed call if reservation succeeded.
|
|
336
|
+
*/
|
|
337
|
+
refundUsage?: RefundUsage;
|
|
338
|
+
}
|
|
339
|
+
interface TokenUsage {
|
|
340
|
+
promptTokens: number;
|
|
341
|
+
completionTokens: number;
|
|
342
|
+
totalTokens: number;
|
|
343
|
+
/**
|
|
344
|
+
* Tokens spent on internal reasoning, a subset of `completionTokens`,
|
|
345
|
+
* never added on top of it. Undefined when the provider's response
|
|
346
|
+
* doesn't report a separate reasoning figure, e.g. Bedrock Converse
|
|
347
|
+
* without an explicit `additionalModelResponseFieldPaths` request.
|
|
348
|
+
*/
|
|
349
|
+
reasoningTokens?: number;
|
|
350
|
+
requestId: string;
|
|
351
|
+
model: string;
|
|
352
|
+
/**
|
|
353
|
+
* The provider target that produced this usage. See `VernLLMOptions['name']`,
|
|
354
|
+
* default `'primary'`. Optional so consumers constructing a `TokenUsage`
|
|
355
|
+
* themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
|
|
356
|
+
* populates it. Absent means the same as `'primary'` if you need a value.
|
|
357
|
+
*/
|
|
358
|
+
provider?: string;
|
|
359
|
+
/**
|
|
360
|
+
* Whether this usage came from a fallback target rather than the
|
|
361
|
+
* primary. Optional for the same reason `provider` is: `VernLLM`
|
|
362
|
+
* always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
|
|
363
|
+
* isn't forced to.
|
|
364
|
+
*/
|
|
365
|
+
usedFallback?: boolean;
|
|
366
|
+
}
|
|
367
|
+
type OnUsage = (usage: TokenUsage) => void;
|
|
368
|
+
/**
|
|
369
|
+
* Called when a provider response arrives but VernLLM's own post-processing
|
|
370
|
+
* then fails, after usage data was already present in that response. Covers
|
|
371
|
+
* any error thrown after usage extraction, not just parse/validation, since
|
|
372
|
+
* everything in that path only runs once a response, and real spend, has
|
|
373
|
+
* already arrived. Fires once per failed attempt with extractable usage,
|
|
374
|
+
* never for transport failures, where no response means no honest number
|
|
375
|
+
* to report.
|
|
376
|
+
*/
|
|
377
|
+
type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
|
|
378
|
+
//#endregion
|
|
296
379
|
//#region src/types/events.d.ts
|
|
297
380
|
/**
|
|
298
381
|
* Reports what happened during a call. Fire and forget, mirroring
|
|
@@ -359,6 +442,26 @@ type VernLLMEvent = {
|
|
|
359
442
|
hook: 'transform' | 'wrap_short_circuit' | 'enabled_skip';
|
|
360
443
|
/** For `hook: 'transform'` only: which top-level fields the merged patch touched. */
|
|
361
444
|
patchedFields?: string[];
|
|
445
|
+
} | {
|
|
446
|
+
/**
|
|
447
|
+
* Reported once a call fully succeeds. Same data `VernLLMOptions.onUsage`
|
|
448
|
+
* receives; that option is sugar over this event, not a second
|
|
449
|
+
* reporting path, see `makeEventReporter`.
|
|
450
|
+
*/
|
|
451
|
+
kind: 'usage';
|
|
452
|
+
requestId: string;
|
|
453
|
+
usage: TokenUsage;
|
|
454
|
+
} | {
|
|
455
|
+
/**
|
|
456
|
+
* A provider response arrived, carrying real usage, and VernLLM's own
|
|
457
|
+
* post-processing then failed. Fires once per failed attempt with
|
|
458
|
+
* extractable usage, matching `VernLLMOptions.onUsageFailure`'s own
|
|
459
|
+
* granularity, which this event is sugar over, not a second path.
|
|
460
|
+
*/
|
|
461
|
+
kind: 'usage_failure';
|
|
462
|
+
requestId: string;
|
|
463
|
+
usage: TokenUsage;
|
|
464
|
+
error: LLMError;
|
|
362
465
|
};
|
|
363
466
|
type OnEvent = (event: VernLLMEvent) => void;
|
|
364
467
|
//#endregion
|
|
@@ -372,6 +475,19 @@ interface MiddlewareCapabilities {
|
|
|
372
475
|
*/
|
|
373
476
|
supportsJsonObjectMode: boolean;
|
|
374
477
|
}
|
|
478
|
+
/**
|
|
479
|
+
* Not exported. Distinguishes `MiddlewareStateKey<T>` from
|
|
480
|
+
* `MiddlewareRef` and from a plain `{ debugName }` object literal at
|
|
481
|
+
* the type level, even though all three have the identical runtime
|
|
482
|
+
* shape. Without this, `MiddlewareStateKey<T>`/`MiddlewareRef` are
|
|
483
|
+
* structurally just `{ debugName: string }`, so TypeScript would treat
|
|
484
|
+
* a state key as a valid middleware ref (or vice versa), and would let
|
|
485
|
+
* anyone hand-write `{ debugName: 'auth' }` in place of a real
|
|
486
|
+
* `createMiddlewareRef` result. Neither is possible once this brand is
|
|
487
|
+
* required: only `createStateKey`, which alone has access to this
|
|
488
|
+
* symbol, can produce a value satisfying `MiddlewareStateKey<T>`.
|
|
489
|
+
*/
|
|
490
|
+
declare const stateKeyBrand: unique symbol;
|
|
375
491
|
/**
|
|
376
492
|
* A typed reference to one slot in `ctx.state`. Create one with
|
|
377
493
|
* `createStateKey`, export it, and import the same reference wherever
|
|
@@ -382,6 +498,7 @@ interface MiddlewareCapabilities {
|
|
|
382
498
|
*/
|
|
383
499
|
interface MiddlewareStateKey<T> {
|
|
384
500
|
readonly debugName: string;
|
|
501
|
+
readonly [stateKeyBrand]: true;
|
|
385
502
|
/**
|
|
386
503
|
* Never set at runtime; exists purely so `T` is actually used
|
|
387
504
|
* somewhere in this interface's shape (a phantom type), which is what
|
|
@@ -392,7 +509,43 @@ interface MiddlewareStateKey<T> {
|
|
|
392
509
|
readonly __phantom?: T;
|
|
393
510
|
}
|
|
394
511
|
/** Creates a new, distinct `MiddlewareStateKey`. `debugName` is used only in log lines and the `'middleware'` event; it never affects equality. */
|
|
395
|
-
declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
|
|
512
|
+
export declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
|
|
513
|
+
/** Not exported. See `stateKeyBrand`; same reasoning, distinct symbol, so the two token types can't be cross-assigned either. */
|
|
514
|
+
declare const middlewareRefBrand: unique symbol;
|
|
515
|
+
/**
|
|
516
|
+
* A typed reference to one middleware's identity, for `runsAfter`/
|
|
517
|
+
* `runsBefore` to target. Purely an ordering concern: unlike `name`,
|
|
518
|
+
* `ref` is never used as a display label anywhere (`name` still covers
|
|
519
|
+
* that), only as a `runsAfter`/`runsBefore` match target. Create one
|
|
520
|
+
* with `createMiddlewareRef`, export it from the package that owns the
|
|
521
|
+
* middleware, and have any dependent import the same reference instead
|
|
522
|
+
* of typing a matching `name` string. Same reasoning as
|
|
523
|
+
* `MiddlewareStateKey`: a typo becomes a missing import, a compile
|
|
524
|
+
* error, instead of a silently unresolved (or worse, silently
|
|
525
|
+
* colliding) string.
|
|
526
|
+
*/
|
|
527
|
+
interface MiddlewareRef {
|
|
528
|
+
readonly debugName: string;
|
|
529
|
+
readonly [middlewareRefBrand]: true;
|
|
530
|
+
}
|
|
531
|
+
/** Creates a new, distinct `MiddlewareRef`. `debugName` is used only in error messages when a reference doesn't resolve; it never affects equality, so two refs with the same `debugName` never collide. */
|
|
532
|
+
export declare function createMiddlewareRef(debugName: string): MiddlewareRef;
|
|
533
|
+
/**
|
|
534
|
+
* A `runsAfter`/`runsBefore` entry that escalates an unresolved
|
|
535
|
+
* reference from a warning to a construction-time throw. Wrap a
|
|
536
|
+
* `MiddlewareRef` with `requireRef` when the dependency isn't optional:
|
|
537
|
+
* a bare `MiddlewareRef` in `runsAfter`/`runsBefore` means "order
|
|
538
|
+
* relative to this if it's registered," which is the right default for
|
|
539
|
+
* a dependency a third party may reasonably not have installed. A
|
|
540
|
+
* `RequiredMiddlewareRef` means "this middleware must not run without
|
|
541
|
+
* that dependency having already run". The app should fail to start
|
|
542
|
+
* rather than run with a silently-missing ordering guarantee.
|
|
543
|
+
*/
|
|
544
|
+
interface RequiredMiddlewareRef {
|
|
545
|
+
readonly ref: MiddlewareRef;
|
|
546
|
+
}
|
|
547
|
+
/** Wraps `ref` so `runsAfter`/`runsBefore` throws at `VernLLM` construction time if it doesn't resolve, instead of warning and continuing. */
|
|
548
|
+
export declare function requireRef(ref: MiddlewareRef): RequiredMiddlewareRef;
|
|
396
549
|
/**
|
|
397
550
|
* Typed, per-logical-call storage two middleware can deliberately share a
|
|
398
551
|
* value through (a span ID one sets, another reads). Backed by a plain
|
|
@@ -404,7 +557,7 @@ interface MiddlewareStateBag {
|
|
|
404
557
|
set<T>(key: MiddlewareStateKey<T>, value: T): void;
|
|
405
558
|
}
|
|
406
559
|
/** A plain, `Map`-backed `MiddlewareStateBag`. */
|
|
407
|
-
declare function createMiddlewareStateBag(): MiddlewareStateBag;
|
|
560
|
+
export declare function createMiddlewareStateBag(): MiddlewareStateBag;
|
|
408
561
|
/** Fields every `MiddlewareContext` variant carries, regardless of `stage`. */
|
|
409
562
|
interface MiddlewareContextBase {
|
|
410
563
|
requestId: string;
|
|
@@ -554,22 +707,36 @@ interface CallResult<T = unknown> {
|
|
|
554
707
|
interface VernLLMMiddleware {
|
|
555
708
|
/** Used in log lines and the `'middleware'` event. Defaults to this entry's array position when omitted. */
|
|
556
709
|
name?: string;
|
|
710
|
+
/**
|
|
711
|
+
* This entry's own identity, purely for another middleware's
|
|
712
|
+
* `runsAfter`/`runsBefore` to target. Create with `createMiddlewareRef`,
|
|
713
|
+
* export it, and have a dependent import the same reference. Optional:
|
|
714
|
+
* only needed if something else must be able to depend on this
|
|
715
|
+
* specific entry. Unrelated to `name`: `ref` is never shown in logs,
|
|
716
|
+
* `name` is never matched against for ordering.
|
|
717
|
+
*/
|
|
718
|
+
ref?: MiddlewareRef;
|
|
557
719
|
/** Sort key for composition order, ascending, ties broken by array order. See the middleware docs for what "lower runs first" means for `wrap`. */
|
|
558
720
|
priority?: number;
|
|
559
721
|
/**
|
|
560
|
-
*
|
|
561
|
-
* `priority` alone can't express.
|
|
562
|
-
*
|
|
563
|
-
*
|
|
564
|
-
*
|
|
565
|
-
*
|
|
722
|
+
* Other middleware this entry must run after, breaking ties
|
|
723
|
+
* `priority` alone can't express. Matched by `ref` identity, so a
|
|
724
|
+
* typo or a stale copy simply fails to resolve instead of silently
|
|
725
|
+
* matching the wrong entry. A bare `MiddlewareRef` that doesn't
|
|
726
|
+
* resolve is dropped, not an error, since a third party may
|
|
727
|
+
* reasonably reference a well known middleware that isn't installed
|
|
728
|
+
* everywhere; wrap it with `requireRef` to make that same target
|
|
729
|
+
* mandatory instead, throwing at `VernLLM` construction time if it's
|
|
730
|
+
* missing. A cycle across `runsAfter`/`runsBefore` always throws,
|
|
731
|
+
* regardless of whether any individual entry is required.
|
|
566
732
|
*/
|
|
567
|
-
runsAfter?:
|
|
733
|
+
runsAfter?: (MiddlewareRef | RequiredMiddlewareRef)[];
|
|
568
734
|
/**
|
|
569
|
-
*
|
|
570
|
-
*
|
|
735
|
+
* Other middleware this entry must run before. See `runsAfter`; a
|
|
736
|
+
* bare reference is dropped if unresolved, a `requireRef`-wrapped one
|
|
737
|
+
* throws.
|
|
571
738
|
*/
|
|
572
|
-
runsBefore?:
|
|
739
|
+
runsBefore?: (MiddlewareRef | RequiredMiddlewareRef)[];
|
|
573
740
|
/**
|
|
574
741
|
* Pins this entry's slot in `wrap` nesting only, independent of
|
|
575
742
|
* `priority`/`runsAfter`/`runsBefore`, which still govern
|
|
@@ -611,17 +778,21 @@ interface CircuitBreakerCallContext {
|
|
|
611
778
|
/** Omitted for calls before any attempt exists, like `assertClosed`'s pre-dispatch check. */
|
|
612
779
|
attempt?: number;
|
|
613
780
|
}
|
|
781
|
+
/**
|
|
782
|
+
* Fires after every real state change, never a no-op transition. `model`
|
|
783
|
+
* is the resolved model of whichever call triggered it. With
|
|
784
|
+
* `isolateByModel` off, failures are still counted across every model.
|
|
785
|
+
* Shared by `CircuitBreakerOptions` and `CircuitBreakerAdapter`, so a
|
|
786
|
+
* custom adapter reports state changes the same way the built in
|
|
787
|
+
* `CircuitBreaker` does.
|
|
788
|
+
*/
|
|
789
|
+
type CircuitBreakerStateChangeHandler = (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
|
|
614
790
|
interface CircuitBreakerOptions {
|
|
615
791
|
/** Consecutive failures before the circuit opens, default 5 */
|
|
616
792
|
threshold?: number;
|
|
617
793
|
/** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
|
|
618
794
|
cooldownMs?: number;
|
|
619
|
-
|
|
620
|
-
* Fires after every real state change, never a no-op transition. `model`
|
|
621
|
-
* is the resolved model of whichever call triggered it. With
|
|
622
|
-
* `isolateByModel` off, failures are still counted across every model.
|
|
623
|
-
*/
|
|
624
|
-
onStateChange?: (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
|
|
795
|
+
onStateChange?: CircuitBreakerStateChangeHandler;
|
|
625
796
|
/**
|
|
626
797
|
* Track a separate circuit per resolved model instead of one shared
|
|
627
798
|
* circuit. Default false. A call that omits `model` falls into one
|
|
@@ -681,7 +852,7 @@ interface TrippingPolicy {
|
|
|
681
852
|
*/
|
|
682
853
|
forget?(key: string): void;
|
|
683
854
|
}
|
|
684
|
-
declare class ConsecutiveTripping implements TrippingPolicy {
|
|
855
|
+
export declare class ConsecutiveTripping implements TrippingPolicy {
|
|
685
856
|
private readonly threshold;
|
|
686
857
|
private failuresByKey;
|
|
687
858
|
constructor(threshold: number);
|
|
@@ -690,7 +861,7 @@ declare class ConsecutiveTripping implements TrippingPolicy {
|
|
|
690
861
|
reset(key: string): void;
|
|
691
862
|
forget(key: string): void;
|
|
692
863
|
}
|
|
693
|
-
declare class RollingTripping implements TrippingPolicy {
|
|
864
|
+
export declare class RollingTripping implements TrippingPolicy {
|
|
694
865
|
private readonly windowMs;
|
|
695
866
|
private readonly minCalls;
|
|
696
867
|
private readonly failureRatio;
|
|
@@ -713,14 +884,65 @@ type TrippingOption = {
|
|
|
713
884
|
failureRatio: number;
|
|
714
885
|
} | TrippingPolicy;
|
|
715
886
|
type CircuitState = 'closed' | 'open' | 'half-open';
|
|
887
|
+
/**
|
|
888
|
+
* What VernLLM's dispatch layer needs from a breaker. `CircuitBreaker`
|
|
889
|
+
* implements this; a caller wanting cross process coordination can hand
|
|
890
|
+
* over their own instance instead.
|
|
891
|
+
*
|
|
892
|
+
* `assertClosed`, `recordSuccess`, `recordFailure`, and `onStateChange`
|
|
893
|
+
* are required, mirroring `RateLimiterAdapter`'s four required methods.
|
|
894
|
+
* `onStateChange` is required so `circuit_state` events can't go
|
|
895
|
+
* silently missing; a no-op `() => {}` is fine if you don't care.
|
|
896
|
+
*
|
|
897
|
+
* `getState`, `getFailureBreakdown`, `isolateByModel`, `open`, and
|
|
898
|
+
* `close` are optional. Omitting one makes the matching call a no-op
|
|
899
|
+
* or return `undefined`/`false`, same as no breaker configured.
|
|
900
|
+
* `open`/`close` are optional since they let VernLLM force a
|
|
901
|
+
* transition, control a distributed adapter may not want to grant.
|
|
902
|
+
*/
|
|
903
|
+
interface CircuitBreakerAdapter {
|
|
904
|
+
/** Throws when the circuit is open (or half open with no trial slot free) for `model`. */
|
|
905
|
+
assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
|
|
906
|
+
recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
|
|
907
|
+
/** `code`, when present, is the failing call's `LLMErrorCode`. */
|
|
908
|
+
recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
|
|
909
|
+
getState?(model?: string): CircuitState;
|
|
910
|
+
/** Failure counts by `LLMErrorCode` for `model`'s bucket, `'unknown'` for one that carried no code. */
|
|
911
|
+
getFailureBreakdown?(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
|
|
912
|
+
/** Whether this adapter tracks failures per model, mirroring `CircuitBreakerOptions.isolateByModel`. Read by `warnIfModelUnsupported`'s diagnostic warning and by `VernLLM.getCircuitStates()`'s public output; omit if the notion doesn't apply to your adapter, `false` is assumed. */
|
|
913
|
+
isolateByModel?: boolean;
|
|
914
|
+
/** Manually opens the circuit, as if enough consecutive failures had just happened. Optional: an adapter that doesn't want external callers forcing a transition can omit it. */
|
|
915
|
+
open?(model?: string, context?: CircuitBreakerCallContext): void;
|
|
916
|
+
/** Manually closes the circuit, without requiring a real success first. Same opt-in reasoning as `open`. */
|
|
917
|
+
close?(model?: string, context?: CircuitBreakerCallContext): void;
|
|
918
|
+
/** Gives back a half-open trial slot when a call ends without `recordSuccess` or `recordFailure`. Idempotent, and a no-op for a call that holds no slot. */
|
|
919
|
+
releaseTrial?(model?: string, context?: CircuitBreakerCallContext): void;
|
|
920
|
+
/** Awaited right before `assertClosed` to refresh local state. Never blocks or fails a call: a rejection or `prepareTimeoutMs` is logged and the call carries on. */
|
|
921
|
+
prepare?(model?: string, context?: CircuitBreakerCallContext): Promise<void>;
|
|
922
|
+
/** How long to wait for `prepare`, in ms. Default 1000. */
|
|
923
|
+
prepareTimeoutMs?: number;
|
|
924
|
+
/** Live counterpart of `getState`, read by `VernLLM.readCircuitStates()`. */
|
|
925
|
+
readState?(model?: string): Promise<CircuitState>;
|
|
926
|
+
/** Receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
|
|
927
|
+
setLogger?(logger: Logger): void;
|
|
928
|
+
/**
|
|
929
|
+
* Called after every real state change, never a no-op transition. VernLLM
|
|
930
|
+
* wraps it the same way it wraps the built in `CircuitBreaker`'s
|
|
931
|
+
* `onStateChange`: every call still reports a `circuit_state` event
|
|
932
|
+
* first, then this hook is chained after that, wrapped so a throw here
|
|
933
|
+
* can't break the call that triggered it.
|
|
934
|
+
*/
|
|
935
|
+
onStateChange: CircuitBreakerStateChangeHandler;
|
|
936
|
+
}
|
|
716
937
|
/**
|
|
717
938
|
* Per retry VernLLM-instance circuit breaker. Tracks consecutive failures
|
|
718
939
|
* across calls. Once the threshold is hit, short-circuits new calls with
|
|
719
940
|
* LLMError('circuit_open') until the cooldown elapses and a trial succeeds.
|
|
720
941
|
*/
|
|
721
|
-
declare class CircuitBreaker {
|
|
942
|
+
export declare class CircuitBreaker implements CircuitBreakerAdapter {
|
|
722
943
|
private readonly cooldownMs;
|
|
723
|
-
|
|
944
|
+
/** Satisfies `CircuitBreakerAdapter.onStateChange`, required there. Defaults to a no-op when `options.onStateChange` is omitted. */
|
|
945
|
+
readonly onStateChange: CircuitBreakerStateChangeHandler;
|
|
724
946
|
/** Whether this breaker tracks failures per model instead of one shared circuit. */
|
|
725
947
|
readonly isolateByModel: boolean;
|
|
726
948
|
private readonly halfOpenProbes;
|
|
@@ -739,6 +961,14 @@ declare class CircuitBreaker {
|
|
|
739
961
|
recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
|
|
740
962
|
/** `code`, when present, is the failing `LLMError`'s `code`. Missing attributes to `'unknown'`. */
|
|
741
963
|
recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
|
|
964
|
+
/**
|
|
965
|
+
* Gives back the half-open trial slot `context`'s call claimed, when
|
|
966
|
+
* that call ended without recording an outcome. No-op without a
|
|
967
|
+
* `context`, when the bucket isn't half-open, or when the call's permit
|
|
968
|
+
* is stale or already spent (an outcome was recorded), so calling it
|
|
969
|
+
* defensively on every failure path is safe.
|
|
970
|
+
*/
|
|
971
|
+
releaseTrial(model?: string, context?: CircuitBreakerCallContext): void;
|
|
742
972
|
/** With `isolateByModel` off, `model` is ignored and the shared circuit's state is returned. */
|
|
743
973
|
getState(model?: string): CircuitState;
|
|
744
974
|
/** Failure counts by `LLMErrorCode` for `model`'s bucket. Returned as a plain object copy. */
|
|
@@ -801,7 +1031,7 @@ interface RetryBudgetOptions {
|
|
|
801
1031
|
* the same primitive `RollingTripping` is built on, rather than a second
|
|
802
1032
|
* hand rolled window.
|
|
803
1033
|
*/
|
|
804
|
-
declare class RetryBudget {
|
|
1034
|
+
export declare class RetryBudget {
|
|
805
1035
|
private readonly options;
|
|
806
1036
|
private readonly ratio;
|
|
807
1037
|
constructor(options: RetryBudgetOptions);
|
|
@@ -820,7 +1050,7 @@ declare class RetryBudget {
|
|
|
820
1050
|
};
|
|
821
1051
|
}
|
|
822
1052
|
//#endregion
|
|
823
|
-
//#region src/internal/utils/rateLimitHint.utils.d.ts
|
|
1053
|
+
//#region src/internal/utils/rate-limit/rateLimitHint.utils.d.ts
|
|
824
1054
|
/** A normalized read of a provider's rate limit headers. */
|
|
825
1055
|
interface ProviderRateLimitHint {
|
|
826
1056
|
remainingRequests?: number;
|
|
@@ -911,7 +1141,7 @@ interface RateLimitAcquireResult {
|
|
|
911
1141
|
reason?: RateLimitReason;
|
|
912
1142
|
}
|
|
913
1143
|
/** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
|
|
914
|
-
declare function defaultEstimateTokens(request: WireRequest): number;
|
|
1144
|
+
export declare function defaultEstimateTokens(request: WireRequest): number;
|
|
915
1145
|
/**
|
|
916
1146
|
* What VernLLM's dispatch layer needs from a limiter. `RateLimiter`
|
|
917
1147
|
* implements this; a caller wanting cross-process coordination can hand
|
|
@@ -926,6 +1156,10 @@ interface RateLimiterAdapter {
|
|
|
926
1156
|
reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
|
|
927
1157
|
/** Optional: current bucket levels, for introspection. Omit if the adapter has no state worth reporting. */
|
|
928
1158
|
getState?(): RateLimitState;
|
|
1159
|
+
/** Optional: live bucket levels for `VernLLM.readRateLimitState()`, the async counterpart of `getState`. */
|
|
1160
|
+
readState?(): Promise<RateLimitState>;
|
|
1161
|
+
/** Optional: receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
|
|
1162
|
+
setLogger?(logger: Logger): void;
|
|
929
1163
|
}
|
|
930
1164
|
/**
|
|
931
1165
|
* Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
|
|
@@ -933,7 +1167,7 @@ interface RateLimiterAdapter {
|
|
|
933
1167
|
* stream of small ones. Any bucket omitted from `options` has infinite
|
|
934
1168
|
* capacity and never blocks.
|
|
935
1169
|
*/
|
|
936
|
-
declare class RateLimiter implements RateLimiterAdapter {
|
|
1170
|
+
export declare class RateLimiter implements RateLimiterAdapter {
|
|
937
1171
|
private readonly requests?;
|
|
938
1172
|
private readonly tokens?;
|
|
939
1173
|
private readonly concurrency?;
|
|
@@ -1019,7 +1253,7 @@ declare class RateLimiter implements RateLimiterAdapter {
|
|
|
1019
1253
|
getState(): RateLimitState;
|
|
1020
1254
|
}
|
|
1021
1255
|
//#endregion
|
|
1022
|
-
//#region src/internal/utils/rateLimitAdapter.utils.d.ts
|
|
1256
|
+
//#region src/internal/utils/rate-limit/rateLimitAdapter.utils.d.ts
|
|
1023
1257
|
/** Not exported. Internal shorthand only, so this union isn't duplicated between the public option fields and `buildRateLimit`'s own signature. */
|
|
1024
1258
|
type RateLimitOption = RateLimitOptions | RateLimiterAdapter;
|
|
1025
1259
|
//#endregion
|
|
@@ -1121,7 +1355,7 @@ type FallbackOn = (error: LLMError, context: {
|
|
|
1121
1355
|
* The default `fallbackOn` policy. Exported so a caller can wrap rather
|
|
1122
1356
|
* than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
|
|
1123
1357
|
*/
|
|
1124
|
-
declare const defaultFallbackOn: FallbackOn;
|
|
1358
|
+
export declare const defaultFallbackOn: FallbackOn;
|
|
1125
1359
|
/**
|
|
1126
1360
|
* Thrown when the chain gives up, whether because the last target failed
|
|
1127
1361
|
* or `fallbackOn` chose to stop early. Carries each attempt in order so
|
|
@@ -1131,7 +1365,7 @@ declare const defaultFallbackOn: FallbackOn;
|
|
|
1131
1365
|
* so existing type-based handling, including reading `retryAfterMs` on an
|
|
1132
1366
|
* `'api'`-typed error, keeps working on a fallback-exhausted error too.
|
|
1133
1367
|
*/
|
|
1134
|
-
declare class FallbackExhaustedError extends LLMError {
|
|
1368
|
+
export declare class FallbackExhaustedError extends LLMError {
|
|
1135
1369
|
readonly attempts: FallbackAttempt[];
|
|
1136
1370
|
constructor(attempts: FallbackAttempt[]);
|
|
1137
1371
|
/**
|
|
@@ -1143,7 +1377,7 @@ declare class FallbackExhaustedError extends LLMError {
|
|
|
1143
1377
|
get retryable(): boolean;
|
|
1144
1378
|
}
|
|
1145
1379
|
/** Narrows `err` to {@link FallbackExhaustedError}, for direct access to its `attempts` (`provider`/`model` per failed target) without a manual `instanceof` check. */
|
|
1146
|
-
declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
|
|
1380
|
+
export declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
|
|
1147
1381
|
/**
|
|
1148
1382
|
* Creates an empty ref box to pass as `CallParams['meta']`, so a caller can
|
|
1149
1383
|
* read the `CallMeta` written by `call()` on the same line as the result
|
|
@@ -1154,7 +1388,7 @@ declare function isFallbackExhaustedError(err: unknown): err is FallbackExhauste
|
|
|
1154
1388
|
* const result = await vern.call({ userContent: '...', meta });
|
|
1155
1389
|
* meta.current?.provider;
|
|
1156
1390
|
*/
|
|
1157
|
-
declare function metaRef(): {
|
|
1391
|
+
export declare function metaRef(): {
|
|
1158
1392
|
current?: CallMeta;
|
|
1159
1393
|
};
|
|
1160
1394
|
//#endregion
|
|
@@ -1229,7 +1463,7 @@ interface ToolDefinition<Name extends string = string, Args = unknown> {
|
|
|
1229
1463
|
* in `defineTool()` preserves the literal `name` type without requiring
|
|
1230
1464
|
* `as const` at every call site.
|
|
1231
1465
|
*/
|
|
1232
|
-
declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
|
|
1466
|
+
export declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
|
|
1233
1467
|
/** Maps a single `ToolDefinition` to its matching `ToolCall` shape. */
|
|
1234
1468
|
type ToolCallFor<T> = T extends ToolDefinition<infer N, infer A> ? {
|
|
1235
1469
|
id: string;
|
|
@@ -1290,77 +1524,12 @@ type ResolvedTools<Tools, R> = [Tools] extends [never] ? ExtractTools<R> : Tools
|
|
|
1290
1524
|
*
|
|
1291
1525
|
* Pass `Tools` explicitly to override inference, e.g. `isToolCallResult<typeof tools>(result)`.
|
|
1292
1526
|
*/
|
|
1293
|
-
declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
|
|
1527
|
+
export declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
|
|
1294
1528
|
/** What the model should do about tools on a given call. */
|
|
1295
1529
|
type ToolChoice = 'auto' | 'none' | 'required' | {
|
|
1296
1530
|
name: string;
|
|
1297
1531
|
};
|
|
1298
1532
|
//#endregion
|
|
1299
|
-
//#region src/types/usage.d.ts
|
|
1300
|
-
type ReserveUsage = (params: {
|
|
1301
|
-
coalesced: boolean;
|
|
1302
|
-
signal?: AbortSignal;
|
|
1303
|
-
}) => Promise<void>;
|
|
1304
|
-
type RefundUsage = (params: {
|
|
1305
|
-
coalesced: boolean;
|
|
1306
|
-
signal?: AbortSignal;
|
|
1307
|
-
}) => Promise<void>;
|
|
1308
|
-
/**
|
|
1309
|
-
* The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
|
|
1310
|
-
* and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
|
|
1311
|
-
* has one definition instead of being redeclared at each use site.
|
|
1312
|
-
*/
|
|
1313
|
-
interface UsageHooks {
|
|
1314
|
-
/**
|
|
1315
|
-
* Reserves usage before the request. Failures become
|
|
1316
|
-
* LLMError('quota_exceeded').
|
|
1317
|
-
*/
|
|
1318
|
-
reserveUsage?: ReserveUsage;
|
|
1319
|
-
/**
|
|
1320
|
-
* Refunds usage after a failed call if reservation succeeded.
|
|
1321
|
-
*/
|
|
1322
|
-
refundUsage?: RefundUsage;
|
|
1323
|
-
}
|
|
1324
|
-
interface TokenUsage {
|
|
1325
|
-
promptTokens: number;
|
|
1326
|
-
completionTokens: number;
|
|
1327
|
-
totalTokens: number;
|
|
1328
|
-
/**
|
|
1329
|
-
* Tokens spent on internal reasoning, a subset of `completionTokens`,
|
|
1330
|
-
* never added on top of it. Undefined when the provider's response
|
|
1331
|
-
* doesn't report a separate reasoning figure, e.g. Bedrock Converse
|
|
1332
|
-
* without an explicit `additionalModelResponseFieldPaths` request.
|
|
1333
|
-
*/
|
|
1334
|
-
reasoningTokens?: number;
|
|
1335
|
-
requestId: string;
|
|
1336
|
-
model: string;
|
|
1337
|
-
/**
|
|
1338
|
-
* The provider target that produced this usage. See `VernLLMOptions['name']`,
|
|
1339
|
-
* default `'primary'`. Optional so consumers constructing a `TokenUsage`
|
|
1340
|
-
* themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
|
|
1341
|
-
* populates it. Absent means the same as `'primary'` if you need a value.
|
|
1342
|
-
*/
|
|
1343
|
-
provider?: string;
|
|
1344
|
-
/**
|
|
1345
|
-
* Whether this usage came from a fallback target rather than the
|
|
1346
|
-
* primary. Optional for the same reason `provider` is: `VernLLM`
|
|
1347
|
-
* always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
|
|
1348
|
-
* isn't forced to.
|
|
1349
|
-
*/
|
|
1350
|
-
usedFallback?: boolean;
|
|
1351
|
-
}
|
|
1352
|
-
type OnUsage = (usage: TokenUsage) => void;
|
|
1353
|
-
/**
|
|
1354
|
-
* Called when a provider response arrives but VernLLM's own post-processing
|
|
1355
|
-
* then fails, after usage data was already present in that response. Covers
|
|
1356
|
-
* any error thrown after usage extraction, not just parse/validation, since
|
|
1357
|
-
* everything in that path only runs once a response, and real spend, has
|
|
1358
|
-
* already arrived. Fires once per failed attempt with extractable usage,
|
|
1359
|
-
* never for transport failures, where no response means no honest number
|
|
1360
|
-
* to report.
|
|
1361
|
-
*/
|
|
1362
|
-
type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
|
|
1363
|
-
//#endregion
|
|
1364
1533
|
//#region src/types/call.d.ts
|
|
1365
1534
|
/**
|
|
1366
1535
|
* Any valid JSON value: a primitive, `null`, or a JSON array/object made
|
|
@@ -1701,6 +1870,13 @@ interface SoftFailureMeta {
|
|
|
1701
1870
|
isFallback: boolean;
|
|
1702
1871
|
/** 1-based, matching `CallMeta.attempts`. */
|
|
1703
1872
|
attempt: number;
|
|
1873
|
+
/**
|
|
1874
|
+
* Token usage for this attempt, if the provider reported it on this
|
|
1875
|
+
* response. `undefined` when the provider omitted usage, not when
|
|
1876
|
+
* usage was zero, so a cost check should treat a missing value as
|
|
1877
|
+
* unknown rather than as free.
|
|
1878
|
+
*/
|
|
1879
|
+
usage?: TokenUsage;
|
|
1704
1880
|
}
|
|
1705
1881
|
/**
|
|
1706
1882
|
* Inspects an otherwise-successful result and optionally reclassifies it
|
|
@@ -1775,7 +1951,7 @@ type ExtractStreamValue<R> = Extract<R, StreamCallResult<unknown>> extends Strea
|
|
|
1775
1951
|
* }
|
|
1776
1952
|
* ```
|
|
1777
1953
|
*/
|
|
1778
|
-
declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
|
|
1954
|
+
export declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
|
|
1779
1955
|
/**
|
|
1780
1956
|
* `StreamEnabledCallParams` with `jsonMode: false`. Selects the streaming
|
|
1781
1957
|
* `call()` overload whose `finalResult` resolves to a plain `string`.
|
|
@@ -2053,7 +2229,7 @@ interface LLMClient {
|
|
|
2053
2229
|
};
|
|
2054
2230
|
}
|
|
2055
2231
|
//#endregion
|
|
2056
|
-
//#region src/internal/utils/cacheAdapter.utils.d.ts
|
|
2232
|
+
//#region src/internal/utils/cache/cacheAdapter.utils.d.ts
|
|
2057
2233
|
/**
|
|
2058
2234
|
* Not exported. Internal shorthand for `VernLLMOptions.cache`, so the
|
|
2059
2235
|
* union isn't duplicated between that field and `buildCache`'s own
|
|
@@ -2065,23 +2241,15 @@ type CacheOption = {
|
|
|
2065
2241
|
eviction?: EvictionOption;
|
|
2066
2242
|
} | CacheAdapter;
|
|
2067
2243
|
//#endregion
|
|
2068
|
-
//#region src/
|
|
2069
|
-
interface Logger {
|
|
2070
|
-
debug(message: string): void;
|
|
2071
|
-
warn(message: string): void;
|
|
2072
|
-
error(message: string, meta?: Record<string, unknown>): void;
|
|
2073
|
-
}
|
|
2244
|
+
//#region src/internal/utils/circuit-breaker/circuitBreakerAdapter.utils.d.ts
|
|
2074
2245
|
/**
|
|
2075
|
-
*
|
|
2076
|
-
*
|
|
2246
|
+
* Not re-exported from the package root, imported directly from this
|
|
2247
|
+
* internal module by `VernLLMOptions.circuitBreaker`'s own type (see
|
|
2248
|
+
* options.ts) so that union isn't duplicated between the public option
|
|
2249
|
+
* field and `buildCircuitBreaker`'s own signature below, same pattern
|
|
2250
|
+
* `CacheOption` and `RateLimitOption` already use for their own options.
|
|
2077
2251
|
*/
|
|
2078
|
-
|
|
2079
|
-
private debugEnabled;
|
|
2080
|
-
constructor(debugEnabled: boolean);
|
|
2081
|
-
debug(message: string): void;
|
|
2082
|
-
warn(message: string): void;
|
|
2083
|
-
error(message: string, meta?: Record<string, unknown>): void;
|
|
2084
|
-
}
|
|
2252
|
+
type CircuitBreakerOption = boolean | CircuitBreakerOptions | CircuitBreakerAdapter;
|
|
2085
2253
|
//#endregion
|
|
2086
2254
|
//#region src/types/options.d.ts
|
|
2087
2255
|
interface VernLLMOptions {
|
|
@@ -2196,9 +2364,11 @@ interface VernLLMOptions {
|
|
|
2196
2364
|
/**
|
|
2197
2365
|
* Enables a circuit breaker that short-circuits calls after repeated
|
|
2198
2366
|
* consecutive failures, instead of continuing to hammer a down provider
|
|
2199
|
-
* Pass `true` for defaults, or an options object to tune threshold/cooldown
|
|
2367
|
+
* Pass `true` for defaults, or an options object to tune threshold/cooldown.
|
|
2368
|
+
* Pass a `CircuitBreakerAdapter` instead for cross-process coordination,
|
|
2369
|
+
* the same pattern `cache` and `rateLimit` already support.
|
|
2200
2370
|
*/
|
|
2201
|
-
circuitBreaker?:
|
|
2371
|
+
circuitBreaker?: CircuitBreakerOption;
|
|
2202
2372
|
/**
|
|
2203
2373
|
* Reports retries and circuit-breaker state transitions as they happen.
|
|
2204
2374
|
* Fire and forget: a throwing handler is caught and logged, and its
|
|
@@ -2296,7 +2466,7 @@ type CreateMiddlewareOptions = Omit<VernLLMMiddleware, 'wrap'> & {
|
|
|
2296
2466
|
* error afterward, so `onError` never changes what the call itself
|
|
2297
2467
|
* returns or throws, only what gets observed about it.
|
|
2298
2468
|
*/
|
|
2299
|
-
declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMiddleware;
|
|
2469
|
+
export declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMiddleware;
|
|
2300
2470
|
//#endregion
|
|
2301
2471
|
//#region src/vernLLM.d.ts
|
|
2302
2472
|
/**
|
|
@@ -2307,7 +2477,7 @@ declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMidd
|
|
|
2307
2477
|
* tracking, and an optional response cache. All configurable, all opt-in
|
|
2308
2478
|
* beyond sensible defaults.
|
|
2309
2479
|
*/
|
|
2310
|
-
declare class VernLLM {
|
|
2480
|
+
export declare class VernLLM {
|
|
2311
2481
|
private readonly logger;
|
|
2312
2482
|
/**
|
|
2313
2483
|
* One `CallExecutor` per provider target: index 0 is the primary,
|
|
@@ -2478,6 +2648,22 @@ declare class VernLLM {
|
|
|
2478
2648
|
* @returns Every target's state, in chain order.
|
|
2479
2649
|
*/
|
|
2480
2650
|
getCircuitStates(model?: string): TargetCircuitState[];
|
|
2651
|
+
/**
|
|
2652
|
+
* The live counterpart of `getRateLimitState`, asking the limiter for its current levels.
|
|
2653
|
+
*
|
|
2654
|
+
* @param target.index Which target to read. Defaults to the primary.
|
|
2655
|
+
* @returns This target's live rate limit levels, or `undefined` if that target has no limiter
|
|
2656
|
+
* configured.
|
|
2657
|
+
* @throws {RangeError} If `target.index` names no target.
|
|
2658
|
+
*/
|
|
2659
|
+
readRateLimitState(target?: Pick<CircuitTarget, 'index'>): Promise<RateLimitState | undefined>;
|
|
2660
|
+
/**
|
|
2661
|
+
* The live counterpart of `getCircuitStates`, asking each breaker for its current state.
|
|
2662
|
+
*
|
|
2663
|
+
* @param model Which model bucket to read, for targets that isolate by model.
|
|
2664
|
+
* @returns Every target's state, in chain order.
|
|
2665
|
+
*/
|
|
2666
|
+
readCircuitStates(model?: string): Promise<TargetCircuitState[]>;
|
|
2481
2667
|
/**
|
|
2482
2668
|
* Manually opens a target's breaker, e.g. to pull a provider out of
|
|
2483
2669
|
* rotation ahead of known maintenance instead of waiting for it to fail.
|
|
@@ -2518,7 +2704,7 @@ declare class VernLLM {
|
|
|
2518
2704
|
* `T` isn't a parameter here; pin it via `llm.call<T>(params)` as usual.
|
|
2519
2705
|
* `defineCachedCallParams` is the `cachedCall()` counterpart.
|
|
2520
2706
|
*/
|
|
2521
|
-
declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
|
|
2707
|
+
export declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
|
|
2522
2708
|
/**
|
|
2523
2709
|
* The `cachedCall()` counterpart to `defineCallParams`: preserves the
|
|
2524
2710
|
* whole `{ cacheKey, ttl, call }` object, `call.tools` included, in one
|
|
@@ -2533,7 +2719,7 @@ declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
|
|
|
2533
2719
|
* const result = await llm.cachedCall(params);
|
|
2534
2720
|
* ```
|
|
2535
2721
|
*/
|
|
2536
|
-
declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
|
|
2722
|
+
export declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
|
|
2537
2723
|
//#endregion
|
|
2538
2724
|
//#region src/adapters/internal/sse.d.ts
|
|
2539
2725
|
/**
|
|
@@ -2563,14 +2749,14 @@ declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(par
|
|
|
2563
2749
|
* Malformed JSON in a frame throws `LLMError('parse')`, consistent with
|
|
2564
2750
|
* how malformed JSON is handled elsewhere in VernLLM.
|
|
2565
2751
|
*/
|
|
2566
|
-
declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
|
|
2752
|
+
export declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
|
|
2567
2753
|
/**
|
|
2568
2754
|
* Sentinel yielded by `parseSseStream` for a comment-only frame (no
|
|
2569
2755
|
* `data:` payload), the mechanism providers use for SSE keep-alive
|
|
2570
2756
|
* pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
|
|
2571
2757
|
* alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
|
|
2572
2758
|
*/
|
|
2573
|
-
declare const SSE_PING: unique symbol;
|
|
2759
|
+
export declare const SSE_PING: unique symbol;
|
|
2574
2760
|
//#endregion
|
|
2575
2761
|
//#region src/adapters/internal/imageFormat.d.ts
|
|
2576
2762
|
/**
|
|
@@ -2817,7 +3003,7 @@ interface AnthropicAdapterOptions {
|
|
|
2817
3003
|
* instead, which maps to a real constraint either way (native
|
|
2818
3004
|
* `output_config.format` or a forced tool call).
|
|
2819
3005
|
*/
|
|
2820
|
-
declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
|
|
3006
|
+
export declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
|
|
2821
3007
|
//#endregion
|
|
2822
3008
|
//#region src/adapters/gemini.d.ts
|
|
2823
3009
|
/**
|
|
@@ -3029,7 +3215,7 @@ interface GeminiAdapterOptions {
|
|
|
3029
3215
|
*/
|
|
3030
3216
|
thinkingLevelModels?: ModelCapabilityOverride;
|
|
3031
3217
|
}
|
|
3032
|
-
declare function fromGemini(client: GeminiClient, options?: GeminiAdapterOptions): LLMClient;
|
|
3218
|
+
export declare function fromGemini(client: GeminiClient, options?: GeminiAdapterOptions): LLMClient;
|
|
3033
3219
|
//#endregion
|
|
3034
3220
|
//#region src/adapters/bedrock.d.ts
|
|
3035
3221
|
/** Bedrock Converse's supported inline image formats. */
|
|
@@ -3388,7 +3574,7 @@ interface AwsSendClient {
|
|
|
3388
3574
|
* `finalizeResponse`'s `content` path exactly like the non-streaming
|
|
3389
3575
|
* `create` branch above unwraps it.
|
|
3390
3576
|
*/
|
|
3391
|
-
declare function fromBedrock(bedrockClient: BedrockConverseClient | AwsSendClient, options?: BedrockAdapterOptions): LLMClient;
|
|
3577
|
+
export declare function fromBedrock(bedrockClient: BedrockConverseClient | AwsSendClient, options?: BedrockAdapterOptions): LLMClient;
|
|
3392
3578
|
//#endregion
|
|
3393
3579
|
//#region src/adapters/fetch.d.ts
|
|
3394
3580
|
/** The chat-completion-shaped request VernLLM builds internally */
|
|
@@ -3550,7 +3736,7 @@ interface FetchAdapterConfig {
|
|
|
3550
3736
|
* `LLMError('validation')` instead of quietly using unrelated native
|
|
3551
3737
|
* `fetch`.
|
|
3552
3738
|
*/
|
|
3553
|
-
declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
3739
|
+
export declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
3554
3740
|
//#endregion
|
|
3555
3741
|
//#region src/adapters/openaiCompatible.d.ts
|
|
3556
3742
|
/**
|
|
@@ -3617,7 +3803,7 @@ interface OpenAICompatibleAdapterOptions {
|
|
|
3617
3803
|
*/
|
|
3618
3804
|
supportsWithResponse?: boolean;
|
|
3619
3805
|
}
|
|
3620
|
-
declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
|
|
3806
|
+
export declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
|
|
3621
3807
|
/**
|
|
3622
3808
|
* Named alias for the OpenAI SDK itself. A raw `new OpenAI(...)` instance
|
|
3623
3809
|
* structurally matches most of `LLMClient`, but newer `openai` SDK major
|
|
@@ -3635,9 +3821,9 @@ declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibl
|
|
|
3635
3821
|
* official `openai` package's client versus a fake or a test double.
|
|
3636
3822
|
* Pass `supportsWithResponse: true` once you've confirmed it.
|
|
3637
3823
|
*/
|
|
3638
|
-
declare const fromOpenAI: typeof fromOpenAICompatible;
|
|
3824
|
+
export declare const fromOpenAI: typeof fromOpenAICompatible;
|
|
3639
3825
|
/** Groqs SDK matches the OpenAI wire format */
|
|
3640
|
-
declare const fromGroq: typeof fromOpenAICompatible;
|
|
3826
|
+
export declare const fromGroq: typeof fromOpenAICompatible;
|
|
3641
3827
|
/**
|
|
3642
3828
|
* Mistrals `chat.completions`-shaped client (or their OpenAI-compat
|
|
3643
3829
|
* endpoint). Mistral supports `stream_options.include_usage` (added after
|
|
@@ -3645,88 +3831,88 @@ declare const fromGroq: typeof fromOpenAICompatible;
|
|
|
3645
3831
|
* Mistral's changelog and streaming docs), so this is a plain alias like
|
|
3646
3832
|
* the others, `supportsStreamUsage` defaults to `true`.
|
|
3647
3833
|
*/
|
|
3648
|
-
declare const fromMistral: typeof fromOpenAICompatible;
|
|
3834
|
+
export declare const fromMistral: typeof fromOpenAICompatible;
|
|
3649
3835
|
/** DeepSeeks API is OpenAI-compatible */
|
|
3650
|
-
declare const fromDeepSeek: typeof fromOpenAICompatible;
|
|
3836
|
+
export declare const fromDeepSeek: typeof fromOpenAICompatible;
|
|
3651
3837
|
/** Cerebras inference API is OpenAI-compatible */
|
|
3652
|
-
declare const fromCerebras: typeof fromOpenAICompatible;
|
|
3838
|
+
export declare const fromCerebras: typeof fromOpenAICompatible;
|
|
3653
3839
|
/** Together AIs API is OpenAI-compatible */
|
|
3654
|
-
declare const fromTogether: typeof fromOpenAICompatible;
|
|
3840
|
+
export declare const fromTogether: typeof fromOpenAICompatible;
|
|
3655
3841
|
/** Fireworks AIs API is OpenAI-compatible */
|
|
3656
|
-
declare const fromFireworks: typeof fromOpenAICompatible;
|
|
3842
|
+
export declare const fromFireworks: typeof fromOpenAICompatible;
|
|
3657
3843
|
/**
|
|
3658
3844
|
* Ollama exposes an OpenAI-compatible endpoint at `/v1/chat/completions`
|
|
3659
3845
|
* (as opposed to its native `/api/chat` format, which differs). Point an
|
|
3660
3846
|
* OpenAI SDK instances `baseURL` at your Ollama server and pass it here:
|
|
3661
3847
|
* this does not talk to Ollamas native API directly.
|
|
3662
3848
|
*/
|
|
3663
|
-
declare const fromOllama: typeof fromOpenAICompatible;
|
|
3849
|
+
export declare const fromOllama: typeof fromOpenAICompatible;
|
|
3664
3850
|
/** OpenRouter's API is OpenAI-compatible */
|
|
3665
|
-
declare const fromOpenRouter: typeof fromOpenAICompatible;
|
|
3851
|
+
export declare const fromOpenRouter: typeof fromOpenAICompatible;
|
|
3666
3852
|
/** Perplexity's API is OpenAI-compatible */
|
|
3667
|
-
declare const fromPerplexity: typeof fromOpenAICompatible;
|
|
3853
|
+
export declare const fromPerplexity: typeof fromOpenAICompatible;
|
|
3668
3854
|
/** DeepInfra's API is OpenAI-compatible */
|
|
3669
|
-
declare const fromDeepInfra: typeof fromOpenAICompatible;
|
|
3855
|
+
export declare const fromDeepInfra: typeof fromOpenAICompatible;
|
|
3670
3856
|
/** Novita's API is OpenAI-compatible */
|
|
3671
|
-
declare const fromNovita: typeof fromOpenAICompatible;
|
|
3857
|
+
export declare const fromNovita: typeof fromOpenAICompatible;
|
|
3672
3858
|
/** Hyperbolic's API is OpenAI-compatible */
|
|
3673
|
-
declare const fromHyperbolic: typeof fromOpenAICompatible;
|
|
3859
|
+
export declare const fromHyperbolic: typeof fromOpenAICompatible;
|
|
3674
3860
|
/** Moonshot's (Kimi) API is OpenAI-compatible */
|
|
3675
|
-
declare const fromMoonshot: typeof fromOpenAICompatible;
|
|
3861
|
+
export declare const fromMoonshot: typeof fromOpenAICompatible;
|
|
3676
3862
|
/** Zhipu's (GLM) API is OpenAI-compatible */
|
|
3677
|
-
declare const fromZhipu: typeof fromOpenAICompatible;
|
|
3863
|
+
export declare const fromZhipu: typeof fromOpenAICompatible;
|
|
3678
3864
|
/**
|
|
3679
3865
|
* LM Studio exposes an OpenAI-compatible endpoint at `/v1/chat/completions`.
|
|
3680
3866
|
* Point an OpenAI SDK instance's `baseURL` at your local LM Studio server.
|
|
3681
3867
|
*/
|
|
3682
|
-
declare const fromLMStudio: typeof fromOpenAICompatible;
|
|
3868
|
+
export declare const fromLMStudio: typeof fromOpenAICompatible;
|
|
3683
3869
|
/**
|
|
3684
3870
|
* vLLM's OpenAI-compatible server mode exposes `/v1/chat/completions`.
|
|
3685
3871
|
* Point an OpenAI SDK instance's `baseURL` at your vLLM server.
|
|
3686
3872
|
*/
|
|
3687
|
-
declare const fromVLLM: typeof fromOpenAICompatible;
|
|
3873
|
+
export declare const fromVLLM: typeof fromOpenAICompatible;
|
|
3688
3874
|
/** xAI's Grok API is OpenAI-compatible */
|
|
3689
|
-
declare const fromXAI: typeof fromOpenAICompatible;
|
|
3875
|
+
export declare const fromXAI: typeof fromOpenAICompatible;
|
|
3690
3876
|
/** NVIDIA NIM's hosted and self-hosted endpoints are OpenAI-compatible */
|
|
3691
|
-
declare const fromNvidiaNIM: typeof fromOpenAICompatible;
|
|
3877
|
+
export declare const fromNvidiaNIM: typeof fromOpenAICompatible;
|
|
3692
3878
|
/** Vercel AI Gateway is OpenAI-compatible */
|
|
3693
|
-
declare const fromVercelAIGateway: typeof fromOpenAICompatible;
|
|
3879
|
+
export declare const fromVercelAIGateway: typeof fromOpenAICompatible;
|
|
3694
3880
|
/** Cloudflare Workers AI exposes an OpenAI-compatible endpoint */
|
|
3695
|
-
declare const fromCloudflareWorkersAI: typeof fromOpenAICompatible;
|
|
3881
|
+
export declare const fromCloudflareWorkersAI: typeof fromOpenAICompatible;
|
|
3696
3882
|
/** Nebius AI Studio is OpenAI-compatible */
|
|
3697
|
-
declare const fromNebius: typeof fromOpenAICompatible;
|
|
3883
|
+
export declare const fromNebius: typeof fromOpenAICompatible;
|
|
3698
3884
|
/** SambaNova Cloud's API is OpenAI-compatible */
|
|
3699
|
-
declare const fromSambaNova: typeof fromOpenAICompatible;
|
|
3885
|
+
export declare const fromSambaNova: typeof fromOpenAICompatible;
|
|
3700
3886
|
/** Baseten's model hosting exposes an OpenAI-compatible endpoint */
|
|
3701
|
-
declare const fromBaseten: typeof fromOpenAICompatible;
|
|
3887
|
+
export declare const fromBaseten: typeof fromOpenAICompatible;
|
|
3702
3888
|
/** Featherless AI's API is OpenAI-compatible */
|
|
3703
|
-
declare const fromFeatherless: typeof fromOpenAICompatible;
|
|
3889
|
+
export declare const fromFeatherless: typeof fromOpenAICompatible;
|
|
3704
3890
|
/** Friendli AI's serving endpoint is OpenAI-compatible */
|
|
3705
|
-
declare const fromFriendli: typeof fromOpenAICompatible;
|
|
3891
|
+
export declare const fromFriendli: typeof fromOpenAICompatible;
|
|
3706
3892
|
/** SiliconFlow's API is OpenAI-compatible */
|
|
3707
|
-
declare const fromSiliconFlow: typeof fromOpenAICompatible;
|
|
3893
|
+
export declare const fromSiliconFlow: typeof fromOpenAICompatible;
|
|
3708
3894
|
/** Parasail's inference API is OpenAI-compatible */
|
|
3709
|
-
declare const fromParasail: typeof fromOpenAICompatible;
|
|
3895
|
+
export declare const fromParasail: typeof fromOpenAICompatible;
|
|
3710
3896
|
/** StepFun's API is OpenAI-compatible */
|
|
3711
|
-
declare const fromStepFun: typeof fromOpenAICompatible;
|
|
3897
|
+
export declare const fromStepFun: typeof fromOpenAICompatible;
|
|
3712
3898
|
/** MiniMax's API is OpenAI-compatible */
|
|
3713
|
-
declare const fromMiniMax: typeof fromOpenAICompatible;
|
|
3899
|
+
export declare const fromMiniMax: typeof fromOpenAICompatible;
|
|
3714
3900
|
/** Lambda Labs' Inference API is OpenAI-compatible */
|
|
3715
|
-
declare const fromLambdaLabs: typeof fromOpenAICompatible;
|
|
3901
|
+
export declare const fromLambdaLabs: typeof fromOpenAICompatible;
|
|
3716
3902
|
/** Snowflake Cortex's LLM endpoint is OpenAI-compatible */
|
|
3717
|
-
declare const fromSnowflakeCortex: typeof fromOpenAICompatible;
|
|
3903
|
+
export declare const fromSnowflakeCortex: typeof fromOpenAICompatible;
|
|
3718
3904
|
/** Anyscale Endpoints' API is OpenAI-compatible */
|
|
3719
|
-
declare const fromAnyscale: typeof fromOpenAICompatible;
|
|
3905
|
+
export declare const fromAnyscale: typeof fromOpenAICompatible;
|
|
3720
3906
|
/** Lepton AI's inference API is OpenAI-compatible */
|
|
3721
|
-
declare const fromLepton: typeof fromOpenAICompatible;
|
|
3907
|
+
export declare const fromLepton: typeof fromOpenAICompatible;
|
|
3722
3908
|
/** Inference.net's API is OpenAI-compatible */
|
|
3723
|
-
declare const fromInferenceNet: typeof fromOpenAICompatible;
|
|
3909
|
+
export declare const fromInferenceNet: typeof fromOpenAICompatible;
|
|
3724
3910
|
/** Infermatic's API is OpenAI-compatible */
|
|
3725
|
-
declare const fromInfermatic: typeof fromOpenAICompatible;
|
|
3911
|
+
export declare const fromInfermatic: typeof fromOpenAICompatible;
|
|
3726
3912
|
/** AtlasCloud's inference API is OpenAI-compatible */
|
|
3727
|
-
declare const fromAtlasCloud: typeof fromOpenAICompatible;
|
|
3913
|
+
export declare const fromAtlasCloud: typeof fromOpenAICompatible;
|
|
3728
3914
|
/** 01.AI's (Yi models) API is OpenAI-compatible */
|
|
3729
|
-
declare const from01AI: typeof fromOpenAICompatible;
|
|
3915
|
+
export declare const from01AI: typeof fromOpenAICompatible;
|
|
3730
3916
|
//#endregion
|
|
3731
|
-
export {
|
|
3917
|
+
export type { AnthropicClient, AssistantContent, AttemptContext, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedConditionalToolCallParams, CachedJsonModeDisabledCallParams, CachedJsonModeEnabledCallParams, CachedStreamCallParams, CachedStreamConditionalToolCallParams, CachedStreamJsonModeDisabledCallParams, CachedStreamJsonModeEnabledCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallMeta, CallParams, CallResult, CallWithToolsResult, CircuitBreakerAdapter, CircuitBreakerCallContext, CircuitBreakerOptions, CircuitBreakerStateChangeHandler, CircuitState, CircuitTarget, ConditionalToolCallParams, ContentBlock, ContentResult, ConversationTurn, CooldownBackoff, CreateMiddlewareOptions, DuplicateToolNamesIssue, EvictionOption, ExponentialBackoffOptions, FallbackAttempt, FallbackOn, FallbackTarget, FetchAdapterConfig, GeminiClient, HistoryToolResultIssue, ImageBlock, JsonModeDisabledCallParams, JsonModeEnabledCallParams, JsonSchemaSpec, JsonValue, LLMClient, LLMErrorCode, LLMErrorIssuesByCode, LLMErrorSnapshot, LLMErrorType, LLMRequestShape, LLMRequestSnapshot, Logger, MiddlewareCapabilities, MiddlewareContext, MiddlewareContextBase, MiddlewareRef, MiddlewareStateBag, MiddlewareStateKey, OnEvent, OnUsage, PreDispatchContext, RateLimitAcquireResult, RateLimitOptions, RateLimitReason, RateLimitState, RateLimiterAdapter, RefundUsage, RequiredMiddlewareRef, ReserveUsage, RetryAttempt, RetryBudgetOptions, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, StreamJsonModeDisabledCallParams, StreamJsonModeEnabledCallParams, TargetCircuitState, TextBlock, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolIssue, ToolResult, ToolsDisabledCallParams, TrippingPolicy, UnknownToolChoiceIssue, UnsupportedCapabilityIssue, VernLLMEvent, VernLLMMiddleware, VernLLMOptions, WireCallRequest, WireCallRequestPatch, WireMessage, WireRequest, WireResponseFormat, WireStreamChunk, WireTool, WireToolCall, WireToolChoice };
|
|
3732
3918
|
//# sourceMappingURL=index.d.mts.map
|