vern-llm 2.9.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +58 -24
- package/dist/adapters/index.cjs +12 -0
- package/dist/adapters/index.cjs.map +1 -0
- package/dist/adapters/index.d.cts +582 -0
- package/dist/adapters/index.d.cts.map +1 -0
- package/dist/adapters/index.d.mts +582 -0
- package/dist/adapters/index.d.mts.map +1 -0
- package/dist/adapters/index.mjs +12 -0
- package/dist/adapters/index.mjs.map +1 -0
- package/dist/client-HWxkwVvj.d.cts +1871 -0
- package/dist/client-HWxkwVvj.d.cts.map +1 -0
- package/dist/client-HWxkwVvj.d.mts +1871 -0
- package/dist/client-HWxkwVvj.d.mts.map +1 -0
- package/dist/index.cjs +2 -14
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +176 -3665
- package/dist/index.d.cts.map +1 -1
- package/dist/index.d.mts +176 -3665
- package/dist/index.d.mts.map +1 -1
- package/dist/index.mjs +2 -14
- package/dist/index.mjs.map +1 -1
- package/dist/responseBody.utils-B2tQiGUi.mjs +2 -0
- package/dist/responseBody.utils-B2tQiGUi.mjs.map +1 -0
- package/dist/responseBody.utils-CknznUJA.cjs +2 -0
- package/dist/responseBody.utils-CknznUJA.cjs.map +1 -0
- package/package.json +19 -6
package/dist/index.d.cts
CHANGED
|
@@ -1,2233 +1,99 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
/** One tool call's contract failure, used to report every bad call in a response at once. */
|
|
12
|
-
interface ToolIssue {
|
|
13
|
-
name: string;
|
|
14
|
-
toolCallId: string;
|
|
15
|
-
code: LLMErrorCode;
|
|
16
|
-
detail?: unknown;
|
|
17
|
-
}
|
|
18
|
-
/**
|
|
19
|
-
* The specific values behind a `duplicate_tool_names` failure: the
|
|
20
|
-
* offending call's `tools` array had more than one entry sharing a name.
|
|
21
|
-
*/
|
|
22
|
-
interface DuplicateToolNamesIssue {
|
|
23
|
-
names: string[];
|
|
24
|
-
}
|
|
25
|
-
/**
|
|
26
|
-
* The specific values behind an `unknown_tool_choice` failure: `toolChoice`
|
|
27
|
-
* named a tool that wasn't in the call's own `tools` array.
|
|
28
|
-
*/
|
|
29
|
-
interface UnknownToolChoiceIssue {
|
|
30
|
-
requested: string;
|
|
31
|
-
available: string[];
|
|
32
|
-
}
|
|
33
|
-
/**
|
|
34
|
-
* The specific values behind a `duplicate_tool_result_ids` /
|
|
35
|
-
* `unknown_tool_result_ids` / `missing_tool_results` failure: which
|
|
36
|
-
* `history` turn was affected, and which `toolCallId`s were the problem.
|
|
37
|
-
*/
|
|
38
|
-
interface HistoryToolResultIssue {
|
|
39
|
-
historyIndex: number;
|
|
40
|
-
ids: string[];
|
|
41
|
-
}
|
|
42
|
-
/**
|
|
43
|
-
* The specific values behind an `unsupported_capability` failure: which
|
|
44
|
-
* capability the current adapter/client/model doesn't support.
|
|
45
|
-
*/
|
|
46
|
-
interface UnsupportedCapabilityIssue {
|
|
47
|
-
capability: string;
|
|
48
|
-
}
|
|
49
|
-
/**
|
|
50
|
-
* Maps each `LLMErrorCode` that carries structured `issues` to that
|
|
51
|
-
* payload's exact shape. Not every code appears here: most `invalid_params`
|
|
52
|
-
* failures are a single deterministic fact the `message` already states in
|
|
53
|
-
* full, so adding a typed `issues` entry for them would only duplicate the
|
|
54
|
-
* message into a field, the same near-duplicate-code problem `code` itself
|
|
55
|
-
* avoids. Codes that repeat here are exactly the ones whose `message`
|
|
56
|
-
* already string-joins a list a caller might want to consume directly
|
|
57
|
-
* rather than re-parse out of prose, or that otherwise want a place to
|
|
58
|
-
* report the exact captured values of a failure.
|
|
59
|
-
*
|
|
60
|
-
* Deliberately not a mapped type over the whole `LLMErrorCode` union: a
|
|
61
|
-
* schema-validation failure's `issues` (the caller's own Zod-compatible
|
|
62
|
-
* validator's error object) has no code and no shape VernLLM could know in
|
|
63
|
-
* advance, so it stays untyped on `LLMError.issues` itself rather than
|
|
64
|
-
* forcing every code into this table.
|
|
65
|
-
*/
|
|
66
|
-
interface LLMErrorIssuesByCode {
|
|
67
|
-
unknown_tool: ToolIssue[];
|
|
68
|
-
duplicate_tool_call_id: ToolIssue[];
|
|
69
|
-
duplicate_tool_names: DuplicateToolNamesIssue;
|
|
70
|
-
unknown_tool_choice: UnknownToolChoiceIssue;
|
|
71
|
-
duplicate_tool_result_ids: HistoryToolResultIssue;
|
|
72
|
-
unknown_tool_result_ids: HistoryToolResultIssue;
|
|
73
|
-
missing_tool_results: HistoryToolResultIssue;
|
|
74
|
-
unsupported_capability: UnsupportedCapabilityIssue;
|
|
75
|
-
}
|
|
76
|
-
/**
|
|
77
|
-
* Point-in-time copy of an `LLMError`'s fields, produced by
|
|
78
|
-
* `LLMError.toSnapshot()`. This is what `RetryAttempt.error` holds
|
|
79
|
-
* instead of a live `LLMError`.
|
|
80
|
-
*
|
|
81
|
-
* A past attempt only needs to be describable (message, type, code,
|
|
82
|
-
* whether it was retryable), never thrown again. So it skips `Error`'s
|
|
83
|
-
* behavior, `instanceof` identity, and any live getter. Using the full
|
|
84
|
-
* `LLMError` class here would also make the type self referential
|
|
85
|
-
* through its own `attempts` field.
|
|
86
|
-
*
|
|
87
|
-
* Has no `cause`. `cause` is `unknown` and never validated by VernLLM,
|
|
88
|
-
* and it is meant to be read directly on the live error you just
|
|
89
|
-
* caught, not carried indefinitely inside history. `type`, `code`,
|
|
90
|
-
* `status`, and `issues` are the structured fields a snapshot carries
|
|
91
|
-
* instead.
|
|
92
|
-
*
|
|
93
|
-
* `attempts` is still present, since a recorded attempt can itself be
|
|
94
|
-
* the terminal failure of an inner retry loop with its own history (see
|
|
95
|
-
* `FallbackAttempt`). That's a tree of past data, not a cycle.
|
|
96
|
-
*/
|
|
97
|
-
interface LLMErrorSnapshot {
|
|
98
|
-
message: string;
|
|
99
|
-
type: LLMErrorType;
|
|
100
|
-
status?: number;
|
|
101
|
-
issues?: unknown;
|
|
102
|
-
retryAfterMs?: number;
|
|
103
|
-
code?: LLMErrorCode;
|
|
104
|
-
/** Computed once, at snapshot time, since a snapshot has no live getter. */
|
|
105
|
-
retryable: boolean;
|
|
106
|
-
/** This attempt's own prior attempts, if it was itself the terminal failure of a retry loop. */
|
|
107
|
-
attempts?: RetryAttempt[];
|
|
108
|
-
}
|
|
109
|
-
/**
|
|
110
|
-
* Point-in-time copy of the request an attempt sent, produced by
|
|
111
|
-
* `toRequestSnapshot()`. This is what `RetryAttempt.request` holds.
|
|
112
|
-
* Mirrors `LLMErrorSnapshot`: plain data, never thrown or dispatched
|
|
113
|
-
* again, safe to serialize and store.
|
|
114
|
-
*/
|
|
115
|
-
interface LLMRequestSnapshot {
|
|
116
|
-
/** Provider id this attempt targeted, e.g. "openai". */
|
|
117
|
-
provider: string;
|
|
118
|
-
/** Model id this attempt targeted. */
|
|
119
|
-
model: string;
|
|
120
|
-
/** The payload as actually sent for this attempt, after any transform/repair. Passed through `safeBody`. */
|
|
121
|
-
body: unknown;
|
|
122
|
-
/** Non sensitive request headers. Auth headers are stripped before the snapshot is built, never included. */
|
|
123
|
-
headers?: Record<string, string>;
|
|
124
|
-
/** Wall clock time the attempt started, ms since epoch. */
|
|
125
|
-
startedAt: number;
|
|
126
|
-
}
|
|
127
|
-
/**
|
|
128
|
-
* One failed attempt on the way to a terminal error: which attempt index
|
|
129
|
-
* it was, and a snapshot of the error it failed with. The base shape
|
|
130
|
-
* every richer attempt record (e.g. `FallbackAttempt`) extends, rather
|
|
131
|
-
* than duplicates.
|
|
132
|
-
*/
|
|
133
|
-
interface RetryAttempt {
|
|
134
|
-
index: number;
|
|
135
|
-
error: LLMErrorSnapshot;
|
|
136
|
-
/** What was sent for this attempt. Optional: absent for attempts predating this field. */
|
|
137
|
-
request?: LLMRequestSnapshot;
|
|
138
|
-
}
|
|
139
|
-
/** Optional fields for constructing an {@link LLMError}. `message` and `type` stay positional since every throw site sets both. */
|
|
140
|
-
interface LLMErrorOptions {
|
|
141
|
-
status?: number;
|
|
142
|
-
issues?: unknown;
|
|
143
|
-
cause?: unknown;
|
|
144
|
-
retryAfterMs?: number;
|
|
145
|
-
/** Stable discriminator within `type`. Absent on errors predating it. */
|
|
146
|
-
code?: LLMErrorCode;
|
|
147
|
-
/** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
|
|
148
|
-
attempts?: RetryAttempt[];
|
|
149
|
-
}
|
|
150
|
-
export declare class LLMError extends Error {
|
|
151
|
-
type: LLMErrorType;
|
|
152
|
-
status?: number;
|
|
153
|
-
issues?: unknown;
|
|
154
|
-
cause?: unknown;
|
|
155
|
-
retryAfterMs?: number;
|
|
156
|
-
/** Stable discriminator within `type`. Absent on errors predating it. */
|
|
157
|
-
code?: LLMErrorCode;
|
|
158
|
-
/** Every attempt made before this error was thrown, in order. Absent when nothing was retried. */
|
|
159
|
-
attempts?: RetryAttempt[];
|
|
160
|
-
constructor(message: string, type: LLMErrorType, options?: LLMErrorOptions);
|
|
161
|
-
/**
|
|
162
|
-
* Computed purely from `type`/`code`, independent of any specific call's
|
|
163
|
-
* `nonRetryableStatus` list. False for `parse`/`validation`/
|
|
164
|
-
* `invalid_params`/`aborted` types (the caller's own input, the model's
|
|
165
|
-
* own response, or intentional cancellation, none of which are the
|
|
166
|
-
* provider being unhealthy), the tool contract codes, the local
|
|
167
|
-
* rate limit codes, and the middleware timeout code.
|
|
168
|
-
* Subclasses (see `FallbackExhaustedError`) may override this when `type`
|
|
169
|
-
* alone carries no retry signal.
|
|
170
|
-
*/
|
|
171
|
-
get retryable(): boolean;
|
|
172
|
-
/**
|
|
173
|
-
* Whether this failure should count toward the circuit breaker's
|
|
174
|
-
* failure threshold. Not the same question as `retryable`:
|
|
175
|
-
* `quota_exceeded` is retryable but says nothing about provider
|
|
176
|
-
* health, so it's excluded here even though `retryable` is true for
|
|
177
|
-
* it. Always false whenever `retryable` is false.
|
|
178
|
-
*/
|
|
179
|
-
get countsTowardBreaker(): boolean;
|
|
180
|
-
/**
|
|
181
|
-
* Copies this error's fields into an {@link LLMErrorSnapshot}, for
|
|
182
|
-
* recording as a `RetryAttempt`/`FallbackAttempt`. `retryable` is
|
|
183
|
-
* captured here since a snapshot has no getter of its own. `cause` is
|
|
184
|
-
* not copied, see `LLMErrorSnapshot`'s own doc. `issues` and every
|
|
185
|
-
* nested `attempts` entry's own `issues` go through `safeAttempts`,
|
|
186
|
-
* since a schema validation failure's `issues` is a caller supplied
|
|
187
|
-
* value, not controlled by VernLLM, and `attempts` is itself a public
|
|
188
|
-
* constructor option a caller can hand build.
|
|
189
|
-
*/
|
|
190
|
-
toSnapshot(): LLMErrorSnapshot;
|
|
191
|
-
/**
|
|
192
|
-
* Controls what `JSON.stringify(err)` produces. Omits `cause` for the
|
|
193
|
-
* same reason `toSnapshot()` does: `cause` is `unknown` and never
|
|
194
|
-
* validated by VernLLM, and some SDK errors carry circular structures
|
|
195
|
-
* `JSON.stringify` cannot serialize at all. Read `err.cause` directly
|
|
196
|
-
* instead. `issues`, including every nested `attempts` entry's own
|
|
197
|
-
* `issues`, goes through `safeAttempts` for the same reason: a schema
|
|
198
|
-
* validation failure's `issues` is caller supplied and not guaranteed
|
|
199
|
-
* circular free. Also includes `message` and `retryable`, which a
|
|
200
|
-
* plain property walk would otherwise miss: `message` is
|
|
201
|
-
* non-enumerable on `Error`, and `retryable` is a getter, not an own
|
|
202
|
-
* property.
|
|
203
|
-
*/
|
|
204
|
-
toJSON(): Record<string, unknown>;
|
|
205
|
-
}
|
|
206
|
-
export declare function isLLMError(err: unknown): err is LLMError;
|
|
207
|
-
/**
|
|
208
|
-
* Narrows `err.issues` to the exact shape {@link LLMErrorIssuesByCode} maps
|
|
209
|
-
* `code` to, for any code listed there. `code` stays the only discriminator
|
|
210
|
-
* VernLLM uses; this just gives that existing check a typed return instead
|
|
211
|
-
* of requiring a manual cast of `issues`:
|
|
212
|
-
*
|
|
213
|
-
* ```ts
|
|
214
|
-
* if (isLLMError(err) && hasIssues(err, 'duplicate_tool_names')) {
|
|
215
|
-
* console.log(err.issues.names); // string[], no cast needed
|
|
216
|
-
* }
|
|
217
|
-
* ```
|
|
218
|
-
*/
|
|
219
|
-
export declare function hasIssues<C extends keyof LLMErrorIssuesByCode>(err: LLMError, code: C): err is LLMError & {
|
|
220
|
-
code: C;
|
|
221
|
-
issues: LLMErrorIssuesByCode[C];
|
|
222
|
-
};
|
|
223
|
-
//#endregion
|
|
224
|
-
//#region src/types/cache.d.ts
|
|
225
|
-
interface CacheAdapter<T = unknown> {
|
|
226
|
-
get(key: string): Promise<{
|
|
227
|
-
hit: boolean;
|
|
228
|
-
value: T | null;
|
|
229
|
-
}>;
|
|
230
|
-
set(key: string, value: T, ttl: number): Promise<void>;
|
|
231
|
-
delete?(key: string): Promise<void>;
|
|
232
|
-
resolveKey?(key: string): Promise<string>;
|
|
233
|
-
}
|
|
234
|
-
/**
|
|
235
|
-
* Which entry `InMemoryCacheAdapter` evicts once `maxSize` is exceeded.
|
|
236
|
-
* `'fifo'` (default) drops the oldest inserted entry. `'lru'` drops the
|
|
237
|
-
* least recently read or written entry.
|
|
238
|
-
*/
|
|
239
|
-
type EvictionOption = 'fifo' | 'lru';
|
|
240
|
-
/**
|
|
241
|
-
* Trivial default so the package works out of the box with no external deps.
|
|
242
|
-
* Not shared across processes, swap in Redis/Upstash/etc for production.
|
|
243
|
-
*/
|
|
244
|
-
export declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
245
|
-
private readonly maxSize;
|
|
246
|
-
private store;
|
|
247
|
-
private readonly eviction;
|
|
248
|
-
constructor(maxSize?: number, eviction?: EvictionOption);
|
|
249
|
-
get(key: string): Promise<{
|
|
250
|
-
hit: boolean;
|
|
251
|
-
value: T | null;
|
|
252
|
-
}>;
|
|
253
|
-
set(key: string, value: T, ttl: number): Promise<void>;
|
|
254
|
-
delete(key: string): Promise<void>;
|
|
255
|
-
private cleanupExpiredEntries;
|
|
256
|
-
private enforceSizeLimit;
|
|
257
|
-
}
|
|
258
|
-
/**
|
|
259
|
-
* Normalizes keys before caching to avoid duplicate entries from formatting differences.
|
|
260
|
-
*/
|
|
261
|
-
export declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
262
|
-
private readonly inner;
|
|
263
|
-
constructor(inner?: CacheAdapter<T>);
|
|
264
|
-
private normalize;
|
|
265
|
-
resolveKey(key: string): Promise<string>;
|
|
266
|
-
get(key: string): Promise<{
|
|
267
|
-
hit: boolean;
|
|
268
|
-
value: T | null;
|
|
269
|
-
}>;
|
|
270
|
-
set(key: string, value: T, ttl: number): Promise<void>;
|
|
271
|
-
delete(key: string): Promise<void>;
|
|
272
|
-
}
|
|
273
|
-
/**
|
|
274
|
-
* Two-tier cache with fast local L1 and shared L2.
|
|
275
|
-
* L2 hits are promoted back to L1.
|
|
276
|
-
*/
|
|
277
|
-
export declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
278
|
-
private readonly l1;
|
|
279
|
-
private readonly l2;
|
|
280
|
-
private readonly l1Ttl?;
|
|
281
|
-
constructor(l1: CacheAdapter<T>, l2: CacheAdapter<T>, l1Ttl?: number | undefined);
|
|
282
|
-
/**
|
|
283
|
-
* Forwards to L1's `resolveKey` if it has one, otherwise L2's. L1 is
|
|
284
|
-
* preferred since `get()` checks L1 first, so its notion of "the same
|
|
285
|
-
* key" is the one that determines whether a lookup can skip L2 entirely.
|
|
286
|
-
*/
|
|
287
|
-
resolveKey(key: string): Promise<string>;
|
|
288
|
-
get(key: string): Promise<{
|
|
289
|
-
hit: boolean;
|
|
290
|
-
value: T | null;
|
|
291
|
-
}>;
|
|
292
|
-
set(key: string, value: T, ttl: number): Promise<void>;
|
|
293
|
-
delete(key: string): Promise<void>;
|
|
294
|
-
}
|
|
295
|
-
//#endregion
|
|
296
|
-
//#region src/logger.d.ts
|
|
297
|
-
interface Logger {
|
|
298
|
-
debug(message: string): void;
|
|
299
|
-
warn(message: string): void;
|
|
300
|
-
error(message: string, meta?: Record<string, unknown>): void;
|
|
301
|
-
}
|
|
302
|
-
/**
|
|
303
|
-
* Default logger. `debug` is gated by the `debug` option on VernLLM
|
|
304
|
-
* warn/error always fire since they indicate real problems (retries, cache failures)
|
|
305
|
-
*/
|
|
306
|
-
export declare class ConsoleLogger implements Logger {
|
|
307
|
-
private debugEnabled;
|
|
308
|
-
constructor(debugEnabled: boolean);
|
|
309
|
-
debug(message: string): void;
|
|
310
|
-
warn(message: string): void;
|
|
311
|
-
error(message: string, meta?: Record<string, unknown>): void;
|
|
312
|
-
}
|
|
313
|
-
//#endregion
|
|
314
|
-
//#region src/types/usage.d.ts
|
|
315
|
-
type ReserveUsage = (params: {
|
|
316
|
-
coalesced: boolean;
|
|
317
|
-
signal?: AbortSignal;
|
|
318
|
-
}) => Promise<void>;
|
|
319
|
-
type RefundUsage = (params: {
|
|
320
|
-
coalesced: boolean;
|
|
321
|
-
signal?: AbortSignal;
|
|
322
|
-
}) => Promise<void>;
|
|
323
|
-
/**
|
|
324
|
-
* The reserve/refund usage hooks shared by `CallParams`, `CachedCallParams`,
|
|
325
|
-
* and `VernLLM`'s internal `withReservedUsage`. Centralized here so the pair
|
|
326
|
-
* has one definition instead of being redeclared at each use site.
|
|
327
|
-
*/
|
|
328
|
-
interface UsageHooks {
|
|
329
|
-
/**
|
|
330
|
-
* Reserves usage before the request. Failures become
|
|
331
|
-
* LLMError('quota_exceeded').
|
|
332
|
-
*/
|
|
333
|
-
reserveUsage?: ReserveUsage;
|
|
334
|
-
/**
|
|
335
|
-
* Refunds usage after a failed call if reservation succeeded.
|
|
336
|
-
*/
|
|
337
|
-
refundUsage?: RefundUsage;
|
|
338
|
-
}
|
|
339
|
-
interface TokenUsage {
|
|
340
|
-
promptTokens: number;
|
|
341
|
-
completionTokens: number;
|
|
342
|
-
totalTokens: number;
|
|
343
|
-
/**
|
|
344
|
-
* Tokens spent on internal reasoning, a subset of `completionTokens`,
|
|
345
|
-
* never added on top of it. Undefined when the provider's response
|
|
346
|
-
* doesn't report a separate reasoning figure, e.g. Bedrock Converse
|
|
347
|
-
* without an explicit `additionalModelResponseFieldPaths` request.
|
|
348
|
-
*/
|
|
349
|
-
reasoningTokens?: number;
|
|
350
|
-
requestId: string;
|
|
351
|
-
model: string;
|
|
352
|
-
/**
|
|
353
|
-
* The provider target that produced this usage. See `VernLLMOptions['name']`,
|
|
354
|
-
* default `'primary'`. Optional so consumers constructing a `TokenUsage`
|
|
355
|
-
* themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
|
|
356
|
-
* populates it. Absent means the same as `'primary'` if you need a value.
|
|
357
|
-
*/
|
|
358
|
-
provider?: string;
|
|
359
|
-
/**
|
|
360
|
-
* Whether this usage came from a fallback target rather than the
|
|
361
|
-
* primary. Optional for the same reason `provider` is: `VernLLM`
|
|
362
|
-
* always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
|
|
363
|
-
* isn't forced to.
|
|
364
|
-
*/
|
|
365
|
-
usedFallback?: boolean;
|
|
366
|
-
}
|
|
367
|
-
type OnUsage = (usage: TokenUsage) => void;
|
|
368
|
-
/**
|
|
369
|
-
* Called when a provider response arrives but VernLLM's own post-processing
|
|
370
|
-
* then fails, after usage data was already present in that response. Covers
|
|
371
|
-
* any error thrown after usage extraction, not just parse/validation, since
|
|
372
|
-
* everything in that path only runs once a response, and real spend, has
|
|
373
|
-
* already arrived. Fires once per failed attempt with extractable usage,
|
|
374
|
-
* never for transport failures, where no response means no honest number
|
|
375
|
-
* to report.
|
|
376
|
-
*/
|
|
377
|
-
type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
|
|
378
|
-
//#endregion
|
|
379
|
-
//#region src/types/events.d.ts
|
|
380
|
-
/**
|
|
381
|
-
* Reports what happened during a call. Fire and forget, mirroring
|
|
382
|
-
* `onUsage`: the return value is never read and a throwing handler cannot
|
|
383
|
-
* change what the call does, only what gets reported about it.
|
|
384
|
-
*/
|
|
385
|
-
type VernLLMEvent = {
|
|
386
|
-
kind: 'retry';
|
|
387
|
-
requestId: string;
|
|
388
|
-
provider: string;
|
|
389
|
-
/** The model actually resolved for this call (honors a per-call `model` override). */
|
|
390
|
-
model: string;
|
|
391
|
-
/** The 1-based retry ordinal (the 1st retry is `1`, not the overall attempt count). */
|
|
392
|
-
attempt: number;
|
|
393
|
-
maxRetries: number;
|
|
394
|
-
delayMs: number;
|
|
395
|
-
retryAfterHonored: boolean;
|
|
396
|
-
error: LLMError;
|
|
397
|
-
} | {
|
|
398
|
-
kind: 'circuit_state';
|
|
399
|
-
provider: string;
|
|
400
|
-
/**
|
|
401
|
-
* The model of the call that triggered this specific transition
|
|
402
|
-
* (whatever was passed to the `assertClosed`/`recordSuccess`/
|
|
403
|
-
* `recordFailure` call that caused it), not a property of the
|
|
404
|
-
* circuit itself: the breaker still counts failures across every
|
|
405
|
-
* model together, so a threshold crossing can be the sum of
|
|
406
|
-
* several different models' failures even though only the
|
|
407
|
-
* triggering call's `model` is reported here.
|
|
408
|
-
*/
|
|
409
|
-
model: string;
|
|
410
|
-
from: CircuitState;
|
|
411
|
-
to: CircuitState;
|
|
412
|
-
consecutiveFailures: number;
|
|
413
|
-
} | {
|
|
414
|
-
kind: 'fallback';
|
|
415
|
-
requestId: string;
|
|
416
|
-
/** Provider name of the target that just failed. */
|
|
417
|
-
from: string;
|
|
418
|
-
/** Provider name of the target about to be tried next. */
|
|
419
|
-
to: string;
|
|
420
|
-
/** `-1` for the primary target, otherwise the index into `fallback`. */
|
|
421
|
-
fromIndex: number;
|
|
422
|
-
toIndex: number;
|
|
423
|
-
/** The normalized error that caused `from` to be abandoned. */
|
|
424
|
-
error: LLMError;
|
|
425
|
-
/** Time spent on `from`, including its own retries, before giving up. */
|
|
426
|
-
elapsedMs: number;
|
|
427
|
-
} | {
|
|
428
|
-
kind: 'rate_limited';
|
|
429
|
-
requestId: string;
|
|
430
|
-
provider: string;
|
|
431
|
-
/** The model actually resolved for this call (honors a per-call `model` override). */
|
|
432
|
-
model: string;
|
|
433
|
-
/** How long this attempt sat queued for capacity before it was let through. */
|
|
434
|
-
waitedMs: number;
|
|
435
|
-
/** Which configured bucket was blocking this attempt just before it cleared. */
|
|
436
|
-
reason: 'concurrency' | 'rpm' | 'tpm';
|
|
437
|
-
} | {
|
|
438
|
-
kind: 'middleware';
|
|
439
|
-
requestId: string;
|
|
440
|
-
/** This middleware's `name`, or its array position if unnamed. */
|
|
441
|
-
middleware: string;
|
|
442
|
-
hook: 'transform' | 'wrap_short_circuit' | 'enabled_skip';
|
|
443
|
-
/** For `hook: 'transform'` only: which top-level fields the merged patch touched. */
|
|
444
|
-
patchedFields?: string[];
|
|
445
|
-
} | {
|
|
446
|
-
/**
|
|
447
|
-
* Reported once a call fully succeeds. Same data `VernLLMOptions.onUsage`
|
|
448
|
-
* receives; that option is sugar over this event, not a second
|
|
449
|
-
* reporting path, see `makeEventReporter`.
|
|
450
|
-
*/
|
|
451
|
-
kind: 'usage';
|
|
452
|
-
requestId: string;
|
|
453
|
-
usage: TokenUsage;
|
|
454
|
-
} | {
|
|
455
|
-
/**
|
|
456
|
-
* A provider response arrived, carrying real usage, and VernLLM's own
|
|
457
|
-
* post-processing then failed. Fires once per failed attempt with
|
|
458
|
-
* extractable usage, matching `VernLLMOptions.onUsageFailure`'s own
|
|
459
|
-
* granularity, which this event is sugar over, not a second path.
|
|
460
|
-
*/
|
|
461
|
-
kind: 'usage_failure';
|
|
462
|
-
requestId: string;
|
|
463
|
-
usage: TokenUsage;
|
|
464
|
-
error: LLMError;
|
|
465
|
-
};
|
|
466
|
-
type OnEvent = (event: VernLLMEvent) => void;
|
|
467
|
-
//#endregion
|
|
468
|
-
//#region src/types/middleware.d.ts
|
|
469
|
-
/** Capabilities of the target a middleware hook is currently looking at. */
|
|
470
|
-
interface MiddlewareCapabilities {
|
|
471
|
-
/**
|
|
472
|
-
* Whether this target honors `response_format: { type: 'json_object' }`
|
|
473
|
-
* as a real constraint. Mirrors `LLMClient.supportsJsonObjectMode`.
|
|
474
|
-
* `false` for `fromAnthropic` and `fromBedrock`.
|
|
475
|
-
*/
|
|
476
|
-
supportsJsonObjectMode: boolean;
|
|
477
|
-
}
|
|
478
|
-
/**
|
|
479
|
-
* Not exported. Distinguishes `MiddlewareStateKey<T>` from
|
|
480
|
-
* `MiddlewareRef` and from a plain `{ debugName }` object literal at
|
|
481
|
-
* the type level, even though all three have the identical runtime
|
|
482
|
-
* shape. Without this, `MiddlewareStateKey<T>`/`MiddlewareRef` are
|
|
483
|
-
* structurally just `{ debugName: string }`, so TypeScript would treat
|
|
484
|
-
* a state key as a valid middleware ref (or vice versa), and would let
|
|
485
|
-
* anyone hand-write `{ debugName: 'auth' }` in place of a real
|
|
486
|
-
* `createMiddlewareRef` result. Neither is possible once this brand is
|
|
487
|
-
* required: only `createStateKey`, which alone has access to this
|
|
488
|
-
* symbol, can produce a value satisfying `MiddlewareStateKey<T>`.
|
|
489
|
-
*/
|
|
490
|
-
declare const stateKeyBrand: unique symbol;
|
|
491
|
-
/**
|
|
492
|
-
* A typed reference to one slot in `ctx.state`. Create one with
|
|
493
|
-
* `createStateKey`, export it, and import the same reference wherever
|
|
494
|
-
* another middleware needs to read or write the same value. There's no
|
|
495
|
-
* string key anywhere in this path, so a typo becomes a missing import
|
|
496
|
-
* or an undefined variable, a compile error, instead of a silently
|
|
497
|
-
* created new property.
|
|
498
|
-
*/
|
|
499
|
-
interface MiddlewareStateKey<T> {
|
|
500
|
-
readonly debugName: string;
|
|
501
|
-
readonly [stateKeyBrand]: true;
|
|
502
|
-
/**
|
|
503
|
-
* Never set at runtime; exists purely so `T` is actually used
|
|
504
|
-
* somewhere in this interface's shape (a phantom type), which is what
|
|
505
|
-
* lets `MiddlewareStateBag.get`/`set` infer the right type for a given
|
|
506
|
-
* key instead of two `MiddlewareStateKey<string>` and
|
|
507
|
-
* `MiddlewareStateKey<number>` keys being structurally identical.
|
|
508
|
-
*/
|
|
509
|
-
readonly __phantom?: T;
|
|
510
|
-
}
|
|
511
|
-
/** Creates a new, distinct `MiddlewareStateKey`. `debugName` is used only in log lines and the `'middleware'` event; it never affects equality. */
|
|
512
|
-
export declare function createStateKey<T>(debugName: string): MiddlewareStateKey<T>;
|
|
513
|
-
/** Not exported. See `stateKeyBrand`; same reasoning, distinct symbol, so the two token types can't be cross-assigned either. */
|
|
514
|
-
declare const middlewareRefBrand: unique symbol;
|
|
515
|
-
/**
|
|
516
|
-
* A typed reference to one middleware's identity, for `runsAfter`/
|
|
517
|
-
* `runsBefore` to target. Purely an ordering concern: unlike `name`,
|
|
518
|
-
* `ref` is never used as a display label anywhere (`name` still covers
|
|
519
|
-
* that), only as a `runsAfter`/`runsBefore` match target. Create one
|
|
520
|
-
* with `createMiddlewareRef`, export it from the package that owns the
|
|
521
|
-
* middleware, and have any dependent import the same reference instead
|
|
522
|
-
* of typing a matching `name` string. Same reasoning as
|
|
523
|
-
* `MiddlewareStateKey`: a typo becomes a missing import, a compile
|
|
524
|
-
* error, instead of a silently unresolved (or worse, silently
|
|
525
|
-
* colliding) string.
|
|
526
|
-
*/
|
|
527
|
-
interface MiddlewareRef {
|
|
528
|
-
readonly debugName: string;
|
|
529
|
-
readonly [middlewareRefBrand]: true;
|
|
530
|
-
}
|
|
531
|
-
/** Creates a new, distinct `MiddlewareRef`. `debugName` is used only in error messages when a reference doesn't resolve; it never affects equality, so two refs with the same `debugName` never collide. */
|
|
532
|
-
export declare function createMiddlewareRef(debugName: string): MiddlewareRef;
|
|
533
|
-
/**
|
|
534
|
-
* A `runsAfter`/`runsBefore` entry that escalates an unresolved
|
|
535
|
-
* reference from a warning to a construction-time throw. Wrap a
|
|
536
|
-
* `MiddlewareRef` with `requireRef` when the dependency isn't optional:
|
|
537
|
-
* a bare `MiddlewareRef` in `runsAfter`/`runsBefore` means "order
|
|
538
|
-
* relative to this if it's registered," which is the right default for
|
|
539
|
-
* a dependency a third party may reasonably not have installed. A
|
|
540
|
-
* `RequiredMiddlewareRef` means "this middleware must not run without
|
|
541
|
-
* that dependency having already run". The app should fail to start
|
|
542
|
-
* rather than run with a silently-missing ordering guarantee.
|
|
543
|
-
*/
|
|
544
|
-
interface RequiredMiddlewareRef {
|
|
545
|
-
readonly ref: MiddlewareRef;
|
|
546
|
-
}
|
|
547
|
-
/** Wraps `ref` so `runsAfter`/`runsBefore` throws at `VernLLM` construction time if it doesn't resolve, instead of warning and continuing. */
|
|
548
|
-
export declare function requireRef(ref: MiddlewareRef): RequiredMiddlewareRef;
|
|
549
|
-
/**
|
|
550
|
-
* Typed, per-logical-call storage two middleware can deliberately share a
|
|
551
|
-
* value through (a span ID one sets, another reads). Backed by a plain
|
|
552
|
-
* `Map` internally, created once per logical call and never read or
|
|
553
|
-
* written by VernLLM itself.
|
|
554
|
-
*/
|
|
555
|
-
interface MiddlewareStateBag {
|
|
556
|
-
get<T>(key: MiddlewareStateKey<T>): T | undefined;
|
|
557
|
-
set<T>(key: MiddlewareStateKey<T>, value: T): void;
|
|
558
|
-
}
|
|
559
|
-
/** A plain, `Map`-backed `MiddlewareStateBag`. */
|
|
560
|
-
export declare function createMiddlewareStateBag(): MiddlewareStateBag;
|
|
561
|
-
/** Fields every `MiddlewareContext` variant carries, regardless of `stage`. */
|
|
562
|
-
interface MiddlewareContextBase {
|
|
563
|
-
requestId: string;
|
|
564
|
-
/** Capabilities of the target this stage's identity fields describe. */
|
|
565
|
-
capabilities: MiddlewareCapabilities;
|
|
566
|
-
signal?: AbortSignal;
|
|
567
|
-
/** Shared, collision-proof state for two middleware to deliberately coordinate through. See `MiddlewareStateBag`. */
|
|
568
|
-
state: MiddlewareStateBag;
|
|
569
|
-
/** Simple, string-keyed scratch space, pre-namespaced to this one middleware so two middleware can never collide here even by accident. */
|
|
570
|
-
own: Record<string, unknown>;
|
|
571
|
-
/**
|
|
572
|
-
* Every registered middleware's resolved label, in `transformOrder`,
|
|
573
|
-
* frozen. Lets a middleware make an informed call, like skipping a
|
|
574
|
-
* duplicate action when it detects another known middleware by name
|
|
575
|
-
* already handles it, without needing to know anything else about
|
|
576
|
-
* that middleware's own configuration.
|
|
577
|
-
*/
|
|
578
|
-
registeredMiddlewareNames: readonly string[];
|
|
579
|
-
}
|
|
580
|
-
/**
|
|
581
|
-
* The `ctx` `transform` receives, and every attempt-scoped event context
|
|
582
|
-
* (`'retry'`, `'fallback'`, `'circuit_state'`, `'middleware'`). Built once
|
|
583
|
-
* a specific target has actually been selected for this attempt, so every
|
|
584
|
-
* field describes the real target, not a placeholder.
|
|
585
|
-
*/
|
|
586
|
-
interface AttemptContext extends MiddlewareContextBase {
|
|
587
|
-
stage: 'attempt';
|
|
588
|
-
/** The target this attempt is actually dispatched to. */
|
|
589
|
-
requestedProvider: string;
|
|
590
|
-
requestedModel: string;
|
|
591
|
-
isFallbackAttempt: boolean;
|
|
592
|
-
/**
|
|
593
|
-
* The real, current attempt number for this dispatch.
|
|
594
|
-
*
|
|
595
|
-
* Exception: on a `'circuit_state'` event triggered by a pre-dispatch
|
|
596
|
-
* check (`assertClosed`, before any attempt has been made), this is
|
|
597
|
-
* `1` regardless of which attempt is about to run, since no attempt
|
|
598
|
-
* exists yet to report. Every other `'circuit_state'` event, and every
|
|
599
|
-
* other attempt-scoped event, reports the real attempt number.
|
|
600
|
-
*/
|
|
601
|
-
attempt: number;
|
|
602
|
-
}
|
|
603
|
-
/**
|
|
604
|
-
* The `ctx` `wrap` receives before `next()` resolves (and `onError`'s own
|
|
605
|
-
* `ctx`, built the same way under the hood). Built once, before any
|
|
606
|
-
* fallback target is chosen, so it only ever describes the primary
|
|
607
|
-
* target. There is no real "requested" target yet, and no attempt count,
|
|
608
|
-
* fallback flag, or per-attempt capability to report. Read `next()`'s
|
|
609
|
-
* resolved `CallResult.meta` once you need to know what actually
|
|
610
|
-
* happened.
|
|
611
|
-
*/
|
|
612
|
-
interface PreDispatchContext extends MiddlewareContextBase {
|
|
613
|
-
stage: 'pre-dispatch';
|
|
614
|
-
/** The primary target only, not necessarily who ends up answering. */
|
|
615
|
-
primaryProvider: string;
|
|
616
|
-
primaryModel: string;
|
|
617
|
-
}
|
|
618
|
-
/**
|
|
619
|
-
* `enabled` and `onEvent` are called from both stages (gating/observing
|
|
620
|
-
* `transform` as well as `wrap`), so they receive this union and must
|
|
621
|
-
* narrow on `ctx.stage` before reading stage-specific fields.
|
|
622
|
-
* `transform` and `wrap` themselves receive the single variant that's
|
|
623
|
-
* always accurate for them (`AttemptContext`/`PreDispatchContext`
|
|
624
|
-
* respectively). See `VernLLMMiddleware`.
|
|
625
|
-
*/
|
|
626
|
-
type MiddlewareContext = AttemptContext | PreDispatchContext;
|
|
627
|
-
/** The `response_format` shape `RequestBuilder` can put on the wire. */
|
|
628
|
-
type WireResponseFormat = {
|
|
629
|
-
type: 'json_object';
|
|
630
|
-
} | {
|
|
631
|
-
type: 'json_schema';
|
|
632
|
-
json_schema: {
|
|
633
|
-
name: string;
|
|
634
|
-
schema: Record<string, unknown>;
|
|
635
|
-
strict?: boolean;
|
|
636
|
-
description?: string;
|
|
637
|
-
};
|
|
638
|
-
};
|
|
639
|
-
/** A tool as it appears on the wire, OpenAI's `function`-wrapped shape. */
|
|
640
|
-
interface WireTool {
|
|
641
|
-
type: 'function';
|
|
642
|
-
function: {
|
|
643
|
-
name: string;
|
|
644
|
-
description: string;
|
|
645
|
-
parameters: Record<string, unknown>;
|
|
646
|
-
};
|
|
647
|
-
}
|
|
648
|
-
/**
|
|
649
|
-
* The wire-shaped request `RequestBuilder.build()` produces for one call
|
|
650
|
-
* attempt, before dispatch. Read only inside `transform`; return a patch
|
|
651
|
-
* of the fields you want to change instead of the whole object.
|
|
652
|
-
*/
|
|
653
|
-
interface WireCallRequest {
|
|
654
|
-
model: string;
|
|
655
|
-
temperature?: number;
|
|
656
|
-
max_tokens: number;
|
|
657
|
-
response_format?: WireResponseFormat;
|
|
658
|
-
reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
659
|
-
budget_tokens?: number;
|
|
660
|
-
tools?: WireTool[];
|
|
661
|
-
tool_choice?: WireToolChoice;
|
|
662
|
-
messages: WireMessage[];
|
|
663
|
-
}
|
|
664
|
-
/**
|
|
665
|
-
* What `transform` returns: a patch merged onto the request that
|
|
666
|
-
* `RequestBuilder.build()` (plus every earlier middleware's own patch)
|
|
667
|
-
* already produced, not a replacement for it. `model` and
|
|
668
|
-
* `response_format` can't be expressed here at all, since everything
|
|
669
|
-
* downstream that attributes a call to a target keys off the values
|
|
670
|
-
* `RequestBuilder` already resolved for those two fields, not off
|
|
671
|
-
* whatever ends up on the wire request. `messages`/`tools` are joined by
|
|
672
|
-
* a separate `add*` field, appended rather than replaced, so two
|
|
673
|
-
* independently written middleware can each add to the list without one
|
|
674
|
-
* silently clobbering what the other already added.
|
|
675
|
-
*/
|
|
676
|
-
interface WireCallRequestPatch {
|
|
677
|
-
temperature?: number;
|
|
678
|
-
max_tokens?: number;
|
|
679
|
-
reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
680
|
-
budget_tokens?: number;
|
|
681
|
-
tool_choice?: WireToolChoice;
|
|
682
|
-
/** Replaces the whole message list. Prefer `addMessages` unless a full replace is genuinely the intent. */
|
|
683
|
-
messages?: WireMessage[];
|
|
684
|
-
/** Appended after whatever earlier middleware already added. Never clobbers a prior addition. */
|
|
685
|
-
addMessages?: WireMessage[];
|
|
686
|
-
/** Replaces the whole tool list. Prefer `addTools`, same reasoning as `messages`/`addMessages`. */
|
|
687
|
-
tools?: WireTool[];
|
|
688
|
-
/** Appended after whatever earlier middleware already added. Never clobbers a prior addition. */
|
|
689
|
-
addTools?: WireTool[];
|
|
690
|
-
}
|
|
691
|
-
/**
|
|
692
|
-
* The settled outcome of one logical call, passed to `wrap`'s `next()`.
|
|
693
|
-
* `meta` is populated once a target has actually answered, for both
|
|
694
|
-
* streaming and non-streaming calls (`undefined` only on a cache hit,
|
|
695
|
-
* where nothing was actually spent).
|
|
696
|
-
*/
|
|
697
|
-
interface CallResult<T = unknown> {
|
|
698
|
-
value: T;
|
|
699
|
-
meta?: CallMeta;
|
|
700
|
-
}
|
|
701
|
-
/**
|
|
702
|
-
* One entry in `VernLLMOptions.middleware`. All four hooks are optional;
|
|
703
|
-
* an entry that sets none of them is inert. See the middleware docs for
|
|
704
|
-
* how `transform`, `wrap`, `onEvent`, and `enabled` compose across
|
|
705
|
-
* several entries.
|
|
706
|
-
*/
|
|
707
|
-
interface VernLLMMiddleware {
|
|
708
|
-
/** Used in log lines and the `'middleware'` event. Defaults to this entry's array position when omitted. */
|
|
709
|
-
name?: string;
|
|
710
|
-
/**
|
|
711
|
-
* This entry's own identity, purely for another middleware's
|
|
712
|
-
* `runsAfter`/`runsBefore` to target. Create with `createMiddlewareRef`,
|
|
713
|
-
* export it, and have a dependent import the same reference. Optional:
|
|
714
|
-
* only needed if something else must be able to depend on this
|
|
715
|
-
* specific entry. Unrelated to `name`: `ref` is never shown in logs,
|
|
716
|
-
* `name` is never matched against for ordering.
|
|
717
|
-
*/
|
|
718
|
-
ref?: MiddlewareRef;
|
|
719
|
-
/** Sort key for composition order, ascending, ties broken by array order. See the middleware docs for what "lower runs first" means for `wrap`. */
|
|
720
|
-
priority?: number;
|
|
721
|
-
/**
|
|
722
|
-
* Other middleware this entry must run after, breaking ties
|
|
723
|
-
* `priority` alone can't express. Matched by `ref` identity, so a
|
|
724
|
-
* typo or a stale copy simply fails to resolve instead of silently
|
|
725
|
-
* matching the wrong entry. A bare `MiddlewareRef` that doesn't
|
|
726
|
-
* resolve is dropped, not an error, since a third party may
|
|
727
|
-
* reasonably reference a well known middleware that isn't installed
|
|
728
|
-
* everywhere; wrap it with `requireRef` to make that same target
|
|
729
|
-
* mandatory instead, throwing at `VernLLM` construction time if it's
|
|
730
|
-
* missing. A cycle across `runsAfter`/`runsBefore` always throws,
|
|
731
|
-
* regardless of whether any individual entry is required.
|
|
732
|
-
*/
|
|
733
|
-
runsAfter?: (MiddlewareRef | RequiredMiddlewareRef)[];
|
|
734
|
-
/**
|
|
735
|
-
* Other middleware this entry must run before. See `runsAfter`; a
|
|
736
|
-
* bare reference is dropped if unresolved, a `requireRef`-wrapped one
|
|
737
|
-
* throws.
|
|
738
|
-
*/
|
|
739
|
-
runsBefore?: (MiddlewareRef | RequiredMiddlewareRef)[];
|
|
740
|
-
/**
|
|
741
|
-
* Pins this entry's slot in `wrap` nesting only, independent of
|
|
742
|
-
* `priority`/`runsAfter`/`runsBefore`, which still govern
|
|
743
|
-
* `transform`/`onEvent` order. `'outermost'` sees the net
|
|
744
|
-
* `CallResult` of every retry, fallback, and other middleware's
|
|
745
|
-
* `wrap`; `'innermost'` sits closest to the real dispatch. A numeric
|
|
746
|
-
* value behaves like `priority`, but only for `wrap` nesting.
|
|
747
|
-
*/
|
|
748
|
-
position?: 'outermost' | 'innermost' | number;
|
|
749
|
-
/**
|
|
750
|
-
* Boolean for a static on/off switch, or a predicate evaluated per
|
|
751
|
-
* call. A throwing, rejecting, or timed-out predicate is logged and
|
|
752
|
-
* treated as `false` for that call.
|
|
753
|
-
*/
|
|
754
|
-
enabled?: boolean | ((ctx: MiddlewareContext) => boolean | Promise<boolean>);
|
|
755
|
-
/** Per-middleware override of the instance-level `middlewareTimeoutMs`, applied to this entry's `transform` and function `enabled`. `<= 0` means unbounded (no timer at all). */
|
|
756
|
-
timeoutMs?: number;
|
|
757
|
-
/** Transforms the outgoing wire request for one attempt. Runs once per attempt, including retries. `ctx` is always accurate to the real target for this attempt. */
|
|
758
|
-
transform?: (request: Readonly<WireCallRequest>, ctx: AttemptContext) => WireCallRequestPatch | Promise<WireCallRequestPatch>;
|
|
759
|
-
/**
|
|
760
|
-
* Wraps one whole logical call, exactly once, regardless of how many
|
|
761
|
-
* retries or fallback targets ran underneath it. `ctx` is built once,
|
|
762
|
-
* before any fallback target is chosen, so it only describes the
|
|
763
|
-
* primary target. There is no `requestedProvider`/`isFallbackAttempt`/
|
|
764
|
-
* `attempt` to read here. Read `next()`'s resolved `CallResult.meta`
|
|
765
|
-
* for what actually happened.
|
|
766
|
-
*/
|
|
767
|
-
wrap?: (request: Readonly<WireCallRequest>, next: () => Promise<CallResult>, ctx: PreDispatchContext) => Promise<CallResult>;
|
|
768
|
-
/** Observes the same events reported on `VernLLMOptions.onEvent`, filtered by this middleware's own `enabled`. Called from both stages; narrow on `ctx.stage` before reading stage-specific fields. */
|
|
769
|
-
onEvent?: (event: VernLLMEvent, ctx: MiddlewareContext) => void;
|
|
770
|
-
}
|
|
771
|
-
//#endregion
|
|
772
|
-
//#region src/circuitBreaker.d.ts
|
|
773
|
-
/** The call this mutation happened as part of, forwarded to `onStateChange` untouched. */
|
|
774
|
-
interface CircuitBreakerCallContext {
|
|
775
|
-
requestId: string;
|
|
776
|
-
state: MiddlewareStateBag;
|
|
777
|
-
signal?: AbortSignal;
|
|
778
|
-
/** Omitted for calls before any attempt exists, like `assertClosed`'s pre-dispatch check. */
|
|
779
|
-
attempt?: number;
|
|
780
|
-
}
|
|
781
|
-
/**
|
|
782
|
-
* Fires after every real state change, never a no-op transition. `model`
|
|
783
|
-
* is the resolved model of whichever call triggered it. With
|
|
784
|
-
* `isolateByModel` off, failures are still counted across every model.
|
|
785
|
-
* Shared by `CircuitBreakerOptions` and `CircuitBreakerAdapter`, so a
|
|
786
|
-
* custom adapter reports state changes the same way the built in
|
|
787
|
-
* `CircuitBreaker` does.
|
|
788
|
-
*/
|
|
789
|
-
type CircuitBreakerStateChangeHandler = (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string, context?: CircuitBreakerCallContext) => void;
|
|
790
|
-
interface CircuitBreakerOptions {
|
|
791
|
-
/** Consecutive failures before the circuit opens, default 5 */
|
|
792
|
-
threshold?: number;
|
|
793
|
-
/** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
|
|
794
|
-
cooldownMs?: number;
|
|
795
|
-
onStateChange?: CircuitBreakerStateChangeHandler;
|
|
796
|
-
/**
|
|
797
|
-
* Track a separate circuit per resolved model instead of one shared
|
|
798
|
-
* circuit. Default false. A call that omits `model` falls into one
|
|
799
|
-
* shared bucket alongside every other call that also omits it.
|
|
800
|
-
*/
|
|
801
|
-
isolateByModel?: boolean;
|
|
802
|
-
/** Trial calls allowed through per half-open cycle. Default 1, clamped to at least 1. */
|
|
803
|
-
halfOpenProbes?: number;
|
|
804
|
-
/** Fraction of `halfOpenProbes` that must succeed to close the circuit. Default 1, clamped to `[0, 1]`. */
|
|
805
|
-
halfOpenSuccessRatio?: number;
|
|
806
|
-
/**
|
|
807
|
-
* Grows `cooldownMs` on each repeat open instead of a fixed wait.
|
|
808
|
-
* `{ multiplier, maxMs }` covers exponential growth; a `CooldownBackoff`
|
|
809
|
-
* function covers anything else. Omitted means `cooldownMs` stays fixed.
|
|
810
|
-
*/
|
|
811
|
-
cooldownBackoff?: ExponentialBackoffOptions | CooldownBackoff;
|
|
812
|
-
/**
|
|
813
|
-
* Decides when a bucket's failures should open the circuit.
|
|
814
|
-
* `{ kind: 'consecutive', threshold }` (the default) opens after that
|
|
815
|
-
* many failures in a row. `{ kind: 'rolling', windowMs, minCalls,
|
|
816
|
-
* failureRatio }` opens once at least `minCalls` calls have landed in
|
|
817
|
-
* the trailing `windowMs` and the failure ratio reaches `failureRatio`.
|
|
818
|
-
* `minCalls` must be a non-negative integer; `failureRatio` must be
|
|
819
|
-
* finite and within `[0, 1]`. Both are validated at construction,
|
|
820
|
-
* thrown as `RangeError`. A `TrippingPolicy` covers anything else, one
|
|
821
|
-
* instance shared across every model automatically under
|
|
822
|
-
* `isolateByModel`, since it tracks its own state per key rather than
|
|
823
|
-
* owning one flat counter.
|
|
824
|
-
*/
|
|
825
|
-
tripping?: TrippingOption;
|
|
826
|
-
}
|
|
827
|
-
/** Computes the cooldown for a bucket's `reopenCount`-th repeat open. */
|
|
828
|
-
type CooldownBackoff = (reopenCount: number, baseCooldownMs: number) => number;
|
|
829
|
-
interface ExponentialBackoffOptions {
|
|
830
|
-
/** Growth factor applied per repeat open, e.g. 2 doubles each time. */
|
|
831
|
-
multiplier: number;
|
|
832
|
-
/** Upper bound on the computed cooldown, in ms. Default `Infinity`. */
|
|
833
|
-
maxMs?: number;
|
|
834
|
-
}
|
|
835
|
-
/**
|
|
836
|
-
* Decides when a bucket's failures should open the circuit. Keyed by
|
|
837
|
-
* `key` (a resolved model, or the shared bucket's key when
|
|
838
|
-
* `isolateByModel` is off) rather than holding one flat counter, so a
|
|
839
|
-
* single `TrippingPolicy` instance is always safe to share across every
|
|
840
|
-
* bucket: `CircuitBreaker` never needs to clone or construct a fresh one
|
|
841
|
-
* per model, `isolateByModel` isolation falls out of `key` alone.
|
|
842
|
-
*/
|
|
843
|
-
interface TrippingPolicy {
|
|
844
|
-
onSuccess(key: string): void;
|
|
845
|
-
/** Returns true if this failure should open the circuit for `key`. */
|
|
846
|
-
onFailure(key: string): boolean;
|
|
847
|
-
reset(key: string): void;
|
|
848
|
-
/**
|
|
849
|
-
* Called when `key`'s bucket is discarded (closed and idle, under
|
|
850
|
-
* `isolateByModel`), so a keyed policy can release that key's state.
|
|
851
|
-
* Optional: omit if there's nothing to release.
|
|
852
|
-
*/
|
|
853
|
-
forget?(key: string): void;
|
|
854
|
-
}
|
|
855
|
-
export declare class ConsecutiveTripping implements TrippingPolicy {
|
|
856
|
-
private readonly threshold;
|
|
857
|
-
private failuresByKey;
|
|
858
|
-
constructor(threshold: number);
|
|
859
|
-
onSuccess(key: string): void;
|
|
860
|
-
onFailure(key: string): boolean;
|
|
861
|
-
reset(key: string): void;
|
|
862
|
-
forget(key: string): void;
|
|
863
|
-
}
|
|
864
|
-
export declare class RollingTripping implements TrippingPolicy {
|
|
865
|
-
private readonly windowMs;
|
|
866
|
-
private readonly minCalls;
|
|
867
|
-
private readonly failureRatio;
|
|
868
|
-
private ratiosByKey;
|
|
869
|
-
constructor(windowMs: number, minCalls: number, failureRatio: number);
|
|
870
|
-
private ratioFor;
|
|
871
|
-
onSuccess(key: string): void;
|
|
872
|
-
onFailure(key: string): boolean;
|
|
873
|
-
reset(key: string): void;
|
|
874
|
-
forget(key: string): void;
|
|
875
|
-
}
|
|
876
|
-
/** Not exported. Internal shorthand union for `CircuitBreakerOptions.tripping`. */
|
|
877
|
-
type TrippingOption = {
|
|
878
|
-
kind: 'consecutive';
|
|
879
|
-
threshold: number;
|
|
880
|
-
} | {
|
|
881
|
-
kind: 'rolling';
|
|
882
|
-
windowMs: number;
|
|
883
|
-
minCalls: number;
|
|
884
|
-
failureRatio: number;
|
|
885
|
-
} | TrippingPolicy;
|
|
886
|
-
type CircuitState = 'closed' | 'open' | 'half-open';
|
|
887
|
-
/**
|
|
888
|
-
* What VernLLM's dispatch layer needs from a breaker. `CircuitBreaker`
|
|
889
|
-
* implements this; a caller wanting cross process coordination can hand
|
|
890
|
-
* over their own instance instead.
|
|
891
|
-
*
|
|
892
|
-
* `assertClosed`, `recordSuccess`, `recordFailure`, and `onStateChange`
|
|
893
|
-
* are required, mirroring `RateLimiterAdapter`'s four required methods.
|
|
894
|
-
* `onStateChange` is required so `circuit_state` events can't go
|
|
895
|
-
* silently missing; a no-op `() => {}` is fine if you don't care.
|
|
896
|
-
*
|
|
897
|
-
* `getState`, `getFailureBreakdown`, `isolateByModel`, `open`, and
|
|
898
|
-
* `close` are optional. Omitting one makes the matching call a no-op
|
|
899
|
-
* or return `undefined`/`false`, same as no breaker configured.
|
|
900
|
-
* `open`/`close` are optional since they let VernLLM force a
|
|
901
|
-
* transition, control a distributed adapter may not want to grant.
|
|
902
|
-
*/
|
|
903
|
-
interface CircuitBreakerAdapter {
|
|
904
|
-
/** Throws when the circuit is open (or half open with no trial slot free) for `model`. */
|
|
905
|
-
assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
|
|
906
|
-
recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
|
|
907
|
-
/** `code`, when present, is the failing call's `LLMErrorCode`. */
|
|
908
|
-
recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
|
|
909
|
-
getState?(model?: string): CircuitState;
|
|
910
|
-
/** Failure counts by `LLMErrorCode` for `model`'s bucket, `'unknown'` for one that carried no code. */
|
|
911
|
-
getFailureBreakdown?(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
|
|
912
|
-
/** Whether this adapter tracks failures per model, mirroring `CircuitBreakerOptions.isolateByModel`. Read by `warnIfModelUnsupported`'s diagnostic warning and by `VernLLM.getCircuitStates()`'s public output; omit if the notion doesn't apply to your adapter, `false` is assumed. */
|
|
913
|
-
isolateByModel?: boolean;
|
|
914
|
-
/** Manually opens the circuit, as if enough consecutive failures had just happened. Optional: an adapter that doesn't want external callers forcing a transition can omit it. */
|
|
915
|
-
open?(model?: string, context?: CircuitBreakerCallContext): void;
|
|
916
|
-
/** Manually closes the circuit, without requiring a real success first. Same opt-in reasoning as `open`. */
|
|
917
|
-
close?(model?: string, context?: CircuitBreakerCallContext): void;
|
|
918
|
-
/** Gives back a half-open trial slot when a call ends without `recordSuccess` or `recordFailure`. Idempotent, and a no-op for a call that holds no slot. */
|
|
919
|
-
releaseTrial?(model?: string, context?: CircuitBreakerCallContext): void;
|
|
920
|
-
/** Awaited right before `assertClosed` to refresh local state. Never blocks or fails a call: a rejection or `prepareTimeoutMs` is logged and the call carries on. */
|
|
921
|
-
prepare?(model?: string, context?: CircuitBreakerCallContext): Promise<void>;
|
|
922
|
-
/** How long to wait for `prepare`, in ms. Default 1000. */
|
|
923
|
-
prepareTimeoutMs?: number;
|
|
924
|
-
/** Live counterpart of `getState`, read by `VernLLM.readCircuitStates()`. */
|
|
925
|
-
readState?(model?: string): Promise<CircuitState>;
|
|
926
|
-
/** Receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
|
|
927
|
-
setLogger?(logger: Logger): void;
|
|
928
|
-
/**
|
|
929
|
-
* Called after every real state change, never a no-op transition. VernLLM
|
|
930
|
-
* wraps it the same way it wraps the built in `CircuitBreaker`'s
|
|
931
|
-
* `onStateChange`: every call still reports a `circuit_state` event
|
|
932
|
-
* first, then this hook is chained after that, wrapped so a throw here
|
|
933
|
-
* can't break the call that triggered it.
|
|
934
|
-
*/
|
|
935
|
-
onStateChange: CircuitBreakerStateChangeHandler;
|
|
936
|
-
}
|
|
937
|
-
/**
|
|
938
|
-
* Per retry VernLLM-instance circuit breaker. Tracks consecutive failures
|
|
939
|
-
* across calls. Once the threshold is hit, short-circuits new calls with
|
|
940
|
-
* LLMError('circuit_open') until the cooldown elapses and a trial succeeds.
|
|
941
|
-
*/
|
|
942
|
-
export declare class CircuitBreaker implements CircuitBreakerAdapter {
|
|
943
|
-
private readonly cooldownMs;
|
|
944
|
-
/** Satisfies `CircuitBreakerAdapter.onStateChange`, required there. Defaults to a no-op when `options.onStateChange` is omitted. */
|
|
945
|
-
readonly onStateChange: CircuitBreakerStateChangeHandler;
|
|
946
|
-
/** Whether this breaker tracks failures per model instead of one shared circuit. */
|
|
947
|
-
readonly isolateByModel: boolean;
|
|
948
|
-
private readonly halfOpenProbes;
|
|
949
|
-
private readonly halfOpenSuccessRatio;
|
|
950
|
-
private readonly cooldownBackoff?;
|
|
951
|
-
/** One instance, keyed per model internally. See `TrippingPolicy`. */
|
|
952
|
-
private readonly tripping;
|
|
953
|
-
private readonly sharedBucket;
|
|
954
|
-
private readonly bucketsByModel;
|
|
955
|
-
constructor(options?: CircuitBreakerOptions);
|
|
956
|
-
/**
|
|
957
|
-
* Throws if the circuit is open and the cooldown hasn't elapsed, or if
|
|
958
|
-
* half-open with every trial slot claimed. Otherwise claims a trial slot.
|
|
959
|
-
*/
|
|
960
|
-
assertClosed(model?: string, context?: CircuitBreakerCallContext): void;
|
|
961
|
-
recordSuccess(model?: string, context?: CircuitBreakerCallContext): void;
|
|
962
|
-
/** `code`, when present, is the failing `LLMError`'s `code`. Missing attributes to `'unknown'`. */
|
|
963
|
-
recordFailure(model?: string, context?: CircuitBreakerCallContext, code?: LLMErrorCode): void;
|
|
964
|
-
/**
|
|
965
|
-
* Gives back the half-open trial slot `context`'s call claimed, when
|
|
966
|
-
* that call ended without recording an outcome. No-op without a
|
|
967
|
-
* `context`, when the bucket isn't half-open, or when the call's permit
|
|
968
|
-
* is stale or already spent (an outcome was recorded), so calling it
|
|
969
|
-
* defensively on every failure path is safe.
|
|
970
|
-
*/
|
|
971
|
-
releaseTrial(model?: string, context?: CircuitBreakerCallContext): void;
|
|
972
|
-
/** With `isolateByModel` off, `model` is ignored and the shared circuit's state is returned. */
|
|
973
|
-
getState(model?: string): CircuitState;
|
|
974
|
-
/** Failure counts by `LLMErrorCode` for `model`'s bucket. Returned as a plain object copy. */
|
|
975
|
-
getFailureBreakdown(model?: string): Partial<Record<LLMErrorCode | 'unknown', number>>;
|
|
976
|
-
/** Manually opens the circuit, as if `threshold` consecutive failures had just happened. */
|
|
977
|
-
open(model?: string, context?: CircuitBreakerCallContext): void;
|
|
978
|
-
/** Manually closes the circuit and resets its failure count, without requiring a real success first. */
|
|
979
|
-
close(model?: string, context?: CircuitBreakerCallContext): void;
|
|
980
|
-
/**
|
|
981
|
-
* Opens `bucket`: stamps `openedAt`/`cooldownMsForOpen` and transitions
|
|
982
|
-
* to `open`. Shared by `recordFailure`'s trip, `settleTrialIfComplete`'s
|
|
983
|
-
* reopen, and the manual `open()`, all of which reach this with
|
|
984
|
-
* `bucket.trial` already `null`.
|
|
985
|
-
*/
|
|
986
|
-
private openBucket;
|
|
987
|
-
/** Computes and clamps the cooldown for `bucket`'s current `reopenCount`. Called once, on open. */
|
|
988
|
-
private computeCooldown;
|
|
989
|
-
/** Returns the bucket for a model if one already exists, without allocating. */
|
|
990
|
-
private lookupBucket;
|
|
991
|
-
/**
|
|
992
|
-
* The key `tripping` is called with. Real per-model isolation under
|
|
993
|
-
* `isolateByModel`, matching `ensureBucketFor`/`lookupBucket`'s own
|
|
994
|
-
* per-model key. Otherwise one fixed shared key regardless of what
|
|
995
|
-
* `model` was passed, matching `sharedBucket` being the one and only
|
|
996
|
-
* bucket in that mode: `model` is never allowed to split tripping state
|
|
997
|
-
* when `isolateByModel` is off, the same way it never splits which
|
|
998
|
-
* bucket a call lands in.
|
|
999
|
-
*/
|
|
1000
|
-
private trippingKeyFor;
|
|
1001
|
-
/** Creates and stores a bucket for a model when the first mutation needs one. */
|
|
1002
|
-
private ensureBucketFor;
|
|
1003
|
-
/** Drops an idle model's bucket and lets `tripping` release that key's state too. */
|
|
1004
|
-
private forgetModel;
|
|
1005
|
-
/** Every state mutation routes through here, so `onStateChange` fires exactly once per real change. */
|
|
1006
|
-
private transition;
|
|
1007
|
-
/** Once every admitted trial has reported in, closes or reopens based on `halfOpenSuccessRatio`. */
|
|
1008
|
-
private settleTrialIfComplete;
|
|
1009
|
-
}
|
|
1010
|
-
//#endregion
|
|
1011
|
-
//#region src/internal/retryBudget.d.ts
|
|
1012
|
-
/**
|
|
1013
|
-
* Tunables for a `RetryBudget`. `windowMs`/`minCalls` behave the same as
|
|
1014
|
-
* `RollingTripping`'s (see `circuitBreaker.ts`): `minCalls` gates the
|
|
1015
|
-
* check so a cold start with too little traffic to judge doesn't trip.
|
|
1016
|
-
* `retryRatio` is the max fraction of calls in the window allowed to be
|
|
1017
|
-
* retries before the budget stops allowing more. `minCalls` must be a
|
|
1018
|
-
* non-negative integer; `retryRatio` must be finite and within `[0, 1]`.
|
|
1019
|
-
* Both are validated at construction, thrown as `RangeError`.
|
|
1020
|
-
*/
|
|
1021
|
-
interface RetryBudgetOptions {
|
|
1022
|
-
windowMs: number;
|
|
1023
|
-
minCalls: number;
|
|
1024
|
-
retryRatio: number;
|
|
1025
|
-
}
|
|
1026
|
-
/**
|
|
1027
|
-
* Caps how much of a target's recent traffic is allowed to be retries,
|
|
1028
|
-
* independent of the circuit breaker. The breaker asks whether the
|
|
1029
|
-
* provider is healthy; this asks whether retrying is still worth the
|
|
1030
|
-
* capacity it costs, regardless of provider health. Reuses `RollingRatio`,
|
|
1031
|
-
* the same primitive `RollingTripping` is built on, rather than a second
|
|
1032
|
-
* hand rolled window.
|
|
1033
|
-
*/
|
|
1034
|
-
export declare class RetryBudget {
|
|
1035
|
-
private readonly options;
|
|
1036
|
-
private readonly ratio;
|
|
1037
|
-
constructor(options: RetryBudgetOptions);
|
|
1038
|
-
/**
|
|
1039
|
-
* Throws `LLMError('retry_budget_exhausted')` once at least `minCalls`
|
|
1040
|
-
* calls have landed in the trailing `windowMs` and the retry ratio
|
|
1041
|
-
* among them has reached `retryRatio`. A no-op otherwise.
|
|
1042
|
-
*/
|
|
1043
|
-
assertAvailable(): void;
|
|
1044
|
-
/** Records one attempt. `isRetry` is false for a call's first attempt, true for every attempt after it. */
|
|
1045
|
-
recordAttempt(isRetry: boolean): void;
|
|
1046
|
-
/** Current traffic and retry ratio in the trailing window. */
|
|
1047
|
-
getSnapshot(): {
|
|
1048
|
-
attempts: number;
|
|
1049
|
-
retryRatio: number;
|
|
1050
|
-
};
|
|
1051
|
-
}
|
|
1052
|
-
//#endregion
|
|
1053
|
-
//#region src/internal/utils/rate-limit/rateLimitHint.utils.d.ts
|
|
1054
|
-
/** A normalized read of a provider's rate limit headers. */
|
|
1055
|
-
interface ProviderRateLimitHint {
|
|
1056
|
-
remainingRequests?: number;
|
|
1057
|
-
limitRequests?: number;
|
|
1058
|
-
resetAfterMs?: number;
|
|
1059
|
-
}
|
|
1060
|
-
//#endregion
|
|
1061
|
-
//#region src/rateLimit.d.ts
|
|
1062
|
-
/** The request shape sent to `LLMClient['chat']['completions']['create']`, used for token estimation. */
|
|
1063
|
-
type WireRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
|
|
1064
|
-
/** Which configured bucket is currently blocking a call. */
|
|
1065
|
-
type RateLimitReason = 'concurrency' | 'rpm' | 'tpm';
|
|
1066
|
-
interface RateLimitOptions {
|
|
1067
|
-
/** Max requests per minute. Omit for unlimited. */
|
|
1068
|
-
requestsPerMinute?: number;
|
|
1069
|
-
/**
|
|
1070
|
-
* Max tokens per minute. Enforced against a pre-flight estimate, then
|
|
1071
|
-
* reconciled against reported usage once the call completes. Omit for
|
|
1072
|
-
* unlimited.
|
|
1073
|
-
*/
|
|
1074
|
-
tokensPerMinute?: number;
|
|
1075
|
-
/** Max requests in flight at once. Default 0, meaning unlimited. */
|
|
1076
|
-
maxConcurrent?: number;
|
|
1077
|
-
/**
|
|
1078
|
-
* Max time a call may sit queued waiting for capacity, in ms. Exceeding
|
|
1079
|
-
* it throws rather than hanging forever. Default 30000. Pass 0 to wait
|
|
1080
|
-
* indefinitely.
|
|
1081
|
-
*/
|
|
1082
|
-
maxQueueMs?: number;
|
|
1083
|
-
/** Max queued calls before new ones reject immediately instead of queueing. Default 0, unbounded. */
|
|
1084
|
-
maxQueueSize?: number;
|
|
1085
|
-
/**
|
|
1086
|
-
* Pre-flight token estimate for `tokensPerMinute`. Defaults to a
|
|
1087
|
-
* chars/4 heuristic over message content plus `max_tokens`.
|
|
1088
|
-
*/
|
|
1089
|
-
estimateTokens?: (request: WireRequest) => number;
|
|
1090
|
-
/**
|
|
1091
|
-
* Scales the pre-flight estimate down before it's reserved against
|
|
1092
|
-
* `tokensPerMinute`, since most calls don't use their full `max_tokens`
|
|
1093
|
-
* budget. Applied after `estimateTokens`, as rate-limiter bookkeeping
|
|
1094
|
-
* only; never changes the `max_tokens` sent to the provider.
|
|
1095
|
-
* `release`'s `actualTokens` still reconciles against real usage
|
|
1096
|
-
* afterward. Default `1` (today's behavior, no scaling). Must be a
|
|
1097
|
-
* finite number greater than `0`; values above `1` are clamped to `1`.
|
|
1098
|
-
*/
|
|
1099
|
-
estimateFraction?: number;
|
|
1100
|
-
/**
|
|
1101
|
-
* AIMD against the `requestsPerMinute` bucket. Omit for a fixed
|
|
1102
|
-
* ceiling, today's behavior. Requires `requestsPerMinute`.
|
|
1103
|
-
*/
|
|
1104
|
-
aimd?: AimdOptions;
|
|
1105
|
-
}
|
|
1106
|
-
interface AimdOptions {
|
|
1107
|
-
/** Added to the requests-per-minute ceiling on every clean release. */
|
|
1108
|
-
increaseBy: number;
|
|
1109
|
-
/** Multiplied against the ceiling on a rate-limit signal. Must be greater than `0` and at most `1`; clamped otherwise. */
|
|
1110
|
-
decreaseFactor: number;
|
|
1111
|
-
/** Floor the ceiling never shrinks below. */
|
|
1112
|
-
minCapacity: number;
|
|
1113
|
-
/** Ceiling the bucket never grows above. */
|
|
1114
|
-
maxCapacity: number;
|
|
1115
|
-
/**
|
|
1116
|
-
* Shrink proactively once a provider hint reports `remainingRequests`
|
|
1117
|
-
* at or below this, before a real 429 happens. Default 0, meaning
|
|
1118
|
-
* off.
|
|
1119
|
-
*/
|
|
1120
|
-
proactiveFloor?: number;
|
|
1121
|
-
}
|
|
1122
|
-
interface RateLimitState {
|
|
1123
|
-
/** Requests still available this window, or `undefined` if `requestsPerMinute` isn't configured. */
|
|
1124
|
-
requestsRemaining?: number;
|
|
1125
|
-
/** Tokens still available this window, or `undefined` if `tokensPerMinute` isn't configured. */
|
|
1126
|
-
tokensRemaining?: number;
|
|
1127
|
-
/** Concurrency slots currently in use, or `undefined` if `maxConcurrent` isn't configured. */
|
|
1128
|
-
concurrentInFlight?: number;
|
|
1129
|
-
}
|
|
1130
|
-
interface RateLimitAcquireResult {
|
|
1131
|
-
/**
|
|
1132
|
-
* Releases the concurrency slot this attempt held and reconciles the
|
|
1133
|
-
* token bucket against real usage, when `actualTokens` is supplied.
|
|
1134
|
-
* Idempotent: only the first call does anything. Must run in a
|
|
1135
|
-
* `finally` block so a slot is never leaked on a failed attempt.
|
|
1136
|
-
*/
|
|
1137
|
-
release: (actualTokens?: number, success?: boolean) => void;
|
|
1138
|
-
/** How long this attempt waited in queue before capacity was available. */
|
|
1139
|
-
waitedMs: number;
|
|
1140
|
-
/** Which bucket was blocking this attempt just before it cleared, if any wait happened. */
|
|
1141
|
-
reason?: RateLimitReason;
|
|
1142
|
-
}
|
|
1143
|
-
/** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
|
|
1144
|
-
export declare function defaultEstimateTokens(request: WireRequest): number;
|
|
1145
|
-
/**
|
|
1146
|
-
* What VernLLM's dispatch layer needs from a limiter. `RateLimiter`
|
|
1147
|
-
* implements this; a caller wanting cross-process coordination can hand
|
|
1148
|
-
* over their own instance instead, see `buildRateLimit`. Every method is
|
|
1149
|
-
* required, `RateLimiter` itself already no-ops the AIMD methods when
|
|
1150
|
-
* `aimd` isn't configured, so a custom limiter follows the same pattern.
|
|
1151
|
-
*/
|
|
1152
|
-
interface RateLimiterAdapter {
|
|
1153
|
-
estimate(request: WireRequest): number;
|
|
1154
|
-
acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
|
|
1155
|
-
signalRateLimit(): void;
|
|
1156
|
-
reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
|
|
1157
|
-
/** Optional: current bucket levels, for introspection. Omit if the adapter has no state worth reporting. */
|
|
1158
|
-
getState?(): RateLimitState;
|
|
1159
|
-
/** Optional: live bucket levels for `VernLLM.readRateLimitState()`, the async counterpart of `getState`. */
|
|
1160
|
-
readState?(): Promise<RateLimitState>;
|
|
1161
|
-
/** Optional: receives the instance's `Logger` once, when `VernLLM` wires this adapter in. */
|
|
1162
|
-
setLogger?(logger: Logger): void;
|
|
1163
|
-
}
|
|
1164
|
-
/**
|
|
1165
|
-
* Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
|
|
1166
|
-
* concurrency) behind one FIFO queue, so a large call isn't starved by a
|
|
1167
|
-
* stream of small ones. Any bucket omitted from `options` has infinite
|
|
1168
|
-
* capacity and never blocks.
|
|
1169
|
-
*/
|
|
1170
|
-
export declare class RateLimiter implements RateLimiterAdapter {
|
|
1171
|
-
private readonly requests?;
|
|
1172
|
-
private readonly tokens?;
|
|
1173
|
-
private readonly concurrency?;
|
|
1174
|
-
/** Buckets in acquire precedence order (concurrency, rpm, tpm), omitted ones filtered out. Built once so order can't drift between `tryAcquireBuckets` and `scheduleWake`. */
|
|
1175
|
-
private readonly buckets;
|
|
1176
|
-
private readonly maxQueueMs;
|
|
1177
|
-
private readonly maxQueueSize;
|
|
1178
|
-
private readonly estimateTokensFn;
|
|
1179
|
-
private readonly estimateFraction;
|
|
1180
|
-
private readonly aimd?;
|
|
1181
|
-
private readonly queue;
|
|
1182
|
-
/**
|
|
1183
|
-
* A single scheduled re-check for the head of the queue when it's
|
|
1184
|
-
* blocked on a bucket that refills on its own clock (rpm/tpm), so a
|
|
1185
|
-
* queue that nobody calls `acquire`/`release` on again isn't stuck
|
|
1186
|
-
* forever waiting for an external trigger to re-drain it. Not needed
|
|
1187
|
-
* for a concurrency block, which only clears via `release`.
|
|
1188
|
-
*/
|
|
1189
|
-
private wakeTimer?;
|
|
1190
|
-
constructor(options: RateLimitOptions);
|
|
1191
|
-
/**
|
|
1192
|
-
* Pre-flight token estimate for a request, per the configured (or
|
|
1193
|
-
* default) heuristic, scaled by `estimateFraction`. This is the sole
|
|
1194
|
-
* value reserved against `tokensPerMinute` and later reconciled in
|
|
1195
|
-
* `release`; the provider-facing `max_tokens` on the request itself is
|
|
1196
|
-
* never touched.
|
|
1197
|
-
*/
|
|
1198
|
-
estimate(request: WireRequest): number;
|
|
1199
|
-
/**
|
|
1200
|
-
* Waits for capacity in every configured bucket, then takes from each.
|
|
1201
|
-
* The returned `release` gives the concurrency slot back and reconciles
|
|
1202
|
-
* the token bucket against real usage; it must run in a `finally` block.
|
|
1203
|
-
*/
|
|
1204
|
-
acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
|
|
1205
|
-
private queueFullError;
|
|
1206
|
-
private enqueue;
|
|
1207
|
-
/** Takes from every configured bucket as one atomic unit, in `this.buckets`' order. Rolls back whatever was already taken if any bucket lacks capacity. */
|
|
1208
|
-
private tryAcquireBuckets;
|
|
1209
|
-
/** Drains the queue head first. Stops at the first waiter that still can't proceed, so no one is starved out of turn. */
|
|
1210
|
-
private drain;
|
|
1211
|
-
/**
|
|
1212
|
-
* Schedules a one-shot re-check of the queue for whenever the bucket
|
|
1213
|
-
* that's currently blocking the head waiter should next have enough
|
|
1214
|
-
* capacity. A no-op for a concurrency block (only `release` can clear
|
|
1215
|
-
* that) or while a wake is already pending.
|
|
1216
|
-
*/
|
|
1217
|
-
private scheduleWake;
|
|
1218
|
-
/**
|
|
1219
|
-
* Builds the one-shot release closure for an acquired slot. Only the
|
|
1220
|
-
* concurrency bucket is given back on release; the requests-per-minute
|
|
1221
|
-
* bucket is a real spend that only recovers via its own refill, and the
|
|
1222
|
-
* tokens bucket is reconciled against `actualTokens` rather than fully
|
|
1223
|
-
* refunded, since real tokens really were spent.
|
|
1224
|
-
*
|
|
1225
|
-
* `success` defaults to `false`: the AIMD ceiling only grows when the
|
|
1226
|
-
* caller explicitly confirms a successful attempt. A failed or
|
|
1227
|
-
* rate-limited attempt still releases its slot (so nothing leaks), but
|
|
1228
|
-
* must not also grow the ceiling right back up after
|
|
1229
|
-
* `signalRateLimit()` just shrank it.
|
|
1230
|
-
*/
|
|
1231
|
-
private makeRelease;
|
|
1232
|
-
/** Shared guard and resize call behind both AIMD halves below; only the arithmetic differs. */
|
|
1233
|
-
private resizeRequestsCeiling;
|
|
1234
|
-
/** AIMD's additive-increase half: grows the ceiling by `aimd.increaseBy` on a clean release. No-op without `aimd`/`requestsPerMinute`. */
|
|
1235
|
-
private growOnSuccess;
|
|
1236
|
-
/**
|
|
1237
|
-
* AIMD's multiplicative-decrease half. Called on a real 429, and,
|
|
1238
|
-
* where an adapter can produce a hint, proactively via
|
|
1239
|
-
* `reactToRateLimitHint`. Never throws or blocks a call itself, only
|
|
1240
|
-
* adjusts the ceiling as a side effect.
|
|
1241
|
-
*/
|
|
1242
|
-
signalRateLimit(): void;
|
|
1243
|
-
/**
|
|
1244
|
-
* AIMD's proactive entry point: shrinks via `signalRateLimit()` if
|
|
1245
|
-
* `hint.remainingRequests` is at or below `aimd.proactiveFloor`.
|
|
1246
|
-
*/
|
|
1247
|
-
reactToRateLimitHint(hint: ProviderRateLimitHint | undefined): void;
|
|
1248
|
-
/**
|
|
1249
|
-
* Current bucket levels, read live rather than cached. `concurrency`
|
|
1250
|
-
* tracks free slots internally, so `concurrentInFlight` is reported as
|
|
1251
|
-
* `capacity - available`, the inverse of what the bucket itself holds.
|
|
1252
|
-
*/
|
|
1253
|
-
getState(): RateLimitState;
|
|
1254
|
-
}
|
|
1255
|
-
//#endregion
|
|
1256
|
-
//#region src/internal/utils/rate-limit/rateLimitAdapter.utils.d.ts
|
|
1257
|
-
/** Not exported. Internal shorthand only, so this union isn't duplicated between the public option fields and `buildRateLimit`'s own signature. */
|
|
1258
|
-
type RateLimitOption = RateLimitOptions | RateLimiterAdapter;
|
|
1259
|
-
//#endregion
|
|
1260
|
-
//#region src/types/fallback.d.ts
|
|
1261
|
-
/**
|
|
1262
|
-
* One provider to try after the primary (or after an earlier fallback
|
|
1263
|
-
* target) fails. Order is the policy: VernLLM never reorders, scores, or
|
|
1264
|
-
* selects a target, it only walks the list as given.
|
|
1265
|
-
*
|
|
1266
|
-
* Most per-target overrides fall back to the parent `VernLLM` instance's
|
|
1267
|
-
* own option when omitted, so a target only needs to specify what's
|
|
1268
|
-
* actually different about it (a different client/model is the common
|
|
1269
|
-
* case). `circuitBreaker`, `rateLimit`, and `retryBudget` are the
|
|
1270
|
-
* exception: they are never inherited from the parent, since a breaker,
|
|
1271
|
-
* limiter, or budget tuned for the primary provider's limits is rarely
|
|
1272
|
-
* right for a fallback's. Leave them unset on a target to run it without
|
|
1273
|
-
* one, even if the parent has one configured.
|
|
1274
|
-
*/
|
|
1275
|
-
interface FallbackTarget {
|
|
1276
|
-
client: LLMClient;
|
|
1277
|
-
model: string;
|
|
1278
|
-
/** Label for events, errors, and `TokenUsage.provider`. Default `` `fallback[${index}]` ``. */
|
|
1279
|
-
name?: string;
|
|
1280
|
-
maxRetries?: number;
|
|
1281
|
-
timeoutMs?: number;
|
|
1282
|
-
chunkIdleTimeoutMs?: number;
|
|
1283
|
-
baseDelayMs?: number;
|
|
1284
|
-
defaultMaxTokens?: number;
|
|
1285
|
-
defaultTemperature?: number | null;
|
|
1286
|
-
defaultReasoningEffort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
1287
|
-
defaultBudgetTokens?: number;
|
|
1288
|
-
nonRetryableStatus?: number[];
|
|
1289
|
-
/** This target's own circuit breaker, independent of every other target's. Not inherited from the parent's `circuitBreaker`. */
|
|
1290
|
-
circuitBreaker?: boolean | CircuitBreakerOptions;
|
|
1291
|
-
/** This target's own rate limiter, independent of every other target's. Not inherited from the parent's `rateLimit`. */
|
|
1292
|
-
rateLimit?: RateLimitOption;
|
|
1293
|
-
/** This target's own retry budget, independent of every other target's. Not inherited from the parent's `retryBudget`. */
|
|
1294
|
-
retryBudget?: RetryBudgetOptions;
|
|
1295
|
-
/**
|
|
1296
|
-
* Reclassifies an otherwise-successful result from this target as a
|
|
1297
|
-
* failure. Falls back to the parent `VernLLM` instance's own
|
|
1298
|
-
* `detectSoftFailure` when omitted, same as most other per-target
|
|
1299
|
-
* options (unlike `circuitBreaker`/`rateLimit`, which never inherit).
|
|
1300
|
-
*/
|
|
1301
|
-
detectSoftFailure?: DetectSoftFailure;
|
|
1302
|
-
}
|
|
1303
|
-
/**
|
|
1304
|
-
* Written into `CallParams['meta']` once `call()` resolves, so a caller
|
|
1305
|
-
* who wants provider identity on the same line as the result doesn't need
|
|
1306
|
-
* to read it back out of `onUsage`.
|
|
1307
|
-
*/
|
|
1308
|
-
interface CallMeta {
|
|
1309
|
-
provider: string;
|
|
1310
|
-
model: string;
|
|
1311
|
-
/** `-1` if the primary target answered, otherwise the index into `fallback`. */
|
|
1312
|
-
fallbackIndex: number;
|
|
1313
|
-
usedFallback: boolean;
|
|
1314
|
-
/** Attempts made against the target that ultimately answered, including the successful one. */
|
|
1315
|
-
attempts: number;
|
|
1316
|
-
}
|
|
1317
|
-
/** One target's circuit state, as returned by `VernLLM.getCircuitStates()`. */
|
|
1318
|
-
interface TargetCircuitState {
|
|
1319
|
-
provider: string;
|
|
1320
|
-
/** Position in the chain: `0` for the primary, `1`+ for fallback targets. */
|
|
1321
|
-
index: number;
|
|
1322
|
-
isFallback: boolean;
|
|
1323
|
-
/** Whether this target tracks failures per model. `false` means `model` on `getCircuitStates` had no effect on this entry. */
|
|
1324
|
-
isolateByModel: boolean;
|
|
1325
|
-
/** `undefined` if that target has no circuit breaker configured. */
|
|
1326
|
-
state: CircuitState | undefined;
|
|
1327
|
-
}
|
|
1328
|
-
/** Which target/model `VernLLM.getCircuitState`, `openCircuit`, and `closeCircuit` act on. */
|
|
1329
|
-
interface CircuitTarget {
|
|
1330
|
-
/** Which target to act on. `0` is the primary, `1`+ are fallbacks. Defaults to `0`. */
|
|
1331
|
-
index?: number;
|
|
1332
|
-
/** Which model bucket to act on, if the resolved target isolates by model. */
|
|
1333
|
-
model?: string;
|
|
1334
|
-
}
|
|
1335
|
-
/**
|
|
1336
|
-
* One target's failure, recorded on the way to either the next target or
|
|
1337
|
-
* `FallbackExhaustedError`. Extends `RetryAttempt`: `index` is `-1` for
|
|
1338
|
-
* the primary target here (rather than a plain retry count), and
|
|
1339
|
-
* `provider`/`model` identify which target failed.
|
|
1340
|
-
*/
|
|
1341
|
-
interface FallbackAttempt extends RetryAttempt {
|
|
1342
|
-
provider: string;
|
|
1343
|
-
model: string;
|
|
1344
|
-
}
|
|
1345
|
-
/**
|
|
1346
|
-
* Decides what happens after a target's own retries are exhausted or
|
|
1347
|
-
* abandoned early. Called once per failed target. `'retry'` is not a
|
|
1348
|
-
* valid return here: retrying already happened inside the target, this
|
|
1349
|
-
* only decides whether to move on to the next one or stop.
|
|
1350
|
-
*/
|
|
1351
|
-
type FallbackOn = (error: LLMError, context: {
|
|
1352
|
-
isLastTarget: boolean;
|
|
1353
|
-
}) => 'next' | 'stop';
|
|
1354
|
-
/**
|
|
1355
|
-
* The default `fallbackOn` policy. Exported so a caller can wrap rather
|
|
1356
|
-
* than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
|
|
1357
|
-
*/
|
|
1358
|
-
export declare const defaultFallbackOn: FallbackOn;
|
|
1359
|
-
/**
|
|
1360
|
-
* Thrown when the chain gives up, whether because the last target failed
|
|
1361
|
-
* or `fallbackOn` chose to stop early. Carries each attempt in order so
|
|
1362
|
-
* an outage across providers stays debuggable without reproducing it.
|
|
1363
|
-
* Extends `LLMError` so `isLLMError` and any `instanceof LLMError` check
|
|
1364
|
-
* still passes, inheriting the last failure's `type`/`status`/`retryAfterMs`
|
|
1365
|
-
* so existing type-based handling, including reading `retryAfterMs` on an
|
|
1366
|
-
* `'api'`-typed error, keeps working on a fallback-exhausted error too.
|
|
1367
|
-
*/
|
|
1368
|
-
export declare class FallbackExhaustedError extends LLMError {
|
|
1369
|
-
readonly attempts: FallbackAttempt[];
|
|
1370
|
-
constructor(attempts: FallbackAttempt[]);
|
|
1371
|
-
/**
|
|
1372
|
-
* `type: 'fallback_exhausted'` by itself says nothing about whether
|
|
1373
|
-
* retrying could help; the reason the last target failed does. Defers to
|
|
1374
|
-
* that attempt's own `retryable` instead of anything about this class's
|
|
1375
|
-
* own type.
|
|
1376
|
-
*/
|
|
1377
|
-
get retryable(): boolean;
|
|
1378
|
-
}
|
|
1379
|
-
/** Narrows `err` to {@link FallbackExhaustedError}, for direct access to its `attempts` (`provider`/`model` per failed target) without a manual `instanceof` check. */
|
|
1380
|
-
export declare function isFallbackExhaustedError(err: unknown): err is FallbackExhaustedError;
|
|
1381
|
-
/**
|
|
1382
|
-
* Creates an empty ref box to pass as `CallParams['meta']`, so a caller can
|
|
1383
|
-
* read the `CallMeta` written by `call()` on the same line as the result
|
|
1384
|
-
* instead of pre-declaring a `{ current?: CallMeta }` by hand.
|
|
1385
|
-
*
|
|
1386
|
-
* @example
|
|
1387
|
-
* const meta = metaRef();
|
|
1388
|
-
* const result = await vern.call({ userContent: '...', meta });
|
|
1389
|
-
* meta.current?.provider;
|
|
1390
|
-
*/
|
|
1391
|
-
export declare function metaRef(): {
|
|
1392
|
-
current?: CallMeta;
|
|
1393
|
-
};
|
|
1394
|
-
//#endregion
|
|
1395
|
-
//#region src/types/schema.d.ts
|
|
1396
|
-
/**
|
|
1397
|
-
* Minimal structural type for a Zod-like schema, so this package doesnt need
|
|
1398
|
-
* a hard dependency on a specific Zod major version. Any object exposing
|
|
1399
|
-
* `safeParse` (Zod v3/v4, and most Zod-compatible validators) should satisfy this
|
|
1400
|
-
*/
|
|
1401
|
-
interface SchemaLike<T> {
|
|
1402
|
-
safeParse(data: unknown): {
|
|
1403
|
-
success: true;
|
|
1404
|
-
data: T;
|
|
1405
|
-
} | {
|
|
1406
|
-
success: false;
|
|
1407
|
-
error: unknown;
|
|
1408
|
-
};
|
|
1409
|
-
}
|
|
1410
|
-
/**
|
|
1411
|
-
* A provider-native JSON Schema for structured outputs (OpenAI/Groq
|
|
1412
|
-
* `response_format: { type: 'json_schema' }`) This is the wire-format
|
|
1413
|
-
* schema the model is constrained to generate against, distinct from
|
|
1414
|
-
* `schema`, which is a client-side Zod validator run on the parsed result
|
|
1415
|
-
* You can use one, both, or neither; using both gets you provider-level
|
|
1416
|
-
* constraint plus client-side type inference/validation as a safety net
|
|
1417
|
-
*/
|
|
1418
|
-
interface JsonSchemaSpec {
|
|
1419
|
-
name: string;
|
|
1420
|
-
schema: Record<string, unknown>;
|
|
1421
|
-
/** Enforces the schema strictly (OpenAI-specific), default true when supported */
|
|
1422
|
-
strict?: boolean;
|
|
1423
|
-
description?: string;
|
|
1424
|
-
}
|
|
1425
|
-
//#endregion
|
|
1426
|
-
//#region src/types/tools.d.ts
|
|
1427
|
-
/**
|
|
1428
|
-
* Describes a capability the model may request, not the capability
|
|
1429
|
-
* itself. VernLLM transports this to the provider and parses what comes
|
|
1430
|
-
* back; it never executes anything.
|
|
1431
|
-
*/
|
|
1432
|
-
interface ToolDefinition<Name extends string = string, Args = unknown> {
|
|
1433
|
-
name: Name;
|
|
1434
|
-
description: string;
|
|
1435
|
-
/** JSON Schema for the tool's input. */
|
|
1436
|
-
parameters: Record<string, unknown>;
|
|
1437
|
-
/**
|
|
1438
|
-
* Optional client-side validator run on the parsed `arguments` before
|
|
1439
|
-
* they're handed back to the caller, mirroring the `schema: SchemaLike<T>`
|
|
1440
|
-
* pattern already used for response validation (see `types/schema.ts`).
|
|
1441
|
-
* Reuses that zero-dependency, `safeParse`-compatible shape instead of
|
|
1442
|
-
* requiring a JSON Schema validator (e.g. ajv) as a new dependency.
|
|
1443
|
-
* Failed validation throws `LLMError('validation')`. If omitted, VernLLM
|
|
1444
|
-
* parses arguments as JSON but does not validate them further.
|
|
1445
|
-
*
|
|
1446
|
-
* When set, `Args` (and therefore `Name`) flow into the `ToolCall`s
|
|
1447
|
-
* returned by `call()`/`cachedCall()`, provided the tool was declared
|
|
1448
|
-
* with `defineTool()` or otherwise has a literal `name`; see
|
|
1449
|
-
* `defineTool()` below for why a plain object literal often doesn't.
|
|
1450
|
-
*/
|
|
1451
|
-
argumentsSchema?: SchemaLike<Args>;
|
|
1452
|
-
}
|
|
1453
|
-
/**
|
|
1454
|
-
* Preserves a tool definition's literal `name` (and its `argumentsSchema`'s
|
|
1455
|
-
* inferred `Args`) so it can discriminate a `ToolCall` union later.
|
|
1456
|
-
*
|
|
1457
|
-
* A plain object literal like `{ name: 'get_weather', ... }` widens `name`
|
|
1458
|
-
* to `string` unless annotated `as const`, which silently defeats
|
|
1459
|
-
* `ToolCall` narrowing the moment a second tool is added to the same
|
|
1460
|
-
* `tools: [...]` array (single-tool arrays still narrow fine even without
|
|
1461
|
-
* this, since there's nothing to discriminate against but that stops
|
|
1462
|
-
* being true as soon as a second tool shows up). Wrapping the same object
|
|
1463
|
-
* in `defineTool()` preserves the literal `name` type without requiring
|
|
1464
|
-
* `as const` at every call site.
|
|
1465
|
-
*/
|
|
1466
|
-
export declare function defineTool<const Name extends string, Args = unknown>(tool: ToolDefinition<Name, Args>): ToolDefinition<Name, Args>;
|
|
1467
|
-
/** Maps a single `ToolDefinition` to its matching `ToolCall` shape. */
|
|
1468
|
-
type ToolCallFor<T> = T extends ToolDefinition<infer N, infer A> ? {
|
|
1469
|
-
id: string;
|
|
1470
|
-
name: N;
|
|
1471
|
-
arguments: A;
|
|
1472
|
-
} : never;
|
|
1473
|
-
/**
|
|
1474
|
-
* A single tool invocation requested by the model.
|
|
1475
|
-
*
|
|
1476
|
-
* When `Tools` is a literal tuple (e.g. inferred from `tools: [getWeather,
|
|
1477
|
-
* cancelOrder]` at a `call()`/`cachedCall()` site), this is a discriminated
|
|
1478
|
-
* union keyed by `name`. Checking `call.name === 'get_weather'` narrows
|
|
1479
|
-
* `call.arguments` to that tool's `Args` with no cast needed. Without a
|
|
1480
|
-
* literal `Tools` (the default), this collapses back to today's
|
|
1481
|
-
* `{ id: string; name: string; arguments: unknown }`.
|
|
1482
|
-
*/
|
|
1483
|
-
type ToolCall<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ToolCallFor<Tools[number]>;
|
|
1484
|
-
/** The application's result of executing a `ToolCall`, sent back to the model. */
|
|
1485
|
-
interface ToolResult {
|
|
1486
|
-
toolCallId: string;
|
|
1487
|
-
content: unknown;
|
|
1488
|
-
/**
|
|
1489
|
-
* Signals a failed tool execution back to the model (matches Anthropic's
|
|
1490
|
-
* native `is_error` on tool_result blocks). Only `fromAnthropic` honors
|
|
1491
|
-
* this today, Gemini and Bedrock have no equivalent wire concept, so
|
|
1492
|
-
* other adapters ignore it silently.
|
|
1493
|
-
*/
|
|
1494
|
-
isError?: boolean;
|
|
1495
|
-
}
|
|
1496
|
-
/** `call()` result when `tools` was set and the model produced a normal answer. */
|
|
1497
|
-
interface ContentResult<T> {
|
|
1498
|
-
type: 'content';
|
|
1499
|
-
content: T;
|
|
1500
|
-
}
|
|
1501
|
-
/** `call()` result when `tools` was set and the model requested one or more tools. */
|
|
1502
|
-
interface ToolCallResult<Tools extends readonly ToolDefinition[] = ToolDefinition[]> {
|
|
1503
|
-
type: 'tool_calls';
|
|
1504
|
-
toolCalls: ToolCall<Tools>[];
|
|
1505
|
-
/** Any text the model produced alongside the tool request, if present. */
|
|
1506
|
-
content?: string;
|
|
1507
|
-
}
|
|
1508
|
-
type CallWithToolsResult<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ContentResult<T> | ToolCallResult<Tools>;
|
|
1509
|
-
/** Recovers `Tools` from a `result` already typed `ContentResult<T> | ToolCallResult<Tools>`. Falls back to `ToolDefinition[]`. */
|
|
1510
|
-
type ExtractTools<R> = Extract<R, ToolCallResult<ToolDefinition[]>> extends ToolCallResult<infer Tools> ? Tools : ToolDefinition[];
|
|
1511
|
-
/** Explicit `Tools` type argument if given, otherwise inferred via `ExtractTools`. `never` marks "unset". */
|
|
1512
|
-
type ResolvedTools<Tools, R> = [Tools] extends [never] ? ExtractTools<R> : Tools;
|
|
1513
|
-
/**
|
|
1514
|
-
* Runtime check for whether a `call()` result is a `tool_calls` result. Use
|
|
1515
|
-
* this instead of trusting static narrowing whenever `tools` was set
|
|
1516
|
-
* conditionally, see `ConditionalToolCallParams`.
|
|
1517
|
-
*
|
|
1518
|
-
* ```ts
|
|
1519
|
-
* const result = await llm.call({ userContent: '...', tools: someCondition ? [myTool] : undefined });
|
|
1520
|
-
* if (isToolCallResult(result)) {
|
|
1521
|
-
* // result.toolCalls[number].arguments typed per tool, inferred automatically
|
|
1522
|
-
* }
|
|
1523
|
-
* ```
|
|
1524
|
-
*
|
|
1525
|
-
* Pass `Tools` explicitly to override inference, e.g. `isToolCallResult<typeof tools>(result)`.
|
|
1526
|
-
*/
|
|
1527
|
-
export declare function isToolCallResult<Tools extends readonly ToolDefinition[] | undefined = never, R = unknown>(result: R): result is R & ToolCallResult<NonNullable<ResolvedTools<Tools, R>>>;
|
|
1528
|
-
/** What the model should do about tools on a given call. */
|
|
1529
|
-
type ToolChoice = 'auto' | 'none' | 'required' | {
|
|
1530
|
-
name: string;
|
|
1531
|
-
};
|
|
1532
|
-
//#endregion
|
|
1533
|
-
//#region src/types/call.d.ts
|
|
1534
|
-
/**
|
|
1535
|
-
* Any valid JSON value: a primitive, `null`, or a JSON array/object made
|
|
1536
|
-
* of the same. This is what `call()` returns when `jsonMode: true`.
|
|
1537
|
-
*/
|
|
1538
|
-
type JsonValue = string | number | boolean | null | JsonValue[] | {
|
|
1539
|
-
[key: string]: JsonValue;
|
|
1540
|
-
};
|
|
1541
|
-
/**
|
|
1542
|
-
* Content for an `assistant` turn in `history`. Accepts a string or a
|
|
1543
|
-
* parsed `JsonValue`, so a prior `jsonMode: true` response can be pushed
|
|
1544
|
-
* straight back into history. Request construction stringifies non-string
|
|
1545
|
-
* content before it's sent to the provider.
|
|
1546
|
-
*/
|
|
1547
|
-
type AssistantContent = string | JsonValue;
|
|
1548
|
-
/**
|
|
1549
|
-
* A single prior turn in a multi-turn conversation, passed via `history`.
|
|
1550
|
-
*
|
|
1551
|
-
* Supports normal user/assistant messages and tool continuations: an assistant
|
|
1552
|
-
* turn may include `toolCalls`, and a tool turn carries the matching
|
|
1553
|
-
* `toolResults`. A tool turn must immediately follow an assistant tool call
|
|
1554
|
-
* turn, and every requested tool call must have a result.
|
|
1555
|
-
*/
|
|
1556
|
-
type ConversationTurn = {
|
|
1557
|
-
role: 'user';
|
|
1558
|
-
content: string;
|
|
1559
|
-
} | {
|
|
1560
|
-
role: 'assistant';
|
|
1561
|
-
content?: AssistantContent;
|
|
1562
|
-
toolCalls?: ToolCall[];
|
|
1563
|
-
} | {
|
|
1564
|
-
role: 'tool';
|
|
1565
|
-
toolResults: ToolResult[];
|
|
1566
|
-
};
|
|
1567
|
-
/** A plain text segment of a multimodal `userContent` array. */
|
|
1568
|
-
interface TextBlock {
|
|
1569
|
-
type: 'text';
|
|
1570
|
-
text: string;
|
|
1571
|
-
}
|
|
1572
|
-
/**
|
|
1573
|
-
* An inline image segment of a multimodal `userContent` array.
|
|
1574
|
-
*
|
|
1575
|
-
* `data` is the raw base64-encoded image bytes, with no `data:` URL prefix
|
|
1576
|
-
* (adapters that need a data URL, e.g. OpenAI-compatible `image_url`, build
|
|
1577
|
-
* it themselves from `mimeType` + `data`; adapters that need raw bytes, e.g.
|
|
1578
|
-
* Bedrock, decode the base64 themselves).
|
|
1579
|
-
*/
|
|
1580
|
-
interface ImageBlock {
|
|
1581
|
-
type: 'image';
|
|
1582
|
-
/** Base64-encoded image bytes, no `data:` prefix */
|
|
1583
|
-
data: string;
|
|
1584
|
-
/** e.g. 'image/png', 'image/jpeg', 'image/webp', 'image/gif' */
|
|
1585
|
-
mimeType: string;
|
|
1586
|
-
}
|
|
1587
|
-
/** A single segment of multimodal `userContent`. */
|
|
1588
|
-
type ContentBlock = TextBlock | ImageBlock;
|
|
1589
|
-
/**
|
|
1590
|
-
* Every field of a call request except the `reserveUsage`/`refundUsage`
|
|
1591
|
-
* hooks from `UsageHooks`. `CallParams` is this plus `UsageHooks`; the
|
|
1592
|
-
* `Cached*` param types below are call sites that want the request shape
|
|
1593
|
-
* without those two hooks (usage is metered once, at the `cachedCall`
|
|
1594
|
-
* level, not per-request), and use this directly instead of re-deriving
|
|
1595
|
-
* it with `Omit<CallParams<T>, 'reserveUsage' | 'refundUsage'>` each time.
|
|
1596
|
-
*/
|
|
1597
|
-
interface LLMRequestShape<T = unknown, Tools extends readonly ToolDefinition[] = ToolDefinition[]> {
|
|
1598
|
-
systemPrompt?: string;
|
|
1599
|
-
/** Current user message, as text or multimodal content blocks. */
|
|
1600
|
-
userContent: string | ContentBlock[];
|
|
1601
|
-
/**
|
|
1602
|
-
* Previous conversation turns. Must alternate roles; tool turns must follow
|
|
1603
|
-
* assistant tool calls. Invalid history throws LLMError('invalid_params').
|
|
1604
|
-
*/
|
|
1605
|
-
history?: ConversationTurn[];
|
|
1606
|
-
/**
|
|
1607
|
-
* Generation temperature. Default 0.2, not the provider's own default.
|
|
1608
|
-
* Pass `null` to omit `temperature` from the request entirely, so the
|
|
1609
|
-
* provider applies its own default instead.
|
|
1610
|
-
*/
|
|
1611
|
-
temperature?: number | null;
|
|
1612
|
-
jsonMode?: boolean;
|
|
1613
|
-
maxTokens?: number;
|
|
1614
|
-
requestId?: string;
|
|
1615
|
-
signal?: AbortSignal;
|
|
1616
|
-
/**
|
|
1617
|
-
* Total time budget in ms for this whole call, across every retry and
|
|
1618
|
-
* every fallback target. Unlike timeoutMs, which resets on each attempt,
|
|
1619
|
-
* this is a single clock starting when call is invoked. The call is
|
|
1620
|
-
* aborted once this elapses, even mid retry or mid fallback, the same
|
|
1621
|
-
* way an aborted signal is today. Omit for no overall deadline, only
|
|
1622
|
-
* the existing per attempt timeoutMs applies.
|
|
1623
|
-
*
|
|
1624
|
-
* Only bounds getting to a final result: choosing a target, retrying,
|
|
1625
|
-
* and opening a stream. It does not extend to the time spent reading a
|
|
1626
|
-
* stream after it has opened. Use chunkIdleTimeoutMs for gaps between
|
|
1627
|
-
* chunks once a stream is open.
|
|
1628
|
-
*/
|
|
1629
|
-
deadlineMs?: number;
|
|
1630
|
-
/**
|
|
1631
|
-
* Per-call override for the instance's `chunkIdleTimeoutMs` (max gap
|
|
1632
|
-
* between stream chunks once opened). Only applies when `stream: true`.
|
|
1633
|
-
* Useful for routes using reasoning-heavy models with documented long
|
|
1634
|
-
* silent gaps mid-stream. Pass 0 to disable the idle timeout for this
|
|
1635
|
-
* call.
|
|
1636
|
-
*/
|
|
1637
|
-
chunkIdleTimeoutMs?: number;
|
|
1638
|
-
/** Overrides the instance model for this call. */
|
|
1639
|
-
model?: string;
|
|
1640
|
-
/**
|
|
1641
|
-
* Reasoning effort for supported reasoning models. Pass `null` to
|
|
1642
|
-
* explicitly skip an instance-level `defaultReasoningEffort` for this
|
|
1643
|
-
* one call (e.g. a call using a forced `toolChoice`, which Anthropic
|
|
1644
|
-
* rejects alongside any reasoning at all), the same way `temperature:
|
|
1645
|
-
* null` opts a call out of `defaultTemperature`. Omitting the field
|
|
1646
|
-
* entirely (`undefined`) defers to the instance default instead.
|
|
1647
|
-
*/
|
|
1648
|
-
reasoningEffort?: 'minimal' | 'low' | 'medium' | 'high' | null;
|
|
1649
|
-
/**
|
|
1650
|
-
* Token budget for internal reasoning, for models with a native numeric
|
|
1651
|
-
* budget (Anthropic's `budget_tokens`, Gemini's `thinkingBudget`). On a
|
|
1652
|
-
* provider that only understands `reasoningEffort` tiers (OpenAI-
|
|
1653
|
-
* compatible), this is converted to the nearest tier instead of sent as
|
|
1654
|
-
* a raw number. When both `budgetTokens` and `reasoningEffort` are set,
|
|
1655
|
-
* each adapter prefers whichever field it natively understands and
|
|
1656
|
-
* ignores the other. See the reasoning budget docs for the conversion
|
|
1657
|
-
* table used in each direction. Pass `null` to explicitly skip an
|
|
1658
|
-
* instance-level `defaultBudgetTokens` for this one call, mirroring
|
|
1659
|
-
* `reasoningEffort: null` above; omitting the field entirely defers to
|
|
1660
|
-
* the instance default.
|
|
1661
|
-
*/
|
|
1662
|
-
budgetTokens?: number | null;
|
|
1663
|
-
/**
|
|
1664
|
-
* Provider-native JSON Schema output constraint. Implies jsonMode: true.
|
|
1665
|
-
*/
|
|
1666
|
-
jsonSchema?: JsonSchemaSpec;
|
|
1667
|
-
/**
|
|
1668
|
-
* Validates parsed JSON output. Failure throws LLMError('validation').
|
|
1669
|
-
* Implies jsonMode: true.
|
|
1670
|
-
*/
|
|
1671
|
-
schema?: SchemaLike<T>;
|
|
1672
|
-
/**
|
|
1673
|
-
* Tools the model may call. When set, `call()` returns a
|
|
1674
|
-
* `CallWithToolsResult<T>` union instead of `T` directly. Combining with
|
|
1675
|
-
* `jsonSchema` is provider-dependent; see the Tool Calling docs.
|
|
1676
|
-
*
|
|
1677
|
-
* Passed as a literal array (or via `defineTool()`-wrapped entries, see
|
|
1678
|
-
* `types/tools.ts`), this also drives the `Tools` type parameter, which
|
|
1679
|
-
* narrows `CallWithToolsResult`'s `toolCalls[number].arguments` per tool.
|
|
1680
|
-
*/
|
|
1681
|
-
tools?: Tools;
|
|
1682
|
-
/** Defaults to `'auto'` when `tools` is set. */
|
|
1683
|
-
toolChoice?: ToolChoice;
|
|
1684
|
-
/**
|
|
1685
|
-
* Streams the response incrementally instead of resolving once. Default:
|
|
1686
|
-
* false. Requires a client/adapter that implements `createStream`.
|
|
1687
|
-
* Retry/timeout/circuit-breaker guarantees apply only to opening the
|
|
1688
|
-
* stream (through the first chunk); a failure after that point rejects
|
|
1689
|
-
* `finalResult` directly and is not retried, since a mid-stream failure
|
|
1690
|
-
* isn't connection-time evidence for the circuit breaker, the attempt
|
|
1691
|
-
* already counted as a success once the first chunk arrived. Once the
|
|
1692
|
-
* stream opens successfully, `finalResult` still resolves to the same
|
|
1693
|
-
* validated `T`/`CallWithToolsResult<T>` shape `call()` would have
|
|
1694
|
-
* returned for the same params with `stream` omitted. See
|
|
1695
|
-
* `StreamCallResult`.
|
|
1696
|
-
*/
|
|
1697
|
-
stream?: boolean;
|
|
1698
|
-
/**
|
|
1699
|
-
* Optional out-parameter for provider identity. Pass `{}` (or any object
|
|
1700
|
-
* with a mutable `current` property) and `call()` writes a `CallMeta`
|
|
1701
|
-
* into `meta.current` before returning, alongside whatever `onUsage`
|
|
1702
|
-
* already reports. This includes `stream: true`: the target is chosen
|
|
1703
|
-
* once the stream opens, which is also the point `call()` itself
|
|
1704
|
-
* returns `{ chunks, finalResult }`, so `meta.current` is already set
|
|
1705
|
-
* by then. `TokenUsage.provider`/`usedFallback` from `onUsage` reports
|
|
1706
|
-
* the same information asynchronously, for both streaming and
|
|
1707
|
-
* non-streaming calls.
|
|
1708
|
-
*
|
|
1709
|
-
* `meta.current` is only written once execution actually reaches and
|
|
1710
|
-
* selects a provider target. A `wrap` middleware that short-circuits
|
|
1711
|
-
* without calling `next()` never reaches that point, so `meta.current`
|
|
1712
|
-
* is left untouched; if the same holder object is reused across calls,
|
|
1713
|
-
* it can still hold a prior call's target.
|
|
1714
|
-
*/
|
|
1715
|
-
meta?: {
|
|
1716
|
-
current?: CallMeta;
|
|
1717
|
-
};
|
|
1718
|
-
}
|
|
1719
|
-
interface CallParams<T = unknown, Tools extends readonly ToolDefinition[] = ToolDefinition[]> extends LLMRequestShape<T, Tools>, UsageHooks {}
|
|
1720
|
-
/**
|
|
1721
|
-
* A `CallParams` variant where tool calling is explicitly enabled.
|
|
1722
|
-
*
|
|
1723
|
-
* Requiring `tools` to be present allows TypeScript to select the
|
|
1724
|
-
* tool-aware `call()` overload and return `CallWithToolsResult<T>` instead
|
|
1725
|
-
* of the normal `T` response type.
|
|
1726
|
-
*/
|
|
1727
|
-
type ToolEnabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
|
|
1728
|
-
tools: NonNullable<CallParams<T, Tools>['tools']>;
|
|
1729
|
-
};
|
|
1730
|
-
/**
|
|
1731
|
-
* A `CallParams` variant for tools set conditionally, e.g. `tools:
|
|
1732
|
-
* someCondition ? [myTool] : undefined`. Selects the `call()` overload
|
|
1733
|
-
* returning the honest union `T | CallWithToolsResult<T, Tools>` instead of
|
|
1734
|
-
* falling through to plain `T` (which is what happened before this type
|
|
1735
|
-
* existed, since `ToolDefinition[] | undefined` matched neither
|
|
1736
|
-
* `ToolEnabledCallParams` nor `ToolsDisabledCallParams`). Forces an
|
|
1737
|
-
* `isToolCallResult()` check before treating the result as plain
|
|
1738
|
-
* content. Omitting `tools` entirely still resolves to plain `T`, since
|
|
1739
|
-
* tools genuinely cannot have run there.
|
|
1740
|
-
*
|
|
1741
|
-
* `Tools` still can't reliably infer a literal tuple here the way
|
|
1742
|
-
* `ToolEnabledCallParams` does for an inline array (a ternary/variable
|
|
1743
|
-
* expression doesn't carry the same `const`-literal preservation), so
|
|
1744
|
-
* getting typed `arguments` out of a conditional-tools result also needs
|
|
1745
|
-
* an explicit `Tools` type argument on `isToolCallResult<Tools>()` when
|
|
1746
|
-
* narrowing, see its docs.
|
|
1747
|
-
*/
|
|
1748
|
-
type ConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
|
|
1749
|
-
tools: Tools | undefined;
|
|
1750
|
-
};
|
|
1751
|
-
/** Conditional tool-call parameters whose non-tool result is plain text. */
|
|
1752
|
-
type ConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = ConditionalToolCallParams<string, Tools> & {
|
|
1753
|
-
jsonMode: false;
|
|
1754
|
-
};
|
|
1755
|
-
/**
|
|
1756
|
-
* A `CallParams` variant where tools are offered but the model is barred
|
|
1757
|
-
* from calling one. `toolChoice: 'none'` guarantees the response can never
|
|
1758
|
-
* be a `tool_calls` result, so `call()` can narrow straight to
|
|
1759
|
-
* `ContentResult<T>` instead of the full `CallWithToolsResult<T>` union.
|
|
1760
|
-
* A call site that already knows it forced `'none'` no longer needs a
|
|
1761
|
-
* runtime `isToolCallResult` check, or to remember that `String(result)`
|
|
1762
|
-
* on the wrapper object silently produces `"[object Object]"` instead of
|
|
1763
|
-
* throwing. The type itself rules that shape out.
|
|
1764
|
-
*/
|
|
1765
|
-
type ToolsDisabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
|
|
1766
|
-
tools: NonNullable<CallParams<T, Tools>['tools']>;
|
|
1767
|
-
toolChoice: 'none';
|
|
1768
|
-
};
|
|
1769
|
-
/**
|
|
1770
|
-
* `CallParams` with `jsonMode: false`. Selects the `call()` overload
|
|
1771
|
-
* that returns a plain `string`. `jsonSchema` is typed `never` here: a
|
|
1772
|
-
* truthy `jsonSchema` forces JSON parsing at runtime regardless of
|
|
1773
|
-
* `jsonMode` (see `RequestBuilder.build()`), so `jsonMode: false` +
|
|
1774
|
-
* `jsonSchema` together would otherwise still match this overload and
|
|
1775
|
-
* falsely promise a `string`.
|
|
1776
|
-
*/
|
|
1777
|
-
type JsonModeDisabledCallParams = Omit<CallParams<unknown>, 'jsonSchema'> & {
|
|
1778
|
-
jsonMode: false;
|
|
1779
|
-
jsonSchema?: never;
|
|
1780
|
-
};
|
|
1781
|
-
/**
|
|
1782
|
-
* `CallParams` with `jsonMode: true` and no `schema`. Selects the
|
|
1783
|
-
* `call()` overload that returns a `JsonValue`.
|
|
1784
|
-
*
|
|
1785
|
-
* `schema` is explicitly typed `never` here, not just omitted: `CallParams<JsonValue>['schema']`
|
|
1786
|
-
* would be `SchemaLike<JsonValue> | undefined`, and a schema whose inferred result type is
|
|
1787
|
-
* itself structurally assignable to `JsonValue` (e.g. a schema for `string[]` or
|
|
1788
|
-
* `Record<string, string>`) would still satisfy that shape, incorrectly selecting this
|
|
1789
|
-
* overload over the schema-aware generic one and widening the result to `JsonValue`. Forcing
|
|
1790
|
-
* `schema?: never` makes any call that sets `schema` fail this overload's structural check
|
|
1791
|
-
* regardless of the schema's result type, so it always falls through to the generic
|
|
1792
|
-
* `CallParams<T>` overload and infers `T` from the schema instead.
|
|
1793
|
-
*/
|
|
1794
|
-
type JsonModeEnabledCallParams = Omit<CallParams<JsonValue>, 'schema'> & {
|
|
1795
|
-
jsonMode: true;
|
|
1796
|
-
schema?: never;
|
|
1797
|
-
};
|
|
1798
|
-
/** Shared cache-configuration fields, minus the internal `fn` primitive. */
|
|
1799
|
-
interface CachedCallInput extends UsageHooks {
|
|
1800
|
-
cacheKey: string;
|
|
1801
|
-
ttl: number;
|
|
1802
|
-
signal?: AbortSignal;
|
|
1
|
+
import { $ as SchemaLike, $t as createStateKey, A as ConditionalToolCallParams, At as ConsecutiveTripping, B as ThinkingBlock, Bt as MiddlewareRef, C as CachedConditionalToolCallParams, Cn as hasIssues, Ct as RetryBudgetOptions, D as CallContext, Dt as CircuitBreakerOptions, E as CachedToolCallParams, Et as CircuitBreakerCallContext, F as JsonModeDisabledCallParams, Ft as AttemptContext, G as ToolCall, Gt as RequiredMiddlewareRef, H as ToolsDisabledCallParams, Ht as MiddlewareStateEntry, I as JsonModeEnabledCallParams, It as CallResult, J as ToolDefinition, Jt as WireCallRequestPatch, K as ToolCallResult, Kt as VernLLMMiddleware, L as JsonValue, Lt as MiddlewareCapabilities, M as ConversationTurn, Mt as ExponentialBackoffOptions, N as DetectSoftFailure, Nt as RollingTripping, O as CallParams, Ot as CircuitBreakerStateChangeHandler, P as ImageBlock, Pt as TrippingPolicy, Q as JsonSchemaSpec, Qt as createMiddlewareStateBag, R as LLMRequestShape, Rt as MiddlewareContext, S as CachedConditionalStringToolCallParams, Sn as UnsupportedCapabilityIssue, St as RetryBudget, T as CachedJsonModeEnabledCallParams, Tt as CircuitBreakerAdapter, U as CallWithToolsResult, Ut as MiddlewareStateKey, V as ToolEnabledCallParams, Vt as MiddlewareStateBag, W as ContentResult, Wt as PreDispatchContext, X as defineTool, Xt as WireTool, Y as ToolResult, Yt as WireResponseFormat, Z as isToolCallResult, Zt as createMiddlewareRef, _ as StreamJsonModeEnabledCallParams, _n as LLMErrorType, _t as RateLimiter, a as WireToolChoice, an as OnUsageFailure, at as FallbackOnContext, b as AssistantContent, bn as ToolIssue, c as CachedStreamConditionalToolCallParams, cn as TokenUsage, ct as TargetInfo, d as CachedStreamToolCallParams, dn as DuplicateToolNamesIssue, dt as metaRef, en as requireRef, et as CallMeta, f as StreamCallResult, fn as HistoryToolResultIssue, ft as RateLimitOption, g as StreamJsonModeDisabledCallParams, gn as LLMErrorSnapshot, gt as RateLimitState, h as StreamEnabledCallParams, hn as LLMErrorIssuesByCode, ht as RateLimitReason, i as WireToolCall, in as OnUsage, it as FallbackOn, j as ContentBlock, jt as CooldownBackoff, k as ConditionalStringToolCallParams, kt as CircuitState, l as CachedStreamJsonModeDisabledCallParams, ln as ConsoleLogger, lt as defaultFallbackOn, m as StreamConditionalStringToolCallParams, mn as LLMErrorCode, mt as RateLimitOptions, n as LLMClient, nn as OnEvent, nt as FallbackAttempt, o as CachedStreamCallParams, on as RefundUsage, ot as FallbackTarget, p as StreamChunk, pn as LLMError, pt as RateLimitAcquireResult, q as ToolChoice, qt as WireCallRequest, r as WireMessage, rn as VernLLMEvent, rt as FallbackExhaustedError, s as CachedStreamConditionalStringToolCallParams, sn as ReserveUsage, st as TargetCircuitState, t as AdapterInfo, tn as stateEntry, tt as CircuitTarget, u as CachedStreamJsonModeEnabledCallParams, un as Logger, ut as isFallbackExhaustedError, v as WireStreamChunk, vn as LLMRequestSnapshot, vt as RateLimiterAdapter, w as CachedJsonModeDisabledCallParams, wn as isLLMError, wt as CircuitBreaker, x as CachedCallParams, xn as UnknownToolChoiceIssue, xt as CircuitBreakerOption, y as isStreamResult, yn as RetryAttempt, yt as WireRequest, z as TextBlock, zt as MiddlewareContextBase } from "./client-HWxkwVvj.cjs";
|
|
2
|
+
//#region src/types/cache.d.ts
|
|
3
|
+
interface CacheAdapter<T = unknown> {
|
|
4
|
+
get(key: string): Promise<{
|
|
5
|
+
hit: boolean;
|
|
6
|
+
value: T | null;
|
|
7
|
+
}>;
|
|
8
|
+
set(key: string, value: T, ttl: number): Promise<void>;
|
|
9
|
+
delete?(key: string): Promise<void>;
|
|
10
|
+
resolveKey?(key: string): Promise<string>;
|
|
1803
11
|
}
|
|
1804
12
|
/**
|
|
1805
|
-
*
|
|
1806
|
-
*
|
|
1807
|
-
*
|
|
1808
|
-
* the caching docs for why.
|
|
1809
|
-
*/
|
|
1810
|
-
type CachedCallParams<T> = CachedCallInput & {
|
|
1811
|
-
call: LLMRequestShape<T>;
|
|
1812
|
-
};
|
|
1813
|
-
/**
|
|
1814
|
-
* Parameters for a cached LLM call with tool calling enabled.
|
|
1815
|
-
*
|
|
1816
|
-
* The cached value includes the full `CallWithToolsResult<T>`, meaning
|
|
1817
|
-
* tool requests and normal content responses are cached exactly as returned
|
|
1818
|
-
* by the model.
|
|
1819
|
-
*
|
|
1820
|
-
* See `CachedCallParams` for why `reserveUsage`/`refundUsage` are omitted
|
|
1821
|
-
* from `call`'s type here too.
|
|
1822
|
-
*/
|
|
1823
|
-
type CachedToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
|
|
1824
|
-
call: LLMRequestShape<T, Tools> & {
|
|
1825
|
-
tools: NonNullable<LLMRequestShape<T, Tools>['tools']>;
|
|
1826
|
-
};
|
|
1827
|
-
};
|
|
1828
|
-
/**
|
|
1829
|
-
* Parameters for a cached LLM call with `call.tools` set conditionally.
|
|
1830
|
-
* Selects the `cachedCall()` overload that returns the honest union
|
|
1831
|
-
* `T | CallWithToolsResult<T>` instead of narrowing to plain `T`. See
|
|
1832
|
-
* `ConditionalToolCallParams` for why this overload exists.
|
|
1833
|
-
*/
|
|
1834
|
-
type CachedConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
|
|
1835
|
-
call: LLMRequestShape<T, Tools> & {
|
|
1836
|
-
tools: Tools | undefined;
|
|
1837
|
-
};
|
|
1838
|
-
};
|
|
1839
|
-
/** Cached conditional tool-call parameters whose non-tool result is plain text. */
|
|
1840
|
-
type CachedConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedConditionalToolCallParams<string, Tools> & {
|
|
1841
|
-
call: {
|
|
1842
|
-
jsonMode: false;
|
|
1843
|
-
};
|
|
1844
|
-
};
|
|
1845
|
-
/**
|
|
1846
|
-
* Parameters for a cached LLM call with `jsonMode: false`. Selects the
|
|
1847
|
-
* `cachedCall()` overload that returns a plain `string`.
|
|
13
|
+
* Which entry `InMemoryCacheAdapter` evicts once `maxSize` is exceeded.
|
|
14
|
+
* `'fifo'` (default) drops the oldest inserted entry. `'lru'` drops the
|
|
15
|
+
* least recently read or written entry.
|
|
1848
16
|
*/
|
|
1849
|
-
type
|
|
1850
|
-
call: Omit<LLMRequestShape<unknown>, 'jsonSchema'> & {
|
|
1851
|
-
jsonMode: false;
|
|
1852
|
-
jsonSchema?: never;
|
|
1853
|
-
};
|
|
1854
|
-
};
|
|
17
|
+
type EvictionOption = 'fifo' | 'lru';
|
|
1855
18
|
/**
|
|
1856
|
-
*
|
|
1857
|
-
*
|
|
19
|
+
* The default in-process cache. Not shared across processes; use a shared backend in production.
|
|
20
|
+
* Values are copied on `set` and `get`, so mutating a result never changes the next hit. A value
|
|
21
|
+
* that can't be copied faithfully, or a missing, NaN, zero or negative `ttl`, is not stored and
|
|
22
|
+
* drops any existing entry.
|
|
1858
23
|
*/
|
|
1859
|
-
|
|
1860
|
-
|
|
1861
|
-
|
|
1862
|
-
|
|
1863
|
-
|
|
1864
|
-
|
|
1865
|
-
|
|
1866
|
-
|
|
1867
|
-
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1871
|
-
|
|
1872
|
-
attempt: number;
|
|
1873
|
-
/**
|
|
1874
|
-
* Token usage for this attempt, if the provider reported it on this
|
|
1875
|
-
* response. `undefined` when the provider omitted usage, not when
|
|
1876
|
-
* usage was zero, so a cost check should treat a missing value as
|
|
1877
|
-
* unknown rather than as free.
|
|
1878
|
-
*/
|
|
1879
|
-
usage?: TokenUsage;
|
|
24
|
+
export declare class InMemoryCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
25
|
+
private readonly maxSize;
|
|
26
|
+
private store;
|
|
27
|
+
private readonly eviction;
|
|
28
|
+
constructor(maxSize?: number, eviction?: EvictionOption);
|
|
29
|
+
get(key: string): Promise<{
|
|
30
|
+
hit: boolean;
|
|
31
|
+
value: T | null;
|
|
32
|
+
}>;
|
|
33
|
+
set(key: string, value: T, ttl: number): Promise<void>;
|
|
34
|
+
delete(key: string): Promise<void>;
|
|
35
|
+
private cleanupExpiredEntries;
|
|
36
|
+
private enforceSizeLimit;
|
|
1880
37
|
}
|
|
1881
38
|
/**
|
|
1882
|
-
*
|
|
1883
|
-
*
|
|
1884
|
-
*
|
|
1885
|
-
* the same retry and circuit-breaker paths a thrown error would. A
|
|
1886
|
-
* result that parses fine but is empty, truncated, or a low-confidence
|
|
1887
|
-
* refusal is otherwise invisible to both.
|
|
1888
|
-
*/
|
|
1889
|
-
type DetectSoftFailure<T = unknown> = (result: T | CallWithToolsResult<T>, meta: SoftFailureMeta) => LLMErrorCode | undefined;
|
|
1890
|
-
//#endregion
|
|
1891
|
-
//#region src/types/stream.d.ts
|
|
1892
|
-
/** One incremental unit of a streaming response, as delivered to the caller. */
|
|
1893
|
-
type StreamChunk = {
|
|
1894
|
-
type: 'text-delta';
|
|
1895
|
-
delta: string;
|
|
1896
|
-
} | {
|
|
1897
|
-
type: 'tool_call_delta';
|
|
1898
|
-
index: number;
|
|
1899
|
-
id?: string;
|
|
1900
|
-
name?: string;
|
|
1901
|
-
argsDelta?: string;
|
|
1902
|
-
/**
|
|
1903
|
-
* True when `argsDelta` is the whole set of arguments, not a
|
|
1904
|
-
* fragment. Set for Gemini (its API returns function-call args
|
|
1905
|
-
* whole in one chunk) and for cache/replay chunks, which are
|
|
1906
|
-
* one-shot too. Omitted or `false` for a genuine fragment from
|
|
1907
|
-
* providers that do stream incrementally (OpenAI-compatible,
|
|
1908
|
-
* Anthropic, Bedrock).
|
|
1909
|
-
*/
|
|
1910
|
-
complete?: boolean;
|
|
1911
|
-
} | {
|
|
1912
|
-
type: 'usage';
|
|
1913
|
-
usage: TokenUsage;
|
|
1914
|
-
};
|
|
1915
|
-
/**
|
|
1916
|
-
* What `call()` returns when `stream: true`. `finalResult` resolves to
|
|
1917
|
-
* the same shape `call()` would have returned with `stream` omitted.
|
|
1918
|
-
* `chunks` is single-use and buffered; see the streaming docs for the
|
|
1919
|
-
* full consumption/backpressure semantics.
|
|
39
|
+
* Normalizes keys to avoid duplicates from formatting that can't change a prompt's meaning: Unicode
|
|
40
|
+
* composition, line endings and outer whitespace. Case, punctuation and inner whitespace are kept,
|
|
41
|
+
* since `"2+2"` and `"2-2"` must never share an answer.
|
|
1920
42
|
*/
|
|
1921
|
-
|
|
1922
|
-
|
|
1923
|
-
|
|
43
|
+
export declare class NormalizedCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
44
|
+
private readonly inner;
|
|
45
|
+
constructor(inner?: CacheAdapter<T>);
|
|
46
|
+
private normalize;
|
|
47
|
+
resolveKey(key: string): Promise<string>;
|
|
48
|
+
get(key: string): Promise<{
|
|
49
|
+
hit: boolean;
|
|
50
|
+
value: T | null;
|
|
51
|
+
}>;
|
|
52
|
+
set(key: string, value: T, ttl: number): Promise<void>;
|
|
53
|
+
delete(key: string): Promise<void>;
|
|
1924
54
|
}
|
|
1925
55
|
/**
|
|
1926
|
-
*
|
|
1927
|
-
*
|
|
1928
|
-
*
|
|
1929
|
-
* select the streaming `call()` overload and return `StreamCallResult<...>`
|
|
1930
|
-
* instead of the normal, single-shot response type.
|
|
1931
|
-
*/
|
|
1932
|
-
type StreamEnabledCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CallParams<T, Tools> & {
|
|
1933
|
-
stream: true;
|
|
1934
|
-
};
|
|
1935
|
-
/** Streaming conditional tool-call parameters whose non-tool result is text. */
|
|
1936
|
-
type StreamConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = StreamEnabledCallParams<string, Tools> & ConditionalStringToolCallParams<Tools>;
|
|
1937
|
-
/** Recovers `T` from a `result` already typed `T | StreamCallResult<T>`. Falls back to `unknown`. */
|
|
1938
|
-
type ExtractStreamValue<R> = Extract<R, StreamCallResult<unknown>> extends StreamCallResult<infer V> ? V : unknown;
|
|
1939
|
-
/**
|
|
1940
|
-
* Runtime check for whether a `call()` result is a `StreamCallResult`
|
|
1941
|
-
* (`{ chunks, finalResult }`) rather than the resolved value directly.
|
|
1942
|
-
* Useful when `stream` was computed conditionally and cast/narrowed
|
|
1943
|
-
* manually, since TypeScript's `call()` overloads only pick the streaming
|
|
1944
|
-
* shape for a literal `stream: true` at the call site.
|
|
1945
|
-
*
|
|
1946
|
-
* ```ts
|
|
1947
|
-
* const params = someCondition ? { userContent: '...', stream: true } : { userContent: '...' };
|
|
1948
|
-
* const result = await llm.call(params as CallParams<string> | (CallParams<string> & { stream: true }));
|
|
1949
|
-
* if (isStreamResult(result)) {
|
|
1950
|
-
* for await (const chunk of result.chunks) { ... }
|
|
1951
|
-
* }
|
|
1952
|
-
* ```
|
|
1953
|
-
*/
|
|
1954
|
-
export declare function isStreamResult<R = unknown>(result: R): result is R & StreamCallResult<ExtractStreamValue<R>>;
|
|
1955
|
-
/**
|
|
1956
|
-
* `StreamEnabledCallParams` with `jsonMode: false`. Selects the streaming
|
|
1957
|
-
* `call()` overload whose `finalResult` resolves to a plain `string`.
|
|
1958
|
-
* `jsonSchema` is typed `never` for the same reason as
|
|
1959
|
-
* `JsonModeDisabledCallParams`.
|
|
56
|
+
* Two tier cache: fast local L1, shared L2, with L2 hits promoted to L1. An L1 entry never outlives
|
|
57
|
+
* an L2 entry this adapter wrote. A promoted entry another process wrote keeps `l1Ttl`, since L2
|
|
58
|
+
* doesn't report remaining ttl.
|
|
1960
59
|
*/
|
|
1961
|
-
|
|
1962
|
-
|
|
1963
|
-
|
|
1964
|
-
|
|
1965
|
-
/**
|
|
1966
|
-
|
|
1967
|
-
* the streaming `call()` overload whose `finalResult` resolves to a
|
|
1968
|
-
* `JsonValue`.
|
|
1969
|
-
*
|
|
1970
|
-
* `schema` is explicitly `never` here for the same reason as
|
|
1971
|
-
* `JsonModeEnabledCallParams`: a schema whose result type is itself
|
|
1972
|
-
* structurally assignable to `JsonValue` would otherwise still satisfy this
|
|
1973
|
-
* overload's shape and incorrectly widen the result to `JsonValue` instead
|
|
1974
|
-
* of the schema's real type.
|
|
1975
|
-
*/
|
|
1976
|
-
type StreamJsonModeEnabledCallParams = Omit<StreamEnabledCallParams<JsonValue>, 'schema'> & {
|
|
1977
|
-
jsonMode: true;
|
|
1978
|
-
schema?: never;
|
|
1979
|
-
};
|
|
1980
|
-
/**
|
|
1981
|
-
* The adapter-facing, pre-normalization shape a `createStream` client
|
|
1982
|
-
* implementation emits, analogous to how `WireMessage`/`WireToolCall`
|
|
1983
|
-
* already sit between `CallParams` and each provider's own wire format.
|
|
1984
|
-
*/
|
|
1985
|
-
type WireStreamChunk = {
|
|
1986
|
-
type: 'text-delta';
|
|
1987
|
-
delta: string;
|
|
1988
|
-
} | {
|
|
1989
|
-
type: 'tool_call_delta';
|
|
1990
|
-
index: number;
|
|
1991
|
-
id?: string;
|
|
1992
|
-
name?: string;
|
|
1993
|
-
argumentsDelta?: string;
|
|
1994
|
-
/** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
|
|
1995
|
-
complete?: boolean;
|
|
1996
|
-
} | {
|
|
1997
|
-
type: 'usage';
|
|
1998
|
-
usage: {
|
|
1999
|
-
prompt_tokens?: number;
|
|
2000
|
-
completion_tokens?: number;
|
|
2001
|
-
total_tokens?: number;
|
|
2002
|
-
completion_tokens_details?: {
|
|
2003
|
-
reasoning_tokens?: number;
|
|
2004
|
-
};
|
|
2005
|
-
};
|
|
2006
|
-
} | {
|
|
60
|
+
export declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
61
|
+
private readonly l1;
|
|
62
|
+
private readonly l2;
|
|
63
|
+
private readonly l1Ttl?;
|
|
64
|
+
/** L2 expiry (epoch ms) per key this adapter wrote, oldest write first. */
|
|
65
|
+
private readonly l2ExpiresAt;
|
|
2007
66
|
/**
|
|
2008
|
-
*
|
|
2009
|
-
*
|
|
2010
|
-
* Adapters yield this so the stream loop resets its idle timeout.
|
|
2011
|
-
* Never surfaced to callers as a `StreamChunk`.
|
|
67
|
+
* Expiries as a min heap, so a write only visits expired records. Stale heap entries are skipped
|
|
68
|
+
* unless the map still holds that exact expiry.
|
|
2012
69
|
*/
|
|
2013
|
-
|
|
2014
|
-
|
|
70
|
+
private expiryHeap;
|
|
71
|
+
constructor(l1: CacheAdapter<T>, l2: CacheAdapter<T>, l1Ttl?: number | undefined);
|
|
2015
72
|
/**
|
|
2016
|
-
*
|
|
2017
|
-
*
|
|
2018
|
-
*
|
|
2019
|
-
* for the non-streaming path, just carried as a chunk instead of a
|
|
2020
|
-
* hidden property on a response object, since a stream has no
|
|
2021
|
-
* single response value to attach one to. Never surfaced to
|
|
2022
|
-
* callers as a `StreamChunk`.
|
|
73
|
+
* Forwards to L1's `resolveKey` if it has one, otherwise L2's. L1 is
|
|
74
|
+
* preferred since `get()` checks L1 first, so its notion of "the same
|
|
75
|
+
* key" is the one that determines whether a lookup can skip L2 entirely.
|
|
2023
76
|
*/
|
|
2024
|
-
|
|
2025
|
-
|
|
2026
|
-
|
|
2027
|
-
|
|
2028
|
-
|
|
2029
|
-
|
|
2030
|
-
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2034
|
-
|
|
2035
|
-
|
|
2036
|
-
* `reserveUsage`/`refundUsage` are omitted from `call`'s type; see
|
|
2037
|
-
* `CachedCallParams` for why they belong at the top level here too.
|
|
2038
|
-
*/
|
|
2039
|
-
type CachedStreamCallParams<T> = CachedCallInput & {
|
|
2040
|
-
call: LLMRequestShape<T> & {
|
|
2041
|
-
stream: true;
|
|
2042
|
-
};
|
|
2043
|
-
};
|
|
2044
|
-
/**
|
|
2045
|
-
* Parameters for a cached, streaming LLM call with tool calling enabled.
|
|
2046
|
-
*
|
|
2047
|
-
* The cached value is the full `CallWithToolsResult<T>`, same as
|
|
2048
|
-
* `CachedToolCallParams<T>`, with the same live-chunks-on-miss,
|
|
2049
|
-
* replayed-chunks-on-hit behavior as `CachedStreamCallParams<T>`.
|
|
2050
|
-
*/
|
|
2051
|
-
type CachedStreamToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
|
|
2052
|
-
call: LLMRequestShape<T, Tools> & {
|
|
2053
|
-
stream: true;
|
|
2054
|
-
tools: NonNullable<LLMRequestShape<T, Tools>['tools']>;
|
|
2055
|
-
};
|
|
2056
|
-
};
|
|
2057
|
-
/**
|
|
2058
|
-
* Parameters for a cached, streaming LLM call with `call.tools` set
|
|
2059
|
-
* conditionally. Selects the `cachedCall()` overload whose `finalResult`
|
|
2060
|
-
* (on a miss) or cached value (on a hit) is the honest union
|
|
2061
|
-
* `T | CallWithToolsResult<T>` instead of narrowing to plain `T`. See
|
|
2062
|
-
* `ConditionalToolCallParams` for why this overload exists.
|
|
2063
|
-
*/
|
|
2064
|
-
type CachedStreamConditionalToolCallParams<T, Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedCallInput & {
|
|
2065
|
-
call: LLMRequestShape<T, Tools> & {
|
|
2066
|
-
stream: true;
|
|
2067
|
-
tools: Tools | undefined;
|
|
2068
|
-
};
|
|
2069
|
-
};
|
|
2070
|
-
/** Cached streaming conditional tool-call parameters whose non-tool result is text. */
|
|
2071
|
-
type CachedStreamConditionalStringToolCallParams<Tools extends readonly ToolDefinition[] = ToolDefinition[]> = CachedStreamConditionalToolCallParams<string, Tools> & {
|
|
2072
|
-
call: {
|
|
2073
|
-
jsonMode: false;
|
|
2074
|
-
};
|
|
2075
|
-
};
|
|
2076
|
-
/**
|
|
2077
|
-
* Parameters for a cached, streaming LLM call with `jsonMode: false`.
|
|
2078
|
-
* Selects the `cachedCall()` overload whose `finalResult` (on a miss) or
|
|
2079
|
-
* cached value (on a hit) is a plain `string`.
|
|
2080
|
-
*/
|
|
2081
|
-
type CachedStreamJsonModeDisabledCallParams = CachedCallInput & {
|
|
2082
|
-
call: Omit<LLMRequestShape<unknown>, 'jsonSchema'> & {
|
|
2083
|
-
stream: true;
|
|
2084
|
-
jsonMode: false;
|
|
2085
|
-
jsonSchema?: never;
|
|
2086
|
-
};
|
|
2087
|
-
};
|
|
2088
|
-
/**
|
|
2089
|
-
* Parameters for a cached, streaming LLM call with `jsonMode: true` and no
|
|
2090
|
-
* `schema`. Selects the `cachedCall()` overload whose `finalResult` (on a
|
|
2091
|
-
* miss) or cached value (on a hit) is a `JsonValue`.
|
|
2092
|
-
*/
|
|
2093
|
-
type CachedStreamJsonModeEnabledCallParams = CachedCallInput & {
|
|
2094
|
-
call: Omit<LLMRequestShape<JsonValue>, 'schema'> & {
|
|
2095
|
-
stream: true;
|
|
2096
|
-
jsonMode: true;
|
|
2097
|
-
schema?: never;
|
|
2098
|
-
};
|
|
2099
|
-
};
|
|
2100
|
-
//#endregion
|
|
2101
|
-
//#region src/types/client.d.ts
|
|
2102
|
-
/** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
|
|
2103
|
-
interface WireToolCall {
|
|
2104
|
-
id: string;
|
|
2105
|
-
type: 'function';
|
|
2106
|
-
function: {
|
|
2107
|
-
name: string;
|
|
2108
|
-
/** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
|
|
2109
|
-
arguments: string;
|
|
2110
|
-
};
|
|
77
|
+
resolveKey(key: string): Promise<string>;
|
|
78
|
+
get(key: string): Promise<{
|
|
79
|
+
hit: boolean;
|
|
80
|
+
value: T | null;
|
|
81
|
+
}>;
|
|
82
|
+
set(key: string, value: T, ttl: number): Promise<void>;
|
|
83
|
+
delete(key: string): Promise<void>;
|
|
84
|
+
/** L1 ttl for a promoted entry, capped at the seconds L2 has left when that is known. */
|
|
85
|
+
private promotionTtl;
|
|
86
|
+
private trackExpiry;
|
|
87
|
+
/** Drops every tracked record that has expired, visiting only those. */
|
|
88
|
+
private pruneExpired;
|
|
2111
89
|
}
|
|
2112
|
-
|
|
2113
|
-
|
|
2114
|
-
role: 'system';
|
|
2115
|
-
content: string;
|
|
2116
|
-
} | {
|
|
2117
|
-
role: 'user';
|
|
2118
|
-
content: string | ContentBlock[];
|
|
2119
|
-
} | {
|
|
2120
|
-
role: 'assistant';
|
|
2121
|
-
/** Optional: an assistant turn that only requested tools has no text. */
|
|
2122
|
-
content?: string;
|
|
2123
|
-
tool_calls?: WireToolCall[];
|
|
2124
|
-
} | {
|
|
2125
|
-
role: 'tool';
|
|
2126
|
-
tool_call_id: string;
|
|
2127
|
-
content: string;
|
|
2128
|
-
/** Honored by `fromAnthropic` (maps to `tool_result.is_error`) and `fromBedrock` (maps to `toolResult.status`); other adapters ignore it. */
|
|
2129
|
-
is_error?: boolean;
|
|
2130
|
-
};
|
|
2131
|
-
/** The OpenAI-shaped wire `tool_choice`. */
|
|
2132
|
-
type WireToolChoice = 'auto' | 'none' | 'required' | {
|
|
2133
|
-
type: 'function';
|
|
2134
|
-
function: {
|
|
2135
|
-
name: string;
|
|
2136
|
-
};
|
|
2137
|
-
};
|
|
90
|
+
//#endregion
|
|
91
|
+
//#region src/internal/utils/rate-limit/tokenEstimate.utils.d.ts
|
|
2138
92
|
/**
|
|
2139
|
-
*
|
|
2140
|
-
* `
|
|
2141
|
-
* providers that don't support them will just ignore fields they don't recognize,
|
|
2142
|
-
* but not every SDKs TS types accept them, hence this being a structural type
|
|
2143
|
-
* rather than importing the SDKs own params type
|
|
93
|
+
* Default `estimateTokens`: chars/4 over every message's text, plus
|
|
94
|
+
* `IMAGE_TOKEN_ESTIMATE` per image, plus the requested `max_tokens`.
|
|
2144
95
|
*/
|
|
2145
|
-
|
|
2146
|
-
/**
|
|
2147
|
-
* Whether this client supports OpenAI's `response_format: { type:
|
|
2148
|
-
* 'json_object' }` as a real, API-level constraint. Defaults to `true`
|
|
2149
|
-
* when omitted (every OpenAI-compatible client and `fromGemini` map it to
|
|
2150
|
-
* a real field). `fromAnthropic` and `fromBedrock` set this to `false`:
|
|
2151
|
-
* neither provider has a field that mechanically guarantees JSON output
|
|
2152
|
-
* for this mode, so `RequestBuilder` downgrades a *default* (unset)
|
|
2153
|
-
* `jsonMode` to plain text for these clients instead of requesting
|
|
2154
|
-
* `json_object` and getting an unenforced, provider-side no-op back. An
|
|
2155
|
-
* *explicit* `jsonMode: true` still throws for such clients, since that's
|
|
2156
|
-
* a caller deliberately asking for a guarantee the client can't provide.
|
|
2157
|
-
*/
|
|
2158
|
-
supportsJsonObjectMode?: boolean;
|
|
2159
|
-
chat: {
|
|
2160
|
-
completions: {
|
|
2161
|
-
create(params: {
|
|
2162
|
-
model: string;
|
|
2163
|
-
temperature?: number;
|
|
2164
|
-
max_tokens: number;
|
|
2165
|
-
response_format?: {
|
|
2166
|
-
type: 'json_object';
|
|
2167
|
-
} | {
|
|
2168
|
-
type: 'json_schema';
|
|
2169
|
-
json_schema: {
|
|
2170
|
-
name: string;
|
|
2171
|
-
schema: Record<string, unknown>;
|
|
2172
|
-
strict?: boolean;
|
|
2173
|
-
description?: string;
|
|
2174
|
-
};
|
|
2175
|
-
};
|
|
2176
|
-
/** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
|
|
2177
|
-
reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
2178
|
-
/**
|
|
2179
|
-
* Numeric reasoning token budget, for providers with a native
|
|
2180
|
-
* budget field (Anthropic, Gemini). Ignored by clients that only
|
|
2181
|
-
* understand `reasoning_effort` tiers, use that field instead for
|
|
2182
|
-
* those.
|
|
2183
|
-
*/
|
|
2184
|
-
budget_tokens?: number;
|
|
2185
|
-
/** Tools the model may call, OpenAI's `function`-wrapped shape. */
|
|
2186
|
-
tools?: Array<{
|
|
2187
|
-
type: 'function';
|
|
2188
|
-
function: {
|
|
2189
|
-
name: string;
|
|
2190
|
-
description: string;
|
|
2191
|
-
parameters: Record<string, unknown>;
|
|
2192
|
-
};
|
|
2193
|
-
}>;
|
|
2194
|
-
tool_choice?: WireToolChoice;
|
|
2195
|
-
/**
|
|
2196
|
-
* Wire-format messages. Breaking change for custom adapters:
|
|
2197
|
-
* implementations must handle tool messages and assistant tool_calls.
|
|
2198
|
-
* Exhaustive switches over only system/user/assistant roles may no longer compile.
|
|
2199
|
-
*/
|
|
2200
|
-
messages: WireMessage[];
|
|
2201
|
-
}, options: {
|
|
2202
|
-
signal: AbortSignal;
|
|
2203
|
-
}): Promise<{
|
|
2204
|
-
choices?: Array<{
|
|
2205
|
-
message?: {
|
|
2206
|
-
content?: string | null;
|
|
2207
|
-
tool_calls?: WireToolCall[];
|
|
2208
|
-
};
|
|
2209
|
-
}>;
|
|
2210
|
-
usage?: {
|
|
2211
|
-
prompt_tokens?: number;
|
|
2212
|
-
completion_tokens?: number;
|
|
2213
|
-
total_tokens?: number;
|
|
2214
|
-
completion_tokens_details?: {
|
|
2215
|
-
reasoning_tokens?: number;
|
|
2216
|
-
};
|
|
2217
|
-
};
|
|
2218
|
-
}>;
|
|
2219
|
-
/**
|
|
2220
|
-
* Optional. Required only for `stream: true` calls. Adapters/clients
|
|
2221
|
-
* that don't implement this make `stream: true` throw a clear
|
|
2222
|
-
* `LLMError('validation')` rather than a confusing runtime failure.
|
|
2223
|
-
* Takes the same request shape as `create`, minus the response type.
|
|
2224
|
-
*/
|
|
2225
|
-
createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
|
|
2226
|
-
signal: AbortSignal;
|
|
2227
|
-
}): AsyncIterable<WireStreamChunk>;
|
|
2228
|
-
};
|
|
2229
|
-
};
|
|
2230
|
-
}
|
|
96
|
+
export declare function defaultEstimateTokens(request: WireRequest): number;
|
|
2231
97
|
//#endregion
|
|
2232
98
|
//#region src/internal/utils/cache/cacheAdapter.utils.d.ts
|
|
2233
99
|
/**
|
|
@@ -2241,16 +107,6 @@ type CacheOption = {
|
|
|
2241
107
|
eviction?: EvictionOption;
|
|
2242
108
|
} | CacheAdapter;
|
|
2243
109
|
//#endregion
|
|
2244
|
-
//#region src/internal/utils/circuit-breaker/circuitBreakerAdapter.utils.d.ts
|
|
2245
|
-
/**
|
|
2246
|
-
* Not re-exported from the package root, imported directly from this
|
|
2247
|
-
* internal module by `VernLLMOptions.circuitBreaker`'s own type (see
|
|
2248
|
-
* options.ts) so that union isn't duplicated between the public option
|
|
2249
|
-
* field and `buildCircuitBreaker`'s own signature below, same pattern
|
|
2250
|
-
* `CacheOption` and `RateLimitOption` already use for their own options.
|
|
2251
|
-
*/
|
|
2252
|
-
type CircuitBreakerOption = boolean | CircuitBreakerOptions | CircuitBreakerAdapter;
|
|
2253
|
-
//#endregion
|
|
2254
110
|
//#region src/types/options.d.ts
|
|
2255
111
|
interface VernLLMOptions {
|
|
2256
112
|
client: LLMClient;
|
|
@@ -2265,17 +121,24 @@ interface VernLLMOptions {
|
|
|
2265
121
|
/** Per-attempt timeout in ms. Default 25000 */
|
|
2266
122
|
timeoutMs?: number;
|
|
2267
123
|
/**
|
|
2268
|
-
*
|
|
2269
|
-
* stream
|
|
2270
|
-
*
|
|
2271
|
-
* chunk; this covers every gap after that. Also counts as a
|
|
2272
|
-
* circuit-breaker failure, unlike other mid-stream errors, since a
|
|
2273
|
-
* provider that streams one chunk then stalls should still trip it.
|
|
2274
|
-
* Default 30000. Pass 0 or negative to disable.
|
|
124
|
+
* Max gap between stream chunks once open, in ms. Resets on every chunk, pings included. Unlike
|
|
125
|
+
* other mid-stream errors it counts toward the breaker, so a provider that stalls after one chunk
|
|
126
|
+
* still trips it. Default 30000; 0 or negative disables.
|
|
2275
127
|
*/
|
|
2276
128
|
chunkIdleTimeoutMs?: number;
|
|
129
|
+
/**
|
|
130
|
+
* How long a `chunks` reader may stop pulling on a full buffer before it is detached. Its next
|
|
131
|
+
* pull rejects with code `reader_stall_timeout`, while the stream finishes so `finalResult`
|
|
132
|
+
* settles and its slot is freed. Off by default.
|
|
133
|
+
*/
|
|
134
|
+
readerStallTimeoutMs?: number;
|
|
2277
135
|
/** Base delay for exponential backoff in ms. Default 500 */
|
|
2278
136
|
baseDelayMs?: number;
|
|
137
|
+
/**
|
|
138
|
+
* Longest `Retry-After` wait honored, in ms. Also caps `LLMError.retryAfterMs`. `0` retries at
|
|
139
|
+
* once; `Infinity` removes the cap. Default 10000. Negative or NaN throws.
|
|
140
|
+
*/
|
|
141
|
+
maxRetryAfterMs?: number;
|
|
2279
142
|
/** Default max_tokens for calls that don't override it. Default 1000 */
|
|
2280
143
|
defaultMaxTokens?: number;
|
|
2281
144
|
/**
|
|
@@ -2284,76 +147,42 @@ interface VernLLMOptions {
|
|
|
2284
147
|
* request entirely, so the provider applies its own default instead.
|
|
2285
148
|
*/
|
|
2286
149
|
defaultTemperature?: number | null;
|
|
2287
|
-
/**
|
|
2288
|
-
* Default reasoning effort for calls that don't override it. Not sent
|
|
2289
|
-
* when omitted, same as leaving `reasoningEffort` unset on a call. See
|
|
2290
|
-
* `budgetTokens`/`reasoningEffort` on `CallParams` for how the two
|
|
2291
|
-
* relate and how each adapter converts between them.
|
|
2292
|
-
*/
|
|
150
|
+
/** Default reasoning effort for calls that don't set one. Not sent when omitted. */
|
|
2293
151
|
defaultReasoningEffort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
2294
|
-
/**
|
|
2295
|
-
* Default reasoning token budget for calls that don't override it. Not
|
|
2296
|
-
* sent when omitted. If both this and `defaultReasoningEffort` are set,
|
|
2297
|
-
* each adapter still prefers whichever field it natively understands,
|
|
2298
|
-
* same as at the per-call level.
|
|
2299
|
-
*/
|
|
152
|
+
/** Default reasoning token budget for calls that don't set one. Not sent when omitted. */
|
|
2300
153
|
defaultBudgetTokens?: number;
|
|
2301
154
|
/**
|
|
2302
|
-
*
|
|
2303
|
-
*
|
|
2304
|
-
* default `ConsoleLogger`: when a custom `logger` is supplied instead,
|
|
2305
|
-
* that logger's own `debug()` implementation decides whether messages
|
|
2306
|
-
* are emitted, and this option has no effect on it.
|
|
155
|
+
* Logs raw model output (up to 800 chars) and provider errors. Only affects the default
|
|
156
|
+
* `ConsoleLogger`; a custom `logger` decides for itself.
|
|
2307
157
|
*/
|
|
2308
158
|
debug?: boolean;
|
|
2309
159
|
/**
|
|
2310
|
-
* Applied before
|
|
2311
|
-
*
|
|
2312
|
-
*
|
|
2313
|
-
* intercept itself, since it's a direct call into `logger.debug`
|
|
2314
|
-
* rather than something routed through `onEvent`/`onUsage`; anything
|
|
2315
|
-
* caught elsewhere (events, `LLMError.cause`) already passes through
|
|
2316
|
-
* the app's own callback and can be redacted there instead. Runs
|
|
2317
|
-
* before `logger.debug()` regardless of whether that call ends up
|
|
2318
|
-
* emitting anything, so with a custom `logger`, `redact` still applies
|
|
2319
|
-
* even without `debug: true`; see `debug` for why. Default: identity
|
|
2320
|
-
* (no redaction).
|
|
160
|
+
* Applied to model output and provider errors before VernLLM's own `logger.debug()` calls, the
|
|
161
|
+
* one log path an app can't intercept. Runs even without `debug: true`, since a custom logger may
|
|
162
|
+
* emit debug anyway. Default: no redaction.
|
|
2321
163
|
*/
|
|
2322
164
|
redact?: (text: string) => string;
|
|
2323
165
|
/**
|
|
2324
|
-
* Cache for cachedCall
|
|
2325
|
-
*
|
|
2326
|
-
* `CacheAdapter` directly for a real backend. Default: in-memory,
|
|
2327
|
-
* maxSize 1000, fifo.
|
|
166
|
+
* Cache for `cachedCall`. `{ maxSize, eviction }` configures the built in adapter; pass a
|
|
167
|
+
* `CacheAdapter` for a real backend. Default in memory, 1000 entries, fifo.
|
|
2328
168
|
*/
|
|
2329
169
|
cache?: CacheOption;
|
|
2330
170
|
/**
|
|
2331
|
-
* Reclassifies an otherwise
|
|
2332
|
-
*
|
|
2333
|
-
*
|
|
2334
|
-
* `undefined` leaves the result untouched; returning an
|
|
2335
|
-
* `LLMErrorCode` fails that attempt with it, feeding the same retry
|
|
2336
|
-
* and circuit-breaker paths a thrown error would. A throwing hook is
|
|
2337
|
-
* caught, logged, and treated as no soft failure, so a broken hook
|
|
2338
|
-
* degrades safely instead of failing every call.
|
|
171
|
+
* Reclassifies an otherwise successful result as a failure. Runs once per attempt after
|
|
172
|
+
* validation; returning an `LLMErrorCode` fails the attempt through the normal retry and breaker
|
|
173
|
+
* paths. A throwing hook is logged and ignored.
|
|
2339
174
|
*/
|
|
2340
175
|
detectSoftFailure?: DetectSoftFailure;
|
|
2341
|
-
/** HTTP status codes that should fail fast without retrying. Default [400, 401, 403, 404, 422] */
|
|
176
|
+
/** HTTP status codes that should fail fast without retrying. Default [400, 401, 402, 403, 404, 413, 422] */
|
|
2342
177
|
nonRetryableStatus?: number[];
|
|
2343
178
|
/** Custom JSON parser. Must return undefined/null on failure. Default: JSON.parse wrapped in try/catch */
|
|
2344
179
|
parseJson?: (content: string) => unknown;
|
|
2345
180
|
/** Called after every successful call with token usage, if the provider reports it */
|
|
2346
181
|
onUsage?: OnUsage;
|
|
2347
182
|
/**
|
|
2348
|
-
*
|
|
2349
|
-
*
|
|
2350
|
-
*
|
|
2351
|
-
*
|
|
2352
|
-
* For non-streaming calls, never fires for transport failures (timeout,
|
|
2353
|
-
* network error, non-retryable status), since no response means no usage
|
|
2354
|
-
* to report. For streaming calls, this is not guaranteed: a stream can
|
|
2355
|
-
* deliver a usage chunk and then fail later (e.g. an idle timeout waiting
|
|
2356
|
-
* for the final close), in which case this does fire.
|
|
183
|
+
* Fires when a response carried usage but post-processing then failed. `onUsage` only fires on
|
|
184
|
+
* success. A non-streaming transport failure has no usage to report; a stream can deliver usage
|
|
185
|
+
* and fail later, which does fire this.
|
|
2357
186
|
*/
|
|
2358
187
|
onUsageFailure?: OnUsageFailure;
|
|
2359
188
|
/**
|
|
@@ -2362,11 +191,8 @@ interface VernLLMOptions {
|
|
|
2362
191
|
*/
|
|
2363
192
|
logger?: Logger | 'silent';
|
|
2364
193
|
/**
|
|
2365
|
-
*
|
|
2366
|
-
*
|
|
2367
|
-
* Pass `true` for defaults, or an options object to tune threshold/cooldown.
|
|
2368
|
-
* Pass a `CircuitBreakerAdapter` instead for cross-process coordination,
|
|
2369
|
-
* the same pattern `cache` and `rateLimit` already support.
|
|
194
|
+
* Short-circuits calls after repeated failures. `true` for defaults, options to tune, or a
|
|
195
|
+
* `CircuitBreakerAdapter` for cross-process state.
|
|
2370
196
|
*/
|
|
2371
197
|
circuitBreaker?: CircuitBreakerOption;
|
|
2372
198
|
/**
|
|
@@ -2376,175 +202,95 @@ interface VernLLMOptions {
|
|
|
2376
202
|
*/
|
|
2377
203
|
onEvent?: OnEvent;
|
|
2378
204
|
/**
|
|
2379
|
-
* Client
|
|
2380
|
-
*
|
|
2381
|
-
*
|
|
2382
|
-
* handling already applied to a provider 429: this avoids tripping the
|
|
2383
|
-
* limit in the first place. Omit for unlimited (the default).
|
|
2384
|
-
*
|
|
2385
|
-
* A plain config object builds an in-process limiter. Pass a
|
|
2386
|
-
* `RateLimiterAdapter` instead for cross-process coordination.
|
|
205
|
+
* Client side rate limiting: queues calls to stay under request, token or concurrency caps rather
|
|
206
|
+
* than letting the provider reject them. A config object builds an in-process limiter; pass a
|
|
207
|
+
* `RateLimiterAdapter` for cross-process. Omit for unlimited.
|
|
2387
208
|
*/
|
|
2388
209
|
rateLimit?: RateLimitOption;
|
|
2389
210
|
/**
|
|
2390
|
-
* Caps
|
|
2391
|
-
*
|
|
2392
|
-
*
|
|
2393
|
-
* among them reaches `retryRatio`, further retries against this target
|
|
2394
|
-
* throw `LLMError('retry_budget_exhausted')` instead of retrying,
|
|
2395
|
-
* protecting the target's real capacity even while its breaker is
|
|
2396
|
-
* still closed. Omit for no budget (the default). Never inherited by
|
|
2397
|
-
* `fallback` targets, same as `circuitBreaker`/`rateLimit`.
|
|
211
|
+
* Caps the share of this target's recent traffic that may be retries. Once `minCalls` calls land
|
|
212
|
+
* in `windowMs` and the retry ratio reaches `retryRatio`, retries throw `retry_budget_exhausted`,
|
|
213
|
+
* even while the breaker is closed. Not inherited by fallback targets.
|
|
2398
214
|
*/
|
|
2399
215
|
retryBudget?: RetryBudgetOptions;
|
|
2400
216
|
/**
|
|
2401
|
-
*
|
|
2402
|
-
* own retries
|
|
2403
|
-
* never reorders, scores, or selects between targets. Each target keeps
|
|
2404
|
-
* its own retry state, circuit breaker, and rate limiter, independent
|
|
2405
|
-
* of every other target's. A single `FallbackTarget` is equivalent to
|
|
2406
|
-
* `[target]`.
|
|
217
|
+
* Targets tried in order after the primary is exhausted. VernLLM never reorders them. Each keeps
|
|
218
|
+
* its own retries, breaker and limiter. A single target equals `[target]`.
|
|
2407
219
|
*/
|
|
2408
220
|
fallback?: FallbackTarget | FallbackTarget[];
|
|
2409
221
|
/**
|
|
2410
|
-
*
|
|
2411
|
-
*
|
|
2412
|
-
*
|
|
2413
|
-
* once per failed target, after that target's own retries are
|
|
2414
|
-
* exhausted or abandoned early, so `'retry'` is never a valid return
|
|
2415
|
-
* here. Defaults to `defaultFallbackOn`, which stops on
|
|
2416
|
-
* parse/validation/aborted/quota errors and on tool-contract failures
|
|
2417
|
-
* (the model ignoring the request, not the provider being unhealthy),
|
|
2418
|
-
* and moves on for everything else.
|
|
222
|
+
* Whether a failed target moves on (`'next'`) or ends the call (`'stop'`). Called once per failed
|
|
223
|
+
* target after its own retries. Defaults to `defaultFallbackOn`; see its docs for what it stops
|
|
224
|
+
* on.
|
|
2419
225
|
*/
|
|
2420
226
|
fallbackOn?: FallbackOn;
|
|
2421
227
|
/**
|
|
2422
|
-
*
|
|
2423
|
-
*
|
|
2424
|
-
* Defaults to an empty array. See `VernLLMMiddleware` for the four
|
|
2425
|
-
* available hooks (`transform`, `wrap`, `onEvent`, `enabled`).
|
|
228
|
+
* Request transforms and call wrappers that leave retry, breaker and fallback internals alone.
|
|
229
|
+
* See `VernLLMMiddleware` for the hooks.
|
|
2426
230
|
*/
|
|
2427
231
|
middleware?: VernLLMMiddleware[];
|
|
2428
232
|
/**
|
|
2429
|
-
* Bounds `transform` and a function `enabled
|
|
2430
|
-
*
|
|
2431
|
-
* Overridable per middleware via that entry's own `timeoutMs`.
|
|
2432
|
-
* `<= 0` means unbounded (no timer at all). Default 5000.
|
|
233
|
+
* Bounds `transform` and a function `enabled`. Overridable per middleware with `timeoutMs`. `<=
|
|
234
|
+
* 0` means unbounded. Default 5000.
|
|
2433
235
|
*/
|
|
2434
236
|
middlewareTimeoutMs?: number;
|
|
2435
237
|
}
|
|
2436
238
|
//#endregion
|
|
2437
239
|
//#region src/types/createMiddleware.d.ts
|
|
2438
240
|
/**
|
|
2439
|
-
* `VernLLMMiddleware` plus `onError`,
|
|
2440
|
-
*
|
|
2441
|
-
* the resulting `VernLLMMiddleware` unchanged; setting `wrap` directly
|
|
2442
|
-
* alongside `onError` is an error, since `onError` builds its own `wrap`
|
|
2443
|
-
* under the hood, and building it around a `wrap` you also supplied
|
|
2444
|
-
* would silently drop one of the two.
|
|
241
|
+
* `VernLLMMiddleware` plus `onError`, for middleware that only cares about failures. `wrap` can't
|
|
242
|
+
* be set alongside it, since `onError` builds its own.
|
|
2445
243
|
*/
|
|
2446
244
|
type CreateMiddlewareOptions = Omit<VernLLMMiddleware, 'wrap'> & {
|
|
2447
245
|
wrap?: undefined;
|
|
2448
246
|
/**
|
|
2449
|
-
* Called with
|
|
2450
|
-
*
|
|
2451
|
-
*
|
|
2452
|
-
* already swallowed by short-circuiting with its own `CallResult`.
|
|
2453
|
-
* The original error is always rethrown afterward, `onError` only
|
|
2454
|
-
* observes it, exactly like `onUsage`/`onEvent` elsewhere: a throwing
|
|
2455
|
-
* `onError` is discarded (not logged, this helper has no `Logger` of
|
|
2456
|
-
* its own to log through) and otherwise has no effect on the call.
|
|
2457
|
-
* `ctx` is `wrap`'s own pre-dispatch context (`onError` builds a `wrap`
|
|
2458
|
-
* under the hood), so it only describes the primary target.
|
|
247
|
+
* Called with the call's terminal error. Not called on success or when another `wrap` swallowed
|
|
248
|
+
* the failure. Only observes: the error is always rethrown and a throwing `onError` is discarded.
|
|
249
|
+
* `ctx` is `wrap`'s pre-dispatch context.
|
|
2459
250
|
*/
|
|
2460
251
|
onError?: (error: LLMError, ctx: PreDispatchContext) => void | Promise<void>;
|
|
2461
252
|
};
|
|
2462
253
|
/**
|
|
2463
|
-
* Builds a
|
|
2464
|
-
*
|
|
2465
|
-
* reports `onError` on a rejection, and always rethrows the original
|
|
2466
|
-
* error afterward, so `onError` never changes what the call itself
|
|
2467
|
-
* returns or throws, only what gets observed about it.
|
|
254
|
+
* Builds a middleware entry. With `onError`, adds a `wrap` that reports rejections and always
|
|
255
|
+
* rethrows, so the call's outcome never changes.
|
|
2468
256
|
*/
|
|
2469
257
|
export declare function createMiddleware(options: CreateMiddlewareOptions): VernLLMMiddleware;
|
|
2470
258
|
//#endregion
|
|
2471
259
|
//#region src/vernLLM.d.ts
|
|
2472
260
|
/**
|
|
2473
|
-
*
|
|
2474
|
-
*
|
|
2475
|
-
* Adds retry with backoff and jitter, per-attempt timeouts, an optional
|
|
2476
|
-
* circuit breaker, JSON parsing with optional schema validation, usage
|
|
2477
|
-
* tracking, and an optional response cache. All configurable, all opt-in
|
|
2478
|
-
* beyond sensible defaults.
|
|
261
|
+
* The LLM call framework: retries, timeouts, circuit breaking, fallback, rate limiting, caching and
|
|
262
|
+
* middleware around any provider adapter.
|
|
2479
263
|
*/
|
|
2480
264
|
export declare class VernLLM {
|
|
2481
265
|
private readonly logger;
|
|
2482
|
-
/**
|
|
2483
|
-
* One `CallExecutor` per provider target: index 0 is the primary,
|
|
2484
|
-
* everything after it is a `fallback` target, in the order declared.
|
|
2485
|
-
* Walked by `runFallbackChain`, moving to the next entry only when
|
|
2486
|
-
* `fallbackOn` says to.
|
|
2487
|
-
*/
|
|
266
|
+
/** One per target: index 0 is the primary, then each fallback in declared order. */
|
|
2488
267
|
private readonly executors;
|
|
2489
|
-
/**
|
|
2490
|
-
private readonly fallbackOn;
|
|
2491
|
-
/** Reports a `'fallback'` event when the chain moves to the next target. Shared `onEvent` plumbing, same as every executor's. */
|
|
2492
|
-
private readonly reportEvent;
|
|
2493
|
-
/** Owns cache reads/writes and in-flight coalescing for `cachedCall()`. Only calls back into `this.call()` as an opaque function. */
|
|
268
|
+
/** Owns cache reads, writes and in-flight coalescing for `cachedCall()`. */
|
|
2494
269
|
private readonly cacheOrchestrator;
|
|
2495
|
-
/**
|
|
2496
|
-
|
|
2497
|
-
|
|
2498
|
-
|
|
2499
|
-
|
|
2500
|
-
|
|
2501
|
-
|
|
2502
|
-
|
|
2503
|
-
|
|
2504
|
-
|
|
2505
|
-
|
|
2506
|
-
* Maps `cachedCall()`'s inner `this.call(...)` params to its own
|
|
2507
|
-
* `middlewareState`, so that call's own `runOperation` skips wrapping
|
|
2508
|
-
* again and reuses the same state bag `wrap` just ran with (so a
|
|
2509
|
-
* value `wrap` sets is visible to `transform`, same as a direct
|
|
2510
|
-
* call). Keyed by object identity, not `requestId`, since two
|
|
2511
|
-
* concurrent `cachedCall()`s can share an explicit `requestId`.
|
|
270
|
+
/** Every target in declared order, the pool `targets` selects from. */
|
|
271
|
+
private readonly declaredTargets;
|
|
272
|
+
/** Everything but `targets`, which each call sets for its own order. */
|
|
273
|
+
private readonly logicalCallDependencies;
|
|
274
|
+
private readonly runOperationDependencies;
|
|
275
|
+
/** What each call's scope needs to deliver a `ctx.emit` event. */
|
|
276
|
+
private readonly callScopeDependencies;
|
|
277
|
+
/**
|
|
278
|
+
* Marks `cachedCall()`'s inner call params with its state bag, so the inner
|
|
279
|
+
* `runOperation` skips `wrap` and reuses that bag. Keyed by object identity,
|
|
280
|
+
* since concurrent `cachedCall()`s can share an explicit `requestId`.
|
|
2512
281
|
*/
|
|
2513
282
|
private readonly cachedCallInnerParams;
|
|
2514
|
-
/**
|
|
2515
|
-
* Shares one `CallMeta` holder across every `cachedCall()` in flight
|
|
2516
|
-
* for the same resolved cache key, so a joining invocation (never
|
|
2517
|
-
* calls `call()` itself) reports the trigger's real metadata instead
|
|
2518
|
-
* of `undefined`. A true cache hit never creates an entry, so it
|
|
2519
|
-
* still reports no metadata correctly.
|
|
2520
|
-
*/
|
|
283
|
+
/** Shared `CallMeta` holders per resolved cache key. See `claimMetaHolder`. */
|
|
2521
284
|
private readonly cachedCallMeta;
|
|
2522
|
-
/**
|
|
2523
|
-
* @param options Client, model, and tunables. Defaults: `maxRetries` 1,
|
|
2524
|
-
* `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
|
|
2525
|
-
* `defaultTemperature` 0.2, `cache` an in-memory adapter,
|
|
2526
|
-
* `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
|
|
2527
|
-
*/
|
|
285
|
+
/** @param options Client, model and tunables. See `VernLLMOptions` for each default. */
|
|
2528
286
|
constructor(options: VernLLMOptions);
|
|
2529
|
-
/** Logs a failed refundUsage attempt via the configured logger. */
|
|
2530
|
-
private logRefundError;
|
|
2531
|
-
/** Everything `executeLogicalCall`/`executeLogicalStreamCall` (in `logicalCall.ts`) need from this instance, gathered once so `call()` doesn't rebuild it per invocation. */
|
|
2532
|
-
private get logicalCallDependencies();
|
|
2533
|
-
/** Everything `runOperation` (in `runOperation.ts`) needs from this instance, gathered once so `call()`/`cachedCall()` don't rebuild it per invocation. */
|
|
2534
|
-
private get runOperationDependencies();
|
|
2535
287
|
/**
|
|
2536
|
-
* Makes
|
|
2537
|
-
*
|
|
2538
|
-
* aborted. Rejects with a normalized `LLMError` on exhausted retries.
|
|
2539
|
-
*
|
|
2540
|
-
* Supports `tools`, `stream`, and JSON mode/schema, in any combination.
|
|
2541
|
-
* See the Tool Calling and Streaming docs for return-shape details and
|
|
2542
|
-
* the TypeScript overloads that select between them.
|
|
288
|
+
* Makes one logical call, with retries, fallback and the breaker applied. Rejects with a
|
|
289
|
+
* normalized `LLMError`. See the Tool Calling and Streaming docs for the return shapes.
|
|
2543
290
|
*
|
|
2544
|
-
* @param params
|
|
2545
|
-
* @returns The parsed response
|
|
2546
|
-
*
|
|
2547
|
-
* finalResult }` `StreamCallResult` when `stream: true`. See `StreamCallResult`.
|
|
291
|
+
* @param params Content plus per call overrides. See `CallParams`.
|
|
292
|
+
* @returns The parsed response, a `CallWithToolsResult<T>` with `tools`, or a `StreamCallResult`
|
|
293
|
+
* with `stream: true`.
|
|
2548
294
|
*/
|
|
2549
295
|
call<T = unknown, const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: StreamEnabledCallParams<T, Tools> & ToolsDisabledCallParams<T, Tools>): Promise<StreamCallResult<ContentResult<T>>>;
|
|
2550
296
|
call<T = unknown, const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: StreamEnabledCallParams<T, Tools> & ToolEnabledCallParams<T, Tools>): Promise<StreamCallResult<CallWithToolsResult<T, Tools>>>;
|
|
@@ -2561,37 +307,26 @@ export declare class VernLLM {
|
|
|
2561
307
|
call(params: JsonModeEnabledCallParams): Promise<JsonValue>;
|
|
2562
308
|
call<T = unknown>(params: CallParams<T>): Promise<T>;
|
|
2563
309
|
/**
|
|
2564
|
-
*
|
|
2565
|
-
*
|
|
2566
|
-
* directly by white-box tests, independent of the public `cachedCall()`
|
|
2567
|
-
* surface.
|
|
310
|
+
* The order a call starts with: `requested` by name, or every target as declared. Throws before
|
|
311
|
+
* any timer or hook exists, so an invalid order never starts a call.
|
|
2568
312
|
*/
|
|
313
|
+
private resolveTargets;
|
|
314
|
+
/** Kept on `VernLLM` since tests drive the caching core directly through it. */
|
|
2569
315
|
private runCached;
|
|
2570
316
|
/**
|
|
2571
|
-
* Removes a cached response
|
|
2572
|
-
* supports deletion. Cache invalidation is the caller's responsibility;
|
|
2573
|
-
* only the application knows when cached data is stale.
|
|
317
|
+
* Removes a cached response when the adapter supports deletion. Invalidation is up to the app.
|
|
2574
318
|
*
|
|
2575
|
-
* @param key The raw cache key
|
|
2576
|
-
* `resolveKey`, if any, before deletion).
|
|
319
|
+
* @param key The raw cache key, resolved through the adapter's `resolveKey` first.
|
|
2577
320
|
*/
|
|
2578
321
|
deleteCache(key: string): Promise<void>;
|
|
2579
322
|
/**
|
|
2580
|
-
*
|
|
2581
|
-
*
|
|
2582
|
-
*
|
|
2583
|
-
*
|
|
323
|
+
* `call()` with caching. Concurrent misses for one `cacheKey` share a single in-flight call.
|
|
324
|
+
* Works with `stream` and `tools`; with tools the whole result is cached, tool call decisions
|
|
325
|
+
* included. A caller's own abort or deadline only ends its wait; the shared request is aborted
|
|
326
|
+
* once every caller has left.
|
|
2584
327
|
*
|
|
2585
|
-
*
|
|
2586
|
-
*
|
|
2587
|
-
* separate `cacheKey` if a tool's result shouldn't be reused across calls.
|
|
2588
|
-
*
|
|
2589
|
-
* @param params `cacheKey`, `ttl`, optional
|
|
2590
|
-
* `reserveUsage`/`refundUsage`/`signal`, plus `call`, the `CallParams`
|
|
2591
|
-
* to pass through to `this.call(...)`. The top-level `signal` governs
|
|
2592
|
-
* the cached operation and its usage hooks only; to also abort the
|
|
2593
|
-
* underlying provider request, set `signal` inside `call`.
|
|
2594
|
-
* @returns The cached value on a hit, or the freshly-called result on a miss.
|
|
328
|
+
* @param params Cache settings plus `call`, the `CallParams` for the underlying call.
|
|
329
|
+
* @returns The cached value on a hit, or the fresh result on a miss.
|
|
2595
330
|
*/
|
|
2596
331
|
cachedCall<T, const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: CachedStreamToolCallParams<T, Tools>): Promise<StreamCallResult<CallWithToolsResult<T, Tools>>>;
|
|
2597
332
|
cachedCall<const Tools extends readonly ToolDefinition[] = ToolDefinition[]>(params: CachedStreamConditionalStringToolCallParams<Tools>): Promise<StreamCallResult<string | CallWithToolsResult<string, Tools>>>;
|
|
@@ -2609,9 +344,7 @@ export declare class VernLLM {
|
|
|
2609
344
|
* @param target.index Which target to read. Defaults to the primary.
|
|
2610
345
|
* @param target.model Which model bucket to read, if the target isolates by model.
|
|
2611
346
|
* @returns The breaker state, or `undefined` if that target has no breaker.
|
|
2612
|
-
* @throws {RangeError} If `target.index` names no target.
|
|
2613
|
-
* target with no breaker (`undefined`) stay distinguishable from a
|
|
2614
|
-
* target that doesn't exist.
|
|
347
|
+
* @throws {RangeError} If `target.index` names no target.
|
|
2615
348
|
*/
|
|
2616
349
|
getCircuitState(target?: CircuitTarget): CircuitState | undefined;
|
|
2617
350
|
/**
|
|
@@ -2624,10 +357,8 @@ export declare class VernLLM {
|
|
|
2624
357
|
getFailureBreakdown(target?: CircuitTarget): Partial<Record<LLMErrorCode | 'unknown', number>> | undefined;
|
|
2625
358
|
/**
|
|
2626
359
|
* @param target.index Which target to read. Defaults to the primary.
|
|
2627
|
-
* @returns
|
|
2628
|
-
*
|
|
2629
|
-
* configured. A budget is target-scoped, not model-scoped, so unlike
|
|
2630
|
-
* `getFailureBreakdown` there's no `target.model` to pass.
|
|
360
|
+
* @returns The retry traffic and ratio in the trailing window, or `undefined` without a budget.
|
|
361
|
+
* Budgets are per target, not per model.
|
|
2631
362
|
* @throws {RangeError} If `target.index` names no target.
|
|
2632
363
|
*/
|
|
2633
364
|
getRetryBudgetState(target?: Pick<CircuitTarget, 'index'>): {
|
|
@@ -2636,10 +367,8 @@ export declare class VernLLM {
|
|
|
2636
367
|
} | undefined;
|
|
2637
368
|
/**
|
|
2638
369
|
* @param target.index Which target to read. Defaults to the primary.
|
|
2639
|
-
* @returns
|
|
2640
|
-
*
|
|
2641
|
-
* not model-scoped, so unlike `getFailureBreakdown` there's no
|
|
2642
|
-
* `target.model` to pass.
|
|
370
|
+
* @returns Current rate limit levels, or `undefined` without a limiter. Limiters are per target,
|
|
371
|
+
* not per model.
|
|
2643
372
|
* @throws {RangeError} If `target.index` names no target.
|
|
2644
373
|
*/
|
|
2645
374
|
getRateLimitState(target?: Pick<CircuitTarget, 'index'>): RateLimitState | undefined;
|
|
@@ -2686,1233 +415,15 @@ export declare class VernLLM {
|
|
|
2686
415
|
//#endregion
|
|
2687
416
|
//#region src/paramsHelpers.d.ts
|
|
2688
417
|
/**
|
|
2689
|
-
*
|
|
2690
|
-
*
|
|
2691
|
-
* `ConditionalToolCallParams<T>` overload for `tools: someCondition ?
|
|
2692
|
-
* [tool] : undefined`. Use it when you need `call()` params in a named,
|
|
2693
|
-
* reusable variable; skip it when you can pass the object inline.
|
|
2694
|
-
*
|
|
2695
|
-
* ```ts
|
|
2696
|
-
* const params = defineCallParams({
|
|
2697
|
-
* userContent: 'What is the weather?',
|
|
2698
|
-
* tools: someCondition ? [weatherTool] : undefined,
|
|
2699
|
-
* });
|
|
2700
|
-
* const result = await llm.call(params);
|
|
2701
|
-
* // result: unknown | CallWithToolsResult<unknown>, same as inline
|
|
2702
|
-
* ```
|
|
2703
|
-
*
|
|
2704
|
-
* `T` isn't a parameter here; pin it via `llm.call<T>(params)` as usual.
|
|
2705
|
-
* `defineCachedCallParams` is the `cachedCall()` counterpart.
|
|
418
|
+
* Keeps `params`' precise type for a reusable variable. A `CallParams<T>` annotation would widen
|
|
419
|
+
* `tools` and lose the conditional tools overload. Pin `T` through `llm.call<T>(params)`.
|
|
2706
420
|
*/
|
|
2707
421
|
export declare function defineCallParams<P extends CallParams<unknown>>(params: P): P;
|
|
2708
422
|
/**
|
|
2709
|
-
*
|
|
2710
|
-
* whole `{ cacheKey, ttl, call }` object, `call.tools` included, in one
|
|
2711
|
-
* named variable.
|
|
2712
|
-
*
|
|
2713
|
-
* ```ts
|
|
2714
|
-
* const params = defineCachedCallParams({
|
|
2715
|
-
* cacheKey: 'weather-ny',
|
|
2716
|
-
* ttl: 60,
|
|
2717
|
-
* call: { userContent: 'What is the weather?', tools: someCondition ? [weatherTool] : undefined },
|
|
2718
|
-
* });
|
|
2719
|
-
* const result = await llm.cachedCall(params);
|
|
2720
|
-
* ```
|
|
2721
|
-
*/
|
|
2722
|
-
export declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
|
|
2723
|
-
//#endregion
|
|
2724
|
-
//#region src/adapters/internal/sse.d.ts
|
|
2725
|
-
/**
|
|
2726
|
-
* Parses a Server-Sent-Events byte/text stream into the JSON payload of
|
|
2727
|
-
* each `data:` frame, in arrival order. Generic over transport: works with
|
|
2728
|
-
* anything that hands back progressively-arriving `Uint8Array` or `string`
|
|
2729
|
-
* chunks via async iteration: native `fetch`'s `response.body` (wrapped
|
|
2730
|
-
* to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
|
|
2731
|
-
* Node `Readable` (already async-iterable, no wrapping needed), etc, so
|
|
2732
|
-
* this framing layer doesn't care which transport produced the bytes.
|
|
2733
|
-
*
|
|
2734
|
-
* Follows the SSE spec's frame-delimiting rules closely enough for LLM
|
|
2735
|
-
* streaming responses: frames are separated by a blank line, each frame
|
|
2736
|
-
* may carry one or more `data:` lines (joined with `\n` per spec when
|
|
2737
|
-
* there's more than one), `:`-prefixed lines are comments and ignored, and
|
|
2738
|
-
* other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
|
|
2739
|
-
* only needs the payload. A frame whose data is exactly `[DONE]` (the
|
|
2740
|
-
* sentinel several providers, notably OpenAI, send to mark stream end)
|
|
2741
|
-
* ends iteration without yielding it.
|
|
2742
|
-
*
|
|
2743
|
-
* Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
|
|
2744
|
-
* to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
|
|
2745
|
-
* alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
|
|
2746
|
-
* stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
|
|
2747
|
-
* lines.
|
|
2748
|
-
*
|
|
2749
|
-
* Malformed JSON in a frame throws `LLMError('parse')`, consistent with
|
|
2750
|
-
* how malformed JSON is handled elsewhere in VernLLM.
|
|
2751
|
-
*/
|
|
2752
|
-
export declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
|
|
2753
|
-
/**
|
|
2754
|
-
* Sentinel yielded by `parseSseStream` for a comment-only frame (no
|
|
2755
|
-
* `data:` payload), the mechanism providers use for SSE keep-alive
|
|
2756
|
-
* pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
|
|
2757
|
-
* alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
|
|
2758
|
-
*/
|
|
2759
|
-
export declare const SSE_PING: unique symbol;
|
|
2760
|
-
//#endregion
|
|
2761
|
-
//#region src/adapters/internal/imageFormat.d.ts
|
|
2762
|
-
/**
|
|
2763
|
-
* MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
|
|
2764
|
-
* the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
|
|
2765
|
-
* Converse all natively support, so a `ContentBlock[]` that validates for
|
|
2766
|
-
* one provider validates for all of them.
|
|
2767
|
-
*/
|
|
2768
|
-
declare const SUPPORTED_IMAGE_MIME_TYPES: readonly ["image/png", "image/jpeg", "image/gif", "image/webp"];
|
|
2769
|
-
type SupportedImageMimeType = (typeof SUPPORTED_IMAGE_MIME_TYPES)[number];
|
|
2770
|
-
//#endregion
|
|
2771
|
-
//#region src/adapters/internal/nativeStructuredOutput.d.ts
|
|
2772
|
-
/**
|
|
2773
|
-
* A static allow-list or predicate naming which models support native,
|
|
2774
|
-
* schema-constrained output as its own request field. Anthropic's
|
|
2775
|
-
* `output_config.format` and Bedrock's `outputConfig.textFormat` separate
|
|
2776
|
-
* from `tools`/`tool_choice`, so it can be combined with real,
|
|
2777
|
-
* caller-supplied `tools` in the same request.
|
|
2778
|
-
*
|
|
2779
|
-
* There is no built-in default list here. Which models support this is
|
|
2780
|
-
* Anthropic's and Bedrock's call to make, not this package's, and it
|
|
2781
|
-
* changes over time; hardcoding a guessed list would risk silently
|
|
2782
|
-
* routing a request onto a field a given model doesn't actually support,
|
|
2783
|
-
* trading a clear `LLMError('invalid_params')` with
|
|
2784
|
-
* `code: 'unsupported_capability'` for a confusing error from the provider
|
|
2785
|
-
* instead. So this is opt-in: pass the model IDs you've verified against the
|
|
2786
|
-
* provider's own docs (or a predicate). Left unset, no model is treated as
|
|
2787
|
-
* native-capable, `jsonSchema` keeps using the older forced-single-tool-call
|
|
2788
|
-
* emulation, and combining it with `tools` throws the coded capability error,
|
|
2789
|
-
* exactly this package's behavior before native support was added.
|
|
2790
|
-
*/
|
|
2791
|
-
type ModelCapabilityOverride = string[] | ((model: string) => boolean);
|
|
2792
|
-
//#endregion
|
|
2793
|
-
//#region src/adapters/internal/reasoningBudget.utils.d.ts
|
|
2794
|
-
/**
|
|
2795
|
-
* Shared conversion between the two reasoning controls VernLLM exposes:
|
|
2796
|
-
* `reasoningEffort` (a tier string, OpenAI's native shape) and
|
|
2797
|
-
* `budgetTokens` (a raw integer, Anthropic's and Gemini's native shape).
|
|
2798
|
-
*
|
|
2799
|
-
* Every adapter prefers its own native field when the caller set it, and
|
|
2800
|
-
* only calls into this table when the caller set the other one instead.
|
|
2801
|
-
* The numbers here are a guess, not a provider guarantee, callers who
|
|
2802
|
-
* need a precise budget on a specific model should set `budgetTokens`
|
|
2803
|
-
* directly rather than relying on this table's `reasoningEffort` mapping.
|
|
2804
|
-
*
|
|
2805
|
-
* The table itself is overridable per adapter instance, via
|
|
2806
|
-
* `reasoningEffortTokens` on each `from*` adapter's options (see
|
|
2807
|
-
* `AnthropicAdapterOptions`, `GeminiAdapterOptions`,
|
|
2808
|
-
* `OpenAICompatibleAdapterOptions`, `BedrockAdapterOptions`), for callers
|
|
2809
|
-
* who want `reasoningEffort` tiers to map onto different token counts
|
|
2810
|
-
* than the defaults below, e.g. a model whose useful reasoning range
|
|
2811
|
-
* doesn't match these numbers.
|
|
2812
|
-
*/
|
|
2813
|
-
type EffortTokenTable = Record<'minimal' | 'low' | 'medium' | 'high', number>;
|
|
2814
|
-
//#endregion
|
|
2815
|
-
//#region src/adapters/anthropic.d.ts
|
|
2816
|
-
/** Anthropic's native per-block content shape for a message. */
|
|
2817
|
-
type AnthropicContentBlock = {
|
|
2818
|
-
type: 'text';
|
|
2819
|
-
text: string;
|
|
2820
|
-
} | {
|
|
2821
|
-
type: 'image';
|
|
2822
|
-
source: {
|
|
2823
|
-
type: 'base64';
|
|
2824
|
-
media_type: SupportedImageMimeType;
|
|
2825
|
-
data: string;
|
|
2826
|
-
};
|
|
2827
|
-
} | {
|
|
2828
|
-
type: 'tool_use';
|
|
2829
|
-
id: string;
|
|
2830
|
-
name: string;
|
|
2831
|
-
input: unknown;
|
|
2832
|
-
} | {
|
|
2833
|
-
type: 'tool_result';
|
|
2834
|
-
tool_use_id: string;
|
|
2835
|
-
content: string;
|
|
2836
|
-
is_error?: boolean;
|
|
2837
|
-
};
|
|
2838
|
-
/** Minimal structural type for the Anthropic SDK's `messages.create` */
|
|
2839
|
-
interface AnthropicClient {
|
|
2840
|
-
messages: {
|
|
2841
|
-
create(params: {
|
|
2842
|
-
model: string;
|
|
2843
|
-
max_tokens: number;
|
|
2844
|
-
temperature?: number;
|
|
2845
|
-
system?: string;
|
|
2846
|
-
messages: Array<{
|
|
2847
|
-
role: 'user' | 'assistant';
|
|
2848
|
-
content: string | AnthropicContentBlock[];
|
|
2849
|
-
}>;
|
|
2850
|
-
tools?: Array<{
|
|
2851
|
-
name: string;
|
|
2852
|
-
description?: string;
|
|
2853
|
-
input_schema: {
|
|
2854
|
-
type: 'object';
|
|
2855
|
-
[key: string]: unknown;
|
|
2856
|
-
};
|
|
2857
|
-
strict?: boolean;
|
|
2858
|
-
}>;
|
|
2859
|
-
tool_choice?: {
|
|
2860
|
-
type: 'auto';
|
|
2861
|
-
} | {
|
|
2862
|
-
type: 'any';
|
|
2863
|
-
} | {
|
|
2864
|
-
type: 'none';
|
|
2865
|
-
} | {
|
|
2866
|
-
type: 'tool';
|
|
2867
|
-
name: string;
|
|
2868
|
-
};
|
|
2869
|
-
/**
|
|
2870
|
-
* Native, schema-constrained output: a separate request field from
|
|
2871
|
-
* `tools`/`tool_choice`, so it can be sent alongside real tool
|
|
2872
|
-
* calls. Only built by this adapter for models covered by
|
|
2873
|
-
* `nativeStructuredOutputModels` (opt-in, see
|
|
2874
|
-
* `AnthropicAdapterOptions`); other models keep getting
|
|
2875
|
-
* `jsonSchema` emulated as a forced single tool call, the
|
|
2876
|
-
* pre-existing behavior.
|
|
2877
|
-
*
|
|
2878
|
-
* Matches the real Anthropic API's `output_config.format` shape
|
|
2879
|
-
* exactly: just `type` and `schema`, no `name`/`description`/
|
|
2880
|
-
* `strict`. Those three exist on VernLLM's own `jsonSchema` API
|
|
2881
|
-
* (and are still forwarded on the legacy forced-tool-call path,
|
|
2882
|
-
* where they're real `Tool` fields), but the native structured-
|
|
2883
|
-
* output endpoint has no equivalent for any of them.
|
|
2884
|
-
*/
|
|
2885
|
-
output_config?: {
|
|
2886
|
-
format?: {
|
|
2887
|
-
type: 'json_schema';
|
|
2888
|
-
schema: Record<string, unknown>;
|
|
2889
|
-
};
|
|
2890
|
-
/**
|
|
2891
|
-
* Effort control for adaptive thinking, on models where manual
|
|
2892
|
-
* `budget_tokens` thinking is no longer accepted (see
|
|
2893
|
-
* `supportsManualThinkingBudget` in
|
|
2894
|
-
* `adapters/internal/reasoningBudget.utils.ts`). Sibling to
|
|
2895
|
-
* `format`, either or both may be present independently.
|
|
2896
|
-
*/
|
|
2897
|
-
effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
2898
|
-
};
|
|
2899
|
-
/**
|
|
2900
|
-
* Native reasoning control. `{ type: 'enabled', budget_tokens }`
|
|
2901
|
-
* is built directly from `CallParams.budgetTokens`, or converted
|
|
2902
|
-
* from `reasoningEffort`, on models that still accept a manual
|
|
2903
|
-
* token budget. `{ type: 'adaptive' }` is sent instead, paired
|
|
2904
|
-
* with `output_config.effort`, on models that only support
|
|
2905
|
-
* adaptive thinking. See
|
|
2906
|
-
* `adapters/internal/reasoningBudget.utils.ts`.
|
|
2907
|
-
*/
|
|
2908
|
-
thinking?: {
|
|
2909
|
-
type: 'enabled';
|
|
2910
|
-
budget_tokens: number;
|
|
2911
|
-
} | {
|
|
2912
|
-
type: 'adaptive';
|
|
2913
|
-
};
|
|
2914
|
-
}, options: {
|
|
2915
|
-
signal: AbortSignal;
|
|
2916
|
-
}): Promise<{
|
|
2917
|
-
content: Array<{
|
|
2918
|
-
type: string;
|
|
2919
|
-
text?: string;
|
|
2920
|
-
id?: string;
|
|
2921
|
-
name?: string;
|
|
2922
|
-
input?: unknown;
|
|
2923
|
-
}>;
|
|
2924
|
-
usage?: {
|
|
2925
|
-
input_tokens?: number;
|
|
2926
|
-
output_tokens?: number;
|
|
2927
|
-
output_tokens_details?: {
|
|
2928
|
-
thinking_tokens?: number;
|
|
2929
|
-
} | null;
|
|
2930
|
-
};
|
|
2931
|
-
}>;
|
|
2932
|
-
};
|
|
2933
|
-
}
|
|
2934
|
-
/** Optional configuration for `fromAnthropic`. */
|
|
2935
|
-
interface AnthropicAdapterOptions {
|
|
2936
|
-
/**
|
|
2937
|
-
* Which models support native, schema-constrained output
|
|
2938
|
-
* (`output_config.format`), independent of `tools`/`tool_choice`, so it
|
|
2939
|
-
* can be combined with real `tools` in one request. Pass a static list
|
|
2940
|
-
* of model IDs (verified against Anthropic's own docs) or a predicate.
|
|
2941
|
-
*
|
|
2942
|
-
* There is no built-in default here (see `supportsNativeStructuredOutput`
|
|
2943
|
-
* for why). Left unset, every model uses the older forced-single-tool-
|
|
2944
|
-
* call emulation, and `tools` + `jsonSchema` together is rejected,
|
|
2945
|
-
* exactly this adapter's behavior before native support was added.
|
|
2946
|
-
*/
|
|
2947
|
-
nativeStructuredOutputModels?: ModelCapabilityOverride;
|
|
2948
|
-
/**
|
|
2949
|
-
* Overrides the token count `reasoningEffort` tiers map onto when the
|
|
2950
|
-
* caller sets `reasoningEffort` but not `budgetTokens` (Claude has no
|
|
2951
|
-
* tier concept of its own, see `adapters/internal/reasoningBudget.utils.ts`).
|
|
2952
|
-
* Only the tiers listed are changed; any omitted tier keeps the
|
|
2953
|
-
* built-in default. Has no effect when `budgetTokens` is set directly.
|
|
2954
|
-
*/
|
|
2955
|
-
reasoningEffortTokens?: Partial<EffortTokenTable>;
|
|
2956
|
-
/**
|
|
2957
|
-
* Marks additional models as adaptive-only, on top of this package's
|
|
2958
|
-
* own built-in rule (Claude Opus 4.7 and later, every Claude 5 tier
|
|
2959
|
-
* model, see `isAdaptiveOnlyModel` in
|
|
2960
|
-
* `adapters/internal/reasoningBudget.utils.ts`). Additive, not a
|
|
2961
|
-
* replacement: it can correct a false negative (a newer model this
|
|
2962
|
-
* package doesn't know about yet), it can't un-mark a model the
|
|
2963
|
-
* built-in rule already caught. Pass a static list of model IDs or a
|
|
2964
|
-
* predicate.
|
|
2965
|
-
*/
|
|
2966
|
-
adaptiveOnlyModels?: ModelCapabilityOverride;
|
|
2967
|
-
/**
|
|
2968
|
-
* Whether the client's `messages.create` supports `.withResponse()`
|
|
2969
|
-
* (needed for AIMD's proactive path). Default `false`, since
|
|
2970
|
-
* `AnthropicClient` is structural and a test fake or thin wrapper
|
|
2971
|
-
* won't implement it.
|
|
2972
|
-
*/
|
|
2973
|
-
supportsWithResponse?: boolean;
|
|
2974
|
-
}
|
|
2975
|
-
/**
|
|
2976
|
-
* Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
|
|
2977
|
-
* interface VernLLM uses for OpenAI/Groq.
|
|
2978
|
-
*
|
|
2979
|
-
* `response_format: json_schema`, on a model covered by
|
|
2980
|
-
* `options.nativeStructuredOutputModels`, is sent as `output_config.format`,
|
|
2981
|
-
* its own request field, independent of `tools`/`tool_choice`, so it can be
|
|
2982
|
-
* combined with real, caller-supplied `tools` in the same request. Only
|
|
2983
|
-
* `type` and `schema` are sent on this path, the real Anthropic API's
|
|
2984
|
-
* `output_config.format` has no `name`/`description`/`strict` fields.
|
|
2985
|
-
*
|
|
2986
|
-
* On any other model (the default, since `nativeStructuredOutputModels` is
|
|
2987
|
-
* opt-in), `response_format: json_schema` is mapped to Anthropic's forced
|
|
2988
|
-
* tool-use instead: a single tool is defined with `input_schema` set to
|
|
2989
|
-
* the caller's schema, `description` forwarded when provided, and `strict`
|
|
2990
|
-
* forwarded when set, and `tool_choice` forces the model to call it. This
|
|
2991
|
-
* legacy path cannot be combined with real `tools` (both would need the
|
|
2992
|
-
* same `tools`/`tool_choice` field), and a call that tries throws
|
|
2993
|
-
* `LLMError('invalid_params')` with `code: 'unsupported_capability'` and
|
|
2994
|
-
* `issues: { capability: 'tools_with_json_schema' }` before reaching the API. Provider-constrained
|
|
2995
|
-
* schema matching applies only when `strict: true` is forwarded and
|
|
2996
|
-
* supported.
|
|
2997
|
-
*
|
|
2998
|
-
* `response_format: json_object` throws `LLMError('validation')`. Anthropic
|
|
2999
|
-
* has no API-level field that mechanically guarantees JSON output the way
|
|
3000
|
-
* OpenAI's `json_object` mode does; the only way to emulate it was a
|
|
3001
|
-
* system-prompt instruction with no actual enforcement behind it, a
|
|
3002
|
-
* guarantee this adapter no longer pretends to make. Use `jsonSchema`
|
|
3003
|
-
* instead, which maps to a real constraint either way (native
|
|
3004
|
-
* `output_config.format` or a forced tool call).
|
|
3005
|
-
*/
|
|
3006
|
-
export declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
|
|
3007
|
-
//#endregion
|
|
3008
|
-
//#region src/adapters/gemini.d.ts
|
|
3009
|
-
/**
|
|
3010
|
-
* Gemini's native per-part content shape for a `contents` entry.
|
|
3011
|
-
* `functionCall.args` and `functionResponse.response` are typed as
|
|
3012
|
-
* `Record<string, unknown>` (not `unknown`) to match the real SDK's
|
|
3013
|
-
* `FunctionCall.args` / `FunctionResponse.response`, see the doc comment
|
|
3014
|
-
* on {@link GeminiClient}.
|
|
3015
|
-
*/
|
|
3016
|
-
type GeminiPart = {
|
|
3017
|
-
text: string;
|
|
3018
|
-
} | {
|
|
3019
|
-
inlineData: {
|
|
3020
|
-
mimeType: string;
|
|
3021
|
-
data: string;
|
|
3022
|
-
};
|
|
3023
|
-
} | {
|
|
3024
|
-
functionCall: {
|
|
3025
|
-
id?: string;
|
|
3026
|
-
name: string;
|
|
3027
|
-
args: Record<string, unknown>;
|
|
3028
|
-
};
|
|
3029
|
-
} | {
|
|
3030
|
-
functionResponse: {
|
|
3031
|
-
id?: string;
|
|
3032
|
-
name: string;
|
|
3033
|
-
response: Record<string, unknown>;
|
|
3034
|
-
};
|
|
3035
|
-
};
|
|
3036
|
-
/**
|
|
3037
|
-
* Structural type matching the real `@google/genai` SDK, in either shape
|
|
3038
|
-
* it's commonly held in: the callable model methods directly (`ai.models`),
|
|
3039
|
-
* or the complete top-level client (`ai`, via the optional `models` field
|
|
3040
|
-
* below). Both work with `fromGemini` directly, with no cast:
|
|
3041
|
-
*
|
|
3042
|
-
* ```ts
|
|
3043
|
-
* import { GoogleGenAI } from '@google/genai';
|
|
3044
|
-
* const ai = new GoogleGenAI({ apiKey: '...' });
|
|
3045
|
-
* const llm = new VernLLM({ client: fromGemini(ai), model: 'gemini-2.5-flash' });
|
|
3046
|
-
* ```
|
|
3047
|
-
*
|
|
3048
|
-
* `generateContent` is optional so a `{ models: ... }`-shaped value is
|
|
3049
|
-
* still a structural `GeminiClient`; `fromGemini` resolves `models` at
|
|
3050
|
-
* runtime and throws if nothing callable results.
|
|
3051
|
-
*
|
|
3052
|
-
* Every field is shaped to be structurally assignable from the real SDK's
|
|
3053
|
-
* generated types without importing them, so provider SDKs stay optional:
|
|
3054
|
-
* `model` is required (the real SDK requires it), `functionCall.args` /
|
|
3055
|
-
* `functionResponse.response` are `Record<string, unknown>` (matching the
|
|
3056
|
-
* real SDK, not `unknown`), `toolConfig...mode` is `any` (TypeScript never
|
|
3057
|
-
* treats a string-literal union as assignable to the real SDK's string
|
|
3058
|
-
* enum), and response-side `functionCall.name` is optional (matching the
|
|
3059
|
-
* real SDK).
|
|
3060
|
-
*/
|
|
3061
|
-
interface GeminiClient {
|
|
3062
|
-
/** Present when this is the whole top-level SDK client, not `ai.models`. `fromGemini` unwraps it at runtime. */
|
|
3063
|
-
models?: GeminiClient;
|
|
3064
|
-
generateContent?(params: {
|
|
3065
|
-
model: string;
|
|
3066
|
-
contents: Array<{
|
|
3067
|
-
role: 'user' | 'model';
|
|
3068
|
-
parts: GeminiPart[];
|
|
3069
|
-
}>;
|
|
3070
|
-
config?: {
|
|
3071
|
-
systemInstruction?: {
|
|
3072
|
-
parts: Array<{
|
|
3073
|
-
text: string;
|
|
3074
|
-
}>;
|
|
3075
|
-
};
|
|
3076
|
-
temperature?: number;
|
|
3077
|
-
maxOutputTokens?: number;
|
|
3078
|
-
responseMimeType?: string;
|
|
3079
|
-
responseSchema?: Record<string, unknown>;
|
|
3080
|
-
tools?: Array<{
|
|
3081
|
-
functionDeclarations: Array<{
|
|
3082
|
-
name: string;
|
|
3083
|
-
description?: string;
|
|
3084
|
-
parameters: Record<string, unknown>;
|
|
3085
|
-
}>;
|
|
3086
|
-
}>;
|
|
3087
|
-
toolConfig?: {
|
|
3088
|
-
functionCallingConfig: {
|
|
3089
|
-
mode: any;
|
|
3090
|
-
allowedFunctionNames?: string[];
|
|
3091
|
-
};
|
|
3092
|
-
};
|
|
3093
|
-
/**
|
|
3094
|
-
* Native reasoning control. `thinkingBudget` is built from
|
|
3095
|
-
* `CallParams.budgetTokens` directly when set (0 disables thinking,
|
|
3096
|
-
* -1 requests automatic budgeting, both passed through unchanged),
|
|
3097
|
-
* or converted from `reasoningEffort`, on Gemini 2.5 and earlier
|
|
3098
|
-
* models. `thinkingLevel` is used instead on Gemini 3 and later,
|
|
3099
|
-
* which use a level-based control rather than a numeric budget.
|
|
3100
|
-
* `any`, same reason as `toolConfig...mode` above, see class doc
|
|
3101
|
-
* comment. See `usesGeminiThinkingLevel` in
|
|
3102
|
-
* `adapters/internal/reasoningBudget.utils.ts`.
|
|
3103
|
-
*/
|
|
3104
|
-
thinkingConfig?: {
|
|
3105
|
-
thinkingBudget?: number;
|
|
3106
|
-
thinkingLevel?: any;
|
|
3107
|
-
};
|
|
3108
|
-
abortSignal?: AbortSignal;
|
|
3109
|
-
};
|
|
3110
|
-
}): Promise<{
|
|
3111
|
-
candidates?: Array<{
|
|
3112
|
-
content?: {
|
|
3113
|
-
parts?: Array<{
|
|
3114
|
-
text?: string;
|
|
3115
|
-
functionCall?: {
|
|
3116
|
-
id?: string;
|
|
3117
|
-
name?: string;
|
|
3118
|
-
args?: unknown;
|
|
3119
|
-
};
|
|
3120
|
-
}>;
|
|
3121
|
-
};
|
|
3122
|
-
}>;
|
|
3123
|
-
usageMetadata?: {
|
|
3124
|
-
promptTokenCount?: number;
|
|
3125
|
-
candidatesTokenCount?: number;
|
|
3126
|
-
totalTokenCount?: number;
|
|
3127
|
-
thoughtsTokenCount?: number;
|
|
3128
|
-
};
|
|
3129
|
-
}>;
|
|
3130
|
-
/**
|
|
3131
|
-
* Optional. Required only for `stream: true` calls. Takes the same
|
|
3132
|
-
* request shape as `generateContent`. Matching the real SDK's own
|
|
3133
|
-
* `generateContentStream`, this resolves to an `AsyncIterable` (rather
|
|
3134
|
-
* than returning one synchronously) of partial responses, each chunk
|
|
3135
|
-
* holding the same `candidates[].content.parts[]` structure as
|
|
3136
|
-
* `generateContent`'s response, just incremental.
|
|
3137
|
-
*/
|
|
3138
|
-
generateContentStream?(params: Parameters<NonNullable<GeminiClient['generateContent']>>[0]): Promise<AsyncIterable<{
|
|
3139
|
-
candidates?: Array<{
|
|
3140
|
-
content?: {
|
|
3141
|
-
parts?: Array<{
|
|
3142
|
-
text?: string;
|
|
3143
|
-
functionCall?: {
|
|
3144
|
-
id?: string;
|
|
3145
|
-
name?: string;
|
|
3146
|
-
args?: unknown;
|
|
3147
|
-
};
|
|
3148
|
-
}>;
|
|
3149
|
-
};
|
|
3150
|
-
}>;
|
|
3151
|
-
usageMetadata?: {
|
|
3152
|
-
promptTokenCount?: number;
|
|
3153
|
-
candidatesTokenCount?: number;
|
|
3154
|
-
totalTokenCount?: number;
|
|
3155
|
-
thoughtsTokenCount?: number;
|
|
3156
|
-
};
|
|
3157
|
-
}>>;
|
|
3158
|
-
}
|
|
3159
|
-
/**
|
|
3160
|
-
* Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
|
|
3161
|
-
* uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
|
|
3162
|
-
* axis: a `contents` array instead of `messages`, a separate
|
|
3163
|
-
* `systemInstruction` field instead of a `system` role message,
|
|
3164
|
-
* `generationConfig` instead of top-level `temperature`/`max_tokens`, and
|
|
3165
|
-
* native JSON Schema support via `responseMimeType: 'application/json'` +
|
|
3166
|
-
* `responseSchema`. `reasoning_effort` has no native Gemini equivalent, so
|
|
3167
|
-
* it's converted to a `thinkingConfig.thinkingBudget` token count; `budget_tokens`
|
|
3168
|
-
* maps to `thinkingBudget` directly, Gemini's native reasoning control. See
|
|
3169
|
-
* `adapters/internal/reasoningBudget.utils.ts`.
|
|
3170
|
-
*
|
|
3171
|
-
* `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
|
|
3172
|
-
* `tool_choice` maps to `toolConfig.functionCallingConfig`. Gemini accepts
|
|
3173
|
-
* `responseSchema` and `tools` in the same request natively, so both are
|
|
3174
|
-
* set independently here and no special-casing is needed for the
|
|
3175
|
-
* combination, unlike `fromAnthropic`/`fromBedrock`.
|
|
3176
|
-
*
|
|
3177
|
-
* `createStream` calls `generateContentStream` (optional on `GeminiClient`
|
|
3178
|
-
*, required only if the caller sets `stream: true`) and translates each
|
|
3179
|
-
* partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
|
|
3180
|
-
* Gemini's own function-calling API doesn't stream tool-call arguments
|
|
3181
|
-
* incrementally: a `functionCall` part always arrives whole in one chunk,
|
|
3182
|
-
* so each one is emitted as a single, complete `tool_call_delta` (a
|
|
3183
|
-
* one-shot "delta" containing the full arguments) rather than accumulated
|
|
3184
|
-
* fragments, that's a real difference in the underlying API, not
|
|
3185
|
-
* something this adapter can smooth over. `usageMetadata` is (per Gemini's
|
|
3186
|
-
* own behavior) only reliably present on the last chunk, so the `usage`
|
|
3187
|
-
* `WireStreamChunk` is emitted once, after the stream completes, from
|
|
3188
|
-
* whichever chunk's `usageMetadata` was seen last.
|
|
3189
|
-
*
|
|
3190
|
-
* Accepts a `GeminiClient` in either shape it structurally covers: the
|
|
3191
|
-
* callable model methods directly (`ai.models`), or the complete
|
|
3192
|
-
* top-level client (`ai`), unwrapping `.models` internally when present.
|
|
3193
|
-
* Both work with no cast: `fromGemini(ai.models)` and `fromGemini(ai)`.
|
|
3194
|
-
* Throws `LLMError('invalid_params')` up front if nothing callable
|
|
3195
|
-
* results.
|
|
3196
|
-
*/
|
|
3197
|
-
interface GeminiAdapterOptions {
|
|
3198
|
-
/**
|
|
3199
|
-
* Overrides the token count `reasoningEffort` tiers map onto when the
|
|
3200
|
-
* caller sets `reasoningEffort` but not `budgetTokens` (Gemini has no
|
|
3201
|
-
* tier string of its own, see `adapters/internal/reasoningBudget.utils.ts`).
|
|
3202
|
-
* Only the tiers listed are changed; any omitted tier keeps the
|
|
3203
|
-
* built-in default. Has no effect when `budgetTokens` is set directly.
|
|
3204
|
-
*/
|
|
3205
|
-
reasoningEffortTokens?: Partial<EffortTokenTable>;
|
|
3206
|
-
/**
|
|
3207
|
-
* Marks additional models as using `thinkingLevel` instead of
|
|
3208
|
-
* `thinkingBudget`, on top of this package's own built-in rule (every
|
|
3209
|
-
* Gemini 3 series model and later, see `usesGeminiThinkingLevel` in
|
|
3210
|
-
* `adapters/internal/reasoningBudget.utils.ts`). Additive, not a
|
|
3211
|
-
* replacement: it can correct a false negative (a newer model this
|
|
3212
|
-
* package doesn't know about yet), it can't un-mark a model the
|
|
3213
|
-
* built-in rule already caught. Pass a static list of model IDs or a
|
|
3214
|
-
* predicate.
|
|
3215
|
-
*/
|
|
3216
|
-
thinkingLevelModels?: ModelCapabilityOverride;
|
|
3217
|
-
}
|
|
3218
|
-
export declare function fromGemini(client: GeminiClient, options?: GeminiAdapterOptions): LLMClient;
|
|
3219
|
-
//#endregion
|
|
3220
|
-
//#region src/adapters/bedrock.d.ts
|
|
3221
|
-
/** Bedrock Converse's supported inline image formats. */
|
|
3222
|
-
type BedrockImageFormat = 'png' | 'jpeg' | 'gif' | 'webp';
|
|
3223
|
-
/** Bedrock Converse's native per-block content shape for a message. */
|
|
3224
|
-
type BedrockContentBlock = {
|
|
3225
|
-
text: string;
|
|
3226
|
-
} | {
|
|
3227
|
-
image: {
|
|
3228
|
-
format: BedrockImageFormat;
|
|
3229
|
-
source: {
|
|
3230
|
-
bytes: Uint8Array;
|
|
3231
|
-
};
|
|
3232
|
-
};
|
|
3233
|
-
} | {
|
|
3234
|
-
toolUse: {
|
|
3235
|
-
toolUseId: string;
|
|
3236
|
-
name: string;
|
|
3237
|
-
input: unknown;
|
|
3238
|
-
};
|
|
3239
|
-
} | {
|
|
3240
|
-
toolResult: {
|
|
3241
|
-
toolUseId: string;
|
|
3242
|
-
content: Array<{
|
|
3243
|
-
text: string;
|
|
3244
|
-
}>;
|
|
3245
|
-
status?: 'success' | 'error';
|
|
3246
|
-
};
|
|
3247
|
-
};
|
|
3248
|
-
/**
|
|
3249
|
-
* Minimal structural type matching AWS Bedrock's Converse API. This is
|
|
3250
|
-
* intentionally NOT `BedrockRuntimeClient` itself, the AWS SDK v3 client
|
|
3251
|
-
* exposes `.send(command)`, not a direct `.converse()` method, and pulling
|
|
3252
|
-
* in `@aws-sdk/client-bedrock-runtime` as a dependency just for its types
|
|
3253
|
-
* isn't worth it for a structural adapter. Wrap your client, e.g:
|
|
3254
|
-
*
|
|
3255
|
-
* ```ts
|
|
3256
|
-
* import { BedrockRuntimeClient, ConverseCommand } from '@aws-sdk/client-bedrock-runtime';
|
|
3257
|
-
* const client = new BedrockRuntimeClient({ region: 'us-east-1' });
|
|
3258
|
-
* const converseClient = {
|
|
3259
|
-
* converse: (params, options) =>
|
|
3260
|
-
* client.send(new ConverseCommand(params), { abortSignal: options.signal }),
|
|
3261
|
-
* };
|
|
3262
|
-
* ```
|
|
3263
|
-
*/
|
|
3264
|
-
interface BedrockConverseClient {
|
|
3265
|
-
converse(params: {
|
|
3266
|
-
modelId: string;
|
|
3267
|
-
messages: Array<{
|
|
3268
|
-
role: 'user' | 'assistant';
|
|
3269
|
-
content: BedrockContentBlock[];
|
|
3270
|
-
}>;
|
|
3271
|
-
system?: Array<{
|
|
3272
|
-
text: string;
|
|
3273
|
-
}>;
|
|
3274
|
-
inferenceConfig?: {
|
|
3275
|
-
temperature?: number;
|
|
3276
|
-
maxTokens?: number;
|
|
3277
|
-
};
|
|
3278
|
-
toolConfig?: {
|
|
3279
|
-
tools: Array<{
|
|
3280
|
-
toolSpec: {
|
|
3281
|
-
name: string;
|
|
3282
|
-
description?: string;
|
|
3283
|
-
inputSchema: {
|
|
3284
|
-
json: Record<string, unknown>;
|
|
3285
|
-
};
|
|
3286
|
-
strict?: boolean;
|
|
3287
|
-
};
|
|
3288
|
-
}>;
|
|
3289
|
-
toolChoice?: {
|
|
3290
|
-
tool: {
|
|
3291
|
-
name: string;
|
|
3292
|
-
};
|
|
3293
|
-
} | {
|
|
3294
|
-
auto: Record<string, never>;
|
|
3295
|
-
} | {
|
|
3296
|
-
any: Record<string, never>;
|
|
3297
|
-
};
|
|
3298
|
-
};
|
|
3299
|
-
/**
|
|
3300
|
-
* Native, schema-constrained output: a separate request field from
|
|
3301
|
-
* `toolConfig`, so it can be sent alongside real tool calls. Only
|
|
3302
|
-
* built by this adapter for models covered by
|
|
3303
|
-
* `nativeStructuredOutputModels` (opt-in, see
|
|
3304
|
-
* `BedrockAdapterOptions`); other models keep getting `jsonSchema`
|
|
3305
|
-
* emulated as a forced single tool call via `toolConfig`, the
|
|
3306
|
-
* pre-existing behavior.
|
|
3307
|
-
*
|
|
3308
|
-
* Matches the real Bedrock Converse API's `outputConfig.textFormat`
|
|
3309
|
-
* shape exactly: the schema itself is nested one level deeper, under
|
|
3310
|
-
* `structure.jsonSchema`, not flat on `textFormat`, and `schema` is
|
|
3311
|
-
* a JSON-encoded *string*, not a parsed object, unlike every other
|
|
3312
|
-
* schema field this adapter builds (`toolSpec.inputSchema.json`
|
|
3313
|
-
* included). There is no `strict` field here, unlike `toolSpec`.
|
|
3314
|
-
*/
|
|
3315
|
-
outputConfig?: {
|
|
3316
|
-
textFormat?: {
|
|
3317
|
-
type: 'json_schema';
|
|
3318
|
-
structure: {
|
|
3319
|
-
jsonSchema: {
|
|
3320
|
-
schema: string;
|
|
3321
|
-
name?: string;
|
|
3322
|
-
description?: string;
|
|
3323
|
-
};
|
|
3324
|
-
};
|
|
3325
|
-
};
|
|
3326
|
-
/**
|
|
3327
|
-
* Effort control for adaptive thinking, on Claude models where
|
|
3328
|
-
* manual `budget_tokens` thinking is no longer accepted (see
|
|
3329
|
-
* `supportsManualThinkingBudget` in
|
|
3330
|
-
* `adapters/internal/reasoningBudget.utils.ts`). Sibling to
|
|
3331
|
-
* `textFormat`, either or both may be present independently.
|
|
3332
|
-
*/
|
|
3333
|
-
effort?: 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
3334
|
-
};
|
|
3335
|
-
/**
|
|
3336
|
-
* Model-specific passthrough. Converse has no reasoning-budget field
|
|
3337
|
-
* of its own, so a token budget for a Claude model on Bedrock is
|
|
3338
|
-
* forwarded here under Anthropic's own key, `{ thinking: { type:
|
|
3339
|
-
* 'enabled', budget_tokens } }`. Non-Claude models get nothing here,
|
|
3340
|
-
* there is no equivalent field to reach for.
|
|
3341
|
-
*/
|
|
3342
|
-
additionalModelRequestFields?: Record<string, unknown>;
|
|
3343
|
-
}, options: {
|
|
3344
|
-
signal: AbortSignal;
|
|
3345
|
-
}): Promise<{
|
|
3346
|
-
output?: {
|
|
3347
|
-
message?: {
|
|
3348
|
-
content?: Array<{
|
|
3349
|
-
text?: string;
|
|
3350
|
-
toolUse?: {
|
|
3351
|
-
toolUseId?: string;
|
|
3352
|
-
name?: string;
|
|
3353
|
-
input?: unknown;
|
|
3354
|
-
};
|
|
3355
|
-
}>;
|
|
3356
|
-
};
|
|
3357
|
-
};
|
|
3358
|
-
usage?: {
|
|
3359
|
-
inputTokens?: number;
|
|
3360
|
-
outputTokens?: number;
|
|
3361
|
-
totalTokens?: number;
|
|
3362
|
-
};
|
|
3363
|
-
}>;
|
|
3364
|
-
/**
|
|
3365
|
-
* Optional. Required only for `stream: true` calls. Takes the same
|
|
3366
|
-
* request shape `converse` does, returning `{ stream }`, matching
|
|
3367
|
-
* `ConverseStreamCommand`'s real AWS SDK v3 output shape, an
|
|
3368
|
-
* `AsyncIterable` of incremental events under a `stream` property,
|
|
3369
|
-
* rather than the whole response being the iterable directly.
|
|
3370
|
-
*/
|
|
3371
|
-
converseStream?(params: Parameters<BedrockConverseClient['converse']>[0], options: {
|
|
3372
|
-
signal: AbortSignal;
|
|
3373
|
-
}): Promise<{
|
|
3374
|
-
stream: AsyncIterable<BedrockConverseStreamEvent>;
|
|
3375
|
-
}>;
|
|
3376
|
-
}
|
|
3377
|
-
/**
|
|
3378
|
-
* One event of a Bedrock `ConverseStreamCommand` response's `stream`.
|
|
3379
|
-
* Content blocks (text or toolUse) are identified by `contentBlockIndex`,
|
|
3380
|
-
* Converse's own convention for correlating start/delta/stop events across
|
|
3381
|
-
* possibly-interleaved blocks, mirrored directly by VernLLM's
|
|
3382
|
-
* `tool_call_delta.index`.
|
|
3383
|
-
*/
|
|
3384
|
-
type BedrockConverseStreamEvent = {
|
|
3385
|
-
messageStart: {
|
|
3386
|
-
role: 'assistant';
|
|
3387
|
-
};
|
|
3388
|
-
} | {
|
|
3389
|
-
contentBlockStart: {
|
|
3390
|
-
contentBlockIndex: number;
|
|
3391
|
-
start?: {
|
|
3392
|
-
toolUse?: {
|
|
3393
|
-
toolUseId?: string;
|
|
3394
|
-
name?: string;
|
|
3395
|
-
};
|
|
3396
|
-
};
|
|
3397
|
-
};
|
|
3398
|
-
} | {
|
|
3399
|
-
contentBlockDelta: {
|
|
3400
|
-
contentBlockIndex: number;
|
|
3401
|
-
delta?: {
|
|
3402
|
-
text?: string;
|
|
3403
|
-
} | {
|
|
3404
|
-
toolUse?: {
|
|
3405
|
-
input?: string;
|
|
3406
|
-
};
|
|
3407
|
-
};
|
|
3408
|
-
};
|
|
3409
|
-
} | {
|
|
3410
|
-
contentBlockStop: {
|
|
3411
|
-
contentBlockIndex: number;
|
|
3412
|
-
};
|
|
3413
|
-
} | {
|
|
3414
|
-
messageStop: {
|
|
3415
|
-
stopReason?: string;
|
|
3416
|
-
};
|
|
3417
|
-
} | {
|
|
3418
|
-
metadata: {
|
|
3419
|
-
usage?: {
|
|
3420
|
-
inputTokens?: number;
|
|
3421
|
-
outputTokens?: number;
|
|
3422
|
-
totalTokens?: number;
|
|
3423
|
-
};
|
|
3424
|
-
};
|
|
3425
|
-
} | {
|
|
3426
|
-
internalServerException: {
|
|
3427
|
-
message?: string;
|
|
3428
|
-
};
|
|
3429
|
-
} | {
|
|
3430
|
-
modelStreamErrorException: {
|
|
3431
|
-
message?: string;
|
|
3432
|
-
originalStatusCode?: number;
|
|
3433
|
-
};
|
|
3434
|
-
} | {
|
|
3435
|
-
validationException: {
|
|
3436
|
-
message?: string;
|
|
3437
|
-
};
|
|
3438
|
-
} | {
|
|
3439
|
-
throttlingException: {
|
|
3440
|
-
message?: string;
|
|
3441
|
-
};
|
|
3442
|
-
} | {
|
|
3443
|
-
serviceUnavailableException: {
|
|
3444
|
-
message?: string;
|
|
3445
|
-
};
|
|
3446
|
-
};
|
|
3447
|
-
/**
|
|
3448
|
-
* Optional configuration for `fromBedrock`.
|
|
3449
|
-
*/
|
|
3450
|
-
interface BedrockAdapterOptions {
|
|
3451
|
-
/**
|
|
3452
|
-
* Optional preflight check for tool-use support, needed whenever a
|
|
3453
|
-
* `jsonSchema` call ends up sending Converse `toolConfig`, either the
|
|
3454
|
-
* legacy forced-single-tool-call emulation, or real `tools` sent
|
|
3455
|
-
* alongside native structured output (`outputConfig`). VernLLM never
|
|
3456
|
-
* guesses capability from a failed call's error message (AWS's error
|
|
3457
|
-
* text isn't a documented, stable contract), so this is opt-in: pass
|
|
3458
|
-
* either a static list of tool-use-capable model IDs, or a predicate
|
|
3459
|
-
* function, and VernLLM will reject unsupported models with a clear
|
|
3460
|
-
* `LLMError('validation')` *before* dispatching the request, instead of
|
|
3461
|
-
* on the wire.
|
|
3462
|
-
*
|
|
3463
|
-
* Left unset (default), no preflight check runs, and a `jsonSchema` call
|
|
3464
|
-
* to an unsupported model surfaces Bedrock's raw `converse` error as-is.
|
|
3465
|
-
*/
|
|
3466
|
-
toolUseSupportedModels?: string[] | ((modelId: string) => boolean);
|
|
3467
|
-
/**
|
|
3468
|
-
* Which models support native, schema-constrained output
|
|
3469
|
-
* (`outputConfig.textFormat`), independent of `toolConfig`, so it can be
|
|
3470
|
-
* combined with real `tools` in one request. Pass a static list of
|
|
3471
|
-
* model IDs (verified against Bedrock's own docs) or a predicate.
|
|
3472
|
-
*
|
|
3473
|
-
* There is no built-in default here (see `supportsNativeStructuredOutput`
|
|
3474
|
-
* for why). Left unset, every model uses the older forced-single-tool-
|
|
3475
|
-
* call emulation via `toolConfig`, and `tools` + `jsonSchema` together is
|
|
3476
|
-
* rejected, exactly this adapter's behavior before native support was
|
|
3477
|
-
* added.
|
|
3478
|
-
*/
|
|
3479
|
-
nativeStructuredOutputModels?: ModelCapabilityOverride;
|
|
3480
|
-
/**
|
|
3481
|
-
* Overrides the token count `reasoningEffort` tiers map onto when the
|
|
3482
|
-
* caller sets `reasoningEffort` but not `budgetTokens` (Converse has no
|
|
3483
|
-
* tier string of its own, see `adapters/internal/reasoningBudget.utils.ts`).
|
|
3484
|
-
* Only the tiers listed are changed; any omitted tier keeps the
|
|
3485
|
-
* built-in default. Has no effect when `budgetTokens` is set directly,
|
|
3486
|
-
* or when the target model isn't a Claude model.
|
|
3487
|
-
*/
|
|
3488
|
-
reasoningEffortTokens?: Partial<EffortTokenTable>;
|
|
3489
|
-
/**
|
|
3490
|
-
* Marks additional models as adaptive-only, on top of this package's
|
|
3491
|
-
* own built-in rule (Claude Opus 4.7 and later, every Claude 5 tier
|
|
3492
|
-
* model, see `isAdaptiveOnlyModel` in
|
|
3493
|
-
* `adapters/internal/reasoningBudget.utils.ts`). Additive, not a
|
|
3494
|
-
* replacement: it can correct a false negative (a newer model this
|
|
3495
|
-
* package doesn't know about yet), it can't un-mark a model the
|
|
3496
|
-
* built-in rule already caught. Pass a static list of model IDs or a
|
|
3497
|
-
* predicate.
|
|
3498
|
-
*/
|
|
3499
|
-
adaptiveOnlyModels?: ModelCapabilityOverride;
|
|
3500
|
-
}
|
|
3501
|
-
/**
|
|
3502
|
-
* Minimal structural shape of an AWS SDK v3 client that exposes `.send()`,
|
|
3503
|
-
* matching `BedrockRuntimeClient` (and its abort-signal-aware call
|
|
3504
|
-
* convention). Avoids importing `@aws-sdk/client-bedrock-runtime` for the
|
|
423
|
+
* `cachedCall()` counterpart to `defineCallParams`, keeping the whole `{ cacheKey, ttl, call }`
|
|
3505
424
|
* type.
|
|
3506
425
|
*/
|
|
3507
|
-
|
|
3508
|
-
send(command: unknown, options?: {
|
|
3509
|
-
abortSignal?: AbortSignal;
|
|
3510
|
-
}): Promise<unknown>;
|
|
3511
|
-
}
|
|
3512
|
-
/**
|
|
3513
|
-
* Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
|
|
3514
|
-
* interface VernLLM uses for OpenAI/Groq. The Converse API is unified
|
|
3515
|
-
* across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
|
|
3516
|
-
* so unlike raw per-model Bedrock invocation, this one adapter works
|
|
3517
|
-
* regardless of which underlying model `modelId` points at, as long as
|
|
3518
|
-
* that model supports Converse (most current-generation ones do)
|
|
3519
|
-
*
|
|
3520
|
-
* `bedrockClient` accepts either a hand-written `BedrockConverseClient`
|
|
3521
|
-
* (a `.converse()`/`.converseStream()` wrapper you provide) or a real AWS
|
|
3522
|
-
* SDK v3 client (anything with `.send()`, matching `BedrockRuntimeClient`)
|
|
3523
|
-
* directly, detected structurally. Passing a raw AWS client skips the
|
|
3524
|
-
* hand-written wrapper entirely, internally doing what it would
|
|
3525
|
-
* (`send(new ConverseCommand(...))`, `send(new
|
|
3526
|
-
* ConverseStreamCommand(...))`). See `wrapAwsSendClient` for how that path
|
|
3527
|
-
* is implemented, including why `@aws-sdk/client-bedrock-runtime` stays
|
|
3528
|
-
* out of this package's dependencies either way.
|
|
3529
|
-
*
|
|
3530
|
-
* `response_format: json_schema`, on a model covered by
|
|
3531
|
-
* `options.nativeStructuredOutputModels` (opt-in, unset by default), is
|
|
3532
|
-
* sent as `outputConfig.textFormat`, its own request field, independent of
|
|
3533
|
-
* `toolConfig`, so it can be combined with real, caller-supplied `tools`
|
|
3534
|
-
* in the same request. Matches the real Converse API's shape exactly: the
|
|
3535
|
-
* schema is nested under `structure.jsonSchema` and JSON-encoded as a
|
|
3536
|
-
* string, not the parsed object `toolConfig`'s tool schemas use, and there
|
|
3537
|
-
* is no `strict` field on this path.
|
|
3538
|
-
*
|
|
3539
|
-
* On any other model (the default), `response_format: json_schema` is
|
|
3540
|
-
* mapped to Converse's `toolConfig` instead: a single tool is defined from
|
|
3541
|
-
* the schema, description, and strictness settings, and `toolChoice`
|
|
3542
|
-
* forces the model to call it. This legacy path cannot be combined with
|
|
3543
|
-
* real `tools` (both would need the same `toolConfig`), and a call that
|
|
3544
|
-
* tries throws `LLMError('invalid_params')` with `code: 'unsupported_capability'`
|
|
3545
|
-
* and `issues: { capability: 'tools_with_json_schema' }` before reaching the API.
|
|
3546
|
-
* Provider-constrained schema matching applies only when `strict: true` is
|
|
3547
|
-
* forwarded and supported. Native tool support varies by model family;
|
|
3548
|
-
* pass `toolUseSupportedModels` to preflight-check it (see
|
|
3549
|
-
* `BedrockAdapterOptions`), otherwise a `jsonSchema` call to an
|
|
3550
|
-
* unsupported model surfaces Bedrock's raw error unchanged.
|
|
3551
|
-
*
|
|
3552
|
-
* `response_format: json_object` throws `LLMError('validation')`: Converse
|
|
3553
|
-
* has no field that mechanically guarantees JSON output, and the only way
|
|
3554
|
-
* to emulate it was an unenforced system-prompt instruction, a guarantee
|
|
3555
|
-
* this adapter no longer pretends to make. Use `jsonSchema` instead.
|
|
3556
|
-
* `reasoning_effort` (no Converse equivalent) is converted to a token
|
|
3557
|
-
* budget and forwarded via `additionalModelRequestFields` for Claude
|
|
3558
|
-
* models only; `budget_tokens` is forwarded the same way directly. Both
|
|
3559
|
-
* are silently dropped for non-Claude models, which have no equivalent
|
|
3560
|
-
* field to reach for. See `adapters/internal/reasoningBudget.utils.ts`.
|
|
3561
|
-
*
|
|
3562
|
-
* `tools` alone maps to Converse's native `toolConfig`/`toolUse`/
|
|
3563
|
-
* `toolResult`; `tool_choice` maps to `toolConfig.toolChoice`.
|
|
3564
|
-
*
|
|
3565
|
-
* `createStream` calls `converseStream` (optional on `BedrockConverseClient`
|
|
3566
|
-
*, required only if the caller sets `stream: true`) and translates its
|
|
3567
|
-
* `contentBlockStart`/`contentBlockDelta`/`metadata` events into
|
|
3568
|
-
* `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
|
|
3569
|
-
* same as `fromAnthropic`'s block-index tracking (Converse's streaming
|
|
3570
|
-
* shape is structurally close to Anthropic's own, both being tool-use-aware
|
|
3571
|
-
* content-block streams), including the same `json-tool` unwrapping: a
|
|
3572
|
-
* `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
|
|
3573
|
-
* `text-delta`, not `tool_call_delta`, so the accumulated result lands in
|
|
3574
|
-
* `finalizeResponse`'s `content` path exactly like the non-streaming
|
|
3575
|
-
* `create` branch above unwraps it.
|
|
3576
|
-
*/
|
|
3577
|
-
export declare function fromBedrock(bedrockClient: BedrockConverseClient | AwsSendClient, options?: BedrockAdapterOptions): LLMClient;
|
|
3578
|
-
//#endregion
|
|
3579
|
-
//#region src/adapters/fetch.d.ts
|
|
3580
|
-
/** The chat-completion-shaped request VernLLM builds internally */
|
|
3581
|
-
type ChatRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
|
|
3582
|
-
/**
|
|
3583
|
-
* The minimal shape the fetch adapter needs from a response object.
|
|
3584
|
-
* Native `fetch`'s `Response` satisfies this, but so do wrappers around
|
|
3585
|
-
* `axios`, `node-fetch`, `undici`, etc, which makes `request` swappable
|
|
3586
|
-
* without forcing consumers to polyfill the full `Response` interface
|
|
3587
|
-
*/
|
|
3588
|
-
interface ResponseLike {
|
|
3589
|
-
ok: boolean;
|
|
3590
|
-
status: number;
|
|
3591
|
-
headers: {
|
|
3592
|
-
get(name: string): string | null;
|
|
3593
|
-
};
|
|
3594
|
-
text(): Promise<string>;
|
|
3595
|
-
json(): Promise<unknown>;
|
|
3596
|
-
}
|
|
3597
|
-
/** A fetch-compatible request function; defaults to native `fetch` */
|
|
3598
|
-
type RequestLike = (url: string, init: {
|
|
3599
|
-
method: string;
|
|
3600
|
-
headers: Record<string, string>;
|
|
3601
|
-
body?: string;
|
|
3602
|
-
signal?: AbortSignal;
|
|
3603
|
-
}) => Promise<ResponseLike>;
|
|
3604
|
-
/**
|
|
3605
|
-
* A streaming-capable request function. Unlike `RequestLike`, which returns
|
|
3606
|
-
* a fully-buffered `ResponseLike`, this resolves to an `AsyncIterable` of
|
|
3607
|
-
* progressively-arriving chunks, the common ground across transports:
|
|
3608
|
-
* native `fetch`'s `response.body` (wrapped to be iterable; see
|
|
3609
|
-
* `webStreamToAsyncIterable` below), axios's Node `Readable` in
|
|
3610
|
-
* `responseType: 'stream'` mode (already async-iterable, no wrapping
|
|
3611
|
-
* needed), `node-fetch`, `undici`, etc, all satisfy this with little or no
|
|
3612
|
-
* glue code. Defaults to native `fetch`.
|
|
3613
|
-
*/
|
|
3614
|
-
type StreamRequestLike = (url: string, init: {
|
|
3615
|
-
method: string;
|
|
3616
|
-
headers: Record<string, string>;
|
|
3617
|
-
body?: string;
|
|
3618
|
-
signal?: AbortSignal;
|
|
3619
|
-
}) => Promise<AsyncIterable<Uint8Array | string>>;
|
|
3620
|
-
interface FetchAdapterConfig {
|
|
3621
|
-
/** Endpoint URL, or a function of the request in case it depends on model/params */
|
|
3622
|
-
url: string | ((params: ChatRequest) => string);
|
|
3623
|
-
/** Static headers, or a function (sync or async) for things like refreshed auth tokens */
|
|
3624
|
-
headers?: Record<string, string> | (() => Record<string, string> | Promise<Record<string, string>>);
|
|
3625
|
-
/** HTTP method. Default 'POST' */
|
|
3626
|
-
method?: string;
|
|
3627
|
-
/**
|
|
3628
|
-
* The function used to make the HTTP request. Defaults to native `fetch`.
|
|
3629
|
-
* Swap in `axios`, `node-fetch`, or any other transport, as long as it
|
|
3630
|
-
* resolves to a `ResponseLike` object
|
|
3631
|
-
*/
|
|
3632
|
-
request?: RequestLike;
|
|
3633
|
-
/** Maps VernLLMs internal chat-completion request into the providers raw request body */
|
|
3634
|
-
mapRequest: (params: ChatRequest) => unknown;
|
|
3635
|
-
/**
|
|
3636
|
-
* Maps the providers raw JSON response into `{ content, usage?, toolCalls? }`
|
|
3637
|
-
* `content` is the assistants text (JSON string when JSON mode was requested).
|
|
3638
|
-
* `content` may be empty/omitted when the model responded with only tool
|
|
3639
|
-
* calls and no text.
|
|
3640
|
-
*
|
|
3641
|
-
* `toolCalls`, when the model requested one or more tools, is the list of
|
|
3642
|
-
* calls as flat `{ id, name, arguments }` entries (matching this config's
|
|
3643
|
-
* own `toolCalls?: Array<{ id: string; name: string; arguments: string }>`
|
|
3644
|
-
* return type below), each entry's `arguments` already JSON-*encoded* as a
|
|
3645
|
-
* string (not the parsed object), mirroring the wire format every
|
|
3646
|
-
* OpenAI-compatible provider uses. `fromFetch` itself converts these into
|
|
3647
|
-
* `WireToolCall`'s `type`/`function`-wrapped shape before returning them
|
|
3648
|
-
* from `create`. VernLLM parses (and validates, if `argumentsSchema` was
|
|
3649
|
-
* set) the arguments string internally, mapResponse doesn't need to do
|
|
3650
|
-
* that itself.
|
|
3651
|
-
*/
|
|
3652
|
-
mapResponse: (json: unknown) => {
|
|
3653
|
-
content?: string;
|
|
3654
|
-
usage?: {
|
|
3655
|
-
promptTokens?: number;
|
|
3656
|
-
completionTokens?: number;
|
|
3657
|
-
totalTokens?: number;
|
|
3658
|
-
};
|
|
3659
|
-
toolCalls?: Array<{
|
|
3660
|
-
id: string;
|
|
3661
|
-
name: string;
|
|
3662
|
-
arguments: string;
|
|
3663
|
-
}>;
|
|
3664
|
-
};
|
|
3665
|
-
/**
|
|
3666
|
-
* Optional. Required only for `stream: true` calls. The function used to
|
|
3667
|
-
* open a streaming HTTP request. Takes the same request shape as
|
|
3668
|
-
* `request`, but resolves to an `AsyncIterable` of progressively-arriving
|
|
3669
|
-
* `Uint8Array` or `string` chunks instead of a buffered `ResponseLike`.
|
|
3670
|
-
* Defaults to native `fetch`.
|
|
3671
|
-
*/
|
|
3672
|
-
requestStream?: StreamRequestLike;
|
|
3673
|
-
/**
|
|
3674
|
-
* Optional. How the raw stream bytes are split into individual event
|
|
3675
|
-
* payloads. Defaults to Server-Sent Events framing (`data: ...` blocks
|
|
3676
|
-
* separated by a blank line, `[DONE]` sentinel honored, see
|
|
3677
|
-
* `parseSseStream`), which covers the large majority of LLM providers'
|
|
3678
|
-
* streaming HTTP endpoints. Override this for a provider that frames its
|
|
3679
|
-
* stream differently, e.g. newline-delimited JSON (NDJSON) with no SSE
|
|
3680
|
-
* envelope.
|
|
3681
|
-
*/
|
|
3682
|
-
parseStreamFrames?: (chunks: AsyncIterable<Uint8Array | string>) => AsyncIterable<unknown>;
|
|
3683
|
-
/**
|
|
3684
|
-
* Optional. Required only for `stream: true` calls. Maps one parsed
|
|
3685
|
-
* stream event (already extracted from its frame by `parseStreamFrames`)
|
|
3686
|
-
* into zero, one, or more `WireStreamChunk`s, mirrors `mapResponse`'s
|
|
3687
|
-
* role for the non-streaming path, just per-event instead of once for
|
|
3688
|
-
* the whole body. Return `undefined` to skip an event that carries
|
|
3689
|
-
* nothing VernLLM needs (e.g. a provider's keep-alive ping). Configs
|
|
3690
|
-
* that don't implement this make `stream: true` throw a clear
|
|
3691
|
-
* `LLMError('validation')` rather than a confusing runtime failure or a
|
|
3692
|
-
* silently empty stream.
|
|
3693
|
-
*/
|
|
3694
|
-
mapStreamEvent?: (event: unknown) => WireStreamChunk | WireStreamChunk[] | undefined;
|
|
3695
|
-
/**
|
|
3696
|
-
* Optional. How to read AIMD's proactive rate limit hint off a
|
|
3697
|
-
* successful response. Defaults to OpenAI's header set.
|
|
3698
|
-
*/
|
|
3699
|
-
parseRateLimitHint?: (headers: ResponseLike['headers']) => ProviderRateLimitHint;
|
|
3700
|
-
}
|
|
3701
|
-
/**
|
|
3702
|
-
* A fetch-based escape hatch for providers with no SDK, or where pulling one
|
|
3703
|
-
* in isnt worth it. You supply the URL, headers, and two small mapping
|
|
3704
|
-
* functions; this handles the HTTP call and slots the result into the same
|
|
3705
|
-
* `LLMClient` shape every other adapter produces, so retries, timeouts,
|
|
3706
|
-
* the circuit breaker, and JSON/schema handling all still work unmodified
|
|
3707
|
-
*
|
|
3708
|
-
* Non-2xx responses throw an error with `.status` set to the HTTP status
|
|
3709
|
-
* code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
|
|
3710
|
-
* 401/403) applies here too
|
|
3711
|
-
*
|
|
3712
|
-
* Tool calling works the same way as every other adapter: `mapRequest`
|
|
3713
|
-
* receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
|
|
3714
|
-
* translate them into whatever shape the provider's wire format expects
|
|
3715
|
-
* (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
|
|
3716
|
-
* field). On the way back, `mapResponse` may return a `toolCalls` array
|
|
3717
|
-
* (id/name/JSON-encoded-arguments-string per call) alongside or instead of
|
|
3718
|
-
* `content`; VernLLM parses and (if `argumentsSchema` was set) validates
|
|
3719
|
-
* those arguments the same way it does for every other adapter. For
|
|
3720
|
-
* `stream: true`, tool-call deltas go through the existing
|
|
3721
|
-
* `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
|
|
3722
|
-
* no separate config is needed for streaming vs non-streaming tool calls.
|
|
3723
|
-
*
|
|
3724
|
-
* `createStream` requires `mapStreamEvent` (there's no non-streaming
|
|
3725
|
-
* response to fall back on, unlike the other three optional streaming
|
|
3726
|
-
* seams). It opens the request via `requestStream` (defaults to native
|
|
3727
|
-
* `fetch`), splits the raw bytes into individual events via
|
|
3728
|
-
* `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
|
|
3729
|
-
* and translates each event into `WireStreamChunk`(s) via
|
|
3730
|
-
* `mapStreamEvent`. Both seams are overridable per-config for providers
|
|
3731
|
-
* that don't fit the SSE-over-fetch default. If a custom `request`
|
|
3732
|
-
* transport is configured, `requestStream` must be configured too,
|
|
3733
|
-
* `requestStream` never silently falls back to `request` (see
|
|
3734
|
-
* `createStream`'s own comment for why), so a `stream: true` call with
|
|
3735
|
-
* `request` set but no `requestStream` throws a clear
|
|
3736
|
-
* `LLMError('validation')` instead of quietly using unrelated native
|
|
3737
|
-
* `fetch`.
|
|
3738
|
-
*/
|
|
3739
|
-
export declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
3740
|
-
//#endregion
|
|
3741
|
-
//#region src/adapters/openaiCompatible.d.ts
|
|
3742
|
-
/**
|
|
3743
|
-
* Adapter for any SDK/client whose `chat.completions.create` already
|
|
3744
|
-
* matches the OpenAI wire format: this covers most hosted inference
|
|
3745
|
-
* providers, since "OpenAI-compatible" is a de facto standard for chat
|
|
3746
|
-
* completion APIs. Almost everything passes straight through untouched,
|
|
3747
|
-
* this exists purely so call sites read clearly (`fromMistral(client)` vs
|
|
3748
|
-
* handing a Mistral client to something typed for OpenAI) and so a real
|
|
3749
|
-
* transformation could be added later, per-provider, without a breaking
|
|
3750
|
-
* change.
|
|
3751
|
-
*
|
|
3752
|
-
* The one thing that isn't a pure passthrough: a `ContentBlock[]`
|
|
3753
|
-
* `userContent` is translated into OpenAI's native `image_url` content-part
|
|
3754
|
-
* shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
|
|
3755
|
-
* rather than a copy of any one provider's wire format.
|
|
3756
|
-
*
|
|
3757
|
-
* Not every SDKs own TypeScript types line up exactly with `LLMClient`
|
|
3758
|
-
* (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
|
|
3759
|
-
* the actual compatibility contract is the JSON each provider sends and
|
|
3760
|
-
* receives over the wire, not the SDKs TS types.
|
|
3761
|
-
*
|
|
3762
|
-
* `createStream` is implemented by calling the same underlying
|
|
3763
|
-
* `chat.completions.create` with `stream: true` (and, for providers that
|
|
3764
|
-
* support it, `stream_options: { include_usage: true }`, so a final usage
|
|
3765
|
-
* block arrives), the OpenAI SDK, and every OpenAI-compatible client
|
|
3766
|
-
* modeled on it, returns an `AsyncIterable` of SSE chunks instead of a
|
|
3767
|
-
* single completion object when `stream: true` is set. Each chunk is
|
|
3768
|
-
* translated into `WireStreamChunk`(s) via `toWireStreamChunks`.
|
|
3769
|
-
*
|
|
3770
|
-
* Note on long-running reasoning models: this adapter consumes the
|
|
3771
|
-
* underlying SDK's already-parsed stream rather than raw SSE bytes, so
|
|
3772
|
-
* unlike `fromFetch`/`fromAnthropic` it cannot see comment-only keep-alive
|
|
3773
|
-
* ping frames. Combined with `chunkIdleTimeoutMs`'s 30 second default and
|
|
3774
|
-
* `reasoningEffort` (documented to have long silent gaps for o-series and
|
|
3775
|
-
* similar models), a long-running reasoning call on this adapter can trip
|
|
3776
|
-
* the idle timeout even though the provider is still working. Raise or
|
|
3777
|
-
* disable `chunkIdleTimeoutMs` per call for those routes, see `CallParams`.
|
|
3778
|
-
*/
|
|
3779
|
-
interface OpenAICompatibleAdapterOptions {
|
|
3780
|
-
/**
|
|
3781
|
-
* Whether the provider supports `stream_options.include_usage`. Not
|
|
3782
|
-
* every "OpenAI-compatible" provider is guaranteed to, so this defaults
|
|
3783
|
-
* to `true` (matching OpenAI, Groq, Mistral, and most others observed)
|
|
3784
|
-
* and should be set to `false` for a provider verified not to support
|
|
3785
|
-
* it. When `false`, `stream_options` is omitted entirely and no usage
|
|
3786
|
-
* block will arrive on the stream; callers relying on streamed `usage`
|
|
3787
|
-
* with such a provider won't get one.
|
|
3788
|
-
*/
|
|
3789
|
-
supportsStreamUsage?: boolean;
|
|
3790
|
-
/**
|
|
3791
|
-
* Overrides the token count `budgetTokens` buckets into when the caller
|
|
3792
|
-
* sets `budgetTokens` but not `reasoningEffort` (OpenAI-compatible
|
|
3793
|
-
* clients have no numeric budget field of their own, see
|
|
3794
|
-
* `adapters/internal/reasoningBudget.utils.ts`). Only the tiers listed
|
|
3795
|
-
* are changed; any omitted tier keeps the built-in default. Has no
|
|
3796
|
-
* effect when `reasoningEffort` is set directly.
|
|
3797
|
-
*/
|
|
3798
|
-
reasoningEffortTokens?: Partial<EffortTokenTable>;
|
|
3799
|
-
/**
|
|
3800
|
-
* Whether the client's request builder supports `.withResponse()`
|
|
3801
|
-
* (needed for AIMD's proactive path). Default `false`, since not
|
|
3802
|
-
* every "OpenAI-compatible" client is confirmed to support it.
|
|
3803
|
-
*/
|
|
3804
|
-
supportsWithResponse?: boolean;
|
|
3805
|
-
}
|
|
3806
|
-
export declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
|
|
3807
|
-
/**
|
|
3808
|
-
* Named alias for the OpenAI SDK itself. A raw `new OpenAI(...)` instance
|
|
3809
|
-
* structurally matches most of `LLMClient`, but newer `openai` SDK major
|
|
3810
|
-
* versions have widened `ChatCompletionContentPart` (e.g. adding a `file`
|
|
3811
|
-
* variant) in ways that no longer structurally satisfy VernLLM's
|
|
3812
|
-
* provider-agnostic `ContentBlock[]` on `userContent`, so passing the SDK
|
|
3813
|
-
* instance directly can fail to typecheck depending on the installed
|
|
3814
|
-
* `openai` version. Wrapping with `fromOpenAI()` sidesteps that by
|
|
3815
|
-
* translating through `unknown` at the boundary, and also picks up
|
|
3816
|
-
* multimodal image translation and `createStream` wiring that a raw
|
|
3817
|
-
* client doesn't have. See Migration Notes for details.
|
|
3818
|
-
*
|
|
3819
|
-
* `supportsWithResponse` defaults to `false` here too:
|
|
3820
|
-
* `client` is `unknown`, so there's no way to verify it's really the
|
|
3821
|
-
* official `openai` package's client versus a fake or a test double.
|
|
3822
|
-
* Pass `supportsWithResponse: true` once you've confirmed it.
|
|
3823
|
-
*/
|
|
3824
|
-
export declare const fromOpenAI: typeof fromOpenAICompatible;
|
|
3825
|
-
/** Groqs SDK matches the OpenAI wire format */
|
|
3826
|
-
export declare const fromGroq: typeof fromOpenAICompatible;
|
|
3827
|
-
/**
|
|
3828
|
-
* Mistrals `chat.completions`-shaped client (or their OpenAI-compat
|
|
3829
|
-
* endpoint). Mistral supports `stream_options.include_usage` (added after
|
|
3830
|
-
* an earlier period where it returned a 422 for unrecognized fields, per
|
|
3831
|
-
* Mistral's changelog and streaming docs), so this is a plain alias like
|
|
3832
|
-
* the others, `supportsStreamUsage` defaults to `true`.
|
|
3833
|
-
*/
|
|
3834
|
-
export declare const fromMistral: typeof fromOpenAICompatible;
|
|
3835
|
-
/** DeepSeeks API is OpenAI-compatible */
|
|
3836
|
-
export declare const fromDeepSeek: typeof fromOpenAICompatible;
|
|
3837
|
-
/** Cerebras inference API is OpenAI-compatible */
|
|
3838
|
-
export declare const fromCerebras: typeof fromOpenAICompatible;
|
|
3839
|
-
/** Together AIs API is OpenAI-compatible */
|
|
3840
|
-
export declare const fromTogether: typeof fromOpenAICompatible;
|
|
3841
|
-
/** Fireworks AIs API is OpenAI-compatible */
|
|
3842
|
-
export declare const fromFireworks: typeof fromOpenAICompatible;
|
|
3843
|
-
/**
|
|
3844
|
-
* Ollama exposes an OpenAI-compatible endpoint at `/v1/chat/completions`
|
|
3845
|
-
* (as opposed to its native `/api/chat` format, which differs). Point an
|
|
3846
|
-
* OpenAI SDK instances `baseURL` at your Ollama server and pass it here:
|
|
3847
|
-
* this does not talk to Ollamas native API directly.
|
|
3848
|
-
*/
|
|
3849
|
-
export declare const fromOllama: typeof fromOpenAICompatible;
|
|
3850
|
-
/** OpenRouter's API is OpenAI-compatible */
|
|
3851
|
-
export declare const fromOpenRouter: typeof fromOpenAICompatible;
|
|
3852
|
-
/** Perplexity's API is OpenAI-compatible */
|
|
3853
|
-
export declare const fromPerplexity: typeof fromOpenAICompatible;
|
|
3854
|
-
/** DeepInfra's API is OpenAI-compatible */
|
|
3855
|
-
export declare const fromDeepInfra: typeof fromOpenAICompatible;
|
|
3856
|
-
/** Novita's API is OpenAI-compatible */
|
|
3857
|
-
export declare const fromNovita: typeof fromOpenAICompatible;
|
|
3858
|
-
/** Hyperbolic's API is OpenAI-compatible */
|
|
3859
|
-
export declare const fromHyperbolic: typeof fromOpenAICompatible;
|
|
3860
|
-
/** Moonshot's (Kimi) API is OpenAI-compatible */
|
|
3861
|
-
export declare const fromMoonshot: typeof fromOpenAICompatible;
|
|
3862
|
-
/** Zhipu's (GLM) API is OpenAI-compatible */
|
|
3863
|
-
export declare const fromZhipu: typeof fromOpenAICompatible;
|
|
3864
|
-
/**
|
|
3865
|
-
* LM Studio exposes an OpenAI-compatible endpoint at `/v1/chat/completions`.
|
|
3866
|
-
* Point an OpenAI SDK instance's `baseURL` at your local LM Studio server.
|
|
3867
|
-
*/
|
|
3868
|
-
export declare const fromLMStudio: typeof fromOpenAICompatible;
|
|
3869
|
-
/**
|
|
3870
|
-
* vLLM's OpenAI-compatible server mode exposes `/v1/chat/completions`.
|
|
3871
|
-
* Point an OpenAI SDK instance's `baseURL` at your vLLM server.
|
|
3872
|
-
*/
|
|
3873
|
-
export declare const fromVLLM: typeof fromOpenAICompatible;
|
|
3874
|
-
/** xAI's Grok API is OpenAI-compatible */
|
|
3875
|
-
export declare const fromXAI: typeof fromOpenAICompatible;
|
|
3876
|
-
/** NVIDIA NIM's hosted and self-hosted endpoints are OpenAI-compatible */
|
|
3877
|
-
export declare const fromNvidiaNIM: typeof fromOpenAICompatible;
|
|
3878
|
-
/** Vercel AI Gateway is OpenAI-compatible */
|
|
3879
|
-
export declare const fromVercelAIGateway: typeof fromOpenAICompatible;
|
|
3880
|
-
/** Cloudflare Workers AI exposes an OpenAI-compatible endpoint */
|
|
3881
|
-
export declare const fromCloudflareWorkersAI: typeof fromOpenAICompatible;
|
|
3882
|
-
/** Nebius AI Studio is OpenAI-compatible */
|
|
3883
|
-
export declare const fromNebius: typeof fromOpenAICompatible;
|
|
3884
|
-
/** SambaNova Cloud's API is OpenAI-compatible */
|
|
3885
|
-
export declare const fromSambaNova: typeof fromOpenAICompatible;
|
|
3886
|
-
/** Baseten's model hosting exposes an OpenAI-compatible endpoint */
|
|
3887
|
-
export declare const fromBaseten: typeof fromOpenAICompatible;
|
|
3888
|
-
/** Featherless AI's API is OpenAI-compatible */
|
|
3889
|
-
export declare const fromFeatherless: typeof fromOpenAICompatible;
|
|
3890
|
-
/** Friendli AI's serving endpoint is OpenAI-compatible */
|
|
3891
|
-
export declare const fromFriendli: typeof fromOpenAICompatible;
|
|
3892
|
-
/** SiliconFlow's API is OpenAI-compatible */
|
|
3893
|
-
export declare const fromSiliconFlow: typeof fromOpenAICompatible;
|
|
3894
|
-
/** Parasail's inference API is OpenAI-compatible */
|
|
3895
|
-
export declare const fromParasail: typeof fromOpenAICompatible;
|
|
3896
|
-
/** StepFun's API is OpenAI-compatible */
|
|
3897
|
-
export declare const fromStepFun: typeof fromOpenAICompatible;
|
|
3898
|
-
/** MiniMax's API is OpenAI-compatible */
|
|
3899
|
-
export declare const fromMiniMax: typeof fromOpenAICompatible;
|
|
3900
|
-
/** Lambda Labs' Inference API is OpenAI-compatible */
|
|
3901
|
-
export declare const fromLambdaLabs: typeof fromOpenAICompatible;
|
|
3902
|
-
/** Snowflake Cortex's LLM endpoint is OpenAI-compatible */
|
|
3903
|
-
export declare const fromSnowflakeCortex: typeof fromOpenAICompatible;
|
|
3904
|
-
/** Anyscale Endpoints' API is OpenAI-compatible */
|
|
3905
|
-
export declare const fromAnyscale: typeof fromOpenAICompatible;
|
|
3906
|
-
/** Lepton AI's inference API is OpenAI-compatible */
|
|
3907
|
-
export declare const fromLepton: typeof fromOpenAICompatible;
|
|
3908
|
-
/** Inference.net's API is OpenAI-compatible */
|
|
3909
|
-
export declare const fromInferenceNet: typeof fromOpenAICompatible;
|
|
3910
|
-
/** Infermatic's API is OpenAI-compatible */
|
|
3911
|
-
export declare const fromInfermatic: typeof fromOpenAICompatible;
|
|
3912
|
-
/** AtlasCloud's inference API is OpenAI-compatible */
|
|
3913
|
-
export declare const fromAtlasCloud: typeof fromOpenAICompatible;
|
|
3914
|
-
/** 01.AI's (Yi models) API is OpenAI-compatible */
|
|
3915
|
-
export declare const from01AI: typeof fromOpenAICompatible;
|
|
426
|
+
export declare function defineCachedCallParams<P extends CachedCallParams<unknown>>(params: P): P;
|
|
3916
427
|
//#endregion
|
|
3917
|
-
export type
|
|
428
|
+
export { type AdapterInfo, type AssistantContent, type AttemptContext, type CacheAdapter, type CachedCallParams, type CachedConditionalToolCallParams, type CachedJsonModeDisabledCallParams, type CachedJsonModeEnabledCallParams, type CachedStreamCallParams, type CachedStreamConditionalToolCallParams, type CachedStreamJsonModeDisabledCallParams, type CachedStreamJsonModeEnabledCallParams, type CachedStreamToolCallParams, type CachedToolCallParams, type CallContext, type CallMeta, type CallParams, type CallResult, type CallWithToolsResult, CircuitBreaker, type CircuitBreakerAdapter, type CircuitBreakerCallContext, type CircuitBreakerOptions, type CircuitBreakerStateChangeHandler, type CircuitState, type CircuitTarget, type ConditionalToolCallParams, ConsecutiveTripping, ConsoleLogger, type ContentBlock, type ContentResult, type ConversationTurn, type CooldownBackoff, type CreateMiddlewareOptions, type DuplicateToolNamesIssue, type EvictionOption, type ExponentialBackoffOptions, type FallbackAttempt, FallbackExhaustedError, type FallbackOn, type FallbackOnContext, type FallbackTarget, type HistoryToolResultIssue, type ImageBlock, type JsonModeDisabledCallParams, type JsonModeEnabledCallParams, type JsonSchemaSpec, type JsonValue, type LLMClient, LLMError, type LLMErrorCode, type LLMErrorIssuesByCode, type LLMErrorSnapshot, type LLMErrorType, type LLMRequestShape, type LLMRequestSnapshot, type Logger, type MiddlewareCapabilities, type MiddlewareContext, type MiddlewareContextBase, type MiddlewareRef, type MiddlewareStateBag, type MiddlewareStateEntry, type MiddlewareStateKey, type OnEvent, type OnUsage, type PreDispatchContext, type RateLimitAcquireResult, type RateLimitOptions, type RateLimitReason, type RateLimitState, RateLimiter, type RateLimiterAdapter, type RefundUsage, type RequiredMiddlewareRef, type ReserveUsage, type RetryAttempt, RetryBudget, type RetryBudgetOptions, RollingTripping, type SchemaLike, type StreamCallResult, type StreamChunk, type StreamEnabledCallParams, type StreamJsonModeDisabledCallParams, type StreamJsonModeEnabledCallParams, type TargetCircuitState, type TargetInfo, type TextBlock, type ThinkingBlock, type TokenUsage, type ToolCall, type ToolCallResult, type ToolChoice, type ToolDefinition, type ToolEnabledCallParams, type ToolIssue, type ToolResult, type ToolsDisabledCallParams, type TrippingPolicy, type UnknownToolChoiceIssue, type UnsupportedCapabilityIssue, type VernLLMEvent, type VernLLMMiddleware, type VernLLMOptions, type WireCallRequest, type WireCallRequestPatch, type WireMessage, type WireRequest, type WireResponseFormat, type WireStreamChunk, type WireTool, type WireToolCall, type WireToolChoice, createMiddlewareRef, createMiddlewareStateBag, createStateKey, defaultFallbackOn, defineTool, hasIssues, isFallbackExhaustedError, isLLMError, isStreamResult, isToolCallResult, metaRef, requireRef, stateEntry };
|
|
3918
429
|
//# sourceMappingURL=index.d.cts.map
|