vern-llm 1.7.1 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -11
- package/dist/index.cjs +3166 -700
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1542 -194
- package/dist/index.d.cts.map +1 -1
- package/dist/index.d.ts +1542 -194
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3160 -701
- package/dist/index.js.map +1 -1
- package/package.json +8 -1
package/dist/index.d.cts
CHANGED
|
@@ -1,14 +1,109 @@
|
|
|
1
|
+
//#region src/circuitBreaker.d.ts
|
|
2
|
+
interface CircuitBreakerOptions {
|
|
3
|
+
/** Consecutive failures before the circuit opens, default 5 */
|
|
4
|
+
threshold?: number;
|
|
5
|
+
/** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
|
|
6
|
+
cooldownMs?: number;
|
|
7
|
+
/**
|
|
8
|
+
* Called after every real state change, never for a no-op transition
|
|
9
|
+
* (e.g. open to open). `model` is the resolved model of whichever call
|
|
10
|
+
* triggered this specific transition (the `model` passed to whichever
|
|
11
|
+
* of `assertClosed`/`recordSuccess`/`recordFailure` caused it).
|
|
12
|
+
*
|
|
13
|
+
* With `isolateByModel` off (the default), this is a label only: the
|
|
14
|
+
* breaker still counts failures across every model together, so a
|
|
15
|
+
* threshold crossing can be the sum of several different models'
|
|
16
|
+
* failures even though only the triggering call's `model` is reported
|
|
17
|
+
* here. With `isolateByModel` on, it's exact: each model has its own
|
|
18
|
+
* counter, so the transition really was caused solely by that model.
|
|
19
|
+
*/
|
|
20
|
+
onStateChange?: (from: CircuitState, to: CircuitState, consecutiveFailures: number, model?: string) => void;
|
|
21
|
+
/**
|
|
22
|
+
* Track a separate circuit per resolved model instead of one shared
|
|
23
|
+
* circuit for the whole instance. A failure on one model then never
|
|
24
|
+
* opens another model's circuit, at the cost of slower detection for
|
|
25
|
+
* an outage spread across many distinct models (each model's counter
|
|
26
|
+
* must independently cross `threshold`). Default false: one shared
|
|
27
|
+
* circuit, matching every version before this option existed.
|
|
28
|
+
*
|
|
29
|
+
* A call that omits `model` (only possible calling `CircuitBreaker`
|
|
30
|
+
* directly, `VernLLM` always passes one) falls into one shared bucket
|
|
31
|
+
* alongside every other call that also omits it.
|
|
32
|
+
*/
|
|
33
|
+
isolateByModel?: boolean;
|
|
34
|
+
}
|
|
35
|
+
type CircuitState = 'closed' | 'open' | 'half-open';
|
|
36
|
+
/**
|
|
37
|
+
* Per retry VernLLM-instance circuit breaker. Tracks consecutive failures across
|
|
38
|
+
* calls. Once the threshold is hit, short-circuits new calls with an
|
|
39
|
+
* LLMError('circuit_open') instead of hitting the provider, until the
|
|
40
|
+
* cooldown elapses and a single trial call is allowed through
|
|
41
|
+
*/
|
|
42
|
+
declare class CircuitBreaker {
|
|
43
|
+
private readonly threshold;
|
|
44
|
+
private readonly cooldownMs;
|
|
45
|
+
private readonly onStateChange?;
|
|
46
|
+
private readonly isolateByModel;
|
|
47
|
+
private readonly sharedBucket;
|
|
48
|
+
private readonly bucketsByModel;
|
|
49
|
+
constructor(options?: CircuitBreakerOptions);
|
|
50
|
+
/** Returns the bucket for a model if one already exists, without allocating. */
|
|
51
|
+
private lookupBucket;
|
|
52
|
+
/** Creates and stores a bucket for a model when the first mutation needs one. */
|
|
53
|
+
private ensureBucketFor;
|
|
54
|
+
/** Every state mutation routes through here, so `onStateChange` fires exactly once per real change. */
|
|
55
|
+
private transition;
|
|
56
|
+
/**
|
|
57
|
+
* Throws if the circuit is open and the cooldown hasn't elapsed, or if
|
|
58
|
+
* the circuit is half-open and a trial call is already in flight.
|
|
59
|
+
* Otherwise, if the circuit just became eligible for a trial (cooldown
|
|
60
|
+
* elapsed, or half-open with no trial currently running), this call
|
|
61
|
+
* becomes that trial
|
|
62
|
+
*/
|
|
63
|
+
assertClosed(model?: string): void;
|
|
64
|
+
recordSuccess(model?: string): void;
|
|
65
|
+
recordFailure(model?: string): void;
|
|
66
|
+
/**
|
|
67
|
+
* With `isolateByModel` off (the default), `model` is ignored and the
|
|
68
|
+
* one shared circuit's state is returned, unchanged from every version
|
|
69
|
+
* before this option existed. With `isolateByModel` on, returns that
|
|
70
|
+
* model's own state, `'closed'` for a model never seen yet, same as a
|
|
71
|
+
* fresh breaker.
|
|
72
|
+
*/
|
|
73
|
+
getState(model?: string): CircuitState;
|
|
74
|
+
} //#endregion
|
|
1
75
|
//#region src/types/errors.d.ts
|
|
76
|
+
|
|
77
|
+
//# sourceMappingURL=circuitBreaker.d.ts.map
|
|
2
78
|
type LLMErrorType = 'timeout' | 'api' | 'parse' | 'validation' | 'circuit_open' | 'quota_exceeded' | 'unknown' | 'aborted';
|
|
79
|
+
/**
|
|
80
|
+
* Machine readable discriminator within a `type`, for cases where `type`
|
|
81
|
+
* alone is too coarse to act on. Optional and additive: errors thrown
|
|
82
|
+
* before a given code existed simply omit it.
|
|
83
|
+
*/
|
|
84
|
+
type LLMErrorCode = 'unknown_tool' | 'duplicate_tool_call_id' | 'local_rate_limit' | 'provider_rate_limited' | 'fallback_exhausted';
|
|
85
|
+
/** One tool call's contract failure, used to report every bad call in a response at once. */
|
|
86
|
+
interface ToolIssue {
|
|
87
|
+
name: string;
|
|
88
|
+
toolCallId: string;
|
|
89
|
+
code: LLMErrorCode;
|
|
90
|
+
detail?: unknown;
|
|
91
|
+
}
|
|
3
92
|
declare class LLMError extends Error {
|
|
4
93
|
type: LLMErrorType;
|
|
5
94
|
status?: number | undefined;
|
|
6
95
|
issues?: unknown | undefined;
|
|
7
96
|
cause?: unknown | undefined;
|
|
8
97
|
retryAfterMs?: number | undefined;
|
|
9
|
-
|
|
98
|
+
/** Stable discriminator within `type`. Absent on errors predating it. */
|
|
99
|
+
code?: LLMErrorCode | undefined;
|
|
100
|
+
constructor(message: string, type: LLMErrorType, status?: number | undefined, issues?: unknown | undefined, cause?: unknown | undefined, retryAfterMs?: number | undefined, /** Stable discriminator within `type`. Absent on errors predating it. */
|
|
101
|
+
code?: LLMErrorCode | undefined);
|
|
102
|
+
/** Every tool contract failure in one response, when there is more than one. */
|
|
103
|
+
toolIssues?: ToolIssue[];
|
|
10
104
|
}
|
|
11
105
|
declare function isLLMError(err: unknown): err is LLMError;
|
|
106
|
+
|
|
12
107
|
//#endregion
|
|
13
108
|
//#region src/types/cache.d.ts
|
|
14
109
|
//# sourceMappingURL=errors.d.ts.map
|
|
@@ -77,8 +172,206 @@ declare class TieredCacheAdapter<T = unknown> implements CacheAdapter<T> {
|
|
|
77
172
|
}
|
|
78
173
|
|
|
79
174
|
//#endregion
|
|
80
|
-
//#region src/
|
|
175
|
+
//#region src/rateLimit.d.ts
|
|
81
176
|
//# sourceMappingURL=cache.d.ts.map
|
|
177
|
+
/** The request shape sent to `LLMClient['chat']['completions']['create']`, used for token estimation. */
|
|
178
|
+
type WireRequest = Parameters<LLMClient['chat']['completions']['create']>[0];
|
|
179
|
+
/** Which configured bucket is currently blocking a call. */
|
|
180
|
+
type RateLimitReason = 'concurrency' | 'rpm' | 'tpm';
|
|
181
|
+
interface RateLimitOptions {
|
|
182
|
+
/** Max requests per minute. Omit for unlimited. */
|
|
183
|
+
requestsPerMinute?: number;
|
|
184
|
+
/**
|
|
185
|
+
* Max tokens per minute. Enforced against a pre-flight estimate, then
|
|
186
|
+
* reconciled against reported usage once the call completes. Omit for
|
|
187
|
+
* unlimited.
|
|
188
|
+
*/
|
|
189
|
+
tokensPerMinute?: number;
|
|
190
|
+
/** Max requests in flight at once. Default 0, meaning unlimited. */
|
|
191
|
+
maxConcurrent?: number;
|
|
192
|
+
/**
|
|
193
|
+
* Max time a call may sit queued waiting for capacity, in ms. Exceeding
|
|
194
|
+
* it throws rather than hanging forever. Default 30000. Pass 0 to wait
|
|
195
|
+
* indefinitely.
|
|
196
|
+
*/
|
|
197
|
+
maxQueueMs?: number;
|
|
198
|
+
/** Max queued calls before new ones reject immediately instead of queueing. Default 0, unbounded. */
|
|
199
|
+
maxQueueSize?: number;
|
|
200
|
+
/**
|
|
201
|
+
* Pre-flight token estimate for `tokensPerMinute`. Defaults to a
|
|
202
|
+
* chars/4 heuristic over message content plus `max_tokens`.
|
|
203
|
+
*/
|
|
204
|
+
estimateTokens?: (request: WireRequest) => number;
|
|
205
|
+
}
|
|
206
|
+
interface RateLimitAcquireResult {
|
|
207
|
+
/**
|
|
208
|
+
* Releases the concurrency slot this attempt held and reconciles the
|
|
209
|
+
* token bucket against real usage, when `actualTokens` is supplied.
|
|
210
|
+
* Idempotent: only the first call does anything. Must run in a
|
|
211
|
+
* `finally` block so a slot is never leaked on a failed attempt.
|
|
212
|
+
*/
|
|
213
|
+
release: (actualTokens?: number) => void;
|
|
214
|
+
/** How long this attempt waited in queue before capacity was available. */
|
|
215
|
+
waitedMs: number;
|
|
216
|
+
/** Which bucket was blocking this attempt just before it cleared, if any wait happened. */
|
|
217
|
+
reason?: RateLimitReason;
|
|
218
|
+
}
|
|
219
|
+
/** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
|
|
220
|
+
declare function defaultEstimateTokens(request: WireRequest): number;
|
|
221
|
+
/**
|
|
222
|
+
* Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
|
|
223
|
+
* concurrency) behind one FIFO queue, so a large call isn't starved by a
|
|
224
|
+
* stream of small ones. Any bucket omitted from `options` has infinite
|
|
225
|
+
* capacity and never blocks.
|
|
226
|
+
*/
|
|
227
|
+
declare class RateLimiter {
|
|
228
|
+
private readonly requests?;
|
|
229
|
+
private readonly tokens?;
|
|
230
|
+
private readonly concurrency?;
|
|
231
|
+
private readonly maxQueueMs;
|
|
232
|
+
private readonly maxQueueSize;
|
|
233
|
+
private readonly estimateTokensFn;
|
|
234
|
+
private readonly queue;
|
|
235
|
+
/**
|
|
236
|
+
* A single scheduled re-check for the head of the queue when it's
|
|
237
|
+
* blocked on a bucket that refills on its own clock (rpm/tpm), so a
|
|
238
|
+
* queue that nobody calls `acquire`/`release` on again isn't stuck
|
|
239
|
+
* forever waiting for an external trigger to re-drain it. Not needed
|
|
240
|
+
* for a concurrency block, which only clears via `release`.
|
|
241
|
+
*/
|
|
242
|
+
private wakeTimer?;
|
|
243
|
+
constructor(options: RateLimitOptions);
|
|
244
|
+
/** Pre-flight token estimate for a request, per the configured (or default) heuristic. */
|
|
245
|
+
estimate(request: WireRequest): number;
|
|
246
|
+
/**
|
|
247
|
+
* Waits for capacity in every configured bucket, then takes from each.
|
|
248
|
+
* The returned `release` gives the concurrency slot back and reconciles
|
|
249
|
+
* the token bucket against real usage; it must run in a `finally` block.
|
|
250
|
+
*/
|
|
251
|
+
acquire(estimatedTokens: number, signal?: AbortSignal): Promise<RateLimitAcquireResult>;
|
|
252
|
+
private queueFullError;
|
|
253
|
+
private enqueue;
|
|
254
|
+
/**
|
|
255
|
+
* Checks and takes from every configured bucket as one atomic unit: if
|
|
256
|
+
* any bucket lacks capacity, whatever was already taken from the
|
|
257
|
+
* earlier ones in this attempt is rolled back before reporting which
|
|
258
|
+
* bucket blocked.
|
|
259
|
+
*/
|
|
260
|
+
private tryAcquireBuckets;
|
|
261
|
+
/** Drains the queue head first. Stops at the first waiter that still can't proceed, so no one is starved out of turn. */
|
|
262
|
+
private drain;
|
|
263
|
+
/**
|
|
264
|
+
* Schedules a one-shot re-check of the queue for whenever the bucket
|
|
265
|
+
* that's currently blocking the head waiter should next have enough
|
|
266
|
+
* capacity. A no-op for a concurrency block (only `release` can clear
|
|
267
|
+
* that) or while a wake is already pending.
|
|
268
|
+
*/
|
|
269
|
+
private scheduleWake;
|
|
270
|
+
/**
|
|
271
|
+
* Builds the one-shot release closure for an acquired slot. Only the
|
|
272
|
+
* concurrency bucket is given back on release; the requests-per-minute
|
|
273
|
+
* bucket is a real spend that only recovers via its own refill, and the
|
|
274
|
+
* tokens bucket is reconciled against `actualTokens` rather than fully
|
|
275
|
+
* refunded, since real tokens really were spent.
|
|
276
|
+
*/
|
|
277
|
+
private makeRelease;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
//#endregion
|
|
281
|
+
//#region src/types/fallback.d.ts
|
|
282
|
+
//# sourceMappingURL=rateLimit.d.ts.map
|
|
283
|
+
/**
|
|
284
|
+
* One provider to try after the primary (or after an earlier fallback
|
|
285
|
+
* target) fails. Order is the policy: VernLLM never reorders, scores, or
|
|
286
|
+
* selects a target, it only walks the list as given.
|
|
287
|
+
*
|
|
288
|
+
* Most per-target overrides fall back to the parent `VernLLM` instance's
|
|
289
|
+
* own option when omitted, so a target only needs to specify what's
|
|
290
|
+
* actually different about it (a different client/model is the common
|
|
291
|
+
* case). `circuitBreaker` and `rateLimit` are the exception: they are
|
|
292
|
+
* never inherited from the parent, since a breaker or limiter tuned for
|
|
293
|
+
* the primary provider's limits is rarely right for a fallback's. Leave
|
|
294
|
+
* them unset on a target to run it without one, even if the parent has
|
|
295
|
+
* one configured.
|
|
296
|
+
*/
|
|
297
|
+
interface FallbackTarget {
|
|
298
|
+
client: LLMClient;
|
|
299
|
+
model: string;
|
|
300
|
+
/** Label for events, errors, and `TokenUsage.provider`. Default `` `fallback[${index}]` ``. */
|
|
301
|
+
name?: string;
|
|
302
|
+
maxRetries?: number;
|
|
303
|
+
timeoutMs?: number;
|
|
304
|
+
chunkIdleTimeoutMs?: number;
|
|
305
|
+
baseDelayMs?: number;
|
|
306
|
+
defaultMaxTokens?: number;
|
|
307
|
+
defaultTemperature?: number | null;
|
|
308
|
+
nonRetryableStatus?: number[];
|
|
309
|
+
/** This target's own circuit breaker, independent of every other target's. Not inherited from the parent's `circuitBreaker`. */
|
|
310
|
+
circuitBreaker?: boolean | CircuitBreakerOptions;
|
|
311
|
+
/** This target's own rate limiter, independent of every other target's. Not inherited from the parent's `rateLimit`. */
|
|
312
|
+
rateLimit?: RateLimitOptions;
|
|
313
|
+
}
|
|
314
|
+
/**
|
|
315
|
+
* Written into `CallParams['meta']` once `call()` resolves, so a caller
|
|
316
|
+
* who wants provider identity on the same line as the result doesn't need
|
|
317
|
+
* to read it back out of `onUsage`.
|
|
318
|
+
*/
|
|
319
|
+
interface CallMeta {
|
|
320
|
+
provider: string;
|
|
321
|
+
model: string;
|
|
322
|
+
/** `-1` if the primary target answered, otherwise the index into `fallback`. */
|
|
323
|
+
fallbackIndex: number;
|
|
324
|
+
usedFallback: boolean;
|
|
325
|
+
/** Attempts made against the target that ultimately answered, including the successful one. */
|
|
326
|
+
attempts: number;
|
|
327
|
+
}
|
|
328
|
+
/** One target's circuit state, as returned by `VernLLM.getCircuitStates()`. */
|
|
329
|
+
interface TargetCircuitState {
|
|
330
|
+
provider: string;
|
|
331
|
+
/** Position in the chain: `0` for the primary, `1`+ for fallback targets. */
|
|
332
|
+
index: number;
|
|
333
|
+
isFallback: boolean;
|
|
334
|
+
/** `undefined` if that target has no circuit breaker configured. */
|
|
335
|
+
state: CircuitState | undefined;
|
|
336
|
+
}
|
|
337
|
+
/** One target's failure, recorded on the way to either the next target or `FallbackExhaustedError`. */
|
|
338
|
+
interface FallbackAttempt {
|
|
339
|
+
/** `-1` for the primary target. */
|
|
340
|
+
index: number;
|
|
341
|
+
provider: string;
|
|
342
|
+
model: string;
|
|
343
|
+
error: LLMError;
|
|
344
|
+
}
|
|
345
|
+
/**
|
|
346
|
+
* Decides what happens after a target's own retries are exhausted or
|
|
347
|
+
* abandoned early. Called once per failed target. `'retry'` is not a
|
|
348
|
+
* valid return here: retrying already happened inside the target, this
|
|
349
|
+
* only decides whether to move on to the next one or stop.
|
|
350
|
+
*/
|
|
351
|
+
type FallbackOn = (error: LLMError, context: {
|
|
352
|
+
isLastTarget: boolean;
|
|
353
|
+
}) => 'next' | 'stop';
|
|
354
|
+
/**
|
|
355
|
+
* The default `fallbackOn` policy. Exported so a caller can wrap rather
|
|
356
|
+
* than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
|
|
357
|
+
*/
|
|
358
|
+
declare const defaultFallbackOn: FallbackOn;
|
|
359
|
+
/**
|
|
360
|
+
* Thrown when the chain gives up, whether because the last target failed
|
|
361
|
+
* or `fallbackOn` chose to stop early. Carries each attempt in order so
|
|
362
|
+
* an outage across providers stays debuggable without reproducing it.
|
|
363
|
+
* Extends `LLMError` so `isLLMError` and any `instanceof LLMError` check
|
|
364
|
+
* still passes, inheriting the last failure's `type` so existing
|
|
365
|
+
* type-based handling keeps working on a fallback-exhausted error too.
|
|
366
|
+
*/
|
|
367
|
+
declare class FallbackExhaustedError extends LLMError {
|
|
368
|
+
readonly attempts: FallbackAttempt[];
|
|
369
|
+
constructor(attempts: FallbackAttempt[]);
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
//#endregion
|
|
373
|
+
//#region src/types/schema.d.ts
|
|
374
|
+
//# sourceMappingURL=fallback.d.ts.map
|
|
82
375
|
/**
|
|
83
376
|
* Minimal structural type for a Zod-like schema, so this package doesnt need
|
|
84
377
|
* a hard dependency on a specific Zod major version. Any object exposing
|
|
@@ -110,8 +403,79 @@ interface JsonSchemaSpec {
|
|
|
110
403
|
}
|
|
111
404
|
|
|
112
405
|
//#endregion
|
|
113
|
-
//#region src/types/
|
|
406
|
+
//#region src/types/tools.d.ts
|
|
114
407
|
//# sourceMappingURL=schema.d.ts.map
|
|
408
|
+
/**
|
|
409
|
+
* Describes a capability the model may request, not the capability
|
|
410
|
+
* itself. VernLLM transports this to the provider and parses what comes
|
|
411
|
+
* back; it never executes anything.
|
|
412
|
+
*/
|
|
413
|
+
interface ToolDefinition {
|
|
414
|
+
name: string;
|
|
415
|
+
description: string;
|
|
416
|
+
/** JSON Schema for the tool's input. */
|
|
417
|
+
parameters: Record<string, unknown>;
|
|
418
|
+
/**
|
|
419
|
+
* Optional client-side validator run on the parsed `arguments` before
|
|
420
|
+
* they're handed back to the caller, mirroring the `schema: SchemaLike<T>`
|
|
421
|
+
* pattern already used for response validation (see `types/schema.ts`).
|
|
422
|
+
* Reuses that zero-dependency, `safeParse`-compatible shape instead of
|
|
423
|
+
* requiring a JSON Schema validator (e.g. ajv) as a new dependency.
|
|
424
|
+
* Failed validation throws `LLMError('validation')`. If omitted, VernLLM
|
|
425
|
+
* parses arguments as JSON but does not validate them further.
|
|
426
|
+
*/
|
|
427
|
+
argumentsSchema?: SchemaLike<unknown>;
|
|
428
|
+
}
|
|
429
|
+
/** A single tool invocation requested by the model. */
|
|
430
|
+
interface ToolCall {
|
|
431
|
+
id: string;
|
|
432
|
+
name: string;
|
|
433
|
+
/** Parsed JSON arguments (and validated, if `argumentsSchema` was set). */
|
|
434
|
+
arguments: unknown;
|
|
435
|
+
}
|
|
436
|
+
/** The application's result of executing a `ToolCall`, sent back to the model. */
|
|
437
|
+
interface ToolResult {
|
|
438
|
+
toolCallId: string;
|
|
439
|
+
content: unknown;
|
|
440
|
+
/**
|
|
441
|
+
* Signals a failed tool execution back to the model (matches Anthropic's
|
|
442
|
+
* native `is_error` on tool_result blocks). Only `fromAnthropic` honors
|
|
443
|
+
* this today, Gemini and Bedrock have no equivalent wire concept, so
|
|
444
|
+
* other adapters ignore it silently.
|
|
445
|
+
*/
|
|
446
|
+
isError?: boolean;
|
|
447
|
+
}
|
|
448
|
+
/** `call()` result when `tools` was set and the model produced a normal answer. */
|
|
449
|
+
interface ContentResult<T> {
|
|
450
|
+
type: 'content';
|
|
451
|
+
content: T;
|
|
452
|
+
}
|
|
453
|
+
/** `call()` result when `tools` was set and the model requested one or more tools. */
|
|
454
|
+
interface ToolCallResult {
|
|
455
|
+
type: 'tool_calls';
|
|
456
|
+
toolCalls: ToolCall[];
|
|
457
|
+
/** Any text the model produced alongside the tool request, if present. */
|
|
458
|
+
content?: string;
|
|
459
|
+
}
|
|
460
|
+
type CallWithToolsResult<T> = ContentResult<T> | ToolCallResult;
|
|
461
|
+
/**
|
|
462
|
+
* Runtime-safe check for whether a `call()` result is a `tool_calls`
|
|
463
|
+
* result. Prefer this over relying on TypeScript's static narrowing
|
|
464
|
+
* whenever `params` passed to `call()` wasn't a literal with `tools`
|
|
465
|
+
* inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
|
|
466
|
+
* that case TS may have typed the result as plain `T` even though it's
|
|
467
|
+
* actually a `CallWithToolsResult<T>` at runtime, and this check works
|
|
468
|
+
* either way.
|
|
469
|
+
*/
|
|
470
|
+
declare function isToolCallResult(result: unknown): result is ToolCallResult;
|
|
471
|
+
/** What the model should do about tools on a given call. */
|
|
472
|
+
type ToolChoice = 'auto' | 'none' | 'required' | {
|
|
473
|
+
name: string;
|
|
474
|
+
};
|
|
475
|
+
|
|
476
|
+
//#endregion
|
|
477
|
+
//#region src/types/usage.d.ts
|
|
478
|
+
//# sourceMappingURL=tools.d.ts.map
|
|
115
479
|
type ReserveUsage = (params: {
|
|
116
480
|
coalesced: boolean;
|
|
117
481
|
signal?: AbortSignal;
|
|
@@ -142,17 +506,55 @@ interface TokenUsage {
|
|
|
142
506
|
totalTokens: number;
|
|
143
507
|
requestId: string;
|
|
144
508
|
model: string;
|
|
509
|
+
/**
|
|
510
|
+
* The provider target that produced this usage. See `VernLLMOptions['name']`,
|
|
511
|
+
* default `'primary'`. Optional so consumers constructing a `TokenUsage`
|
|
512
|
+
* themselves (e.g. in tests) aren't forced to supply it; `VernLLM` always
|
|
513
|
+
* populates it. Absent means the same as `'primary'` if you need a value.
|
|
514
|
+
*/
|
|
515
|
+
provider?: string;
|
|
516
|
+
/**
|
|
517
|
+
* Whether this usage came from a fallback target rather than the
|
|
518
|
+
* primary. Optional for the same reason `provider` is: `VernLLM`
|
|
519
|
+
* always populates it, a hand-constructed `TokenUsage` (e.g. in tests)
|
|
520
|
+
* isn't forced to.
|
|
521
|
+
*/
|
|
522
|
+
usedFallback?: boolean;
|
|
145
523
|
}
|
|
146
524
|
type OnUsage = (usage: TokenUsage) => void;
|
|
525
|
+
/**
|
|
526
|
+
* Called when a provider response arrives but VernLLM's own post-processing
|
|
527
|
+
* then fails, after usage data was already present in that response. Covers
|
|
528
|
+
* any error thrown after usage extraction, not just parse/validation, since
|
|
529
|
+
* everything in that path only runs once a response, and real spend, has
|
|
530
|
+
* already arrived. Fires once per failed attempt with extractable usage,
|
|
531
|
+
* never for transport failures, where no response means no honest number
|
|
532
|
+
* to report.
|
|
533
|
+
*/
|
|
534
|
+
type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
|
|
147
535
|
|
|
148
536
|
//#endregion
|
|
149
537
|
//#region src/types/call.d.ts
|
|
150
538
|
//# sourceMappingURL=usage.d.ts.map
|
|
151
|
-
/**
|
|
152
|
-
|
|
153
|
-
|
|
539
|
+
/**
|
|
540
|
+
* A single prior turn in a multi-turn conversation, passed via `history`.
|
|
541
|
+
*
|
|
542
|
+
* Supports normal user/assistant messages and tool continuations: an assistant
|
|
543
|
+
* turn may include `toolCalls`, and a tool turn carries the matching
|
|
544
|
+
* `toolResults`. A tool turn must immediately follow an assistant tool call
|
|
545
|
+
* turn, and every requested tool call must have a result.
|
|
546
|
+
*/
|
|
547
|
+
type ConversationTurn = {
|
|
548
|
+
role: 'user';
|
|
154
549
|
content: string;
|
|
155
|
-
}
|
|
550
|
+
} | {
|
|
551
|
+
role: 'assistant';
|
|
552
|
+
content?: string;
|
|
553
|
+
toolCalls?: ToolCall[];
|
|
554
|
+
} | {
|
|
555
|
+
role: 'tool';
|
|
556
|
+
toolResults: ToolResult[];
|
|
557
|
+
};
|
|
156
558
|
/** A plain text segment of a multimodal `userContent` array. */
|
|
157
559
|
interface TextBlock {
|
|
158
560
|
type: 'text';
|
|
@@ -180,15 +582,28 @@ interface CallParams<T = unknown> extends UsageHooks {
|
|
|
180
582
|
/** Current user message, as text or multimodal content blocks. */
|
|
181
583
|
userContent: string | ContentBlock[];
|
|
182
584
|
/**
|
|
183
|
-
* Previous conversation turns. Must alternate
|
|
184
|
-
*
|
|
585
|
+
* Previous conversation turns. Must alternate roles; tool turns must follow
|
|
586
|
+
* assistant tool calls. Invalid history throws LLMError('validation').
|
|
185
587
|
*/
|
|
186
588
|
history?: ConversationTurn[];
|
|
187
|
-
|
|
589
|
+
/**
|
|
590
|
+
* Generation temperature. Default 0.2, not the provider's own default.
|
|
591
|
+
* Pass `null` to omit `temperature` from the request entirely, so the
|
|
592
|
+
* provider applies its own default instead.
|
|
593
|
+
*/
|
|
594
|
+
temperature?: number | null;
|
|
188
595
|
jsonMode?: boolean;
|
|
189
596
|
maxTokens?: number;
|
|
190
597
|
requestId?: string;
|
|
191
598
|
signal?: AbortSignal;
|
|
599
|
+
/**
|
|
600
|
+
* Per-call override for the instance's `chunkIdleTimeoutMs` (max gap
|
|
601
|
+
* between stream chunks once opened). Only applies when `stream: true`.
|
|
602
|
+
* Useful for routes using reasoning-heavy models with documented long
|
|
603
|
+
* silent gaps mid-stream. Pass 0 to disable the idle timeout for this
|
|
604
|
+
* call.
|
|
605
|
+
*/
|
|
606
|
+
chunkIdleTimeoutMs?: number;
|
|
192
607
|
/** Overrides the instance model for this call. */
|
|
193
608
|
model?: string;
|
|
194
609
|
/** Reasoning effort for supported reasoning models. */
|
|
@@ -202,17 +617,253 @@ interface CallParams<T = unknown> extends UsageHooks {
|
|
|
202
617
|
* Implies jsonMode: true.
|
|
203
618
|
*/
|
|
204
619
|
schema?: SchemaLike<T>;
|
|
620
|
+
/**
|
|
621
|
+
* Tools the model may call. When set, `call()` always returns a
|
|
622
|
+
* `CallWithToolsResult<T>` discriminated union instead of `T` directly
|
|
623
|
+
* (see `CallWithToolsResult`), a breaking-change point: omitting `tools`
|
|
624
|
+
* keeps `call()`'s old `Promise<T>` behavior exactly.
|
|
625
|
+
*
|
|
626
|
+
* Can be combined with `jsonSchema` on Gemini and OpenAI-compatible
|
|
627
|
+
* clients unconditionally (neither ever restricted the combination:
|
|
628
|
+
* Gemini builds `responseSchema`/`tools` as independent fields, OpenAI-
|
|
629
|
+
* compatible clients pass both straight through). On Anthropic and
|
|
630
|
+
* Bedrock, combining the two is opt-in per call site, via each
|
|
631
|
+
* adapter's `nativeStructuredOutputModels` option: models not covered
|
|
632
|
+
* by it still throw `LLMError('validation')`, since `jsonSchema` falls
|
|
633
|
+
* back to a forced single-tool call there, which would collide with
|
|
634
|
+
* real tools. See `fromAnthropic`/`fromBedrock`.
|
|
635
|
+
*
|
|
636
|
+
* `schema` (client-side validation, distinct from `jsonSchema`) was
|
|
637
|
+
* never restricted from combining with `tools` on any provider.
|
|
638
|
+
*/
|
|
639
|
+
tools?: ToolDefinition[];
|
|
640
|
+
/** Defaults to `'auto'` when `tools` is set. */
|
|
641
|
+
toolChoice?: ToolChoice;
|
|
642
|
+
/**
|
|
643
|
+
* Streams the response incrementally instead of resolving once. Default:
|
|
644
|
+
* false. Requires a client/adapter that implements `createStream`.
|
|
645
|
+
* Retry/timeout/circuit-breaker guarantees apply only to opening the
|
|
646
|
+
* stream (through the first chunk); a failure after that point rejects
|
|
647
|
+
* `finalResult` directly and is not retried, since a mid-stream failure
|
|
648
|
+
* isn't connection-time evidence for the circuit breaker, the attempt
|
|
649
|
+
* already counted as a success once the first chunk arrived. Once the
|
|
650
|
+
* stream opens successfully, `finalResult` still resolves to the same
|
|
651
|
+
* validated `T`/`CallWithToolsResult<T>` shape `call()` would have
|
|
652
|
+
* returned for the same params with `stream` omitted. See
|
|
653
|
+
* `StreamCallResult`.
|
|
654
|
+
*/
|
|
655
|
+
stream?: boolean;
|
|
656
|
+
/**
|
|
657
|
+
* Optional out-parameter for provider identity. Pass `{}` (or any object
|
|
658
|
+
* with a mutable `current` property) and `call()` writes a `CallMeta`
|
|
659
|
+
* into `meta.current` before returning, alongside whatever `onUsage`
|
|
660
|
+
* already reports. Ignored for `stream: true`, since `call()` returns
|
|
661
|
+
* before the outcome (and so the target that answered) is known; read
|
|
662
|
+
* `TokenUsage.provider`/`usedFallback` from `onUsage` for streaming
|
|
663
|
+
* calls instead.
|
|
664
|
+
*/
|
|
665
|
+
meta?: {
|
|
666
|
+
current?: CallMeta;
|
|
667
|
+
};
|
|
205
668
|
}
|
|
206
|
-
|
|
669
|
+
/**
|
|
670
|
+
* A `CallParams` variant where tool calling is explicitly enabled.
|
|
671
|
+
*
|
|
672
|
+
* Requiring `tools` to be present allows TypeScript to select the
|
|
673
|
+
* tool-aware `call()` overload and return `CallWithToolsResult<T>` instead
|
|
674
|
+
* of the normal `T` response type.
|
|
675
|
+
*/
|
|
676
|
+
type ToolEnabledCallParams<T> = CallParams<T> & {
|
|
677
|
+
tools: NonNullable<CallParams<T>['tools']>;
|
|
678
|
+
};
|
|
679
|
+
/** Shared cache-configuration fields, minus the internal `fn` primitive. */
|
|
680
|
+
interface CachedCallInput extends UsageHooks {
|
|
207
681
|
cacheKey: string;
|
|
208
682
|
ttl: number;
|
|
209
|
-
fn: () => Promise<T>;
|
|
210
683
|
signal?: AbortSignal;
|
|
211
684
|
}
|
|
685
|
+
/**
|
|
686
|
+
* Parameters for a cached LLM call without tool calling.
|
|
687
|
+
*
|
|
688
|
+
* Combines the cache configuration with the `CallParams` passed to
|
|
689
|
+
* `VernLLM.call()`. The cached value is the normal LLM response type `T`.
|
|
690
|
+
*/
|
|
691
|
+
type CachedCallParams<T> = CachedCallInput & {
|
|
692
|
+
call: CallParams<T>;
|
|
693
|
+
};
|
|
694
|
+
/**
|
|
695
|
+
* Parameters for a cached LLM call with tool calling enabled.
|
|
696
|
+
*
|
|
697
|
+
* The cached value includes the full `CallWithToolsResult<T>`, meaning
|
|
698
|
+
* tool requests and normal content responses are cached exactly as returned
|
|
699
|
+
* by the model.
|
|
700
|
+
*/
|
|
701
|
+
type CachedToolCallParams<T> = CachedCallInput & {
|
|
702
|
+
call: ToolEnabledCallParams<T>;
|
|
703
|
+
};
|
|
212
704
|
|
|
213
705
|
//#endregion
|
|
214
|
-
//#region src/types/
|
|
706
|
+
//#region src/types/stream.d.ts
|
|
215
707
|
//# sourceMappingURL=call.d.ts.map
|
|
708
|
+
/** One incremental unit of a streaming response, as delivered to the caller. */
|
|
709
|
+
type StreamChunk = {
|
|
710
|
+
type: 'text-delta';
|
|
711
|
+
delta: string;
|
|
712
|
+
} | {
|
|
713
|
+
type: 'tool_call_delta';
|
|
714
|
+
index: number;
|
|
715
|
+
id?: string;
|
|
716
|
+
name?: string;
|
|
717
|
+
argsDelta?: string;
|
|
718
|
+
/**
|
|
719
|
+
* True when `argsDelta` is the whole set of arguments, not a
|
|
720
|
+
* fragment. Set for Gemini (its API returns function-call args
|
|
721
|
+
* whole in one chunk) and for cache/replay chunks, which are
|
|
722
|
+
* one-shot too. Omitted or `false` for a genuine fragment from
|
|
723
|
+
* providers that do stream incrementally (OpenAI-compatible,
|
|
724
|
+
* Anthropic, Bedrock).
|
|
725
|
+
*/
|
|
726
|
+
complete?: boolean;
|
|
727
|
+
} | {
|
|
728
|
+
type: 'usage';
|
|
729
|
+
usage: TokenUsage;
|
|
730
|
+
};
|
|
731
|
+
/**
|
|
732
|
+
* What `call()` returns when `stream: true`. `chunks` is for live rendering;
|
|
733
|
+
* `finalResult` resolves to the same validated `T`/`CallWithToolsResult<T>`
|
|
734
|
+
* shape `call()` would have returned had `stream` been omitted, once the
|
|
735
|
+
* stream completes successfully.
|
|
736
|
+
*
|
|
737
|
+
* `chunks` is single-use and supports only one consumer: iterating it more
|
|
738
|
+
* than once, or from more than one place concurrently, shares the same
|
|
739
|
+
* underlying buffered stream rather than replaying or forking it, which can
|
|
740
|
+
* split chunks unpredictably between consumers. Stopping iteration early
|
|
741
|
+
* (e.g. `break`ing out of a `for await`) does not cancel or otherwise
|
|
742
|
+
* signal the underlying stream, the background pump keeps running to
|
|
743
|
+
* completion regardless, buffering any chunks emitted after that point, so
|
|
744
|
+
* `finalResult` still settles normally even if `chunks` is abandoned or
|
|
745
|
+
* never read at all.
|
|
746
|
+
*
|
|
747
|
+
* Unread chunks are buffered internally for the duration of one stream,
|
|
748
|
+
* this is what lets a caller start iterating `chunks` after the stream has
|
|
749
|
+
* already progressed (or finished) and still see everything. That backlog
|
|
750
|
+
* is capped: an unusually large stream whose `chunks` is never read at all
|
|
751
|
+
* has its oldest buffered chunks dropped once the backlog grows past
|
|
752
|
+
* roughly twice a fixed internal limit, trimmed back down to that limit in
|
|
753
|
+
* one batch rather than one-at-a-time, bounding both peak memory and the
|
|
754
|
+
* eviction work itself for that pathological case instead of the array
|
|
755
|
+
* growing (or being trimmed) proportional to the whole stream's output.
|
|
756
|
+
* Ordinary consumption, even started somewhat late, stays far under the
|
|
757
|
+
* limit and is unaffected.
|
|
758
|
+
*/
|
|
759
|
+
interface StreamCallResult<R> {
|
|
760
|
+
chunks: AsyncIterable<StreamChunk>;
|
|
761
|
+
finalResult: Promise<R>;
|
|
762
|
+
}
|
|
763
|
+
/**
|
|
764
|
+
* A `CallParams` variant where streaming is explicitly enabled.
|
|
765
|
+
*
|
|
766
|
+
* Requiring `stream: true` to be statically present allows TypeScript to
|
|
767
|
+
* select the streaming `call()` overload and return `StreamCallResult<...>`
|
|
768
|
+
* instead of the normal, single-shot response type.
|
|
769
|
+
*/
|
|
770
|
+
type StreamEnabledCallParams<T> = CallParams<T> & {
|
|
771
|
+
stream: true;
|
|
772
|
+
};
|
|
773
|
+
/**
|
|
774
|
+
* The adapter-facing, pre-normalization shape a `createStream` client
|
|
775
|
+
* implementation emits, analogous to how `WireMessage`/`WireToolCall`
|
|
776
|
+
* already sit between `CallParams` and each provider's own wire format.
|
|
777
|
+
*/
|
|
778
|
+
type WireStreamChunk = {
|
|
779
|
+
type: 'text-delta';
|
|
780
|
+
delta: string;
|
|
781
|
+
} | {
|
|
782
|
+
type: 'tool_call_delta';
|
|
783
|
+
index: number;
|
|
784
|
+
id?: string;
|
|
785
|
+
name?: string;
|
|
786
|
+
argumentsDelta?: string;
|
|
787
|
+
/** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
|
|
788
|
+
complete?: boolean;
|
|
789
|
+
} | {
|
|
790
|
+
type: 'usage';
|
|
791
|
+
usage: {
|
|
792
|
+
prompt_tokens?: number;
|
|
793
|
+
completion_tokens?: number;
|
|
794
|
+
total_tokens?: number;
|
|
795
|
+
};
|
|
796
|
+
} | {
|
|
797
|
+
/**
|
|
798
|
+
* A provider keep-alive signal with no content of its own (e.g.
|
|
799
|
+
* Anthropic's `ping` events, an SSE comment-line heartbeat).
|
|
800
|
+
* Adapters yield this so the stream loop resets its idle timeout.
|
|
801
|
+
* Never surfaced to callers as a `StreamChunk`.
|
|
802
|
+
*/
|
|
803
|
+
type: 'ping';
|
|
804
|
+
};
|
|
805
|
+
/**
|
|
806
|
+
* Parameters for a cached, streaming LLM call without tool calling.
|
|
807
|
+
*
|
|
808
|
+
* The cached value is `T`, same as `CachedCallParams<T>`, but a miss
|
|
809
|
+
* relays live `chunks` to the caller while the result is being generated,
|
|
810
|
+
* and a hit synthesizes a one-shot `chunks` replay from the cached value
|
|
811
|
+
* (see `VernLLM.cachedCall`'s docs for exactly what that replay looks
|
|
812
|
+
* like).
|
|
813
|
+
*/
|
|
814
|
+
type CachedStreamCallParams<T> = CachedCallInput & {
|
|
815
|
+
call: StreamEnabledCallParams<T>;
|
|
816
|
+
};
|
|
817
|
+
/**
|
|
818
|
+
* Parameters for a cached, streaming LLM call with tool calling enabled.
|
|
819
|
+
*
|
|
820
|
+
* The cached value is the full `CallWithToolsResult<T>`, same as
|
|
821
|
+
* `CachedToolCallParams<T>`, with the same live-chunks-on-miss,
|
|
822
|
+
* replayed-chunks-on-hit behavior as `CachedStreamCallParams<T>`.
|
|
823
|
+
*/
|
|
824
|
+
type CachedStreamToolCallParams<T> = CachedCallInput & {
|
|
825
|
+
call: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>;
|
|
826
|
+
};
|
|
827
|
+
|
|
828
|
+
//#endregion
|
|
829
|
+
//#region src/types/client.d.ts
|
|
830
|
+
//# sourceMappingURL=stream.d.ts.map
|
|
831
|
+
/** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
|
|
832
|
+
interface WireToolCall {
|
|
833
|
+
id: string;
|
|
834
|
+
type: 'function';
|
|
835
|
+
function: {
|
|
836
|
+
name: string;
|
|
837
|
+
/** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
|
|
838
|
+
arguments: string;
|
|
839
|
+
};
|
|
840
|
+
}
|
|
841
|
+
/** One entry of `LLMClient`'s `messages` array, named so callers building it can annotate against it. */
|
|
842
|
+
type WireMessage = {
|
|
843
|
+
role: 'system';
|
|
844
|
+
content: string;
|
|
845
|
+
} | {
|
|
846
|
+
role: 'user';
|
|
847
|
+
content: string | ContentBlock[];
|
|
848
|
+
} | {
|
|
849
|
+
role: 'assistant';
|
|
850
|
+
/** Optional: an assistant turn that only requested tools has no text. */
|
|
851
|
+
content?: string;
|
|
852
|
+
tool_calls?: WireToolCall[];
|
|
853
|
+
} | {
|
|
854
|
+
role: 'tool';
|
|
855
|
+
tool_call_id: string;
|
|
856
|
+
content: string;
|
|
857
|
+
/** Only honored by `fromAnthropic` today (maps to `tool_result.is_error`); other adapters ignore it. */
|
|
858
|
+
is_error?: boolean;
|
|
859
|
+
};
|
|
860
|
+
/** The OpenAI-shaped wire `tool_choice`. */
|
|
861
|
+
type WireToolChoice = 'auto' | 'none' | 'required' | {
|
|
862
|
+
type: 'function';
|
|
863
|
+
function: {
|
|
864
|
+
name: string;
|
|
865
|
+
};
|
|
866
|
+
};
|
|
216
867
|
/**
|
|
217
868
|
* Minimal shape compatible with the OpenAI SDKs chat.completions.create,
|
|
218
869
|
* so consumers can pass an OpenAI client directly
|
|
@@ -226,7 +877,7 @@ interface LLMClient {
|
|
|
226
877
|
completions: {
|
|
227
878
|
create(params: {
|
|
228
879
|
model: string;
|
|
229
|
-
temperature
|
|
880
|
+
temperature?: number;
|
|
230
881
|
max_tokens: number;
|
|
231
882
|
response_format?: {
|
|
232
883
|
type: 'json_object';
|
|
@@ -241,19 +892,29 @@ interface LLMClient {
|
|
|
241
892
|
};
|
|
242
893
|
/** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
|
|
243
894
|
reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
895
|
+
/** Tools the model may call, OpenAI's `function`-wrapped shape. */
|
|
896
|
+
tools?: Array<{
|
|
897
|
+
type: 'function';
|
|
898
|
+
function: {
|
|
899
|
+
name: string;
|
|
900
|
+
description: string;
|
|
901
|
+
parameters: Record<string, unknown>;
|
|
902
|
+
};
|
|
250
903
|
}>;
|
|
904
|
+
tool_choice?: WireToolChoice;
|
|
905
|
+
/**
|
|
906
|
+
* Wire-format messages. Breaking change for custom adapters:
|
|
907
|
+
* implementations must handle tool messages and assistant tool_calls.
|
|
908
|
+
* Exhaustive switches over only system/user/assistant roles may no longer compile.
|
|
909
|
+
*/
|
|
910
|
+
messages: WireMessage[];
|
|
251
911
|
}, options: {
|
|
252
912
|
signal: AbortSignal;
|
|
253
913
|
}): Promise<{
|
|
254
914
|
choices?: Array<{
|
|
255
915
|
message?: {
|
|
256
916
|
content?: string | null;
|
|
917
|
+
tool_calls?: WireToolCall[];
|
|
257
918
|
};
|
|
258
919
|
}>;
|
|
259
920
|
usage?: {
|
|
@@ -262,54 +923,22 @@ interface LLMClient {
|
|
|
262
923
|
total_tokens?: number;
|
|
263
924
|
};
|
|
264
925
|
}>;
|
|
926
|
+
/**
|
|
927
|
+
* Optional. Required only for `stream: true` calls. Adapters/clients
|
|
928
|
+
* that don't implement this make `stream: true` throw a clear
|
|
929
|
+
* `LLMError('validation')` rather than a confusing runtime failure.
|
|
930
|
+
* Takes the same request shape as `create`, minus the response type.
|
|
931
|
+
*/
|
|
932
|
+
createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
|
|
933
|
+
signal: AbortSignal;
|
|
934
|
+
}): AsyncIterable<WireStreamChunk>;
|
|
265
935
|
};
|
|
266
936
|
};
|
|
267
937
|
}
|
|
268
938
|
|
|
269
|
-
//#endregion
|
|
270
|
-
//#region src/circuitBreaker.d.ts
|
|
271
|
-
//# sourceMappingURL=client.d.ts.map
|
|
272
|
-
interface CircuitBreakerOptions {
|
|
273
|
-
/** Consecutive failures before the circuit opens, default 5 */
|
|
274
|
-
threshold?: number;
|
|
275
|
-
/** How long the circuit stays open before allowing a trial request, in ms. Default 30000 */
|
|
276
|
-
cooldownMs?: number;
|
|
277
|
-
}
|
|
278
|
-
type CircuitState = 'closed' | 'open' | 'half-open';
|
|
279
|
-
/**
|
|
280
|
-
* Per retry VernLLM-instance circuit breaker. Tracks consecutive failures across
|
|
281
|
-
* calls. Once the threshold is hit, short-circuits new calls with an
|
|
282
|
-
* LLMError('circuit_open') instead of hitting the provider, until the
|
|
283
|
-
* cooldown elapses and a single trial call is allowed through
|
|
284
|
-
*/
|
|
285
|
-
declare class CircuitBreaker {
|
|
286
|
-
private state;
|
|
287
|
-
private consecutiveFailures;
|
|
288
|
-
private openedAt;
|
|
289
|
-
private threshold;
|
|
290
|
-
private cooldownMs;
|
|
291
|
-
/**
|
|
292
|
-
* True while a single half-open trial call is in flight. Guards against
|
|
293
|
-
* multiple concurrent callers all treating themselves as "the" trial once
|
|
294
|
-
* the cooldown elapses
|
|
295
|
-
*/
|
|
296
|
-
private trialInFlight;
|
|
297
|
-
constructor(options?: CircuitBreakerOptions);
|
|
298
|
-
/**
|
|
299
|
-
* Throws if the circuit is open and the cooldown hasn't elapsed, or if
|
|
300
|
-
* the circuit is half-open and a trial call is already in flight.
|
|
301
|
-
* Otherwise, if the circuit just became eligible for a trial (cooldown
|
|
302
|
-
* elapsed, or half-open with no trial currently running), this call
|
|
303
|
-
* becomes that trial
|
|
304
|
-
*/
|
|
305
|
-
assertClosed(): void;
|
|
306
|
-
recordSuccess(): void;
|
|
307
|
-
recordFailure(): void;
|
|
308
|
-
getState(): CircuitState;
|
|
309
|
-
}
|
|
310
|
-
|
|
311
939
|
//#endregion
|
|
312
940
|
//#region src/logger.d.ts
|
|
941
|
+
//# sourceMappingURL=client.d.ts.map
|
|
313
942
|
interface Logger {
|
|
314
943
|
debug(message: string): void;
|
|
315
944
|
warn(message: string): void;
|
|
@@ -328,22 +957,125 @@ declare class ConsoleLogger implements Logger {
|
|
|
328
957
|
}
|
|
329
958
|
|
|
330
959
|
//#endregion
|
|
331
|
-
//#region src/types/
|
|
960
|
+
//#region src/types/events.d.ts
|
|
332
961
|
//# sourceMappingURL=logger.d.ts.map
|
|
962
|
+
/**
|
|
963
|
+
* Reports what happened during a call. Fire and forget, mirroring
|
|
964
|
+
* `onUsage`: the return value is never read and a throwing handler cannot
|
|
965
|
+
* change what the call does, only what gets reported about it.
|
|
966
|
+
*/
|
|
967
|
+
type VernLLMEvent = {
|
|
968
|
+
kind: 'retry';
|
|
969
|
+
requestId: string;
|
|
970
|
+
provider: string;
|
|
971
|
+
/** The model actually resolved for this call (honors a per-call `model` override). */
|
|
972
|
+
model: string;
|
|
973
|
+
/** The 1-based retry ordinal (the 1st retry is `1`, not the overall attempt count). */
|
|
974
|
+
attempt: number;
|
|
975
|
+
maxRetries: number;
|
|
976
|
+
delayMs: number;
|
|
977
|
+
retryAfterHonored: boolean;
|
|
978
|
+
error: LLMError;
|
|
979
|
+
} | {
|
|
980
|
+
kind: 'circuit_state';
|
|
981
|
+
provider: string;
|
|
982
|
+
/**
|
|
983
|
+
* The model of the call that triggered this specific transition
|
|
984
|
+
* (whatever was passed to the `assertClosed`/`recordSuccess`/
|
|
985
|
+
* `recordFailure` call that caused it), not a property of the
|
|
986
|
+
* circuit itself: the breaker still counts failures across every
|
|
987
|
+
* model together, so a threshold crossing can be the sum of
|
|
988
|
+
* several different models' failures even though only the
|
|
989
|
+
* triggering call's `model` is reported here.
|
|
990
|
+
*/
|
|
991
|
+
model: string;
|
|
992
|
+
from: CircuitState;
|
|
993
|
+
to: CircuitState;
|
|
994
|
+
consecutiveFailures: number;
|
|
995
|
+
} | {
|
|
996
|
+
kind: 'fallback';
|
|
997
|
+
requestId: string;
|
|
998
|
+
/** Provider name of the target that just failed. */
|
|
999
|
+
from: string;
|
|
1000
|
+
/** Provider name of the target about to be tried next. */
|
|
1001
|
+
to: string;
|
|
1002
|
+
/** `-1` for the primary target, otherwise the index into `fallback`. */
|
|
1003
|
+
fromIndex: number;
|
|
1004
|
+
toIndex: number;
|
|
1005
|
+
/** The normalized error that caused `from` to be abandoned. */
|
|
1006
|
+
error: LLMError;
|
|
1007
|
+
/** Time spent on `from`, including its own retries, before giving up. */
|
|
1008
|
+
elapsedMs: number;
|
|
1009
|
+
} | {
|
|
1010
|
+
kind: 'rate_limited';
|
|
1011
|
+
requestId: string;
|
|
1012
|
+
provider: string;
|
|
1013
|
+
/** The model actually resolved for this call (honors a per-call `model` override). */
|
|
1014
|
+
model: string;
|
|
1015
|
+
/** How long this attempt sat queued for capacity before it was let through. */
|
|
1016
|
+
waitedMs: number;
|
|
1017
|
+
/** Which configured bucket was blocking this attempt just before it cleared. */
|
|
1018
|
+
reason: 'concurrency' | 'rpm' | 'tpm';
|
|
1019
|
+
};
|
|
1020
|
+
type OnEvent = (event: VernLLMEvent) => void;
|
|
1021
|
+
|
|
1022
|
+
//#endregion
|
|
1023
|
+
//#region src/types/options.d.ts
|
|
1024
|
+
//# sourceMappingURL=events.d.ts.map
|
|
333
1025
|
interface VernLLMOptions {
|
|
334
1026
|
client: LLMClient;
|
|
335
1027
|
model: string;
|
|
1028
|
+
/**
|
|
1029
|
+
* Label for this provider in usage (`TokenUsage.provider`) and events.
|
|
1030
|
+
* Default `'primary'`.
|
|
1031
|
+
*/
|
|
1032
|
+
name?: string;
|
|
336
1033
|
/** Max retries after the first attempt. Default 1 (2 attempts total) */
|
|
337
1034
|
maxRetries?: number;
|
|
338
1035
|
/** Per-attempt timeout in ms. Default 25000 */
|
|
339
1036
|
timeoutMs?: number;
|
|
1037
|
+
/**
|
|
1038
|
+
* For `stream: true` calls: max gap allowed between chunks once the
|
|
1039
|
+
* stream has opened, in ms. Resets on every chunk, including keep-alive
|
|
1040
|
+
* pings. `timeoutMs` only covers opening the stream and its first
|
|
1041
|
+
* chunk; this covers every gap after that. Also counts as a
|
|
1042
|
+
* circuit-breaker failure, unlike other mid-stream errors, since a
|
|
1043
|
+
* provider that streams one chunk then stalls should still trip it.
|
|
1044
|
+
* Default 30000. Pass 0 or negative to disable.
|
|
1045
|
+
*/
|
|
1046
|
+
chunkIdleTimeoutMs?: number;
|
|
340
1047
|
/** Base delay for exponential backoff in ms. Default 500 */
|
|
341
1048
|
baseDelayMs?: number;
|
|
342
1049
|
/** Default max_tokens for calls that don't override it. Default 1000 */
|
|
343
1050
|
defaultMaxTokens?: number;
|
|
344
|
-
/**
|
|
345
|
-
*
|
|
1051
|
+
/**
|
|
1052
|
+
* Default temperature for calls that don't override it. Default 0.2, not
|
|
1053
|
+
* the provider's own default. Pass `null` to omit `temperature` from the
|
|
1054
|
+
* request entirely, so the provider applies its own default instead.
|
|
1055
|
+
*/
|
|
1056
|
+
defaultTemperature?: number | null;
|
|
1057
|
+
/**
|
|
1058
|
+
* Enables debug logging of raw model output (logs up to 800 chars of each
|
|
1059
|
+
* response) and provider errors. Off by default. Only controls the
|
|
1060
|
+
* default `ConsoleLogger`: when a custom `logger` is supplied instead,
|
|
1061
|
+
* that logger's own `debug()` implementation decides whether messages
|
|
1062
|
+
* are emitted, and this option has no effect on it.
|
|
1063
|
+
*/
|
|
346
1064
|
debug?: boolean;
|
|
1065
|
+
/**
|
|
1066
|
+
* Applied before every internal `logger.debug()` call: the raw output
|
|
1067
|
+
* logged on success, and the provider error logged on a failed call or
|
|
1068
|
+
* a failed stream open. This is the one piece of logging an app can't
|
|
1069
|
+
* intercept itself, since it's a direct call into `logger.debug`
|
|
1070
|
+
* rather than something routed through `onEvent`/`onUsage`; anything
|
|
1071
|
+
* caught elsewhere (events, `LLMError.cause`) already passes through
|
|
1072
|
+
* the app's own callback and can be redacted there instead. Runs
|
|
1073
|
+
* before `logger.debug()` regardless of whether that call ends up
|
|
1074
|
+
* emitting anything, so with a custom `logger`, `redact` still applies
|
|
1075
|
+
* even without `debug: true`; see `debug` for why. Default: identity
|
|
1076
|
+
* (no redaction).
|
|
1077
|
+
*/
|
|
1078
|
+
redact?: (text: string) => string;
|
|
347
1079
|
/** Cache adapter for cachedCall. Defaults to an in-memory adapter */
|
|
348
1080
|
cache?: CacheAdapter;
|
|
349
1081
|
/** HTTP status codes that should fail fast without retrying. Default [400, 401, 403, 404, 422] */
|
|
@@ -352,6 +1084,18 @@ interface VernLLMOptions {
|
|
|
352
1084
|
parseJson?: (content: string) => unknown;
|
|
353
1085
|
/** Called after every successful call with token usage, if the provider reports it */
|
|
354
1086
|
onUsage?: OnUsage;
|
|
1087
|
+
/**
|
|
1088
|
+
* Called when a provider response arrives but VernLLM's own post-processing
|
|
1089
|
+
* then fails, after usage data was already present in that response.
|
|
1090
|
+
* Separate from `onUsage`, which only fires on full success.
|
|
1091
|
+
*
|
|
1092
|
+
* For non-streaming calls, never fires for transport failures (timeout,
|
|
1093
|
+
* network error, non-retryable status), since no response means no usage
|
|
1094
|
+
* to report. For streaming calls, this is not guaranteed: a stream can
|
|
1095
|
+
* deliver a usage chunk and then fail later (e.g. an idle timeout waiting
|
|
1096
|
+
* for the final close), in which case this does fire.
|
|
1097
|
+
*/
|
|
1098
|
+
onUsageFailure?: OnUsageFailure;
|
|
355
1099
|
/** Injectable logger. Defaults to a console-based logger gated by `debug` */
|
|
356
1100
|
logger?: Logger;
|
|
357
1101
|
/**
|
|
@@ -360,136 +1104,289 @@ interface VernLLMOptions {
|
|
|
360
1104
|
* Pass `true` for defaults, or an options object to tune threshold/cooldown
|
|
361
1105
|
*/
|
|
362
1106
|
circuitBreaker?: boolean | CircuitBreakerOptions;
|
|
1107
|
+
/**
|
|
1108
|
+
* Reports retries and circuit-breaker state transitions as they happen.
|
|
1109
|
+
* Fire and forget: a throwing handler is caught and logged, and its
|
|
1110
|
+
* return value is never read, so it cannot influence the call.
|
|
1111
|
+
*/
|
|
1112
|
+
onEvent?: OnEvent;
|
|
1113
|
+
/**
|
|
1114
|
+
* Client-side rate limiting. Queues calls locally to stay under the
|
|
1115
|
+
* configured requests/tokens-per-minute or concurrency caps, instead of
|
|
1116
|
+
* letting the provider reject them. Independent of the `Retry-After`
|
|
1117
|
+
* handling already applied to a provider 429: this avoids tripping the
|
|
1118
|
+
* limit in the first place. Omit for unlimited (the default).
|
|
1119
|
+
*/
|
|
1120
|
+
rateLimit?: RateLimitOptions;
|
|
1121
|
+
/**
|
|
1122
|
+
* Ordered targets tried after the primary, in order, once it (and its
|
|
1123
|
+
* own retries) is exhausted or abandoned. Order is the policy: VernLLM
|
|
1124
|
+
* never reorders, scores, or selects between targets. Each target keeps
|
|
1125
|
+
* its own retry state, circuit breaker, and rate limiter, independent
|
|
1126
|
+
* of every other target's. A single `FallbackTarget` is equivalent to
|
|
1127
|
+
* `[target]`.
|
|
1128
|
+
*/
|
|
1129
|
+
fallback?: FallbackTarget | FallbackTarget[];
|
|
1130
|
+
/**
|
|
1131
|
+
* Decides what happens after a target fails: `'next'` to move on to
|
|
1132
|
+
* the following target (or throw, if it was the last one), `'stop'` to
|
|
1133
|
+
* give up immediately without trying any remaining targets. Called
|
|
1134
|
+
* once per failed target, after that target's own retries are
|
|
1135
|
+
* exhausted or abandoned early, so `'retry'` is never a valid return
|
|
1136
|
+
* here. Defaults to `defaultFallbackOn`, which stops on
|
|
1137
|
+
* parse/validation/aborted/quota errors and on tool-contract failures
|
|
1138
|
+
* (the model ignoring the request, not the provider being unhealthy),
|
|
1139
|
+
* and moves on for everything else.
|
|
1140
|
+
*/
|
|
1141
|
+
fallbackOn?: FallbackOn;
|
|
363
1142
|
}
|
|
364
1143
|
|
|
365
1144
|
//#endregion
|
|
366
1145
|
//#region src/vernLLM.d.ts
|
|
367
1146
|
//# sourceMappingURL=options.d.ts.map
|
|
368
1147
|
/**
|
|
369
|
-
* A resilient layer around an LLM chat completions client
|
|
1148
|
+
* A resilient layer around an LLM chat completions client. This is VernLLM!
|
|
370
1149
|
*
|
|
371
|
-
* Adds retry with backoff
|
|
372
|
-
* JSON parsing with optional schema validation, usage
|
|
373
|
-
* optional response cache
|
|
374
|
-
* defaults.
|
|
1150
|
+
* Adds retry with backoff and jitter, per-attempt timeouts, an optional
|
|
1151
|
+
* circuit breaker, JSON parsing with optional schema validation, usage
|
|
1152
|
+
* tracking, and an optional response cache. All configurable, all opt-in
|
|
1153
|
+
* beyond sensible defaults.
|
|
375
1154
|
*/
|
|
376
1155
|
declare class VernLLM {
|
|
377
|
-
private readonly client;
|
|
378
|
-
private readonly model;
|
|
379
|
-
private readonly maxRetries;
|
|
380
|
-
private readonly timeoutMs;
|
|
381
|
-
private readonly baseDelayMs;
|
|
382
|
-
private readonly defaultMaxTokens;
|
|
383
|
-
private readonly cache;
|
|
384
|
-
private readonly nonRetryableStatus;
|
|
385
|
-
private readonly inFlight;
|
|
386
|
-
private readonly parseJson;
|
|
387
|
-
private readonly onUsage?;
|
|
388
1156
|
private readonly logger;
|
|
389
|
-
private readonly breaker?;
|
|
390
1157
|
/**
|
|
391
|
-
*
|
|
392
|
-
*
|
|
393
|
-
*
|
|
394
|
-
* `
|
|
1158
|
+
* One `CallExecutor` per provider target: index 0 is the primary,
|
|
1159
|
+
* everything after it is a `fallback` target, in the order declared.
|
|
1160
|
+
* Each owns its own request building, retry/timeout, circuit breaker,
|
|
1161
|
+
* and rate limiter. `call()` walks this array in `runFallbackChain`,
|
|
1162
|
+
* moving to the next entry only when `fallbackOn` says to.
|
|
395
1163
|
*/
|
|
396
|
-
|
|
397
|
-
/**
|
|
398
|
-
private
|
|
1164
|
+
private readonly executors;
|
|
1165
|
+
/** Decides whether a failed target is followed by the next one or the chain stops. See `VernLLMOptions['fallbackOn']`. */
|
|
1166
|
+
private readonly fallbackOn;
|
|
1167
|
+
/** Reports a `'fallback'` event when the chain moves to the next target. Shared `onEvent` plumbing, same as every executor's. */
|
|
1168
|
+
private readonly reportEvent;
|
|
399
1169
|
/**
|
|
400
|
-
*
|
|
401
|
-
*
|
|
402
|
-
*
|
|
403
|
-
* with a normalized LLMError.
|
|
404
|
-
*
|
|
405
|
-
* @param params - System/user content plus per-call overrides (model,
|
|
406
|
-
* temperature, jsonMode, schema, signal, etc). See `CallParams`.
|
|
407
|
-
* @returns The parsed (and optionally schema-validated) response, or the
|
|
408
|
-
* raw string content when `jsonMode` is false and no `jsonSchema` is set.
|
|
1170
|
+
* Owns cache key resolution, cache reads/writes, and in-flight
|
|
1171
|
+
* coalescing for `cachedCall()`. Independent of `executor`: it only
|
|
1172
|
+
* ever calls back into `this.call()` as an opaque function.
|
|
409
1173
|
*/
|
|
410
|
-
|
|
411
|
-
/** Runs `fn`, retrying with backoff according to `shouldRetry`. */
|
|
412
|
-
private retryWithBackoff;
|
|
1174
|
+
private readonly cacheOrchestrator;
|
|
413
1175
|
/**
|
|
414
|
-
*
|
|
415
|
-
*
|
|
416
|
-
*
|
|
1176
|
+
* @param options Client, model, and tunables. Defaults: `maxRetries` 1,
|
|
1177
|
+
* `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
|
|
1178
|
+
* `defaultTemperature` 0.2, `cache` an in-memory adapter,
|
|
1179
|
+
* `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
|
|
417
1180
|
*/
|
|
418
|
-
|
|
1181
|
+
constructor(options: VernLLMOptions);
|
|
1182
|
+
/** Logs a failed refundUsage attempt via the configured logger. */
|
|
1183
|
+
private logRefundError;
|
|
419
1184
|
/**
|
|
420
|
-
*
|
|
421
|
-
*
|
|
1185
|
+
* Walks `this.executors` in order, running `attempt` against each until
|
|
1186
|
+
* one succeeds or every target has failed. `run` on a lone target
|
|
1187
|
+
* (no `fallback` configured) throws exactly what it throws today: the
|
|
1188
|
+
* loop's single iteration path is unchanged from pre-fallback behavior.
|
|
1189
|
+
*
|
|
1190
|
+
* For streaming, `attempt` is `executor.runStream`, whose own retries
|
|
1191
|
+
* only cover *opening* the stream (see `CallExecutor.runStream`). A
|
|
1192
|
+
* mid-stream failure surfaces through `finalResult` after this function
|
|
1193
|
+
* has already returned, so it's never seen here and never falls over,
|
|
1194
|
+
* per the streaming limitation: splicing a second model's output into a
|
|
1195
|
+
* response the consumer has already partially rendered would corrupt
|
|
1196
|
+
* it.
|
|
422
1197
|
*/
|
|
423
|
-
private
|
|
424
|
-
/** Applies per-call defaults and shapes params into the client's request object. */
|
|
425
|
-
private buildRequestPayload;
|
|
1198
|
+
private runFallbackChain;
|
|
426
1199
|
/**
|
|
427
|
-
*
|
|
428
|
-
*
|
|
429
|
-
*
|
|
430
|
-
*
|
|
1200
|
+
* Makes a single logical LLM call, retrying on failure per the configured
|
|
1201
|
+
* policy. Fails fast if the breaker is open or the signal is already
|
|
1202
|
+
* aborted. Rejects with a normalized LLMError on exhausted retries.
|
|
1203
|
+
*
|
|
1204
|
+
* When `tools` is set, returns a `CallWithToolsResult<T>` instead of `T`:
|
|
1205
|
+
* `{ type: 'content', content }` or `{ type: 'tool_calls', toolCalls,
|
|
1206
|
+
* content? }`. VernLLM never executes tools; run them yourself and
|
|
1207
|
+
* continue via `history` (see `ConversationTurn`). Mutually exclusive
|
|
1208
|
+
* with `jsonSchema`/`schema`.
|
|
1209
|
+
*
|
|
1210
|
+
* TypeScript only picks the tools-aware overload when `tools` is
|
|
1211
|
+
* statically present on `params`. If set conditionally on a plain
|
|
1212
|
+
* `CallParams<T>`, use `isToolCallResult()` to check the shape at
|
|
1213
|
+
* runtime instead. See the Tool Calling docs for details.
|
|
1214
|
+
*
|
|
1215
|
+
* The same static-vs-dynamic caveat applies to `stream`: TypeScript only
|
|
1216
|
+
* selects the streaming overload (returning `StreamCallResult<...>`) when
|
|
1217
|
+
* `stream: true` is statically present on `params`. A `stream` value set
|
|
1218
|
+
* conditionally on a plain `CallParams<T>` still resolves to `Promise<T>`
|
|
1219
|
+
* (or `Promise<CallWithToolsResult<T>>`) at the type level even though
|
|
1220
|
+
* the actual runtime result is the `{ chunks, finalResult }` streaming
|
|
1221
|
+
* shape whenever `stream` evaluates to `true`, callers doing this should
|
|
1222
|
+
* narrow/cast accordingly rather than relying on the static return type.
|
|
1223
|
+
*
|
|
1224
|
+
* @param params System/user content plus per-call overrides. See `CallParams`.
|
|
1225
|
+
* @returns Without `tools` or `stream`: the parsed response, or raw
|
|
1226
|
+
* string if `jsonMode` is false. With `tools`: a `CallWithToolsResult<T>`.
|
|
1227
|
+
* With `stream: true` (statically): a `{ chunks, finalResult }`
|
|
1228
|
+
* `StreamCallResult`, `finalResult` resolving to whichever of the above
|
|
1229
|
+
* shapes applies once the stream completes. See `StreamCallResult`.
|
|
431
1230
|
*/
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
private parseAndValidate;
|
|
1231
|
+
call<T = unknown>(params: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
|
|
1232
|
+
call<T = unknown>(params: StreamEnabledCallParams<T>): Promise<StreamCallResult<T>>;
|
|
1233
|
+
call<T = unknown>(params: ToolEnabledCallParams<T>): Promise<CallWithToolsResult<T>>;
|
|
1234
|
+
call<T = unknown>(params: CallParams<T>): Promise<T>;
|
|
437
1235
|
/**
|
|
438
|
-
*
|
|
439
|
-
*
|
|
440
|
-
*
|
|
441
|
-
*
|
|
1236
|
+
* Thin delegator kept private on `VernLLM` (rather than only existing on
|
|
1237
|
+
* `CacheOrchestrator`) since it's the one caching primitive exercised
|
|
1238
|
+
* directly by white-box tests, independent of the public `cachedCall()`
|
|
1239
|
+
* surface.
|
|
442
1240
|
*/
|
|
443
|
-
private
|
|
444
|
-
/** Decides whether a failed attempt is worth retrying. */
|
|
445
|
-
private shouldRetry;
|
|
1241
|
+
private runCached;
|
|
446
1242
|
/**
|
|
447
1243
|
* Removes a cached response by key when the configured cache adapter
|
|
448
1244
|
* supports deletion. Cache invalidation is the caller's responsibility;
|
|
449
1245
|
* only the application knows when cached data is stale.
|
|
450
1246
|
*
|
|
451
|
-
* @param key
|
|
1247
|
+
* @param key The raw cache key (resolved through the adapter's
|
|
452
1248
|
* `resolveKey`, if any, before deletion).
|
|
453
1249
|
*/
|
|
454
1250
|
deleteCache(key: string): Promise<void>;
|
|
455
1251
|
/**
|
|
456
|
-
* Cache wrapper
|
|
457
|
-
* same `cacheKey` share a single in-flight call, avoiding cache stampedes.
|
|
458
|
-
*
|
|
459
|
-
* @param params - `cacheKey`, `ttl`, `fn` (the work to run on a cache
|
|
460
|
-
* miss, typically `() => this.call(...)`), and optional
|
|
461
|
-
* `reserveUsage`/`refundUsage`/`signal`. See `CachedCallParams`.
|
|
462
|
-
* @returns The cached value on a hit, or the result of `fn()` on a miss.
|
|
463
|
-
*/
|
|
464
|
-
cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
|
|
465
|
-
/** Starts the shared fn() call for a cache miss and tracks it in the in-flight map until it settles. */
|
|
466
|
-
private registerTrigger;
|
|
467
|
-
/** Runs `fn` and writes its result to the cache. */
|
|
468
|
-
private runAndCache;
|
|
469
|
-
/** Logs a failed refundUsage attempt via the configured logger. */
|
|
470
|
-
private logRefundError;
|
|
471
|
-
/**
|
|
472
|
-
* Convenience wrapper composing `call` + `cachedCall`, so cached LLM calls
|
|
1252
|
+
* Cache wrapper composing `call` + caching, so cached LLM calls
|
|
473
1253
|
* automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
|
|
474
|
-
* `refundUsage` are read from the top-level params only.
|
|
1254
|
+
* `refundUsage` are read from the top-level params only. Concurrent misses
|
|
1255
|
+
* for the same `cacheKey` share a single in-flight call, avoiding cache
|
|
1256
|
+
* stampedes. Supports `stream: true` and `tools` in any combination.
|
|
475
1257
|
*
|
|
476
|
-
*
|
|
477
|
-
*
|
|
1258
|
+
* When `call.tools` is set, this caches the whole `CallWithToolsResult`,
|
|
1259
|
+
* including `tool_calls` results, not just final answers. Whether
|
|
1260
|
+
* that's appropriate depends on the tool: caching "the model decided to
|
|
1261
|
+
* call get_weather" is usually fine to reuse briefly, but caching a
|
|
1262
|
+
* decision made under permissions or account state that can change
|
|
1263
|
+
* between calls is not. Use a short `ttl` or a separate `cacheKey` for
|
|
1264
|
+
* such tools if this distinction matters.
|
|
1265
|
+
*
|
|
1266
|
+
* There is no public way to cache an arbitrary non-LLM function through
|
|
1267
|
+
* `VernLLM`. This method always composes with `call()`. For
|
|
1268
|
+
* general-purpose caching unrelated to an LLM call, use a dedicated
|
|
1269
|
+
* caching library at the application level instead.
|
|
1270
|
+
*
|
|
1271
|
+
* @param params `cacheKey`, `ttl`, and optional
|
|
1272
|
+
* `reserveUsage`/`refundUsage`/`signal`, plus `call`, the `CallParams`
|
|
1273
|
+
* (optionally with `tools` and/or `stream`) to pass through to
|
|
1274
|
+
* `this.call(...)`. The top-level `signal` governs the cached operation
|
|
1275
|
+
* and its usage hooks only; to also abort the underlying provider
|
|
1276
|
+
* request, set `signal` inside `call`.
|
|
478
1277
|
* @returns The cached value on a hit, or the freshly-called result on a miss.
|
|
479
1278
|
*/
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
1279
|
+
cachedCall<T>(params: CachedStreamToolCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
|
|
1280
|
+
cachedCall<T>(params: CachedStreamCallParams<T>): Promise<StreamCallResult<T>>;
|
|
1281
|
+
cachedCall<T>(params: CachedToolCallParams<T>): Promise<CallWithToolsResult<T>>;
|
|
1282
|
+
cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
|
|
483
1283
|
/**
|
|
1284
|
+
* @param model With `circuitBreaker.isolateByModel` on, returns that
|
|
1285
|
+
* model's own circuit state instead of the shared one. Ignored
|
|
1286
|
+
* otherwise. Omit for the shared circuit (the default) or, under
|
|
1287
|
+
* isolation, the state of calls that didn't resolve a model.
|
|
484
1288
|
* @returns The current circuit breaker state (`'closed' | 'open' |
|
|
485
1289
|
* 'half-open'`), or undefined if no circuit breaker was configured.
|
|
486
1290
|
*/
|
|
487
|
-
getCircuitState():
|
|
1291
|
+
getCircuitState(model?: string): CircuitState | undefined;
|
|
1292
|
+
/**
|
|
1293
|
+
* @param model With `circuitBreaker.isolateByModel` on, returns each
|
|
1294
|
+
* target's circuit state for that model instead of its shared state.
|
|
1295
|
+
* Ignored otherwise. Omit for the shared circuit (the default) or, under
|
|
1296
|
+
* isolation, the state of calls that didn't resolve a model.
|
|
1297
|
+
* @returns The current circuit state for every target in declaration
|
|
1298
|
+
* order, including the primary and all fallback targets. Each entry
|
|
1299
|
+
* includes the target's provider name, chain index, whether it is a
|
|
1300
|
+
* fallback, and its circuit state, or undefined if that target has no
|
|
1301
|
+
* circuit breaker configured.
|
|
1302
|
+
*/
|
|
1303
|
+
getCircuitStates(model?: string): TargetCircuitState[];
|
|
488
1304
|
}
|
|
489
1305
|
|
|
490
1306
|
//#endregion
|
|
491
|
-
//#region src/adapters/
|
|
1307
|
+
//#region src/adapters/internal/sse.d.ts
|
|
492
1308
|
//# sourceMappingURL=vernLLM.d.ts.map
|
|
1309
|
+
/**
|
|
1310
|
+
* Parses a Server-Sent-Events byte/text stream into the JSON payload of
|
|
1311
|
+
* each `data:` frame, in arrival order. Generic over transport: works with
|
|
1312
|
+
* anything that hands back progressively-arriving `Uint8Array` or `string`
|
|
1313
|
+
* chunks via async iteration: native `fetch`'s `response.body` (wrapped
|
|
1314
|
+
* to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
|
|
1315
|
+
* Node `Readable` (already async-iterable, no wrapping needed), etc, so
|
|
1316
|
+
* this framing layer doesn't care which transport produced the bytes.
|
|
1317
|
+
*
|
|
1318
|
+
* Follows the SSE spec's frame-delimiting rules closely enough for LLM
|
|
1319
|
+
* streaming responses: frames are separated by a blank line, each frame
|
|
1320
|
+
* may carry one or more `data:` lines (joined with `\n` per spec when
|
|
1321
|
+
* there's more than one), `:`-prefixed lines are comments and ignored, and
|
|
1322
|
+
* other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
|
|
1323
|
+
* only needs the payload. A frame whose data is exactly `[DONE]` (the
|
|
1324
|
+
* sentinel several providers, notably OpenAI, send to mark stream end)
|
|
1325
|
+
* ends iteration without yielding it.
|
|
1326
|
+
*
|
|
1327
|
+
* Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
|
|
1328
|
+
* to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
|
|
1329
|
+
* alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
|
|
1330
|
+
* stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
|
|
1331
|
+
* lines.
|
|
1332
|
+
*
|
|
1333
|
+
* Malformed JSON in a frame throws `LLMError('parse')`, consistent with
|
|
1334
|
+
* how malformed JSON is handled elsewhere in VernLLM.
|
|
1335
|
+
*/
|
|
1336
|
+
declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
|
|
1337
|
+
/**
|
|
1338
|
+
* Sentinel yielded by `parseSseStream` for a comment-only frame (no
|
|
1339
|
+
* `data:` payload), the mechanism providers use for SSE keep-alive
|
|
1340
|
+
* pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
|
|
1341
|
+
* alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
|
|
1342
|
+
*/
|
|
1343
|
+
declare const SSE_PING: unique symbol;
|
|
1344
|
+
|
|
1345
|
+
//#endregion
|
|
1346
|
+
//#region src/adapters/internal/imageFormat.d.ts
|
|
1347
|
+
//# sourceMappingURL=sse.d.ts.map
|
|
1348
|
+
/**
|
|
1349
|
+
* MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
|
|
1350
|
+
* the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
|
|
1351
|
+
* Converse all natively support, so a `ContentBlock[]` that validates for
|
|
1352
|
+
* one provider validates for all of them.
|
|
1353
|
+
*/
|
|
1354
|
+
declare const SUPPORTED_IMAGE_MIME_TYPES: readonly ["image/png", "image/jpeg", "image/gif", "image/webp"];
|
|
1355
|
+
type SupportedImageMimeType = (typeof SUPPORTED_IMAGE_MIME_TYPES)[number];
|
|
1356
|
+
|
|
1357
|
+
//#endregion
|
|
1358
|
+
//#region src/adapters/internal/nativeStructuredOutput.d.ts
|
|
1359
|
+
/**
|
|
1360
|
+
* Validates an `ImageBlock.mimeType` against the shared supported set.
|
|
1361
|
+
* Throws a non-retryable `LLMError('validation')`, since an unsupported
|
|
1362
|
+
* mimeType is a permanent failure, retrying the same input can't fix it,
|
|
1363
|
+
* the same way a schema-validation or JSON-parse failure isn't retried.
|
|
1364
|
+
*/
|
|
1365
|
+
|
|
1366
|
+
/**
|
|
1367
|
+
* A static allow-list or predicate naming which models support native,
|
|
1368
|
+
* schema-constrained output as its own request field — Anthropic's
|
|
1369
|
+
* `output_config.format`, Bedrock's `outputConfig.textFormat` — separate
|
|
1370
|
+
* from `tools`/`tool_choice`, so it can be combined with real,
|
|
1371
|
+
* caller-supplied `tools` in the same request.
|
|
1372
|
+
*
|
|
1373
|
+
* There is no built-in default list here. Which models support this is
|
|
1374
|
+
* Anthropic's and Bedrock's call to make, not this package's, and it
|
|
1375
|
+
* changes over time; hardcoding a guessed list would risk silently
|
|
1376
|
+
* routing a request onto a field a given model doesn't actually support,
|
|
1377
|
+
* trading a clear `LLMError('validation')` for a confusing error from the
|
|
1378
|
+
* provider instead. So this is opt-in: pass the model IDs you've verified
|
|
1379
|
+
* against the provider's own docs (or a predicate). Left unset, no model
|
|
1380
|
+
* is treated as native-capable, `jsonSchema` keeps using the older
|
|
1381
|
+
* forced-single-tool-call emulation, and `tools` + `jsonSchema` together
|
|
1382
|
+
* is rejected, exactly this package's behavior before native support was
|
|
1383
|
+
* added.
|
|
1384
|
+
*/
|
|
1385
|
+
type ModelCapabilityOverride = string[] | ((model: string) => boolean);
|
|
1386
|
+
|
|
1387
|
+
//#endregion
|
|
1388
|
+
//#region src/adapters/anthropic.d.ts
|
|
1389
|
+
/** Resolves whether `model` is covered by a caller-supplied allow-list/predicate. */
|
|
493
1390
|
/** Anthropic's native per-block content shape for a message. */
|
|
494
1391
|
type AnthropicContentBlock = {
|
|
495
1392
|
type: 'text';
|
|
@@ -498,9 +1395,19 @@ type AnthropicContentBlock = {
|
|
|
498
1395
|
type: 'image';
|
|
499
1396
|
source: {
|
|
500
1397
|
type: 'base64';
|
|
501
|
-
media_type:
|
|
1398
|
+
media_type: SupportedImageMimeType;
|
|
502
1399
|
data: string;
|
|
503
1400
|
};
|
|
1401
|
+
} | {
|
|
1402
|
+
type: 'tool_use';
|
|
1403
|
+
id: string;
|
|
1404
|
+
name: string;
|
|
1405
|
+
input: unknown;
|
|
1406
|
+
} | {
|
|
1407
|
+
type: 'tool_result';
|
|
1408
|
+
tool_use_id: string;
|
|
1409
|
+
content: string;
|
|
1410
|
+
is_error?: boolean;
|
|
504
1411
|
};
|
|
505
1412
|
/** Minimal structural type for the Anthropic SDK's `messages.create` */
|
|
506
1413
|
interface AnthropicClient {
|
|
@@ -517,19 +1424,51 @@ interface AnthropicClient {
|
|
|
517
1424
|
tools?: Array<{
|
|
518
1425
|
name: string;
|
|
519
1426
|
description?: string;
|
|
520
|
-
input_schema:
|
|
1427
|
+
input_schema: {
|
|
1428
|
+
type: 'object';
|
|
1429
|
+
[key: string]: unknown;
|
|
1430
|
+
};
|
|
521
1431
|
strict?: boolean;
|
|
522
1432
|
}>;
|
|
523
1433
|
tool_choice?: {
|
|
1434
|
+
type: 'auto';
|
|
1435
|
+
} | {
|
|
1436
|
+
type: 'any';
|
|
1437
|
+
} | {
|
|
1438
|
+
type: 'none';
|
|
1439
|
+
} | {
|
|
524
1440
|
type: 'tool';
|
|
525
1441
|
name: string;
|
|
526
1442
|
};
|
|
1443
|
+
/**
|
|
1444
|
+
* Native, schema-constrained output: a separate request field from
|
|
1445
|
+
* `tools`/`tool_choice`, so it can be sent alongside real tool
|
|
1446
|
+
* calls. Only built by this adapter for models covered by
|
|
1447
|
+
* `nativeStructuredOutputModels` (opt-in, see
|
|
1448
|
+
* `AnthropicAdapterOptions`); other models keep getting
|
|
1449
|
+
* `jsonSchema` emulated as a forced single tool call, the
|
|
1450
|
+
* pre-existing behavior.
|
|
1451
|
+
*
|
|
1452
|
+
* Matches the real Anthropic API's `output_config.format` shape
|
|
1453
|
+
* exactly: just `type` and `schema`, no `name`/`description`/
|
|
1454
|
+
* `strict`. Those three exist on VernLLM's own `jsonSchema` API
|
|
1455
|
+
* (and are still forwarded on the legacy forced-tool-call path,
|
|
1456
|
+
* where they're real `Tool` fields), but the native structured-
|
|
1457
|
+
* output endpoint has no equivalent for any of them.
|
|
1458
|
+
*/
|
|
1459
|
+
output_config?: {
|
|
1460
|
+
format: {
|
|
1461
|
+
type: 'json_schema';
|
|
1462
|
+
schema: Record<string, unknown>;
|
|
1463
|
+
};
|
|
1464
|
+
};
|
|
527
1465
|
}, options: {
|
|
528
1466
|
signal: AbortSignal;
|
|
529
1467
|
}): Promise<{
|
|
530
1468
|
content: Array<{
|
|
531
1469
|
type: string;
|
|
532
1470
|
text?: string;
|
|
1471
|
+
id?: string;
|
|
533
1472
|
name?: string;
|
|
534
1473
|
input?: unknown;
|
|
535
1474
|
}>;
|
|
@@ -540,21 +1479,50 @@ interface AnthropicClient {
|
|
|
540
1479
|
}>;
|
|
541
1480
|
};
|
|
542
1481
|
}
|
|
1482
|
+
/** Optional configuration for `fromAnthropic`. */
|
|
1483
|
+
interface AnthropicAdapterOptions {
|
|
1484
|
+
/**
|
|
1485
|
+
* Which models support native, schema-constrained output
|
|
1486
|
+
* (`output_config.format`), independent of `tools`/`tool_choice`, so it
|
|
1487
|
+
* can be combined with real `tools` in one request. Pass a static list
|
|
1488
|
+
* of model IDs (verified against Anthropic's own docs) or a predicate.
|
|
1489
|
+
*
|
|
1490
|
+
* There is no built-in default here (see `supportsNativeStructuredOutput`
|
|
1491
|
+
* for why). Left unset, every model uses the older forced-single-tool-
|
|
1492
|
+
* call emulation, and `tools` + `jsonSchema` together is rejected,
|
|
1493
|
+
* exactly this adapter's behavior before native support was added.
|
|
1494
|
+
*/
|
|
1495
|
+
nativeStructuredOutputModels?: ModelCapabilityOverride;
|
|
1496
|
+
}
|
|
543
1497
|
/**
|
|
544
1498
|
* Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
|
|
545
1499
|
* interface VernLLM uses for OpenAI/Groq.
|
|
546
1500
|
*
|
|
547
|
-
* `response_format: json_schema
|
|
548
|
-
*
|
|
549
|
-
*
|
|
550
|
-
* `
|
|
551
|
-
*
|
|
1501
|
+
* `response_format: json_schema`, on a model covered by
|
|
1502
|
+
* `options.nativeStructuredOutputModels`, is sent as `output_config.format`,
|
|
1503
|
+
* its own request field, independent of `tools`/`tool_choice`, so it can be
|
|
1504
|
+
* combined with real, caller-supplied `tools` in the same request. Only
|
|
1505
|
+
* `type` and `schema` are sent on this path, the real Anthropic API's
|
|
1506
|
+
* `output_config.format` has no `name`/`description`/`strict` fields.
|
|
1507
|
+
*
|
|
1508
|
+
* On any other model (the default, since `nativeStructuredOutputModels` is
|
|
1509
|
+
* opt-in), `response_format: json_schema` is mapped to Anthropic's forced
|
|
1510
|
+
* tool-use instead: a single tool is defined with `input_schema` set to
|
|
1511
|
+
* the caller's schema, `description` forwarded when provided, and `strict`
|
|
1512
|
+
* forwarded when set, and `tool_choice` forces the model to call it. This
|
|
1513
|
+
* legacy path cannot be combined with real `tools` (both would need the
|
|
1514
|
+
* same `tools`/`tool_choice` field), and a call that tries throws
|
|
1515
|
+
* `LLMError('validation')` before reaching the API. Provider-constrained
|
|
1516
|
+
* schema matching applies only when `strict: true` is forwarded and
|
|
1517
|
+
* supported.
|
|
552
1518
|
*
|
|
553
1519
|
* `response_format: json_object` (no schema to build a tool from) falls
|
|
554
1520
|
* back to a system-prompt instruction, since there's nothing to constrain
|
|
555
|
-
* generation against.
|
|
1521
|
+
* generation against. Unlike `jsonSchema`, this combines with real `tools`
|
|
1522
|
+
* freely on every model: it's a prompt nudge, not a request field, so
|
|
1523
|
+
* there's nothing for it to collide with.
|
|
556
1524
|
*/
|
|
557
|
-
declare function fromAnthropic(anthropicClient: AnthropicClient): LLMClient;
|
|
1525
|
+
declare function fromAnthropic(anthropicClient: AnthropicClient, options?: AnthropicAdapterOptions): LLMClient;
|
|
558
1526
|
|
|
559
1527
|
//#endregion
|
|
560
1528
|
//#region src/adapters/gemini.d.ts
|
|
@@ -566,13 +1534,30 @@ type GeminiPart = {
|
|
|
566
1534
|
mimeType: string;
|
|
567
1535
|
data: string;
|
|
568
1536
|
};
|
|
1537
|
+
} | {
|
|
1538
|
+
functionCall: {
|
|
1539
|
+
name: string;
|
|
1540
|
+
args: unknown;
|
|
1541
|
+
};
|
|
1542
|
+
} | {
|
|
1543
|
+
functionResponse: {
|
|
1544
|
+
name: string;
|
|
1545
|
+
response: unknown;
|
|
1546
|
+
};
|
|
569
1547
|
};
|
|
570
1548
|
/**
|
|
571
|
-
*
|
|
572
|
-
* `generateContent
|
|
573
|
-
*
|
|
574
|
-
* `
|
|
575
|
-
*
|
|
1549
|
+
* Structural type matching the real `@google/genai` SDK's `ai.models`
|
|
1550
|
+
* object: `generateContent`/`generateContentStream` both take a single
|
|
1551
|
+
* `{ model, contents, config }` argument (config carries
|
|
1552
|
+
* `systemInstruction`, `tools`, `toolConfig`, generation settings, and
|
|
1553
|
+
* `abortSignal` all together), matching the real SDK closely enough that
|
|
1554
|
+
* `fromGemini(ai.models)` works directly, e.g:
|
|
1555
|
+
*
|
|
1556
|
+
* ```ts
|
|
1557
|
+
* import { GoogleGenAI } from '@google/genai';
|
|
1558
|
+
* const ai = new GoogleGenAI({ apiKey: '...' });
|
|
1559
|
+
* const llm = new VernLLM({ client: fromGemini(ai.models), model: 'gemini-2.5-flash' });
|
|
1560
|
+
* ```
|
|
576
1561
|
*/
|
|
577
1562
|
interface GeminiClient {
|
|
578
1563
|
generateContent(params: {
|
|
@@ -581,24 +1566,40 @@ interface GeminiClient {
|
|
|
581
1566
|
role: 'user' | 'model';
|
|
582
1567
|
parts: GeminiPart[];
|
|
583
1568
|
}>;
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
1569
|
+
config?: {
|
|
1570
|
+
systemInstruction?: {
|
|
1571
|
+
parts: Array<{
|
|
1572
|
+
text: string;
|
|
1573
|
+
}>;
|
|
1574
|
+
};
|
|
590
1575
|
temperature?: number;
|
|
591
1576
|
maxOutputTokens?: number;
|
|
592
1577
|
responseMimeType?: string;
|
|
593
1578
|
responseSchema?: Record<string, unknown>;
|
|
1579
|
+
tools?: Array<{
|
|
1580
|
+
functionDeclarations: Array<{
|
|
1581
|
+
name: string;
|
|
1582
|
+
description?: string;
|
|
1583
|
+
parameters: Record<string, unknown>;
|
|
1584
|
+
}>;
|
|
1585
|
+
}>;
|
|
1586
|
+
toolConfig?: {
|
|
1587
|
+
functionCallingConfig: {
|
|
1588
|
+
mode: 'AUTO' | 'ANY' | 'NONE';
|
|
1589
|
+
allowedFunctionNames?: string[];
|
|
1590
|
+
};
|
|
1591
|
+
};
|
|
1592
|
+
abortSignal?: AbortSignal;
|
|
594
1593
|
};
|
|
595
|
-
}, options: {
|
|
596
|
-
signal: AbortSignal;
|
|
597
1594
|
}): Promise<{
|
|
598
1595
|
candidates?: Array<{
|
|
599
1596
|
content?: {
|
|
600
1597
|
parts?: Array<{
|
|
601
1598
|
text?: string;
|
|
1599
|
+
functionCall?: {
|
|
1600
|
+
name: string;
|
|
1601
|
+
args: unknown;
|
|
1602
|
+
};
|
|
602
1603
|
}>;
|
|
603
1604
|
};
|
|
604
1605
|
}>;
|
|
@@ -608,6 +1609,32 @@ interface GeminiClient {
|
|
|
608
1609
|
totalTokenCount?: number;
|
|
609
1610
|
};
|
|
610
1611
|
}>;
|
|
1612
|
+
/**
|
|
1613
|
+
* Optional. Required only for `stream: true` calls. Takes the same
|
|
1614
|
+
* request shape as `generateContent`. Matching the real SDK's own
|
|
1615
|
+
* `generateContentStream`, this resolves to an `AsyncIterable` (rather
|
|
1616
|
+
* than returning one synchronously) of partial responses, each chunk
|
|
1617
|
+
* holding the same `candidates[].content.parts[]` structure as
|
|
1618
|
+
* `generateContent`'s response, just incremental.
|
|
1619
|
+
*/
|
|
1620
|
+
generateContentStream?(params: Parameters<GeminiClient['generateContent']>[0]): Promise<AsyncIterable<{
|
|
1621
|
+
candidates?: Array<{
|
|
1622
|
+
content?: {
|
|
1623
|
+
parts?: Array<{
|
|
1624
|
+
text?: string;
|
|
1625
|
+
functionCall?: {
|
|
1626
|
+
name: string;
|
|
1627
|
+
args: unknown;
|
|
1628
|
+
};
|
|
1629
|
+
}>;
|
|
1630
|
+
};
|
|
1631
|
+
}>;
|
|
1632
|
+
usageMetadata?: {
|
|
1633
|
+
promptTokenCount?: number;
|
|
1634
|
+
candidatesTokenCount?: number;
|
|
1635
|
+
totalTokenCount?: number;
|
|
1636
|
+
};
|
|
1637
|
+
}>>;
|
|
611
1638
|
}
|
|
612
1639
|
/**
|
|
613
1640
|
* Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
|
|
@@ -619,6 +1646,25 @@ interface GeminiClient {
|
|
|
619
1646
|
* `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
|
|
620
1647
|
* models use a token budget, not an effort tier, so it's dropped, same as
|
|
621
1648
|
* Anthropic.
|
|
1649
|
+
*
|
|
1650
|
+
* `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
|
|
1651
|
+
* `tool_choice` maps to `toolConfig.functionCallingConfig`. Gemini accepts
|
|
1652
|
+
* `responseSchema` and `tools` in the same request natively, so both are
|
|
1653
|
+
* set independently here and no special-casing is needed for the
|
|
1654
|
+
* combination, unlike `fromAnthropic`/`fromBedrock`.
|
|
1655
|
+
*
|
|
1656
|
+
* `createStream` calls `generateContentStream` (optional on `GeminiClient`
|
|
1657
|
+
*, required only if the caller sets `stream: true`) and translates each
|
|
1658
|
+
* partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
|
|
1659
|
+
* Gemini's own function-calling API doesn't stream tool-call arguments
|
|
1660
|
+
* incrementally: a `functionCall` part always arrives whole in one chunk,
|
|
1661
|
+
* so each one is emitted as a single, complete `tool_call_delta` (a
|
|
1662
|
+
* one-shot "delta" containing the full arguments) rather than accumulated
|
|
1663
|
+
* fragments, that's a real difference in the underlying API, not
|
|
1664
|
+
* something this adapter can smooth over. `usageMetadata` is (per Gemini's
|
|
1665
|
+
* own behavior) only reliably present on the last chunk, so the `usage`
|
|
1666
|
+
* `WireStreamChunk` is emitted once, after the stream completes, from
|
|
1667
|
+
* whichever chunk's `usageMetadata` was seen last.
|
|
622
1668
|
*/
|
|
623
1669
|
declare function fromGemini(geminiClient: GeminiClient): LLMClient;
|
|
624
1670
|
|
|
@@ -636,6 +1682,20 @@ type BedrockContentBlock = {
|
|
|
636
1682
|
bytes: Uint8Array;
|
|
637
1683
|
};
|
|
638
1684
|
};
|
|
1685
|
+
} | {
|
|
1686
|
+
toolUse: {
|
|
1687
|
+
toolUseId: string;
|
|
1688
|
+
name: string;
|
|
1689
|
+
input: unknown;
|
|
1690
|
+
};
|
|
1691
|
+
} | {
|
|
1692
|
+
toolResult: {
|
|
1693
|
+
toolUseId: string;
|
|
1694
|
+
content: Array<{
|
|
1695
|
+
text: string;
|
|
1696
|
+
}>;
|
|
1697
|
+
status?: 'success' | 'error';
|
|
1698
|
+
};
|
|
639
1699
|
};
|
|
640
1700
|
/**
|
|
641
1701
|
* Minimal structural type matching AWS Bedrock's Converse API. This is
|
|
@@ -682,6 +1742,38 @@ interface BedrockConverseClient {
|
|
|
682
1742
|
tool: {
|
|
683
1743
|
name: string;
|
|
684
1744
|
};
|
|
1745
|
+
} | {
|
|
1746
|
+
auto: Record<string, never>;
|
|
1747
|
+
} | {
|
|
1748
|
+
any: Record<string, never>;
|
|
1749
|
+
};
|
|
1750
|
+
};
|
|
1751
|
+
/**
|
|
1752
|
+
* Native, schema-constrained output: a separate request field from
|
|
1753
|
+
* `toolConfig`, so it can be sent alongside real tool calls. Only
|
|
1754
|
+
* built by this adapter for models covered by
|
|
1755
|
+
* `nativeStructuredOutputModels` (opt-in, see
|
|
1756
|
+
* `BedrockAdapterOptions`); other models keep getting `jsonSchema`
|
|
1757
|
+
* emulated as a forced single tool call via `toolConfig`, the
|
|
1758
|
+
* pre-existing behavior.
|
|
1759
|
+
*
|
|
1760
|
+
* Matches the real Bedrock Converse API's `outputConfig.textFormat`
|
|
1761
|
+
* shape exactly: the schema itself is nested one level deeper, under
|
|
1762
|
+
* `structure.jsonSchema`, not flat on `textFormat`, and `schema` is
|
|
1763
|
+
* a JSON-encoded *string*, not a parsed object, unlike every other
|
|
1764
|
+
* schema field this adapter builds (`toolSpec.inputSchema.json`
|
|
1765
|
+
* included). There is no `strict` field here, unlike `toolSpec`.
|
|
1766
|
+
*/
|
|
1767
|
+
outputConfig?: {
|
|
1768
|
+
textFormat: {
|
|
1769
|
+
type: 'json_schema';
|
|
1770
|
+
structure: {
|
|
1771
|
+
jsonSchema: {
|
|
1772
|
+
schema: string;
|
|
1773
|
+
name?: string;
|
|
1774
|
+
description?: string;
|
|
1775
|
+
};
|
|
1776
|
+
};
|
|
685
1777
|
};
|
|
686
1778
|
};
|
|
687
1779
|
}, options: {
|
|
@@ -692,6 +1784,7 @@ interface BedrockConverseClient {
|
|
|
692
1784
|
content?: Array<{
|
|
693
1785
|
text?: string;
|
|
694
1786
|
toolUse?: {
|
|
1787
|
+
toolUseId?: string;
|
|
695
1788
|
name?: string;
|
|
696
1789
|
input?: unknown;
|
|
697
1790
|
};
|
|
@@ -704,24 +1797,122 @@ interface BedrockConverseClient {
|
|
|
704
1797
|
totalTokens?: number;
|
|
705
1798
|
};
|
|
706
1799
|
}>;
|
|
1800
|
+
/**
|
|
1801
|
+
* Optional. Required only for `stream: true` calls. Takes the same
|
|
1802
|
+
* request shape `converse` does, returning `{ stream }`, matching
|
|
1803
|
+
* `ConverseStreamCommand`'s real AWS SDK v3 output shape, an
|
|
1804
|
+
* `AsyncIterable` of incremental events under a `stream` property,
|
|
1805
|
+
* rather than the whole response being the iterable directly.
|
|
1806
|
+
*/
|
|
1807
|
+
converseStream?(params: Parameters<BedrockConverseClient['converse']>[0], options: {
|
|
1808
|
+
signal: AbortSignal;
|
|
1809
|
+
}): Promise<{
|
|
1810
|
+
stream: AsyncIterable<BedrockConverseStreamEvent>;
|
|
1811
|
+
}>;
|
|
707
1812
|
}
|
|
1813
|
+
/**
|
|
1814
|
+
* One event of a Bedrock `ConverseStreamCommand` response's `stream`.
|
|
1815
|
+
* Content blocks (text or toolUse) are identified by `contentBlockIndex`,
|
|
1816
|
+
* Converse's own convention for correlating start/delta/stop events across
|
|
1817
|
+
* possibly-interleaved blocks, mirrored directly by VernLLM's
|
|
1818
|
+
* `tool_call_delta.index`.
|
|
1819
|
+
*/
|
|
1820
|
+
type BedrockConverseStreamEvent = {
|
|
1821
|
+
messageStart: {
|
|
1822
|
+
role: 'assistant';
|
|
1823
|
+
};
|
|
1824
|
+
} | {
|
|
1825
|
+
contentBlockStart: {
|
|
1826
|
+
contentBlockIndex: number;
|
|
1827
|
+
start?: {
|
|
1828
|
+
toolUse?: {
|
|
1829
|
+
toolUseId?: string;
|
|
1830
|
+
name?: string;
|
|
1831
|
+
};
|
|
1832
|
+
};
|
|
1833
|
+
};
|
|
1834
|
+
} | {
|
|
1835
|
+
contentBlockDelta: {
|
|
1836
|
+
contentBlockIndex: number;
|
|
1837
|
+
delta?: {
|
|
1838
|
+
text?: string;
|
|
1839
|
+
} | {
|
|
1840
|
+
toolUse?: {
|
|
1841
|
+
input?: string;
|
|
1842
|
+
};
|
|
1843
|
+
};
|
|
1844
|
+
};
|
|
1845
|
+
} | {
|
|
1846
|
+
contentBlockStop: {
|
|
1847
|
+
contentBlockIndex: number;
|
|
1848
|
+
};
|
|
1849
|
+
} | {
|
|
1850
|
+
messageStop: {
|
|
1851
|
+
stopReason?: string;
|
|
1852
|
+
};
|
|
1853
|
+
} | {
|
|
1854
|
+
metadata: {
|
|
1855
|
+
usage?: {
|
|
1856
|
+
inputTokens?: number;
|
|
1857
|
+
outputTokens?: number;
|
|
1858
|
+
totalTokens?: number;
|
|
1859
|
+
};
|
|
1860
|
+
};
|
|
1861
|
+
} | {
|
|
1862
|
+
internalServerException: {
|
|
1863
|
+
message?: string;
|
|
1864
|
+
};
|
|
1865
|
+
} | {
|
|
1866
|
+
modelStreamErrorException: {
|
|
1867
|
+
message?: string;
|
|
1868
|
+
originalStatusCode?: number;
|
|
1869
|
+
};
|
|
1870
|
+
} | {
|
|
1871
|
+
validationException: {
|
|
1872
|
+
message?: string;
|
|
1873
|
+
};
|
|
1874
|
+
} | {
|
|
1875
|
+
throttlingException: {
|
|
1876
|
+
message?: string;
|
|
1877
|
+
};
|
|
1878
|
+
} | {
|
|
1879
|
+
serviceUnavailableException: {
|
|
1880
|
+
message?: string;
|
|
1881
|
+
};
|
|
1882
|
+
};
|
|
708
1883
|
/**
|
|
709
1884
|
* Optional configuration for `fromBedrock`.
|
|
710
1885
|
*/
|
|
711
1886
|
interface BedrockAdapterOptions {
|
|
712
1887
|
/**
|
|
713
|
-
* Optional preflight check for tool-use support, needed
|
|
714
|
-
*
|
|
715
|
-
* call
|
|
716
|
-
*
|
|
717
|
-
*
|
|
718
|
-
*
|
|
719
|
-
*
|
|
1888
|
+
* Optional preflight check for tool-use support, needed whenever a
|
|
1889
|
+
* `jsonSchema` call ends up sending Converse `toolConfig` — either the
|
|
1890
|
+
* legacy forced-single-tool-call emulation, or real `tools` sent
|
|
1891
|
+
* alongside native structured output (`outputConfig`). VernLLM never
|
|
1892
|
+
* guesses capability from a failed call's error message (AWS's error
|
|
1893
|
+
* text isn't a documented, stable contract), so this is opt-in: pass
|
|
1894
|
+
* either a static list of tool-use-capable model IDs, or a predicate
|
|
1895
|
+
* function, and VernLLM will reject unsupported models with a clear
|
|
1896
|
+
* `LLMError('validation')` *before* dispatching the request, instead of
|
|
1897
|
+
* on the wire.
|
|
720
1898
|
*
|
|
721
1899
|
* Left unset (default), no preflight check runs, and a `jsonSchema` call
|
|
722
1900
|
* to an unsupported model surfaces Bedrock's raw `converse` error as-is.
|
|
723
1901
|
*/
|
|
724
1902
|
toolUseSupportedModels?: string[] | ((modelId: string) => boolean);
|
|
1903
|
+
/**
|
|
1904
|
+
* Which models support native, schema-constrained output
|
|
1905
|
+
* (`outputConfig.textFormat`), independent of `toolConfig`, so it can be
|
|
1906
|
+
* combined with real `tools` in one request. Pass a static list of
|
|
1907
|
+
* model IDs (verified against Bedrock's own docs) or a predicate.
|
|
1908
|
+
*
|
|
1909
|
+
* There is no built-in default here (see `supportsNativeStructuredOutput`
|
|
1910
|
+
* for why). Left unset, every model uses the older forced-single-tool-
|
|
1911
|
+
* call emulation via `toolConfig`, and `tools` + `jsonSchema` together is
|
|
1912
|
+
* rejected, exactly this adapter's behavior before native support was
|
|
1913
|
+
* added.
|
|
1914
|
+
*/
|
|
1915
|
+
nativeStructuredOutputModels?: ModelCapabilityOverride;
|
|
725
1916
|
}
|
|
726
1917
|
/**
|
|
727
1918
|
* Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
|
|
@@ -731,18 +1922,48 @@ interface BedrockAdapterOptions {
|
|
|
731
1922
|
* regardless of which underlying model `modelId` points at, as long as
|
|
732
1923
|
* that model supports Converse (most current-generation ones do)
|
|
733
1924
|
*
|
|
734
|
-
* `response_format: json_schema
|
|
735
|
-
*
|
|
736
|
-
*
|
|
737
|
-
*
|
|
738
|
-
*
|
|
739
|
-
* `
|
|
1925
|
+
* `response_format: json_schema`, on a model covered by
|
|
1926
|
+
* `options.nativeStructuredOutputModels` (opt-in, unset by default), is
|
|
1927
|
+
* sent as `outputConfig.textFormat`, its own request field, independent of
|
|
1928
|
+
* `toolConfig`, so it can be combined with real, caller-supplied `tools`
|
|
1929
|
+
* in the same request. Matches the real Converse API's shape exactly: the
|
|
1930
|
+
* schema is nested under `structure.jsonSchema` and JSON-encoded as a
|
|
1931
|
+
* string, not the parsed object `toolConfig`'s tool schemas use, and there
|
|
1932
|
+
* is no `strict` field on this path.
|
|
1933
|
+
*
|
|
1934
|
+
* On any other model (the default), `response_format: json_schema` is
|
|
1935
|
+
* mapped to Converse's `toolConfig` instead: a single tool is defined from
|
|
1936
|
+
* the schema, description, and strictness settings, and `toolChoice`
|
|
1937
|
+
* forces the model to call it. This legacy path cannot be combined with
|
|
1938
|
+
* real `tools` (both would need the same `toolConfig`), and a call that
|
|
1939
|
+
* tries throws `LLMError('validation')` before reaching the API.
|
|
1940
|
+
* Provider-constrained schema matching applies only when `strict: true` is
|
|
1941
|
+
* forwarded and supported. Native tool support varies by model family;
|
|
1942
|
+
* pass `toolUseSupportedModels` to preflight-check it (see
|
|
740
1943
|
* `BedrockAdapterOptions`), otherwise a `jsonSchema` call to an
|
|
741
1944
|
* unsupported model surfaces Bedrock's raw error unchanged.
|
|
742
1945
|
*
|
|
743
1946
|
* `response_format: json_object` (no schema to build a tool from) and
|
|
744
1947
|
* `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
|
|
745
|
-
* instruction and are dropped respectively.
|
|
1948
|
+
* instruction and are dropped respectively. Unlike `jsonSchema`,
|
|
1949
|
+
* `json_object` combines with real `tools` freely on every model: it's a
|
|
1950
|
+
* prompt nudge, not a request field, so there's nothing for it to collide
|
|
1951
|
+
* with.
|
|
1952
|
+
*
|
|
1953
|
+
* `tools` alone maps to Converse's native `toolConfig`/`toolUse`/
|
|
1954
|
+
* `toolResult`; `tool_choice` maps to `toolConfig.toolChoice`.
|
|
1955
|
+
*
|
|
1956
|
+
* `createStream` calls `converseStream` (optional on `BedrockConverseClient`
|
|
1957
|
+
*, required only if the caller sets `stream: true`) and translates its
|
|
1958
|
+
* `contentBlockStart`/`contentBlockDelta`/`metadata` events into
|
|
1959
|
+
* `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
|
|
1960
|
+
* same as `fromAnthropic`'s block-index tracking (Converse's streaming
|
|
1961
|
+
* shape is structurally close to Anthropic's own, both being tool-use-aware
|
|
1962
|
+
* content-block streams), including the same `json-tool` unwrapping: a
|
|
1963
|
+
* `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
|
|
1964
|
+
* `text-delta`, not `tool_call_delta`, so the accumulated result lands in
|
|
1965
|
+
* `finalizeResponse`'s `content` path exactly like the non-streaming
|
|
1966
|
+
* `create` branch above unwraps it.
|
|
746
1967
|
*/
|
|
747
1968
|
declare function fromBedrock(bedrockClient: BedrockConverseClient, options?: BedrockAdapterOptions): LLMClient;
|
|
748
1969
|
|
|
@@ -772,6 +1993,22 @@ type RequestLike = (url: string, init: {
|
|
|
772
1993
|
body?: string;
|
|
773
1994
|
signal?: AbortSignal;
|
|
774
1995
|
}) => Promise<ResponseLike>;
|
|
1996
|
+
/**
|
|
1997
|
+
* A streaming-capable request function. Unlike `RequestLike`, which returns
|
|
1998
|
+
* a fully-buffered `ResponseLike`, this resolves to an `AsyncIterable` of
|
|
1999
|
+
* progressively-arriving chunks, the common ground across transports:
|
|
2000
|
+
* native `fetch`'s `response.body` (wrapped to be iterable; see
|
|
2001
|
+
* `webStreamToAsyncIterable` below), axios's Node `Readable` in
|
|
2002
|
+
* `responseType: 'stream'` mode (already async-iterable, no wrapping
|
|
2003
|
+
* needed), `node-fetch`, `undici`, etc, all satisfy this with little or no
|
|
2004
|
+
* glue code. Defaults to native `fetch`.
|
|
2005
|
+
*/
|
|
2006
|
+
type StreamRequestLike = (url: string, init: {
|
|
2007
|
+
method: string;
|
|
2008
|
+
headers: Record<string, string>;
|
|
2009
|
+
body?: string;
|
|
2010
|
+
signal?: AbortSignal;
|
|
2011
|
+
}) => Promise<AsyncIterable<Uint8Array | string>>;
|
|
775
2012
|
interface FetchAdapterConfig {
|
|
776
2013
|
/** Endpoint URL, or a function of the request in case it depends on model/params */
|
|
777
2014
|
url: string | ((params: ChatRequest) => string);
|
|
@@ -788,17 +2025,65 @@ interface FetchAdapterConfig {
|
|
|
788
2025
|
/** Maps VernLLMs internal chat-completion request into the providers raw request body */
|
|
789
2026
|
mapRequest: (params: ChatRequest) => unknown;
|
|
790
2027
|
/**
|
|
791
|
-
* Maps the providers raw JSON response into `{ content, usage? }`
|
|
792
|
-
* `content` is the assistants text (JSON string when JSON mode was requested)
|
|
2028
|
+
* Maps the providers raw JSON response into `{ content, usage?, toolCalls? }`
|
|
2029
|
+
* `content` is the assistants text (JSON string when JSON mode was requested).
|
|
2030
|
+
* `content` may be empty/omitted when the model responded with only tool
|
|
2031
|
+
* calls and no text.
|
|
2032
|
+
*
|
|
2033
|
+
* `toolCalls`, when the model requested one or more tools, is the list of
|
|
2034
|
+
* calls as flat `{ id, name, arguments }` entries (matching this config's
|
|
2035
|
+
* own `toolCalls?: Array<{ id: string; name: string; arguments: string }>`
|
|
2036
|
+
* return type below), each entry's `arguments` already JSON-*encoded* as a
|
|
2037
|
+
* string (not the parsed object), mirroring the wire format every
|
|
2038
|
+
* OpenAI-compatible provider uses. `fromFetch` itself converts these into
|
|
2039
|
+
* `WireToolCall`'s `type`/`function`-wrapped shape before returning them
|
|
2040
|
+
* from `create`. VernLLM parses (and validates, if `argumentsSchema` was
|
|
2041
|
+
* set) the arguments string internally, mapResponse doesn't need to do
|
|
2042
|
+
* that itself.
|
|
793
2043
|
*/
|
|
794
2044
|
mapResponse: (json: unknown) => {
|
|
795
|
-
content
|
|
2045
|
+
content?: string;
|
|
796
2046
|
usage?: {
|
|
797
2047
|
promptTokens?: number;
|
|
798
2048
|
completionTokens?: number;
|
|
799
2049
|
totalTokens?: number;
|
|
800
2050
|
};
|
|
2051
|
+
toolCalls?: Array<{
|
|
2052
|
+
id: string;
|
|
2053
|
+
name: string;
|
|
2054
|
+
arguments: string;
|
|
2055
|
+
}>;
|
|
801
2056
|
};
|
|
2057
|
+
/**
|
|
2058
|
+
* Optional. Required only for `stream: true` calls. The function used to
|
|
2059
|
+
* open a streaming HTTP request. Takes the same request shape as
|
|
2060
|
+
* `request`, but resolves to an `AsyncIterable` of progressively-arriving
|
|
2061
|
+
* `Uint8Array` or `string` chunks instead of a buffered `ResponseLike`.
|
|
2062
|
+
* Defaults to native `fetch`.
|
|
2063
|
+
*/
|
|
2064
|
+
requestStream?: StreamRequestLike;
|
|
2065
|
+
/**
|
|
2066
|
+
* Optional. How the raw stream bytes are split into individual event
|
|
2067
|
+
* payloads. Defaults to Server-Sent Events framing (`data: ...` blocks
|
|
2068
|
+
* separated by a blank line, `[DONE]` sentinel honored, see
|
|
2069
|
+
* `parseSseStream`), which covers the large majority of LLM providers'
|
|
2070
|
+
* streaming HTTP endpoints. Override this for a provider that frames its
|
|
2071
|
+
* stream differently, e.g. newline-delimited JSON (NDJSON) with no SSE
|
|
2072
|
+
* envelope.
|
|
2073
|
+
*/
|
|
2074
|
+
parseStreamFrames?: (chunks: AsyncIterable<Uint8Array | string>) => AsyncIterable<unknown>;
|
|
2075
|
+
/**
|
|
2076
|
+
* Optional. Required only for `stream: true` calls. Maps one parsed
|
|
2077
|
+
* stream event (already extracted from its frame by `parseStreamFrames`)
|
|
2078
|
+
* into zero, one, or more `WireStreamChunk`s, mirrors `mapResponse`'s
|
|
2079
|
+
* role for the non-streaming path, just per-event instead of once for
|
|
2080
|
+
* the whole body. Return `undefined` to skip an event that carries
|
|
2081
|
+
* nothing VernLLM needs (e.g. a provider's keep-alive ping). Configs
|
|
2082
|
+
* that don't implement this make `stream: true` throw a clear
|
|
2083
|
+
* `LLMError('validation')` rather than a confusing runtime failure or a
|
|
2084
|
+
* silently empty stream.
|
|
2085
|
+
*/
|
|
2086
|
+
mapStreamEvent?: (event: unknown) => WireStreamChunk | WireStreamChunk[] | undefined;
|
|
802
2087
|
}
|
|
803
2088
|
/**
|
|
804
2089
|
* A fetch-based escape hatch for providers with no SDK, or where pulling one
|
|
@@ -810,6 +2095,34 @@ interface FetchAdapterConfig {
|
|
|
810
2095
|
* Non-2xx responses throw an error with `.status` set to the HTTP status
|
|
811
2096
|
* code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
|
|
812
2097
|
* 401/403) applies here too
|
|
2098
|
+
*
|
|
2099
|
+
* Tool calling works the same way as every other adapter: `mapRequest`
|
|
2100
|
+
* receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
|
|
2101
|
+
* translate them into whatever shape the provider's wire format expects
|
|
2102
|
+
* (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
|
|
2103
|
+
* field). On the way back, `mapResponse` may return a `toolCalls` array
|
|
2104
|
+
* (id/name/JSON-encoded-arguments-string per call) alongside or instead of
|
|
2105
|
+
* `content`; VernLLM parses and (if `argumentsSchema` was set) validates
|
|
2106
|
+
* those arguments the same way it does for every other adapter. For
|
|
2107
|
+
* `stream: true`, tool-call deltas go through the existing
|
|
2108
|
+
* `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
|
|
2109
|
+
* no separate config is needed for streaming vs non-streaming tool calls.
|
|
2110
|
+
*
|
|
2111
|
+
|
|
2112
|
+
* `createStream` requires `mapStreamEvent` (there's no non-streaming
|
|
2113
|
+
* response to fall back on, unlike the other three optional streaming
|
|
2114
|
+
* seams). It opens the request via `requestStream` (defaults to native
|
|
2115
|
+
* `fetch`), splits the raw bytes into individual events via
|
|
2116
|
+
* `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
|
|
2117
|
+
* and translates each event into `WireStreamChunk`(s) via
|
|
2118
|
+
* `mapStreamEvent`. Both seams are overridable per-config for providers
|
|
2119
|
+
* that don't fit the SSE-over-fetch default. If a custom `request`
|
|
2120
|
+
* transport is configured, `requestStream` must be configured too,
|
|
2121
|
+
* `requestStream` never silently falls back to `request` (see
|
|
2122
|
+
* `createStream`'s own comment for why), so a `stream: true` call with
|
|
2123
|
+
* `request` set but no `requestStream` throws a clear
|
|
2124
|
+
* `LLMError('validation')` instead of quietly using unrelated native
|
|
2125
|
+
* `fetch`.
|
|
813
2126
|
*/
|
|
814
2127
|
declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
815
2128
|
|
|
@@ -834,11 +2147,46 @@ declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
|
834
2147
|
* (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
|
|
835
2148
|
* the actual compatibility contract is the JSON each provider sends and
|
|
836
2149
|
* receives over the wire, not the SDKs TS types.
|
|
2150
|
+
*
|
|
2151
|
+
* `createStream` is implemented by calling the same underlying
|
|
2152
|
+
* `chat.completions.create` with `stream: true` (and, for providers that
|
|
2153
|
+
* support it, `stream_options: { include_usage: true }`, so a final usage
|
|
2154
|
+
* block arrives), the OpenAI SDK, and every OpenAI-compatible client
|
|
2155
|
+
* modeled on it, returns an `AsyncIterable` of SSE chunks instead of a
|
|
2156
|
+
* single completion object when `stream: true` is set. Each chunk is
|
|
2157
|
+
* translated into `WireStreamChunk`(s) via `toWireStreamChunks`.
|
|
2158
|
+
*
|
|
2159
|
+
* Note on long-running reasoning models: this adapter consumes the
|
|
2160
|
+
* underlying SDK's already-parsed stream rather than raw SSE bytes, so
|
|
2161
|
+
* unlike `fromFetch`/`fromAnthropic` it cannot see comment-only keep-alive
|
|
2162
|
+
* ping frames. Combined with `chunkIdleTimeoutMs`'s 30 second default and
|
|
2163
|
+
* `reasoningEffort` (documented to have long silent gaps for o-series and
|
|
2164
|
+
* similar models), a long-running reasoning call on this adapter can trip
|
|
2165
|
+
* the idle timeout even though the provider is still working. Raise or
|
|
2166
|
+
* disable `chunkIdleTimeoutMs` per call for those routes, see `CallParams`.
|
|
837
2167
|
*/
|
|
838
|
-
|
|
2168
|
+
interface OpenAICompatibleAdapterOptions {
|
|
2169
|
+
/**
|
|
2170
|
+
* Whether the provider supports `stream_options.include_usage`. Not
|
|
2171
|
+
* every "OpenAI-compatible" provider is guaranteed to, so this defaults
|
|
2172
|
+
* to `true` (matching OpenAI, Groq, Mistral, and most others observed)
|
|
2173
|
+
* and should be set to `false` for a provider verified not to support
|
|
2174
|
+
* it. When `false`, `stream_options` is omitted entirely and no usage
|
|
2175
|
+
* block will arrive on the stream; callers relying on streamed `usage`
|
|
2176
|
+
* with such a provider won't get one.
|
|
2177
|
+
*/
|
|
2178
|
+
supportsStreamUsage?: boolean;
|
|
2179
|
+
}
|
|
2180
|
+
declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
|
|
839
2181
|
/** Groqs SDK matches the OpenAI wire format */
|
|
840
2182
|
declare const fromGroq: typeof fromOpenAICompatible;
|
|
841
|
-
/**
|
|
2183
|
+
/**
|
|
2184
|
+
* Mistrals `chat.completions`-shaped client (or their OpenAI-compat
|
|
2185
|
+
* endpoint). Mistral supports `stream_options.include_usage` (added after
|
|
2186
|
+
* an earlier period where it returned a 422 for unrecognized fields, per
|
|
2187
|
+
* Mistral's changelog and streaming docs), so this is a plain alias like
|
|
2188
|
+
* the others, `supportsStreamUsage` defaults to `true`.
|
|
2189
|
+
*/
|
|
842
2190
|
declare const fromMistral: typeof fromOpenAICompatible;
|
|
843
2191
|
/** DeepSeeks API is OpenAI-compatible */
|
|
844
2192
|
declare const fromDeepSeek: typeof fromOpenAICompatible;
|
|
@@ -929,5 +2277,5 @@ declare const from01AI: typeof fromOpenAICompatible;
|
|
|
929
2277
|
//#endregion
|
|
930
2278
|
//# sourceMappingURL=openaiCompatible.d.ts.map
|
|
931
2279
|
|
|
932
|
-
export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CallParams, CircuitBreaker, CircuitBreakerOptions, ConsoleLogger, ContentBlock, ConversationTurn, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorType, Logger, NormalizedCacheAdapter, OnUsage, RefundUsage, ReserveUsage, SchemaLike, TextBlock, TieredCacheAdapter, TokenUsage, VernLLM, VernLLMOptions, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError };
|
|
2280
|
+
export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedStreamCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallMeta, CallParams, CallWithToolsResult, CircuitBreaker, CircuitBreakerOptions, CircuitState, ConsoleLogger, ContentBlock, ContentResult, ConversationTurn, FallbackAttempt, FallbackExhaustedError, FallbackOn, FallbackTarget, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorCode, LLMErrorType, Logger, NormalizedCacheAdapter, OnEvent, OnUsage, RateLimitAcquireResult, RateLimitOptions, RateLimitReason, RateLimiter, RefundUsage, ReserveUsage, SSE_PING, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, TargetCircuitState, TextBlock, TieredCacheAdapter, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolIssue, ToolResult, VernLLM, VernLLMEvent, VernLLMOptions, WireMessage, WireRequest, WireStreamChunk, WireToolCall, WireToolChoice, defaultEstimateTokens, defaultFallbackOn, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError, isToolCallResult, parseSseStream };
|
|
933
2281
|
//# sourceMappingURL=index.d.cts.map
|