vern-llm 1.7.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -6
- package/dist/index.cjs +1784 -272
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1009 -73
- package/dist/index.d.cts.map +1 -1
- package/dist/index.d.ts +1009 -73
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1782 -273
- package/dist/index.js.map +1 -1
- package/package.json +9 -13
package/dist/index.d.cts
CHANGED
|
@@ -110,8 +110,79 @@ interface JsonSchemaSpec {
|
|
|
110
110
|
}
|
|
111
111
|
|
|
112
112
|
//#endregion
|
|
113
|
-
//#region src/types/
|
|
113
|
+
//#region src/types/tools.d.ts
|
|
114
114
|
//# sourceMappingURL=schema.d.ts.map
|
|
115
|
+
/**
|
|
116
|
+
* Describes a capability the model may request, not the capability
|
|
117
|
+
* itself. VernLLM transports this to the provider and parses what comes
|
|
118
|
+
* back; it never executes anything.
|
|
119
|
+
*/
|
|
120
|
+
interface ToolDefinition {
|
|
121
|
+
name: string;
|
|
122
|
+
description: string;
|
|
123
|
+
/** JSON Schema for the tool's input. */
|
|
124
|
+
parameters: Record<string, unknown>;
|
|
125
|
+
/**
|
|
126
|
+
* Optional client-side validator run on the parsed `arguments` before
|
|
127
|
+
* they're handed back to the caller, mirroring the `schema: SchemaLike<T>`
|
|
128
|
+
* pattern already used for response validation (see `types/schema.ts`).
|
|
129
|
+
* Reuses that zero-dependency, `safeParse`-compatible shape instead of
|
|
130
|
+
* requiring a JSON Schema validator (e.g. ajv) as a new dependency.
|
|
131
|
+
* Failed validation throws `LLMError('validation')`. If omitted, VernLLM
|
|
132
|
+
* parses arguments as JSON but does not validate them further.
|
|
133
|
+
*/
|
|
134
|
+
argumentsSchema?: SchemaLike<unknown>;
|
|
135
|
+
}
|
|
136
|
+
/** A single tool invocation requested by the model. */
|
|
137
|
+
interface ToolCall {
|
|
138
|
+
id: string;
|
|
139
|
+
name: string;
|
|
140
|
+
/** Parsed JSON arguments (and validated, if `argumentsSchema` was set). */
|
|
141
|
+
arguments: unknown;
|
|
142
|
+
}
|
|
143
|
+
/** The application's result of executing a `ToolCall`, sent back to the model. */
|
|
144
|
+
interface ToolResult {
|
|
145
|
+
toolCallId: string;
|
|
146
|
+
content: unknown;
|
|
147
|
+
/**
|
|
148
|
+
* Signals a failed tool execution back to the model (matches Anthropic's
|
|
149
|
+
* native `is_error` on tool_result blocks). Only `fromAnthropic` honors
|
|
150
|
+
* this today, Gemini and Bedrock have no equivalent wire concept, so
|
|
151
|
+
* other adapters ignore it silently.
|
|
152
|
+
*/
|
|
153
|
+
isError?: boolean;
|
|
154
|
+
}
|
|
155
|
+
/** `call()` result when `tools` was set and the model produced a normal answer. */
|
|
156
|
+
interface ContentResult<T> {
|
|
157
|
+
type: 'content';
|
|
158
|
+
content: T;
|
|
159
|
+
}
|
|
160
|
+
/** `call()` result when `tools` was set and the model requested one or more tools. */
|
|
161
|
+
interface ToolCallResult {
|
|
162
|
+
type: 'tool_calls';
|
|
163
|
+
toolCalls: ToolCall[];
|
|
164
|
+
/** Any text the model produced alongside the tool request, if present. */
|
|
165
|
+
content?: string;
|
|
166
|
+
}
|
|
167
|
+
type CallWithToolsResult<T> = ContentResult<T> | ToolCallResult;
|
|
168
|
+
/**
|
|
169
|
+
* Runtime-safe check for whether a `call()` result is a `tool_calls`
|
|
170
|
+
* result. Prefer this over relying on TypeScript's static narrowing
|
|
171
|
+
* whenever `params` passed to `call()` wasn't a literal with `tools`
|
|
172
|
+
* inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
|
|
173
|
+
* that case TS may have typed the result as plain `T` even though it's
|
|
174
|
+
* actually a `CallWithToolsResult<T>` at runtime, and this check works
|
|
175
|
+
* either way.
|
|
176
|
+
*/
|
|
177
|
+
declare function isToolCallResult(result: unknown): result is ToolCallResult;
|
|
178
|
+
/** What the model should do about tools on a given call. */
|
|
179
|
+
type ToolChoice = 'auto' | 'none' | 'required' | {
|
|
180
|
+
name: string;
|
|
181
|
+
};
|
|
182
|
+
|
|
183
|
+
//#endregion
|
|
184
|
+
//#region src/types/usage.d.ts
|
|
185
|
+
//# sourceMappingURL=tools.d.ts.map
|
|
115
186
|
type ReserveUsage = (params: {
|
|
116
187
|
coalesced: boolean;
|
|
117
188
|
signal?: AbortSignal;
|
|
@@ -144,15 +215,39 @@ interface TokenUsage {
|
|
|
144
215
|
model: string;
|
|
145
216
|
}
|
|
146
217
|
type OnUsage = (usage: TokenUsage) => void;
|
|
218
|
+
/**
|
|
219
|
+
* Called when a provider response arrives but VernLLM's own post-processing
|
|
220
|
+
* then fails, after usage data was already present in that response. Covers
|
|
221
|
+
* any error thrown after usage extraction, not just parse/validation, since
|
|
222
|
+
* everything in that path only runs once a response, and real spend, has
|
|
223
|
+
* already arrived. Fires once per failed attempt with extractable usage,
|
|
224
|
+
* never for transport failures, where no response means no honest number
|
|
225
|
+
* to report.
|
|
226
|
+
*/
|
|
227
|
+
type OnUsageFailure = (usage: TokenUsage, error: LLMError) => void;
|
|
147
228
|
|
|
148
229
|
//#endregion
|
|
149
230
|
//#region src/types/call.d.ts
|
|
150
231
|
//# sourceMappingURL=usage.d.ts.map
|
|
151
|
-
/**
|
|
152
|
-
|
|
153
|
-
|
|
232
|
+
/**
|
|
233
|
+
* A single prior turn in a multi-turn conversation, passed via `history`.
|
|
234
|
+
*
|
|
235
|
+
* Supports normal user/assistant messages and tool continuations: an assistant
|
|
236
|
+
* turn may include `toolCalls`, and a tool turn carries the matching
|
|
237
|
+
* `toolResults`. A tool turn must immediately follow an assistant tool call
|
|
238
|
+
* turn, and every requested tool call must have a result.
|
|
239
|
+
*/
|
|
240
|
+
type ConversationTurn = {
|
|
241
|
+
role: 'user';
|
|
154
242
|
content: string;
|
|
155
|
-
}
|
|
243
|
+
} | {
|
|
244
|
+
role: 'assistant';
|
|
245
|
+
content?: string;
|
|
246
|
+
toolCalls?: ToolCall[];
|
|
247
|
+
} | {
|
|
248
|
+
role: 'tool';
|
|
249
|
+
toolResults: ToolResult[];
|
|
250
|
+
};
|
|
156
251
|
/** A plain text segment of a multimodal `userContent` array. */
|
|
157
252
|
interface TextBlock {
|
|
158
253
|
type: 'text';
|
|
@@ -180,15 +275,28 @@ interface CallParams<T = unknown> extends UsageHooks {
|
|
|
180
275
|
/** Current user message, as text or multimodal content blocks. */
|
|
181
276
|
userContent: string | ContentBlock[];
|
|
182
277
|
/**
|
|
183
|
-
* Previous conversation turns. Must alternate
|
|
184
|
-
*
|
|
278
|
+
* Previous conversation turns. Must alternate roles; tool turns must follow
|
|
279
|
+
* assistant tool calls. Invalid history throws LLMError('validation').
|
|
185
280
|
*/
|
|
186
281
|
history?: ConversationTurn[];
|
|
187
|
-
|
|
282
|
+
/**
|
|
283
|
+
* Generation temperature. Default 0.2, not the provider's own default.
|
|
284
|
+
* Pass `null` to omit `temperature` from the request entirely, so the
|
|
285
|
+
* provider applies its own default instead.
|
|
286
|
+
*/
|
|
287
|
+
temperature?: number | null;
|
|
188
288
|
jsonMode?: boolean;
|
|
189
289
|
maxTokens?: number;
|
|
190
290
|
requestId?: string;
|
|
191
291
|
signal?: AbortSignal;
|
|
292
|
+
/**
|
|
293
|
+
* Per-call override for the instance's `chunkIdleTimeoutMs` (max gap
|
|
294
|
+
* between stream chunks once opened). Only applies when `stream: true`.
|
|
295
|
+
* Useful for routes using reasoning-heavy models with documented long
|
|
296
|
+
* silent gaps mid-stream. Pass 0 to disable the idle timeout for this
|
|
297
|
+
* call.
|
|
298
|
+
*/
|
|
299
|
+
chunkIdleTimeoutMs?: number;
|
|
192
300
|
/** Overrides the instance model for this call. */
|
|
193
301
|
model?: string;
|
|
194
302
|
/** Reasoning effort for supported reasoning models. */
|
|
@@ -202,17 +310,233 @@ interface CallParams<T = unknown> extends UsageHooks {
|
|
|
202
310
|
* Implies jsonMode: true.
|
|
203
311
|
*/
|
|
204
312
|
schema?: SchemaLike<T>;
|
|
313
|
+
/**
|
|
314
|
+
* Tools the model may call. When set, `call()` always returns a
|
|
315
|
+
* `CallWithToolsResult<T>` discriminated union instead of `T` directly
|
|
316
|
+
* (see `CallWithToolsResult`), a breaking-change point: omitting `tools`
|
|
317
|
+
* keeps `call()`'s old `Promise<T>` behavior exactly.
|
|
318
|
+
*
|
|
319
|
+
* Mutually exclusive with `jsonSchema`/`schema`: on Anthropic and
|
|
320
|
+
* Bedrock, `jsonSchema` is implemented internally as a forced single-tool
|
|
321
|
+
* call, which would collide with real tools. Setting both throws
|
|
322
|
+
* `LLMError('validation')`.
|
|
323
|
+
*/
|
|
324
|
+
tools?: ToolDefinition[];
|
|
325
|
+
/** Defaults to `'auto'` when `tools` is set. */
|
|
326
|
+
toolChoice?: ToolChoice;
|
|
327
|
+
/**
|
|
328
|
+
* Streams the response incrementally instead of resolving once. Default:
|
|
329
|
+
* false. Requires a client/adapter that implements `createStream`.
|
|
330
|
+
* Retry/timeout/circuit-breaker guarantees apply only to opening the
|
|
331
|
+
* stream (through the first chunk); a failure after that point rejects
|
|
332
|
+
* `finalResult` directly and is not retried, since a mid-stream failure
|
|
333
|
+
* isn't connection-time evidence for the circuit breaker, the attempt
|
|
334
|
+
* already counted as a success once the first chunk arrived. Once the
|
|
335
|
+
* stream opens successfully, `finalResult` still resolves to the same
|
|
336
|
+
* validated `T`/`CallWithToolsResult<T>` shape `call()` would have
|
|
337
|
+
* returned for the same params with `stream` omitted. See
|
|
338
|
+
* `StreamCallResult`.
|
|
339
|
+
*/
|
|
340
|
+
stream?: boolean;
|
|
205
341
|
}
|
|
206
|
-
|
|
342
|
+
/**
|
|
343
|
+
* A `CallParams` variant where tool calling is explicitly enabled.
|
|
344
|
+
*
|
|
345
|
+
* Requiring `tools` to be present allows TypeScript to select the
|
|
346
|
+
* tool-aware `call()` overload and return `CallWithToolsResult<T>` instead
|
|
347
|
+
* of the normal `T` response type.
|
|
348
|
+
*/
|
|
349
|
+
type ToolEnabledCallParams<T> = CallParams<T> & {
|
|
350
|
+
tools: NonNullable<CallParams<T>['tools']>;
|
|
351
|
+
};
|
|
352
|
+
/** Shared cache-configuration fields, minus the internal `fn` primitive. */
|
|
353
|
+
interface CachedCallInput extends UsageHooks {
|
|
207
354
|
cacheKey: string;
|
|
208
355
|
ttl: number;
|
|
209
|
-
fn: () => Promise<T>;
|
|
210
356
|
signal?: AbortSignal;
|
|
211
357
|
}
|
|
358
|
+
/**
|
|
359
|
+
* Parameters for a cached LLM call without tool calling.
|
|
360
|
+
*
|
|
361
|
+
* Combines the cache configuration with the `CallParams` passed to
|
|
362
|
+
* `VernLLM.call()`. The cached value is the normal LLM response type `T`.
|
|
363
|
+
*/
|
|
364
|
+
type CachedCallParams<T> = CachedCallInput & {
|
|
365
|
+
call: CallParams<T>;
|
|
366
|
+
};
|
|
367
|
+
/**
|
|
368
|
+
* Parameters for a cached LLM call with tool calling enabled.
|
|
369
|
+
*
|
|
370
|
+
* The cached value includes the full `CallWithToolsResult<T>`, meaning
|
|
371
|
+
* tool requests and normal content responses are cached exactly as returned
|
|
372
|
+
* by the model.
|
|
373
|
+
*/
|
|
374
|
+
type CachedToolCallParams<T> = CachedCallInput & {
|
|
375
|
+
call: ToolEnabledCallParams<T>;
|
|
376
|
+
};
|
|
212
377
|
|
|
213
378
|
//#endregion
|
|
214
|
-
//#region src/types/
|
|
379
|
+
//#region src/types/stream.d.ts
|
|
215
380
|
//# sourceMappingURL=call.d.ts.map
|
|
381
|
+
/** One incremental unit of a streaming response, as delivered to the caller. */
|
|
382
|
+
type StreamChunk = {
|
|
383
|
+
type: 'text-delta';
|
|
384
|
+
delta: string;
|
|
385
|
+
} | {
|
|
386
|
+
type: 'tool_call_delta';
|
|
387
|
+
index: number;
|
|
388
|
+
id?: string;
|
|
389
|
+
name?: string;
|
|
390
|
+
argsDelta?: string;
|
|
391
|
+
/**
|
|
392
|
+
* True when `argsDelta` is the whole set of arguments, not a
|
|
393
|
+
* fragment. Set for Gemini (its API returns function-call args
|
|
394
|
+
* whole in one chunk) and for cache/replay chunks, which are
|
|
395
|
+
* one-shot too. Omitted or `false` for a genuine fragment from
|
|
396
|
+
* providers that do stream incrementally (OpenAI-compatible,
|
|
397
|
+
* Anthropic, Bedrock).
|
|
398
|
+
*/
|
|
399
|
+
complete?: boolean;
|
|
400
|
+
} | {
|
|
401
|
+
type: 'usage';
|
|
402
|
+
usage: TokenUsage;
|
|
403
|
+
};
|
|
404
|
+
/**
|
|
405
|
+
* What `call()` returns when `stream: true`. `chunks` is for live rendering;
|
|
406
|
+
* `finalResult` resolves to the same validated `T`/`CallWithToolsResult<T>`
|
|
407
|
+
* shape `call()` would have returned had `stream` been omitted, once the
|
|
408
|
+
* stream completes successfully.
|
|
409
|
+
*
|
|
410
|
+
* `chunks` is single-use and supports only one consumer: iterating it more
|
|
411
|
+
* than once, or from more than one place concurrently, shares the same
|
|
412
|
+
* underlying buffered stream rather than replaying or forking it, which can
|
|
413
|
+
* split chunks unpredictably between consumers. Stopping iteration early
|
|
414
|
+
* (e.g. `break`ing out of a `for await`) does not cancel or otherwise
|
|
415
|
+
* signal the underlying stream, the background pump keeps running to
|
|
416
|
+
* completion regardless, buffering any chunks emitted after that point, so
|
|
417
|
+
* `finalResult` still settles normally even if `chunks` is abandoned or
|
|
418
|
+
* never read at all.
|
|
419
|
+
*
|
|
420
|
+
* Unread chunks are buffered internally for the duration of one stream,
|
|
421
|
+
* this is what lets a caller start iterating `chunks` after the stream has
|
|
422
|
+
* already progressed (or finished) and still see everything. That backlog
|
|
423
|
+
* is capped: an unusually large stream whose `chunks` is never read at all
|
|
424
|
+
* has its oldest buffered chunks dropped once the backlog grows past
|
|
425
|
+
* roughly twice a fixed internal limit, trimmed back down to that limit in
|
|
426
|
+
* one batch rather than one-at-a-time, bounding both peak memory and the
|
|
427
|
+
* eviction work itself for that pathological case instead of the array
|
|
428
|
+
* growing (or being trimmed) proportional to the whole stream's output.
|
|
429
|
+
* Ordinary consumption, even started somewhat late, stays far under the
|
|
430
|
+
* limit and is unaffected.
|
|
431
|
+
*/
|
|
432
|
+
interface StreamCallResult<R> {
|
|
433
|
+
chunks: AsyncIterable<StreamChunk>;
|
|
434
|
+
finalResult: Promise<R>;
|
|
435
|
+
}
|
|
436
|
+
/**
|
|
437
|
+
* A `CallParams` variant where streaming is explicitly enabled.
|
|
438
|
+
*
|
|
439
|
+
* Requiring `stream: true` to be statically present allows TypeScript to
|
|
440
|
+
* select the streaming `call()` overload and return `StreamCallResult<...>`
|
|
441
|
+
* instead of the normal, single-shot response type.
|
|
442
|
+
*/
|
|
443
|
+
type StreamEnabledCallParams<T> = CallParams<T> & {
|
|
444
|
+
stream: true;
|
|
445
|
+
};
|
|
446
|
+
/**
|
|
447
|
+
* The adapter-facing, pre-normalization shape a `createStream` client
|
|
448
|
+
* implementation emits, analogous to how `WireMessage`/`WireToolCall`
|
|
449
|
+
* already sit between `CallParams` and each provider's own wire format.
|
|
450
|
+
*/
|
|
451
|
+
type WireStreamChunk = {
|
|
452
|
+
type: 'text-delta';
|
|
453
|
+
delta: string;
|
|
454
|
+
} | {
|
|
455
|
+
type: 'tool_call_delta';
|
|
456
|
+
index: number;
|
|
457
|
+
id?: string;
|
|
458
|
+
name?: string;
|
|
459
|
+
argumentsDelta?: string;
|
|
460
|
+
/** Same meaning as `StreamChunk`'s `tool_call_delta.complete`. */
|
|
461
|
+
complete?: boolean;
|
|
462
|
+
} | {
|
|
463
|
+
type: 'usage';
|
|
464
|
+
usage: {
|
|
465
|
+
prompt_tokens?: number;
|
|
466
|
+
completion_tokens?: number;
|
|
467
|
+
total_tokens?: number;
|
|
468
|
+
};
|
|
469
|
+
} | {
|
|
470
|
+
/**
|
|
471
|
+
* A provider keep-alive signal with no content of its own (e.g.
|
|
472
|
+
* Anthropic's `ping` events, an SSE comment-line heartbeat).
|
|
473
|
+
* Adapters yield this so the stream loop resets its idle timeout.
|
|
474
|
+
* Never surfaced to callers as a `StreamChunk`.
|
|
475
|
+
*/
|
|
476
|
+
type: 'ping';
|
|
477
|
+
};
|
|
478
|
+
/**
|
|
479
|
+
* Parameters for a cached, streaming LLM call without tool calling.
|
|
480
|
+
*
|
|
481
|
+
* The cached value is `T`, same as `CachedCallParams<T>`, but a miss
|
|
482
|
+
* relays live `chunks` to the caller while the result is being generated,
|
|
483
|
+
* and a hit synthesizes a one-shot `chunks` replay from the cached value
|
|
484
|
+
* (see `VernLLM.cachedCall`'s docs for exactly what that replay looks
|
|
485
|
+
* like).
|
|
486
|
+
*/
|
|
487
|
+
type CachedStreamCallParams<T> = CachedCallInput & {
|
|
488
|
+
call: StreamEnabledCallParams<T>;
|
|
489
|
+
};
|
|
490
|
+
/**
|
|
491
|
+
* Parameters for a cached, streaming LLM call with tool calling enabled.
|
|
492
|
+
*
|
|
493
|
+
* The cached value is the full `CallWithToolsResult<T>`, same as
|
|
494
|
+
* `CachedToolCallParams<T>`, with the same live-chunks-on-miss,
|
|
495
|
+
* replayed-chunks-on-hit behavior as `CachedStreamCallParams<T>`.
|
|
496
|
+
*/
|
|
497
|
+
type CachedStreamToolCallParams<T> = CachedCallInput & {
|
|
498
|
+
call: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>;
|
|
499
|
+
};
|
|
500
|
+
|
|
501
|
+
//#endregion
|
|
502
|
+
//#region src/types/client.d.ts
|
|
503
|
+
//# sourceMappingURL=stream.d.ts.map
|
|
504
|
+
/** A tool call as it appears on the wire, OpenAI's `function`-wrapped shape. */
|
|
505
|
+
interface WireToolCall {
|
|
506
|
+
id: string;
|
|
507
|
+
type: 'function';
|
|
508
|
+
function: {
|
|
509
|
+
name: string;
|
|
510
|
+
/** JSON-encoded arguments, matching every OpenAI-compatible provider's wire format. */
|
|
511
|
+
arguments: string;
|
|
512
|
+
};
|
|
513
|
+
}
|
|
514
|
+
/** One entry of `LLMClient`'s `messages` array, named so callers building it can annotate against it. */
|
|
515
|
+
type WireMessage = {
|
|
516
|
+
role: 'system';
|
|
517
|
+
content: string;
|
|
518
|
+
} | {
|
|
519
|
+
role: 'user';
|
|
520
|
+
content: string | ContentBlock[];
|
|
521
|
+
} | {
|
|
522
|
+
role: 'assistant';
|
|
523
|
+
/** Optional: an assistant turn that only requested tools has no text. */
|
|
524
|
+
content?: string;
|
|
525
|
+
tool_calls?: WireToolCall[];
|
|
526
|
+
} | {
|
|
527
|
+
role: 'tool';
|
|
528
|
+
tool_call_id: string;
|
|
529
|
+
content: string;
|
|
530
|
+
/** Only honored by `fromAnthropic` today (maps to `tool_result.is_error`); other adapters ignore it. */
|
|
531
|
+
is_error?: boolean;
|
|
532
|
+
};
|
|
533
|
+
/** The OpenAI-shaped wire `tool_choice`. */
|
|
534
|
+
type WireToolChoice = 'auto' | 'none' | 'required' | {
|
|
535
|
+
type: 'function';
|
|
536
|
+
function: {
|
|
537
|
+
name: string;
|
|
538
|
+
};
|
|
539
|
+
};
|
|
216
540
|
/**
|
|
217
541
|
* Minimal shape compatible with the OpenAI SDKs chat.completions.create,
|
|
218
542
|
* so consumers can pass an OpenAI client directly
|
|
@@ -226,7 +550,7 @@ interface LLMClient {
|
|
|
226
550
|
completions: {
|
|
227
551
|
create(params: {
|
|
228
552
|
model: string;
|
|
229
|
-
temperature
|
|
553
|
+
temperature?: number;
|
|
230
554
|
max_tokens: number;
|
|
231
555
|
response_format?: {
|
|
232
556
|
type: 'json_object';
|
|
@@ -241,19 +565,29 @@ interface LLMClient {
|
|
|
241
565
|
};
|
|
242
566
|
/** OpenAI reasoning-model param (o-series, gpt-5), ignored by providers that don't support it */
|
|
243
567
|
reasoning_effort?: 'minimal' | 'low' | 'medium' | 'high';
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
568
|
+
/** Tools the model may call, OpenAI's `function`-wrapped shape. */
|
|
569
|
+
tools?: Array<{
|
|
570
|
+
type: 'function';
|
|
571
|
+
function: {
|
|
572
|
+
name: string;
|
|
573
|
+
description: string;
|
|
574
|
+
parameters: Record<string, unknown>;
|
|
575
|
+
};
|
|
250
576
|
}>;
|
|
577
|
+
tool_choice?: WireToolChoice;
|
|
578
|
+
/**
|
|
579
|
+
* Wire-format messages. Breaking change for custom adapters:
|
|
580
|
+
* implementations must handle tool messages and assistant tool_calls.
|
|
581
|
+
* Exhaustive switches over only system/user/assistant roles may no longer compile.
|
|
582
|
+
*/
|
|
583
|
+
messages: WireMessage[];
|
|
251
584
|
}, options: {
|
|
252
585
|
signal: AbortSignal;
|
|
253
586
|
}): Promise<{
|
|
254
587
|
choices?: Array<{
|
|
255
588
|
message?: {
|
|
256
589
|
content?: string | null;
|
|
590
|
+
tool_calls?: WireToolCall[];
|
|
257
591
|
};
|
|
258
592
|
}>;
|
|
259
593
|
usage?: {
|
|
@@ -262,6 +596,15 @@ interface LLMClient {
|
|
|
262
596
|
total_tokens?: number;
|
|
263
597
|
};
|
|
264
598
|
}>;
|
|
599
|
+
/**
|
|
600
|
+
* Optional. Required only for `stream: true` calls. Adapters/clients
|
|
601
|
+
* that don't implement this make `stream: true` throw a clear
|
|
602
|
+
* `LLMError('validation')` rather than a confusing runtime failure.
|
|
603
|
+
* Takes the same request shape as `create`, minus the response type.
|
|
604
|
+
*/
|
|
605
|
+
createStream?(params: Parameters<LLMClient['chat']['completions']['create']>[0], options: {
|
|
606
|
+
signal: AbortSignal;
|
|
607
|
+
}): AsyncIterable<WireStreamChunk>;
|
|
265
608
|
};
|
|
266
609
|
};
|
|
267
610
|
}
|
|
@@ -337,10 +680,26 @@ interface VernLLMOptions {
|
|
|
337
680
|
maxRetries?: number;
|
|
338
681
|
/** Per-attempt timeout in ms. Default 25000 */
|
|
339
682
|
timeoutMs?: number;
|
|
683
|
+
/**
|
|
684
|
+
* For `stream: true` calls: max gap allowed between chunks once the
|
|
685
|
+
* stream has opened, in ms. Resets on every chunk, including keep-alive
|
|
686
|
+
* pings. `timeoutMs` only covers opening the stream and its first
|
|
687
|
+
* chunk; this covers every gap after that. Also counts as a
|
|
688
|
+
* circuit-breaker failure, unlike other mid-stream errors, since a
|
|
689
|
+
* provider that streams one chunk then stalls should still trip it.
|
|
690
|
+
* Default 30000. Pass 0 or negative to disable.
|
|
691
|
+
*/
|
|
692
|
+
chunkIdleTimeoutMs?: number;
|
|
340
693
|
/** Base delay for exponential backoff in ms. Default 500 */
|
|
341
694
|
baseDelayMs?: number;
|
|
342
695
|
/** Default max_tokens for calls that don't override it. Default 1000 */
|
|
343
696
|
defaultMaxTokens?: number;
|
|
697
|
+
/**
|
|
698
|
+
* Default temperature for calls that don't override it. Default 0.2, not
|
|
699
|
+
* the provider's own default. Pass `null` to omit `temperature` from the
|
|
700
|
+
* request entirely, so the provider applies its own default instead.
|
|
701
|
+
*/
|
|
702
|
+
defaultTemperature?: number | null;
|
|
344
703
|
/** Enables debug logging of raw model output (logs up to 800 chars of each
|
|
345
704
|
* response). Off by default */
|
|
346
705
|
debug?: boolean;
|
|
@@ -352,6 +711,18 @@ interface VernLLMOptions {
|
|
|
352
711
|
parseJson?: (content: string) => unknown;
|
|
353
712
|
/** Called after every successful call with token usage, if the provider reports it */
|
|
354
713
|
onUsage?: OnUsage;
|
|
714
|
+
/**
|
|
715
|
+
* Called when a provider response arrives but VernLLM's own post-processing
|
|
716
|
+
* then fails, after usage data was already present in that response.
|
|
717
|
+
* Separate from `onUsage`, which only fires on full success.
|
|
718
|
+
*
|
|
719
|
+
* For non-streaming calls, never fires for transport failures (timeout,
|
|
720
|
+
* network error, non-retryable status), since no response means no usage
|
|
721
|
+
* to report. For streaming calls, this is not guaranteed: a stream can
|
|
722
|
+
* deliver a usage chunk and then fail later (e.g. an idle timeout waiting
|
|
723
|
+
* for the final close), in which case this does fire.
|
|
724
|
+
*/
|
|
725
|
+
onUsageFailure?: OnUsageFailure;
|
|
355
726
|
/** Injectable logger. Defaults to a console-based logger gated by `debug` */
|
|
356
727
|
logger?: Logger;
|
|
357
728
|
/**
|
|
@@ -366,31 +737,34 @@ interface VernLLMOptions {
|
|
|
366
737
|
//#region src/vernLLM.d.ts
|
|
367
738
|
//# sourceMappingURL=options.d.ts.map
|
|
368
739
|
/**
|
|
369
|
-
* A resilient layer around an LLM chat completions client
|
|
740
|
+
* A resilient layer around an LLM chat completions client. This is VernLLM!
|
|
370
741
|
*
|
|
371
|
-
* Adds retry with backoff
|
|
372
|
-
* JSON parsing with optional schema validation, usage
|
|
373
|
-
* optional response cache
|
|
374
|
-
* defaults.
|
|
742
|
+
* Adds retry with backoff and jitter, per-attempt timeouts, an optional
|
|
743
|
+
* circuit breaker, JSON parsing with optional schema validation, usage
|
|
744
|
+
* tracking, and an optional response cache. All configurable, all opt-in
|
|
745
|
+
* beyond sensible defaults.
|
|
375
746
|
*/
|
|
376
747
|
declare class VernLLM {
|
|
377
748
|
private readonly client;
|
|
378
749
|
private readonly model;
|
|
379
750
|
private readonly maxRetries;
|
|
380
751
|
private readonly timeoutMs;
|
|
752
|
+
private readonly chunkIdleTimeoutMs;
|
|
381
753
|
private readonly baseDelayMs;
|
|
382
754
|
private readonly defaultMaxTokens;
|
|
755
|
+
private readonly defaultTemperature;
|
|
383
756
|
private readonly cache;
|
|
384
757
|
private readonly nonRetryableStatus;
|
|
385
758
|
private readonly inFlight;
|
|
386
759
|
private readonly parseJson;
|
|
387
760
|
private readonly onUsage?;
|
|
761
|
+
private readonly onUsageFailure?;
|
|
388
762
|
private readonly logger;
|
|
389
763
|
private readonly breaker?;
|
|
390
764
|
/**
|
|
391
|
-
* @param options
|
|
392
|
-
* `
|
|
393
|
-
*
|
|
765
|
+
* @param options Client, model, and tunables. Defaults: `maxRetries` 1,
|
|
766
|
+
* `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
|
|
767
|
+
* `defaultTemperature` 0.2, `cache` an in-memory adapter,
|
|
394
768
|
* `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
|
|
395
769
|
*/
|
|
396
770
|
constructor(options: VernLLMOptions);
|
|
@@ -399,23 +773,103 @@ declare class VernLLM {
|
|
|
399
773
|
/**
|
|
400
774
|
* Makes a single logical LLM call, retrying on failure per the configured
|
|
401
775
|
* policy. Fails fast if the breaker is open or the signal is already
|
|
402
|
-
* aborted.
|
|
403
|
-
*
|
|
776
|
+
* aborted. Rejects with a normalized LLMError on exhausted retries.
|
|
777
|
+
*
|
|
778
|
+
* When `tools` is set, returns a `CallWithToolsResult<T>` instead of `T`:
|
|
779
|
+
* `{ type: 'content', content }` or `{ type: 'tool_calls', toolCalls,
|
|
780
|
+
* content? }`. VernLLM never executes tools; run them yourself and
|
|
781
|
+
* continue via `history` (see `ConversationTurn`). Mutually exclusive
|
|
782
|
+
* with `jsonSchema`/`schema`.
|
|
783
|
+
*
|
|
784
|
+
* TypeScript only picks the tools-aware overload when `tools` is
|
|
785
|
+
* statically present on `params`. If set conditionally on a plain
|
|
786
|
+
* `CallParams<T>`, use `isToolCallResult()` to check the shape at
|
|
787
|
+
* runtime instead. See the Tool Calling docs for details.
|
|
404
788
|
*
|
|
405
|
-
*
|
|
406
|
-
*
|
|
407
|
-
*
|
|
408
|
-
*
|
|
789
|
+
* The same static-vs-dynamic caveat applies to `stream`: TypeScript only
|
|
790
|
+
* selects the streaming overload (returning `StreamCallResult<...>`) when
|
|
791
|
+
* `stream: true` is statically present on `params`. A `stream` value set
|
|
792
|
+
* conditionally on a plain `CallParams<T>` still resolves to `Promise<T>`
|
|
793
|
+
* (or `Promise<CallWithToolsResult<T>>`) at the type level even though
|
|
794
|
+
* the actual runtime result is the `{ chunks, finalResult }` streaming
|
|
795
|
+
* shape whenever `stream` evaluates to `true`, callers doing this should
|
|
796
|
+
* narrow/cast accordingly rather than relying on the static return type.
|
|
797
|
+
*
|
|
798
|
+
* @param params System/user content plus per-call overrides. See `CallParams`.
|
|
799
|
+
* @returns Without `tools` or `stream`: the parsed response, or raw
|
|
800
|
+
* string if `jsonMode` is false. With `tools`: a `CallWithToolsResult<T>`.
|
|
801
|
+
* With `stream: true` (statically): a `{ chunks, finalResult }`
|
|
802
|
+
* `StreamCallResult`, `finalResult` resolving to whichever of the above
|
|
803
|
+
* shapes applies once the stream completes. See `StreamCallResult`.
|
|
409
804
|
*/
|
|
805
|
+
call<T = unknown>(params: StreamEnabledCallParams<T> & ToolEnabledCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
|
|
806
|
+
call<T = unknown>(params: StreamEnabledCallParams<T>): Promise<StreamCallResult<T>>;
|
|
807
|
+
call<T = unknown>(params: ToolEnabledCallParams<T>): Promise<CallWithToolsResult<T>>;
|
|
410
808
|
call<T = unknown>(params: CallParams<T>): Promise<T>;
|
|
411
|
-
/** Runs `fn`, retrying with backoff according to `shouldRetry`. */
|
|
412
|
-
private retryWithBackoff;
|
|
413
809
|
/**
|
|
414
|
-
* Performs a single attempt: builds the request
|
|
415
|
-
*
|
|
810
|
+
* Performs a single attempt: builds the request (translating `tools` to
|
|
811
|
+
* wire shape when present), dispatches it with a timeout, and shapes the
|
|
812
|
+
* response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
|
|
813
|
+
* set. Throws on an empty response (no text and no tool_calls) so the
|
|
416
814
|
* retry loop treats it like any other transient failure.
|
|
417
815
|
*/
|
|
418
816
|
private executeCall;
|
|
817
|
+
/**
|
|
818
|
+
* Shapes a fully-arrived response (content and/or tool_calls, already
|
|
819
|
+
* extracted from the provider's payload) into `T` or a
|
|
820
|
+
* `CallWithToolsResult<T>`. Reused by the streaming path once it has
|
|
821
|
+
* buffered the full text/tool-call deltas, so there's no separate
|
|
822
|
+
* parsing/validation logic for streaming.
|
|
823
|
+
*
|
|
824
|
+
* Normalizes and reports usage failure on error itself, so every caller
|
|
825
|
+
* gets identical error handling without duplicating it.
|
|
826
|
+
*/
|
|
827
|
+
private finalizeResponse;
|
|
828
|
+
/**
|
|
829
|
+
* Opens a stream for a single attempt: builds the request exactly like
|
|
830
|
+
* `executeCall`, then requires `createStream` on the client (a clear
|
|
831
|
+
* `validation` error if the adapter doesn't support it). The timeout
|
|
832
|
+
* wraps stream construction and the first `.next()` together, not just
|
|
833
|
+
* construction: calling an `async function*` returns an iterator
|
|
834
|
+
* synchronously without running its body until `.next()` is first
|
|
835
|
+
* invoked, so timing only construction would time an operation that's
|
|
836
|
+
* always instant, not the actual connection. Both are folded into a
|
|
837
|
+
* single `withTimeout` so the same abort signal reaches whatever the
|
|
838
|
+
* adapter's `createStream` uses internally for its first network
|
|
839
|
+
* round-trip.
|
|
840
|
+
*
|
|
841
|
+
* Circuit-breaker success is recorded once the stream fully completes,
|
|
842
|
+
* not on the first chunk arriving, so a connection that opens but then
|
|
843
|
+
* dies mid-stream isn't masked as a success (see `buildStreamResult`).
|
|
844
|
+
*/
|
|
845
|
+
private executeStreamCall;
|
|
846
|
+
/**
|
|
847
|
+
* The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
|
|
848
|
+
* an async generator that yields translated `StreamChunk`s to the caller
|
|
849
|
+
* live, as they arrive, with no per-chunk timeout and no bound on total
|
|
850
|
+
* duration, and accumulates text/tool-call deltas internally so that
|
|
851
|
+
* `finalizeResponse` can produce `finalResult` once the stream completes.
|
|
852
|
+
*
|
|
853
|
+
* Two separate try/catches: the iteration loop's catch handles errors
|
|
854
|
+
* the transport itself throws, which aren't normalized yet, so that
|
|
855
|
+
* happens here along with the one `reportUsageFailure` call for them.
|
|
856
|
+
* The second catch, around `finalizeResponse`, does not re-normalize or
|
|
857
|
+
* re-report since `finalizeResponse` already does both internally.
|
|
858
|
+
* Circuit-breaker success is only recorded once the stream fully
|
|
859
|
+
* completes, not when the first chunk arrives, so a connection that
|
|
860
|
+
* opens and then dies mid-way still counts as a failure below instead
|
|
861
|
+
* of masking it.
|
|
862
|
+
*/
|
|
863
|
+
private buildStreamResult;
|
|
864
|
+
/**
|
|
865
|
+
* Checks every `ToolCall` against the `tools` that were offered, catching
|
|
866
|
+
* a hallucinated tool name early instead of letting it reach the
|
|
867
|
+
* application's dispatch table. Then runs each tool's `argumentsSchema`,
|
|
868
|
+
* if present, throwing `LLMError('validation')` on failure.
|
|
869
|
+
*/
|
|
870
|
+
private validateToolCallArguments;
|
|
871
|
+
/** Runs `fn`, retrying with backoff according to `shouldRetry`. */
|
|
872
|
+
private retryWithBackoff;
|
|
419
873
|
/**
|
|
420
874
|
* Validates `history` alternates user/assistant turns, since providers
|
|
421
875
|
* like Anthropic/Gemini reject or mishandle consecutive same-role turns.
|
|
@@ -423,6 +877,16 @@ declare class VernLLM {
|
|
|
423
877
|
private validateHistory;
|
|
424
878
|
/** Applies per-call defaults and shapes params into the client's request object. */
|
|
425
879
|
private buildRequestPayload;
|
|
880
|
+
/** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
|
|
881
|
+
private buildWireToolChoice;
|
|
882
|
+
/**
|
|
883
|
+
* Expands one `ConversationTurn` into one or more wire messages. Plain
|
|
884
|
+
* user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
|
|
885
|
+
* to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
|
|
886
|
+
* turn expands into one wire `tool` message per `toolResult`, since
|
|
887
|
+
* OpenAI-shaped wire format wants one message per tool_call_id.
|
|
888
|
+
*/
|
|
889
|
+
private turnToWireMessages;
|
|
426
890
|
/**
|
|
427
891
|
* Chooses the response format: a provider-native `jsonSchema` takes
|
|
428
892
|
* priority when supplied (constrains generation directly), otherwise
|
|
@@ -430,8 +894,23 @@ declare class VernLLM {
|
|
|
430
894
|
* requested, or no format at all for plain text responses.
|
|
431
895
|
*/
|
|
432
896
|
private buildResponseFormat;
|
|
433
|
-
/**
|
|
434
|
-
|
|
897
|
+
/**
|
|
898
|
+
* Pulls `TokenUsage` out of a raw response, if the provider reported it.
|
|
899
|
+
* Extraction doesn't depend on what happens to the response afterward, so
|
|
900
|
+
* a malformed body can still yield usage if the provider's usage block
|
|
901
|
+
* itself came through intact.
|
|
902
|
+
*/
|
|
903
|
+
private extractUsage;
|
|
904
|
+
/** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
|
|
905
|
+
private reportUsage;
|
|
906
|
+
/**
|
|
907
|
+
* Reports token usage spent on an attempt that then failed, so it isn't
|
|
908
|
+
* dropped alongside the error. Covers any error thrown after usage
|
|
909
|
+
* extraction, since all of them happen only after a response (real
|
|
910
|
+
* spend) already arrived. Swallows and logs any error `onUsageFailure`
|
|
911
|
+
* itself throws.
|
|
912
|
+
*/
|
|
913
|
+
private reportUsageFailure;
|
|
435
914
|
/** Parses response content as JSON and validates it against `schema` when supplied. */
|
|
436
915
|
private parseAndValidate;
|
|
437
916
|
/**
|
|
@@ -448,38 +927,97 @@ declare class VernLLM {
|
|
|
448
927
|
* supports deletion. Cache invalidation is the caller's responsibility;
|
|
449
928
|
* only the application knows when cached data is stale.
|
|
450
929
|
*
|
|
451
|
-
* @param key
|
|
930
|
+
* @param key The raw cache key (resolved through the adapter's
|
|
452
931
|
* `resolveKey`, if any, before deletion).
|
|
453
932
|
*/
|
|
454
933
|
deleteCache(key: string): Promise<void>;
|
|
455
934
|
/**
|
|
456
|
-
*
|
|
457
|
-
* same `cacheKey` share a single in-flight call, avoiding cache
|
|
935
|
+
* Internal cache primitive around caller-supplied logic. Concurrent misses
|
|
936
|
+
* for the same `cacheKey` share a single in-flight call, avoiding cache
|
|
937
|
+
* stampedes.
|
|
938
|
+
*
|
|
939
|
+
* Not part of the public API. Backs the public `cachedCall()`, which
|
|
940
|
+
* always composes this with `call()` so cached results get the same
|
|
941
|
+
* retry/timeout/circuit-breaker guarantees as any other LLM call.
|
|
458
942
|
*
|
|
459
|
-
* @param params
|
|
943
|
+
* @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
|
|
460
944
|
* miss, typically `() => this.call(...)`), and optional
|
|
461
|
-
* `reserveUsage`/`refundUsage`/`signal`. See `
|
|
945
|
+
* `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
|
|
462
946
|
* @returns The cached value on a hit, or the result of `fn()` on a miss.
|
|
463
947
|
*/
|
|
464
|
-
|
|
948
|
+
private runCached;
|
|
465
949
|
/** Starts the shared fn() call for a cache miss and tracks it in the in-flight map until it settles. */
|
|
466
950
|
private registerTrigger;
|
|
467
951
|
/** Runs `fn` and writes its result to the cache. */
|
|
468
952
|
private runAndCache;
|
|
953
|
+
/**
|
|
954
|
+
* Streaming counterpart to `runCached`. Three cases:
|
|
955
|
+
*
|
|
956
|
+
* - Hit: no live generation to relay. Returns immediately with
|
|
957
|
+
* `finalResult` resolved to the cached value and a one-shot `chunks`
|
|
958
|
+
* replay built from it, so `for await (const c of chunks)` call sites
|
|
959
|
+
* work identically on a hit or a miss. No usage hooks fire, since
|
|
960
|
+
* nothing was actually spent.
|
|
961
|
+
* - Miss, nothing else in flight for this key: delegates to
|
|
962
|
+
* `registerStreamTrigger`, which opens the stream and relays its
|
|
963
|
+
* `chunks` live.
|
|
964
|
+
* - Miss, but another call for the same key is already in flight: this
|
|
965
|
+
* call has no live chunks of its own to relay, so it's treated like a
|
|
966
|
+
* delayed hit. `finalResult` shares the trigger's in-flight promise
|
|
967
|
+
* (the same `this.inFlight` map non-streaming `runCached` uses, so
|
|
968
|
+
* streaming and non-streaming `cachedCall`s for the same key coalesce
|
|
969
|
+
* against each other too), and `chunks` is a one-shot replay built
|
|
970
|
+
* once that promise resolves.
|
|
971
|
+
*/
|
|
972
|
+
private runCachedStream;
|
|
973
|
+
/**
|
|
974
|
+
* Opens the shared stream for a cache miss and tracks its settled value
|
|
975
|
+
* in `this.inFlight` until it resolves or rejects. Writes to the cache
|
|
976
|
+
* on success only, matching `runAndCache`.
|
|
977
|
+
*
|
|
978
|
+
* Registers the in-flight promise synchronously, before anything async
|
|
979
|
+
* runs, so a concurrent `cachedCall` for the same key always sees it in
|
|
980
|
+
* time to join instead of triggering its own stream. Settlement is
|
|
981
|
+
* wired onto the whole `withReservedUsageForStream` call rather than a
|
|
982
|
+
* line inside its callback, so any failure point (reserving usage,
|
|
983
|
+
* opening the stream, or the stream itself) reliably settles the
|
|
984
|
+
* in-flight entry instead of leaving it stuck.
|
|
985
|
+
*/
|
|
986
|
+
private registerStreamTrigger;
|
|
469
987
|
/** Logs a failed refundUsage attempt via the configured logger. */
|
|
470
988
|
private logRefundError;
|
|
471
989
|
/**
|
|
472
|
-
*
|
|
990
|
+
* Cache wrapper composing `call` + caching, so cached LLM calls
|
|
473
991
|
* automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
|
|
474
|
-
* `refundUsage` are read from the top-level params only.
|
|
992
|
+
* `refundUsage` are read from the top-level params only. Concurrent misses
|
|
993
|
+
* for the same `cacheKey` share a single in-flight call, avoiding cache
|
|
994
|
+
* stampedes. Supports `stream: true` and `tools` in any combination.
|
|
995
|
+
*
|
|
996
|
+
* When `call.tools` is set, this caches the whole `CallWithToolsResult`,
|
|
997
|
+
* including `tool_calls` results, not just final answers. Whether
|
|
998
|
+
* that's appropriate depends on the tool: caching "the model decided to
|
|
999
|
+
* call get_weather" is usually fine to reuse briefly, but caching a
|
|
1000
|
+
* decision made under permissions or account state that can change
|
|
1001
|
+
* between calls is not. Use a short `ttl` or a separate `cacheKey` for
|
|
1002
|
+
* such tools if this distinction matters.
|
|
1003
|
+
*
|
|
1004
|
+
* There is no public way to cache an arbitrary non-LLM function through
|
|
1005
|
+
* `VernLLM`. This method always composes with `call()`. For
|
|
1006
|
+
* general-purpose caching unrelated to an LLM call, use a dedicated
|
|
1007
|
+
* caching library at the application level instead.
|
|
475
1008
|
*
|
|
476
|
-
* @param params
|
|
477
|
-
* plus `call`, the `CallParams`
|
|
1009
|
+
* @param params `cacheKey`, `ttl`, and optional
|
|
1010
|
+
* `reserveUsage`/`refundUsage`/`signal`, plus `call`, the `CallParams`
|
|
1011
|
+
* (optionally with `tools` and/or `stream`) to pass through to
|
|
1012
|
+
* `this.call(...)`. The top-level `signal` governs the cached operation
|
|
1013
|
+
* and its usage hooks only; to also abort the underlying provider
|
|
1014
|
+
* request, set `signal` inside `call`.
|
|
478
1015
|
* @returns The cached value on a hit, or the freshly-called result on a miss.
|
|
479
1016
|
*/
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
1017
|
+
cachedCall<T>(params: CachedStreamToolCallParams<T>): Promise<StreamCallResult<CallWithToolsResult<T>>>;
|
|
1018
|
+
cachedCall<T>(params: CachedStreamCallParams<T>): Promise<StreamCallResult<T>>;
|
|
1019
|
+
cachedCall<T>(params: CachedToolCallParams<T>): Promise<CallWithToolsResult<T>>;
|
|
1020
|
+
cachedCall<T>(params: CachedCallParams<T>): Promise<T>;
|
|
483
1021
|
/**
|
|
484
1022
|
* @returns The current circuit breaker state (`'closed' | 'open' |
|
|
485
1023
|
* 'half-open'`), or undefined if no circuit breaker was configured.
|
|
@@ -488,8 +1026,64 @@ declare class VernLLM {
|
|
|
488
1026
|
}
|
|
489
1027
|
|
|
490
1028
|
//#endregion
|
|
491
|
-
//#region src/
|
|
1029
|
+
//#region src/internal/sse.d.ts
|
|
492
1030
|
//# sourceMappingURL=vernLLM.d.ts.map
|
|
1031
|
+
/**
|
|
1032
|
+
* Parses a Server-Sent-Events byte/text stream into the JSON payload of
|
|
1033
|
+
* each `data:` frame, in arrival order. Generic over transport: works with
|
|
1034
|
+
* anything that hands back progressively-arriving `Uint8Array` or `string`
|
|
1035
|
+
* chunks via async iteration: native `fetch`'s `response.body` (wrapped
|
|
1036
|
+
* to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
|
|
1037
|
+
* Node `Readable` (already async-iterable, no wrapping needed), etc, so
|
|
1038
|
+
* this framing layer doesn't care which transport produced the bytes.
|
|
1039
|
+
*
|
|
1040
|
+
* Follows the SSE spec's frame-delimiting rules closely enough for LLM
|
|
1041
|
+
* streaming responses: frames are separated by a blank line, each frame
|
|
1042
|
+
* may carry one or more `data:` lines (joined with `\n` per spec when
|
|
1043
|
+
* there's more than one), `:`-prefixed lines are comments and ignored, and
|
|
1044
|
+
* other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
|
|
1045
|
+
* only needs the payload. A frame whose data is exactly `[DONE]` (the
|
|
1046
|
+
* sentinel several providers, notably OpenAI, send to mark stream end)
|
|
1047
|
+
* ends iteration without yielding it.
|
|
1048
|
+
*
|
|
1049
|
+
* Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
|
|
1050
|
+
* to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
|
|
1051
|
+
* alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
|
|
1052
|
+
* stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
|
|
1053
|
+
* lines.
|
|
1054
|
+
*
|
|
1055
|
+
* Malformed JSON in a frame throws `LLMError('parse')`, consistent with
|
|
1056
|
+
* how malformed JSON is handled elsewhere in VernLLM.
|
|
1057
|
+
*/
|
|
1058
|
+
declare function parseSseStream(source: AsyncIterable<Uint8Array | string>): AsyncGenerator<unknown>;
|
|
1059
|
+
/**
|
|
1060
|
+
* Sentinel yielded by `parseSseStream` for a comment-only frame (no
|
|
1061
|
+
* `data:` payload), the mechanism providers use for SSE keep-alive
|
|
1062
|
+
* pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
|
|
1063
|
+
* alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
|
|
1064
|
+
*/
|
|
1065
|
+
declare const SSE_PING: unique symbol;
|
|
1066
|
+
|
|
1067
|
+
//#endregion
|
|
1068
|
+
//#region src/internal/imageFormat.d.ts
|
|
1069
|
+
//# sourceMappingURL=sse.d.ts.map
|
|
1070
|
+
/**
|
|
1071
|
+
* MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
|
|
1072
|
+
* the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
|
|
1073
|
+
* Converse all natively support, so a `ContentBlock[]` that validates for
|
|
1074
|
+
* one provider validates for all of them.
|
|
1075
|
+
*/
|
|
1076
|
+
declare const SUPPORTED_IMAGE_MIME_TYPES: readonly ["image/png", "image/jpeg", "image/gif", "image/webp"];
|
|
1077
|
+
type SupportedImageMimeType = (typeof SUPPORTED_IMAGE_MIME_TYPES)[number];
|
|
1078
|
+
|
|
1079
|
+
//#endregion
|
|
1080
|
+
//#region src/adapters/anthropic.d.ts
|
|
1081
|
+
/**
|
|
1082
|
+
* Validates an `ImageBlock.mimeType` against the shared supported set.
|
|
1083
|
+
* Throws a non-retryable `LLMError('validation')`, since an unsupported
|
|
1084
|
+
* mimeType is a permanent failure, retrying the same input can't fix it,
|
|
1085
|
+
* the same way a schema-validation or JSON-parse failure isn't retried.
|
|
1086
|
+
*/
|
|
493
1087
|
/** Anthropic's native per-block content shape for a message. */
|
|
494
1088
|
type AnthropicContentBlock = {
|
|
495
1089
|
type: 'text';
|
|
@@ -498,9 +1092,19 @@ type AnthropicContentBlock = {
|
|
|
498
1092
|
type: 'image';
|
|
499
1093
|
source: {
|
|
500
1094
|
type: 'base64';
|
|
501
|
-
media_type:
|
|
1095
|
+
media_type: SupportedImageMimeType;
|
|
502
1096
|
data: string;
|
|
503
1097
|
};
|
|
1098
|
+
} | {
|
|
1099
|
+
type: 'tool_use';
|
|
1100
|
+
id: string;
|
|
1101
|
+
name: string;
|
|
1102
|
+
input: unknown;
|
|
1103
|
+
} | {
|
|
1104
|
+
type: 'tool_result';
|
|
1105
|
+
tool_use_id: string;
|
|
1106
|
+
content: string;
|
|
1107
|
+
is_error?: boolean;
|
|
504
1108
|
};
|
|
505
1109
|
/** Minimal structural type for the Anthropic SDK's `messages.create` */
|
|
506
1110
|
interface AnthropicClient {
|
|
@@ -517,10 +1121,19 @@ interface AnthropicClient {
|
|
|
517
1121
|
tools?: Array<{
|
|
518
1122
|
name: string;
|
|
519
1123
|
description?: string;
|
|
520
|
-
input_schema:
|
|
1124
|
+
input_schema: {
|
|
1125
|
+
type: 'object';
|
|
1126
|
+
[key: string]: unknown;
|
|
1127
|
+
};
|
|
521
1128
|
strict?: boolean;
|
|
522
1129
|
}>;
|
|
523
1130
|
tool_choice?: {
|
|
1131
|
+
type: 'auto';
|
|
1132
|
+
} | {
|
|
1133
|
+
type: 'any';
|
|
1134
|
+
} | {
|
|
1135
|
+
type: 'none';
|
|
1136
|
+
} | {
|
|
524
1137
|
type: 'tool';
|
|
525
1138
|
name: string;
|
|
526
1139
|
};
|
|
@@ -530,6 +1143,7 @@ interface AnthropicClient {
|
|
|
530
1143
|
content: Array<{
|
|
531
1144
|
type: string;
|
|
532
1145
|
text?: string;
|
|
1146
|
+
id?: string;
|
|
533
1147
|
name?: string;
|
|
534
1148
|
input?: unknown;
|
|
535
1149
|
}>;
|
|
@@ -566,13 +1180,30 @@ type GeminiPart = {
|
|
|
566
1180
|
mimeType: string;
|
|
567
1181
|
data: string;
|
|
568
1182
|
};
|
|
1183
|
+
} | {
|
|
1184
|
+
functionCall: {
|
|
1185
|
+
name: string;
|
|
1186
|
+
args: unknown;
|
|
1187
|
+
};
|
|
1188
|
+
} | {
|
|
1189
|
+
functionResponse: {
|
|
1190
|
+
name: string;
|
|
1191
|
+
response: unknown;
|
|
1192
|
+
};
|
|
569
1193
|
};
|
|
570
1194
|
/**
|
|
571
|
-
*
|
|
572
|
-
* `generateContent
|
|
573
|
-
*
|
|
574
|
-
* `
|
|
575
|
-
*
|
|
1195
|
+
* Structural type matching the real `@google/genai` SDK's `ai.models`
|
|
1196
|
+
* object: `generateContent`/`generateContentStream` both take a single
|
|
1197
|
+
* `{ model, contents, config }` argument (config carries
|
|
1198
|
+
* `systemInstruction`, `tools`, `toolConfig`, generation settings, and
|
|
1199
|
+
* `abortSignal` all together), matching the real SDK closely enough that
|
|
1200
|
+
* `fromGemini(ai.models)` works directly, e.g:
|
|
1201
|
+
*
|
|
1202
|
+
* ```ts
|
|
1203
|
+
* import { GoogleGenAI } from '@google/genai';
|
|
1204
|
+
* const ai = new GoogleGenAI({ apiKey: '...' });
|
|
1205
|
+
* const llm = new VernLLM({ client: fromGemini(ai.models), model: 'gemini-2.5-flash' });
|
|
1206
|
+
* ```
|
|
576
1207
|
*/
|
|
577
1208
|
interface GeminiClient {
|
|
578
1209
|
generateContent(params: {
|
|
@@ -581,24 +1212,40 @@ interface GeminiClient {
|
|
|
581
1212
|
role: 'user' | 'model';
|
|
582
1213
|
parts: GeminiPart[];
|
|
583
1214
|
}>;
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
1215
|
+
config?: {
|
|
1216
|
+
systemInstruction?: {
|
|
1217
|
+
parts: Array<{
|
|
1218
|
+
text: string;
|
|
1219
|
+
}>;
|
|
1220
|
+
};
|
|
590
1221
|
temperature?: number;
|
|
591
1222
|
maxOutputTokens?: number;
|
|
592
1223
|
responseMimeType?: string;
|
|
593
1224
|
responseSchema?: Record<string, unknown>;
|
|
1225
|
+
tools?: Array<{
|
|
1226
|
+
functionDeclarations: Array<{
|
|
1227
|
+
name: string;
|
|
1228
|
+
description?: string;
|
|
1229
|
+
parameters: Record<string, unknown>;
|
|
1230
|
+
}>;
|
|
1231
|
+
}>;
|
|
1232
|
+
toolConfig?: {
|
|
1233
|
+
functionCallingConfig: {
|
|
1234
|
+
mode: 'AUTO' | 'ANY' | 'NONE';
|
|
1235
|
+
allowedFunctionNames?: string[];
|
|
1236
|
+
};
|
|
1237
|
+
};
|
|
1238
|
+
abortSignal?: AbortSignal;
|
|
594
1239
|
};
|
|
595
|
-
}, options: {
|
|
596
|
-
signal: AbortSignal;
|
|
597
1240
|
}): Promise<{
|
|
598
1241
|
candidates?: Array<{
|
|
599
1242
|
content?: {
|
|
600
1243
|
parts?: Array<{
|
|
601
1244
|
text?: string;
|
|
1245
|
+
functionCall?: {
|
|
1246
|
+
name: string;
|
|
1247
|
+
args: unknown;
|
|
1248
|
+
};
|
|
602
1249
|
}>;
|
|
603
1250
|
};
|
|
604
1251
|
}>;
|
|
@@ -608,6 +1255,32 @@ interface GeminiClient {
|
|
|
608
1255
|
totalTokenCount?: number;
|
|
609
1256
|
};
|
|
610
1257
|
}>;
|
|
1258
|
+
/**
|
|
1259
|
+
* Optional. Required only for `stream: true` calls. Takes the same
|
|
1260
|
+
* request shape as `generateContent`. Matching the real SDK's own
|
|
1261
|
+
* `generateContentStream`, this resolves to an `AsyncIterable` (rather
|
|
1262
|
+
* than returning one synchronously) of partial responses, each chunk
|
|
1263
|
+
* holding the same `candidates[].content.parts[]` structure as
|
|
1264
|
+
* `generateContent`'s response, just incremental.
|
|
1265
|
+
*/
|
|
1266
|
+
generateContentStream?(params: Parameters<GeminiClient['generateContent']>[0]): Promise<AsyncIterable<{
|
|
1267
|
+
candidates?: Array<{
|
|
1268
|
+
content?: {
|
|
1269
|
+
parts?: Array<{
|
|
1270
|
+
text?: string;
|
|
1271
|
+
functionCall?: {
|
|
1272
|
+
name: string;
|
|
1273
|
+
args: unknown;
|
|
1274
|
+
};
|
|
1275
|
+
}>;
|
|
1276
|
+
};
|
|
1277
|
+
}>;
|
|
1278
|
+
usageMetadata?: {
|
|
1279
|
+
promptTokenCount?: number;
|
|
1280
|
+
candidatesTokenCount?: number;
|
|
1281
|
+
totalTokenCount?: number;
|
|
1282
|
+
};
|
|
1283
|
+
}>>;
|
|
611
1284
|
}
|
|
612
1285
|
/**
|
|
613
1286
|
* Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
|
|
@@ -619,6 +1292,25 @@ interface GeminiClient {
|
|
|
619
1292
|
* `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
|
|
620
1293
|
* models use a token budget, not an effort tier, so it's dropped, same as
|
|
621
1294
|
* Anthropic.
|
|
1295
|
+
*
|
|
1296
|
+
* `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
|
|
1297
|
+
* `tool_choice` maps to `toolConfig.functionCallingConfig`. `jsonSchema`
|
|
1298
|
+
* and `tools` are mutually exclusive by the time a call reaches here
|
|
1299
|
+
* (enforced in vernLLM.ts), so `responseSchema` and `tools` never
|
|
1300
|
+
* both apply.
|
|
1301
|
+
*
|
|
1302
|
+
* `createStream` calls `generateContentStream` (optional on `GeminiClient`
|
|
1303
|
+
*, required only if the caller sets `stream: true`) and translates each
|
|
1304
|
+
* partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
|
|
1305
|
+
* Gemini's own function-calling API doesn't stream tool-call arguments
|
|
1306
|
+
* incrementally: a `functionCall` part always arrives whole in one chunk,
|
|
1307
|
+
* so each one is emitted as a single, complete `tool_call_delta` (a
|
|
1308
|
+
* one-shot "delta" containing the full arguments) rather than accumulated
|
|
1309
|
+
* fragments, that's a real difference in the underlying API, not
|
|
1310
|
+
* something this adapter can smooth over. `usageMetadata` is (per Gemini's
|
|
1311
|
+
* own behavior) only reliably present on the last chunk, so the `usage`
|
|
1312
|
+
* `WireStreamChunk` is emitted once, after the stream completes, from
|
|
1313
|
+
* whichever chunk's `usageMetadata` was seen last.
|
|
622
1314
|
*/
|
|
623
1315
|
declare function fromGemini(geminiClient: GeminiClient): LLMClient;
|
|
624
1316
|
|
|
@@ -636,6 +1328,20 @@ type BedrockContentBlock = {
|
|
|
636
1328
|
bytes: Uint8Array;
|
|
637
1329
|
};
|
|
638
1330
|
};
|
|
1331
|
+
} | {
|
|
1332
|
+
toolUse: {
|
|
1333
|
+
toolUseId: string;
|
|
1334
|
+
name: string;
|
|
1335
|
+
input: unknown;
|
|
1336
|
+
};
|
|
1337
|
+
} | {
|
|
1338
|
+
toolResult: {
|
|
1339
|
+
toolUseId: string;
|
|
1340
|
+
content: Array<{
|
|
1341
|
+
text: string;
|
|
1342
|
+
}>;
|
|
1343
|
+
status?: 'success' | 'error';
|
|
1344
|
+
};
|
|
639
1345
|
};
|
|
640
1346
|
/**
|
|
641
1347
|
* Minimal structural type matching AWS Bedrock's Converse API. This is
|
|
@@ -682,6 +1388,10 @@ interface BedrockConverseClient {
|
|
|
682
1388
|
tool: {
|
|
683
1389
|
name: string;
|
|
684
1390
|
};
|
|
1391
|
+
} | {
|
|
1392
|
+
auto: Record<string, never>;
|
|
1393
|
+
} | {
|
|
1394
|
+
any: Record<string, never>;
|
|
685
1395
|
};
|
|
686
1396
|
};
|
|
687
1397
|
}, options: {
|
|
@@ -692,6 +1402,7 @@ interface BedrockConverseClient {
|
|
|
692
1402
|
content?: Array<{
|
|
693
1403
|
text?: string;
|
|
694
1404
|
toolUse?: {
|
|
1405
|
+
toolUseId?: string;
|
|
695
1406
|
name?: string;
|
|
696
1407
|
input?: unknown;
|
|
697
1408
|
};
|
|
@@ -704,7 +1415,89 @@ interface BedrockConverseClient {
|
|
|
704
1415
|
totalTokens?: number;
|
|
705
1416
|
};
|
|
706
1417
|
}>;
|
|
1418
|
+
/**
|
|
1419
|
+
* Optional. Required only for `stream: true` calls. Takes the same
|
|
1420
|
+
* request shape `converse` does, returning `{ stream }`, matching
|
|
1421
|
+
* `ConverseStreamCommand`'s real AWS SDK v3 output shape, an
|
|
1422
|
+
* `AsyncIterable` of incremental events under a `stream` property,
|
|
1423
|
+
* rather than the whole response being the iterable directly.
|
|
1424
|
+
*/
|
|
1425
|
+
converseStream?(params: Parameters<BedrockConverseClient['converse']>[0], options: {
|
|
1426
|
+
signal: AbortSignal;
|
|
1427
|
+
}): Promise<{
|
|
1428
|
+
stream: AsyncIterable<BedrockConverseStreamEvent>;
|
|
1429
|
+
}>;
|
|
707
1430
|
}
|
|
1431
|
+
/**
|
|
1432
|
+
* One event of a Bedrock `ConverseStreamCommand` response's `stream`.
|
|
1433
|
+
* Content blocks (text or toolUse) are identified by `contentBlockIndex`,
|
|
1434
|
+
* Converse's own convention for correlating start/delta/stop events across
|
|
1435
|
+
* possibly-interleaved blocks, mirrored directly by VernLLM's
|
|
1436
|
+
* `tool_call_delta.index`.
|
|
1437
|
+
*/
|
|
1438
|
+
type BedrockConverseStreamEvent = {
|
|
1439
|
+
messageStart: {
|
|
1440
|
+
role: 'assistant';
|
|
1441
|
+
};
|
|
1442
|
+
} | {
|
|
1443
|
+
contentBlockStart: {
|
|
1444
|
+
contentBlockIndex: number;
|
|
1445
|
+
start?: {
|
|
1446
|
+
toolUse?: {
|
|
1447
|
+
toolUseId?: string;
|
|
1448
|
+
name?: string;
|
|
1449
|
+
};
|
|
1450
|
+
};
|
|
1451
|
+
};
|
|
1452
|
+
} | {
|
|
1453
|
+
contentBlockDelta: {
|
|
1454
|
+
contentBlockIndex: number;
|
|
1455
|
+
delta?: {
|
|
1456
|
+
text?: string;
|
|
1457
|
+
} | {
|
|
1458
|
+
toolUse?: {
|
|
1459
|
+
input?: string;
|
|
1460
|
+
};
|
|
1461
|
+
};
|
|
1462
|
+
};
|
|
1463
|
+
} | {
|
|
1464
|
+
contentBlockStop: {
|
|
1465
|
+
contentBlockIndex: number;
|
|
1466
|
+
};
|
|
1467
|
+
} | {
|
|
1468
|
+
messageStop: {
|
|
1469
|
+
stopReason?: string;
|
|
1470
|
+
};
|
|
1471
|
+
} | {
|
|
1472
|
+
metadata: {
|
|
1473
|
+
usage?: {
|
|
1474
|
+
inputTokens?: number;
|
|
1475
|
+
outputTokens?: number;
|
|
1476
|
+
totalTokens?: number;
|
|
1477
|
+
};
|
|
1478
|
+
};
|
|
1479
|
+
} | {
|
|
1480
|
+
internalServerException: {
|
|
1481
|
+
message?: string;
|
|
1482
|
+
};
|
|
1483
|
+
} | {
|
|
1484
|
+
modelStreamErrorException: {
|
|
1485
|
+
message?: string;
|
|
1486
|
+
originalStatusCode?: number;
|
|
1487
|
+
};
|
|
1488
|
+
} | {
|
|
1489
|
+
validationException: {
|
|
1490
|
+
message?: string;
|
|
1491
|
+
};
|
|
1492
|
+
} | {
|
|
1493
|
+
throttlingException: {
|
|
1494
|
+
message?: string;
|
|
1495
|
+
};
|
|
1496
|
+
} | {
|
|
1497
|
+
serviceUnavailableException: {
|
|
1498
|
+
message?: string;
|
|
1499
|
+
};
|
|
1500
|
+
};
|
|
708
1501
|
/**
|
|
709
1502
|
* Optional configuration for `fromBedrock`.
|
|
710
1503
|
*/
|
|
@@ -743,6 +1536,22 @@ interface BedrockAdapterOptions {
|
|
|
743
1536
|
* `response_format: json_object` (no schema to build a tool from) and
|
|
744
1537
|
* `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
|
|
745
1538
|
* instruction and are dropped respectively.
|
|
1539
|
+
*
|
|
1540
|
+
* `tools` maps to Converse's native `toolConfig`/`toolUse`/`toolResult`;
|
|
1541
|
+
* `tool_choice` maps to `toolConfig.toolChoice`. Mutually exclusive with
|
|
1542
|
+
* `jsonSchema` by the time a call reaches here (enforced in vernLLM.ts).
|
|
1543
|
+
*
|
|
1544
|
+
* `createStream` calls `converseStream` (optional on `BedrockConverseClient`
|
|
1545
|
+
*, required only if the caller sets `stream: true`) and translates its
|
|
1546
|
+
* `contentBlockStart`/`contentBlockDelta`/`metadata` events into
|
|
1547
|
+
* `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
|
|
1548
|
+
* same as `fromAnthropic`'s block-index tracking (Converse's streaming
|
|
1549
|
+
* shape is structurally close to Anthropic's own, both being tool-use-aware
|
|
1550
|
+
* content-block streams), including the same `json-tool` unwrapping: a
|
|
1551
|
+
* `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
|
|
1552
|
+
* `text-delta`, not `tool_call_delta`, so the accumulated result lands in
|
|
1553
|
+
* `finalizeResponse`'s `content` path exactly like the non-streaming
|
|
1554
|
+
* `create` branch above unwraps it.
|
|
746
1555
|
*/
|
|
747
1556
|
declare function fromBedrock(bedrockClient: BedrockConverseClient, options?: BedrockAdapterOptions): LLMClient;
|
|
748
1557
|
|
|
@@ -772,6 +1581,22 @@ type RequestLike = (url: string, init: {
|
|
|
772
1581
|
body?: string;
|
|
773
1582
|
signal?: AbortSignal;
|
|
774
1583
|
}) => Promise<ResponseLike>;
|
|
1584
|
+
/**
|
|
1585
|
+
* A streaming-capable request function. Unlike `RequestLike`, which returns
|
|
1586
|
+
* a fully-buffered `ResponseLike`, this resolves to an `AsyncIterable` of
|
|
1587
|
+
* progressively-arriving chunks, the common ground across transports:
|
|
1588
|
+
* native `fetch`'s `response.body` (wrapped to be iterable; see
|
|
1589
|
+
* `webStreamToAsyncIterable` below), axios's Node `Readable` in
|
|
1590
|
+
* `responseType: 'stream'` mode (already async-iterable, no wrapping
|
|
1591
|
+
* needed), `node-fetch`, `undici`, etc, all satisfy this with little or no
|
|
1592
|
+
* glue code. Defaults to native `fetch`.
|
|
1593
|
+
*/
|
|
1594
|
+
type StreamRequestLike = (url: string, init: {
|
|
1595
|
+
method: string;
|
|
1596
|
+
headers: Record<string, string>;
|
|
1597
|
+
body?: string;
|
|
1598
|
+
signal?: AbortSignal;
|
|
1599
|
+
}) => Promise<AsyncIterable<Uint8Array | string>>;
|
|
775
1600
|
interface FetchAdapterConfig {
|
|
776
1601
|
/** Endpoint URL, or a function of the request in case it depends on model/params */
|
|
777
1602
|
url: string | ((params: ChatRequest) => string);
|
|
@@ -788,17 +1613,65 @@ interface FetchAdapterConfig {
|
|
|
788
1613
|
/** Maps VernLLMs internal chat-completion request into the providers raw request body */
|
|
789
1614
|
mapRequest: (params: ChatRequest) => unknown;
|
|
790
1615
|
/**
|
|
791
|
-
* Maps the providers raw JSON response into `{ content, usage? }`
|
|
792
|
-
* `content` is the assistants text (JSON string when JSON mode was requested)
|
|
1616
|
+
* Maps the providers raw JSON response into `{ content, usage?, toolCalls? }`
|
|
1617
|
+
* `content` is the assistants text (JSON string when JSON mode was requested).
|
|
1618
|
+
* `content` may be empty/omitted when the model responded with only tool
|
|
1619
|
+
* calls and no text.
|
|
1620
|
+
*
|
|
1621
|
+
* `toolCalls`, when the model requested one or more tools, is the list of
|
|
1622
|
+
* calls as flat `{ id, name, arguments }` entries (matching this config's
|
|
1623
|
+
* own `toolCalls?: Array<{ id: string; name: string; arguments: string }>`
|
|
1624
|
+
* return type below), each entry's `arguments` already JSON-*encoded* as a
|
|
1625
|
+
* string (not the parsed object), mirroring the wire format every
|
|
1626
|
+
* OpenAI-compatible provider uses. `fromFetch` itself converts these into
|
|
1627
|
+
* `WireToolCall`'s `type`/`function`-wrapped shape before returning them
|
|
1628
|
+
* from `create`. VernLLM parses (and validates, if `argumentsSchema` was
|
|
1629
|
+
* set) the arguments string internally, mapResponse doesn't need to do
|
|
1630
|
+
* that itself.
|
|
793
1631
|
*/
|
|
794
1632
|
mapResponse: (json: unknown) => {
|
|
795
|
-
content
|
|
1633
|
+
content?: string;
|
|
796
1634
|
usage?: {
|
|
797
1635
|
promptTokens?: number;
|
|
798
1636
|
completionTokens?: number;
|
|
799
1637
|
totalTokens?: number;
|
|
800
1638
|
};
|
|
1639
|
+
toolCalls?: Array<{
|
|
1640
|
+
id: string;
|
|
1641
|
+
name: string;
|
|
1642
|
+
arguments: string;
|
|
1643
|
+
}>;
|
|
801
1644
|
};
|
|
1645
|
+
/**
|
|
1646
|
+
* Optional. Required only for `stream: true` calls. The function used to
|
|
1647
|
+
* open a streaming HTTP request. Takes the same request shape as
|
|
1648
|
+
* `request`, but resolves to an `AsyncIterable` of progressively-arriving
|
|
1649
|
+
* `Uint8Array` or `string` chunks instead of a buffered `ResponseLike`.
|
|
1650
|
+
* Defaults to native `fetch`.
|
|
1651
|
+
*/
|
|
1652
|
+
requestStream?: StreamRequestLike;
|
|
1653
|
+
/**
|
|
1654
|
+
* Optional. How the raw stream bytes are split into individual event
|
|
1655
|
+
* payloads. Defaults to Server-Sent Events framing (`data: ...` blocks
|
|
1656
|
+
* separated by a blank line, `[DONE]` sentinel honored, see
|
|
1657
|
+
* `parseSseStream`), which covers the large majority of LLM providers'
|
|
1658
|
+
* streaming HTTP endpoints. Override this for a provider that frames its
|
|
1659
|
+
* stream differently, e.g. newline-delimited JSON (NDJSON) with no SSE
|
|
1660
|
+
* envelope.
|
|
1661
|
+
*/
|
|
1662
|
+
parseStreamFrames?: (chunks: AsyncIterable<Uint8Array | string>) => AsyncIterable<unknown>;
|
|
1663
|
+
/**
|
|
1664
|
+
* Optional. Required only for `stream: true` calls. Maps one parsed
|
|
1665
|
+
* stream event (already extracted from its frame by `parseStreamFrames`)
|
|
1666
|
+
* into zero, one, or more `WireStreamChunk`s, mirrors `mapResponse`'s
|
|
1667
|
+
* role for the non-streaming path, just per-event instead of once for
|
|
1668
|
+
* the whole body. Return `undefined` to skip an event that carries
|
|
1669
|
+
* nothing VernLLM needs (e.g. a provider's keep-alive ping). Configs
|
|
1670
|
+
* that don't implement this make `stream: true` throw a clear
|
|
1671
|
+
* `LLMError('validation')` rather than a confusing runtime failure or a
|
|
1672
|
+
* silently empty stream.
|
|
1673
|
+
*/
|
|
1674
|
+
mapStreamEvent?: (event: unknown) => WireStreamChunk | WireStreamChunk[] | undefined;
|
|
802
1675
|
}
|
|
803
1676
|
/**
|
|
804
1677
|
* A fetch-based escape hatch for providers with no SDK, or where pulling one
|
|
@@ -810,6 +1683,34 @@ interface FetchAdapterConfig {
|
|
|
810
1683
|
* Non-2xx responses throw an error with `.status` set to the HTTP status
|
|
811
1684
|
* code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
|
|
812
1685
|
* 401/403) applies here too
|
|
1686
|
+
*
|
|
1687
|
+
* Tool calling works the same way as every other adapter: `mapRequest`
|
|
1688
|
+
* receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
|
|
1689
|
+
* translate them into whatever shape the provider's wire format expects
|
|
1690
|
+
* (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
|
|
1691
|
+
* field). On the way back, `mapResponse` may return a `toolCalls` array
|
|
1692
|
+
* (id/name/JSON-encoded-arguments-string per call) alongside or instead of
|
|
1693
|
+
* `content`; VernLLM parses and (if `argumentsSchema` was set) validates
|
|
1694
|
+
* those arguments the same way it does for every other adapter. For
|
|
1695
|
+
* `stream: true`, tool-call deltas go through the existing
|
|
1696
|
+
* `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
|
|
1697
|
+
* no separate config is needed for streaming vs non-streaming tool calls.
|
|
1698
|
+
*
|
|
1699
|
+
|
|
1700
|
+
* `createStream` requires `mapStreamEvent` (there's no non-streaming
|
|
1701
|
+
* response to fall back on, unlike the other three optional streaming
|
|
1702
|
+
* seams). It opens the request via `requestStream` (defaults to native
|
|
1703
|
+
* `fetch`), splits the raw bytes into individual events via
|
|
1704
|
+
* `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
|
|
1705
|
+
* and translates each event into `WireStreamChunk`(s) via
|
|
1706
|
+
* `mapStreamEvent`. Both seams are overridable per-config for providers
|
|
1707
|
+
* that don't fit the SSE-over-fetch default. If a custom `request`
|
|
1708
|
+
* transport is configured, `requestStream` must be configured too,
|
|
1709
|
+
* `requestStream` never silently falls back to `request` (see
|
|
1710
|
+
* `createStream`'s own comment for why), so a `stream: true` call with
|
|
1711
|
+
* `request` set but no `requestStream` throws a clear
|
|
1712
|
+
* `LLMError('validation')` instead of quietly using unrelated native
|
|
1713
|
+
* `fetch`.
|
|
813
1714
|
*/
|
|
814
1715
|
declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
815
1716
|
|
|
@@ -834,11 +1735,46 @@ declare function fromFetch(config: FetchAdapterConfig): LLMClient;
|
|
|
834
1735
|
* (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
|
|
835
1736
|
* the actual compatibility contract is the JSON each provider sends and
|
|
836
1737
|
* receives over the wire, not the SDKs TS types.
|
|
1738
|
+
*
|
|
1739
|
+
* `createStream` is implemented by calling the same underlying
|
|
1740
|
+
* `chat.completions.create` with `stream: true` (and, for providers that
|
|
1741
|
+
* support it, `stream_options: { include_usage: true }`, so a final usage
|
|
1742
|
+
* block arrives), the OpenAI SDK, and every OpenAI-compatible client
|
|
1743
|
+
* modeled on it, returns an `AsyncIterable` of SSE chunks instead of a
|
|
1744
|
+
* single completion object when `stream: true` is set. Each chunk is
|
|
1745
|
+
* translated into `WireStreamChunk`(s) via `toWireStreamChunks`.
|
|
1746
|
+
*
|
|
1747
|
+
* Note on long-running reasoning models: this adapter consumes the
|
|
1748
|
+
* underlying SDK's already-parsed stream rather than raw SSE bytes, so
|
|
1749
|
+
* unlike `fromFetch`/`fromAnthropic` it cannot see comment-only keep-alive
|
|
1750
|
+
* ping frames. Combined with `chunkIdleTimeoutMs`'s 30 second default and
|
|
1751
|
+
* `reasoningEffort` (documented to have long silent gaps for o-series and
|
|
1752
|
+
* similar models), a long-running reasoning call on this adapter can trip
|
|
1753
|
+
* the idle timeout even though the provider is still working. Raise or
|
|
1754
|
+
* disable `chunkIdleTimeoutMs` per call for those routes, see `CallParams`.
|
|
837
1755
|
*/
|
|
838
|
-
|
|
1756
|
+
interface OpenAICompatibleAdapterOptions {
|
|
1757
|
+
/**
|
|
1758
|
+
* Whether the provider supports `stream_options.include_usage`. Not
|
|
1759
|
+
* every "OpenAI-compatible" provider is guaranteed to, so this defaults
|
|
1760
|
+
* to `true` (matching OpenAI, Groq, Mistral, and most others observed)
|
|
1761
|
+
* and should be set to `false` for a provider verified not to support
|
|
1762
|
+
* it. When `false`, `stream_options` is omitted entirely and no usage
|
|
1763
|
+
* block will arrive on the stream; callers relying on streamed `usage`
|
|
1764
|
+
* with such a provider won't get one.
|
|
1765
|
+
*/
|
|
1766
|
+
supportsStreamUsage?: boolean;
|
|
1767
|
+
}
|
|
1768
|
+
declare function fromOpenAICompatible(client: unknown, options?: OpenAICompatibleAdapterOptions): LLMClient;
|
|
839
1769
|
/** Groqs SDK matches the OpenAI wire format */
|
|
840
1770
|
declare const fromGroq: typeof fromOpenAICompatible;
|
|
841
|
-
/**
|
|
1771
|
+
/**
|
|
1772
|
+
* Mistrals `chat.completions`-shaped client (or their OpenAI-compat
|
|
1773
|
+
* endpoint). Mistral supports `stream_options.include_usage` (added after
|
|
1774
|
+
* an earlier period where it returned a 422 for unrecognized fields, per
|
|
1775
|
+
* Mistral's changelog and streaming docs), so this is a plain alias like
|
|
1776
|
+
* the others, `supportsStreamUsage` defaults to `true`.
|
|
1777
|
+
*/
|
|
842
1778
|
declare const fromMistral: typeof fromOpenAICompatible;
|
|
843
1779
|
/** DeepSeeks API is OpenAI-compatible */
|
|
844
1780
|
declare const fromDeepSeek: typeof fromOpenAICompatible;
|
|
@@ -929,5 +1865,5 @@ declare const from01AI: typeof fromOpenAICompatible;
|
|
|
929
1865
|
//#endregion
|
|
930
1866
|
//# sourceMappingURL=openaiCompatible.d.ts.map
|
|
931
1867
|
|
|
932
|
-
export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CallParams, CircuitBreaker, CircuitBreakerOptions, ConsoleLogger, ContentBlock, ConversationTurn, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorType, Logger, NormalizedCacheAdapter, OnUsage, RefundUsage, ReserveUsage, SchemaLike, TextBlock, TieredCacheAdapter, TokenUsage, VernLLM, VernLLMOptions, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError };
|
|
1868
|
+
export { AnthropicClient, BedrockConverseClient, CacheAdapter, CachedCallParams, CachedStreamCallParams, CachedStreamToolCallParams, CachedToolCallParams, CallParams, CallWithToolsResult, CircuitBreaker, CircuitBreakerOptions, ConsoleLogger, ContentBlock, ContentResult, ConversationTurn, FetchAdapterConfig, GeminiClient, ImageBlock, InMemoryCacheAdapter, JsonSchemaSpec, LLMClient, LLMError, LLMErrorType, Logger, NormalizedCacheAdapter, OnUsage, RefundUsage, ReserveUsage, SSE_PING, SchemaLike, StreamCallResult, StreamChunk, StreamEnabledCallParams, TextBlock, TieredCacheAdapter, TokenUsage, ToolCall, ToolCallResult, ToolChoice, ToolDefinition, ToolEnabledCallParams, ToolResult, VernLLM, VernLLMOptions, WireMessage, WireStreamChunk, WireToolCall, WireToolChoice, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError, isToolCallResult, parseSseStream };
|
|
933
1869
|
//# sourceMappingURL=index.d.cts.map
|