vern-llm 1.7.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -6
- package/dist/index.cjs +1784 -272
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1009 -73
- package/dist/index.d.cts.map +1 -1
- package/dist/index.d.ts +1009 -73
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1782 -273
- package/dist/index.js.map +1 -1
- package/package.json +9 -13
package/dist/index.js
CHANGED
|
@@ -51,7 +51,7 @@ var CircuitBreaker = class {
|
|
|
51
51
|
if (this.state === "closed") return;
|
|
52
52
|
if (this.state === "open") {
|
|
53
53
|
const elapsed = Date.now() - this.openedAt;
|
|
54
|
-
if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open
|
|
54
|
+
if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open, provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
|
|
55
55
|
this.state = "half-open";
|
|
56
56
|
this.trialInFlight = true;
|
|
57
57
|
return;
|
|
@@ -84,6 +84,48 @@ var CircuitBreaker = class {
|
|
|
84
84
|
|
|
85
85
|
//#endregion
|
|
86
86
|
//#region src/internal/vernLLM.utils.ts
|
|
87
|
+
/** Translates app-facing `ToolDefinition[]` into the OpenAI-shaped wire tools array. */
|
|
88
|
+
function toWireTools(tools) {
|
|
89
|
+
return tools.map((tool) => ({
|
|
90
|
+
type: "function",
|
|
91
|
+
function: {
|
|
92
|
+
name: tool.name,
|
|
93
|
+
description: tool.description,
|
|
94
|
+
parameters: tool.parameters
|
|
95
|
+
}
|
|
96
|
+
}));
|
|
97
|
+
}
|
|
98
|
+
/** Translates app-facing `ToolCall[]` (e.g. from a replayed assistant turn) into wire tool_calls. */
|
|
99
|
+
function toWireToolCalls(toolCalls) {
|
|
100
|
+
return toolCalls.map((tc) => ({
|
|
101
|
+
id: tc.id,
|
|
102
|
+
type: "function",
|
|
103
|
+
function: {
|
|
104
|
+
name: tc.name,
|
|
105
|
+
arguments: JSON.stringify(tc.arguments ?? {})
|
|
106
|
+
}
|
|
107
|
+
}));
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Parses the provider's wire-shaped `tool_calls` back into VernLLM's
|
|
111
|
+
* `ToolCall[]`. Malformed argument JSON is a `'parse'` error, same
|
|
112
|
+
* convention as malformed JSON response bodies elsewhere in VernLLM.
|
|
113
|
+
*/
|
|
114
|
+
function parseWireToolCalls(wireToolCalls) {
|
|
115
|
+
return wireToolCalls.map((wc) => {
|
|
116
|
+
let parsedArgs;
|
|
117
|
+
try {
|
|
118
|
+
parsedArgs = wc.function.arguments.trim() ? JSON.parse(wc.function.arguments) : {};
|
|
119
|
+
} catch {
|
|
120
|
+
throw new LLMError(`Invalid JSON arguments for tool call "${wc.function.name}"`, "parse");
|
|
121
|
+
}
|
|
122
|
+
return {
|
|
123
|
+
id: wc.id,
|
|
124
|
+
name: wc.function.name,
|
|
125
|
+
arguments: parsedArgs
|
|
126
|
+
};
|
|
127
|
+
});
|
|
128
|
+
}
|
|
87
129
|
function defaultParseJson(content) {
|
|
88
130
|
try {
|
|
89
131
|
return JSON.parse(content);
|
|
@@ -94,7 +136,9 @@ function defaultParseJson(content) {
|
|
|
94
136
|
/**
|
|
95
137
|
* Looks inside an unknown error value and pulls out an http status code
|
|
96
138
|
* if one is present. Checks the status field first then the status code
|
|
97
|
-
* field since different client libraries use different names for this
|
|
139
|
+
* field since different client libraries use different names for this,
|
|
140
|
+
* falling back to AWS SDK v3's `$metadata.httpStatusCode` (e.g. Bedrock's
|
|
141
|
+
* `ThrottlingException`), which doesn't set either of the other two.
|
|
98
142
|
* Returns undefined when the error is not an object or carries no status
|
|
99
143
|
*/
|
|
100
144
|
function extractStatus(err) {
|
|
@@ -102,6 +146,7 @@ function extractStatus(err) {
|
|
|
102
146
|
const error = err;
|
|
103
147
|
if (typeof error.status === "number") return error.status;
|
|
104
148
|
if (typeof error.statusCode === "number") return error.statusCode;
|
|
149
|
+
if (typeof error.$metadata?.httpStatusCode === "number") return error.$metadata.httpStatusCode;
|
|
105
150
|
return void 0;
|
|
106
151
|
}
|
|
107
152
|
function formatSafely(value) {
|
|
@@ -131,6 +176,21 @@ function describeError(err) {
|
|
|
131
176
|
return formatSafely(err);
|
|
132
177
|
}
|
|
133
178
|
/**
|
|
179
|
+
* `setTimeout` silently clamps any delay above this (~24.8 days) or
|
|
180
|
+
* `Infinity` down to ~1ms instead of erroring, so a caller passing
|
|
181
|
+
* `Infinity` as "no timeout" gets the opposite of what they asked for.
|
|
182
|
+
* Both timeout helpers below guard against this explicitly.
|
|
183
|
+
*/
|
|
184
|
+
const MAX_SETTIMEOUT_MS = 2147483647;
|
|
185
|
+
/** True when a timeout value should be treated as "disabled" rather than passed to `setTimeout`. */
|
|
186
|
+
function isTimeoutDisabled(ms) {
|
|
187
|
+
return !ms || ms <= 0 || ms === Infinity;
|
|
188
|
+
}
|
|
189
|
+
/** Caps a timeout at the largest delay `setTimeout` actually honors. */
|
|
190
|
+
function clampTimeoutMs(ms) {
|
|
191
|
+
return Math.min(ms, MAX_SETTIMEOUT_MS);
|
|
192
|
+
}
|
|
193
|
+
/**
|
|
134
194
|
* Runs an async function and cancels it if it takes longer than the given
|
|
135
195
|
* timeout. Creates an internal abort controller that fires after the
|
|
136
196
|
* timeout elapses, and combines it with any external signal the caller
|
|
@@ -140,12 +200,15 @@ function describeError(err) {
|
|
|
140
200
|
* continue to propagate as aborted errors. The internal timer is always
|
|
141
201
|
* cleared afterward, whether the function succeeds, fails, or is aborted,
|
|
142
202
|
* so nothing is left running in the background.
|
|
203
|
+
*
|
|
204
|
+
* `timeoutMs` of `Infinity` (or any value beyond what `setTimeout` can
|
|
205
|
+
* represent) disables the timeout rather than firing almost immediately.
|
|
143
206
|
*/
|
|
144
207
|
async function withTimeout(fn, timeoutMs, externalSignal) {
|
|
145
208
|
const controller = new AbortController();
|
|
146
|
-
const timer = setTimeout(() => {
|
|
209
|
+
const timer = isTimeoutDisabled(timeoutMs) ? void 0 : setTimeout(() => {
|
|
147
210
|
controller.abort();
|
|
148
|
-
}, timeoutMs);
|
|
211
|
+
}, clampTimeoutMs(timeoutMs));
|
|
149
212
|
const signal = externalSignal ? AbortSignal.any([externalSignal, controller.signal]) : controller.signal;
|
|
150
213
|
try {
|
|
151
214
|
return await fn(signal);
|
|
@@ -157,6 +220,54 @@ async function withTimeout(fn, timeoutMs, externalSignal) {
|
|
|
157
220
|
}
|
|
158
221
|
}
|
|
159
222
|
/**
|
|
223
|
+
* Races one `iterator.next()` call against a per-call idle timer, to
|
|
224
|
+
* bound the gap *between* chunks (unlike `withTimeout`, which only bounds
|
|
225
|
+
* opening the stream and its first chunk). Without this, a connection
|
|
226
|
+
* that streams one chunk then hangs would never fail.
|
|
227
|
+
*
|
|
228
|
+
* `timeoutMs` of 0/undefined/`Infinity` disables the check. Otherwise
|
|
229
|
+
* rejects with `LLMError('timeout')` if `next()` doesn't settle in time.
|
|
230
|
+
* The clock resets on every call, so the window is measured from the most
|
|
231
|
+
* recent chunk, not from stream start.
|
|
232
|
+
*
|
|
233
|
+
* `onIdle`, if given, is called the moment the timer fires (before the
|
|
234
|
+
* rejection), so callers can abort the underlying transport instead of
|
|
235
|
+
* just walking away from an unread promise. `logger`, if given, records a
|
|
236
|
+
* debug line if `next()` still settles *after* the idle timeout already
|
|
237
|
+
* rejected. `resolve`/`reject` on an already-settled promise is otherwise
|
|
238
|
+
* a silent no-op, so without this the late chunk (possibly the final
|
|
239
|
+
* usage chunk) would vanish with no trace.
|
|
240
|
+
*/
|
|
241
|
+
function withChunkIdleTimeout(next, timeoutMs, onIdle, logger) {
|
|
242
|
+
if (isTimeoutDisabled(timeoutMs)) return next();
|
|
243
|
+
const activeTimeoutMs = timeoutMs;
|
|
244
|
+
let settled = false;
|
|
245
|
+
return new Promise((resolve, reject) => {
|
|
246
|
+
const timer = setTimeout(() => {
|
|
247
|
+
settled = true;
|
|
248
|
+
onIdle?.();
|
|
249
|
+
reject(new LLMError(`No stream chunk received for ${activeTimeoutMs}ms (idle timeout)`, "timeout"));
|
|
250
|
+
}, clampTimeoutMs(activeTimeoutMs));
|
|
251
|
+
next().then((result) => {
|
|
252
|
+
clearTimeout(timer);
|
|
253
|
+
if (settled) {
|
|
254
|
+
logger?.debug("[VernLLM] chunk resolved after idle timeout already fired; discarding");
|
|
255
|
+
return;
|
|
256
|
+
}
|
|
257
|
+
settled = true;
|
|
258
|
+
resolve(result);
|
|
259
|
+
}, (error) => {
|
|
260
|
+
clearTimeout(timer);
|
|
261
|
+
if (settled) {
|
|
262
|
+
logger?.debug("[VernLLM] chunk rejection arrived after idle timeout already fired; discarding");
|
|
263
|
+
return;
|
|
264
|
+
}
|
|
265
|
+
settled = true;
|
|
266
|
+
reject(error);
|
|
267
|
+
});
|
|
268
|
+
});
|
|
269
|
+
}
|
|
270
|
+
/**
|
|
160
271
|
* Default cap (ms) for both exponential backoff and honored Retry-After
|
|
161
272
|
* values, so a misbehaving/adversarial Retry-After can't stall a caller
|
|
162
273
|
* indefinitely
|
|
@@ -271,6 +382,143 @@ async function withReservedUsage(params, coalesced, getResult, signal, onRefundE
|
|
|
271
382
|
}
|
|
272
383
|
return result;
|
|
273
384
|
}
|
|
385
|
+
/**
|
|
386
|
+
* Streaming counterpart to `withReservedUsage`. `withReservedUsage` assumes
|
|
387
|
+
* `getResult()` settling *is* the operation's final outcome, awaiting it
|
|
388
|
+
* synchronously before reserve/refund resolve. Streaming can't satisfy that:
|
|
389
|
+
* `call()` must return `{ chunks, finalResult }` as soon as the stream
|
|
390
|
+
* opens, well before the real outcome (validation, schema/tool-call checks)
|
|
391
|
+
* is known.
|
|
392
|
+
*
|
|
393
|
+
* Reserves usage before `openStream` runs, same failure mode as the
|
|
394
|
+
* non-streaming path if `reserveUsage` itself throws (mapped to
|
|
395
|
+
* `quota_exceeded`, nothing opened). If `openStream` itself throws (stream
|
|
396
|
+
* never opened), refunds synchronously and rethrows, exactly like
|
|
397
|
+
* `withReservedUsage` does today. If it succeeds, returns `{ chunks,
|
|
398
|
+
* finalResult }` immediately, refund/report is deferred onto
|
|
399
|
+
* `finalResult`'s continuation, since that's the only point the real
|
|
400
|
+
* outcome is known. This means `onUsageFailure` (and any refund) can fire
|
|
401
|
+
* well after this function itself has returned.
|
|
402
|
+
*/
|
|
403
|
+
async function withReservedUsageForStream(params, openStream, signal, onRefundError) {
|
|
404
|
+
if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
|
|
405
|
+
let reserved = false;
|
|
406
|
+
try {
|
|
407
|
+
if (params.reserveUsage) {
|
|
408
|
+
await params.reserveUsage({
|
|
409
|
+
coalesced: false,
|
|
410
|
+
signal
|
|
411
|
+
});
|
|
412
|
+
reserved = true;
|
|
413
|
+
}
|
|
414
|
+
} catch (error) {
|
|
415
|
+
if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
|
|
416
|
+
throw new LLMError(error instanceof Error ? error.message : "Usage reservation failed", "quota_exceeded", void 0, void 0, error);
|
|
417
|
+
}
|
|
418
|
+
const refund = async (logMessage) => {
|
|
419
|
+
try {
|
|
420
|
+
await params.refundUsage?.({
|
|
421
|
+
coalesced: false,
|
|
422
|
+
signal
|
|
423
|
+
});
|
|
424
|
+
} catch (refundError) {
|
|
425
|
+
onRefundError(logMessage, refundError);
|
|
426
|
+
}
|
|
427
|
+
};
|
|
428
|
+
let opened;
|
|
429
|
+
try {
|
|
430
|
+
opened = await openStream();
|
|
431
|
+
} catch (error) {
|
|
432
|
+
if (reserved) await refund("[VernLLM] refundUsage failed after stream-open failure");
|
|
433
|
+
throw error;
|
|
434
|
+
}
|
|
435
|
+
const finalResult = opened.finalResult.then((value) => value, async (error) => {
|
|
436
|
+
if (reserved) await refund("[VernLLM] refundUsage failed after stream error");
|
|
437
|
+
throw error;
|
|
438
|
+
});
|
|
439
|
+
finalResult.catch(() => {});
|
|
440
|
+
return {
|
|
441
|
+
chunks: opened.chunks,
|
|
442
|
+
finalResult
|
|
443
|
+
};
|
|
444
|
+
}
|
|
445
|
+
/**
|
|
446
|
+
* Converts an already-known cache value back into a plausible "text" form
|
|
447
|
+
* for a one-shot replay chunk: passed through unchanged if it's already a
|
|
448
|
+
* string (the `jsonMode: false` case), otherwise `JSON.stringify`'d (the
|
|
449
|
+
* `jsonMode: true` case, where the cached value is the *parsed* result, not
|
|
450
|
+
* the original raw text). This is a reasonable reconstruction, not a
|
|
451
|
+
* byte-identical replay of whatever text the model originally streamed,
|
|
452
|
+
* good enough for `for await (const c of chunks)` call sites that don't
|
|
453
|
+
* branch on hit vs. miss, which is the only thing a cache-hit replay needs
|
|
454
|
+
* to support.
|
|
455
|
+
*/
|
|
456
|
+
function toReplayText(value) {
|
|
457
|
+
return typeof value === "string" ? value : JSON.stringify(value);
|
|
458
|
+
}
|
|
459
|
+
/**
|
|
460
|
+
* Builds a trivially-exhausted one-shot `chunks` iterable from an
|
|
461
|
+
* already-known value, used for a `cachedCall` cache hit, where there's no
|
|
462
|
+
* live generation to relay (see `VernLLM.cachedCall`'s docs). No `usage`
|
|
463
|
+
* chunk is emitted: a cache hit spent no real tokens, so there's nothing to
|
|
464
|
+
* report, matching how non-streaming `cachedCall` never calls `onUsage` on
|
|
465
|
+
* a hit either.
|
|
466
|
+
*
|
|
467
|
+
* `hasTools` must reflect whether the *original* call that produced this
|
|
468
|
+
* cached value had `tools` set, that's what determines whether `value` is
|
|
469
|
+
* `T` directly or a `CallWithToolsResult<T>` wrapper, and it isn't
|
|
470
|
+
* something that can be reliably guessed from the value's shape alone
|
|
471
|
+
* (a `schema`-validated `T` could coincidentally look like a
|
|
472
|
+
* `CallWithToolsResult`).
|
|
473
|
+
*/
|
|
474
|
+
function buildReplayChunks(value, hasTools) {
|
|
475
|
+
const items = [];
|
|
476
|
+
if (hasTools) {
|
|
477
|
+
const result = value;
|
|
478
|
+
if (result.type === "tool_calls") {
|
|
479
|
+
result.toolCalls.forEach((toolCall, index) => {
|
|
480
|
+
items.push({
|
|
481
|
+
type: "tool_call_delta",
|
|
482
|
+
index,
|
|
483
|
+
id: toolCall.id,
|
|
484
|
+
name: toolCall.name,
|
|
485
|
+
argsDelta: JSON.stringify(toolCall.arguments ?? {}),
|
|
486
|
+
complete: true
|
|
487
|
+
});
|
|
488
|
+
});
|
|
489
|
+
if (result.content) items.push({
|
|
490
|
+
type: "text-delta",
|
|
491
|
+
delta: result.content
|
|
492
|
+
});
|
|
493
|
+
} else items.push({
|
|
494
|
+
type: "text-delta",
|
|
495
|
+
delta: toReplayText(result.content)
|
|
496
|
+
});
|
|
497
|
+
} else items.push({
|
|
498
|
+
type: "text-delta",
|
|
499
|
+
delta: toReplayText(value)
|
|
500
|
+
});
|
|
501
|
+
return { async *[Symbol.asyncIterator]() {
|
|
502
|
+
for (const item of items) yield item;
|
|
503
|
+
} };
|
|
504
|
+
}
|
|
505
|
+
/**
|
|
506
|
+
* Streaming counterpart to `buildReplayChunks` for a `cachedCall` that
|
|
507
|
+
* *joined* an already-in-flight call for the same key rather than
|
|
508
|
+
* triggering one itself (see `runCachedStream`'s in-flight-coalescing
|
|
509
|
+
* path): there's no live stream to relay (it isn't this call's stream to
|
|
510
|
+
* relay, see the joiner-path comment in `runCachedStream`), but there's
|
|
511
|
+
* also no value yet, only a pending promise for one. Waits for `promise`,
|
|
512
|
+
* then delegates to `buildReplayChunks`. If `promise` rejects, iterating
|
|
513
|
+
* `chunks` throws that same error, consistent with how a live stream's
|
|
514
|
+
* `chunks` throws on a mid-stream failure.
|
|
515
|
+
*/
|
|
516
|
+
function buildReplayChunksFromPromise(promise, hasTools) {
|
|
517
|
+
return { async *[Symbol.asyncIterator]() {
|
|
518
|
+
const value = await promise;
|
|
519
|
+
yield* buildReplayChunks(value, hasTools);
|
|
520
|
+
} };
|
|
521
|
+
}
|
|
274
522
|
|
|
275
523
|
//#endregion
|
|
276
524
|
//#region src/logger.ts
|
|
@@ -403,34 +651,52 @@ var TieredCacheAdapter = class {
|
|
|
403
651
|
}
|
|
404
652
|
};
|
|
405
653
|
|
|
654
|
+
//#endregion
|
|
655
|
+
//#region src/types/tools.ts
|
|
656
|
+
/**
|
|
657
|
+
* Runtime-safe check for whether a `call()` result is a `tool_calls`
|
|
658
|
+
* result. Prefer this over relying on TypeScript's static narrowing
|
|
659
|
+
* whenever `params` passed to `call()` wasn't a literal with `tools`
|
|
660
|
+
* inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
|
|
661
|
+
* that case TS may have typed the result as plain `T` even though it's
|
|
662
|
+
* actually a `CallWithToolsResult<T>` at runtime, and this check works
|
|
663
|
+
* either way.
|
|
664
|
+
*/
|
|
665
|
+
function isToolCallResult(result) {
|
|
666
|
+
return typeof result === "object" && result !== null && "type" in result && result.type === "tool_calls" && Array.isArray(result.toolCalls);
|
|
667
|
+
}
|
|
668
|
+
|
|
406
669
|
//#endregion
|
|
407
670
|
//#region src/vernLLM.ts
|
|
408
671
|
/**
|
|
409
|
-
* A resilient layer around an LLM chat completions client
|
|
672
|
+
* A resilient layer around an LLM chat completions client. This is VernLLM!
|
|
410
673
|
*
|
|
411
|
-
* Adds retry with backoff
|
|
412
|
-
* JSON parsing with optional schema validation, usage
|
|
413
|
-
* optional response cache
|
|
414
|
-
* defaults.
|
|
674
|
+
* Adds retry with backoff and jitter, per-attempt timeouts, an optional
|
|
675
|
+
* circuit breaker, JSON parsing with optional schema validation, usage
|
|
676
|
+
* tracking, and an optional response cache. All configurable, all opt-in
|
|
677
|
+
* beyond sensible defaults.
|
|
415
678
|
*/
|
|
416
679
|
var VernLLM = class {
|
|
417
680
|
client;
|
|
418
681
|
model;
|
|
419
682
|
maxRetries;
|
|
420
683
|
timeoutMs;
|
|
684
|
+
chunkIdleTimeoutMs;
|
|
421
685
|
baseDelayMs;
|
|
422
686
|
defaultMaxTokens;
|
|
687
|
+
defaultTemperature;
|
|
423
688
|
cache;
|
|
424
689
|
nonRetryableStatus;
|
|
425
690
|
inFlight = new Map();
|
|
426
691
|
parseJson;
|
|
427
692
|
onUsage;
|
|
693
|
+
onUsageFailure;
|
|
428
694
|
logger;
|
|
429
695
|
breaker;
|
|
430
696
|
/**
|
|
431
|
-
* @param options
|
|
432
|
-
* `
|
|
433
|
-
*
|
|
697
|
+
* @param options Client, model, and tunables. Defaults: `maxRetries` 1,
|
|
698
|
+
* `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
|
|
699
|
+
* `defaultTemperature` 0.2, `cache` an in-memory adapter,
|
|
434
700
|
* `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
|
|
435
701
|
*/
|
|
436
702
|
constructor(options) {
|
|
@@ -438,8 +704,10 @@ var VernLLM = class {
|
|
|
438
704
|
this.model = options.model;
|
|
439
705
|
this.maxRetries = options.maxRetries ?? 1;
|
|
440
706
|
this.timeoutMs = options.timeoutMs ?? 25e3;
|
|
707
|
+
this.chunkIdleTimeoutMs = options.chunkIdleTimeoutMs ?? 3e4;
|
|
441
708
|
this.baseDelayMs = options.baseDelayMs ?? 500;
|
|
442
709
|
this.defaultMaxTokens = options.defaultMaxTokens ?? 1e3;
|
|
710
|
+
this.defaultTemperature = options.defaultTemperature === void 0 ? .2 : options.defaultTemperature;
|
|
443
711
|
this.cache = options.cache ?? new InMemoryCacheAdapter();
|
|
444
712
|
this.nonRetryableStatus = options.nonRetryableStatus ?? [
|
|
445
713
|
400,
|
|
@@ -450,6 +718,7 @@ var VernLLM = class {
|
|
|
450
718
|
];
|
|
451
719
|
this.parseJson = options.parseJson ?? defaultParseJson;
|
|
452
720
|
this.onUsage = options.onUsage;
|
|
721
|
+
this.onUsageFailure = options.onUsageFailure;
|
|
453
722
|
this.logger = options.logger ?? new ConsoleLogger(options.debug ?? false);
|
|
454
723
|
this.breaker = options.circuitBreaker ? new CircuitBreaker(options.circuitBreaker === true ? void 0 : options.circuitBreaker) : void 0;
|
|
455
724
|
}
|
|
@@ -457,38 +726,307 @@ var VernLLM = class {
|
|
|
457
726
|
async resolveCacheKey(key) {
|
|
458
727
|
return this.cache.resolveKey ? await this.cache.resolveKey(key) : key;
|
|
459
728
|
}
|
|
460
|
-
/**
|
|
461
|
-
* Makes a single logical LLM call, retrying on failure per the configured
|
|
462
|
-
* policy. Fails fast if the breaker is open or the signal is already
|
|
463
|
-
* aborted. On exhausting retries, records a breaker failure and rejects
|
|
464
|
-
* with a normalized LLMError.
|
|
465
|
-
*
|
|
466
|
-
* @param params - System/user content plus per-call overrides (model,
|
|
467
|
-
* temperature, jsonMode, schema, signal, etc). See `CallParams`.
|
|
468
|
-
* @returns The parsed (and optionally schema-validated) response, or the
|
|
469
|
-
* raw string content when `jsonMode` is false and no `jsonSchema` is set.
|
|
470
|
-
*/
|
|
471
729
|
async call(params) {
|
|
472
730
|
this.breaker?.assertClosed();
|
|
473
731
|
if (params.signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
|
|
474
732
|
const requestId = params.requestId ?? randomUUID();
|
|
733
|
+
if (params.stream) return withReservedUsageForStream(params, async () => {
|
|
734
|
+
try {
|
|
735
|
+
return await this.retryWithBackoff((attempt) => this.executeStreamCall(params, requestId, attempt), requestId, params.signal);
|
|
736
|
+
} catch (error) {
|
|
737
|
+
const normalized = normalizeError(error, params.signal);
|
|
738
|
+
if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
|
|
739
|
+
this.logger.debug(`[VernLLM:${requestId}] stream-open error:\n${describeError(error)}`);
|
|
740
|
+
throw normalized;
|
|
741
|
+
}
|
|
742
|
+
}, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
475
743
|
return withReservedUsage(params, false, async () => {
|
|
476
744
|
try {
|
|
477
|
-
return await this.retryWithBackoff(() => this.executeCall(params, requestId), requestId, params.signal);
|
|
745
|
+
return await this.retryWithBackoff((attempt) => this.executeCall(params, requestId, attempt), requestId, params.signal);
|
|
478
746
|
} catch (error) {
|
|
479
747
|
const normalized = normalizeError(error, params.signal);
|
|
480
748
|
if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
|
|
481
|
-
this.logger.debug(`[
|
|
749
|
+
this.logger.debug(`[VernLLM:${requestId}] error:\n${describeError(error)}`);
|
|
482
750
|
throw normalized;
|
|
483
751
|
}
|
|
484
752
|
}, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
485
753
|
}
|
|
754
|
+
/**
|
|
755
|
+
* Performs a single attempt: builds the request (translating `tools` to
|
|
756
|
+
* wire shape when present), dispatches it with a timeout, and shapes the
|
|
757
|
+
* response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
|
|
758
|
+
* set. Throws on an empty response (no text and no tool_calls) so the
|
|
759
|
+
* retry loop treats it like any other transient failure.
|
|
760
|
+
*/
|
|
761
|
+
async executeCall(params, requestId, attempt) {
|
|
762
|
+
const { useJson, model, request } = this.buildRequestPayload(params);
|
|
763
|
+
const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
|
|
764
|
+
const usage = this.extractUsage(response, requestId, model);
|
|
765
|
+
const rawContent = response.choices?.[0]?.message?.content;
|
|
766
|
+
const wireToolCalls = response.choices?.[0]?.message?.tool_calls;
|
|
767
|
+
return this.finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt);
|
|
768
|
+
}
|
|
769
|
+
/**
|
|
770
|
+
* Shapes a fully-arrived response (content and/or tool_calls, already
|
|
771
|
+
* extracted from the provider's payload) into `T` or a
|
|
772
|
+
* `CallWithToolsResult<T>`. Reused by the streaming path once it has
|
|
773
|
+
* buffered the full text/tool-call deltas, so there's no separate
|
|
774
|
+
* parsing/validation logic for streaming.
|
|
775
|
+
*
|
|
776
|
+
* Normalizes and reports usage failure on error itself, so every caller
|
|
777
|
+
* gets identical error handling without duplicating it.
|
|
778
|
+
*/
|
|
779
|
+
finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt) {
|
|
780
|
+
try {
|
|
781
|
+
const content = rawContent?.trim();
|
|
782
|
+
if (!content && !wireToolCalls?.length) throw new LLMError("Empty LLM response", "api");
|
|
783
|
+
this.logger.debug(`[VernLLM:${requestId}] output:\n${(content ?? `[${wireToolCalls?.length ?? 0} tool call(s)]`).slice(0, 800)}`);
|
|
784
|
+
if (wireToolCalls?.length) {
|
|
785
|
+
if (!params.tools) throw new LLMError("Provider returned tool_calls but no `tools` were sent with this call.", "api");
|
|
786
|
+
const toolCalls = parseWireToolCalls(wireToolCalls);
|
|
787
|
+
this.validateToolCallArguments(toolCalls, params.tools);
|
|
788
|
+
this.breaker?.recordSuccess();
|
|
789
|
+
this.reportUsage(usage);
|
|
790
|
+
return {
|
|
791
|
+
type: "tool_calls",
|
|
792
|
+
toolCalls,
|
|
793
|
+
...content ? { content } : {}
|
|
794
|
+
};
|
|
795
|
+
}
|
|
796
|
+
const textContent = content ?? "";
|
|
797
|
+
if (!useJson) {
|
|
798
|
+
this.breaker?.recordSuccess();
|
|
799
|
+
this.reportUsage(usage);
|
|
800
|
+
return params.tools ? {
|
|
801
|
+
type: "content",
|
|
802
|
+
content: textContent
|
|
803
|
+
} : textContent;
|
|
804
|
+
}
|
|
805
|
+
const result = this.parseAndValidate(textContent, params.schema);
|
|
806
|
+
this.breaker?.recordSuccess();
|
|
807
|
+
this.reportUsage(usage);
|
|
808
|
+
return params.tools ? {
|
|
809
|
+
type: "content",
|
|
810
|
+
content: result
|
|
811
|
+
} : result;
|
|
812
|
+
} catch (error) {
|
|
813
|
+
const normalized = normalizeError(error, params.signal);
|
|
814
|
+
if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt);
|
|
815
|
+
throw normalized;
|
|
816
|
+
}
|
|
817
|
+
}
|
|
818
|
+
/**
|
|
819
|
+
* Opens a stream for a single attempt: builds the request exactly like
|
|
820
|
+
* `executeCall`, then requires `createStream` on the client (a clear
|
|
821
|
+
* `validation` error if the adapter doesn't support it). The timeout
|
|
822
|
+
* wraps stream construction and the first `.next()` together, not just
|
|
823
|
+
* construction: calling an `async function*` returns an iterator
|
|
824
|
+
* synchronously without running its body until `.next()` is first
|
|
825
|
+
* invoked, so timing only construction would time an operation that's
|
|
826
|
+
* always instant, not the actual connection. Both are folded into a
|
|
827
|
+
* single `withTimeout` so the same abort signal reaches whatever the
|
|
828
|
+
* adapter's `createStream` uses internally for its first network
|
|
829
|
+
* round-trip.
|
|
830
|
+
*
|
|
831
|
+
* Circuit-breaker success is recorded once the stream fully completes,
|
|
832
|
+
* not on the first chunk arriving, so a connection that opens but then
|
|
833
|
+
* dies mid-stream isn't masked as a success (see `buildStreamResult`).
|
|
834
|
+
*/
|
|
835
|
+
async executeStreamCall(params, requestId, attempt) {
|
|
836
|
+
const { useJson, model, request } = this.buildRequestPayload(params);
|
|
837
|
+
const createStream = this.client.chat.completions.createStream;
|
|
838
|
+
if (!createStream) throw new LLMError("stream: true requires a client/adapter with createStream", "validation");
|
|
839
|
+
const streamController = new AbortController();
|
|
840
|
+
const combinedExternal = params.signal ? AbortSignal.any([params.signal, streamController.signal]) : streamController.signal;
|
|
841
|
+
const { iterator, first } = await withTimeout(async (attemptSignal) => {
|
|
842
|
+
const streamIterator = createStream(request, { signal: attemptSignal })[Symbol.asyncIterator]();
|
|
843
|
+
const firstResult = await streamIterator.next();
|
|
844
|
+
return {
|
|
845
|
+
iterator: streamIterator,
|
|
846
|
+
first: firstResult
|
|
847
|
+
};
|
|
848
|
+
}, this.timeoutMs, combinedExternal);
|
|
849
|
+
if (first.done) throw new LLMError("Empty LLM response", "api");
|
|
850
|
+
return this.buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController);
|
|
851
|
+
}
|
|
852
|
+
/**
|
|
853
|
+
* The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
|
|
854
|
+
* an async generator that yields translated `StreamChunk`s to the caller
|
|
855
|
+
* live, as they arrive, with no per-chunk timeout and no bound on total
|
|
856
|
+
* duration, and accumulates text/tool-call deltas internally so that
|
|
857
|
+
* `finalizeResponse` can produce `finalResult` once the stream completes.
|
|
858
|
+
*
|
|
859
|
+
* Two separate try/catches: the iteration loop's catch handles errors
|
|
860
|
+
* the transport itself throws, which aren't normalized yet, so that
|
|
861
|
+
* happens here along with the one `reportUsageFailure` call for them.
|
|
862
|
+
* The second catch, around `finalizeResponse`, does not re-normalize or
|
|
863
|
+
* re-report since `finalizeResponse` already does both internally.
|
|
864
|
+
* Circuit-breaker success is only recorded once the stream fully
|
|
865
|
+
* completes, not when the first chunk arrives, so a connection that
|
|
866
|
+
* opens and then dies mid-way still counts as a failure below instead
|
|
867
|
+
* of masking it.
|
|
868
|
+
*/
|
|
869
|
+
buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController) {
|
|
870
|
+
let resolveFinal;
|
|
871
|
+
let rejectFinal;
|
|
872
|
+
const finalResult = new Promise((resolve, reject) => {
|
|
873
|
+
resolveFinal = resolve;
|
|
874
|
+
rejectFinal = reject;
|
|
875
|
+
});
|
|
876
|
+
finalResult.catch(() => {});
|
|
877
|
+
const MAX_BUFFERED_CHUNKS = 1e4;
|
|
878
|
+
const buffered = [];
|
|
879
|
+
const pending = [];
|
|
880
|
+
let streamDone = false;
|
|
881
|
+
let streamError;
|
|
882
|
+
let hasLoggedEviction = false;
|
|
883
|
+
const push = (chunk) => {
|
|
884
|
+
const waiter = pending.shift();
|
|
885
|
+
if (waiter) {
|
|
886
|
+
waiter.resolve({
|
|
887
|
+
done: false,
|
|
888
|
+
value: chunk
|
|
889
|
+
});
|
|
890
|
+
return;
|
|
891
|
+
}
|
|
892
|
+
buffered.push(chunk);
|
|
893
|
+
if (buffered.length > MAX_BUFFERED_CHUNKS * 2) {
|
|
894
|
+
if (!hasLoggedEviction) {
|
|
895
|
+
hasLoggedEviction = true;
|
|
896
|
+
this.logger.debug(`[VernLLM] stream chunk buffer exceeded cap (${MAX_BUFFERED_CHUNKS}), evicting ${buffered.length - MAX_BUFFERED_CHUNKS} oldest chunk(s); buffered=${buffered.length}. The chunks iterable was never read (or fell far behind) for this stream.`);
|
|
897
|
+
}
|
|
898
|
+
buffered.splice(0, buffered.length - MAX_BUFFERED_CHUNKS);
|
|
899
|
+
}
|
|
900
|
+
};
|
|
901
|
+
const finish = () => {
|
|
902
|
+
streamDone = true;
|
|
903
|
+
for (const waiter of pending.splice(0)) waiter.resolve({
|
|
904
|
+
done: true,
|
|
905
|
+
value: void 0
|
|
906
|
+
});
|
|
907
|
+
};
|
|
908
|
+
const fail = (error) => {
|
|
909
|
+
streamDone = true;
|
|
910
|
+
streamError = error;
|
|
911
|
+
for (const waiter of pending.splice(0)) waiter.reject(error);
|
|
912
|
+
};
|
|
913
|
+
const chunks = { [Symbol.asyncIterator]() {
|
|
914
|
+
return { next() {
|
|
915
|
+
if (buffered.length) return Promise.resolve({
|
|
916
|
+
done: false,
|
|
917
|
+
value: buffered.shift()
|
|
918
|
+
});
|
|
919
|
+
if (streamDone) return streamError ? Promise.reject(streamError) : Promise.resolve({
|
|
920
|
+
done: true,
|
|
921
|
+
value: void 0
|
|
922
|
+
});
|
|
923
|
+
return new Promise((resolve, reject) => {
|
|
924
|
+
pending.push({
|
|
925
|
+
resolve,
|
|
926
|
+
reject
|
|
927
|
+
});
|
|
928
|
+
});
|
|
929
|
+
} };
|
|
930
|
+
} };
|
|
931
|
+
const toolCallAcc = new Map();
|
|
932
|
+
let textAcc = "";
|
|
933
|
+
let usage;
|
|
934
|
+
(async () => {
|
|
935
|
+
try {
|
|
936
|
+
let result = first;
|
|
937
|
+
while (!result.done) {
|
|
938
|
+
const wireChunk = result.value;
|
|
939
|
+
if (wireChunk.type === "ping") {} else if (wireChunk.type === "text-delta") {
|
|
940
|
+
textAcc += wireChunk.delta;
|
|
941
|
+
push({
|
|
942
|
+
type: "text-delta",
|
|
943
|
+
delta: wireChunk.delta
|
|
944
|
+
});
|
|
945
|
+
} else if (wireChunk.type === "tool_call_delta") {
|
|
946
|
+
const entry = toolCallAcc.get(wireChunk.index) ?? { args: "" };
|
|
947
|
+
entry.id ??= wireChunk.id;
|
|
948
|
+
entry.name ??= wireChunk.name;
|
|
949
|
+
entry.args += wireChunk.argumentsDelta ?? "";
|
|
950
|
+
toolCallAcc.set(wireChunk.index, entry);
|
|
951
|
+
push({
|
|
952
|
+
type: "tool_call_delta",
|
|
953
|
+
index: wireChunk.index,
|
|
954
|
+
id: wireChunk.id,
|
|
955
|
+
name: wireChunk.name,
|
|
956
|
+
argsDelta: wireChunk.argumentsDelta,
|
|
957
|
+
complete: wireChunk.complete
|
|
958
|
+
});
|
|
959
|
+
} else if (wireChunk.type === "usage") {
|
|
960
|
+
usage = {
|
|
961
|
+
promptTokens: wireChunk.usage.prompt_tokens ?? 0,
|
|
962
|
+
completionTokens: wireChunk.usage.completion_tokens ?? 0,
|
|
963
|
+
totalTokens: wireChunk.usage.total_tokens ?? 0,
|
|
964
|
+
requestId,
|
|
965
|
+
model
|
|
966
|
+
};
|
|
967
|
+
push({
|
|
968
|
+
type: "usage",
|
|
969
|
+
usage
|
|
970
|
+
});
|
|
971
|
+
}
|
|
972
|
+
result = await withChunkIdleTimeout(() => iterator.next(), params.chunkIdleTimeoutMs ?? this.chunkIdleTimeoutMs, () => streamController.abort(), this.logger);
|
|
973
|
+
}
|
|
974
|
+
} catch (error) {
|
|
975
|
+
try {
|
|
976
|
+
await iterator.return?.();
|
|
977
|
+
} catch {}
|
|
978
|
+
streamController.abort();
|
|
979
|
+
const normalized = normalizeError(error, params.signal);
|
|
980
|
+
if (normalized.type === "timeout") this.breaker?.recordFailure();
|
|
981
|
+
if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt, true);
|
|
982
|
+
fail(normalized);
|
|
983
|
+
rejectFinal(normalized);
|
|
984
|
+
return;
|
|
985
|
+
}
|
|
986
|
+
finish();
|
|
987
|
+
this.breaker?.recordSuccess();
|
|
988
|
+
try {
|
|
989
|
+
const wireToolCalls = toolCallAcc.size ? [...toolCallAcc.entries()].sort(([indexA], [indexB]) => indexA - indexB).map(([, entry]) => ({
|
|
990
|
+
id: entry.id ?? "",
|
|
991
|
+
type: "function",
|
|
992
|
+
function: {
|
|
993
|
+
name: entry.name ?? "",
|
|
994
|
+
arguments: entry.args
|
|
995
|
+
}
|
|
996
|
+
})) : void 0;
|
|
997
|
+
const finalized = this.finalizeResponse(textAcc, wireToolCalls, params, useJson, usage, requestId, attempt);
|
|
998
|
+
resolveFinal(finalized);
|
|
999
|
+
} catch (error) {
|
|
1000
|
+
rejectFinal(error);
|
|
1001
|
+
}
|
|
1002
|
+
})();
|
|
1003
|
+
return {
|
|
1004
|
+
chunks,
|
|
1005
|
+
finalResult
|
|
1006
|
+
};
|
|
1007
|
+
}
|
|
1008
|
+
/**
|
|
1009
|
+
* Checks every `ToolCall` against the `tools` that were offered, catching
|
|
1010
|
+
* a hallucinated tool name early instead of letting it reach the
|
|
1011
|
+
* application's dispatch table. Then runs each tool's `argumentsSchema`,
|
|
1012
|
+
* if present, throwing `LLMError('validation')` on failure.
|
|
1013
|
+
*/
|
|
1014
|
+
validateToolCallArguments(toolCalls, tools) {
|
|
1015
|
+
const knownNames = new Set(tools.map((t) => t.name));
|
|
1016
|
+
for (const call of toolCalls) {
|
|
1017
|
+
if (!knownNames.has(call.name)) throw new LLMError(`Model requested tool "${call.name}", which was not in the tools offered ([${[...knownNames].join(", ")}]).`, "api");
|
|
1018
|
+
const definition = tools.find((t) => t.name === call.name);
|
|
1019
|
+
if (!definition?.argumentsSchema) continue;
|
|
1020
|
+
const result = definition.argumentsSchema.safeParse(call.arguments);
|
|
1021
|
+
if (!result.success) throw new LLMError(`Arguments for tool call "${call.name}" failed validation`, "validation", void 0, result.error);
|
|
1022
|
+
}
|
|
1023
|
+
}
|
|
486
1024
|
/** Runs `fn`, retrying with backoff according to `shouldRetry`. */
|
|
487
1025
|
async retryWithBackoff(fn, requestId, signal) {
|
|
488
1026
|
let lastError;
|
|
489
1027
|
for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
|
|
490
1028
|
if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
|
|
491
|
-
return await fn();
|
|
1029
|
+
return await fn(attempt);
|
|
492
1030
|
} catch (error) {
|
|
493
1031
|
lastError = error;
|
|
494
1032
|
if (!this.shouldRetry(error, signal)) break;
|
|
@@ -496,60 +1034,73 @@ var VernLLM = class {
|
|
|
496
1034
|
throw lastError;
|
|
497
1035
|
}
|
|
498
1036
|
/**
|
|
499
|
-
* Performs a single attempt: builds the request, dispatches it with a
|
|
500
|
-
* timeout, and shapes the response. Throws on an empty response so the
|
|
501
|
-
* retry loop treats it like any other transient failure.
|
|
502
|
-
*/
|
|
503
|
-
async executeCall(params, requestId) {
|
|
504
|
-
const { useJson, model, request } = this.buildRequestPayload(params);
|
|
505
|
-
const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
|
|
506
|
-
const content = response.choices?.[0]?.message?.content?.trim();
|
|
507
|
-
if (!content) throw new LLMError("Empty LLM response", "api");
|
|
508
|
-
this.logger.debug(`[vern:${requestId}] output:\n${content.slice(0, 800)}`);
|
|
509
|
-
this.recordUsage(response, requestId, model);
|
|
510
|
-
if (!useJson) {
|
|
511
|
-
this.breaker?.recordSuccess();
|
|
512
|
-
return content;
|
|
513
|
-
}
|
|
514
|
-
const result = this.parseAndValidate(content, params.schema);
|
|
515
|
-
this.breaker?.recordSuccess();
|
|
516
|
-
return result;
|
|
517
|
-
}
|
|
518
|
-
/**
|
|
519
1037
|
* Validates `history` alternates user/assistant turns, since providers
|
|
520
1038
|
* like Anthropic/Gemini reject or mishandle consecutive same-role turns.
|
|
521
1039
|
*/
|
|
522
1040
|
validateHistory(history) {
|
|
523
|
-
let
|
|
1041
|
+
let previousTurn;
|
|
524
1042
|
for (const [index, turn] of history.entries()) {
|
|
525
|
-
if (turn.role
|
|
526
|
-
|
|
527
|
-
|
|
1043
|
+
if (turn.role === "tool") {
|
|
1044
|
+
if (previousTurn?.role !== "assistant" || !previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] is a "tool" turn, but must immediately follow an "assistant" turn that requested tools`, "validation");
|
|
1045
|
+
if (!turn.toolResults?.length) throw new LLMError(`history[${index}] is a "tool" turn but has no toolResults`, "validation");
|
|
1046
|
+
const requestedIds = new Set(previousTurn.toolCalls.map((tc) => tc.id));
|
|
1047
|
+
const resultIds = turn.toolResults.map((tr) => tr.toolCallId);
|
|
1048
|
+
const unknownIds = resultIds.filter((id) => !requestedIds.has(id));
|
|
1049
|
+
if (unknownIds.length) throw new LLMError(`history[${index}].toolResults references unknown toolCallId(s) [${unknownIds.join(", ")}]`, "validation");
|
|
1050
|
+
const seenIds = new Set();
|
|
1051
|
+
const duplicateIds = new Set();
|
|
1052
|
+
for (const id of resultIds) {
|
|
1053
|
+
if (seenIds.has(id)) duplicateIds.add(id);
|
|
1054
|
+
seenIds.add(id);
|
|
1055
|
+
}
|
|
1056
|
+
if (duplicateIds.size) throw new LLMError(`history[${index}].toolResults has duplicate toolCallId(s) [${[...duplicateIds].join(", ")}]`, "validation");
|
|
1057
|
+
const missingIds = [...requestedIds].filter((id) => !resultIds.includes(id));
|
|
1058
|
+
if (missingIds.length) throw new LLMError(`history[${index}] is missing toolResults for toolCallId(s) [${missingIds.join(", ")}]`, "validation");
|
|
1059
|
+
} else {
|
|
1060
|
+
if (turn.role === previousTurn?.role) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
|
|
1061
|
+
if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] follows an assistant tool request without tool results`, "validation");
|
|
1062
|
+
}
|
|
1063
|
+
previousTurn = turn;
|
|
528
1064
|
}
|
|
529
|
-
if (
|
|
1065
|
+
if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError("The last entry in history is an assistant tool request without tool results", "validation");
|
|
1066
|
+
if (previousTurn?.role === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn.", "validation");
|
|
530
1067
|
}
|
|
531
1068
|
/** Applies per-call defaults and shapes params into the client's request object. */
|
|
532
1069
|
buildRequestPayload(params) {
|
|
533
|
-
const { systemPrompt, userContent, history = [],
|
|
1070
|
+
const { systemPrompt, userContent, history = [], maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema, tools, toolChoice } = params;
|
|
1071
|
+
const temperature = params.temperature === void 0 ? this.defaultTemperature : params.temperature;
|
|
1072
|
+
if (tools && (jsonSchema || params.schema)) throw new LLMError("`tools` cannot be combined with `jsonSchema`/`schema`: on Anthropic and Bedrock, jsonSchema is implemented internally as a forced single-tool call, which would collide with real tools. Use one or the other.", "validation");
|
|
1073
|
+
if (tools && tools.length === 0) throw new LLMError("`tools` was an empty array. This is almost always a bug (e.g. a filtered tool list that ended up empty). An empty `tools` array still switches on tool-call mode (response shape, jsonMode default, wire format) with nothing for the model to call. Omit `tools` entirely for a normal call, or make sure the array is non-empty.", "validation");
|
|
1074
|
+
if (tools) {
|
|
1075
|
+
const seen = new Set();
|
|
1076
|
+
const duplicates = new Set();
|
|
1077
|
+
for (const tool of tools) {
|
|
1078
|
+
if (seen.has(tool.name)) duplicates.add(tool.name);
|
|
1079
|
+
seen.add(tool.name);
|
|
1080
|
+
}
|
|
1081
|
+
if (duplicates.size) throw new LLMError(`\`tools\` has duplicate name(s): [${[...duplicates].join(", ")}]. Tool names must be unique.`, "validation");
|
|
1082
|
+
}
|
|
1083
|
+
if (toolChoice && !tools) throw new LLMError("`toolChoice` was set without `tools`. There is nothing for it to choose between. Set `tools`, or remove `toolChoice`.", "validation");
|
|
1084
|
+
if (tools && typeof toolChoice === "object" && !tools.some((t) => t.name === toolChoice.name)) throw new LLMError(`toolChoice names "${toolChoice.name}", which is not in \`tools\` ([${tools.map((t) => t.name).join(", ")}]).`, "validation");
|
|
1085
|
+
const jsonMode = params.jsonMode ?? (tools ? false : true);
|
|
534
1086
|
const useJson = jsonMode || Boolean(jsonSchema);
|
|
535
1087
|
if (params.schema && !useJson) throw new LLMError("schema was provided but jsonMode: false disables JSON parsing, so nothing would validate it. Remove jsonMode: false, set jsonSchema, or remove schema.", "validation");
|
|
536
1088
|
const responseFormat = this.buildResponseFormat(jsonSchema, useJson);
|
|
537
1089
|
this.validateHistory(history);
|
|
538
1090
|
const request = {
|
|
539
1091
|
model,
|
|
540
|
-
temperature,
|
|
1092
|
+
...temperature !== null ? { temperature } : {},
|
|
541
1093
|
max_tokens: maxTokens,
|
|
542
1094
|
...responseFormat ? { response_format: responseFormat } : {},
|
|
543
1095
|
...reasoningEffort ? { reasoning_effort: reasoningEffort } : {},
|
|
1096
|
+
...tools ? { tools: toWireTools(tools) } : {},
|
|
1097
|
+
...tools ? { tool_choice: this.buildWireToolChoice(toolChoice) } : {},
|
|
544
1098
|
messages: [
|
|
545
1099
|
...systemPrompt ? [{
|
|
546
1100
|
role: "system",
|
|
547
1101
|
content: systemPrompt
|
|
548
1102
|
}] : [],
|
|
549
|
-
...history.
|
|
550
|
-
role: turn.role,
|
|
551
|
-
content: turn.content
|
|
552
|
-
})),
|
|
1103
|
+
...history.flatMap((turn) => this.turnToWireMessages(turn)),
|
|
553
1104
|
{
|
|
554
1105
|
role: "user",
|
|
555
1106
|
content: userContent
|
|
@@ -562,6 +1113,39 @@ var VernLLM = class {
|
|
|
562
1113
|
request
|
|
563
1114
|
};
|
|
564
1115
|
}
|
|
1116
|
+
/** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
|
|
1117
|
+
buildWireToolChoice(toolChoice) {
|
|
1118
|
+
if (!toolChoice || toolChoice === "auto") return "auto";
|
|
1119
|
+
if (toolChoice === "none" || toolChoice === "required") return toolChoice;
|
|
1120
|
+
return {
|
|
1121
|
+
type: "function",
|
|
1122
|
+
function: { name: toolChoice.name }
|
|
1123
|
+
};
|
|
1124
|
+
}
|
|
1125
|
+
/**
|
|
1126
|
+
* Expands one `ConversationTurn` into one or more wire messages. Plain
|
|
1127
|
+
* user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
|
|
1128
|
+
* to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
|
|
1129
|
+
* turn expands into one wire `tool` message per `toolResult`, since
|
|
1130
|
+
* OpenAI-shaped wire format wants one message per tool_call_id.
|
|
1131
|
+
*/
|
|
1132
|
+
turnToWireMessages(turn) {
|
|
1133
|
+
if (turn.role === "tool") return (turn.toolResults ?? []).map((tr) => ({
|
|
1134
|
+
role: "tool",
|
|
1135
|
+
tool_call_id: tr.toolCallId,
|
|
1136
|
+
content: typeof tr.content === "string" ? tr.content : JSON.stringify(tr.content ?? null),
|
|
1137
|
+
...tr.isError ? { is_error: true } : {}
|
|
1138
|
+
}));
|
|
1139
|
+
if (turn.role === "assistant" && turn.toolCalls?.length) return [{
|
|
1140
|
+
role: "assistant",
|
|
1141
|
+
...turn.content ? { content: turn.content } : {},
|
|
1142
|
+
tool_calls: toWireToolCalls(turn.toolCalls)
|
|
1143
|
+
}];
|
|
1144
|
+
return [{
|
|
1145
|
+
role: turn.role,
|
|
1146
|
+
content: turn.content ?? ""
|
|
1147
|
+
}];
|
|
1148
|
+
}
|
|
565
1149
|
/**
|
|
566
1150
|
* Chooses the response format: a provider-native `jsonSchema` takes
|
|
567
1151
|
* priority when supplied (constrains generation directly), otherwise
|
|
@@ -580,21 +1164,49 @@ var VernLLM = class {
|
|
|
580
1164
|
};
|
|
581
1165
|
return useJson ? { type: "json_object" } : void 0;
|
|
582
1166
|
}
|
|
583
|
-
/**
|
|
584
|
-
|
|
585
|
-
|
|
1167
|
+
/**
|
|
1168
|
+
* Pulls `TokenUsage` out of a raw response, if the provider reported it.
|
|
1169
|
+
* Extraction doesn't depend on what happens to the response afterward, so
|
|
1170
|
+
* a malformed body can still yield usage if the provider's usage block
|
|
1171
|
+
* itself came through intact.
|
|
1172
|
+
*/
|
|
1173
|
+
extractUsage(response, requestId, model) {
|
|
1174
|
+
if (!response.usage) return void 0;
|
|
1175
|
+
return {
|
|
1176
|
+
promptTokens: response.usage.prompt_tokens ?? 0,
|
|
1177
|
+
completionTokens: response.usage.completion_tokens ?? 0,
|
|
1178
|
+
totalTokens: response.usage.total_tokens ?? 0,
|
|
1179
|
+
requestId,
|
|
1180
|
+
model
|
|
1181
|
+
};
|
|
1182
|
+
}
|
|
1183
|
+
/** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
|
|
1184
|
+
reportUsage(usage) {
|
|
1185
|
+
if (!usage || !this.onUsage) return;
|
|
586
1186
|
try {
|
|
587
|
-
this.onUsage(
|
|
588
|
-
promptTokens: response.usage.prompt_tokens ?? 0,
|
|
589
|
-
completionTokens: response.usage.completion_tokens ?? 0,
|
|
590
|
-
totalTokens: response.usage.total_tokens ?? 0,
|
|
591
|
-
requestId,
|
|
592
|
-
model
|
|
593
|
-
});
|
|
1187
|
+
this.onUsage(usage);
|
|
594
1188
|
} catch (error) {
|
|
595
1189
|
this.logger.error("[VernLLM] onUsage failed", { message: error instanceof Error ? error.message : "unknown" });
|
|
596
1190
|
}
|
|
597
1191
|
}
|
|
1192
|
+
/**
|
|
1193
|
+
* Reports token usage spent on an attempt that then failed, so it isn't
|
|
1194
|
+
* dropped alongside the error. Covers any error thrown after usage
|
|
1195
|
+
* extraction, since all of them happen only after a response (real
|
|
1196
|
+
* spend) already arrived. Swallows and logs any error `onUsageFailure`
|
|
1197
|
+
* itself throws.
|
|
1198
|
+
*/
|
|
1199
|
+
reportUsageFailure(usage, error, attempt, terminal = false) {
|
|
1200
|
+
const displayTokens = usage.totalTokens || usage.promptTokens + usage.completionTokens;
|
|
1201
|
+
const attemptText = terminal ? "mid-stream failure (terminal, no further attempts)" : `attempt ${attempt + 1}/${this.maxRetries + 1}`;
|
|
1202
|
+
this.logger.warn(`[VernLLM:${usage.requestId}] usage failure, ${attemptText}: type=${error.type} tokens=${displayTokens}`);
|
|
1203
|
+
if (!this.onUsageFailure) return;
|
|
1204
|
+
try {
|
|
1205
|
+
this.onUsageFailure(usage, error);
|
|
1206
|
+
} catch (hookError) {
|
|
1207
|
+
this.logger.error("[VernLLM] onUsageFailure failed", { message: hookError instanceof Error ? hookError.message : "unknown" });
|
|
1208
|
+
}
|
|
1209
|
+
}
|
|
598
1210
|
/** Parses response content as JSON and validates it against `schema` when supplied. */
|
|
599
1211
|
parseAndValidate(content, schema) {
|
|
600
1212
|
let parsed;
|
|
@@ -618,7 +1230,7 @@ var VernLLM = class {
|
|
|
618
1230
|
async recoverDelay(requestId, attempt, error, signal) {
|
|
619
1231
|
const retryAfterMs = extractRetryAfterMs(error);
|
|
620
1232
|
const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
|
|
621
|
-
this.logger.warn(`[
|
|
1233
|
+
this.logger.warn(`[VernLLM:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
|
|
622
1234
|
await waitForRetry(delay, signal);
|
|
623
1235
|
}
|
|
624
1236
|
/** Decides whether a failed attempt is worth retrying. */
|
|
@@ -633,7 +1245,7 @@ var VernLLM = class {
|
|
|
633
1245
|
* supports deletion. Cache invalidation is the caller's responsibility;
|
|
634
1246
|
* only the application knows when cached data is stale.
|
|
635
1247
|
*
|
|
636
|
-
* @param key
|
|
1248
|
+
* @param key The raw cache key (resolved through the adapter's
|
|
637
1249
|
* `resolveKey`, if any, before deletion).
|
|
638
1250
|
*/
|
|
639
1251
|
async deleteCache(key) {
|
|
@@ -641,15 +1253,20 @@ var VernLLM = class {
|
|
|
641
1253
|
await this.cache.delete(await this.resolveCacheKey(key));
|
|
642
1254
|
}
|
|
643
1255
|
/**
|
|
644
|
-
*
|
|
645
|
-
* same `cacheKey` share a single in-flight call, avoiding cache
|
|
1256
|
+
* Internal cache primitive around caller-supplied logic. Concurrent misses
|
|
1257
|
+
* for the same `cacheKey` share a single in-flight call, avoiding cache
|
|
1258
|
+
* stampedes.
|
|
646
1259
|
*
|
|
647
|
-
*
|
|
1260
|
+
* Not part of the public API. Backs the public `cachedCall()`, which
|
|
1261
|
+
* always composes this with `call()` so cached results get the same
|
|
1262
|
+
* retry/timeout/circuit-breaker guarantees as any other LLM call.
|
|
1263
|
+
*
|
|
1264
|
+
* @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
|
|
648
1265
|
* miss, typically `() => this.call(...)`), and optional
|
|
649
|
-
* `reserveUsage`/`refundUsage`/`signal`. See `
|
|
1266
|
+
* `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
|
|
650
1267
|
* @returns The cached value on a hit, or the result of `fn()` on a miss.
|
|
651
1268
|
*/
|
|
652
|
-
async
|
|
1269
|
+
async runCached(params) {
|
|
653
1270
|
const resolvedKey = await this.resolveCacheKey(params.cacheKey);
|
|
654
1271
|
const resolvedParams = resolvedKey === params.cacheKey ? params : {
|
|
655
1272
|
...params,
|
|
@@ -680,24 +1297,113 @@ var VernLLM = class {
|
|
|
680
1297
|
}
|
|
681
1298
|
return result;
|
|
682
1299
|
}
|
|
683
|
-
/**
|
|
684
|
-
|
|
685
|
-
|
|
1300
|
+
/**
|
|
1301
|
+
* Streaming counterpart to `runCached`. Three cases:
|
|
1302
|
+
*
|
|
1303
|
+
* - Hit: no live generation to relay. Returns immediately with
|
|
1304
|
+
* `finalResult` resolved to the cached value and a one-shot `chunks`
|
|
1305
|
+
* replay built from it, so `for await (const c of chunks)` call sites
|
|
1306
|
+
* work identically on a hit or a miss. No usage hooks fire, since
|
|
1307
|
+
* nothing was actually spent.
|
|
1308
|
+
* - Miss, nothing else in flight for this key: delegates to
|
|
1309
|
+
* `registerStreamTrigger`, which opens the stream and relays its
|
|
1310
|
+
* `chunks` live.
|
|
1311
|
+
* - Miss, but another call for the same key is already in flight: this
|
|
1312
|
+
* call has no live chunks of its own to relay, so it's treated like a
|
|
1313
|
+
* delayed hit. `finalResult` shares the trigger's in-flight promise
|
|
1314
|
+
* (the same `this.inFlight` map non-streaming `runCached` uses, so
|
|
1315
|
+
* streaming and non-streaming `cachedCall`s for the same key coalesce
|
|
1316
|
+
* against each other too), and `chunks` is a one-shot replay built
|
|
1317
|
+
* once that promise resolves.
|
|
1318
|
+
*/
|
|
1319
|
+
async runCachedStream(params, hasTools) {
|
|
1320
|
+
const resolvedKey = await this.resolveCacheKey(params.cacheKey);
|
|
1321
|
+
const resolvedParams = resolvedKey === params.cacheKey ? params : {
|
|
1322
|
+
...params,
|
|
1323
|
+
cacheKey: resolvedKey
|
|
1324
|
+
};
|
|
1325
|
+
const cached = await this.cache.get(resolvedKey);
|
|
1326
|
+
if (cached.hit) {
|
|
1327
|
+
const value = cached.value;
|
|
1328
|
+
return {
|
|
1329
|
+
chunks: buildReplayChunks(value, hasTools),
|
|
1330
|
+
finalResult: Promise.resolve(value)
|
|
1331
|
+
};
|
|
1332
|
+
}
|
|
1333
|
+
const existing = this.inFlight.get(resolvedKey);
|
|
1334
|
+
if (existing) {
|
|
1335
|
+
const finalResult = withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
1336
|
+
return {
|
|
1337
|
+
chunks: buildReplayChunksFromPromise(finalResult, hasTools),
|
|
1338
|
+
finalResult
|
|
1339
|
+
};
|
|
1340
|
+
}
|
|
1341
|
+
return this.registerStreamTrigger(resolvedParams);
|
|
686
1342
|
}
|
|
687
1343
|
/**
|
|
688
|
-
*
|
|
689
|
-
*
|
|
690
|
-
*
|
|
1344
|
+
* Opens the shared stream for a cache miss and tracks its settled value
|
|
1345
|
+
* in `this.inFlight` until it resolves or rejects. Writes to the cache
|
|
1346
|
+
* on success only, matching `runAndCache`.
|
|
691
1347
|
*
|
|
692
|
-
*
|
|
693
|
-
*
|
|
694
|
-
*
|
|
1348
|
+
* Registers the in-flight promise synchronously, before anything async
|
|
1349
|
+
* runs, so a concurrent `cachedCall` for the same key always sees it in
|
|
1350
|
+
* time to join instead of triggering its own stream. Settlement is
|
|
1351
|
+
* wired onto the whole `withReservedUsageForStream` call rather than a
|
|
1352
|
+
* line inside its callback, so any failure point (reserving usage,
|
|
1353
|
+
* opening the stream, or the stream itself) reliably settles the
|
|
1354
|
+
* in-flight entry instead of leaving it stuck.
|
|
695
1355
|
*/
|
|
696
|
-
|
|
1356
|
+
registerStreamTrigger(params) {
|
|
1357
|
+
let resolveInFlight;
|
|
1358
|
+
let rejectInFlight;
|
|
1359
|
+
const inFlightResult = new Promise((resolve, reject) => {
|
|
1360
|
+
resolveInFlight = resolve;
|
|
1361
|
+
rejectInFlight = reject;
|
|
1362
|
+
});
|
|
1363
|
+
this.inFlight.set(params.cacheKey, inFlightResult);
|
|
1364
|
+
inFlightResult.catch(() => {}).finally(() => {
|
|
1365
|
+
this.inFlight.delete(params.cacheKey);
|
|
1366
|
+
});
|
|
1367
|
+
const streamPromise = withReservedUsageForStream(params, async () => {
|
|
1368
|
+
const opened = await params.openStream();
|
|
1369
|
+
const trackedResult = opened.finalResult.then(async (value) => {
|
|
1370
|
+
try {
|
|
1371
|
+
await this.cache.set(params.cacheKey, value, params.ttl);
|
|
1372
|
+
} catch (error) {
|
|
1373
|
+
this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
|
|
1374
|
+
}
|
|
1375
|
+
return value;
|
|
1376
|
+
}, (error) => {
|
|
1377
|
+
throw error;
|
|
1378
|
+
});
|
|
1379
|
+
return {
|
|
1380
|
+
chunks: opened.chunks,
|
|
1381
|
+
finalResult: trackedResult
|
|
1382
|
+
};
|
|
1383
|
+
}, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
1384
|
+
streamPromise.then((opened) => {
|
|
1385
|
+
opened.finalResult.then(resolveInFlight, rejectInFlight);
|
|
1386
|
+
}, (error) => {
|
|
1387
|
+
rejectInFlight(error);
|
|
1388
|
+
});
|
|
1389
|
+
return streamPromise;
|
|
1390
|
+
}
|
|
1391
|
+
/** Logs a failed refundUsage attempt via the configured logger. */
|
|
1392
|
+
logRefundError(logMessage, error) {
|
|
1393
|
+
this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
|
|
1394
|
+
}
|
|
1395
|
+
async cachedCall(params) {
|
|
697
1396
|
const { call: callParams,...cacheParams } = params;
|
|
698
|
-
const { reserveUsage
|
|
699
|
-
if (
|
|
700
|
-
|
|
1397
|
+
const { reserveUsage, refundUsage,...restCallParams } = callParams;
|
|
1398
|
+
if (reserveUsage || refundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedCall; set them at the top level instead.");
|
|
1399
|
+
if (restCallParams.stream) {
|
|
1400
|
+
const streamParams = restCallParams;
|
|
1401
|
+
return this.runCachedStream({
|
|
1402
|
+
...cacheParams,
|
|
1403
|
+
openStream: () => this.call(streamParams)
|
|
1404
|
+
}, Boolean(restCallParams.tools));
|
|
1405
|
+
}
|
|
1406
|
+
return this.runCached({
|
|
701
1407
|
...cacheParams,
|
|
702
1408
|
fn: () => this.call(restCallParams)
|
|
703
1409
|
});
|
|
@@ -711,6 +1417,108 @@ var VernLLM = class {
|
|
|
711
1417
|
}
|
|
712
1418
|
};
|
|
713
1419
|
|
|
1420
|
+
//#endregion
|
|
1421
|
+
//#region src/internal/sse.ts
|
|
1422
|
+
/**
|
|
1423
|
+
* Parses a Server-Sent-Events byte/text stream into the JSON payload of
|
|
1424
|
+
* each `data:` frame, in arrival order. Generic over transport: works with
|
|
1425
|
+
* anything that hands back progressively-arriving `Uint8Array` or `string`
|
|
1426
|
+
* chunks via async iteration: native `fetch`'s `response.body` (wrapped
|
|
1427
|
+
* to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
|
|
1428
|
+
* Node `Readable` (already async-iterable, no wrapping needed), etc, so
|
|
1429
|
+
* this framing layer doesn't care which transport produced the bytes.
|
|
1430
|
+
*
|
|
1431
|
+
* Follows the SSE spec's frame-delimiting rules closely enough for LLM
|
|
1432
|
+
* streaming responses: frames are separated by a blank line, each frame
|
|
1433
|
+
* may carry one or more `data:` lines (joined with `\n` per spec when
|
|
1434
|
+
* there's more than one), `:`-prefixed lines are comments and ignored, and
|
|
1435
|
+
* other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
|
|
1436
|
+
* only needs the payload. A frame whose data is exactly `[DONE]` (the
|
|
1437
|
+
* sentinel several providers, notably OpenAI, send to mark stream end)
|
|
1438
|
+
* ends iteration without yielding it.
|
|
1439
|
+
*
|
|
1440
|
+
* Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
|
|
1441
|
+
* to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
|
|
1442
|
+
* alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
|
|
1443
|
+
* stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
|
|
1444
|
+
* lines.
|
|
1445
|
+
*
|
|
1446
|
+
* Malformed JSON in a frame throws `LLMError('parse')`, consistent with
|
|
1447
|
+
* how malformed JSON is handled elsewhere in VernLLM.
|
|
1448
|
+
*/
|
|
1449
|
+
async function* parseSseStream(source) {
|
|
1450
|
+
const decoder = new TextDecoder("utf-8", { fatal: true });
|
|
1451
|
+
let buffer = "";
|
|
1452
|
+
for await (const chunk of source) {
|
|
1453
|
+
let text;
|
|
1454
|
+
try {
|
|
1455
|
+
text = typeof chunk === "string" ? chunk : decoder.decode(chunk, { stream: true });
|
|
1456
|
+
} catch (cause) {
|
|
1457
|
+
throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
|
|
1458
|
+
}
|
|
1459
|
+
buffer = (buffer + text).replace(/\r\n/g, "\n").replace(/\r(?!$)/g, "\n");
|
|
1460
|
+
let boundary$1 = buffer.indexOf("\n\n");
|
|
1461
|
+
while (boundary$1 !== -1) {
|
|
1462
|
+
const frame = buffer.slice(0, boundary$1);
|
|
1463
|
+
buffer = buffer.slice(boundary$1 + 2);
|
|
1464
|
+
const event = parseSseFrame(frame);
|
|
1465
|
+
if (event === DONE) return;
|
|
1466
|
+
if (event !== NO_DATA) yield event;
|
|
1467
|
+
boundary$1 = buffer.indexOf("\n\n");
|
|
1468
|
+
}
|
|
1469
|
+
}
|
|
1470
|
+
try {
|
|
1471
|
+
buffer += decoder.decode();
|
|
1472
|
+
} catch (cause) {
|
|
1473
|
+
throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
|
|
1474
|
+
}
|
|
1475
|
+
buffer = buffer.replace(/\r$/, "\n");
|
|
1476
|
+
let boundary = buffer.indexOf("\n\n");
|
|
1477
|
+
while (boundary !== -1) {
|
|
1478
|
+
const frame = buffer.slice(0, boundary);
|
|
1479
|
+
buffer = buffer.slice(boundary + 2);
|
|
1480
|
+
const event = parseSseFrame(frame);
|
|
1481
|
+
if (event === DONE) return;
|
|
1482
|
+
if (event !== NO_DATA) yield event;
|
|
1483
|
+
boundary = buffer.indexOf("\n\n");
|
|
1484
|
+
}
|
|
1485
|
+
const trailing = buffer.trim();
|
|
1486
|
+
if (trailing) {
|
|
1487
|
+
const event = parseSseFrame(trailing);
|
|
1488
|
+
if (event !== DONE && event !== NO_DATA) yield event;
|
|
1489
|
+
}
|
|
1490
|
+
}
|
|
1491
|
+
const DONE = Symbol("sse-stream-done");
|
|
1492
|
+
const NO_DATA = Symbol("sse-frame-no-data");
|
|
1493
|
+
/**
|
|
1494
|
+
* Sentinel yielded by `parseSseStream` for a comment-only frame (no
|
|
1495
|
+
* `data:` payload), the mechanism providers use for SSE keep-alive
|
|
1496
|
+
* pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
|
|
1497
|
+
* alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
|
|
1498
|
+
*/
|
|
1499
|
+
const SSE_PING = Symbol("sse-frame-ping");
|
|
1500
|
+
/** Extracts and JSON-parses the `data:` payload of one SSE frame (the text between two blank lines). */
|
|
1501
|
+
function parseSseFrame(frame) {
|
|
1502
|
+
const dataLines = [];
|
|
1503
|
+
let sawComment = false;
|
|
1504
|
+
for (const line of frame.split("\n")) {
|
|
1505
|
+
if (line.startsWith(":")) {
|
|
1506
|
+
sawComment = true;
|
|
1507
|
+
continue;
|
|
1508
|
+
}
|
|
1509
|
+
if (!line.startsWith("data:")) continue;
|
|
1510
|
+
dataLines.push(line.startsWith("data: ") ? line.slice(6) : line.slice(5));
|
|
1511
|
+
}
|
|
1512
|
+
if (!dataLines.length) return sawComment ? SSE_PING : NO_DATA;
|
|
1513
|
+
const data = dataLines.join("\n");
|
|
1514
|
+
if (data === "[DONE]") return DONE;
|
|
1515
|
+
try {
|
|
1516
|
+
return JSON.parse(data);
|
|
1517
|
+
} catch (cause) {
|
|
1518
|
+
throw new LLMError(`Invalid JSON in SSE frame: ${data.slice(0, 200)}`, "parse", void 0, void 0, cause);
|
|
1519
|
+
}
|
|
1520
|
+
}
|
|
1521
|
+
|
|
714
1522
|
//#endregion
|
|
715
1523
|
//#region src/internal/imageFormat.ts
|
|
716
1524
|
/**
|
|
@@ -758,6 +1566,93 @@ function toAnthropicContent(blocks) {
|
|
|
758
1566
|
});
|
|
759
1567
|
}
|
|
760
1568
|
/**
|
|
1569
|
+
* Asserts a caller-supplied JSON Schema is an object schema before it's
|
|
1570
|
+
* used as Anthropic's `Tool.input_schema`, which (like every other
|
|
1571
|
+
* provider's function-calling API) requires `type: 'object'`. VernLLM's own
|
|
1572
|
+
* public `tools`/`jsonSchema` APIs accept freeform `Record<string,
|
|
1573
|
+
* unknown>` JSON Schema, so nothing upstream guarantees this at compile
|
|
1574
|
+
* time; this is the runtime check that stands in for that, so a schema
|
|
1575
|
+
* missing (or mistyping) `type: 'object'` fails loudly and immediately
|
|
1576
|
+
* instead of being silently forwarded to Anthropic malformed.
|
|
1577
|
+
*/
|
|
1578
|
+
function assertObjectSchema(schema, toolName) {
|
|
1579
|
+
if (schema.type !== "object") throw new LLMError(`Tool "${toolName}"'s schema must have "type": "object" (Anthropic requires object-shaped tool parameters).`, "validation");
|
|
1580
|
+
return schema;
|
|
1581
|
+
}
|
|
1582
|
+
/**
|
|
1583
|
+
* Translates VernLLM's OpenAI-shaped wire `tool_choice` into Anthropic's
|
|
1584
|
+
* `{ type: 'auto' | 'any' | 'none' | 'tool', name? }` shape. `'required'`
|
|
1585
|
+
* maps to `'any'` (Anthropic's "must call some tool" equivalent).
|
|
1586
|
+
*/
|
|
1587
|
+
function toAnthropicToolChoice(toolChoice) {
|
|
1588
|
+
if (!toolChoice || toolChoice === "auto") return { type: "auto" };
|
|
1589
|
+
if (toolChoice === "none") return { type: "none" };
|
|
1590
|
+
if (toolChoice === "required") return { type: "any" };
|
|
1591
|
+
return {
|
|
1592
|
+
type: "tool",
|
|
1593
|
+
name: toolChoice.function.name
|
|
1594
|
+
};
|
|
1595
|
+
}
|
|
1596
|
+
/**
|
|
1597
|
+
* Builds the Anthropic-shaped request body from VernLLM's wire params,
|
|
1598
|
+
* shared between `create` and `createStream` so both go through identical
|
|
1599
|
+
* translation (system prompt, message shaping, and the jsonSchema →
|
|
1600
|
+
* forced-single-tool mapping all happen exactly once, not once per entry
|
|
1601
|
+
* point).
|
|
1602
|
+
*
|
|
1603
|
+
* Returns `toolName` alongside the body: when set, the model was forced to
|
|
1604
|
+
* call a single synthetic tool standing in for `jsonSchema` output, and
|
|
1605
|
+
* both `create` and `createStream` need to know this so they can unwrap
|
|
1606
|
+
* that tool call back into plain text content instead of treating it like
|
|
1607
|
+
* a real tool call.
|
|
1608
|
+
*/
|
|
1609
|
+
function buildAnthropicRequestBody(params) {
|
|
1610
|
+
const systemMessage = params.messages.find((m) => m.role === "system");
|
|
1611
|
+
const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
|
|
1612
|
+
const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name.trim() : void 0;
|
|
1613
|
+
if (params.response_format?.type === "json_schema" && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
|
|
1614
|
+
let jsonInstruction;
|
|
1615
|
+
let tools;
|
|
1616
|
+
let toolChoice;
|
|
1617
|
+
if (params.response_format?.type === "json_schema" && toolName) {
|
|
1618
|
+
const { schema, description, strict } = params.response_format.json_schema;
|
|
1619
|
+
tools = [{
|
|
1620
|
+
name: toolName,
|
|
1621
|
+
description,
|
|
1622
|
+
input_schema: assertObjectSchema(schema, toolName),
|
|
1623
|
+
strict
|
|
1624
|
+
}];
|
|
1625
|
+
toolChoice = {
|
|
1626
|
+
type: "tool",
|
|
1627
|
+
name: toolName
|
|
1628
|
+
};
|
|
1629
|
+
} else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
|
|
1630
|
+
else if (params.tools?.length) {
|
|
1631
|
+
tools = params.tools.map((t) => ({
|
|
1632
|
+
name: t.function.name,
|
|
1633
|
+
description: t.function.description,
|
|
1634
|
+
input_schema: assertObjectSchema(t.function.parameters, t.function.name)
|
|
1635
|
+
}));
|
|
1636
|
+
toolChoice = toAnthropicToolChoice(params.tool_choice);
|
|
1637
|
+
}
|
|
1638
|
+
const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
|
|
1639
|
+
const body = {
|
|
1640
|
+
model: params.model,
|
|
1641
|
+
max_tokens: params.max_tokens,
|
|
1642
|
+
...params.temperature !== void 0 ? { temperature: params.temperature } : {},
|
|
1643
|
+
system: system || void 0,
|
|
1644
|
+
messages: mergeConsecutiveToolResults$1(conversationMessages.map((m) => toAnthropicMessage(m))),
|
|
1645
|
+
...tools ? {
|
|
1646
|
+
tools,
|
|
1647
|
+
tool_choice: toolChoice
|
|
1648
|
+
} : {}
|
|
1649
|
+
};
|
|
1650
|
+
return {
|
|
1651
|
+
body,
|
|
1652
|
+
toolName
|
|
1653
|
+
};
|
|
1654
|
+
}
|
|
1655
|
+
/**
|
|
761
1656
|
* Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
|
|
762
1657
|
* interface VernLLM uses for OpenAI/Groq.
|
|
763
1658
|
*
|
|
@@ -772,53 +1667,161 @@ function toAnthropicContent(blocks) {
|
|
|
772
1667
|
* generation against.
|
|
773
1668
|
*/
|
|
774
1669
|
function fromAnthropic(anthropicClient) {
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
1670
|
+
const rawMessagesCreate = anthropicClient.messages.create.bind(anthropicClient.messages);
|
|
1671
|
+
return { chat: { completions: {
|
|
1672
|
+
async create(params, options) {
|
|
1673
|
+
const { body, toolName } = buildAnthropicRequestBody(params);
|
|
1674
|
+
const response = await anthropicClient.messages.create(body, options);
|
|
1675
|
+
let text;
|
|
1676
|
+
let wireToolCalls;
|
|
1677
|
+
if (toolName) {
|
|
1678
|
+
const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
|
|
1679
|
+
if (!toolUse) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
|
|
1680
|
+
if (!toolUse.input || typeof toolUse.input !== "object" || Array.isArray(toolUse.input)) throw new LLMError(`Anthropic returned invalid structured output for tool "${toolName}". Expected an object.`, "validation");
|
|
1681
|
+
text = JSON.stringify(toolUse.input);
|
|
1682
|
+
} else {
|
|
1683
|
+
text = response.content.filter((block) => block.type === "text").map((block) => block.text ?? "").join("");
|
|
1684
|
+
const toolUses = response.content.filter((block) => block.type === "tool_use");
|
|
1685
|
+
if (toolUses.length) wireToolCalls = toolUses.map((block) => ({
|
|
1686
|
+
id: block.id,
|
|
1687
|
+
type: "function",
|
|
1688
|
+
function: {
|
|
1689
|
+
name: block.name,
|
|
1690
|
+
arguments: JSON.stringify(block.input ?? {})
|
|
1691
|
+
}
|
|
1692
|
+
}));
|
|
1693
|
+
}
|
|
1694
|
+
return {
|
|
1695
|
+
choices: [{ message: {
|
|
1696
|
+
content: text,
|
|
1697
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
1698
|
+
} }],
|
|
1699
|
+
usage: {
|
|
1700
|
+
prompt_tokens: response.usage?.input_tokens,
|
|
1701
|
+
completion_tokens: response.usage?.output_tokens,
|
|
1702
|
+
total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
|
|
805
1703
|
}
|
|
806
|
-
}
|
|
807
|
-
},
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
const
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
1704
|
+
};
|
|
1705
|
+
},
|
|
1706
|
+
async *createStream(params, options) {
|
|
1707
|
+
const { body, toolName } = buildAnthropicRequestBody(params);
|
|
1708
|
+
const stream = await rawMessagesCreate({
|
|
1709
|
+
...body,
|
|
1710
|
+
stream: true
|
|
1711
|
+
}, options);
|
|
1712
|
+
const blockKinds = new Map();
|
|
1713
|
+
let inputTokens = 0;
|
|
1714
|
+
let sawJsonTool = false;
|
|
1715
|
+
for await (const event of stream) if (event.type === "message_start") inputTokens = event.message.usage?.input_tokens ?? 0;
|
|
1716
|
+
else if (event.type === "content_block_start") if (event.content_block.type === "tool_use") {
|
|
1717
|
+
const kind = event.content_block.name === toolName ? "json-tool" : "tool_use";
|
|
1718
|
+
blockKinds.set(event.index, kind);
|
|
1719
|
+
if (kind === "json-tool") sawJsonTool = true;
|
|
1720
|
+
else if (!toolName) yield {
|
|
1721
|
+
type: "tool_call_delta",
|
|
1722
|
+
index: event.index,
|
|
1723
|
+
id: event.content_block.id,
|
|
1724
|
+
name: event.content_block.name
|
|
1725
|
+
};
|
|
1726
|
+
} else blockKinds.set(event.index, "text");
|
|
1727
|
+
else if (event.type === "content_block_delta") {
|
|
1728
|
+
if (event.delta.type === "text_delta") {
|
|
1729
|
+
if (!toolName) yield {
|
|
1730
|
+
type: "text-delta",
|
|
1731
|
+
delta: event.delta.text
|
|
1732
|
+
};
|
|
1733
|
+
} else if (event.delta.type === "input_json_delta") {
|
|
1734
|
+
const kind = blockKinds.get(event.index);
|
|
1735
|
+
if (kind === "json-tool") yield {
|
|
1736
|
+
type: "text-delta",
|
|
1737
|
+
delta: event.delta.partial_json
|
|
1738
|
+
};
|
|
1739
|
+
else if (!toolName) yield {
|
|
1740
|
+
type: "tool_call_delta",
|
|
1741
|
+
index: event.index,
|
|
1742
|
+
argumentsDelta: event.delta.partial_json
|
|
1743
|
+
};
|
|
1744
|
+
}
|
|
1745
|
+
} else if (event.type === "message_delta") {
|
|
1746
|
+
const outputTokens = event.usage?.output_tokens ?? 0;
|
|
1747
|
+
yield {
|
|
1748
|
+
type: "usage",
|
|
1749
|
+
usage: {
|
|
1750
|
+
prompt_tokens: inputTokens,
|
|
1751
|
+
completion_tokens: outputTokens,
|
|
1752
|
+
total_tokens: inputTokens + outputTokens
|
|
1753
|
+
}
|
|
1754
|
+
};
|
|
1755
|
+
} else if (event.type === "ping") yield { type: "ping" };
|
|
1756
|
+
if (toolName && !sawJsonTool) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
|
|
1757
|
+
}
|
|
1758
|
+
} } };
|
|
1759
|
+
}
|
|
1760
|
+
/**
|
|
1761
|
+
* Anthropic requires strict role alternation, so the per-wire-message
|
|
1762
|
+
* mapping above (one `{role:'user', content:[tool_result]}` per VernLLM
|
|
1763
|
+
* wire tool message) needs merging back together when an assistant turn
|
|
1764
|
+
* requested more than one tool: multiple consecutive user turns would
|
|
1765
|
+
* violate that alternation, and Anthropic's API rejects it outright. This
|
|
1766
|
+
* merges any run of tool-result-only user messages into one, with all
|
|
1767
|
+
* their tool_result blocks combined, the shape Anthropic expects for "here
|
|
1768
|
+
* are the results of everything you just asked for."
|
|
1769
|
+
*/
|
|
1770
|
+
function mergeConsecutiveToolResults$1(messages) {
|
|
1771
|
+
const isToolResultOnly = (m) => m.role === "user" && Array.isArray(m.content) && m.content.length > 0 && m.content.every((b) => b.type === "tool_result");
|
|
1772
|
+
const merged = [];
|
|
1773
|
+
for (const m of messages) {
|
|
1774
|
+
const prev = merged.at(-1);
|
|
1775
|
+
if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
|
|
1776
|
+
else merged.push(m);
|
|
1777
|
+
}
|
|
1778
|
+
return merged;
|
|
1779
|
+
}
|
|
1780
|
+
/**
|
|
1781
|
+
* Translates one VernLLM wire message (OpenAI-shaped: plain user/assistant
|
|
1782
|
+
* turns, an assistant turn with `tool_calls`, or a `tool` turn) into
|
|
1783
|
+
* Anthropic's `{ role: 'user' | 'assistant', content }` shape.
|
|
1784
|
+
*/
|
|
1785
|
+
function toAnthropicMessage(m) {
|
|
1786
|
+
if (m.role === "tool") return {
|
|
1787
|
+
role: "user",
|
|
1788
|
+
content: [{
|
|
1789
|
+
type: "tool_result",
|
|
1790
|
+
tool_use_id: m.tool_call_id,
|
|
1791
|
+
content: m.content,
|
|
1792
|
+
...m.is_error ? { is_error: true } : {}
|
|
1793
|
+
}]
|
|
1794
|
+
};
|
|
1795
|
+
if (m.role === "assistant" && m.tool_calls?.length) {
|
|
1796
|
+
const blocks = [];
|
|
1797
|
+
if (m.content) blocks.push({
|
|
1798
|
+
type: "text",
|
|
1799
|
+
text: m.content
|
|
1800
|
+
});
|
|
1801
|
+
for (const tc of m.tool_calls) {
|
|
1802
|
+
let input;
|
|
1803
|
+
try {
|
|
1804
|
+
input = tc.function.arguments.trim() ? JSON.parse(tc.function.arguments) : {};
|
|
1805
|
+
} catch (cause) {
|
|
1806
|
+
throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
|
|
819
1807
|
}
|
|
1808
|
+
if (input === null || Array.isArray(input) || typeof input !== "object") throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) arguments must be a JSON object.`, "validation");
|
|
1809
|
+
blocks.push({
|
|
1810
|
+
type: "tool_use",
|
|
1811
|
+
id: tc.id,
|
|
1812
|
+
name: tc.function.name,
|
|
1813
|
+
input
|
|
1814
|
+
});
|
|
1815
|
+
}
|
|
1816
|
+
return {
|
|
1817
|
+
role: "assistant",
|
|
1818
|
+
content: blocks
|
|
820
1819
|
};
|
|
821
|
-
}
|
|
1820
|
+
}
|
|
1821
|
+
return {
|
|
1822
|
+
role: m.role,
|
|
1823
|
+
content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content ?? ""
|
|
1824
|
+
};
|
|
822
1825
|
}
|
|
823
1826
|
|
|
824
1827
|
//#endregion
|
|
@@ -835,6 +1838,122 @@ function toGeminiParts(blocks) {
|
|
|
835
1838
|
data: block.data
|
|
836
1839
|
} } : { text: block.text });
|
|
837
1840
|
}
|
|
1841
|
+
/** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Gemini's `functionCallingConfig`. */
|
|
1842
|
+
function toGeminiToolConfig(toolChoice) {
|
|
1843
|
+
if (!toolChoice || toolChoice === "auto") return { functionCallingConfig: { mode: "AUTO" } };
|
|
1844
|
+
if (toolChoice === "none") return { functionCallingConfig: { mode: "NONE" } };
|
|
1845
|
+
if (toolChoice === "required") return { functionCallingConfig: { mode: "ANY" } };
|
|
1846
|
+
return { functionCallingConfig: {
|
|
1847
|
+
mode: "ANY",
|
|
1848
|
+
allowedFunctionNames: [toolChoice.function.name]
|
|
1849
|
+
} };
|
|
1850
|
+
}
|
|
1851
|
+
/**
|
|
1852
|
+
* Translates one VernLLM wire message into a Gemini `contents` entry.
|
|
1853
|
+
* Gemini has no separate 'tool' role: a prior assistant tool request
|
|
1854
|
+
* becomes a `'model'` turn with `functionCall` parts, and its result
|
|
1855
|
+
* becomes a `'user'` turn with `functionResponse` parts.
|
|
1856
|
+
*/
|
|
1857
|
+
function toGeminiContent(m) {
|
|
1858
|
+
if (m.role === "tool") return {
|
|
1859
|
+
role: "user",
|
|
1860
|
+
parts: [{ functionResponse: {
|
|
1861
|
+
name: m.tool_call_id,
|
|
1862
|
+
response: parseToolResult(m.content)
|
|
1863
|
+
} }]
|
|
1864
|
+
};
|
|
1865
|
+
if (m.role === "assistant" && m.tool_calls?.length) {
|
|
1866
|
+
const parts = [];
|
|
1867
|
+
if (typeof m.content === "string" && m.content) parts.push({ text: m.content });
|
|
1868
|
+
parts.push(...m.tool_calls.map((tc) => ({ functionCall: {
|
|
1869
|
+
name: tc.function.name,
|
|
1870
|
+
args: parseToolArguments(tc.function.arguments, tc.function.name)
|
|
1871
|
+
} })));
|
|
1872
|
+
return {
|
|
1873
|
+
role: "model",
|
|
1874
|
+
parts
|
|
1875
|
+
};
|
|
1876
|
+
}
|
|
1877
|
+
return {
|
|
1878
|
+
role: m.role === "assistant" ? "model" : "user",
|
|
1879
|
+
parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content ?? "" }]
|
|
1880
|
+
};
|
|
1881
|
+
}
|
|
1882
|
+
function parseToolArguments(text, toolName) {
|
|
1883
|
+
let parsed;
|
|
1884
|
+
try {
|
|
1885
|
+
parsed = text.trim() ? JSON.parse(text) : {};
|
|
1886
|
+
} catch (cause) {
|
|
1887
|
+
throw new LLMError(`Tool call "${toolName}" arguments are not valid JSON.`, "validation", void 0, void 0, cause);
|
|
1888
|
+
}
|
|
1889
|
+
if (!parsed || Array.isArray(parsed) || typeof parsed !== "object") throw new LLMError(`Tool call "${toolName}" arguments must be a JSON object.`, "validation");
|
|
1890
|
+
return parsed;
|
|
1891
|
+
}
|
|
1892
|
+
function parseToolResult(text) {
|
|
1893
|
+
try {
|
|
1894
|
+
return text.trim() ? JSON.parse(text) : "";
|
|
1895
|
+
} catch {
|
|
1896
|
+
return text;
|
|
1897
|
+
}
|
|
1898
|
+
}
|
|
1899
|
+
/**
|
|
1900
|
+
* Gemini expects the results of everything the model asked for in one turn
|
|
1901
|
+
* to arrive together as multiple `functionResponse` parts on a single
|
|
1902
|
+
* `'user'` entry, not as separate consecutive `'user'` entries. The
|
|
1903
|
+
* per-wire-message mapping above produces one `'user'` entry per VernLLM
|
|
1904
|
+
* wire tool message, so when an assistant turn requested more than one
|
|
1905
|
+
* tool, this merges the resulting run of functionResponse-only `'user'`
|
|
1906
|
+
* entries back into one.
|
|
1907
|
+
*/
|
|
1908
|
+
function mergeConsecutiveFunctionResponses(contents) {
|
|
1909
|
+
const isFunctionResponseOnly = (c) => c.role === "user" && c.parts.length > 0 && c.parts.every((p) => "functionResponse" in p);
|
|
1910
|
+
const merged = [];
|
|
1911
|
+
for (const c of contents) {
|
|
1912
|
+
const prev = merged.at(-1);
|
|
1913
|
+
if (isFunctionResponseOnly(c) && prev && isFunctionResponseOnly(prev)) prev.parts.push(...c.parts);
|
|
1914
|
+
else merged.push(c);
|
|
1915
|
+
}
|
|
1916
|
+
return merged;
|
|
1917
|
+
}
|
|
1918
|
+
/**
|
|
1919
|
+
* Builds the Gemini-shaped request from VernLLM's wire params, shared
|
|
1920
|
+
* between `create` and `createStream` so both go through identical
|
|
1921
|
+
* translation (contents shaping, `responseSchema`/`responseMimeType`
|
|
1922
|
+
* mapping, and tool/toolConfig translation all happen exactly once).
|
|
1923
|
+
* `abortSignal` is folded into `config` by the caller (`create`/
|
|
1924
|
+
* `createStream`), once the request options are available.
|
|
1925
|
+
*/
|
|
1926
|
+
function buildGeminiRequest(params) {
|
|
1927
|
+
const systemMessage = params.messages.find((m) => m.role === "system");
|
|
1928
|
+
const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
|
|
1929
|
+
const wantsJson = Boolean(params.response_format);
|
|
1930
|
+
const config = {
|
|
1931
|
+
...params.temperature !== void 0 ? { temperature: params.temperature } : {},
|
|
1932
|
+
maxOutputTokens: params.max_tokens,
|
|
1933
|
+
...systemMessage ? { systemInstruction: { parts: [{ text: systemMessage.content }] } } : {}
|
|
1934
|
+
};
|
|
1935
|
+
if (wantsJson) config.responseMimeType = "application/json";
|
|
1936
|
+
if (params.response_format?.type === "json_schema") {
|
|
1937
|
+
const { schema, description } = params.response_format.json_schema;
|
|
1938
|
+
config.responseSchema = {
|
|
1939
|
+
...schema,
|
|
1940
|
+
...description ? { description } : {}
|
|
1941
|
+
};
|
|
1942
|
+
}
|
|
1943
|
+
if (params.tools?.length) {
|
|
1944
|
+
config.tools = [{ functionDeclarations: params.tools.map((t) => ({
|
|
1945
|
+
name: t.function.name,
|
|
1946
|
+
description: t.function.description,
|
|
1947
|
+
parameters: t.function.parameters
|
|
1948
|
+
})) }];
|
|
1949
|
+
config.toolConfig = toGeminiToolConfig(params.tool_choice);
|
|
1950
|
+
}
|
|
1951
|
+
return {
|
|
1952
|
+
model: params.model,
|
|
1953
|
+
contents: mergeConsecutiveFunctionResponses(conversationMessages.map((m) => toGeminiContent(m))),
|
|
1954
|
+
config
|
|
1955
|
+
};
|
|
1956
|
+
}
|
|
838
1957
|
/**
|
|
839
1958
|
* Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
|
|
840
1959
|
* uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
|
|
@@ -845,43 +1964,100 @@ function toGeminiParts(blocks) {
|
|
|
845
1964
|
* `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
|
|
846
1965
|
* models use a token budget, not an effort tier, so it's dropped, same as
|
|
847
1966
|
* Anthropic.
|
|
1967
|
+
*
|
|
1968
|
+
* `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
|
|
1969
|
+
* `tool_choice` maps to `toolConfig.functionCallingConfig`. `jsonSchema`
|
|
1970
|
+
* and `tools` are mutually exclusive by the time a call reaches here
|
|
1971
|
+
* (enforced in vernLLM.ts), so `responseSchema` and `tools` never
|
|
1972
|
+
* both apply.
|
|
1973
|
+
*
|
|
1974
|
+
* `createStream` calls `generateContentStream` (optional on `GeminiClient`
|
|
1975
|
+
*, required only if the caller sets `stream: true`) and translates each
|
|
1976
|
+
* partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
|
|
1977
|
+
* Gemini's own function-calling API doesn't stream tool-call arguments
|
|
1978
|
+
* incrementally: a `functionCall` part always arrives whole in one chunk,
|
|
1979
|
+
* so each one is emitted as a single, complete `tool_call_delta` (a
|
|
1980
|
+
* one-shot "delta" containing the full arguments) rather than accumulated
|
|
1981
|
+
* fragments, that's a real difference in the underlying API, not
|
|
1982
|
+
* something this adapter can smooth over. `usageMetadata` is (per Gemini's
|
|
1983
|
+
* own behavior) only reliably present on the last chunk, so the `usage`
|
|
1984
|
+
* `WireStreamChunk` is emitted once, after the stream completes, from
|
|
1985
|
+
* whichever chunk's `usageMetadata` was seen last.
|
|
848
1986
|
*/
|
|
849
1987
|
function fromGemini(geminiClient) {
|
|
850
|
-
return { chat: { completions: {
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
maxOutputTokens: params.max_tokens
|
|
857
|
-
};
|
|
858
|
-
if (wantsJson) generationConfig.responseMimeType = "application/json";
|
|
859
|
-
if (params.response_format?.type === "json_schema") {
|
|
860
|
-
const { schema, description } = params.response_format.json_schema;
|
|
861
|
-
generationConfig.responseSchema = {
|
|
862
|
-
...schema,
|
|
863
|
-
...description ? { description } : {}
|
|
1988
|
+
return { chat: { completions: {
|
|
1989
|
+
async create(params, options) {
|
|
1990
|
+
const request = buildGeminiRequest(params);
|
|
1991
|
+
request.config = {
|
|
1992
|
+
...request.config,
|
|
1993
|
+
abortSignal: options.signal
|
|
864
1994
|
};
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
1995
|
+
const response = await geminiClient.generateContent(request);
|
|
1996
|
+
const parts = response.candidates?.[0]?.content?.parts ?? [];
|
|
1997
|
+
const text = parts.map((p) => p.text ?? "").join("");
|
|
1998
|
+
const functionCalls = parts.filter((p) => p.functionCall);
|
|
1999
|
+
let wireToolCalls;
|
|
2000
|
+
if (functionCalls.length) wireToolCalls = functionCalls.map((p) => ({
|
|
2001
|
+
id: p.functionCall.name,
|
|
2002
|
+
type: "function",
|
|
2003
|
+
function: {
|
|
2004
|
+
name: p.functionCall.name,
|
|
2005
|
+
arguments: JSON.stringify(p.functionCall.args ?? {})
|
|
2006
|
+
}
|
|
2007
|
+
}));
|
|
2008
|
+
return {
|
|
2009
|
+
choices: [{ message: {
|
|
2010
|
+
content: text,
|
|
2011
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
2012
|
+
} }],
|
|
2013
|
+
usage: {
|
|
2014
|
+
prompt_tokens: response.usageMetadata?.promptTokenCount,
|
|
2015
|
+
completion_tokens: response.usageMetadata?.candidatesTokenCount,
|
|
2016
|
+
total_tokens: response.usageMetadata?.totalTokenCount
|
|
2017
|
+
}
|
|
2018
|
+
};
|
|
2019
|
+
},
|
|
2020
|
+
async *createStream(params, options) {
|
|
2021
|
+
if (!geminiClient.generateContentStream) throw new LLMError("stream: true requires a Gemini client with generateContentStream", "validation");
|
|
2022
|
+
const request = buildGeminiRequest(params);
|
|
2023
|
+
request.config = {
|
|
2024
|
+
...request.config,
|
|
2025
|
+
abortSignal: options.signal
|
|
2026
|
+
};
|
|
2027
|
+
const stream = await geminiClient.generateContentStream(request);
|
|
2028
|
+
let toolCallIndex = 0;
|
|
2029
|
+
let lastUsage;
|
|
2030
|
+
for await (const chunk of stream) {
|
|
2031
|
+
const parts = chunk.candidates?.[0]?.content?.parts ?? [];
|
|
2032
|
+
for (const part of parts) {
|
|
2033
|
+
if (part.text) yield {
|
|
2034
|
+
type: "text-delta",
|
|
2035
|
+
delta: part.text
|
|
2036
|
+
};
|
|
2037
|
+
if (part.functionCall) {
|
|
2038
|
+
yield {
|
|
2039
|
+
type: "tool_call_delta",
|
|
2040
|
+
index: toolCallIndex,
|
|
2041
|
+
id: part.functionCall.name,
|
|
2042
|
+
name: part.functionCall.name,
|
|
2043
|
+
argumentsDelta: JSON.stringify(part.functionCall.args ?? {}),
|
|
2044
|
+
complete: true
|
|
2045
|
+
};
|
|
2046
|
+
toolCallIndex++;
|
|
2047
|
+
}
|
|
2048
|
+
}
|
|
2049
|
+
if (chunk.usageMetadata) lastUsage = chunk.usageMetadata;
|
|
882
2050
|
}
|
|
883
|
-
|
|
884
|
-
|
|
2051
|
+
if (lastUsage) yield {
|
|
2052
|
+
type: "usage",
|
|
2053
|
+
usage: {
|
|
2054
|
+
prompt_tokens: lastUsage.promptTokenCount,
|
|
2055
|
+
completion_tokens: lastUsage.candidatesTokenCount,
|
|
2056
|
+
total_tokens: lastUsage.totalTokenCount
|
|
2057
|
+
}
|
|
2058
|
+
};
|
|
2059
|
+
}
|
|
2060
|
+
} } };
|
|
885
2061
|
}
|
|
886
2062
|
|
|
887
2063
|
//#endregion
|
|
@@ -917,6 +2093,67 @@ function toBedrockContent(blocks) {
|
|
|
917
2093
|
} } : { text: block.text });
|
|
918
2094
|
}
|
|
919
2095
|
/**
|
|
2096
|
+
* Builds the Converse-shaped request from VernLLM's wire params, shared
|
|
2097
|
+
* between `create` and `createStream` so both go through identical
|
|
2098
|
+
* translation (system prompt, message shaping, the jsonSchema →
|
|
2099
|
+
* forced-single-tool mapping, and the `toolUseSupportedModels` preflight
|
|
2100
|
+
* check all happen exactly once).
|
|
2101
|
+
*
|
|
2102
|
+
* Returns `toolName` alongside the request: when set, the model was forced
|
|
2103
|
+
* to call a single synthetic tool standing in for `jsonSchema` output, and
|
|
2104
|
+
* both `create` and `createStream` need to know this so they can unwrap
|
|
2105
|
+
* that tool call back into plain text content instead of treating it like
|
|
2106
|
+
* a real tool call.
|
|
2107
|
+
*/
|
|
2108
|
+
function buildBedrockRequest(params, toolUseSupportedModels) {
|
|
2109
|
+
const systemMessage = params.messages.find((m) => m.role === "system");
|
|
2110
|
+
const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
|
|
2111
|
+
const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
|
|
2112
|
+
const toolName = jsonSchema?.name.trim();
|
|
2113
|
+
if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
|
|
2114
|
+
let jsonInstruction;
|
|
2115
|
+
let toolConfig;
|
|
2116
|
+
if (jsonSchema) {
|
|
2117
|
+
const { schema, description, strict } = jsonSchema;
|
|
2118
|
+
toolConfig = {
|
|
2119
|
+
tools: [{ toolSpec: {
|
|
2120
|
+
name: toolName,
|
|
2121
|
+
description,
|
|
2122
|
+
inputSchema: { json: schema },
|
|
2123
|
+
strict
|
|
2124
|
+
} }],
|
|
2125
|
+
toolChoice: { tool: { name: toolName } }
|
|
2126
|
+
};
|
|
2127
|
+
} else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
|
|
2128
|
+
else if (params.tools?.length) toolConfig = {
|
|
2129
|
+
tools: params.tools.map((t) => ({ toolSpec: {
|
|
2130
|
+
name: t.function.name,
|
|
2131
|
+
description: t.function.description,
|
|
2132
|
+
inputSchema: { json: t.function.parameters }
|
|
2133
|
+
} })),
|
|
2134
|
+
toolChoice: toBedrockToolChoice(params.tool_choice)
|
|
2135
|
+
};
|
|
2136
|
+
if (jsonSchema && toolUseSupportedModels) {
|
|
2137
|
+
const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
|
|
2138
|
+
if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
|
|
2139
|
+
}
|
|
2140
|
+
const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
|
|
2141
|
+
const request = {
|
|
2142
|
+
modelId: params.model,
|
|
2143
|
+
messages: mergeConsecutiveToolResults(conversationMessages.map((m) => toBedrockMessage(m))),
|
|
2144
|
+
system: systemParts.length ? systemParts.map((text) => ({ text })) : void 0,
|
|
2145
|
+
inferenceConfig: {
|
|
2146
|
+
...params.temperature !== void 0 ? { temperature: params.temperature } : {},
|
|
2147
|
+
maxTokens: params.max_tokens
|
|
2148
|
+
},
|
|
2149
|
+
...toolConfig ? { toolConfig } : {}
|
|
2150
|
+
};
|
|
2151
|
+
return {
|
|
2152
|
+
request,
|
|
2153
|
+
toolName
|
|
2154
|
+
};
|
|
2155
|
+
}
|
|
2156
|
+
/**
|
|
920
2157
|
* Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
|
|
921
2158
|
* interface VernLLM uses for OpenAI/Groq. The Converse API is unified
|
|
922
2159
|
* across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
|
|
@@ -936,66 +2173,237 @@ function toBedrockContent(blocks) {
|
|
|
936
2173
|
* `response_format: json_object` (no schema to build a tool from) and
|
|
937
2174
|
* `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
|
|
938
2175
|
* instruction and are dropped respectively.
|
|
2176
|
+
*
|
|
2177
|
+
* `tools` maps to Converse's native `toolConfig`/`toolUse`/`toolResult`;
|
|
2178
|
+
* `tool_choice` maps to `toolConfig.toolChoice`. Mutually exclusive with
|
|
2179
|
+
* `jsonSchema` by the time a call reaches here (enforced in vernLLM.ts).
|
|
2180
|
+
*
|
|
2181
|
+
* `createStream` calls `converseStream` (optional on `BedrockConverseClient`
|
|
2182
|
+
*, required only if the caller sets `stream: true`) and translates its
|
|
2183
|
+
* `contentBlockStart`/`contentBlockDelta`/`metadata` events into
|
|
2184
|
+
* `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
|
|
2185
|
+
* same as `fromAnthropic`'s block-index tracking (Converse's streaming
|
|
2186
|
+
* shape is structurally close to Anthropic's own, both being tool-use-aware
|
|
2187
|
+
* content-block streams), including the same `json-tool` unwrapping: a
|
|
2188
|
+
* `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
|
|
2189
|
+
* `text-delta`, not `tool_call_delta`, so the accumulated result lands in
|
|
2190
|
+
* `finalizeResponse`'s `content` path exactly like the non-streaming
|
|
2191
|
+
* `create` branch above unwraps it.
|
|
939
2192
|
*/
|
|
940
2193
|
function fromBedrock(bedrockClient, options) {
|
|
941
2194
|
const toolUseSupportedModels = options?.toolUseSupportedModels;
|
|
942
|
-
return { chat: { completions: {
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
2195
|
+
return { chat: { completions: {
|
|
2196
|
+
async create(params, requestOptions) {
|
|
2197
|
+
const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
|
|
2198
|
+
const response = await bedrockClient.converse(request, requestOptions);
|
|
2199
|
+
let text;
|
|
2200
|
+
let wireToolCalls;
|
|
2201
|
+
if (toolName) {
|
|
2202
|
+
const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
|
|
2203
|
+
text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
|
|
2204
|
+
} else {
|
|
2205
|
+
const blocks = response.output?.message?.content ?? [];
|
|
2206
|
+
text = blocks.map((c) => c.text ?? "").join("");
|
|
2207
|
+
const toolUses = blocks.filter((block) => Boolean(block.toolUse));
|
|
2208
|
+
if (toolUses.length) wireToolCalls = toolUses.map((block, i) => {
|
|
2209
|
+
const toolUse = block.toolUse;
|
|
2210
|
+
if (!toolUse.name) throw new LLMError(`Bedrock returned a toolUse block without a name at index ${i}.`, "validation");
|
|
2211
|
+
return {
|
|
2212
|
+
id: toolUse.toolUseId ?? `${toolUse.name}_${i}`,
|
|
2213
|
+
type: "function",
|
|
2214
|
+
function: {
|
|
2215
|
+
name: toolUse.name,
|
|
2216
|
+
arguments: JSON.stringify(toolUse.input ?? {})
|
|
2217
|
+
}
|
|
2218
|
+
};
|
|
2219
|
+
});
|
|
2220
|
+
}
|
|
2221
|
+
return {
|
|
2222
|
+
choices: [{ message: {
|
|
2223
|
+
content: text,
|
|
2224
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
958
2225
|
} }],
|
|
959
|
-
|
|
2226
|
+
usage: {
|
|
2227
|
+
prompt_tokens: response.usage?.inputTokens,
|
|
2228
|
+
completion_tokens: response.usage?.outputTokens,
|
|
2229
|
+
total_tokens: response.usage?.totalTokens
|
|
2230
|
+
}
|
|
2231
|
+
};
|
|
2232
|
+
},
|
|
2233
|
+
async *createStream(params, requestOptions) {
|
|
2234
|
+
if (!bedrockClient.converseStream) throw new LLMError("stream: true requires a Bedrock client with converseStream", "validation");
|
|
2235
|
+
const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
|
|
2236
|
+
const { stream } = await bedrockClient.converseStream(request, requestOptions);
|
|
2237
|
+
const blockKinds = new Map();
|
|
2238
|
+
for await (const event of stream) if ("contentBlockStart" in event) {
|
|
2239
|
+
const { contentBlockIndex, start } = event.contentBlockStart;
|
|
2240
|
+
if (start?.toolUse) {
|
|
2241
|
+
const kind = start.toolUse.name === toolName ? "json-tool" : "tool_use";
|
|
2242
|
+
blockKinds.set(contentBlockIndex, kind);
|
|
2243
|
+
if (kind === "tool_use" && !toolName) yield {
|
|
2244
|
+
type: "tool_call_delta",
|
|
2245
|
+
index: contentBlockIndex,
|
|
2246
|
+
id: start.toolUse.toolUseId,
|
|
2247
|
+
name: start.toolUse.name
|
|
2248
|
+
};
|
|
2249
|
+
} else blockKinds.set(contentBlockIndex, "text");
|
|
2250
|
+
} else if ("contentBlockDelta" in event) {
|
|
2251
|
+
const { contentBlockIndex, delta } = event.contentBlockDelta;
|
|
2252
|
+
if (delta && "text" in delta && delta.text !== void 0 && !toolName) yield {
|
|
2253
|
+
type: "text-delta",
|
|
2254
|
+
delta: delta.text
|
|
2255
|
+
};
|
|
2256
|
+
else if (delta && "toolUse" in delta && delta.toolUse?.input !== void 0) {
|
|
2257
|
+
const kind = blockKinds.get(contentBlockIndex);
|
|
2258
|
+
if (kind === "json-tool") yield {
|
|
2259
|
+
type: "text-delta",
|
|
2260
|
+
delta: delta.toolUse.input
|
|
2261
|
+
};
|
|
2262
|
+
else if (!toolName) yield {
|
|
2263
|
+
type: "tool_call_delta",
|
|
2264
|
+
index: contentBlockIndex,
|
|
2265
|
+
argumentsDelta: delta.toolUse.input
|
|
2266
|
+
};
|
|
2267
|
+
}
|
|
2268
|
+
} else if ("metadata" in event && event.metadata.usage) yield {
|
|
2269
|
+
type: "usage",
|
|
2270
|
+
usage: {
|
|
2271
|
+
prompt_tokens: event.metadata.usage.inputTokens,
|
|
2272
|
+
completion_tokens: event.metadata.usage.outputTokens,
|
|
2273
|
+
total_tokens: event.metadata.usage.totalTokens
|
|
2274
|
+
}
|
|
960
2275
|
};
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
2276
|
+
else if ("throttlingException" in event) throw new LLMError(event.throttlingException.message ?? "Bedrock throttled the request mid-stream", "api", 429);
|
|
2277
|
+
else if ("validationException" in event) throw new LLMError(event.validationException.message ?? "Bedrock rejected the request mid-stream", "validation");
|
|
2278
|
+
else if ("internalServerException" in event || "serviceUnavailableException" in event || "modelStreamErrorException" in event) {
|
|
2279
|
+
const detail = "internalServerException" in event && event.internalServerException.message || "serviceUnavailableException" in event && event.serviceUnavailableException.message || "modelStreamErrorException" in event && event.modelStreamErrorException.message || "Bedrock reported a mid-stream error";
|
|
2280
|
+
const status = "modelStreamErrorException" in event && event.modelStreamErrorException.originalStatusCode || "serviceUnavailableException" in event && 503 || 500;
|
|
2281
|
+
throw new LLMError(detail, "api", status);
|
|
2282
|
+
}
|
|
965
2283
|
}
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
2284
|
+
} } };
|
|
2285
|
+
}
|
|
2286
|
+
/** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Converse's `toolChoice`. */
|
|
2287
|
+
function toBedrockToolChoice(toolChoice) {
|
|
2288
|
+
if (!toolChoice || toolChoice === "auto") return { auto: {} };
|
|
2289
|
+
if (toolChoice === "required") return { any: {} };
|
|
2290
|
+
if (toolChoice === "none") throw new LLMError("'none' is not supported by fromBedrock: Bedrock Converse has no `tool_choice` equivalent to forbidding tool use while tools are still offered. Omit `tools` entirely for this call instead.", "validation");
|
|
2291
|
+
return { tool: { name: toolChoice.function.name } };
|
|
2292
|
+
}
|
|
2293
|
+
/**
|
|
2294
|
+
* Translates one VernLLM wire message into Converse's
|
|
2295
|
+
* `{ role: 'user' | 'assistant', content }` shape.
|
|
2296
|
+
*/
|
|
2297
|
+
function toBedrockMessage(m) {
|
|
2298
|
+
if (m.role === "tool") return {
|
|
2299
|
+
role: "user",
|
|
2300
|
+
content: [{ toolResult: {
|
|
2301
|
+
toolUseId: m.tool_call_id,
|
|
2302
|
+
content: [{ text: m.content }],
|
|
2303
|
+
status: m.is_error ? "error" : "success"
|
|
2304
|
+
} }]
|
|
2305
|
+
};
|
|
2306
|
+
if (m.role === "assistant" && m.tool_calls?.length) {
|
|
2307
|
+
const blocks = [];
|
|
2308
|
+
if (m.content) blocks.push({ text: m.content });
|
|
2309
|
+
for (const tc of m.tool_calls) {
|
|
2310
|
+
let input;
|
|
2311
|
+
if (!tc.function.arguments.trim()) input = {};
|
|
2312
|
+
else try {
|
|
2313
|
+
input = JSON.parse(tc.function.arguments);
|
|
2314
|
+
} catch (cause) {
|
|
2315
|
+
throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
|
|
991
2316
|
}
|
|
2317
|
+
blocks.push({ toolUse: {
|
|
2318
|
+
toolUseId: tc.id,
|
|
2319
|
+
name: tc.function.name,
|
|
2320
|
+
input
|
|
2321
|
+
} });
|
|
2322
|
+
}
|
|
2323
|
+
return {
|
|
2324
|
+
role: "assistant",
|
|
2325
|
+
content: blocks
|
|
992
2326
|
};
|
|
993
|
-
}
|
|
2327
|
+
}
|
|
2328
|
+
return {
|
|
2329
|
+
role: m.role,
|
|
2330
|
+
content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content ?? "" }]
|
|
2331
|
+
};
|
|
2332
|
+
}
|
|
2333
|
+
/**
|
|
2334
|
+
* Converse expects the results of everything the model asked for in one
|
|
2335
|
+
* turn to arrive together as multiple `toolResult` content blocks on a
|
|
2336
|
+
* single `'user'` message, not as separate consecutive `'user'` messages.
|
|
2337
|
+
* The per-wire-message mapping above produces one `'user'` message per
|
|
2338
|
+
* VernLLM wire tool message, so when an assistant turn requested more than
|
|
2339
|
+
* one tool, this merges the resulting run of toolResult-only `'user'`
|
|
2340
|
+
* messages back into one.
|
|
2341
|
+
*/
|
|
2342
|
+
function mergeConsecutiveToolResults(messages) {
|
|
2343
|
+
const isToolResultOnly = (m) => m.role === "user" && m.content.length > 0 && m.content.every((b) => "toolResult" in b);
|
|
2344
|
+
const merged = [];
|
|
2345
|
+
for (const m of messages) {
|
|
2346
|
+
const prev = merged.at(-1);
|
|
2347
|
+
if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
|
|
2348
|
+
else merged.push(m);
|
|
2349
|
+
}
|
|
2350
|
+
return merged;
|
|
994
2351
|
}
|
|
995
2352
|
|
|
996
2353
|
//#endregion
|
|
997
2354
|
//#region src/adapters/fetch.ts
|
|
998
2355
|
/**
|
|
2356
|
+
* Wraps a WHATWG `ReadableStream` (what `response.body` is) so it can be
|
|
2357
|
+
* consumed with `for await`. Implemented via `getReader()` rather than
|
|
2358
|
+
* relying on `ReadableStream` having a native `Symbol.asyncIterator`,
|
|
2359
|
+
* that support varies across runtimes/versions, and this works everywhere
|
|
2360
|
+
* a `ReadableStream` does.
|
|
2361
|
+
*/
|
|
2362
|
+
async function* webStreamToAsyncIterable(stream) {
|
|
2363
|
+
const reader = stream.getReader();
|
|
2364
|
+
try {
|
|
2365
|
+
for (;;) {
|
|
2366
|
+
const { done, value } = await reader.read();
|
|
2367
|
+
if (done) return;
|
|
2368
|
+
if (value) yield value;
|
|
2369
|
+
}
|
|
2370
|
+
} finally {
|
|
2371
|
+
try {
|
|
2372
|
+
await reader.cancel();
|
|
2373
|
+
} catch {}
|
|
2374
|
+
reader.releaseLock();
|
|
2375
|
+
}
|
|
2376
|
+
}
|
|
2377
|
+
/** Default `requestStream`: native `fetch`, with the same error/`.status` contract non-streaming errors get. */
|
|
2378
|
+
async function defaultRequestStream(url, init) {
|
|
2379
|
+
const res = await fetch(url, init);
|
|
2380
|
+
if (!res.ok) {
|
|
2381
|
+
const body = await res.text().catch(() => "");
|
|
2382
|
+
const err = new Error(`Fetch adapter stream request failed (${res.status}): ${body.slice(0, 500)}`);
|
|
2383
|
+
err.status = res.status;
|
|
2384
|
+
err.headers = res.headers;
|
|
2385
|
+
throw err;
|
|
2386
|
+
}
|
|
2387
|
+
if (!res.body) throw new Error("Fetch adapter stream request received a response with no body.");
|
|
2388
|
+
return webStreamToAsyncIterable(res.body);
|
|
2389
|
+
}
|
|
2390
|
+
/** Builds the shared `{ method, headers, body? }` request-init for both `create` and `createStream`. */
|
|
2391
|
+
async function buildRequestInit(config, params, requestBody) {
|
|
2392
|
+
const url = typeof config.url === "function" ? config.url(params) : config.url;
|
|
2393
|
+
const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
|
|
2394
|
+
const method = config.method ?? "POST";
|
|
2395
|
+
const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
|
|
2396
|
+
return {
|
|
2397
|
+
url,
|
|
2398
|
+
method,
|
|
2399
|
+
headers: supportsBody ? {
|
|
2400
|
+
"Content-Type": "application/json",
|
|
2401
|
+
...headers
|
|
2402
|
+
} : { ...headers },
|
|
2403
|
+
...supportsBody ? { body: JSON.stringify(requestBody) } : {}
|
|
2404
|
+
};
|
|
2405
|
+
}
|
|
2406
|
+
/**
|
|
999
2407
|
* A fetch-based escape hatch for providers with no SDK, or where pulling one
|
|
1000
2408
|
* in isnt worth it. You supply the URL, headers, and two small mapping
|
|
1001
2409
|
* functions; this handles the HTTP call and slots the result into the same
|
|
@@ -1005,41 +2413,99 @@ function fromBedrock(bedrockClient, options) {
|
|
|
1005
2413
|
* Non-2xx responses throw an error with `.status` set to the HTTP status
|
|
1006
2414
|
* code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
|
|
1007
2415
|
* 401/403) applies here too
|
|
2416
|
+
*
|
|
2417
|
+
* Tool calling works the same way as every other adapter: `mapRequest`
|
|
2418
|
+
* receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
|
|
2419
|
+
* translate them into whatever shape the provider's wire format expects
|
|
2420
|
+
* (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
|
|
2421
|
+
* field). On the way back, `mapResponse` may return a `toolCalls` array
|
|
2422
|
+
* (id/name/JSON-encoded-arguments-string per call) alongside or instead of
|
|
2423
|
+
* `content`; VernLLM parses and (if `argumentsSchema` was set) validates
|
|
2424
|
+
* those arguments the same way it does for every other adapter. For
|
|
2425
|
+
* `stream: true`, tool-call deltas go through the existing
|
|
2426
|
+
* `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
|
|
2427
|
+
* no separate config is needed for streaming vs non-streaming tool calls.
|
|
2428
|
+
*
|
|
2429
|
+
|
|
2430
|
+
* `createStream` requires `mapStreamEvent` (there's no non-streaming
|
|
2431
|
+
* response to fall back on, unlike the other three optional streaming
|
|
2432
|
+
* seams). It opens the request via `requestStream` (defaults to native
|
|
2433
|
+
* `fetch`), splits the raw bytes into individual events via
|
|
2434
|
+
* `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
|
|
2435
|
+
* and translates each event into `WireStreamChunk`(s) via
|
|
2436
|
+
* `mapStreamEvent`. Both seams are overridable per-config for providers
|
|
2437
|
+
* that don't fit the SSE-over-fetch default. If a custom `request`
|
|
2438
|
+
* transport is configured, `requestStream` must be configured too,
|
|
2439
|
+
* `requestStream` never silently falls back to `request` (see
|
|
2440
|
+
* `createStream`'s own comment for why), so a `stream: true` call with
|
|
2441
|
+
* `request` set but no `requestStream` throws a clear
|
|
2442
|
+
* `LLMError('validation')` instead of quietly using unrelated native
|
|
2443
|
+
* `fetch`.
|
|
1008
2444
|
*/
|
|
1009
2445
|
function fromFetch(config) {
|
|
1010
|
-
return { chat: { completions: {
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
const
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
2446
|
+
return { chat: { completions: {
|
|
2447
|
+
async create(params, options) {
|
|
2448
|
+
const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
|
|
2449
|
+
const request = config.request ?? fetch;
|
|
2450
|
+
const res = await request(url, {
|
|
2451
|
+
method,
|
|
2452
|
+
headers,
|
|
2453
|
+
body,
|
|
2454
|
+
signal: options.signal
|
|
2455
|
+
});
|
|
2456
|
+
if (!res.ok) {
|
|
2457
|
+
const responseBody = await res.text().catch(() => "");
|
|
2458
|
+
const err = new Error(`Fetch adapter request failed (${res.status}): ${responseBody.slice(0, 500)}`);
|
|
2459
|
+
err.status = res.status;
|
|
2460
|
+
err.headers = res.headers;
|
|
2461
|
+
throw err;
|
|
2462
|
+
}
|
|
2463
|
+
const json = await res.json();
|
|
2464
|
+
const { content, usage, toolCalls } = config.mapResponse(json);
|
|
2465
|
+
const wireToolCalls = toolCalls?.length ? toolCalls.map((tc) => ({
|
|
2466
|
+
id: tc.id,
|
|
2467
|
+
type: "function",
|
|
2468
|
+
function: {
|
|
2469
|
+
name: tc.name,
|
|
2470
|
+
arguments: tc.arguments
|
|
2471
|
+
}
|
|
2472
|
+
})) : void 0;
|
|
2473
|
+
return {
|
|
2474
|
+
choices: [{ message: {
|
|
2475
|
+
content,
|
|
2476
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
2477
|
+
} }],
|
|
2478
|
+
usage: usage ? {
|
|
2479
|
+
prompt_tokens: usage.promptTokens,
|
|
2480
|
+
completion_tokens: usage.completionTokens,
|
|
2481
|
+
total_tokens: usage.totalTokens
|
|
2482
|
+
} : void 0
|
|
2483
|
+
};
|
|
2484
|
+
},
|
|
2485
|
+
async *createStream(params, options) {
|
|
2486
|
+
if (!config.mapStreamEvent) throw new LLMError("stream: true requires mapStreamEvent to be configured on fromFetch", "validation");
|
|
2487
|
+
if (config.request && !config.requestStream) throw new LLMError("`stream: true` requires `requestStream` to be configured on fromFetch when a custom `request` transport is set. `requestStream` does not fall back to `request` (it needs an async-iterable byte stream, which `RequestLike`'s buffered `ResponseLike` has no way to provide), without it, `stream: true` would silently use plain native `fetch` instead of your configured transport. Add a `requestStream` that opens the same connection your `request` does, or omit `request` if native `fetch` is fine for both.", "validation");
|
|
2488
|
+
const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
|
|
2489
|
+
const requestStream = config.requestStream ?? defaultRequestStream;
|
|
2490
|
+
const parseFrames = config.parseStreamFrames ?? parseSseStream;
|
|
2491
|
+
const byteStream = await requestStream(url, {
|
|
2492
|
+
method,
|
|
2493
|
+
headers,
|
|
2494
|
+
body,
|
|
2495
|
+
signal: options.signal
|
|
2496
|
+
});
|
|
2497
|
+
for await (const event of parseFrames(byteStream)) {
|
|
2498
|
+
if (event === SSE_PING) {
|
|
2499
|
+
yield { type: "ping" };
|
|
2500
|
+
continue;
|
|
2501
|
+
}
|
|
2502
|
+
const wireChunks = config.mapStreamEvent(event);
|
|
2503
|
+
if (!wireChunks) continue;
|
|
2504
|
+
if (Array.isArray(wireChunks)) yield* wireChunks;
|
|
2505
|
+
else yield wireChunks;
|
|
2506
|
+
}
|
|
1031
2507
|
}
|
|
1032
|
-
|
|
1033
|
-
const { content, usage } = config.mapResponse(json);
|
|
1034
|
-
return {
|
|
1035
|
-
choices: [{ message: { content } }],
|
|
1036
|
-
usage: usage ? {
|
|
1037
|
-
prompt_tokens: usage.promptTokens,
|
|
1038
|
-
completion_tokens: usage.completionTokens,
|
|
1039
|
-
total_tokens: usage.totalTokens
|
|
1040
|
-
} : void 0
|
|
1041
|
-
};
|
|
1042
|
-
} } } };
|
|
2508
|
+
} } };
|
|
1043
2509
|
}
|
|
1044
2510
|
|
|
1045
2511
|
//#endregion
|
|
@@ -1062,41 +2528,84 @@ function toOpenAIContent(blocks) {
|
|
|
1062
2528
|
});
|
|
1063
2529
|
}
|
|
1064
2530
|
/**
|
|
1065
|
-
*
|
|
1066
|
-
*
|
|
1067
|
-
*
|
|
1068
|
-
*
|
|
1069
|
-
* this exists purely so call sites read clearly (`fromMistral(client)` vs
|
|
1070
|
-
* handing a Mistral client to something typed for OpenAI) and so a real
|
|
1071
|
-
* transformation could be added later, per-provider, without a breaking
|
|
1072
|
-
* change.
|
|
1073
|
-
*
|
|
1074
|
-
* The one thing that isn't a pure passthrough: a `ContentBlock[]`
|
|
1075
|
-
* `userContent` is translated into OpenAI's native `image_url` content-part
|
|
1076
|
-
* shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
|
|
1077
|
-
* rather than a copy of any one provider's wire format.
|
|
1078
|
-
*
|
|
1079
|
-
* Not every SDKs own TypeScript types line up exactly with `LLMClient`
|
|
1080
|
-
* (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
|
|
1081
|
-
* the actual compatibility contract is the JSON each provider sends and
|
|
1082
|
-
* receives over the wire, not the SDKs TS types.
|
|
2531
|
+
* Translates VernLLM's provider-agnostic `messages` (the one part of a
|
|
2532
|
+
* request that isn't a pure passthrough for OpenAI-compatible clients) into
|
|
2533
|
+
* OpenAI's native wire shape. Shared between `create` and `createStream` so
|
|
2534
|
+
* both go through identical message translation.
|
|
1083
2535
|
*/
|
|
1084
|
-
function
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
|
|
2536
|
+
function toOpenAIMessages(params) {
|
|
2537
|
+
return params.messages.map((m) => {
|
|
2538
|
+
if (m.role === "user" && Array.isArray(m.content)) return {
|
|
1088
2539
|
...m,
|
|
1089
2540
|
content: toOpenAIContent(m.content)
|
|
1090
|
-
}
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
}
|
|
1095
|
-
|
|
2541
|
+
};
|
|
2542
|
+
if (m.role === "tool") {
|
|
2543
|
+
const { is_error: _isError,...openAIToolMessage } = m;
|
|
2544
|
+
return openAIToolMessage;
|
|
2545
|
+
}
|
|
2546
|
+
return m;
|
|
2547
|
+
});
|
|
2548
|
+
}
|
|
2549
|
+
/**
|
|
2550
|
+
* Translates one OpenAI-shaped SSE chunk into zero or more `WireStreamChunk`s.
|
|
2551
|
+
* A single chunk can carry a text delta, one or more tool-call argument
|
|
2552
|
+
* deltas (each keyed by `index`, OpenAI's own convention for streaming
|
|
2553
|
+
* parallel tool calls, mirrored directly by VernLLM's `tool_call_delta`
|
|
2554
|
+
* shape so accumulation composes without translation), and/or a final
|
|
2555
|
+
* usage block (present only when `stream_options.include_usage` is set,
|
|
2556
|
+
* which this adapter always sets).
|
|
2557
|
+
*/
|
|
2558
|
+
function* toWireStreamChunks(chunk) {
|
|
2559
|
+
const delta = chunk.choices?.[0]?.delta;
|
|
2560
|
+
if (delta?.content) yield {
|
|
2561
|
+
type: "text-delta",
|
|
2562
|
+
delta: delta.content
|
|
2563
|
+
};
|
|
2564
|
+
if (delta?.tool_calls?.length) for (const toolCall of delta.tool_calls) yield {
|
|
2565
|
+
type: "tool_call_delta",
|
|
2566
|
+
index: toolCall.index,
|
|
2567
|
+
id: toolCall.id,
|
|
2568
|
+
name: toolCall.function?.name,
|
|
2569
|
+
argumentsDelta: toolCall.function?.arguments
|
|
2570
|
+
};
|
|
2571
|
+
if (chunk.usage) yield {
|
|
2572
|
+
type: "usage",
|
|
2573
|
+
usage: chunk.usage
|
|
2574
|
+
};
|
|
2575
|
+
}
|
|
2576
|
+
function fromOpenAICompatible(client, options = {}) {
|
|
2577
|
+
const raw = client;
|
|
2578
|
+
const { supportsStreamUsage = true } = options;
|
|
2579
|
+
const rawCreate = raw.chat.completions.create.bind(raw.chat.completions);
|
|
2580
|
+
return { chat: { completions: {
|
|
2581
|
+
async create(params, options$1) {
|
|
2582
|
+
const messages = toOpenAIMessages(params);
|
|
2583
|
+
return raw.chat.completions.create({
|
|
2584
|
+
...params,
|
|
2585
|
+
messages
|
|
2586
|
+
}, options$1);
|
|
2587
|
+
},
|
|
2588
|
+
async *createStream(params, options$1) {
|
|
2589
|
+
const messages = toOpenAIMessages(params);
|
|
2590
|
+
const stream = await rawCreate({
|
|
2591
|
+
...params,
|
|
2592
|
+
messages,
|
|
2593
|
+
stream: true,
|
|
2594
|
+
...supportsStreamUsage ? { stream_options: { include_usage: true } } : {}
|
|
2595
|
+
}, options$1);
|
|
2596
|
+
for await (const chunk of stream) yield* toWireStreamChunks(chunk);
|
|
2597
|
+
}
|
|
2598
|
+
} } };
|
|
1096
2599
|
}
|
|
1097
2600
|
/** Groqs SDK matches the OpenAI wire format */
|
|
1098
2601
|
const fromGroq = fromOpenAICompatible;
|
|
1099
|
-
/**
|
|
2602
|
+
/**
|
|
2603
|
+
* Mistrals `chat.completions`-shaped client (or their OpenAI-compat
|
|
2604
|
+
* endpoint). Mistral supports `stream_options.include_usage` (added after
|
|
2605
|
+
* an earlier period where it returned a 422 for unrecognized fields, per
|
|
2606
|
+
* Mistral's changelog and streaming docs), so this is a plain alias like
|
|
2607
|
+
* the others, `supportsStreamUsage` defaults to `true`.
|
|
2608
|
+
*/
|
|
1100
2609
|
const fromMistral = fromOpenAICompatible;
|
|
1101
2610
|
/** DeepSeeks API is OpenAI-compatible */
|
|
1102
2611
|
const fromDeepSeek = fromOpenAICompatible;
|
|
@@ -1185,5 +2694,5 @@ const fromAtlasCloud = fromOpenAICompatible;
|
|
|
1185
2694
|
const from01AI = fromOpenAICompatible;
|
|
1186
2695
|
|
|
1187
2696
|
//#endregion
|
|
1188
|
-
export { CircuitBreaker, ConsoleLogger, InMemoryCacheAdapter, LLMError, NormalizedCacheAdapter, TieredCacheAdapter, VernLLM, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError };
|
|
2697
|
+
export { CircuitBreaker, ConsoleLogger, InMemoryCacheAdapter, LLMError, NormalizedCacheAdapter, SSE_PING, TieredCacheAdapter, VernLLM, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError, isToolCallResult, parseSseStream };
|
|
1189
2698
|
//# sourceMappingURL=index.js.map
|