vern-llm 1.7.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -6
- package/dist/index.cjs +1784 -272
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1009 -73
- package/dist/index.d.cts.map +1 -1
- package/dist/index.d.ts +1009 -73
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1782 -273
- package/dist/index.js.map +1 -1
- package/package.json +9 -13
package/dist/index.cjs
CHANGED
|
@@ -75,7 +75,7 @@ var CircuitBreaker = class {
|
|
|
75
75
|
if (this.state === "closed") return;
|
|
76
76
|
if (this.state === "open") {
|
|
77
77
|
const elapsed = Date.now() - this.openedAt;
|
|
78
|
-
if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open
|
|
78
|
+
if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open, provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
|
|
79
79
|
this.state = "half-open";
|
|
80
80
|
this.trialInFlight = true;
|
|
81
81
|
return;
|
|
@@ -108,6 +108,48 @@ var CircuitBreaker = class {
|
|
|
108
108
|
|
|
109
109
|
//#endregion
|
|
110
110
|
//#region src/internal/vernLLM.utils.ts
|
|
111
|
+
/** Translates app-facing `ToolDefinition[]` into the OpenAI-shaped wire tools array. */
|
|
112
|
+
function toWireTools(tools) {
|
|
113
|
+
return tools.map((tool) => ({
|
|
114
|
+
type: "function",
|
|
115
|
+
function: {
|
|
116
|
+
name: tool.name,
|
|
117
|
+
description: tool.description,
|
|
118
|
+
parameters: tool.parameters
|
|
119
|
+
}
|
|
120
|
+
}));
|
|
121
|
+
}
|
|
122
|
+
/** Translates app-facing `ToolCall[]` (e.g. from a replayed assistant turn) into wire tool_calls. */
|
|
123
|
+
function toWireToolCalls(toolCalls) {
|
|
124
|
+
return toolCalls.map((tc) => ({
|
|
125
|
+
id: tc.id,
|
|
126
|
+
type: "function",
|
|
127
|
+
function: {
|
|
128
|
+
name: tc.name,
|
|
129
|
+
arguments: JSON.stringify(tc.arguments ?? {})
|
|
130
|
+
}
|
|
131
|
+
}));
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Parses the provider's wire-shaped `tool_calls` back into VernLLM's
|
|
135
|
+
* `ToolCall[]`. Malformed argument JSON is a `'parse'` error, same
|
|
136
|
+
* convention as malformed JSON response bodies elsewhere in VernLLM.
|
|
137
|
+
*/
|
|
138
|
+
function parseWireToolCalls(wireToolCalls) {
|
|
139
|
+
return wireToolCalls.map((wc) => {
|
|
140
|
+
let parsedArgs;
|
|
141
|
+
try {
|
|
142
|
+
parsedArgs = wc.function.arguments.trim() ? JSON.parse(wc.function.arguments) : {};
|
|
143
|
+
} catch {
|
|
144
|
+
throw new LLMError(`Invalid JSON arguments for tool call "${wc.function.name}"`, "parse");
|
|
145
|
+
}
|
|
146
|
+
return {
|
|
147
|
+
id: wc.id,
|
|
148
|
+
name: wc.function.name,
|
|
149
|
+
arguments: parsedArgs
|
|
150
|
+
};
|
|
151
|
+
});
|
|
152
|
+
}
|
|
111
153
|
function defaultParseJson(content) {
|
|
112
154
|
try {
|
|
113
155
|
return JSON.parse(content);
|
|
@@ -118,7 +160,9 @@ function defaultParseJson(content) {
|
|
|
118
160
|
/**
|
|
119
161
|
* Looks inside an unknown error value and pulls out an http status code
|
|
120
162
|
* if one is present. Checks the status field first then the status code
|
|
121
|
-
* field since different client libraries use different names for this
|
|
163
|
+
* field since different client libraries use different names for this,
|
|
164
|
+
* falling back to AWS SDK v3's `$metadata.httpStatusCode` (e.g. Bedrock's
|
|
165
|
+
* `ThrottlingException`), which doesn't set either of the other two.
|
|
122
166
|
* Returns undefined when the error is not an object or carries no status
|
|
123
167
|
*/
|
|
124
168
|
function extractStatus(err) {
|
|
@@ -126,6 +170,7 @@ function extractStatus(err) {
|
|
|
126
170
|
const error = err;
|
|
127
171
|
if (typeof error.status === "number") return error.status;
|
|
128
172
|
if (typeof error.statusCode === "number") return error.statusCode;
|
|
173
|
+
if (typeof error.$metadata?.httpStatusCode === "number") return error.$metadata.httpStatusCode;
|
|
129
174
|
return void 0;
|
|
130
175
|
}
|
|
131
176
|
function formatSafely(value) {
|
|
@@ -155,6 +200,21 @@ function describeError(err) {
|
|
|
155
200
|
return formatSafely(err);
|
|
156
201
|
}
|
|
157
202
|
/**
|
|
203
|
+
* `setTimeout` silently clamps any delay above this (~24.8 days) or
|
|
204
|
+
* `Infinity` down to ~1ms instead of erroring, so a caller passing
|
|
205
|
+
* `Infinity` as "no timeout" gets the opposite of what they asked for.
|
|
206
|
+
* Both timeout helpers below guard against this explicitly.
|
|
207
|
+
*/
|
|
208
|
+
const MAX_SETTIMEOUT_MS = 2147483647;
|
|
209
|
+
/** True when a timeout value should be treated as "disabled" rather than passed to `setTimeout`. */
|
|
210
|
+
function isTimeoutDisabled(ms) {
|
|
211
|
+
return !ms || ms <= 0 || ms === Infinity;
|
|
212
|
+
}
|
|
213
|
+
/** Caps a timeout at the largest delay `setTimeout` actually honors. */
|
|
214
|
+
function clampTimeoutMs(ms) {
|
|
215
|
+
return Math.min(ms, MAX_SETTIMEOUT_MS);
|
|
216
|
+
}
|
|
217
|
+
/**
|
|
158
218
|
* Runs an async function and cancels it if it takes longer than the given
|
|
159
219
|
* timeout. Creates an internal abort controller that fires after the
|
|
160
220
|
* timeout elapses, and combines it with any external signal the caller
|
|
@@ -164,12 +224,15 @@ function describeError(err) {
|
|
|
164
224
|
* continue to propagate as aborted errors. The internal timer is always
|
|
165
225
|
* cleared afterward, whether the function succeeds, fails, or is aborted,
|
|
166
226
|
* so nothing is left running in the background.
|
|
227
|
+
*
|
|
228
|
+
* `timeoutMs` of `Infinity` (or any value beyond what `setTimeout` can
|
|
229
|
+
* represent) disables the timeout rather than firing almost immediately.
|
|
167
230
|
*/
|
|
168
231
|
async function withTimeout(fn, timeoutMs, externalSignal) {
|
|
169
232
|
const controller = new AbortController();
|
|
170
|
-
const timer = setTimeout(() => {
|
|
233
|
+
const timer = isTimeoutDisabled(timeoutMs) ? void 0 : setTimeout(() => {
|
|
171
234
|
controller.abort();
|
|
172
|
-
}, timeoutMs);
|
|
235
|
+
}, clampTimeoutMs(timeoutMs));
|
|
173
236
|
const signal = externalSignal ? AbortSignal.any([externalSignal, controller.signal]) : controller.signal;
|
|
174
237
|
try {
|
|
175
238
|
return await fn(signal);
|
|
@@ -181,6 +244,54 @@ async function withTimeout(fn, timeoutMs, externalSignal) {
|
|
|
181
244
|
}
|
|
182
245
|
}
|
|
183
246
|
/**
|
|
247
|
+
* Races one `iterator.next()` call against a per-call idle timer, to
|
|
248
|
+
* bound the gap *between* chunks (unlike `withTimeout`, which only bounds
|
|
249
|
+
* opening the stream and its first chunk). Without this, a connection
|
|
250
|
+
* that streams one chunk then hangs would never fail.
|
|
251
|
+
*
|
|
252
|
+
* `timeoutMs` of 0/undefined/`Infinity` disables the check. Otherwise
|
|
253
|
+
* rejects with `LLMError('timeout')` if `next()` doesn't settle in time.
|
|
254
|
+
* The clock resets on every call, so the window is measured from the most
|
|
255
|
+
* recent chunk, not from stream start.
|
|
256
|
+
*
|
|
257
|
+
* `onIdle`, if given, is called the moment the timer fires (before the
|
|
258
|
+
* rejection), so callers can abort the underlying transport instead of
|
|
259
|
+
* just walking away from an unread promise. `logger`, if given, records a
|
|
260
|
+
* debug line if `next()` still settles *after* the idle timeout already
|
|
261
|
+
* rejected. `resolve`/`reject` on an already-settled promise is otherwise
|
|
262
|
+
* a silent no-op, so without this the late chunk (possibly the final
|
|
263
|
+
* usage chunk) would vanish with no trace.
|
|
264
|
+
*/
|
|
265
|
+
function withChunkIdleTimeout(next, timeoutMs, onIdle, logger) {
|
|
266
|
+
if (isTimeoutDisabled(timeoutMs)) return next();
|
|
267
|
+
const activeTimeoutMs = timeoutMs;
|
|
268
|
+
let settled = false;
|
|
269
|
+
return new Promise((resolve, reject) => {
|
|
270
|
+
const timer = setTimeout(() => {
|
|
271
|
+
settled = true;
|
|
272
|
+
onIdle?.();
|
|
273
|
+
reject(new LLMError(`No stream chunk received for ${activeTimeoutMs}ms (idle timeout)`, "timeout"));
|
|
274
|
+
}, clampTimeoutMs(activeTimeoutMs));
|
|
275
|
+
next().then((result) => {
|
|
276
|
+
clearTimeout(timer);
|
|
277
|
+
if (settled) {
|
|
278
|
+
logger?.debug("[VernLLM] chunk resolved after idle timeout already fired; discarding");
|
|
279
|
+
return;
|
|
280
|
+
}
|
|
281
|
+
settled = true;
|
|
282
|
+
resolve(result);
|
|
283
|
+
}, (error) => {
|
|
284
|
+
clearTimeout(timer);
|
|
285
|
+
if (settled) {
|
|
286
|
+
logger?.debug("[VernLLM] chunk rejection arrived after idle timeout already fired; discarding");
|
|
287
|
+
return;
|
|
288
|
+
}
|
|
289
|
+
settled = true;
|
|
290
|
+
reject(error);
|
|
291
|
+
});
|
|
292
|
+
});
|
|
293
|
+
}
|
|
294
|
+
/**
|
|
184
295
|
* Default cap (ms) for both exponential backoff and honored Retry-After
|
|
185
296
|
* values, so a misbehaving/adversarial Retry-After can't stall a caller
|
|
186
297
|
* indefinitely
|
|
@@ -295,6 +406,143 @@ async function withReservedUsage(params, coalesced, getResult, signal, onRefundE
|
|
|
295
406
|
}
|
|
296
407
|
return result;
|
|
297
408
|
}
|
|
409
|
+
/**
|
|
410
|
+
* Streaming counterpart to `withReservedUsage`. `withReservedUsage` assumes
|
|
411
|
+
* `getResult()` settling *is* the operation's final outcome, awaiting it
|
|
412
|
+
* synchronously before reserve/refund resolve. Streaming can't satisfy that:
|
|
413
|
+
* `call()` must return `{ chunks, finalResult }` as soon as the stream
|
|
414
|
+
* opens, well before the real outcome (validation, schema/tool-call checks)
|
|
415
|
+
* is known.
|
|
416
|
+
*
|
|
417
|
+
* Reserves usage before `openStream` runs, same failure mode as the
|
|
418
|
+
* non-streaming path if `reserveUsage` itself throws (mapped to
|
|
419
|
+
* `quota_exceeded`, nothing opened). If `openStream` itself throws (stream
|
|
420
|
+
* never opened), refunds synchronously and rethrows, exactly like
|
|
421
|
+
* `withReservedUsage` does today. If it succeeds, returns `{ chunks,
|
|
422
|
+
* finalResult }` immediately, refund/report is deferred onto
|
|
423
|
+
* `finalResult`'s continuation, since that's the only point the real
|
|
424
|
+
* outcome is known. This means `onUsageFailure` (and any refund) can fire
|
|
425
|
+
* well after this function itself has returned.
|
|
426
|
+
*/
|
|
427
|
+
async function withReservedUsageForStream(params, openStream, signal, onRefundError) {
|
|
428
|
+
if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
|
|
429
|
+
let reserved = false;
|
|
430
|
+
try {
|
|
431
|
+
if (params.reserveUsage) {
|
|
432
|
+
await params.reserveUsage({
|
|
433
|
+
coalesced: false,
|
|
434
|
+
signal
|
|
435
|
+
});
|
|
436
|
+
reserved = true;
|
|
437
|
+
}
|
|
438
|
+
} catch (error) {
|
|
439
|
+
if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
|
|
440
|
+
throw new LLMError(error instanceof Error ? error.message : "Usage reservation failed", "quota_exceeded", void 0, void 0, error);
|
|
441
|
+
}
|
|
442
|
+
const refund = async (logMessage) => {
|
|
443
|
+
try {
|
|
444
|
+
await params.refundUsage?.({
|
|
445
|
+
coalesced: false,
|
|
446
|
+
signal
|
|
447
|
+
});
|
|
448
|
+
} catch (refundError) {
|
|
449
|
+
onRefundError(logMessage, refundError);
|
|
450
|
+
}
|
|
451
|
+
};
|
|
452
|
+
let opened;
|
|
453
|
+
try {
|
|
454
|
+
opened = await openStream();
|
|
455
|
+
} catch (error) {
|
|
456
|
+
if (reserved) await refund("[VernLLM] refundUsage failed after stream-open failure");
|
|
457
|
+
throw error;
|
|
458
|
+
}
|
|
459
|
+
const finalResult = opened.finalResult.then((value) => value, async (error) => {
|
|
460
|
+
if (reserved) await refund("[VernLLM] refundUsage failed after stream error");
|
|
461
|
+
throw error;
|
|
462
|
+
});
|
|
463
|
+
finalResult.catch(() => {});
|
|
464
|
+
return {
|
|
465
|
+
chunks: opened.chunks,
|
|
466
|
+
finalResult
|
|
467
|
+
};
|
|
468
|
+
}
|
|
469
|
+
/**
|
|
470
|
+
* Converts an already-known cache value back into a plausible "text" form
|
|
471
|
+
* for a one-shot replay chunk: passed through unchanged if it's already a
|
|
472
|
+
* string (the `jsonMode: false` case), otherwise `JSON.stringify`'d (the
|
|
473
|
+
* `jsonMode: true` case, where the cached value is the *parsed* result, not
|
|
474
|
+
* the original raw text). This is a reasonable reconstruction, not a
|
|
475
|
+
* byte-identical replay of whatever text the model originally streamed,
|
|
476
|
+
* good enough for `for await (const c of chunks)` call sites that don't
|
|
477
|
+
* branch on hit vs. miss, which is the only thing a cache-hit replay needs
|
|
478
|
+
* to support.
|
|
479
|
+
*/
|
|
480
|
+
function toReplayText(value) {
|
|
481
|
+
return typeof value === "string" ? value : JSON.stringify(value);
|
|
482
|
+
}
|
|
483
|
+
/**
|
|
484
|
+
* Builds a trivially-exhausted one-shot `chunks` iterable from an
|
|
485
|
+
* already-known value, used for a `cachedCall` cache hit, where there's no
|
|
486
|
+
* live generation to relay (see `VernLLM.cachedCall`'s docs). No `usage`
|
|
487
|
+
* chunk is emitted: a cache hit spent no real tokens, so there's nothing to
|
|
488
|
+
* report, matching how non-streaming `cachedCall` never calls `onUsage` on
|
|
489
|
+
* a hit either.
|
|
490
|
+
*
|
|
491
|
+
* `hasTools` must reflect whether the *original* call that produced this
|
|
492
|
+
* cached value had `tools` set, that's what determines whether `value` is
|
|
493
|
+
* `T` directly or a `CallWithToolsResult<T>` wrapper, and it isn't
|
|
494
|
+
* something that can be reliably guessed from the value's shape alone
|
|
495
|
+
* (a `schema`-validated `T` could coincidentally look like a
|
|
496
|
+
* `CallWithToolsResult`).
|
|
497
|
+
*/
|
|
498
|
+
function buildReplayChunks(value, hasTools) {
|
|
499
|
+
const items = [];
|
|
500
|
+
if (hasTools) {
|
|
501
|
+
const result = value;
|
|
502
|
+
if (result.type === "tool_calls") {
|
|
503
|
+
result.toolCalls.forEach((toolCall, index) => {
|
|
504
|
+
items.push({
|
|
505
|
+
type: "tool_call_delta",
|
|
506
|
+
index,
|
|
507
|
+
id: toolCall.id,
|
|
508
|
+
name: toolCall.name,
|
|
509
|
+
argsDelta: JSON.stringify(toolCall.arguments ?? {}),
|
|
510
|
+
complete: true
|
|
511
|
+
});
|
|
512
|
+
});
|
|
513
|
+
if (result.content) items.push({
|
|
514
|
+
type: "text-delta",
|
|
515
|
+
delta: result.content
|
|
516
|
+
});
|
|
517
|
+
} else items.push({
|
|
518
|
+
type: "text-delta",
|
|
519
|
+
delta: toReplayText(result.content)
|
|
520
|
+
});
|
|
521
|
+
} else items.push({
|
|
522
|
+
type: "text-delta",
|
|
523
|
+
delta: toReplayText(value)
|
|
524
|
+
});
|
|
525
|
+
return { async *[Symbol.asyncIterator]() {
|
|
526
|
+
for (const item of items) yield item;
|
|
527
|
+
} };
|
|
528
|
+
}
|
|
529
|
+
/**
|
|
530
|
+
* Streaming counterpart to `buildReplayChunks` for a `cachedCall` that
|
|
531
|
+
* *joined* an already-in-flight call for the same key rather than
|
|
532
|
+
* triggering one itself (see `runCachedStream`'s in-flight-coalescing
|
|
533
|
+
* path): there's no live stream to relay (it isn't this call's stream to
|
|
534
|
+
* relay, see the joiner-path comment in `runCachedStream`), but there's
|
|
535
|
+
* also no value yet, only a pending promise for one. Waits for `promise`,
|
|
536
|
+
* then delegates to `buildReplayChunks`. If `promise` rejects, iterating
|
|
537
|
+
* `chunks` throws that same error, consistent with how a live stream's
|
|
538
|
+
* `chunks` throws on a mid-stream failure.
|
|
539
|
+
*/
|
|
540
|
+
function buildReplayChunksFromPromise(promise, hasTools) {
|
|
541
|
+
return { async *[Symbol.asyncIterator]() {
|
|
542
|
+
const value = await promise;
|
|
543
|
+
yield* buildReplayChunks(value, hasTools);
|
|
544
|
+
} };
|
|
545
|
+
}
|
|
298
546
|
|
|
299
547
|
//#endregion
|
|
300
548
|
//#region src/logger.ts
|
|
@@ -427,34 +675,52 @@ var TieredCacheAdapter = class {
|
|
|
427
675
|
}
|
|
428
676
|
};
|
|
429
677
|
|
|
678
|
+
//#endregion
|
|
679
|
+
//#region src/types/tools.ts
|
|
680
|
+
/**
|
|
681
|
+
* Runtime-safe check for whether a `call()` result is a `tool_calls`
|
|
682
|
+
* result. Prefer this over relying on TypeScript's static narrowing
|
|
683
|
+
* whenever `params` passed to `call()` wasn't a literal with `tools`
|
|
684
|
+
* inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
|
|
685
|
+
* that case TS may have typed the result as plain `T` even though it's
|
|
686
|
+
* actually a `CallWithToolsResult<T>` at runtime, and this check works
|
|
687
|
+
* either way.
|
|
688
|
+
*/
|
|
689
|
+
function isToolCallResult(result) {
|
|
690
|
+
return typeof result === "object" && result !== null && "type" in result && result.type === "tool_calls" && Array.isArray(result.toolCalls);
|
|
691
|
+
}
|
|
692
|
+
|
|
430
693
|
//#endregion
|
|
431
694
|
//#region src/vernLLM.ts
|
|
432
695
|
/**
|
|
433
|
-
* A resilient layer around an LLM chat completions client
|
|
696
|
+
* A resilient layer around an LLM chat completions client. This is VernLLM!
|
|
434
697
|
*
|
|
435
|
-
* Adds retry with backoff
|
|
436
|
-
* JSON parsing with optional schema validation, usage
|
|
437
|
-
* optional response cache
|
|
438
|
-
* defaults.
|
|
698
|
+
* Adds retry with backoff and jitter, per-attempt timeouts, an optional
|
|
699
|
+
* circuit breaker, JSON parsing with optional schema validation, usage
|
|
700
|
+
* tracking, and an optional response cache. All configurable, all opt-in
|
|
701
|
+
* beyond sensible defaults.
|
|
439
702
|
*/
|
|
440
703
|
var VernLLM = class {
|
|
441
704
|
client;
|
|
442
705
|
model;
|
|
443
706
|
maxRetries;
|
|
444
707
|
timeoutMs;
|
|
708
|
+
chunkIdleTimeoutMs;
|
|
445
709
|
baseDelayMs;
|
|
446
710
|
defaultMaxTokens;
|
|
711
|
+
defaultTemperature;
|
|
447
712
|
cache;
|
|
448
713
|
nonRetryableStatus;
|
|
449
714
|
inFlight = new Map();
|
|
450
715
|
parseJson;
|
|
451
716
|
onUsage;
|
|
717
|
+
onUsageFailure;
|
|
452
718
|
logger;
|
|
453
719
|
breaker;
|
|
454
720
|
/**
|
|
455
|
-
* @param options
|
|
456
|
-
* `
|
|
457
|
-
*
|
|
721
|
+
* @param options Client, model, and tunables. Defaults: `maxRetries` 1,
|
|
722
|
+
* `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
|
|
723
|
+
* `defaultTemperature` 0.2, `cache` an in-memory adapter,
|
|
458
724
|
* `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
|
|
459
725
|
*/
|
|
460
726
|
constructor(options) {
|
|
@@ -462,8 +728,10 @@ var VernLLM = class {
|
|
|
462
728
|
this.model = options.model;
|
|
463
729
|
this.maxRetries = options.maxRetries ?? 1;
|
|
464
730
|
this.timeoutMs = options.timeoutMs ?? 25e3;
|
|
731
|
+
this.chunkIdleTimeoutMs = options.chunkIdleTimeoutMs ?? 3e4;
|
|
465
732
|
this.baseDelayMs = options.baseDelayMs ?? 500;
|
|
466
733
|
this.defaultMaxTokens = options.defaultMaxTokens ?? 1e3;
|
|
734
|
+
this.defaultTemperature = options.defaultTemperature === void 0 ? .2 : options.defaultTemperature;
|
|
467
735
|
this.cache = options.cache ?? new InMemoryCacheAdapter();
|
|
468
736
|
this.nonRetryableStatus = options.nonRetryableStatus ?? [
|
|
469
737
|
400,
|
|
@@ -474,6 +742,7 @@ var VernLLM = class {
|
|
|
474
742
|
];
|
|
475
743
|
this.parseJson = options.parseJson ?? defaultParseJson;
|
|
476
744
|
this.onUsage = options.onUsage;
|
|
745
|
+
this.onUsageFailure = options.onUsageFailure;
|
|
477
746
|
this.logger = options.logger ?? new ConsoleLogger(options.debug ?? false);
|
|
478
747
|
this.breaker = options.circuitBreaker ? new CircuitBreaker(options.circuitBreaker === true ? void 0 : options.circuitBreaker) : void 0;
|
|
479
748
|
}
|
|
@@ -481,38 +750,307 @@ var VernLLM = class {
|
|
|
481
750
|
async resolveCacheKey(key) {
|
|
482
751
|
return this.cache.resolveKey ? await this.cache.resolveKey(key) : key;
|
|
483
752
|
}
|
|
484
|
-
/**
|
|
485
|
-
* Makes a single logical LLM call, retrying on failure per the configured
|
|
486
|
-
* policy. Fails fast if the breaker is open or the signal is already
|
|
487
|
-
* aborted. On exhausting retries, records a breaker failure and rejects
|
|
488
|
-
* with a normalized LLMError.
|
|
489
|
-
*
|
|
490
|
-
* @param params - System/user content plus per-call overrides (model,
|
|
491
|
-
* temperature, jsonMode, schema, signal, etc). See `CallParams`.
|
|
492
|
-
* @returns The parsed (and optionally schema-validated) response, or the
|
|
493
|
-
* raw string content when `jsonMode` is false and no `jsonSchema` is set.
|
|
494
|
-
*/
|
|
495
753
|
async call(params) {
|
|
496
754
|
this.breaker?.assertClosed();
|
|
497
755
|
if (params.signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
|
|
498
756
|
const requestId = params.requestId ?? (0, crypto.randomUUID)();
|
|
757
|
+
if (params.stream) return withReservedUsageForStream(params, async () => {
|
|
758
|
+
try {
|
|
759
|
+
return await this.retryWithBackoff((attempt) => this.executeStreamCall(params, requestId, attempt), requestId, params.signal);
|
|
760
|
+
} catch (error) {
|
|
761
|
+
const normalized = normalizeError(error, params.signal);
|
|
762
|
+
if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
|
|
763
|
+
this.logger.debug(`[VernLLM:${requestId}] stream-open error:\n${describeError(error)}`);
|
|
764
|
+
throw normalized;
|
|
765
|
+
}
|
|
766
|
+
}, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
499
767
|
return withReservedUsage(params, false, async () => {
|
|
500
768
|
try {
|
|
501
|
-
return await this.retryWithBackoff(() => this.executeCall(params, requestId), requestId, params.signal);
|
|
769
|
+
return await this.retryWithBackoff((attempt) => this.executeCall(params, requestId, attempt), requestId, params.signal);
|
|
502
770
|
} catch (error) {
|
|
503
771
|
const normalized = normalizeError(error, params.signal);
|
|
504
772
|
if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
|
|
505
|
-
this.logger.debug(`[
|
|
773
|
+
this.logger.debug(`[VernLLM:${requestId}] error:\n${describeError(error)}`);
|
|
506
774
|
throw normalized;
|
|
507
775
|
}
|
|
508
776
|
}, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
509
777
|
}
|
|
778
|
+
/**
|
|
779
|
+
* Performs a single attempt: builds the request (translating `tools` to
|
|
780
|
+
* wire shape when present), dispatches it with a timeout, and shapes the
|
|
781
|
+
* response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
|
|
782
|
+
* set. Throws on an empty response (no text and no tool_calls) so the
|
|
783
|
+
* retry loop treats it like any other transient failure.
|
|
784
|
+
*/
|
|
785
|
+
async executeCall(params, requestId, attempt) {
|
|
786
|
+
const { useJson, model, request } = this.buildRequestPayload(params);
|
|
787
|
+
const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
|
|
788
|
+
const usage = this.extractUsage(response, requestId, model);
|
|
789
|
+
const rawContent = response.choices?.[0]?.message?.content;
|
|
790
|
+
const wireToolCalls = response.choices?.[0]?.message?.tool_calls;
|
|
791
|
+
return this.finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt);
|
|
792
|
+
}
|
|
793
|
+
/**
|
|
794
|
+
* Shapes a fully-arrived response (content and/or tool_calls, already
|
|
795
|
+
* extracted from the provider's payload) into `T` or a
|
|
796
|
+
* `CallWithToolsResult<T>`. Reused by the streaming path once it has
|
|
797
|
+
* buffered the full text/tool-call deltas, so there's no separate
|
|
798
|
+
* parsing/validation logic for streaming.
|
|
799
|
+
*
|
|
800
|
+
* Normalizes and reports usage failure on error itself, so every caller
|
|
801
|
+
* gets identical error handling without duplicating it.
|
|
802
|
+
*/
|
|
803
|
+
finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt) {
|
|
804
|
+
try {
|
|
805
|
+
const content = rawContent?.trim();
|
|
806
|
+
if (!content && !wireToolCalls?.length) throw new LLMError("Empty LLM response", "api");
|
|
807
|
+
this.logger.debug(`[VernLLM:${requestId}] output:\n${(content ?? `[${wireToolCalls?.length ?? 0} tool call(s)]`).slice(0, 800)}`);
|
|
808
|
+
if (wireToolCalls?.length) {
|
|
809
|
+
if (!params.tools) throw new LLMError("Provider returned tool_calls but no `tools` were sent with this call.", "api");
|
|
810
|
+
const toolCalls = parseWireToolCalls(wireToolCalls);
|
|
811
|
+
this.validateToolCallArguments(toolCalls, params.tools);
|
|
812
|
+
this.breaker?.recordSuccess();
|
|
813
|
+
this.reportUsage(usage);
|
|
814
|
+
return {
|
|
815
|
+
type: "tool_calls",
|
|
816
|
+
toolCalls,
|
|
817
|
+
...content ? { content } : {}
|
|
818
|
+
};
|
|
819
|
+
}
|
|
820
|
+
const textContent = content ?? "";
|
|
821
|
+
if (!useJson) {
|
|
822
|
+
this.breaker?.recordSuccess();
|
|
823
|
+
this.reportUsage(usage);
|
|
824
|
+
return params.tools ? {
|
|
825
|
+
type: "content",
|
|
826
|
+
content: textContent
|
|
827
|
+
} : textContent;
|
|
828
|
+
}
|
|
829
|
+
const result = this.parseAndValidate(textContent, params.schema);
|
|
830
|
+
this.breaker?.recordSuccess();
|
|
831
|
+
this.reportUsage(usage);
|
|
832
|
+
return params.tools ? {
|
|
833
|
+
type: "content",
|
|
834
|
+
content: result
|
|
835
|
+
} : result;
|
|
836
|
+
} catch (error) {
|
|
837
|
+
const normalized = normalizeError(error, params.signal);
|
|
838
|
+
if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt);
|
|
839
|
+
throw normalized;
|
|
840
|
+
}
|
|
841
|
+
}
|
|
842
|
+
/**
|
|
843
|
+
* Opens a stream for a single attempt: builds the request exactly like
|
|
844
|
+
* `executeCall`, then requires `createStream` on the client (a clear
|
|
845
|
+
* `validation` error if the adapter doesn't support it). The timeout
|
|
846
|
+
* wraps stream construction and the first `.next()` together, not just
|
|
847
|
+
* construction: calling an `async function*` returns an iterator
|
|
848
|
+
* synchronously without running its body until `.next()` is first
|
|
849
|
+
* invoked, so timing only construction would time an operation that's
|
|
850
|
+
* always instant, not the actual connection. Both are folded into a
|
|
851
|
+
* single `withTimeout` so the same abort signal reaches whatever the
|
|
852
|
+
* adapter's `createStream` uses internally for its first network
|
|
853
|
+
* round-trip.
|
|
854
|
+
*
|
|
855
|
+
* Circuit-breaker success is recorded once the stream fully completes,
|
|
856
|
+
* not on the first chunk arriving, so a connection that opens but then
|
|
857
|
+
* dies mid-stream isn't masked as a success (see `buildStreamResult`).
|
|
858
|
+
*/
|
|
859
|
+
async executeStreamCall(params, requestId, attempt) {
|
|
860
|
+
const { useJson, model, request } = this.buildRequestPayload(params);
|
|
861
|
+
const createStream = this.client.chat.completions.createStream;
|
|
862
|
+
if (!createStream) throw new LLMError("stream: true requires a client/adapter with createStream", "validation");
|
|
863
|
+
const streamController = new AbortController();
|
|
864
|
+
const combinedExternal = params.signal ? AbortSignal.any([params.signal, streamController.signal]) : streamController.signal;
|
|
865
|
+
const { iterator, first } = await withTimeout(async (attemptSignal) => {
|
|
866
|
+
const streamIterator = createStream(request, { signal: attemptSignal })[Symbol.asyncIterator]();
|
|
867
|
+
const firstResult = await streamIterator.next();
|
|
868
|
+
return {
|
|
869
|
+
iterator: streamIterator,
|
|
870
|
+
first: firstResult
|
|
871
|
+
};
|
|
872
|
+
}, this.timeoutMs, combinedExternal);
|
|
873
|
+
if (first.done) throw new LLMError("Empty LLM response", "api");
|
|
874
|
+
return this.buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController);
|
|
875
|
+
}
|
|
876
|
+
/**
|
|
877
|
+
* The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
|
|
878
|
+
* an async generator that yields translated `StreamChunk`s to the caller
|
|
879
|
+
* live, as they arrive, with no per-chunk timeout and no bound on total
|
|
880
|
+
* duration, and accumulates text/tool-call deltas internally so that
|
|
881
|
+
* `finalizeResponse` can produce `finalResult` once the stream completes.
|
|
882
|
+
*
|
|
883
|
+
* Two separate try/catches: the iteration loop's catch handles errors
|
|
884
|
+
* the transport itself throws, which aren't normalized yet, so that
|
|
885
|
+
* happens here along with the one `reportUsageFailure` call for them.
|
|
886
|
+
* The second catch, around `finalizeResponse`, does not re-normalize or
|
|
887
|
+
* re-report since `finalizeResponse` already does both internally.
|
|
888
|
+
* Circuit-breaker success is only recorded once the stream fully
|
|
889
|
+
* completes, not when the first chunk arrives, so a connection that
|
|
890
|
+
* opens and then dies mid-way still counts as a failure below instead
|
|
891
|
+
* of masking it.
|
|
892
|
+
*/
|
|
893
|
+
buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController) {
|
|
894
|
+
let resolveFinal;
|
|
895
|
+
let rejectFinal;
|
|
896
|
+
const finalResult = new Promise((resolve, reject) => {
|
|
897
|
+
resolveFinal = resolve;
|
|
898
|
+
rejectFinal = reject;
|
|
899
|
+
});
|
|
900
|
+
finalResult.catch(() => {});
|
|
901
|
+
const MAX_BUFFERED_CHUNKS = 1e4;
|
|
902
|
+
const buffered = [];
|
|
903
|
+
const pending = [];
|
|
904
|
+
let streamDone = false;
|
|
905
|
+
let streamError;
|
|
906
|
+
let hasLoggedEviction = false;
|
|
907
|
+
const push = (chunk) => {
|
|
908
|
+
const waiter = pending.shift();
|
|
909
|
+
if (waiter) {
|
|
910
|
+
waiter.resolve({
|
|
911
|
+
done: false,
|
|
912
|
+
value: chunk
|
|
913
|
+
});
|
|
914
|
+
return;
|
|
915
|
+
}
|
|
916
|
+
buffered.push(chunk);
|
|
917
|
+
if (buffered.length > MAX_BUFFERED_CHUNKS * 2) {
|
|
918
|
+
if (!hasLoggedEviction) {
|
|
919
|
+
hasLoggedEviction = true;
|
|
920
|
+
this.logger.debug(`[VernLLM] stream chunk buffer exceeded cap (${MAX_BUFFERED_CHUNKS}), evicting ${buffered.length - MAX_BUFFERED_CHUNKS} oldest chunk(s); buffered=${buffered.length}. The chunks iterable was never read (or fell far behind) for this stream.`);
|
|
921
|
+
}
|
|
922
|
+
buffered.splice(0, buffered.length - MAX_BUFFERED_CHUNKS);
|
|
923
|
+
}
|
|
924
|
+
};
|
|
925
|
+
const finish = () => {
|
|
926
|
+
streamDone = true;
|
|
927
|
+
for (const waiter of pending.splice(0)) waiter.resolve({
|
|
928
|
+
done: true,
|
|
929
|
+
value: void 0
|
|
930
|
+
});
|
|
931
|
+
};
|
|
932
|
+
const fail = (error) => {
|
|
933
|
+
streamDone = true;
|
|
934
|
+
streamError = error;
|
|
935
|
+
for (const waiter of pending.splice(0)) waiter.reject(error);
|
|
936
|
+
};
|
|
937
|
+
const chunks = { [Symbol.asyncIterator]() {
|
|
938
|
+
return { next() {
|
|
939
|
+
if (buffered.length) return Promise.resolve({
|
|
940
|
+
done: false,
|
|
941
|
+
value: buffered.shift()
|
|
942
|
+
});
|
|
943
|
+
if (streamDone) return streamError ? Promise.reject(streamError) : Promise.resolve({
|
|
944
|
+
done: true,
|
|
945
|
+
value: void 0
|
|
946
|
+
});
|
|
947
|
+
return new Promise((resolve, reject) => {
|
|
948
|
+
pending.push({
|
|
949
|
+
resolve,
|
|
950
|
+
reject
|
|
951
|
+
});
|
|
952
|
+
});
|
|
953
|
+
} };
|
|
954
|
+
} };
|
|
955
|
+
const toolCallAcc = new Map();
|
|
956
|
+
let textAcc = "";
|
|
957
|
+
let usage;
|
|
958
|
+
(async () => {
|
|
959
|
+
try {
|
|
960
|
+
let result = first;
|
|
961
|
+
while (!result.done) {
|
|
962
|
+
const wireChunk = result.value;
|
|
963
|
+
if (wireChunk.type === "ping") {} else if (wireChunk.type === "text-delta") {
|
|
964
|
+
textAcc += wireChunk.delta;
|
|
965
|
+
push({
|
|
966
|
+
type: "text-delta",
|
|
967
|
+
delta: wireChunk.delta
|
|
968
|
+
});
|
|
969
|
+
} else if (wireChunk.type === "tool_call_delta") {
|
|
970
|
+
const entry = toolCallAcc.get(wireChunk.index) ?? { args: "" };
|
|
971
|
+
entry.id ??= wireChunk.id;
|
|
972
|
+
entry.name ??= wireChunk.name;
|
|
973
|
+
entry.args += wireChunk.argumentsDelta ?? "";
|
|
974
|
+
toolCallAcc.set(wireChunk.index, entry);
|
|
975
|
+
push({
|
|
976
|
+
type: "tool_call_delta",
|
|
977
|
+
index: wireChunk.index,
|
|
978
|
+
id: wireChunk.id,
|
|
979
|
+
name: wireChunk.name,
|
|
980
|
+
argsDelta: wireChunk.argumentsDelta,
|
|
981
|
+
complete: wireChunk.complete
|
|
982
|
+
});
|
|
983
|
+
} else if (wireChunk.type === "usage") {
|
|
984
|
+
usage = {
|
|
985
|
+
promptTokens: wireChunk.usage.prompt_tokens ?? 0,
|
|
986
|
+
completionTokens: wireChunk.usage.completion_tokens ?? 0,
|
|
987
|
+
totalTokens: wireChunk.usage.total_tokens ?? 0,
|
|
988
|
+
requestId,
|
|
989
|
+
model
|
|
990
|
+
};
|
|
991
|
+
push({
|
|
992
|
+
type: "usage",
|
|
993
|
+
usage
|
|
994
|
+
});
|
|
995
|
+
}
|
|
996
|
+
result = await withChunkIdleTimeout(() => iterator.next(), params.chunkIdleTimeoutMs ?? this.chunkIdleTimeoutMs, () => streamController.abort(), this.logger);
|
|
997
|
+
}
|
|
998
|
+
} catch (error) {
|
|
999
|
+
try {
|
|
1000
|
+
await iterator.return?.();
|
|
1001
|
+
} catch {}
|
|
1002
|
+
streamController.abort();
|
|
1003
|
+
const normalized = normalizeError(error, params.signal);
|
|
1004
|
+
if (normalized.type === "timeout") this.breaker?.recordFailure();
|
|
1005
|
+
if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt, true);
|
|
1006
|
+
fail(normalized);
|
|
1007
|
+
rejectFinal(normalized);
|
|
1008
|
+
return;
|
|
1009
|
+
}
|
|
1010
|
+
finish();
|
|
1011
|
+
this.breaker?.recordSuccess();
|
|
1012
|
+
try {
|
|
1013
|
+
const wireToolCalls = toolCallAcc.size ? [...toolCallAcc.entries()].sort(([indexA], [indexB]) => indexA - indexB).map(([, entry]) => ({
|
|
1014
|
+
id: entry.id ?? "",
|
|
1015
|
+
type: "function",
|
|
1016
|
+
function: {
|
|
1017
|
+
name: entry.name ?? "",
|
|
1018
|
+
arguments: entry.args
|
|
1019
|
+
}
|
|
1020
|
+
})) : void 0;
|
|
1021
|
+
const finalized = this.finalizeResponse(textAcc, wireToolCalls, params, useJson, usage, requestId, attempt);
|
|
1022
|
+
resolveFinal(finalized);
|
|
1023
|
+
} catch (error) {
|
|
1024
|
+
rejectFinal(error);
|
|
1025
|
+
}
|
|
1026
|
+
})();
|
|
1027
|
+
return {
|
|
1028
|
+
chunks,
|
|
1029
|
+
finalResult
|
|
1030
|
+
};
|
|
1031
|
+
}
|
|
1032
|
+
/**
|
|
1033
|
+
* Checks every `ToolCall` against the `tools` that were offered, catching
|
|
1034
|
+
* a hallucinated tool name early instead of letting it reach the
|
|
1035
|
+
* application's dispatch table. Then runs each tool's `argumentsSchema`,
|
|
1036
|
+
* if present, throwing `LLMError('validation')` on failure.
|
|
1037
|
+
*/
|
|
1038
|
+
validateToolCallArguments(toolCalls, tools) {
|
|
1039
|
+
const knownNames = new Set(tools.map((t) => t.name));
|
|
1040
|
+
for (const call of toolCalls) {
|
|
1041
|
+
if (!knownNames.has(call.name)) throw new LLMError(`Model requested tool "${call.name}", which was not in the tools offered ([${[...knownNames].join(", ")}]).`, "api");
|
|
1042
|
+
const definition = tools.find((t) => t.name === call.name);
|
|
1043
|
+
if (!definition?.argumentsSchema) continue;
|
|
1044
|
+
const result = definition.argumentsSchema.safeParse(call.arguments);
|
|
1045
|
+
if (!result.success) throw new LLMError(`Arguments for tool call "${call.name}" failed validation`, "validation", void 0, result.error);
|
|
1046
|
+
}
|
|
1047
|
+
}
|
|
510
1048
|
/** Runs `fn`, retrying with backoff according to `shouldRetry`. */
|
|
511
1049
|
async retryWithBackoff(fn, requestId, signal) {
|
|
512
1050
|
let lastError;
|
|
513
1051
|
for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
|
|
514
1052
|
if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
|
|
515
|
-
return await fn();
|
|
1053
|
+
return await fn(attempt);
|
|
516
1054
|
} catch (error) {
|
|
517
1055
|
lastError = error;
|
|
518
1056
|
if (!this.shouldRetry(error, signal)) break;
|
|
@@ -520,60 +1058,73 @@ var VernLLM = class {
|
|
|
520
1058
|
throw lastError;
|
|
521
1059
|
}
|
|
522
1060
|
/**
|
|
523
|
-
* Performs a single attempt: builds the request, dispatches it with a
|
|
524
|
-
* timeout, and shapes the response. Throws on an empty response so the
|
|
525
|
-
* retry loop treats it like any other transient failure.
|
|
526
|
-
*/
|
|
527
|
-
async executeCall(params, requestId) {
|
|
528
|
-
const { useJson, model, request } = this.buildRequestPayload(params);
|
|
529
|
-
const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
|
|
530
|
-
const content = response.choices?.[0]?.message?.content?.trim();
|
|
531
|
-
if (!content) throw new LLMError("Empty LLM response", "api");
|
|
532
|
-
this.logger.debug(`[vern:${requestId}] output:\n${content.slice(0, 800)}`);
|
|
533
|
-
this.recordUsage(response, requestId, model);
|
|
534
|
-
if (!useJson) {
|
|
535
|
-
this.breaker?.recordSuccess();
|
|
536
|
-
return content;
|
|
537
|
-
}
|
|
538
|
-
const result = this.parseAndValidate(content, params.schema);
|
|
539
|
-
this.breaker?.recordSuccess();
|
|
540
|
-
return result;
|
|
541
|
-
}
|
|
542
|
-
/**
|
|
543
1061
|
* Validates `history` alternates user/assistant turns, since providers
|
|
544
1062
|
* like Anthropic/Gemini reject or mishandle consecutive same-role turns.
|
|
545
1063
|
*/
|
|
546
1064
|
validateHistory(history) {
|
|
547
|
-
let
|
|
1065
|
+
let previousTurn;
|
|
548
1066
|
for (const [index, turn] of history.entries()) {
|
|
549
|
-
if (turn.role
|
|
550
|
-
|
|
551
|
-
|
|
1067
|
+
if (turn.role === "tool") {
|
|
1068
|
+
if (previousTurn?.role !== "assistant" || !previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] is a "tool" turn, but must immediately follow an "assistant" turn that requested tools`, "validation");
|
|
1069
|
+
if (!turn.toolResults?.length) throw new LLMError(`history[${index}] is a "tool" turn but has no toolResults`, "validation");
|
|
1070
|
+
const requestedIds = new Set(previousTurn.toolCalls.map((tc) => tc.id));
|
|
1071
|
+
const resultIds = turn.toolResults.map((tr) => tr.toolCallId);
|
|
1072
|
+
const unknownIds = resultIds.filter((id) => !requestedIds.has(id));
|
|
1073
|
+
if (unknownIds.length) throw new LLMError(`history[${index}].toolResults references unknown toolCallId(s) [${unknownIds.join(", ")}]`, "validation");
|
|
1074
|
+
const seenIds = new Set();
|
|
1075
|
+
const duplicateIds = new Set();
|
|
1076
|
+
for (const id of resultIds) {
|
|
1077
|
+
if (seenIds.has(id)) duplicateIds.add(id);
|
|
1078
|
+
seenIds.add(id);
|
|
1079
|
+
}
|
|
1080
|
+
if (duplicateIds.size) throw new LLMError(`history[${index}].toolResults has duplicate toolCallId(s) [${[...duplicateIds].join(", ")}]`, "validation");
|
|
1081
|
+
const missingIds = [...requestedIds].filter((id) => !resultIds.includes(id));
|
|
1082
|
+
if (missingIds.length) throw new LLMError(`history[${index}] is missing toolResults for toolCallId(s) [${missingIds.join(", ")}]`, "validation");
|
|
1083
|
+
} else {
|
|
1084
|
+
if (turn.role === previousTurn?.role) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
|
|
1085
|
+
if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] follows an assistant tool request without tool results`, "validation");
|
|
1086
|
+
}
|
|
1087
|
+
previousTurn = turn;
|
|
552
1088
|
}
|
|
553
|
-
if (
|
|
1089
|
+
if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError("The last entry in history is an assistant tool request without tool results", "validation");
|
|
1090
|
+
if (previousTurn?.role === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn.", "validation");
|
|
554
1091
|
}
|
|
555
1092
|
/** Applies per-call defaults and shapes params into the client's request object. */
|
|
556
1093
|
buildRequestPayload(params) {
|
|
557
|
-
const { systemPrompt, userContent, history = [],
|
|
1094
|
+
const { systemPrompt, userContent, history = [], maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema, tools, toolChoice } = params;
|
|
1095
|
+
const temperature = params.temperature === void 0 ? this.defaultTemperature : params.temperature;
|
|
1096
|
+
if (tools && (jsonSchema || params.schema)) throw new LLMError("`tools` cannot be combined with `jsonSchema`/`schema`: on Anthropic and Bedrock, jsonSchema is implemented internally as a forced single-tool call, which would collide with real tools. Use one or the other.", "validation");
|
|
1097
|
+
if (tools && tools.length === 0) throw new LLMError("`tools` was an empty array. This is almost always a bug (e.g. a filtered tool list that ended up empty). An empty `tools` array still switches on tool-call mode (response shape, jsonMode default, wire format) with nothing for the model to call. Omit `tools` entirely for a normal call, or make sure the array is non-empty.", "validation");
|
|
1098
|
+
if (tools) {
|
|
1099
|
+
const seen = new Set();
|
|
1100
|
+
const duplicates = new Set();
|
|
1101
|
+
for (const tool of tools) {
|
|
1102
|
+
if (seen.has(tool.name)) duplicates.add(tool.name);
|
|
1103
|
+
seen.add(tool.name);
|
|
1104
|
+
}
|
|
1105
|
+
if (duplicates.size) throw new LLMError(`\`tools\` has duplicate name(s): [${[...duplicates].join(", ")}]. Tool names must be unique.`, "validation");
|
|
1106
|
+
}
|
|
1107
|
+
if (toolChoice && !tools) throw new LLMError("`toolChoice` was set without `tools`. There is nothing for it to choose between. Set `tools`, or remove `toolChoice`.", "validation");
|
|
1108
|
+
if (tools && typeof toolChoice === "object" && !tools.some((t) => t.name === toolChoice.name)) throw new LLMError(`toolChoice names "${toolChoice.name}", which is not in \`tools\` ([${tools.map((t) => t.name).join(", ")}]).`, "validation");
|
|
1109
|
+
const jsonMode = params.jsonMode ?? (tools ? false : true);
|
|
558
1110
|
const useJson = jsonMode || Boolean(jsonSchema);
|
|
559
1111
|
if (params.schema && !useJson) throw new LLMError("schema was provided but jsonMode: false disables JSON parsing, so nothing would validate it. Remove jsonMode: false, set jsonSchema, or remove schema.", "validation");
|
|
560
1112
|
const responseFormat = this.buildResponseFormat(jsonSchema, useJson);
|
|
561
1113
|
this.validateHistory(history);
|
|
562
1114
|
const request = {
|
|
563
1115
|
model,
|
|
564
|
-
temperature,
|
|
1116
|
+
...temperature !== null ? { temperature } : {},
|
|
565
1117
|
max_tokens: maxTokens,
|
|
566
1118
|
...responseFormat ? { response_format: responseFormat } : {},
|
|
567
1119
|
...reasoningEffort ? { reasoning_effort: reasoningEffort } : {},
|
|
1120
|
+
...tools ? { tools: toWireTools(tools) } : {},
|
|
1121
|
+
...tools ? { tool_choice: this.buildWireToolChoice(toolChoice) } : {},
|
|
568
1122
|
messages: [
|
|
569
1123
|
...systemPrompt ? [{
|
|
570
1124
|
role: "system",
|
|
571
1125
|
content: systemPrompt
|
|
572
1126
|
}] : [],
|
|
573
|
-
...history.
|
|
574
|
-
role: turn.role,
|
|
575
|
-
content: turn.content
|
|
576
|
-
})),
|
|
1127
|
+
...history.flatMap((turn) => this.turnToWireMessages(turn)),
|
|
577
1128
|
{
|
|
578
1129
|
role: "user",
|
|
579
1130
|
content: userContent
|
|
@@ -586,6 +1137,39 @@ var VernLLM = class {
|
|
|
586
1137
|
request
|
|
587
1138
|
};
|
|
588
1139
|
}
|
|
1140
|
+
/** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
|
|
1141
|
+
buildWireToolChoice(toolChoice) {
|
|
1142
|
+
if (!toolChoice || toolChoice === "auto") return "auto";
|
|
1143
|
+
if (toolChoice === "none" || toolChoice === "required") return toolChoice;
|
|
1144
|
+
return {
|
|
1145
|
+
type: "function",
|
|
1146
|
+
function: { name: toolChoice.name }
|
|
1147
|
+
};
|
|
1148
|
+
}
|
|
1149
|
+
/**
|
|
1150
|
+
* Expands one `ConversationTurn` into one or more wire messages. Plain
|
|
1151
|
+
* user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
|
|
1152
|
+
* to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
|
|
1153
|
+
* turn expands into one wire `tool` message per `toolResult`, since
|
|
1154
|
+
* OpenAI-shaped wire format wants one message per tool_call_id.
|
|
1155
|
+
*/
|
|
1156
|
+
turnToWireMessages(turn) {
|
|
1157
|
+
if (turn.role === "tool") return (turn.toolResults ?? []).map((tr) => ({
|
|
1158
|
+
role: "tool",
|
|
1159
|
+
tool_call_id: tr.toolCallId,
|
|
1160
|
+
content: typeof tr.content === "string" ? tr.content : JSON.stringify(tr.content ?? null),
|
|
1161
|
+
...tr.isError ? { is_error: true } : {}
|
|
1162
|
+
}));
|
|
1163
|
+
if (turn.role === "assistant" && turn.toolCalls?.length) return [{
|
|
1164
|
+
role: "assistant",
|
|
1165
|
+
...turn.content ? { content: turn.content } : {},
|
|
1166
|
+
tool_calls: toWireToolCalls(turn.toolCalls)
|
|
1167
|
+
}];
|
|
1168
|
+
return [{
|
|
1169
|
+
role: turn.role,
|
|
1170
|
+
content: turn.content ?? ""
|
|
1171
|
+
}];
|
|
1172
|
+
}
|
|
589
1173
|
/**
|
|
590
1174
|
* Chooses the response format: a provider-native `jsonSchema` takes
|
|
591
1175
|
* priority when supplied (constrains generation directly), otherwise
|
|
@@ -604,21 +1188,49 @@ var VernLLM = class {
|
|
|
604
1188
|
};
|
|
605
1189
|
return useJson ? { type: "json_object" } : void 0;
|
|
606
1190
|
}
|
|
607
|
-
/**
|
|
608
|
-
|
|
609
|
-
|
|
1191
|
+
/**
|
|
1192
|
+
* Pulls `TokenUsage` out of a raw response, if the provider reported it.
|
|
1193
|
+
* Extraction doesn't depend on what happens to the response afterward, so
|
|
1194
|
+
* a malformed body can still yield usage if the provider's usage block
|
|
1195
|
+
* itself came through intact.
|
|
1196
|
+
*/
|
|
1197
|
+
extractUsage(response, requestId, model) {
|
|
1198
|
+
if (!response.usage) return void 0;
|
|
1199
|
+
return {
|
|
1200
|
+
promptTokens: response.usage.prompt_tokens ?? 0,
|
|
1201
|
+
completionTokens: response.usage.completion_tokens ?? 0,
|
|
1202
|
+
totalTokens: response.usage.total_tokens ?? 0,
|
|
1203
|
+
requestId,
|
|
1204
|
+
model
|
|
1205
|
+
};
|
|
1206
|
+
}
|
|
1207
|
+
/** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
|
|
1208
|
+
reportUsage(usage) {
|
|
1209
|
+
if (!usage || !this.onUsage) return;
|
|
610
1210
|
try {
|
|
611
|
-
this.onUsage(
|
|
612
|
-
promptTokens: response.usage.prompt_tokens ?? 0,
|
|
613
|
-
completionTokens: response.usage.completion_tokens ?? 0,
|
|
614
|
-
totalTokens: response.usage.total_tokens ?? 0,
|
|
615
|
-
requestId,
|
|
616
|
-
model
|
|
617
|
-
});
|
|
1211
|
+
this.onUsage(usage);
|
|
618
1212
|
} catch (error) {
|
|
619
1213
|
this.logger.error("[VernLLM] onUsage failed", { message: error instanceof Error ? error.message : "unknown" });
|
|
620
1214
|
}
|
|
621
1215
|
}
|
|
1216
|
+
/**
|
|
1217
|
+
* Reports token usage spent on an attempt that then failed, so it isn't
|
|
1218
|
+
* dropped alongside the error. Covers any error thrown after usage
|
|
1219
|
+
* extraction, since all of them happen only after a response (real
|
|
1220
|
+
* spend) already arrived. Swallows and logs any error `onUsageFailure`
|
|
1221
|
+
* itself throws.
|
|
1222
|
+
*/
|
|
1223
|
+
reportUsageFailure(usage, error, attempt, terminal = false) {
|
|
1224
|
+
const displayTokens = usage.totalTokens || usage.promptTokens + usage.completionTokens;
|
|
1225
|
+
const attemptText = terminal ? "mid-stream failure (terminal, no further attempts)" : `attempt ${attempt + 1}/${this.maxRetries + 1}`;
|
|
1226
|
+
this.logger.warn(`[VernLLM:${usage.requestId}] usage failure, ${attemptText}: type=${error.type} tokens=${displayTokens}`);
|
|
1227
|
+
if (!this.onUsageFailure) return;
|
|
1228
|
+
try {
|
|
1229
|
+
this.onUsageFailure(usage, error);
|
|
1230
|
+
} catch (hookError) {
|
|
1231
|
+
this.logger.error("[VernLLM] onUsageFailure failed", { message: hookError instanceof Error ? hookError.message : "unknown" });
|
|
1232
|
+
}
|
|
1233
|
+
}
|
|
622
1234
|
/** Parses response content as JSON and validates it against `schema` when supplied. */
|
|
623
1235
|
parseAndValidate(content, schema) {
|
|
624
1236
|
let parsed;
|
|
@@ -642,7 +1254,7 @@ var VernLLM = class {
|
|
|
642
1254
|
async recoverDelay(requestId, attempt, error, signal) {
|
|
643
1255
|
const retryAfterMs = extractRetryAfterMs(error);
|
|
644
1256
|
const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
|
|
645
|
-
this.logger.warn(`[
|
|
1257
|
+
this.logger.warn(`[VernLLM:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
|
|
646
1258
|
await waitForRetry(delay, signal);
|
|
647
1259
|
}
|
|
648
1260
|
/** Decides whether a failed attempt is worth retrying. */
|
|
@@ -657,7 +1269,7 @@ var VernLLM = class {
|
|
|
657
1269
|
* supports deletion. Cache invalidation is the caller's responsibility;
|
|
658
1270
|
* only the application knows when cached data is stale.
|
|
659
1271
|
*
|
|
660
|
-
* @param key
|
|
1272
|
+
* @param key The raw cache key (resolved through the adapter's
|
|
661
1273
|
* `resolveKey`, if any, before deletion).
|
|
662
1274
|
*/
|
|
663
1275
|
async deleteCache(key) {
|
|
@@ -665,15 +1277,20 @@ var VernLLM = class {
|
|
|
665
1277
|
await this.cache.delete(await this.resolveCacheKey(key));
|
|
666
1278
|
}
|
|
667
1279
|
/**
|
|
668
|
-
*
|
|
669
|
-
* same `cacheKey` share a single in-flight call, avoiding cache
|
|
1280
|
+
* Internal cache primitive around caller-supplied logic. Concurrent misses
|
|
1281
|
+
* for the same `cacheKey` share a single in-flight call, avoiding cache
|
|
1282
|
+
* stampedes.
|
|
670
1283
|
*
|
|
671
|
-
*
|
|
1284
|
+
* Not part of the public API. Backs the public `cachedCall()`, which
|
|
1285
|
+
* always composes this with `call()` so cached results get the same
|
|
1286
|
+
* retry/timeout/circuit-breaker guarantees as any other LLM call.
|
|
1287
|
+
*
|
|
1288
|
+
* @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
|
|
672
1289
|
* miss, typically `() => this.call(...)`), and optional
|
|
673
|
-
* `reserveUsage`/`refundUsage`/`signal`. See `
|
|
1290
|
+
* `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
|
|
674
1291
|
* @returns The cached value on a hit, or the result of `fn()` on a miss.
|
|
675
1292
|
*/
|
|
676
|
-
async
|
|
1293
|
+
async runCached(params) {
|
|
677
1294
|
const resolvedKey = await this.resolveCacheKey(params.cacheKey);
|
|
678
1295
|
const resolvedParams = resolvedKey === params.cacheKey ? params : {
|
|
679
1296
|
...params,
|
|
@@ -704,24 +1321,113 @@ var VernLLM = class {
|
|
|
704
1321
|
}
|
|
705
1322
|
return result;
|
|
706
1323
|
}
|
|
707
|
-
/**
|
|
708
|
-
|
|
709
|
-
|
|
1324
|
+
/**
|
|
1325
|
+
* Streaming counterpart to `runCached`. Three cases:
|
|
1326
|
+
*
|
|
1327
|
+
* - Hit: no live generation to relay. Returns immediately with
|
|
1328
|
+
* `finalResult` resolved to the cached value and a one-shot `chunks`
|
|
1329
|
+
* replay built from it, so `for await (const c of chunks)` call sites
|
|
1330
|
+
* work identically on a hit or a miss. No usage hooks fire, since
|
|
1331
|
+
* nothing was actually spent.
|
|
1332
|
+
* - Miss, nothing else in flight for this key: delegates to
|
|
1333
|
+
* `registerStreamTrigger`, which opens the stream and relays its
|
|
1334
|
+
* `chunks` live.
|
|
1335
|
+
* - Miss, but another call for the same key is already in flight: this
|
|
1336
|
+
* call has no live chunks of its own to relay, so it's treated like a
|
|
1337
|
+
* delayed hit. `finalResult` shares the trigger's in-flight promise
|
|
1338
|
+
* (the same `this.inFlight` map non-streaming `runCached` uses, so
|
|
1339
|
+
* streaming and non-streaming `cachedCall`s for the same key coalesce
|
|
1340
|
+
* against each other too), and `chunks` is a one-shot replay built
|
|
1341
|
+
* once that promise resolves.
|
|
1342
|
+
*/
|
|
1343
|
+
async runCachedStream(params, hasTools) {
|
|
1344
|
+
const resolvedKey = await this.resolveCacheKey(params.cacheKey);
|
|
1345
|
+
const resolvedParams = resolvedKey === params.cacheKey ? params : {
|
|
1346
|
+
...params,
|
|
1347
|
+
cacheKey: resolvedKey
|
|
1348
|
+
};
|
|
1349
|
+
const cached = await this.cache.get(resolvedKey);
|
|
1350
|
+
if (cached.hit) {
|
|
1351
|
+
const value = cached.value;
|
|
1352
|
+
return {
|
|
1353
|
+
chunks: buildReplayChunks(value, hasTools),
|
|
1354
|
+
finalResult: Promise.resolve(value)
|
|
1355
|
+
};
|
|
1356
|
+
}
|
|
1357
|
+
const existing = this.inFlight.get(resolvedKey);
|
|
1358
|
+
if (existing) {
|
|
1359
|
+
const finalResult = withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
1360
|
+
return {
|
|
1361
|
+
chunks: buildReplayChunksFromPromise(finalResult, hasTools),
|
|
1362
|
+
finalResult
|
|
1363
|
+
};
|
|
1364
|
+
}
|
|
1365
|
+
return this.registerStreamTrigger(resolvedParams);
|
|
710
1366
|
}
|
|
711
1367
|
/**
|
|
712
|
-
*
|
|
713
|
-
*
|
|
714
|
-
*
|
|
1368
|
+
* Opens the shared stream for a cache miss and tracks its settled value
|
|
1369
|
+
* in `this.inFlight` until it resolves or rejects. Writes to the cache
|
|
1370
|
+
* on success only, matching `runAndCache`.
|
|
715
1371
|
*
|
|
716
|
-
*
|
|
717
|
-
*
|
|
718
|
-
*
|
|
1372
|
+
* Registers the in-flight promise synchronously, before anything async
|
|
1373
|
+
* runs, so a concurrent `cachedCall` for the same key always sees it in
|
|
1374
|
+
* time to join instead of triggering its own stream. Settlement is
|
|
1375
|
+
* wired onto the whole `withReservedUsageForStream` call rather than a
|
|
1376
|
+
* line inside its callback, so any failure point (reserving usage,
|
|
1377
|
+
* opening the stream, or the stream itself) reliably settles the
|
|
1378
|
+
* in-flight entry instead of leaving it stuck.
|
|
719
1379
|
*/
|
|
720
|
-
|
|
1380
|
+
registerStreamTrigger(params) {
|
|
1381
|
+
let resolveInFlight;
|
|
1382
|
+
let rejectInFlight;
|
|
1383
|
+
const inFlightResult = new Promise((resolve, reject) => {
|
|
1384
|
+
resolveInFlight = resolve;
|
|
1385
|
+
rejectInFlight = reject;
|
|
1386
|
+
});
|
|
1387
|
+
this.inFlight.set(params.cacheKey, inFlightResult);
|
|
1388
|
+
inFlightResult.catch(() => {}).finally(() => {
|
|
1389
|
+
this.inFlight.delete(params.cacheKey);
|
|
1390
|
+
});
|
|
1391
|
+
const streamPromise = withReservedUsageForStream(params, async () => {
|
|
1392
|
+
const opened = await params.openStream();
|
|
1393
|
+
const trackedResult = opened.finalResult.then(async (value) => {
|
|
1394
|
+
try {
|
|
1395
|
+
await this.cache.set(params.cacheKey, value, params.ttl);
|
|
1396
|
+
} catch (error) {
|
|
1397
|
+
this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
|
|
1398
|
+
}
|
|
1399
|
+
return value;
|
|
1400
|
+
}, (error) => {
|
|
1401
|
+
throw error;
|
|
1402
|
+
});
|
|
1403
|
+
return {
|
|
1404
|
+
chunks: opened.chunks,
|
|
1405
|
+
finalResult: trackedResult
|
|
1406
|
+
};
|
|
1407
|
+
}, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
|
|
1408
|
+
streamPromise.then((opened) => {
|
|
1409
|
+
opened.finalResult.then(resolveInFlight, rejectInFlight);
|
|
1410
|
+
}, (error) => {
|
|
1411
|
+
rejectInFlight(error);
|
|
1412
|
+
});
|
|
1413
|
+
return streamPromise;
|
|
1414
|
+
}
|
|
1415
|
+
/** Logs a failed refundUsage attempt via the configured logger. */
|
|
1416
|
+
logRefundError(logMessage, error) {
|
|
1417
|
+
this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
|
|
1418
|
+
}
|
|
1419
|
+
async cachedCall(params) {
|
|
721
1420
|
const { call: callParams,...cacheParams } = params;
|
|
722
|
-
const { reserveUsage
|
|
723
|
-
if (
|
|
724
|
-
|
|
1421
|
+
const { reserveUsage, refundUsage,...restCallParams } = callParams;
|
|
1422
|
+
if (reserveUsage || refundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedCall; set them at the top level instead.");
|
|
1423
|
+
if (restCallParams.stream) {
|
|
1424
|
+
const streamParams = restCallParams;
|
|
1425
|
+
return this.runCachedStream({
|
|
1426
|
+
...cacheParams,
|
|
1427
|
+
openStream: () => this.call(streamParams)
|
|
1428
|
+
}, Boolean(restCallParams.tools));
|
|
1429
|
+
}
|
|
1430
|
+
return this.runCached({
|
|
725
1431
|
...cacheParams,
|
|
726
1432
|
fn: () => this.call(restCallParams)
|
|
727
1433
|
});
|
|
@@ -735,6 +1441,108 @@ var VernLLM = class {
|
|
|
735
1441
|
}
|
|
736
1442
|
};
|
|
737
1443
|
|
|
1444
|
+
//#endregion
|
|
1445
|
+
//#region src/internal/sse.ts
|
|
1446
|
+
/**
|
|
1447
|
+
* Parses a Server-Sent-Events byte/text stream into the JSON payload of
|
|
1448
|
+
* each `data:` frame, in arrival order. Generic over transport: works with
|
|
1449
|
+
* anything that hands back progressively-arriving `Uint8Array` or `string`
|
|
1450
|
+
* chunks via async iteration: native `fetch`'s `response.body` (wrapped
|
|
1451
|
+
* to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
|
|
1452
|
+
* Node `Readable` (already async-iterable, no wrapping needed), etc, so
|
|
1453
|
+
* this framing layer doesn't care which transport produced the bytes.
|
|
1454
|
+
*
|
|
1455
|
+
* Follows the SSE spec's frame-delimiting rules closely enough for LLM
|
|
1456
|
+
* streaming responses: frames are separated by a blank line, each frame
|
|
1457
|
+
* may carry one or more `data:` lines (joined with `\n` per spec when
|
|
1458
|
+
* there's more than one), `:`-prefixed lines are comments and ignored, and
|
|
1459
|
+
* other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
|
|
1460
|
+
* only needs the payload. A frame whose data is exactly `[DONE]` (the
|
|
1461
|
+
* sentinel several providers, notably OpenAI, send to mark stream end)
|
|
1462
|
+
* ends iteration without yielding it.
|
|
1463
|
+
*
|
|
1464
|
+
* Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
|
|
1465
|
+
* to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
|
|
1466
|
+
* alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
|
|
1467
|
+
* stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
|
|
1468
|
+
* lines.
|
|
1469
|
+
*
|
|
1470
|
+
* Malformed JSON in a frame throws `LLMError('parse')`, consistent with
|
|
1471
|
+
* how malformed JSON is handled elsewhere in VernLLM.
|
|
1472
|
+
*/
|
|
1473
|
+
async function* parseSseStream(source) {
|
|
1474
|
+
const decoder = new TextDecoder("utf-8", { fatal: true });
|
|
1475
|
+
let buffer = "";
|
|
1476
|
+
for await (const chunk of source) {
|
|
1477
|
+
let text;
|
|
1478
|
+
try {
|
|
1479
|
+
text = typeof chunk === "string" ? chunk : decoder.decode(chunk, { stream: true });
|
|
1480
|
+
} catch (cause) {
|
|
1481
|
+
throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
|
|
1482
|
+
}
|
|
1483
|
+
buffer = (buffer + text).replace(/\r\n/g, "\n").replace(/\r(?!$)/g, "\n");
|
|
1484
|
+
let boundary$1 = buffer.indexOf("\n\n");
|
|
1485
|
+
while (boundary$1 !== -1) {
|
|
1486
|
+
const frame = buffer.slice(0, boundary$1);
|
|
1487
|
+
buffer = buffer.slice(boundary$1 + 2);
|
|
1488
|
+
const event = parseSseFrame(frame);
|
|
1489
|
+
if (event === DONE) return;
|
|
1490
|
+
if (event !== NO_DATA) yield event;
|
|
1491
|
+
boundary$1 = buffer.indexOf("\n\n");
|
|
1492
|
+
}
|
|
1493
|
+
}
|
|
1494
|
+
try {
|
|
1495
|
+
buffer += decoder.decode();
|
|
1496
|
+
} catch (cause) {
|
|
1497
|
+
throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
|
|
1498
|
+
}
|
|
1499
|
+
buffer = buffer.replace(/\r$/, "\n");
|
|
1500
|
+
let boundary = buffer.indexOf("\n\n");
|
|
1501
|
+
while (boundary !== -1) {
|
|
1502
|
+
const frame = buffer.slice(0, boundary);
|
|
1503
|
+
buffer = buffer.slice(boundary + 2);
|
|
1504
|
+
const event = parseSseFrame(frame);
|
|
1505
|
+
if (event === DONE) return;
|
|
1506
|
+
if (event !== NO_DATA) yield event;
|
|
1507
|
+
boundary = buffer.indexOf("\n\n");
|
|
1508
|
+
}
|
|
1509
|
+
const trailing = buffer.trim();
|
|
1510
|
+
if (trailing) {
|
|
1511
|
+
const event = parseSseFrame(trailing);
|
|
1512
|
+
if (event !== DONE && event !== NO_DATA) yield event;
|
|
1513
|
+
}
|
|
1514
|
+
}
|
|
1515
|
+
const DONE = Symbol("sse-stream-done");
|
|
1516
|
+
const NO_DATA = Symbol("sse-frame-no-data");
|
|
1517
|
+
/**
|
|
1518
|
+
* Sentinel yielded by `parseSseStream` for a comment-only frame (no
|
|
1519
|
+
* `data:` payload), the mechanism providers use for SSE keep-alive
|
|
1520
|
+
* pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
|
|
1521
|
+
* alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
|
|
1522
|
+
*/
|
|
1523
|
+
const SSE_PING = Symbol("sse-frame-ping");
|
|
1524
|
+
/** Extracts and JSON-parses the `data:` payload of one SSE frame (the text between two blank lines). */
|
|
1525
|
+
function parseSseFrame(frame) {
|
|
1526
|
+
const dataLines = [];
|
|
1527
|
+
let sawComment = false;
|
|
1528
|
+
for (const line of frame.split("\n")) {
|
|
1529
|
+
if (line.startsWith(":")) {
|
|
1530
|
+
sawComment = true;
|
|
1531
|
+
continue;
|
|
1532
|
+
}
|
|
1533
|
+
if (!line.startsWith("data:")) continue;
|
|
1534
|
+
dataLines.push(line.startsWith("data: ") ? line.slice(6) : line.slice(5));
|
|
1535
|
+
}
|
|
1536
|
+
if (!dataLines.length) return sawComment ? SSE_PING : NO_DATA;
|
|
1537
|
+
const data = dataLines.join("\n");
|
|
1538
|
+
if (data === "[DONE]") return DONE;
|
|
1539
|
+
try {
|
|
1540
|
+
return JSON.parse(data);
|
|
1541
|
+
} catch (cause) {
|
|
1542
|
+
throw new LLMError(`Invalid JSON in SSE frame: ${data.slice(0, 200)}`, "parse", void 0, void 0, cause);
|
|
1543
|
+
}
|
|
1544
|
+
}
|
|
1545
|
+
|
|
738
1546
|
//#endregion
|
|
739
1547
|
//#region src/internal/imageFormat.ts
|
|
740
1548
|
/**
|
|
@@ -782,6 +1590,93 @@ function toAnthropicContent(blocks) {
|
|
|
782
1590
|
});
|
|
783
1591
|
}
|
|
784
1592
|
/**
|
|
1593
|
+
* Asserts a caller-supplied JSON Schema is an object schema before it's
|
|
1594
|
+
* used as Anthropic's `Tool.input_schema`, which (like every other
|
|
1595
|
+
* provider's function-calling API) requires `type: 'object'`. VernLLM's own
|
|
1596
|
+
* public `tools`/`jsonSchema` APIs accept freeform `Record<string,
|
|
1597
|
+
* unknown>` JSON Schema, so nothing upstream guarantees this at compile
|
|
1598
|
+
* time; this is the runtime check that stands in for that, so a schema
|
|
1599
|
+
* missing (or mistyping) `type: 'object'` fails loudly and immediately
|
|
1600
|
+
* instead of being silently forwarded to Anthropic malformed.
|
|
1601
|
+
*/
|
|
1602
|
+
function assertObjectSchema(schema, toolName) {
|
|
1603
|
+
if (schema.type !== "object") throw new LLMError(`Tool "${toolName}"'s schema must have "type": "object" (Anthropic requires object-shaped tool parameters).`, "validation");
|
|
1604
|
+
return schema;
|
|
1605
|
+
}
|
|
1606
|
+
/**
|
|
1607
|
+
* Translates VernLLM's OpenAI-shaped wire `tool_choice` into Anthropic's
|
|
1608
|
+
* `{ type: 'auto' | 'any' | 'none' | 'tool', name? }` shape. `'required'`
|
|
1609
|
+
* maps to `'any'` (Anthropic's "must call some tool" equivalent).
|
|
1610
|
+
*/
|
|
1611
|
+
function toAnthropicToolChoice(toolChoice) {
|
|
1612
|
+
if (!toolChoice || toolChoice === "auto") return { type: "auto" };
|
|
1613
|
+
if (toolChoice === "none") return { type: "none" };
|
|
1614
|
+
if (toolChoice === "required") return { type: "any" };
|
|
1615
|
+
return {
|
|
1616
|
+
type: "tool",
|
|
1617
|
+
name: toolChoice.function.name
|
|
1618
|
+
};
|
|
1619
|
+
}
|
|
1620
|
+
/**
|
|
1621
|
+
* Builds the Anthropic-shaped request body from VernLLM's wire params,
|
|
1622
|
+
* shared between `create` and `createStream` so both go through identical
|
|
1623
|
+
* translation (system prompt, message shaping, and the jsonSchema →
|
|
1624
|
+
* forced-single-tool mapping all happen exactly once, not once per entry
|
|
1625
|
+
* point).
|
|
1626
|
+
*
|
|
1627
|
+
* Returns `toolName` alongside the body: when set, the model was forced to
|
|
1628
|
+
* call a single synthetic tool standing in for `jsonSchema` output, and
|
|
1629
|
+
* both `create` and `createStream` need to know this so they can unwrap
|
|
1630
|
+
* that tool call back into plain text content instead of treating it like
|
|
1631
|
+
* a real tool call.
|
|
1632
|
+
*/
|
|
1633
|
+
function buildAnthropicRequestBody(params) {
|
|
1634
|
+
const systemMessage = params.messages.find((m) => m.role === "system");
|
|
1635
|
+
const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
|
|
1636
|
+
const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name.trim() : void 0;
|
|
1637
|
+
if (params.response_format?.type === "json_schema" && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
|
|
1638
|
+
let jsonInstruction;
|
|
1639
|
+
let tools;
|
|
1640
|
+
let toolChoice;
|
|
1641
|
+
if (params.response_format?.type === "json_schema" && toolName) {
|
|
1642
|
+
const { schema, description, strict } = params.response_format.json_schema;
|
|
1643
|
+
tools = [{
|
|
1644
|
+
name: toolName,
|
|
1645
|
+
description,
|
|
1646
|
+
input_schema: assertObjectSchema(schema, toolName),
|
|
1647
|
+
strict
|
|
1648
|
+
}];
|
|
1649
|
+
toolChoice = {
|
|
1650
|
+
type: "tool",
|
|
1651
|
+
name: toolName
|
|
1652
|
+
};
|
|
1653
|
+
} else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
|
|
1654
|
+
else if (params.tools?.length) {
|
|
1655
|
+
tools = params.tools.map((t) => ({
|
|
1656
|
+
name: t.function.name,
|
|
1657
|
+
description: t.function.description,
|
|
1658
|
+
input_schema: assertObjectSchema(t.function.parameters, t.function.name)
|
|
1659
|
+
}));
|
|
1660
|
+
toolChoice = toAnthropicToolChoice(params.tool_choice);
|
|
1661
|
+
}
|
|
1662
|
+
const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
|
|
1663
|
+
const body = {
|
|
1664
|
+
model: params.model,
|
|
1665
|
+
max_tokens: params.max_tokens,
|
|
1666
|
+
...params.temperature !== void 0 ? { temperature: params.temperature } : {},
|
|
1667
|
+
system: system || void 0,
|
|
1668
|
+
messages: mergeConsecutiveToolResults$1(conversationMessages.map((m) => toAnthropicMessage(m))),
|
|
1669
|
+
...tools ? {
|
|
1670
|
+
tools,
|
|
1671
|
+
tool_choice: toolChoice
|
|
1672
|
+
} : {}
|
|
1673
|
+
};
|
|
1674
|
+
return {
|
|
1675
|
+
body,
|
|
1676
|
+
toolName
|
|
1677
|
+
};
|
|
1678
|
+
}
|
|
1679
|
+
/**
|
|
785
1680
|
* Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
|
|
786
1681
|
* interface VernLLM uses for OpenAI/Groq.
|
|
787
1682
|
*
|
|
@@ -796,53 +1691,161 @@ function toAnthropicContent(blocks) {
|
|
|
796
1691
|
* generation against.
|
|
797
1692
|
*/
|
|
798
1693
|
function fromAnthropic(anthropicClient) {
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
1694
|
+
const rawMessagesCreate = anthropicClient.messages.create.bind(anthropicClient.messages);
|
|
1695
|
+
return { chat: { completions: {
|
|
1696
|
+
async create(params, options) {
|
|
1697
|
+
const { body, toolName } = buildAnthropicRequestBody(params);
|
|
1698
|
+
const response = await anthropicClient.messages.create(body, options);
|
|
1699
|
+
let text;
|
|
1700
|
+
let wireToolCalls;
|
|
1701
|
+
if (toolName) {
|
|
1702
|
+
const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
|
|
1703
|
+
if (!toolUse) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
|
|
1704
|
+
if (!toolUse.input || typeof toolUse.input !== "object" || Array.isArray(toolUse.input)) throw new LLMError(`Anthropic returned invalid structured output for tool "${toolName}". Expected an object.`, "validation");
|
|
1705
|
+
text = JSON.stringify(toolUse.input);
|
|
1706
|
+
} else {
|
|
1707
|
+
text = response.content.filter((block) => block.type === "text").map((block) => block.text ?? "").join("");
|
|
1708
|
+
const toolUses = response.content.filter((block) => block.type === "tool_use");
|
|
1709
|
+
if (toolUses.length) wireToolCalls = toolUses.map((block) => ({
|
|
1710
|
+
id: block.id,
|
|
1711
|
+
type: "function",
|
|
1712
|
+
function: {
|
|
1713
|
+
name: block.name,
|
|
1714
|
+
arguments: JSON.stringify(block.input ?? {})
|
|
1715
|
+
}
|
|
1716
|
+
}));
|
|
1717
|
+
}
|
|
1718
|
+
return {
|
|
1719
|
+
choices: [{ message: {
|
|
1720
|
+
content: text,
|
|
1721
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
1722
|
+
} }],
|
|
1723
|
+
usage: {
|
|
1724
|
+
prompt_tokens: response.usage?.input_tokens,
|
|
1725
|
+
completion_tokens: response.usage?.output_tokens,
|
|
1726
|
+
total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
|
|
829
1727
|
}
|
|
830
|
-
}
|
|
831
|
-
},
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
const
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
1728
|
+
};
|
|
1729
|
+
},
|
|
1730
|
+
async *createStream(params, options) {
|
|
1731
|
+
const { body, toolName } = buildAnthropicRequestBody(params);
|
|
1732
|
+
const stream = await rawMessagesCreate({
|
|
1733
|
+
...body,
|
|
1734
|
+
stream: true
|
|
1735
|
+
}, options);
|
|
1736
|
+
const blockKinds = new Map();
|
|
1737
|
+
let inputTokens = 0;
|
|
1738
|
+
let sawJsonTool = false;
|
|
1739
|
+
for await (const event of stream) if (event.type === "message_start") inputTokens = event.message.usage?.input_tokens ?? 0;
|
|
1740
|
+
else if (event.type === "content_block_start") if (event.content_block.type === "tool_use") {
|
|
1741
|
+
const kind = event.content_block.name === toolName ? "json-tool" : "tool_use";
|
|
1742
|
+
blockKinds.set(event.index, kind);
|
|
1743
|
+
if (kind === "json-tool") sawJsonTool = true;
|
|
1744
|
+
else if (!toolName) yield {
|
|
1745
|
+
type: "tool_call_delta",
|
|
1746
|
+
index: event.index,
|
|
1747
|
+
id: event.content_block.id,
|
|
1748
|
+
name: event.content_block.name
|
|
1749
|
+
};
|
|
1750
|
+
} else blockKinds.set(event.index, "text");
|
|
1751
|
+
else if (event.type === "content_block_delta") {
|
|
1752
|
+
if (event.delta.type === "text_delta") {
|
|
1753
|
+
if (!toolName) yield {
|
|
1754
|
+
type: "text-delta",
|
|
1755
|
+
delta: event.delta.text
|
|
1756
|
+
};
|
|
1757
|
+
} else if (event.delta.type === "input_json_delta") {
|
|
1758
|
+
const kind = blockKinds.get(event.index);
|
|
1759
|
+
if (kind === "json-tool") yield {
|
|
1760
|
+
type: "text-delta",
|
|
1761
|
+
delta: event.delta.partial_json
|
|
1762
|
+
};
|
|
1763
|
+
else if (!toolName) yield {
|
|
1764
|
+
type: "tool_call_delta",
|
|
1765
|
+
index: event.index,
|
|
1766
|
+
argumentsDelta: event.delta.partial_json
|
|
1767
|
+
};
|
|
1768
|
+
}
|
|
1769
|
+
} else if (event.type === "message_delta") {
|
|
1770
|
+
const outputTokens = event.usage?.output_tokens ?? 0;
|
|
1771
|
+
yield {
|
|
1772
|
+
type: "usage",
|
|
1773
|
+
usage: {
|
|
1774
|
+
prompt_tokens: inputTokens,
|
|
1775
|
+
completion_tokens: outputTokens,
|
|
1776
|
+
total_tokens: inputTokens + outputTokens
|
|
1777
|
+
}
|
|
1778
|
+
};
|
|
1779
|
+
} else if (event.type === "ping") yield { type: "ping" };
|
|
1780
|
+
if (toolName && !sawJsonTool) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
|
|
1781
|
+
}
|
|
1782
|
+
} } };
|
|
1783
|
+
}
|
|
1784
|
+
/**
|
|
1785
|
+
* Anthropic requires strict role alternation, so the per-wire-message
|
|
1786
|
+
* mapping above (one `{role:'user', content:[tool_result]}` per VernLLM
|
|
1787
|
+
* wire tool message) needs merging back together when an assistant turn
|
|
1788
|
+
* requested more than one tool: multiple consecutive user turns would
|
|
1789
|
+
* violate that alternation, and Anthropic's API rejects it outright. This
|
|
1790
|
+
* merges any run of tool-result-only user messages into one, with all
|
|
1791
|
+
* their tool_result blocks combined, the shape Anthropic expects for "here
|
|
1792
|
+
* are the results of everything you just asked for."
|
|
1793
|
+
*/
|
|
1794
|
+
function mergeConsecutiveToolResults$1(messages) {
|
|
1795
|
+
const isToolResultOnly = (m) => m.role === "user" && Array.isArray(m.content) && m.content.length > 0 && m.content.every((b) => b.type === "tool_result");
|
|
1796
|
+
const merged = [];
|
|
1797
|
+
for (const m of messages) {
|
|
1798
|
+
const prev = merged.at(-1);
|
|
1799
|
+
if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
|
|
1800
|
+
else merged.push(m);
|
|
1801
|
+
}
|
|
1802
|
+
return merged;
|
|
1803
|
+
}
|
|
1804
|
+
/**
|
|
1805
|
+
* Translates one VernLLM wire message (OpenAI-shaped: plain user/assistant
|
|
1806
|
+
* turns, an assistant turn with `tool_calls`, or a `tool` turn) into
|
|
1807
|
+
* Anthropic's `{ role: 'user' | 'assistant', content }` shape.
|
|
1808
|
+
*/
|
|
1809
|
+
function toAnthropicMessage(m) {
|
|
1810
|
+
if (m.role === "tool") return {
|
|
1811
|
+
role: "user",
|
|
1812
|
+
content: [{
|
|
1813
|
+
type: "tool_result",
|
|
1814
|
+
tool_use_id: m.tool_call_id,
|
|
1815
|
+
content: m.content,
|
|
1816
|
+
...m.is_error ? { is_error: true } : {}
|
|
1817
|
+
}]
|
|
1818
|
+
};
|
|
1819
|
+
if (m.role === "assistant" && m.tool_calls?.length) {
|
|
1820
|
+
const blocks = [];
|
|
1821
|
+
if (m.content) blocks.push({
|
|
1822
|
+
type: "text",
|
|
1823
|
+
text: m.content
|
|
1824
|
+
});
|
|
1825
|
+
for (const tc of m.tool_calls) {
|
|
1826
|
+
let input;
|
|
1827
|
+
try {
|
|
1828
|
+
input = tc.function.arguments.trim() ? JSON.parse(tc.function.arguments) : {};
|
|
1829
|
+
} catch (cause) {
|
|
1830
|
+
throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
|
|
843
1831
|
}
|
|
1832
|
+
if (input === null || Array.isArray(input) || typeof input !== "object") throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) arguments must be a JSON object.`, "validation");
|
|
1833
|
+
blocks.push({
|
|
1834
|
+
type: "tool_use",
|
|
1835
|
+
id: tc.id,
|
|
1836
|
+
name: tc.function.name,
|
|
1837
|
+
input
|
|
1838
|
+
});
|
|
1839
|
+
}
|
|
1840
|
+
return {
|
|
1841
|
+
role: "assistant",
|
|
1842
|
+
content: blocks
|
|
844
1843
|
};
|
|
845
|
-
}
|
|
1844
|
+
}
|
|
1845
|
+
return {
|
|
1846
|
+
role: m.role,
|
|
1847
|
+
content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content ?? ""
|
|
1848
|
+
};
|
|
846
1849
|
}
|
|
847
1850
|
|
|
848
1851
|
//#endregion
|
|
@@ -859,6 +1862,122 @@ function toGeminiParts(blocks) {
|
|
|
859
1862
|
data: block.data
|
|
860
1863
|
} } : { text: block.text });
|
|
861
1864
|
}
|
|
1865
|
+
/** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Gemini's `functionCallingConfig`. */
|
|
1866
|
+
function toGeminiToolConfig(toolChoice) {
|
|
1867
|
+
if (!toolChoice || toolChoice === "auto") return { functionCallingConfig: { mode: "AUTO" } };
|
|
1868
|
+
if (toolChoice === "none") return { functionCallingConfig: { mode: "NONE" } };
|
|
1869
|
+
if (toolChoice === "required") return { functionCallingConfig: { mode: "ANY" } };
|
|
1870
|
+
return { functionCallingConfig: {
|
|
1871
|
+
mode: "ANY",
|
|
1872
|
+
allowedFunctionNames: [toolChoice.function.name]
|
|
1873
|
+
} };
|
|
1874
|
+
}
|
|
1875
|
+
/**
|
|
1876
|
+
* Translates one VernLLM wire message into a Gemini `contents` entry.
|
|
1877
|
+
* Gemini has no separate 'tool' role: a prior assistant tool request
|
|
1878
|
+
* becomes a `'model'` turn with `functionCall` parts, and its result
|
|
1879
|
+
* becomes a `'user'` turn with `functionResponse` parts.
|
|
1880
|
+
*/
|
|
1881
|
+
function toGeminiContent(m) {
|
|
1882
|
+
if (m.role === "tool") return {
|
|
1883
|
+
role: "user",
|
|
1884
|
+
parts: [{ functionResponse: {
|
|
1885
|
+
name: m.tool_call_id,
|
|
1886
|
+
response: parseToolResult(m.content)
|
|
1887
|
+
} }]
|
|
1888
|
+
};
|
|
1889
|
+
if (m.role === "assistant" && m.tool_calls?.length) {
|
|
1890
|
+
const parts = [];
|
|
1891
|
+
if (typeof m.content === "string" && m.content) parts.push({ text: m.content });
|
|
1892
|
+
parts.push(...m.tool_calls.map((tc) => ({ functionCall: {
|
|
1893
|
+
name: tc.function.name,
|
|
1894
|
+
args: parseToolArguments(tc.function.arguments, tc.function.name)
|
|
1895
|
+
} })));
|
|
1896
|
+
return {
|
|
1897
|
+
role: "model",
|
|
1898
|
+
parts
|
|
1899
|
+
};
|
|
1900
|
+
}
|
|
1901
|
+
return {
|
|
1902
|
+
role: m.role === "assistant" ? "model" : "user",
|
|
1903
|
+
parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content ?? "" }]
|
|
1904
|
+
};
|
|
1905
|
+
}
|
|
1906
|
+
function parseToolArguments(text, toolName) {
|
|
1907
|
+
let parsed;
|
|
1908
|
+
try {
|
|
1909
|
+
parsed = text.trim() ? JSON.parse(text) : {};
|
|
1910
|
+
} catch (cause) {
|
|
1911
|
+
throw new LLMError(`Tool call "${toolName}" arguments are not valid JSON.`, "validation", void 0, void 0, cause);
|
|
1912
|
+
}
|
|
1913
|
+
if (!parsed || Array.isArray(parsed) || typeof parsed !== "object") throw new LLMError(`Tool call "${toolName}" arguments must be a JSON object.`, "validation");
|
|
1914
|
+
return parsed;
|
|
1915
|
+
}
|
|
1916
|
+
function parseToolResult(text) {
|
|
1917
|
+
try {
|
|
1918
|
+
return text.trim() ? JSON.parse(text) : "";
|
|
1919
|
+
} catch {
|
|
1920
|
+
return text;
|
|
1921
|
+
}
|
|
1922
|
+
}
|
|
1923
|
+
/**
|
|
1924
|
+
* Gemini expects the results of everything the model asked for in one turn
|
|
1925
|
+
* to arrive together as multiple `functionResponse` parts on a single
|
|
1926
|
+
* `'user'` entry, not as separate consecutive `'user'` entries. The
|
|
1927
|
+
* per-wire-message mapping above produces one `'user'` entry per VernLLM
|
|
1928
|
+
* wire tool message, so when an assistant turn requested more than one
|
|
1929
|
+
* tool, this merges the resulting run of functionResponse-only `'user'`
|
|
1930
|
+
* entries back into one.
|
|
1931
|
+
*/
|
|
1932
|
+
function mergeConsecutiveFunctionResponses(contents) {
|
|
1933
|
+
const isFunctionResponseOnly = (c) => c.role === "user" && c.parts.length > 0 && c.parts.every((p) => "functionResponse" in p);
|
|
1934
|
+
const merged = [];
|
|
1935
|
+
for (const c of contents) {
|
|
1936
|
+
const prev = merged.at(-1);
|
|
1937
|
+
if (isFunctionResponseOnly(c) && prev && isFunctionResponseOnly(prev)) prev.parts.push(...c.parts);
|
|
1938
|
+
else merged.push(c);
|
|
1939
|
+
}
|
|
1940
|
+
return merged;
|
|
1941
|
+
}
|
|
1942
|
+
/**
|
|
1943
|
+
* Builds the Gemini-shaped request from VernLLM's wire params, shared
|
|
1944
|
+
* between `create` and `createStream` so both go through identical
|
|
1945
|
+
* translation (contents shaping, `responseSchema`/`responseMimeType`
|
|
1946
|
+
* mapping, and tool/toolConfig translation all happen exactly once).
|
|
1947
|
+
* `abortSignal` is folded into `config` by the caller (`create`/
|
|
1948
|
+
* `createStream`), once the request options are available.
|
|
1949
|
+
*/
|
|
1950
|
+
function buildGeminiRequest(params) {
|
|
1951
|
+
const systemMessage = params.messages.find((m) => m.role === "system");
|
|
1952
|
+
const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
|
|
1953
|
+
const wantsJson = Boolean(params.response_format);
|
|
1954
|
+
const config = {
|
|
1955
|
+
...params.temperature !== void 0 ? { temperature: params.temperature } : {},
|
|
1956
|
+
maxOutputTokens: params.max_tokens,
|
|
1957
|
+
...systemMessage ? { systemInstruction: { parts: [{ text: systemMessage.content }] } } : {}
|
|
1958
|
+
};
|
|
1959
|
+
if (wantsJson) config.responseMimeType = "application/json";
|
|
1960
|
+
if (params.response_format?.type === "json_schema") {
|
|
1961
|
+
const { schema, description } = params.response_format.json_schema;
|
|
1962
|
+
config.responseSchema = {
|
|
1963
|
+
...schema,
|
|
1964
|
+
...description ? { description } : {}
|
|
1965
|
+
};
|
|
1966
|
+
}
|
|
1967
|
+
if (params.tools?.length) {
|
|
1968
|
+
config.tools = [{ functionDeclarations: params.tools.map((t) => ({
|
|
1969
|
+
name: t.function.name,
|
|
1970
|
+
description: t.function.description,
|
|
1971
|
+
parameters: t.function.parameters
|
|
1972
|
+
})) }];
|
|
1973
|
+
config.toolConfig = toGeminiToolConfig(params.tool_choice);
|
|
1974
|
+
}
|
|
1975
|
+
return {
|
|
1976
|
+
model: params.model,
|
|
1977
|
+
contents: mergeConsecutiveFunctionResponses(conversationMessages.map((m) => toGeminiContent(m))),
|
|
1978
|
+
config
|
|
1979
|
+
};
|
|
1980
|
+
}
|
|
862
1981
|
/**
|
|
863
1982
|
* Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
|
|
864
1983
|
* uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
|
|
@@ -869,43 +1988,100 @@ function toGeminiParts(blocks) {
|
|
|
869
1988
|
* `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
|
|
870
1989
|
* models use a token budget, not an effort tier, so it's dropped, same as
|
|
871
1990
|
* Anthropic.
|
|
1991
|
+
*
|
|
1992
|
+
* `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
|
|
1993
|
+
* `tool_choice` maps to `toolConfig.functionCallingConfig`. `jsonSchema`
|
|
1994
|
+
* and `tools` are mutually exclusive by the time a call reaches here
|
|
1995
|
+
* (enforced in vernLLM.ts), so `responseSchema` and `tools` never
|
|
1996
|
+
* both apply.
|
|
1997
|
+
*
|
|
1998
|
+
* `createStream` calls `generateContentStream` (optional on `GeminiClient`
|
|
1999
|
+
*, required only if the caller sets `stream: true`) and translates each
|
|
2000
|
+
* partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
|
|
2001
|
+
* Gemini's own function-calling API doesn't stream tool-call arguments
|
|
2002
|
+
* incrementally: a `functionCall` part always arrives whole in one chunk,
|
|
2003
|
+
* so each one is emitted as a single, complete `tool_call_delta` (a
|
|
2004
|
+
* one-shot "delta" containing the full arguments) rather than accumulated
|
|
2005
|
+
* fragments, that's a real difference in the underlying API, not
|
|
2006
|
+
* something this adapter can smooth over. `usageMetadata` is (per Gemini's
|
|
2007
|
+
* own behavior) only reliably present on the last chunk, so the `usage`
|
|
2008
|
+
* `WireStreamChunk` is emitted once, after the stream completes, from
|
|
2009
|
+
* whichever chunk's `usageMetadata` was seen last.
|
|
872
2010
|
*/
|
|
873
2011
|
function fromGemini(geminiClient) {
|
|
874
|
-
return { chat: { completions: {
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
maxOutputTokens: params.max_tokens
|
|
881
|
-
};
|
|
882
|
-
if (wantsJson) generationConfig.responseMimeType = "application/json";
|
|
883
|
-
if (params.response_format?.type === "json_schema") {
|
|
884
|
-
const { schema, description } = params.response_format.json_schema;
|
|
885
|
-
generationConfig.responseSchema = {
|
|
886
|
-
...schema,
|
|
887
|
-
...description ? { description } : {}
|
|
2012
|
+
return { chat: { completions: {
|
|
2013
|
+
async create(params, options) {
|
|
2014
|
+
const request = buildGeminiRequest(params);
|
|
2015
|
+
request.config = {
|
|
2016
|
+
...request.config,
|
|
2017
|
+
abortSignal: options.signal
|
|
888
2018
|
};
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
2019
|
+
const response = await geminiClient.generateContent(request);
|
|
2020
|
+
const parts = response.candidates?.[0]?.content?.parts ?? [];
|
|
2021
|
+
const text = parts.map((p) => p.text ?? "").join("");
|
|
2022
|
+
const functionCalls = parts.filter((p) => p.functionCall);
|
|
2023
|
+
let wireToolCalls;
|
|
2024
|
+
if (functionCalls.length) wireToolCalls = functionCalls.map((p) => ({
|
|
2025
|
+
id: p.functionCall.name,
|
|
2026
|
+
type: "function",
|
|
2027
|
+
function: {
|
|
2028
|
+
name: p.functionCall.name,
|
|
2029
|
+
arguments: JSON.stringify(p.functionCall.args ?? {})
|
|
2030
|
+
}
|
|
2031
|
+
}));
|
|
2032
|
+
return {
|
|
2033
|
+
choices: [{ message: {
|
|
2034
|
+
content: text,
|
|
2035
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
2036
|
+
} }],
|
|
2037
|
+
usage: {
|
|
2038
|
+
prompt_tokens: response.usageMetadata?.promptTokenCount,
|
|
2039
|
+
completion_tokens: response.usageMetadata?.candidatesTokenCount,
|
|
2040
|
+
total_tokens: response.usageMetadata?.totalTokenCount
|
|
2041
|
+
}
|
|
2042
|
+
};
|
|
2043
|
+
},
|
|
2044
|
+
async *createStream(params, options) {
|
|
2045
|
+
if (!geminiClient.generateContentStream) throw new LLMError("stream: true requires a Gemini client with generateContentStream", "validation");
|
|
2046
|
+
const request = buildGeminiRequest(params);
|
|
2047
|
+
request.config = {
|
|
2048
|
+
...request.config,
|
|
2049
|
+
abortSignal: options.signal
|
|
2050
|
+
};
|
|
2051
|
+
const stream = await geminiClient.generateContentStream(request);
|
|
2052
|
+
let toolCallIndex = 0;
|
|
2053
|
+
let lastUsage;
|
|
2054
|
+
for await (const chunk of stream) {
|
|
2055
|
+
const parts = chunk.candidates?.[0]?.content?.parts ?? [];
|
|
2056
|
+
for (const part of parts) {
|
|
2057
|
+
if (part.text) yield {
|
|
2058
|
+
type: "text-delta",
|
|
2059
|
+
delta: part.text
|
|
2060
|
+
};
|
|
2061
|
+
if (part.functionCall) {
|
|
2062
|
+
yield {
|
|
2063
|
+
type: "tool_call_delta",
|
|
2064
|
+
index: toolCallIndex,
|
|
2065
|
+
id: part.functionCall.name,
|
|
2066
|
+
name: part.functionCall.name,
|
|
2067
|
+
argumentsDelta: JSON.stringify(part.functionCall.args ?? {}),
|
|
2068
|
+
complete: true
|
|
2069
|
+
};
|
|
2070
|
+
toolCallIndex++;
|
|
2071
|
+
}
|
|
2072
|
+
}
|
|
2073
|
+
if (chunk.usageMetadata) lastUsage = chunk.usageMetadata;
|
|
906
2074
|
}
|
|
907
|
-
|
|
908
|
-
|
|
2075
|
+
if (lastUsage) yield {
|
|
2076
|
+
type: "usage",
|
|
2077
|
+
usage: {
|
|
2078
|
+
prompt_tokens: lastUsage.promptTokenCount,
|
|
2079
|
+
completion_tokens: lastUsage.candidatesTokenCount,
|
|
2080
|
+
total_tokens: lastUsage.totalTokenCount
|
|
2081
|
+
}
|
|
2082
|
+
};
|
|
2083
|
+
}
|
|
2084
|
+
} } };
|
|
909
2085
|
}
|
|
910
2086
|
|
|
911
2087
|
//#endregion
|
|
@@ -941,6 +2117,67 @@ function toBedrockContent(blocks) {
|
|
|
941
2117
|
} } : { text: block.text });
|
|
942
2118
|
}
|
|
943
2119
|
/**
|
|
2120
|
+
* Builds the Converse-shaped request from VernLLM's wire params, shared
|
|
2121
|
+
* between `create` and `createStream` so both go through identical
|
|
2122
|
+
* translation (system prompt, message shaping, the jsonSchema →
|
|
2123
|
+
* forced-single-tool mapping, and the `toolUseSupportedModels` preflight
|
|
2124
|
+
* check all happen exactly once).
|
|
2125
|
+
*
|
|
2126
|
+
* Returns `toolName` alongside the request: when set, the model was forced
|
|
2127
|
+
* to call a single synthetic tool standing in for `jsonSchema` output, and
|
|
2128
|
+
* both `create` and `createStream` need to know this so they can unwrap
|
|
2129
|
+
* that tool call back into plain text content instead of treating it like
|
|
2130
|
+
* a real tool call.
|
|
2131
|
+
*/
|
|
2132
|
+
function buildBedrockRequest(params, toolUseSupportedModels) {
|
|
2133
|
+
const systemMessage = params.messages.find((m) => m.role === "system");
|
|
2134
|
+
const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
|
|
2135
|
+
const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
|
|
2136
|
+
const toolName = jsonSchema?.name.trim();
|
|
2137
|
+
if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
|
|
2138
|
+
let jsonInstruction;
|
|
2139
|
+
let toolConfig;
|
|
2140
|
+
if (jsonSchema) {
|
|
2141
|
+
const { schema, description, strict } = jsonSchema;
|
|
2142
|
+
toolConfig = {
|
|
2143
|
+
tools: [{ toolSpec: {
|
|
2144
|
+
name: toolName,
|
|
2145
|
+
description,
|
|
2146
|
+
inputSchema: { json: schema },
|
|
2147
|
+
strict
|
|
2148
|
+
} }],
|
|
2149
|
+
toolChoice: { tool: { name: toolName } }
|
|
2150
|
+
};
|
|
2151
|
+
} else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
|
|
2152
|
+
else if (params.tools?.length) toolConfig = {
|
|
2153
|
+
tools: params.tools.map((t) => ({ toolSpec: {
|
|
2154
|
+
name: t.function.name,
|
|
2155
|
+
description: t.function.description,
|
|
2156
|
+
inputSchema: { json: t.function.parameters }
|
|
2157
|
+
} })),
|
|
2158
|
+
toolChoice: toBedrockToolChoice(params.tool_choice)
|
|
2159
|
+
};
|
|
2160
|
+
if (jsonSchema && toolUseSupportedModels) {
|
|
2161
|
+
const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
|
|
2162
|
+
if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
|
|
2163
|
+
}
|
|
2164
|
+
const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
|
|
2165
|
+
const request = {
|
|
2166
|
+
modelId: params.model,
|
|
2167
|
+
messages: mergeConsecutiveToolResults(conversationMessages.map((m) => toBedrockMessage(m))),
|
|
2168
|
+
system: systemParts.length ? systemParts.map((text) => ({ text })) : void 0,
|
|
2169
|
+
inferenceConfig: {
|
|
2170
|
+
...params.temperature !== void 0 ? { temperature: params.temperature } : {},
|
|
2171
|
+
maxTokens: params.max_tokens
|
|
2172
|
+
},
|
|
2173
|
+
...toolConfig ? { toolConfig } : {}
|
|
2174
|
+
};
|
|
2175
|
+
return {
|
|
2176
|
+
request,
|
|
2177
|
+
toolName
|
|
2178
|
+
};
|
|
2179
|
+
}
|
|
2180
|
+
/**
|
|
944
2181
|
* Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
|
|
945
2182
|
* interface VernLLM uses for OpenAI/Groq. The Converse API is unified
|
|
946
2183
|
* across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
|
|
@@ -960,66 +2197,237 @@ function toBedrockContent(blocks) {
|
|
|
960
2197
|
* `response_format: json_object` (no schema to build a tool from) and
|
|
961
2198
|
* `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
|
|
962
2199
|
* instruction and are dropped respectively.
|
|
2200
|
+
*
|
|
2201
|
+
* `tools` maps to Converse's native `toolConfig`/`toolUse`/`toolResult`;
|
|
2202
|
+
* `tool_choice` maps to `toolConfig.toolChoice`. Mutually exclusive with
|
|
2203
|
+
* `jsonSchema` by the time a call reaches here (enforced in vernLLM.ts).
|
|
2204
|
+
*
|
|
2205
|
+
* `createStream` calls `converseStream` (optional on `BedrockConverseClient`
|
|
2206
|
+
*, required only if the caller sets `stream: true`) and translates its
|
|
2207
|
+
* `contentBlockStart`/`contentBlockDelta`/`metadata` events into
|
|
2208
|
+
* `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
|
|
2209
|
+
* same as `fromAnthropic`'s block-index tracking (Converse's streaming
|
|
2210
|
+
* shape is structurally close to Anthropic's own, both being tool-use-aware
|
|
2211
|
+
* content-block streams), including the same `json-tool` unwrapping: a
|
|
2212
|
+
* `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
|
|
2213
|
+
* `text-delta`, not `tool_call_delta`, so the accumulated result lands in
|
|
2214
|
+
* `finalizeResponse`'s `content` path exactly like the non-streaming
|
|
2215
|
+
* `create` branch above unwraps it.
|
|
963
2216
|
*/
|
|
964
2217
|
function fromBedrock(bedrockClient, options) {
|
|
965
2218
|
const toolUseSupportedModels = options?.toolUseSupportedModels;
|
|
966
|
-
return { chat: { completions: {
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
2219
|
+
return { chat: { completions: {
|
|
2220
|
+
async create(params, requestOptions) {
|
|
2221
|
+
const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
|
|
2222
|
+
const response = await bedrockClient.converse(request, requestOptions);
|
|
2223
|
+
let text;
|
|
2224
|
+
let wireToolCalls;
|
|
2225
|
+
if (toolName) {
|
|
2226
|
+
const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
|
|
2227
|
+
text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
|
|
2228
|
+
} else {
|
|
2229
|
+
const blocks = response.output?.message?.content ?? [];
|
|
2230
|
+
text = blocks.map((c) => c.text ?? "").join("");
|
|
2231
|
+
const toolUses = blocks.filter((block) => Boolean(block.toolUse));
|
|
2232
|
+
if (toolUses.length) wireToolCalls = toolUses.map((block, i) => {
|
|
2233
|
+
const toolUse = block.toolUse;
|
|
2234
|
+
if (!toolUse.name) throw new LLMError(`Bedrock returned a toolUse block without a name at index ${i}.`, "validation");
|
|
2235
|
+
return {
|
|
2236
|
+
id: toolUse.toolUseId ?? `${toolUse.name}_${i}`,
|
|
2237
|
+
type: "function",
|
|
2238
|
+
function: {
|
|
2239
|
+
name: toolUse.name,
|
|
2240
|
+
arguments: JSON.stringify(toolUse.input ?? {})
|
|
2241
|
+
}
|
|
2242
|
+
};
|
|
2243
|
+
});
|
|
2244
|
+
}
|
|
2245
|
+
return {
|
|
2246
|
+
choices: [{ message: {
|
|
2247
|
+
content: text,
|
|
2248
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
982
2249
|
} }],
|
|
983
|
-
|
|
2250
|
+
usage: {
|
|
2251
|
+
prompt_tokens: response.usage?.inputTokens,
|
|
2252
|
+
completion_tokens: response.usage?.outputTokens,
|
|
2253
|
+
total_tokens: response.usage?.totalTokens
|
|
2254
|
+
}
|
|
2255
|
+
};
|
|
2256
|
+
},
|
|
2257
|
+
async *createStream(params, requestOptions) {
|
|
2258
|
+
if (!bedrockClient.converseStream) throw new LLMError("stream: true requires a Bedrock client with converseStream", "validation");
|
|
2259
|
+
const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
|
|
2260
|
+
const { stream } = await bedrockClient.converseStream(request, requestOptions);
|
|
2261
|
+
const blockKinds = new Map();
|
|
2262
|
+
for await (const event of stream) if ("contentBlockStart" in event) {
|
|
2263
|
+
const { contentBlockIndex, start } = event.contentBlockStart;
|
|
2264
|
+
if (start?.toolUse) {
|
|
2265
|
+
const kind = start.toolUse.name === toolName ? "json-tool" : "tool_use";
|
|
2266
|
+
blockKinds.set(contentBlockIndex, kind);
|
|
2267
|
+
if (kind === "tool_use" && !toolName) yield {
|
|
2268
|
+
type: "tool_call_delta",
|
|
2269
|
+
index: contentBlockIndex,
|
|
2270
|
+
id: start.toolUse.toolUseId,
|
|
2271
|
+
name: start.toolUse.name
|
|
2272
|
+
};
|
|
2273
|
+
} else blockKinds.set(contentBlockIndex, "text");
|
|
2274
|
+
} else if ("contentBlockDelta" in event) {
|
|
2275
|
+
const { contentBlockIndex, delta } = event.contentBlockDelta;
|
|
2276
|
+
if (delta && "text" in delta && delta.text !== void 0 && !toolName) yield {
|
|
2277
|
+
type: "text-delta",
|
|
2278
|
+
delta: delta.text
|
|
2279
|
+
};
|
|
2280
|
+
else if (delta && "toolUse" in delta && delta.toolUse?.input !== void 0) {
|
|
2281
|
+
const kind = blockKinds.get(contentBlockIndex);
|
|
2282
|
+
if (kind === "json-tool") yield {
|
|
2283
|
+
type: "text-delta",
|
|
2284
|
+
delta: delta.toolUse.input
|
|
2285
|
+
};
|
|
2286
|
+
else if (!toolName) yield {
|
|
2287
|
+
type: "tool_call_delta",
|
|
2288
|
+
index: contentBlockIndex,
|
|
2289
|
+
argumentsDelta: delta.toolUse.input
|
|
2290
|
+
};
|
|
2291
|
+
}
|
|
2292
|
+
} else if ("metadata" in event && event.metadata.usage) yield {
|
|
2293
|
+
type: "usage",
|
|
2294
|
+
usage: {
|
|
2295
|
+
prompt_tokens: event.metadata.usage.inputTokens,
|
|
2296
|
+
completion_tokens: event.metadata.usage.outputTokens,
|
|
2297
|
+
total_tokens: event.metadata.usage.totalTokens
|
|
2298
|
+
}
|
|
984
2299
|
};
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
2300
|
+
else if ("throttlingException" in event) throw new LLMError(event.throttlingException.message ?? "Bedrock throttled the request mid-stream", "api", 429);
|
|
2301
|
+
else if ("validationException" in event) throw new LLMError(event.validationException.message ?? "Bedrock rejected the request mid-stream", "validation");
|
|
2302
|
+
else if ("internalServerException" in event || "serviceUnavailableException" in event || "modelStreamErrorException" in event) {
|
|
2303
|
+
const detail = "internalServerException" in event && event.internalServerException.message || "serviceUnavailableException" in event && event.serviceUnavailableException.message || "modelStreamErrorException" in event && event.modelStreamErrorException.message || "Bedrock reported a mid-stream error";
|
|
2304
|
+
const status = "modelStreamErrorException" in event && event.modelStreamErrorException.originalStatusCode || "serviceUnavailableException" in event && 503 || 500;
|
|
2305
|
+
throw new LLMError(detail, "api", status);
|
|
2306
|
+
}
|
|
989
2307
|
}
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
2308
|
+
} } };
|
|
2309
|
+
}
|
|
2310
|
+
/** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Converse's `toolChoice`. */
|
|
2311
|
+
function toBedrockToolChoice(toolChoice) {
|
|
2312
|
+
if (!toolChoice || toolChoice === "auto") return { auto: {} };
|
|
2313
|
+
if (toolChoice === "required") return { any: {} };
|
|
2314
|
+
if (toolChoice === "none") throw new LLMError("'none' is not supported by fromBedrock: Bedrock Converse has no `tool_choice` equivalent to forbidding tool use while tools are still offered. Omit `tools` entirely for this call instead.", "validation");
|
|
2315
|
+
return { tool: { name: toolChoice.function.name } };
|
|
2316
|
+
}
|
|
2317
|
+
/**
|
|
2318
|
+
* Translates one VernLLM wire message into Converse's
|
|
2319
|
+
* `{ role: 'user' | 'assistant', content }` shape.
|
|
2320
|
+
*/
|
|
2321
|
+
function toBedrockMessage(m) {
|
|
2322
|
+
if (m.role === "tool") return {
|
|
2323
|
+
role: "user",
|
|
2324
|
+
content: [{ toolResult: {
|
|
2325
|
+
toolUseId: m.tool_call_id,
|
|
2326
|
+
content: [{ text: m.content }],
|
|
2327
|
+
status: m.is_error ? "error" : "success"
|
|
2328
|
+
} }]
|
|
2329
|
+
};
|
|
2330
|
+
if (m.role === "assistant" && m.tool_calls?.length) {
|
|
2331
|
+
const blocks = [];
|
|
2332
|
+
if (m.content) blocks.push({ text: m.content });
|
|
2333
|
+
for (const tc of m.tool_calls) {
|
|
2334
|
+
let input;
|
|
2335
|
+
if (!tc.function.arguments.trim()) input = {};
|
|
2336
|
+
else try {
|
|
2337
|
+
input = JSON.parse(tc.function.arguments);
|
|
2338
|
+
} catch (cause) {
|
|
2339
|
+
throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
|
|
1015
2340
|
}
|
|
2341
|
+
blocks.push({ toolUse: {
|
|
2342
|
+
toolUseId: tc.id,
|
|
2343
|
+
name: tc.function.name,
|
|
2344
|
+
input
|
|
2345
|
+
} });
|
|
2346
|
+
}
|
|
2347
|
+
return {
|
|
2348
|
+
role: "assistant",
|
|
2349
|
+
content: blocks
|
|
1016
2350
|
};
|
|
1017
|
-
}
|
|
2351
|
+
}
|
|
2352
|
+
return {
|
|
2353
|
+
role: m.role,
|
|
2354
|
+
content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content ?? "" }]
|
|
2355
|
+
};
|
|
2356
|
+
}
|
|
2357
|
+
/**
|
|
2358
|
+
* Converse expects the results of everything the model asked for in one
|
|
2359
|
+
* turn to arrive together as multiple `toolResult` content blocks on a
|
|
2360
|
+
* single `'user'` message, not as separate consecutive `'user'` messages.
|
|
2361
|
+
* The per-wire-message mapping above produces one `'user'` message per
|
|
2362
|
+
* VernLLM wire tool message, so when an assistant turn requested more than
|
|
2363
|
+
* one tool, this merges the resulting run of toolResult-only `'user'`
|
|
2364
|
+
* messages back into one.
|
|
2365
|
+
*/
|
|
2366
|
+
function mergeConsecutiveToolResults(messages) {
|
|
2367
|
+
const isToolResultOnly = (m) => m.role === "user" && m.content.length > 0 && m.content.every((b) => "toolResult" in b);
|
|
2368
|
+
const merged = [];
|
|
2369
|
+
for (const m of messages) {
|
|
2370
|
+
const prev = merged.at(-1);
|
|
2371
|
+
if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
|
|
2372
|
+
else merged.push(m);
|
|
2373
|
+
}
|
|
2374
|
+
return merged;
|
|
1018
2375
|
}
|
|
1019
2376
|
|
|
1020
2377
|
//#endregion
|
|
1021
2378
|
//#region src/adapters/fetch.ts
|
|
1022
2379
|
/**
|
|
2380
|
+
* Wraps a WHATWG `ReadableStream` (what `response.body` is) so it can be
|
|
2381
|
+
* consumed with `for await`. Implemented via `getReader()` rather than
|
|
2382
|
+
* relying on `ReadableStream` having a native `Symbol.asyncIterator`,
|
|
2383
|
+
* that support varies across runtimes/versions, and this works everywhere
|
|
2384
|
+
* a `ReadableStream` does.
|
|
2385
|
+
*/
|
|
2386
|
+
async function* webStreamToAsyncIterable(stream) {
|
|
2387
|
+
const reader = stream.getReader();
|
|
2388
|
+
try {
|
|
2389
|
+
for (;;) {
|
|
2390
|
+
const { done, value } = await reader.read();
|
|
2391
|
+
if (done) return;
|
|
2392
|
+
if (value) yield value;
|
|
2393
|
+
}
|
|
2394
|
+
} finally {
|
|
2395
|
+
try {
|
|
2396
|
+
await reader.cancel();
|
|
2397
|
+
} catch {}
|
|
2398
|
+
reader.releaseLock();
|
|
2399
|
+
}
|
|
2400
|
+
}
|
|
2401
|
+
/** Default `requestStream`: native `fetch`, with the same error/`.status` contract non-streaming errors get. */
|
|
2402
|
+
async function defaultRequestStream(url, init) {
|
|
2403
|
+
const res = await fetch(url, init);
|
|
2404
|
+
if (!res.ok) {
|
|
2405
|
+
const body = await res.text().catch(() => "");
|
|
2406
|
+
const err = new Error(`Fetch adapter stream request failed (${res.status}): ${body.slice(0, 500)}`);
|
|
2407
|
+
err.status = res.status;
|
|
2408
|
+
err.headers = res.headers;
|
|
2409
|
+
throw err;
|
|
2410
|
+
}
|
|
2411
|
+
if (!res.body) throw new Error("Fetch adapter stream request received a response with no body.");
|
|
2412
|
+
return webStreamToAsyncIterable(res.body);
|
|
2413
|
+
}
|
|
2414
|
+
/** Builds the shared `{ method, headers, body? }` request-init for both `create` and `createStream`. */
|
|
2415
|
+
async function buildRequestInit(config, params, requestBody) {
|
|
2416
|
+
const url = typeof config.url === "function" ? config.url(params) : config.url;
|
|
2417
|
+
const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
|
|
2418
|
+
const method = config.method ?? "POST";
|
|
2419
|
+
const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
|
|
2420
|
+
return {
|
|
2421
|
+
url,
|
|
2422
|
+
method,
|
|
2423
|
+
headers: supportsBody ? {
|
|
2424
|
+
"Content-Type": "application/json",
|
|
2425
|
+
...headers
|
|
2426
|
+
} : { ...headers },
|
|
2427
|
+
...supportsBody ? { body: JSON.stringify(requestBody) } : {}
|
|
2428
|
+
};
|
|
2429
|
+
}
|
|
2430
|
+
/**
|
|
1023
2431
|
* A fetch-based escape hatch for providers with no SDK, or where pulling one
|
|
1024
2432
|
* in isnt worth it. You supply the URL, headers, and two small mapping
|
|
1025
2433
|
* functions; this handles the HTTP call and slots the result into the same
|
|
@@ -1029,41 +2437,99 @@ function fromBedrock(bedrockClient, options) {
|
|
|
1029
2437
|
* Non-2xx responses throw an error with `.status` set to the HTTP status
|
|
1030
2438
|
* code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
|
|
1031
2439
|
* 401/403) applies here too
|
|
2440
|
+
*
|
|
2441
|
+
* Tool calling works the same way as every other adapter: `mapRequest`
|
|
2442
|
+
* receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
|
|
2443
|
+
* translate them into whatever shape the provider's wire format expects
|
|
2444
|
+
* (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
|
|
2445
|
+
* field). On the way back, `mapResponse` may return a `toolCalls` array
|
|
2446
|
+
* (id/name/JSON-encoded-arguments-string per call) alongside or instead of
|
|
2447
|
+
* `content`; VernLLM parses and (if `argumentsSchema` was set) validates
|
|
2448
|
+
* those arguments the same way it does for every other adapter. For
|
|
2449
|
+
* `stream: true`, tool-call deltas go through the existing
|
|
2450
|
+
* `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
|
|
2451
|
+
* no separate config is needed for streaming vs non-streaming tool calls.
|
|
2452
|
+
*
|
|
2453
|
+
|
|
2454
|
+
* `createStream` requires `mapStreamEvent` (there's no non-streaming
|
|
2455
|
+
* response to fall back on, unlike the other three optional streaming
|
|
2456
|
+
* seams). It opens the request via `requestStream` (defaults to native
|
|
2457
|
+
* `fetch`), splits the raw bytes into individual events via
|
|
2458
|
+
* `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
|
|
2459
|
+
* and translates each event into `WireStreamChunk`(s) via
|
|
2460
|
+
* `mapStreamEvent`. Both seams are overridable per-config for providers
|
|
2461
|
+
* that don't fit the SSE-over-fetch default. If a custom `request`
|
|
2462
|
+
* transport is configured, `requestStream` must be configured too,
|
|
2463
|
+
* `requestStream` never silently falls back to `request` (see
|
|
2464
|
+
* `createStream`'s own comment for why), so a `stream: true` call with
|
|
2465
|
+
* `request` set but no `requestStream` throws a clear
|
|
2466
|
+
* `LLMError('validation')` instead of quietly using unrelated native
|
|
2467
|
+
* `fetch`.
|
|
1032
2468
|
*/
|
|
1033
2469
|
function fromFetch(config) {
|
|
1034
|
-
return { chat: { completions: {
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
const
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
2470
|
+
return { chat: { completions: {
|
|
2471
|
+
async create(params, options) {
|
|
2472
|
+
const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
|
|
2473
|
+
const request = config.request ?? fetch;
|
|
2474
|
+
const res = await request(url, {
|
|
2475
|
+
method,
|
|
2476
|
+
headers,
|
|
2477
|
+
body,
|
|
2478
|
+
signal: options.signal
|
|
2479
|
+
});
|
|
2480
|
+
if (!res.ok) {
|
|
2481
|
+
const responseBody = await res.text().catch(() => "");
|
|
2482
|
+
const err = new Error(`Fetch adapter request failed (${res.status}): ${responseBody.slice(0, 500)}`);
|
|
2483
|
+
err.status = res.status;
|
|
2484
|
+
err.headers = res.headers;
|
|
2485
|
+
throw err;
|
|
2486
|
+
}
|
|
2487
|
+
const json = await res.json();
|
|
2488
|
+
const { content, usage, toolCalls } = config.mapResponse(json);
|
|
2489
|
+
const wireToolCalls = toolCalls?.length ? toolCalls.map((tc) => ({
|
|
2490
|
+
id: tc.id,
|
|
2491
|
+
type: "function",
|
|
2492
|
+
function: {
|
|
2493
|
+
name: tc.name,
|
|
2494
|
+
arguments: tc.arguments
|
|
2495
|
+
}
|
|
2496
|
+
})) : void 0;
|
|
2497
|
+
return {
|
|
2498
|
+
choices: [{ message: {
|
|
2499
|
+
content,
|
|
2500
|
+
...wireToolCalls ? { tool_calls: wireToolCalls } : {}
|
|
2501
|
+
} }],
|
|
2502
|
+
usage: usage ? {
|
|
2503
|
+
prompt_tokens: usage.promptTokens,
|
|
2504
|
+
completion_tokens: usage.completionTokens,
|
|
2505
|
+
total_tokens: usage.totalTokens
|
|
2506
|
+
} : void 0
|
|
2507
|
+
};
|
|
2508
|
+
},
|
|
2509
|
+
async *createStream(params, options) {
|
|
2510
|
+
if (!config.mapStreamEvent) throw new LLMError("stream: true requires mapStreamEvent to be configured on fromFetch", "validation");
|
|
2511
|
+
if (config.request && !config.requestStream) throw new LLMError("`stream: true` requires `requestStream` to be configured on fromFetch when a custom `request` transport is set. `requestStream` does not fall back to `request` (it needs an async-iterable byte stream, which `RequestLike`'s buffered `ResponseLike` has no way to provide), without it, `stream: true` would silently use plain native `fetch` instead of your configured transport. Add a `requestStream` that opens the same connection your `request` does, or omit `request` if native `fetch` is fine for both.", "validation");
|
|
2512
|
+
const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
|
|
2513
|
+
const requestStream = config.requestStream ?? defaultRequestStream;
|
|
2514
|
+
const parseFrames = config.parseStreamFrames ?? parseSseStream;
|
|
2515
|
+
const byteStream = await requestStream(url, {
|
|
2516
|
+
method,
|
|
2517
|
+
headers,
|
|
2518
|
+
body,
|
|
2519
|
+
signal: options.signal
|
|
2520
|
+
});
|
|
2521
|
+
for await (const event of parseFrames(byteStream)) {
|
|
2522
|
+
if (event === SSE_PING) {
|
|
2523
|
+
yield { type: "ping" };
|
|
2524
|
+
continue;
|
|
2525
|
+
}
|
|
2526
|
+
const wireChunks = config.mapStreamEvent(event);
|
|
2527
|
+
if (!wireChunks) continue;
|
|
2528
|
+
if (Array.isArray(wireChunks)) yield* wireChunks;
|
|
2529
|
+
else yield wireChunks;
|
|
2530
|
+
}
|
|
1055
2531
|
}
|
|
1056
|
-
|
|
1057
|
-
const { content, usage } = config.mapResponse(json);
|
|
1058
|
-
return {
|
|
1059
|
-
choices: [{ message: { content } }],
|
|
1060
|
-
usage: usage ? {
|
|
1061
|
-
prompt_tokens: usage.promptTokens,
|
|
1062
|
-
completion_tokens: usage.completionTokens,
|
|
1063
|
-
total_tokens: usage.totalTokens
|
|
1064
|
-
} : void 0
|
|
1065
|
-
};
|
|
1066
|
-
} } } };
|
|
2532
|
+
} } };
|
|
1067
2533
|
}
|
|
1068
2534
|
|
|
1069
2535
|
//#endregion
|
|
@@ -1086,41 +2552,84 @@ function toOpenAIContent(blocks) {
|
|
|
1086
2552
|
});
|
|
1087
2553
|
}
|
|
1088
2554
|
/**
|
|
1089
|
-
*
|
|
1090
|
-
*
|
|
1091
|
-
*
|
|
1092
|
-
*
|
|
1093
|
-
* this exists purely so call sites read clearly (`fromMistral(client)` vs
|
|
1094
|
-
* handing a Mistral client to something typed for OpenAI) and so a real
|
|
1095
|
-
* transformation could be added later, per-provider, without a breaking
|
|
1096
|
-
* change.
|
|
1097
|
-
*
|
|
1098
|
-
* The one thing that isn't a pure passthrough: a `ContentBlock[]`
|
|
1099
|
-
* `userContent` is translated into OpenAI's native `image_url` content-part
|
|
1100
|
-
* shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
|
|
1101
|
-
* rather than a copy of any one provider's wire format.
|
|
1102
|
-
*
|
|
1103
|
-
* Not every SDKs own TypeScript types line up exactly with `LLMClient`
|
|
1104
|
-
* (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
|
|
1105
|
-
* the actual compatibility contract is the JSON each provider sends and
|
|
1106
|
-
* receives over the wire, not the SDKs TS types.
|
|
2555
|
+
* Translates VernLLM's provider-agnostic `messages` (the one part of a
|
|
2556
|
+
* request that isn't a pure passthrough for OpenAI-compatible clients) into
|
|
2557
|
+
* OpenAI's native wire shape. Shared between `create` and `createStream` so
|
|
2558
|
+
* both go through identical message translation.
|
|
1107
2559
|
*/
|
|
1108
|
-
function
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
|
|
2560
|
+
function toOpenAIMessages(params) {
|
|
2561
|
+
return params.messages.map((m) => {
|
|
2562
|
+
if (m.role === "user" && Array.isArray(m.content)) return {
|
|
1112
2563
|
...m,
|
|
1113
2564
|
content: toOpenAIContent(m.content)
|
|
1114
|
-
}
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
}
|
|
1119
|
-
|
|
2565
|
+
};
|
|
2566
|
+
if (m.role === "tool") {
|
|
2567
|
+
const { is_error: _isError,...openAIToolMessage } = m;
|
|
2568
|
+
return openAIToolMessage;
|
|
2569
|
+
}
|
|
2570
|
+
return m;
|
|
2571
|
+
});
|
|
2572
|
+
}
|
|
2573
|
+
/**
|
|
2574
|
+
* Translates one OpenAI-shaped SSE chunk into zero or more `WireStreamChunk`s.
|
|
2575
|
+
* A single chunk can carry a text delta, one or more tool-call argument
|
|
2576
|
+
* deltas (each keyed by `index`, OpenAI's own convention for streaming
|
|
2577
|
+
* parallel tool calls, mirrored directly by VernLLM's `tool_call_delta`
|
|
2578
|
+
* shape so accumulation composes without translation), and/or a final
|
|
2579
|
+
* usage block (present only when `stream_options.include_usage` is set,
|
|
2580
|
+
* which this adapter always sets).
|
|
2581
|
+
*/
|
|
2582
|
+
function* toWireStreamChunks(chunk) {
|
|
2583
|
+
const delta = chunk.choices?.[0]?.delta;
|
|
2584
|
+
if (delta?.content) yield {
|
|
2585
|
+
type: "text-delta",
|
|
2586
|
+
delta: delta.content
|
|
2587
|
+
};
|
|
2588
|
+
if (delta?.tool_calls?.length) for (const toolCall of delta.tool_calls) yield {
|
|
2589
|
+
type: "tool_call_delta",
|
|
2590
|
+
index: toolCall.index,
|
|
2591
|
+
id: toolCall.id,
|
|
2592
|
+
name: toolCall.function?.name,
|
|
2593
|
+
argumentsDelta: toolCall.function?.arguments
|
|
2594
|
+
};
|
|
2595
|
+
if (chunk.usage) yield {
|
|
2596
|
+
type: "usage",
|
|
2597
|
+
usage: chunk.usage
|
|
2598
|
+
};
|
|
2599
|
+
}
|
|
2600
|
+
function fromOpenAICompatible(client, options = {}) {
|
|
2601
|
+
const raw = client;
|
|
2602
|
+
const { supportsStreamUsage = true } = options;
|
|
2603
|
+
const rawCreate = raw.chat.completions.create.bind(raw.chat.completions);
|
|
2604
|
+
return { chat: { completions: {
|
|
2605
|
+
async create(params, options$1) {
|
|
2606
|
+
const messages = toOpenAIMessages(params);
|
|
2607
|
+
return raw.chat.completions.create({
|
|
2608
|
+
...params,
|
|
2609
|
+
messages
|
|
2610
|
+
}, options$1);
|
|
2611
|
+
},
|
|
2612
|
+
async *createStream(params, options$1) {
|
|
2613
|
+
const messages = toOpenAIMessages(params);
|
|
2614
|
+
const stream = await rawCreate({
|
|
2615
|
+
...params,
|
|
2616
|
+
messages,
|
|
2617
|
+
stream: true,
|
|
2618
|
+
...supportsStreamUsage ? { stream_options: { include_usage: true } } : {}
|
|
2619
|
+
}, options$1);
|
|
2620
|
+
for await (const chunk of stream) yield* toWireStreamChunks(chunk);
|
|
2621
|
+
}
|
|
2622
|
+
} } };
|
|
1120
2623
|
}
|
|
1121
2624
|
/** Groqs SDK matches the OpenAI wire format */
|
|
1122
2625
|
const fromGroq = fromOpenAICompatible;
|
|
1123
|
-
/**
|
|
2626
|
+
/**
|
|
2627
|
+
* Mistrals `chat.completions`-shaped client (or their OpenAI-compat
|
|
2628
|
+
* endpoint). Mistral supports `stream_options.include_usage` (added after
|
|
2629
|
+
* an earlier period where it returned a 422 for unrecognized fields, per
|
|
2630
|
+
* Mistral's changelog and streaming docs), so this is a plain alias like
|
|
2631
|
+
* the others, `supportsStreamUsage` defaults to `true`.
|
|
2632
|
+
*/
|
|
1124
2633
|
const fromMistral = fromOpenAICompatible;
|
|
1125
2634
|
/** DeepSeeks API is OpenAI-compatible */
|
|
1126
2635
|
const fromDeepSeek = fromOpenAICompatible;
|
|
@@ -1214,6 +2723,7 @@ exports.ConsoleLogger = ConsoleLogger
|
|
|
1214
2723
|
exports.InMemoryCacheAdapter = InMemoryCacheAdapter
|
|
1215
2724
|
exports.LLMError = LLMError
|
|
1216
2725
|
exports.NormalizedCacheAdapter = NormalizedCacheAdapter
|
|
2726
|
+
exports.SSE_PING = SSE_PING
|
|
1217
2727
|
exports.TieredCacheAdapter = TieredCacheAdapter
|
|
1218
2728
|
exports.VernLLM = VernLLM
|
|
1219
2729
|
exports.from01AI = from01AI
|
|
@@ -1261,4 +2771,6 @@ exports.fromVercelAIGateway = fromVercelAIGateway
|
|
|
1261
2771
|
exports.fromXAI = fromXAI
|
|
1262
2772
|
exports.fromZhipu = fromZhipu
|
|
1263
2773
|
exports.isLLMError = isLLMError
|
|
2774
|
+
exports.isToolCallResult = isToolCallResult
|
|
2775
|
+
exports.parseSseStream = parseSseStream
|
|
1264
2776
|
//# sourceMappingURL=index.cjs.map
|