vern-llm 1.7.1 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -26,297 +26,23 @@ const crypto = __toESM(require("crypto"));
26
26
 
27
27
  //#region src/types/errors.ts
28
28
  var LLMError = class extends Error {
29
- constructor(message, type, status, issues, cause, retryAfterMs) {
29
+ constructor(message, type, status, issues, cause, retryAfterMs, code) {
30
30
  super(message);
31
31
  this.type = type;
32
32
  this.status = status;
33
33
  this.issues = issues;
34
34
  this.cause = cause;
35
35
  this.retryAfterMs = retryAfterMs;
36
+ this.code = code;
36
37
  this.name = "LLMError";
37
38
  }
39
+ /** Every tool contract failure in one response, when there is more than one. */
40
+ toolIssues;
38
41
  };
39
42
  function isLLMError(err) {
40
43
  return err instanceof LLMError;
41
44
  }
42
45
 
43
- //#endregion
44
- //#region src/circuitBreaker.ts
45
- /**
46
- * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures across
47
- * calls. Once the threshold is hit, short-circuits new calls with an
48
- * LLMError('circuit_open') instead of hitting the provider, until the
49
- * cooldown elapses and a single trial call is allowed through
50
- */
51
- var CircuitBreaker = class {
52
- state = "closed";
53
- consecutiveFailures = 0;
54
- openedAt = 0;
55
- threshold;
56
- cooldownMs;
57
- /**
58
- * True while a single half-open trial call is in flight. Guards against
59
- * multiple concurrent callers all treating themselves as "the" trial once
60
- * the cooldown elapses
61
- */
62
- trialInFlight = false;
63
- constructor(options = {}) {
64
- this.threshold = options.threshold ?? 5;
65
- this.cooldownMs = options.cooldownMs ?? 3e4;
66
- }
67
- /**
68
- * Throws if the circuit is open and the cooldown hasn't elapsed, or if
69
- * the circuit is half-open and a trial call is already in flight.
70
- * Otherwise, if the circuit just became eligible for a trial (cooldown
71
- * elapsed, or half-open with no trial currently running), this call
72
- * becomes that trial
73
- */
74
- assertClosed() {
75
- if (this.state === "closed") return;
76
- if (this.state === "open") {
77
- const elapsed = Date.now() - this.openedAt;
78
- if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
79
- this.state = "half-open";
80
- this.trialInFlight = true;
81
- return;
82
- }
83
- if (this.trialInFlight) throw new LLMError("Circuit half-open. A trial request is already in flight. Try again shortly.", "circuit_open");
84
- this.trialInFlight = true;
85
- }
86
- recordSuccess() {
87
- this.consecutiveFailures = 0;
88
- this.state = "closed";
89
- this.trialInFlight = false;
90
- }
91
- recordFailure() {
92
- this.consecutiveFailures += 1;
93
- this.trialInFlight = false;
94
- if (this.state === "half-open") {
95
- this.state = "open";
96
- this.openedAt = Date.now();
97
- return;
98
- }
99
- if (this.consecutiveFailures >= this.threshold) {
100
- this.state = "open";
101
- this.openedAt = Date.now();
102
- }
103
- }
104
- getState() {
105
- return this.state;
106
- }
107
- };
108
-
109
- //#endregion
110
- //#region src/internal/vernLLM.utils.ts
111
- function defaultParseJson(content) {
112
- try {
113
- return JSON.parse(content);
114
- } catch {
115
- return void 0;
116
- }
117
- }
118
- /**
119
- * Looks inside an unknown error value and pulls out an http status code
120
- * if one is present. Checks the status field first then the status code
121
- * field since different client libraries use different names for this.
122
- * Returns undefined when the error is not an object or carries no status
123
- */
124
- function extractStatus(err) {
125
- if (!err || typeof err !== "object") return void 0;
126
- const error = err;
127
- if (typeof error.status === "number") return error.status;
128
- if (typeof error.statusCode === "number") return error.statusCode;
129
- return void 0;
130
- }
131
- function formatSafely(value) {
132
- try {
133
- return JSON.stringify(value, null, 2) ?? String(value);
134
- } catch {
135
- try {
136
- return String(value);
137
- } catch {
138
- return "[unprintable error]";
139
- }
140
- }
141
- }
142
- /**
143
- * Looks inside an unknown thrown value and pulls out a human-readable
144
- * description of it. Checks the `error` field first (the provider's raw
145
- * rejection body, JSON-stringified if possible) then falls back to the
146
- * message` field. Always returns a safe string, even when the thrown value
147
- * has hostile properties or cannot be serialized normally.
148
- */
149
- function describeError(err) {
150
- if (err && typeof err === "object") try {
151
- const error = err;
152
- if (error.error !== void 0) return formatSafely(error.error);
153
- if (typeof error.message === "string") return error.message;
154
- } catch {}
155
- return formatSafely(err);
156
- }
157
- /**
158
- * Runs an async function and cancels it if it takes longer than the given
159
- * timeout. Creates an internal abort controller that fires after the
160
- * timeout elapses, and combines it with any external signal the caller
161
- * passed in so either one can cancel the underlying call. If the internal
162
- * timeout triggers and the underlying operation aborts, the error is
163
- * converted into an LLMError with type "timeout". External cancellations
164
- * continue to propagate as aborted errors. The internal timer is always
165
- * cleared afterward, whether the function succeeds, fails, or is aborted,
166
- * so nothing is left running in the background.
167
- */
168
- async function withTimeout(fn, timeoutMs, externalSignal) {
169
- const controller = new AbortController();
170
- const timer = setTimeout(() => {
171
- controller.abort();
172
- }, timeoutMs);
173
- const signal = externalSignal ? AbortSignal.any([externalSignal, controller.signal]) : controller.signal;
174
- try {
175
- return await fn(signal);
176
- } catch (err) {
177
- if (controller.signal.aborted && !externalSignal?.aborted && err instanceof DOMException && err.name === "AbortError") throw new LLMError("Request timed out", "timeout");
178
- throw err;
179
- } finally {
180
- clearTimeout(timer);
181
- }
182
- }
183
- /**
184
- * Default cap (ms) for both exponential backoff and honored Retry-After
185
- * values, so a misbehaving/adversarial Retry-After can't stall a caller
186
- * indefinitely
187
- */
188
- const DEFAULT_MAX_DELAY_MS = 1e4;
189
- /**
190
- * Looks inside an unknown error value for a Retry-After header and
191
- * converts it to milliseconds. Checks `.headers` first (fetch-style,
192
- * Headers-like with `.get()`), then `.response.headers` (axios-style,
193
- * plain object) since different client libraries surface headers
194
- * differently. Supports both the delta-seconds form ("30") and the
195
- * HTTP-date form ("Wed, 21 Oct 2015 07:28:00 GMT"). The result is capped
196
- * at maxDelayMs. Returns undefined when no usable Retry-After is present
197
- */
198
- function extractRetryAfterMs(err, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
199
- if (!err || typeof err !== "object") return void 0;
200
- const error = err;
201
- const headers = error.headers ?? error.response?.headers;
202
- if (!headers || typeof headers !== "object") return void 0;
203
- const getter = headers;
204
- const raw = typeof getter.get === "function" ? getter.get("Retry-After") : Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.at(1);
205
- if (typeof raw !== "string" || raw.trim() === "") return void 0;
206
- const trimmed = raw.trim();
207
- if (/^\d+$/.test(trimmed)) return Math.max(0, Math.min(Number(trimmed) * 1e3, maxDelayMs));
208
- const dateMs = Date.parse(trimmed);
209
- if (!Number.isNaN(dateMs)) return Math.max(0, Math.min(dateMs - Date.now(), maxDelayMs));
210
- return void 0;
211
- }
212
- /** Converts any thrown value into a well-typed LLMError. */
213
- function normalizeError(error, signal) {
214
- if (signal?.aborted) return new LLMError("LLM request aborted", "aborted");
215
- if (error instanceof LLMError) return error;
216
- const status = extractStatus(error);
217
- const retryAfterMs = extractRetryAfterMs(error);
218
- if (status !== void 0) return new LLMError("LLM request failed", "api", status, void 0, error, retryAfterMs);
219
- return new LLMError("LLM request failed", "unknown", void 0, void 0, error, retryAfterMs);
220
- }
221
- /**
222
- * Exponential backoff with jitter, capped at maxDelayMs.
223
- * Jitter avoids thundering-herd retries when many callers back off in lockstep,
224
- * the cap prevents unbounded delays when maxRetries is high
225
- */
226
- function getBackoffDelay(baseDelayMs, attempt, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
227
- const exp = Math.min(baseDelayMs * 2 ** attempt, maxDelayMs);
228
- return exp / 2 + Math.random() * (exp / 2);
229
- }
230
- /**
231
- * Pauses execution for the given delay before a retry attempt. If an
232
- * abort signal is provided and it fires while waiting, the pending
233
- * timer is cancelled immediately and the wait rejects right away with
234
- * an aborted error instead of continuing to sit idle until the delay
235
- * would have finished on its own
236
- */
237
- async function waitForRetry(delay, signal) {
238
- await new Promise((resolve, reject) => {
239
- const onAbort = () => {
240
- clearTimeout(timer);
241
- reject(new LLMError("Operation aborted", "aborted"));
242
- };
243
- const timer = setTimeout(() => {
244
- signal?.removeEventListener("abort", onAbort);
245
- resolve();
246
- }, delay);
247
- signal?.addEventListener("abort", onAbort, { once: true });
248
- });
249
- }
250
- /**
251
- * Runs `getResult` after reserving usage, if a `reserveUsage` hook was
252
- * provided. `refundUsage` fires only if a reservation was actually made.
253
- * `onRefundError` is called (instead of throwing) whenever a refund attempt
254
- * itself fails, so a broken refund hook never masks the original error.
255
- */
256
- async function withReservedUsage(params, coalesced, getResult, signal, onRefundError) {
257
- if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
258
- let reserved = false;
259
- try {
260
- if (params.reserveUsage) {
261
- await params.reserveUsage({
262
- coalesced,
263
- signal
264
- });
265
- reserved = true;
266
- }
267
- } catch (error) {
268
- if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
269
- throw new LLMError(error instanceof Error ? error.message : "Usage reservation failed", "quota_exceeded", void 0, void 0, error);
270
- }
271
- const refund = async (logMessage) => {
272
- try {
273
- await params.refundUsage?.({
274
- coalesced,
275
- signal
276
- });
277
- } catch (refundError) {
278
- onRefundError(logMessage, refundError);
279
- }
280
- };
281
- if (signal?.aborted) {
282
- if (reserved) await refund("[VernLLM] refundUsage failed after abort");
283
- throw new LLMError("LLM request aborted", "aborted");
284
- }
285
- let result;
286
- try {
287
- result = await getResult();
288
- } catch (error) {
289
- if (reserved) await refund("[VernLLM] refundUsage failed");
290
- throw error;
291
- }
292
- if (signal?.aborted) {
293
- if (reserved) await refund("[VernLLM] refundUsage failed after abort");
294
- throw new LLMError("LLM request aborted", "aborted");
295
- }
296
- return result;
297
- }
298
-
299
- //#endregion
300
- //#region src/logger.ts
301
- /**
302
- * Default logger. `debug` is gated by the `debug` option on VernLLM
303
- * warn/error always fire since they indicate real problems (retries, cache failures)
304
- */
305
- var ConsoleLogger = class {
306
- constructor(debugEnabled) {
307
- this.debugEnabled = debugEnabled;
308
- }
309
- debug(message) {
310
- if (this.debugEnabled) console.debug(message);
311
- }
312
- warn(message) {
313
- console.warn(message);
314
- }
315
- error(message, meta) {
316
- console.error(message, meta ?? "");
317
- }
318
- };
319
-
320
46
  //#endregion
321
47
  //#region src/types/cache.ts
322
48
  /**
@@ -428,91 +154,1495 @@ var TieredCacheAdapter = class {
428
154
  };
429
155
 
430
156
  //#endregion
431
- //#region src/vernLLM.ts
157
+ //#region src/types/tools.ts
432
158
  /**
433
- * A resilient layer around an LLM chat completions client, this is VernLLM!
434
- *
435
- * Adds retry with backoff/jitter, per-attempt timeouts, an optional circuit breaker,
436
- * JSON parsing with optional schema validation, usage tracking, and an
437
- * optional response cache, all configurable, all opt-in beyond sensible
438
- * defaults.
159
+ * Runtime-safe check for whether a `call()` result is a `tool_calls`
160
+ * result. Prefer this over relying on TypeScript's static narrowing
161
+ * whenever `params` passed to `call()` wasn't a literal with `tools`
162
+ * inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
163
+ * that case TS may have typed the result as plain `T` even though it's
164
+ * actually a `CallWithToolsResult<T>` at runtime, and this check works
165
+ * either way.
439
166
  */
440
- var VernLLM = class {
441
- client;
442
- model;
443
- maxRetries;
444
- timeoutMs;
445
- baseDelayMs;
446
- defaultMaxTokens;
447
- cache;
448
- nonRetryableStatus;
449
- inFlight = new Map();
450
- parseJson;
451
- onUsage;
452
- logger;
453
- breaker;
454
- /**
455
- * @param options - Client, model, and tunables. Notable defaults:
456
- * `maxRetries` 1, `timeoutMs` 25000, `baseDelayMs` 500 (exponential backoff
457
- * base), `defaultMaxTokens` 1000, `cache` an in-memory adapter,
458
- * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
459
- */
460
- constructor(options) {
461
- this.client = options.client;
462
- this.model = options.model;
463
- this.maxRetries = options.maxRetries ?? 1;
464
- this.timeoutMs = options.timeoutMs ?? 25e3;
465
- this.baseDelayMs = options.baseDelayMs ?? 500;
466
- this.defaultMaxTokens = options.defaultMaxTokens ?? 1e3;
467
- this.cache = options.cache ?? new InMemoryCacheAdapter();
468
- this.nonRetryableStatus = options.nonRetryableStatus ?? [
469
- 400,
470
- 401,
471
- 403,
472
- 404,
473
- 422
474
- ];
475
- this.parseJson = options.parseJson ?? defaultParseJson;
476
- this.onUsage = options.onUsage;
477
- this.logger = options.logger ?? new ConsoleLogger(options.debug ?? false);
478
- this.breaker = options.circuitBreaker ? new CircuitBreaker(options.circuitBreaker === true ? void 0 : options.circuitBreaker) : void 0;
479
- }
167
+ function isToolCallResult(result) {
168
+ return typeof result === "object" && result !== null && "type" in result && result.type === "tool_calls" && Array.isArray(result.toolCalls);
169
+ }
170
+
171
+ //#endregion
172
+ //#region src/types/fallback.ts
173
+ /** Tool contract failures are the model ignoring the request, not a sick provider: repeating it elsewhere can't help. */
174
+ const TOOL_CONTRACT_CODES = new Set(["unknown_tool", "duplicate_tool_call_id"]);
175
+ /**
176
+ * The default `fallbackOn` policy. Exported so a caller can wrap rather
177
+ * than replace it, e.g. `fallbackOn: (e, ctx) => myCheck(e) ? 'stop' : defaultFallbackOn(e, ctx)`.
178
+ */
179
+ const defaultFallbackOn = (error) => {
180
+ if (error.type === "parse" || error.type === "validation" || error.type === "aborted") return "stop";
181
+ if (error.type === "quota_exceeded") return "stop";
182
+ if (error.code && TOOL_CONTRACT_CODES.has(error.code)) return "stop";
183
+ return "next";
184
+ };
185
+ /**
186
+ * Thrown when the chain gives up, whether because the last target failed
187
+ * or `fallbackOn` chose to stop early. Carries each attempt in order so
188
+ * an outage across providers stays debuggable without reproducing it.
189
+ * Extends `LLMError` so `isLLMError` and any `instanceof LLMError` check
190
+ * still passes, inheriting the last failure's `type` so existing
191
+ * type-based handling keeps working on a fallback-exhausted error too.
192
+ */
193
+ var FallbackExhaustedError = class extends LLMError {
194
+ constructor(attempts) {
195
+ const last = attempts[attempts.length - 1]?.error;
196
+ super(`${attempts.length} provider${attempts.length === 1 ? "" : "s"} attempted and failed: ${attempts.map((a) => `${a.provider}(${a.error.type})`).join(" then ")}`, last?.type ?? "unknown", last?.status, void 0, last, void 0, "fallback_exhausted");
197
+ this.attempts = attempts;
198
+ }
199
+ };
200
+
201
+ //#endregion
202
+ //#region src/internal/execution/usage.utils.ts
203
+ /**
204
+ * Calls `params.reserveUsage`, if present, mapping any failure to a
205
+ * `quota_exceeded` LLMError (or an aborted error, if the signal fired
206
+ * during reservation). Returns whether a reservation was actually made,
207
+ * so callers know whether a later refund is needed. Shared by
208
+ * `withReservedUsage` and `withReservedUsageForStream`, which differ only
209
+ * in whether `coalesced` is caller-supplied or always `false`.
210
+ */
211
+ async function reserve(params, coalesced, signal) {
212
+ if (!params.reserveUsage) return false;
213
+ try {
214
+ await params.reserveUsage({
215
+ coalesced,
216
+ signal
217
+ });
218
+ return true;
219
+ } catch (error) {
220
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
221
+ throw new LLMError(error instanceof Error ? error.message : "Usage reservation failed", "quota_exceeded", void 0, void 0, error);
222
+ }
223
+ }
224
+ /**
225
+ * Builds a `(logMessage) => Promise<void>` refund function bound to the
226
+ * given hooks/coalesced/signal, reporting (instead of throwing) any error
227
+ * the refund hook itself raises, so a broken refund hook never masks the
228
+ * original error it was called to clean up after.
229
+ */
230
+ function makeRefund(params, coalesced, signal, onRefundError) {
231
+ return async (logMessage) => {
232
+ try {
233
+ await params.refundUsage?.({
234
+ coalesced,
235
+ signal
236
+ });
237
+ } catch (refundError) {
238
+ onRefundError(logMessage, refundError);
239
+ }
240
+ };
241
+ }
242
+ /**
243
+ * Runs `getResult` after reserving usage, if a `reserveUsage` hook was
244
+ * provided. `refundUsage` fires only if a reservation was actually made.
245
+ * `onRefundError` is called (instead of throwing) whenever a refund attempt
246
+ * itself fails, so a broken refund hook never masks the original error.
247
+ */
248
+ async function withReservedUsage(params, coalesced, getResult, signal, onRefundError) {
249
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
250
+ const reserved = await reserve(params, coalesced, signal);
251
+ const refund = makeRefund(params, coalesced, signal, onRefundError);
252
+ if (signal?.aborted) {
253
+ if (reserved) await refund("[VernLLM] refundUsage failed after abort");
254
+ throw new LLMError("LLM request aborted", "aborted");
255
+ }
256
+ let result;
257
+ try {
258
+ result = await getResult();
259
+ } catch (error) {
260
+ if (reserved) await refund("[VernLLM] refundUsage failed");
261
+ throw error;
262
+ }
263
+ if (signal?.aborted) {
264
+ if (reserved) await refund("[VernLLM] refundUsage failed after abort");
265
+ throw new LLMError("LLM request aborted", "aborted");
266
+ }
267
+ return result;
268
+ }
269
+ /**
270
+ * Streaming counterpart to `withReservedUsage`. `withReservedUsage` assumes
271
+ * `getResult()` settling *is* the operation's final outcome, awaiting it
272
+ * synchronously before reserve/refund resolve. Streaming can't satisfy that:
273
+ * `call()` must return `{ chunks, finalResult }` as soon as the stream
274
+ * opens, well before the real outcome (validation, schema/tool-call checks)
275
+ * is known.
276
+ *
277
+ * Reserves usage before `openStream` runs, same failure mode as the
278
+ * non-streaming path if `reserveUsage` itself throws (mapped to
279
+ * `quota_exceeded`, nothing opened). If `openStream` itself throws (stream
280
+ * never opened), refunds synchronously and rethrows, exactly like
281
+ * `withReservedUsage` does today. If it succeeds, returns `{ chunks,
282
+ * finalResult }` immediately, refund/report is deferred onto
283
+ * `finalResult`'s continuation, since that's the only point the real
284
+ * outcome is known. This means `onUsageFailure` (and any refund) can fire
285
+ * well after this function itself has returned.
286
+ */
287
+ async function withReservedUsageForStream(params, openStream, signal, onRefundError) {
288
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
289
+ const reserved = await reserve(params, false, signal);
290
+ const refund = makeRefund(params, false, signal, onRefundError);
291
+ if (signal?.aborted) {
292
+ if (reserved) await refund("[VernLLM] refundUsage failed after abort");
293
+ throw new LLMError("LLM request aborted", "aborted");
294
+ }
295
+ let opened;
296
+ try {
297
+ opened = await openStream();
298
+ } catch (error) {
299
+ if (reserved) await refund("[VernLLM] refundUsage failed after stream-open failure");
300
+ throw error;
301
+ }
302
+ const finalResult = opened.finalResult.then((value) => value, async (error) => {
303
+ if (reserved) await refund("[VernLLM] refundUsage failed after stream error");
304
+ throw error;
305
+ });
306
+ finalResult.catch(() => {});
307
+ return {
308
+ chunks: opened.chunks,
309
+ finalResult
310
+ };
311
+ }
312
+
313
+ //#endregion
314
+ //#region src/internal/cache/replay.utils.ts
315
+ /**
316
+ * Converts an already-known cache value back into a plausible "text" form
317
+ * for a one-shot replay chunk: passed through unchanged if it's already a
318
+ * string (the `jsonMode: false` case), otherwise `JSON.stringify`'d (the
319
+ * `jsonMode: true` case, where the cached value is the *parsed* result, not
320
+ * the original raw text). This is a reasonable reconstruction, not a
321
+ * byte-identical replay of whatever text the model originally streamed,
322
+ * good enough for `for await (const c of chunks)` call sites that don't
323
+ * branch on hit vs. miss, which is the only thing a cache-hit replay needs
324
+ * to support.
325
+ */
326
+ function toReplayText(value) {
327
+ return typeof value === "string" ? value : JSON.stringify(value) ?? "";
328
+ }
329
+ /**
330
+ * Builds a trivially-exhausted one-shot `chunks` iterable from an
331
+ * already-known value, used for a `cachedCall` cache hit, where there's no
332
+ * live generation to relay (see `VernLLM.cachedCall`'s docs). No `usage`
333
+ * chunk is emitted: a cache hit spent no real tokens, so there's nothing to
334
+ * report, matching how non-streaming `cachedCall` never calls `onUsage` on
335
+ * a hit either.
336
+ *
337
+ * `hasTools` must reflect whether the *original* call that produced this
338
+ * cached value had `tools` set, that's what determines whether `value` is
339
+ * `T` directly or a `CallWithToolsResult<T>` wrapper, and it isn't
340
+ * something that can be reliably guessed from the value's shape alone
341
+ * (a `schema`-validated `T` could coincidentally look like a
342
+ * `CallWithToolsResult`).
343
+ */
344
+ function buildReplayChunks(value, hasTools) {
345
+ const items = [];
346
+ if (hasTools) {
347
+ const result = value;
348
+ if (result.type === "tool_calls") {
349
+ result.toolCalls.forEach((toolCall, index) => {
350
+ items.push({
351
+ type: "tool_call_delta",
352
+ index,
353
+ id: toolCall.id,
354
+ name: toolCall.name,
355
+ argsDelta: JSON.stringify(toolCall.arguments ?? {}),
356
+ complete: true
357
+ });
358
+ });
359
+ if (result.content) items.push({
360
+ type: "text-delta",
361
+ delta: result.content
362
+ });
363
+ } else items.push({
364
+ type: "text-delta",
365
+ delta: toReplayText(result.content)
366
+ });
367
+ } else items.push({
368
+ type: "text-delta",
369
+ delta: toReplayText(value)
370
+ });
371
+ return { async *[Symbol.asyncIterator]() {
372
+ for (const item of items) yield item;
373
+ } };
374
+ }
375
+ /**
376
+ * Streaming counterpart to `buildReplayChunks` for a `cachedCall` that
377
+ * *joined* an already-in-flight call for the same key rather than
378
+ * triggering one itself (see `runCachedStream`'s in-flight-coalescing
379
+ * path): there's no live stream to relay (it isn't this call's stream to
380
+ * relay, see the joiner-path comment in `runCachedStream`), but there's
381
+ * also no value yet, only a pending promise for one. Waits for `promise`,
382
+ * then delegates to `buildReplayChunks`. If `promise` rejects, iterating
383
+ * `chunks` throws that same error, consistent with how a live stream's
384
+ * `chunks` throws on a mid-stream failure.
385
+ */
386
+ function buildReplayChunksFromPromise(promise, hasTools) {
387
+ return { async *[Symbol.asyncIterator]() {
388
+ const value = await promise;
389
+ yield* buildReplayChunks(value, hasTools);
390
+ } };
391
+ }
392
+
393
+ //#endregion
394
+ //#region src/internal/cache/cacheOrchestrator.ts
395
+ /**
396
+ * Owns cache key resolution, cache reads/writes, and in-flight coalescing
397
+ * for concurrent misses on the same key. Doesn't know about `CallExecutor`,
398
+ * retries, or providers at all: `fn`/`openStream` are opaque callbacks
399
+ * (`VernLLM.cachedCall` passes `() => this.call(...)`), so this class only
400
+ * needs the cache adapter and a logger. Extracted from `VernLLM` since
401
+ * caching and per-target call mechanics are independent concerns that
402
+ * happened to live on the same class.
403
+ */
404
+ var CacheOrchestrator = class {
405
+ inFlight = new Map();
406
+ constructor(cache, logger) {
407
+ this.cache = cache;
408
+ this.logger = logger;
409
+ }
480
410
  /** Resolves a cache key through the adapter when it supports normalization. */
481
411
  async resolveCacheKey(key) {
482
412
  return this.cache.resolveKey ? await this.cache.resolveKey(key) : key;
483
413
  }
484
414
  /**
485
- * Makes a single logical LLM call, retrying on failure per the configured
486
- * policy. Fails fast if the breaker is open or the signal is already
487
- * aborted. On exhausting retries, records a breaker failure and rejects
488
- * with a normalized LLMError.
415
+ * Removes a cached response by key when the configured cache adapter
416
+ * supports deletion. Cache invalidation is the caller's responsibility;
417
+ * only the application knows when cached data is stale.
418
+ */
419
+ async deleteCache(key) {
420
+ if (!this.cache.delete) return;
421
+ await this.cache.delete(await this.resolveCacheKey(key));
422
+ }
423
+ /** Logs a failed refundUsage attempt via the configured logger. */
424
+ logRefundError(logMessage, error) {
425
+ this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
426
+ }
427
+ /**
428
+ * Internal cache primitive around caller-supplied logic. Concurrent misses
429
+ * for the same `cacheKey` share a single in-flight call, avoiding cache
430
+ * stampedes.
431
+ *
432
+ * Backs the public `VernLLM.cachedCall()`, which always composes this
433
+ * with `call()` so cached results get the same retry/timeout/
434
+ * circuit-breaker guarantees as any other LLM call.
435
+ *
436
+ * @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
437
+ * miss, typically `() => this.call(...)`), and optional
438
+ * `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
439
+ * @returns The cached value on a hit, or the result of `fn()` on a miss.
440
+ */
441
+ async runCached(params) {
442
+ const resolvedKey = await this.resolveCacheKey(params.cacheKey);
443
+ const resolvedParams = resolvedKey === params.cacheKey ? params : {
444
+ ...params,
445
+ cacheKey: resolvedKey
446
+ };
447
+ const cached = await this.cache.get(resolvedKey);
448
+ if (cached.hit) return cached.value;
449
+ const existing = this.inFlight.get(resolvedKey);
450
+ if (existing) return withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
451
+ return this.registerTrigger(resolvedParams);
452
+ }
453
+ /** Starts the shared fn() call for a cache miss and tracks it in the in-flight map until it settles. */
454
+ registerTrigger(params) {
455
+ const resultPromise = withReservedUsage(params, false, () => this.runAndCache(params), params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
456
+ this.inFlight.set(params.cacheKey, resultPromise);
457
+ resultPromise.catch(() => {}).finally(() => {
458
+ this.inFlight.delete(params.cacheKey);
459
+ });
460
+ return resultPromise;
461
+ }
462
+ /** Runs `fn` and writes its result to the cache. */
463
+ async runAndCache(params) {
464
+ const result = await params.fn();
465
+ try {
466
+ await this.cache.set(params.cacheKey, result, params.ttl);
467
+ } catch (error) {
468
+ this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
469
+ }
470
+ return result;
471
+ }
472
+ /**
473
+ * Streaming counterpart to `runCached`. Three cases:
474
+ *
475
+ * - Hit: no live generation to relay. Returns immediately with
476
+ * `finalResult` resolved to the cached value and a one-shot `chunks`
477
+ * replay built from it, so `for await (const c of chunks)` call sites
478
+ * work identically on a hit or a miss. No usage hooks fire, since
479
+ * nothing was actually spent.
480
+ * - Miss, nothing else in flight for this key: delegates to
481
+ * `registerStreamTrigger`, which opens the stream and relays its
482
+ * `chunks` live.
483
+ * - Miss, but another call for the same key is already in flight: this
484
+ * call has no live chunks of its own to relay, so it's treated like a
485
+ * delayed hit. `finalResult` shares the trigger's in-flight promise
486
+ * (the same in-flight map non-streaming `runCached` uses, so
487
+ * streaming and non-streaming calls for the same key coalesce
488
+ * against each other too), and `chunks` is a one-shot replay built
489
+ * once that promise resolves.
490
+ */
491
+ async runCachedStream(params, hasTools) {
492
+ const resolvedKey = await this.resolveCacheKey(params.cacheKey);
493
+ const resolvedParams = resolvedKey === params.cacheKey ? params : {
494
+ ...params,
495
+ cacheKey: resolvedKey
496
+ };
497
+ const cached = await this.cache.get(resolvedKey);
498
+ if (cached.hit) {
499
+ const value = cached.value;
500
+ return {
501
+ chunks: buildReplayChunks(value, hasTools),
502
+ finalResult: Promise.resolve(value)
503
+ };
504
+ }
505
+ const existing = this.inFlight.get(resolvedKey);
506
+ if (existing) {
507
+ const finalResult = withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
508
+ finalResult.catch(() => {});
509
+ return {
510
+ chunks: buildReplayChunksFromPromise(finalResult, hasTools),
511
+ finalResult
512
+ };
513
+ }
514
+ return this.registerStreamTrigger(resolvedParams);
515
+ }
516
+ /**
517
+ * Opens the shared stream for a cache miss and tracks its settled value
518
+ * in the in-flight map until it resolves or rejects. Writes to the cache
519
+ * on success only, matching `runAndCache`.
520
+ *
521
+ * Registers the in-flight promise synchronously, before anything async
522
+ * runs, so a concurrent `cachedCall` for the same key always sees it in
523
+ * time to join instead of triggering its own stream. Settlement is
524
+ * wired onto the whole `withReservedUsageForStream` call rather than a
525
+ * line inside its callback, so any failure point (reserving usage,
526
+ * opening the stream, or the stream itself) reliably settles the
527
+ * in-flight entry instead of leaving it stuck.
528
+ */
529
+ registerStreamTrigger(params) {
530
+ let resolveInFlight;
531
+ let rejectInFlight;
532
+ const inFlightResult = new Promise((resolve, reject) => {
533
+ resolveInFlight = resolve;
534
+ rejectInFlight = reject;
535
+ });
536
+ this.inFlight.set(params.cacheKey, inFlightResult);
537
+ inFlightResult.catch(() => {}).finally(() => {
538
+ this.inFlight.delete(params.cacheKey);
539
+ });
540
+ const streamPromise = withReservedUsageForStream(params, async () => {
541
+ const opened = await params.openStream();
542
+ const trackedResult = opened.finalResult.then(async (value) => {
543
+ try {
544
+ await this.cache.set(params.cacheKey, value, params.ttl);
545
+ } catch (error) {
546
+ this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
547
+ }
548
+ return value;
549
+ }, (error) => {
550
+ throw error;
551
+ });
552
+ return {
553
+ chunks: opened.chunks,
554
+ finalResult: trackedResult
555
+ };
556
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
557
+ streamPromise.then((opened) => {
558
+ opened.finalResult.then(resolveInFlight, rejectInFlight);
559
+ }, (error) => {
560
+ rejectInFlight(error);
561
+ });
562
+ return streamPromise;
563
+ }
564
+ };
565
+
566
+ //#endregion
567
+ //#region src/circuitBreaker.ts
568
+ function newBucket() {
569
+ return {
570
+ state: "closed",
571
+ consecutiveFailures: 0,
572
+ openedAt: 0,
573
+ trialInFlight: false
574
+ };
575
+ }
576
+ /** Key a bucket lookup falls into when the call omitted `model` under `isolateByModel`. */
577
+ const UNLABELED_MODEL = "";
578
+ /**
579
+ * Per retry VernLLM-instance circuit breaker. Tracks consecutive failures across
580
+ * calls. Once the threshold is hit, short-circuits new calls with an
581
+ * LLMError('circuit_open') instead of hitting the provider, until the
582
+ * cooldown elapses and a single trial call is allowed through
583
+ */
584
+ var CircuitBreaker = class {
585
+ threshold;
586
+ cooldownMs;
587
+ onStateChange;
588
+ isolateByModel;
589
+ sharedBucket = newBucket();
590
+ bucketsByModel = new Map();
591
+ constructor(options = {}) {
592
+ this.threshold = options.threshold ?? 5;
593
+ this.cooldownMs = options.cooldownMs ?? 3e4;
594
+ this.onStateChange = options.onStateChange;
595
+ this.isolateByModel = options.isolateByModel ?? false;
596
+ }
597
+ /** Returns the bucket for a model if one already exists, without allocating. */
598
+ lookupBucket(model) {
599
+ if (!this.isolateByModel) return this.sharedBucket;
600
+ const key = model ?? UNLABELED_MODEL;
601
+ return this.bucketsByModel.get(key);
602
+ }
603
+ /** Creates and stores a bucket for a model when the first mutation needs one. */
604
+ ensureBucketFor(model) {
605
+ if (!this.isolateByModel) return this.sharedBucket;
606
+ const key = model ?? UNLABELED_MODEL;
607
+ let bucket = this.bucketsByModel.get(key);
608
+ if (!bucket) {
609
+ bucket = newBucket();
610
+ this.bucketsByModel.set(key, bucket);
611
+ }
612
+ return bucket;
613
+ }
614
+ /** Every state mutation routes through here, so `onStateChange` fires exactly once per real change. */
615
+ transition(bucket, to, model) {
616
+ if (to === bucket.state) return;
617
+ const from = bucket.state;
618
+ bucket.state = to;
619
+ this.onStateChange?.(from, to, bucket.consecutiveFailures, model);
620
+ }
621
+ /**
622
+ * Throws if the circuit is open and the cooldown hasn't elapsed, or if
623
+ * the circuit is half-open and a trial call is already in flight.
624
+ * Otherwise, if the circuit just became eligible for a trial (cooldown
625
+ * elapsed, or half-open with no trial currently running), this call
626
+ * becomes that trial
627
+ */
628
+ assertClosed(model) {
629
+ const bucket = this.ensureBucketFor(model);
630
+ if (bucket.state === "closed") return;
631
+ if (bucket.state === "open") {
632
+ const elapsed = Date.now() - bucket.openedAt;
633
+ if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open, provider has failed ${bucket.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
634
+ bucket.trialInFlight = true;
635
+ this.transition(bucket, "half-open", model);
636
+ return;
637
+ }
638
+ if (bucket.trialInFlight) throw new LLMError("Circuit half-open. A trial request is already in flight. Try again shortly.", "circuit_open");
639
+ bucket.trialInFlight = true;
640
+ }
641
+ recordSuccess(model) {
642
+ const bucket = this.lookupBucket(model);
643
+ if (!bucket) return;
644
+ bucket.consecutiveFailures = 0;
645
+ bucket.trialInFlight = false;
646
+ this.transition(bucket, "closed", model);
647
+ if (this.isolateByModel && bucket.state === "closed" && bucket.consecutiveFailures === 0) this.bucketsByModel.delete(model ?? UNLABELED_MODEL);
648
+ }
649
+ recordFailure(model) {
650
+ const bucket = this.ensureBucketFor(model);
651
+ bucket.consecutiveFailures += 1;
652
+ bucket.trialInFlight = false;
653
+ if (bucket.state === "half-open") {
654
+ bucket.openedAt = Date.now();
655
+ this.transition(bucket, "open", model);
656
+ return;
657
+ }
658
+ if (bucket.consecutiveFailures >= this.threshold) {
659
+ bucket.openedAt = Date.now();
660
+ this.transition(bucket, "open", model);
661
+ }
662
+ }
663
+ /**
664
+ * With `isolateByModel` off (the default), `model` is ignored and the
665
+ * one shared circuit's state is returned, unchanged from every version
666
+ * before this option existed. With `isolateByModel` on, returns that
667
+ * model's own state, `'closed'` for a model never seen yet, same as a
668
+ * fresh breaker.
669
+ */
670
+ getState(model) {
671
+ return this.lookupBucket(model)?.state ?? "closed";
672
+ }
673
+ };
674
+
675
+ //#endregion
676
+ //#region src/internal/circuitBreaker.utils.ts
677
+ /**
678
+ * Builds a `(event) => void` reporter that no-ops when `onEvent` is unset,
679
+ * and otherwise calls it, swallowing and logging any error the handler
680
+ * throws so a broken `onEvent` can't break the call that triggered it.
681
+ * Shared by `buildCircuitBreaker` (which needs to report before any
682
+ * executor exists) and `CallExecutor.reportEvent`, kept independent of the
683
+ * executor for that reason.
684
+ */
685
+ function makeEventReporter(onEvent, logger) {
686
+ return (event) => {
687
+ if (!onEvent) return;
688
+ try {
689
+ onEvent(event);
690
+ } catch (error) {
691
+ logger.error("[VernLLM] onEvent failed", { message: error instanceof Error ? error.message : "unknown" });
692
+ }
693
+ };
694
+ }
695
+ /**
696
+ * Builds the optional circuit breaker for one provider target, wiring its
697
+ * `onStateChange` to emit a `circuit_state` event and chain any
698
+ * caller-supplied `onStateChange`. Returns `undefined` when
699
+ * `circuitBreakerOption` is falsy, matching the option's own semantics.
700
+ *
701
+ * Lives outside `CallExecutor` (and outside `VernLLM`, once this were
702
+ * inlined) because the breaker has to exist *before* the executor it's
703
+ * passed into, so its construction can't be an executor concern.
704
+ * `onEvent` is called directly rather than through the executor for the
705
+ * same reason: nothing executor-shaped exists yet at this point.
706
+ *
707
+ * Takes the specific fields it needs (rather than a full `VernLLMOptions`)
708
+ * so it works identically for the primary target and for each fallback
709
+ * target, which carry their own `circuitBreaker` override alongside the
710
+ * shared `onEvent`.
711
+ */
712
+ function buildCircuitBreaker(circuitBreakerOption, providerName, defaultModel, onEvent, logger) {
713
+ if (!circuitBreakerOption) return void 0;
714
+ const breakerOptions = typeof circuitBreakerOption === "object" ? circuitBreakerOption : void 0;
715
+ const userOnStateChange = breakerOptions?.onStateChange;
716
+ const reportEvent = makeEventReporter(onEvent, logger);
717
+ return new CircuitBreaker({
718
+ ...breakerOptions,
719
+ onStateChange: (from, to, consecutiveFailures, model) => {
720
+ reportEvent({
721
+ kind: "circuit_state",
722
+ provider: providerName,
723
+ model: model ?? defaultModel,
724
+ from,
725
+ to,
726
+ consecutiveFailures
727
+ });
728
+ if (!userOnStateChange) return;
729
+ try {
730
+ userOnStateChange(from, to, consecutiveFailures, model);
731
+ } catch (error) {
732
+ logger.error("[VernLLM] circuitBreaker.onStateChange failed", { message: error instanceof Error ? error.message : "unknown" });
733
+ }
734
+ }
735
+ });
736
+ }
737
+
738
+ //#endregion
739
+ //#region src/internal/execution/retry.utils.ts
740
+ /**
741
+ * Default cap (ms) for both exponential backoff and honored Retry-After
742
+ * values, so a misbehaving/adversarial Retry-After can't stall a caller
743
+ * indefinitely
744
+ */
745
+ const DEFAULT_MAX_DELAY_MS = 1e4;
746
+ /**
747
+ * `setTimeout` silently clamps any delay above this (~24.8 days) or
748
+ * `Infinity` down to ~1ms instead of erroring, so a caller passing
749
+ * `Infinity` as "no timeout" gets the opposite of what they asked for.
750
+ * Both timeout helpers below guard against this explicitly.
751
+ */
752
+ const MAX_SETTIMEOUT_MS = 2147483647;
753
+ /**
754
+ * Resolves a timeout value to the number `setTimeout` should actually use,
755
+ * or `undefined` when the timeout should be treated as disabled (0,
756
+ * negative, or `Infinity`). Returning the resolved value directly, rather
757
+ * than a boolean, lets callers narrow `number | undefined` to `number`
758
+ * without an `as number` cast.
759
+ */
760
+ function resolveActiveTimeoutMs(ms) {
761
+ return !ms || ms <= 0 || ms === Infinity ? void 0 : ms;
762
+ }
763
+ /** Caps a timeout at the largest delay `setTimeout` actually honors. */
764
+ function clampTimeoutMs(ms) {
765
+ return Math.min(ms, MAX_SETTIMEOUT_MS);
766
+ }
767
+ /**
768
+ * Runs an async function and cancels it if it takes longer than the given
769
+ * timeout. Creates an internal abort controller that fires after the
770
+ * timeout elapses, and combines it with any external signal the caller
771
+ * passed in so either one can cancel the underlying call. If the internal
772
+ * timeout triggers and the underlying operation aborts, the error is
773
+ * converted into an LLMError with type "timeout". External cancellations
774
+ * continue to propagate as aborted errors. The internal timer is always
775
+ * cleared afterward, whether the function succeeds, fails, or is aborted,
776
+ * so nothing is left running in the background.
777
+ *
778
+ * `timeoutMs` of `Infinity` (or any value beyond what `setTimeout` can
779
+ * represent) disables the timeout rather than firing almost immediately.
780
+ */
781
+ async function withTimeout(fn, timeoutMs, externalSignal) {
782
+ const controller = new AbortController();
783
+ const activeTimeoutMs = resolveActiveTimeoutMs(timeoutMs);
784
+ const timer = activeTimeoutMs === void 0 ? void 0 : setTimeout(() => {
785
+ controller.abort();
786
+ }, clampTimeoutMs(activeTimeoutMs));
787
+ const signal = externalSignal ? AbortSignal.any([externalSignal, controller.signal]) : controller.signal;
788
+ try {
789
+ return await fn(signal);
790
+ } catch (err) {
791
+ if (controller.signal.aborted && !externalSignal?.aborted && err instanceof DOMException && err.name === "AbortError") throw new LLMError("Request timed out", "timeout");
792
+ throw err;
793
+ } finally {
794
+ clearTimeout(timer);
795
+ }
796
+ }
797
+ /**
798
+ * Races one `iterator.next()` call against a per-call idle timer, to
799
+ * bound the gap *between* chunks (unlike `withTimeout`, which only bounds
800
+ * opening the stream and its first chunk). Without this, a connection
801
+ * that streams one chunk then hangs would never fail.
802
+ *
803
+ * `timeoutMs` of 0/undefined/`Infinity` disables the check. Otherwise
804
+ * rejects with `LLMError('timeout')` if `next()` doesn't settle in time.
805
+ * The clock resets on every call, so the window is measured from the most
806
+ * recent chunk, not from stream start.
807
+ *
808
+ * `onIdle`, if given, is called the moment the timer fires (before the
809
+ * rejection), so callers can abort the underlying transport instead of
810
+ * just walking away from an unread promise. `logger`, if given, records a
811
+ * debug line if `next()` still settles *after* the idle timeout already
812
+ * rejected. `resolve`/`reject` on an already-settled promise is otherwise
813
+ * a silent no-op, so without this the late chunk (possibly the final
814
+ * usage chunk) would vanish with no trace.
815
+ */
816
+ function withChunkIdleTimeout(next, timeoutMs, onIdle, logger) {
817
+ const activeTimeoutMs = resolveActiveTimeoutMs(timeoutMs);
818
+ if (activeTimeoutMs === void 0) return next();
819
+ let settled = false;
820
+ return new Promise((resolve, reject) => {
821
+ const timer = setTimeout(() => {
822
+ settled = true;
823
+ onIdle?.();
824
+ reject(new LLMError(`No stream chunk received for ${activeTimeoutMs}ms (idle timeout)`, "timeout"));
825
+ }, clampTimeoutMs(activeTimeoutMs));
826
+ next().then((result) => {
827
+ clearTimeout(timer);
828
+ if (settled) {
829
+ logger?.debug("[VernLLM] chunk resolved after idle timeout already fired; discarding");
830
+ return;
831
+ }
832
+ settled = true;
833
+ resolve(result);
834
+ }, (error) => {
835
+ clearTimeout(timer);
836
+ if (settled) {
837
+ logger?.debug("[VernLLM] chunk rejection arrived after idle timeout already fired; discarding");
838
+ return;
839
+ }
840
+ settled = true;
841
+ reject(error);
842
+ });
843
+ });
844
+ }
845
+ /**
846
+ * Looks inside an unknown error value for a Retry-After header and
847
+ * converts it to milliseconds. Checks `.headers` first (fetch-style,
848
+ * Headers-like with `.get()`), then `.response.headers` (axios-style,
849
+ * plain object) since different client libraries surface headers
850
+ * differently. Supports both the delta-seconds form ("30") and the
851
+ * HTTP-date form ("Wed, 21 Oct 2015 07:28:00 GMT"). The result is capped
852
+ * at maxDelayMs. Returns undefined when no usable Retry-After is present
853
+ */
854
+ function extractRetryAfterMs(err, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
855
+ if (!err || typeof err !== "object") return void 0;
856
+ const error = err;
857
+ const headers = error.headers ?? error.response?.headers;
858
+ if (!headers || typeof headers !== "object") return void 0;
859
+ const getter = headers;
860
+ const raw = typeof getter.get === "function" ? getter.get("Retry-After") : Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.at(1);
861
+ if (typeof raw !== "string" || raw.trim() === "") return void 0;
862
+ const trimmed = raw.trim();
863
+ if (/^\d+$/.test(trimmed)) return Math.max(0, Math.min(Number(trimmed) * 1e3, maxDelayMs));
864
+ const dateMs = Date.parse(trimmed);
865
+ if (!Number.isNaN(dateMs)) return Math.max(0, Math.min(dateMs - Date.now(), maxDelayMs));
866
+ return void 0;
867
+ }
868
+ /**
869
+ * Exponential backoff with jitter, capped at maxDelayMs.
870
+ * Jitter avoids thundering-herd retries when many callers back off in lockstep,
871
+ * the cap prevents unbounded delays when maxRetries is high
872
+ */
873
+ function getBackoffDelay(baseDelayMs, attempt, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
874
+ const exp = Math.min(baseDelayMs * 2 ** attempt, maxDelayMs);
875
+ return exp / 2 + Math.random() * (exp / 2);
876
+ }
877
+ /**
878
+ * Pauses execution for the given delay before a retry attempt. If an
879
+ * abort signal is provided and it fires while waiting, the pending
880
+ * timer is cancelled immediately and the wait rejects right away with
881
+ * an aborted error instead of continuing to sit idle until the delay
882
+ * would have finished on its own
883
+ */
884
+ async function waitForRetry(delay, signal) {
885
+ if (signal?.aborted) throw new LLMError("Operation aborted", "aborted");
886
+ await new Promise((resolve, reject) => {
887
+ const onAbort = () => {
888
+ clearTimeout(timer);
889
+ reject(new LLMError("Operation aborted", "aborted"));
890
+ };
891
+ const timer = setTimeout(() => {
892
+ signal?.removeEventListener("abort", onAbort);
893
+ resolve();
894
+ }, delay);
895
+ signal?.addEventListener("abort", onAbort, { once: true });
896
+ });
897
+ }
898
+
899
+ //#endregion
900
+ //#region src/internal/execution/errors.utils.ts
901
+ /**
902
+ * Looks inside an unknown error value and pulls out an http status code
903
+ * if one is present. Checks the status field first then the status code
904
+ * field since different client libraries use different names for this,
905
+ * falling back to AWS SDK v3's `$metadata.httpStatusCode` (e.g. Bedrock's
906
+ * `ThrottlingException`), which doesn't set either of the other two.
907
+ * Returns undefined when the error is not an object or carries no status
908
+ */
909
+ function extractStatus(err) {
910
+ if (!err || typeof err !== "object") return void 0;
911
+ const error = err;
912
+ if (typeof error.status === "number") return error.status;
913
+ if (typeof error.statusCode === "number") return error.statusCode;
914
+ if (typeof error.$metadata?.httpStatusCode === "number") return error.$metadata.httpStatusCode;
915
+ return void 0;
916
+ }
917
+ function formatSafely(value) {
918
+ try {
919
+ return JSON.stringify(value, null, 2) ?? String(value);
920
+ } catch {
921
+ try {
922
+ return String(value);
923
+ } catch {
924
+ return "[unprintable error]";
925
+ }
926
+ }
927
+ }
928
+ /**
929
+ * Looks inside an unknown thrown value and pulls out a human-readable
930
+ * description of it. Checks the `error` field first (the provider's raw
931
+ * rejection body, JSON-stringified if possible) then falls back to the
932
+ * message` field. Always returns a safe string, even when the thrown value
933
+ * has hostile properties or cannot be serialized normally.
934
+ */
935
+ function describeError(err) {
936
+ if (err && typeof err === "object") try {
937
+ const error = err;
938
+ if (error.error !== void 0) return formatSafely(error.error);
939
+ if (typeof error.message === "string") return error.message;
940
+ } catch {}
941
+ return formatSafely(err);
942
+ }
943
+ /** Converts any thrown value into a well-typed LLMError. */
944
+ function normalizeError(error, signal) {
945
+ if (signal?.aborted) return new LLMError("LLM request aborted", "aborted");
946
+ if (error instanceof LLMError) {
947
+ if (error.status === 429 && error.code === void 0) error.code = "provider_rate_limited";
948
+ return error;
949
+ }
950
+ const status = extractStatus(error);
951
+ const retryAfterMs = extractRetryAfterMs(error);
952
+ if (status !== void 0) return new LLMError("LLM request failed", "api", status, void 0, error, retryAfterMs, status === 429 ? "provider_rate_limited" : void 0);
953
+ return new LLMError("LLM request failed", "unknown", void 0, void 0, error, retryAfterMs);
954
+ }
955
+
956
+ //#endregion
957
+ //#region src/internal/execution/parse.utils.ts
958
+ /** Default `parseJson`: `JSON.parse` wrapped in try/catch, returning `undefined` on failure. */
959
+ function defaultParseJson(content) {
960
+ try {
961
+ return JSON.parse(content);
962
+ } catch {
963
+ return void 0;
964
+ }
965
+ }
966
+
967
+ //#endregion
968
+ //#region src/internal/execution/wire.utils.ts
969
+ /** Translates app-facing `ToolDefinition[]` into the OpenAI-shaped wire tools array. */
970
+ function toWireTools(tools) {
971
+ return tools.map((tool) => ({
972
+ type: "function",
973
+ function: {
974
+ name: tool.name,
975
+ description: tool.description,
976
+ parameters: tool.parameters
977
+ }
978
+ }));
979
+ }
980
+ /** Translates app-facing `ToolCall[]` (e.g. from a replayed assistant turn) into wire tool_calls. */
981
+ function toWireToolCalls(toolCalls) {
982
+ return toolCalls.map((tc) => ({
983
+ id: tc.id,
984
+ type: "function",
985
+ function: {
986
+ name: tc.name,
987
+ arguments: JSON.stringify(tc.arguments ?? {})
988
+ }
989
+ }));
990
+ }
991
+ /**
992
+ * Parses the provider's wire-shaped `tool_calls` back into VernLLM's
993
+ * `ToolCall[]`. Malformed argument JSON is a `'parse'` error, same
994
+ * convention as malformed JSON response bodies elsewhere in VernLLM.
995
+ */
996
+ function parseWireToolCalls(wireToolCalls) {
997
+ return wireToolCalls.map((wc) => {
998
+ let parsedArgs;
999
+ try {
1000
+ parsedArgs = wc.function.arguments.trim() ? JSON.parse(wc.function.arguments) : {};
1001
+ } catch {
1002
+ throw new LLMError(`Invalid JSON arguments for tool call "${wc.function.name}"`, "parse");
1003
+ }
1004
+ return {
1005
+ id: wc.id,
1006
+ name: wc.function.name,
1007
+ arguments: parsedArgs
1008
+ };
1009
+ });
1010
+ }
1011
+
1012
+ //#endregion
1013
+ //#region src/internal/execution/requestBuilder.ts
1014
+ /**
1015
+ * Builds the wire request object for one call, applying per-instance
1016
+ * defaults (model, max tokens, temperature) and per-call overrides.
1017
+ * Owns every validation that depends only on shape, not on execution:
1018
+ * history alternation, duplicate/empty tool lists, `toolChoice` naming a
1019
+ * real tool. Has no knowledge of retry, timeouts, or the breaker, only
1020
+ * the three defaults a `FallbackTarget` can override per-target (see the
1021
+ * `defaultMaxTokens`/`defaultTemperature` overrides in the fallback
1022
+ * design), which is what keeps it separable from `CallExecutor`.
1023
+ */
1024
+ var RequestBuilder = class {
1025
+ model;
1026
+ defaultMaxTokens;
1027
+ defaultTemperature;
1028
+ constructor(options) {
1029
+ this.model = options.model;
1030
+ this.defaultMaxTokens = options.defaultMaxTokens;
1031
+ this.defaultTemperature = options.defaultTemperature;
1032
+ }
1033
+ /** Applies per-call defaults and shapes params into the client's request object. */
1034
+ build(params) {
1035
+ const { systemPrompt, userContent, history = [], maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema, tools, toolChoice } = params;
1036
+ const temperature = params.temperature === void 0 ? this.defaultTemperature : params.temperature;
1037
+ if (tools && tools.length === 0) throw new LLMError("`tools` was an empty array. This is almost always a bug (e.g. a filtered tool list that ended up empty). An empty `tools` array still switches on tool-call mode (response shape, jsonMode default, wire format) with nothing for the model to call. Omit `tools` entirely for a normal call, or make sure the array is non-empty.", "validation");
1038
+ if (tools) {
1039
+ const seen = new Set();
1040
+ const duplicates = new Set();
1041
+ for (const tool of tools) {
1042
+ if (seen.has(tool.name)) duplicates.add(tool.name);
1043
+ seen.add(tool.name);
1044
+ }
1045
+ if (duplicates.size) throw new LLMError(`\`tools\` has duplicate name(s): [${[...duplicates].join(", ")}]. Tool names must be unique.`, "validation");
1046
+ }
1047
+ if (toolChoice && !tools) throw new LLMError("`toolChoice` was set without `tools`. There is nothing for it to choose between. Set `tools`, or remove `toolChoice`.", "validation");
1048
+ if (tools && typeof toolChoice === "object" && !tools.some((t) => t.name === toolChoice.name)) throw new LLMError(`toolChoice names "${toolChoice.name}", which is not in \`tools\` ([${tools.map((t) => t.name).join(", ")}]).`, "validation");
1049
+ const jsonMode = params.jsonMode ?? (tools ? false : true);
1050
+ const useJson = jsonMode || Boolean(jsonSchema);
1051
+ if (params.schema && !useJson) throw new LLMError("schema was provided but jsonMode: false disables JSON parsing, so nothing would validate it. Remove jsonMode: false, set jsonSchema, or remove schema.", "validation");
1052
+ const responseFormat = this.buildResponseFormat(jsonSchema, useJson);
1053
+ this.validateHistory(history);
1054
+ const request = {
1055
+ model,
1056
+ ...temperature !== null ? { temperature } : {},
1057
+ max_tokens: maxTokens,
1058
+ ...responseFormat ? { response_format: responseFormat } : {},
1059
+ ...reasoningEffort ? { reasoning_effort: reasoningEffort } : {},
1060
+ ...tools ? { tools: toWireTools(tools) } : {},
1061
+ ...tools ? { tool_choice: this.buildWireToolChoice(toolChoice) } : {},
1062
+ messages: [
1063
+ ...systemPrompt ? [{
1064
+ role: "system",
1065
+ content: systemPrompt
1066
+ }] : [],
1067
+ ...history.flatMap((turn) => this.turnToWireMessages(turn)),
1068
+ {
1069
+ role: "user",
1070
+ content: userContent
1071
+ }
1072
+ ]
1073
+ };
1074
+ return {
1075
+ useJson,
1076
+ model,
1077
+ request
1078
+ };
1079
+ }
1080
+ /**
1081
+ * Validates `history` alternates user/assistant turns, since providers
1082
+ * like Anthropic/Gemini reject or mishandle consecutive same-role turns.
1083
+ */
1084
+ validateHistory(history) {
1085
+ let previousTurn;
1086
+ for (const [index, turn] of history.entries()) {
1087
+ if (turn.role === "tool") {
1088
+ if (previousTurn?.role !== "assistant" || !previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] is a "tool" turn, but must immediately follow an "assistant" turn that requested tools`, "validation");
1089
+ if (!turn.toolResults?.length) throw new LLMError(`history[${index}] is a "tool" turn but has no toolResults`, "validation");
1090
+ const requestedIds = new Set(previousTurn.toolCalls.map((tc) => tc.id));
1091
+ const resultIds = turn.toolResults.map((tr) => tr.toolCallId);
1092
+ const unknownIds = resultIds.filter((id) => !requestedIds.has(id));
1093
+ if (unknownIds.length) throw new LLMError(`history[${index}].toolResults references unknown toolCallId(s) [${unknownIds.join(", ")}]`, "validation");
1094
+ const seenIds = new Set();
1095
+ const duplicateIds = new Set();
1096
+ for (const id of resultIds) {
1097
+ if (seenIds.has(id)) duplicateIds.add(id);
1098
+ seenIds.add(id);
1099
+ }
1100
+ if (duplicateIds.size) throw new LLMError(`history[${index}].toolResults has duplicate toolCallId(s) [${[...duplicateIds].join(", ")}]`, "validation");
1101
+ const missingIds = [...requestedIds].filter((id) => !resultIds.includes(id));
1102
+ if (missingIds.length) throw new LLMError(`history[${index}] is missing toolResults for toolCallId(s) [${missingIds.join(", ")}]`, "validation");
1103
+ } else {
1104
+ if (turn.role === previousTurn?.role) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
1105
+ if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] follows an assistant tool request without tool results`, "validation");
1106
+ }
1107
+ previousTurn = turn;
1108
+ }
1109
+ if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError("The last entry in history is an assistant tool request without tool results", "validation");
1110
+ if (previousTurn?.role === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn.", "validation");
1111
+ }
1112
+ /** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
1113
+ buildWireToolChoice(toolChoice) {
1114
+ if (!toolChoice || toolChoice === "auto") return "auto";
1115
+ if (toolChoice === "none" || toolChoice === "required") return toolChoice;
1116
+ return {
1117
+ type: "function",
1118
+ function: { name: toolChoice.name }
1119
+ };
1120
+ }
1121
+ /**
1122
+ * Expands one `ConversationTurn` into one or more wire messages. Plain
1123
+ * user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
1124
+ * to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
1125
+ * turn expands into one wire `tool` message per `toolResult`, since
1126
+ * OpenAI-shaped wire format wants one message per tool_call_id.
1127
+ */
1128
+ turnToWireMessages(turn) {
1129
+ if (turn.role === "tool") return (turn.toolResults ?? []).map((tr) => ({
1130
+ role: "tool",
1131
+ tool_call_id: tr.toolCallId,
1132
+ content: typeof tr.content === "string" ? tr.content : JSON.stringify(tr.content ?? null),
1133
+ ...tr.isError ? { is_error: true } : {}
1134
+ }));
1135
+ if (turn.role === "assistant" && turn.toolCalls?.length) return [{
1136
+ role: "assistant",
1137
+ ...turn.content ? { content: turn.content } : {},
1138
+ tool_calls: toWireToolCalls(turn.toolCalls)
1139
+ }];
1140
+ return [{
1141
+ role: turn.role,
1142
+ content: turn.content ?? ""
1143
+ }];
1144
+ }
1145
+ /**
1146
+ * Chooses the response format: a provider-native `jsonSchema` takes
1147
+ * priority when supplied (constrains generation directly), otherwise
1148
+ * falls back to the looser `json_object` mode when JSON output is
1149
+ * requested, or no format at all for plain text responses.
1150
+ */
1151
+ buildResponseFormat(jsonSchema, useJson) {
1152
+ if (jsonSchema) return {
1153
+ type: "json_schema",
1154
+ json_schema: {
1155
+ name: jsonSchema.name,
1156
+ schema: jsonSchema.schema,
1157
+ strict: jsonSchema.strict ?? true,
1158
+ description: jsonSchema.description
1159
+ }
1160
+ };
1161
+ return useJson ? { type: "json_object" } : void 0;
1162
+ }
1163
+ };
1164
+
1165
+ //#endregion
1166
+ //#region src/internal/execution/streamAccumulator.ts
1167
+ /**
1168
+ * The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
1169
+ * an async generator that yields translated `StreamChunk`s to the caller
1170
+ * live, as they arrive, with no per-chunk timeout and no bound on total
1171
+ * duration, and accumulates text/tool-call deltas internally so that
1172
+ * `finalize` can produce `finalResult` once the stream completes.
1173
+ *
1174
+ * Two separate try/catches: the iteration loop's catch handles errors
1175
+ * the transport itself throws, which aren't normalized yet, so that
1176
+ * happens here, alongside the one `onStreamFailure` call for them. The
1177
+ * second catch, around `finalize`, does not re-normalize or re-report,
1178
+ * since `finalize`'s caller (`finalizeResponse`) already does both
1179
+ * internally.
1180
+ */
1181
+ function buildStreamResult(iterator, first, options) {
1182
+ const { requestId, model, providerName, isFallback, chunkIdleTimeoutMs, streamController, logger, signal } = options;
1183
+ let resolveFinal;
1184
+ let rejectFinal;
1185
+ const finalResult = new Promise((resolve, reject) => {
1186
+ resolveFinal = resolve;
1187
+ rejectFinal = reject;
1188
+ });
1189
+ finalResult.catch(() => {});
1190
+ const MAX_BUFFERED_CHUNKS = 1e4;
1191
+ const buffered = [];
1192
+ const pending = [];
1193
+ let streamDone = false;
1194
+ let streamError;
1195
+ let hasLoggedEviction = false;
1196
+ const push = (chunk) => {
1197
+ const waiter = pending.shift();
1198
+ if (waiter) {
1199
+ waiter.resolve({
1200
+ done: false,
1201
+ value: chunk
1202
+ });
1203
+ return;
1204
+ }
1205
+ buffered.push(chunk);
1206
+ if (buffered.length > MAX_BUFFERED_CHUNKS * 2) {
1207
+ if (!hasLoggedEviction) {
1208
+ hasLoggedEviction = true;
1209
+ logger.warn(`[VernLLM] stream chunk buffer exceeded cap (${MAX_BUFFERED_CHUNKS}), evicting ${buffered.length - MAX_BUFFERED_CHUNKS} oldest chunk(s); buffered=${buffered.length}. The chunks iterable was never read (or fell far behind) for this stream.`);
1210
+ }
1211
+ buffered.splice(0, buffered.length - MAX_BUFFERED_CHUNKS);
1212
+ }
1213
+ };
1214
+ const finish = () => {
1215
+ streamDone = true;
1216
+ for (const waiter of pending.splice(0)) waiter.resolve({
1217
+ done: true,
1218
+ value: void 0
1219
+ });
1220
+ };
1221
+ const fail = (error) => {
1222
+ streamDone = true;
1223
+ streamError = error;
1224
+ for (const waiter of pending.splice(0)) waiter.reject(error);
1225
+ };
1226
+ const chunks = { [Symbol.asyncIterator]() {
1227
+ return { next() {
1228
+ if (buffered.length) return Promise.resolve({
1229
+ done: false,
1230
+ value: buffered.shift()
1231
+ });
1232
+ if (streamDone) return streamError ? Promise.reject(streamError) : Promise.resolve({
1233
+ done: true,
1234
+ value: void 0
1235
+ });
1236
+ return new Promise((resolve, reject) => {
1237
+ pending.push({
1238
+ resolve,
1239
+ reject
1240
+ });
1241
+ });
1242
+ } };
1243
+ } };
1244
+ const toolCallAcc = new Map();
1245
+ let textAcc = "";
1246
+ let usage;
1247
+ (async () => {
1248
+ try {
1249
+ let result = first;
1250
+ while (!result.done) {
1251
+ const wireChunk = result.value;
1252
+ if (wireChunk.type === "ping") {} else if (wireChunk.type === "text-delta") {
1253
+ textAcc += wireChunk.delta;
1254
+ push({
1255
+ type: "text-delta",
1256
+ delta: wireChunk.delta
1257
+ });
1258
+ } else if (wireChunk.type === "tool_call_delta") {
1259
+ const entry = toolCallAcc.get(wireChunk.index) ?? { args: "" };
1260
+ entry.id ??= wireChunk.id;
1261
+ entry.name ??= wireChunk.name;
1262
+ entry.args += wireChunk.argumentsDelta ?? "";
1263
+ toolCallAcc.set(wireChunk.index, entry);
1264
+ push({
1265
+ type: "tool_call_delta",
1266
+ index: wireChunk.index,
1267
+ id: wireChunk.id,
1268
+ name: wireChunk.name,
1269
+ argsDelta: wireChunk.argumentsDelta,
1270
+ complete: wireChunk.complete
1271
+ });
1272
+ } else if (wireChunk.type === "usage") {
1273
+ usage = {
1274
+ promptTokens: wireChunk.usage.prompt_tokens ?? 0,
1275
+ completionTokens: wireChunk.usage.completion_tokens ?? 0,
1276
+ totalTokens: wireChunk.usage.total_tokens ?? 0,
1277
+ requestId,
1278
+ model,
1279
+ provider: providerName,
1280
+ usedFallback: isFallback
1281
+ };
1282
+ push({
1283
+ type: "usage",
1284
+ usage
1285
+ });
1286
+ }
1287
+ result = await withChunkIdleTimeout(() => iterator.next(), chunkIdleTimeoutMs, () => streamController.abort(), logger);
1288
+ }
1289
+ } catch (error) {
1290
+ try {
1291
+ await iterator.return?.();
1292
+ } catch {}
1293
+ streamController.abort();
1294
+ const normalized = normalizeError(error, signal);
1295
+ try {
1296
+ options.onStreamFailure(normalized, usage);
1297
+ } catch {}
1298
+ fail(normalized);
1299
+ rejectFinal(normalized);
1300
+ return;
1301
+ }
1302
+ finish();
1303
+ try {
1304
+ options.onStreamSuccess(usage);
1305
+ } catch {}
1306
+ try {
1307
+ const wireToolCalls = toolCallAcc.size ? [...toolCallAcc.entries()].sort(([indexA], [indexB]) => indexA - indexB).map(([, entry]) => ({
1308
+ id: entry.id ?? "",
1309
+ type: "function",
1310
+ function: {
1311
+ name: entry.name ?? "",
1312
+ arguments: entry.args
1313
+ }
1314
+ })) : void 0;
1315
+ const finalized = options.finalize(textAcc, wireToolCalls, usage);
1316
+ resolveFinal(finalized);
1317
+ } catch (error) {
1318
+ rejectFinal(error);
1319
+ }
1320
+ })();
1321
+ return {
1322
+ chunks,
1323
+ finalResult
1324
+ };
1325
+ }
1326
+
1327
+ //#endregion
1328
+ //#region src/internal/execution/callExecutor.ts
1329
+ /**
1330
+ * Everything one provider target needs to attempt a call: request
1331
+ * building, retry with backoff, the per-target breaker, the per-target
1332
+ * limiter. Never exported publicly. `VernLLM` holds one per target and
1333
+ * owns the fallback loop and caching on top.
1334
+ */
1335
+ var CallExecutor = class {
1336
+ maxRetries;
1337
+ timeoutMs;
1338
+ chunkIdleTimeoutMs;
1339
+ baseDelayMs;
1340
+ nonRetryableStatus;
1341
+ parseJson;
1342
+ logger;
1343
+ redact;
1344
+ onUsage;
1345
+ onUsageFailure;
1346
+ reportEvent;
1347
+ breaker;
1348
+ limiter;
1349
+ isFallback;
1350
+ requestBuilder;
1351
+ constructor(providerName, client, model, options) {
1352
+ this.providerName = providerName;
1353
+ this.client = client;
1354
+ this.model = model;
1355
+ this.maxRetries = options.maxRetries;
1356
+ this.timeoutMs = options.timeoutMs;
1357
+ this.chunkIdleTimeoutMs = options.chunkIdleTimeoutMs;
1358
+ this.baseDelayMs = options.baseDelayMs;
1359
+ this.nonRetryableStatus = options.nonRetryableStatus;
1360
+ this.parseJson = options.parseJson ?? defaultParseJson;
1361
+ this.logger = options.logger;
1362
+ this.redact = options.redact;
1363
+ this.onUsage = options.onUsage;
1364
+ this.onUsageFailure = options.onUsageFailure;
1365
+ this.reportEvent = makeEventReporter(options.onEvent, this.logger);
1366
+ this.breaker = options.breaker;
1367
+ this.limiter = options.limiter;
1368
+ this.isFallback = options.isFallback ?? false;
1369
+ this.requestBuilder = new RequestBuilder({
1370
+ model,
1371
+ defaultMaxTokens: options.defaultMaxTokens,
1372
+ defaultTemperature: options.defaultTemperature
1373
+ });
1374
+ }
1375
+ getCircuitState(model) {
1376
+ return this.breaker?.getState(model);
1377
+ }
1378
+ /**
1379
+ * Throws if the breaker is open for this target/model, exactly like the
1380
+ * check `run`/`runStream` used to make internally. Exposed so `VernLLM`
1381
+ * can gate on it before reserving usage, avoiding a reserve-then-refund
1382
+ * round trip on a call that was never going to be attempted. `assertClosed`
1383
+ * has a stateful side effect (claiming a half-open trial slot), so it must
1384
+ * run exactly once per logical call: `run`/`runStream` no longer call it
1385
+ * themselves, this is now the only call site.
1386
+ */
1387
+ assertBreakerClosed(model) {
1388
+ this.breaker?.assertClosed(model ?? this.model);
1389
+ }
1390
+ /**
1391
+ * Runs a single logical call against this target: retry with backoff,
1392
+ * normalized error on exhaustion. Mirrors the old `VernLLM.call`'s
1393
+ * non-streaming branch, minus cache/usage-reservation and the breaker
1394
+ * check, which stay one layer up since they aren't per-target concerns
1395
+ * (see `assertBreakerClosed`).
1396
+ */
1397
+ async run(params, requestId, onAttempt) {
1398
+ const model = params.model ?? this.model;
1399
+ try {
1400
+ return await this.retryWithBackoff((attempt) => this.executeCall(params, requestId, attempt), requestId, model, params.signal, onAttempt);
1401
+ } catch (error) {
1402
+ const normalized = normalizeError(error, params.signal);
1403
+ if (this.countsTowardBreaker(normalized)) this.breaker?.recordFailure(model);
1404
+ this.logger.debug(`[VernLLM:${requestId}] error:\n${this.redactText(describeError(error))}`);
1405
+ throw normalized;
1406
+ }
1407
+ }
1408
+ /** Streaming counterpart to `run`. Mirrors the old streaming branch of `VernLLM.call`. */
1409
+ async runStream(params, requestId, onAttempt) {
1410
+ const model = params.model ?? this.model;
1411
+ try {
1412
+ return await this.retryWithBackoff((attempt) => this.executeStreamCall(params, requestId, attempt), requestId, model, params.signal, onAttempt);
1413
+ } catch (error) {
1414
+ const normalized = normalizeError(error, params.signal);
1415
+ if (this.countsTowardBreaker(normalized)) this.breaker?.recordFailure(model);
1416
+ this.logger.debug(`[VernLLM:${requestId}] stream-open error:\n${this.redactText(describeError(error))}`);
1417
+ throw normalized;
1418
+ }
1419
+ }
1420
+ /**
1421
+ * Performs a single attempt: builds the request (translating `tools` to
1422
+ * wire shape when present), dispatches it with a timeout, and shapes the
1423
+ * response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
1424
+ * set. Throws on an empty response (no text and no tool_calls) so the
1425
+ * retry loop treats it like any other transient failure.
1426
+ */
1427
+ async executeCall(params, requestId, attempt) {
1428
+ const { useJson, model, request } = this.requestBuilder.build(params);
1429
+ let release;
1430
+ if (this.limiter) {
1431
+ const acquired = await this.limiter.acquire(this.limiter.estimate(request), params.signal);
1432
+ release = acquired.release;
1433
+ if (acquired.waitedMs > 0) this.reportEvent({
1434
+ kind: "rate_limited",
1435
+ requestId,
1436
+ provider: this.providerName,
1437
+ model,
1438
+ waitedMs: acquired.waitedMs,
1439
+ reason: acquired.reason ?? "rpm"
1440
+ });
1441
+ }
1442
+ try {
1443
+ const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
1444
+ const usage = this.extractUsage(response, requestId, model);
1445
+ release?.(this.actualTokensFor(usage));
1446
+ release = void 0;
1447
+ const rawContent = response.choices?.[0]?.message?.content;
1448
+ const wireToolCalls = response.choices?.[0]?.message?.tool_calls;
1449
+ return this.finalizeResponse(rawContent, wireToolCalls, params, useJson, model, usage, requestId, attempt);
1450
+ } finally {
1451
+ release?.();
1452
+ }
1453
+ }
1454
+ /** Applies `redact` (if configured); otherwise returns `text` unchanged. */
1455
+ redactText(text) {
1456
+ return this.redact ? this.redact(text) : text;
1457
+ }
1458
+ /**
1459
+ * Applies `redact` (if configured) to whatever the debug log is about
1460
+ * to show: real content when there is any, otherwise the tool-call
1461
+ * placeholder, which carries no user data and passes through
1462
+ * `redact` unchanged in practice but is included for a caller whose
1463
+ * `redact` does something structural (e.g. adding a marker) rather
1464
+ * than just scrubbing PII.
1465
+ */
1466
+ redactedOutput(content, wireToolCalls) {
1467
+ return this.redactText(content ?? `[${wireToolCalls?.length ?? 0} tool call(s)]`);
1468
+ }
1469
+ /**
1470
+ * Shapes a fully-arrived response (content and/or tool_calls, already
1471
+ * extracted from the provider's payload) into `T` or a
1472
+ * `CallWithToolsResult<T>`. Reused by the streaming path once it has
1473
+ * buffered the full text/tool-call deltas, so there's no separate
1474
+ * parsing/validation logic for streaming.
489
1475
  *
490
- * @param params - System/user content plus per-call overrides (model,
491
- * temperature, jsonMode, schema, signal, etc). See `CallParams`.
492
- * @returns The parsed (and optionally schema-validated) response, or the
493
- * raw string content when `jsonMode` is false and no `jsonSchema` is set.
1476
+ * Normalizes and reports usage failure on error itself, so every caller
1477
+ * gets identical error handling without duplicating it.
494
1478
  */
495
- async call(params) {
496
- this.breaker?.assertClosed();
497
- if (params.signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
498
- const requestId = params.requestId ?? (0, crypto.randomUUID)();
499
- return withReservedUsage(params, false, async () => {
500
- try {
501
- return await this.retryWithBackoff(() => this.executeCall(params, requestId), requestId, params.signal);
502
- } catch (error) {
503
- const normalized = normalizeError(error, params.signal);
504
- if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
505
- this.logger.debug(`[vern:${requestId}] error:\n${describeError(error)}`);
506
- throw normalized;
1479
+ finalizeResponse(rawContent, wireToolCalls, params, useJson, model, usage, requestId, attempt) {
1480
+ try {
1481
+ const content = rawContent?.trim();
1482
+ if (!content && !wireToolCalls?.length) throw new LLMError("Empty LLM response", "api");
1483
+ this.logger.debug(`[VernLLM:${requestId}] output:\n${this.redactedOutput(content, wireToolCalls).slice(0, 800)}`);
1484
+ if (wireToolCalls?.length) {
1485
+ if (!params.tools) throw new LLMError("Provider returned tool_calls but no `tools` were sent with this call.", "api");
1486
+ const toolCalls = parseWireToolCalls(wireToolCalls);
1487
+ this.validateToolCallArguments(toolCalls, params.tools);
1488
+ this.breaker?.recordSuccess(model);
1489
+ this.reportUsage(usage);
1490
+ return {
1491
+ type: "tool_calls",
1492
+ toolCalls,
1493
+ ...content ? { content } : {}
1494
+ };
507
1495
  }
508
- }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
1496
+ const textContent = content ?? "";
1497
+ if (!useJson) {
1498
+ this.breaker?.recordSuccess(model);
1499
+ this.reportUsage(usage);
1500
+ return params.tools ? {
1501
+ type: "content",
1502
+ content: textContent
1503
+ } : textContent;
1504
+ }
1505
+ const result = this.parseAndValidate(textContent, params.schema);
1506
+ this.breaker?.recordSuccess(model);
1507
+ this.reportUsage(usage);
1508
+ return params.tools ? {
1509
+ type: "content",
1510
+ content: result
1511
+ } : result;
1512
+ } catch (error) {
1513
+ const normalized = normalizeError(error, params.signal);
1514
+ if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt);
1515
+ throw normalized;
1516
+ }
1517
+ }
1518
+ /**
1519
+ * Opens a stream for a single attempt: builds the request exactly like
1520
+ * `executeCall`, then requires `createStream` on the client (a clear
1521
+ * `validation` error if the adapter doesn't support it). The timeout
1522
+ * wraps stream construction and the first `.next()` together, not just
1523
+ * construction: calling an `async function*` returns an iterator
1524
+ * synchronously without running its body until `.next()` is first
1525
+ * invoked, so timing only construction would time an operation that's
1526
+ * always instant, not the actual connection. Both are folded into a
1527
+ * single `withTimeout` so the same abort signal reaches whatever the
1528
+ * adapter's `createStream` uses internally for its first network
1529
+ * round-trip.
1530
+ *
1531
+ * Circuit-breaker success is recorded once the stream fully completes,
1532
+ * not on the first chunk arriving, so a connection that opens but then
1533
+ * dies mid-stream isn't masked as a success (see `buildStreamResult`).
1534
+ */
1535
+ async executeStreamCall(params, requestId, attempt) {
1536
+ const { useJson, model, request } = this.requestBuilder.build(params);
1537
+ const completions = this.client.chat.completions;
1538
+ if (!completions.createStream) throw new LLMError("stream: true requires a client/adapter with createStream", "validation");
1539
+ const createStream = completions.createStream.bind(completions);
1540
+ let release;
1541
+ if (this.limiter) {
1542
+ const acquired = await this.limiter.acquire(this.limiter.estimate(request), params.signal);
1543
+ release = acquired.release;
1544
+ if (acquired.waitedMs > 0) this.reportEvent({
1545
+ kind: "rate_limited",
1546
+ requestId,
1547
+ provider: this.providerName,
1548
+ model,
1549
+ waitedMs: acquired.waitedMs,
1550
+ reason: acquired.reason ?? "rpm"
1551
+ });
1552
+ }
1553
+ const streamController = new AbortController();
1554
+ const combinedExternal = params.signal ? AbortSignal.any([params.signal, streamController.signal]) : streamController.signal;
1555
+ try {
1556
+ const { iterator, first } = await withTimeout(async (attemptSignal) => {
1557
+ const streamIterator = createStream(request, { signal: attemptSignal })[Symbol.asyncIterator]();
1558
+ const firstResult = await streamIterator.next();
1559
+ return {
1560
+ iterator: streamIterator,
1561
+ first: firstResult
1562
+ };
1563
+ }, this.timeoutMs, combinedExternal);
1564
+ if (first.done) throw new LLMError("Empty LLM response", "api");
1565
+ const releaseAtOpen = release;
1566
+ const result = buildStreamResult(iterator, first, {
1567
+ requestId,
1568
+ model,
1569
+ providerName: this.providerName,
1570
+ isFallback: this.isFallback,
1571
+ chunkIdleTimeoutMs: params.chunkIdleTimeoutMs ?? this.chunkIdleTimeoutMs,
1572
+ streamController,
1573
+ logger: this.logger,
1574
+ signal: params.signal,
1575
+ onStreamSuccess: (usage) => {
1576
+ this.breaker?.recordSuccess(model);
1577
+ releaseAtOpen?.(this.actualTokensFor(usage));
1578
+ },
1579
+ onStreamFailure: (normalized, usage) => {
1580
+ if (normalized.type === "timeout") this.breaker?.recordFailure(model);
1581
+ if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt, true);
1582
+ releaseAtOpen?.(this.actualTokensFor(usage));
1583
+ },
1584
+ finalize: (textAcc, wireToolCalls, usage) => this.finalizeResponse(textAcc, wireToolCalls, params, useJson, model, usage, requestId, attempt)
1585
+ });
1586
+ release = void 0;
1587
+ return result;
1588
+ } finally {
1589
+ release?.();
1590
+ }
1591
+ }
1592
+ /**
1593
+ * Checks every `ToolCall` against the `tools` that were offered, catching
1594
+ * a hallucinated tool name and a duplicate call id before either reaches
1595
+ * the application's dispatch table, then runs each tool's
1596
+ * `argumentsSchema`, if present.
1597
+ *
1598
+ * Contract failures (unknown name, duplicate id) are collected across
1599
+ * every call and thrown together, since retrying a request that already
1600
+ * has these errors cannot help (`shouldRetry` excludes them by `code`)
1601
+ * and a caller fixing them wants to see every one, not just the first.
1602
+ * Schema failures keep the original single-error, `type: 'validation'`
1603
+ * shape rather than being folded into the aggregate, since they're a
1604
+ * distinct failure kind from the contract failures above (also excluded
1605
+ * from retry, by `type` rather than `code`; see `shouldRetry`).
1606
+ */
1607
+ validateToolCallArguments(toolCalls, tools) {
1608
+ const known = new Map(tools.map((t) => [t.name, t]));
1609
+ const seenIds = new Set();
1610
+ const toolIssues = [];
1611
+ for (const call of toolCalls) {
1612
+ if (seenIds.has(call.id)) toolIssues.push({
1613
+ name: call.name,
1614
+ toolCallId: call.id,
1615
+ code: "duplicate_tool_call_id"
1616
+ });
1617
+ seenIds.add(call.id);
1618
+ if (!known.has(call.name)) toolIssues.push({
1619
+ name: call.name,
1620
+ toolCallId: call.id,
1621
+ code: "unknown_tool"
1622
+ });
1623
+ }
1624
+ if (toolIssues.length > 0) {
1625
+ const unknownTool = toolIssues.find((i) => i.code === "unknown_tool");
1626
+ const primary = unknownTool ? `Model requested tool "${unknownTool.name}", which was not in the tools offered ([${[...known.keys()].join(", ")}]).` : `Duplicate tool call id "${toolIssues[0].toolCallId}" in the model's response.`;
1627
+ const message = toolIssues.length > 1 ? `${primary} (${toolIssues.length} tool call issues total, see toolIssues.)` : primary;
1628
+ const error = new LLMError(message, "api", void 0, void 0, void 0, void 0, unknownTool ? "unknown_tool" : "duplicate_tool_call_id");
1629
+ error.toolIssues = toolIssues;
1630
+ throw error;
1631
+ }
1632
+ for (const call of toolCalls) {
1633
+ const definition = known.get(call.name);
1634
+ if (!definition?.argumentsSchema) continue;
1635
+ const result = definition.argumentsSchema.safeParse(call.arguments);
1636
+ if (!result.success) throw new LLMError(`Arguments for tool call "${call.name}" failed validation`, "validation", void 0, result.error);
1637
+ }
509
1638
  }
510
1639
  /** Runs `fn`, retrying with backoff according to `shouldRetry`. */
511
- async retryWithBackoff(fn, requestId, signal) {
1640
+ async retryWithBackoff(fn, requestId, model, signal, onAttempt) {
512
1641
  let lastError;
513
1642
  for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
514
- if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
515
- return await fn();
1643
+ if (attempt > 0) await this.recoverDelay(requestId, model, attempt, lastError, signal);
1644
+ onAttempt?.();
1645
+ return await fn(attempt);
516
1646
  } catch (error) {
517
1647
  lastError = error;
518
1648
  if (!this.shouldRetry(error, signal)) break;
@@ -520,105 +1650,60 @@ var VernLLM = class {
520
1650
  throw lastError;
521
1651
  }
522
1652
  /**
523
- * Performs a single attempt: builds the request, dispatches it with a
524
- * timeout, and shapes the response. Throws on an empty response so the
525
- * retry loop treats it like any other transient failure.
526
- */
527
- async executeCall(params, requestId) {
528
- const { useJson, model, request } = this.buildRequestPayload(params);
529
- const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
530
- const content = response.choices?.[0]?.message?.content?.trim();
531
- if (!content) throw new LLMError("Empty LLM response", "api");
532
- this.logger.debug(`[vern:${requestId}] output:\n${content.slice(0, 800)}`);
533
- this.recordUsage(response, requestId, model);
534
- if (!useJson) {
535
- this.breaker?.recordSuccess();
536
- return content;
537
- }
538
- const result = this.parseAndValidate(content, params.schema);
539
- this.breaker?.recordSuccess();
540
- return result;
541
- }
542
- /**
543
- * Validates `history` alternates user/assistant turns, since providers
544
- * like Anthropic/Gemini reject or mishandle consecutive same-role turns.
1653
+ * Pulls `TokenUsage` out of a raw response, if the provider reported it.
1654
+ * Extraction doesn't depend on what happens to the response afterward, so
1655
+ * a malformed body can still yield usage if the provider's usage block
1656
+ * itself came through intact.
545
1657
  */
546
- validateHistory(history) {
547
- let previousRole;
548
- for (const [index, turn] of history.entries()) {
549
- if (turn.role !== "user" && turn.role !== "assistant") throw new LLMError(`Invalid history[${index}].role "${turn.role}": must be "user" or "assistant"`, "validation");
550
- if (turn.role === previousRole) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
551
- previousRole = turn.role;
552
- }
553
- if (previousRole === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn. history must end with an \"assistant\" turn (or be empty).", "validation");
554
- }
555
- /** Applies per-call defaults and shapes params into the client's request object. */
556
- buildRequestPayload(params) {
557
- const { systemPrompt, userContent, history = [], temperature = .2, jsonMode = true, maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema } = params;
558
- const useJson = jsonMode || Boolean(jsonSchema);
559
- if (params.schema && !useJson) throw new LLMError("schema was provided but jsonMode: false disables JSON parsing, so nothing would validate it. Remove jsonMode: false, set jsonSchema, or remove schema.", "validation");
560
- const responseFormat = this.buildResponseFormat(jsonSchema, useJson);
561
- this.validateHistory(history);
562
- const request = {
563
- model,
564
- temperature,
565
- max_tokens: maxTokens,
566
- ...responseFormat ? { response_format: responseFormat } : {},
567
- ...reasoningEffort ? { reasoning_effort: reasoningEffort } : {},
568
- messages: [
569
- ...systemPrompt ? [{
570
- role: "system",
571
- content: systemPrompt
572
- }] : [],
573
- ...history.map((turn) => ({
574
- role: turn.role,
575
- content: turn.content
576
- })),
577
- {
578
- role: "user",
579
- content: userContent
580
- }
581
- ]
582
- };
1658
+ extractUsage(response, requestId, model) {
1659
+ if (!response.usage) return void 0;
583
1660
  return {
584
- useJson,
1661
+ promptTokens: response.usage.prompt_tokens ?? 0,
1662
+ completionTokens: response.usage.completion_tokens ?? 0,
1663
+ totalTokens: response.usage.total_tokens ?? 0,
1664
+ requestId,
585
1665
  model,
586
- request
1666
+ provider: this.providerName,
1667
+ usedFallback: this.isFallback
587
1668
  };
588
1669
  }
589
1670
  /**
590
- * Chooses the response format: a provider-native `jsonSchema` takes
591
- * priority when supplied (constrains generation directly), otherwise
592
- * falls back to the looser `json_object` mode when JSON output is
593
- * requested, or no format at all for plain text responses.
1671
+ * The token count to reconcile the rate limiter against for a finished
1672
+ * attempt: `totalTokens` when reported, otherwise the sum of prompt and
1673
+ * completion tokens, matching `reportUsageFailure`'s own fallback below
1674
+ * for a hand-rolled client that reports the parts but omits the total.
594
1675
  */
595
- buildResponseFormat(jsonSchema, useJson) {
596
- if (jsonSchema) return {
597
- type: "json_schema",
598
- json_schema: {
599
- name: jsonSchema.name,
600
- schema: jsonSchema.schema,
601
- strict: jsonSchema.strict ?? true,
602
- description: jsonSchema.description
603
- }
604
- };
605
- return useJson ? { type: "json_object" } : void 0;
1676
+ actualTokensFor(usage) {
1677
+ if (!usage) return void 0;
1678
+ return usage.totalTokens || usage.promptTokens + usage.completionTokens;
606
1679
  }
607
- /** Reports token usage to `onUsage`, swallowing and logging any error it throws. */
608
- recordUsage(response, requestId, model) {
609
- if (!response.usage || !this.onUsage) return;
1680
+ /** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
1681
+ reportUsage(usage) {
1682
+ if (!usage || !this.onUsage) return;
610
1683
  try {
611
- this.onUsage({
612
- promptTokens: response.usage.prompt_tokens ?? 0,
613
- completionTokens: response.usage.completion_tokens ?? 0,
614
- totalTokens: response.usage.total_tokens ?? 0,
615
- requestId,
616
- model
617
- });
1684
+ this.onUsage(usage);
618
1685
  } catch (error) {
619
1686
  this.logger.error("[VernLLM] onUsage failed", { message: error instanceof Error ? error.message : "unknown" });
620
1687
  }
621
1688
  }
1689
+ /**
1690
+ * Reports token usage spent on an attempt that then failed, so it isn't
1691
+ * dropped alongside the error. Covers any error thrown after usage
1692
+ * extraction, since all of them happen only after a response (real
1693
+ * spend) already arrived. Swallows and logs any error `onUsageFailure`
1694
+ * itself throws.
1695
+ */
1696
+ reportUsageFailure(usage, error, attempt, terminal = false) {
1697
+ const displayTokens = usage.totalTokens || usage.promptTokens + usage.completionTokens;
1698
+ const attemptText = terminal ? "mid-stream failure (terminal, no further attempts)" : `attempt ${attempt + 1}/${this.maxRetries + 1}`;
1699
+ this.logger.warn(`[VernLLM:${usage.requestId}] usage failure, ${attemptText}: type=${error.type} tokens=${displayTokens}`);
1700
+ if (!this.onUsageFailure) return;
1701
+ try {
1702
+ this.onUsageFailure(usage, error);
1703
+ } catch (hookError) {
1704
+ this.logger.error("[VernLLM] onUsageFailure failed", { message: hookError instanceof Error ? hookError.message : "unknown" });
1705
+ }
1706
+ }
622
1707
  /** Parses response content as JSON and validates it against `schema` when supplied. */
623
1708
  parseAndValidate(content, schema) {
624
1709
  let parsed;
@@ -637,106 +1722,681 @@ var VernLLM = class {
637
1722
  * Waits out the backoff delay for a retry attempt, honoring a
638
1723
  * Retry-After header on the failed attempt's error when present.
639
1724
  * Both Retry-After and plain exponential backoff are capped at the same
640
- * max delay (see `DEFAULT_MAX_DELAY_MS` in `vernLLM.utils.ts`).
1725
+ * max delay (see `DEFAULT_MAX_DELAY_MS` in `retry.utils.ts`).
641
1726
  */
642
- async recoverDelay(requestId, attempt, error, signal) {
1727
+ async recoverDelay(requestId, model, attempt, error, signal) {
643
1728
  const retryAfterMs = extractRetryAfterMs(error);
644
1729
  const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
645
- this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
1730
+ const retryAfterHonored = retryAfterMs !== void 0;
1731
+ this.logger.warn(`[VernLLM:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterHonored ? " (honoring Retry-After)" : ""));
1732
+ this.reportEvent({
1733
+ kind: "retry",
1734
+ requestId,
1735
+ provider: this.providerName,
1736
+ model,
1737
+ attempt,
1738
+ maxRetries: this.maxRetries,
1739
+ delayMs: delay,
1740
+ retryAfterHonored,
1741
+ error: normalizeError(error, signal)
1742
+ });
646
1743
  await waitForRetry(delay, signal);
647
1744
  }
1745
+ isNonRetryableToolContractError(error) {
1746
+ return error instanceof LLMError && (error.code === "unknown_tool" || error.code === "duplicate_tool_call_id");
1747
+ }
648
1748
  /** Decides whether a failed attempt is worth retrying. */
649
1749
  shouldRetry(error, signal) {
650
1750
  if (signal?.aborted) return false;
651
1751
  if (error instanceof LLMError && (error.type === "parse" || error.type === "validation")) return false;
1752
+ if (error instanceof LLMError && error.code === "local_rate_limit") return false;
1753
+ if (this.isNonRetryableToolContractError(error)) return false;
652
1754
  const status = extractStatus(error);
653
1755
  return !(status !== void 0 && this.nonRetryableStatus.includes(status));
654
1756
  }
655
1757
  /**
1758
+ * Decides whether a failed attempt should count toward the circuit
1759
+ * breaker's failure threshold. A model hallucinating a tool name or
1760
+ * reusing a call id isn't the provider being unhealthy, it's a model
1761
+ * response defect that will very likely recur regardless of provider
1762
+ * health, so it shouldn't push a healthy provider's circuit toward
1763
+ * opening. Mirrors the same reasoning `shouldRetry` already applies to
1764
+ * `parse`/`validation`/these same tool-contract codes.
1765
+ */
1766
+ countsTowardBreaker(error) {
1767
+ if (error.type === "validation" || error.type === "parse" || error.type === "aborted" || error.code === "local_rate_limit" || this.isNonRetryableToolContractError(error)) return false;
1768
+ return true;
1769
+ }
1770
+ };
1771
+
1772
+ //#endregion
1773
+ //#region src/logger.ts
1774
+ /**
1775
+ * Default logger. `debug` is gated by the `debug` option on VernLLM
1776
+ * warn/error always fire since they indicate real problems (retries, cache failures)
1777
+ */
1778
+ var ConsoleLogger = class {
1779
+ constructor(debugEnabled) {
1780
+ this.debugEnabled = debugEnabled;
1781
+ }
1782
+ debug(message) {
1783
+ if (this.debugEnabled) console.debug(message);
1784
+ }
1785
+ warn(message) {
1786
+ console.warn(message);
1787
+ }
1788
+ error(message, meta) {
1789
+ console.error(message, meta ?? "");
1790
+ }
1791
+ };
1792
+
1793
+ //#endregion
1794
+ //#region src/rateLimit.ts
1795
+ /** Default `estimateTokens`: chars/4 over every message's content, plus the requested `max_tokens`. */
1796
+ function defaultEstimateTokens(request) {
1797
+ const messagesChars = request.messages.reduce((sum, message) => {
1798
+ const content = message.content;
1799
+ if (typeof content === "string") return sum + content.length;
1800
+ if (content === void 0 || content === null) return sum;
1801
+ try {
1802
+ return sum + JSON.stringify(content).length;
1803
+ } catch {
1804
+ return sum;
1805
+ }
1806
+ }, 0);
1807
+ return Math.ceil(messagesChars / 4) + (request.max_tokens ?? 0);
1808
+ }
1809
+ /**
1810
+ * A capacity that refills continuously. Used for requests per minute and
1811
+ * tokens per minute, where `refillPerMs` is `capacity / 60000`, and for
1812
+ * concurrency, where `refillPerMs` is 0 and every release calls
1813
+ * `give(1)` instead of relying on the clock.
1814
+ */
1815
+ var TokenBucket = class {
1816
+ available;
1817
+ lastRefill = Date.now();
1818
+ constructor(capacity, refillPerMs) {
1819
+ this.capacity = capacity;
1820
+ this.refillPerMs = refillPerMs;
1821
+ this.available = capacity;
1822
+ }
1823
+ refill() {
1824
+ if (this.refillPerMs === 0) return;
1825
+ const now = Date.now();
1826
+ this.available = Math.min(this.capacity, this.available + (now - this.lastRefill) * this.refillPerMs);
1827
+ this.lastRefill = now;
1828
+ }
1829
+ /** Refills, then takes `amount` if available. Leaves the bucket untouched if it can't. */
1830
+ tryTake(amount) {
1831
+ this.refill();
1832
+ if (this.available < amount) return false;
1833
+ this.available -= amount;
1834
+ return true;
1835
+ }
1836
+ /**
1837
+ * Refills, then reports how many ms until this bucket could supply
1838
+ * `amount`, assuming nothing else takes from it meanwhile. Returns 0 if
1839
+ * it already can, `Infinity` if it never will on its own (a
1840
+ * concurrency bucket, `refillPerMs === 0`, only frees via `give`).
1841
+ */
1842
+ msUntilAvailable(amount) {
1843
+ this.refill();
1844
+ if (this.available >= amount) return 0;
1845
+ if (this.refillPerMs === 0) return Infinity;
1846
+ return (amount - this.available) / this.refillPerMs;
1847
+ }
1848
+ /**
1849
+ * Gives capacity back. Not floored at 0: a bad token-usage estimate can
1850
+ * push `available` negative, and it self-corrects on the next refill
1851
+ * rather than being clamped away immediately. Only ceilinged at
1852
+ * `capacity`, so a give can never overfill the bucket.
1853
+ */
1854
+ give(amount) {
1855
+ this.available = Math.min(this.capacity, this.available + amount);
1856
+ }
1857
+ /** The bucket's ceiling, e.g. so a request that could never fit can fail fast instead of queueing forever. */
1858
+ getCapacity() {
1859
+ return this.capacity;
1860
+ }
1861
+ };
1862
+ /**
1863
+ * `setTimeout` silently clamps any delay above this (~24.8 days) instead
1864
+ * of erroring, so an uncapped delay derived from a very small
1865
+ * `requestsPerMinute`/`tokensPerMinute` could wrap around to firing
1866
+ * almost immediately instead of waiting. Mirrors the same guard in
1867
+ * `withTimeout`/`withChunkIdleTimeout`.
1868
+ */
1869
+ const MAX_WAKE_DELAY_MS = 2147483647;
1870
+ /**
1871
+ * Per-target rate limiter. Up to three buckets (requests/min, tokens/min,
1872
+ * concurrency) behind one FIFO queue, so a large call isn't starved by a
1873
+ * stream of small ones. Any bucket omitted from `options` has infinite
1874
+ * capacity and never blocks.
1875
+ */
1876
+ var RateLimiter = class {
1877
+ requests;
1878
+ tokens;
1879
+ concurrency;
1880
+ maxQueueMs;
1881
+ maxQueueSize;
1882
+ estimateTokensFn;
1883
+ queue = [];
1884
+ /**
1885
+ * A single scheduled re-check for the head of the queue when it's
1886
+ * blocked on a bucket that refills on its own clock (rpm/tpm), so a
1887
+ * queue that nobody calls `acquire`/`release` on again isn't stuck
1888
+ * forever waiting for an external trigger to re-drain it. Not needed
1889
+ * for a concurrency block, which only clears via `release`.
1890
+ */
1891
+ wakeTimer;
1892
+ constructor(options) {
1893
+ if (options.requestsPerMinute) this.requests = new TokenBucket(options.requestsPerMinute, options.requestsPerMinute / 6e4);
1894
+ if (options.tokensPerMinute) this.tokens = new TokenBucket(options.tokensPerMinute, options.tokensPerMinute / 6e4);
1895
+ if (options.maxConcurrent) this.concurrency = new TokenBucket(options.maxConcurrent, 0);
1896
+ this.maxQueueMs = options.maxQueueMs ?? 3e4;
1897
+ this.maxQueueSize = options.maxQueueSize ?? 0;
1898
+ this.estimateTokensFn = options.estimateTokens ?? defaultEstimateTokens;
1899
+ }
1900
+ /** Pre-flight token estimate for a request, per the configured (or default) heuristic. */
1901
+ estimate(request) {
1902
+ return this.estimateTokensFn(request);
1903
+ }
1904
+ /**
1905
+ * Waits for capacity in every configured bucket, then takes from each.
1906
+ * The returned `release` gives the concurrency slot back and reconciles
1907
+ * the token bucket against real usage; it must run in a `finally` block.
1908
+ */
1909
+ async acquire(estimatedTokens, signal) {
1910
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
1911
+ if (!Number.isFinite(estimatedTokens) || estimatedTokens < 0) throw new LLMError(`estimatedTokens must be a finite, non-negative number, got ${String(estimatedTokens)}`, "validation");
1912
+ if (this.tokens && estimatedTokens > this.tokens.getCapacity()) throw new LLMError(`estimatedTokens (${estimatedTokens}) exceeds the configured tokensPerMinute capacity (${this.tokens.getCapacity()}); this call could never acquire capacity.`, "quota_exceeded", void 0, void 0, void 0, void 0, "local_rate_limit");
1913
+ if (this.queue.length === 0) {
1914
+ const attempt = this.tryAcquireBuckets(estimatedTokens);
1915
+ if (attempt.ok) return {
1916
+ release: this.makeRelease(estimatedTokens),
1917
+ waitedMs: 0
1918
+ };
1919
+ if (this.maxQueueSize > 0 && this.queue.length >= this.maxQueueSize) throw this.queueFullError();
1920
+ return this.enqueue(estimatedTokens, attempt.reason, signal);
1921
+ }
1922
+ if (this.maxQueueSize > 0 && this.queue.length >= this.maxQueueSize) throw this.queueFullError();
1923
+ return this.enqueue(estimatedTokens, void 0, signal);
1924
+ }
1925
+ queueFullError() {
1926
+ return new LLMError("Rate limit queue is full", "quota_exceeded", void 0, void 0, void 0, void 0, "local_rate_limit");
1927
+ }
1928
+ enqueue(estimatedTokens, initialReason, signal) {
1929
+ return new Promise((resolvePromise, rejectPromise) => {
1930
+ const waiter = {
1931
+ estimatedTokens,
1932
+ enqueuedAt: Date.now(),
1933
+ lastReason: initialReason,
1934
+ resolve: (result) => {
1935
+ cleanup();
1936
+ resolvePromise(result);
1937
+ },
1938
+ reject: (error) => {
1939
+ cleanup();
1940
+ if (this.wakeTimer) {
1941
+ clearTimeout(this.wakeTimer);
1942
+ this.wakeTimer = void 0;
1943
+ }
1944
+ this.drain();
1945
+ rejectPromise(error);
1946
+ }
1947
+ };
1948
+ let queueTimer;
1949
+ const onAbort = () => {
1950
+ waiter.reject(new LLMError("LLM request aborted", "aborted"));
1951
+ };
1952
+ const cleanup = () => {
1953
+ if (queueTimer) clearTimeout(queueTimer);
1954
+ signal?.removeEventListener("abort", onAbort);
1955
+ const index = this.queue.indexOf(waiter);
1956
+ if (index !== -1) this.queue.splice(index, 1);
1957
+ };
1958
+ if (this.maxQueueMs > 0) queueTimer = setTimeout(() => {
1959
+ waiter.reject(new LLMError("Rate limit queue timed out before capacity was available", "quota_exceeded", void 0, void 0, void 0, void 0, "local_rate_limit"));
1960
+ }, this.maxQueueMs);
1961
+ signal?.addEventListener("abort", onAbort, { once: true });
1962
+ this.queue.push(waiter);
1963
+ this.drain();
1964
+ });
1965
+ }
1966
+ /**
1967
+ * Checks and takes from every configured bucket as one atomic unit: if
1968
+ * any bucket lacks capacity, whatever was already taken from the
1969
+ * earlier ones in this attempt is rolled back before reporting which
1970
+ * bucket blocked.
1971
+ */
1972
+ tryAcquireBuckets(estimatedTokens) {
1973
+ const taken = [];
1974
+ const take = (bucket, amount) => {
1975
+ if (!bucket) return true;
1976
+ if (!bucket.tryTake(amount)) return false;
1977
+ taken.push({
1978
+ bucket,
1979
+ amount
1980
+ });
1981
+ return true;
1982
+ };
1983
+ if (!take(this.concurrency, 1)) return {
1984
+ ok: false,
1985
+ reason: "concurrency"
1986
+ };
1987
+ if (!take(this.requests, 1)) {
1988
+ for (const entry of taken) entry.bucket.give(entry.amount);
1989
+ return {
1990
+ ok: false,
1991
+ reason: "rpm"
1992
+ };
1993
+ }
1994
+ if (!take(this.tokens, estimatedTokens)) {
1995
+ for (const entry of taken) entry.bucket.give(entry.amount);
1996
+ return {
1997
+ ok: false,
1998
+ reason: "tpm"
1999
+ };
2000
+ }
2001
+ return { ok: true };
2002
+ }
2003
+ /** Drains the queue head first. Stops at the first waiter that still can't proceed, so no one is starved out of turn. */
2004
+ drain() {
2005
+ while (this.queue.length > 0) {
2006
+ const waiter = this.queue[0];
2007
+ const attempt = this.tryAcquireBuckets(waiter.estimatedTokens);
2008
+ if (!attempt.ok) {
2009
+ waiter.lastReason = attempt.reason;
2010
+ this.scheduleWake(attempt.reason, waiter.estimatedTokens);
2011
+ return;
2012
+ }
2013
+ const waitedMs = Date.now() - waiter.enqueuedAt;
2014
+ waiter.resolve({
2015
+ release: this.makeRelease(waiter.estimatedTokens),
2016
+ waitedMs,
2017
+ reason: waiter.lastReason
2018
+ });
2019
+ }
2020
+ }
2021
+ /**
2022
+ * Schedules a one-shot re-check of the queue for whenever the bucket
2023
+ * that's currently blocking the head waiter should next have enough
2024
+ * capacity. A no-op for a concurrency block (only `release` can clear
2025
+ * that) or while a wake is already pending.
2026
+ */
2027
+ scheduleWake(reason, estimatedTokens) {
2028
+ if (this.wakeTimer) return;
2029
+ const ms = reason === "rpm" ? this.requests?.msUntilAvailable(1) : reason === "tpm" ? this.tokens?.msUntilAvailable(estimatedTokens) : void 0;
2030
+ if (ms === void 0 || !Number.isFinite(ms)) return;
2031
+ const delay = Math.min(Math.max(1, Math.ceil(ms)), MAX_WAKE_DELAY_MS);
2032
+ this.wakeTimer = setTimeout(() => {
2033
+ this.wakeTimer = void 0;
2034
+ this.drain();
2035
+ }, delay);
2036
+ }
2037
+ /**
2038
+ * Builds the one-shot release closure for an acquired slot. Only the
2039
+ * concurrency bucket is given back on release; the requests-per-minute
2040
+ * bucket is a real spend that only recovers via its own refill, and the
2041
+ * tokens bucket is reconciled against `actualTokens` rather than fully
2042
+ * refunded, since real tokens really were spent.
2043
+ */
2044
+ makeRelease(estimatedTokens) {
2045
+ let released = false;
2046
+ return (actualTokens) => {
2047
+ if (released) return;
2048
+ released = true;
2049
+ this.concurrency?.give(1);
2050
+ if (this.tokens && actualTokens !== void 0 && Number.isFinite(actualTokens)) this.tokens.give(estimatedTokens - actualTokens);
2051
+ this.drain();
2052
+ };
2053
+ }
2054
+ };
2055
+
2056
+ //#endregion
2057
+ //#region src/vernLLM.ts
2058
+ /**
2059
+ * A resilient layer around an LLM chat completions client. This is VernLLM!
2060
+ *
2061
+ * Adds retry with backoff and jitter, per-attempt timeouts, an optional
2062
+ * circuit breaker, JSON parsing with optional schema validation, usage
2063
+ * tracking, and an optional response cache. All configurable, all opt-in
2064
+ * beyond sensible defaults.
2065
+ */
2066
+ var VernLLM = class {
2067
+ logger;
2068
+ /**
2069
+ * One `CallExecutor` per provider target: index 0 is the primary,
2070
+ * everything after it is a `fallback` target, in the order declared.
2071
+ * Each owns its own request building, retry/timeout, circuit breaker,
2072
+ * and rate limiter. `call()` walks this array in `runFallbackChain`,
2073
+ * moving to the next entry only when `fallbackOn` says to.
2074
+ */
2075
+ executors;
2076
+ /** Decides whether a failed target is followed by the next one or the chain stops. See `VernLLMOptions['fallbackOn']`. */
2077
+ fallbackOn;
2078
+ /** Reports a `'fallback'` event when the chain moves to the next target. Shared `onEvent` plumbing, same as every executor's. */
2079
+ reportEvent;
2080
+ /**
2081
+ * Owns cache key resolution, cache reads/writes, and in-flight
2082
+ * coalescing for `cachedCall()`. Independent of `executor`: it only
2083
+ * ever calls back into `this.call()` as an opaque function.
2084
+ */
2085
+ cacheOrchestrator;
2086
+ /**
2087
+ * @param options Client, model, and tunables. Defaults: `maxRetries` 1,
2088
+ * `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
2089
+ * `defaultTemperature` 0.2, `cache` an in-memory adapter,
2090
+ * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
2091
+ */
2092
+ constructor(options) {
2093
+ this.logger = options.logger ?? new ConsoleLogger(options.debug ?? false);
2094
+ const providerName = options.name ?? "primary";
2095
+ this.cacheOrchestrator = new CacheOrchestrator(options.cache ?? new InMemoryCacheAdapter(), this.logger);
2096
+ this.fallbackOn = options.fallbackOn ?? defaultFallbackOn;
2097
+ this.reportEvent = makeEventReporter(options.onEvent, this.logger);
2098
+ const primaryDefaultTemperature = options.defaultTemperature === void 0 ? .2 : options.defaultTemperature;
2099
+ const primaryTarget = {
2100
+ client: options.client,
2101
+ model: options.model,
2102
+ name: providerName,
2103
+ maxRetries: options.maxRetries,
2104
+ timeoutMs: options.timeoutMs,
2105
+ chunkIdleTimeoutMs: options.chunkIdleTimeoutMs,
2106
+ baseDelayMs: options.baseDelayMs,
2107
+ defaultMaxTokens: options.defaultMaxTokens,
2108
+ defaultTemperature: primaryDefaultTemperature,
2109
+ nonRetryableStatus: options.nonRetryableStatus,
2110
+ circuitBreaker: options.circuitBreaker,
2111
+ rateLimit: options.rateLimit
2112
+ };
2113
+ const declaredFallbacks = Array.isArray(options.fallback) ? options.fallback : options.fallback ? [options.fallback] : [];
2114
+ const targets = [primaryTarget, ...declaredFallbacks];
2115
+ this.executors = targets.map((target, i) => {
2116
+ const isFallback = i > 0;
2117
+ const name = target.name ?? (isFallback ? `fallback[${i - 1}]` : providerName);
2118
+ const breaker = buildCircuitBreaker(target.circuitBreaker, name, target.model, options.onEvent, this.logger);
2119
+ return new CallExecutor(name, target.client, target.model, {
2120
+ maxRetries: target.maxRetries ?? options.maxRetries ?? 1,
2121
+ timeoutMs: target.timeoutMs ?? options.timeoutMs ?? 25e3,
2122
+ chunkIdleTimeoutMs: target.chunkIdleTimeoutMs ?? options.chunkIdleTimeoutMs ?? 3e4,
2123
+ baseDelayMs: target.baseDelayMs ?? options.baseDelayMs ?? 500,
2124
+ defaultMaxTokens: target.defaultMaxTokens ?? options.defaultMaxTokens ?? 1e3,
2125
+ defaultTemperature: target.defaultTemperature === void 0 ? primaryDefaultTemperature : target.defaultTemperature,
2126
+ nonRetryableStatus: target.nonRetryableStatus ?? options.nonRetryableStatus ?? [
2127
+ 400,
2128
+ 401,
2129
+ 403,
2130
+ 404,
2131
+ 422
2132
+ ],
2133
+ parseJson: options.parseJson,
2134
+ logger: this.logger,
2135
+ redact: options.redact,
2136
+ onUsage: options.onUsage,
2137
+ onUsageFailure: options.onUsageFailure,
2138
+ onEvent: options.onEvent,
2139
+ breaker,
2140
+ limiter: target.rateLimit ? new RateLimiter(target.rateLimit) : void 0,
2141
+ isFallback
2142
+ });
2143
+ });
2144
+ }
2145
+ /** Logs a failed refundUsage attempt via the configured logger. */
2146
+ logRefundError(logMessage, error) {
2147
+ this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
2148
+ }
2149
+ /**
2150
+ * Walks `this.executors` in order, running `attempt` against each until
2151
+ * one succeeds or every target has failed. `run` on a lone target
2152
+ * (no `fallback` configured) throws exactly what it throws today: the
2153
+ * loop's single iteration path is unchanged from pre-fallback behavior.
2154
+ *
2155
+ * For streaming, `attempt` is `executor.runStream`, whose own retries
2156
+ * only cover *opening* the stream (see `CallExecutor.runStream`). A
2157
+ * mid-stream failure surfaces through `finalResult` after this function
2158
+ * has already returned, so it's never seen here and never falls over,
2159
+ * per the streaming limitation: splicing a second model's output into a
2160
+ * response the consumer has already partially rendered would corrupt
2161
+ * it.
2162
+ */
2163
+ async runFallbackChain(params, requestId, attempt, skipBreakerCheckForFirst = false) {
2164
+ const attempts = [];
2165
+ for (let i = 0; i < this.executors.length; i++) {
2166
+ const executor = this.executors[i];
2167
+ const startedAt = Date.now();
2168
+ let attemptCount = 0;
2169
+ try {
2170
+ if (!(i === 0 && skipBreakerCheckForFirst)) executor.assertBreakerClosed(params.model);
2171
+ const result = await attempt(executor, () => {
2172
+ attemptCount += 1;
2173
+ });
2174
+ return {
2175
+ result,
2176
+ executor,
2177
+ index: i,
2178
+ attemptCount
2179
+ };
2180
+ } catch (error) {
2181
+ const normalized = normalizeError(error, params.signal);
2182
+ attempts.push({
2183
+ index: i - 1,
2184
+ provider: executor.providerName,
2185
+ model: params.model ?? executor.model,
2186
+ error: normalized
2187
+ });
2188
+ const isLast = i === this.executors.length - 1;
2189
+ const policyDecision = this.fallbackOn(normalized, { isLastTarget: isLast });
2190
+ const decision = isLast ? "stop" : policyDecision;
2191
+ if (decision === "stop") throw attempts.length > 1 ? new FallbackExhaustedError(attempts) : normalized;
2192
+ const next = this.executors[i + 1];
2193
+ this.reportEvent({
2194
+ kind: "fallback",
2195
+ requestId,
2196
+ from: executor.providerName,
2197
+ to: next.providerName,
2198
+ fromIndex: i - 1,
2199
+ toIndex: i,
2200
+ error: normalized,
2201
+ elapsedMs: Date.now() - startedAt
2202
+ });
2203
+ }
2204
+ }
2205
+ throw new LLMError("No provider targets configured", "unknown");
2206
+ }
2207
+ async call(params) {
2208
+ if (params.signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
2209
+ const requestId = params.requestId ?? (0, crypto.randomUUID)();
2210
+ const soleTarget = this.executors.length === 1;
2211
+ if (soleTarget) this.executors[0].assertBreakerClosed(params.model);
2212
+ if (params.stream) return withReservedUsageForStream(params, async () => {
2213
+ const { result } = await this.runFallbackChain(params, requestId, (executor, onAttempt) => executor.runStream(params, requestId, onAttempt), soleTarget);
2214
+ return result;
2215
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
2216
+ return withReservedUsage(params, false, async () => {
2217
+ const { result, executor, index, attemptCount } = await this.runFallbackChain(params, requestId, (target, onAttempt) => target.run(params, requestId, onAttempt), soleTarget);
2218
+ if (params.meta) params.meta.current = {
2219
+ provider: executor.providerName,
2220
+ model: params.model ?? executor.model,
2221
+ fallbackIndex: index - 1,
2222
+ usedFallback: index > 0,
2223
+ attempts: attemptCount
2224
+ };
2225
+ return result;
2226
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
2227
+ }
2228
+ /**
2229
+ * Thin delegator kept private on `VernLLM` (rather than only existing on
2230
+ * `CacheOrchestrator`) since it's the one caching primitive exercised
2231
+ * directly by white-box tests, independent of the public `cachedCall()`
2232
+ * surface.
2233
+ */
2234
+ runCached(params) {
2235
+ return this.cacheOrchestrator.runCached(params);
2236
+ }
2237
+ /**
656
2238
  * Removes a cached response by key when the configured cache adapter
657
2239
  * supports deletion. Cache invalidation is the caller's responsibility;
658
2240
  * only the application knows when cached data is stale.
659
2241
  *
660
- * @param key - The raw cache key (resolved through the adapter's
2242
+ * @param key The raw cache key (resolved through the adapter's
661
2243
  * `resolveKey`, if any, before deletion).
662
2244
  */
663
2245
  async deleteCache(key) {
664
- if (!this.cache.delete) return;
665
- await this.cache.delete(await this.resolveCacheKey(key));
2246
+ await this.cacheOrchestrator.deleteCache(key);
666
2247
  }
667
- /**
668
- * Cache wrapper around caller-supplied logic. Concurrent misses for the
669
- * same `cacheKey` share a single in-flight call, avoiding cache stampedes.
670
- *
671
- * @param params - `cacheKey`, `ttl`, `fn` (the work to run on a cache
672
- * miss, typically `() => this.call(...)`), and optional
673
- * `reserveUsage`/`refundUsage`/`signal`. See `CachedCallParams`.
674
- * @returns The cached value on a hit, or the result of `fn()` on a miss.
675
- */
676
2248
  async cachedCall(params) {
677
- const resolvedKey = await this.resolveCacheKey(params.cacheKey);
678
- const resolvedParams = resolvedKey === params.cacheKey ? params : {
679
- ...params,
680
- cacheKey: resolvedKey
681
- };
682
- const cached = await this.cache.get(resolvedKey);
683
- if (cached.hit) return cached.value;
684
- const existing = this.inFlight.get(resolvedKey);
685
- if (existing) return withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
686
- return this.registerTrigger(resolvedParams, false);
687
- }
688
- /** Starts the shared fn() call for a cache miss and tracks it in the in-flight map until it settles. */
689
- registerTrigger(params, coalesced) {
690
- const resultPromise = withReservedUsage(params, coalesced, () => this.runAndCache(params), params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
691
- this.inFlight.set(params.cacheKey, resultPromise);
692
- resultPromise.catch(() => {}).finally(() => {
693
- this.inFlight.delete(params.cacheKey);
694
- });
695
- return resultPromise;
696
- }
697
- /** Runs `fn` and writes its result to the cache. */
698
- async runAndCache(params) {
699
- const result = await params.fn();
700
- try {
701
- await this.cache.set(params.cacheKey, result, params.ttl);
702
- } catch (error) {
703
- this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
704
- }
705
- return result;
706
- }
707
- /** Logs a failed refundUsage attempt via the configured logger. */
708
- logRefundError(logMessage, error) {
709
- this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
710
- }
711
- /**
712
- * Convenience wrapper composing `call` + `cachedCall`, so cached LLM calls
713
- * automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
714
- * `refundUsage` are read from the top-level params only.
715
- *
716
- * @param params - `cachedCall` params (`cacheKey`, `ttl`, etc, minus `fn`)
717
- * plus `call`, the `CallParams` to pass through to `this.call(...)`.
718
- * @returns The cached value on a hit, or the freshly-called result on a miss.
719
- */
720
- async cachedLLMCall(params) {
721
2249
  const { call: callParams,...cacheParams } = params;
722
- const { reserveUsage: innerReserveUsage, refundUsage: innerRefundUsage,...restCallParams } = callParams;
723
- if (innerReserveUsage || innerRefundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedLLMCall; set them at the top level instead.");
724
- return this.cachedCall({
2250
+ const { reserveUsage, refundUsage,...restCallParams } = callParams;
2251
+ if (reserveUsage || refundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedCall; set them at the top level instead.");
2252
+ if (restCallParams.stream) {
2253
+ const streamParams = restCallParams;
2254
+ return this.cacheOrchestrator.runCachedStream({
2255
+ ...cacheParams,
2256
+ openStream: () => this.call(streamParams)
2257
+ }, Boolean(restCallParams.tools));
2258
+ }
2259
+ return this.runCached({
725
2260
  ...cacheParams,
726
2261
  fn: () => this.call(restCallParams)
727
2262
  });
728
2263
  }
729
2264
  /**
2265
+ * @param model With `circuitBreaker.isolateByModel` on, returns that
2266
+ * model's own circuit state instead of the shared one. Ignored
2267
+ * otherwise. Omit for the shared circuit (the default) or, under
2268
+ * isolation, the state of calls that didn't resolve a model.
730
2269
  * @returns The current circuit breaker state (`'closed' | 'open' |
731
2270
  * 'half-open'`), or undefined if no circuit breaker was configured.
732
2271
  */
733
- getCircuitState() {
734
- return this.breaker?.getState();
2272
+ getCircuitState(model) {
2273
+ return this.executors[0].getCircuitState(model);
2274
+ }
2275
+ /**
2276
+ * @param model With `circuitBreaker.isolateByModel` on, returns each
2277
+ * target's circuit state for that model instead of its shared state.
2278
+ * Ignored otherwise. Omit for the shared circuit (the default) or, under
2279
+ * isolation, the state of calls that didn't resolve a model.
2280
+ * @returns The current circuit state for every target in declaration
2281
+ * order, including the primary and all fallback targets. Each entry
2282
+ * includes the target's provider name, chain index, whether it is a
2283
+ * fallback, and its circuit state, or undefined if that target has no
2284
+ * circuit breaker configured.
2285
+ */
2286
+ getCircuitStates(model) {
2287
+ return this.executors.map((executor, index) => ({
2288
+ provider: executor.providerName,
2289
+ index,
2290
+ isFallback: index > 0,
2291
+ state: executor.getCircuitState(model)
2292
+ }));
735
2293
  }
736
2294
  };
737
2295
 
738
2296
  //#endregion
739
- //#region src/internal/imageFormat.ts
2297
+ //#region src/adapters/internal/sse.ts
2298
+ /**
2299
+ * Parses a Server-Sent-Events byte/text stream into the JSON payload of
2300
+ * each `data:` frame, in arrival order. Generic over transport: works with
2301
+ * anything that hands back progressively-arriving `Uint8Array` or `string`
2302
+ * chunks via async iteration: native `fetch`'s `response.body` (wrapped
2303
+ * to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
2304
+ * Node `Readable` (already async-iterable, no wrapping needed), etc, so
2305
+ * this framing layer doesn't care which transport produced the bytes.
2306
+ *
2307
+ * Follows the SSE spec's frame-delimiting rules closely enough for LLM
2308
+ * streaming responses: frames are separated by a blank line, each frame
2309
+ * may carry one or more `data:` lines (joined with `\n` per spec when
2310
+ * there's more than one), `:`-prefixed lines are comments and ignored, and
2311
+ * other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
2312
+ * only needs the payload. A frame whose data is exactly `[DONE]` (the
2313
+ * sentinel several providers, notably OpenAI, send to mark stream end)
2314
+ * ends iteration without yielding it.
2315
+ *
2316
+ * Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
2317
+ * to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
2318
+ * alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
2319
+ * stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
2320
+ * lines.
2321
+ *
2322
+ * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
2323
+ * how malformed JSON is handled elsewhere in VernLLM.
2324
+ */
2325
+ async function* parseSseStream(source) {
2326
+ const decoder = new TextDecoder("utf-8", { fatal: true });
2327
+ let buffer = "";
2328
+ for await (const chunk of source) {
2329
+ let text;
2330
+ try {
2331
+ text = typeof chunk === "string" ? chunk : decoder.decode(chunk, { stream: true });
2332
+ } catch (cause) {
2333
+ throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
2334
+ }
2335
+ buffer = (buffer + text).replace(/\r\n/g, "\n").replace(/\r(?!$)/g, "\n");
2336
+ let boundary$1 = buffer.indexOf("\n\n");
2337
+ while (boundary$1 !== -1) {
2338
+ const frame = buffer.slice(0, boundary$1);
2339
+ buffer = buffer.slice(boundary$1 + 2);
2340
+ const event = parseSseFrame(frame);
2341
+ if (event === DONE) return;
2342
+ if (event !== NO_DATA) yield event;
2343
+ boundary$1 = buffer.indexOf("\n\n");
2344
+ }
2345
+ }
2346
+ try {
2347
+ buffer += decoder.decode();
2348
+ } catch (cause) {
2349
+ throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
2350
+ }
2351
+ buffer = buffer.replace(/\r$/, "\n");
2352
+ let boundary = buffer.indexOf("\n\n");
2353
+ while (boundary !== -1) {
2354
+ const frame = buffer.slice(0, boundary);
2355
+ buffer = buffer.slice(boundary + 2);
2356
+ const event = parseSseFrame(frame);
2357
+ if (event === DONE) return;
2358
+ if (event !== NO_DATA) yield event;
2359
+ boundary = buffer.indexOf("\n\n");
2360
+ }
2361
+ const trailing = buffer.trim();
2362
+ if (trailing) {
2363
+ const event = parseSseFrame(trailing);
2364
+ if (event !== DONE && event !== NO_DATA) yield event;
2365
+ }
2366
+ }
2367
+ const DONE = Symbol("sse-stream-done");
2368
+ const NO_DATA = Symbol("sse-frame-no-data");
2369
+ /**
2370
+ * Sentinel yielded by `parseSseStream` for a comment-only frame (no
2371
+ * `data:` payload), the mechanism providers use for SSE keep-alive
2372
+ * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
2373
+ * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
2374
+ */
2375
+ const SSE_PING = Symbol("sse-frame-ping");
2376
+ /** Extracts and JSON-parses the `data:` payload of one SSE frame (the text between two blank lines). */
2377
+ function parseSseFrame(frame) {
2378
+ const dataLines = [];
2379
+ let sawComment = false;
2380
+ for (const line of frame.split("\n")) {
2381
+ if (line.startsWith(":")) {
2382
+ sawComment = true;
2383
+ continue;
2384
+ }
2385
+ if (!line.startsWith("data:")) continue;
2386
+ dataLines.push(line.startsWith("data: ") ? line.slice(6) : line.slice(5));
2387
+ }
2388
+ if (!dataLines.length) return sawComment ? SSE_PING : NO_DATA;
2389
+ const data = dataLines.join("\n");
2390
+ if (data === "[DONE]") return DONE;
2391
+ try {
2392
+ return JSON.parse(data);
2393
+ } catch (cause) {
2394
+ throw new LLMError(`Invalid JSON in SSE frame: ${data.slice(0, 200)}`, "parse", void 0, void 0, cause);
2395
+ }
2396
+ }
2397
+
2398
+ //#endregion
2399
+ //#region src/adapters/internal/imageFormat.ts
740
2400
  /**
741
2401
  * MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
742
2402
  * the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
@@ -760,6 +2420,14 @@ function assertSupportedImageMimeType(mimeType) {
760
2420
  throw new LLMError(`Unsupported image mimeType "${mimeType}": expected one of ${SUPPORTED_IMAGE_MIME_TYPES.join(", ")}`, "validation");
761
2421
  }
762
2422
 
2423
+ //#endregion
2424
+ //#region src/adapters/internal/nativeStructuredOutput.ts
2425
+ /** Resolves whether `model` is covered by a caller-supplied allow-list/predicate. */
2426
+ function supportsNativeStructuredOutput(model, override) {
2427
+ if (!override) return false;
2428
+ return Array.isArray(override) ? override.includes(model) : override(model);
2429
+ }
2430
+
763
2431
  //#endregion
764
2432
  //#region src/adapters/anthropic.ts
765
2433
  /**
@@ -782,67 +2450,305 @@ function toAnthropicContent(blocks) {
782
2450
  });
783
2451
  }
784
2452
  /**
2453
+ * Asserts a caller-supplied JSON Schema is an object schema before it's
2454
+ * used as Anthropic's `Tool.input_schema`, which (like every other
2455
+ * provider's function-calling API) requires `type: 'object'`. VernLLM's own
2456
+ * public `tools`/`jsonSchema` APIs accept freeform `Record<string,
2457
+ * unknown>` JSON Schema, so nothing upstream guarantees this at compile
2458
+ * time; this is the runtime check that stands in for that, so a schema
2459
+ * missing (or mistyping) `type: 'object'` fails loudly and immediately
2460
+ * instead of being silently forwarded to Anthropic malformed.
2461
+ */
2462
+ function assertObjectSchema(schema, toolName) {
2463
+ if (schema.type !== "object") throw new LLMError(`Tool "${toolName}"'s schema must have "type": "object" (Anthropic requires object-shaped tool parameters).`, "validation");
2464
+ return schema;
2465
+ }
2466
+ /**
2467
+ * Translates VernLLM's OpenAI-shaped wire `tool_choice` into Anthropic's
2468
+ * `{ type: 'auto' | 'any' | 'none' | 'tool', name? }` shape. `'required'`
2469
+ * maps to `'any'` (Anthropic's "must call some tool" equivalent).
2470
+ */
2471
+ function toAnthropicToolChoice(toolChoice) {
2472
+ if (!toolChoice || toolChoice === "auto") return { type: "auto" };
2473
+ if (toolChoice === "none") return { type: "none" };
2474
+ if (toolChoice === "required") return { type: "any" };
2475
+ return {
2476
+ type: "tool",
2477
+ name: toolChoice.function.name
2478
+ };
2479
+ }
2480
+ /**
2481
+ * Maps VernLLM's OpenAI-shaped wire `tools`/`tool_choice` into Anthropic's
2482
+ * `tools`/`tool_choice` shape. Shared by the two call sites that build real
2483
+ * (non-schema-forced) tool definitions: the plain tools-only branch, and
2484
+ * the native-structured-output branch, which sends real tools alongside
2485
+ * `output_config` rather than instead of it.
2486
+ */
2487
+ function buildAnthropicTools(tools, toolChoiceParam) {
2488
+ return {
2489
+ tools: tools.map((t) => ({
2490
+ name: t.function.name,
2491
+ description: t.function.description,
2492
+ input_schema: assertObjectSchema(t.function.parameters, t.function.name)
2493
+ })),
2494
+ toolChoice: toAnthropicToolChoice(toolChoiceParam)
2495
+ };
2496
+ }
2497
+ /**
2498
+ * Builds the Anthropic-shaped request body from VernLLM's wire params,
2499
+ * shared between `create` and `createStream` so both go through identical
2500
+ * translation (system prompt, message shaping, and the jsonSchema →
2501
+ * forced-single-tool mapping all happen exactly once, not once per entry
2502
+ * point).
2503
+ *
2504
+ * Returns `toolName` alongside the body: when set, the model was forced to
2505
+ * call a single synthetic tool standing in for `jsonSchema` output (the
2506
+ * legacy path, for models without native structured-output support), and
2507
+ * both `create` and `createStream` need to know this so they can unwrap
2508
+ * that tool call back into plain text content instead of treating it like
2509
+ * a real tool call. On the native path (model supports `output_config`),
2510
+ * `toolName` is `undefined`: the schema-conforming JSON already arrives as
2511
+ * ordinary text content, nothing to unwrap, and any real tool calls in
2512
+ * `params.tools` are left for the normal, non-forced tool-call handling
2513
+ * both `create` and `createStream` already do when `toolName` is unset.
2514
+ */
2515
+ function buildAnthropicRequestBody(params, nativeStructuredOutputModels) {
2516
+ const systemMessage = params.messages.find((m) => m.role === "system");
2517
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
2518
+ const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
2519
+ const schemaName = jsonSchema?.name.trim();
2520
+ if (jsonSchema && !schemaName) throw new LLMError("json_schema.name must not be empty.", "validation");
2521
+ const isNative = Boolean(jsonSchema) && supportsNativeStructuredOutput(params.model, nativeStructuredOutputModels);
2522
+ if (jsonSchema && params.tools?.length && !isNative) throw new LLMError(`Anthropic model "${params.model}" is not covered by nativeStructuredOutputModels, so \`jsonSchema\` is emulated as a forced single tool call there, which collides with the \`tools\` you also provided. Either drop \`tools\` or \`jsonSchema\` for this call, or pass this model in fromAnthropic's \`nativeStructuredOutputModels\` option once you've confirmed it supports Anthropic's \`output_config.format\`.`, "validation");
2523
+ let toolName;
2524
+ let jsonInstruction;
2525
+ let outputFormat;
2526
+ let tools;
2527
+ let toolChoice;
2528
+ if (jsonSchema && isNative) {
2529
+ outputFormat = {
2530
+ type: "json_schema",
2531
+ schema: jsonSchema.schema
2532
+ };
2533
+ if (params.tools?.length) ({tools, toolChoice} = buildAnthropicTools(params.tools, params.tool_choice));
2534
+ } else if (jsonSchema && schemaName) {
2535
+ const { schema, description, strict } = jsonSchema;
2536
+ toolName = schemaName;
2537
+ tools = [{
2538
+ name: toolName,
2539
+ description,
2540
+ input_schema: assertObjectSchema(schema, toolName),
2541
+ strict
2542
+ }];
2543
+ toolChoice = {
2544
+ type: "tool",
2545
+ name: toolName
2546
+ };
2547
+ } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
2548
+ if (!jsonSchema && params.tools?.length) ({tools, toolChoice} = buildAnthropicTools(params.tools, params.tool_choice));
2549
+ const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
2550
+ const body = {
2551
+ model: params.model,
2552
+ max_tokens: params.max_tokens,
2553
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
2554
+ system: system || void 0,
2555
+ messages: mergeConsecutiveToolResults$1(conversationMessages.map((m) => toAnthropicMessage(m))),
2556
+ ...tools ? {
2557
+ tools,
2558
+ tool_choice: toolChoice
2559
+ } : {},
2560
+ ...outputFormat ? { output_config: { format: outputFormat } } : {}
2561
+ };
2562
+ return {
2563
+ body,
2564
+ toolName
2565
+ };
2566
+ }
2567
+ /**
785
2568
  * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
786
2569
  * interface VernLLM uses for OpenAI/Groq.
787
2570
  *
788
- * `response_format: json_schema` is mapped to Anthropic's forced tool-use:
789
- * a single tool is defined with `input_schema` set to the caller's schema,
790
- * `description` forwarded when provided, and `strict` forwarded when set.
791
- * `tool_choice` forces the model to call it. Provider-constrained schema
792
- * matching applies only when `strict: true` is forwarded and supported.
2571
+ * `response_format: json_schema`, on a model covered by
2572
+ * `options.nativeStructuredOutputModels`, is sent as `output_config.format`,
2573
+ * its own request field, independent of `tools`/`tool_choice`, so it can be
2574
+ * combined with real, caller-supplied `tools` in the same request. Only
2575
+ * `type` and `schema` are sent on this path, the real Anthropic API's
2576
+ * `output_config.format` has no `name`/`description`/`strict` fields.
2577
+ *
2578
+ * On any other model (the default, since `nativeStructuredOutputModels` is
2579
+ * opt-in), `response_format: json_schema` is mapped to Anthropic's forced
2580
+ * tool-use instead: a single tool is defined with `input_schema` set to
2581
+ * the caller's schema, `description` forwarded when provided, and `strict`
2582
+ * forwarded when set, and `tool_choice` forces the model to call it. This
2583
+ * legacy path cannot be combined with real `tools` (both would need the
2584
+ * same `tools`/`tool_choice` field), and a call that tries throws
2585
+ * `LLMError('validation')` before reaching the API. Provider-constrained
2586
+ * schema matching applies only when `strict: true` is forwarded and
2587
+ * supported.
793
2588
  *
794
2589
  * `response_format: json_object` (no schema to build a tool from) falls
795
2590
  * back to a system-prompt instruction, since there's nothing to constrain
796
- * generation against.
797
- */
798
- function fromAnthropic(anthropicClient) {
799
- return { chat: { completions: { async create(params, options) {
800
- const systemMessage = params.messages.find((m) => m.role === "system");
801
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
802
- const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name : void 0;
803
- let jsonInstruction;
804
- let tools;
805
- if (params.response_format?.type === "json_schema" && toolName) {
806
- const { schema, description, strict } = params.response_format.json_schema;
807
- tools = [{
808
- name: toolName,
809
- description,
810
- input_schema: schema,
811
- strict
812
- }];
813
- } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
814
- const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
815
- const response = await anthropicClient.messages.create({
816
- model: params.model,
817
- max_tokens: params.max_tokens,
818
- temperature: params.temperature,
819
- system: system || void 0,
820
- messages: conversationMessages.map((m) => ({
821
- role: m.role,
822
- content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content
823
- })),
824
- ...tools ? {
825
- tools,
826
- tool_choice: {
827
- type: "tool",
828
- name: toolName
2591
+ * generation against. Unlike `jsonSchema`, this combines with real `tools`
2592
+ * freely on every model: it's a prompt nudge, not a request field, so
2593
+ * there's nothing for it to collide with.
2594
+ */
2595
+ function fromAnthropic(anthropicClient, options) {
2596
+ const nativeStructuredOutputModels = options?.nativeStructuredOutputModels;
2597
+ const rawMessagesCreate = anthropicClient.messages.create.bind(anthropicClient.messages);
2598
+ return { chat: { completions: {
2599
+ async create(params, options$1) {
2600
+ const { body, toolName } = buildAnthropicRequestBody(params, nativeStructuredOutputModels);
2601
+ const response = await anthropicClient.messages.create(body, options$1);
2602
+ let text;
2603
+ let wireToolCalls;
2604
+ if (toolName) {
2605
+ const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
2606
+ if (!toolUse) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
2607
+ if (!toolUse.input || typeof toolUse.input !== "object" || Array.isArray(toolUse.input)) throw new LLMError(`Anthropic returned invalid structured output for tool "${toolName}". Expected an object.`, "validation");
2608
+ text = JSON.stringify(toolUse.input);
2609
+ } else {
2610
+ text = response.content.filter((block) => block.type === "text").map((block) => block.text ?? "").join("");
2611
+ const toolUses = response.content.filter((block) => block.type === "tool_use");
2612
+ if (toolUses.length) wireToolCalls = toolUses.map((block) => ({
2613
+ id: block.id,
2614
+ type: "function",
2615
+ function: {
2616
+ name: block.name,
2617
+ arguments: JSON.stringify(block.input ?? {})
2618
+ }
2619
+ }));
2620
+ }
2621
+ return {
2622
+ choices: [{ message: {
2623
+ content: text,
2624
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
2625
+ } }],
2626
+ usage: {
2627
+ prompt_tokens: response.usage?.input_tokens,
2628
+ completion_tokens: response.usage?.output_tokens,
2629
+ total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
829
2630
  }
830
- } : {}
831
- }, options);
832
- let text;
833
- if (toolName) {
834
- const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
835
- text = toolUse ? JSON.stringify(toolUse.input) : "";
836
- } else text = response.content.find((block) => block.type === "text")?.text ?? "";
837
- return {
838
- choices: [{ message: { content: text } }],
839
- usage: {
840
- prompt_tokens: response.usage?.input_tokens,
841
- completion_tokens: response.usage?.output_tokens,
842
- total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
2631
+ };
2632
+ },
2633
+ async *createStream(params, options$1) {
2634
+ const { body, toolName } = buildAnthropicRequestBody(params, nativeStructuredOutputModels);
2635
+ const stream = await rawMessagesCreate({
2636
+ ...body,
2637
+ stream: true
2638
+ }, options$1);
2639
+ const blockKinds = new Map();
2640
+ let inputTokens = 0;
2641
+ let sawJsonTool = false;
2642
+ for await (const event of stream) if (event.type === "message_start") inputTokens = event.message.usage?.input_tokens ?? 0;
2643
+ else if (event.type === "content_block_start") if (event.content_block.type === "tool_use") {
2644
+ const kind = event.content_block.name === toolName ? "json-tool" : "tool_use";
2645
+ blockKinds.set(event.index, kind);
2646
+ if (kind === "json-tool") sawJsonTool = true;
2647
+ else if (!toolName) yield {
2648
+ type: "tool_call_delta",
2649
+ index: event.index,
2650
+ id: event.content_block.id,
2651
+ name: event.content_block.name
2652
+ };
2653
+ } else blockKinds.set(event.index, "text");
2654
+ else if (event.type === "content_block_delta") {
2655
+ if (event.delta.type === "text_delta") {
2656
+ if (!toolName) yield {
2657
+ type: "text-delta",
2658
+ delta: event.delta.text
2659
+ };
2660
+ } else if (event.delta.type === "input_json_delta") {
2661
+ const kind = blockKinds.get(event.index);
2662
+ if (kind === "json-tool") yield {
2663
+ type: "text-delta",
2664
+ delta: event.delta.partial_json
2665
+ };
2666
+ else if (!toolName) yield {
2667
+ type: "tool_call_delta",
2668
+ index: event.index,
2669
+ argumentsDelta: event.delta.partial_json
2670
+ };
2671
+ }
2672
+ } else if (event.type === "message_delta") {
2673
+ const outputTokens = event.usage?.output_tokens ?? 0;
2674
+ yield {
2675
+ type: "usage",
2676
+ usage: {
2677
+ prompt_tokens: inputTokens,
2678
+ completion_tokens: outputTokens,
2679
+ total_tokens: inputTokens + outputTokens
2680
+ }
2681
+ };
2682
+ } else if (event.type === "ping") yield { type: "ping" };
2683
+ if (toolName && !sawJsonTool) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
2684
+ }
2685
+ } } };
2686
+ }
2687
+ /**
2688
+ * Anthropic requires strict role alternation, so the per-wire-message
2689
+ * mapping above (one `{role:'user', content:[tool_result]}` per VernLLM
2690
+ * wire tool message) needs merging back together when an assistant turn
2691
+ * requested more than one tool: multiple consecutive user turns would
2692
+ * violate that alternation, and Anthropic's API rejects it outright. This
2693
+ * merges any run of tool-result-only user messages into one, with all
2694
+ * their tool_result blocks combined, the shape Anthropic expects for "here
2695
+ * are the results of everything you just asked for."
2696
+ */
2697
+ function mergeConsecutiveToolResults$1(messages) {
2698
+ const isToolResultOnly = (m) => m.role === "user" && Array.isArray(m.content) && m.content.length > 0 && m.content.every((b) => b.type === "tool_result");
2699
+ const merged = [];
2700
+ for (const m of messages) {
2701
+ const prev = merged.at(-1);
2702
+ if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
2703
+ else merged.push(m);
2704
+ }
2705
+ return merged;
2706
+ }
2707
+ /**
2708
+ * Translates one VernLLM wire message (OpenAI-shaped: plain user/assistant
2709
+ * turns, an assistant turn with `tool_calls`, or a `tool` turn) into
2710
+ * Anthropic's `{ role: 'user' | 'assistant', content }` shape.
2711
+ */
2712
+ function toAnthropicMessage(m) {
2713
+ if (m.role === "tool") return {
2714
+ role: "user",
2715
+ content: [{
2716
+ type: "tool_result",
2717
+ tool_use_id: m.tool_call_id,
2718
+ content: m.content,
2719
+ ...m.is_error ? { is_error: true } : {}
2720
+ }]
2721
+ };
2722
+ if (m.role === "assistant" && m.tool_calls?.length) {
2723
+ const blocks = [];
2724
+ if (m.content) blocks.push({
2725
+ type: "text",
2726
+ text: m.content
2727
+ });
2728
+ for (const tc of m.tool_calls) {
2729
+ let input;
2730
+ try {
2731
+ input = tc.function.arguments.trim() ? JSON.parse(tc.function.arguments) : {};
2732
+ } catch (cause) {
2733
+ throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
843
2734
  }
2735
+ if (input === null || Array.isArray(input) || typeof input !== "object") throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) arguments must be a JSON object.`, "validation");
2736
+ blocks.push({
2737
+ type: "tool_use",
2738
+ id: tc.id,
2739
+ name: tc.function.name,
2740
+ input
2741
+ });
2742
+ }
2743
+ return {
2744
+ role: "assistant",
2745
+ content: blocks
844
2746
  };
845
- } } } };
2747
+ }
2748
+ return {
2749
+ role: m.role,
2750
+ content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content ?? ""
2751
+ };
846
2752
  }
847
2753
 
848
2754
  //#endregion
@@ -859,6 +2765,122 @@ function toGeminiParts(blocks) {
859
2765
  data: block.data
860
2766
  } } : { text: block.text });
861
2767
  }
2768
+ /** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Gemini's `functionCallingConfig`. */
2769
+ function toGeminiToolConfig(toolChoice) {
2770
+ if (!toolChoice || toolChoice === "auto") return { functionCallingConfig: { mode: "AUTO" } };
2771
+ if (toolChoice === "none") return { functionCallingConfig: { mode: "NONE" } };
2772
+ if (toolChoice === "required") return { functionCallingConfig: { mode: "ANY" } };
2773
+ return { functionCallingConfig: {
2774
+ mode: "ANY",
2775
+ allowedFunctionNames: [toolChoice.function.name]
2776
+ } };
2777
+ }
2778
+ /**
2779
+ * Translates one VernLLM wire message into a Gemini `contents` entry.
2780
+ * Gemini has no separate 'tool' role: a prior assistant tool request
2781
+ * becomes a `'model'` turn with `functionCall` parts, and its result
2782
+ * becomes a `'user'` turn with `functionResponse` parts.
2783
+ */
2784
+ function toGeminiContent(m) {
2785
+ if (m.role === "tool") return {
2786
+ role: "user",
2787
+ parts: [{ functionResponse: {
2788
+ name: m.tool_call_id,
2789
+ response: parseToolResult(m.content)
2790
+ } }]
2791
+ };
2792
+ if (m.role === "assistant" && m.tool_calls?.length) {
2793
+ const parts = [];
2794
+ if (typeof m.content === "string" && m.content) parts.push({ text: m.content });
2795
+ parts.push(...m.tool_calls.map((tc) => ({ functionCall: {
2796
+ name: tc.function.name,
2797
+ args: parseToolArguments(tc.function.arguments, tc.function.name)
2798
+ } })));
2799
+ return {
2800
+ role: "model",
2801
+ parts
2802
+ };
2803
+ }
2804
+ return {
2805
+ role: m.role === "assistant" ? "model" : "user",
2806
+ parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content ?? "" }]
2807
+ };
2808
+ }
2809
+ function parseToolArguments(text, toolName) {
2810
+ let parsed;
2811
+ try {
2812
+ parsed = text.trim() ? JSON.parse(text) : {};
2813
+ } catch (cause) {
2814
+ throw new LLMError(`Tool call "${toolName}" arguments are not valid JSON.`, "validation", void 0, void 0, cause);
2815
+ }
2816
+ if (!parsed || Array.isArray(parsed) || typeof parsed !== "object") throw new LLMError(`Tool call "${toolName}" arguments must be a JSON object.`, "validation");
2817
+ return parsed;
2818
+ }
2819
+ function parseToolResult(text) {
2820
+ try {
2821
+ return text.trim() ? JSON.parse(text) : "";
2822
+ } catch {
2823
+ return text;
2824
+ }
2825
+ }
2826
+ /**
2827
+ * Gemini expects the results of everything the model asked for in one turn
2828
+ * to arrive together as multiple `functionResponse` parts on a single
2829
+ * `'user'` entry, not as separate consecutive `'user'` entries. The
2830
+ * per-wire-message mapping above produces one `'user'` entry per VernLLM
2831
+ * wire tool message, so when an assistant turn requested more than one
2832
+ * tool, this merges the resulting run of functionResponse-only `'user'`
2833
+ * entries back into one.
2834
+ */
2835
+ function mergeConsecutiveFunctionResponses(contents) {
2836
+ const isFunctionResponseOnly = (c) => c.role === "user" && c.parts.length > 0 && c.parts.every((p) => "functionResponse" in p);
2837
+ const merged = [];
2838
+ for (const c of contents) {
2839
+ const prev = merged.at(-1);
2840
+ if (isFunctionResponseOnly(c) && prev && isFunctionResponseOnly(prev)) prev.parts.push(...c.parts);
2841
+ else merged.push(c);
2842
+ }
2843
+ return merged;
2844
+ }
2845
+ /**
2846
+ * Builds the Gemini-shaped request from VernLLM's wire params, shared
2847
+ * between `create` and `createStream` so both go through identical
2848
+ * translation (contents shaping, `responseSchema`/`responseMimeType`
2849
+ * mapping, and tool/toolConfig translation all happen exactly once).
2850
+ * `abortSignal` is folded into `config` by the caller (`create`/
2851
+ * `createStream`), once the request options are available.
2852
+ */
2853
+ function buildGeminiRequest(params) {
2854
+ const systemMessage = params.messages.find((m) => m.role === "system");
2855
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
2856
+ const wantsJson = Boolean(params.response_format);
2857
+ const config = {
2858
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
2859
+ maxOutputTokens: params.max_tokens,
2860
+ ...systemMessage ? { systemInstruction: { parts: [{ text: systemMessage.content }] } } : {}
2861
+ };
2862
+ if (wantsJson) config.responseMimeType = "application/json";
2863
+ if (params.response_format?.type === "json_schema") {
2864
+ const { schema, description } = params.response_format.json_schema;
2865
+ config.responseSchema = {
2866
+ ...schema,
2867
+ ...description ? { description } : {}
2868
+ };
2869
+ }
2870
+ if (params.tools?.length) {
2871
+ config.tools = [{ functionDeclarations: params.tools.map((t) => ({
2872
+ name: t.function.name,
2873
+ description: t.function.description,
2874
+ parameters: t.function.parameters
2875
+ })) }];
2876
+ config.toolConfig = toGeminiToolConfig(params.tool_choice);
2877
+ }
2878
+ return {
2879
+ model: params.model,
2880
+ contents: mergeConsecutiveFunctionResponses(conversationMessages.map((m) => toGeminiContent(m))),
2881
+ config
2882
+ };
2883
+ }
862
2884
  /**
863
2885
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
864
2886
  * uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
@@ -869,43 +2891,100 @@ function toGeminiParts(blocks) {
869
2891
  * `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
870
2892
  * models use a token budget, not an effort tier, so it's dropped, same as
871
2893
  * Anthropic.
2894
+ *
2895
+ * `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
2896
+ * `tool_choice` maps to `toolConfig.functionCallingConfig`. Gemini accepts
2897
+ * `responseSchema` and `tools` in the same request natively, so both are
2898
+ * set independently here and no special-casing is needed for the
2899
+ * combination, unlike `fromAnthropic`/`fromBedrock`.
2900
+ *
2901
+ * `createStream` calls `generateContentStream` (optional on `GeminiClient`
2902
+ *, required only if the caller sets `stream: true`) and translates each
2903
+ * partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
2904
+ * Gemini's own function-calling API doesn't stream tool-call arguments
2905
+ * incrementally: a `functionCall` part always arrives whole in one chunk,
2906
+ * so each one is emitted as a single, complete `tool_call_delta` (a
2907
+ * one-shot "delta" containing the full arguments) rather than accumulated
2908
+ * fragments, that's a real difference in the underlying API, not
2909
+ * something this adapter can smooth over. `usageMetadata` is (per Gemini's
2910
+ * own behavior) only reliably present on the last chunk, so the `usage`
2911
+ * `WireStreamChunk` is emitted once, after the stream completes, from
2912
+ * whichever chunk's `usageMetadata` was seen last.
872
2913
  */
873
2914
  function fromGemini(geminiClient) {
874
- return { chat: { completions: { async create(params, options) {
875
- const systemMessage = params.messages.find((m) => m.role === "system");
876
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
877
- const wantsJson = Boolean(params.response_format);
878
- const generationConfig = {
879
- temperature: params.temperature,
880
- maxOutputTokens: params.max_tokens
881
- };
882
- if (wantsJson) generationConfig.responseMimeType = "application/json";
883
- if (params.response_format?.type === "json_schema") {
884
- const { schema, description } = params.response_format.json_schema;
885
- generationConfig.responseSchema = {
886
- ...schema,
887
- ...description ? { description } : {}
2915
+ return { chat: { completions: {
2916
+ async create(params, options) {
2917
+ const request = buildGeminiRequest(params);
2918
+ request.config = {
2919
+ ...request.config,
2920
+ abortSignal: options.signal
888
2921
  };
889
- }
890
- const response = await geminiClient.generateContent({
891
- model: params.model,
892
- contents: conversationMessages.map((m) => ({
893
- role: m.role === "assistant" ? "model" : "user",
894
- parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content }]
895
- })),
896
- systemInstruction: systemMessage ? { parts: [{ text: systemMessage.content }] } : void 0,
897
- generationConfig
898
- }, options);
899
- const text = response.candidates?.[0]?.content?.parts?.map((p) => p.text ?? "").join("") ?? "";
900
- return {
901
- choices: [{ message: { content: text } }],
902
- usage: {
903
- prompt_tokens: response.usageMetadata?.promptTokenCount,
904
- completion_tokens: response.usageMetadata?.candidatesTokenCount,
905
- total_tokens: response.usageMetadata?.totalTokenCount
2922
+ const response = await geminiClient.generateContent(request);
2923
+ const parts = response.candidates?.[0]?.content?.parts ?? [];
2924
+ const text = parts.map((p) => p.text ?? "").join("");
2925
+ const functionCalls = parts.filter((p) => p.functionCall);
2926
+ let wireToolCalls;
2927
+ if (functionCalls.length) wireToolCalls = functionCalls.map((p) => ({
2928
+ id: p.functionCall.name,
2929
+ type: "function",
2930
+ function: {
2931
+ name: p.functionCall.name,
2932
+ arguments: JSON.stringify(p.functionCall.args ?? {})
2933
+ }
2934
+ }));
2935
+ return {
2936
+ choices: [{ message: {
2937
+ content: text,
2938
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
2939
+ } }],
2940
+ usage: {
2941
+ prompt_tokens: response.usageMetadata?.promptTokenCount,
2942
+ completion_tokens: response.usageMetadata?.candidatesTokenCount,
2943
+ total_tokens: response.usageMetadata?.totalTokenCount
2944
+ }
2945
+ };
2946
+ },
2947
+ async *createStream(params, options) {
2948
+ if (!geminiClient.generateContentStream) throw new LLMError("stream: true requires a Gemini client with generateContentStream", "validation");
2949
+ const request = buildGeminiRequest(params);
2950
+ request.config = {
2951
+ ...request.config,
2952
+ abortSignal: options.signal
2953
+ };
2954
+ const stream = await geminiClient.generateContentStream(request);
2955
+ let toolCallIndex = 0;
2956
+ let lastUsage;
2957
+ for await (const chunk of stream) {
2958
+ const parts = chunk.candidates?.[0]?.content?.parts ?? [];
2959
+ for (const part of parts) {
2960
+ if (part.text) yield {
2961
+ type: "text-delta",
2962
+ delta: part.text
2963
+ };
2964
+ if (part.functionCall) {
2965
+ yield {
2966
+ type: "tool_call_delta",
2967
+ index: toolCallIndex,
2968
+ id: part.functionCall.name,
2969
+ name: part.functionCall.name,
2970
+ argumentsDelta: JSON.stringify(part.functionCall.args ?? {}),
2971
+ complete: true
2972
+ };
2973
+ toolCallIndex++;
2974
+ }
2975
+ }
2976
+ if (chunk.usageMetadata) lastUsage = chunk.usageMetadata;
906
2977
  }
907
- };
908
- } } } };
2978
+ if (lastUsage) yield {
2979
+ type: "usage",
2980
+ usage: {
2981
+ prompt_tokens: lastUsage.promptTokenCount,
2982
+ completion_tokens: lastUsage.candidatesTokenCount,
2983
+ total_tokens: lastUsage.totalTokenCount
2984
+ }
2985
+ };
2986
+ }
2987
+ } } };
909
2988
  }
910
2989
 
911
2990
  //#endregion
@@ -941,6 +3020,99 @@ function toBedrockContent(blocks) {
941
3020
  } } : { text: block.text });
942
3021
  }
943
3022
  /**
3023
+ * Maps VernLLM's OpenAI-shaped wire `tools`/`tool_choice` into Converse's
3024
+ * `toolConfig` shape. Shared by the two call sites that build real
3025
+ * (non-schema-forced) tool definitions: the plain tools-only branch, and
3026
+ * the native-structured-output branch, which sends real tools alongside
3027
+ * `outputConfig` rather than instead of it.
3028
+ */
3029
+ function buildBedrockToolConfig(tools, toolChoiceParam) {
3030
+ return {
3031
+ tools: tools.map((t) => ({ toolSpec: {
3032
+ name: t.function.name,
3033
+ description: t.function.description,
3034
+ inputSchema: { json: t.function.parameters }
3035
+ } })),
3036
+ toolChoice: toBedrockToolChoice(toolChoiceParam)
3037
+ };
3038
+ }
3039
+ /**
3040
+ * Builds the Converse-shaped request from VernLLM's wire params, shared
3041
+ * between `create` and `createStream` so both go through identical
3042
+ * translation (system prompt, message shaping, the jsonSchema →
3043
+ * forced-single-tool mapping, and the `toolUseSupportedModels` preflight
3044
+ * check all happen exactly once).
3045
+ *
3046
+ * Returns `toolName` alongside the request: when set, the model was forced
3047
+ * to call a single synthetic tool standing in for `jsonSchema` output (the
3048
+ * legacy path, for models not covered by `nativeStructuredOutputModels`),
3049
+ * and both `create` and `createStream` need to know this so they can
3050
+ * unwrap that tool call back into plain text content instead of treating
3051
+ * it like a real tool call. On the native path (model covered by
3052
+ * `nativeStructuredOutputModels`), `toolName` is `undefined`: the
3053
+ * schema-conforming JSON already arrives as ordinary text content, nothing
3054
+ * to unwrap, and any real tool calls in `params.tools` are left for the
3055
+ * normal, non-forced tool-call handling both `create` and `createStream`
3056
+ * already do when `toolName` is unset.
3057
+ */
3058
+ function buildBedrockRequest(params, toolUseSupportedModels, nativeStructuredOutputModels) {
3059
+ const systemMessage = params.messages.find((m) => m.role === "system");
3060
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
3061
+ const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
3062
+ const schemaName = jsonSchema?.name.trim();
3063
+ if (jsonSchema && !schemaName) throw new LLMError("json_schema.name must not be empty.", "validation");
3064
+ const isNative = Boolean(jsonSchema) && supportsNativeStructuredOutput(params.model, nativeStructuredOutputModels);
3065
+ if (jsonSchema && params.tools?.length && !isNative) throw new LLMError(`Bedrock model "${params.model}" is not covered by nativeStructuredOutputModels, so \`jsonSchema\` is emulated as a forced single tool call there (via \`toolConfig\`), which collides with the \`tools\` you also provided. Either drop \`tools\` or \`jsonSchema\` for this call, or pass this model in fromBedrock's \`nativeStructuredOutputModels\` option once you've confirmed it supports Converse's \`outputConfig.textFormat\`.`, "validation");
3066
+ let toolName;
3067
+ let jsonInstruction;
3068
+ let toolConfig;
3069
+ let outputConfig;
3070
+ if (jsonSchema && isNative) {
3071
+ const { schema, description } = jsonSchema;
3072
+ outputConfig = { textFormat: {
3073
+ type: "json_schema",
3074
+ structure: { jsonSchema: {
3075
+ schema: JSON.stringify(schema),
3076
+ name: schemaName,
3077
+ description
3078
+ } }
3079
+ } };
3080
+ } else if (jsonSchema && schemaName) {
3081
+ const { schema, description, strict } = jsonSchema;
3082
+ toolName = schemaName;
3083
+ toolConfig = {
3084
+ tools: [{ toolSpec: {
3085
+ name: toolName,
3086
+ description,
3087
+ inputSchema: { json: schema },
3088
+ strict
3089
+ } }],
3090
+ toolChoice: { tool: { name: toolName } }
3091
+ };
3092
+ } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
3093
+ if (params.tools?.length && !toolName) toolConfig = buildBedrockToolConfig(params.tools, params.tool_choice);
3094
+ if (jsonSchema && toolConfig && toolUseSupportedModels) {
3095
+ const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
3096
+ if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but this call requires Converse tool use (either jsonSchema emulated as a forced tool call, or real \`tools\` sent alongside native structured output).`, "validation");
3097
+ }
3098
+ const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
3099
+ const request = {
3100
+ modelId: params.model,
3101
+ messages: mergeConsecutiveToolResults(conversationMessages.map((m) => toBedrockMessage(m))),
3102
+ system: systemParts.length ? systemParts.map((text) => ({ text })) : void 0,
3103
+ inferenceConfig: {
3104
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
3105
+ maxTokens: params.max_tokens
3106
+ },
3107
+ ...toolConfig ? { toolConfig } : {},
3108
+ ...outputConfig ? { outputConfig } : {}
3109
+ };
3110
+ return {
3111
+ request,
3112
+ toolName
3113
+ };
3114
+ }
3115
+ /**
944
3116
  * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
945
3117
  * interface VernLLM uses for OpenAI/Groq. The Converse API is unified
946
3118
  * across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
@@ -948,78 +3120,264 @@ function toBedrockContent(blocks) {
948
3120
  * regardless of which underlying model `modelId` points at, as long as
949
3121
  * that model supports Converse (most current-generation ones do)
950
3122
  *
951
- * `response_format: json_schema` is mapped to Converse's `toolConfig`: a
952
- * single tool is defined from the schema, description, and strictness settings,
953
- * and `toolChoice` forces the model to call it. Provider-constrained schema
954
- * matching applies only when `strict: true` is forwarded and supported.
955
- * Native tool support varies by model family; pass
956
- * `toolUseSupportedModels` to preflight-check it (see
3123
+ * `response_format: json_schema`, on a model covered by
3124
+ * `options.nativeStructuredOutputModels` (opt-in, unset by default), is
3125
+ * sent as `outputConfig.textFormat`, its own request field, independent of
3126
+ * `toolConfig`, so it can be combined with real, caller-supplied `tools`
3127
+ * in the same request. Matches the real Converse API's shape exactly: the
3128
+ * schema is nested under `structure.jsonSchema` and JSON-encoded as a
3129
+ * string, not the parsed object `toolConfig`'s tool schemas use, and there
3130
+ * is no `strict` field on this path.
3131
+ *
3132
+ * On any other model (the default), `response_format: json_schema` is
3133
+ * mapped to Converse's `toolConfig` instead: a single tool is defined from
3134
+ * the schema, description, and strictness settings, and `toolChoice`
3135
+ * forces the model to call it. This legacy path cannot be combined with
3136
+ * real `tools` (both would need the same `toolConfig`), and a call that
3137
+ * tries throws `LLMError('validation')` before reaching the API.
3138
+ * Provider-constrained schema matching applies only when `strict: true` is
3139
+ * forwarded and supported. Native tool support varies by model family;
3140
+ * pass `toolUseSupportedModels` to preflight-check it (see
957
3141
  * `BedrockAdapterOptions`), otherwise a `jsonSchema` call to an
958
3142
  * unsupported model surfaces Bedrock's raw error unchanged.
959
3143
  *
960
3144
  * `response_format: json_object` (no schema to build a tool from) and
961
3145
  * `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
962
- * instruction and are dropped respectively.
3146
+ * instruction and are dropped respectively. Unlike `jsonSchema`,
3147
+ * `json_object` combines with real `tools` freely on every model: it's a
3148
+ * prompt nudge, not a request field, so there's nothing for it to collide
3149
+ * with.
3150
+ *
3151
+ * `tools` alone maps to Converse's native `toolConfig`/`toolUse`/
3152
+ * `toolResult`; `tool_choice` maps to `toolConfig.toolChoice`.
3153
+ *
3154
+ * `createStream` calls `converseStream` (optional on `BedrockConverseClient`
3155
+ *, required only if the caller sets `stream: true`) and translates its
3156
+ * `contentBlockStart`/`contentBlockDelta`/`metadata` events into
3157
+ * `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
3158
+ * same as `fromAnthropic`'s block-index tracking (Converse's streaming
3159
+ * shape is structurally close to Anthropic's own, both being tool-use-aware
3160
+ * content-block streams), including the same `json-tool` unwrapping: a
3161
+ * `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
3162
+ * `text-delta`, not `tool_call_delta`, so the accumulated result lands in
3163
+ * `finalizeResponse`'s `content` path exactly like the non-streaming
3164
+ * `create` branch above unwraps it.
963
3165
  */
964
3166
  function fromBedrock(bedrockClient, options) {
965
3167
  const toolUseSupportedModels = options?.toolUseSupportedModels;
966
- return { chat: { completions: { async create(params, requestOptions) {
967
- const systemMessage = params.messages.find((m) => m.role === "system");
968
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
969
- const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
970
- const toolName = jsonSchema?.name.trim();
971
- if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
972
- let jsonInstruction;
973
- let toolConfig;
974
- if (jsonSchema) {
975
- const { schema, description, strict } = jsonSchema;
976
- toolConfig = {
977
- tools: [{ toolSpec: {
978
- name: toolName,
979
- description,
980
- inputSchema: { json: schema },
981
- strict
3168
+ const nativeStructuredOutputModels = options?.nativeStructuredOutputModels;
3169
+ return { chat: { completions: {
3170
+ async create(params, requestOptions) {
3171
+ const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels, nativeStructuredOutputModels);
3172
+ const response = await bedrockClient.converse(request, requestOptions);
3173
+ let text;
3174
+ let wireToolCalls;
3175
+ if (toolName) {
3176
+ const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
3177
+ text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
3178
+ } else {
3179
+ const blocks = response.output?.message?.content ?? [];
3180
+ text = blocks.map((c) => c.text ?? "").join("");
3181
+ const toolUses = blocks.filter((block) => Boolean(block.toolUse));
3182
+ if (toolUses.length) wireToolCalls = toolUses.map((block, i) => {
3183
+ const toolUse = block.toolUse;
3184
+ if (!toolUse.name) throw new LLMError(`Bedrock returned a toolUse block without a name at index ${i}.`, "validation");
3185
+ return {
3186
+ id: toolUse.toolUseId ?? `${toolUse.name}_${i}`,
3187
+ type: "function",
3188
+ function: {
3189
+ name: toolUse.name,
3190
+ arguments: JSON.stringify(toolUse.input ?? {})
3191
+ }
3192
+ };
3193
+ });
3194
+ }
3195
+ return {
3196
+ choices: [{ message: {
3197
+ content: text,
3198
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
982
3199
  } }],
983
- toolChoice: { tool: { name: toolName } }
3200
+ usage: {
3201
+ prompt_tokens: response.usage?.inputTokens,
3202
+ completion_tokens: response.usage?.outputTokens,
3203
+ total_tokens: response.usage?.totalTokens
3204
+ }
3205
+ };
3206
+ },
3207
+ async *createStream(params, requestOptions) {
3208
+ if (!bedrockClient.converseStream) throw new LLMError("stream: true requires a Bedrock client with converseStream", "validation");
3209
+ const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels, nativeStructuredOutputModels);
3210
+ const { stream } = await bedrockClient.converseStream(request, requestOptions);
3211
+ const blockKinds = new Map();
3212
+ for await (const event of stream) if ("contentBlockStart" in event) {
3213
+ const { contentBlockIndex, start } = event.contentBlockStart;
3214
+ if (start?.toolUse) {
3215
+ const kind = start.toolUse.name === toolName ? "json-tool" : "tool_use";
3216
+ blockKinds.set(contentBlockIndex, kind);
3217
+ if (kind === "tool_use" && !toolName) yield {
3218
+ type: "tool_call_delta",
3219
+ index: contentBlockIndex,
3220
+ id: start.toolUse.toolUseId,
3221
+ name: start.toolUse.name
3222
+ };
3223
+ } else blockKinds.set(contentBlockIndex, "text");
3224
+ } else if ("contentBlockDelta" in event) {
3225
+ const { contentBlockIndex, delta } = event.contentBlockDelta;
3226
+ if (delta && "text" in delta && delta.text !== void 0 && !toolName) yield {
3227
+ type: "text-delta",
3228
+ delta: delta.text
3229
+ };
3230
+ else if (delta && "toolUse" in delta && delta.toolUse?.input !== void 0) {
3231
+ const kind = blockKinds.get(contentBlockIndex);
3232
+ if (kind === "json-tool") yield {
3233
+ type: "text-delta",
3234
+ delta: delta.toolUse.input
3235
+ };
3236
+ else if (!toolName) yield {
3237
+ type: "tool_call_delta",
3238
+ index: contentBlockIndex,
3239
+ argumentsDelta: delta.toolUse.input
3240
+ };
3241
+ }
3242
+ } else if ("metadata" in event && event.metadata.usage) yield {
3243
+ type: "usage",
3244
+ usage: {
3245
+ prompt_tokens: event.metadata.usage.inputTokens,
3246
+ completion_tokens: event.metadata.usage.outputTokens,
3247
+ total_tokens: event.metadata.usage.totalTokens
3248
+ }
984
3249
  };
985
- } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
986
- if (jsonSchema && toolUseSupportedModels) {
987
- const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
988
- if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
3250
+ else if ("throttlingException" in event) throw new LLMError(event.throttlingException.message ?? "Bedrock throttled the request mid-stream", "api", 429);
3251
+ else if ("validationException" in event) throw new LLMError(event.validationException.message ?? "Bedrock rejected the request mid-stream", "validation");
3252
+ else if ("internalServerException" in event || "serviceUnavailableException" in event || "modelStreamErrorException" in event) {
3253
+ const detail = "internalServerException" in event && event.internalServerException.message || "serviceUnavailableException" in event && event.serviceUnavailableException.message || "modelStreamErrorException" in event && event.modelStreamErrorException.message || "Bedrock reported a mid-stream error";
3254
+ const status = "modelStreamErrorException" in event && event.modelStreamErrorException.originalStatusCode || "serviceUnavailableException" in event && 503 || 500;
3255
+ throw new LLMError(detail, "api", status);
3256
+ }
989
3257
  }
990
- const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
991
- const response = await bedrockClient.converse({
992
- modelId: params.model,
993
- messages: conversationMessages.map((m) => ({
994
- role: m.role,
995
- content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content }]
996
- })),
997
- system: systemParts.length ? systemParts.map((text$1) => ({ text: text$1 })) : void 0,
998
- inferenceConfig: {
999
- temperature: params.temperature,
1000
- maxTokens: params.max_tokens
1001
- },
1002
- ...toolConfig ? { toolConfig } : {}
1003
- }, requestOptions);
1004
- let text;
1005
- if (toolName) {
1006
- const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
1007
- text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
1008
- } else text = response.output?.message?.content?.map((c) => c.text ?? "").join("") ?? "";
1009
- return {
1010
- choices: [{ message: { content: text } }],
1011
- usage: {
1012
- prompt_tokens: response.usage?.inputTokens,
1013
- completion_tokens: response.usage?.outputTokens,
1014
- total_tokens: response.usage?.totalTokens
3258
+ } } };
3259
+ }
3260
+ /** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Converse's `toolChoice`. */
3261
+ function toBedrockToolChoice(toolChoice) {
3262
+ if (!toolChoice || toolChoice === "auto") return { auto: {} };
3263
+ if (toolChoice === "required") return { any: {} };
3264
+ if (toolChoice === "none") throw new LLMError("'none' is not supported by fromBedrock: Bedrock Converse has no `tool_choice` equivalent to forbidding tool use while tools are still offered. Omit `tools` entirely for this call instead.", "validation");
3265
+ return { tool: { name: toolChoice.function.name } };
3266
+ }
3267
+ /**
3268
+ * Translates one VernLLM wire message into Converse's
3269
+ * `{ role: 'user' | 'assistant', content }` shape.
3270
+ */
3271
+ function toBedrockMessage(m) {
3272
+ if (m.role === "tool") return {
3273
+ role: "user",
3274
+ content: [{ toolResult: {
3275
+ toolUseId: m.tool_call_id,
3276
+ content: [{ text: m.content }],
3277
+ status: m.is_error ? "error" : "success"
3278
+ } }]
3279
+ };
3280
+ if (m.role === "assistant" && m.tool_calls?.length) {
3281
+ const blocks = [];
3282
+ if (m.content) blocks.push({ text: m.content });
3283
+ for (const tc of m.tool_calls) {
3284
+ let input;
3285
+ if (!tc.function.arguments.trim()) input = {};
3286
+ else try {
3287
+ input = JSON.parse(tc.function.arguments);
3288
+ } catch (cause) {
3289
+ throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
1015
3290
  }
3291
+ blocks.push({ toolUse: {
3292
+ toolUseId: tc.id,
3293
+ name: tc.function.name,
3294
+ input
3295
+ } });
3296
+ }
3297
+ return {
3298
+ role: "assistant",
3299
+ content: blocks
1016
3300
  };
1017
- } } } };
3301
+ }
3302
+ return {
3303
+ role: m.role,
3304
+ content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content ?? "" }]
3305
+ };
3306
+ }
3307
+ /**
3308
+ * Converse expects the results of everything the model asked for in one
3309
+ * turn to arrive together as multiple `toolResult` content blocks on a
3310
+ * single `'user'` message, not as separate consecutive `'user'` messages.
3311
+ * The per-wire-message mapping above produces one `'user'` message per
3312
+ * VernLLM wire tool message, so when an assistant turn requested more than
3313
+ * one tool, this merges the resulting run of toolResult-only `'user'`
3314
+ * messages back into one.
3315
+ */
3316
+ function mergeConsecutiveToolResults(messages) {
3317
+ const isToolResultOnly = (m) => m.role === "user" && m.content.length > 0 && m.content.every((b) => "toolResult" in b);
3318
+ const merged = [];
3319
+ for (const m of messages) {
3320
+ const prev = merged.at(-1);
3321
+ if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
3322
+ else merged.push(m);
3323
+ }
3324
+ return merged;
1018
3325
  }
1019
3326
 
1020
3327
  //#endregion
1021
3328
  //#region src/adapters/fetch.ts
1022
3329
  /**
3330
+ * Wraps a WHATWG `ReadableStream` (what `response.body` is) so it can be
3331
+ * consumed with `for await`. Implemented via `getReader()` rather than
3332
+ * relying on `ReadableStream` having a native `Symbol.asyncIterator`,
3333
+ * that support varies across runtimes/versions, and this works everywhere
3334
+ * a `ReadableStream` does.
3335
+ */
3336
+ async function* webStreamToAsyncIterable(stream) {
3337
+ const reader = stream.getReader();
3338
+ try {
3339
+ for (;;) {
3340
+ const { done, value } = await reader.read();
3341
+ if (done) return;
3342
+ if (value) yield value;
3343
+ }
3344
+ } finally {
3345
+ try {
3346
+ await reader.cancel();
3347
+ } catch {}
3348
+ reader.releaseLock();
3349
+ }
3350
+ }
3351
+ /** Default `requestStream`: native `fetch`, with the same error/`.status` contract non-streaming errors get. */
3352
+ async function defaultRequestStream(url, init) {
3353
+ const res = await fetch(url, init);
3354
+ if (!res.ok) {
3355
+ const body = await res.text().catch(() => "");
3356
+ const err = new Error(`Fetch adapter stream request failed (${res.status}): ${body.slice(0, 500)}`);
3357
+ err.status = res.status;
3358
+ err.headers = res.headers;
3359
+ throw err;
3360
+ }
3361
+ if (!res.body) throw new Error("Fetch adapter stream request received a response with no body.");
3362
+ return webStreamToAsyncIterable(res.body);
3363
+ }
3364
+ /** Builds the shared `{ method, headers, body? }` request-init for both `create` and `createStream`. */
3365
+ async function buildRequestInit(config, params, requestBody) {
3366
+ const url = typeof config.url === "function" ? config.url(params) : config.url;
3367
+ const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
3368
+ const method = config.method ?? "POST";
3369
+ const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
3370
+ return {
3371
+ url,
3372
+ method,
3373
+ headers: supportsBody ? {
3374
+ "Content-Type": "application/json",
3375
+ ...headers
3376
+ } : { ...headers },
3377
+ ...supportsBody ? { body: JSON.stringify(requestBody) } : {}
3378
+ };
3379
+ }
3380
+ /**
1023
3381
  * A fetch-based escape hatch for providers with no SDK, or where pulling one
1024
3382
  * in isnt worth it. You supply the URL, headers, and two small mapping
1025
3383
  * functions; this handles the HTTP call and slots the result into the same
@@ -1029,41 +3387,99 @@ function fromBedrock(bedrockClient, options) {
1029
3387
  * Non-2xx responses throw an error with `.status` set to the HTTP status
1030
3388
  * code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
1031
3389
  * 401/403) applies here too
3390
+ *
3391
+ * Tool calling works the same way as every other adapter: `mapRequest`
3392
+ * receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
3393
+ * translate them into whatever shape the provider's wire format expects
3394
+ * (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
3395
+ * field). On the way back, `mapResponse` may return a `toolCalls` array
3396
+ * (id/name/JSON-encoded-arguments-string per call) alongside or instead of
3397
+ * `content`; VernLLM parses and (if `argumentsSchema` was set) validates
3398
+ * those arguments the same way it does for every other adapter. For
3399
+ * `stream: true`, tool-call deltas go through the existing
3400
+ * `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
3401
+ * no separate config is needed for streaming vs non-streaming tool calls.
3402
+ *
3403
+
3404
+ * `createStream` requires `mapStreamEvent` (there's no non-streaming
3405
+ * response to fall back on, unlike the other three optional streaming
3406
+ * seams). It opens the request via `requestStream` (defaults to native
3407
+ * `fetch`), splits the raw bytes into individual events via
3408
+ * `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
3409
+ * and translates each event into `WireStreamChunk`(s) via
3410
+ * `mapStreamEvent`. Both seams are overridable per-config for providers
3411
+ * that don't fit the SSE-over-fetch default. If a custom `request`
3412
+ * transport is configured, `requestStream` must be configured too,
3413
+ * `requestStream` never silently falls back to `request` (see
3414
+ * `createStream`'s own comment for why), so a `stream: true` call with
3415
+ * `request` set but no `requestStream` throws a clear
3416
+ * `LLMError('validation')` instead of quietly using unrelated native
3417
+ * `fetch`.
1032
3418
  */
1033
3419
  function fromFetch(config) {
1034
- return { chat: { completions: { async create(params, options) {
1035
- const url = typeof config.url === "function" ? config.url(params) : config.url;
1036
- const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
1037
- const method = config.method ?? "POST";
1038
- const request = config.request ?? fetch;
1039
- const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
1040
- const res = await request(url, {
1041
- method,
1042
- headers: supportsBody ? {
1043
- "Content-Type": "application/json",
1044
- ...headers
1045
- } : { ...headers },
1046
- ...supportsBody ? { body: JSON.stringify(config.mapRequest(params)) } : {},
1047
- signal: options.signal
1048
- });
1049
- if (!res.ok) {
1050
- const body = await res.text().catch(() => "");
1051
- const err = new Error(`Fetch adapter request failed (${res.status}): ${body.slice(0, 500)}`);
1052
- err.status = res.status;
1053
- err.headers = res.headers;
1054
- throw err;
3420
+ return { chat: { completions: {
3421
+ async create(params, options) {
3422
+ const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
3423
+ const request = config.request ?? fetch;
3424
+ const res = await request(url, {
3425
+ method,
3426
+ headers,
3427
+ body,
3428
+ signal: options.signal
3429
+ });
3430
+ if (!res.ok) {
3431
+ const responseBody = await res.text().catch(() => "");
3432
+ const err = new Error(`Fetch adapter request failed (${res.status}): ${responseBody.slice(0, 500)}`);
3433
+ err.status = res.status;
3434
+ err.headers = res.headers;
3435
+ throw err;
3436
+ }
3437
+ const json = await res.json();
3438
+ const { content, usage, toolCalls } = config.mapResponse(json);
3439
+ const wireToolCalls = toolCalls?.length ? toolCalls.map((tc) => ({
3440
+ id: tc.id,
3441
+ type: "function",
3442
+ function: {
3443
+ name: tc.name,
3444
+ arguments: tc.arguments
3445
+ }
3446
+ })) : void 0;
3447
+ return {
3448
+ choices: [{ message: {
3449
+ content,
3450
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
3451
+ } }],
3452
+ usage: usage ? {
3453
+ prompt_tokens: usage.promptTokens,
3454
+ completion_tokens: usage.completionTokens,
3455
+ total_tokens: usage.totalTokens
3456
+ } : void 0
3457
+ };
3458
+ },
3459
+ async *createStream(params, options) {
3460
+ if (!config.mapStreamEvent) throw new LLMError("stream: true requires mapStreamEvent to be configured on fromFetch", "validation");
3461
+ if (config.request && !config.requestStream) throw new LLMError("`stream: true` requires `requestStream` to be configured on fromFetch when a custom `request` transport is set. `requestStream` does not fall back to `request` (it needs an async-iterable byte stream, which `RequestLike`'s buffered `ResponseLike` has no way to provide), without it, `stream: true` would silently use plain native `fetch` instead of your configured transport. Add a `requestStream` that opens the same connection your `request` does, or omit `request` if native `fetch` is fine for both.", "validation");
3462
+ const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
3463
+ const requestStream = config.requestStream ?? defaultRequestStream;
3464
+ const parseFrames = config.parseStreamFrames ?? parseSseStream;
3465
+ const byteStream = await requestStream(url, {
3466
+ method,
3467
+ headers,
3468
+ body,
3469
+ signal: options.signal
3470
+ });
3471
+ for await (const event of parseFrames(byteStream)) {
3472
+ if (event === SSE_PING) {
3473
+ yield { type: "ping" };
3474
+ continue;
3475
+ }
3476
+ const wireChunks = config.mapStreamEvent(event);
3477
+ if (!wireChunks) continue;
3478
+ if (Array.isArray(wireChunks)) yield* wireChunks;
3479
+ else yield wireChunks;
3480
+ }
1055
3481
  }
1056
- const json = await res.json();
1057
- const { content, usage } = config.mapResponse(json);
1058
- return {
1059
- choices: [{ message: { content } }],
1060
- usage: usage ? {
1061
- prompt_tokens: usage.promptTokens,
1062
- completion_tokens: usage.completionTokens,
1063
- total_tokens: usage.totalTokens
1064
- } : void 0
1065
- };
1066
- } } } };
3482
+ } } };
1067
3483
  }
1068
3484
 
1069
3485
  //#endregion
@@ -1086,41 +3502,84 @@ function toOpenAIContent(blocks) {
1086
3502
  });
1087
3503
  }
1088
3504
  /**
1089
- * Adapter for any SDK/client whose `chat.completions.create` already
1090
- * matches the OpenAI wire format: this covers most hosted inference
1091
- * providers, since "OpenAI-compatible" is a de facto standard for chat
1092
- * completion APIs. Almost everything passes straight through untouched,
1093
- * this exists purely so call sites read clearly (`fromMistral(client)` vs
1094
- * handing a Mistral client to something typed for OpenAI) and so a real
1095
- * transformation could be added later, per-provider, without a breaking
1096
- * change.
1097
- *
1098
- * The one thing that isn't a pure passthrough: a `ContentBlock[]`
1099
- * `userContent` is translated into OpenAI's native `image_url` content-part
1100
- * shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
1101
- * rather than a copy of any one provider's wire format.
1102
- *
1103
- * Not every SDKs own TypeScript types line up exactly with `LLMClient`
1104
- * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
1105
- * the actual compatibility contract is the JSON each provider sends and
1106
- * receives over the wire, not the SDKs TS types.
3505
+ * Translates VernLLM's provider-agnostic `messages` (the one part of a
3506
+ * request that isn't a pure passthrough for OpenAI-compatible clients) into
3507
+ * OpenAI's native wire shape. Shared between `create` and `createStream` so
3508
+ * both go through identical message translation.
1107
3509
  */
1108
- function fromOpenAICompatible(client) {
1109
- const raw = client;
1110
- return { chat: { completions: { async create(params, options) {
1111
- const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
3510
+ function toOpenAIMessages(params) {
3511
+ return params.messages.map((m) => {
3512
+ if (m.role === "user" && Array.isArray(m.content)) return {
1112
3513
  ...m,
1113
3514
  content: toOpenAIContent(m.content)
1114
- } : m);
1115
- return raw.chat.completions.create({
1116
- ...params,
1117
- messages
1118
- }, options);
1119
- } } } };
3515
+ };
3516
+ if (m.role === "tool") {
3517
+ const { is_error: _isError,...openAIToolMessage } = m;
3518
+ return openAIToolMessage;
3519
+ }
3520
+ return m;
3521
+ });
3522
+ }
3523
+ /**
3524
+ * Translates one OpenAI-shaped SSE chunk into zero or more `WireStreamChunk`s.
3525
+ * A single chunk can carry a text delta, one or more tool-call argument
3526
+ * deltas (each keyed by `index`, OpenAI's own convention for streaming
3527
+ * parallel tool calls, mirrored directly by VernLLM's `tool_call_delta`
3528
+ * shape so accumulation composes without translation), and/or a final
3529
+ * usage block (present only when `stream_options.include_usage` is set,
3530
+ * which this adapter always sets).
3531
+ */
3532
+ function* toWireStreamChunks(chunk) {
3533
+ const delta = chunk.choices?.[0]?.delta;
3534
+ if (delta?.content) yield {
3535
+ type: "text-delta",
3536
+ delta: delta.content
3537
+ };
3538
+ if (delta?.tool_calls?.length) for (const toolCall of delta.tool_calls) yield {
3539
+ type: "tool_call_delta",
3540
+ index: toolCall.index,
3541
+ id: toolCall.id,
3542
+ name: toolCall.function?.name,
3543
+ argumentsDelta: toolCall.function?.arguments
3544
+ };
3545
+ if (chunk.usage) yield {
3546
+ type: "usage",
3547
+ usage: chunk.usage
3548
+ };
3549
+ }
3550
+ function fromOpenAICompatible(client, options = {}) {
3551
+ const raw = client;
3552
+ const { supportsStreamUsage = true } = options;
3553
+ const rawCreate = raw.chat.completions.create.bind(raw.chat.completions);
3554
+ return { chat: { completions: {
3555
+ async create(params, options$1) {
3556
+ const messages = toOpenAIMessages(params);
3557
+ return raw.chat.completions.create({
3558
+ ...params,
3559
+ messages
3560
+ }, options$1);
3561
+ },
3562
+ async *createStream(params, options$1) {
3563
+ const messages = toOpenAIMessages(params);
3564
+ const stream = await rawCreate({
3565
+ ...params,
3566
+ messages,
3567
+ stream: true,
3568
+ ...supportsStreamUsage ? { stream_options: { include_usage: true } } : {}
3569
+ }, options$1);
3570
+ for await (const chunk of stream) yield* toWireStreamChunks(chunk);
3571
+ }
3572
+ } } };
1120
3573
  }
1121
3574
  /** Groqs SDK matches the OpenAI wire format */
1122
3575
  const fromGroq = fromOpenAICompatible;
1123
- /** Mistrals `chat.completions`-shaped client (or their OpenAI-compat endpoint) */
3576
+ /**
3577
+ * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
3578
+ * endpoint). Mistral supports `stream_options.include_usage` (added after
3579
+ * an earlier period where it returned a 422 for unrecognized fields, per
3580
+ * Mistral's changelog and streaming docs), so this is a plain alias like
3581
+ * the others, `supportsStreamUsage` defaults to `true`.
3582
+ */
1124
3583
  const fromMistral = fromOpenAICompatible;
1125
3584
  /** DeepSeeks API is OpenAI-compatible */
1126
3585
  const fromDeepSeek = fromOpenAICompatible;
@@ -1211,11 +3670,16 @@ const from01AI = fromOpenAICompatible;
1211
3670
  //#endregion
1212
3671
  exports.CircuitBreaker = CircuitBreaker
1213
3672
  exports.ConsoleLogger = ConsoleLogger
3673
+ exports.FallbackExhaustedError = FallbackExhaustedError
1214
3674
  exports.InMemoryCacheAdapter = InMemoryCacheAdapter
1215
3675
  exports.LLMError = LLMError
1216
3676
  exports.NormalizedCacheAdapter = NormalizedCacheAdapter
3677
+ exports.RateLimiter = RateLimiter
3678
+ exports.SSE_PING = SSE_PING
1217
3679
  exports.TieredCacheAdapter = TieredCacheAdapter
1218
3680
  exports.VernLLM = VernLLM
3681
+ exports.defaultEstimateTokens = defaultEstimateTokens
3682
+ exports.defaultFallbackOn = defaultFallbackOn
1219
3683
  exports.from01AI = from01AI
1220
3684
  exports.fromAnthropic = fromAnthropic
1221
3685
  exports.fromAnyscale = fromAnyscale
@@ -1261,4 +3725,6 @@ exports.fromVercelAIGateway = fromVercelAIGateway
1261
3725
  exports.fromXAI = fromXAI
1262
3726
  exports.fromZhipu = fromZhipu
1263
3727
  exports.isLLMError = isLLMError
3728
+ exports.isToolCallResult = isToolCallResult
3729
+ exports.parseSseStream = parseSseStream
1264
3730
  //# sourceMappingURL=index.cjs.map