vern-llm 1.7.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -75,7 +75,7 @@ var CircuitBreaker = class {
75
75
  if (this.state === "closed") return;
76
76
  if (this.state === "open") {
77
77
  const elapsed = Date.now() - this.openedAt;
78
- if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
78
+ if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open, provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
79
79
  this.state = "half-open";
80
80
  this.trialInFlight = true;
81
81
  return;
@@ -108,6 +108,48 @@ var CircuitBreaker = class {
108
108
 
109
109
  //#endregion
110
110
  //#region src/internal/vernLLM.utils.ts
111
+ /** Translates app-facing `ToolDefinition[]` into the OpenAI-shaped wire tools array. */
112
+ function toWireTools(tools) {
113
+ return tools.map((tool) => ({
114
+ type: "function",
115
+ function: {
116
+ name: tool.name,
117
+ description: tool.description,
118
+ parameters: tool.parameters
119
+ }
120
+ }));
121
+ }
122
+ /** Translates app-facing `ToolCall[]` (e.g. from a replayed assistant turn) into wire tool_calls. */
123
+ function toWireToolCalls(toolCalls) {
124
+ return toolCalls.map((tc) => ({
125
+ id: tc.id,
126
+ type: "function",
127
+ function: {
128
+ name: tc.name,
129
+ arguments: JSON.stringify(tc.arguments ?? {})
130
+ }
131
+ }));
132
+ }
133
+ /**
134
+ * Parses the provider's wire-shaped `tool_calls` back into VernLLM's
135
+ * `ToolCall[]`. Malformed argument JSON is a `'parse'` error, same
136
+ * convention as malformed JSON response bodies elsewhere in VernLLM.
137
+ */
138
+ function parseWireToolCalls(wireToolCalls) {
139
+ return wireToolCalls.map((wc) => {
140
+ let parsedArgs;
141
+ try {
142
+ parsedArgs = wc.function.arguments.trim() ? JSON.parse(wc.function.arguments) : {};
143
+ } catch {
144
+ throw new LLMError(`Invalid JSON arguments for tool call "${wc.function.name}"`, "parse");
145
+ }
146
+ return {
147
+ id: wc.id,
148
+ name: wc.function.name,
149
+ arguments: parsedArgs
150
+ };
151
+ });
152
+ }
111
153
  function defaultParseJson(content) {
112
154
  try {
113
155
  return JSON.parse(content);
@@ -118,7 +160,9 @@ function defaultParseJson(content) {
118
160
  /**
119
161
  * Looks inside an unknown error value and pulls out an http status code
120
162
  * if one is present. Checks the status field first then the status code
121
- * field since different client libraries use different names for this.
163
+ * field since different client libraries use different names for this,
164
+ * falling back to AWS SDK v3's `$metadata.httpStatusCode` (e.g. Bedrock's
165
+ * `ThrottlingException`), which doesn't set either of the other two.
122
166
  * Returns undefined when the error is not an object or carries no status
123
167
  */
124
168
  function extractStatus(err) {
@@ -126,6 +170,7 @@ function extractStatus(err) {
126
170
  const error = err;
127
171
  if (typeof error.status === "number") return error.status;
128
172
  if (typeof error.statusCode === "number") return error.statusCode;
173
+ if (typeof error.$metadata?.httpStatusCode === "number") return error.$metadata.httpStatusCode;
129
174
  return void 0;
130
175
  }
131
176
  function formatSafely(value) {
@@ -155,6 +200,21 @@ function describeError(err) {
155
200
  return formatSafely(err);
156
201
  }
157
202
  /**
203
+ * `setTimeout` silently clamps any delay above this (~24.8 days) or
204
+ * `Infinity` down to ~1ms instead of erroring, so a caller passing
205
+ * `Infinity` as "no timeout" gets the opposite of what they asked for.
206
+ * Both timeout helpers below guard against this explicitly.
207
+ */
208
+ const MAX_SETTIMEOUT_MS = 2147483647;
209
+ /** True when a timeout value should be treated as "disabled" rather than passed to `setTimeout`. */
210
+ function isTimeoutDisabled(ms) {
211
+ return !ms || ms <= 0 || ms === Infinity;
212
+ }
213
+ /** Caps a timeout at the largest delay `setTimeout` actually honors. */
214
+ function clampTimeoutMs(ms) {
215
+ return Math.min(ms, MAX_SETTIMEOUT_MS);
216
+ }
217
+ /**
158
218
  * Runs an async function and cancels it if it takes longer than the given
159
219
  * timeout. Creates an internal abort controller that fires after the
160
220
  * timeout elapses, and combines it with any external signal the caller
@@ -164,12 +224,15 @@ function describeError(err) {
164
224
  * continue to propagate as aborted errors. The internal timer is always
165
225
  * cleared afterward, whether the function succeeds, fails, or is aborted,
166
226
  * so nothing is left running in the background.
227
+ *
228
+ * `timeoutMs` of `Infinity` (or any value beyond what `setTimeout` can
229
+ * represent) disables the timeout rather than firing almost immediately.
167
230
  */
168
231
  async function withTimeout(fn, timeoutMs, externalSignal) {
169
232
  const controller = new AbortController();
170
- const timer = setTimeout(() => {
233
+ const timer = isTimeoutDisabled(timeoutMs) ? void 0 : setTimeout(() => {
171
234
  controller.abort();
172
- }, timeoutMs);
235
+ }, clampTimeoutMs(timeoutMs));
173
236
  const signal = externalSignal ? AbortSignal.any([externalSignal, controller.signal]) : controller.signal;
174
237
  try {
175
238
  return await fn(signal);
@@ -181,6 +244,54 @@ async function withTimeout(fn, timeoutMs, externalSignal) {
181
244
  }
182
245
  }
183
246
  /**
247
+ * Races one `iterator.next()` call against a per-call idle timer, to
248
+ * bound the gap *between* chunks (unlike `withTimeout`, which only bounds
249
+ * opening the stream and its first chunk). Without this, a connection
250
+ * that streams one chunk then hangs would never fail.
251
+ *
252
+ * `timeoutMs` of 0/undefined/`Infinity` disables the check. Otherwise
253
+ * rejects with `LLMError('timeout')` if `next()` doesn't settle in time.
254
+ * The clock resets on every call, so the window is measured from the most
255
+ * recent chunk, not from stream start.
256
+ *
257
+ * `onIdle`, if given, is called the moment the timer fires (before the
258
+ * rejection), so callers can abort the underlying transport instead of
259
+ * just walking away from an unread promise. `logger`, if given, records a
260
+ * debug line if `next()` still settles *after* the idle timeout already
261
+ * rejected. `resolve`/`reject` on an already-settled promise is otherwise
262
+ * a silent no-op, so without this the late chunk (possibly the final
263
+ * usage chunk) would vanish with no trace.
264
+ */
265
+ function withChunkIdleTimeout(next, timeoutMs, onIdle, logger) {
266
+ if (isTimeoutDisabled(timeoutMs)) return next();
267
+ const activeTimeoutMs = timeoutMs;
268
+ let settled = false;
269
+ return new Promise((resolve, reject) => {
270
+ const timer = setTimeout(() => {
271
+ settled = true;
272
+ onIdle?.();
273
+ reject(new LLMError(`No stream chunk received for ${activeTimeoutMs}ms (idle timeout)`, "timeout"));
274
+ }, clampTimeoutMs(activeTimeoutMs));
275
+ next().then((result) => {
276
+ clearTimeout(timer);
277
+ if (settled) {
278
+ logger?.debug("[VernLLM] chunk resolved after idle timeout already fired; discarding");
279
+ return;
280
+ }
281
+ settled = true;
282
+ resolve(result);
283
+ }, (error) => {
284
+ clearTimeout(timer);
285
+ if (settled) {
286
+ logger?.debug("[VernLLM] chunk rejection arrived after idle timeout already fired; discarding");
287
+ return;
288
+ }
289
+ settled = true;
290
+ reject(error);
291
+ });
292
+ });
293
+ }
294
+ /**
184
295
  * Default cap (ms) for both exponential backoff and honored Retry-After
185
296
  * values, so a misbehaving/adversarial Retry-After can't stall a caller
186
297
  * indefinitely
@@ -295,6 +406,143 @@ async function withReservedUsage(params, coalesced, getResult, signal, onRefundE
295
406
  }
296
407
  return result;
297
408
  }
409
+ /**
410
+ * Streaming counterpart to `withReservedUsage`. `withReservedUsage` assumes
411
+ * `getResult()` settling *is* the operation's final outcome, awaiting it
412
+ * synchronously before reserve/refund resolve. Streaming can't satisfy that:
413
+ * `call()` must return `{ chunks, finalResult }` as soon as the stream
414
+ * opens, well before the real outcome (validation, schema/tool-call checks)
415
+ * is known.
416
+ *
417
+ * Reserves usage before `openStream` runs, same failure mode as the
418
+ * non-streaming path if `reserveUsage` itself throws (mapped to
419
+ * `quota_exceeded`, nothing opened). If `openStream` itself throws (stream
420
+ * never opened), refunds synchronously and rethrows, exactly like
421
+ * `withReservedUsage` does today. If it succeeds, returns `{ chunks,
422
+ * finalResult }` immediately, refund/report is deferred onto
423
+ * `finalResult`'s continuation, since that's the only point the real
424
+ * outcome is known. This means `onUsageFailure` (and any refund) can fire
425
+ * well after this function itself has returned.
426
+ */
427
+ async function withReservedUsageForStream(params, openStream, signal, onRefundError) {
428
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
429
+ let reserved = false;
430
+ try {
431
+ if (params.reserveUsage) {
432
+ await params.reserveUsage({
433
+ coalesced: false,
434
+ signal
435
+ });
436
+ reserved = true;
437
+ }
438
+ } catch (error) {
439
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
440
+ throw new LLMError(error instanceof Error ? error.message : "Usage reservation failed", "quota_exceeded", void 0, void 0, error);
441
+ }
442
+ const refund = async (logMessage) => {
443
+ try {
444
+ await params.refundUsage?.({
445
+ coalesced: false,
446
+ signal
447
+ });
448
+ } catch (refundError) {
449
+ onRefundError(logMessage, refundError);
450
+ }
451
+ };
452
+ let opened;
453
+ try {
454
+ opened = await openStream();
455
+ } catch (error) {
456
+ if (reserved) await refund("[VernLLM] refundUsage failed after stream-open failure");
457
+ throw error;
458
+ }
459
+ const finalResult = opened.finalResult.then((value) => value, async (error) => {
460
+ if (reserved) await refund("[VernLLM] refundUsage failed after stream error");
461
+ throw error;
462
+ });
463
+ finalResult.catch(() => {});
464
+ return {
465
+ chunks: opened.chunks,
466
+ finalResult
467
+ };
468
+ }
469
+ /**
470
+ * Converts an already-known cache value back into a plausible "text" form
471
+ * for a one-shot replay chunk: passed through unchanged if it's already a
472
+ * string (the `jsonMode: false` case), otherwise `JSON.stringify`'d (the
473
+ * `jsonMode: true` case, where the cached value is the *parsed* result, not
474
+ * the original raw text). This is a reasonable reconstruction, not a
475
+ * byte-identical replay of whatever text the model originally streamed,
476
+ * good enough for `for await (const c of chunks)` call sites that don't
477
+ * branch on hit vs. miss, which is the only thing a cache-hit replay needs
478
+ * to support.
479
+ */
480
+ function toReplayText(value) {
481
+ return typeof value === "string" ? value : JSON.stringify(value);
482
+ }
483
+ /**
484
+ * Builds a trivially-exhausted one-shot `chunks` iterable from an
485
+ * already-known value, used for a `cachedCall` cache hit, where there's no
486
+ * live generation to relay (see `VernLLM.cachedCall`'s docs). No `usage`
487
+ * chunk is emitted: a cache hit spent no real tokens, so there's nothing to
488
+ * report, matching how non-streaming `cachedCall` never calls `onUsage` on
489
+ * a hit either.
490
+ *
491
+ * `hasTools` must reflect whether the *original* call that produced this
492
+ * cached value had `tools` set, that's what determines whether `value` is
493
+ * `T` directly or a `CallWithToolsResult<T>` wrapper, and it isn't
494
+ * something that can be reliably guessed from the value's shape alone
495
+ * (a `schema`-validated `T` could coincidentally look like a
496
+ * `CallWithToolsResult`).
497
+ */
498
+ function buildReplayChunks(value, hasTools) {
499
+ const items = [];
500
+ if (hasTools) {
501
+ const result = value;
502
+ if (result.type === "tool_calls") {
503
+ result.toolCalls.forEach((toolCall, index) => {
504
+ items.push({
505
+ type: "tool_call_delta",
506
+ index,
507
+ id: toolCall.id,
508
+ name: toolCall.name,
509
+ argsDelta: JSON.stringify(toolCall.arguments ?? {}),
510
+ complete: true
511
+ });
512
+ });
513
+ if (result.content) items.push({
514
+ type: "text-delta",
515
+ delta: result.content
516
+ });
517
+ } else items.push({
518
+ type: "text-delta",
519
+ delta: toReplayText(result.content)
520
+ });
521
+ } else items.push({
522
+ type: "text-delta",
523
+ delta: toReplayText(value)
524
+ });
525
+ return { async *[Symbol.asyncIterator]() {
526
+ for (const item of items) yield item;
527
+ } };
528
+ }
529
+ /**
530
+ * Streaming counterpart to `buildReplayChunks` for a `cachedCall` that
531
+ * *joined* an already-in-flight call for the same key rather than
532
+ * triggering one itself (see `runCachedStream`'s in-flight-coalescing
533
+ * path): there's no live stream to relay (it isn't this call's stream to
534
+ * relay, see the joiner-path comment in `runCachedStream`), but there's
535
+ * also no value yet, only a pending promise for one. Waits for `promise`,
536
+ * then delegates to `buildReplayChunks`. If `promise` rejects, iterating
537
+ * `chunks` throws that same error, consistent with how a live stream's
538
+ * `chunks` throws on a mid-stream failure.
539
+ */
540
+ function buildReplayChunksFromPromise(promise, hasTools) {
541
+ return { async *[Symbol.asyncIterator]() {
542
+ const value = await promise;
543
+ yield* buildReplayChunks(value, hasTools);
544
+ } };
545
+ }
298
546
 
299
547
  //#endregion
300
548
  //#region src/logger.ts
@@ -427,34 +675,52 @@ var TieredCacheAdapter = class {
427
675
  }
428
676
  };
429
677
 
678
+ //#endregion
679
+ //#region src/types/tools.ts
680
+ /**
681
+ * Runtime-safe check for whether a `call()` result is a `tool_calls`
682
+ * result. Prefer this over relying on TypeScript's static narrowing
683
+ * whenever `params` passed to `call()` wasn't a literal with `tools`
684
+ * inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
685
+ * that case TS may have typed the result as plain `T` even though it's
686
+ * actually a `CallWithToolsResult<T>` at runtime, and this check works
687
+ * either way.
688
+ */
689
+ function isToolCallResult(result) {
690
+ return typeof result === "object" && result !== null && "type" in result && result.type === "tool_calls" && Array.isArray(result.toolCalls);
691
+ }
692
+
430
693
  //#endregion
431
694
  //#region src/vernLLM.ts
432
695
  /**
433
- * A resilient layer around an LLM chat completions client, this is VernLLM!
696
+ * A resilient layer around an LLM chat completions client. This is VernLLM!
434
697
  *
435
- * Adds retry with backoff/jitter, per-attempt timeouts, an optional circuit breaker,
436
- * JSON parsing with optional schema validation, usage tracking, and an
437
- * optional response cache, all configurable, all opt-in beyond sensible
438
- * defaults.
698
+ * Adds retry with backoff and jitter, per-attempt timeouts, an optional
699
+ * circuit breaker, JSON parsing with optional schema validation, usage
700
+ * tracking, and an optional response cache. All configurable, all opt-in
701
+ * beyond sensible defaults.
439
702
  */
440
703
  var VernLLM = class {
441
704
  client;
442
705
  model;
443
706
  maxRetries;
444
707
  timeoutMs;
708
+ chunkIdleTimeoutMs;
445
709
  baseDelayMs;
446
710
  defaultMaxTokens;
711
+ defaultTemperature;
447
712
  cache;
448
713
  nonRetryableStatus;
449
714
  inFlight = new Map();
450
715
  parseJson;
451
716
  onUsage;
717
+ onUsageFailure;
452
718
  logger;
453
719
  breaker;
454
720
  /**
455
- * @param options - Client, model, and tunables. Notable defaults:
456
- * `maxRetries` 1, `timeoutMs` 25000, `baseDelayMs` 500 (exponential backoff
457
- * base), `defaultMaxTokens` 1000, `cache` an in-memory adapter,
721
+ * @param options Client, model, and tunables. Defaults: `maxRetries` 1,
722
+ * `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
723
+ * `defaultTemperature` 0.2, `cache` an in-memory adapter,
458
724
  * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
459
725
  */
460
726
  constructor(options) {
@@ -462,8 +728,10 @@ var VernLLM = class {
462
728
  this.model = options.model;
463
729
  this.maxRetries = options.maxRetries ?? 1;
464
730
  this.timeoutMs = options.timeoutMs ?? 25e3;
731
+ this.chunkIdleTimeoutMs = options.chunkIdleTimeoutMs ?? 3e4;
465
732
  this.baseDelayMs = options.baseDelayMs ?? 500;
466
733
  this.defaultMaxTokens = options.defaultMaxTokens ?? 1e3;
734
+ this.defaultTemperature = options.defaultTemperature === void 0 ? .2 : options.defaultTemperature;
467
735
  this.cache = options.cache ?? new InMemoryCacheAdapter();
468
736
  this.nonRetryableStatus = options.nonRetryableStatus ?? [
469
737
  400,
@@ -474,6 +742,7 @@ var VernLLM = class {
474
742
  ];
475
743
  this.parseJson = options.parseJson ?? defaultParseJson;
476
744
  this.onUsage = options.onUsage;
745
+ this.onUsageFailure = options.onUsageFailure;
477
746
  this.logger = options.logger ?? new ConsoleLogger(options.debug ?? false);
478
747
  this.breaker = options.circuitBreaker ? new CircuitBreaker(options.circuitBreaker === true ? void 0 : options.circuitBreaker) : void 0;
479
748
  }
@@ -481,38 +750,307 @@ var VernLLM = class {
481
750
  async resolveCacheKey(key) {
482
751
  return this.cache.resolveKey ? await this.cache.resolveKey(key) : key;
483
752
  }
484
- /**
485
- * Makes a single logical LLM call, retrying on failure per the configured
486
- * policy. Fails fast if the breaker is open or the signal is already
487
- * aborted. On exhausting retries, records a breaker failure and rejects
488
- * with a normalized LLMError.
489
- *
490
- * @param params - System/user content plus per-call overrides (model,
491
- * temperature, jsonMode, schema, signal, etc). See `CallParams`.
492
- * @returns The parsed (and optionally schema-validated) response, or the
493
- * raw string content when `jsonMode` is false and no `jsonSchema` is set.
494
- */
495
753
  async call(params) {
496
754
  this.breaker?.assertClosed();
497
755
  if (params.signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
498
756
  const requestId = params.requestId ?? (0, crypto.randomUUID)();
757
+ if (params.stream) return withReservedUsageForStream(params, async () => {
758
+ try {
759
+ return await this.retryWithBackoff((attempt) => this.executeStreamCall(params, requestId, attempt), requestId, params.signal);
760
+ } catch (error) {
761
+ const normalized = normalizeError(error, params.signal);
762
+ if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
763
+ this.logger.debug(`[VernLLM:${requestId}] stream-open error:\n${describeError(error)}`);
764
+ throw normalized;
765
+ }
766
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
499
767
  return withReservedUsage(params, false, async () => {
500
768
  try {
501
- return await this.retryWithBackoff(() => this.executeCall(params, requestId), requestId, params.signal);
769
+ return await this.retryWithBackoff((attempt) => this.executeCall(params, requestId, attempt), requestId, params.signal);
502
770
  } catch (error) {
503
771
  const normalized = normalizeError(error, params.signal);
504
772
  if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
505
- this.logger.debug(`[vern:${requestId}] error:\n${describeError(error)}`);
773
+ this.logger.debug(`[VernLLM:${requestId}] error:\n${describeError(error)}`);
506
774
  throw normalized;
507
775
  }
508
776
  }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
509
777
  }
778
+ /**
779
+ * Performs a single attempt: builds the request (translating `tools` to
780
+ * wire shape when present), dispatches it with a timeout, and shapes the
781
+ * response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
782
+ * set. Throws on an empty response (no text and no tool_calls) so the
783
+ * retry loop treats it like any other transient failure.
784
+ */
785
+ async executeCall(params, requestId, attempt) {
786
+ const { useJson, model, request } = this.buildRequestPayload(params);
787
+ const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
788
+ const usage = this.extractUsage(response, requestId, model);
789
+ const rawContent = response.choices?.[0]?.message?.content;
790
+ const wireToolCalls = response.choices?.[0]?.message?.tool_calls;
791
+ return this.finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt);
792
+ }
793
+ /**
794
+ * Shapes a fully-arrived response (content and/or tool_calls, already
795
+ * extracted from the provider's payload) into `T` or a
796
+ * `CallWithToolsResult<T>`. Reused by the streaming path once it has
797
+ * buffered the full text/tool-call deltas, so there's no separate
798
+ * parsing/validation logic for streaming.
799
+ *
800
+ * Normalizes and reports usage failure on error itself, so every caller
801
+ * gets identical error handling without duplicating it.
802
+ */
803
+ finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt) {
804
+ try {
805
+ const content = rawContent?.trim();
806
+ if (!content && !wireToolCalls?.length) throw new LLMError("Empty LLM response", "api");
807
+ this.logger.debug(`[VernLLM:${requestId}] output:\n${(content ?? `[${wireToolCalls?.length ?? 0} tool call(s)]`).slice(0, 800)}`);
808
+ if (wireToolCalls?.length) {
809
+ if (!params.tools) throw new LLMError("Provider returned tool_calls but no `tools` were sent with this call.", "api");
810
+ const toolCalls = parseWireToolCalls(wireToolCalls);
811
+ this.validateToolCallArguments(toolCalls, params.tools);
812
+ this.breaker?.recordSuccess();
813
+ this.reportUsage(usage);
814
+ return {
815
+ type: "tool_calls",
816
+ toolCalls,
817
+ ...content ? { content } : {}
818
+ };
819
+ }
820
+ const textContent = content ?? "";
821
+ if (!useJson) {
822
+ this.breaker?.recordSuccess();
823
+ this.reportUsage(usage);
824
+ return params.tools ? {
825
+ type: "content",
826
+ content: textContent
827
+ } : textContent;
828
+ }
829
+ const result = this.parseAndValidate(textContent, params.schema);
830
+ this.breaker?.recordSuccess();
831
+ this.reportUsage(usage);
832
+ return params.tools ? {
833
+ type: "content",
834
+ content: result
835
+ } : result;
836
+ } catch (error) {
837
+ const normalized = normalizeError(error, params.signal);
838
+ if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt);
839
+ throw normalized;
840
+ }
841
+ }
842
+ /**
843
+ * Opens a stream for a single attempt: builds the request exactly like
844
+ * `executeCall`, then requires `createStream` on the client (a clear
845
+ * `validation` error if the adapter doesn't support it). The timeout
846
+ * wraps stream construction and the first `.next()` together, not just
847
+ * construction: calling an `async function*` returns an iterator
848
+ * synchronously without running its body until `.next()` is first
849
+ * invoked, so timing only construction would time an operation that's
850
+ * always instant, not the actual connection. Both are folded into a
851
+ * single `withTimeout` so the same abort signal reaches whatever the
852
+ * adapter's `createStream` uses internally for its first network
853
+ * round-trip.
854
+ *
855
+ * Circuit-breaker success is recorded once the stream fully completes,
856
+ * not on the first chunk arriving, so a connection that opens but then
857
+ * dies mid-stream isn't masked as a success (see `buildStreamResult`).
858
+ */
859
+ async executeStreamCall(params, requestId, attempt) {
860
+ const { useJson, model, request } = this.buildRequestPayload(params);
861
+ const createStream = this.client.chat.completions.createStream;
862
+ if (!createStream) throw new LLMError("stream: true requires a client/adapter with createStream", "validation");
863
+ const streamController = new AbortController();
864
+ const combinedExternal = params.signal ? AbortSignal.any([params.signal, streamController.signal]) : streamController.signal;
865
+ const { iterator, first } = await withTimeout(async (attemptSignal) => {
866
+ const streamIterator = createStream(request, { signal: attemptSignal })[Symbol.asyncIterator]();
867
+ const firstResult = await streamIterator.next();
868
+ return {
869
+ iterator: streamIterator,
870
+ first: firstResult
871
+ };
872
+ }, this.timeoutMs, combinedExternal);
873
+ if (first.done) throw new LLMError("Empty LLM response", "api");
874
+ return this.buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController);
875
+ }
876
+ /**
877
+ * The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
878
+ * an async generator that yields translated `StreamChunk`s to the caller
879
+ * live, as they arrive, with no per-chunk timeout and no bound on total
880
+ * duration, and accumulates text/tool-call deltas internally so that
881
+ * `finalizeResponse` can produce `finalResult` once the stream completes.
882
+ *
883
+ * Two separate try/catches: the iteration loop's catch handles errors
884
+ * the transport itself throws, which aren't normalized yet, so that
885
+ * happens here along with the one `reportUsageFailure` call for them.
886
+ * The second catch, around `finalizeResponse`, does not re-normalize or
887
+ * re-report since `finalizeResponse` already does both internally.
888
+ * Circuit-breaker success is only recorded once the stream fully
889
+ * completes, not when the first chunk arrives, so a connection that
890
+ * opens and then dies mid-way still counts as a failure below instead
891
+ * of masking it.
892
+ */
893
+ buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController) {
894
+ let resolveFinal;
895
+ let rejectFinal;
896
+ const finalResult = new Promise((resolve, reject) => {
897
+ resolveFinal = resolve;
898
+ rejectFinal = reject;
899
+ });
900
+ finalResult.catch(() => {});
901
+ const MAX_BUFFERED_CHUNKS = 1e4;
902
+ const buffered = [];
903
+ const pending = [];
904
+ let streamDone = false;
905
+ let streamError;
906
+ let hasLoggedEviction = false;
907
+ const push = (chunk) => {
908
+ const waiter = pending.shift();
909
+ if (waiter) {
910
+ waiter.resolve({
911
+ done: false,
912
+ value: chunk
913
+ });
914
+ return;
915
+ }
916
+ buffered.push(chunk);
917
+ if (buffered.length > MAX_BUFFERED_CHUNKS * 2) {
918
+ if (!hasLoggedEviction) {
919
+ hasLoggedEviction = true;
920
+ this.logger.debug(`[VernLLM] stream chunk buffer exceeded cap (${MAX_BUFFERED_CHUNKS}), evicting ${buffered.length - MAX_BUFFERED_CHUNKS} oldest chunk(s); buffered=${buffered.length}. The chunks iterable was never read (or fell far behind) for this stream.`);
921
+ }
922
+ buffered.splice(0, buffered.length - MAX_BUFFERED_CHUNKS);
923
+ }
924
+ };
925
+ const finish = () => {
926
+ streamDone = true;
927
+ for (const waiter of pending.splice(0)) waiter.resolve({
928
+ done: true,
929
+ value: void 0
930
+ });
931
+ };
932
+ const fail = (error) => {
933
+ streamDone = true;
934
+ streamError = error;
935
+ for (const waiter of pending.splice(0)) waiter.reject(error);
936
+ };
937
+ const chunks = { [Symbol.asyncIterator]() {
938
+ return { next() {
939
+ if (buffered.length) return Promise.resolve({
940
+ done: false,
941
+ value: buffered.shift()
942
+ });
943
+ if (streamDone) return streamError ? Promise.reject(streamError) : Promise.resolve({
944
+ done: true,
945
+ value: void 0
946
+ });
947
+ return new Promise((resolve, reject) => {
948
+ pending.push({
949
+ resolve,
950
+ reject
951
+ });
952
+ });
953
+ } };
954
+ } };
955
+ const toolCallAcc = new Map();
956
+ let textAcc = "";
957
+ let usage;
958
+ (async () => {
959
+ try {
960
+ let result = first;
961
+ while (!result.done) {
962
+ const wireChunk = result.value;
963
+ if (wireChunk.type === "ping") {} else if (wireChunk.type === "text-delta") {
964
+ textAcc += wireChunk.delta;
965
+ push({
966
+ type: "text-delta",
967
+ delta: wireChunk.delta
968
+ });
969
+ } else if (wireChunk.type === "tool_call_delta") {
970
+ const entry = toolCallAcc.get(wireChunk.index) ?? { args: "" };
971
+ entry.id ??= wireChunk.id;
972
+ entry.name ??= wireChunk.name;
973
+ entry.args += wireChunk.argumentsDelta ?? "";
974
+ toolCallAcc.set(wireChunk.index, entry);
975
+ push({
976
+ type: "tool_call_delta",
977
+ index: wireChunk.index,
978
+ id: wireChunk.id,
979
+ name: wireChunk.name,
980
+ argsDelta: wireChunk.argumentsDelta,
981
+ complete: wireChunk.complete
982
+ });
983
+ } else if (wireChunk.type === "usage") {
984
+ usage = {
985
+ promptTokens: wireChunk.usage.prompt_tokens ?? 0,
986
+ completionTokens: wireChunk.usage.completion_tokens ?? 0,
987
+ totalTokens: wireChunk.usage.total_tokens ?? 0,
988
+ requestId,
989
+ model
990
+ };
991
+ push({
992
+ type: "usage",
993
+ usage
994
+ });
995
+ }
996
+ result = await withChunkIdleTimeout(() => iterator.next(), params.chunkIdleTimeoutMs ?? this.chunkIdleTimeoutMs, () => streamController.abort(), this.logger);
997
+ }
998
+ } catch (error) {
999
+ try {
1000
+ await iterator.return?.();
1001
+ } catch {}
1002
+ streamController.abort();
1003
+ const normalized = normalizeError(error, params.signal);
1004
+ if (normalized.type === "timeout") this.breaker?.recordFailure();
1005
+ if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt, true);
1006
+ fail(normalized);
1007
+ rejectFinal(normalized);
1008
+ return;
1009
+ }
1010
+ finish();
1011
+ this.breaker?.recordSuccess();
1012
+ try {
1013
+ const wireToolCalls = toolCallAcc.size ? [...toolCallAcc.entries()].sort(([indexA], [indexB]) => indexA - indexB).map(([, entry]) => ({
1014
+ id: entry.id ?? "",
1015
+ type: "function",
1016
+ function: {
1017
+ name: entry.name ?? "",
1018
+ arguments: entry.args
1019
+ }
1020
+ })) : void 0;
1021
+ const finalized = this.finalizeResponse(textAcc, wireToolCalls, params, useJson, usage, requestId, attempt);
1022
+ resolveFinal(finalized);
1023
+ } catch (error) {
1024
+ rejectFinal(error);
1025
+ }
1026
+ })();
1027
+ return {
1028
+ chunks,
1029
+ finalResult
1030
+ };
1031
+ }
1032
+ /**
1033
+ * Checks every `ToolCall` against the `tools` that were offered, catching
1034
+ * a hallucinated tool name early instead of letting it reach the
1035
+ * application's dispatch table. Then runs each tool's `argumentsSchema`,
1036
+ * if present, throwing `LLMError('validation')` on failure.
1037
+ */
1038
+ validateToolCallArguments(toolCalls, tools) {
1039
+ const knownNames = new Set(tools.map((t) => t.name));
1040
+ for (const call of toolCalls) {
1041
+ if (!knownNames.has(call.name)) throw new LLMError(`Model requested tool "${call.name}", which was not in the tools offered ([${[...knownNames].join(", ")}]).`, "api");
1042
+ const definition = tools.find((t) => t.name === call.name);
1043
+ if (!definition?.argumentsSchema) continue;
1044
+ const result = definition.argumentsSchema.safeParse(call.arguments);
1045
+ if (!result.success) throw new LLMError(`Arguments for tool call "${call.name}" failed validation`, "validation", void 0, result.error);
1046
+ }
1047
+ }
510
1048
  /** Runs `fn`, retrying with backoff according to `shouldRetry`. */
511
1049
  async retryWithBackoff(fn, requestId, signal) {
512
1050
  let lastError;
513
1051
  for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
514
1052
  if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
515
- return await fn();
1053
+ return await fn(attempt);
516
1054
  } catch (error) {
517
1055
  lastError = error;
518
1056
  if (!this.shouldRetry(error, signal)) break;
@@ -520,60 +1058,73 @@ var VernLLM = class {
520
1058
  throw lastError;
521
1059
  }
522
1060
  /**
523
- * Performs a single attempt: builds the request, dispatches it with a
524
- * timeout, and shapes the response. Throws on an empty response so the
525
- * retry loop treats it like any other transient failure.
526
- */
527
- async executeCall(params, requestId) {
528
- const { useJson, model, request } = this.buildRequestPayload(params);
529
- const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
530
- const content = response.choices?.[0]?.message?.content?.trim();
531
- if (!content) throw new LLMError("Empty LLM response", "api");
532
- this.logger.debug(`[vern:${requestId}] output:\n${content.slice(0, 800)}`);
533
- this.recordUsage(response, requestId, model);
534
- if (!useJson) {
535
- this.breaker?.recordSuccess();
536
- return content;
537
- }
538
- const result = this.parseAndValidate(content, params.schema);
539
- this.breaker?.recordSuccess();
540
- return result;
541
- }
542
- /**
543
1061
  * Validates `history` alternates user/assistant turns, since providers
544
1062
  * like Anthropic/Gemini reject or mishandle consecutive same-role turns.
545
1063
  */
546
1064
  validateHistory(history) {
547
- let previousRole;
1065
+ let previousTurn;
548
1066
  for (const [index, turn] of history.entries()) {
549
- if (turn.role !== "user" && turn.role !== "assistant") throw new LLMError(`Invalid history[${index}].role "${turn.role}": must be "user" or "assistant"`, "validation");
550
- if (turn.role === previousRole) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
551
- previousRole = turn.role;
1067
+ if (turn.role === "tool") {
1068
+ if (previousTurn?.role !== "assistant" || !previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] is a "tool" turn, but must immediately follow an "assistant" turn that requested tools`, "validation");
1069
+ if (!turn.toolResults?.length) throw new LLMError(`history[${index}] is a "tool" turn but has no toolResults`, "validation");
1070
+ const requestedIds = new Set(previousTurn.toolCalls.map((tc) => tc.id));
1071
+ const resultIds = turn.toolResults.map((tr) => tr.toolCallId);
1072
+ const unknownIds = resultIds.filter((id) => !requestedIds.has(id));
1073
+ if (unknownIds.length) throw new LLMError(`history[${index}].toolResults references unknown toolCallId(s) [${unknownIds.join(", ")}]`, "validation");
1074
+ const seenIds = new Set();
1075
+ const duplicateIds = new Set();
1076
+ for (const id of resultIds) {
1077
+ if (seenIds.has(id)) duplicateIds.add(id);
1078
+ seenIds.add(id);
1079
+ }
1080
+ if (duplicateIds.size) throw new LLMError(`history[${index}].toolResults has duplicate toolCallId(s) [${[...duplicateIds].join(", ")}]`, "validation");
1081
+ const missingIds = [...requestedIds].filter((id) => !resultIds.includes(id));
1082
+ if (missingIds.length) throw new LLMError(`history[${index}] is missing toolResults for toolCallId(s) [${missingIds.join(", ")}]`, "validation");
1083
+ } else {
1084
+ if (turn.role === previousTurn?.role) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
1085
+ if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] follows an assistant tool request without tool results`, "validation");
1086
+ }
1087
+ previousTurn = turn;
552
1088
  }
553
- if (previousRole === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn. history must end with an \"assistant\" turn (or be empty).", "validation");
1089
+ if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError("The last entry in history is an assistant tool request without tool results", "validation");
1090
+ if (previousTurn?.role === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn.", "validation");
554
1091
  }
555
1092
  /** Applies per-call defaults and shapes params into the client's request object. */
556
1093
  buildRequestPayload(params) {
557
- const { systemPrompt, userContent, history = [], temperature = .2, jsonMode = true, maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema } = params;
1094
+ const { systemPrompt, userContent, history = [], maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema, tools, toolChoice } = params;
1095
+ const temperature = params.temperature === void 0 ? this.defaultTemperature : params.temperature;
1096
+ if (tools && (jsonSchema || params.schema)) throw new LLMError("`tools` cannot be combined with `jsonSchema`/`schema`: on Anthropic and Bedrock, jsonSchema is implemented internally as a forced single-tool call, which would collide with real tools. Use one or the other.", "validation");
1097
+ if (tools && tools.length === 0) throw new LLMError("`tools` was an empty array. This is almost always a bug (e.g. a filtered tool list that ended up empty). An empty `tools` array still switches on tool-call mode (response shape, jsonMode default, wire format) with nothing for the model to call. Omit `tools` entirely for a normal call, or make sure the array is non-empty.", "validation");
1098
+ if (tools) {
1099
+ const seen = new Set();
1100
+ const duplicates = new Set();
1101
+ for (const tool of tools) {
1102
+ if (seen.has(tool.name)) duplicates.add(tool.name);
1103
+ seen.add(tool.name);
1104
+ }
1105
+ if (duplicates.size) throw new LLMError(`\`tools\` has duplicate name(s): [${[...duplicates].join(", ")}]. Tool names must be unique.`, "validation");
1106
+ }
1107
+ if (toolChoice && !tools) throw new LLMError("`toolChoice` was set without `tools`. There is nothing for it to choose between. Set `tools`, or remove `toolChoice`.", "validation");
1108
+ if (tools && typeof toolChoice === "object" && !tools.some((t) => t.name === toolChoice.name)) throw new LLMError(`toolChoice names "${toolChoice.name}", which is not in \`tools\` ([${tools.map((t) => t.name).join(", ")}]).`, "validation");
1109
+ const jsonMode = params.jsonMode ?? (tools ? false : true);
558
1110
  const useJson = jsonMode || Boolean(jsonSchema);
559
1111
  if (params.schema && !useJson) throw new LLMError("schema was provided but jsonMode: false disables JSON parsing, so nothing would validate it. Remove jsonMode: false, set jsonSchema, or remove schema.", "validation");
560
1112
  const responseFormat = this.buildResponseFormat(jsonSchema, useJson);
561
1113
  this.validateHistory(history);
562
1114
  const request = {
563
1115
  model,
564
- temperature,
1116
+ ...temperature !== null ? { temperature } : {},
565
1117
  max_tokens: maxTokens,
566
1118
  ...responseFormat ? { response_format: responseFormat } : {},
567
1119
  ...reasoningEffort ? { reasoning_effort: reasoningEffort } : {},
1120
+ ...tools ? { tools: toWireTools(tools) } : {},
1121
+ ...tools ? { tool_choice: this.buildWireToolChoice(toolChoice) } : {},
568
1122
  messages: [
569
1123
  ...systemPrompt ? [{
570
1124
  role: "system",
571
1125
  content: systemPrompt
572
1126
  }] : [],
573
- ...history.map((turn) => ({
574
- role: turn.role,
575
- content: turn.content
576
- })),
1127
+ ...history.flatMap((turn) => this.turnToWireMessages(turn)),
577
1128
  {
578
1129
  role: "user",
579
1130
  content: userContent
@@ -586,6 +1137,39 @@ var VernLLM = class {
586
1137
  request
587
1138
  };
588
1139
  }
1140
+ /** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
1141
+ buildWireToolChoice(toolChoice) {
1142
+ if (!toolChoice || toolChoice === "auto") return "auto";
1143
+ if (toolChoice === "none" || toolChoice === "required") return toolChoice;
1144
+ return {
1145
+ type: "function",
1146
+ function: { name: toolChoice.name }
1147
+ };
1148
+ }
1149
+ /**
1150
+ * Expands one `ConversationTurn` into one or more wire messages. Plain
1151
+ * user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
1152
+ * to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
1153
+ * turn expands into one wire `tool` message per `toolResult`, since
1154
+ * OpenAI-shaped wire format wants one message per tool_call_id.
1155
+ */
1156
+ turnToWireMessages(turn) {
1157
+ if (turn.role === "tool") return (turn.toolResults ?? []).map((tr) => ({
1158
+ role: "tool",
1159
+ tool_call_id: tr.toolCallId,
1160
+ content: typeof tr.content === "string" ? tr.content : JSON.stringify(tr.content ?? null),
1161
+ ...tr.isError ? { is_error: true } : {}
1162
+ }));
1163
+ if (turn.role === "assistant" && turn.toolCalls?.length) return [{
1164
+ role: "assistant",
1165
+ ...turn.content ? { content: turn.content } : {},
1166
+ tool_calls: toWireToolCalls(turn.toolCalls)
1167
+ }];
1168
+ return [{
1169
+ role: turn.role,
1170
+ content: turn.content ?? ""
1171
+ }];
1172
+ }
589
1173
  /**
590
1174
  * Chooses the response format: a provider-native `jsonSchema` takes
591
1175
  * priority when supplied (constrains generation directly), otherwise
@@ -604,21 +1188,49 @@ var VernLLM = class {
604
1188
  };
605
1189
  return useJson ? { type: "json_object" } : void 0;
606
1190
  }
607
- /** Reports token usage to `onUsage`, swallowing and logging any error it throws. */
608
- recordUsage(response, requestId, model) {
609
- if (!response.usage || !this.onUsage) return;
1191
+ /**
1192
+ * Pulls `TokenUsage` out of a raw response, if the provider reported it.
1193
+ * Extraction doesn't depend on what happens to the response afterward, so
1194
+ * a malformed body can still yield usage if the provider's usage block
1195
+ * itself came through intact.
1196
+ */
1197
+ extractUsage(response, requestId, model) {
1198
+ if (!response.usage) return void 0;
1199
+ return {
1200
+ promptTokens: response.usage.prompt_tokens ?? 0,
1201
+ completionTokens: response.usage.completion_tokens ?? 0,
1202
+ totalTokens: response.usage.total_tokens ?? 0,
1203
+ requestId,
1204
+ model
1205
+ };
1206
+ }
1207
+ /** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
1208
+ reportUsage(usage) {
1209
+ if (!usage || !this.onUsage) return;
610
1210
  try {
611
- this.onUsage({
612
- promptTokens: response.usage.prompt_tokens ?? 0,
613
- completionTokens: response.usage.completion_tokens ?? 0,
614
- totalTokens: response.usage.total_tokens ?? 0,
615
- requestId,
616
- model
617
- });
1211
+ this.onUsage(usage);
618
1212
  } catch (error) {
619
1213
  this.logger.error("[VernLLM] onUsage failed", { message: error instanceof Error ? error.message : "unknown" });
620
1214
  }
621
1215
  }
1216
+ /**
1217
+ * Reports token usage spent on an attempt that then failed, so it isn't
1218
+ * dropped alongside the error. Covers any error thrown after usage
1219
+ * extraction, since all of them happen only after a response (real
1220
+ * spend) already arrived. Swallows and logs any error `onUsageFailure`
1221
+ * itself throws.
1222
+ */
1223
+ reportUsageFailure(usage, error, attempt, terminal = false) {
1224
+ const displayTokens = usage.totalTokens || usage.promptTokens + usage.completionTokens;
1225
+ const attemptText = terminal ? "mid-stream failure (terminal, no further attempts)" : `attempt ${attempt + 1}/${this.maxRetries + 1}`;
1226
+ this.logger.warn(`[VernLLM:${usage.requestId}] usage failure, ${attemptText}: type=${error.type} tokens=${displayTokens}`);
1227
+ if (!this.onUsageFailure) return;
1228
+ try {
1229
+ this.onUsageFailure(usage, error);
1230
+ } catch (hookError) {
1231
+ this.logger.error("[VernLLM] onUsageFailure failed", { message: hookError instanceof Error ? hookError.message : "unknown" });
1232
+ }
1233
+ }
622
1234
  /** Parses response content as JSON and validates it against `schema` when supplied. */
623
1235
  parseAndValidate(content, schema) {
624
1236
  let parsed;
@@ -642,7 +1254,7 @@ var VernLLM = class {
642
1254
  async recoverDelay(requestId, attempt, error, signal) {
643
1255
  const retryAfterMs = extractRetryAfterMs(error);
644
1256
  const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
645
- this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
1257
+ this.logger.warn(`[VernLLM:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
646
1258
  await waitForRetry(delay, signal);
647
1259
  }
648
1260
  /** Decides whether a failed attempt is worth retrying. */
@@ -657,7 +1269,7 @@ var VernLLM = class {
657
1269
  * supports deletion. Cache invalidation is the caller's responsibility;
658
1270
  * only the application knows when cached data is stale.
659
1271
  *
660
- * @param key - The raw cache key (resolved through the adapter's
1272
+ * @param key The raw cache key (resolved through the adapter's
661
1273
  * `resolveKey`, if any, before deletion).
662
1274
  */
663
1275
  async deleteCache(key) {
@@ -665,15 +1277,20 @@ var VernLLM = class {
665
1277
  await this.cache.delete(await this.resolveCacheKey(key));
666
1278
  }
667
1279
  /**
668
- * Cache wrapper around caller-supplied logic. Concurrent misses for the
669
- * same `cacheKey` share a single in-flight call, avoiding cache stampedes.
1280
+ * Internal cache primitive around caller-supplied logic. Concurrent misses
1281
+ * for the same `cacheKey` share a single in-flight call, avoiding cache
1282
+ * stampedes.
670
1283
  *
671
- * @param params - `cacheKey`, `ttl`, `fn` (the work to run on a cache
1284
+ * Not part of the public API. Backs the public `cachedCall()`, which
1285
+ * always composes this with `call()` so cached results get the same
1286
+ * retry/timeout/circuit-breaker guarantees as any other LLM call.
1287
+ *
1288
+ * @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
672
1289
  * miss, typically `() => this.call(...)`), and optional
673
- * `reserveUsage`/`refundUsage`/`signal`. See `CachedCallParams`.
1290
+ * `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
674
1291
  * @returns The cached value on a hit, or the result of `fn()` on a miss.
675
1292
  */
676
- async cachedCall(params) {
1293
+ async runCached(params) {
677
1294
  const resolvedKey = await this.resolveCacheKey(params.cacheKey);
678
1295
  const resolvedParams = resolvedKey === params.cacheKey ? params : {
679
1296
  ...params,
@@ -704,24 +1321,113 @@ var VernLLM = class {
704
1321
  }
705
1322
  return result;
706
1323
  }
707
- /** Logs a failed refundUsage attempt via the configured logger. */
708
- logRefundError(logMessage, error) {
709
- this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
1324
+ /**
1325
+ * Streaming counterpart to `runCached`. Three cases:
1326
+ *
1327
+ * - Hit: no live generation to relay. Returns immediately with
1328
+ * `finalResult` resolved to the cached value and a one-shot `chunks`
1329
+ * replay built from it, so `for await (const c of chunks)` call sites
1330
+ * work identically on a hit or a miss. No usage hooks fire, since
1331
+ * nothing was actually spent.
1332
+ * - Miss, nothing else in flight for this key: delegates to
1333
+ * `registerStreamTrigger`, which opens the stream and relays its
1334
+ * `chunks` live.
1335
+ * - Miss, but another call for the same key is already in flight: this
1336
+ * call has no live chunks of its own to relay, so it's treated like a
1337
+ * delayed hit. `finalResult` shares the trigger's in-flight promise
1338
+ * (the same `this.inFlight` map non-streaming `runCached` uses, so
1339
+ * streaming and non-streaming `cachedCall`s for the same key coalesce
1340
+ * against each other too), and `chunks` is a one-shot replay built
1341
+ * once that promise resolves.
1342
+ */
1343
+ async runCachedStream(params, hasTools) {
1344
+ const resolvedKey = await this.resolveCacheKey(params.cacheKey);
1345
+ const resolvedParams = resolvedKey === params.cacheKey ? params : {
1346
+ ...params,
1347
+ cacheKey: resolvedKey
1348
+ };
1349
+ const cached = await this.cache.get(resolvedKey);
1350
+ if (cached.hit) {
1351
+ const value = cached.value;
1352
+ return {
1353
+ chunks: buildReplayChunks(value, hasTools),
1354
+ finalResult: Promise.resolve(value)
1355
+ };
1356
+ }
1357
+ const existing = this.inFlight.get(resolvedKey);
1358
+ if (existing) {
1359
+ const finalResult = withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
1360
+ return {
1361
+ chunks: buildReplayChunksFromPromise(finalResult, hasTools),
1362
+ finalResult
1363
+ };
1364
+ }
1365
+ return this.registerStreamTrigger(resolvedParams);
710
1366
  }
711
1367
  /**
712
- * Convenience wrapper composing `call` + `cachedCall`, so cached LLM calls
713
- * automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
714
- * `refundUsage` are read from the top-level params only.
1368
+ * Opens the shared stream for a cache miss and tracks its settled value
1369
+ * in `this.inFlight` until it resolves or rejects. Writes to the cache
1370
+ * on success only, matching `runAndCache`.
715
1371
  *
716
- * @param params - `cachedCall` params (`cacheKey`, `ttl`, etc, minus `fn`)
717
- * plus `call`, the `CallParams` to pass through to `this.call(...)`.
718
- * @returns The cached value on a hit, or the freshly-called result on a miss.
1372
+ * Registers the in-flight promise synchronously, before anything async
1373
+ * runs, so a concurrent `cachedCall` for the same key always sees it in
1374
+ * time to join instead of triggering its own stream. Settlement is
1375
+ * wired onto the whole `withReservedUsageForStream` call rather than a
1376
+ * line inside its callback, so any failure point (reserving usage,
1377
+ * opening the stream, or the stream itself) reliably settles the
1378
+ * in-flight entry instead of leaving it stuck.
719
1379
  */
720
- async cachedLLMCall(params) {
1380
+ registerStreamTrigger(params) {
1381
+ let resolveInFlight;
1382
+ let rejectInFlight;
1383
+ const inFlightResult = new Promise((resolve, reject) => {
1384
+ resolveInFlight = resolve;
1385
+ rejectInFlight = reject;
1386
+ });
1387
+ this.inFlight.set(params.cacheKey, inFlightResult);
1388
+ inFlightResult.catch(() => {}).finally(() => {
1389
+ this.inFlight.delete(params.cacheKey);
1390
+ });
1391
+ const streamPromise = withReservedUsageForStream(params, async () => {
1392
+ const opened = await params.openStream();
1393
+ const trackedResult = opened.finalResult.then(async (value) => {
1394
+ try {
1395
+ await this.cache.set(params.cacheKey, value, params.ttl);
1396
+ } catch (error) {
1397
+ this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
1398
+ }
1399
+ return value;
1400
+ }, (error) => {
1401
+ throw error;
1402
+ });
1403
+ return {
1404
+ chunks: opened.chunks,
1405
+ finalResult: trackedResult
1406
+ };
1407
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
1408
+ streamPromise.then((opened) => {
1409
+ opened.finalResult.then(resolveInFlight, rejectInFlight);
1410
+ }, (error) => {
1411
+ rejectInFlight(error);
1412
+ });
1413
+ return streamPromise;
1414
+ }
1415
+ /** Logs a failed refundUsage attempt via the configured logger. */
1416
+ logRefundError(logMessage, error) {
1417
+ this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
1418
+ }
1419
+ async cachedCall(params) {
721
1420
  const { call: callParams,...cacheParams } = params;
722
- const { reserveUsage: innerReserveUsage, refundUsage: innerRefundUsage,...restCallParams } = callParams;
723
- if (innerReserveUsage || innerRefundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedLLMCall; set them at the top level instead.");
724
- return this.cachedCall({
1421
+ const { reserveUsage, refundUsage,...restCallParams } = callParams;
1422
+ if (reserveUsage || refundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedCall; set them at the top level instead.");
1423
+ if (restCallParams.stream) {
1424
+ const streamParams = restCallParams;
1425
+ return this.runCachedStream({
1426
+ ...cacheParams,
1427
+ openStream: () => this.call(streamParams)
1428
+ }, Boolean(restCallParams.tools));
1429
+ }
1430
+ return this.runCached({
725
1431
  ...cacheParams,
726
1432
  fn: () => this.call(restCallParams)
727
1433
  });
@@ -735,6 +1441,108 @@ var VernLLM = class {
735
1441
  }
736
1442
  };
737
1443
 
1444
+ //#endregion
1445
+ //#region src/internal/sse.ts
1446
+ /**
1447
+ * Parses a Server-Sent-Events byte/text stream into the JSON payload of
1448
+ * each `data:` frame, in arrival order. Generic over transport: works with
1449
+ * anything that hands back progressively-arriving `Uint8Array` or `string`
1450
+ * chunks via async iteration: native `fetch`'s `response.body` (wrapped
1451
+ * to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
1452
+ * Node `Readable` (already async-iterable, no wrapping needed), etc, so
1453
+ * this framing layer doesn't care which transport produced the bytes.
1454
+ *
1455
+ * Follows the SSE spec's frame-delimiting rules closely enough for LLM
1456
+ * streaming responses: frames are separated by a blank line, each frame
1457
+ * may carry one or more `data:` lines (joined with `\n` per spec when
1458
+ * there's more than one), `:`-prefixed lines are comments and ignored, and
1459
+ * other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
1460
+ * only needs the payload. A frame whose data is exactly `[DONE]` (the
1461
+ * sentinel several providers, notably OpenAI, send to mark stream end)
1462
+ * ends iteration without yielding it.
1463
+ *
1464
+ * Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
1465
+ * to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
1466
+ * alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
1467
+ * stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
1468
+ * lines.
1469
+ *
1470
+ * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
1471
+ * how malformed JSON is handled elsewhere in VernLLM.
1472
+ */
1473
+ async function* parseSseStream(source) {
1474
+ const decoder = new TextDecoder("utf-8", { fatal: true });
1475
+ let buffer = "";
1476
+ for await (const chunk of source) {
1477
+ let text;
1478
+ try {
1479
+ text = typeof chunk === "string" ? chunk : decoder.decode(chunk, { stream: true });
1480
+ } catch (cause) {
1481
+ throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
1482
+ }
1483
+ buffer = (buffer + text).replace(/\r\n/g, "\n").replace(/\r(?!$)/g, "\n");
1484
+ let boundary$1 = buffer.indexOf("\n\n");
1485
+ while (boundary$1 !== -1) {
1486
+ const frame = buffer.slice(0, boundary$1);
1487
+ buffer = buffer.slice(boundary$1 + 2);
1488
+ const event = parseSseFrame(frame);
1489
+ if (event === DONE) return;
1490
+ if (event !== NO_DATA) yield event;
1491
+ boundary$1 = buffer.indexOf("\n\n");
1492
+ }
1493
+ }
1494
+ try {
1495
+ buffer += decoder.decode();
1496
+ } catch (cause) {
1497
+ throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
1498
+ }
1499
+ buffer = buffer.replace(/\r$/, "\n");
1500
+ let boundary = buffer.indexOf("\n\n");
1501
+ while (boundary !== -1) {
1502
+ const frame = buffer.slice(0, boundary);
1503
+ buffer = buffer.slice(boundary + 2);
1504
+ const event = parseSseFrame(frame);
1505
+ if (event === DONE) return;
1506
+ if (event !== NO_DATA) yield event;
1507
+ boundary = buffer.indexOf("\n\n");
1508
+ }
1509
+ const trailing = buffer.trim();
1510
+ if (trailing) {
1511
+ const event = parseSseFrame(trailing);
1512
+ if (event !== DONE && event !== NO_DATA) yield event;
1513
+ }
1514
+ }
1515
+ const DONE = Symbol("sse-stream-done");
1516
+ const NO_DATA = Symbol("sse-frame-no-data");
1517
+ /**
1518
+ * Sentinel yielded by `parseSseStream` for a comment-only frame (no
1519
+ * `data:` payload), the mechanism providers use for SSE keep-alive
1520
+ * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
1521
+ * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
1522
+ */
1523
+ const SSE_PING = Symbol("sse-frame-ping");
1524
+ /** Extracts and JSON-parses the `data:` payload of one SSE frame (the text between two blank lines). */
1525
+ function parseSseFrame(frame) {
1526
+ const dataLines = [];
1527
+ let sawComment = false;
1528
+ for (const line of frame.split("\n")) {
1529
+ if (line.startsWith(":")) {
1530
+ sawComment = true;
1531
+ continue;
1532
+ }
1533
+ if (!line.startsWith("data:")) continue;
1534
+ dataLines.push(line.startsWith("data: ") ? line.slice(6) : line.slice(5));
1535
+ }
1536
+ if (!dataLines.length) return sawComment ? SSE_PING : NO_DATA;
1537
+ const data = dataLines.join("\n");
1538
+ if (data === "[DONE]") return DONE;
1539
+ try {
1540
+ return JSON.parse(data);
1541
+ } catch (cause) {
1542
+ throw new LLMError(`Invalid JSON in SSE frame: ${data.slice(0, 200)}`, "parse", void 0, void 0, cause);
1543
+ }
1544
+ }
1545
+
738
1546
  //#endregion
739
1547
  //#region src/internal/imageFormat.ts
740
1548
  /**
@@ -782,6 +1590,93 @@ function toAnthropicContent(blocks) {
782
1590
  });
783
1591
  }
784
1592
  /**
1593
+ * Asserts a caller-supplied JSON Schema is an object schema before it's
1594
+ * used as Anthropic's `Tool.input_schema`, which (like every other
1595
+ * provider's function-calling API) requires `type: 'object'`. VernLLM's own
1596
+ * public `tools`/`jsonSchema` APIs accept freeform `Record<string,
1597
+ * unknown>` JSON Schema, so nothing upstream guarantees this at compile
1598
+ * time; this is the runtime check that stands in for that, so a schema
1599
+ * missing (or mistyping) `type: 'object'` fails loudly and immediately
1600
+ * instead of being silently forwarded to Anthropic malformed.
1601
+ */
1602
+ function assertObjectSchema(schema, toolName) {
1603
+ if (schema.type !== "object") throw new LLMError(`Tool "${toolName}"'s schema must have "type": "object" (Anthropic requires object-shaped tool parameters).`, "validation");
1604
+ return schema;
1605
+ }
1606
+ /**
1607
+ * Translates VernLLM's OpenAI-shaped wire `tool_choice` into Anthropic's
1608
+ * `{ type: 'auto' | 'any' | 'none' | 'tool', name? }` shape. `'required'`
1609
+ * maps to `'any'` (Anthropic's "must call some tool" equivalent).
1610
+ */
1611
+ function toAnthropicToolChoice(toolChoice) {
1612
+ if (!toolChoice || toolChoice === "auto") return { type: "auto" };
1613
+ if (toolChoice === "none") return { type: "none" };
1614
+ if (toolChoice === "required") return { type: "any" };
1615
+ return {
1616
+ type: "tool",
1617
+ name: toolChoice.function.name
1618
+ };
1619
+ }
1620
+ /**
1621
+ * Builds the Anthropic-shaped request body from VernLLM's wire params,
1622
+ * shared between `create` and `createStream` so both go through identical
1623
+ * translation (system prompt, message shaping, and the jsonSchema →
1624
+ * forced-single-tool mapping all happen exactly once, not once per entry
1625
+ * point).
1626
+ *
1627
+ * Returns `toolName` alongside the body: when set, the model was forced to
1628
+ * call a single synthetic tool standing in for `jsonSchema` output, and
1629
+ * both `create` and `createStream` need to know this so they can unwrap
1630
+ * that tool call back into plain text content instead of treating it like
1631
+ * a real tool call.
1632
+ */
1633
+ function buildAnthropicRequestBody(params) {
1634
+ const systemMessage = params.messages.find((m) => m.role === "system");
1635
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
1636
+ const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name.trim() : void 0;
1637
+ if (params.response_format?.type === "json_schema" && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
1638
+ let jsonInstruction;
1639
+ let tools;
1640
+ let toolChoice;
1641
+ if (params.response_format?.type === "json_schema" && toolName) {
1642
+ const { schema, description, strict } = params.response_format.json_schema;
1643
+ tools = [{
1644
+ name: toolName,
1645
+ description,
1646
+ input_schema: assertObjectSchema(schema, toolName),
1647
+ strict
1648
+ }];
1649
+ toolChoice = {
1650
+ type: "tool",
1651
+ name: toolName
1652
+ };
1653
+ } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
1654
+ else if (params.tools?.length) {
1655
+ tools = params.tools.map((t) => ({
1656
+ name: t.function.name,
1657
+ description: t.function.description,
1658
+ input_schema: assertObjectSchema(t.function.parameters, t.function.name)
1659
+ }));
1660
+ toolChoice = toAnthropicToolChoice(params.tool_choice);
1661
+ }
1662
+ const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
1663
+ const body = {
1664
+ model: params.model,
1665
+ max_tokens: params.max_tokens,
1666
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
1667
+ system: system || void 0,
1668
+ messages: mergeConsecutiveToolResults$1(conversationMessages.map((m) => toAnthropicMessage(m))),
1669
+ ...tools ? {
1670
+ tools,
1671
+ tool_choice: toolChoice
1672
+ } : {}
1673
+ };
1674
+ return {
1675
+ body,
1676
+ toolName
1677
+ };
1678
+ }
1679
+ /**
785
1680
  * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
786
1681
  * interface VernLLM uses for OpenAI/Groq.
787
1682
  *
@@ -796,53 +1691,161 @@ function toAnthropicContent(blocks) {
796
1691
  * generation against.
797
1692
  */
798
1693
  function fromAnthropic(anthropicClient) {
799
- return { chat: { completions: { async create(params, options) {
800
- const systemMessage = params.messages.find((m) => m.role === "system");
801
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
802
- const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name : void 0;
803
- let jsonInstruction;
804
- let tools;
805
- if (params.response_format?.type === "json_schema" && toolName) {
806
- const { schema, description, strict } = params.response_format.json_schema;
807
- tools = [{
808
- name: toolName,
809
- description,
810
- input_schema: schema,
811
- strict
812
- }];
813
- } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
814
- const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
815
- const response = await anthropicClient.messages.create({
816
- model: params.model,
817
- max_tokens: params.max_tokens,
818
- temperature: params.temperature,
819
- system: system || void 0,
820
- messages: conversationMessages.map((m) => ({
821
- role: m.role,
822
- content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content
823
- })),
824
- ...tools ? {
825
- tools,
826
- tool_choice: {
827
- type: "tool",
828
- name: toolName
1694
+ const rawMessagesCreate = anthropicClient.messages.create.bind(anthropicClient.messages);
1695
+ return { chat: { completions: {
1696
+ async create(params, options) {
1697
+ const { body, toolName } = buildAnthropicRequestBody(params);
1698
+ const response = await anthropicClient.messages.create(body, options);
1699
+ let text;
1700
+ let wireToolCalls;
1701
+ if (toolName) {
1702
+ const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
1703
+ if (!toolUse) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
1704
+ if (!toolUse.input || typeof toolUse.input !== "object" || Array.isArray(toolUse.input)) throw new LLMError(`Anthropic returned invalid structured output for tool "${toolName}". Expected an object.`, "validation");
1705
+ text = JSON.stringify(toolUse.input);
1706
+ } else {
1707
+ text = response.content.filter((block) => block.type === "text").map((block) => block.text ?? "").join("");
1708
+ const toolUses = response.content.filter((block) => block.type === "tool_use");
1709
+ if (toolUses.length) wireToolCalls = toolUses.map((block) => ({
1710
+ id: block.id,
1711
+ type: "function",
1712
+ function: {
1713
+ name: block.name,
1714
+ arguments: JSON.stringify(block.input ?? {})
1715
+ }
1716
+ }));
1717
+ }
1718
+ return {
1719
+ choices: [{ message: {
1720
+ content: text,
1721
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
1722
+ } }],
1723
+ usage: {
1724
+ prompt_tokens: response.usage?.input_tokens,
1725
+ completion_tokens: response.usage?.output_tokens,
1726
+ total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
829
1727
  }
830
- } : {}
831
- }, options);
832
- let text;
833
- if (toolName) {
834
- const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
835
- text = toolUse ? JSON.stringify(toolUse.input) : "";
836
- } else text = response.content.find((block) => block.type === "text")?.text ?? "";
837
- return {
838
- choices: [{ message: { content: text } }],
839
- usage: {
840
- prompt_tokens: response.usage?.input_tokens,
841
- completion_tokens: response.usage?.output_tokens,
842
- total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
1728
+ };
1729
+ },
1730
+ async *createStream(params, options) {
1731
+ const { body, toolName } = buildAnthropicRequestBody(params);
1732
+ const stream = await rawMessagesCreate({
1733
+ ...body,
1734
+ stream: true
1735
+ }, options);
1736
+ const blockKinds = new Map();
1737
+ let inputTokens = 0;
1738
+ let sawJsonTool = false;
1739
+ for await (const event of stream) if (event.type === "message_start") inputTokens = event.message.usage?.input_tokens ?? 0;
1740
+ else if (event.type === "content_block_start") if (event.content_block.type === "tool_use") {
1741
+ const kind = event.content_block.name === toolName ? "json-tool" : "tool_use";
1742
+ blockKinds.set(event.index, kind);
1743
+ if (kind === "json-tool") sawJsonTool = true;
1744
+ else if (!toolName) yield {
1745
+ type: "tool_call_delta",
1746
+ index: event.index,
1747
+ id: event.content_block.id,
1748
+ name: event.content_block.name
1749
+ };
1750
+ } else blockKinds.set(event.index, "text");
1751
+ else if (event.type === "content_block_delta") {
1752
+ if (event.delta.type === "text_delta") {
1753
+ if (!toolName) yield {
1754
+ type: "text-delta",
1755
+ delta: event.delta.text
1756
+ };
1757
+ } else if (event.delta.type === "input_json_delta") {
1758
+ const kind = blockKinds.get(event.index);
1759
+ if (kind === "json-tool") yield {
1760
+ type: "text-delta",
1761
+ delta: event.delta.partial_json
1762
+ };
1763
+ else if (!toolName) yield {
1764
+ type: "tool_call_delta",
1765
+ index: event.index,
1766
+ argumentsDelta: event.delta.partial_json
1767
+ };
1768
+ }
1769
+ } else if (event.type === "message_delta") {
1770
+ const outputTokens = event.usage?.output_tokens ?? 0;
1771
+ yield {
1772
+ type: "usage",
1773
+ usage: {
1774
+ prompt_tokens: inputTokens,
1775
+ completion_tokens: outputTokens,
1776
+ total_tokens: inputTokens + outputTokens
1777
+ }
1778
+ };
1779
+ } else if (event.type === "ping") yield { type: "ping" };
1780
+ if (toolName && !sawJsonTool) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
1781
+ }
1782
+ } } };
1783
+ }
1784
+ /**
1785
+ * Anthropic requires strict role alternation, so the per-wire-message
1786
+ * mapping above (one `{role:'user', content:[tool_result]}` per VernLLM
1787
+ * wire tool message) needs merging back together when an assistant turn
1788
+ * requested more than one tool: multiple consecutive user turns would
1789
+ * violate that alternation, and Anthropic's API rejects it outright. This
1790
+ * merges any run of tool-result-only user messages into one, with all
1791
+ * their tool_result blocks combined, the shape Anthropic expects for "here
1792
+ * are the results of everything you just asked for."
1793
+ */
1794
+ function mergeConsecutiveToolResults$1(messages) {
1795
+ const isToolResultOnly = (m) => m.role === "user" && Array.isArray(m.content) && m.content.length > 0 && m.content.every((b) => b.type === "tool_result");
1796
+ const merged = [];
1797
+ for (const m of messages) {
1798
+ const prev = merged.at(-1);
1799
+ if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
1800
+ else merged.push(m);
1801
+ }
1802
+ return merged;
1803
+ }
1804
+ /**
1805
+ * Translates one VernLLM wire message (OpenAI-shaped: plain user/assistant
1806
+ * turns, an assistant turn with `tool_calls`, or a `tool` turn) into
1807
+ * Anthropic's `{ role: 'user' | 'assistant', content }` shape.
1808
+ */
1809
+ function toAnthropicMessage(m) {
1810
+ if (m.role === "tool") return {
1811
+ role: "user",
1812
+ content: [{
1813
+ type: "tool_result",
1814
+ tool_use_id: m.tool_call_id,
1815
+ content: m.content,
1816
+ ...m.is_error ? { is_error: true } : {}
1817
+ }]
1818
+ };
1819
+ if (m.role === "assistant" && m.tool_calls?.length) {
1820
+ const blocks = [];
1821
+ if (m.content) blocks.push({
1822
+ type: "text",
1823
+ text: m.content
1824
+ });
1825
+ for (const tc of m.tool_calls) {
1826
+ let input;
1827
+ try {
1828
+ input = tc.function.arguments.trim() ? JSON.parse(tc.function.arguments) : {};
1829
+ } catch (cause) {
1830
+ throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
843
1831
  }
1832
+ if (input === null || Array.isArray(input) || typeof input !== "object") throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) arguments must be a JSON object.`, "validation");
1833
+ blocks.push({
1834
+ type: "tool_use",
1835
+ id: tc.id,
1836
+ name: tc.function.name,
1837
+ input
1838
+ });
1839
+ }
1840
+ return {
1841
+ role: "assistant",
1842
+ content: blocks
844
1843
  };
845
- } } } };
1844
+ }
1845
+ return {
1846
+ role: m.role,
1847
+ content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content ?? ""
1848
+ };
846
1849
  }
847
1850
 
848
1851
  //#endregion
@@ -859,6 +1862,122 @@ function toGeminiParts(blocks) {
859
1862
  data: block.data
860
1863
  } } : { text: block.text });
861
1864
  }
1865
+ /** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Gemini's `functionCallingConfig`. */
1866
+ function toGeminiToolConfig(toolChoice) {
1867
+ if (!toolChoice || toolChoice === "auto") return { functionCallingConfig: { mode: "AUTO" } };
1868
+ if (toolChoice === "none") return { functionCallingConfig: { mode: "NONE" } };
1869
+ if (toolChoice === "required") return { functionCallingConfig: { mode: "ANY" } };
1870
+ return { functionCallingConfig: {
1871
+ mode: "ANY",
1872
+ allowedFunctionNames: [toolChoice.function.name]
1873
+ } };
1874
+ }
1875
+ /**
1876
+ * Translates one VernLLM wire message into a Gemini `contents` entry.
1877
+ * Gemini has no separate 'tool' role: a prior assistant tool request
1878
+ * becomes a `'model'` turn with `functionCall` parts, and its result
1879
+ * becomes a `'user'` turn with `functionResponse` parts.
1880
+ */
1881
+ function toGeminiContent(m) {
1882
+ if (m.role === "tool") return {
1883
+ role: "user",
1884
+ parts: [{ functionResponse: {
1885
+ name: m.tool_call_id,
1886
+ response: parseToolResult(m.content)
1887
+ } }]
1888
+ };
1889
+ if (m.role === "assistant" && m.tool_calls?.length) {
1890
+ const parts = [];
1891
+ if (typeof m.content === "string" && m.content) parts.push({ text: m.content });
1892
+ parts.push(...m.tool_calls.map((tc) => ({ functionCall: {
1893
+ name: tc.function.name,
1894
+ args: parseToolArguments(tc.function.arguments, tc.function.name)
1895
+ } })));
1896
+ return {
1897
+ role: "model",
1898
+ parts
1899
+ };
1900
+ }
1901
+ return {
1902
+ role: m.role === "assistant" ? "model" : "user",
1903
+ parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content ?? "" }]
1904
+ };
1905
+ }
1906
+ function parseToolArguments(text, toolName) {
1907
+ let parsed;
1908
+ try {
1909
+ parsed = text.trim() ? JSON.parse(text) : {};
1910
+ } catch (cause) {
1911
+ throw new LLMError(`Tool call "${toolName}" arguments are not valid JSON.`, "validation", void 0, void 0, cause);
1912
+ }
1913
+ if (!parsed || Array.isArray(parsed) || typeof parsed !== "object") throw new LLMError(`Tool call "${toolName}" arguments must be a JSON object.`, "validation");
1914
+ return parsed;
1915
+ }
1916
+ function parseToolResult(text) {
1917
+ try {
1918
+ return text.trim() ? JSON.parse(text) : "";
1919
+ } catch {
1920
+ return text;
1921
+ }
1922
+ }
1923
+ /**
1924
+ * Gemini expects the results of everything the model asked for in one turn
1925
+ * to arrive together as multiple `functionResponse` parts on a single
1926
+ * `'user'` entry, not as separate consecutive `'user'` entries. The
1927
+ * per-wire-message mapping above produces one `'user'` entry per VernLLM
1928
+ * wire tool message, so when an assistant turn requested more than one
1929
+ * tool, this merges the resulting run of functionResponse-only `'user'`
1930
+ * entries back into one.
1931
+ */
1932
+ function mergeConsecutiveFunctionResponses(contents) {
1933
+ const isFunctionResponseOnly = (c) => c.role === "user" && c.parts.length > 0 && c.parts.every((p) => "functionResponse" in p);
1934
+ const merged = [];
1935
+ for (const c of contents) {
1936
+ const prev = merged.at(-1);
1937
+ if (isFunctionResponseOnly(c) && prev && isFunctionResponseOnly(prev)) prev.parts.push(...c.parts);
1938
+ else merged.push(c);
1939
+ }
1940
+ return merged;
1941
+ }
1942
+ /**
1943
+ * Builds the Gemini-shaped request from VernLLM's wire params, shared
1944
+ * between `create` and `createStream` so both go through identical
1945
+ * translation (contents shaping, `responseSchema`/`responseMimeType`
1946
+ * mapping, and tool/toolConfig translation all happen exactly once).
1947
+ * `abortSignal` is folded into `config` by the caller (`create`/
1948
+ * `createStream`), once the request options are available.
1949
+ */
1950
+ function buildGeminiRequest(params) {
1951
+ const systemMessage = params.messages.find((m) => m.role === "system");
1952
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
1953
+ const wantsJson = Boolean(params.response_format);
1954
+ const config = {
1955
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
1956
+ maxOutputTokens: params.max_tokens,
1957
+ ...systemMessage ? { systemInstruction: { parts: [{ text: systemMessage.content }] } } : {}
1958
+ };
1959
+ if (wantsJson) config.responseMimeType = "application/json";
1960
+ if (params.response_format?.type === "json_schema") {
1961
+ const { schema, description } = params.response_format.json_schema;
1962
+ config.responseSchema = {
1963
+ ...schema,
1964
+ ...description ? { description } : {}
1965
+ };
1966
+ }
1967
+ if (params.tools?.length) {
1968
+ config.tools = [{ functionDeclarations: params.tools.map((t) => ({
1969
+ name: t.function.name,
1970
+ description: t.function.description,
1971
+ parameters: t.function.parameters
1972
+ })) }];
1973
+ config.toolConfig = toGeminiToolConfig(params.tool_choice);
1974
+ }
1975
+ return {
1976
+ model: params.model,
1977
+ contents: mergeConsecutiveFunctionResponses(conversationMessages.map((m) => toGeminiContent(m))),
1978
+ config
1979
+ };
1980
+ }
862
1981
  /**
863
1982
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
864
1983
  * uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
@@ -869,43 +1988,100 @@ function toGeminiParts(blocks) {
869
1988
  * `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
870
1989
  * models use a token budget, not an effort tier, so it's dropped, same as
871
1990
  * Anthropic.
1991
+ *
1992
+ * `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
1993
+ * `tool_choice` maps to `toolConfig.functionCallingConfig`. `jsonSchema`
1994
+ * and `tools` are mutually exclusive by the time a call reaches here
1995
+ * (enforced in vernLLM.ts), so `responseSchema` and `tools` never
1996
+ * both apply.
1997
+ *
1998
+ * `createStream` calls `generateContentStream` (optional on `GeminiClient`
1999
+ *, required only if the caller sets `stream: true`) and translates each
2000
+ * partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
2001
+ * Gemini's own function-calling API doesn't stream tool-call arguments
2002
+ * incrementally: a `functionCall` part always arrives whole in one chunk,
2003
+ * so each one is emitted as a single, complete `tool_call_delta` (a
2004
+ * one-shot "delta" containing the full arguments) rather than accumulated
2005
+ * fragments, that's a real difference in the underlying API, not
2006
+ * something this adapter can smooth over. `usageMetadata` is (per Gemini's
2007
+ * own behavior) only reliably present on the last chunk, so the `usage`
2008
+ * `WireStreamChunk` is emitted once, after the stream completes, from
2009
+ * whichever chunk's `usageMetadata` was seen last.
872
2010
  */
873
2011
  function fromGemini(geminiClient) {
874
- return { chat: { completions: { async create(params, options) {
875
- const systemMessage = params.messages.find((m) => m.role === "system");
876
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
877
- const wantsJson = Boolean(params.response_format);
878
- const generationConfig = {
879
- temperature: params.temperature,
880
- maxOutputTokens: params.max_tokens
881
- };
882
- if (wantsJson) generationConfig.responseMimeType = "application/json";
883
- if (params.response_format?.type === "json_schema") {
884
- const { schema, description } = params.response_format.json_schema;
885
- generationConfig.responseSchema = {
886
- ...schema,
887
- ...description ? { description } : {}
2012
+ return { chat: { completions: {
2013
+ async create(params, options) {
2014
+ const request = buildGeminiRequest(params);
2015
+ request.config = {
2016
+ ...request.config,
2017
+ abortSignal: options.signal
888
2018
  };
889
- }
890
- const response = await geminiClient.generateContent({
891
- model: params.model,
892
- contents: conversationMessages.map((m) => ({
893
- role: m.role === "assistant" ? "model" : "user",
894
- parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content }]
895
- })),
896
- systemInstruction: systemMessage ? { parts: [{ text: systemMessage.content }] } : void 0,
897
- generationConfig
898
- }, options);
899
- const text = response.candidates?.[0]?.content?.parts?.map((p) => p.text ?? "").join("") ?? "";
900
- return {
901
- choices: [{ message: { content: text } }],
902
- usage: {
903
- prompt_tokens: response.usageMetadata?.promptTokenCount,
904
- completion_tokens: response.usageMetadata?.candidatesTokenCount,
905
- total_tokens: response.usageMetadata?.totalTokenCount
2019
+ const response = await geminiClient.generateContent(request);
2020
+ const parts = response.candidates?.[0]?.content?.parts ?? [];
2021
+ const text = parts.map((p) => p.text ?? "").join("");
2022
+ const functionCalls = parts.filter((p) => p.functionCall);
2023
+ let wireToolCalls;
2024
+ if (functionCalls.length) wireToolCalls = functionCalls.map((p) => ({
2025
+ id: p.functionCall.name,
2026
+ type: "function",
2027
+ function: {
2028
+ name: p.functionCall.name,
2029
+ arguments: JSON.stringify(p.functionCall.args ?? {})
2030
+ }
2031
+ }));
2032
+ return {
2033
+ choices: [{ message: {
2034
+ content: text,
2035
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
2036
+ } }],
2037
+ usage: {
2038
+ prompt_tokens: response.usageMetadata?.promptTokenCount,
2039
+ completion_tokens: response.usageMetadata?.candidatesTokenCount,
2040
+ total_tokens: response.usageMetadata?.totalTokenCount
2041
+ }
2042
+ };
2043
+ },
2044
+ async *createStream(params, options) {
2045
+ if (!geminiClient.generateContentStream) throw new LLMError("stream: true requires a Gemini client with generateContentStream", "validation");
2046
+ const request = buildGeminiRequest(params);
2047
+ request.config = {
2048
+ ...request.config,
2049
+ abortSignal: options.signal
2050
+ };
2051
+ const stream = await geminiClient.generateContentStream(request);
2052
+ let toolCallIndex = 0;
2053
+ let lastUsage;
2054
+ for await (const chunk of stream) {
2055
+ const parts = chunk.candidates?.[0]?.content?.parts ?? [];
2056
+ for (const part of parts) {
2057
+ if (part.text) yield {
2058
+ type: "text-delta",
2059
+ delta: part.text
2060
+ };
2061
+ if (part.functionCall) {
2062
+ yield {
2063
+ type: "tool_call_delta",
2064
+ index: toolCallIndex,
2065
+ id: part.functionCall.name,
2066
+ name: part.functionCall.name,
2067
+ argumentsDelta: JSON.stringify(part.functionCall.args ?? {}),
2068
+ complete: true
2069
+ };
2070
+ toolCallIndex++;
2071
+ }
2072
+ }
2073
+ if (chunk.usageMetadata) lastUsage = chunk.usageMetadata;
906
2074
  }
907
- };
908
- } } } };
2075
+ if (lastUsage) yield {
2076
+ type: "usage",
2077
+ usage: {
2078
+ prompt_tokens: lastUsage.promptTokenCount,
2079
+ completion_tokens: lastUsage.candidatesTokenCount,
2080
+ total_tokens: lastUsage.totalTokenCount
2081
+ }
2082
+ };
2083
+ }
2084
+ } } };
909
2085
  }
910
2086
 
911
2087
  //#endregion
@@ -941,6 +2117,67 @@ function toBedrockContent(blocks) {
941
2117
  } } : { text: block.text });
942
2118
  }
943
2119
  /**
2120
+ * Builds the Converse-shaped request from VernLLM's wire params, shared
2121
+ * between `create` and `createStream` so both go through identical
2122
+ * translation (system prompt, message shaping, the jsonSchema →
2123
+ * forced-single-tool mapping, and the `toolUseSupportedModels` preflight
2124
+ * check all happen exactly once).
2125
+ *
2126
+ * Returns `toolName` alongside the request: when set, the model was forced
2127
+ * to call a single synthetic tool standing in for `jsonSchema` output, and
2128
+ * both `create` and `createStream` need to know this so they can unwrap
2129
+ * that tool call back into plain text content instead of treating it like
2130
+ * a real tool call.
2131
+ */
2132
+ function buildBedrockRequest(params, toolUseSupportedModels) {
2133
+ const systemMessage = params.messages.find((m) => m.role === "system");
2134
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
2135
+ const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
2136
+ const toolName = jsonSchema?.name.trim();
2137
+ if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
2138
+ let jsonInstruction;
2139
+ let toolConfig;
2140
+ if (jsonSchema) {
2141
+ const { schema, description, strict } = jsonSchema;
2142
+ toolConfig = {
2143
+ tools: [{ toolSpec: {
2144
+ name: toolName,
2145
+ description,
2146
+ inputSchema: { json: schema },
2147
+ strict
2148
+ } }],
2149
+ toolChoice: { tool: { name: toolName } }
2150
+ };
2151
+ } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
2152
+ else if (params.tools?.length) toolConfig = {
2153
+ tools: params.tools.map((t) => ({ toolSpec: {
2154
+ name: t.function.name,
2155
+ description: t.function.description,
2156
+ inputSchema: { json: t.function.parameters }
2157
+ } })),
2158
+ toolChoice: toBedrockToolChoice(params.tool_choice)
2159
+ };
2160
+ if (jsonSchema && toolUseSupportedModels) {
2161
+ const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
2162
+ if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
2163
+ }
2164
+ const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
2165
+ const request = {
2166
+ modelId: params.model,
2167
+ messages: mergeConsecutiveToolResults(conversationMessages.map((m) => toBedrockMessage(m))),
2168
+ system: systemParts.length ? systemParts.map((text) => ({ text })) : void 0,
2169
+ inferenceConfig: {
2170
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
2171
+ maxTokens: params.max_tokens
2172
+ },
2173
+ ...toolConfig ? { toolConfig } : {}
2174
+ };
2175
+ return {
2176
+ request,
2177
+ toolName
2178
+ };
2179
+ }
2180
+ /**
944
2181
  * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
945
2182
  * interface VernLLM uses for OpenAI/Groq. The Converse API is unified
946
2183
  * across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
@@ -960,66 +2197,237 @@ function toBedrockContent(blocks) {
960
2197
  * `response_format: json_object` (no schema to build a tool from) and
961
2198
  * `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
962
2199
  * instruction and are dropped respectively.
2200
+ *
2201
+ * `tools` maps to Converse's native `toolConfig`/`toolUse`/`toolResult`;
2202
+ * `tool_choice` maps to `toolConfig.toolChoice`. Mutually exclusive with
2203
+ * `jsonSchema` by the time a call reaches here (enforced in vernLLM.ts).
2204
+ *
2205
+ * `createStream` calls `converseStream` (optional on `BedrockConverseClient`
2206
+ *, required only if the caller sets `stream: true`) and translates its
2207
+ * `contentBlockStart`/`contentBlockDelta`/`metadata` events into
2208
+ * `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
2209
+ * same as `fromAnthropic`'s block-index tracking (Converse's streaming
2210
+ * shape is structurally close to Anthropic's own, both being tool-use-aware
2211
+ * content-block streams), including the same `json-tool` unwrapping: a
2212
+ * `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
2213
+ * `text-delta`, not `tool_call_delta`, so the accumulated result lands in
2214
+ * `finalizeResponse`'s `content` path exactly like the non-streaming
2215
+ * `create` branch above unwraps it.
963
2216
  */
964
2217
  function fromBedrock(bedrockClient, options) {
965
2218
  const toolUseSupportedModels = options?.toolUseSupportedModels;
966
- return { chat: { completions: { async create(params, requestOptions) {
967
- const systemMessage = params.messages.find((m) => m.role === "system");
968
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
969
- const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
970
- const toolName = jsonSchema?.name.trim();
971
- if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
972
- let jsonInstruction;
973
- let toolConfig;
974
- if (jsonSchema) {
975
- const { schema, description, strict } = jsonSchema;
976
- toolConfig = {
977
- tools: [{ toolSpec: {
978
- name: toolName,
979
- description,
980
- inputSchema: { json: schema },
981
- strict
2219
+ return { chat: { completions: {
2220
+ async create(params, requestOptions) {
2221
+ const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
2222
+ const response = await bedrockClient.converse(request, requestOptions);
2223
+ let text;
2224
+ let wireToolCalls;
2225
+ if (toolName) {
2226
+ const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
2227
+ text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
2228
+ } else {
2229
+ const blocks = response.output?.message?.content ?? [];
2230
+ text = blocks.map((c) => c.text ?? "").join("");
2231
+ const toolUses = blocks.filter((block) => Boolean(block.toolUse));
2232
+ if (toolUses.length) wireToolCalls = toolUses.map((block, i) => {
2233
+ const toolUse = block.toolUse;
2234
+ if (!toolUse.name) throw new LLMError(`Bedrock returned a toolUse block without a name at index ${i}.`, "validation");
2235
+ return {
2236
+ id: toolUse.toolUseId ?? `${toolUse.name}_${i}`,
2237
+ type: "function",
2238
+ function: {
2239
+ name: toolUse.name,
2240
+ arguments: JSON.stringify(toolUse.input ?? {})
2241
+ }
2242
+ };
2243
+ });
2244
+ }
2245
+ return {
2246
+ choices: [{ message: {
2247
+ content: text,
2248
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
982
2249
  } }],
983
- toolChoice: { tool: { name: toolName } }
2250
+ usage: {
2251
+ prompt_tokens: response.usage?.inputTokens,
2252
+ completion_tokens: response.usage?.outputTokens,
2253
+ total_tokens: response.usage?.totalTokens
2254
+ }
2255
+ };
2256
+ },
2257
+ async *createStream(params, requestOptions) {
2258
+ if (!bedrockClient.converseStream) throw new LLMError("stream: true requires a Bedrock client with converseStream", "validation");
2259
+ const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
2260
+ const { stream } = await bedrockClient.converseStream(request, requestOptions);
2261
+ const blockKinds = new Map();
2262
+ for await (const event of stream) if ("contentBlockStart" in event) {
2263
+ const { contentBlockIndex, start } = event.contentBlockStart;
2264
+ if (start?.toolUse) {
2265
+ const kind = start.toolUse.name === toolName ? "json-tool" : "tool_use";
2266
+ blockKinds.set(contentBlockIndex, kind);
2267
+ if (kind === "tool_use" && !toolName) yield {
2268
+ type: "tool_call_delta",
2269
+ index: contentBlockIndex,
2270
+ id: start.toolUse.toolUseId,
2271
+ name: start.toolUse.name
2272
+ };
2273
+ } else blockKinds.set(contentBlockIndex, "text");
2274
+ } else if ("contentBlockDelta" in event) {
2275
+ const { contentBlockIndex, delta } = event.contentBlockDelta;
2276
+ if (delta && "text" in delta && delta.text !== void 0 && !toolName) yield {
2277
+ type: "text-delta",
2278
+ delta: delta.text
2279
+ };
2280
+ else if (delta && "toolUse" in delta && delta.toolUse?.input !== void 0) {
2281
+ const kind = blockKinds.get(contentBlockIndex);
2282
+ if (kind === "json-tool") yield {
2283
+ type: "text-delta",
2284
+ delta: delta.toolUse.input
2285
+ };
2286
+ else if (!toolName) yield {
2287
+ type: "tool_call_delta",
2288
+ index: contentBlockIndex,
2289
+ argumentsDelta: delta.toolUse.input
2290
+ };
2291
+ }
2292
+ } else if ("metadata" in event && event.metadata.usage) yield {
2293
+ type: "usage",
2294
+ usage: {
2295
+ prompt_tokens: event.metadata.usage.inputTokens,
2296
+ completion_tokens: event.metadata.usage.outputTokens,
2297
+ total_tokens: event.metadata.usage.totalTokens
2298
+ }
984
2299
  };
985
- } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
986
- if (jsonSchema && toolUseSupportedModels) {
987
- const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
988
- if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
2300
+ else if ("throttlingException" in event) throw new LLMError(event.throttlingException.message ?? "Bedrock throttled the request mid-stream", "api", 429);
2301
+ else if ("validationException" in event) throw new LLMError(event.validationException.message ?? "Bedrock rejected the request mid-stream", "validation");
2302
+ else if ("internalServerException" in event || "serviceUnavailableException" in event || "modelStreamErrorException" in event) {
2303
+ const detail = "internalServerException" in event && event.internalServerException.message || "serviceUnavailableException" in event && event.serviceUnavailableException.message || "modelStreamErrorException" in event && event.modelStreamErrorException.message || "Bedrock reported a mid-stream error";
2304
+ const status = "modelStreamErrorException" in event && event.modelStreamErrorException.originalStatusCode || "serviceUnavailableException" in event && 503 || 500;
2305
+ throw new LLMError(detail, "api", status);
2306
+ }
989
2307
  }
990
- const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
991
- const response = await bedrockClient.converse({
992
- modelId: params.model,
993
- messages: conversationMessages.map((m) => ({
994
- role: m.role,
995
- content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content }]
996
- })),
997
- system: systemParts.length ? systemParts.map((text$1) => ({ text: text$1 })) : void 0,
998
- inferenceConfig: {
999
- temperature: params.temperature,
1000
- maxTokens: params.max_tokens
1001
- },
1002
- ...toolConfig ? { toolConfig } : {}
1003
- }, requestOptions);
1004
- let text;
1005
- if (toolName) {
1006
- const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
1007
- text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
1008
- } else text = response.output?.message?.content?.map((c) => c.text ?? "").join("") ?? "";
1009
- return {
1010
- choices: [{ message: { content: text } }],
1011
- usage: {
1012
- prompt_tokens: response.usage?.inputTokens,
1013
- completion_tokens: response.usage?.outputTokens,
1014
- total_tokens: response.usage?.totalTokens
2308
+ } } };
2309
+ }
2310
+ /** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Converse's `toolChoice`. */
2311
+ function toBedrockToolChoice(toolChoice) {
2312
+ if (!toolChoice || toolChoice === "auto") return { auto: {} };
2313
+ if (toolChoice === "required") return { any: {} };
2314
+ if (toolChoice === "none") throw new LLMError("'none' is not supported by fromBedrock: Bedrock Converse has no `tool_choice` equivalent to forbidding tool use while tools are still offered. Omit `tools` entirely for this call instead.", "validation");
2315
+ return { tool: { name: toolChoice.function.name } };
2316
+ }
2317
+ /**
2318
+ * Translates one VernLLM wire message into Converse's
2319
+ * `{ role: 'user' | 'assistant', content }` shape.
2320
+ */
2321
+ function toBedrockMessage(m) {
2322
+ if (m.role === "tool") return {
2323
+ role: "user",
2324
+ content: [{ toolResult: {
2325
+ toolUseId: m.tool_call_id,
2326
+ content: [{ text: m.content }],
2327
+ status: m.is_error ? "error" : "success"
2328
+ } }]
2329
+ };
2330
+ if (m.role === "assistant" && m.tool_calls?.length) {
2331
+ const blocks = [];
2332
+ if (m.content) blocks.push({ text: m.content });
2333
+ for (const tc of m.tool_calls) {
2334
+ let input;
2335
+ if (!tc.function.arguments.trim()) input = {};
2336
+ else try {
2337
+ input = JSON.parse(tc.function.arguments);
2338
+ } catch (cause) {
2339
+ throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
1015
2340
  }
2341
+ blocks.push({ toolUse: {
2342
+ toolUseId: tc.id,
2343
+ name: tc.function.name,
2344
+ input
2345
+ } });
2346
+ }
2347
+ return {
2348
+ role: "assistant",
2349
+ content: blocks
1016
2350
  };
1017
- } } } };
2351
+ }
2352
+ return {
2353
+ role: m.role,
2354
+ content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content ?? "" }]
2355
+ };
2356
+ }
2357
+ /**
2358
+ * Converse expects the results of everything the model asked for in one
2359
+ * turn to arrive together as multiple `toolResult` content blocks on a
2360
+ * single `'user'` message, not as separate consecutive `'user'` messages.
2361
+ * The per-wire-message mapping above produces one `'user'` message per
2362
+ * VernLLM wire tool message, so when an assistant turn requested more than
2363
+ * one tool, this merges the resulting run of toolResult-only `'user'`
2364
+ * messages back into one.
2365
+ */
2366
+ function mergeConsecutiveToolResults(messages) {
2367
+ const isToolResultOnly = (m) => m.role === "user" && m.content.length > 0 && m.content.every((b) => "toolResult" in b);
2368
+ const merged = [];
2369
+ for (const m of messages) {
2370
+ const prev = merged.at(-1);
2371
+ if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
2372
+ else merged.push(m);
2373
+ }
2374
+ return merged;
1018
2375
  }
1019
2376
 
1020
2377
  //#endregion
1021
2378
  //#region src/adapters/fetch.ts
1022
2379
  /**
2380
+ * Wraps a WHATWG `ReadableStream` (what `response.body` is) so it can be
2381
+ * consumed with `for await`. Implemented via `getReader()` rather than
2382
+ * relying on `ReadableStream` having a native `Symbol.asyncIterator`,
2383
+ * that support varies across runtimes/versions, and this works everywhere
2384
+ * a `ReadableStream` does.
2385
+ */
2386
+ async function* webStreamToAsyncIterable(stream) {
2387
+ const reader = stream.getReader();
2388
+ try {
2389
+ for (;;) {
2390
+ const { done, value } = await reader.read();
2391
+ if (done) return;
2392
+ if (value) yield value;
2393
+ }
2394
+ } finally {
2395
+ try {
2396
+ await reader.cancel();
2397
+ } catch {}
2398
+ reader.releaseLock();
2399
+ }
2400
+ }
2401
+ /** Default `requestStream`: native `fetch`, with the same error/`.status` contract non-streaming errors get. */
2402
+ async function defaultRequestStream(url, init) {
2403
+ const res = await fetch(url, init);
2404
+ if (!res.ok) {
2405
+ const body = await res.text().catch(() => "");
2406
+ const err = new Error(`Fetch adapter stream request failed (${res.status}): ${body.slice(0, 500)}`);
2407
+ err.status = res.status;
2408
+ err.headers = res.headers;
2409
+ throw err;
2410
+ }
2411
+ if (!res.body) throw new Error("Fetch adapter stream request received a response with no body.");
2412
+ return webStreamToAsyncIterable(res.body);
2413
+ }
2414
+ /** Builds the shared `{ method, headers, body? }` request-init for both `create` and `createStream`. */
2415
+ async function buildRequestInit(config, params, requestBody) {
2416
+ const url = typeof config.url === "function" ? config.url(params) : config.url;
2417
+ const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
2418
+ const method = config.method ?? "POST";
2419
+ const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
2420
+ return {
2421
+ url,
2422
+ method,
2423
+ headers: supportsBody ? {
2424
+ "Content-Type": "application/json",
2425
+ ...headers
2426
+ } : { ...headers },
2427
+ ...supportsBody ? { body: JSON.stringify(requestBody) } : {}
2428
+ };
2429
+ }
2430
+ /**
1023
2431
  * A fetch-based escape hatch for providers with no SDK, or where pulling one
1024
2432
  * in isnt worth it. You supply the URL, headers, and two small mapping
1025
2433
  * functions; this handles the HTTP call and slots the result into the same
@@ -1029,41 +2437,99 @@ function fromBedrock(bedrockClient, options) {
1029
2437
  * Non-2xx responses throw an error with `.status` set to the HTTP status
1030
2438
  * code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
1031
2439
  * 401/403) applies here too
2440
+ *
2441
+ * Tool calling works the same way as every other adapter: `mapRequest`
2442
+ * receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
2443
+ * translate them into whatever shape the provider's wire format expects
2444
+ * (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
2445
+ * field). On the way back, `mapResponse` may return a `toolCalls` array
2446
+ * (id/name/JSON-encoded-arguments-string per call) alongside or instead of
2447
+ * `content`; VernLLM parses and (if `argumentsSchema` was set) validates
2448
+ * those arguments the same way it does for every other adapter. For
2449
+ * `stream: true`, tool-call deltas go through the existing
2450
+ * `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
2451
+ * no separate config is needed for streaming vs non-streaming tool calls.
2452
+ *
2453
+
2454
+ * `createStream` requires `mapStreamEvent` (there's no non-streaming
2455
+ * response to fall back on, unlike the other three optional streaming
2456
+ * seams). It opens the request via `requestStream` (defaults to native
2457
+ * `fetch`), splits the raw bytes into individual events via
2458
+ * `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
2459
+ * and translates each event into `WireStreamChunk`(s) via
2460
+ * `mapStreamEvent`. Both seams are overridable per-config for providers
2461
+ * that don't fit the SSE-over-fetch default. If a custom `request`
2462
+ * transport is configured, `requestStream` must be configured too,
2463
+ * `requestStream` never silently falls back to `request` (see
2464
+ * `createStream`'s own comment for why), so a `stream: true` call with
2465
+ * `request` set but no `requestStream` throws a clear
2466
+ * `LLMError('validation')` instead of quietly using unrelated native
2467
+ * `fetch`.
1032
2468
  */
1033
2469
  function fromFetch(config) {
1034
- return { chat: { completions: { async create(params, options) {
1035
- const url = typeof config.url === "function" ? config.url(params) : config.url;
1036
- const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
1037
- const method = config.method ?? "POST";
1038
- const request = config.request ?? fetch;
1039
- const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
1040
- const res = await request(url, {
1041
- method,
1042
- headers: supportsBody ? {
1043
- "Content-Type": "application/json",
1044
- ...headers
1045
- } : { ...headers },
1046
- ...supportsBody ? { body: JSON.stringify(config.mapRequest(params)) } : {},
1047
- signal: options.signal
1048
- });
1049
- if (!res.ok) {
1050
- const body = await res.text().catch(() => "");
1051
- const err = new Error(`Fetch adapter request failed (${res.status}): ${body.slice(0, 500)}`);
1052
- err.status = res.status;
1053
- err.headers = res.headers;
1054
- throw err;
2470
+ return { chat: { completions: {
2471
+ async create(params, options) {
2472
+ const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
2473
+ const request = config.request ?? fetch;
2474
+ const res = await request(url, {
2475
+ method,
2476
+ headers,
2477
+ body,
2478
+ signal: options.signal
2479
+ });
2480
+ if (!res.ok) {
2481
+ const responseBody = await res.text().catch(() => "");
2482
+ const err = new Error(`Fetch adapter request failed (${res.status}): ${responseBody.slice(0, 500)}`);
2483
+ err.status = res.status;
2484
+ err.headers = res.headers;
2485
+ throw err;
2486
+ }
2487
+ const json = await res.json();
2488
+ const { content, usage, toolCalls } = config.mapResponse(json);
2489
+ const wireToolCalls = toolCalls?.length ? toolCalls.map((tc) => ({
2490
+ id: tc.id,
2491
+ type: "function",
2492
+ function: {
2493
+ name: tc.name,
2494
+ arguments: tc.arguments
2495
+ }
2496
+ })) : void 0;
2497
+ return {
2498
+ choices: [{ message: {
2499
+ content,
2500
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
2501
+ } }],
2502
+ usage: usage ? {
2503
+ prompt_tokens: usage.promptTokens,
2504
+ completion_tokens: usage.completionTokens,
2505
+ total_tokens: usage.totalTokens
2506
+ } : void 0
2507
+ };
2508
+ },
2509
+ async *createStream(params, options) {
2510
+ if (!config.mapStreamEvent) throw new LLMError("stream: true requires mapStreamEvent to be configured on fromFetch", "validation");
2511
+ if (config.request && !config.requestStream) throw new LLMError("`stream: true` requires `requestStream` to be configured on fromFetch when a custom `request` transport is set. `requestStream` does not fall back to `request` (it needs an async-iterable byte stream, which `RequestLike`'s buffered `ResponseLike` has no way to provide), without it, `stream: true` would silently use plain native `fetch` instead of your configured transport. Add a `requestStream` that opens the same connection your `request` does, or omit `request` if native `fetch` is fine for both.", "validation");
2512
+ const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
2513
+ const requestStream = config.requestStream ?? defaultRequestStream;
2514
+ const parseFrames = config.parseStreamFrames ?? parseSseStream;
2515
+ const byteStream = await requestStream(url, {
2516
+ method,
2517
+ headers,
2518
+ body,
2519
+ signal: options.signal
2520
+ });
2521
+ for await (const event of parseFrames(byteStream)) {
2522
+ if (event === SSE_PING) {
2523
+ yield { type: "ping" };
2524
+ continue;
2525
+ }
2526
+ const wireChunks = config.mapStreamEvent(event);
2527
+ if (!wireChunks) continue;
2528
+ if (Array.isArray(wireChunks)) yield* wireChunks;
2529
+ else yield wireChunks;
2530
+ }
1055
2531
  }
1056
- const json = await res.json();
1057
- const { content, usage } = config.mapResponse(json);
1058
- return {
1059
- choices: [{ message: { content } }],
1060
- usage: usage ? {
1061
- prompt_tokens: usage.promptTokens,
1062
- completion_tokens: usage.completionTokens,
1063
- total_tokens: usage.totalTokens
1064
- } : void 0
1065
- };
1066
- } } } };
2532
+ } } };
1067
2533
  }
1068
2534
 
1069
2535
  //#endregion
@@ -1086,41 +2552,84 @@ function toOpenAIContent(blocks) {
1086
2552
  });
1087
2553
  }
1088
2554
  /**
1089
- * Adapter for any SDK/client whose `chat.completions.create` already
1090
- * matches the OpenAI wire format: this covers most hosted inference
1091
- * providers, since "OpenAI-compatible" is a de facto standard for chat
1092
- * completion APIs. Almost everything passes straight through untouched,
1093
- * this exists purely so call sites read clearly (`fromMistral(client)` vs
1094
- * handing a Mistral client to something typed for OpenAI) and so a real
1095
- * transformation could be added later, per-provider, without a breaking
1096
- * change.
1097
- *
1098
- * The one thing that isn't a pure passthrough: a `ContentBlock[]`
1099
- * `userContent` is translated into OpenAI's native `image_url` content-part
1100
- * shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
1101
- * rather than a copy of any one provider's wire format.
1102
- *
1103
- * Not every SDKs own TypeScript types line up exactly with `LLMClient`
1104
- * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
1105
- * the actual compatibility contract is the JSON each provider sends and
1106
- * receives over the wire, not the SDKs TS types.
2555
+ * Translates VernLLM's provider-agnostic `messages` (the one part of a
2556
+ * request that isn't a pure passthrough for OpenAI-compatible clients) into
2557
+ * OpenAI's native wire shape. Shared between `create` and `createStream` so
2558
+ * both go through identical message translation.
1107
2559
  */
1108
- function fromOpenAICompatible(client) {
1109
- const raw = client;
1110
- return { chat: { completions: { async create(params, options) {
1111
- const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
2560
+ function toOpenAIMessages(params) {
2561
+ return params.messages.map((m) => {
2562
+ if (m.role === "user" && Array.isArray(m.content)) return {
1112
2563
  ...m,
1113
2564
  content: toOpenAIContent(m.content)
1114
- } : m);
1115
- return raw.chat.completions.create({
1116
- ...params,
1117
- messages
1118
- }, options);
1119
- } } } };
2565
+ };
2566
+ if (m.role === "tool") {
2567
+ const { is_error: _isError,...openAIToolMessage } = m;
2568
+ return openAIToolMessage;
2569
+ }
2570
+ return m;
2571
+ });
2572
+ }
2573
+ /**
2574
+ * Translates one OpenAI-shaped SSE chunk into zero or more `WireStreamChunk`s.
2575
+ * A single chunk can carry a text delta, one or more tool-call argument
2576
+ * deltas (each keyed by `index`, OpenAI's own convention for streaming
2577
+ * parallel tool calls, mirrored directly by VernLLM's `tool_call_delta`
2578
+ * shape so accumulation composes without translation), and/or a final
2579
+ * usage block (present only when `stream_options.include_usage` is set,
2580
+ * which this adapter always sets).
2581
+ */
2582
+ function* toWireStreamChunks(chunk) {
2583
+ const delta = chunk.choices?.[0]?.delta;
2584
+ if (delta?.content) yield {
2585
+ type: "text-delta",
2586
+ delta: delta.content
2587
+ };
2588
+ if (delta?.tool_calls?.length) for (const toolCall of delta.tool_calls) yield {
2589
+ type: "tool_call_delta",
2590
+ index: toolCall.index,
2591
+ id: toolCall.id,
2592
+ name: toolCall.function?.name,
2593
+ argumentsDelta: toolCall.function?.arguments
2594
+ };
2595
+ if (chunk.usage) yield {
2596
+ type: "usage",
2597
+ usage: chunk.usage
2598
+ };
2599
+ }
2600
+ function fromOpenAICompatible(client, options = {}) {
2601
+ const raw = client;
2602
+ const { supportsStreamUsage = true } = options;
2603
+ const rawCreate = raw.chat.completions.create.bind(raw.chat.completions);
2604
+ return { chat: { completions: {
2605
+ async create(params, options$1) {
2606
+ const messages = toOpenAIMessages(params);
2607
+ return raw.chat.completions.create({
2608
+ ...params,
2609
+ messages
2610
+ }, options$1);
2611
+ },
2612
+ async *createStream(params, options$1) {
2613
+ const messages = toOpenAIMessages(params);
2614
+ const stream = await rawCreate({
2615
+ ...params,
2616
+ messages,
2617
+ stream: true,
2618
+ ...supportsStreamUsage ? { stream_options: { include_usage: true } } : {}
2619
+ }, options$1);
2620
+ for await (const chunk of stream) yield* toWireStreamChunks(chunk);
2621
+ }
2622
+ } } };
1120
2623
  }
1121
2624
  /** Groqs SDK matches the OpenAI wire format */
1122
2625
  const fromGroq = fromOpenAICompatible;
1123
- /** Mistrals `chat.completions`-shaped client (or their OpenAI-compat endpoint) */
2626
+ /**
2627
+ * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
2628
+ * endpoint). Mistral supports `stream_options.include_usage` (added after
2629
+ * an earlier period where it returned a 422 for unrecognized fields, per
2630
+ * Mistral's changelog and streaming docs), so this is a plain alias like
2631
+ * the others, `supportsStreamUsage` defaults to `true`.
2632
+ */
1124
2633
  const fromMistral = fromOpenAICompatible;
1125
2634
  /** DeepSeeks API is OpenAI-compatible */
1126
2635
  const fromDeepSeek = fromOpenAICompatible;
@@ -1214,6 +2723,7 @@ exports.ConsoleLogger = ConsoleLogger
1214
2723
  exports.InMemoryCacheAdapter = InMemoryCacheAdapter
1215
2724
  exports.LLMError = LLMError
1216
2725
  exports.NormalizedCacheAdapter = NormalizedCacheAdapter
2726
+ exports.SSE_PING = SSE_PING
1217
2727
  exports.TieredCacheAdapter = TieredCacheAdapter
1218
2728
  exports.VernLLM = VernLLM
1219
2729
  exports.from01AI = from01AI
@@ -1261,4 +2771,6 @@ exports.fromVercelAIGateway = fromVercelAIGateway
1261
2771
  exports.fromXAI = fromXAI
1262
2772
  exports.fromZhipu = fromZhipu
1263
2773
  exports.isLLMError = isLLMError
2774
+ exports.isToolCallResult = isToolCallResult
2775
+ exports.parseSseStream = parseSseStream
1264
2776
  //# sourceMappingURL=index.cjs.map