vern-llm 1.7.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -51,7 +51,7 @@ var CircuitBreaker = class {
51
51
  if (this.state === "closed") return;
52
52
  if (this.state === "open") {
53
53
  const elapsed = Date.now() - this.openedAt;
54
- if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
54
+ if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open, provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
55
55
  this.state = "half-open";
56
56
  this.trialInFlight = true;
57
57
  return;
@@ -84,6 +84,48 @@ var CircuitBreaker = class {
84
84
 
85
85
  //#endregion
86
86
  //#region src/internal/vernLLM.utils.ts
87
+ /** Translates app-facing `ToolDefinition[]` into the OpenAI-shaped wire tools array. */
88
+ function toWireTools(tools) {
89
+ return tools.map((tool) => ({
90
+ type: "function",
91
+ function: {
92
+ name: tool.name,
93
+ description: tool.description,
94
+ parameters: tool.parameters
95
+ }
96
+ }));
97
+ }
98
+ /** Translates app-facing `ToolCall[]` (e.g. from a replayed assistant turn) into wire tool_calls. */
99
+ function toWireToolCalls(toolCalls) {
100
+ return toolCalls.map((tc) => ({
101
+ id: tc.id,
102
+ type: "function",
103
+ function: {
104
+ name: tc.name,
105
+ arguments: JSON.stringify(tc.arguments ?? {})
106
+ }
107
+ }));
108
+ }
109
+ /**
110
+ * Parses the provider's wire-shaped `tool_calls` back into VernLLM's
111
+ * `ToolCall[]`. Malformed argument JSON is a `'parse'` error, same
112
+ * convention as malformed JSON response bodies elsewhere in VernLLM.
113
+ */
114
+ function parseWireToolCalls(wireToolCalls) {
115
+ return wireToolCalls.map((wc) => {
116
+ let parsedArgs;
117
+ try {
118
+ parsedArgs = wc.function.arguments.trim() ? JSON.parse(wc.function.arguments) : {};
119
+ } catch {
120
+ throw new LLMError(`Invalid JSON arguments for tool call "${wc.function.name}"`, "parse");
121
+ }
122
+ return {
123
+ id: wc.id,
124
+ name: wc.function.name,
125
+ arguments: parsedArgs
126
+ };
127
+ });
128
+ }
87
129
  function defaultParseJson(content) {
88
130
  try {
89
131
  return JSON.parse(content);
@@ -94,7 +136,9 @@ function defaultParseJson(content) {
94
136
  /**
95
137
  * Looks inside an unknown error value and pulls out an http status code
96
138
  * if one is present. Checks the status field first then the status code
97
- * field since different client libraries use different names for this.
139
+ * field since different client libraries use different names for this,
140
+ * falling back to AWS SDK v3's `$metadata.httpStatusCode` (e.g. Bedrock's
141
+ * `ThrottlingException`), which doesn't set either of the other two.
98
142
  * Returns undefined when the error is not an object or carries no status
99
143
  */
100
144
  function extractStatus(err) {
@@ -102,6 +146,7 @@ function extractStatus(err) {
102
146
  const error = err;
103
147
  if (typeof error.status === "number") return error.status;
104
148
  if (typeof error.statusCode === "number") return error.statusCode;
149
+ if (typeof error.$metadata?.httpStatusCode === "number") return error.$metadata.httpStatusCode;
105
150
  return void 0;
106
151
  }
107
152
  function formatSafely(value) {
@@ -131,6 +176,21 @@ function describeError(err) {
131
176
  return formatSafely(err);
132
177
  }
133
178
  /**
179
+ * `setTimeout` silently clamps any delay above this (~24.8 days) or
180
+ * `Infinity` down to ~1ms instead of erroring, so a caller passing
181
+ * `Infinity` as "no timeout" gets the opposite of what they asked for.
182
+ * Both timeout helpers below guard against this explicitly.
183
+ */
184
+ const MAX_SETTIMEOUT_MS = 2147483647;
185
+ /** True when a timeout value should be treated as "disabled" rather than passed to `setTimeout`. */
186
+ function isTimeoutDisabled(ms) {
187
+ return !ms || ms <= 0 || ms === Infinity;
188
+ }
189
+ /** Caps a timeout at the largest delay `setTimeout` actually honors. */
190
+ function clampTimeoutMs(ms) {
191
+ return Math.min(ms, MAX_SETTIMEOUT_MS);
192
+ }
193
+ /**
134
194
  * Runs an async function and cancels it if it takes longer than the given
135
195
  * timeout. Creates an internal abort controller that fires after the
136
196
  * timeout elapses, and combines it with any external signal the caller
@@ -140,12 +200,15 @@ function describeError(err) {
140
200
  * continue to propagate as aborted errors. The internal timer is always
141
201
  * cleared afterward, whether the function succeeds, fails, or is aborted,
142
202
  * so nothing is left running in the background.
203
+ *
204
+ * `timeoutMs` of `Infinity` (or any value beyond what `setTimeout` can
205
+ * represent) disables the timeout rather than firing almost immediately.
143
206
  */
144
207
  async function withTimeout(fn, timeoutMs, externalSignal) {
145
208
  const controller = new AbortController();
146
- const timer = setTimeout(() => {
209
+ const timer = isTimeoutDisabled(timeoutMs) ? void 0 : setTimeout(() => {
147
210
  controller.abort();
148
- }, timeoutMs);
211
+ }, clampTimeoutMs(timeoutMs));
149
212
  const signal = externalSignal ? AbortSignal.any([externalSignal, controller.signal]) : controller.signal;
150
213
  try {
151
214
  return await fn(signal);
@@ -157,6 +220,54 @@ async function withTimeout(fn, timeoutMs, externalSignal) {
157
220
  }
158
221
  }
159
222
  /**
223
+ * Races one `iterator.next()` call against a per-call idle timer, to
224
+ * bound the gap *between* chunks (unlike `withTimeout`, which only bounds
225
+ * opening the stream and its first chunk). Without this, a connection
226
+ * that streams one chunk then hangs would never fail.
227
+ *
228
+ * `timeoutMs` of 0/undefined/`Infinity` disables the check. Otherwise
229
+ * rejects with `LLMError('timeout')` if `next()` doesn't settle in time.
230
+ * The clock resets on every call, so the window is measured from the most
231
+ * recent chunk, not from stream start.
232
+ *
233
+ * `onIdle`, if given, is called the moment the timer fires (before the
234
+ * rejection), so callers can abort the underlying transport instead of
235
+ * just walking away from an unread promise. `logger`, if given, records a
236
+ * debug line if `next()` still settles *after* the idle timeout already
237
+ * rejected. `resolve`/`reject` on an already-settled promise is otherwise
238
+ * a silent no-op, so without this the late chunk (possibly the final
239
+ * usage chunk) would vanish with no trace.
240
+ */
241
+ function withChunkIdleTimeout(next, timeoutMs, onIdle, logger) {
242
+ if (isTimeoutDisabled(timeoutMs)) return next();
243
+ const activeTimeoutMs = timeoutMs;
244
+ let settled = false;
245
+ return new Promise((resolve, reject) => {
246
+ const timer = setTimeout(() => {
247
+ settled = true;
248
+ onIdle?.();
249
+ reject(new LLMError(`No stream chunk received for ${activeTimeoutMs}ms (idle timeout)`, "timeout"));
250
+ }, clampTimeoutMs(activeTimeoutMs));
251
+ next().then((result) => {
252
+ clearTimeout(timer);
253
+ if (settled) {
254
+ logger?.debug("[VernLLM] chunk resolved after idle timeout already fired; discarding");
255
+ return;
256
+ }
257
+ settled = true;
258
+ resolve(result);
259
+ }, (error) => {
260
+ clearTimeout(timer);
261
+ if (settled) {
262
+ logger?.debug("[VernLLM] chunk rejection arrived after idle timeout already fired; discarding");
263
+ return;
264
+ }
265
+ settled = true;
266
+ reject(error);
267
+ });
268
+ });
269
+ }
270
+ /**
160
271
  * Default cap (ms) for both exponential backoff and honored Retry-After
161
272
  * values, so a misbehaving/adversarial Retry-After can't stall a caller
162
273
  * indefinitely
@@ -271,6 +382,143 @@ async function withReservedUsage(params, coalesced, getResult, signal, onRefundE
271
382
  }
272
383
  return result;
273
384
  }
385
+ /**
386
+ * Streaming counterpart to `withReservedUsage`. `withReservedUsage` assumes
387
+ * `getResult()` settling *is* the operation's final outcome, awaiting it
388
+ * synchronously before reserve/refund resolve. Streaming can't satisfy that:
389
+ * `call()` must return `{ chunks, finalResult }` as soon as the stream
390
+ * opens, well before the real outcome (validation, schema/tool-call checks)
391
+ * is known.
392
+ *
393
+ * Reserves usage before `openStream` runs, same failure mode as the
394
+ * non-streaming path if `reserveUsage` itself throws (mapped to
395
+ * `quota_exceeded`, nothing opened). If `openStream` itself throws (stream
396
+ * never opened), refunds synchronously and rethrows, exactly like
397
+ * `withReservedUsage` does today. If it succeeds, returns `{ chunks,
398
+ * finalResult }` immediately, refund/report is deferred onto
399
+ * `finalResult`'s continuation, since that's the only point the real
400
+ * outcome is known. This means `onUsageFailure` (and any refund) can fire
401
+ * well after this function itself has returned.
402
+ */
403
+ async function withReservedUsageForStream(params, openStream, signal, onRefundError) {
404
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
405
+ let reserved = false;
406
+ try {
407
+ if (params.reserveUsage) {
408
+ await params.reserveUsage({
409
+ coalesced: false,
410
+ signal
411
+ });
412
+ reserved = true;
413
+ }
414
+ } catch (error) {
415
+ if (signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
416
+ throw new LLMError(error instanceof Error ? error.message : "Usage reservation failed", "quota_exceeded", void 0, void 0, error);
417
+ }
418
+ const refund = async (logMessage) => {
419
+ try {
420
+ await params.refundUsage?.({
421
+ coalesced: false,
422
+ signal
423
+ });
424
+ } catch (refundError) {
425
+ onRefundError(logMessage, refundError);
426
+ }
427
+ };
428
+ let opened;
429
+ try {
430
+ opened = await openStream();
431
+ } catch (error) {
432
+ if (reserved) await refund("[VernLLM] refundUsage failed after stream-open failure");
433
+ throw error;
434
+ }
435
+ const finalResult = opened.finalResult.then((value) => value, async (error) => {
436
+ if (reserved) await refund("[VernLLM] refundUsage failed after stream error");
437
+ throw error;
438
+ });
439
+ finalResult.catch(() => {});
440
+ return {
441
+ chunks: opened.chunks,
442
+ finalResult
443
+ };
444
+ }
445
+ /**
446
+ * Converts an already-known cache value back into a plausible "text" form
447
+ * for a one-shot replay chunk: passed through unchanged if it's already a
448
+ * string (the `jsonMode: false` case), otherwise `JSON.stringify`'d (the
449
+ * `jsonMode: true` case, where the cached value is the *parsed* result, not
450
+ * the original raw text). This is a reasonable reconstruction, not a
451
+ * byte-identical replay of whatever text the model originally streamed,
452
+ * good enough for `for await (const c of chunks)` call sites that don't
453
+ * branch on hit vs. miss, which is the only thing a cache-hit replay needs
454
+ * to support.
455
+ */
456
+ function toReplayText(value) {
457
+ return typeof value === "string" ? value : JSON.stringify(value);
458
+ }
459
+ /**
460
+ * Builds a trivially-exhausted one-shot `chunks` iterable from an
461
+ * already-known value, used for a `cachedCall` cache hit, where there's no
462
+ * live generation to relay (see `VernLLM.cachedCall`'s docs). No `usage`
463
+ * chunk is emitted: a cache hit spent no real tokens, so there's nothing to
464
+ * report, matching how non-streaming `cachedCall` never calls `onUsage` on
465
+ * a hit either.
466
+ *
467
+ * `hasTools` must reflect whether the *original* call that produced this
468
+ * cached value had `tools` set, that's what determines whether `value` is
469
+ * `T` directly or a `CallWithToolsResult<T>` wrapper, and it isn't
470
+ * something that can be reliably guessed from the value's shape alone
471
+ * (a `schema`-validated `T` could coincidentally look like a
472
+ * `CallWithToolsResult`).
473
+ */
474
+ function buildReplayChunks(value, hasTools) {
475
+ const items = [];
476
+ if (hasTools) {
477
+ const result = value;
478
+ if (result.type === "tool_calls") {
479
+ result.toolCalls.forEach((toolCall, index) => {
480
+ items.push({
481
+ type: "tool_call_delta",
482
+ index,
483
+ id: toolCall.id,
484
+ name: toolCall.name,
485
+ argsDelta: JSON.stringify(toolCall.arguments ?? {}),
486
+ complete: true
487
+ });
488
+ });
489
+ if (result.content) items.push({
490
+ type: "text-delta",
491
+ delta: result.content
492
+ });
493
+ } else items.push({
494
+ type: "text-delta",
495
+ delta: toReplayText(result.content)
496
+ });
497
+ } else items.push({
498
+ type: "text-delta",
499
+ delta: toReplayText(value)
500
+ });
501
+ return { async *[Symbol.asyncIterator]() {
502
+ for (const item of items) yield item;
503
+ } };
504
+ }
505
+ /**
506
+ * Streaming counterpart to `buildReplayChunks` for a `cachedCall` that
507
+ * *joined* an already-in-flight call for the same key rather than
508
+ * triggering one itself (see `runCachedStream`'s in-flight-coalescing
509
+ * path): there's no live stream to relay (it isn't this call's stream to
510
+ * relay, see the joiner-path comment in `runCachedStream`), but there's
511
+ * also no value yet, only a pending promise for one. Waits for `promise`,
512
+ * then delegates to `buildReplayChunks`. If `promise` rejects, iterating
513
+ * `chunks` throws that same error, consistent with how a live stream's
514
+ * `chunks` throws on a mid-stream failure.
515
+ */
516
+ function buildReplayChunksFromPromise(promise, hasTools) {
517
+ return { async *[Symbol.asyncIterator]() {
518
+ const value = await promise;
519
+ yield* buildReplayChunks(value, hasTools);
520
+ } };
521
+ }
274
522
 
275
523
  //#endregion
276
524
  //#region src/logger.ts
@@ -403,34 +651,52 @@ var TieredCacheAdapter = class {
403
651
  }
404
652
  };
405
653
 
654
+ //#endregion
655
+ //#region src/types/tools.ts
656
+ /**
657
+ * Runtime-safe check for whether a `call()` result is a `tool_calls`
658
+ * result. Prefer this over relying on TypeScript's static narrowing
659
+ * whenever `params` passed to `call()` wasn't a literal with `tools`
660
+ * inlined (see the "note on the overload" in `VernLLM.call`'s docs), in
661
+ * that case TS may have typed the result as plain `T` even though it's
662
+ * actually a `CallWithToolsResult<T>` at runtime, and this check works
663
+ * either way.
664
+ */
665
+ function isToolCallResult(result) {
666
+ return typeof result === "object" && result !== null && "type" in result && result.type === "tool_calls" && Array.isArray(result.toolCalls);
667
+ }
668
+
406
669
  //#endregion
407
670
  //#region src/vernLLM.ts
408
671
  /**
409
- * A resilient layer around an LLM chat completions client, this is VernLLM!
672
+ * A resilient layer around an LLM chat completions client. This is VernLLM!
410
673
  *
411
- * Adds retry with backoff/jitter, per-attempt timeouts, an optional circuit breaker,
412
- * JSON parsing with optional schema validation, usage tracking, and an
413
- * optional response cache, all configurable, all opt-in beyond sensible
414
- * defaults.
674
+ * Adds retry with backoff and jitter, per-attempt timeouts, an optional
675
+ * circuit breaker, JSON parsing with optional schema validation, usage
676
+ * tracking, and an optional response cache. All configurable, all opt-in
677
+ * beyond sensible defaults.
415
678
  */
416
679
  var VernLLM = class {
417
680
  client;
418
681
  model;
419
682
  maxRetries;
420
683
  timeoutMs;
684
+ chunkIdleTimeoutMs;
421
685
  baseDelayMs;
422
686
  defaultMaxTokens;
687
+ defaultTemperature;
423
688
  cache;
424
689
  nonRetryableStatus;
425
690
  inFlight = new Map();
426
691
  parseJson;
427
692
  onUsage;
693
+ onUsageFailure;
428
694
  logger;
429
695
  breaker;
430
696
  /**
431
- * @param options - Client, model, and tunables. Notable defaults:
432
- * `maxRetries` 1, `timeoutMs` 25000, `baseDelayMs` 500 (exponential backoff
433
- * base), `defaultMaxTokens` 1000, `cache` an in-memory adapter,
697
+ * @param options Client, model, and tunables. Defaults: `maxRetries` 1,
698
+ * `timeoutMs` 25000, `baseDelayMs` 500, `defaultMaxTokens` 1000,
699
+ * `defaultTemperature` 0.2, `cache` an in-memory adapter,
434
700
  * `nonRetryableStatus` `[400, 401, 403, 404, 422]`, `debug` false.
435
701
  */
436
702
  constructor(options) {
@@ -438,8 +704,10 @@ var VernLLM = class {
438
704
  this.model = options.model;
439
705
  this.maxRetries = options.maxRetries ?? 1;
440
706
  this.timeoutMs = options.timeoutMs ?? 25e3;
707
+ this.chunkIdleTimeoutMs = options.chunkIdleTimeoutMs ?? 3e4;
441
708
  this.baseDelayMs = options.baseDelayMs ?? 500;
442
709
  this.defaultMaxTokens = options.defaultMaxTokens ?? 1e3;
710
+ this.defaultTemperature = options.defaultTemperature === void 0 ? .2 : options.defaultTemperature;
443
711
  this.cache = options.cache ?? new InMemoryCacheAdapter();
444
712
  this.nonRetryableStatus = options.nonRetryableStatus ?? [
445
713
  400,
@@ -450,6 +718,7 @@ var VernLLM = class {
450
718
  ];
451
719
  this.parseJson = options.parseJson ?? defaultParseJson;
452
720
  this.onUsage = options.onUsage;
721
+ this.onUsageFailure = options.onUsageFailure;
453
722
  this.logger = options.logger ?? new ConsoleLogger(options.debug ?? false);
454
723
  this.breaker = options.circuitBreaker ? new CircuitBreaker(options.circuitBreaker === true ? void 0 : options.circuitBreaker) : void 0;
455
724
  }
@@ -457,38 +726,307 @@ var VernLLM = class {
457
726
  async resolveCacheKey(key) {
458
727
  return this.cache.resolveKey ? await this.cache.resolveKey(key) : key;
459
728
  }
460
- /**
461
- * Makes a single logical LLM call, retrying on failure per the configured
462
- * policy. Fails fast if the breaker is open or the signal is already
463
- * aborted. On exhausting retries, records a breaker failure and rejects
464
- * with a normalized LLMError.
465
- *
466
- * @param params - System/user content plus per-call overrides (model,
467
- * temperature, jsonMode, schema, signal, etc). See `CallParams`.
468
- * @returns The parsed (and optionally schema-validated) response, or the
469
- * raw string content when `jsonMode` is false and no `jsonSchema` is set.
470
- */
471
729
  async call(params) {
472
730
  this.breaker?.assertClosed();
473
731
  if (params.signal?.aborted) throw new LLMError("LLM request aborted", "aborted");
474
732
  const requestId = params.requestId ?? randomUUID();
733
+ if (params.stream) return withReservedUsageForStream(params, async () => {
734
+ try {
735
+ return await this.retryWithBackoff((attempt) => this.executeStreamCall(params, requestId, attempt), requestId, params.signal);
736
+ } catch (error) {
737
+ const normalized = normalizeError(error, params.signal);
738
+ if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
739
+ this.logger.debug(`[VernLLM:${requestId}] stream-open error:\n${describeError(error)}`);
740
+ throw normalized;
741
+ }
742
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
475
743
  return withReservedUsage(params, false, async () => {
476
744
  try {
477
- return await this.retryWithBackoff(() => this.executeCall(params, requestId), requestId, params.signal);
745
+ return await this.retryWithBackoff((attempt) => this.executeCall(params, requestId, attempt), requestId, params.signal);
478
746
  } catch (error) {
479
747
  const normalized = normalizeError(error, params.signal);
480
748
  if (normalized.type !== "validation" && normalized.type !== "parse" && normalized.type !== "aborted") this.breaker?.recordFailure();
481
- this.logger.debug(`[vern:${requestId}] error:\n${describeError(error)}`);
749
+ this.logger.debug(`[VernLLM:${requestId}] error:\n${describeError(error)}`);
482
750
  throw normalized;
483
751
  }
484
752
  }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
485
753
  }
754
+ /**
755
+ * Performs a single attempt: builds the request (translating `tools` to
756
+ * wire shape when present), dispatches it with a timeout, and shapes the
757
+ * response into `T` or a `CallWithToolsResult<T>` when `params.tools` was
758
+ * set. Throws on an empty response (no text and no tool_calls) so the
759
+ * retry loop treats it like any other transient failure.
760
+ */
761
+ async executeCall(params, requestId, attempt) {
762
+ const { useJson, model, request } = this.buildRequestPayload(params);
763
+ const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
764
+ const usage = this.extractUsage(response, requestId, model);
765
+ const rawContent = response.choices?.[0]?.message?.content;
766
+ const wireToolCalls = response.choices?.[0]?.message?.tool_calls;
767
+ return this.finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt);
768
+ }
769
+ /**
770
+ * Shapes a fully-arrived response (content and/or tool_calls, already
771
+ * extracted from the provider's payload) into `T` or a
772
+ * `CallWithToolsResult<T>`. Reused by the streaming path once it has
773
+ * buffered the full text/tool-call deltas, so there's no separate
774
+ * parsing/validation logic for streaming.
775
+ *
776
+ * Normalizes and reports usage failure on error itself, so every caller
777
+ * gets identical error handling without duplicating it.
778
+ */
779
+ finalizeResponse(rawContent, wireToolCalls, params, useJson, usage, requestId, attempt) {
780
+ try {
781
+ const content = rawContent?.trim();
782
+ if (!content && !wireToolCalls?.length) throw new LLMError("Empty LLM response", "api");
783
+ this.logger.debug(`[VernLLM:${requestId}] output:\n${(content ?? `[${wireToolCalls?.length ?? 0} tool call(s)]`).slice(0, 800)}`);
784
+ if (wireToolCalls?.length) {
785
+ if (!params.tools) throw new LLMError("Provider returned tool_calls but no `tools` were sent with this call.", "api");
786
+ const toolCalls = parseWireToolCalls(wireToolCalls);
787
+ this.validateToolCallArguments(toolCalls, params.tools);
788
+ this.breaker?.recordSuccess();
789
+ this.reportUsage(usage);
790
+ return {
791
+ type: "tool_calls",
792
+ toolCalls,
793
+ ...content ? { content } : {}
794
+ };
795
+ }
796
+ const textContent = content ?? "";
797
+ if (!useJson) {
798
+ this.breaker?.recordSuccess();
799
+ this.reportUsage(usage);
800
+ return params.tools ? {
801
+ type: "content",
802
+ content: textContent
803
+ } : textContent;
804
+ }
805
+ const result = this.parseAndValidate(textContent, params.schema);
806
+ this.breaker?.recordSuccess();
807
+ this.reportUsage(usage);
808
+ return params.tools ? {
809
+ type: "content",
810
+ content: result
811
+ } : result;
812
+ } catch (error) {
813
+ const normalized = normalizeError(error, params.signal);
814
+ if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt);
815
+ throw normalized;
816
+ }
817
+ }
818
+ /**
819
+ * Opens a stream for a single attempt: builds the request exactly like
820
+ * `executeCall`, then requires `createStream` on the client (a clear
821
+ * `validation` error if the adapter doesn't support it). The timeout
822
+ * wraps stream construction and the first `.next()` together, not just
823
+ * construction: calling an `async function*` returns an iterator
824
+ * synchronously without running its body until `.next()` is first
825
+ * invoked, so timing only construction would time an operation that's
826
+ * always instant, not the actual connection. Both are folded into a
827
+ * single `withTimeout` so the same abort signal reaches whatever the
828
+ * adapter's `createStream` uses internally for its first network
829
+ * round-trip.
830
+ *
831
+ * Circuit-breaker success is recorded once the stream fully completes,
832
+ * not on the first chunk arriving, so a connection that opens but then
833
+ * dies mid-stream isn't masked as a success (see `buildStreamResult`).
834
+ */
835
+ async executeStreamCall(params, requestId, attempt) {
836
+ const { useJson, model, request } = this.buildRequestPayload(params);
837
+ const createStream = this.client.chat.completions.createStream;
838
+ if (!createStream) throw new LLMError("stream: true requires a client/adapter with createStream", "validation");
839
+ const streamController = new AbortController();
840
+ const combinedExternal = params.signal ? AbortSignal.any([params.signal, streamController.signal]) : streamController.signal;
841
+ const { iterator, first } = await withTimeout(async (attemptSignal) => {
842
+ const streamIterator = createStream(request, { signal: attemptSignal })[Symbol.asyncIterator]();
843
+ const firstResult = await streamIterator.next();
844
+ return {
845
+ iterator: streamIterator,
846
+ first: firstResult
847
+ };
848
+ }, this.timeoutMs, combinedExternal);
849
+ if (first.done) throw new LLMError("Empty LLM response", "api");
850
+ return this.buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController);
851
+ }
852
+ /**
853
+ * The streaming accumulator: wraps the raw `WireStreamChunk` iterator in
854
+ * an async generator that yields translated `StreamChunk`s to the caller
855
+ * live, as they arrive, with no per-chunk timeout and no bound on total
856
+ * duration, and accumulates text/tool-call deltas internally so that
857
+ * `finalizeResponse` can produce `finalResult` once the stream completes.
858
+ *
859
+ * Two separate try/catches: the iteration loop's catch handles errors
860
+ * the transport itself throws, which aren't normalized yet, so that
861
+ * happens here along with the one `reportUsageFailure` call for them.
862
+ * The second catch, around `finalizeResponse`, does not re-normalize or
863
+ * re-report since `finalizeResponse` already does both internally.
864
+ * Circuit-breaker success is only recorded once the stream fully
865
+ * completes, not when the first chunk arrives, so a connection that
866
+ * opens and then dies mid-way still counts as a failure below instead
867
+ * of masking it.
868
+ */
869
+ buildStreamResult(iterator, first, params, useJson, requestId, model, attempt, streamController) {
870
+ let resolveFinal;
871
+ let rejectFinal;
872
+ const finalResult = new Promise((resolve, reject) => {
873
+ resolveFinal = resolve;
874
+ rejectFinal = reject;
875
+ });
876
+ finalResult.catch(() => {});
877
+ const MAX_BUFFERED_CHUNKS = 1e4;
878
+ const buffered = [];
879
+ const pending = [];
880
+ let streamDone = false;
881
+ let streamError;
882
+ let hasLoggedEviction = false;
883
+ const push = (chunk) => {
884
+ const waiter = pending.shift();
885
+ if (waiter) {
886
+ waiter.resolve({
887
+ done: false,
888
+ value: chunk
889
+ });
890
+ return;
891
+ }
892
+ buffered.push(chunk);
893
+ if (buffered.length > MAX_BUFFERED_CHUNKS * 2) {
894
+ if (!hasLoggedEviction) {
895
+ hasLoggedEviction = true;
896
+ this.logger.debug(`[VernLLM] stream chunk buffer exceeded cap (${MAX_BUFFERED_CHUNKS}), evicting ${buffered.length - MAX_BUFFERED_CHUNKS} oldest chunk(s); buffered=${buffered.length}. The chunks iterable was never read (or fell far behind) for this stream.`);
897
+ }
898
+ buffered.splice(0, buffered.length - MAX_BUFFERED_CHUNKS);
899
+ }
900
+ };
901
+ const finish = () => {
902
+ streamDone = true;
903
+ for (const waiter of pending.splice(0)) waiter.resolve({
904
+ done: true,
905
+ value: void 0
906
+ });
907
+ };
908
+ const fail = (error) => {
909
+ streamDone = true;
910
+ streamError = error;
911
+ for (const waiter of pending.splice(0)) waiter.reject(error);
912
+ };
913
+ const chunks = { [Symbol.asyncIterator]() {
914
+ return { next() {
915
+ if (buffered.length) return Promise.resolve({
916
+ done: false,
917
+ value: buffered.shift()
918
+ });
919
+ if (streamDone) return streamError ? Promise.reject(streamError) : Promise.resolve({
920
+ done: true,
921
+ value: void 0
922
+ });
923
+ return new Promise((resolve, reject) => {
924
+ pending.push({
925
+ resolve,
926
+ reject
927
+ });
928
+ });
929
+ } };
930
+ } };
931
+ const toolCallAcc = new Map();
932
+ let textAcc = "";
933
+ let usage;
934
+ (async () => {
935
+ try {
936
+ let result = first;
937
+ while (!result.done) {
938
+ const wireChunk = result.value;
939
+ if (wireChunk.type === "ping") {} else if (wireChunk.type === "text-delta") {
940
+ textAcc += wireChunk.delta;
941
+ push({
942
+ type: "text-delta",
943
+ delta: wireChunk.delta
944
+ });
945
+ } else if (wireChunk.type === "tool_call_delta") {
946
+ const entry = toolCallAcc.get(wireChunk.index) ?? { args: "" };
947
+ entry.id ??= wireChunk.id;
948
+ entry.name ??= wireChunk.name;
949
+ entry.args += wireChunk.argumentsDelta ?? "";
950
+ toolCallAcc.set(wireChunk.index, entry);
951
+ push({
952
+ type: "tool_call_delta",
953
+ index: wireChunk.index,
954
+ id: wireChunk.id,
955
+ name: wireChunk.name,
956
+ argsDelta: wireChunk.argumentsDelta,
957
+ complete: wireChunk.complete
958
+ });
959
+ } else if (wireChunk.type === "usage") {
960
+ usage = {
961
+ promptTokens: wireChunk.usage.prompt_tokens ?? 0,
962
+ completionTokens: wireChunk.usage.completion_tokens ?? 0,
963
+ totalTokens: wireChunk.usage.total_tokens ?? 0,
964
+ requestId,
965
+ model
966
+ };
967
+ push({
968
+ type: "usage",
969
+ usage
970
+ });
971
+ }
972
+ result = await withChunkIdleTimeout(() => iterator.next(), params.chunkIdleTimeoutMs ?? this.chunkIdleTimeoutMs, () => streamController.abort(), this.logger);
973
+ }
974
+ } catch (error) {
975
+ try {
976
+ await iterator.return?.();
977
+ } catch {}
978
+ streamController.abort();
979
+ const normalized = normalizeError(error, params.signal);
980
+ if (normalized.type === "timeout") this.breaker?.recordFailure();
981
+ if (usage && normalized.type !== "aborted") this.reportUsageFailure(usage, normalized, attempt, true);
982
+ fail(normalized);
983
+ rejectFinal(normalized);
984
+ return;
985
+ }
986
+ finish();
987
+ this.breaker?.recordSuccess();
988
+ try {
989
+ const wireToolCalls = toolCallAcc.size ? [...toolCallAcc.entries()].sort(([indexA], [indexB]) => indexA - indexB).map(([, entry]) => ({
990
+ id: entry.id ?? "",
991
+ type: "function",
992
+ function: {
993
+ name: entry.name ?? "",
994
+ arguments: entry.args
995
+ }
996
+ })) : void 0;
997
+ const finalized = this.finalizeResponse(textAcc, wireToolCalls, params, useJson, usage, requestId, attempt);
998
+ resolveFinal(finalized);
999
+ } catch (error) {
1000
+ rejectFinal(error);
1001
+ }
1002
+ })();
1003
+ return {
1004
+ chunks,
1005
+ finalResult
1006
+ };
1007
+ }
1008
+ /**
1009
+ * Checks every `ToolCall` against the `tools` that were offered, catching
1010
+ * a hallucinated tool name early instead of letting it reach the
1011
+ * application's dispatch table. Then runs each tool's `argumentsSchema`,
1012
+ * if present, throwing `LLMError('validation')` on failure.
1013
+ */
1014
+ validateToolCallArguments(toolCalls, tools) {
1015
+ const knownNames = new Set(tools.map((t) => t.name));
1016
+ for (const call of toolCalls) {
1017
+ if (!knownNames.has(call.name)) throw new LLMError(`Model requested tool "${call.name}", which was not in the tools offered ([${[...knownNames].join(", ")}]).`, "api");
1018
+ const definition = tools.find((t) => t.name === call.name);
1019
+ if (!definition?.argumentsSchema) continue;
1020
+ const result = definition.argumentsSchema.safeParse(call.arguments);
1021
+ if (!result.success) throw new LLMError(`Arguments for tool call "${call.name}" failed validation`, "validation", void 0, result.error);
1022
+ }
1023
+ }
486
1024
  /** Runs `fn`, retrying with backoff according to `shouldRetry`. */
487
1025
  async retryWithBackoff(fn, requestId, signal) {
488
1026
  let lastError;
489
1027
  for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
490
1028
  if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
491
- return await fn();
1029
+ return await fn(attempt);
492
1030
  } catch (error) {
493
1031
  lastError = error;
494
1032
  if (!this.shouldRetry(error, signal)) break;
@@ -496,60 +1034,73 @@ var VernLLM = class {
496
1034
  throw lastError;
497
1035
  }
498
1036
  /**
499
- * Performs a single attempt: builds the request, dispatches it with a
500
- * timeout, and shapes the response. Throws on an empty response so the
501
- * retry loop treats it like any other transient failure.
502
- */
503
- async executeCall(params, requestId) {
504
- const { useJson, model, request } = this.buildRequestPayload(params);
505
- const response = await withTimeout((attemptSignal) => this.client.chat.completions.create(request, { signal: attemptSignal }), this.timeoutMs, params.signal);
506
- const content = response.choices?.[0]?.message?.content?.trim();
507
- if (!content) throw new LLMError("Empty LLM response", "api");
508
- this.logger.debug(`[vern:${requestId}] output:\n${content.slice(0, 800)}`);
509
- this.recordUsage(response, requestId, model);
510
- if (!useJson) {
511
- this.breaker?.recordSuccess();
512
- return content;
513
- }
514
- const result = this.parseAndValidate(content, params.schema);
515
- this.breaker?.recordSuccess();
516
- return result;
517
- }
518
- /**
519
1037
  * Validates `history` alternates user/assistant turns, since providers
520
1038
  * like Anthropic/Gemini reject or mishandle consecutive same-role turns.
521
1039
  */
522
1040
  validateHistory(history) {
523
- let previousRole;
1041
+ let previousTurn;
524
1042
  for (const [index, turn] of history.entries()) {
525
- if (turn.role !== "user" && turn.role !== "assistant") throw new LLMError(`Invalid history[${index}].role "${turn.role}": must be "user" or "assistant"`, "validation");
526
- if (turn.role === previousRole) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
527
- previousRole = turn.role;
1043
+ if (turn.role === "tool") {
1044
+ if (previousTurn?.role !== "assistant" || !previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] is a "tool" turn, but must immediately follow an "assistant" turn that requested tools`, "validation");
1045
+ if (!turn.toolResults?.length) throw new LLMError(`history[${index}] is a "tool" turn but has no toolResults`, "validation");
1046
+ const requestedIds = new Set(previousTurn.toolCalls.map((tc) => tc.id));
1047
+ const resultIds = turn.toolResults.map((tr) => tr.toolCallId);
1048
+ const unknownIds = resultIds.filter((id) => !requestedIds.has(id));
1049
+ if (unknownIds.length) throw new LLMError(`history[${index}].toolResults references unknown toolCallId(s) [${unknownIds.join(", ")}]`, "validation");
1050
+ const seenIds = new Set();
1051
+ const duplicateIds = new Set();
1052
+ for (const id of resultIds) {
1053
+ if (seenIds.has(id)) duplicateIds.add(id);
1054
+ seenIds.add(id);
1055
+ }
1056
+ if (duplicateIds.size) throw new LLMError(`history[${index}].toolResults has duplicate toolCallId(s) [${[...duplicateIds].join(", ")}]`, "validation");
1057
+ const missingIds = [...requestedIds].filter((id) => !resultIds.includes(id));
1058
+ if (missingIds.length) throw new LLMError(`history[${index}] is missing toolResults for toolCallId(s) [${missingIds.join(", ")}]`, "validation");
1059
+ } else {
1060
+ if (turn.role === previousTurn?.role) throw new LLMError(`history must alternate user/assistant turns: consecutive "${turn.role}" turns at history[${index - 1}] and history[${index}]`, "validation");
1061
+ if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError(`history[${index}] follows an assistant tool request without tool results`, "validation");
1062
+ }
1063
+ previousTurn = turn;
528
1064
  }
529
- if (previousRole === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn. history must end with an \"assistant\" turn (or be empty).", "validation");
1065
+ if (previousTurn?.role === "assistant" && previousTurn.toolCalls?.length) throw new LLMError("The last entry in history is an assistant tool request without tool results", "validation");
1066
+ if (previousTurn?.role === "user") throw new LLMError("The last entry in history is a \"user\" turn, which would collide with the current userContent turn.", "validation");
530
1067
  }
531
1068
  /** Applies per-call defaults and shapes params into the client's request object. */
532
1069
  buildRequestPayload(params) {
533
- const { systemPrompt, userContent, history = [], temperature = .2, jsonMode = true, maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema } = params;
1070
+ const { systemPrompt, userContent, history = [], maxTokens = this.defaultMaxTokens, model = this.model, reasoningEffort, jsonSchema, tools, toolChoice } = params;
1071
+ const temperature = params.temperature === void 0 ? this.defaultTemperature : params.temperature;
1072
+ if (tools && (jsonSchema || params.schema)) throw new LLMError("`tools` cannot be combined with `jsonSchema`/`schema`: on Anthropic and Bedrock, jsonSchema is implemented internally as a forced single-tool call, which would collide with real tools. Use one or the other.", "validation");
1073
+ if (tools && tools.length === 0) throw new LLMError("`tools` was an empty array. This is almost always a bug (e.g. a filtered tool list that ended up empty). An empty `tools` array still switches on tool-call mode (response shape, jsonMode default, wire format) with nothing for the model to call. Omit `tools` entirely for a normal call, or make sure the array is non-empty.", "validation");
1074
+ if (tools) {
1075
+ const seen = new Set();
1076
+ const duplicates = new Set();
1077
+ for (const tool of tools) {
1078
+ if (seen.has(tool.name)) duplicates.add(tool.name);
1079
+ seen.add(tool.name);
1080
+ }
1081
+ if (duplicates.size) throw new LLMError(`\`tools\` has duplicate name(s): [${[...duplicates].join(", ")}]. Tool names must be unique.`, "validation");
1082
+ }
1083
+ if (toolChoice && !tools) throw new LLMError("`toolChoice` was set without `tools`. There is nothing for it to choose between. Set `tools`, or remove `toolChoice`.", "validation");
1084
+ if (tools && typeof toolChoice === "object" && !tools.some((t) => t.name === toolChoice.name)) throw new LLMError(`toolChoice names "${toolChoice.name}", which is not in \`tools\` ([${tools.map((t) => t.name).join(", ")}]).`, "validation");
1085
+ const jsonMode = params.jsonMode ?? (tools ? false : true);
534
1086
  const useJson = jsonMode || Boolean(jsonSchema);
535
1087
  if (params.schema && !useJson) throw new LLMError("schema was provided but jsonMode: false disables JSON parsing, so nothing would validate it. Remove jsonMode: false, set jsonSchema, or remove schema.", "validation");
536
1088
  const responseFormat = this.buildResponseFormat(jsonSchema, useJson);
537
1089
  this.validateHistory(history);
538
1090
  const request = {
539
1091
  model,
540
- temperature,
1092
+ ...temperature !== null ? { temperature } : {},
541
1093
  max_tokens: maxTokens,
542
1094
  ...responseFormat ? { response_format: responseFormat } : {},
543
1095
  ...reasoningEffort ? { reasoning_effort: reasoningEffort } : {},
1096
+ ...tools ? { tools: toWireTools(tools) } : {},
1097
+ ...tools ? { tool_choice: this.buildWireToolChoice(toolChoice) } : {},
544
1098
  messages: [
545
1099
  ...systemPrompt ? [{
546
1100
  role: "system",
547
1101
  content: systemPrompt
548
1102
  }] : [],
549
- ...history.map((turn) => ({
550
- role: turn.role,
551
- content: turn.content
552
- })),
1103
+ ...history.flatMap((turn) => this.turnToWireMessages(turn)),
553
1104
  {
554
1105
  role: "user",
555
1106
  content: userContent
@@ -562,6 +1113,39 @@ var VernLLM = class {
562
1113
  request
563
1114
  };
564
1115
  }
1116
+ /** Maps VernLLM's app-facing `ToolChoice` onto the OpenAI-shaped wire `tool_choice`. */
1117
+ buildWireToolChoice(toolChoice) {
1118
+ if (!toolChoice || toolChoice === "auto") return "auto";
1119
+ if (toolChoice === "none" || toolChoice === "required") return toolChoice;
1120
+ return {
1121
+ type: "function",
1122
+ function: { name: toolChoice.name }
1123
+ };
1124
+ }
1125
+ /**
1126
+ * Expands one `ConversationTurn` into one or more wire messages. Plain
1127
+ * user/assistant turns map 1:1. An assistant turn with `toolCalls` maps
1128
+ * to an assistant message carrying wire-shaped `tool_calls`. A `'tool'`
1129
+ * turn expands into one wire `tool` message per `toolResult`, since
1130
+ * OpenAI-shaped wire format wants one message per tool_call_id.
1131
+ */
1132
+ turnToWireMessages(turn) {
1133
+ if (turn.role === "tool") return (turn.toolResults ?? []).map((tr) => ({
1134
+ role: "tool",
1135
+ tool_call_id: tr.toolCallId,
1136
+ content: typeof tr.content === "string" ? tr.content : JSON.stringify(tr.content ?? null),
1137
+ ...tr.isError ? { is_error: true } : {}
1138
+ }));
1139
+ if (turn.role === "assistant" && turn.toolCalls?.length) return [{
1140
+ role: "assistant",
1141
+ ...turn.content ? { content: turn.content } : {},
1142
+ tool_calls: toWireToolCalls(turn.toolCalls)
1143
+ }];
1144
+ return [{
1145
+ role: turn.role,
1146
+ content: turn.content ?? ""
1147
+ }];
1148
+ }
565
1149
  /**
566
1150
  * Chooses the response format: a provider-native `jsonSchema` takes
567
1151
  * priority when supplied (constrains generation directly), otherwise
@@ -580,21 +1164,49 @@ var VernLLM = class {
580
1164
  };
581
1165
  return useJson ? { type: "json_object" } : void 0;
582
1166
  }
583
- /** Reports token usage to `onUsage`, swallowing and logging any error it throws. */
584
- recordUsage(response, requestId, model) {
585
- if (!response.usage || !this.onUsage) return;
1167
+ /**
1168
+ * Pulls `TokenUsage` out of a raw response, if the provider reported it.
1169
+ * Extraction doesn't depend on what happens to the response afterward, so
1170
+ * a malformed body can still yield usage if the provider's usage block
1171
+ * itself came through intact.
1172
+ */
1173
+ extractUsage(response, requestId, model) {
1174
+ if (!response.usage) return void 0;
1175
+ return {
1176
+ promptTokens: response.usage.prompt_tokens ?? 0,
1177
+ completionTokens: response.usage.completion_tokens ?? 0,
1178
+ totalTokens: response.usage.total_tokens ?? 0,
1179
+ requestId,
1180
+ model
1181
+ };
1182
+ }
1183
+ /** Reports token usage for a successful call, swallowing and logging any error `onUsage` throws. */
1184
+ reportUsage(usage) {
1185
+ if (!usage || !this.onUsage) return;
586
1186
  try {
587
- this.onUsage({
588
- promptTokens: response.usage.prompt_tokens ?? 0,
589
- completionTokens: response.usage.completion_tokens ?? 0,
590
- totalTokens: response.usage.total_tokens ?? 0,
591
- requestId,
592
- model
593
- });
1187
+ this.onUsage(usage);
594
1188
  } catch (error) {
595
1189
  this.logger.error("[VernLLM] onUsage failed", { message: error instanceof Error ? error.message : "unknown" });
596
1190
  }
597
1191
  }
1192
+ /**
1193
+ * Reports token usage spent on an attempt that then failed, so it isn't
1194
+ * dropped alongside the error. Covers any error thrown after usage
1195
+ * extraction, since all of them happen only after a response (real
1196
+ * spend) already arrived. Swallows and logs any error `onUsageFailure`
1197
+ * itself throws.
1198
+ */
1199
+ reportUsageFailure(usage, error, attempt, terminal = false) {
1200
+ const displayTokens = usage.totalTokens || usage.promptTokens + usage.completionTokens;
1201
+ const attemptText = terminal ? "mid-stream failure (terminal, no further attempts)" : `attempt ${attempt + 1}/${this.maxRetries + 1}`;
1202
+ this.logger.warn(`[VernLLM:${usage.requestId}] usage failure, ${attemptText}: type=${error.type} tokens=${displayTokens}`);
1203
+ if (!this.onUsageFailure) return;
1204
+ try {
1205
+ this.onUsageFailure(usage, error);
1206
+ } catch (hookError) {
1207
+ this.logger.error("[VernLLM] onUsageFailure failed", { message: hookError instanceof Error ? hookError.message : "unknown" });
1208
+ }
1209
+ }
598
1210
  /** Parses response content as JSON and validates it against `schema` when supplied. */
599
1211
  parseAndValidate(content, schema) {
600
1212
  let parsed;
@@ -618,7 +1230,7 @@ var VernLLM = class {
618
1230
  async recoverDelay(requestId, attempt, error, signal) {
619
1231
  const retryAfterMs = extractRetryAfterMs(error);
620
1232
  const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
621
- this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
1233
+ this.logger.warn(`[VernLLM:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
622
1234
  await waitForRetry(delay, signal);
623
1235
  }
624
1236
  /** Decides whether a failed attempt is worth retrying. */
@@ -633,7 +1245,7 @@ var VernLLM = class {
633
1245
  * supports deletion. Cache invalidation is the caller's responsibility;
634
1246
  * only the application knows when cached data is stale.
635
1247
  *
636
- * @param key - The raw cache key (resolved through the adapter's
1248
+ * @param key The raw cache key (resolved through the adapter's
637
1249
  * `resolveKey`, if any, before deletion).
638
1250
  */
639
1251
  async deleteCache(key) {
@@ -641,15 +1253,20 @@ var VernLLM = class {
641
1253
  await this.cache.delete(await this.resolveCacheKey(key));
642
1254
  }
643
1255
  /**
644
- * Cache wrapper around caller-supplied logic. Concurrent misses for the
645
- * same `cacheKey` share a single in-flight call, avoiding cache stampedes.
1256
+ * Internal cache primitive around caller-supplied logic. Concurrent misses
1257
+ * for the same `cacheKey` share a single in-flight call, avoiding cache
1258
+ * stampedes.
646
1259
  *
647
- * @param params - `cacheKey`, `ttl`, `fn` (the work to run on a cache
1260
+ * Not part of the public API. Backs the public `cachedCall()`, which
1261
+ * always composes this with `call()` so cached results get the same
1262
+ * retry/timeout/circuit-breaker guarantees as any other LLM call.
1263
+ *
1264
+ * @param params `cacheKey`, `ttl`, `fn` (the work to run on a cache
648
1265
  * miss, typically `() => this.call(...)`), and optional
649
- * `reserveUsage`/`refundUsage`/`signal`. See `CachedCallParams`.
1266
+ * `reserveUsage`/`refundUsage`/`signal`. See `InternalCacheParams`.
650
1267
  * @returns The cached value on a hit, or the result of `fn()` on a miss.
651
1268
  */
652
- async cachedCall(params) {
1269
+ async runCached(params) {
653
1270
  const resolvedKey = await this.resolveCacheKey(params.cacheKey);
654
1271
  const resolvedParams = resolvedKey === params.cacheKey ? params : {
655
1272
  ...params,
@@ -680,24 +1297,113 @@ var VernLLM = class {
680
1297
  }
681
1298
  return result;
682
1299
  }
683
- /** Logs a failed refundUsage attempt via the configured logger. */
684
- logRefundError(logMessage, error) {
685
- this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
1300
+ /**
1301
+ * Streaming counterpart to `runCached`. Three cases:
1302
+ *
1303
+ * - Hit: no live generation to relay. Returns immediately with
1304
+ * `finalResult` resolved to the cached value and a one-shot `chunks`
1305
+ * replay built from it, so `for await (const c of chunks)` call sites
1306
+ * work identically on a hit or a miss. No usage hooks fire, since
1307
+ * nothing was actually spent.
1308
+ * - Miss, nothing else in flight for this key: delegates to
1309
+ * `registerStreamTrigger`, which opens the stream and relays its
1310
+ * `chunks` live.
1311
+ * - Miss, but another call for the same key is already in flight: this
1312
+ * call has no live chunks of its own to relay, so it's treated like a
1313
+ * delayed hit. `finalResult` shares the trigger's in-flight promise
1314
+ * (the same `this.inFlight` map non-streaming `runCached` uses, so
1315
+ * streaming and non-streaming `cachedCall`s for the same key coalesce
1316
+ * against each other too), and `chunks` is a one-shot replay built
1317
+ * once that promise resolves.
1318
+ */
1319
+ async runCachedStream(params, hasTools) {
1320
+ const resolvedKey = await this.resolveCacheKey(params.cacheKey);
1321
+ const resolvedParams = resolvedKey === params.cacheKey ? params : {
1322
+ ...params,
1323
+ cacheKey: resolvedKey
1324
+ };
1325
+ const cached = await this.cache.get(resolvedKey);
1326
+ if (cached.hit) {
1327
+ const value = cached.value;
1328
+ return {
1329
+ chunks: buildReplayChunks(value, hasTools),
1330
+ finalResult: Promise.resolve(value)
1331
+ };
1332
+ }
1333
+ const existing = this.inFlight.get(resolvedKey);
1334
+ if (existing) {
1335
+ const finalResult = withReservedUsage(resolvedParams, true, () => existing, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
1336
+ return {
1337
+ chunks: buildReplayChunksFromPromise(finalResult, hasTools),
1338
+ finalResult
1339
+ };
1340
+ }
1341
+ return this.registerStreamTrigger(resolvedParams);
686
1342
  }
687
1343
  /**
688
- * Convenience wrapper composing `call` + `cachedCall`, so cached LLM calls
689
- * automatically get retry/timeout/circuit-breaker behavior. `reserveUsage`/
690
- * `refundUsage` are read from the top-level params only.
1344
+ * Opens the shared stream for a cache miss and tracks its settled value
1345
+ * in `this.inFlight` until it resolves or rejects. Writes to the cache
1346
+ * on success only, matching `runAndCache`.
691
1347
  *
692
- * @param params - `cachedCall` params (`cacheKey`, `ttl`, etc, minus `fn`)
693
- * plus `call`, the `CallParams` to pass through to `this.call(...)`.
694
- * @returns The cached value on a hit, or the freshly-called result on a miss.
1348
+ * Registers the in-flight promise synchronously, before anything async
1349
+ * runs, so a concurrent `cachedCall` for the same key always sees it in
1350
+ * time to join instead of triggering its own stream. Settlement is
1351
+ * wired onto the whole `withReservedUsageForStream` call rather than a
1352
+ * line inside its callback, so any failure point (reserving usage,
1353
+ * opening the stream, or the stream itself) reliably settles the
1354
+ * in-flight entry instead of leaving it stuck.
695
1355
  */
696
- async cachedLLMCall(params) {
1356
+ registerStreamTrigger(params) {
1357
+ let resolveInFlight;
1358
+ let rejectInFlight;
1359
+ const inFlightResult = new Promise((resolve, reject) => {
1360
+ resolveInFlight = resolve;
1361
+ rejectInFlight = reject;
1362
+ });
1363
+ this.inFlight.set(params.cacheKey, inFlightResult);
1364
+ inFlightResult.catch(() => {}).finally(() => {
1365
+ this.inFlight.delete(params.cacheKey);
1366
+ });
1367
+ const streamPromise = withReservedUsageForStream(params, async () => {
1368
+ const opened = await params.openStream();
1369
+ const trackedResult = opened.finalResult.then(async (value) => {
1370
+ try {
1371
+ await this.cache.set(params.cacheKey, value, params.ttl);
1372
+ } catch (error) {
1373
+ this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
1374
+ }
1375
+ return value;
1376
+ }, (error) => {
1377
+ throw error;
1378
+ });
1379
+ return {
1380
+ chunks: opened.chunks,
1381
+ finalResult: trackedResult
1382
+ };
1383
+ }, params.signal, (logMessage, error) => this.logRefundError(logMessage, error));
1384
+ streamPromise.then((opened) => {
1385
+ opened.finalResult.then(resolveInFlight, rejectInFlight);
1386
+ }, (error) => {
1387
+ rejectInFlight(error);
1388
+ });
1389
+ return streamPromise;
1390
+ }
1391
+ /** Logs a failed refundUsage attempt via the configured logger. */
1392
+ logRefundError(logMessage, error) {
1393
+ this.logger.error(logMessage, { message: error instanceof Error ? error.message : "unknown" });
1394
+ }
1395
+ async cachedCall(params) {
697
1396
  const { call: callParams,...cacheParams } = params;
698
- const { reserveUsage: innerReserveUsage, refundUsage: innerRefundUsage,...restCallParams } = callParams;
699
- if (innerReserveUsage || innerRefundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedLLMCall; set them at the top level instead.");
700
- return this.cachedCall({
1397
+ const { reserveUsage, refundUsage,...restCallParams } = callParams;
1398
+ if (reserveUsage || refundUsage) this.logger.warn("[VernLLM] reserveUsage/refundUsage on `call` are ignored by cachedCall; set them at the top level instead.");
1399
+ if (restCallParams.stream) {
1400
+ const streamParams = restCallParams;
1401
+ return this.runCachedStream({
1402
+ ...cacheParams,
1403
+ openStream: () => this.call(streamParams)
1404
+ }, Boolean(restCallParams.tools));
1405
+ }
1406
+ return this.runCached({
701
1407
  ...cacheParams,
702
1408
  fn: () => this.call(restCallParams)
703
1409
  });
@@ -711,6 +1417,108 @@ var VernLLM = class {
711
1417
  }
712
1418
  };
713
1419
 
1420
+ //#endregion
1421
+ //#region src/internal/sse.ts
1422
+ /**
1423
+ * Parses a Server-Sent-Events byte/text stream into the JSON payload of
1424
+ * each `data:` frame, in arrival order. Generic over transport: works with
1425
+ * anything that hands back progressively-arriving `Uint8Array` or `string`
1426
+ * chunks via async iteration: native `fetch`'s `response.body` (wrapped
1427
+ * to be iterable, see `webStreamToAsyncIterable` in `fetch.ts`), axios's
1428
+ * Node `Readable` (already async-iterable, no wrapping needed), etc, so
1429
+ * this framing layer doesn't care which transport produced the bytes.
1430
+ *
1431
+ * Follows the SSE spec's frame-delimiting rules closely enough for LLM
1432
+ * streaming responses: frames are separated by a blank line, each frame
1433
+ * may carry one or more `data:` lines (joined with `\n` per spec when
1434
+ * there's more than one), `:`-prefixed lines are comments and ignored, and
1435
+ * other SSE fields (`event:`, `id:`, `retry:`) are ignored since VernLLM
1436
+ * only needs the payload. A frame whose data is exactly `[DONE]` (the
1437
+ * sentinel several providers, notably OpenAI, send to mark stream end)
1438
+ * ends iteration without yielding it.
1439
+ *
1440
+ * Line endings: `\r\n` and bare `\r` (both legal per the SSE spec, alongside `\n`) are normalized
1441
+ * to `\n` before frame splitting. A `\r` at the very end of the currently-buffered text is left
1442
+ * alone until either more text arrives (in case it's the first half of a split `\r\n` pair) or the
1443
+ * stream ends, so a `\r\n` pair split across two transport chunks is never misread as two blank
1444
+ * lines.
1445
+ *
1446
+ * Malformed JSON in a frame throws `LLMError('parse')`, consistent with
1447
+ * how malformed JSON is handled elsewhere in VernLLM.
1448
+ */
1449
+ async function* parseSseStream(source) {
1450
+ const decoder = new TextDecoder("utf-8", { fatal: true });
1451
+ let buffer = "";
1452
+ for await (const chunk of source) {
1453
+ let text;
1454
+ try {
1455
+ text = typeof chunk === "string" ? chunk : decoder.decode(chunk, { stream: true });
1456
+ } catch (cause) {
1457
+ throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
1458
+ }
1459
+ buffer = (buffer + text).replace(/\r\n/g, "\n").replace(/\r(?!$)/g, "\n");
1460
+ let boundary$1 = buffer.indexOf("\n\n");
1461
+ while (boundary$1 !== -1) {
1462
+ const frame = buffer.slice(0, boundary$1);
1463
+ buffer = buffer.slice(boundary$1 + 2);
1464
+ const event = parseSseFrame(frame);
1465
+ if (event === DONE) return;
1466
+ if (event !== NO_DATA) yield event;
1467
+ boundary$1 = buffer.indexOf("\n\n");
1468
+ }
1469
+ }
1470
+ try {
1471
+ buffer += decoder.decode();
1472
+ } catch (cause) {
1473
+ throw new LLMError("Invalid UTF-8 in SSE stream", "parse", void 0, void 0, cause);
1474
+ }
1475
+ buffer = buffer.replace(/\r$/, "\n");
1476
+ let boundary = buffer.indexOf("\n\n");
1477
+ while (boundary !== -1) {
1478
+ const frame = buffer.slice(0, boundary);
1479
+ buffer = buffer.slice(boundary + 2);
1480
+ const event = parseSseFrame(frame);
1481
+ if (event === DONE) return;
1482
+ if (event !== NO_DATA) yield event;
1483
+ boundary = buffer.indexOf("\n\n");
1484
+ }
1485
+ const trailing = buffer.trim();
1486
+ if (trailing) {
1487
+ const event = parseSseFrame(trailing);
1488
+ if (event !== DONE && event !== NO_DATA) yield event;
1489
+ }
1490
+ }
1491
+ const DONE = Symbol("sse-stream-done");
1492
+ const NO_DATA = Symbol("sse-frame-no-data");
1493
+ /**
1494
+ * Sentinel yielded by `parseSseStream` for a comment-only frame (no
1495
+ * `data:` payload), the mechanism providers use for SSE keep-alive
1496
+ * pings. Exported so a consumer (e.g. `fromFetch`) can react to "still
1497
+ * alive" separately from a genuinely empty frame (`NO_DATA`, kept internal).
1498
+ */
1499
+ const SSE_PING = Symbol("sse-frame-ping");
1500
+ /** Extracts and JSON-parses the `data:` payload of one SSE frame (the text between two blank lines). */
1501
+ function parseSseFrame(frame) {
1502
+ const dataLines = [];
1503
+ let sawComment = false;
1504
+ for (const line of frame.split("\n")) {
1505
+ if (line.startsWith(":")) {
1506
+ sawComment = true;
1507
+ continue;
1508
+ }
1509
+ if (!line.startsWith("data:")) continue;
1510
+ dataLines.push(line.startsWith("data: ") ? line.slice(6) : line.slice(5));
1511
+ }
1512
+ if (!dataLines.length) return sawComment ? SSE_PING : NO_DATA;
1513
+ const data = dataLines.join("\n");
1514
+ if (data === "[DONE]") return DONE;
1515
+ try {
1516
+ return JSON.parse(data);
1517
+ } catch (cause) {
1518
+ throw new LLMError(`Invalid JSON in SSE frame: ${data.slice(0, 200)}`, "parse", void 0, void 0, cause);
1519
+ }
1520
+ }
1521
+
714
1522
  //#endregion
715
1523
  //#region src/internal/imageFormat.ts
716
1524
  /**
@@ -758,6 +1566,93 @@ function toAnthropicContent(blocks) {
758
1566
  });
759
1567
  }
760
1568
  /**
1569
+ * Asserts a caller-supplied JSON Schema is an object schema before it's
1570
+ * used as Anthropic's `Tool.input_schema`, which (like every other
1571
+ * provider's function-calling API) requires `type: 'object'`. VernLLM's own
1572
+ * public `tools`/`jsonSchema` APIs accept freeform `Record<string,
1573
+ * unknown>` JSON Schema, so nothing upstream guarantees this at compile
1574
+ * time; this is the runtime check that stands in for that, so a schema
1575
+ * missing (or mistyping) `type: 'object'` fails loudly and immediately
1576
+ * instead of being silently forwarded to Anthropic malformed.
1577
+ */
1578
+ function assertObjectSchema(schema, toolName) {
1579
+ if (schema.type !== "object") throw new LLMError(`Tool "${toolName}"'s schema must have "type": "object" (Anthropic requires object-shaped tool parameters).`, "validation");
1580
+ return schema;
1581
+ }
1582
+ /**
1583
+ * Translates VernLLM's OpenAI-shaped wire `tool_choice` into Anthropic's
1584
+ * `{ type: 'auto' | 'any' | 'none' | 'tool', name? }` shape. `'required'`
1585
+ * maps to `'any'` (Anthropic's "must call some tool" equivalent).
1586
+ */
1587
+ function toAnthropicToolChoice(toolChoice) {
1588
+ if (!toolChoice || toolChoice === "auto") return { type: "auto" };
1589
+ if (toolChoice === "none") return { type: "none" };
1590
+ if (toolChoice === "required") return { type: "any" };
1591
+ return {
1592
+ type: "tool",
1593
+ name: toolChoice.function.name
1594
+ };
1595
+ }
1596
+ /**
1597
+ * Builds the Anthropic-shaped request body from VernLLM's wire params,
1598
+ * shared between `create` and `createStream` so both go through identical
1599
+ * translation (system prompt, message shaping, and the jsonSchema →
1600
+ * forced-single-tool mapping all happen exactly once, not once per entry
1601
+ * point).
1602
+ *
1603
+ * Returns `toolName` alongside the body: when set, the model was forced to
1604
+ * call a single synthetic tool standing in for `jsonSchema` output, and
1605
+ * both `create` and `createStream` need to know this so they can unwrap
1606
+ * that tool call back into plain text content instead of treating it like
1607
+ * a real tool call.
1608
+ */
1609
+ function buildAnthropicRequestBody(params) {
1610
+ const systemMessage = params.messages.find((m) => m.role === "system");
1611
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
1612
+ const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name.trim() : void 0;
1613
+ if (params.response_format?.type === "json_schema" && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
1614
+ let jsonInstruction;
1615
+ let tools;
1616
+ let toolChoice;
1617
+ if (params.response_format?.type === "json_schema" && toolName) {
1618
+ const { schema, description, strict } = params.response_format.json_schema;
1619
+ tools = [{
1620
+ name: toolName,
1621
+ description,
1622
+ input_schema: assertObjectSchema(schema, toolName),
1623
+ strict
1624
+ }];
1625
+ toolChoice = {
1626
+ type: "tool",
1627
+ name: toolName
1628
+ };
1629
+ } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
1630
+ else if (params.tools?.length) {
1631
+ tools = params.tools.map((t) => ({
1632
+ name: t.function.name,
1633
+ description: t.function.description,
1634
+ input_schema: assertObjectSchema(t.function.parameters, t.function.name)
1635
+ }));
1636
+ toolChoice = toAnthropicToolChoice(params.tool_choice);
1637
+ }
1638
+ const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
1639
+ const body = {
1640
+ model: params.model,
1641
+ max_tokens: params.max_tokens,
1642
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
1643
+ system: system || void 0,
1644
+ messages: mergeConsecutiveToolResults$1(conversationMessages.map((m) => toAnthropicMessage(m))),
1645
+ ...tools ? {
1646
+ tools,
1647
+ tool_choice: toolChoice
1648
+ } : {}
1649
+ };
1650
+ return {
1651
+ body,
1652
+ toolName
1653
+ };
1654
+ }
1655
+ /**
761
1656
  * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
762
1657
  * interface VernLLM uses for OpenAI/Groq.
763
1658
  *
@@ -772,53 +1667,161 @@ function toAnthropicContent(blocks) {
772
1667
  * generation against.
773
1668
  */
774
1669
  function fromAnthropic(anthropicClient) {
775
- return { chat: { completions: { async create(params, options) {
776
- const systemMessage = params.messages.find((m) => m.role === "system");
777
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
778
- const toolName = params.response_format?.type === "json_schema" ? params.response_format.json_schema.name : void 0;
779
- let jsonInstruction;
780
- let tools;
781
- if (params.response_format?.type === "json_schema" && toolName) {
782
- const { schema, description, strict } = params.response_format.json_schema;
783
- tools = [{
784
- name: toolName,
785
- description,
786
- input_schema: schema,
787
- strict
788
- }];
789
- } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
790
- const system = [systemMessage?.content, jsonInstruction].filter(Boolean).join("\n\n");
791
- const response = await anthropicClient.messages.create({
792
- model: params.model,
793
- max_tokens: params.max_tokens,
794
- temperature: params.temperature,
795
- system: system || void 0,
796
- messages: conversationMessages.map((m) => ({
797
- role: m.role,
798
- content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content
799
- })),
800
- ...tools ? {
801
- tools,
802
- tool_choice: {
803
- type: "tool",
804
- name: toolName
1670
+ const rawMessagesCreate = anthropicClient.messages.create.bind(anthropicClient.messages);
1671
+ return { chat: { completions: {
1672
+ async create(params, options) {
1673
+ const { body, toolName } = buildAnthropicRequestBody(params);
1674
+ const response = await anthropicClient.messages.create(body, options);
1675
+ let text;
1676
+ let wireToolCalls;
1677
+ if (toolName) {
1678
+ const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
1679
+ if (!toolUse) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
1680
+ if (!toolUse.input || typeof toolUse.input !== "object" || Array.isArray(toolUse.input)) throw new LLMError(`Anthropic returned invalid structured output for tool "${toolName}". Expected an object.`, "validation");
1681
+ text = JSON.stringify(toolUse.input);
1682
+ } else {
1683
+ text = response.content.filter((block) => block.type === "text").map((block) => block.text ?? "").join("");
1684
+ const toolUses = response.content.filter((block) => block.type === "tool_use");
1685
+ if (toolUses.length) wireToolCalls = toolUses.map((block) => ({
1686
+ id: block.id,
1687
+ type: "function",
1688
+ function: {
1689
+ name: block.name,
1690
+ arguments: JSON.stringify(block.input ?? {})
1691
+ }
1692
+ }));
1693
+ }
1694
+ return {
1695
+ choices: [{ message: {
1696
+ content: text,
1697
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
1698
+ } }],
1699
+ usage: {
1700
+ prompt_tokens: response.usage?.input_tokens,
1701
+ completion_tokens: response.usage?.output_tokens,
1702
+ total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
805
1703
  }
806
- } : {}
807
- }, options);
808
- let text;
809
- if (toolName) {
810
- const toolUse = response.content.find((block) => block.type === "tool_use" && block.name === toolName);
811
- text = toolUse ? JSON.stringify(toolUse.input) : "";
812
- } else text = response.content.find((block) => block.type === "text")?.text ?? "";
813
- return {
814
- choices: [{ message: { content: text } }],
815
- usage: {
816
- prompt_tokens: response.usage?.input_tokens,
817
- completion_tokens: response.usage?.output_tokens,
818
- total_tokens: (response.usage?.input_tokens ?? 0) + (response.usage?.output_tokens ?? 0)
1704
+ };
1705
+ },
1706
+ async *createStream(params, options) {
1707
+ const { body, toolName } = buildAnthropicRequestBody(params);
1708
+ const stream = await rawMessagesCreate({
1709
+ ...body,
1710
+ stream: true
1711
+ }, options);
1712
+ const blockKinds = new Map();
1713
+ let inputTokens = 0;
1714
+ let sawJsonTool = false;
1715
+ for await (const event of stream) if (event.type === "message_start") inputTokens = event.message.usage?.input_tokens ?? 0;
1716
+ else if (event.type === "content_block_start") if (event.content_block.type === "tool_use") {
1717
+ const kind = event.content_block.name === toolName ? "json-tool" : "tool_use";
1718
+ blockKinds.set(event.index, kind);
1719
+ if (kind === "json-tool") sawJsonTool = true;
1720
+ else if (!toolName) yield {
1721
+ type: "tool_call_delta",
1722
+ index: event.index,
1723
+ id: event.content_block.id,
1724
+ name: event.content_block.name
1725
+ };
1726
+ } else blockKinds.set(event.index, "text");
1727
+ else if (event.type === "content_block_delta") {
1728
+ if (event.delta.type === "text_delta") {
1729
+ if (!toolName) yield {
1730
+ type: "text-delta",
1731
+ delta: event.delta.text
1732
+ };
1733
+ } else if (event.delta.type === "input_json_delta") {
1734
+ const kind = blockKinds.get(event.index);
1735
+ if (kind === "json-tool") yield {
1736
+ type: "text-delta",
1737
+ delta: event.delta.partial_json
1738
+ };
1739
+ else if (!toolName) yield {
1740
+ type: "tool_call_delta",
1741
+ index: event.index,
1742
+ argumentsDelta: event.delta.partial_json
1743
+ };
1744
+ }
1745
+ } else if (event.type === "message_delta") {
1746
+ const outputTokens = event.usage?.output_tokens ?? 0;
1747
+ yield {
1748
+ type: "usage",
1749
+ usage: {
1750
+ prompt_tokens: inputTokens,
1751
+ completion_tokens: outputTokens,
1752
+ total_tokens: inputTokens + outputTokens
1753
+ }
1754
+ };
1755
+ } else if (event.type === "ping") yield { type: "ping" };
1756
+ if (toolName && !sawJsonTool) throw new LLMError(`Anthropic did not return the required structured output tool "${toolName}".`, "validation");
1757
+ }
1758
+ } } };
1759
+ }
1760
+ /**
1761
+ * Anthropic requires strict role alternation, so the per-wire-message
1762
+ * mapping above (one `{role:'user', content:[tool_result]}` per VernLLM
1763
+ * wire tool message) needs merging back together when an assistant turn
1764
+ * requested more than one tool: multiple consecutive user turns would
1765
+ * violate that alternation, and Anthropic's API rejects it outright. This
1766
+ * merges any run of tool-result-only user messages into one, with all
1767
+ * their tool_result blocks combined, the shape Anthropic expects for "here
1768
+ * are the results of everything you just asked for."
1769
+ */
1770
+ function mergeConsecutiveToolResults$1(messages) {
1771
+ const isToolResultOnly = (m) => m.role === "user" && Array.isArray(m.content) && m.content.length > 0 && m.content.every((b) => b.type === "tool_result");
1772
+ const merged = [];
1773
+ for (const m of messages) {
1774
+ const prev = merged.at(-1);
1775
+ if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
1776
+ else merged.push(m);
1777
+ }
1778
+ return merged;
1779
+ }
1780
+ /**
1781
+ * Translates one VernLLM wire message (OpenAI-shaped: plain user/assistant
1782
+ * turns, an assistant turn with `tool_calls`, or a `tool` turn) into
1783
+ * Anthropic's `{ role: 'user' | 'assistant', content }` shape.
1784
+ */
1785
+ function toAnthropicMessage(m) {
1786
+ if (m.role === "tool") return {
1787
+ role: "user",
1788
+ content: [{
1789
+ type: "tool_result",
1790
+ tool_use_id: m.tool_call_id,
1791
+ content: m.content,
1792
+ ...m.is_error ? { is_error: true } : {}
1793
+ }]
1794
+ };
1795
+ if (m.role === "assistant" && m.tool_calls?.length) {
1796
+ const blocks = [];
1797
+ if (m.content) blocks.push({
1798
+ type: "text",
1799
+ text: m.content
1800
+ });
1801
+ for (const tc of m.tool_calls) {
1802
+ let input;
1803
+ try {
1804
+ input = tc.function.arguments.trim() ? JSON.parse(tc.function.arguments) : {};
1805
+ } catch (cause) {
1806
+ throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
819
1807
  }
1808
+ if (input === null || Array.isArray(input) || typeof input !== "object") throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) arguments must be a JSON object.`, "validation");
1809
+ blocks.push({
1810
+ type: "tool_use",
1811
+ id: tc.id,
1812
+ name: tc.function.name,
1813
+ input
1814
+ });
1815
+ }
1816
+ return {
1817
+ role: "assistant",
1818
+ content: blocks
820
1819
  };
821
- } } } };
1820
+ }
1821
+ return {
1822
+ role: m.role,
1823
+ content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content ?? ""
1824
+ };
822
1825
  }
823
1826
 
824
1827
  //#endregion
@@ -835,6 +1838,122 @@ function toGeminiParts(blocks) {
835
1838
  data: block.data
836
1839
  } } : { text: block.text });
837
1840
  }
1841
+ /** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Gemini's `functionCallingConfig`. */
1842
+ function toGeminiToolConfig(toolChoice) {
1843
+ if (!toolChoice || toolChoice === "auto") return { functionCallingConfig: { mode: "AUTO" } };
1844
+ if (toolChoice === "none") return { functionCallingConfig: { mode: "NONE" } };
1845
+ if (toolChoice === "required") return { functionCallingConfig: { mode: "ANY" } };
1846
+ return { functionCallingConfig: {
1847
+ mode: "ANY",
1848
+ allowedFunctionNames: [toolChoice.function.name]
1849
+ } };
1850
+ }
1851
+ /**
1852
+ * Translates one VernLLM wire message into a Gemini `contents` entry.
1853
+ * Gemini has no separate 'tool' role: a prior assistant tool request
1854
+ * becomes a `'model'` turn with `functionCall` parts, and its result
1855
+ * becomes a `'user'` turn with `functionResponse` parts.
1856
+ */
1857
+ function toGeminiContent(m) {
1858
+ if (m.role === "tool") return {
1859
+ role: "user",
1860
+ parts: [{ functionResponse: {
1861
+ name: m.tool_call_id,
1862
+ response: parseToolResult(m.content)
1863
+ } }]
1864
+ };
1865
+ if (m.role === "assistant" && m.tool_calls?.length) {
1866
+ const parts = [];
1867
+ if (typeof m.content === "string" && m.content) parts.push({ text: m.content });
1868
+ parts.push(...m.tool_calls.map((tc) => ({ functionCall: {
1869
+ name: tc.function.name,
1870
+ args: parseToolArguments(tc.function.arguments, tc.function.name)
1871
+ } })));
1872
+ return {
1873
+ role: "model",
1874
+ parts
1875
+ };
1876
+ }
1877
+ return {
1878
+ role: m.role === "assistant" ? "model" : "user",
1879
+ parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content ?? "" }]
1880
+ };
1881
+ }
1882
+ function parseToolArguments(text, toolName) {
1883
+ let parsed;
1884
+ try {
1885
+ parsed = text.trim() ? JSON.parse(text) : {};
1886
+ } catch (cause) {
1887
+ throw new LLMError(`Tool call "${toolName}" arguments are not valid JSON.`, "validation", void 0, void 0, cause);
1888
+ }
1889
+ if (!parsed || Array.isArray(parsed) || typeof parsed !== "object") throw new LLMError(`Tool call "${toolName}" arguments must be a JSON object.`, "validation");
1890
+ return parsed;
1891
+ }
1892
+ function parseToolResult(text) {
1893
+ try {
1894
+ return text.trim() ? JSON.parse(text) : "";
1895
+ } catch {
1896
+ return text;
1897
+ }
1898
+ }
1899
+ /**
1900
+ * Gemini expects the results of everything the model asked for in one turn
1901
+ * to arrive together as multiple `functionResponse` parts on a single
1902
+ * `'user'` entry, not as separate consecutive `'user'` entries. The
1903
+ * per-wire-message mapping above produces one `'user'` entry per VernLLM
1904
+ * wire tool message, so when an assistant turn requested more than one
1905
+ * tool, this merges the resulting run of functionResponse-only `'user'`
1906
+ * entries back into one.
1907
+ */
1908
+ function mergeConsecutiveFunctionResponses(contents) {
1909
+ const isFunctionResponseOnly = (c) => c.role === "user" && c.parts.length > 0 && c.parts.every((p) => "functionResponse" in p);
1910
+ const merged = [];
1911
+ for (const c of contents) {
1912
+ const prev = merged.at(-1);
1913
+ if (isFunctionResponseOnly(c) && prev && isFunctionResponseOnly(prev)) prev.parts.push(...c.parts);
1914
+ else merged.push(c);
1915
+ }
1916
+ return merged;
1917
+ }
1918
+ /**
1919
+ * Builds the Gemini-shaped request from VernLLM's wire params, shared
1920
+ * between `create` and `createStream` so both go through identical
1921
+ * translation (contents shaping, `responseSchema`/`responseMimeType`
1922
+ * mapping, and tool/toolConfig translation all happen exactly once).
1923
+ * `abortSignal` is folded into `config` by the caller (`create`/
1924
+ * `createStream`), once the request options are available.
1925
+ */
1926
+ function buildGeminiRequest(params) {
1927
+ const systemMessage = params.messages.find((m) => m.role === "system");
1928
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
1929
+ const wantsJson = Boolean(params.response_format);
1930
+ const config = {
1931
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
1932
+ maxOutputTokens: params.max_tokens,
1933
+ ...systemMessage ? { systemInstruction: { parts: [{ text: systemMessage.content }] } } : {}
1934
+ };
1935
+ if (wantsJson) config.responseMimeType = "application/json";
1936
+ if (params.response_format?.type === "json_schema") {
1937
+ const { schema, description } = params.response_format.json_schema;
1938
+ config.responseSchema = {
1939
+ ...schema,
1940
+ ...description ? { description } : {}
1941
+ };
1942
+ }
1943
+ if (params.tools?.length) {
1944
+ config.tools = [{ functionDeclarations: params.tools.map((t) => ({
1945
+ name: t.function.name,
1946
+ description: t.function.description,
1947
+ parameters: t.function.parameters
1948
+ })) }];
1949
+ config.toolConfig = toGeminiToolConfig(params.tool_choice);
1950
+ }
1951
+ return {
1952
+ model: params.model,
1953
+ contents: mergeConsecutiveFunctionResponses(conversationMessages.map((m) => toGeminiContent(m))),
1954
+ config
1955
+ };
1956
+ }
838
1957
  /**
839
1958
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
840
1959
  * uses for OpenAI-compatible APIs. Gemini's shape differs on nearly every
@@ -845,43 +1964,100 @@ function toGeminiParts(blocks) {
845
1964
  * `responseSchema`. `reasoning_effort` has no equivalent. Gemini's thinking
846
1965
  * models use a token budget, not an effort tier, so it's dropped, same as
847
1966
  * Anthropic.
1967
+ *
1968
+ * `tools` maps to Gemini's native `functionDeclarations`/`functionCall`;
1969
+ * `tool_choice` maps to `toolConfig.functionCallingConfig`. `jsonSchema`
1970
+ * and `tools` are mutually exclusive by the time a call reaches here
1971
+ * (enforced in vernLLM.ts), so `responseSchema` and `tools` never
1972
+ * both apply.
1973
+ *
1974
+ * `createStream` calls `generateContentStream` (optional on `GeminiClient`
1975
+ *, required only if the caller sets `stream: true`) and translates each
1976
+ * partial response into `WireStreamChunk`s. Unlike OpenAI/Anthropic,
1977
+ * Gemini's own function-calling API doesn't stream tool-call arguments
1978
+ * incrementally: a `functionCall` part always arrives whole in one chunk,
1979
+ * so each one is emitted as a single, complete `tool_call_delta` (a
1980
+ * one-shot "delta" containing the full arguments) rather than accumulated
1981
+ * fragments, that's a real difference in the underlying API, not
1982
+ * something this adapter can smooth over. `usageMetadata` is (per Gemini's
1983
+ * own behavior) only reliably present on the last chunk, so the `usage`
1984
+ * `WireStreamChunk` is emitted once, after the stream completes, from
1985
+ * whichever chunk's `usageMetadata` was seen last.
848
1986
  */
849
1987
  function fromGemini(geminiClient) {
850
- return { chat: { completions: { async create(params, options) {
851
- const systemMessage = params.messages.find((m) => m.role === "system");
852
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
853
- const wantsJson = Boolean(params.response_format);
854
- const generationConfig = {
855
- temperature: params.temperature,
856
- maxOutputTokens: params.max_tokens
857
- };
858
- if (wantsJson) generationConfig.responseMimeType = "application/json";
859
- if (params.response_format?.type === "json_schema") {
860
- const { schema, description } = params.response_format.json_schema;
861
- generationConfig.responseSchema = {
862
- ...schema,
863
- ...description ? { description } : {}
1988
+ return { chat: { completions: {
1989
+ async create(params, options) {
1990
+ const request = buildGeminiRequest(params);
1991
+ request.config = {
1992
+ ...request.config,
1993
+ abortSignal: options.signal
864
1994
  };
865
- }
866
- const response = await geminiClient.generateContent({
867
- model: params.model,
868
- contents: conversationMessages.map((m) => ({
869
- role: m.role === "assistant" ? "model" : "user",
870
- parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content }]
871
- })),
872
- systemInstruction: systemMessage ? { parts: [{ text: systemMessage.content }] } : void 0,
873
- generationConfig
874
- }, options);
875
- const text = response.candidates?.[0]?.content?.parts?.map((p) => p.text ?? "").join("") ?? "";
876
- return {
877
- choices: [{ message: { content: text } }],
878
- usage: {
879
- prompt_tokens: response.usageMetadata?.promptTokenCount,
880
- completion_tokens: response.usageMetadata?.candidatesTokenCount,
881
- total_tokens: response.usageMetadata?.totalTokenCount
1995
+ const response = await geminiClient.generateContent(request);
1996
+ const parts = response.candidates?.[0]?.content?.parts ?? [];
1997
+ const text = parts.map((p) => p.text ?? "").join("");
1998
+ const functionCalls = parts.filter((p) => p.functionCall);
1999
+ let wireToolCalls;
2000
+ if (functionCalls.length) wireToolCalls = functionCalls.map((p) => ({
2001
+ id: p.functionCall.name,
2002
+ type: "function",
2003
+ function: {
2004
+ name: p.functionCall.name,
2005
+ arguments: JSON.stringify(p.functionCall.args ?? {})
2006
+ }
2007
+ }));
2008
+ return {
2009
+ choices: [{ message: {
2010
+ content: text,
2011
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
2012
+ } }],
2013
+ usage: {
2014
+ prompt_tokens: response.usageMetadata?.promptTokenCount,
2015
+ completion_tokens: response.usageMetadata?.candidatesTokenCount,
2016
+ total_tokens: response.usageMetadata?.totalTokenCount
2017
+ }
2018
+ };
2019
+ },
2020
+ async *createStream(params, options) {
2021
+ if (!geminiClient.generateContentStream) throw new LLMError("stream: true requires a Gemini client with generateContentStream", "validation");
2022
+ const request = buildGeminiRequest(params);
2023
+ request.config = {
2024
+ ...request.config,
2025
+ abortSignal: options.signal
2026
+ };
2027
+ const stream = await geminiClient.generateContentStream(request);
2028
+ let toolCallIndex = 0;
2029
+ let lastUsage;
2030
+ for await (const chunk of stream) {
2031
+ const parts = chunk.candidates?.[0]?.content?.parts ?? [];
2032
+ for (const part of parts) {
2033
+ if (part.text) yield {
2034
+ type: "text-delta",
2035
+ delta: part.text
2036
+ };
2037
+ if (part.functionCall) {
2038
+ yield {
2039
+ type: "tool_call_delta",
2040
+ index: toolCallIndex,
2041
+ id: part.functionCall.name,
2042
+ name: part.functionCall.name,
2043
+ argumentsDelta: JSON.stringify(part.functionCall.args ?? {}),
2044
+ complete: true
2045
+ };
2046
+ toolCallIndex++;
2047
+ }
2048
+ }
2049
+ if (chunk.usageMetadata) lastUsage = chunk.usageMetadata;
882
2050
  }
883
- };
884
- } } } };
2051
+ if (lastUsage) yield {
2052
+ type: "usage",
2053
+ usage: {
2054
+ prompt_tokens: lastUsage.promptTokenCount,
2055
+ completion_tokens: lastUsage.candidatesTokenCount,
2056
+ total_tokens: lastUsage.totalTokenCount
2057
+ }
2058
+ };
2059
+ }
2060
+ } } };
885
2061
  }
886
2062
 
887
2063
  //#endregion
@@ -917,6 +2093,67 @@ function toBedrockContent(blocks) {
917
2093
  } } : { text: block.text });
918
2094
  }
919
2095
  /**
2096
+ * Builds the Converse-shaped request from VernLLM's wire params, shared
2097
+ * between `create` and `createStream` so both go through identical
2098
+ * translation (system prompt, message shaping, the jsonSchema →
2099
+ * forced-single-tool mapping, and the `toolUseSupportedModels` preflight
2100
+ * check all happen exactly once).
2101
+ *
2102
+ * Returns `toolName` alongside the request: when set, the model was forced
2103
+ * to call a single synthetic tool standing in for `jsonSchema` output, and
2104
+ * both `create` and `createStream` need to know this so they can unwrap
2105
+ * that tool call back into plain text content instead of treating it like
2106
+ * a real tool call.
2107
+ */
2108
+ function buildBedrockRequest(params, toolUseSupportedModels) {
2109
+ const systemMessage = params.messages.find((m) => m.role === "system");
2110
+ const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant" || m.role === "tool");
2111
+ const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
2112
+ const toolName = jsonSchema?.name.trim();
2113
+ if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
2114
+ let jsonInstruction;
2115
+ let toolConfig;
2116
+ if (jsonSchema) {
2117
+ const { schema, description, strict } = jsonSchema;
2118
+ toolConfig = {
2119
+ tools: [{ toolSpec: {
2120
+ name: toolName,
2121
+ description,
2122
+ inputSchema: { json: schema },
2123
+ strict
2124
+ } }],
2125
+ toolChoice: { tool: { name: toolName } }
2126
+ };
2127
+ } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
2128
+ else if (params.tools?.length) toolConfig = {
2129
+ tools: params.tools.map((t) => ({ toolSpec: {
2130
+ name: t.function.name,
2131
+ description: t.function.description,
2132
+ inputSchema: { json: t.function.parameters }
2133
+ } })),
2134
+ toolChoice: toBedrockToolChoice(params.tool_choice)
2135
+ };
2136
+ if (jsonSchema && toolUseSupportedModels) {
2137
+ const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
2138
+ if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
2139
+ }
2140
+ const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
2141
+ const request = {
2142
+ modelId: params.model,
2143
+ messages: mergeConsecutiveToolResults(conversationMessages.map((m) => toBedrockMessage(m))),
2144
+ system: systemParts.length ? systemParts.map((text) => ({ text })) : void 0,
2145
+ inferenceConfig: {
2146
+ ...params.temperature !== void 0 ? { temperature: params.temperature } : {},
2147
+ maxTokens: params.max_tokens
2148
+ },
2149
+ ...toolConfig ? { toolConfig } : {}
2150
+ };
2151
+ return {
2152
+ request,
2153
+ toolName
2154
+ };
2155
+ }
2156
+ /**
920
2157
  * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
921
2158
  * interface VernLLM uses for OpenAI/Groq. The Converse API is unified
922
2159
  * across Bedrock's model families (Anthropic, Titan, Llama, Mistral, etc.),
@@ -936,66 +2173,237 @@ function toBedrockContent(blocks) {
936
2173
  * `response_format: json_object` (no schema to build a tool from) and
937
2174
  * `reasoning_effort` (no Converse equivalent) fall back to a system-prompt
938
2175
  * instruction and are dropped respectively.
2176
+ *
2177
+ * `tools` maps to Converse's native `toolConfig`/`toolUse`/`toolResult`;
2178
+ * `tool_choice` maps to `toolConfig.toolChoice`. Mutually exclusive with
2179
+ * `jsonSchema` by the time a call reaches here (enforced in vernLLM.ts).
2180
+ *
2181
+ * `createStream` calls `converseStream` (optional on `BedrockConverseClient`
2182
+ *, required only if the caller sets `stream: true`) and translates its
2183
+ * `contentBlockStart`/`contentBlockDelta`/`metadata` events into
2184
+ * `WireStreamChunk`s. Content blocks are tracked by `contentBlockIndex`,
2185
+ * same as `fromAnthropic`'s block-index tracking (Converse's streaming
2186
+ * shape is structurally close to Anthropic's own, both being tool-use-aware
2187
+ * content-block streams), including the same `json-tool` unwrapping: a
2188
+ * `jsonSchema`-forced tool's `toolUse.input` deltas are re-emitted as
2189
+ * `text-delta`, not `tool_call_delta`, so the accumulated result lands in
2190
+ * `finalizeResponse`'s `content` path exactly like the non-streaming
2191
+ * `create` branch above unwraps it.
939
2192
  */
940
2193
  function fromBedrock(bedrockClient, options) {
941
2194
  const toolUseSupportedModels = options?.toolUseSupportedModels;
942
- return { chat: { completions: { async create(params, requestOptions) {
943
- const systemMessage = params.messages.find((m) => m.role === "system");
944
- const conversationMessages = params.messages.filter((m) => m.role === "user" || m.role === "assistant");
945
- const jsonSchema = params.response_format?.type === "json_schema" ? params.response_format.json_schema : void 0;
946
- const toolName = jsonSchema?.name.trim();
947
- if (jsonSchema && !toolName) throw new LLMError("json_schema.name must not be empty.", "validation");
948
- let jsonInstruction;
949
- let toolConfig;
950
- if (jsonSchema) {
951
- const { schema, description, strict } = jsonSchema;
952
- toolConfig = {
953
- tools: [{ toolSpec: {
954
- name: toolName,
955
- description,
956
- inputSchema: { json: schema },
957
- strict
2195
+ return { chat: { completions: {
2196
+ async create(params, requestOptions) {
2197
+ const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
2198
+ const response = await bedrockClient.converse(request, requestOptions);
2199
+ let text;
2200
+ let wireToolCalls;
2201
+ if (toolName) {
2202
+ const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
2203
+ text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
2204
+ } else {
2205
+ const blocks = response.output?.message?.content ?? [];
2206
+ text = blocks.map((c) => c.text ?? "").join("");
2207
+ const toolUses = blocks.filter((block) => Boolean(block.toolUse));
2208
+ if (toolUses.length) wireToolCalls = toolUses.map((block, i) => {
2209
+ const toolUse = block.toolUse;
2210
+ if (!toolUse.name) throw new LLMError(`Bedrock returned a toolUse block without a name at index ${i}.`, "validation");
2211
+ return {
2212
+ id: toolUse.toolUseId ?? `${toolUse.name}_${i}`,
2213
+ type: "function",
2214
+ function: {
2215
+ name: toolUse.name,
2216
+ arguments: JSON.stringify(toolUse.input ?? {})
2217
+ }
2218
+ };
2219
+ });
2220
+ }
2221
+ return {
2222
+ choices: [{ message: {
2223
+ content: text,
2224
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
958
2225
  } }],
959
- toolChoice: { tool: { name: toolName } }
2226
+ usage: {
2227
+ prompt_tokens: response.usage?.inputTokens,
2228
+ completion_tokens: response.usage?.outputTokens,
2229
+ total_tokens: response.usage?.totalTokens
2230
+ }
2231
+ };
2232
+ },
2233
+ async *createStream(params, requestOptions) {
2234
+ if (!bedrockClient.converseStream) throw new LLMError("stream: true requires a Bedrock client with converseStream", "validation");
2235
+ const { request, toolName } = buildBedrockRequest(params, toolUseSupportedModels);
2236
+ const { stream } = await bedrockClient.converseStream(request, requestOptions);
2237
+ const blockKinds = new Map();
2238
+ for await (const event of stream) if ("contentBlockStart" in event) {
2239
+ const { contentBlockIndex, start } = event.contentBlockStart;
2240
+ if (start?.toolUse) {
2241
+ const kind = start.toolUse.name === toolName ? "json-tool" : "tool_use";
2242
+ blockKinds.set(contentBlockIndex, kind);
2243
+ if (kind === "tool_use" && !toolName) yield {
2244
+ type: "tool_call_delta",
2245
+ index: contentBlockIndex,
2246
+ id: start.toolUse.toolUseId,
2247
+ name: start.toolUse.name
2248
+ };
2249
+ } else blockKinds.set(contentBlockIndex, "text");
2250
+ } else if ("contentBlockDelta" in event) {
2251
+ const { contentBlockIndex, delta } = event.contentBlockDelta;
2252
+ if (delta && "text" in delta && delta.text !== void 0 && !toolName) yield {
2253
+ type: "text-delta",
2254
+ delta: delta.text
2255
+ };
2256
+ else if (delta && "toolUse" in delta && delta.toolUse?.input !== void 0) {
2257
+ const kind = blockKinds.get(contentBlockIndex);
2258
+ if (kind === "json-tool") yield {
2259
+ type: "text-delta",
2260
+ delta: delta.toolUse.input
2261
+ };
2262
+ else if (!toolName) yield {
2263
+ type: "tool_call_delta",
2264
+ index: contentBlockIndex,
2265
+ argumentsDelta: delta.toolUse.input
2266
+ };
2267
+ }
2268
+ } else if ("metadata" in event && event.metadata.usage) yield {
2269
+ type: "usage",
2270
+ usage: {
2271
+ prompt_tokens: event.metadata.usage.inputTokens,
2272
+ completion_tokens: event.metadata.usage.outputTokens,
2273
+ total_tokens: event.metadata.usage.totalTokens
2274
+ }
960
2275
  };
961
- } else if (params.response_format?.type === "json_object") jsonInstruction = "Respond with valid JSON only, no prose or markdown fences.";
962
- if (jsonSchema && toolUseSupportedModels) {
963
- const isSupported = Array.isArray(toolUseSupportedModels) ? toolUseSupportedModels.includes(params.model) : toolUseSupportedModels(params.model);
964
- if (!isSupported) throw new LLMError(`Bedrock model "${params.model}" is not listed in toolUseSupportedModels, but jsonSchema structured output requires Converse tool use.`, "validation");
2276
+ else if ("throttlingException" in event) throw new LLMError(event.throttlingException.message ?? "Bedrock throttled the request mid-stream", "api", 429);
2277
+ else if ("validationException" in event) throw new LLMError(event.validationException.message ?? "Bedrock rejected the request mid-stream", "validation");
2278
+ else if ("internalServerException" in event || "serviceUnavailableException" in event || "modelStreamErrorException" in event) {
2279
+ const detail = "internalServerException" in event && event.internalServerException.message || "serviceUnavailableException" in event && event.serviceUnavailableException.message || "modelStreamErrorException" in event && event.modelStreamErrorException.message || "Bedrock reported a mid-stream error";
2280
+ const status = "modelStreamErrorException" in event && event.modelStreamErrorException.originalStatusCode || "serviceUnavailableException" in event && 503 || 500;
2281
+ throw new LLMError(detail, "api", status);
2282
+ }
965
2283
  }
966
- const systemParts = [systemMessage?.content, jsonInstruction].filter((s) => Boolean(s));
967
- const response = await bedrockClient.converse({
968
- modelId: params.model,
969
- messages: conversationMessages.map((m) => ({
970
- role: m.role,
971
- content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content }]
972
- })),
973
- system: systemParts.length ? systemParts.map((text$1) => ({ text: text$1 })) : void 0,
974
- inferenceConfig: {
975
- temperature: params.temperature,
976
- maxTokens: params.max_tokens
977
- },
978
- ...toolConfig ? { toolConfig } : {}
979
- }, requestOptions);
980
- let text;
981
- if (toolName) {
982
- const toolUseBlock = response.output?.message?.content?.find((block) => block.toolUse?.name === toolName);
983
- text = toolUseBlock?.toolUse ? JSON.stringify(toolUseBlock.toolUse.input) : "";
984
- } else text = response.output?.message?.content?.map((c) => c.text ?? "").join("") ?? "";
985
- return {
986
- choices: [{ message: { content: text } }],
987
- usage: {
988
- prompt_tokens: response.usage?.inputTokens,
989
- completion_tokens: response.usage?.outputTokens,
990
- total_tokens: response.usage?.totalTokens
2284
+ } } };
2285
+ }
2286
+ /** Maps VernLLM's OpenAI-shaped wire `tool_choice` onto Converse's `toolChoice`. */
2287
+ function toBedrockToolChoice(toolChoice) {
2288
+ if (!toolChoice || toolChoice === "auto") return { auto: {} };
2289
+ if (toolChoice === "required") return { any: {} };
2290
+ if (toolChoice === "none") throw new LLMError("'none' is not supported by fromBedrock: Bedrock Converse has no `tool_choice` equivalent to forbidding tool use while tools are still offered. Omit `tools` entirely for this call instead.", "validation");
2291
+ return { tool: { name: toolChoice.function.name } };
2292
+ }
2293
+ /**
2294
+ * Translates one VernLLM wire message into Converse's
2295
+ * `{ role: 'user' | 'assistant', content }` shape.
2296
+ */
2297
+ function toBedrockMessage(m) {
2298
+ if (m.role === "tool") return {
2299
+ role: "user",
2300
+ content: [{ toolResult: {
2301
+ toolUseId: m.tool_call_id,
2302
+ content: [{ text: m.content }],
2303
+ status: m.is_error ? "error" : "success"
2304
+ } }]
2305
+ };
2306
+ if (m.role === "assistant" && m.tool_calls?.length) {
2307
+ const blocks = [];
2308
+ if (m.content) blocks.push({ text: m.content });
2309
+ for (const tc of m.tool_calls) {
2310
+ let input;
2311
+ if (!tc.function.arguments.trim()) input = {};
2312
+ else try {
2313
+ input = JSON.parse(tc.function.arguments);
2314
+ } catch (cause) {
2315
+ throw new LLMError(`Assistant tool call "${tc.function.name}" (${tc.id}) has arguments that are not valid JSON.`, "validation", void 0, void 0, cause);
991
2316
  }
2317
+ blocks.push({ toolUse: {
2318
+ toolUseId: tc.id,
2319
+ name: tc.function.name,
2320
+ input
2321
+ } });
2322
+ }
2323
+ return {
2324
+ role: "assistant",
2325
+ content: blocks
992
2326
  };
993
- } } } };
2327
+ }
2328
+ return {
2329
+ role: m.role,
2330
+ content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content ?? "" }]
2331
+ };
2332
+ }
2333
+ /**
2334
+ * Converse expects the results of everything the model asked for in one
2335
+ * turn to arrive together as multiple `toolResult` content blocks on a
2336
+ * single `'user'` message, not as separate consecutive `'user'` messages.
2337
+ * The per-wire-message mapping above produces one `'user'` message per
2338
+ * VernLLM wire tool message, so when an assistant turn requested more than
2339
+ * one tool, this merges the resulting run of toolResult-only `'user'`
2340
+ * messages back into one.
2341
+ */
2342
+ function mergeConsecutiveToolResults(messages) {
2343
+ const isToolResultOnly = (m) => m.role === "user" && m.content.length > 0 && m.content.every((b) => "toolResult" in b);
2344
+ const merged = [];
2345
+ for (const m of messages) {
2346
+ const prev = merged.at(-1);
2347
+ if (isToolResultOnly(m) && prev && isToolResultOnly(prev)) prev.content.push(...m.content);
2348
+ else merged.push(m);
2349
+ }
2350
+ return merged;
994
2351
  }
995
2352
 
996
2353
  //#endregion
997
2354
  //#region src/adapters/fetch.ts
998
2355
  /**
2356
+ * Wraps a WHATWG `ReadableStream` (what `response.body` is) so it can be
2357
+ * consumed with `for await`. Implemented via `getReader()` rather than
2358
+ * relying on `ReadableStream` having a native `Symbol.asyncIterator`,
2359
+ * that support varies across runtimes/versions, and this works everywhere
2360
+ * a `ReadableStream` does.
2361
+ */
2362
+ async function* webStreamToAsyncIterable(stream) {
2363
+ const reader = stream.getReader();
2364
+ try {
2365
+ for (;;) {
2366
+ const { done, value } = await reader.read();
2367
+ if (done) return;
2368
+ if (value) yield value;
2369
+ }
2370
+ } finally {
2371
+ try {
2372
+ await reader.cancel();
2373
+ } catch {}
2374
+ reader.releaseLock();
2375
+ }
2376
+ }
2377
+ /** Default `requestStream`: native `fetch`, with the same error/`.status` contract non-streaming errors get. */
2378
+ async function defaultRequestStream(url, init) {
2379
+ const res = await fetch(url, init);
2380
+ if (!res.ok) {
2381
+ const body = await res.text().catch(() => "");
2382
+ const err = new Error(`Fetch adapter stream request failed (${res.status}): ${body.slice(0, 500)}`);
2383
+ err.status = res.status;
2384
+ err.headers = res.headers;
2385
+ throw err;
2386
+ }
2387
+ if (!res.body) throw new Error("Fetch adapter stream request received a response with no body.");
2388
+ return webStreamToAsyncIterable(res.body);
2389
+ }
2390
+ /** Builds the shared `{ method, headers, body? }` request-init for both `create` and `createStream`. */
2391
+ async function buildRequestInit(config, params, requestBody) {
2392
+ const url = typeof config.url === "function" ? config.url(params) : config.url;
2393
+ const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
2394
+ const method = config.method ?? "POST";
2395
+ const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
2396
+ return {
2397
+ url,
2398
+ method,
2399
+ headers: supportsBody ? {
2400
+ "Content-Type": "application/json",
2401
+ ...headers
2402
+ } : { ...headers },
2403
+ ...supportsBody ? { body: JSON.stringify(requestBody) } : {}
2404
+ };
2405
+ }
2406
+ /**
999
2407
  * A fetch-based escape hatch for providers with no SDK, or where pulling one
1000
2408
  * in isnt worth it. You supply the URL, headers, and two small mapping
1001
2409
  * functions; this handles the HTTP call and slots the result into the same
@@ -1005,41 +2413,99 @@ function fromBedrock(bedrockClient, options) {
1005
2413
  * Non-2xx responses throw an error with `.status` set to the HTTP status
1006
2414
  * code, so VernLLMs `nonRetryableStatus` handling (e.g. failing fast on
1007
2415
  * 401/403) applies here too
2416
+ *
2417
+ * Tool calling works the same way as every other adapter: `mapRequest`
2418
+ * receives the full `ChatRequest`, including `tools`/`toolChoice`, so it can
2419
+ * translate them into whatever shape the provider's wire format expects
2420
+ * (typically an OpenAI-`function`-wrapped `tools` array plus a `tool_choice`
2421
+ * field). On the way back, `mapResponse` may return a `toolCalls` array
2422
+ * (id/name/JSON-encoded-arguments-string per call) alongside or instead of
2423
+ * `content`; VernLLM parses and (if `argumentsSchema` was set) validates
2424
+ * those arguments the same way it does for every other adapter. For
2425
+ * `stream: true`, tool-call deltas go through the existing
2426
+ * `mapStreamEvent` seam via `WireStreamChunk`'s `tool_call_delta` variant,
2427
+ * no separate config is needed for streaming vs non-streaming tool calls.
2428
+ *
2429
+
2430
+ * `createStream` requires `mapStreamEvent` (there's no non-streaming
2431
+ * response to fall back on, unlike the other three optional streaming
2432
+ * seams). It opens the request via `requestStream` (defaults to native
2433
+ * `fetch`), splits the raw bytes into individual events via
2434
+ * `parseStreamFrames` (defaults to SSE framing, see `parseSseStream`),
2435
+ * and translates each event into `WireStreamChunk`(s) via
2436
+ * `mapStreamEvent`. Both seams are overridable per-config for providers
2437
+ * that don't fit the SSE-over-fetch default. If a custom `request`
2438
+ * transport is configured, `requestStream` must be configured too,
2439
+ * `requestStream` never silently falls back to `request` (see
2440
+ * `createStream`'s own comment for why), so a `stream: true` call with
2441
+ * `request` set but no `requestStream` throws a clear
2442
+ * `LLMError('validation')` instead of quietly using unrelated native
2443
+ * `fetch`.
1008
2444
  */
1009
2445
  function fromFetch(config) {
1010
- return { chat: { completions: { async create(params, options) {
1011
- const url = typeof config.url === "function" ? config.url(params) : config.url;
1012
- const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
1013
- const method = config.method ?? "POST";
1014
- const request = config.request ?? fetch;
1015
- const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
1016
- const res = await request(url, {
1017
- method,
1018
- headers: supportsBody ? {
1019
- "Content-Type": "application/json",
1020
- ...headers
1021
- } : { ...headers },
1022
- ...supportsBody ? { body: JSON.stringify(config.mapRequest(params)) } : {},
1023
- signal: options.signal
1024
- });
1025
- if (!res.ok) {
1026
- const body = await res.text().catch(() => "");
1027
- const err = new Error(`Fetch adapter request failed (${res.status}): ${body.slice(0, 500)}`);
1028
- err.status = res.status;
1029
- err.headers = res.headers;
1030
- throw err;
2446
+ return { chat: { completions: {
2447
+ async create(params, options) {
2448
+ const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
2449
+ const request = config.request ?? fetch;
2450
+ const res = await request(url, {
2451
+ method,
2452
+ headers,
2453
+ body,
2454
+ signal: options.signal
2455
+ });
2456
+ if (!res.ok) {
2457
+ const responseBody = await res.text().catch(() => "");
2458
+ const err = new Error(`Fetch adapter request failed (${res.status}): ${responseBody.slice(0, 500)}`);
2459
+ err.status = res.status;
2460
+ err.headers = res.headers;
2461
+ throw err;
2462
+ }
2463
+ const json = await res.json();
2464
+ const { content, usage, toolCalls } = config.mapResponse(json);
2465
+ const wireToolCalls = toolCalls?.length ? toolCalls.map((tc) => ({
2466
+ id: tc.id,
2467
+ type: "function",
2468
+ function: {
2469
+ name: tc.name,
2470
+ arguments: tc.arguments
2471
+ }
2472
+ })) : void 0;
2473
+ return {
2474
+ choices: [{ message: {
2475
+ content,
2476
+ ...wireToolCalls ? { tool_calls: wireToolCalls } : {}
2477
+ } }],
2478
+ usage: usage ? {
2479
+ prompt_tokens: usage.promptTokens,
2480
+ completion_tokens: usage.completionTokens,
2481
+ total_tokens: usage.totalTokens
2482
+ } : void 0
2483
+ };
2484
+ },
2485
+ async *createStream(params, options) {
2486
+ if (!config.mapStreamEvent) throw new LLMError("stream: true requires mapStreamEvent to be configured on fromFetch", "validation");
2487
+ if (config.request && !config.requestStream) throw new LLMError("`stream: true` requires `requestStream` to be configured on fromFetch when a custom `request` transport is set. `requestStream` does not fall back to `request` (it needs an async-iterable byte stream, which `RequestLike`'s buffered `ResponseLike` has no way to provide), without it, `stream: true` would silently use plain native `fetch` instead of your configured transport. Add a `requestStream` that opens the same connection your `request` does, or omit `request` if native `fetch` is fine for both.", "validation");
2488
+ const { url, method, headers, body } = await buildRequestInit(config, params, config.mapRequest(params));
2489
+ const requestStream = config.requestStream ?? defaultRequestStream;
2490
+ const parseFrames = config.parseStreamFrames ?? parseSseStream;
2491
+ const byteStream = await requestStream(url, {
2492
+ method,
2493
+ headers,
2494
+ body,
2495
+ signal: options.signal
2496
+ });
2497
+ for await (const event of parseFrames(byteStream)) {
2498
+ if (event === SSE_PING) {
2499
+ yield { type: "ping" };
2500
+ continue;
2501
+ }
2502
+ const wireChunks = config.mapStreamEvent(event);
2503
+ if (!wireChunks) continue;
2504
+ if (Array.isArray(wireChunks)) yield* wireChunks;
2505
+ else yield wireChunks;
2506
+ }
1031
2507
  }
1032
- const json = await res.json();
1033
- const { content, usage } = config.mapResponse(json);
1034
- return {
1035
- choices: [{ message: { content } }],
1036
- usage: usage ? {
1037
- prompt_tokens: usage.promptTokens,
1038
- completion_tokens: usage.completionTokens,
1039
- total_tokens: usage.totalTokens
1040
- } : void 0
1041
- };
1042
- } } } };
2508
+ } } };
1043
2509
  }
1044
2510
 
1045
2511
  //#endregion
@@ -1062,41 +2528,84 @@ function toOpenAIContent(blocks) {
1062
2528
  });
1063
2529
  }
1064
2530
  /**
1065
- * Adapter for any SDK/client whose `chat.completions.create` already
1066
- * matches the OpenAI wire format: this covers most hosted inference
1067
- * providers, since "OpenAI-compatible" is a de facto standard for chat
1068
- * completion APIs. Almost everything passes straight through untouched,
1069
- * this exists purely so call sites read clearly (`fromMistral(client)` vs
1070
- * handing a Mistral client to something typed for OpenAI) and so a real
1071
- * transformation could be added later, per-provider, without a breaking
1072
- * change.
1073
- *
1074
- * The one thing that isn't a pure passthrough: a `ContentBlock[]`
1075
- * `userContent` is translated into OpenAI's native `image_url` content-part
1076
- * shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
1077
- * rather than a copy of any one provider's wire format.
1078
- *
1079
- * Not every SDKs own TypeScript types line up exactly with `LLMClient`
1080
- * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
1081
- * the actual compatibility contract is the JSON each provider sends and
1082
- * receives over the wire, not the SDKs TS types.
2531
+ * Translates VernLLM's provider-agnostic `messages` (the one part of a
2532
+ * request that isn't a pure passthrough for OpenAI-compatible clients) into
2533
+ * OpenAI's native wire shape. Shared between `create` and `createStream` so
2534
+ * both go through identical message translation.
1083
2535
  */
1084
- function fromOpenAICompatible(client) {
1085
- const raw = client;
1086
- return { chat: { completions: { async create(params, options) {
1087
- const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
2536
+ function toOpenAIMessages(params) {
2537
+ return params.messages.map((m) => {
2538
+ if (m.role === "user" && Array.isArray(m.content)) return {
1088
2539
  ...m,
1089
2540
  content: toOpenAIContent(m.content)
1090
- } : m);
1091
- return raw.chat.completions.create({
1092
- ...params,
1093
- messages
1094
- }, options);
1095
- } } } };
2541
+ };
2542
+ if (m.role === "tool") {
2543
+ const { is_error: _isError,...openAIToolMessage } = m;
2544
+ return openAIToolMessage;
2545
+ }
2546
+ return m;
2547
+ });
2548
+ }
2549
+ /**
2550
+ * Translates one OpenAI-shaped SSE chunk into zero or more `WireStreamChunk`s.
2551
+ * A single chunk can carry a text delta, one or more tool-call argument
2552
+ * deltas (each keyed by `index`, OpenAI's own convention for streaming
2553
+ * parallel tool calls, mirrored directly by VernLLM's `tool_call_delta`
2554
+ * shape so accumulation composes without translation), and/or a final
2555
+ * usage block (present only when `stream_options.include_usage` is set,
2556
+ * which this adapter always sets).
2557
+ */
2558
+ function* toWireStreamChunks(chunk) {
2559
+ const delta = chunk.choices?.[0]?.delta;
2560
+ if (delta?.content) yield {
2561
+ type: "text-delta",
2562
+ delta: delta.content
2563
+ };
2564
+ if (delta?.tool_calls?.length) for (const toolCall of delta.tool_calls) yield {
2565
+ type: "tool_call_delta",
2566
+ index: toolCall.index,
2567
+ id: toolCall.id,
2568
+ name: toolCall.function?.name,
2569
+ argumentsDelta: toolCall.function?.arguments
2570
+ };
2571
+ if (chunk.usage) yield {
2572
+ type: "usage",
2573
+ usage: chunk.usage
2574
+ };
2575
+ }
2576
+ function fromOpenAICompatible(client, options = {}) {
2577
+ const raw = client;
2578
+ const { supportsStreamUsage = true } = options;
2579
+ const rawCreate = raw.chat.completions.create.bind(raw.chat.completions);
2580
+ return { chat: { completions: {
2581
+ async create(params, options$1) {
2582
+ const messages = toOpenAIMessages(params);
2583
+ return raw.chat.completions.create({
2584
+ ...params,
2585
+ messages
2586
+ }, options$1);
2587
+ },
2588
+ async *createStream(params, options$1) {
2589
+ const messages = toOpenAIMessages(params);
2590
+ const stream = await rawCreate({
2591
+ ...params,
2592
+ messages,
2593
+ stream: true,
2594
+ ...supportsStreamUsage ? { stream_options: { include_usage: true } } : {}
2595
+ }, options$1);
2596
+ for await (const chunk of stream) yield* toWireStreamChunks(chunk);
2597
+ }
2598
+ } } };
1096
2599
  }
1097
2600
  /** Groqs SDK matches the OpenAI wire format */
1098
2601
  const fromGroq = fromOpenAICompatible;
1099
- /** Mistrals `chat.completions`-shaped client (or their OpenAI-compat endpoint) */
2602
+ /**
2603
+ * Mistrals `chat.completions`-shaped client (or their OpenAI-compat
2604
+ * endpoint). Mistral supports `stream_options.include_usage` (added after
2605
+ * an earlier period where it returned a 422 for unrecognized fields, per
2606
+ * Mistral's changelog and streaming docs), so this is a plain alias like
2607
+ * the others, `supportsStreamUsage` defaults to `true`.
2608
+ */
1100
2609
  const fromMistral = fromOpenAICompatible;
1101
2610
  /** DeepSeeks API is OpenAI-compatible */
1102
2611
  const fromDeepSeek = fromOpenAICompatible;
@@ -1185,5 +2694,5 @@ const fromAtlasCloud = fromOpenAICompatible;
1185
2694
  const from01AI = fromOpenAICompatible;
1186
2695
 
1187
2696
  //#endregion
1188
- export { CircuitBreaker, ConsoleLogger, InMemoryCacheAdapter, LLMError, NormalizedCacheAdapter, TieredCacheAdapter, VernLLM, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError };
2697
+ export { CircuitBreaker, ConsoleLogger, InMemoryCacheAdapter, LLMError, NormalizedCacheAdapter, SSE_PING, TieredCacheAdapter, VernLLM, from01AI, fromAnthropic, fromAnyscale, fromAtlasCloud, fromBaseten, fromBedrock, fromCerebras, fromCloudflareWorkersAI, fromDeepInfra, fromDeepSeek, fromFeatherless, fromFetch, fromFireworks, fromFriendli, fromGemini, fromGitHubModels, fromGroq, fromHyperbolic, fromInferenceNet, fromInfermatic, fromKlusterAI, fromLMStudio, fromLambdaLabs, fromLepton, fromMiniMax, fromMistral, fromMoonshot, fromNebius, fromNovita, fromNvidiaNIM, fromOllama, fromOpenAICompatible, fromOpenRouter, fromParasail, fromPerplexity, fromSambaNova, fromSiliconFlow, fromSnowflakeCortex, fromStepFun, fromTogether, fromVLLM, fromVercelAIGateway, fromXAI, fromZhipu, isLLMError, isToolCallResult, parseSseStream };
1189
2698
  //# sourceMappingURL=index.js.map