vern-llm 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -28,26 +28,43 @@ var CircuitBreaker = class {
28
28
  openedAt = 0;
29
29
  threshold;
30
30
  cooldownMs;
31
+ /**
32
+ * True while a single half-open trial call is in flight. Guards against
33
+ * multiple concurrent callers all treating themselves as "the" trial once
34
+ * the cooldown elapses
35
+ */
36
+ trialInFlight = false;
31
37
  constructor(options = {}) {
32
38
  this.threshold = options.threshold ?? 5;
33
39
  this.cooldownMs = options.cooldownMs ?? 3e4;
34
40
  }
35
- /** Throws if the circuit is open and the cooldown hasnt elapsed */
41
+ /**
42
+ * Throws if the circuit is open and the cooldown hasn't elapsed, or if
43
+ * the circuit is half-open and a trial call is already in flight.
44
+ * Otherwise, if the circuit just became eligible for a trial (cooldown
45
+ * elapsed, or half-open with no trial currently running), this call
46
+ * becomes that trial
47
+ */
36
48
  assertClosed() {
37
- if (this.state !== "open") return;
38
- const elapsed = Date.now() - this.openedAt;
39
- if (elapsed >= this.cooldownMs) {
49
+ if (this.state === "closed") return;
50
+ if (this.state === "open") {
51
+ const elapsed = Date.now() - this.openedAt;
52
+ if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
40
53
  this.state = "half-open";
54
+ this.trialInFlight = true;
41
55
  return;
42
56
  }
43
- throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
57
+ if (this.trialInFlight) throw new LLMError("Circuit half-open. A trial request is already in flight. Try again shortly.", "circuit_open");
58
+ this.trialInFlight = true;
44
59
  }
45
60
  recordSuccess() {
46
61
  this.consecutiveFailures = 0;
47
62
  this.state = "closed";
63
+ this.trialInFlight = false;
48
64
  }
49
65
  recordFailure() {
50
66
  this.consecutiveFailures += 1;
67
+ this.trialInFlight = false;
51
68
  if (this.state === "half-open") {
52
69
  this.state = "open";
53
70
  this.openedAt = Date.now();
@@ -112,11 +129,40 @@ async function withTimeout(fn, timeoutMs, externalSignal) {
112
129
  }
113
130
  }
114
131
  /**
132
+ * Default cap (ms) for both exponential backoff and honored Retry-After
133
+ * values, so a misbehaving/adversarial Retry-After can't stall a caller
134
+ * indefinitely
135
+ */
136
+ const DEFAULT_MAX_DELAY_MS = 1e4;
137
+ /**
138
+ * Looks inside an unknown error value for a Retry-After header and
139
+ * converts it to milliseconds. Checks `.headers` first (fetch-style,
140
+ * Headers-like with `.get()`), then `.response.headers` (axios-style,
141
+ * plain object) since different client libraries surface headers
142
+ * differently. Supports both the delta-seconds form ("30") and the
143
+ * HTTP-date form ("Wed, 21 Oct 2015 07:28:00 GMT"). The result is capped
144
+ * at maxDelayMs. Returns undefined when no usable Retry-After is present
145
+ */
146
+ function extractRetryAfterMs(err, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
147
+ if (!err || typeof err !== "object") return void 0;
148
+ const error = err;
149
+ const headers = error.headers ?? error.response?.headers;
150
+ if (!headers || typeof headers !== "object") return void 0;
151
+ const getter = headers;
152
+ const raw = typeof getter.get === "function" ? getter.get("Retry-After") : Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.at(1);
153
+ if (typeof raw !== "string" || raw.trim() === "") return void 0;
154
+ const trimmed = raw.trim();
155
+ if (/^\d+$/.test(trimmed)) return Math.max(0, Math.min(Number(trimmed) * 1e3, maxDelayMs));
156
+ const dateMs = Date.parse(trimmed);
157
+ if (!Number.isNaN(dateMs)) return Math.max(0, Math.min(dateMs - Date.now(), maxDelayMs));
158
+ return void 0;
159
+ }
160
+ /**
115
161
  * Exponential backoff with jitter, capped at maxDelayMs.
116
162
  * Jitter avoids thundering-herd retries when many callers back off in lockstep,
117
163
  * the cap prevents unbounded delays when maxRetries is high
118
164
  */
119
- function getBackoffDelay(baseDelayMs, attempt, maxDelayMs = 1e4) {
165
+ function getBackoffDelay(baseDelayMs, attempt, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
120
166
  const exp = Math.min(baseDelayMs * 2 ** attempt, maxDelayMs);
121
167
  return exp / 2 + Math.random() * (exp / 2);
122
168
  }
@@ -234,6 +280,7 @@ var VernLLM = class {
234
280
  defaultMaxTokens;
235
281
  cache;
236
282
  nonRetryableStatus;
283
+ inFlight = new Map();
237
284
  parseJson;
238
285
  onUsage;
239
286
  logger;
@@ -327,7 +374,7 @@ var VernLLM = class {
327
374
  async retryWithBackoff(fn, requestId, signal) {
328
375
  let lastError;
329
376
  for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
330
- if (attempt > 0) await this.recoverDelay(requestId, attempt, signal);
377
+ if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
331
378
  return await fn();
332
379
  } catch (error) {
333
380
  lastError = error;
@@ -475,12 +522,16 @@ var VernLLM = class {
475
522
  }
476
523
  /**
477
524
  * Waits out the backoff delay for a given retry attempt, logging the
478
- * attempt for observability before the wait begins. Rejects early if
479
- * the signal aborts during the wait
525
+ * attempt for observability before the wait begins. Honors a
526
+ * Retry-After header on the failed attempt's error when present
527
+ * (capped at the same maxDelayMs as backoff), otherwise falls back to
528
+ * exponential backoff exactly as before. Rejects early if the signal
529
+ * aborts during the wait
480
530
  */
481
- async recoverDelay(requestId, attempt, signal) {
482
- const delay = getBackoffDelay(this.baseDelayMs, attempt);
483
- this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms`);
531
+ async recoverDelay(requestId, attempt, error, signal) {
532
+ const retryAfterMs = extractRetryAfterMs(error);
533
+ const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
534
+ this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
484
535
  await waitForRetry(delay, signal);
485
536
  }
486
537
  /**
@@ -509,26 +560,55 @@ var VernLLM = class {
509
560
  await this.cache.delete(key);
510
561
  }
511
562
  /**
512
- * Thin cache wrapper around caller supplied logic. `params.fn` is expected
513
- * to be a call that itself invokes `this.call(...)` (see `cachedLLMCall`
514
- * below for a convenience wrapper that wires this up automatically),
515
- * `cachedCall` does not itself apply retry/timeout policy.
563
+ * Cache wrapper around caller-supplied logic. `params.fn` should invoke
564
+ * `this.call(...)` (see `cachedLLMCall`); retry/timeout handling is left
565
+ * to the caller.
566
+ *
567
+ * Concurrent misses for the same `cacheKey` share a single in-flight call,
568
+ * avoiding cache stampedes. Each caller still receives its own
569
+ * `reserveUsage`/`refundUsage` callbacks with coalescing metadata.
516
570
  */
517
571
  async cachedCall(params) {
518
572
  const cached = await this.cache.get(params.cacheKey);
519
573
  if (cached.hit) return cached.value;
574
+ const existing = this.inFlight.get(params.cacheKey);
575
+ const coalesced = existing !== void 0;
576
+ const resultPromise = existing ?? this.registerTrigger(params, coalesced);
577
+ if (coalesced) return this.withRefundOnFailure(params, coalesced, async () => {
578
+ await params.reserveUsage?.({ coalesced });
579
+ return resultPromise;
580
+ });
581
+ return this.withRefundOnFailure(params, coalesced, () => resultPromise);
582
+ }
583
+ /** Starts the shared fn() call for a cache miss, reserving usage first, and registers it in the in-flight map until it settles */
584
+ registerTrigger(params, coalesced) {
585
+ const resultPromise = (async () => {
586
+ await params.reserveUsage?.({ coalesced });
587
+ return this.runAndCache(params);
588
+ })();
589
+ this.inFlight.set(params.cacheKey, resultPromise);
590
+ resultPromise.catch(() => {}).finally(() => {
591
+ this.inFlight.delete(params.cacheKey);
592
+ });
593
+ return resultPromise;
594
+ }
595
+ /** Runs `fn` and writes its result to the cache. Only ever called once per cacheKey per in-flight window, from registerTrigger */
596
+ async runAndCache(params) {
597
+ const result = await params.fn();
520
598
  try {
521
- await params.reserveUsage?.();
522
- const result = await params.fn();
523
- try {
524
- await this.cache.set(params.cacheKey, result, params.ttl);
525
- } catch (error) {
526
- this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
527
- }
528
- return result;
599
+ await this.cache.set(params.cacheKey, result, params.ttl);
600
+ } catch (error) {
601
+ this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
602
+ }
603
+ return result;
604
+ }
605
+ /** Awaits `run`, calling this caller's own refundUsage (tagged with whether it was coalesced) if it rejects, then rethrows the original error */
606
+ async withRefundOnFailure(params, coalesced, run) {
607
+ try {
608
+ return await run();
529
609
  } catch (error) {
530
610
  try {
531
- await params.refundUsage?.();
611
+ await params.refundUsage?.({ coalesced });
532
612
  } catch (refundError) {
533
613
  this.logger.error("[VernLLM] refundUsage failed", { message: refundError instanceof Error ? refundError.message : "unknown" });
534
614
  }
@@ -556,9 +636,53 @@ var VernLLM = class {
556
636
  }
557
637
  };
558
638
 
639
+ //#endregion
640
+ //#region src/internal/imageFormat.ts
641
+ /**
642
+ * MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
643
+ * the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
644
+ * Converse all natively support, so a `ContentBlock[]` that validates for
645
+ * one provider validates for all of them.
646
+ */
647
+ const SUPPORTED_IMAGE_MIME_TYPES = [
648
+ "image/png",
649
+ "image/jpeg",
650
+ "image/gif",
651
+ "image/webp"
652
+ ];
653
+ /**
654
+ * Validates an `ImageBlock.mimeType` against the shared supported set.
655
+ * Throws a non-retryable `LLMError('validation')`, since an unsupported
656
+ * mimeType is a permanent failure, retrying the same input can't fix it,
657
+ * the same way a schema-validation or JSON-parse failure isn't retried.
658
+ */
659
+ function assertSupportedImageMimeType(mimeType) {
660
+ if (SUPPORTED_IMAGE_MIME_TYPES.includes(mimeType)) return mimeType;
661
+ throw new LLMError(`Unsupported image mimeType "${mimeType}": expected one of ${SUPPORTED_IMAGE_MIME_TYPES.join(", ")}`, "validation");
662
+ }
663
+
559
664
  //#endregion
560
665
  //#region src/adapters/anthropic.ts
561
666
  /**
667
+ * Translates a VernLLM `ContentBlock[]` (our provider-agnostic multimodal
668
+ * shape) into Anthropic's native content-block array: text blocks pass
669
+ * through as-is, image blocks become `{ type: 'image', source: { type:
670
+ * 'base64', media_type, data } }`.
671
+ */
672
+ function toAnthropicContent(blocks) {
673
+ return blocks.map((block) => block.type === "image" ? {
674
+ type: "image",
675
+ source: {
676
+ type: "base64",
677
+ media_type: assertSupportedImageMimeType(block.mimeType),
678
+ data: block.data
679
+ }
680
+ } : {
681
+ type: "text",
682
+ text: block.text
683
+ });
684
+ }
685
+ /**
562
686
  * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
563
687
  * interface VernLLM uses for OpenAI/Groq.
564
688
  *
@@ -596,7 +720,7 @@ function fromAnthropic(anthropicClient) {
596
720
  system: system || void 0,
597
721
  messages: conversationMessages.map((m) => ({
598
722
  role: m.role,
599
- content: m.content
723
+ content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content
600
724
  })),
601
725
  ...tools ? {
602
726
  tools,
@@ -625,6 +749,18 @@ function fromAnthropic(anthropicClient) {
625
749
  //#endregion
626
750
  //#region src/adapters/gemini.ts
627
751
  /**
752
+ * Translates a VernLLM `ContentBlock[]` into Gemini's native `parts` array:
753
+ * text blocks become `{ text }`, image blocks become inline data parts
754
+ * (`{ inlineData: { mimeType, data } }`), Geminis shape for embedding raw
755
+ * base64 image bytes directly in the request.
756
+ */
757
+ function toGeminiParts(blocks) {
758
+ return blocks.map((block) => block.type === "image" ? { inlineData: {
759
+ mimeType: assertSupportedImageMimeType(block.mimeType),
760
+ data: block.data
761
+ } } : { text: block.text });
762
+ }
763
+ /**
628
764
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
629
765
  * uses for OpenAI/Groq. Geminis shape differs on nearly every axis: a
630
766
  * `contents` array instead of `messages`, a separate `systemInstruction`
@@ -650,7 +786,7 @@ function fromGemini(geminiClient) {
650
786
  model: params.model,
651
787
  contents: conversationMessages.map((m) => ({
652
788
  role: m.role === "assistant" ? "model" : "user",
653
- parts: [{ text: m.content }]
789
+ parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content }]
654
790
  })),
655
791
  systemInstruction: systemMessage ? { parts: [{ text: systemMessage.content }] } : void 0,
656
792
  generationConfig
@@ -669,6 +805,36 @@ function fromGemini(geminiClient) {
669
805
 
670
806
  //#endregion
671
807
  //#region src/adapters/bedrock.ts
808
+ /** Maps a `ContentBlock` image MIME type, already validated, to Converse's `format` enum. */
809
+ function toBedrockImageFormat(mimeType) {
810
+ switch (assertSupportedImageMimeType(mimeType)) {
811
+ case "image/png": return "png";
812
+ case "image/jpeg": return "jpeg";
813
+ case "image/gif": return "gif";
814
+ case "image/webp": return "webp";
815
+ }
816
+ }
817
+ /**
818
+ * Decodes base64 image data into the raw `Uint8Array` bytes Converse's
819
+ * `image.source.bytes` expects (unlike Anthropic/Gemini/OpenAI, which all
820
+ * take base64 strings directly). Uses `Buffer`, since this adapter, like
821
+ * the rest of the package, targets Node.
822
+ */
823
+ function decodeBase64(data) {
824
+ return new Uint8Array(Buffer.from(data, "base64"));
825
+ }
826
+ /**
827
+ * Translates a VernLLM `ContentBlock[]` into Converse's native content-block
828
+ * array: text blocks pass through as `{ text }`, image blocks become
829
+ * `{ image: { format, source: { bytes } } }` with the base64 payload decoded
830
+ * to raw bytes, since Converse doesn't accept base64 strings directly.
831
+ */
832
+ function toBedrockContent(blocks) {
833
+ return blocks.map((block) => block.type === "image" ? { image: {
834
+ format: toBedrockImageFormat(block.mimeType),
835
+ source: { bytes: decodeBase64(block.data) }
836
+ } } : { text: block.text });
837
+ }
672
838
  /**
673
839
  * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
674
840
  * interface VernLLM uses for OpenAI/Groq. The Converse API is unified
@@ -710,7 +876,7 @@ function fromBedrock(bedrockClient) {
710
876
  modelId: params.model,
711
877
  messages: conversationMessages.map((m) => ({
712
878
  role: m.role,
713
- content: [{ text: m.content }]
879
+ content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content }]
714
880
  })),
715
881
  system: systemParts.length ? systemParts.map((text$1) => ({ text: text$1 })) : void 0,
716
882
  inferenceConfig: {
@@ -752,19 +918,23 @@ function fromFetch(config) {
752
918
  return { chat: { completions: { async create(params, options) {
753
919
  const url = typeof config.url === "function" ? config.url(params) : config.url;
754
920
  const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
755
- const res = await fetch(url, {
756
- method: config.method ?? "POST",
757
- headers: {
921
+ const method = config.method ?? "POST";
922
+ const request = config.request ?? fetch;
923
+ const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
924
+ const res = await request(url, {
925
+ method,
926
+ headers: supportsBody ? {
758
927
  "Content-Type": "application/json",
759
928
  ...headers
760
- },
761
- body: JSON.stringify(config.mapRequest(params)),
929
+ } : { ...headers },
930
+ ...supportsBody ? { body: JSON.stringify(config.mapRequest(params)) } : {},
762
931
  signal: options.signal
763
932
  });
764
933
  if (!res.ok) {
765
934
  const body = await res.text().catch(() => "");
766
935
  const err = new Error(`Fetch adapter request failed (${res.status}): ${body.slice(0, 500)}`);
767
936
  err.status = res.status;
937
+ err.headers = res.headers;
768
938
  throw err;
769
939
  }
770
940
  const json = await res.json();
@@ -783,13 +953,36 @@ function fromFetch(config) {
783
953
  //#endregion
784
954
  //#region src/adapters/openaiCompatible.ts
785
955
  /**
786
- * Passthrough adapter for any SDK/client whose `chat.completions.create`
787
- * already matches the OpenAI wire format 1:1 : this covers most hosted
788
- * inference providers, since "OpenAI-compatible" is a de facto standard for
789
- * chat completion APIs. No transformation happens here, this exists purely
790
- * so call sites read clearly (`fromMistral(client)` vs handing a Mistral
791
- * client to something typed for OpenAI) and so a real transformation could
792
- * be added later, per-provider, without a breaking change.
956
+ * Translates a VernLLM `ContentBlock[]` into OpenAI's wire-level content
957
+ * array. Text blocks become `{ type: 'text', text }`; image blocks become
958
+ * `{ type: 'image_url', image_url: { url } }` with the base64 payload
959
+ * inlined as a `data:` URL, since our `ContentBlock` shape (`{ type:
960
+ * 'image', data, mimeType }`) is provider-agnostic and doesn't itself match
961
+ * OpenAI's wire format.
962
+ */
963
+ function toOpenAIContent(blocks) {
964
+ return blocks.map((block) => block.type === "image" ? {
965
+ type: "image_url",
966
+ image_url: { url: `data:${assertSupportedImageMimeType(block.mimeType)};base64,${block.data}` }
967
+ } : {
968
+ type: "text",
969
+ text: block.text
970
+ });
971
+ }
972
+ /**
973
+ * Adapter for any SDK/client whose `chat.completions.create` already
974
+ * matches the OpenAI wire format: this covers most hosted inference
975
+ * providers, since "OpenAI-compatible" is a de facto standard for chat
976
+ * completion APIs. Almost everything passes straight through untouched,
977
+ * this exists purely so call sites read clearly (`fromMistral(client)` vs
978
+ * handing a Mistral client to something typed for OpenAI) and so a real
979
+ * transformation could be added later, per-provider, without a breaking
980
+ * change.
981
+ *
982
+ * The one thing that isn't a pure passthrough: a `ContentBlock[]`
983
+ * `userContent` is translated into OpenAI's native `image_url` content-part
984
+ * shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
985
+ * rather than a copy of any one provider's wire format.
793
986
  *
794
987
  * Not every SDKs own TypeScript types line up exactly with `LLMClient`
795
988
  * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
@@ -797,7 +990,17 @@ function fromFetch(config) {
797
990
  * receives over the wire, not the SDKs TS types.
798
991
  */
799
992
  function fromOpenAICompatible(client) {
800
- return client;
993
+ const raw = client;
994
+ return { chat: { completions: { async create(params, options) {
995
+ const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
996
+ ...m,
997
+ content: toOpenAIContent(m.content)
998
+ } : m);
999
+ return raw.chat.completions.create({
1000
+ ...params,
1001
+ messages
1002
+ }, options);
1003
+ } } } };
801
1004
  }
802
1005
  /** Groqs SDK matches the OpenAI wire format */
803
1006
  const fromGroq = fromOpenAICompatible;