vern-llm 1.2.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -28,26 +28,43 @@ var CircuitBreaker = class {
28
28
  openedAt = 0;
29
29
  threshold;
30
30
  cooldownMs;
31
+ /**
32
+ * True while a single half-open trial call is in flight. Guards against
33
+ * multiple concurrent callers all treating themselves as "the" trial once
34
+ * the cooldown elapses
35
+ */
36
+ trialInFlight = false;
31
37
  constructor(options = {}) {
32
38
  this.threshold = options.threshold ?? 5;
33
39
  this.cooldownMs = options.cooldownMs ?? 3e4;
34
40
  }
35
- /** Throws if the circuit is open and the cooldown hasnt elapsed */
41
+ /**
42
+ * Throws if the circuit is open and the cooldown hasn't elapsed, or if
43
+ * the circuit is half-open and a trial call is already in flight.
44
+ * Otherwise, if the circuit just became eligible for a trial (cooldown
45
+ * elapsed, or half-open with no trial currently running), this call
46
+ * becomes that trial
47
+ */
36
48
  assertClosed() {
37
- if (this.state !== "open") return;
38
- const elapsed = Date.now() - this.openedAt;
39
- if (elapsed >= this.cooldownMs) {
49
+ if (this.state === "closed") return;
50
+ if (this.state === "open") {
51
+ const elapsed = Date.now() - this.openedAt;
52
+ if (elapsed < this.cooldownMs) throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
40
53
  this.state = "half-open";
54
+ this.trialInFlight = true;
41
55
  return;
42
56
  }
43
- throw new LLMError(`Circuit open — provider has failed ${this.consecutiveFailures} times in a row. Retry in ${Math.ceil((this.cooldownMs - elapsed) / 1e3)}s.`, "circuit_open");
57
+ if (this.trialInFlight) throw new LLMError("Circuit half-open. A trial request is already in flight. Try again shortly.", "circuit_open");
58
+ this.trialInFlight = true;
44
59
  }
45
60
  recordSuccess() {
46
61
  this.consecutiveFailures = 0;
47
62
  this.state = "closed";
63
+ this.trialInFlight = false;
48
64
  }
49
65
  recordFailure() {
50
66
  this.consecutiveFailures += 1;
67
+ this.trialInFlight = false;
51
68
  if (this.state === "half-open") {
52
69
  this.state = "open";
53
70
  this.openedAt = Date.now();
@@ -112,11 +129,40 @@ async function withTimeout(fn, timeoutMs, externalSignal) {
112
129
  }
113
130
  }
114
131
  /**
132
+ * Default cap (ms) for both exponential backoff and honored Retry-After
133
+ * values, so a misbehaving/adversarial Retry-After can't stall a caller
134
+ * indefinitely
135
+ */
136
+ const DEFAULT_MAX_DELAY_MS = 1e4;
137
+ /**
138
+ * Looks inside an unknown error value for a Retry-After header and
139
+ * converts it to milliseconds. Checks `.headers` first (fetch-style,
140
+ * Headers-like with `.get()`), then `.response.headers` (axios-style,
141
+ * plain object) since different client libraries surface headers
142
+ * differently. Supports both the delta-seconds form ("30") and the
143
+ * HTTP-date form ("Wed, 21 Oct 2015 07:28:00 GMT"). The result is capped
144
+ * at maxDelayMs. Returns undefined when no usable Retry-After is present
145
+ */
146
+ function extractRetryAfterMs(err, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
147
+ if (!err || typeof err !== "object") return void 0;
148
+ const error = err;
149
+ const headers = error.headers ?? error.response?.headers;
150
+ if (!headers || typeof headers !== "object") return void 0;
151
+ const getter = headers;
152
+ const raw = typeof getter.get === "function" ? getter.get("Retry-After") : Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.at(1);
153
+ if (typeof raw !== "string" || raw.trim() === "") return void 0;
154
+ const trimmed = raw.trim();
155
+ if (/^\d+$/.test(trimmed)) return Math.max(0, Math.min(Number(trimmed) * 1e3, maxDelayMs));
156
+ const dateMs = Date.parse(trimmed);
157
+ if (!Number.isNaN(dateMs)) return Math.max(0, Math.min(dateMs - Date.now(), maxDelayMs));
158
+ return void 0;
159
+ }
160
+ /**
115
161
  * Exponential backoff with jitter, capped at maxDelayMs.
116
162
  * Jitter avoids thundering-herd retries when many callers back off in lockstep,
117
163
  * the cap prevents unbounded delays when maxRetries is high
118
164
  */
119
- function getBackoffDelay(baseDelayMs, attempt, maxDelayMs = 1e4) {
165
+ function getBackoffDelay(baseDelayMs, attempt, maxDelayMs = DEFAULT_MAX_DELAY_MS) {
120
166
  const exp = Math.min(baseDelayMs * 2 ** attempt, maxDelayMs);
121
167
  return exp / 2 + Math.random() * (exp / 2);
122
168
  }
@@ -234,6 +280,7 @@ var VernLLM = class {
234
280
  defaultMaxTokens;
235
281
  cache;
236
282
  nonRetryableStatus;
283
+ inFlight = new Map();
237
284
  parseJson;
238
285
  onUsage;
239
286
  logger;
@@ -255,7 +302,9 @@ var VernLLM = class {
255
302
  this.nonRetryableStatus = options.nonRetryableStatus ?? [
256
303
  400,
257
304
  401,
258
- 403
305
+ 403,
306
+ 404,
307
+ 422
259
308
  ];
260
309
  this.parseJson = options.parseJson ?? defaultParseJson;
261
310
  this.onUsage = options.onUsage;
@@ -276,11 +325,11 @@ var VernLLM = class {
276
325
  }
277
326
  /**
278
327
  * Returns the caller supplied logger, or a console-based logger whose
279
- * debug output is gated by the `debug` option (defaulting to on
280
- * outside production)
328
+ * debug output is gated by the `debug` option (defaulting to off,
329
+ * so response content isn't unintentionally written to logs)
281
330
  */
282
331
  resolveLogger(options) {
283
- return options.logger ?? new ConsoleLogger(options.debug ?? process.env.NODE_ENV !== "production");
332
+ return options.logger ?? new ConsoleLogger(options.debug ?? false);
284
333
  }
285
334
  /**
286
335
  * Builds a circuit breaker if `circuitBreaker` is truthy on the
@@ -325,7 +374,7 @@ var VernLLM = class {
325
374
  async retryWithBackoff(fn, requestId, signal) {
326
375
  let lastError;
327
376
  for (let attempt = 0; attempt <= this.maxRetries; attempt++) try {
328
- if (attempt > 0) await this.recoverDelay(requestId, attempt, signal);
377
+ if (attempt > 0) await this.recoverDelay(requestId, attempt, lastError, signal);
329
378
  return await fn();
330
379
  } catch (error) {
331
380
  lastError = error;
@@ -473,12 +522,16 @@ var VernLLM = class {
473
522
  }
474
523
  /**
475
524
  * Waits out the backoff delay for a given retry attempt, logging the
476
- * attempt for observability before the wait begins. Rejects early if
477
- * the signal aborts during the wait
525
+ * attempt for observability before the wait begins. Honors a
526
+ * Retry-After header on the failed attempt's error when present
527
+ * (capped at the same maxDelayMs as backoff), otherwise falls back to
528
+ * exponential backoff exactly as before. Rejects early if the signal
529
+ * aborts during the wait
478
530
  */
479
- async recoverDelay(requestId, attempt, signal) {
480
- const delay = getBackoffDelay(this.baseDelayMs, attempt);
481
- this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms`);
531
+ async recoverDelay(requestId, attempt, error, signal) {
532
+ const retryAfterMs = extractRetryAfterMs(error);
533
+ const delay = retryAfterMs ?? getBackoffDelay(this.baseDelayMs, attempt);
534
+ this.logger.warn(`[vern:${requestId}] recovery attempt ${attempt}/${this.maxRetries}, waiting ${delay}ms` + (retryAfterMs !== void 0 ? " (honoring Retry-After)" : ""));
482
535
  await waitForRetry(delay, signal);
483
536
  }
484
537
  /**
@@ -507,26 +560,55 @@ var VernLLM = class {
507
560
  await this.cache.delete(key);
508
561
  }
509
562
  /**
510
- * Thin cache wrapper around caller supplied logic. `params.fn` is expected
511
- * to be a call that itself invokes `this.call(...)` (see `cachedLLMCall`
512
- * below for a convenience wrapper that wires this up automatically),
513
- * `cachedCall` does not itself apply retry/timeout policy.
563
+ * Cache wrapper around caller-supplied logic. `params.fn` should invoke
564
+ * `this.call(...)` (see `cachedLLMCall`); retry/timeout handling is left
565
+ * to the caller.
566
+ *
567
+ * Concurrent misses for the same `cacheKey` share a single in-flight call,
568
+ * avoiding cache stampedes. Each caller still receives its own
569
+ * `reserveUsage`/`refundUsage` callbacks with coalescing metadata.
514
570
  */
515
571
  async cachedCall(params) {
516
572
  const cached = await this.cache.get(params.cacheKey);
517
573
  if (cached.hit) return cached.value;
574
+ const existing = this.inFlight.get(params.cacheKey);
575
+ const coalesced = existing !== void 0;
576
+ const resultPromise = existing ?? this.registerTrigger(params, coalesced);
577
+ if (coalesced) return this.withRefundOnFailure(params, coalesced, async () => {
578
+ await params.reserveUsage?.({ coalesced });
579
+ return resultPromise;
580
+ });
581
+ return this.withRefundOnFailure(params, coalesced, () => resultPromise);
582
+ }
583
+ /** Starts the shared fn() call for a cache miss, reserving usage first, and registers it in the in-flight map until it settles */
584
+ registerTrigger(params, coalesced) {
585
+ const resultPromise = (async () => {
586
+ await params.reserveUsage?.({ coalesced });
587
+ return this.runAndCache(params);
588
+ })();
589
+ this.inFlight.set(params.cacheKey, resultPromise);
590
+ resultPromise.catch(() => {}).finally(() => {
591
+ this.inFlight.delete(params.cacheKey);
592
+ });
593
+ return resultPromise;
594
+ }
595
+ /** Runs `fn` and writes its result to the cache. Only ever called once per cacheKey per in-flight window, from registerTrigger */
596
+ async runAndCache(params) {
597
+ const result = await params.fn();
518
598
  try {
519
- await params.reserveUsage?.();
520
- const result = await params.fn();
521
- try {
522
- await this.cache.set(params.cacheKey, result, params.ttl);
523
- } catch (error) {
524
- this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
525
- }
526
- return result;
599
+ await this.cache.set(params.cacheKey, result, params.ttl);
600
+ } catch (error) {
601
+ this.logger.error("[VernLLM] cache write failed", { message: error instanceof Error ? error.message : "unknown" });
602
+ }
603
+ return result;
604
+ }
605
+ /** Awaits `run`, calling this caller's own refundUsage (tagged with whether it was coalesced) if it rejects, then rethrows the original error */
606
+ async withRefundOnFailure(params, coalesced, run) {
607
+ try {
608
+ return await run();
527
609
  } catch (error) {
528
610
  try {
529
- await params.refundUsage?.();
611
+ await params.refundUsage?.({ coalesced });
530
612
  } catch (refundError) {
531
613
  this.logger.error("[VernLLM] refundUsage failed", { message: refundError instanceof Error ? refundError.message : "unknown" });
532
614
  }
@@ -554,9 +636,53 @@ var VernLLM = class {
554
636
  }
555
637
  };
556
638
 
639
+ //#endregion
640
+ //#region src/internal/imageFormat.ts
641
+ /**
642
+ * MIME types accepted for `ImageBlock.mimeType` across all adapters. This is
643
+ * the intersection of what Anthropic, Gemini, OpenAI-compatible, and Bedrock
644
+ * Converse all natively support, so a `ContentBlock[]` that validates for
645
+ * one provider validates for all of them.
646
+ */
647
+ const SUPPORTED_IMAGE_MIME_TYPES = [
648
+ "image/png",
649
+ "image/jpeg",
650
+ "image/gif",
651
+ "image/webp"
652
+ ];
653
+ /**
654
+ * Validates an `ImageBlock.mimeType` against the shared supported set.
655
+ * Throws a non-retryable `LLMError('validation')`, since an unsupported
656
+ * mimeType is a permanent failure, retrying the same input can't fix it,
657
+ * the same way a schema-validation or JSON-parse failure isn't retried.
658
+ */
659
+ function assertSupportedImageMimeType(mimeType) {
660
+ if (SUPPORTED_IMAGE_MIME_TYPES.includes(mimeType)) return mimeType;
661
+ throw new LLMError(`Unsupported image mimeType "${mimeType}": expected one of ${SUPPORTED_IMAGE_MIME_TYPES.join(", ")}`, "validation");
662
+ }
663
+
557
664
  //#endregion
558
665
  //#region src/adapters/anthropic.ts
559
666
  /**
667
+ * Translates a VernLLM `ContentBlock[]` (our provider-agnostic multimodal
668
+ * shape) into Anthropic's native content-block array: text blocks pass
669
+ * through as-is, image blocks become `{ type: 'image', source: { type:
670
+ * 'base64', media_type, data } }`.
671
+ */
672
+ function toAnthropicContent(blocks) {
673
+ return blocks.map((block) => block.type === "image" ? {
674
+ type: "image",
675
+ source: {
676
+ type: "base64",
677
+ media_type: assertSupportedImageMimeType(block.mimeType),
678
+ data: block.data
679
+ }
680
+ } : {
681
+ type: "text",
682
+ text: block.text
683
+ });
684
+ }
685
+ /**
560
686
  * Wraps an Anthropic SDK client so it satisfies the same `LLMClient`
561
687
  * interface VernLLM uses for OpenAI/Groq.
562
688
  *
@@ -594,7 +720,7 @@ function fromAnthropic(anthropicClient) {
594
720
  system: system || void 0,
595
721
  messages: conversationMessages.map((m) => ({
596
722
  role: m.role,
597
- content: m.content
723
+ content: Array.isArray(m.content) ? toAnthropicContent(m.content) : m.content
598
724
  })),
599
725
  ...tools ? {
600
726
  tools,
@@ -623,6 +749,18 @@ function fromAnthropic(anthropicClient) {
623
749
  //#endregion
624
750
  //#region src/adapters/gemini.ts
625
751
  /**
752
+ * Translates a VernLLM `ContentBlock[]` into Gemini's native `parts` array:
753
+ * text blocks become `{ text }`, image blocks become inline data parts
754
+ * (`{ inlineData: { mimeType, data } }`), Geminis shape for embedding raw
755
+ * base64 image bytes directly in the request.
756
+ */
757
+ function toGeminiParts(blocks) {
758
+ return blocks.map((block) => block.type === "image" ? { inlineData: {
759
+ mimeType: assertSupportedImageMimeType(block.mimeType),
760
+ data: block.data
761
+ } } : { text: block.text });
762
+ }
763
+ /**
626
764
  * Wraps a Gemini client so it satisfies the `LLMClient` interface VernLLM
627
765
  * uses for OpenAI/Groq. Geminis shape differs on nearly every axis: a
628
766
  * `contents` array instead of `messages`, a separate `systemInstruction`
@@ -648,7 +786,7 @@ function fromGemini(geminiClient) {
648
786
  model: params.model,
649
787
  contents: conversationMessages.map((m) => ({
650
788
  role: m.role === "assistant" ? "model" : "user",
651
- parts: [{ text: m.content }]
789
+ parts: Array.isArray(m.content) ? toGeminiParts(m.content) : [{ text: m.content }]
652
790
  })),
653
791
  systemInstruction: systemMessage ? { parts: [{ text: systemMessage.content }] } : void 0,
654
792
  generationConfig
@@ -667,6 +805,36 @@ function fromGemini(geminiClient) {
667
805
 
668
806
  //#endregion
669
807
  //#region src/adapters/bedrock.ts
808
+ /** Maps a `ContentBlock` image MIME type, already validated, to Converse's `format` enum. */
809
+ function toBedrockImageFormat(mimeType) {
810
+ switch (assertSupportedImageMimeType(mimeType)) {
811
+ case "image/png": return "png";
812
+ case "image/jpeg": return "jpeg";
813
+ case "image/gif": return "gif";
814
+ case "image/webp": return "webp";
815
+ }
816
+ }
817
+ /**
818
+ * Decodes base64 image data into the raw `Uint8Array` bytes Converse's
819
+ * `image.source.bytes` expects (unlike Anthropic/Gemini/OpenAI, which all
820
+ * take base64 strings directly). Uses `Buffer`, since this adapter, like
821
+ * the rest of the package, targets Node.
822
+ */
823
+ function decodeBase64(data) {
824
+ return new Uint8Array(Buffer.from(data, "base64"));
825
+ }
826
+ /**
827
+ * Translates a VernLLM `ContentBlock[]` into Converse's native content-block
828
+ * array: text blocks pass through as `{ text }`, image blocks become
829
+ * `{ image: { format, source: { bytes } } }` with the base64 payload decoded
830
+ * to raw bytes, since Converse doesn't accept base64 strings directly.
831
+ */
832
+ function toBedrockContent(blocks) {
833
+ return blocks.map((block) => block.type === "image" ? { image: {
834
+ format: toBedrockImageFormat(block.mimeType),
835
+ source: { bytes: decodeBase64(block.data) }
836
+ } } : { text: block.text });
837
+ }
670
838
  /**
671
839
  * Wraps a Bedrock Converse-API client so it satisfies the `LLMClient`
672
840
  * interface VernLLM uses for OpenAI/Groq. The Converse API is unified
@@ -708,7 +876,7 @@ function fromBedrock(bedrockClient) {
708
876
  modelId: params.model,
709
877
  messages: conversationMessages.map((m) => ({
710
878
  role: m.role,
711
- content: [{ text: m.content }]
879
+ content: Array.isArray(m.content) ? toBedrockContent(m.content) : [{ text: m.content }]
712
880
  })),
713
881
  system: systemParts.length ? systemParts.map((text$1) => ({ text: text$1 })) : void 0,
714
882
  inferenceConfig: {
@@ -750,19 +918,23 @@ function fromFetch(config) {
750
918
  return { chat: { completions: { async create(params, options) {
751
919
  const url = typeof config.url === "function" ? config.url(params) : config.url;
752
920
  const headers = typeof config.headers === "function" ? await config.headers() : config.headers;
753
- const res = await fetch(url, {
754
- method: config.method ?? "POST",
755
- headers: {
921
+ const method = config.method ?? "POST";
922
+ const request = config.request ?? fetch;
923
+ const supportsBody = !["GET", "HEAD"].includes(method.toUpperCase());
924
+ const res = await request(url, {
925
+ method,
926
+ headers: supportsBody ? {
756
927
  "Content-Type": "application/json",
757
928
  ...headers
758
- },
759
- body: JSON.stringify(config.mapRequest(params)),
929
+ } : { ...headers },
930
+ ...supportsBody ? { body: JSON.stringify(config.mapRequest(params)) } : {},
760
931
  signal: options.signal
761
932
  });
762
933
  if (!res.ok) {
763
934
  const body = await res.text().catch(() => "");
764
935
  const err = new Error(`Fetch adapter request failed (${res.status}): ${body.slice(0, 500)}`);
765
936
  err.status = res.status;
937
+ err.headers = res.headers;
766
938
  throw err;
767
939
  }
768
940
  const json = await res.json();
@@ -781,13 +953,36 @@ function fromFetch(config) {
781
953
  //#endregion
782
954
  //#region src/adapters/openaiCompatible.ts
783
955
  /**
784
- * Passthrough adapter for any SDK/client whose `chat.completions.create`
785
- * already matches the OpenAI wire format 1:1 : this covers most hosted
786
- * inference providers, since "OpenAI-compatible" is a de facto standard for
787
- * chat completion APIs. No transformation happens here, this exists purely
788
- * so call sites read clearly (`fromMistral(client)` vs handing a Mistral
789
- * client to something typed for OpenAI) and so a real transformation could
790
- * be added later, per-provider, without a breaking change.
956
+ * Translates a VernLLM `ContentBlock[]` into OpenAI's wire-level content
957
+ * array. Text blocks become `{ type: 'text', text }`; image blocks become
958
+ * `{ type: 'image_url', image_url: { url } }` with the base64 payload
959
+ * inlined as a `data:` URL, since our `ContentBlock` shape (`{ type:
960
+ * 'image', data, mimeType }`) is provider-agnostic and doesn't itself match
961
+ * OpenAI's wire format.
962
+ */
963
+ function toOpenAIContent(blocks) {
964
+ return blocks.map((block) => block.type === "image" ? {
965
+ type: "image_url",
966
+ image_url: { url: `data:${assertSupportedImageMimeType(block.mimeType)};base64,${block.data}` }
967
+ } : {
968
+ type: "text",
969
+ text: block.text
970
+ });
971
+ }
972
+ /**
973
+ * Adapter for any SDK/client whose `chat.completions.create` already
974
+ * matches the OpenAI wire format: this covers most hosted inference
975
+ * providers, since "OpenAI-compatible" is a de facto standard for chat
976
+ * completion APIs. Almost everything passes straight through untouched,
977
+ * this exists purely so call sites read clearly (`fromMistral(client)` vs
978
+ * handing a Mistral client to something typed for OpenAI) and so a real
979
+ * transformation could be added later, per-provider, without a breaking
980
+ * change.
981
+ *
982
+ * The one thing that isn't a pure passthrough: a `ContentBlock[]`
983
+ * `userContent` is translated into OpenAI's native `image_url` content-part
984
+ * shape, since VernLLM's `ContentBlock` is intentionally provider-agnostic
985
+ * rather than a copy of any one provider's wire format.
791
986
  *
792
987
  * Not every SDKs own TypeScript types line up exactly with `LLMClient`
793
988
  * (extra fields, stricter unions, etc.), so this takes `unknown` and casts:
@@ -795,7 +990,17 @@ function fromFetch(config) {
795
990
  * receives over the wire, not the SDKs TS types.
796
991
  */
797
992
  function fromOpenAICompatible(client) {
798
- return client;
993
+ const raw = client;
994
+ return { chat: { completions: { async create(params, options) {
995
+ const messages = params.messages.map((m) => m.role === "user" && Array.isArray(m.content) ? {
996
+ ...m,
997
+ content: toOpenAIContent(m.content)
998
+ } : m);
999
+ return raw.chat.completions.create({
1000
+ ...params,
1001
+ messages
1002
+ }, options);
1003
+ } } } };
799
1004
  }
800
1005
  /** Groqs SDK matches the OpenAI wire format */
801
1006
  const fromGroq = fromOpenAICompatible;