@adaptic/utils 0.0.1037 → 0.0.1038

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -72367,11 +72367,34 @@ const alpaca = {
72367
72367
  * traffic can neither trip nor be tripped by traffic on the other side of the
72368
72368
  * PD-9 boundary.
72369
72369
  *
72370
+ * How long a tripped breaker stays open depends on WHY it tripped. A provider
72371
+ * that is refusing work because it is momentarily full ("model busy", 429,
72372
+ * 503, 529, or a leg that ran out its budget queued behind other traffic) is
72373
+ * shedding load it expects to take back within seconds; excluding it for a
72374
+ * full minute pushes every call onto the next leg, which is how one provider's
72375
+ * capacity blip becomes the last leg's overload. Such a run opens for the
72376
+ * shorter `capacity_cooldown_ms`. A run that contains any hard failure (a
72377
+ * rejected credential, a malformed request, an unreachable gateway, an answer
72378
+ * that does not parse) says the route is broken rather than busy, and opens
72379
+ * for the full `cooldown_ms`. Either way the route then admits a bounded number
72380
+ * of half-open probes, and one success closes it.
72381
+ *
72370
72382
  * The clock is injected. Breaker behaviour is entirely about elapsed time, and
72371
72383
  * a test that must sleep to observe a cooldown is a test nobody runs.
72372
72384
  *
72373
72385
  * @module llm/circuit-breaker
72374
72386
  */
72387
+ /**
72388
+ * @returns A record for a route with no failures on file.
72389
+ */
72390
+ function freshRecord() {
72391
+ return {
72392
+ consecutiveFailures: 0,
72393
+ openedAtMs: null,
72394
+ probesInFlight: 0,
72395
+ runHasHardFailure: false,
72396
+ };
72397
+ }
72375
72398
  /**
72376
72399
  * Tracks route health and decides whether a leg may be attempted.
72377
72400
  */
@@ -72404,7 +72427,26 @@ class CircuitBreakerRegistry {
72404
72427
  return "closed";
72405
72428
  }
72406
72429
  const elapsed = this.now() - record.openedAtMs;
72407
- return elapsed >= this.config.cooldown_ms ? "half-open" : "open";
72430
+ return elapsed >= this.cooldownFor(record) ? "half-open" : "open";
72431
+ }
72432
+ /**
72433
+ * The cooldown a record's current run earns.
72434
+ *
72435
+ * A run made only of capacity failures earns the capacity cooldown; one hard
72436
+ * failure anywhere in the run earns the full one. Mixed evidence is read as
72437
+ * the worse case, because a route that is both busy and broken is broken.
72438
+ * The capacity cooldown is never allowed to exceed the full one, so a
72439
+ * misconfigured table cannot make busy routes wait longer than broken ones.
72440
+ *
72441
+ * @param record The route's record.
72442
+ * @returns The cooldown in milliseconds.
72443
+ */
72444
+ cooldownFor(record) {
72445
+ const capacityCooldown = this.config.capacity_cooldown_ms ?? this.config.cooldown_ms;
72446
+ if (record.runHasHardFailure) {
72447
+ return this.config.cooldown_ms;
72448
+ }
72449
+ return Math.min(capacityCooldown, this.config.cooldown_ms);
72408
72450
  }
72409
72451
  /**
72410
72452
  * Whether a route may be attempted now.
@@ -72475,11 +72517,7 @@ class CircuitBreakerRegistry {
72475
72517
  * @returns void
72476
72518
  */
72477
72519
  onSuccess(routeKey) {
72478
- this.records.set(routeKey, {
72479
- consecutiveFailures: 0,
72480
- openedAtMs: null,
72481
- probesInFlight: 0,
72482
- });
72520
+ this.records.set(routeKey, freshRecord());
72483
72521
  }
72484
72522
  /**
72485
72523
  * Record a failure, opening the breaker once the threshold is reached.
@@ -72488,13 +72526,19 @@ class CircuitBreakerRegistry {
72488
72526
  * re-accumulate the threshold: the probe was the test, and it failed.
72489
72527
  *
72490
72528
  * @param routeKey The route's stable key.
72529
+ * @param kind Whether the failure was a capacity signal or a hard failure.
72530
+ * Defaults to `hard`, so a caller that cannot tell gets the longer, safer
72531
+ * cooldown.
72491
72532
  * @returns void
72492
72533
  */
72493
- onFailure(routeKey) {
72534
+ onFailure(routeKey, kind = "hard") {
72494
72535
  const wasHalfOpen = this.stateOf(routeKey) === "half-open";
72495
72536
  const record = this.recordFor(routeKey);
72496
72537
  record.probesInFlight = 0;
72497
72538
  record.consecutiveFailures += 1;
72539
+ if (kind === "hard") {
72540
+ record.runHasHardFailure = true;
72541
+ }
72498
72542
  if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
72499
72543
  record.openedAtMs = this.now();
72500
72544
  }
@@ -72506,17 +72550,19 @@ class CircuitBreakerRegistry {
72506
72550
  * @returns A snapshot.
72507
72551
  */
72508
72552
  snapshot(routeKey) {
72509
- const record = this.records.get(routeKey) ?? {
72510
- consecutiveFailures: 0,
72511
- openedAtMs: null,
72512
- probesInFlight: 0,
72513
- };
72553
+ const record = this.records.get(routeKey) ?? freshRecord();
72514
72554
  return {
72515
72555
  routeKey,
72516
72556
  state: this.stateOf(routeKey),
72517
72557
  consecutiveFailures: record.consecutiveFailures,
72518
72558
  openedAtMs: record.openedAtMs,
72519
72559
  probesInFlight: record.probesInFlight,
72560
+ failureKind: record.consecutiveFailures === 0
72561
+ ? null
72562
+ : record.runHasHardFailure
72563
+ ? "hard"
72564
+ : "capacity",
72565
+ cooldownMs: this.cooldownFor(record),
72520
72566
  };
72521
72567
  }
72522
72568
  /**
@@ -72544,7 +72590,7 @@ class CircuitBreakerRegistry {
72544
72590
  recordFor(routeKey) {
72545
72591
  let record = this.records.get(routeKey);
72546
72592
  if (record === undefined) {
72547
- record = { consecutiveFailures: 0, openedAtMs: null, probesInFlight: 0 };
72593
+ record = freshRecord();
72548
72594
  this.records.set(routeKey, record);
72549
72595
  }
72550
72596
  return record;
@@ -72820,7 +72866,17 @@ var providers$1 = {
72820
72866
  max_concurrent: 12,
72821
72867
  acquire_timeout_ms: 15000,
72822
72868
  source: null,
72823
- note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published."
72869
+ note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published.",
72870
+ models: {
72871
+ "claude-haiku-4-5": {
72872
+ basis: "conservative-default",
72873
+ requests_per_minute: 300,
72874
+ max_concurrent: 48,
72875
+ acquire_timeout_ms: 10000,
72876
+ source: null,
72877
+ note: "Raised 2026-09-28 for the closed-incumbent role this model holds on llm.fast, llm.decide and llm.extract. Measured that day: when DeepInfra was capacity-limited ('Model busy, retry later', timeouts at the 30000 ms hot-path budget), both DeepInfra legs of llm.fast opened their breakers and ALL of the alias's traffic fell to this model, where the 12-permit guard could not absorb it. In the 12 minutes to 15:33Z the incumbent leg recorded 18 timeouts, 15 breaker-open and 11 guard refusals ('did not admit the call within 15000 ms'), and the chain exhausted 20-45 times per 5 minutes. Sizing: in-flight demand is arrival rate times leg duration (Little's law), and the leg duration that matters is the worst case the chain permits, 30 s. 12 permits at 30 s serve 24 calls/min; 48 permits serve 96 calls/min at 30 s and up to the 300/min rate bound at haiku's normal single-digit-second latency, which covers the diverted llm.fast load plus normal llm.decide/llm.extract volume on the same guard. requests_per_minute is raised with it (the two bind in series) to 300, which is 30% of the 1,000 requests per minute per model that the scope_source lists for the lowest standard tier, so it remains below any standard tier's ceiling; no Anthropic 429 was observed in the window. acquire_timeout_ms is cut from 15000 to 10000 so an admitted call keeps at least 20 s of its 30 s leg budget: a call admitted after a 15 s wait had half its budget left, timed out at the provider, and that timeout was charged to this route's breaker, which then refused every call for the cooldown. STILL conservative-default: the organisation's tier has not been transcribed from the Claude Console (W3-07). Other Anthropic models keep the provider numbers."
72878
+ }
72879
+ }
72824
72880
  },
72825
72881
  openai: {
72826
72882
  basis: "conservative-default",
@@ -72929,18 +72985,42 @@ const SECONDS_PER_MINUTE = 60;
72929
72985
  const TOKENS_PER_REQUEST = 1;
72930
72986
  const config$1 = limitsConfig;
72931
72987
  /**
72932
- * Resolve the limits that apply to a provider.
72988
+ * Resolve the limits that apply to a provider, or to one of its models.
72933
72989
  *
72934
72990
  * An unregistered provider falls back to the conservative defaults rather than
72935
72991
  * to no limit at all. Treating "unknown" as "unlimited" would make every newly
72936
72992
  * onboarded provider the one most likely to be over-driven, which is exactly
72937
72993
  * backwards: a new provider is the one whose real ceiling is least understood.
72938
72994
  *
72995
+ * A model with an override on a per-model provider runs at the override's
72996
+ * numbers and provenance; every other model runs at the provider's. The
72997
+ * override is ignored for a provider scoped as a whole, because that provider
72998
+ * has one guard and a per-model number cannot be enforced on it.
72999
+ *
72939
73000
  * @param provider The provider key.
73001
+ * @param modelId The model, when the caller knows which one it addresses.
72940
73002
  * @returns Its limits.
72941
73003
  */
72942
- function limitsFor(provider) {
72943
- return config$1.providers[provider] ?? config$1.defaults;
73004
+ function limitsFor(provider, modelId) {
73005
+ const limits = config$1.providers[provider] ?? config$1.defaults;
73006
+ if (limits.scope !== "model" || modelId === undefined || modelId.length === 0) {
73007
+ return limits;
73008
+ }
73009
+ const override = limits.models?.[modelId];
73010
+ if (override === undefined) {
73011
+ return limits;
73012
+ }
73013
+ const { models: _siblings, ...providerLimits } = limits;
73014
+ return {
73015
+ ...providerLimits,
73016
+ basis: override.basis,
73017
+ requests_per_minute: override.requests_per_minute,
73018
+ requests_per_minute_basis: override.requests_per_minute_basis,
73019
+ max_concurrent: override.max_concurrent,
73020
+ acquire_timeout_ms: override.acquire_timeout_ms,
73021
+ source: override.source ?? null,
73022
+ note: override.note,
73023
+ };
72944
73024
  }
72945
73025
  /** Every provider with a recorded limit, plus whether it is published or a default. */
72946
73026
  function limitsInventory() {
@@ -73136,7 +73216,7 @@ const guardIdentities = new Map();
73136
73216
  function rateLimiterFor(identity) {
73137
73217
  let limiter = rateLimiters.get(identity.key);
73138
73218
  if (limiter === undefined) {
73139
- const limits = limitsFor(identity.provider);
73219
+ const limits = limitsFor(identity.provider, identity.modelId);
73140
73220
  limiter = new TokenBucketRateLimiter({
73141
73221
  maxTokens: limits.requests_per_minute,
73142
73222
  refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
@@ -73157,7 +73237,7 @@ function rateLimiterFor(identity) {
73157
73237
  function concurrencyGateFor(identity) {
73158
73238
  let gate = concurrencyGates.get(identity.key);
73159
73239
  if (gate === undefined) {
73160
- gate = new ConcurrencyGate(identity, limitsFor(identity.provider).max_concurrent);
73240
+ gate = new ConcurrencyGate(identity, limitsFor(identity.provider, identity.modelId).max_concurrent);
73161
73241
  concurrencyGates.set(identity.key, gate);
73162
73242
  guardIdentities.set(identity.key, identity);
73163
73243
  }
@@ -73187,8 +73267,8 @@ function concurrencyGateFor(identity) {
73187
73267
  * or the caller stopped waiting first.
73188
73268
  */
73189
73269
  async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
73190
- const limits = limitsFor(provider);
73191
73270
  const identity = guardIdentity(provider, scope.modelId);
73271
+ const limits = limitsFor(provider, identity.modelId);
73192
73272
  const waitBudgetMs = maxWaitMs === undefined
73193
73273
  ? limits.acquire_timeout_ms
73194
73274
  : Math.min(maxWaitMs, limits.acquire_timeout_ms);
@@ -73220,7 +73300,7 @@ function guardSnapshots() {
73220
73300
  return [...guardIdentities.values()]
73221
73301
  .sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))
73222
73302
  .map((identity) => {
73223
- const limits = limitsFor(identity.provider);
73303
+ const limits = limitsFor(identity.provider, identity.modelId);
73224
73304
  const limiter = rateLimiters.get(identity.key);
73225
73305
  const gate = concurrencyGates.get(identity.key);
73226
73306
  return {
@@ -73540,6 +73620,38 @@ async function runLeg(leg, params, execution, budgetMs) {
73540
73620
  execution.callerSignal?.removeEventListener("abort", forwardAbort);
73541
73621
  }
73542
73622
  }
73623
+ /**
73624
+ * HTTP statuses a provider (or the gateway relaying it) uses to say it is full
73625
+ * rather than that the request or the route is wrong: request timeout, too
73626
+ * early, too many requests, service unavailable, and Anthropic's overloaded.
73627
+ */
73628
+ const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
73629
+ /**
73630
+ * Wording providers use for a capacity refusal when the status is lost on the
73631
+ * way (a relayed body, a client library's own error). DeepInfra's is
73632
+ * "Model busy, retry later"; Anthropic's is "Overloaded".
73633
+ */
73634
+ const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
73635
+ /**
73636
+ * Whether a failure is the provider saying it is full rather than broken.
73637
+ *
73638
+ * Read by shape rather than by class, because the same signal reaches the
73639
+ * chain from more than one transport and not every transport's error class is
73640
+ * importable here.
73641
+ *
73642
+ * @param error The thrown value.
73643
+ * @param reason Its message.
73644
+ * @returns Whether it is a capacity signal.
73645
+ */
73646
+ function isCapacitySignal(error, reason) {
73647
+ if (typeof error === "object" && error !== null) {
73648
+ const status = error.status;
73649
+ if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
73650
+ return true;
73651
+ }
73652
+ }
73653
+ return CAPACITY_WORDING.test(reason);
73654
+ }
73543
73655
  /**
73544
73656
  * Classify why a leg failed.
73545
73657
  *
@@ -73548,6 +73660,13 @@ async function runLeg(leg, params, execution, budgetMs) {
73548
73660
  * cancellation as a provider failure would let a burst of user-cancelled
73549
73661
  * requests open the breaker on a perfectly healthy route.
73550
73662
  *
73663
+ * Among failures that do count, a capacity signal (the provider said it is
73664
+ * busy, or the leg ran out its budget waiting on it) is told apart from a hard
73665
+ * failure so the breaker can re-admit a busy route sooner than a broken one. A
73666
+ * timeout is read as capacity: on a reachable provider it is what a full queue
73667
+ * looks like from outside, and a provider that is actually down still costs no
73668
+ * more than one probe per capacity cooldown.
73669
+ *
73551
73670
  * @param error The thrown value.
73552
73671
  * @param callerSignal The caller's cancellation signal, if any.
73553
73672
  * @returns The outcome and whether it counts against route health.
@@ -73558,24 +73677,45 @@ function classify(error, callerSignal) {
73558
73677
  outcome: "skipped",
73559
73678
  reason: "caller cancelled",
73560
73679
  countsAgainstHealth: false,
73680
+ failureKind: "hard",
73561
73681
  };
73562
73682
  }
73563
73683
  if (error instanceof LegTimeoutError) {
73564
- return { outcome: "timeout", reason: error.message, countsAgainstHealth: true };
73684
+ return {
73685
+ outcome: "timeout",
73686
+ reason: error.message,
73687
+ countsAgainstHealth: true,
73688
+ failureKind: "capacity",
73689
+ };
73565
73690
  }
73566
73691
  if (error instanceof UnsupportedCapabilityError) {
73567
- return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
73692
+ return {
73693
+ outcome: "skipped",
73694
+ reason: error.message,
73695
+ countsAgainstHealth: false,
73696
+ failureKind: "hard",
73697
+ };
73568
73698
  }
73569
73699
  if (error instanceof ToolChoiceIgnoredError) {
73570
73700
  // The route answered; it broke a declared guarantee rather than failing to
73571
73701
  // be available, so its breaker is not charged for it.
73572
- return { outcome: "error", reason: error.message, countsAgainstHealth: false };
73702
+ return {
73703
+ outcome: "error",
73704
+ reason: error.message,
73705
+ countsAgainstHealth: false,
73706
+ failureKind: "hard",
73707
+ };
73573
73708
  }
73709
+ // Self-inflicted pacing, not provider ill-health. Counting it would let the
73710
+ // client's own throttling open a breaker on a perfectly healthy provider and
73711
+ // permanently reroute traffic nobody chose to reroute.
73574
73712
  if (error instanceof RateGuardTimeoutError) {
73575
- // Self-inflicted pacing, not provider ill-health. Counting it would let the
73576
- // client's own throttling open a breaker on a perfectly healthy provider
73577
- // and permanently reroute traffic nobody chose to reroute.
73578
- return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
73713
+ return {
73714
+ outcome: "skipped",
73715
+ reason: error.message,
73716
+ countsAgainstHealth: false,
73717
+ failureKind: "hard",
73718
+ };
73579
73719
  }
73580
73720
  const reason = error instanceof Error ? error.message : String(error);
73581
73721
  if (/abort/i.test(reason)) {
@@ -73583,9 +73723,20 @@ function classify(error, callerSignal) {
73583
73723
  outcome: "timeout",
73584
73724
  reason: `aborted: ${reason}`,
73585
73725
  countsAgainstHealth: true,
73726
+ failureKind: "capacity",
73586
73727
  };
73587
73728
  }
73588
- return { outcome: "error", reason, countsAgainstHealth: true };
73729
+ if (error instanceof LlmResponseFormatError) {
73730
+ // The provider answered, badly. That is a route defect, not a full queue,
73731
+ // whatever words the unparseable content happens to contain.
73732
+ return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
73733
+ }
73734
+ return {
73735
+ outcome: "error",
73736
+ reason,
73737
+ countsAgainstHealth: true,
73738
+ failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
73739
+ };
73589
73740
  }
73590
73741
  /**
73591
73742
  * Whether the caller has stopped waiting.
@@ -73704,9 +73855,9 @@ async function executeChain(alias, execution) {
73704
73855
  return { response, servedBy: route, attempts, totalUsage };
73705
73856
  }
73706
73857
  catch (error) {
73707
- const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
73858
+ const { outcome, reason, countsAgainstHealth, failureKind } = classify(error, execution.callerSignal);
73708
73859
  if (countsAgainstHealth) {
73709
- execution.breakers.onFailure(route.routeKey);
73860
+ execution.breakers.onFailure(route.routeKey, failureKind);
73710
73861
  }
73711
73862
  else if (holdsProbe) {
73712
73863
  // No verdict on the route's health, but the probe slot this attempt
@@ -73754,6 +73905,7 @@ var defaults = {
73754
73905
  circuit_breaker: {
73755
73906
  failure_threshold: 5,
73756
73907
  cooldown_ms: 60000,
73908
+ capacity_cooldown_ms: 15000,
73757
73909
  half_open_probes: 1
73758
73910
  }
73759
73911
  };