@adaptic/utils 0.0.1036 → 0.0.1038
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +186 -32
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +186 -32
- package/dist/index.mjs.map +1 -1
- package/dist/types/llm/circuit-breaker.d.ts +45 -1
- package/dist/types/llm/circuit-breaker.d.ts.map +1 -1
- package/dist/types/llm/fallback-chain.d.ts +12 -0
- package/dist/types/llm/fallback-chain.d.ts.map +1 -1
- package/dist/types/llm/index.d.ts +2 -2
- package/dist/types/llm/index.d.ts.map +1 -1
- package/dist/types/llm/rate-guard.d.ts +29 -2
- package/dist/types/llm/rate-guard.d.ts.map +1 -1
- package/dist/types/llm/types.d.ts +6 -0
- package/dist/types/llm/types.d.ts.map +1 -1
- package/dist/types/massive.d.ts.map +1 -1
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -19764,7 +19764,9 @@ const fetchTrades = async (symbol, options) => {
|
|
|
19764
19764
|
await rateLimiters$1.massive.acquire();
|
|
19765
19765
|
const url = `${baseUrl}?${params.toString()}`;
|
|
19766
19766
|
try {
|
|
19767
|
-
|
|
19767
|
+
// Redact the apiKey query param before logging — the raw URL embeds the
|
|
19768
|
+
// live Massive API key (audit #595).
|
|
19769
|
+
logIfDebug(`Fetching trades for ${symbol} from ${url.replace(/([?&](?:apiKey|apikey|api_key)=)[^&]+/gi, "$1***")}`);
|
|
19768
19770
|
const response = await fetchWithRetry(url, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
19769
19771
|
const data = (await response.json());
|
|
19770
19772
|
if ("message" in data) {
|
|
@@ -72365,11 +72367,34 @@ const alpaca = {
|
|
|
72365
72367
|
* traffic can neither trip nor be tripped by traffic on the other side of the
|
|
72366
72368
|
* PD-9 boundary.
|
|
72367
72369
|
*
|
|
72370
|
+
* How long a tripped breaker stays open depends on WHY it tripped. A provider
|
|
72371
|
+
* that is refusing work because it is momentarily full ("model busy", 429,
|
|
72372
|
+
* 503, 529, or a leg that ran out its budget queued behind other traffic) is
|
|
72373
|
+
* shedding load it expects to take back within seconds; excluding it for a
|
|
72374
|
+
* full minute pushes every call onto the next leg, which is how one provider's
|
|
72375
|
+
* capacity blip becomes the last leg's overload. Such a run opens for the
|
|
72376
|
+
* shorter `capacity_cooldown_ms`. A run that contains any hard failure (a
|
|
72377
|
+
* rejected credential, a malformed request, an unreachable gateway, an answer
|
|
72378
|
+
* that does not parse) says the route is broken rather than busy, and opens
|
|
72379
|
+
* for the full `cooldown_ms`. Either way the route then admits a bounded number
|
|
72380
|
+
* of half-open probes, and one success closes it.
|
|
72381
|
+
*
|
|
72368
72382
|
* The clock is injected. Breaker behaviour is entirely about elapsed time, and
|
|
72369
72383
|
* a test that must sleep to observe a cooldown is a test nobody runs.
|
|
72370
72384
|
*
|
|
72371
72385
|
* @module llm/circuit-breaker
|
|
72372
72386
|
*/
|
|
72387
|
+
/**
|
|
72388
|
+
* @returns A record for a route with no failures on file.
|
|
72389
|
+
*/
|
|
72390
|
+
function freshRecord() {
|
|
72391
|
+
return {
|
|
72392
|
+
consecutiveFailures: 0,
|
|
72393
|
+
openedAtMs: null,
|
|
72394
|
+
probesInFlight: 0,
|
|
72395
|
+
runHasHardFailure: false,
|
|
72396
|
+
};
|
|
72397
|
+
}
|
|
72373
72398
|
/**
|
|
72374
72399
|
* Tracks route health and decides whether a leg may be attempted.
|
|
72375
72400
|
*/
|
|
@@ -72402,7 +72427,26 @@ class CircuitBreakerRegistry {
|
|
|
72402
72427
|
return "closed";
|
|
72403
72428
|
}
|
|
72404
72429
|
const elapsed = this.now() - record.openedAtMs;
|
|
72405
|
-
return elapsed >= this.
|
|
72430
|
+
return elapsed >= this.cooldownFor(record) ? "half-open" : "open";
|
|
72431
|
+
}
|
|
72432
|
+
/**
|
|
72433
|
+
* The cooldown a record's current run earns.
|
|
72434
|
+
*
|
|
72435
|
+
* A run made only of capacity failures earns the capacity cooldown; one hard
|
|
72436
|
+
* failure anywhere in the run earns the full one. Mixed evidence is read as
|
|
72437
|
+
* the worse case, because a route that is both busy and broken is broken.
|
|
72438
|
+
* The capacity cooldown is never allowed to exceed the full one, so a
|
|
72439
|
+
* misconfigured table cannot make busy routes wait longer than broken ones.
|
|
72440
|
+
*
|
|
72441
|
+
* @param record The route's record.
|
|
72442
|
+
* @returns The cooldown in milliseconds.
|
|
72443
|
+
*/
|
|
72444
|
+
cooldownFor(record) {
|
|
72445
|
+
const capacityCooldown = this.config.capacity_cooldown_ms ?? this.config.cooldown_ms;
|
|
72446
|
+
if (record.runHasHardFailure) {
|
|
72447
|
+
return this.config.cooldown_ms;
|
|
72448
|
+
}
|
|
72449
|
+
return Math.min(capacityCooldown, this.config.cooldown_ms);
|
|
72406
72450
|
}
|
|
72407
72451
|
/**
|
|
72408
72452
|
* Whether a route may be attempted now.
|
|
@@ -72473,11 +72517,7 @@ class CircuitBreakerRegistry {
|
|
|
72473
72517
|
* @returns void
|
|
72474
72518
|
*/
|
|
72475
72519
|
onSuccess(routeKey) {
|
|
72476
|
-
this.records.set(routeKey,
|
|
72477
|
-
consecutiveFailures: 0,
|
|
72478
|
-
openedAtMs: null,
|
|
72479
|
-
probesInFlight: 0,
|
|
72480
|
-
});
|
|
72520
|
+
this.records.set(routeKey, freshRecord());
|
|
72481
72521
|
}
|
|
72482
72522
|
/**
|
|
72483
72523
|
* Record a failure, opening the breaker once the threshold is reached.
|
|
@@ -72486,13 +72526,19 @@ class CircuitBreakerRegistry {
|
|
|
72486
72526
|
* re-accumulate the threshold: the probe was the test, and it failed.
|
|
72487
72527
|
*
|
|
72488
72528
|
* @param routeKey The route's stable key.
|
|
72529
|
+
* @param kind Whether the failure was a capacity signal or a hard failure.
|
|
72530
|
+
* Defaults to `hard`, so a caller that cannot tell gets the longer, safer
|
|
72531
|
+
* cooldown.
|
|
72489
72532
|
* @returns void
|
|
72490
72533
|
*/
|
|
72491
|
-
onFailure(routeKey) {
|
|
72534
|
+
onFailure(routeKey, kind = "hard") {
|
|
72492
72535
|
const wasHalfOpen = this.stateOf(routeKey) === "half-open";
|
|
72493
72536
|
const record = this.recordFor(routeKey);
|
|
72494
72537
|
record.probesInFlight = 0;
|
|
72495
72538
|
record.consecutiveFailures += 1;
|
|
72539
|
+
if (kind === "hard") {
|
|
72540
|
+
record.runHasHardFailure = true;
|
|
72541
|
+
}
|
|
72496
72542
|
if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
|
|
72497
72543
|
record.openedAtMs = this.now();
|
|
72498
72544
|
}
|
|
@@ -72504,17 +72550,19 @@ class CircuitBreakerRegistry {
|
|
|
72504
72550
|
* @returns A snapshot.
|
|
72505
72551
|
*/
|
|
72506
72552
|
snapshot(routeKey) {
|
|
72507
|
-
const record = this.records.get(routeKey) ??
|
|
72508
|
-
consecutiveFailures: 0,
|
|
72509
|
-
openedAtMs: null,
|
|
72510
|
-
probesInFlight: 0,
|
|
72511
|
-
};
|
|
72553
|
+
const record = this.records.get(routeKey) ?? freshRecord();
|
|
72512
72554
|
return {
|
|
72513
72555
|
routeKey,
|
|
72514
72556
|
state: this.stateOf(routeKey),
|
|
72515
72557
|
consecutiveFailures: record.consecutiveFailures,
|
|
72516
72558
|
openedAtMs: record.openedAtMs,
|
|
72517
72559
|
probesInFlight: record.probesInFlight,
|
|
72560
|
+
failureKind: record.consecutiveFailures === 0
|
|
72561
|
+
? null
|
|
72562
|
+
: record.runHasHardFailure
|
|
72563
|
+
? "hard"
|
|
72564
|
+
: "capacity",
|
|
72565
|
+
cooldownMs: this.cooldownFor(record),
|
|
72518
72566
|
};
|
|
72519
72567
|
}
|
|
72520
72568
|
/**
|
|
@@ -72542,7 +72590,7 @@ class CircuitBreakerRegistry {
|
|
|
72542
72590
|
recordFor(routeKey) {
|
|
72543
72591
|
let record = this.records.get(routeKey);
|
|
72544
72592
|
if (record === undefined) {
|
|
72545
|
-
record =
|
|
72593
|
+
record = freshRecord();
|
|
72546
72594
|
this.records.set(routeKey, record);
|
|
72547
72595
|
}
|
|
72548
72596
|
return record;
|
|
@@ -72818,7 +72866,17 @@ var providers$1 = {
|
|
|
72818
72866
|
max_concurrent: 12,
|
|
72819
72867
|
acquire_timeout_ms: 15000,
|
|
72820
72868
|
source: null,
|
|
72821
|
-
note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published."
|
|
72869
|
+
note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published.",
|
|
72870
|
+
models: {
|
|
72871
|
+
"claude-haiku-4-5": {
|
|
72872
|
+
basis: "conservative-default",
|
|
72873
|
+
requests_per_minute: 300,
|
|
72874
|
+
max_concurrent: 48,
|
|
72875
|
+
acquire_timeout_ms: 10000,
|
|
72876
|
+
source: null,
|
|
72877
|
+
note: "Raised 2026-09-28 for the closed-incumbent role this model holds on llm.fast, llm.decide and llm.extract. Measured that day: when DeepInfra was capacity-limited ('Model busy, retry later', timeouts at the 30000 ms hot-path budget), both DeepInfra legs of llm.fast opened their breakers and ALL of the alias's traffic fell to this model, where the 12-permit guard could not absorb it. In the 12 minutes to 15:33Z the incumbent leg recorded 18 timeouts, 15 breaker-open and 11 guard refusals ('did not admit the call within 15000 ms'), and the chain exhausted 20-45 times per 5 minutes. Sizing: in-flight demand is arrival rate times leg duration (Little's law), and the leg duration that matters is the worst case the chain permits, 30 s. 12 permits at 30 s serve 24 calls/min; 48 permits serve 96 calls/min at 30 s and up to the 300/min rate bound at haiku's normal single-digit-second latency, which covers the diverted llm.fast load plus normal llm.decide/llm.extract volume on the same guard. requests_per_minute is raised with it (the two bind in series) to 300, which is 30% of the 1,000 requests per minute per model that the scope_source lists for the lowest standard tier, so it remains below any standard tier's ceiling; no Anthropic 429 was observed in the window. acquire_timeout_ms is cut from 15000 to 10000 so an admitted call keeps at least 20 s of its 30 s leg budget: a call admitted after a 15 s wait had half its budget left, timed out at the provider, and that timeout was charged to this route's breaker, which then refused every call for the cooldown. STILL conservative-default: the organisation's tier has not been transcribed from the Claude Console (W3-07). Other Anthropic models keep the provider numbers."
|
|
72878
|
+
}
|
|
72879
|
+
}
|
|
72822
72880
|
},
|
|
72823
72881
|
openai: {
|
|
72824
72882
|
basis: "conservative-default",
|
|
@@ -72927,18 +72985,42 @@ const SECONDS_PER_MINUTE = 60;
|
|
|
72927
72985
|
const TOKENS_PER_REQUEST = 1;
|
|
72928
72986
|
const config$1 = limitsConfig;
|
|
72929
72987
|
/**
|
|
72930
|
-
* Resolve the limits that apply to a provider.
|
|
72988
|
+
* Resolve the limits that apply to a provider, or to one of its models.
|
|
72931
72989
|
*
|
|
72932
72990
|
* An unregistered provider falls back to the conservative defaults rather than
|
|
72933
72991
|
* to no limit at all. Treating "unknown" as "unlimited" would make every newly
|
|
72934
72992
|
* onboarded provider the one most likely to be over-driven, which is exactly
|
|
72935
72993
|
* backwards: a new provider is the one whose real ceiling is least understood.
|
|
72936
72994
|
*
|
|
72995
|
+
* A model with an override on a per-model provider runs at the override's
|
|
72996
|
+
* numbers and provenance; every other model runs at the provider's. The
|
|
72997
|
+
* override is ignored for a provider scoped as a whole, because that provider
|
|
72998
|
+
* has one guard and a per-model number cannot be enforced on it.
|
|
72999
|
+
*
|
|
72937
73000
|
* @param provider The provider key.
|
|
73001
|
+
* @param modelId The model, when the caller knows which one it addresses.
|
|
72938
73002
|
* @returns Its limits.
|
|
72939
73003
|
*/
|
|
72940
|
-
function limitsFor(provider) {
|
|
72941
|
-
|
|
73004
|
+
function limitsFor(provider, modelId) {
|
|
73005
|
+
const limits = config$1.providers[provider] ?? config$1.defaults;
|
|
73006
|
+
if (limits.scope !== "model" || modelId === undefined || modelId.length === 0) {
|
|
73007
|
+
return limits;
|
|
73008
|
+
}
|
|
73009
|
+
const override = limits.models?.[modelId];
|
|
73010
|
+
if (override === undefined) {
|
|
73011
|
+
return limits;
|
|
73012
|
+
}
|
|
73013
|
+
const { models: _siblings, ...providerLimits } = limits;
|
|
73014
|
+
return {
|
|
73015
|
+
...providerLimits,
|
|
73016
|
+
basis: override.basis,
|
|
73017
|
+
requests_per_minute: override.requests_per_minute,
|
|
73018
|
+
requests_per_minute_basis: override.requests_per_minute_basis,
|
|
73019
|
+
max_concurrent: override.max_concurrent,
|
|
73020
|
+
acquire_timeout_ms: override.acquire_timeout_ms,
|
|
73021
|
+
source: override.source ?? null,
|
|
73022
|
+
note: override.note,
|
|
73023
|
+
};
|
|
72942
73024
|
}
|
|
72943
73025
|
/** Every provider with a recorded limit, plus whether it is published or a default. */
|
|
72944
73026
|
function limitsInventory() {
|
|
@@ -73134,7 +73216,7 @@ const guardIdentities = new Map();
|
|
|
73134
73216
|
function rateLimiterFor(identity) {
|
|
73135
73217
|
let limiter = rateLimiters.get(identity.key);
|
|
73136
73218
|
if (limiter === undefined) {
|
|
73137
|
-
const limits = limitsFor(identity.provider);
|
|
73219
|
+
const limits = limitsFor(identity.provider, identity.modelId);
|
|
73138
73220
|
limiter = new TokenBucketRateLimiter({
|
|
73139
73221
|
maxTokens: limits.requests_per_minute,
|
|
73140
73222
|
refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
|
|
@@ -73155,7 +73237,7 @@ function rateLimiterFor(identity) {
|
|
|
73155
73237
|
function concurrencyGateFor(identity) {
|
|
73156
73238
|
let gate = concurrencyGates.get(identity.key);
|
|
73157
73239
|
if (gate === undefined) {
|
|
73158
|
-
gate = new ConcurrencyGate(identity, limitsFor(identity.provider).max_concurrent);
|
|
73240
|
+
gate = new ConcurrencyGate(identity, limitsFor(identity.provider, identity.modelId).max_concurrent);
|
|
73159
73241
|
concurrencyGates.set(identity.key, gate);
|
|
73160
73242
|
guardIdentities.set(identity.key, identity);
|
|
73161
73243
|
}
|
|
@@ -73185,8 +73267,8 @@ function concurrencyGateFor(identity) {
|
|
|
73185
73267
|
* or the caller stopped waiting first.
|
|
73186
73268
|
*/
|
|
73187
73269
|
async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
|
|
73188
|
-
const limits = limitsFor(provider);
|
|
73189
73270
|
const identity = guardIdentity(provider, scope.modelId);
|
|
73271
|
+
const limits = limitsFor(provider, identity.modelId);
|
|
73190
73272
|
const waitBudgetMs = maxWaitMs === undefined
|
|
73191
73273
|
? limits.acquire_timeout_ms
|
|
73192
73274
|
: Math.min(maxWaitMs, limits.acquire_timeout_ms);
|
|
@@ -73218,7 +73300,7 @@ function guardSnapshots() {
|
|
|
73218
73300
|
return [...guardIdentities.values()]
|
|
73219
73301
|
.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))
|
|
73220
73302
|
.map((identity) => {
|
|
73221
|
-
const limits = limitsFor(identity.provider);
|
|
73303
|
+
const limits = limitsFor(identity.provider, identity.modelId);
|
|
73222
73304
|
const limiter = rateLimiters.get(identity.key);
|
|
73223
73305
|
const gate = concurrencyGates.get(identity.key);
|
|
73224
73306
|
return {
|
|
@@ -73538,6 +73620,38 @@ async function runLeg(leg, params, execution, budgetMs) {
|
|
|
73538
73620
|
execution.callerSignal?.removeEventListener("abort", forwardAbort);
|
|
73539
73621
|
}
|
|
73540
73622
|
}
|
|
73623
|
+
/**
|
|
73624
|
+
* HTTP statuses a provider (or the gateway relaying it) uses to say it is full
|
|
73625
|
+
* rather than that the request or the route is wrong: request timeout, too
|
|
73626
|
+
* early, too many requests, service unavailable, and Anthropic's overloaded.
|
|
73627
|
+
*/
|
|
73628
|
+
const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
|
|
73629
|
+
/**
|
|
73630
|
+
* Wording providers use for a capacity refusal when the status is lost on the
|
|
73631
|
+
* way (a relayed body, a client library's own error). DeepInfra's is
|
|
73632
|
+
* "Model busy, retry later"; Anthropic's is "Overloaded".
|
|
73633
|
+
*/
|
|
73634
|
+
const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
|
|
73635
|
+
/**
|
|
73636
|
+
* Whether a failure is the provider saying it is full rather than broken.
|
|
73637
|
+
*
|
|
73638
|
+
* Read by shape rather than by class, because the same signal reaches the
|
|
73639
|
+
* chain from more than one transport and not every transport's error class is
|
|
73640
|
+
* importable here.
|
|
73641
|
+
*
|
|
73642
|
+
* @param error The thrown value.
|
|
73643
|
+
* @param reason Its message.
|
|
73644
|
+
* @returns Whether it is a capacity signal.
|
|
73645
|
+
*/
|
|
73646
|
+
function isCapacitySignal(error, reason) {
|
|
73647
|
+
if (typeof error === "object" && error !== null) {
|
|
73648
|
+
const status = error.status;
|
|
73649
|
+
if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
|
|
73650
|
+
return true;
|
|
73651
|
+
}
|
|
73652
|
+
}
|
|
73653
|
+
return CAPACITY_WORDING.test(reason);
|
|
73654
|
+
}
|
|
73541
73655
|
/**
|
|
73542
73656
|
* Classify why a leg failed.
|
|
73543
73657
|
*
|
|
@@ -73546,6 +73660,13 @@ async function runLeg(leg, params, execution, budgetMs) {
|
|
|
73546
73660
|
* cancellation as a provider failure would let a burst of user-cancelled
|
|
73547
73661
|
* requests open the breaker on a perfectly healthy route.
|
|
73548
73662
|
*
|
|
73663
|
+
* Among failures that do count, a capacity signal (the provider said it is
|
|
73664
|
+
* busy, or the leg ran out its budget waiting on it) is told apart from a hard
|
|
73665
|
+
* failure so the breaker can re-admit a busy route sooner than a broken one. A
|
|
73666
|
+
* timeout is read as capacity: on a reachable provider it is what a full queue
|
|
73667
|
+
* looks like from outside, and a provider that is actually down still costs no
|
|
73668
|
+
* more than one probe per capacity cooldown.
|
|
73669
|
+
*
|
|
73549
73670
|
* @param error The thrown value.
|
|
73550
73671
|
* @param callerSignal The caller's cancellation signal, if any.
|
|
73551
73672
|
* @returns The outcome and whether it counts against route health.
|
|
@@ -73556,24 +73677,45 @@ function classify(error, callerSignal) {
|
|
|
73556
73677
|
outcome: "skipped",
|
|
73557
73678
|
reason: "caller cancelled",
|
|
73558
73679
|
countsAgainstHealth: false,
|
|
73680
|
+
failureKind: "hard",
|
|
73559
73681
|
};
|
|
73560
73682
|
}
|
|
73561
73683
|
if (error instanceof LegTimeoutError) {
|
|
73562
|
-
return {
|
|
73684
|
+
return {
|
|
73685
|
+
outcome: "timeout",
|
|
73686
|
+
reason: error.message,
|
|
73687
|
+
countsAgainstHealth: true,
|
|
73688
|
+
failureKind: "capacity",
|
|
73689
|
+
};
|
|
73563
73690
|
}
|
|
73564
73691
|
if (error instanceof UnsupportedCapabilityError) {
|
|
73565
|
-
return {
|
|
73692
|
+
return {
|
|
73693
|
+
outcome: "skipped",
|
|
73694
|
+
reason: error.message,
|
|
73695
|
+
countsAgainstHealth: false,
|
|
73696
|
+
failureKind: "hard",
|
|
73697
|
+
};
|
|
73566
73698
|
}
|
|
73567
73699
|
if (error instanceof ToolChoiceIgnoredError) {
|
|
73568
73700
|
// The route answered; it broke a declared guarantee rather than failing to
|
|
73569
73701
|
// be available, so its breaker is not charged for it.
|
|
73570
|
-
return {
|
|
73702
|
+
return {
|
|
73703
|
+
outcome: "error",
|
|
73704
|
+
reason: error.message,
|
|
73705
|
+
countsAgainstHealth: false,
|
|
73706
|
+
failureKind: "hard",
|
|
73707
|
+
};
|
|
73571
73708
|
}
|
|
73709
|
+
// Self-inflicted pacing, not provider ill-health. Counting it would let the
|
|
73710
|
+
// client's own throttling open a breaker on a perfectly healthy provider and
|
|
73711
|
+
// permanently reroute traffic nobody chose to reroute.
|
|
73572
73712
|
if (error instanceof RateGuardTimeoutError) {
|
|
73573
|
-
|
|
73574
|
-
|
|
73575
|
-
|
|
73576
|
-
|
|
73713
|
+
return {
|
|
73714
|
+
outcome: "skipped",
|
|
73715
|
+
reason: error.message,
|
|
73716
|
+
countsAgainstHealth: false,
|
|
73717
|
+
failureKind: "hard",
|
|
73718
|
+
};
|
|
73577
73719
|
}
|
|
73578
73720
|
const reason = error instanceof Error ? error.message : String(error);
|
|
73579
73721
|
if (/abort/i.test(reason)) {
|
|
@@ -73581,9 +73723,20 @@ function classify(error, callerSignal) {
|
|
|
73581
73723
|
outcome: "timeout",
|
|
73582
73724
|
reason: `aborted: ${reason}`,
|
|
73583
73725
|
countsAgainstHealth: true,
|
|
73726
|
+
failureKind: "capacity",
|
|
73584
73727
|
};
|
|
73585
73728
|
}
|
|
73586
|
-
|
|
73729
|
+
if (error instanceof LlmResponseFormatError) {
|
|
73730
|
+
// The provider answered, badly. That is a route defect, not a full queue,
|
|
73731
|
+
// whatever words the unparseable content happens to contain.
|
|
73732
|
+
return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
|
|
73733
|
+
}
|
|
73734
|
+
return {
|
|
73735
|
+
outcome: "error",
|
|
73736
|
+
reason,
|
|
73737
|
+
countsAgainstHealth: true,
|
|
73738
|
+
failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
|
|
73739
|
+
};
|
|
73587
73740
|
}
|
|
73588
73741
|
/**
|
|
73589
73742
|
* Whether the caller has stopped waiting.
|
|
@@ -73702,9 +73855,9 @@ async function executeChain(alias, execution) {
|
|
|
73702
73855
|
return { response, servedBy: route, attempts, totalUsage };
|
|
73703
73856
|
}
|
|
73704
73857
|
catch (error) {
|
|
73705
|
-
const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
|
|
73858
|
+
const { outcome, reason, countsAgainstHealth, failureKind } = classify(error, execution.callerSignal);
|
|
73706
73859
|
if (countsAgainstHealth) {
|
|
73707
|
-
execution.breakers.onFailure(route.routeKey);
|
|
73860
|
+
execution.breakers.onFailure(route.routeKey, failureKind);
|
|
73708
73861
|
}
|
|
73709
73862
|
else if (holdsProbe) {
|
|
73710
73863
|
// No verdict on the route's health, but the probe slot this attempt
|
|
@@ -73752,6 +73905,7 @@ var defaults = {
|
|
|
73752
73905
|
circuit_breaker: {
|
|
73753
73906
|
failure_threshold: 5,
|
|
73754
73907
|
cooldown_ms: 60000,
|
|
73908
|
+
capacity_cooldown_ms: 15000,
|
|
73755
73909
|
half_open_probes: 1
|
|
73756
73910
|
}
|
|
73757
73911
|
};
|