@adaptic/utils 0.0.1037 → 0.0.1038
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +183 -31
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +183 -31
- package/dist/index.mjs.map +1 -1
- package/dist/types/llm/circuit-breaker.d.ts +45 -1
- package/dist/types/llm/circuit-breaker.d.ts.map +1 -1
- package/dist/types/llm/fallback-chain.d.ts +12 -0
- package/dist/types/llm/fallback-chain.d.ts.map +1 -1
- package/dist/types/llm/index.d.ts +2 -2
- package/dist/types/llm/index.d.ts.map +1 -1
- package/dist/types/llm/rate-guard.d.ts +29 -2
- package/dist/types/llm/rate-guard.d.ts.map +1 -1
- package/dist/types/llm/types.d.ts +6 -0
- package/dist/types/llm/types.d.ts.map +1 -1
- package/package.json +1 -1
package/dist/index.mjs
CHANGED
|
@@ -72345,11 +72345,34 @@ const alpaca = {
|
|
|
72345
72345
|
* traffic can neither trip nor be tripped by traffic on the other side of the
|
|
72346
72346
|
* PD-9 boundary.
|
|
72347
72347
|
*
|
|
72348
|
+
* How long a tripped breaker stays open depends on WHY it tripped. A provider
|
|
72349
|
+
* that is refusing work because it is momentarily full ("model busy", 429,
|
|
72350
|
+
* 503, 529, or a leg that ran out its budget queued behind other traffic) is
|
|
72351
|
+
* shedding load it expects to take back within seconds; excluding it for a
|
|
72352
|
+
* full minute pushes every call onto the next leg, which is how one provider's
|
|
72353
|
+
* capacity blip becomes the last leg's overload. Such a run opens for the
|
|
72354
|
+
* shorter `capacity_cooldown_ms`. A run that contains any hard failure (a
|
|
72355
|
+
* rejected credential, a malformed request, an unreachable gateway, an answer
|
|
72356
|
+
* that does not parse) says the route is broken rather than busy, and opens
|
|
72357
|
+
* for the full `cooldown_ms`. Either way the route then admits a bounded number
|
|
72358
|
+
* of half-open probes, and one success closes it.
|
|
72359
|
+
*
|
|
72348
72360
|
* The clock is injected. Breaker behaviour is entirely about elapsed time, and
|
|
72349
72361
|
* a test that must sleep to observe a cooldown is a test nobody runs.
|
|
72350
72362
|
*
|
|
72351
72363
|
* @module llm/circuit-breaker
|
|
72352
72364
|
*/
|
|
72365
|
+
/**
|
|
72366
|
+
* @returns A record for a route with no failures on file.
|
|
72367
|
+
*/
|
|
72368
|
+
function freshRecord() {
|
|
72369
|
+
return {
|
|
72370
|
+
consecutiveFailures: 0,
|
|
72371
|
+
openedAtMs: null,
|
|
72372
|
+
probesInFlight: 0,
|
|
72373
|
+
runHasHardFailure: false,
|
|
72374
|
+
};
|
|
72375
|
+
}
|
|
72353
72376
|
/**
|
|
72354
72377
|
* Tracks route health and decides whether a leg may be attempted.
|
|
72355
72378
|
*/
|
|
@@ -72382,7 +72405,26 @@ class CircuitBreakerRegistry {
|
|
|
72382
72405
|
return "closed";
|
|
72383
72406
|
}
|
|
72384
72407
|
const elapsed = this.now() - record.openedAtMs;
|
|
72385
|
-
return elapsed >= this.
|
|
72408
|
+
return elapsed >= this.cooldownFor(record) ? "half-open" : "open";
|
|
72409
|
+
}
|
|
72410
|
+
/**
|
|
72411
|
+
* The cooldown a record's current run earns.
|
|
72412
|
+
*
|
|
72413
|
+
* A run made only of capacity failures earns the capacity cooldown; one hard
|
|
72414
|
+
* failure anywhere in the run earns the full one. Mixed evidence is read as
|
|
72415
|
+
* the worse case, because a route that is both busy and broken is broken.
|
|
72416
|
+
* The capacity cooldown is never allowed to exceed the full one, so a
|
|
72417
|
+
* misconfigured table cannot make busy routes wait longer than broken ones.
|
|
72418
|
+
*
|
|
72419
|
+
* @param record The route's record.
|
|
72420
|
+
* @returns The cooldown in milliseconds.
|
|
72421
|
+
*/
|
|
72422
|
+
cooldownFor(record) {
|
|
72423
|
+
const capacityCooldown = this.config.capacity_cooldown_ms ?? this.config.cooldown_ms;
|
|
72424
|
+
if (record.runHasHardFailure) {
|
|
72425
|
+
return this.config.cooldown_ms;
|
|
72426
|
+
}
|
|
72427
|
+
return Math.min(capacityCooldown, this.config.cooldown_ms);
|
|
72386
72428
|
}
|
|
72387
72429
|
/**
|
|
72388
72430
|
* Whether a route may be attempted now.
|
|
@@ -72453,11 +72495,7 @@ class CircuitBreakerRegistry {
|
|
|
72453
72495
|
* @returns void
|
|
72454
72496
|
*/
|
|
72455
72497
|
onSuccess(routeKey) {
|
|
72456
|
-
this.records.set(routeKey,
|
|
72457
|
-
consecutiveFailures: 0,
|
|
72458
|
-
openedAtMs: null,
|
|
72459
|
-
probesInFlight: 0,
|
|
72460
|
-
});
|
|
72498
|
+
this.records.set(routeKey, freshRecord());
|
|
72461
72499
|
}
|
|
72462
72500
|
/**
|
|
72463
72501
|
* Record a failure, opening the breaker once the threshold is reached.
|
|
@@ -72466,13 +72504,19 @@ class CircuitBreakerRegistry {
|
|
|
72466
72504
|
* re-accumulate the threshold: the probe was the test, and it failed.
|
|
72467
72505
|
*
|
|
72468
72506
|
* @param routeKey The route's stable key.
|
|
72507
|
+
* @param kind Whether the failure was a capacity signal or a hard failure.
|
|
72508
|
+
* Defaults to `hard`, so a caller that cannot tell gets the longer, safer
|
|
72509
|
+
* cooldown.
|
|
72469
72510
|
* @returns void
|
|
72470
72511
|
*/
|
|
72471
|
-
onFailure(routeKey) {
|
|
72512
|
+
onFailure(routeKey, kind = "hard") {
|
|
72472
72513
|
const wasHalfOpen = this.stateOf(routeKey) === "half-open";
|
|
72473
72514
|
const record = this.recordFor(routeKey);
|
|
72474
72515
|
record.probesInFlight = 0;
|
|
72475
72516
|
record.consecutiveFailures += 1;
|
|
72517
|
+
if (kind === "hard") {
|
|
72518
|
+
record.runHasHardFailure = true;
|
|
72519
|
+
}
|
|
72476
72520
|
if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
|
|
72477
72521
|
record.openedAtMs = this.now();
|
|
72478
72522
|
}
|
|
@@ -72484,17 +72528,19 @@ class CircuitBreakerRegistry {
|
|
|
72484
72528
|
* @returns A snapshot.
|
|
72485
72529
|
*/
|
|
72486
72530
|
snapshot(routeKey) {
|
|
72487
|
-
const record = this.records.get(routeKey) ??
|
|
72488
|
-
consecutiveFailures: 0,
|
|
72489
|
-
openedAtMs: null,
|
|
72490
|
-
probesInFlight: 0,
|
|
72491
|
-
};
|
|
72531
|
+
const record = this.records.get(routeKey) ?? freshRecord();
|
|
72492
72532
|
return {
|
|
72493
72533
|
routeKey,
|
|
72494
72534
|
state: this.stateOf(routeKey),
|
|
72495
72535
|
consecutiveFailures: record.consecutiveFailures,
|
|
72496
72536
|
openedAtMs: record.openedAtMs,
|
|
72497
72537
|
probesInFlight: record.probesInFlight,
|
|
72538
|
+
failureKind: record.consecutiveFailures === 0
|
|
72539
|
+
? null
|
|
72540
|
+
: record.runHasHardFailure
|
|
72541
|
+
? "hard"
|
|
72542
|
+
: "capacity",
|
|
72543
|
+
cooldownMs: this.cooldownFor(record),
|
|
72498
72544
|
};
|
|
72499
72545
|
}
|
|
72500
72546
|
/**
|
|
@@ -72522,7 +72568,7 @@ class CircuitBreakerRegistry {
|
|
|
72522
72568
|
recordFor(routeKey) {
|
|
72523
72569
|
let record = this.records.get(routeKey);
|
|
72524
72570
|
if (record === undefined) {
|
|
72525
|
-
record =
|
|
72571
|
+
record = freshRecord();
|
|
72526
72572
|
this.records.set(routeKey, record);
|
|
72527
72573
|
}
|
|
72528
72574
|
return record;
|
|
@@ -72798,7 +72844,17 @@ var providers$1 = {
|
|
|
72798
72844
|
max_concurrent: 12,
|
|
72799
72845
|
acquire_timeout_ms: 15000,
|
|
72800
72846
|
source: null,
|
|
72801
|
-
note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published."
|
|
72847
|
+
note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published.",
|
|
72848
|
+
models: {
|
|
72849
|
+
"claude-haiku-4-5": {
|
|
72850
|
+
basis: "conservative-default",
|
|
72851
|
+
requests_per_minute: 300,
|
|
72852
|
+
max_concurrent: 48,
|
|
72853
|
+
acquire_timeout_ms: 10000,
|
|
72854
|
+
source: null,
|
|
72855
|
+
note: "Raised 2026-09-28 for the closed-incumbent role this model holds on llm.fast, llm.decide and llm.extract. Measured that day: when DeepInfra was capacity-limited ('Model busy, retry later', timeouts at the 30000 ms hot-path budget), both DeepInfra legs of llm.fast opened their breakers and ALL of the alias's traffic fell to this model, where the 12-permit guard could not absorb it. In the 12 minutes to 15:33Z the incumbent leg recorded 18 timeouts, 15 breaker-open and 11 guard refusals ('did not admit the call within 15000 ms'), and the chain exhausted 20-45 times per 5 minutes. Sizing: in-flight demand is arrival rate times leg duration (Little's law), and the leg duration that matters is the worst case the chain permits, 30 s. 12 permits at 30 s serve 24 calls/min; 48 permits serve 96 calls/min at 30 s and up to the 300/min rate bound at haiku's normal single-digit-second latency, which covers the diverted llm.fast load plus normal llm.decide/llm.extract volume on the same guard. requests_per_minute is raised with it (the two bind in series) to 300, which is 30% of the 1,000 requests per minute per model that the scope_source lists for the lowest standard tier, so it remains below any standard tier's ceiling; no Anthropic 429 was observed in the window. acquire_timeout_ms is cut from 15000 to 10000 so an admitted call keeps at least 20 s of its 30 s leg budget: a call admitted after a 15 s wait had half its budget left, timed out at the provider, and that timeout was charged to this route's breaker, which then refused every call for the cooldown. STILL conservative-default: the organisation's tier has not been transcribed from the Claude Console (W3-07). Other Anthropic models keep the provider numbers."
|
|
72856
|
+
}
|
|
72857
|
+
}
|
|
72802
72858
|
},
|
|
72803
72859
|
openai: {
|
|
72804
72860
|
basis: "conservative-default",
|
|
@@ -72907,18 +72963,42 @@ const SECONDS_PER_MINUTE = 60;
|
|
|
72907
72963
|
const TOKENS_PER_REQUEST = 1;
|
|
72908
72964
|
const config$1 = limitsConfig;
|
|
72909
72965
|
/**
|
|
72910
|
-
* Resolve the limits that apply to a provider.
|
|
72966
|
+
* Resolve the limits that apply to a provider, or to one of its models.
|
|
72911
72967
|
*
|
|
72912
72968
|
* An unregistered provider falls back to the conservative defaults rather than
|
|
72913
72969
|
* to no limit at all. Treating "unknown" as "unlimited" would make every newly
|
|
72914
72970
|
* onboarded provider the one most likely to be over-driven, which is exactly
|
|
72915
72971
|
* backwards: a new provider is the one whose real ceiling is least understood.
|
|
72916
72972
|
*
|
|
72973
|
+
* A model with an override on a per-model provider runs at the override's
|
|
72974
|
+
* numbers and provenance; every other model runs at the provider's. The
|
|
72975
|
+
* override is ignored for a provider scoped as a whole, because that provider
|
|
72976
|
+
* has one guard and a per-model number cannot be enforced on it.
|
|
72977
|
+
*
|
|
72917
72978
|
* @param provider The provider key.
|
|
72979
|
+
* @param modelId The model, when the caller knows which one it addresses.
|
|
72918
72980
|
* @returns Its limits.
|
|
72919
72981
|
*/
|
|
72920
|
-
function limitsFor(provider) {
|
|
72921
|
-
|
|
72982
|
+
function limitsFor(provider, modelId) {
|
|
72983
|
+
const limits = config$1.providers[provider] ?? config$1.defaults;
|
|
72984
|
+
if (limits.scope !== "model" || modelId === undefined || modelId.length === 0) {
|
|
72985
|
+
return limits;
|
|
72986
|
+
}
|
|
72987
|
+
const override = limits.models?.[modelId];
|
|
72988
|
+
if (override === undefined) {
|
|
72989
|
+
return limits;
|
|
72990
|
+
}
|
|
72991
|
+
const { models: _siblings, ...providerLimits } = limits;
|
|
72992
|
+
return {
|
|
72993
|
+
...providerLimits,
|
|
72994
|
+
basis: override.basis,
|
|
72995
|
+
requests_per_minute: override.requests_per_minute,
|
|
72996
|
+
requests_per_minute_basis: override.requests_per_minute_basis,
|
|
72997
|
+
max_concurrent: override.max_concurrent,
|
|
72998
|
+
acquire_timeout_ms: override.acquire_timeout_ms,
|
|
72999
|
+
source: override.source ?? null,
|
|
73000
|
+
note: override.note,
|
|
73001
|
+
};
|
|
72922
73002
|
}
|
|
72923
73003
|
/** Every provider with a recorded limit, plus whether it is published or a default. */
|
|
72924
73004
|
function limitsInventory() {
|
|
@@ -73114,7 +73194,7 @@ const guardIdentities = new Map();
|
|
|
73114
73194
|
function rateLimiterFor(identity) {
|
|
73115
73195
|
let limiter = rateLimiters.get(identity.key);
|
|
73116
73196
|
if (limiter === undefined) {
|
|
73117
|
-
const limits = limitsFor(identity.provider);
|
|
73197
|
+
const limits = limitsFor(identity.provider, identity.modelId);
|
|
73118
73198
|
limiter = new TokenBucketRateLimiter({
|
|
73119
73199
|
maxTokens: limits.requests_per_minute,
|
|
73120
73200
|
refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
|
|
@@ -73135,7 +73215,7 @@ function rateLimiterFor(identity) {
|
|
|
73135
73215
|
function concurrencyGateFor(identity) {
|
|
73136
73216
|
let gate = concurrencyGates.get(identity.key);
|
|
73137
73217
|
if (gate === undefined) {
|
|
73138
|
-
gate = new ConcurrencyGate(identity, limitsFor(identity.provider).max_concurrent);
|
|
73218
|
+
gate = new ConcurrencyGate(identity, limitsFor(identity.provider, identity.modelId).max_concurrent);
|
|
73139
73219
|
concurrencyGates.set(identity.key, gate);
|
|
73140
73220
|
guardIdentities.set(identity.key, identity);
|
|
73141
73221
|
}
|
|
@@ -73165,8 +73245,8 @@ function concurrencyGateFor(identity) {
|
|
|
73165
73245
|
* or the caller stopped waiting first.
|
|
73166
73246
|
*/
|
|
73167
73247
|
async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
|
|
73168
|
-
const limits = limitsFor(provider);
|
|
73169
73248
|
const identity = guardIdentity(provider, scope.modelId);
|
|
73249
|
+
const limits = limitsFor(provider, identity.modelId);
|
|
73170
73250
|
const waitBudgetMs = maxWaitMs === undefined
|
|
73171
73251
|
? limits.acquire_timeout_ms
|
|
73172
73252
|
: Math.min(maxWaitMs, limits.acquire_timeout_ms);
|
|
@@ -73198,7 +73278,7 @@ function guardSnapshots() {
|
|
|
73198
73278
|
return [...guardIdentities.values()]
|
|
73199
73279
|
.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))
|
|
73200
73280
|
.map((identity) => {
|
|
73201
|
-
const limits = limitsFor(identity.provider);
|
|
73281
|
+
const limits = limitsFor(identity.provider, identity.modelId);
|
|
73202
73282
|
const limiter = rateLimiters.get(identity.key);
|
|
73203
73283
|
const gate = concurrencyGates.get(identity.key);
|
|
73204
73284
|
return {
|
|
@@ -73518,6 +73598,38 @@ async function runLeg(leg, params, execution, budgetMs) {
|
|
|
73518
73598
|
execution.callerSignal?.removeEventListener("abort", forwardAbort);
|
|
73519
73599
|
}
|
|
73520
73600
|
}
|
|
73601
|
+
/**
|
|
73602
|
+
* HTTP statuses a provider (or the gateway relaying it) uses to say it is full
|
|
73603
|
+
* rather than that the request or the route is wrong: request timeout, too
|
|
73604
|
+
* early, too many requests, service unavailable, and Anthropic's overloaded.
|
|
73605
|
+
*/
|
|
73606
|
+
const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
|
|
73607
|
+
/**
|
|
73608
|
+
* Wording providers use for a capacity refusal when the status is lost on the
|
|
73609
|
+
* way (a relayed body, a client library's own error). DeepInfra's is
|
|
73610
|
+
* "Model busy, retry later"; Anthropic's is "Overloaded".
|
|
73611
|
+
*/
|
|
73612
|
+
const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
|
|
73613
|
+
/**
|
|
73614
|
+
* Whether a failure is the provider saying it is full rather than broken.
|
|
73615
|
+
*
|
|
73616
|
+
* Read by shape rather than by class, because the same signal reaches the
|
|
73617
|
+
* chain from more than one transport and not every transport's error class is
|
|
73618
|
+
* importable here.
|
|
73619
|
+
*
|
|
73620
|
+
* @param error The thrown value.
|
|
73621
|
+
* @param reason Its message.
|
|
73622
|
+
* @returns Whether it is a capacity signal.
|
|
73623
|
+
*/
|
|
73624
|
+
function isCapacitySignal(error, reason) {
|
|
73625
|
+
if (typeof error === "object" && error !== null) {
|
|
73626
|
+
const status = error.status;
|
|
73627
|
+
if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
|
|
73628
|
+
return true;
|
|
73629
|
+
}
|
|
73630
|
+
}
|
|
73631
|
+
return CAPACITY_WORDING.test(reason);
|
|
73632
|
+
}
|
|
73521
73633
|
/**
|
|
73522
73634
|
* Classify why a leg failed.
|
|
73523
73635
|
*
|
|
@@ -73526,6 +73638,13 @@ async function runLeg(leg, params, execution, budgetMs) {
|
|
|
73526
73638
|
* cancellation as a provider failure would let a burst of user-cancelled
|
|
73527
73639
|
* requests open the breaker on a perfectly healthy route.
|
|
73528
73640
|
*
|
|
73641
|
+
* Among failures that do count, a capacity signal (the provider said it is
|
|
73642
|
+
* busy, or the leg ran out its budget waiting on it) is told apart from a hard
|
|
73643
|
+
* failure so the breaker can re-admit a busy route sooner than a broken one. A
|
|
73644
|
+
* timeout is read as capacity: on a reachable provider it is what a full queue
|
|
73645
|
+
* looks like from outside, and a provider that is actually down still costs no
|
|
73646
|
+
* more than one probe per capacity cooldown.
|
|
73647
|
+
*
|
|
73529
73648
|
* @param error The thrown value.
|
|
73530
73649
|
* @param callerSignal The caller's cancellation signal, if any.
|
|
73531
73650
|
* @returns The outcome and whether it counts against route health.
|
|
@@ -73536,24 +73655,45 @@ function classify(error, callerSignal) {
|
|
|
73536
73655
|
outcome: "skipped",
|
|
73537
73656
|
reason: "caller cancelled",
|
|
73538
73657
|
countsAgainstHealth: false,
|
|
73658
|
+
failureKind: "hard",
|
|
73539
73659
|
};
|
|
73540
73660
|
}
|
|
73541
73661
|
if (error instanceof LegTimeoutError) {
|
|
73542
|
-
return {
|
|
73662
|
+
return {
|
|
73663
|
+
outcome: "timeout",
|
|
73664
|
+
reason: error.message,
|
|
73665
|
+
countsAgainstHealth: true,
|
|
73666
|
+
failureKind: "capacity",
|
|
73667
|
+
};
|
|
73543
73668
|
}
|
|
73544
73669
|
if (error instanceof UnsupportedCapabilityError) {
|
|
73545
|
-
return {
|
|
73670
|
+
return {
|
|
73671
|
+
outcome: "skipped",
|
|
73672
|
+
reason: error.message,
|
|
73673
|
+
countsAgainstHealth: false,
|
|
73674
|
+
failureKind: "hard",
|
|
73675
|
+
};
|
|
73546
73676
|
}
|
|
73547
73677
|
if (error instanceof ToolChoiceIgnoredError) {
|
|
73548
73678
|
// The route answered; it broke a declared guarantee rather than failing to
|
|
73549
73679
|
// be available, so its breaker is not charged for it.
|
|
73550
|
-
return {
|
|
73680
|
+
return {
|
|
73681
|
+
outcome: "error",
|
|
73682
|
+
reason: error.message,
|
|
73683
|
+
countsAgainstHealth: false,
|
|
73684
|
+
failureKind: "hard",
|
|
73685
|
+
};
|
|
73551
73686
|
}
|
|
73687
|
+
// Self-inflicted pacing, not provider ill-health. Counting it would let the
|
|
73688
|
+
// client's own throttling open a breaker on a perfectly healthy provider and
|
|
73689
|
+
// permanently reroute traffic nobody chose to reroute.
|
|
73552
73690
|
if (error instanceof RateGuardTimeoutError) {
|
|
73553
|
-
|
|
73554
|
-
|
|
73555
|
-
|
|
73556
|
-
|
|
73691
|
+
return {
|
|
73692
|
+
outcome: "skipped",
|
|
73693
|
+
reason: error.message,
|
|
73694
|
+
countsAgainstHealth: false,
|
|
73695
|
+
failureKind: "hard",
|
|
73696
|
+
};
|
|
73557
73697
|
}
|
|
73558
73698
|
const reason = error instanceof Error ? error.message : String(error);
|
|
73559
73699
|
if (/abort/i.test(reason)) {
|
|
@@ -73561,9 +73701,20 @@ function classify(error, callerSignal) {
|
|
|
73561
73701
|
outcome: "timeout",
|
|
73562
73702
|
reason: `aborted: ${reason}`,
|
|
73563
73703
|
countsAgainstHealth: true,
|
|
73704
|
+
failureKind: "capacity",
|
|
73564
73705
|
};
|
|
73565
73706
|
}
|
|
73566
|
-
|
|
73707
|
+
if (error instanceof LlmResponseFormatError) {
|
|
73708
|
+
// The provider answered, badly. That is a route defect, not a full queue,
|
|
73709
|
+
// whatever words the unparseable content happens to contain.
|
|
73710
|
+
return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
|
|
73711
|
+
}
|
|
73712
|
+
return {
|
|
73713
|
+
outcome: "error",
|
|
73714
|
+
reason,
|
|
73715
|
+
countsAgainstHealth: true,
|
|
73716
|
+
failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
|
|
73717
|
+
};
|
|
73567
73718
|
}
|
|
73568
73719
|
/**
|
|
73569
73720
|
* Whether the caller has stopped waiting.
|
|
@@ -73682,9 +73833,9 @@ async function executeChain(alias, execution) {
|
|
|
73682
73833
|
return { response, servedBy: route, attempts, totalUsage };
|
|
73683
73834
|
}
|
|
73684
73835
|
catch (error) {
|
|
73685
|
-
const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
|
|
73836
|
+
const { outcome, reason, countsAgainstHealth, failureKind } = classify(error, execution.callerSignal);
|
|
73686
73837
|
if (countsAgainstHealth) {
|
|
73687
|
-
execution.breakers.onFailure(route.routeKey);
|
|
73838
|
+
execution.breakers.onFailure(route.routeKey, failureKind);
|
|
73688
73839
|
}
|
|
73689
73840
|
else if (holdsProbe) {
|
|
73690
73841
|
// No verdict on the route's health, but the probe slot this attempt
|
|
@@ -73732,6 +73883,7 @@ var defaults = {
|
|
|
73732
73883
|
circuit_breaker: {
|
|
73733
73884
|
failure_threshold: 5,
|
|
73734
73885
|
cooldown_ms: 60000,
|
|
73886
|
+
capacity_cooldown_ms: 15000,
|
|
73735
73887
|
half_open_probes: 1
|
|
73736
73888
|
}
|
|
73737
73889
|
};
|