@adaptic/utils 0.0.1036 → 0.0.1038
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +186 -32
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +186 -32
- package/dist/index.mjs.map +1 -1
- package/dist/types/llm/circuit-breaker.d.ts +45 -1
- package/dist/types/llm/circuit-breaker.d.ts.map +1 -1
- package/dist/types/llm/fallback-chain.d.ts +12 -0
- package/dist/types/llm/fallback-chain.d.ts.map +1 -1
- package/dist/types/llm/index.d.ts +2 -2
- package/dist/types/llm/index.d.ts.map +1 -1
- package/dist/types/llm/rate-guard.d.ts +29 -2
- package/dist/types/llm/rate-guard.d.ts.map +1 -1
- package/dist/types/llm/types.d.ts +6 -0
- package/dist/types/llm/types.d.ts.map +1 -1
- package/dist/types/massive.d.ts.map +1 -1
- package/package.json +1 -1
package/dist/index.mjs
CHANGED
|
@@ -19742,7 +19742,9 @@ const fetchTrades = async (symbol, options) => {
|
|
|
19742
19742
|
await rateLimiters$1.massive.acquire();
|
|
19743
19743
|
const url = `${baseUrl}?${params.toString()}`;
|
|
19744
19744
|
try {
|
|
19745
|
-
|
|
19745
|
+
// Redact the apiKey query param before logging — the raw URL embeds the
|
|
19746
|
+
// live Massive API key (audit #595).
|
|
19747
|
+
logIfDebug(`Fetching trades for ${symbol} from ${url.replace(/([?&](?:apiKey|apikey|api_key)=)[^&]+/gi, "$1***")}`);
|
|
19746
19748
|
const response = await fetchWithRetry(url, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
19747
19749
|
const data = (await response.json());
|
|
19748
19750
|
if ("message" in data) {
|
|
@@ -72343,11 +72345,34 @@ const alpaca = {
|
|
|
72343
72345
|
* traffic can neither trip nor be tripped by traffic on the other side of the
|
|
72344
72346
|
* PD-9 boundary.
|
|
72345
72347
|
*
|
|
72348
|
+
* How long a tripped breaker stays open depends on WHY it tripped. A provider
|
|
72349
|
+
* that is refusing work because it is momentarily full ("model busy", 429,
|
|
72350
|
+
* 503, 529, or a leg that ran out its budget queued behind other traffic) is
|
|
72351
|
+
* shedding load it expects to take back within seconds; excluding it for a
|
|
72352
|
+
* full minute pushes every call onto the next leg, which is how one provider's
|
|
72353
|
+
* capacity blip becomes the last leg's overload. Such a run opens for the
|
|
72354
|
+
* shorter `capacity_cooldown_ms`. A run that contains any hard failure (a
|
|
72355
|
+
* rejected credential, a malformed request, an unreachable gateway, an answer
|
|
72356
|
+
* that does not parse) says the route is broken rather than busy, and opens
|
|
72357
|
+
* for the full `cooldown_ms`. Either way the route then admits a bounded number
|
|
72358
|
+
* of half-open probes, and one success closes it.
|
|
72359
|
+
*
|
|
72346
72360
|
* The clock is injected. Breaker behaviour is entirely about elapsed time, and
|
|
72347
72361
|
* a test that must sleep to observe a cooldown is a test nobody runs.
|
|
72348
72362
|
*
|
|
72349
72363
|
* @module llm/circuit-breaker
|
|
72350
72364
|
*/
|
|
72365
|
+
/**
|
|
72366
|
+
* @returns A record for a route with no failures on file.
|
|
72367
|
+
*/
|
|
72368
|
+
function freshRecord() {
|
|
72369
|
+
return {
|
|
72370
|
+
consecutiveFailures: 0,
|
|
72371
|
+
openedAtMs: null,
|
|
72372
|
+
probesInFlight: 0,
|
|
72373
|
+
runHasHardFailure: false,
|
|
72374
|
+
};
|
|
72375
|
+
}
|
|
72351
72376
|
/**
|
|
72352
72377
|
* Tracks route health and decides whether a leg may be attempted.
|
|
72353
72378
|
*/
|
|
@@ -72380,7 +72405,26 @@ class CircuitBreakerRegistry {
|
|
|
72380
72405
|
return "closed";
|
|
72381
72406
|
}
|
|
72382
72407
|
const elapsed = this.now() - record.openedAtMs;
|
|
72383
|
-
return elapsed >= this.
|
|
72408
|
+
return elapsed >= this.cooldownFor(record) ? "half-open" : "open";
|
|
72409
|
+
}
|
|
72410
|
+
/**
|
|
72411
|
+
* The cooldown a record's current run earns.
|
|
72412
|
+
*
|
|
72413
|
+
* A run made only of capacity failures earns the capacity cooldown; one hard
|
|
72414
|
+
* failure anywhere in the run earns the full one. Mixed evidence is read as
|
|
72415
|
+
* the worse case, because a route that is both busy and broken is broken.
|
|
72416
|
+
* The capacity cooldown is never allowed to exceed the full one, so a
|
|
72417
|
+
* misconfigured table cannot make busy routes wait longer than broken ones.
|
|
72418
|
+
*
|
|
72419
|
+
* @param record The route's record.
|
|
72420
|
+
* @returns The cooldown in milliseconds.
|
|
72421
|
+
*/
|
|
72422
|
+
cooldownFor(record) {
|
|
72423
|
+
const capacityCooldown = this.config.capacity_cooldown_ms ?? this.config.cooldown_ms;
|
|
72424
|
+
if (record.runHasHardFailure) {
|
|
72425
|
+
return this.config.cooldown_ms;
|
|
72426
|
+
}
|
|
72427
|
+
return Math.min(capacityCooldown, this.config.cooldown_ms);
|
|
72384
72428
|
}
|
|
72385
72429
|
/**
|
|
72386
72430
|
* Whether a route may be attempted now.
|
|
@@ -72451,11 +72495,7 @@ class CircuitBreakerRegistry {
|
|
|
72451
72495
|
* @returns void
|
|
72452
72496
|
*/
|
|
72453
72497
|
onSuccess(routeKey) {
|
|
72454
|
-
this.records.set(routeKey,
|
|
72455
|
-
consecutiveFailures: 0,
|
|
72456
|
-
openedAtMs: null,
|
|
72457
|
-
probesInFlight: 0,
|
|
72458
|
-
});
|
|
72498
|
+
this.records.set(routeKey, freshRecord());
|
|
72459
72499
|
}
|
|
72460
72500
|
/**
|
|
72461
72501
|
* Record a failure, opening the breaker once the threshold is reached.
|
|
@@ -72464,13 +72504,19 @@ class CircuitBreakerRegistry {
|
|
|
72464
72504
|
* re-accumulate the threshold: the probe was the test, and it failed.
|
|
72465
72505
|
*
|
|
72466
72506
|
* @param routeKey The route's stable key.
|
|
72507
|
+
* @param kind Whether the failure was a capacity signal or a hard failure.
|
|
72508
|
+
* Defaults to `hard`, so a caller that cannot tell gets the longer, safer
|
|
72509
|
+
* cooldown.
|
|
72467
72510
|
* @returns void
|
|
72468
72511
|
*/
|
|
72469
|
-
onFailure(routeKey) {
|
|
72512
|
+
onFailure(routeKey, kind = "hard") {
|
|
72470
72513
|
const wasHalfOpen = this.stateOf(routeKey) === "half-open";
|
|
72471
72514
|
const record = this.recordFor(routeKey);
|
|
72472
72515
|
record.probesInFlight = 0;
|
|
72473
72516
|
record.consecutiveFailures += 1;
|
|
72517
|
+
if (kind === "hard") {
|
|
72518
|
+
record.runHasHardFailure = true;
|
|
72519
|
+
}
|
|
72474
72520
|
if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
|
|
72475
72521
|
record.openedAtMs = this.now();
|
|
72476
72522
|
}
|
|
@@ -72482,17 +72528,19 @@ class CircuitBreakerRegistry {
|
|
|
72482
72528
|
* @returns A snapshot.
|
|
72483
72529
|
*/
|
|
72484
72530
|
snapshot(routeKey) {
|
|
72485
|
-
const record = this.records.get(routeKey) ??
|
|
72486
|
-
consecutiveFailures: 0,
|
|
72487
|
-
openedAtMs: null,
|
|
72488
|
-
probesInFlight: 0,
|
|
72489
|
-
};
|
|
72531
|
+
const record = this.records.get(routeKey) ?? freshRecord();
|
|
72490
72532
|
return {
|
|
72491
72533
|
routeKey,
|
|
72492
72534
|
state: this.stateOf(routeKey),
|
|
72493
72535
|
consecutiveFailures: record.consecutiveFailures,
|
|
72494
72536
|
openedAtMs: record.openedAtMs,
|
|
72495
72537
|
probesInFlight: record.probesInFlight,
|
|
72538
|
+
failureKind: record.consecutiveFailures === 0
|
|
72539
|
+
? null
|
|
72540
|
+
: record.runHasHardFailure
|
|
72541
|
+
? "hard"
|
|
72542
|
+
: "capacity",
|
|
72543
|
+
cooldownMs: this.cooldownFor(record),
|
|
72496
72544
|
};
|
|
72497
72545
|
}
|
|
72498
72546
|
/**
|
|
@@ -72520,7 +72568,7 @@ class CircuitBreakerRegistry {
|
|
|
72520
72568
|
recordFor(routeKey) {
|
|
72521
72569
|
let record = this.records.get(routeKey);
|
|
72522
72570
|
if (record === undefined) {
|
|
72523
|
-
record =
|
|
72571
|
+
record = freshRecord();
|
|
72524
72572
|
this.records.set(routeKey, record);
|
|
72525
72573
|
}
|
|
72526
72574
|
return record;
|
|
@@ -72796,7 +72844,17 @@ var providers$1 = {
|
|
|
72796
72844
|
max_concurrent: 12,
|
|
72797
72845
|
acquire_timeout_ms: 15000,
|
|
72798
72846
|
source: null,
|
|
72799
|
-
note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published."
|
|
72847
|
+
note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published.",
|
|
72848
|
+
models: {
|
|
72849
|
+
"claude-haiku-4-5": {
|
|
72850
|
+
basis: "conservative-default",
|
|
72851
|
+
requests_per_minute: 300,
|
|
72852
|
+
max_concurrent: 48,
|
|
72853
|
+
acquire_timeout_ms: 10000,
|
|
72854
|
+
source: null,
|
|
72855
|
+
note: "Raised 2026-09-28 for the closed-incumbent role this model holds on llm.fast, llm.decide and llm.extract. Measured that day: when DeepInfra was capacity-limited ('Model busy, retry later', timeouts at the 30000 ms hot-path budget), both DeepInfra legs of llm.fast opened their breakers and ALL of the alias's traffic fell to this model, where the 12-permit guard could not absorb it. In the 12 minutes to 15:33Z the incumbent leg recorded 18 timeouts, 15 breaker-open and 11 guard refusals ('did not admit the call within 15000 ms'), and the chain exhausted 20-45 times per 5 minutes. Sizing: in-flight demand is arrival rate times leg duration (Little's law), and the leg duration that matters is the worst case the chain permits, 30 s. 12 permits at 30 s serve 24 calls/min; 48 permits serve 96 calls/min at 30 s and up to the 300/min rate bound at haiku's normal single-digit-second latency, which covers the diverted llm.fast load plus normal llm.decide/llm.extract volume on the same guard. requests_per_minute is raised with it (the two bind in series) to 300, which is 30% of the 1,000 requests per minute per model that the scope_source lists for the lowest standard tier, so it remains below any standard tier's ceiling; no Anthropic 429 was observed in the window. acquire_timeout_ms is cut from 15000 to 10000 so an admitted call keeps at least 20 s of its 30 s leg budget: a call admitted after a 15 s wait had half its budget left, timed out at the provider, and that timeout was charged to this route's breaker, which then refused every call for the cooldown. STILL conservative-default: the organisation's tier has not been transcribed from the Claude Console (W3-07). Other Anthropic models keep the provider numbers."
|
|
72856
|
+
}
|
|
72857
|
+
}
|
|
72800
72858
|
},
|
|
72801
72859
|
openai: {
|
|
72802
72860
|
basis: "conservative-default",
|
|
@@ -72905,18 +72963,42 @@ const SECONDS_PER_MINUTE = 60;
|
|
|
72905
72963
|
const TOKENS_PER_REQUEST = 1;
|
|
72906
72964
|
const config$1 = limitsConfig;
|
|
72907
72965
|
/**
|
|
72908
|
-
* Resolve the limits that apply to a provider.
|
|
72966
|
+
* Resolve the limits that apply to a provider, or to one of its models.
|
|
72909
72967
|
*
|
|
72910
72968
|
* An unregistered provider falls back to the conservative defaults rather than
|
|
72911
72969
|
* to no limit at all. Treating "unknown" as "unlimited" would make every newly
|
|
72912
72970
|
* onboarded provider the one most likely to be over-driven, which is exactly
|
|
72913
72971
|
* backwards: a new provider is the one whose real ceiling is least understood.
|
|
72914
72972
|
*
|
|
72973
|
+
* A model with an override on a per-model provider runs at the override's
|
|
72974
|
+
* numbers and provenance; every other model runs at the provider's. The
|
|
72975
|
+
* override is ignored for a provider scoped as a whole, because that provider
|
|
72976
|
+
* has one guard and a per-model number cannot be enforced on it.
|
|
72977
|
+
*
|
|
72915
72978
|
* @param provider The provider key.
|
|
72979
|
+
* @param modelId The model, when the caller knows which one it addresses.
|
|
72916
72980
|
* @returns Its limits.
|
|
72917
72981
|
*/
|
|
72918
|
-
function limitsFor(provider) {
|
|
72919
|
-
|
|
72982
|
+
function limitsFor(provider, modelId) {
|
|
72983
|
+
const limits = config$1.providers[provider] ?? config$1.defaults;
|
|
72984
|
+
if (limits.scope !== "model" || modelId === undefined || modelId.length === 0) {
|
|
72985
|
+
return limits;
|
|
72986
|
+
}
|
|
72987
|
+
const override = limits.models?.[modelId];
|
|
72988
|
+
if (override === undefined) {
|
|
72989
|
+
return limits;
|
|
72990
|
+
}
|
|
72991
|
+
const { models: _siblings, ...providerLimits } = limits;
|
|
72992
|
+
return {
|
|
72993
|
+
...providerLimits,
|
|
72994
|
+
basis: override.basis,
|
|
72995
|
+
requests_per_minute: override.requests_per_minute,
|
|
72996
|
+
requests_per_minute_basis: override.requests_per_minute_basis,
|
|
72997
|
+
max_concurrent: override.max_concurrent,
|
|
72998
|
+
acquire_timeout_ms: override.acquire_timeout_ms,
|
|
72999
|
+
source: override.source ?? null,
|
|
73000
|
+
note: override.note,
|
|
73001
|
+
};
|
|
72920
73002
|
}
|
|
72921
73003
|
/** Every provider with a recorded limit, plus whether it is published or a default. */
|
|
72922
73004
|
function limitsInventory() {
|
|
@@ -73112,7 +73194,7 @@ const guardIdentities = new Map();
|
|
|
73112
73194
|
function rateLimiterFor(identity) {
|
|
73113
73195
|
let limiter = rateLimiters.get(identity.key);
|
|
73114
73196
|
if (limiter === undefined) {
|
|
73115
|
-
const limits = limitsFor(identity.provider);
|
|
73197
|
+
const limits = limitsFor(identity.provider, identity.modelId);
|
|
73116
73198
|
limiter = new TokenBucketRateLimiter({
|
|
73117
73199
|
maxTokens: limits.requests_per_minute,
|
|
73118
73200
|
refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
|
|
@@ -73133,7 +73215,7 @@ function rateLimiterFor(identity) {
|
|
|
73133
73215
|
function concurrencyGateFor(identity) {
|
|
73134
73216
|
let gate = concurrencyGates.get(identity.key);
|
|
73135
73217
|
if (gate === undefined) {
|
|
73136
|
-
gate = new ConcurrencyGate(identity, limitsFor(identity.provider).max_concurrent);
|
|
73218
|
+
gate = new ConcurrencyGate(identity, limitsFor(identity.provider, identity.modelId).max_concurrent);
|
|
73137
73219
|
concurrencyGates.set(identity.key, gate);
|
|
73138
73220
|
guardIdentities.set(identity.key, identity);
|
|
73139
73221
|
}
|
|
@@ -73163,8 +73245,8 @@ function concurrencyGateFor(identity) {
|
|
|
73163
73245
|
* or the caller stopped waiting first.
|
|
73164
73246
|
*/
|
|
73165
73247
|
async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
|
|
73166
|
-
const limits = limitsFor(provider);
|
|
73167
73248
|
const identity = guardIdentity(provider, scope.modelId);
|
|
73249
|
+
const limits = limitsFor(provider, identity.modelId);
|
|
73168
73250
|
const waitBudgetMs = maxWaitMs === undefined
|
|
73169
73251
|
? limits.acquire_timeout_ms
|
|
73170
73252
|
: Math.min(maxWaitMs, limits.acquire_timeout_ms);
|
|
@@ -73196,7 +73278,7 @@ function guardSnapshots() {
|
|
|
73196
73278
|
return [...guardIdentities.values()]
|
|
73197
73279
|
.sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))
|
|
73198
73280
|
.map((identity) => {
|
|
73199
|
-
const limits = limitsFor(identity.provider);
|
|
73281
|
+
const limits = limitsFor(identity.provider, identity.modelId);
|
|
73200
73282
|
const limiter = rateLimiters.get(identity.key);
|
|
73201
73283
|
const gate = concurrencyGates.get(identity.key);
|
|
73202
73284
|
return {
|
|
@@ -73516,6 +73598,38 @@ async function runLeg(leg, params, execution, budgetMs) {
|
|
|
73516
73598
|
execution.callerSignal?.removeEventListener("abort", forwardAbort);
|
|
73517
73599
|
}
|
|
73518
73600
|
}
|
|
73601
|
+
/**
|
|
73602
|
+
* HTTP statuses a provider (or the gateway relaying it) uses to say it is full
|
|
73603
|
+
* rather than that the request or the route is wrong: request timeout, too
|
|
73604
|
+
* early, too many requests, service unavailable, and Anthropic's overloaded.
|
|
73605
|
+
*/
|
|
73606
|
+
const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
|
|
73607
|
+
/**
|
|
73608
|
+
* Wording providers use for a capacity refusal when the status is lost on the
|
|
73609
|
+
* way (a relayed body, a client library's own error). DeepInfra's is
|
|
73610
|
+
* "Model busy, retry later"; Anthropic's is "Overloaded".
|
|
73611
|
+
*/
|
|
73612
|
+
const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
|
|
73613
|
+
/**
|
|
73614
|
+
* Whether a failure is the provider saying it is full rather than broken.
|
|
73615
|
+
*
|
|
73616
|
+
* Read by shape rather than by class, because the same signal reaches the
|
|
73617
|
+
* chain from more than one transport and not every transport's error class is
|
|
73618
|
+
* importable here.
|
|
73619
|
+
*
|
|
73620
|
+
* @param error The thrown value.
|
|
73621
|
+
* @param reason Its message.
|
|
73622
|
+
* @returns Whether it is a capacity signal.
|
|
73623
|
+
*/
|
|
73624
|
+
function isCapacitySignal(error, reason) {
|
|
73625
|
+
if (typeof error === "object" && error !== null) {
|
|
73626
|
+
const status = error.status;
|
|
73627
|
+
if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
|
|
73628
|
+
return true;
|
|
73629
|
+
}
|
|
73630
|
+
}
|
|
73631
|
+
return CAPACITY_WORDING.test(reason);
|
|
73632
|
+
}
|
|
73519
73633
|
/**
|
|
73520
73634
|
* Classify why a leg failed.
|
|
73521
73635
|
*
|
|
@@ -73524,6 +73638,13 @@ async function runLeg(leg, params, execution, budgetMs) {
|
|
|
73524
73638
|
* cancellation as a provider failure would let a burst of user-cancelled
|
|
73525
73639
|
* requests open the breaker on a perfectly healthy route.
|
|
73526
73640
|
*
|
|
73641
|
+
* Among failures that do count, a capacity signal (the provider said it is
|
|
73642
|
+
* busy, or the leg ran out its budget waiting on it) is told apart from a hard
|
|
73643
|
+
* failure so the breaker can re-admit a busy route sooner than a broken one. A
|
|
73644
|
+
* timeout is read as capacity: on a reachable provider it is what a full queue
|
|
73645
|
+
* looks like from outside, and a provider that is actually down still costs no
|
|
73646
|
+
* more than one probe per capacity cooldown.
|
|
73647
|
+
*
|
|
73527
73648
|
* @param error The thrown value.
|
|
73528
73649
|
* @param callerSignal The caller's cancellation signal, if any.
|
|
73529
73650
|
* @returns The outcome and whether it counts against route health.
|
|
@@ -73534,24 +73655,45 @@ function classify(error, callerSignal) {
|
|
|
73534
73655
|
outcome: "skipped",
|
|
73535
73656
|
reason: "caller cancelled",
|
|
73536
73657
|
countsAgainstHealth: false,
|
|
73658
|
+
failureKind: "hard",
|
|
73537
73659
|
};
|
|
73538
73660
|
}
|
|
73539
73661
|
if (error instanceof LegTimeoutError) {
|
|
73540
|
-
return {
|
|
73662
|
+
return {
|
|
73663
|
+
outcome: "timeout",
|
|
73664
|
+
reason: error.message,
|
|
73665
|
+
countsAgainstHealth: true,
|
|
73666
|
+
failureKind: "capacity",
|
|
73667
|
+
};
|
|
73541
73668
|
}
|
|
73542
73669
|
if (error instanceof UnsupportedCapabilityError) {
|
|
73543
|
-
return {
|
|
73670
|
+
return {
|
|
73671
|
+
outcome: "skipped",
|
|
73672
|
+
reason: error.message,
|
|
73673
|
+
countsAgainstHealth: false,
|
|
73674
|
+
failureKind: "hard",
|
|
73675
|
+
};
|
|
73544
73676
|
}
|
|
73545
73677
|
if (error instanceof ToolChoiceIgnoredError) {
|
|
73546
73678
|
// The route answered; it broke a declared guarantee rather than failing to
|
|
73547
73679
|
// be available, so its breaker is not charged for it.
|
|
73548
|
-
return {
|
|
73680
|
+
return {
|
|
73681
|
+
outcome: "error",
|
|
73682
|
+
reason: error.message,
|
|
73683
|
+
countsAgainstHealth: false,
|
|
73684
|
+
failureKind: "hard",
|
|
73685
|
+
};
|
|
73549
73686
|
}
|
|
73687
|
+
// Self-inflicted pacing, not provider ill-health. Counting it would let the
|
|
73688
|
+
// client's own throttling open a breaker on a perfectly healthy provider and
|
|
73689
|
+
// permanently reroute traffic nobody chose to reroute.
|
|
73550
73690
|
if (error instanceof RateGuardTimeoutError) {
|
|
73551
|
-
|
|
73552
|
-
|
|
73553
|
-
|
|
73554
|
-
|
|
73691
|
+
return {
|
|
73692
|
+
outcome: "skipped",
|
|
73693
|
+
reason: error.message,
|
|
73694
|
+
countsAgainstHealth: false,
|
|
73695
|
+
failureKind: "hard",
|
|
73696
|
+
};
|
|
73555
73697
|
}
|
|
73556
73698
|
const reason = error instanceof Error ? error.message : String(error);
|
|
73557
73699
|
if (/abort/i.test(reason)) {
|
|
@@ -73559,9 +73701,20 @@ function classify(error, callerSignal) {
|
|
|
73559
73701
|
outcome: "timeout",
|
|
73560
73702
|
reason: `aborted: ${reason}`,
|
|
73561
73703
|
countsAgainstHealth: true,
|
|
73704
|
+
failureKind: "capacity",
|
|
73562
73705
|
};
|
|
73563
73706
|
}
|
|
73564
|
-
|
|
73707
|
+
if (error instanceof LlmResponseFormatError) {
|
|
73708
|
+
// The provider answered, badly. That is a route defect, not a full queue,
|
|
73709
|
+
// whatever words the unparseable content happens to contain.
|
|
73710
|
+
return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
|
|
73711
|
+
}
|
|
73712
|
+
return {
|
|
73713
|
+
outcome: "error",
|
|
73714
|
+
reason,
|
|
73715
|
+
countsAgainstHealth: true,
|
|
73716
|
+
failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
|
|
73717
|
+
};
|
|
73565
73718
|
}
|
|
73566
73719
|
/**
|
|
73567
73720
|
* Whether the caller has stopped waiting.
|
|
@@ -73680,9 +73833,9 @@ async function executeChain(alias, execution) {
|
|
|
73680
73833
|
return { response, servedBy: route, attempts, totalUsage };
|
|
73681
73834
|
}
|
|
73682
73835
|
catch (error) {
|
|
73683
|
-
const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
|
|
73836
|
+
const { outcome, reason, countsAgainstHealth, failureKind } = classify(error, execution.callerSignal);
|
|
73684
73837
|
if (countsAgainstHealth) {
|
|
73685
|
-
execution.breakers.onFailure(route.routeKey);
|
|
73838
|
+
execution.breakers.onFailure(route.routeKey, failureKind);
|
|
73686
73839
|
}
|
|
73687
73840
|
else if (holdsProbe) {
|
|
73688
73841
|
// No verdict on the route's health, but the probe slot this attempt
|
|
@@ -73730,6 +73883,7 @@ var defaults = {
|
|
|
73730
73883
|
circuit_breaker: {
|
|
73731
73884
|
failure_threshold: 5,
|
|
73732
73885
|
cooldown_ms: 60000,
|
|
73886
|
+
capacity_cooldown_ms: 15000,
|
|
73733
73887
|
half_open_probes: 1
|
|
73734
73888
|
}
|
|
73735
73889
|
};
|