@adaptic/utils 0.0.1037 → 0.0.1039

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/index.cjs +1885 -294
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.mjs +1869 -295
  4. package/dist/index.mjs.map +1 -1
  5. package/dist/types/__tests__/llm/client/support/legacy-scenarios.d.ts +48 -0
  6. package/dist/types/__tests__/llm/client/support/legacy-scenarios.d.ts.map +1 -0
  7. package/dist/types/__tests__/llm/client/support/transports.d.ts +18 -0
  8. package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -1
  9. package/dist/types/alpaca/index.d.ts.map +1 -1
  10. package/dist/types/alpaca/trading/index.d.ts +1 -0
  11. package/dist/types/alpaca/trading/index.d.ts.map +1 -1
  12. package/dist/types/alpaca/trading/trail-limits.d.ts +8 -0
  13. package/dist/types/alpaca/trading/trail-limits.d.ts.map +1 -0
  14. package/dist/types/alpaca/trading/trail-unit.d.ts +98 -0
  15. package/dist/types/alpaca/trading/trail-unit.d.ts.map +1 -0
  16. package/dist/types/alpaca/trading/trailing-stops.d.ts +28 -15
  17. package/dist/types/alpaca/trading/trailing-stops.d.ts.map +1 -1
  18. package/dist/types/alpaca-trading-api.d.ts.map +1 -1
  19. package/dist/types/index.d.ts.map +1 -1
  20. package/dist/types/llm/alias-client.d.ts +10 -1
  21. package/dist/types/llm/alias-client.d.ts.map +1 -1
  22. package/dist/types/llm/circuit-breaker.d.ts +109 -3
  23. package/dist/types/llm/circuit-breaker.d.ts.map +1 -1
  24. package/dist/types/llm/fallback-chain.d.ts +113 -15
  25. package/dist/types/llm/fallback-chain.d.ts.map +1 -1
  26. package/dist/types/llm/hedge.d.ts +107 -0
  27. package/dist/types/llm/hedge.d.ts.map +1 -0
  28. package/dist/types/llm/index.d.ts +11 -8
  29. package/dist/types/llm/index.d.ts.map +1 -1
  30. package/dist/types/llm/leg-attempt.d.ts +165 -0
  31. package/dist/types/llm/leg-attempt.d.ts.map +1 -0
  32. package/dist/types/llm/leg-latency-tracker.d.ts +126 -0
  33. package/dist/types/llm/leg-latency-tracker.d.ts.map +1 -0
  34. package/dist/types/llm/rate-guard.d.ts +46 -2
  35. package/dist/types/llm/rate-guard.d.ts.map +1 -1
  36. package/dist/types/llm/route-table.d.ts +17 -1
  37. package/dist/types/llm/route-table.d.ts.map +1 -1
  38. package/dist/types/llm/transports/gateway.d.ts +27 -2
  39. package/dist/types/llm/transports/gateway.d.ts.map +1 -1
  40. package/dist/types/llm/types.d.ts +167 -1
  41. package/dist/types/llm/types.d.ts.map +1 -1
  42. package/dist/types/schemas/alpaca-schemas.d.ts +16 -16
  43. package/dist/types/schemas/massive-schemas.d.ts +10 -10
  44. package/dist/types/trading-policy/schemas/effective-policy.schema.d.ts +6 -6
  45. package/dist/types/trading-policy/schemas/policy-mutation.schema.d.ts +12 -12
  46. package/dist/types/trading-policy/schemas/signal-consumption-prefs.schema.d.ts +8 -8
  47. package/package.json +1 -1
package/dist/index.cjs CHANGED
@@ -10090,6 +10090,160 @@ class AlpacaMarketDataAPI extends require$$0$4.EventEmitter {
10090
10090
  // Export the singleton instance
10091
10091
  const marketDataAPI = AlpacaMarketDataAPI.getInstance();
10092
10092
 
10093
+ /**
10094
+ * Alpaca's hard upper limit for `trail_percent` on trailing-stop orders.
10095
+ * Submissions exceeding this value are rejected with HTTP 422 / code 42210000
10096
+ * ("trail_percent must be <= 25"). See:
10097
+ * https://docs.alpaca.markets/reference/postorder
10098
+ */
10099
+ const ALPACA_MAX_TRAIL_PERCENT = 25;
10100
+
10101
+ /**
10102
+ * Trailing-stop replace-unit contract.
10103
+ *
10104
+ * Alpaca's order replace (`PATCH /v2/orders/{id}`) takes a single unitless
10105
+ * `trail` field. The broker reads it in the unit of the ORIGINAL order: a
10106
+ * `trail_percent` order reads `trail` as a percent, a `trail_price` order
10107
+ * reads it as dollars. A replace cannot change the unit. A caller that holds
10108
+ * a distance in one unit must therefore resolve it against the resting
10109
+ * order's unit before the replace, or the broker stores the number in the
10110
+ * other unit: a $22 dollar distance becomes a 22% trail.
10111
+ *
10112
+ * This module is the pure resolution step. It never guesses a unit, never
10113
+ * defaults a reference price, and refuses rather than send a value the broker
10114
+ * would reject or treat as effectively zero.
10115
+ */
10116
+ /**
10117
+ * Smallest `trail_percent` this module will send. Below it the trail is a
10118
+ * rounding artefact of the broker's tick grid rather than a protective
10119
+ * distance, so a conversion landing under it is refused.
10120
+ */
10121
+ const MIN_CONVERTED_TRAIL_PERCENT = 0.1;
10122
+ /** Percent values are sent to the broker at hundredths of a percent. */
10123
+ const PERCENT_DECIMALS_SCALE = 100;
10124
+ /** A ratio expressed in percent. */
10125
+ const PERCENT_PER_UNIT = 100;
10126
+ /**
10127
+ * Thrown when a trail replace cannot be expressed in the resting order's unit
10128
+ * without guessing. No replace is sent; the resting stop keeps protecting.
10129
+ */
10130
+ class TrailUnitConversionRefusedError extends AdapticUtilsError {
10131
+ /** The order the replace targeted. */
10132
+ orderId;
10133
+ /** Why the replace was refused. */
10134
+ reason;
10135
+ /** The converted percent, or `null` when no conversion was computed. */
10136
+ pct;
10137
+ /** The conversion reference price, or `null` when none was usable. */
10138
+ ref;
10139
+ constructor(params) {
10140
+ super(`Trailing stop replace refused for ${params.orderId} (${params.reason}): ${params.detail}`, "TRAIL_UNIT_REFUSED", "alpaca", false);
10141
+ this.orderId = params.orderId;
10142
+ this.reason = params.reason;
10143
+ this.pct = params.pct;
10144
+ this.ref = params.ref;
10145
+ }
10146
+ }
10147
+ /** Parse a broker decimal string; `null` when absent, non-finite or not positive. */
10148
+ function positiveOrNull(value) {
10149
+ if (value === null || value === undefined || value === "") {
10150
+ return null;
10151
+ }
10152
+ const parsed = Number(value);
10153
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : null;
10154
+ }
10155
+ /**
10156
+ * Read the unit a resting trailing-stop order trails in.
10157
+ *
10158
+ * @param order - The resting order as returned by the broker.
10159
+ * @returns `"price"` for a dollar trail, `"percent"` for a percent trail, or
10160
+ * `null` when neither (or both) unit fields carry a positive value.
10161
+ */
10162
+ function readTrailUnit(order) {
10163
+ const hasPrice = positiveOrNull(order.trail_price) !== null;
10164
+ const hasPercent = positiveOrNull(order.trail_percent) !== null;
10165
+ if (hasPrice === hasPercent) {
10166
+ return null;
10167
+ }
10168
+ return hasPrice ? "price" : "percent";
10169
+ }
10170
+ /**
10171
+ * Resolve the replace `trail` value for a requested trail against the resting
10172
+ * order's unit.
10173
+ *
10174
+ * - A dollar distance on a dollar order is sent unchanged.
10175
+ * - A dollar distance on a percent order is converted against
10176
+ * `ref = max(hwm, stop_price)`. For a long the HWM is at or above the live
10177
+ * price; for a short the stop is above it. So `ref` is at or above live on
10178
+ * both sides, and `distance / ref` is at or below `distance / live`: the
10179
+ * resulting stop is at or tighter than `live ∓ distance`. The percent is
10180
+ * rounded down to hundredths, which only tightens it further.
10181
+ * - A percent on a percent order is sent unchanged.
10182
+ * - A percent on a dollar order is refused. There is no conversion that keeps
10183
+ * the caller's intent without a live price this seam does not own.
10184
+ *
10185
+ * @param orderId - The order being replaced (for the refusal record).
10186
+ * @param order - The resting order as returned by the broker.
10187
+ * @param requested - Exactly one of a positive dollar distance or percent.
10188
+ * @returns The resolved `trail` value and its unit.
10189
+ * @throws {TrailUnitConversionRefusedError} When the unit is unknown, no
10190
+ * finite reference exists, the converted percent falls outside
10191
+ * [{@link MIN_CONVERTED_TRAIL_PERCENT}, {@link ALPACA_MAX_TRAIL_PERCENT}],
10192
+ * or a percent targets a dollar order.
10193
+ */
10194
+ function resolveReplaceTrail(orderId, order, requested) {
10195
+ const unit = readTrailUnit(order);
10196
+ if (unit === null) {
10197
+ throw new TrailUnitConversionRefusedError({
10198
+ orderId,
10199
+ reason: "unit_unknown",
10200
+ pct: null,
10201
+ ref: null,
10202
+ detail: `order carries trail_price=${String(order.trail_price)} trail_percent=${String(order.trail_percent)}; the replace unit cannot be determined`,
10203
+ });
10204
+ }
10205
+ if ("trailPercent" in requested) {
10206
+ if (unit === "price") {
10207
+ throw new TrailUnitConversionRefusedError({
10208
+ orderId,
10209
+ reason: "percent_on_price_order",
10210
+ pct: requested.trailPercent,
10211
+ ref: null,
10212
+ detail: `a ${requested.trailPercent}% trail sent to a dollar-trail order would be stored as $${requested.trailPercent}`,
10213
+ });
10214
+ }
10215
+ return { trail: requested.trailPercent.toString(), unit, referencePrice: null };
10216
+ }
10217
+ if (unit === "price") {
10218
+ return { trail: requested.trailPrice.toString(), unit, referencePrice: null };
10219
+ }
10220
+ const candidates = [positiveOrNull(order.hwm), positiveOrNull(order.stop_price)].filter((value) => value !== null);
10221
+ if (candidates.length === 0) {
10222
+ throw new TrailUnitConversionRefusedError({
10223
+ orderId,
10224
+ reason: "reference_unavailable",
10225
+ pct: null,
10226
+ ref: null,
10227
+ detail: `percent-trail order has no finite hwm (${String(order.hwm)}) or stop_price (${String(order.stop_price)}) to convert $${requested.trailPrice} against`,
10228
+ });
10229
+ }
10230
+ const ref = Math.max(...candidates);
10231
+ const pct = Math.floor((requested.trailPrice / ref) * PERCENT_PER_UNIT * PERCENT_DECIMALS_SCALE) /
10232
+ PERCENT_DECIMALS_SCALE;
10233
+ if (!Number.isFinite(pct) ||
10234
+ pct < MIN_CONVERTED_TRAIL_PERCENT ||
10235
+ pct > ALPACA_MAX_TRAIL_PERCENT) {
10236
+ throw new TrailUnitConversionRefusedError({
10237
+ orderId,
10238
+ reason: "converted_percent_out_of_range",
10239
+ pct,
10240
+ ref,
10241
+ detail: `$${requested.trailPrice} against ref ${ref} is ${pct}%, outside [${MIN_CONVERTED_TRAIL_PERCENT}, ${ALPACA_MAX_TRAIL_PERCENT}]`,
10242
+ });
10243
+ }
10244
+ return { trail: pct.toFixed(2), unit, referencePrice: ref };
10245
+ }
10246
+
10093
10247
  const limitPriceSlippagePercent100 = 0.1; // 0.1%
10094
10248
  /**
10095
10249
  * Alpaca's maximum page size for GET /orders — also our explicit default.
@@ -11039,12 +11193,18 @@ class AlpacaTradingAPI {
11039
11193
  return null;
11040
11194
  }
11041
11195
  const originalOrderId = trailingStopOrder.id;
11196
+ // Alpaca reads the replace `trail` in the resting order's unit. A percent
11197
+ // sent to a dollar-trail order would be stored as dollars, so it is refused
11198
+ // (typed, no replace sent) and the resting dollar trail keeps protecting.
11199
+ const resolvedTrail = resolveReplaceTrail(originalOrderId, trailingStopOrder, {
11200
+ trailPercent: trailPercent100,
11201
+ });
11042
11202
  this.log(`Updating trailing stop for ${symbol} from ${currentTrailPercent}% to ${trailPercent100}% (orderId=${originalOrderId})`, {
11043
11203
  symbol,
11044
11204
  });
11045
11205
  try {
11046
11206
  const updatedOrder = await this.makeRequest(`/orders/${trailingStopOrder.id}`, "PATCH", {
11047
- trail: trailPercent100.toString(),
11207
+ trail: resolvedTrail.trail,
11048
11208
  });
11049
11209
  // Log the replacement: Alpaca replaces orders on PATCH, so new ID is returned
11050
11210
  this.log(`Trailing stop updated for ${symbol}: newOrderId=${updatedOrder.id}, replaces=${updatedOrder.replaces || originalOrderId}`, { symbol });
@@ -63460,13 +63620,6 @@ var orderUtils$1 = /*#__PURE__*/Object.freeze({
63460
63620
  });
63461
63621
 
63462
63622
  const LOG_SOURCE$7 = "TrailingStops";
63463
- /**
63464
- * Alpaca's hard upper limit for `trail_percent` on trailing-stop orders.
63465
- * Submissions exceeding this value are rejected with HTTP 422 / code 42210000
63466
- * ("trail_percent must be <= 25"). See:
63467
- * https://docs.alpaca.markets/reference/postorder
63468
- */
63469
- const ALPACA_MAX_TRAIL_PERCENT = 25;
63470
63623
  /**
63471
63624
  * Internal logging helper with consistent source
63472
63625
  */
@@ -63595,23 +63748,42 @@ async function createTrailingStop(client, params) {
63595
63748
  }
63596
63749
  }
63597
63750
  /**
63598
- * Update an existing trailing stop order
63751
+ * Update the trail distance of an existing trailing stop order.
63752
+ *
63753
+ * ## Unit contract
63599
63754
  *
63600
- * You can update the trail_percent or trail_price of an existing order.
63601
- * Note: Alpaca uses 'trail' parameter for replacements (works for both percent and price).
63755
+ * Alpaca's replace takes a single unitless `trail` field and reads it in the
63756
+ * unit of the ORIGINAL order; a replace cannot change the unit. This function
63757
+ * therefore reads the resting order first and resolves the request against its
63758
+ * unit ({@link resolveReplaceTrail}):
63759
+ *
63760
+ * - `trailPrice` on a dollar-trail order is sent as dollars, unchanged.
63761
+ * - `trailPrice` on a percent-trail order is converted to a percent against
63762
+ * `max(hwm, stop_price)`, which is at or above the live price on both sides,
63763
+ * so the resulting stop is at or tighter than `live ∓ trailPrice`. The
63764
+ * percent is rounded down to hundredths (tighter).
63765
+ * - `trailPercent` on a percent-trail order is sent unchanged.
63766
+ * - `trailPercent` on a dollar-trail order is refused.
63767
+ *
63768
+ * A refusal throws {@link TrailUnitConversionRefusedError} and sends no
63769
+ * replace, so the resting stop keeps protecting. The unit is never guessed and
63770
+ * the conversion reference is never defaulted.
63602
63771
  *
63603
63772
  * @param client - AlpacaClient instance
63604
63773
  * @param orderId - The ID of the order to update
63605
- * @param updates - New trail parameters (specify one of trailPercent or trailPrice)
63606
- * @returns The updated order
63607
- * @throws {Error} If no update parameters provided or update fails
63774
+ * @param updates - New trail parameters (specify exactly one of trailPercent or trailPrice)
63775
+ * @returns The replacement order
63776
+ * @throws {Error} If no/both update parameters are given, or a value is not positive
63777
+ * @throws {TrailUnitConversionRefusedError} If the request cannot be expressed
63778
+ * in the resting order's unit
63779
+ * @throws {AlpacaApiError} If the order read or the replace fails at the broker
63608
63780
  *
63609
63781
  * @example
63610
63782
  * ```typescript
63611
- * // Tighten trailing stop to 1.5%
63783
+ * // Tighten a percent trailing stop to 1.5%
63612
63784
  * await updateTrailingStop(client, 'order-id-123', { trailPercent: 1.5 });
63613
63785
  *
63614
- * // Change to $3 trail
63786
+ * // Pin a $3 trail distance (converted when the order trails in percent)
63615
63787
  * await updateTrailingStop(client, 'order-id-123', { trailPrice: 3.00 });
63616
63788
  * ```
63617
63789
  */
@@ -63631,21 +63803,41 @@ async function updateTrailingStop(client, orderId, updates) {
63631
63803
  throw new Error("trailPrice must be greater than 0");
63632
63804
  }
63633
63805
  const sdk = client.getSDK();
63634
- const updateDescription = updates.trailPercent
63635
- ? `${updates.trailPercent}%`
63636
- : `$${updates.trailPrice?.toFixed(2)}`;
63806
+ const requested = updates.trailPercent !== undefined
63807
+ ? { trailPercent: updates.trailPercent }
63808
+ : { trailPrice: updates.trailPrice };
63809
+ const updateDescription = "trailPercent" in requested
63810
+ ? `${requested.trailPercent}%`
63811
+ : `$${requested.trailPrice.toFixed(2)}`;
63637
63812
  log$g(`Updating trailing stop ${orderId} to trail: ${updateDescription}`, {
63638
63813
  type: "info",
63639
63814
  });
63815
+ let resting;
63640
63816
  try {
63641
- const replaceParams = {};
63642
- // Alpaca's replaceOrder uses 'trail' for both percent and price updates
63643
- if (updates.trailPercent !== undefined) {
63644
- replaceParams.trail = updates.trailPercent.toString();
63645
- }
63646
- else if (updates.trailPrice !== undefined) {
63647
- replaceParams.trail = updates.trailPrice.toString();
63648
- }
63817
+ resting = (await sdk.getOrder(orderId));
63818
+ }
63819
+ catch (error) {
63820
+ const err = error;
63821
+ log$g(`Trailing stop update aborted for ${orderId}: order read failed: ${err.message}`, {
63822
+ type: "error",
63823
+ });
63824
+ throw enrichAlpacaError(new Error(`Failed to read trailing stop ${orderId} before update: ${err.message}`), error);
63825
+ }
63826
+ let resolved;
63827
+ try {
63828
+ resolved = resolveReplaceTrail(orderId, resting, requested);
63829
+ }
63830
+ catch (refusal) {
63831
+ log$g(`Trailing stop update refused for ${orderId}: ${refusal.message}`, {
63832
+ type: "warn",
63833
+ });
63834
+ throw refusal;
63835
+ }
63836
+ if (resolved.referencePrice !== null) {
63837
+ log$g(`Trailing stop ${orderId}: ${updateDescription} on percent order → ${resolved.trail}% (ref ${resolved.referencePrice})`, { type: "info" });
63838
+ }
63839
+ try {
63840
+ const replaceParams = { trail: resolved.trail };
63649
63841
  const order = await sdk.replaceOrder(orderId, replaceParams);
63650
63842
  log$g(`Trailing stop updated: orderId=${order.id}, new replacement created`, {
63651
63843
  type: "info",
@@ -72350,6 +72542,192 @@ const alpaca = {
72350
72542
  streams: streams$1,
72351
72543
  };
72352
72544
 
72545
+ /**
72546
+ * Rolling healthy-latency evidence per (provider, model), split by prompt size.
72547
+ *
72548
+ * The chain's timeouts and hedge points are only as good as its idea of how
72549
+ * long a healthy answer takes. A flat route budget encodes no such idea: it
72550
+ * treats a thirty-second wait on a model whose healthy answers arrive in four
72551
+ * the same as a thirty-second wait on one that needs twenty-five. This tracker
72552
+ * supplies the measured alternative.
72553
+ *
72554
+ * It is keyed by provider and model rather than by alias and role, because
72555
+ * health is a property of the model at its host: the same model reached as one
72556
+ * alias's primary and another alias's secondary is one population, and
72557
+ * splitting it would halve the evidence each side sees. Samples are split by
72558
+ * prompt size, because generation time grows with input and a model that
72559
+ * struggles on large prompts would otherwise have its large-prompt tail hidden
72560
+ * by a crowd of small, fast calls.
72561
+ *
72562
+ * Only healthy (answered) attempts are recorded. A timeout says the answer
72563
+ * took at least the budget, not how long it took, and folding it in would pull
72564
+ * the quantiles toward whatever budget happened to be configured.
72565
+ *
72566
+ * Unknown stays unknown: a cell with fewer than the minimum samples answers
72567
+ * `null`, and every consumer treats `null` as "no evidence", never as a value.
72568
+ *
72569
+ * @module llm/leg-latency-tracker
72570
+ */
72571
+ /** Characters per token used to size a prompt before the provider has counted it. */
72572
+ const CHARS_PER_TOKEN = 4;
72573
+ /** The bucket used when a prompt's size could not be estimated. */
72574
+ const UNKNOWN_BUCKET = "unknown";
72575
+ /**
72576
+ * Estimate a prompt's size in tokens from its serialised length.
72577
+ *
72578
+ * Only used to choose a size bucket, and the same estimate is applied when a
72579
+ * sample is recorded and when it is looked up, so its bias cancels. Content
72580
+ * that cannot be serialised yields `null` rather than a guessed size.
72581
+ *
72582
+ * @param parts The prompt, developer instruction and prior turns.
72583
+ * @returns The estimated token count, or null.
72584
+ */
72585
+ function estimatePromptTokens(parts) {
72586
+ let characters = 0;
72587
+ for (const part of parts) {
72588
+ if (part === undefined) {
72589
+ continue;
72590
+ }
72591
+ if (typeof part === "string") {
72592
+ characters += part.length;
72593
+ continue;
72594
+ }
72595
+ try {
72596
+ const serialised = JSON.stringify(part);
72597
+ if (typeof serialised !== "string") {
72598
+ return null;
72599
+ }
72600
+ characters += serialised.length;
72601
+ }
72602
+ catch {
72603
+ // A circular or otherwise unserialisable part has no knowable size; the
72604
+ // caller files its latency under the unknown bucket rather than a guess.
72605
+ return null;
72606
+ }
72607
+ }
72608
+ return Math.ceil(characters / CHARS_PER_TOKEN);
72609
+ }
72610
+ /**
72611
+ * The nearest-rank quantile of a sorted sample.
72612
+ *
72613
+ * @param sorted Ascending values; must be non-empty.
72614
+ * @param q The quantile, in (0, 1).
72615
+ * @returns The value at that rank.
72616
+ */
72617
+ function nearestRank(sorted, q) {
72618
+ const rank = Math.min(sorted.length, Math.max(1, Math.ceil(q * sorted.length)));
72619
+ return sorted[rank - 1];
72620
+ }
72621
+ /**
72622
+ * Per-process healthy-latency windows.
72623
+ */
72624
+ class LegLatencyTracker {
72625
+ cells = new Map();
72626
+ config;
72627
+ now;
72628
+ /**
72629
+ * @param config Window sizing and prompt-size buckets.
72630
+ * @param now Clock, injected so ageing is testable without waiting.
72631
+ */
72632
+ constructor(config, now = Date.now) {
72633
+ this.config = config;
72634
+ this.now = now;
72635
+ }
72636
+ /**
72637
+ * The size bucket a prompt falls in.
72638
+ *
72639
+ * @param promptTokens Estimated prompt tokens, or null when unknown.
72640
+ * @returns The bucket label.
72641
+ */
72642
+ bucketOf(promptTokens) {
72643
+ if (promptTokens === null || !Number.isFinite(promptTokens)) {
72644
+ return UNKNOWN_BUCKET;
72645
+ }
72646
+ const index = this.config.promptTokenBuckets.findIndex((edge) => promptTokens < edge);
72647
+ return String(index === -1 ? this.config.promptTokenBuckets.length : index);
72648
+ }
72649
+ /**
72650
+ * Record one healthy (answered) attempt.
72651
+ *
72652
+ * @param provider The provider that answered.
72653
+ * @param modelId The model it answered with.
72654
+ * @param promptTokens Estimated prompt tokens, or null.
72655
+ * @param durationMs How long the answer took.
72656
+ * @returns void
72657
+ */
72658
+ record(provider, modelId, promptTokens, durationMs) {
72659
+ if (!Number.isFinite(durationMs) || durationMs < 0) {
72660
+ return;
72661
+ }
72662
+ const key = this.cellKey(provider, modelId, promptTokens);
72663
+ const samples = this.fresh(key);
72664
+ samples.push({ atMs: this.now(), durationMs });
72665
+ while (samples.length > this.config.windowSize) {
72666
+ samples.shift();
72667
+ }
72668
+ this.cells.set(key, samples);
72669
+ }
72670
+ /**
72671
+ * A healthy-latency quantile, or null without enough evidence.
72672
+ *
72673
+ * @param provider The provider.
72674
+ * @param modelId The model.
72675
+ * @param promptTokens Estimated prompt tokens, or null.
72676
+ * @param q The quantile, in (0, 1).
72677
+ * @returns The quantile in milliseconds, or null.
72678
+ */
72679
+ quantile(provider, modelId, promptTokens, q) {
72680
+ const samples = this.fresh(this.cellKey(provider, modelId, promptTokens));
72681
+ if (samples.length < this.config.minSamples) {
72682
+ return null;
72683
+ }
72684
+ const sorted = samples.map((sample) => sample.durationMs).sort((a, b) => a - b);
72685
+ return nearestRank(sorted, q);
72686
+ }
72687
+ /**
72688
+ * How many fresh samples a cell holds.
72689
+ *
72690
+ * @param provider The provider.
72691
+ * @param modelId The model.
72692
+ * @param promptTokens Estimated prompt tokens, or null.
72693
+ * @returns The count.
72694
+ */
72695
+ sampleCount(provider, modelId, promptTokens) {
72696
+ return this.fresh(this.cellKey(provider, modelId, promptTokens)).length;
72697
+ }
72698
+ /**
72699
+ * Discard all evidence.
72700
+ *
72701
+ * @returns void
72702
+ */
72703
+ reset() {
72704
+ this.cells.clear();
72705
+ }
72706
+ /**
72707
+ * @param provider The provider.
72708
+ * @param modelId The model.
72709
+ * @param promptTokens Estimated prompt tokens, or null.
72710
+ * @returns The cell key.
72711
+ */
72712
+ cellKey(provider, modelId, promptTokens) {
72713
+ return `${provider}/${modelId}@${this.bucketOf(promptTokens)}`;
72714
+ }
72715
+ /**
72716
+ * A cell's samples with the stale ones removed.
72717
+ *
72718
+ * @param key The cell key.
72719
+ * @returns The fresh samples (the stored array, pruned in place).
72720
+ */
72721
+ fresh(key) {
72722
+ const samples = this.cells.get(key) ?? [];
72723
+ const oldest = this.now() - this.config.sampleMaxAgeMs;
72724
+ while (samples.length > 0 && samples[0].atMs < oldest) {
72725
+ samples.shift();
72726
+ }
72727
+ return samples;
72728
+ }
72729
+ }
72730
+
72353
72731
  /**
72354
72732
  * Per-route circuit breaker for the alias client.
72355
72733
  *
@@ -72367,16 +72745,57 @@ const alpaca = {
72367
72745
  * traffic can neither trip nor be tripped by traffic on the other side of the
72368
72746
  * PD-9 boundary.
72369
72747
  *
72748
+ * How long a tripped breaker stays open depends on WHY it tripped. A provider
72749
+ * that is refusing work because it is momentarily full ("model busy", 429,
72750
+ * 503, 529, or a leg that ran out its budget queued behind other traffic) is
72751
+ * shedding load it expects to take back within seconds; excluding it for a
72752
+ * full minute pushes every call onto the next leg, which is how one provider's
72753
+ * capacity blip becomes the last leg's overload. Such a run opens for the
72754
+ * shorter `capacity_cooldown_ms`. A run that contains any hard failure (a
72755
+ * rejected credential, a malformed request, an unreachable gateway, an answer
72756
+ * that does not parse) says the route is broken rather than busy, and opens
72757
+ * for the full `cooldown_ms`. Either way the route then admits a bounded number
72758
+ * of half-open probes, and one success closes it.
72759
+ *
72760
+ * A route can also be opened by LATENCY (when `latency_trip` is armed): a
72761
+ * provider whose answers arrive, but later than the latency class's objective
72762
+ * in most recent windows, is failing a hot path without ever producing an
72763
+ * error. A latency-opened breaker is not closed by an answer from an attempt
72764
+ * that started before it opened, because such an answer is exactly the slow
72765
+ * evidence that opened it.
72766
+ *
72767
+ * The half-open probe budget scales with how much traffic the route carried
72768
+ * before it opened (`probe_fraction`), so a route that served forty calls at
72769
+ * once is not re-tested by a single probe whose one slow answer decides it.
72770
+ *
72370
72771
  * The clock is injected. Breaker behaviour is entirely about elapsed time, and
72371
72772
  * a test that must sleep to observe a cooldown is a test nobody runs.
72372
72773
  *
72373
72774
  * @module llm/circuit-breaker
72374
72775
  */
72776
+ /**
72777
+ * @returns A record for a route with no failures on file.
72778
+ */
72779
+ function freshRecord() {
72780
+ return {
72781
+ consecutiveFailures: 0,
72782
+ openedAtMs: null,
72783
+ probesInFlight: 0,
72784
+ runHasHardFailure: false,
72785
+ openedByLatency: false,
72786
+ concurrencyAtOpen: 0,
72787
+ };
72788
+ }
72375
72789
  /**
72376
72790
  * Tracks route health and decides whether a leg may be attempted.
72377
72791
  */
72378
72792
  class CircuitBreakerRegistry {
72379
72793
  records = new Map();
72794
+ latency = new Map();
72795
+ /** Attempts currently in flight per route, for scaling the probe budget. */
72796
+ inFlight = new Map();
72797
+ /** Peak of {@link inFlight} since the route last opened. */
72798
+ peakInFlight = new Map();
72380
72799
  config;
72381
72800
  now;
72382
72801
  /**
@@ -72404,7 +72823,26 @@ class CircuitBreakerRegistry {
72404
72823
  return "closed";
72405
72824
  }
72406
72825
  const elapsed = this.now() - record.openedAtMs;
72407
- return elapsed >= this.config.cooldown_ms ? "half-open" : "open";
72826
+ return elapsed >= this.cooldownFor(record) ? "half-open" : "open";
72827
+ }
72828
+ /**
72829
+ * The cooldown a record's current run earns.
72830
+ *
72831
+ * A run made only of capacity failures earns the capacity cooldown; one hard
72832
+ * failure anywhere in the run earns the full one. Mixed evidence is read as
72833
+ * the worse case, because a route that is both busy and broken is broken.
72834
+ * The capacity cooldown is never allowed to exceed the full one, so a
72835
+ * misconfigured table cannot make busy routes wait longer than broken ones.
72836
+ *
72837
+ * @param record The route's record.
72838
+ * @returns The cooldown in milliseconds.
72839
+ */
72840
+ cooldownFor(record) {
72841
+ const capacityCooldown = this.config.capacity_cooldown_ms ?? this.config.cooldown_ms;
72842
+ if (record.runHasHardFailure) {
72843
+ return this.config.cooldown_ms;
72844
+ }
72845
+ return Math.min(capacityCooldown, this.config.cooldown_ms);
72408
72846
  }
72409
72847
  /**
72410
72848
  * Whether a route may be attempted now.
@@ -72426,7 +72864,20 @@ class CircuitBreakerRegistry {
72426
72864
  return false;
72427
72865
  }
72428
72866
  const record = this.recordFor(routeKey);
72429
- return record.probesInFlight < this.config.half_open_probes;
72867
+ return record.probesInFlight < this.probeBudgetFor(record);
72868
+ }
72869
+ /**
72870
+ * How many half-open probes a route admits at once.
72871
+ *
72872
+ * @param record The route's record.
72873
+ * @returns The larger of the configured floor and the concurrency-scaled budget.
72874
+ */
72875
+ probeBudgetFor(record) {
72876
+ const fraction = this.config.probe_fraction;
72877
+ if (fraction === undefined || fraction <= 0) {
72878
+ return this.config.half_open_probes;
72879
+ }
72880
+ return Math.max(this.config.half_open_probes, Math.ceil(fraction * record.concurrencyAtOpen));
72430
72881
  }
72431
72882
  /**
72432
72883
  * Register that an attempt is starting, so half-open probes stay bounded.
@@ -72437,6 +72888,9 @@ class CircuitBreakerRegistry {
72437
72888
  * {@link onAttemptAbandoned}, or the slot is never returned.
72438
72889
  */
72439
72890
  onAttemptStart(routeKey) {
72891
+ const inFlight = (this.inFlight.get(routeKey) ?? 0) + 1;
72892
+ this.inFlight.set(routeKey, inFlight);
72893
+ this.peakInFlight.set(routeKey, Math.max(this.peakInFlight.get(routeKey) ?? 0, inFlight));
72440
72894
  if (this.stateOf(routeKey) === "half-open") {
72441
72895
  this.recordFor(routeKey).probesInFlight += 1;
72442
72896
  return true;
@@ -72463,6 +72917,17 @@ class CircuitBreakerRegistry {
72463
72917
  record.probesInFlight -= 1;
72464
72918
  }
72465
72919
  }
72920
+ /**
72921
+ * Mark an attempt started with {@link onAttemptStart} as finished, whatever
72922
+ * its outcome, so the in-flight count that scales the probe budget stays true.
72923
+ *
72924
+ * @param routeKey The route's stable key.
72925
+ * @returns void
72926
+ */
72927
+ onAttemptEnd(routeKey) {
72928
+ const inFlight = this.inFlight.get(routeKey) ?? 0;
72929
+ this.inFlight.set(routeKey, Math.max(0, inFlight - 1));
72930
+ }
72466
72931
  /**
72467
72932
  * Record a success, closing the breaker.
72468
72933
  *
@@ -72471,15 +72936,68 @@ class CircuitBreakerRegistry {
72471
72936
  * working call answers it; requiring several would keep a recovered provider
72472
72937
  * excluded while the chain paid for slower legs.
72473
72938
  *
72939
+ * The one exception is a breaker opened by latency: an answer from an
72940
+ * attempt that started before it opened is the slow evidence that opened it,
72941
+ * not evidence of recovery, and is ignored.
72942
+ *
72474
72943
  * @param routeKey The route's stable key.
72944
+ * @param startedAtMs When the answering attempt started, on the registry's clock.
72475
72945
  * @returns void
72476
72946
  */
72477
- onSuccess(routeKey) {
72478
- this.records.set(routeKey, {
72479
- consecutiveFailures: 0,
72480
- openedAtMs: null,
72481
- probesInFlight: 0,
72482
- });
72947
+ onSuccess(routeKey, startedAtMs) {
72948
+ const record = this.records.get(routeKey);
72949
+ if (record !== undefined &&
72950
+ record.openedByLatency &&
72951
+ record.openedAtMs !== null &&
72952
+ startedAtMs !== undefined &&
72953
+ startedAtMs < record.openedAtMs) {
72954
+ return;
72955
+ }
72956
+ this.records.set(routeKey, freshRecord());
72957
+ }
72958
+ /**
72959
+ * Record how long an attempt that reached the provider took.
72960
+ *
72961
+ * Durations fill fixed-size windows; each full window is judged against the
72962
+ * latency class's objective at the configured quantile, and the breaker opens
72963
+ * when enough of the recent windows were over it. Does nothing unless the
72964
+ * latency trip is armed.
72965
+ *
72966
+ * @param routeKey The route's stable key.
72967
+ * @param durationMs The attempt's duration.
72968
+ * @param latencyClass The alias's latency class, which selects the objective.
72969
+ * @returns void
72970
+ */
72971
+ onLatencySample(routeKey, durationMs, latencyClass) {
72972
+ const trip = this.config.latency_trip;
72973
+ if (trip === undefined || !trip.enabled || latencyClass === undefined) {
72974
+ return;
72975
+ }
72976
+ if (!Number.isFinite(durationMs) || durationMs < 0) {
72977
+ return;
72978
+ }
72979
+ let state = this.latency.get(routeKey);
72980
+ if (state === undefined) {
72981
+ state = { samples: [], verdicts: [] };
72982
+ this.latency.set(routeKey, state);
72983
+ }
72984
+ state.samples.push(durationMs);
72985
+ if (state.samples.length < trip.window_size) {
72986
+ return;
72987
+ }
72988
+ const sorted = [...state.samples].sort((a, b) => a - b);
72989
+ state.samples = [];
72990
+ state.verdicts.push(nearestRank(sorted, trip.quantile) > trip.slo_ms[latencyClass]);
72991
+ while (state.verdicts.length > trip.of_windows) {
72992
+ state.verdicts.shift();
72993
+ }
72994
+ const over = state.verdicts.filter(Boolean).length;
72995
+ if (over >= trip.trip_windows && this.stateOf(routeKey) === "closed") {
72996
+ const record = this.recordFor(routeKey);
72997
+ this.open(routeKey, record);
72998
+ record.openedByLatency = true;
72999
+ state.verdicts = [];
73000
+ }
72483
73001
  }
72484
73002
  /**
72485
73003
  * Record a failure, opening the breaker once the threshold is reached.
@@ -72488,17 +73006,36 @@ class CircuitBreakerRegistry {
72488
73006
  * re-accumulate the threshold: the probe was the test, and it failed.
72489
73007
  *
72490
73008
  * @param routeKey The route's stable key.
73009
+ * @param kind Whether the failure was a capacity signal or a hard failure.
73010
+ * Defaults to `hard`, so a caller that cannot tell gets the longer, safer
73011
+ * cooldown.
72491
73012
  * @returns void
72492
73013
  */
72493
- onFailure(routeKey) {
73014
+ onFailure(routeKey, kind = "hard") {
72494
73015
  const wasHalfOpen = this.stateOf(routeKey) === "half-open";
72495
73016
  const record = this.recordFor(routeKey);
72496
73017
  record.probesInFlight = 0;
72497
73018
  record.consecutiveFailures += 1;
73019
+ if (kind === "hard") {
73020
+ record.runHasHardFailure = true;
73021
+ }
72498
73022
  if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
72499
- record.openedAtMs = this.now();
73023
+ this.open(routeKey, record);
73024
+ record.openedByLatency = false;
72500
73025
  }
72501
73026
  }
73027
+ /**
73028
+ * Open a route, capturing the concurrency its probe budget scales with.
73029
+ *
73030
+ * @param routeKey The route's stable key.
73031
+ * @param record Its record.
73032
+ * @returns void
73033
+ */
73034
+ open(routeKey, record) {
73035
+ record.openedAtMs = this.now();
73036
+ record.concurrencyAtOpen = Math.max(record.concurrencyAtOpen, this.peakInFlight.get(routeKey) ?? 0);
73037
+ this.peakInFlight.set(routeKey, this.inFlight.get(routeKey) ?? 0);
73038
+ }
72502
73039
  /**
72503
73040
  * Inspect a route's breaker.
72504
73041
  *
@@ -72506,17 +73043,21 @@ class CircuitBreakerRegistry {
72506
73043
  * @returns A snapshot.
72507
73044
  */
72508
73045
  snapshot(routeKey) {
72509
- const record = this.records.get(routeKey) ?? {
72510
- consecutiveFailures: 0,
72511
- openedAtMs: null,
72512
- probesInFlight: 0,
72513
- };
73046
+ const record = this.records.get(routeKey) ?? freshRecord();
72514
73047
  return {
72515
73048
  routeKey,
72516
73049
  state: this.stateOf(routeKey),
72517
73050
  consecutiveFailures: record.consecutiveFailures,
72518
73051
  openedAtMs: record.openedAtMs,
72519
73052
  probesInFlight: record.probesInFlight,
73053
+ failureKind: record.consecutiveFailures === 0
73054
+ ? null
73055
+ : record.runHasHardFailure
73056
+ ? "hard"
73057
+ : "capacity",
73058
+ cooldownMs: this.cooldownFor(record),
73059
+ openedByLatency: record.openedByLatency,
73060
+ probeBudget: this.probeBudgetFor(record),
72520
73061
  };
72521
73062
  }
72522
73063
  /**
@@ -72536,6 +73077,9 @@ class CircuitBreakerRegistry {
72536
73077
  */
72537
73078
  reset() {
72538
73079
  this.records.clear();
73080
+ this.latency.clear();
73081
+ this.inFlight.clear();
73082
+ this.peakInFlight.clear();
72539
73083
  }
72540
73084
  /**
72541
73085
  * @param routeKey The route's stable key.
@@ -72544,7 +73088,7 @@ class CircuitBreakerRegistry {
72544
73088
  recordFor(routeKey) {
72545
73089
  let record = this.records.get(routeKey);
72546
73090
  if (record === undefined) {
72547
- record = { consecutiveFailures: 0, openedAtMs: null, probesInFlight: 0 };
73091
+ record = freshRecord();
72548
73092
  this.records.set(routeKey, record);
72549
73093
  }
72550
73094
  return record;
@@ -72820,7 +73364,17 @@ var providers$1 = {
72820
73364
  max_concurrent: 12,
72821
73365
  acquire_timeout_ms: 15000,
72822
73366
  source: null,
72823
- note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published."
73367
+ note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published.",
73368
+ models: {
73369
+ "claude-haiku-4-5": {
73370
+ basis: "conservative-default",
73371
+ requests_per_minute: 300,
73372
+ max_concurrent: 48,
73373
+ acquire_timeout_ms: 10000,
73374
+ source: null,
73375
+ note: "Raised 2026-09-28 for the closed-incumbent role this model holds on llm.fast, llm.decide and llm.extract. Measured that day: when DeepInfra was capacity-limited ('Model busy, retry later', timeouts at the 30000 ms hot-path budget), both DeepInfra legs of llm.fast opened their breakers and ALL of the alias's traffic fell to this model, where the 12-permit guard could not absorb it. In the 12 minutes to 15:33Z the incumbent leg recorded 18 timeouts, 15 breaker-open and 11 guard refusals ('did not admit the call within 15000 ms'), and the chain exhausted 20-45 times per 5 minutes. Sizing: in-flight demand is arrival rate times leg duration (Little's law), and the leg duration that matters is the worst case the chain permits, 30 s. 12 permits at 30 s serve 24 calls/min; 48 permits serve 96 calls/min at 30 s and up to the 300/min rate bound at haiku's normal single-digit-second latency, which covers the diverted llm.fast load plus normal llm.decide/llm.extract volume on the same guard. requests_per_minute is raised with it (the two bind in series) to 300, which is 30% of the 1,000 requests per minute per model that the scope_source lists for the lowest standard tier, so it remains below any standard tier's ceiling; no Anthropic 429 was observed in the window. acquire_timeout_ms is cut from 15000 to 10000 so an admitted call keeps at least 20 s of its 30 s leg budget: a call admitted after a 15 s wait had half its budget left, timed out at the provider, and that timeout was charged to this route's breaker, which then refused every call for the cooldown. STILL conservative-default: the organisation's tier has not been transcribed from the Claude Console (W3-07). Other Anthropic models keep the provider numbers."
73376
+ }
73377
+ }
72824
73378
  },
72825
73379
  openai: {
72826
73380
  basis: "conservative-default",
@@ -72929,18 +73483,42 @@ const SECONDS_PER_MINUTE = 60;
72929
73483
  const TOKENS_PER_REQUEST = 1;
72930
73484
  const config$1 = limitsConfig;
72931
73485
  /**
72932
- * Resolve the limits that apply to a provider.
73486
+ * Resolve the limits that apply to a provider, or to one of its models.
72933
73487
  *
72934
73488
  * An unregistered provider falls back to the conservative defaults rather than
72935
73489
  * to no limit at all. Treating "unknown" as "unlimited" would make every newly
72936
73490
  * onboarded provider the one most likely to be over-driven, which is exactly
72937
73491
  * backwards: a new provider is the one whose real ceiling is least understood.
72938
73492
  *
73493
+ * A model with an override on a per-model provider runs at the override's
73494
+ * numbers and provenance; every other model runs at the provider's. The
73495
+ * override is ignored for a provider scoped as a whole, because that provider
73496
+ * has one guard and a per-model number cannot be enforced on it.
73497
+ *
72939
73498
  * @param provider The provider key.
73499
+ * @param modelId The model, when the caller knows which one it addresses.
72940
73500
  * @returns Its limits.
72941
73501
  */
72942
- function limitsFor(provider) {
72943
- return config$1.providers[provider] ?? config$1.defaults;
73502
+ function limitsFor(provider, modelId) {
73503
+ const limits = config$1.providers[provider] ?? config$1.defaults;
73504
+ if (limits.scope !== "model" || modelId === undefined || modelId.length === 0) {
73505
+ return limits;
73506
+ }
73507
+ const override = limits.models?.[modelId];
73508
+ if (override === undefined) {
73509
+ return limits;
73510
+ }
73511
+ const { models: _siblings, ...providerLimits } = limits;
73512
+ return {
73513
+ ...providerLimits,
73514
+ basis: override.basis,
73515
+ requests_per_minute: override.requests_per_minute,
73516
+ requests_per_minute_basis: override.requests_per_minute_basis,
73517
+ max_concurrent: override.max_concurrent,
73518
+ acquire_timeout_ms: override.acquire_timeout_ms,
73519
+ source: override.source ?? null,
73520
+ note: override.note,
73521
+ };
72944
73522
  }
72945
73523
  /** Every provider with a recorded limit, plus whether it is published or a default. */
72946
73524
  function limitsInventory() {
@@ -73136,7 +73714,7 @@ const guardIdentities = new Map();
73136
73714
  function rateLimiterFor(identity) {
73137
73715
  let limiter = rateLimiters.get(identity.key);
73138
73716
  if (limiter === undefined) {
73139
- const limits = limitsFor(identity.provider);
73717
+ const limits = limitsFor(identity.provider, identity.modelId);
73140
73718
  limiter = new TokenBucketRateLimiter({
73141
73719
  maxTokens: limits.requests_per_minute,
73142
73720
  refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
@@ -73157,7 +73735,7 @@ function rateLimiterFor(identity) {
73157
73735
  function concurrencyGateFor(identity) {
73158
73736
  let gate = concurrencyGates.get(identity.key);
73159
73737
  if (gate === undefined) {
73160
- gate = new ConcurrencyGate(identity, limitsFor(identity.provider).max_concurrent);
73738
+ gate = new ConcurrencyGate(identity, limitsFor(identity.provider, identity.modelId).max_concurrent);
73161
73739
  concurrencyGates.set(identity.key, gate);
73162
73740
  guardIdentities.set(identity.key, identity);
73163
73741
  }
@@ -73187,8 +73765,8 @@ function concurrencyGateFor(identity) {
73187
73765
  * or the caller stopped waiting first.
73188
73766
  */
73189
73767
  async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
73190
- const limits = limitsFor(provider);
73191
73768
  const identity = guardIdentity(provider, scope.modelId);
73769
+ const limits = limitsFor(provider, identity.modelId);
73192
73770
  const waitBudgetMs = maxWaitMs === undefined
73193
73771
  ? limits.acquire_timeout_ms
73194
73772
  : Math.min(maxWaitMs, limits.acquire_timeout_ms);
@@ -73211,6 +73789,35 @@ async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
73211
73789
  release();
73212
73790
  }
73213
73791
  }
73792
+ /**
73793
+ * Whether a provider guard has room for a DUPLICATE attempt above a reserve.
73794
+ *
73795
+ * A duplicate is a hedge: a second request for an answer another attempt is
73796
+ * already fetching. It is only worth sending with capacity no first attempt
73797
+ * needs, so it is admitted only when nobody is queued behind either bound and
73798
+ * both the free concurrency (after the duplicate) and the rate tokens stay at
73799
+ * or above the reserved fraction of the guard's ceilings. A provider that is
73800
+ * already busy therefore never sees duplicates, which is when they would do
73801
+ * the most harm.
73802
+ *
73803
+ * @param provider The provider key.
73804
+ * @param modelId The model the duplicate would address.
73805
+ * @param reserveFraction Fraction of each ceiling kept free, in [0, 1).
73806
+ * @returns Whether the duplicate may start.
73807
+ */
73808
+ function hasDuplicateHeadroom(provider, modelId, reserveFraction) {
73809
+ const identity = guardIdentity(provider, modelId);
73810
+ const limits = limitsFor(provider, identity.modelId);
73811
+ const limiter = rateLimiterFor(identity);
73812
+ const gate = concurrencyGateFor(identity);
73813
+ if (limiter.getQueueLength() > 0 || gate.queueLength() > 0) {
73814
+ return false;
73815
+ }
73816
+ const freeAfter = limits.max_concurrent - gate.inFlightCount() - 1;
73817
+ const tokensAfter = limiter.getAvailableTokens() - TOKENS_PER_REQUEST;
73818
+ return (freeAfter >= Math.ceil(limits.max_concurrent * reserveFraction) &&
73819
+ tokensAfter >= Math.ceil(limits.requests_per_minute * reserveFraction));
73820
+ }
73214
73821
  /**
73215
73822
  * Inspect the guards currently in use.
73216
73823
  *
@@ -73220,7 +73827,7 @@ function guardSnapshots() {
73220
73827
  return [...guardIdentities.values()]
73221
73828
  .sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))
73222
73829
  .map((identity) => {
73223
- const limits = limitsFor(identity.provider);
73830
+ const limits = limitsFor(identity.provider, identity.modelId);
73224
73831
  const limiter = rateLimiters.get(identity.key);
73225
73832
  const gate = concurrencyGates.get(identity.key);
73226
73833
  return {
@@ -73353,6 +73960,703 @@ function parseStructuredContent(content, responseFormat, usage) {
73353
73960
  }
73354
73961
  }
73355
73962
 
73963
+ /**
73964
+ * One attempt against one leg: dispatch under a hard timeout, and the
73965
+ * classification of how it ended.
73966
+ *
73967
+ * Split from the chain walker so the same-model hedge runner and the walker
73968
+ * share exactly one definition of what a timeout, a capacity refusal, a
73969
+ * cancellation and a superseded attempt are. Two copies of that
73970
+ * classification would drift, and the breaker would then read the same event
73971
+ * differently depending on which code path produced it.
73972
+ *
73973
+ * @module llm/leg-attempt
73974
+ */
73975
+ /** Raised internally when an attempt exceeds its hard budget. */
73976
+ class LegTimeoutError extends Error {
73977
+ /**
73978
+ * @param routeKey The leg that timed out.
73979
+ * @param budgetMs Its budget in milliseconds.
73980
+ */
73981
+ constructor(routeKey, budgetMs) {
73982
+ super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
73983
+ this.name = "LegTimeoutError";
73984
+ }
73985
+ }
73986
+ /**
73987
+ * Raised internally when an attempt ran past its MEASURED timeout and was
73988
+ * replaced by another attempt on the same model.
73989
+ *
73990
+ * Not a verdict on the provider: the measured timeout is the chain's own
73991
+ * impatience, applied only because a same-model alternative could take over,
73992
+ * so the breaker learns nothing from it.
73993
+ */
73994
+ class AttemptSupersededError extends Error {
73995
+ /**
73996
+ * @param routeKey The attempt's leg.
73997
+ * @param afterMs How long it ran before it was replaced.
73998
+ */
73999
+ constructor(routeKey, afterMs) {
74000
+ super(`route ${routeKey} exceeded its measured ${afterMs} ms attempt timeout and was ` +
74001
+ "superseded by a same-model attempt");
74002
+ this.name = "AttemptSupersededError";
74003
+ }
74004
+ }
74005
+ /**
74006
+ * Raised internally on an attempt that was still running when another
74007
+ * attempt on the same model answered first. Not a verdict on the provider.
74008
+ */
74009
+ class HedgeLoserError extends Error {
74010
+ /**
74011
+ * @param routeKey The losing attempt's leg.
74012
+ */
74013
+ constructor(routeKey) {
74014
+ super(`route ${routeKey} was cancelled: a same-model attempt answered first`);
74015
+ this.name = "HedgeLoserError";
74016
+ }
74017
+ }
74018
+ /**
74019
+ * Start one attempt under a hard timeout, honouring the caller's own cancellation.
74020
+ *
74021
+ * The timer is always cleared and the abort listener always removed, including
74022
+ * on the success path. A long-lived process that leaked one timer per LLM call
74023
+ * would accumulate them at exactly the rate it does useful work.
74024
+ *
74025
+ * @param leg The leg to run.
74026
+ * @param params Normalised parameters for this leg.
74027
+ * @param request The call's request fields and cancellation.
74028
+ * @param budgetMs The attempt's hard budget.
74029
+ * @returns A handle on the attempt.
74030
+ */
74031
+ function startAttempt(leg, params, request, budgetMs) {
74032
+ const controller = new AbortController();
74033
+ let chainReason;
74034
+ let hardTimeout = false;
74035
+ const timer = setTimeout(() => {
74036
+ hardTimeout = true;
74037
+ controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
74038
+ }, budgetMs);
74039
+ const forwardAbort = () => {
74040
+ controller.abort(request.callerSignal?.reason);
74041
+ };
74042
+ if (request.callerSignal !== undefined) {
74043
+ if (request.callerSignal.aborted) {
74044
+ forwardAbort();
74045
+ }
74046
+ else {
74047
+ request.callerSignal.addEventListener("abort", forwardAbort, { once: true });
74048
+ }
74049
+ }
74050
+ const run = async () => {
74051
+ try {
74052
+ // The guards wrap the transport rather than the whole attempt, so the
74053
+ // hard timeout above still bounds the total wait: a caller queued behind
74054
+ // the rate limiter is spending its budget just as surely as one waiting
74055
+ // on the provider, and only one clock should govern both. The attempt's
74056
+ // own signal is handed to the guard as well, so an attempt whose budget
74057
+ // or caller is gone leaves the queue at once instead of holding its place.
74058
+ const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
74059
+ route: leg.route,
74060
+ content: request.content,
74061
+ responseFormat: request.responseFormat,
74062
+ params,
74063
+ developerPrompt: request.developerPrompt,
74064
+ context: request.context,
74065
+ signal: controller.signal,
74066
+ correlationId: request.correlationId,
74067
+ }), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
74068
+ assertToolChoiceHonoured(leg.route, params, response);
74069
+ return response;
74070
+ }
74071
+ finally {
74072
+ clearTimeout(timer);
74073
+ request.callerSignal?.removeEventListener("abort", forwardAbort);
74074
+ }
74075
+ };
74076
+ return {
74077
+ promise: run(),
74078
+ abort: (reason) => {
74079
+ if (chainReason === undefined && !controller.signal.aborted) {
74080
+ chainReason = reason;
74081
+ controller.abort(reason);
74082
+ }
74083
+ },
74084
+ abortReason: () => chainReason,
74085
+ timedOut: () => hardTimeout,
74086
+ };
74087
+ }
74088
+ /**
74089
+ * HTTP statuses a provider (or the gateway relaying it) uses to say it is full
74090
+ * rather than that the request or the route is wrong: request timeout, too
74091
+ * early, too many requests, service unavailable, and Anthropic's overloaded.
74092
+ */
74093
+ const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
74094
+ /**
74095
+ * Wording providers use for a capacity refusal when the status is lost on the
74096
+ * way (a relayed body, a client library's own error). DeepInfra's is
74097
+ * "Model busy, retry later"; Anthropic's is "Overloaded".
74098
+ */
74099
+ const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
74100
+ /**
74101
+ * Whether a failure is the provider saying it is full rather than broken.
74102
+ *
74103
+ * Read by shape rather than by class, because the same signal reaches the
74104
+ * chain from more than one transport and not every transport's error class is
74105
+ * importable here.
74106
+ *
74107
+ * @param error The thrown value.
74108
+ * @param reason Its message.
74109
+ * @returns Whether it is a capacity signal.
74110
+ */
74111
+ function isCapacitySignal(error, reason) {
74112
+ if (typeof error === "object" && error !== null) {
74113
+ const status = error.status;
74114
+ if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
74115
+ return true;
74116
+ }
74117
+ }
74118
+ return CAPACITY_WORDING.test(reason);
74119
+ }
74120
+ /**
74121
+ * Classify why a leg failed.
74122
+ *
74123
+ * The distinction matters to the breaker: a timeout and a 5xx are evidence the
74124
+ * provider is unhealthy, while the caller cancelling is not. Counting a
74125
+ * cancellation as a provider failure would let a burst of user-cancelled
74126
+ * requests open the breaker on a perfectly healthy route.
74127
+ *
74128
+ * Among failures that do count, a capacity signal (the provider said it is
74129
+ * busy, or the leg ran out its budget waiting on it) is told apart from a hard
74130
+ * failure so the breaker can re-admit a busy route sooner than a broken one. A
74131
+ * timeout is read as capacity: on a reachable provider it is what a full queue
74132
+ * looks like from outside, and a provider that is actually down still costs no
74133
+ * more than one probe per capacity cooldown.
74134
+ *
74135
+ * An attempt the chain itself cancelled — replaced after its measured
74136
+ * timeout, or beaten by a same-model attempt — is not a verdict on the
74137
+ * provider either, and is classified by the chain's reason rather than by
74138
+ * whatever the transport happened to throw on the way out.
74139
+ *
74140
+ * @param error The thrown value.
74141
+ * @param callerSignal The caller's cancellation signal, if any.
74142
+ * @returns The outcome and whether it counts against route health.
74143
+ */
74144
+ function classify(error, callerSignal) {
74145
+ if (callerSignal !== undefined && callerSignal.aborted) {
74146
+ return {
74147
+ outcome: "skipped",
74148
+ reason: "caller cancelled",
74149
+ countsAgainstHealth: false,
74150
+ failureKind: "hard",
74151
+ };
74152
+ }
74153
+ if (error instanceof AttemptSupersededError) {
74154
+ return {
74155
+ outcome: "timeout",
74156
+ reason: error.message,
74157
+ countsAgainstHealth: false,
74158
+ failureKind: "capacity",
74159
+ };
74160
+ }
74161
+ if (error instanceof HedgeLoserError) {
74162
+ return {
74163
+ outcome: "skipped",
74164
+ reason: error.message,
74165
+ countsAgainstHealth: false,
74166
+ failureKind: "capacity",
74167
+ };
74168
+ }
74169
+ if (error instanceof LegTimeoutError) {
74170
+ return {
74171
+ outcome: "timeout",
74172
+ reason: error.message,
74173
+ countsAgainstHealth: true,
74174
+ failureKind: "capacity",
74175
+ };
74176
+ }
74177
+ if (error instanceof UnsupportedCapabilityError) {
74178
+ return {
74179
+ outcome: "skipped",
74180
+ reason: error.message,
74181
+ countsAgainstHealth: false,
74182
+ failureKind: "hard",
74183
+ };
74184
+ }
74185
+ if (error instanceof ToolChoiceIgnoredError) {
74186
+ // The route answered; it broke a declared guarantee rather than failing to
74187
+ // be available, so its breaker is not charged for it.
74188
+ return {
74189
+ outcome: "error",
74190
+ reason: error.message,
74191
+ countsAgainstHealth: false,
74192
+ failureKind: "hard",
74193
+ };
74194
+ }
74195
+ // Self-inflicted pacing, not provider ill-health. Counting it would let the
74196
+ // client's own throttling open a breaker on a perfectly healthy provider and
74197
+ // permanently reroute traffic nobody chose to reroute.
74198
+ if (error instanceof RateGuardTimeoutError) {
74199
+ return {
74200
+ outcome: "skipped",
74201
+ reason: error.message,
74202
+ countsAgainstHealth: false,
74203
+ failureKind: "hard",
74204
+ };
74205
+ }
74206
+ const reason = error instanceof Error ? error.message : String(error);
74207
+ if (/abort/i.test(reason)) {
74208
+ return {
74209
+ outcome: "timeout",
74210
+ reason: `aborted: ${reason}`,
74211
+ countsAgainstHealth: true,
74212
+ failureKind: "capacity",
74213
+ };
74214
+ }
74215
+ if (error instanceof LlmResponseFormatError) {
74216
+ // The provider answered, badly. That is a route defect, not a full queue,
74217
+ // whatever words the unparseable content happens to contain.
74218
+ return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
74219
+ }
74220
+ return {
74221
+ outcome: "error",
74222
+ reason,
74223
+ countsAgainstHealth: true,
74224
+ failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
74225
+ };
74226
+ }
74227
+ /**
74228
+ * Whether the caller has stopped waiting.
74229
+ *
74230
+ * Read through a function rather than inline, because `AbortSignal.aborted` is
74231
+ * a live getter: it can flip to true while a leg is in flight, but a compiler
74232
+ * that narrowed it at the top of the loop would prove the later check
74233
+ * unreachable and invite its removal. The check is not redundant — it is the
74234
+ * only thing that stops the chain spending money on an answer nobody will read.
74235
+ *
74236
+ * @param signal The caller's signal, if any.
74237
+ * @returns Whether the call has been cancelled.
74238
+ */
74239
+ function isAborted(signal) {
74240
+ return signal !== undefined && signal.aborted;
74241
+ }
74242
+ /**
74243
+ * The usage a failed leg was billed for, when the leg reached an answer.
74244
+ *
74245
+ * A leg that failed after the provider answered — content that does not parse,
74246
+ * or prose where a tool call was mandatory — was still charged. A leg that
74247
+ * never answered (timeout, outage, skip) carries no usage, and none is invented.
74248
+ *
74249
+ * @param error The thrown value.
74250
+ * @returns The billed usage, or undefined when the leg never produced an answer.
74251
+ */
74252
+ function billedUsageOf(error) {
74253
+ if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
74254
+ return error.usage;
74255
+ }
74256
+ return undefined;
74257
+ }
74258
+
74259
+ /**
74260
+ * Same-model attempts for one leg: hedging, measured attempt timeouts, and
74261
+ * reserving deadline for a same-model alternative.
74262
+ *
74263
+ * A leg is a model. Before the chain gives up on it and reaches a DIFFERENT
74264
+ * model — which changes the answer's quality, not just its latency — it is
74265
+ * worth spending the leg's budget on every way of getting that same model to
74266
+ * answer: the same model at another provider (an "equivalent"), or a second
74267
+ * request to the same provider when that provider has capacity to spare (a
74268
+ * "duplicate"). This module runs those attempts as one group.
74269
+ *
74270
+ * Three mechanics, all bounded by the leg's budget and none of them selecting
74271
+ * a different model:
74272
+ *
74273
+ * - **Hedging.** Once an attempt has run past the model's healthy p90 (from
74274
+ * the latency tracker), the next same-model attempt starts beside it. The
74275
+ * first answer wins and every other attempt is cancelled through its own
74276
+ * abort signal. A hedge that loses, or an attempt cancelled because another
74277
+ * won, says nothing about the provider and never touches its breaker.
74278
+ *
74279
+ * - **Measured attempt timeout.** With latency evidence, an attempt that has
74280
+ * run past `k × p99` of healthy latency is replaced by a same-model
74281
+ * alternative instead of holding the budget to its end. It is replaced only
74282
+ * when an alternative exists: with none, cutting it short would only move
74283
+ * the call to a different model sooner, and it runs to the leg budget as
74284
+ * before.
74285
+ *
74286
+ * - **Deadline reservation.** When a same-model equivalent exists, no single
74287
+ * attempt holds more than `max_attempt_share` of the remaining budget before
74288
+ * the equivalent starts beside it, so a slow first attempt cannot consume the
74289
+ * whole deadline while the equivalent that could have answered never runs.
74290
+ *
74291
+ * With no latency evidence and no equivalent, none of the three can act and
74292
+ * the group is exactly one attempt with the leg's full budget: the serial
74293
+ * chain's behaviour, which a fresh process therefore starts from.
74294
+ *
74295
+ * @module llm/hedge
74296
+ */
74297
+ /**
74298
+ * Read the policy from the table's defaults.
74299
+ *
74300
+ * @param defaults The `hedging` defaults.
74301
+ * @returns The policy.
74302
+ */
74303
+ function sameModelPolicyFrom(defaults) {
74304
+ return {
74305
+ maxExtraAttempts: defaults.max_same_model_hedges,
74306
+ hedgeQuantile: defaults.hedge_quantile,
74307
+ timeoutQuantile: defaults.timeout_quantile,
74308
+ kTimeout: defaults.k_timeout,
74309
+ attemptTimeoutFloorMs: defaults.attempt_timeout_floor_ms,
74310
+ maxAttemptShare: defaults.max_attempt_share,
74311
+ duplicateReserve: defaults.duplicate_headroom_reserve,
74312
+ };
74313
+ }
74314
+ /**
74315
+ * The parameters of a prepared leg, when it can serve.
74316
+ *
74317
+ * @param leg The leg.
74318
+ * @returns Its parameters, or undefined when it cannot serve this request.
74319
+ */
74320
+ function paramsOf(leg) {
74321
+ return leg.params instanceof UnsupportedCapabilityError ? undefined : leg.params;
74322
+ }
74323
+ /**
74324
+ * Run every same-model attempt for one leg until one answers or the budget,
74325
+ * the alternatives, or the caller run out.
74326
+ *
74327
+ * The caller has already found the leg servable (breaker allows it, budget
74328
+ * positive). The returned promise settles only once every attempt it started
74329
+ * has been recorded, so the call's attempt record is complete when it returns.
74330
+ *
74331
+ * @param leg The leg, with its equivalents.
74332
+ * @param groupBudgetMs The leg's budget: its route budget cut to the deadline.
74333
+ * @param budgetIsDeadline Whether that budget was cut by the caller's deadline.
74334
+ * @param ctx The chain around the group.
74335
+ * @returns How the group ended.
74336
+ */
74337
+ function runSameModelGroup(leg, groupBudgetMs, budgetIsDeadline, ctx) {
74338
+ const { breakers, now, policy, tracker, promptTokens, request } = ctx;
74339
+ const endsAt = now() + groupBudgetMs;
74340
+ const equivalents = (leg.equivalents ?? []).filter((equivalent) => paramsOf(equivalent) !== undefined);
74341
+ const live = new Set();
74342
+ const billed = [];
74343
+ const timeoutCharged = new Set();
74344
+ let nextEquivalent = 0;
74345
+ let extraLaunched = 0;
74346
+ let settled = false;
74347
+ let finished = false;
74348
+ let deadlineBound = false;
74349
+ let hedgeTimer;
74350
+ let answer;
74351
+ return new Promise((resolve) => {
74352
+ /**
74353
+ * A healthy-latency quantile for a leg's model, when there is evidence.
74354
+ *
74355
+ * @param route The leg's route.
74356
+ * @param q The quantile.
74357
+ * @returns Milliseconds, or null.
74358
+ */
74359
+ const quantileOf = (route, q) => tracker === undefined ? null : tracker.quantile(route.providerName, route.modelId, promptTokens, q);
74360
+ /**
74361
+ * The next same-model attempt that could start now: an equivalent first,
74362
+ * then a duplicate, which needs latency evidence and spare provider capacity.
74363
+ *
74364
+ * @returns The candidate, or undefined.
74365
+ */
74366
+ const peek = () => {
74367
+ if (policy === undefined || extraLaunched >= policy.maxExtraAttempts) {
74368
+ return undefined;
74369
+ }
74370
+ for (let index = nextEquivalent; index < equivalents.length; index += 1) {
74371
+ const equivalent = equivalents[index];
74372
+ if (breakers.allows(equivalent.route.routeKey)) {
74373
+ return { leg: equivalent, kind: "equivalent", index };
74374
+ }
74375
+ }
74376
+ if (quantileOf(leg.route, policy.hedgeQuantile) !== null &&
74377
+ breakers.allows(leg.route.routeKey) &&
74378
+ ctx.admitDuplicate(leg.route, policy.duplicateReserve)) {
74379
+ return { leg, kind: "duplicate", index: -1 };
74380
+ }
74381
+ return undefined;
74382
+ };
74383
+ /** Resolve once nothing is left in flight. */
74384
+ const finishIfIdle = () => {
74385
+ if (finished || live.size > 0) {
74386
+ return;
74387
+ }
74388
+ finished = true;
74389
+ if (hedgeTimer !== undefined) {
74390
+ clearTimeout(hedgeTimer);
74391
+ }
74392
+ resolve({ answer, billed, deadlineBound });
74393
+ };
74394
+ /**
74395
+ * Arm the hedge for the most recent attempt: at the model's healthy p90,
74396
+ * or — when an equivalent is waiting — no later than the reserved share of
74397
+ * the remaining budget.
74398
+ *
74399
+ * @param latest The attempt just started.
74400
+ */
74401
+ const scheduleHedge = (latest) => {
74402
+ if (hedgeTimer !== undefined) {
74403
+ clearTimeout(hedgeTimer);
74404
+ hedgeTimer = undefined;
74405
+ }
74406
+ const candidate = peek();
74407
+ if (policy === undefined || candidate === undefined) {
74408
+ return;
74409
+ }
74410
+ const measured = quantileOf(latest.leg.route, policy.hedgeQuantile);
74411
+ const reserved = candidate.kind === "equivalent"
74412
+ ? policy.maxAttemptShare * (endsAt - now())
74413
+ : Number.POSITIVE_INFINITY;
74414
+ const delay = Math.min(measured ?? Number.POSITIVE_INFINITY, reserved);
74415
+ if (!Number.isFinite(delay)) {
74416
+ return;
74417
+ }
74418
+ hedgeTimer = setTimeout(() => {
74419
+ hedgeTimer = undefined;
74420
+ if (settled) {
74421
+ return;
74422
+ }
74423
+ const next = peek();
74424
+ if (next !== undefined) {
74425
+ launch(next.leg, next, true);
74426
+ }
74427
+ }, Math.max(0, delay));
74428
+ };
74429
+ /**
74430
+ * Start one attempt.
74431
+ *
74432
+ * @param target The leg to address.
74433
+ * @param candidate The candidate it came from, for a hedge.
74434
+ * @param hedged Whether this is a hedge rather than the leg's first attempt.
74435
+ * @returns Whether it started.
74436
+ */
74437
+ const launch = (target, candidate, hedged) => {
74438
+ const params = paramsOf(target);
74439
+ const budgetMs = endsAt - now();
74440
+ if (params === undefined || budgetMs <= 0 || isAborted(request.callerSignal)) {
74441
+ return false;
74442
+ }
74443
+ if (candidate !== undefined) {
74444
+ extraLaunched += 1;
74445
+ if (candidate.kind === "equivalent") {
74446
+ nextEquivalent = candidate.index + 1;
74447
+ }
74448
+ }
74449
+ const key = target.route.routeKey;
74450
+ const attempt = {
74451
+ leg: target,
74452
+ handle: startAttempt(target, params, request, budgetMs),
74453
+ startedAt: now(),
74454
+ budgetMs,
74455
+ holdsProbe: breakers.onAttemptStart(key),
74456
+ hedged,
74457
+ attemptIndex: ctx.nextAttemptIndex(),
74458
+ closed: false,
74459
+ };
74460
+ live.add(attempt);
74461
+ if (policy !== undefined) {
74462
+ const tail = quantileOf(target.route, policy.timeoutQuantile);
74463
+ if (tail !== null) {
74464
+ const measuredMs = Math.max(policy.attemptTimeoutFloorMs, policy.kTimeout * tail);
74465
+ if (measuredMs < budgetMs) {
74466
+ attempt.softTimer = setTimeout(() => supersede(attempt, measuredMs), measuredMs);
74467
+ }
74468
+ }
74469
+ }
74470
+ scheduleHedge(attempt);
74471
+ attempt.handle.promise.then((response) => onAnswer(attempt, response), (error) => onFailure(attempt, error));
74472
+ return true;
74473
+ };
74474
+ /**
74475
+ * An attempt ran past its measured timeout: replace it if a same-model
74476
+ * alternative can take over, otherwise leave it running to the leg budget.
74477
+ *
74478
+ * @param attempt The slow attempt.
74479
+ * @param afterMs Its measured timeout.
74480
+ */
74481
+ const supersede = (attempt, afterMs) => {
74482
+ attempt.softTimer = undefined;
74483
+ if (settled || !live.has(attempt)) {
74484
+ return;
74485
+ }
74486
+ const othersInFlight = live.size > 1;
74487
+ const candidate = othersInFlight ? undefined : peek();
74488
+ if (!othersInFlight && candidate === undefined) {
74489
+ return;
74490
+ }
74491
+ // Recorded now rather than when its rejection arrives, so a replacement
74492
+ // that answers first cannot find it still live and misfile it as a loser.
74493
+ const superseded = new AttemptSupersededError(attempt.leg.route.routeKey, afterMs);
74494
+ close(attempt, superseded);
74495
+ if (candidate !== undefined) {
74496
+ launch(candidate.leg, candidate, true);
74497
+ }
74498
+ finishIfIdle();
74499
+ };
74500
+ /**
74501
+ * Cancel an attempt the chain no longer wants and record it at once, with
74502
+ * no verdict on the provider.
74503
+ *
74504
+ * @param attempt The attempt.
74505
+ * @param reason Why it was cancelled.
74506
+ */
74507
+ const close = (attempt, reason) => {
74508
+ const fields = baseFields(attempt);
74509
+ attempt.handle.abort(reason);
74510
+ attempt.closed = true;
74511
+ retire(attempt);
74512
+ if (attempt.holdsProbe) {
74513
+ breakers.onAttemptAbandoned(attempt.leg.route.routeKey);
74514
+ }
74515
+ const failure = classify(reason, request.callerSignal);
74516
+ emit(attempt, { ...fields, outcome: failure.outcome, reason: failure.reason });
74517
+ };
74518
+ /**
74519
+ * Stop tracking an attempt and release its breaker bookkeeping.
74520
+ *
74521
+ * @param attempt The attempt.
74522
+ */
74523
+ const retire = (attempt) => {
74524
+ if (attempt.softTimer !== undefined) {
74525
+ clearTimeout(attempt.softTimer);
74526
+ attempt.softTimer = undefined;
74527
+ }
74528
+ live.delete(attempt);
74529
+ breakers.onAttemptEnd(attempt.leg.route.routeKey);
74530
+ };
74531
+ /**
74532
+ * Record one attempt through the chain.
74533
+ *
74534
+ * @param attempt The attempt.
74535
+ * @param fields What happened.
74536
+ * @param servedProvider The provider's own report of who served, if any.
74537
+ */
74538
+ const emit = (attempt, fields, servedProvider) => {
74539
+ ctx.record(attempt.leg.route, fields, { hedged: attempt.hedged, attemptIndex: attempt.attemptIndex }, servedProvider);
74540
+ };
74541
+ /**
74542
+ * Base fields shared by every record of an attempt.
74543
+ *
74544
+ * @param attempt The attempt.
74545
+ * @returns The identity and timing fields.
74546
+ */
74547
+ const baseFields = (attempt) => ({
74548
+ routeKey: attempt.leg.route.routeKey,
74549
+ role: attempt.leg.route.role,
74550
+ provider: attempt.leg.route.providerName,
74551
+ modelId: attempt.leg.route.modelId,
74552
+ durationMs: now() - attempt.startedAt,
74553
+ budgetMs: attempt.budgetMs,
74554
+ });
74555
+ const onAnswer = (attempt, response) => {
74556
+ if (attempt.closed) {
74557
+ return;
74558
+ }
74559
+ const { route } = attempt.leg;
74560
+ const fields = baseFields(attempt);
74561
+ retire(attempt);
74562
+ breakers.onSuccess(route.routeKey, attempt.startedAt);
74563
+ breakers.onLatencySample(route.routeKey, fields.durationMs, route.latencyClass);
74564
+ tracker?.record(route.providerName, route.modelId, promptTokens, fields.durationMs);
74565
+ billed.push(response.usage);
74566
+ if (settled) {
74567
+ // Answered in the same instant as the winner, before its cancellation
74568
+ // arrived. Its spend is real; its answer is not the one returned.
74569
+ emit(attempt, {
74570
+ ...fields,
74571
+ outcome: "skipped",
74572
+ reason: "answered after a same-model attempt had already won",
74573
+ servedModel: response.servedModel ?? null,
74574
+ usage: response.usage,
74575
+ }, response.servedProvider);
74576
+ finishIfIdle();
74577
+ return;
74578
+ }
74579
+ settled = true;
74580
+ answer = { response, route, hedged: attempt.hedged };
74581
+ emit(attempt, {
74582
+ ...fields,
74583
+ outcome: "ok",
74584
+ servedModel: response.servedModel ?? null,
74585
+ usage: response.usage,
74586
+ }, response.servedProvider);
74587
+ for (const loser of [...live]) {
74588
+ // Cancelled through its own signal and recorded now, so the answer is
74589
+ // returned without waiting for a loser still queued at the provider.
74590
+ close(loser, new HedgeLoserError(loser.leg.route.routeKey));
74591
+ }
74592
+ finishIfIdle();
74593
+ };
74594
+ const onFailure = (attempt, error) => {
74595
+ if (attempt.closed) {
74596
+ return;
74597
+ }
74598
+ const { route } = attempt.leg;
74599
+ const key = route.routeKey;
74600
+ const fields = baseFields(attempt);
74601
+ const hardTimeout = attempt.handle.timedOut();
74602
+ retire(attempt);
74603
+ const failure = classify(attempt.handle.abortReason() ?? error, request.callerSignal);
74604
+ let charged = failure.countsAgainstHealth;
74605
+ if (charged && hardTimeout) {
74606
+ // Every attempt of a leg shares the leg's end, so several can time out
74607
+ // together; the provider is charged once for the leg, as before hedging.
74608
+ charged = !timeoutCharged.has(key);
74609
+ timeoutCharged.add(key);
74610
+ if (budgetIsDeadline) {
74611
+ deadlineBound = true;
74612
+ }
74613
+ }
74614
+ if (charged) {
74615
+ breakers.onFailure(key, failure.failureKind);
74616
+ }
74617
+ else if (attempt.holdsProbe) {
74618
+ // No verdict on the route's health, but the probe slot this attempt
74619
+ // took must come back, or a half-open route admits no probe ever again.
74620
+ breakers.onAttemptAbandoned(key);
74621
+ }
74622
+ if (failure.countsAgainstHealth) {
74623
+ breakers.onLatencySample(key, fields.durationMs, route.latencyClass);
74624
+ }
74625
+ // A provider that answered — with unparseable content, or in prose where a
74626
+ // tool call was mandatory — still billed for the answer.
74627
+ const usage = billedUsageOf(error);
74628
+ if (usage !== undefined) {
74629
+ billed.push(usage);
74630
+ }
74631
+ const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
74632
+ emit(attempt, {
74633
+ ...fields,
74634
+ outcome: failure.outcome,
74635
+ reason: failure.reason,
74636
+ ...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
74637
+ ...(usage === undefined ? {} : { usage }),
74638
+ });
74639
+ if (settled || live.size > 0) {
74640
+ finishIfIdle();
74641
+ return;
74642
+ }
74643
+ if (!hardTimeout && !isAborted(request.callerSignal)) {
74644
+ // A failed attempt with time left moves to the same model at another
74645
+ // provider at once. A duplicate on the provider that just failed is not
74646
+ // started here: it would most likely fail the same way.
74647
+ const candidate = peek();
74648
+ if (candidate !== undefined && candidate.kind === "equivalent") {
74649
+ launch(candidate.leg, candidate, true);
74650
+ }
74651
+ }
74652
+ finishIfIdle();
74653
+ };
74654
+ if (!launch(leg, undefined, false)) {
74655
+ finishIfIdle();
74656
+ }
74657
+ });
74658
+ }
74659
+
73356
74660
  /**
73357
74661
  * Ordered execution of an alias's fallback chain (PD-3).
73358
74662
  *
@@ -73365,10 +74669,21 @@ function parseStructuredContent(content, responseFormat, usage) {
73365
74669
  * why the budget is enforced here — at the only place that knows both the
73366
74670
  * caller's deadline and how many legs are left to spend it on.
73367
74671
  *
74672
+ * Each leg is first run as a group of SAME-MODEL attempts (see `hedge.ts`):
74673
+ * hedged at the model's healthy p90, replaced after a measured timeout, and
74674
+ * reaching the same model at another provider before the leg is given up.
74675
+ * Only then does the walk move to the next leg — and a leg that serves a
74676
+ * different model than the configured one runs only when the caller's
74677
+ * cross-model policy allows it. With no latency evidence and no equivalent
74678
+ * configured, each group is a single attempt with the leg's budget, which is
74679
+ * the serial walk exactly.
74680
+ *
73368
74681
  * Nothing here ever substitutes a value for an outcome. When every leg is
73369
- * exhausted the caller gets a typed error naming each leg and why it failed,
73370
- * because a default returned in place of an answer is a wrong answer that
73371
- * nobody is told about.
74682
+ * exhausted the caller gets a typed error naming each leg and why it failed —
74683
+ * `LlmDeadlineExceededError` when the caller's deadline is what ran out,
74684
+ * `ChainExhaustedError` with reason `cross_model_denied` when policy stopped
74685
+ * the walk — because a default returned in place of an answer is a wrong
74686
+ * answer that nobody is told about.
73372
74687
  *
73373
74688
  * @module llm/fallback-chain
73374
74689
  */
@@ -73395,21 +74710,64 @@ class ChainExhaustedError extends Error {
73395
74710
  attempts;
73396
74711
  /** Usage spent across the failed attempts, so the spend is still accounted for. */
73397
74712
  totalUsage;
74713
+ /**
74714
+ * Why the chain ended. `cross_model_denied`: the configured model's attempts
74715
+ * were spent and policy forbade a different model. Callers map every reason
74716
+ * to no decision; the reason says which remedy applies.
74717
+ */
74718
+ reason;
73398
74719
  /**
73399
74720
  * @param alias The alias.
73400
74721
  * @param attempts The attempt record.
73401
74722
  * @param totalUsage Usage spent across all attempts.
74723
+ * @param reason Why the chain ended; defaults to plain exhaustion.
73402
74724
  */
73403
- constructor(alias, attempts, totalUsage) {
74725
+ constructor(alias, attempts, totalUsage, reason = "exhausted") {
73404
74726
  const detail = attempts
73405
74727
  .map((attempt) => `${attempt.role}(${attempt.provider}/${attempt.modelId}): ${attempt.outcome}` +
73406
74728
  (attempt.reason === undefined ? "" : ` — ${attempt.reason}`))
73407
74729
  .join("; ");
73408
- super(`LLM alias "${alias}" exhausted its fallback chain. Attempts: ${detail || "(no leg was servable)"}`);
74730
+ const why = reason === "cross_model_denied"
74731
+ ? " Different-model legs were denied by the caller's cross-model policy."
74732
+ : reason === "deadline_exceeded"
74733
+ ? " The caller's deadline ran out."
74734
+ : "";
74735
+ super(`LLM alias "${alias}" exhausted its fallback chain.${why} Attempts: ${detail || "(no leg was servable)"}`);
73409
74736
  this.name = "ChainExhaustedError";
73410
74737
  this.alias = alias;
73411
74738
  this.attempts = attempts;
73412
74739
  this.totalUsage = totalUsage;
74740
+ this.reason = reason;
74741
+ }
74742
+ }
74743
+ /**
74744
+ * Thrown when the caller's deadline ran out before any leg answered.
74745
+ *
74746
+ * A subclass of {@link ChainExhaustedError}, so a consumer that already treats
74747
+ * exhaustion as "no answer" keeps doing so, while one that needs to tell "the
74748
+ * models failed" from "we ran out of time" can match this class — the two call
74749
+ * for different remedies (a provider problem versus a budget problem), and
74750
+ * both map to no decision, never to a default.
74751
+ */
74752
+ class LlmDeadlineExceededError extends ChainExhaustedError {
74753
+ /** Discriminant for consumers that switch on shape rather than class. */
74754
+ kind = "deadline_exceeded";
74755
+ /** The whole-call budget the chain started with, in milliseconds. */
74756
+ deadlineMs;
74757
+ /** The model class of the last attempt dispatched, or null when none was. */
74758
+ lastModelClass;
74759
+ /**
74760
+ * @param alias The alias.
74761
+ * @param attempts The attempt record.
74762
+ * @param totalUsage Usage spent across all attempts.
74763
+ * @param deadlineMs The budget the chain started with.
74764
+ * @param lastModelClass The last dispatched attempt's model class.
74765
+ */
74766
+ constructor(alias, attempts, totalUsage, deadlineMs, lastModelClass) {
74767
+ super(alias, attempts, totalUsage, "deadline_exceeded");
74768
+ this.name = "LlmDeadlineExceededError";
74769
+ this.deadlineMs = deadlineMs;
74770
+ this.lastModelClass = lastModelClass;
73413
74771
  }
73414
74772
  }
73415
74773
  /**
@@ -73475,148 +74833,70 @@ function legBudgetMs(routeBudgetMs, deadlineAtMs, nowMs) {
73475
74833
  }
73476
74834
  return Math.min(routeBudgetMs, deadlineAtMs - nowMs);
73477
74835
  }
73478
- /** Raised internally when a leg exceeds its budget. */
73479
- class LegTimeoutError extends Error {
73480
- /**
73481
- * @param routeKey The leg that timed out.
73482
- * @param budgetMs Its budget in milliseconds.
73483
- */
73484
- constructor(routeKey, budgetMs) {
73485
- super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
73486
- this.name = "LegTimeoutError";
73487
- }
73488
- }
73489
74836
  /**
73490
- * Run one leg under a hard timeout, honouring the caller's own cancellation.
74837
+ * The model a leg serves, independent of which provider hosts it.
73491
74838
  *
73492
- * The timer is always cleared and the abort listener always removed, including
73493
- * on the success path. A long-lived process that leaked one timer per LLM call
73494
- * would accumulate them at exactly the rate it does useful work.
73495
- *
73496
- * @param leg The leg to run.
73497
- * @param params Normalised parameters for this leg.
73498
- * @param execution The call context.
73499
- * @param budgetMs The leg's budget: its route budget cut to the caller's deadline.
73500
- * @returns The provider's answer.
74839
+ * @param route The leg's route.
74840
+ * @returns Its model class.
73501
74841
  */
73502
- async function runLeg(leg, params, execution, budgetMs) {
73503
- const controller = new AbortController();
73504
- const timer = setTimeout(() => {
73505
- controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
73506
- }, budgetMs);
73507
- const forwardAbort = () => {
73508
- controller.abort(execution.callerSignal?.reason);
73509
- };
73510
- if (execution.callerSignal !== undefined) {
73511
- if (execution.callerSignal.aborted) {
73512
- forwardAbort();
73513
- }
73514
- else {
73515
- execution.callerSignal.addEventListener("abort", forwardAbort, { once: true });
73516
- }
73517
- }
73518
- try {
73519
- // The guards wrap the transport rather than the whole leg, so the per-leg
73520
- // timeout above still bounds the total wait: a caller queued behind the
73521
- // rate limiter is spending its budget just as surely as one waiting on the
73522
- // provider, and only one clock should govern both. The leg's own signal is
73523
- // handed to the guard as well, so a leg whose budget or caller is gone
73524
- // leaves the queue at once instead of holding its place in it.
73525
- const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
73526
- route: leg.route,
73527
- content: execution.content,
73528
- responseFormat: execution.responseFormat,
73529
- params,
73530
- developerPrompt: execution.developerPrompt,
73531
- context: execution.context,
73532
- signal: controller.signal,
73533
- correlationId: execution.correlationId,
73534
- }), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
73535
- assertToolChoiceHonoured(leg.route, params, response);
73536
- return response;
73537
- }
73538
- finally {
73539
- clearTimeout(timer);
73540
- execution.callerSignal?.removeEventListener("abort", forwardAbort);
73541
- }
74842
+ function modelClassOf(route) {
74843
+ return route.modelClass ?? route.modelId;
73542
74844
  }
73543
74845
  /**
73544
- * Classify why a leg failed.
74846
+ * Whether a provider-reported model names the model a route addressed.
73545
74847
  *
73546
- * The distinction matters to the breaker: a timeout and a 5xx are evidence the
73547
- * provider is unhealthy, while the caller cancelling is not. Counting a
73548
- * cancellation as a provider failure would let a burst of user-cancelled
73549
- * requests open the breaker on a perfectly healthy route.
74848
+ * Providers report with or without an organisation prefix and in their own
74849
+ * case, so the comparison is case-insensitive and accepts one side being a
74850
+ * `/`-suffix of the other.
73550
74851
  *
73551
- * @param error The thrown value.
73552
- * @param callerSignal The caller's cancellation signal, if any.
73553
- * @returns The outcome and whether it counts against route health.
74852
+ * @param reported The provider's report.
74853
+ * @param expected The route's model id.
74854
+ * @returns Whether they name the same model.
73554
74855
  */
73555
- function classify(error, callerSignal) {
73556
- if (callerSignal !== undefined && callerSignal.aborted) {
73557
- return {
73558
- outcome: "skipped",
73559
- reason: "caller cancelled",
73560
- countsAgainstHealth: false,
73561
- };
73562
- }
73563
- if (error instanceof LegTimeoutError) {
73564
- return { outcome: "timeout", reason: error.message, countsAgainstHealth: true };
73565
- }
73566
- if (error instanceof UnsupportedCapabilityError) {
73567
- return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
73568
- }
73569
- if (error instanceof ToolChoiceIgnoredError) {
73570
- // The route answered; it broke a declared guarantee rather than failing to
73571
- // be available, so its breaker is not charged for it.
73572
- return { outcome: "error", reason: error.message, countsAgainstHealth: false };
73573
- }
73574
- if (error instanceof RateGuardTimeoutError) {
73575
- // Self-inflicted pacing, not provider ill-health. Counting it would let the
73576
- // client's own throttling open a breaker on a perfectly healthy provider
73577
- // and permanently reroute traffic nobody chose to reroute.
73578
- return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
73579
- }
73580
- const reason = error instanceof Error ? error.message : String(error);
73581
- if (/abort/i.test(reason)) {
73582
- return {
73583
- outcome: "timeout",
73584
- reason: `aborted: ${reason}`,
73585
- countsAgainstHealth: true,
73586
- };
73587
- }
73588
- return { outcome: "error", reason, countsAgainstHealth: true };
74856
+ function isSameReportedModel(reported, expected) {
74857
+ const a = reported.trim().toLowerCase();
74858
+ const b = expected.trim().toLowerCase();
74859
+ return a === b || a.endsWith(`/${b}`) || b.endsWith(`/${a}`);
73589
74860
  }
73590
74861
  /**
73591
- * Whether the caller has stopped waiting.
74862
+ * How an attempt's model relates to the configured one.
73592
74863
  *
73593
- * Read through a function rather than inline, because `AbortSignal.aborted` is
73594
- * a live getter: it can flip to true while a leg is in flight, but a compiler
73595
- * that narrowed it at the top of the loop would prove the later check
73596
- * unreachable and invite its removal. The check is not redundant — it is the
73597
- * only thing that stops the chain spending money on an answer nobody will read.
74864
+ * A leg addressed to a different model class is `different` whatever it
74865
+ * reported. A leg addressed to the configured class is `same` only when it
74866
+ * answered and the provider named the expected model; `different` when the
74867
+ * provider named another; `unknown` when it answered without saying. An
74868
+ * attempt that never answered carries the relation of the model it was
74869
+ * addressed to.
73598
74870
  *
73599
- * @param signal The caller's signal, if any.
73600
- * @returns Whether the call has been cancelled.
74871
+ * @param route The attempt's route.
74872
+ * @param configuredClass The configured model class.
74873
+ * @param fields What the attempt recorded.
74874
+ * @returns The relation.
73601
74875
  */
73602
- function isAborted(signal) {
73603
- return signal !== undefined && signal.aborted;
74876
+ function modelClassRelationOf(route, configuredClass, fields) {
74877
+ if (modelClassOf(route) !== configuredClass) {
74878
+ return "different";
74879
+ }
74880
+ if (fields.outcome !== "ok") {
74881
+ return "same";
74882
+ }
74883
+ if (fields.servedModel === undefined || fields.servedModel === null) {
74884
+ return "unknown";
74885
+ }
74886
+ return isSameReportedModel(fields.servedModel, route.modelId) ? "same" : "different";
73604
74887
  }
73605
74888
  /**
73606
- * The usage a failed leg was billed for, when the leg reached an answer.
74889
+ * The prompt size the latency evidence is bucketed by.
73607
74890
  *
73608
- * A leg that failed after the provider answered — content that does not parse,
73609
- * or prose where a tool call was mandatory — was still charged. A leg that
73610
- * never answered (timeout, outage, skip) carries no usage, and none is invented.
73611
- *
73612
- * @param error The thrown value.
73613
- * @returns The billed usage, or undefined when the leg never produced an answer.
74891
+ * @param execution The call.
74892
+ * @returns Estimated prompt tokens, or null.
73614
74893
  */
73615
- function billedUsageOf(error) {
73616
- if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
73617
- return error.usage;
73618
- }
73619
- return undefined;
74894
+ function promptTokensOf(execution) {
74895
+ return estimatePromptTokens([
74896
+ execution.content,
74897
+ execution.developerPrompt,
74898
+ execution.context,
74899
+ ]);
73620
74900
  }
73621
74901
  /**
73622
74902
  * Walk a chain until a leg answers.
@@ -73624,12 +74904,58 @@ function billedUsageOf(error) {
73624
74904
  * @param alias The alias being served, for error attribution.
73625
74905
  * @param execution The call context.
73626
74906
  * @returns The first successful leg's answer, with the full attempt record.
74907
+ * @throws {LlmDeadlineExceededError} When the caller's deadline ran out first.
73627
74908
  * @throws {ChainExhaustedError} When no leg produced an answer.
73628
74909
  */
73629
74910
  async function executeChain(alias, execution) {
73630
74911
  const now = execution.now ?? Date.now;
74912
+ const startedAt = now();
73631
74913
  const attempts = [];
74914
+ const firstRoute = execution.legs[0]?.route;
74915
+ const configuredClass = execution.configuredModelClass ?? (firstRoute === undefined ? "" : modelClassOf(firstRoute));
74916
+ const policy = execution.crossModelPolicy ?? "allow_record";
74917
+ const promptTokens = execution.latency === undefined ? null : promptTokensOf(execution);
73632
74918
  let totalUsage = EMPTY_USAGE;
74919
+ let dispatchIndex = 0;
74920
+ let deadlineHit = false;
74921
+ let crossModelDenied = false;
74922
+ let lastModelClass = null;
74923
+ /**
74924
+ * Record one attempt with its provenance.
74925
+ *
74926
+ * @param route The attempt's route.
74927
+ * @param fields What happened.
74928
+ * @param dispatch Whether it was a hedge, and its dispatch index.
74929
+ * @param servedProvider The provider's own report of who served, if any.
74930
+ * @returns The record.
74931
+ */
74932
+ const record = (route, fields, dispatch, servedProvider) => {
74933
+ const full = {
74934
+ ...fields,
74935
+ servedProvider: servedProvider ?? route.providerName,
74936
+ modelClass: modelClassOf(route),
74937
+ modelClassRelation: modelClassRelationOf(route, configuredClass, fields),
74938
+ ...(dispatch === undefined
74939
+ ? {}
74940
+ : { hedged: dispatch.hedged, attemptIndex: dispatch.attemptIndex }),
74941
+ };
74942
+ attempts.push(full);
74943
+ execution.onAttempt?.(full);
74944
+ return full;
74945
+ };
74946
+ /**
74947
+ * The identity fields of a leg that was never dispatched.
74948
+ *
74949
+ * @param route The leg's route.
74950
+ * @returns The fields.
74951
+ */
74952
+ const undispatched = (route) => ({
74953
+ routeKey: route.routeKey,
74954
+ role: route.role,
74955
+ provider: route.providerName,
74956
+ modelId: route.modelId,
74957
+ durationMs: 0,
74958
+ });
73633
74959
  for (const leg of execution.legs) {
73634
74960
  const { route } = leg;
73635
74961
  if (isAborted(execution.callerSignal)) {
@@ -73637,108 +74963,80 @@ async function executeChain(alias, execution) {
73637
74963
  // spend money on an answer nobody will read.
73638
74964
  break;
73639
74965
  }
73640
- if (leg.params instanceof UnsupportedCapabilityError) {
73641
- const record = {
73642
- routeKey: route.routeKey,
73643
- role: route.role,
73644
- provider: route.providerName,
73645
- modelId: route.modelId,
74966
+ if (policy === "deny" && modelClassOf(route) !== configuredClass) {
74967
+ crossModelDenied = true;
74968
+ record(route, {
74969
+ ...undispatched(route),
73646
74970
  outcome: "skipped",
73647
- durationMs: 0,
73648
- reason: leg.params.message,
73649
- };
73650
- attempts.push(record);
73651
- execution.onAttempt?.(record);
74971
+ reason: `cross-model leg denied by policy: configured model is ${configuredClass}, this leg serves ${modelClassOf(route)}`,
74972
+ });
74973
+ continue;
74974
+ }
74975
+ if (leg.params instanceof UnsupportedCapabilityError) {
74976
+ record(route, { ...undispatched(route), outcome: "skipped", reason: leg.params.message });
73652
74977
  continue;
73653
74978
  }
73654
74979
  if (!execution.breakers.allows(route.routeKey)) {
73655
- const record = {
73656
- routeKey: route.routeKey,
73657
- role: route.role,
73658
- provider: route.providerName,
73659
- modelId: route.modelId,
74980
+ record(route, {
74981
+ ...undispatched(route),
73660
74982
  outcome: "breaker-open",
73661
- durationMs: 0,
73662
74983
  reason: `circuit breaker is ${execution.breakers.stateOf(route.routeKey)}`,
73663
- };
73664
- attempts.push(record);
73665
- execution.onAttempt?.(record);
74984
+ });
73666
74985
  continue;
73667
74986
  }
73668
74987
  const budgetMs = legBudgetMs(route.timeoutMs, execution.deadlineAtMs, now());
73669
74988
  if (budgetMs <= 0) {
73670
74989
  // The caller's deadline is spent. Dispatching now would start a call
73671
74990
  // that is cancelled the moment it begins, and charge nothing but noise.
73672
- const record = {
73673
- routeKey: route.routeKey,
73674
- role: route.role,
73675
- provider: route.providerName,
73676
- modelId: route.modelId,
74991
+ deadlineHit = true;
74992
+ record(route, {
74993
+ ...undispatched(route),
73677
74994
  outcome: "skipped",
73678
- durationMs: 0,
73679
74995
  reason: "caller deadline exhausted before this leg",
73680
- };
73681
- attempts.push(record);
73682
- execution.onAttempt?.(record);
74996
+ });
73683
74997
  continue;
73684
74998
  }
73685
- const startedAt = now();
73686
- const holdsProbe = execution.breakers.onAttemptStart(route.routeKey);
73687
- try {
73688
- const response = await runLeg(leg, leg.params, execution, budgetMs);
73689
- execution.breakers.onSuccess(route.routeKey);
73690
- totalUsage = sumUsage(totalUsage, response.usage);
73691
- const record = {
73692
- routeKey: route.routeKey,
73693
- role: route.role,
73694
- provider: route.providerName,
73695
- modelId: route.modelId,
73696
- outcome: "ok",
73697
- durationMs: now() - startedAt,
73698
- budgetMs,
73699
- servedModel: response.servedModel ?? null,
73700
- usage: response.usage,
73701
- };
73702
- attempts.push(record);
73703
- execution.onAttempt?.(record);
73704
- return { response, servedBy: route, attempts, totalUsage };
74999
+ lastModelClass = modelClassOf(route);
75000
+ const group = await runSameModelGroup(leg, budgetMs, budgetMs < route.timeoutMs, {
75001
+ request: execution,
75002
+ breakers: execution.breakers,
75003
+ now,
75004
+ policy: execution.hedging,
75005
+ tracker: execution.latency,
75006
+ promptTokens,
75007
+ admitDuplicate: execution.admitDuplicate ?? (() => false),
75008
+ record: (attemptRoute, fields, dispatch, servedProvider) => {
75009
+ record(attemptRoute, fields, dispatch, servedProvider);
75010
+ },
75011
+ nextAttemptIndex: () => {
75012
+ const index = dispatchIndex;
75013
+ dispatchIndex += 1;
75014
+ return index;
75015
+ },
75016
+ });
75017
+ for (const usage of group.billed) {
75018
+ totalUsage = sumUsage(totalUsage, usage);
73705
75019
  }
73706
- catch (error) {
73707
- const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
73708
- if (countsAgainstHealth) {
73709
- execution.breakers.onFailure(route.routeKey);
73710
- }
73711
- else if (holdsProbe) {
73712
- // No verdict on the route's health, but the probe slot this attempt
73713
- // took must come back, or a half-open route admits no probe ever again.
73714
- execution.breakers.onAttemptAbandoned(route.routeKey);
73715
- }
73716
- // A provider that answered — with unparseable content, or in prose where a
73717
- // tool call was mandatory — still billed for the answer; the spend belongs
73718
- // in the total whether or not a later leg serves.
73719
- const billed = billedUsageOf(error);
73720
- const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
73721
- totalUsage = sumUsage(totalUsage, billed);
73722
- const record = {
73723
- routeKey: route.routeKey,
73724
- role: route.role,
73725
- provider: route.providerName,
73726
- modelId: route.modelId,
73727
- outcome,
73728
- durationMs: now() - startedAt,
73729
- budgetMs,
73730
- reason,
73731
- ...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
73732
- ...(billed === undefined ? {} : { usage: billed }),
75020
+ deadlineHit = deadlineHit || group.deadlineBound;
75021
+ if (group.answer !== undefined) {
75022
+ const winner = attempts.find((attempt) => attempt.outcome === "ok" && attempt.routeKey === group.answer?.route.routeKey);
75023
+ return {
75024
+ response: group.answer.response,
75025
+ servedBy: group.answer.route,
75026
+ attempts,
75027
+ totalUsage,
75028
+ modelClassRelation: winner?.modelClassRelation ?? "unknown",
75029
+ hedged: group.answer.hedged,
73733
75030
  };
73734
- attempts.push(record);
73735
- execution.onAttempt?.(record);
73736
- if (outcome === "skipped" && isAborted(execution.callerSignal)) {
73737
- break;
73738
- }
73739
75031
  }
75032
+ if (isAborted(execution.callerSignal)) {
75033
+ break;
75034
+ }
75035
+ }
75036
+ if (deadlineHit && !isAborted(execution.callerSignal) && execution.deadlineAtMs !== undefined) {
75037
+ throw new LlmDeadlineExceededError(alias, attempts, totalUsage, execution.deadlineAtMs - startedAt, lastModelClass);
73740
75038
  }
73741
- throw new ChainExhaustedError(alias, attempts, totalUsage);
75039
+ throw new ChainExhaustedError(alias, attempts, totalUsage, crossModelDenied ? "cross_model_denied" : "exhausted");
73742
75040
  }
73743
75041
 
73744
75042
  var schema_version = 1;
@@ -73754,7 +75052,38 @@ var defaults = {
73754
75052
  circuit_breaker: {
73755
75053
  failure_threshold: 5,
73756
75054
  cooldown_ms: 60000,
73757
- half_open_probes: 1
75055
+ capacity_cooldown_ms: 15000,
75056
+ half_open_probes: 1,
75057
+ probe_fraction: 0.1,
75058
+ latency_trip: {
75059
+ enabled: false,
75060
+ slo_ms: {
75061
+ "hot-path": 20000,
75062
+ background: 60000,
75063
+ batch: 240000
75064
+ },
75065
+ quantile: 0.9,
75066
+ window_size: 20,
75067
+ trip_windows: 3,
75068
+ of_windows: 5
75069
+ }
75070
+ },
75071
+ hedging: {
75072
+ max_same_model_hedges: 1,
75073
+ hedge_quantile: 0.9,
75074
+ timeout_quantile: 0.99,
75075
+ k_timeout: 3,
75076
+ attempt_timeout_floor_ms: 5000,
75077
+ max_attempt_share: 0.5,
75078
+ duplicate_headroom_reserve: 0.25,
75079
+ min_samples: 20,
75080
+ window_size: 200,
75081
+ sample_max_age_ms: 900000,
75082
+ prompt_token_buckets: [
75083
+ 2000,
75084
+ 8000,
75085
+ 32000
75086
+ ]
73758
75087
  }
73759
75088
  };
73760
75089
  var providers = {
@@ -74490,6 +75819,86 @@ const ISOLATED_SUFFIX = ".isolated";
74490
75819
  * caller change every other caller's routing.
74491
75820
  */
74492
75821
  const routeTable = rawTable;
75822
+ /** Separates a leg's route key from the provider of one of its equivalents. */
75823
+ const EQUIVALENT_SEPARATOR = "~";
75824
+ /** Gateway name segment for an equivalent leg. */
75825
+ const EQUIVALENT_NAMESPACE = "equivalent";
75826
+ /**
75827
+ * Whether a number lies in a closed or open range.
75828
+ *
75829
+ * @param value The value.
75830
+ * @param min Lower bound.
75831
+ * @param max Upper bound.
75832
+ * @param open Whether the bounds are excluded.
75833
+ * @returns Whether it is in range.
75834
+ */
75835
+ function inRange(value, min, max, open) {
75836
+ if (typeof value !== "number" || !Number.isFinite(value)) {
75837
+ return false;
75838
+ }
75839
+ return open ? value > min && value < max : value >= min && value <= max;
75840
+ }
75841
+ /**
75842
+ * Bounds violations in the tail-latency defaults (hedging, probe scaling and
75843
+ * the latency trip).
75844
+ *
75845
+ * These values are mechanics rather than routing choices, but a value outside
75846
+ * its bounds turns a mechanic into a routing change — a quantile of 1.0 makes
75847
+ * every hedge wait for the slowest answer ever seen, a zero floor lets a
75848
+ * measured timeout cut an attempt the instant it starts — so they are checked
75849
+ * as strictly as the routing itself.
75850
+ *
75851
+ * @param defaults The table's defaults.
75852
+ * @returns One message per violation; empty when every value is in bounds.
75853
+ */
75854
+ function tailLatencyViolations(defaults) {
75855
+ const violations = [];
75856
+ const check = (ok, message) => {
75857
+ if (!ok) {
75858
+ violations.push(message);
75859
+ }
75860
+ };
75861
+ const hedging = defaults.hedging;
75862
+ if (hedging !== undefined) {
75863
+ check(Number.isInteger(hedging.max_same_model_hedges) && inRange(hedging.max_same_model_hedges, 0, 3, false), "hedging.max_same_model_hedges must be an integer in [0, 3]");
75864
+ check(inRange(hedging.hedge_quantile, 0.5, 1, true), "hedging.hedge_quantile must be in (0.5, 1)");
75865
+ check(inRange(hedging.timeout_quantile, 0.5, 1, true), "hedging.timeout_quantile must be in (0.5, 1)");
75866
+ check(hedging.timeout_quantile >= hedging.hedge_quantile, "hedging.timeout_quantile must not be below hedging.hedge_quantile");
75867
+ check(inRange(hedging.k_timeout, 1, 10, false), "hedging.k_timeout must be in [1, 10]");
75868
+ check(inRange(hedging.attempt_timeout_floor_ms, 1000, Number.MAX_SAFE_INTEGER, false), "hedging.attempt_timeout_floor_ms must be at least 1000");
75869
+ check(inRange(hedging.max_attempt_share, 0.25, 1, false), "hedging.max_attempt_share must be in [0.25, 1]");
75870
+ check(inRange(hedging.duplicate_headroom_reserve, 0, 0.9, false), "hedging.duplicate_headroom_reserve must be in [0, 0.9]");
75871
+ check(Number.isInteger(hedging.min_samples) && inRange(hedging.min_samples, 5, 10_000, false), "hedging.min_samples must be an integer in [5, 10000]");
75872
+ check(Number.isInteger(hedging.window_size) && hedging.window_size >= hedging.min_samples, "hedging.window_size must be an integer no smaller than min_samples");
75873
+ check(inRange(hedging.sample_max_age_ms, 60_000, Number.MAX_SAFE_INTEGER, false), "hedging.sample_max_age_ms must be at least 60000");
75874
+ check(hedging.prompt_token_buckets.every((edge, index, edges) => Number.isFinite(edge) && edge > 0 && (index === 0 || edge > edges[index - 1])), "hedging.prompt_token_buckets must be positive and strictly ascending");
75875
+ }
75876
+ const breaker = defaults.circuit_breaker;
75877
+ if (breaker.probe_fraction !== undefined) {
75878
+ check(inRange(breaker.probe_fraction, 0, 0.5, false), "circuit_breaker.probe_fraction must be in [0, 0.5]");
75879
+ }
75880
+ const trip = breaker.latency_trip;
75881
+ if (trip !== undefined) {
75882
+ check(inRange(trip.quantile, 0.5, 1, true), "circuit_breaker.latency_trip.quantile must be in (0.5, 1)");
75883
+ check(Number.isInteger(trip.window_size) && trip.window_size >= 5, "circuit_breaker.latency_trip.window_size must be an integer of at least 5");
75884
+ check(Number.isInteger(trip.of_windows) &&
75885
+ Number.isInteger(trip.trip_windows) &&
75886
+ trip.trip_windows >= 1 &&
75887
+ trip.trip_windows <= trip.of_windows, "circuit_breaker.latency_trip needs integer 1 <= trip_windows <= of_windows");
75888
+ for (const latencyClass of Object.keys(defaults.request_timeout_ms)) {
75889
+ check(inRange(trip.slo_ms[latencyClass], 1000, defaults.request_timeout_ms[latencyClass], false), `circuit_breaker.latency_trip.slo_ms.${latencyClass} must be in [1000, its request timeout]`);
75890
+ }
75891
+ }
75892
+ return violations;
75893
+ }
75894
+ {
75895
+ // Checked when the table loads: it is bundled, so a violation can only
75896
+ // arrive in a release, and it must stop that release rather than run it.
75897
+ const violations = tailLatencyViolations(routeTable.defaults);
75898
+ if (violations.length > 0) {
75899
+ throw new Error(`alias route table has out-of-bounds tail-latency defaults: ${violations.join("; ")}`);
75900
+ }
75901
+ }
74493
75902
  /**
74494
75903
  * Every alias the table defines.
74495
75904
  *
@@ -74669,6 +76078,8 @@ function resolveChain(alias, options = {}) {
74669
76078
  });
74670
76079
  continue;
74671
76080
  }
76081
+ const routeKey = routeKeyFor(alias, isolated, route.role);
76082
+ const modelClass = route.model_class ?? admission.modelId;
74672
76083
  resolved.push({
74673
76084
  alias,
74674
76085
  isolated,
@@ -74678,13 +76089,67 @@ function resolveChain(alias, options = {}) {
74678
76089
  modelId: admission.modelId,
74679
76090
  lumicModel: route.lumic_model ?? null,
74680
76091
  params: route.params ?? {},
74681
- routeKey: routeKeyFor(alias, isolated, route.role),
76092
+ routeKey,
74682
76093
  timeoutMs,
74683
- retriesPerLeg: routeTable.defaults.retries_per_leg,
76094
+ modelClass,
76095
+ latencyClass: definition.latency_class,
76096
+ equivalents: resolveEquivalents(route, {
76097
+ alias,
76098
+ isolated,
76099
+ routeKey,
76100
+ timeoutMs,
76101
+ modelClass,
76102
+ latencyClass: definition.latency_class,
76103
+ }),
74684
76104
  });
74685
76105
  }
74686
76106
  return { alias, isolated, routes: resolved, exclusions };
74687
76107
  }
76108
+ /**
76109
+ * Resolve a leg's same-model equivalents that can serve today.
76110
+ *
76111
+ * An equivalent is admitted by the leg's own rules — a confirmed model id and a
76112
+ * live provider account — and inherits the leg's model class, because being the
76113
+ * same model is what makes it an equivalent. One that cannot serve is simply
76114
+ * absent: it is never a reason to skip the leg.
76115
+ *
76116
+ * @param route The authored leg.
76117
+ * @param leg The resolved leg's identity.
76118
+ * @param leg.alias The alias.
76119
+ * @param leg.isolated Whether this is the isolated variant.
76120
+ * @param leg.routeKey The leg's route key.
76121
+ * @param leg.timeoutMs The leg's budget.
76122
+ * @param leg.modelClass The leg's model class.
76123
+ * @param leg.latencyClass The alias's latency class.
76124
+ * @returns The servable equivalents, in table order.
76125
+ */
76126
+ function resolveEquivalents(route, leg) {
76127
+ const resolved = [];
76128
+ for (const equivalent of route.equivalents ?? []) {
76129
+ const provider = routeTable.providers[equivalent.provider];
76130
+ if (provider === undefined ||
76131
+ provider.account_status !== "live" ||
76132
+ equivalent.model_id_status !== "confirmed" ||
76133
+ equivalent.model_id === null) {
76134
+ continue;
76135
+ }
76136
+ resolved.push({
76137
+ alias: leg.alias,
76138
+ isolated: leg.isolated,
76139
+ role: route.role,
76140
+ providerName: equivalent.provider,
76141
+ provider,
76142
+ modelId: equivalent.model_id,
76143
+ lumicModel: equivalent.lumic_model ?? null,
76144
+ params: equivalent.params ?? route.params ?? {},
76145
+ routeKey: `${leg.routeKey}${EQUIVALENT_SEPARATOR}${equivalent.provider}`,
76146
+ timeoutMs: leg.timeoutMs,
76147
+ modelClass: leg.modelClass,
76148
+ latencyClass: leg.latencyClass,
76149
+ });
76150
+ }
76151
+ return resolved;
76152
+ }
74688
76153
  /**
74689
76154
  * The permanent closed-incumbent leg of an alias (PD-11).
74690
76155
  *
@@ -74713,6 +76178,12 @@ function closedIncumbentLeg(chain) {
74713
76178
  */
74714
76179
  function gatewayModelNameFor(route, chain) {
74715
76180
  const base = `${route.alias}${route.isolated ? ISOLATED_SUFFIX : ""}`;
76181
+ if (route.routeKey.includes(EQUIVALENT_SEPARATOR)) {
76182
+ // The same model at another provider is its own gateway deployment, named
76183
+ // so the gateway serves exactly that deployment and nothing it might fall
76184
+ // back to on its own.
76185
+ return `${base}.${EQUIVALENT_NAMESPACE}.${route.role}.${route.providerName}`;
76186
+ }
74716
76187
  return chain.routes[0]?.routeKey === route.routeKey
74717
76188
  ? base
74718
76189
  : `${base}.fallback.${route.role}`;
@@ -75000,8 +76471,11 @@ async function resolveDefaultDirectCaller() {
75000
76471
  * gateway by its own model name, so the gateway serves one deployment per leg
75001
76472
  * and needs no fallback of its own. A proxy-side fallback inside a leg would
75002
76473
  * spend the leg's budget on a model the chain did not choose and report the
75003
- * answer as the leg's; the served model is read from the response body so such
75004
- * a substitution stays visible while any remains configured.
76474
+ * answer as the leg's; the served model is read from the gateway's own
76475
+ * served-model header, else from the upstream body's `model`, so such a
76476
+ * substitution stays visible while any remains configured. A body `model` that
76477
+ * merely echoes the name the leg was addressed by is the gateway naming its
76478
+ * model GROUP, not the model that answered, and is reported as unknown.
75005
76479
  *
75006
76480
  * The gateway key is read from the environment by NAME at call time and never
75007
76481
  * stored, logged, or included in an error (PD-2). Reading it per call rather
@@ -75017,6 +76491,14 @@ const ERROR_BODY_EXCERPT = 400;
75017
76491
  const RESPONSE_COST_HEADER = "x-litellm-response-cost";
75018
76492
  /** Response header naming the proxy deployment that served the call. */
75019
76493
  const DEPLOYMENT_ID_HEADER = "x-litellm-model-id";
76494
+ /**
76495
+ * Response header in which the gateway reports the upstream model that
76496
+ * actually answered. Takes precedence over the body's `model`, which a proxy
76497
+ * may overwrite with the model-group name it was addressed by.
76498
+ */
76499
+ const SERVED_MODEL_HEADER = "x-adaptic-served-model";
76500
+ /** Response header in which the gateway reports the upstream provider that answered. */
76501
+ const SERVED_PROVIDER_HEADER = "x-adaptic-served-provider";
75020
76502
  /**
75021
76503
  * Thrown when the gateway itself is unreachable, as opposed to a provider
75022
76504
  * behind it failing.
@@ -75179,6 +76661,7 @@ function createGatewayTransport(config) {
75179
76661
  throw new GatewayResponseError(response.status, await response.text());
75180
76662
  }
75181
76663
  const payload = (await response.json());
76664
+ const addressedAs = body.model;
75182
76665
  const choices = payload.choices;
75183
76666
  const message = choices?.[0]?.message;
75184
76667
  // Usage is read before the content is interpreted. The provider billed for
@@ -75191,14 +76674,36 @@ function createGatewayTransport(config) {
75191
76674
  tool_calls: Array.isArray(message?.tool_calls)
75192
76675
  ? message.tool_calls
75193
76676
  : undefined,
75194
- // The model the provider says answered, which a proxy-side fallback can
75195
- // make differ from the leg's route model; unreported stays null.
75196
- servedModel: nonEmptyOrNull(payload.model),
76677
+ // The model that answered, which a proxy-side fallback can make differ
76678
+ // from the leg's route model; unreported (or only the group echo) stays null.
76679
+ servedModel: servedModelOf(response.headers, payload.model, addressedAs),
75197
76680
  servedDeploymentId: nonEmptyOrNull(response.headers?.get(DEPLOYMENT_ID_HEADER)),
76681
+ servedProvider: nonEmptyOrNull(response.headers?.get(SERVED_PROVIDER_HEADER)),
75198
76682
  };
75199
76683
  },
75200
76684
  };
75201
76685
  }
76686
+ /**
76687
+ * The model that answered, as far as the gateway says.
76688
+ *
76689
+ * The gateway's served-model header is authoritative when present. Otherwise
76690
+ * the body's `model` is used — unless it equals the name the leg was addressed
76691
+ * by, which is the proxy echoing its model group rather than naming the
76692
+ * upstream model, and would otherwise read as a confirmed same-model answer.
76693
+ *
76694
+ * @param headers The response headers.
76695
+ * @param bodyModel The body's `model` field.
76696
+ * @param addressedAs The gateway model name the leg was sent to.
76697
+ * @returns The served model, or null when the gateway did not say.
76698
+ */
76699
+ function servedModelOf(headers, bodyModel, addressedAs) {
76700
+ const fromHeader = nonEmptyOrNull(headers?.get(SERVED_MODEL_HEADER));
76701
+ if (fromHeader !== null) {
76702
+ return fromHeader;
76703
+ }
76704
+ const fromBody = nonEmptyOrNull(bodyModel);
76705
+ return fromBody === null || fromBody === addressedAs ? null : fromBody;
76706
+ }
75202
76707
  /**
75203
76708
  * Compose the message array for a request.
75204
76709
  *
@@ -75262,7 +76767,9 @@ function interpretContent(content, responseFormat, usage) {
75262
76767
  * Every call gets, in order: alias resolution against the canonical route
75263
76768
  * table, per-provider parameter normalisation, a hard per-leg timeout, a
75264
76769
  * per-route circuit breaker, an ordered fallback chain ending at the closed
75265
- * incumbent, and — where the caller supplies a validator — one schema-feedback
76770
+ * incumbent — each leg first hedged and failed over on the SAME model (see
76771
+ * `hedge.ts`), and a different-model leg reached only when the caller's
76772
+ * cross-model policy allows it — and, where the caller supplies a validator, one schema-feedback
75266
76773
  * retry ahead of the chain. The caller's `timeoutMs` is ONE deadline for the
75267
76774
  * whole call: each leg runs for its route budget or for what remains of that
75268
76775
  * deadline, whichever is shorter, so the chain is the single fallback owner and
@@ -75278,6 +76785,34 @@ const GATEWAY_BASE_URL_ENV = "LLM_GATEWAY_BASE_URL";
75278
76785
  const DEFAULT_GATEWAY_KEY_ENV = "LLM_GATEWAY_API_KEY";
75279
76786
  /** Process-wide breaker registry, so route health is shared across call sites. */
75280
76787
  let breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker);
76788
+ /**
76789
+ * Build the process-wide latency tracker from the table's hedging defaults.
76790
+ *
76791
+ * @param now Clock.
76792
+ * @returns The tracker, or undefined when the table configures no hedging.
76793
+ */
76794
+ function buildLatencyTracker(now) {
76795
+ const hedging = routeTable.defaults.hedging;
76796
+ if (hedging === undefined) {
76797
+ return undefined;
76798
+ }
76799
+ return new LegLatencyTracker({
76800
+ minSamples: hedging.min_samples,
76801
+ windowSize: hedging.window_size,
76802
+ sampleMaxAgeMs: hedging.sample_max_age_ms,
76803
+ promptTokenBuckets: hedging.prompt_token_buckets,
76804
+ }, now);
76805
+ }
76806
+ /**
76807
+ * Process-wide healthy-latency evidence, shared across call sites for the same
76808
+ * reason the breakers are: a model's health is one population however many
76809
+ * callers reach it.
76810
+ */
76811
+ let latencyTracker = buildLatencyTracker();
76812
+ /** Same-model hedging policy from the table, or undefined when none is configured. */
76813
+ const hedgingPolicy = routeTable.defaults.hedging === undefined
76814
+ ? undefined
76815
+ : sameModelPolicyFrom(routeTable.defaults.hedging);
75281
76816
  /** Active runtime wiring. */
75282
76817
  let config = {};
75283
76818
  /** Lazily built transports, rebuilt whenever configuration changes. */
@@ -75299,6 +76834,28 @@ function configureLlmClient(next) {
75299
76834
  gatewayTransport = null;
75300
76835
  directTransport = null;
75301
76836
  breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker, next.now);
76837
+ latencyTracker = buildLatencyTracker(next.now);
76838
+ }
76839
+ /**
76840
+ * Inspect the healthy-latency evidence the hedging controls read.
76841
+ *
76842
+ * @returns The live tracker, or undefined when the table configures no hedging.
76843
+ */
76844
+ function llmLatencyTracker() {
76845
+ return latencyTracker;
76846
+ }
76847
+ /**
76848
+ * Whether a duplicate attempt on the same provider may start.
76849
+ *
76850
+ * @param route The leg the duplicate would address.
76851
+ * @param reserveFraction Share of the provider's capacity kept free.
76852
+ * @returns Whether it may start.
76853
+ */
76854
+ function admitDuplicate(route, reserveFraction) {
76855
+ if (config.duplicateAdmission !== undefined) {
76856
+ return config.duplicateAdmission(route, reserveFraction);
76857
+ }
76858
+ return hasDuplicateHeadroom(route.providerName, route.modelId, reserveFraction);
75302
76859
  }
75303
76860
  /**
75304
76861
  * Inspect route health.
@@ -75371,10 +76928,11 @@ function directFor() {
75371
76928
  * @param options The caller's options.
75372
76929
  * @param responseFormat The requested response shape.
75373
76930
  * @param transport The transport to carry every leg.
76931
+ * @param admitEquivalent Which same-model equivalents this transport can reach.
75374
76932
  * @returns Prepared legs, in chain order.
75375
76933
  */
75376
- function prepareLegs(chain, options, responseFormat, transport) {
75377
- return chain.routes.map((route) => {
76934
+ function prepareLegs(chain, options, responseFormat, transport, admitEquivalent = () => true) {
76935
+ const prepare = (route) => {
75378
76936
  try {
75379
76937
  return {
75380
76938
  route,
@@ -75388,7 +76946,11 @@ function prepareLegs(chain, options, responseFormat, transport) {
75388
76946
  }
75389
76947
  throw error;
75390
76948
  }
75391
- });
76949
+ };
76950
+ return chain.routes.map((route) => ({
76951
+ ...prepare(route),
76952
+ equivalents: (route.equivalents ?? []).filter(admitEquivalent).map(prepare),
76953
+ }));
75392
76954
  }
75393
76955
  /**
75394
76956
  * Call a language model by semantic alias.
@@ -75432,6 +76994,26 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75432
76994
  }
75433
76995
  const gateway = gatewayFor();
75434
76996
  const attemptLog = [];
76997
+ // What every execution of this call shares, whichever transport carries it.
76998
+ // The configured model is the head of the full chain, so the degraded path —
76999
+ // whose legs are only the closed incumbents — still knows that its answer is
77000
+ // a different model from the one configured.
77001
+ const shared = {
77002
+ responseFormat,
77003
+ developerPrompt: options.developerPrompt,
77004
+ context: options.context,
77005
+ breakers,
77006
+ correlationId: options.correlationId,
77007
+ callerSignal: options.signal,
77008
+ deadlineAtMs,
77009
+ now: config.now,
77010
+ onAttempt: (record) => attemptLog.push(record),
77011
+ hedging: hedgingPolicy,
77012
+ latency: latencyTracker,
77013
+ admitDuplicate,
77014
+ crossModelPolicy: options.crossModelPolicy,
77015
+ configuredModelClass: modelClassOf(chain.routes[0]),
77016
+ };
75435
77017
  /**
75436
77018
  * Run the chain, falling back from the gateway to the degraded direct path
75437
77019
  * only when the gateway itself is unreachable.
@@ -75444,17 +77026,9 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75444
77026
  if (gateway !== null) {
75445
77027
  try {
75446
77028
  const outcome = await executeChain(options.alias, {
77029
+ ...shared,
75447
77030
  legs: prepareLegs(chain, options, responseFormat, gateway),
75448
77031
  content: boundContent,
75449
- responseFormat,
75450
- developerPrompt: options.developerPrompt,
75451
- context: options.context,
75452
- breakers,
75453
- correlationId: options.correlationId,
75454
- callerSignal: options.signal,
75455
- deadlineAtMs,
75456
- now: config.now,
75457
- onAttempt: (record) => attemptLog.push(record),
75458
77032
  });
75459
77033
  return { ...outcome, degraded: false };
75460
77034
  }
@@ -75479,17 +77053,11 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75479
77053
  });
75480
77054
  }
75481
77055
  const outcome = await executeChain(options.alias, {
75482
- legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct),
77056
+ ...shared,
77057
+ // The direct transport serves closed vendors only, so only a closed
77058
+ // equivalent can be reached on the degraded path.
77059
+ legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct, (route) => route.provider.tier === "closed"),
75483
77060
  content: boundContent,
75484
- responseFormat,
75485
- developerPrompt: options.developerPrompt,
75486
- context: options.context,
75487
- breakers,
75488
- correlationId: options.correlationId,
75489
- callerSignal: options.signal,
75490
- deadlineAtMs,
75491
- now: config.now,
75492
- onAttempt: (record) => attemptLog.push(record),
75493
77061
  });
75494
77062
  return { ...outcome, degraded: true };
75495
77063
  };
@@ -75504,6 +77072,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75504
77072
  attempts: attemptLog,
75505
77073
  degraded: outcome.degraded,
75506
77074
  totalUsage: outcome.totalUsage,
77075
+ modelClassRelation: outcome.modelClassRelation,
77076
+ hedged: outcome.hedged,
75507
77077
  };
75508
77078
  }
75509
77079
  // A validator is only meaningful against a text prompt, because the retry has
@@ -75521,6 +77091,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75521
77091
  servedBy: outcome.servedBy,
75522
77092
  degraded: outcome.degraded,
75523
77093
  totalUsage: outcome.totalUsage,
77094
+ modelClassRelation: outcome.modelClassRelation,
77095
+ hedged: outcome.hedged,
75524
77096
  };
75525
77097
  return outcome.response;
75526
77098
  },
@@ -75538,6 +77110,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75538
77110
  attempts: attemptLog,
75539
77111
  degraded: routing.degraded,
75540
77112
  totalUsage: validated.totalUsage,
77113
+ modelClassRelation: routing.modelClassRelation,
77114
+ hedged: routing.hedged,
75541
77115
  };
75542
77116
  }
75543
77117
  /**
@@ -80999,13 +82573,17 @@ exports.DEFAULT_TRADING_POLICY = DEFAULT_TRADING_POLICY;
80999
82573
  exports.DataFormatError = DataFormatError;
81000
82574
  exports.DirectTransportRefusedError = DirectTransportRefusedError;
81001
82575
  exports.DuplicateClientOrderIdError = DuplicateClientOrderIdError;
82576
+ exports.EQUIVALENT_SEPARATOR = EQUIVALENT_SEPARATOR;
81002
82577
  exports.GatewayResponseError = GatewayResponseError;
81003
82578
  exports.GatewayUnreachableError = GatewayUnreachableError;
81004
82579
  exports.HttpClientError = HttpClientError;
81005
82580
  exports.HttpServerError = HttpServerError;
81006
82581
  exports.KEEP_ALIVE_DEFAULTS = KEEP_ALIVE_DEFAULTS;
82582
+ exports.LegLatencyTracker = LegLatencyTracker;
82583
+ exports.LlmDeadlineExceededError = LlmDeadlineExceededError;
81007
82584
  exports.LlmResponseFormatError = LlmResponseFormatError;
81008
82585
  exports.MARKET_DATA_API = MARKET_DATA_API;
82586
+ exports.MIN_CONVERTED_TRAIL_PERCENT = MIN_CONVERTED_TRAIL_PERCENT;
81009
82587
  exports.MassiveAggregatesResponseSchema = MassiveAggregatesResponseSchema;
81010
82588
  exports.MassiveApiError = MassiveApiError;
81011
82589
  exports.MassiveDailyOpenCloseSchema = MassiveDailyOpenCloseSchema;
@@ -81028,6 +82606,8 @@ exports.RISK_FREE_RATE_TTL_MS = RISK_FREE_RATE_TTL_MS;
81028
82606
  exports.RateGuardTimeoutError = RateGuardTimeoutError;
81029
82607
  exports.RateLimitError = RateLimitError;
81030
82608
  exports.RawMassivePriceDataSchema = RawMassivePriceDataSchema;
82609
+ exports.SERVED_MODEL_HEADER = SERVED_MODEL_HEADER;
82610
+ exports.SERVED_PROVIDER_HEADER = SERVED_PROVIDER_HEADER;
81031
82611
  exports.SchemaRetryExhaustedError = SchemaRetryExhaustedError;
81032
82612
  exports.StampedeProtectedCache = StampedeProtectedCache;
81033
82613
  exports.StreamProviderError = StreamProviderError;
@@ -81037,6 +82617,7 @@ exports.TimeoutError = TimeoutError;
81037
82617
  exports.TokenBucketRateLimiter = TokenBucketRateLimiter;
81038
82618
  exports.ToolChoiceIgnoredError = ToolChoiceIgnoredError;
81039
82619
  exports.TradeError = TradeError;
82620
+ exports.TrailUnitConversionRefusedError = TrailUnitConversionRefusedError;
81040
82621
  exports.TrailingStopValidationError = TrailingStopValidationError;
81041
82622
  exports.USDC_PAIRS = USDC_PAIRS;
81042
82623
  exports.USDT_PAIRS = USDT_PAIRS;
@@ -81122,6 +82703,7 @@ exports.createVerticalSpread = createVerticalSpread$1;
81122
82703
  exports.createVerticalSpreadAdvanced = createVerticalSpread;
81123
82704
  exports.enrichAlpacaError = enrichAlpacaError;
81124
82705
  exports.entryWithPercentStopLoss = entryWithPercentStopLoss;
82706
+ exports.estimatePromptTokens = estimatePromptTokens;
81125
82707
  exports.exerciseOption = exerciseOption;
81126
82708
  exports.extractAlpacaBrokerError = extractAlpacaBrokerError;
81127
82709
  exports.extractGreeks = extractGreeks;
@@ -81224,6 +82806,7 @@ exports.groupOrdersByStatus = groupOrdersByStatus;
81224
82806
  exports.groupOrdersBySymbol = groupOrdersBySymbol;
81225
82807
  exports.guardSnapshots = guardSnapshots;
81226
82808
  exports.hasActiveTrailingStop = hasActiveTrailingStop;
82809
+ exports.hasDuplicateHeadroom = hasDuplicateHeadroom;
81227
82810
  exports.hasOptionLiquidity = hasGoodLiquidity;
81228
82811
  exports.hasStockLiquidity = hasGoodLiquidity$1;
81229
82812
  exports.hasSufficientVolume = hasSufficientVolume;
@@ -81242,6 +82825,7 @@ exports.isOrderFilled = isOrderFilled;
81242
82825
  exports.isOrderOpen = isOrderOpen;
81243
82826
  exports.isOrderTerminalStatus = isOrderTerminal$1;
81244
82827
  exports.isPendingCancelRejection = isPendingCancelRejection;
82828
+ exports.isSameReportedModel = isSameReportedModel;
81245
82829
  exports.isSupportedCryptoPair = isSupportedCryptoPair;
81246
82830
  exports.isTransientNetworkError = isTransientNetworkError;
81247
82831
  exports.legBudgetMs = legBudgetMs;
@@ -81252,6 +82836,9 @@ exports.limitsInventory = limitsInventory;
81252
82836
  exports.listAliases = listAliases;
81253
82837
  exports.llmAliases = llmAliases;
81254
82838
  exports.llmBreakers = llmBreakers;
82839
+ exports.llmLatencyTracker = llmLatencyTracker;
82840
+ exports.modelClassOf = modelClassOf;
82841
+ exports.modelClassRelationOf = modelClassRelationOf;
81255
82842
  exports.normaliseAnthropicStream = normaliseAnthropicStream;
81256
82843
  exports.normaliseOpenAiStream = normaliseOpenAiStream;
81257
82844
  exports.normaliseParams = normaliseParams;
@@ -81266,11 +82853,13 @@ exports.parseOCCSymbol = parseOCCSymbol;
81266
82853
  exports.protectLongPosition = protectLongPosition;
81267
82854
  exports.protectShortPosition = protectShortPosition;
81268
82855
  exports.rateLimiters = rateLimiters$1;
82856
+ exports.readTrailUnit = readTrailUnit;
81269
82857
  exports.resetLogger = resetLogger;
81270
82858
  exports.resetProviderGuards = resetProviderGuards;
81271
82859
  exports.resetRiskFreeRateCache = resetRiskFreeRateCache;
81272
82860
  exports.resolveChain = resolveChain;
81273
82861
  exports.resolveDefaultDirectCaller = resolveDefaultDirectCaller;
82862
+ exports.resolveReplaceTrail = resolveReplaceTrail;
81274
82863
  exports.risk = riskNs;
81275
82864
  exports.rollOptionPosition = rollOptionPosition;
81276
82865
  exports.roundPriceForAlpaca = roundPriceForAlpaca$3;
@@ -81285,12 +82874,14 @@ exports.sellAllCrypto = sellAllCrypto;
81285
82874
  exports.sellCryptoNotional = sellCryptoNotional;
81286
82875
  exports.sellToClose = sellToClose;
81287
82876
  exports.sellToOpen = sellToOpen;
82877
+ exports.servedModelOf = servedModelOf;
81288
82878
  exports.setLogger = setLogger;
81289
82879
  exports.setRiskFreeRate = setRiskFreeRate;
81290
82880
  exports.shortWithStopLoss = shortWithStopLoss;
81291
82881
  exports.sortOrdersByDate = sortOrdersByDate;
81292
82882
  exports.strategy = strategyNs;
81293
82883
  exports.sumUsage = sumUsage;
82884
+ exports.tailLatencyViolations = tailLatencyViolations;
81294
82885
  exports.tradingPolicy = index;
81295
82886
  exports.trailingStops = trailingStops;
81296
82887
  exports.unavailableStatistic = unavailableStatistic;