@adaptic/utils 0.0.1038 → 0.0.1039

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/index.cjs +1780 -341
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.mjs +1764 -342
  4. package/dist/index.mjs.map +1 -1
  5. package/dist/types/__tests__/llm/client/support/legacy-scenarios.d.ts +48 -0
  6. package/dist/types/__tests__/llm/client/support/legacy-scenarios.d.ts.map +1 -0
  7. package/dist/types/__tests__/llm/client/support/transports.d.ts +18 -0
  8. package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -1
  9. package/dist/types/alpaca/index.d.ts.map +1 -1
  10. package/dist/types/alpaca/trading/index.d.ts +1 -0
  11. package/dist/types/alpaca/trading/index.d.ts.map +1 -1
  12. package/dist/types/alpaca/trading/trail-limits.d.ts +8 -0
  13. package/dist/types/alpaca/trading/trail-limits.d.ts.map +1 -0
  14. package/dist/types/alpaca/trading/trail-unit.d.ts +98 -0
  15. package/dist/types/alpaca/trading/trail-unit.d.ts.map +1 -0
  16. package/dist/types/alpaca/trading/trailing-stops.d.ts +28 -15
  17. package/dist/types/alpaca/trading/trailing-stops.d.ts.map +1 -1
  18. package/dist/types/alpaca-trading-api.d.ts.map +1 -1
  19. package/dist/types/index.d.ts.map +1 -1
  20. package/dist/types/llm/alias-client.d.ts +10 -1
  21. package/dist/types/llm/alias-client.d.ts.map +1 -1
  22. package/dist/types/llm/circuit-breaker.d.ts +64 -2
  23. package/dist/types/llm/circuit-breaker.d.ts.map +1 -1
  24. package/dist/types/llm/fallback-chain.d.ts +109 -23
  25. package/dist/types/llm/fallback-chain.d.ts.map +1 -1
  26. package/dist/types/llm/hedge.d.ts +107 -0
  27. package/dist/types/llm/hedge.d.ts.map +1 -0
  28. package/dist/types/llm/index.d.ts +9 -6
  29. package/dist/types/llm/index.d.ts.map +1 -1
  30. package/dist/types/llm/leg-attempt.d.ts +165 -0
  31. package/dist/types/llm/leg-attempt.d.ts.map +1 -0
  32. package/dist/types/llm/leg-latency-tracker.d.ts +126 -0
  33. package/dist/types/llm/leg-latency-tracker.d.ts.map +1 -0
  34. package/dist/types/llm/rate-guard.d.ts +17 -0
  35. package/dist/types/llm/rate-guard.d.ts.map +1 -1
  36. package/dist/types/llm/route-table.d.ts +17 -1
  37. package/dist/types/llm/route-table.d.ts.map +1 -1
  38. package/dist/types/llm/transports/gateway.d.ts +27 -2
  39. package/dist/types/llm/transports/gateway.d.ts.map +1 -1
  40. package/dist/types/llm/types.d.ts +161 -1
  41. package/dist/types/llm/types.d.ts.map +1 -1
  42. package/dist/types/schemas/alpaca-schemas.d.ts +16 -16
  43. package/dist/types/schemas/massive-schemas.d.ts +10 -10
  44. package/dist/types/trading-policy/schemas/effective-policy.schema.d.ts +6 -6
  45. package/dist/types/trading-policy/schemas/policy-mutation.schema.d.ts +12 -12
  46. package/dist/types/trading-policy/schemas/signal-consumption-prefs.schema.d.ts +8 -8
  47. package/package.json +1 -1
package/dist/index.mjs CHANGED
@@ -10068,6 +10068,160 @@ class AlpacaMarketDataAPI extends EventEmitter {
10068
10068
  // Export the singleton instance
10069
10069
  const marketDataAPI = AlpacaMarketDataAPI.getInstance();
10070
10070
 
10071
+ /**
10072
+ * Alpaca's hard upper limit for `trail_percent` on trailing-stop orders.
10073
+ * Submissions exceeding this value are rejected with HTTP 422 / code 42210000
10074
+ * ("trail_percent must be <= 25"). See:
10075
+ * https://docs.alpaca.markets/reference/postorder
10076
+ */
10077
+ const ALPACA_MAX_TRAIL_PERCENT = 25;
10078
+
10079
+ /**
10080
+ * Trailing-stop replace-unit contract.
10081
+ *
10082
+ * Alpaca's order replace (`PATCH /v2/orders/{id}`) takes a single unitless
10083
+ * `trail` field. The broker reads it in the unit of the ORIGINAL order: a
10084
+ * `trail_percent` order reads `trail` as a percent, a `trail_price` order
10085
+ * reads it as dollars. A replace cannot change the unit. A caller that holds
10086
+ * a distance in one unit must therefore resolve it against the resting
10087
+ * order's unit before the replace, or the broker stores the number in the
10088
+ * other unit: a $22 dollar distance becomes a 22% trail.
10089
+ *
10090
+ * This module is the pure resolution step. It never guesses a unit, never
10091
+ * defaults a reference price, and refuses rather than send a value the broker
10092
+ * would reject or treat as effectively zero.
10093
+ */
10094
+ /**
10095
+ * Smallest `trail_percent` this module will send. Below it the trail is a
10096
+ * rounding artefact of the broker's tick grid rather than a protective
10097
+ * distance, so a conversion landing under it is refused.
10098
+ */
10099
+ const MIN_CONVERTED_TRAIL_PERCENT = 0.1;
10100
+ /** Percent values are sent to the broker at hundredths of a percent. */
10101
+ const PERCENT_DECIMALS_SCALE = 100;
10102
+ /** A ratio expressed in percent. */
10103
+ const PERCENT_PER_UNIT = 100;
10104
+ /**
10105
+ * Thrown when a trail replace cannot be expressed in the resting order's unit
10106
+ * without guessing. No replace is sent; the resting stop keeps protecting.
10107
+ */
10108
+ class TrailUnitConversionRefusedError extends AdapticUtilsError {
10109
+ /** The order the replace targeted. */
10110
+ orderId;
10111
+ /** Why the replace was refused. */
10112
+ reason;
10113
+ /** The converted percent, or `null` when no conversion was computed. */
10114
+ pct;
10115
+ /** The conversion reference price, or `null` when none was usable. */
10116
+ ref;
10117
+ constructor(params) {
10118
+ super(`Trailing stop replace refused for ${params.orderId} (${params.reason}): ${params.detail}`, "TRAIL_UNIT_REFUSED", "alpaca", false);
10119
+ this.orderId = params.orderId;
10120
+ this.reason = params.reason;
10121
+ this.pct = params.pct;
10122
+ this.ref = params.ref;
10123
+ }
10124
+ }
10125
+ /** Parse a broker decimal string; `null` when absent, non-finite or not positive. */
10126
+ function positiveOrNull(value) {
10127
+ if (value === null || value === undefined || value === "") {
10128
+ return null;
10129
+ }
10130
+ const parsed = Number(value);
10131
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : null;
10132
+ }
10133
+ /**
10134
+ * Read the unit a resting trailing-stop order trails in.
10135
+ *
10136
+ * @param order - The resting order as returned by the broker.
10137
+ * @returns `"price"` for a dollar trail, `"percent"` for a percent trail, or
10138
+ * `null` when neither (or both) unit fields carry a positive value.
10139
+ */
10140
+ function readTrailUnit(order) {
10141
+ const hasPrice = positiveOrNull(order.trail_price) !== null;
10142
+ const hasPercent = positiveOrNull(order.trail_percent) !== null;
10143
+ if (hasPrice === hasPercent) {
10144
+ return null;
10145
+ }
10146
+ return hasPrice ? "price" : "percent";
10147
+ }
10148
+ /**
10149
+ * Resolve the replace `trail` value for a requested trail against the resting
10150
+ * order's unit.
10151
+ *
10152
+ * - A dollar distance on a dollar order is sent unchanged.
10153
+ * - A dollar distance on a percent order is converted against
10154
+ * `ref = max(hwm, stop_price)`. For a long the HWM is at or above the live
10155
+ * price; for a short the stop is above it. So `ref` is at or above live on
10156
+ * both sides, and `distance / ref` is at or below `distance / live`: the
10157
+ * resulting stop is at or tighter than `live ∓ distance`. The percent is
10158
+ * rounded down to hundredths, which only tightens it further.
10159
+ * - A percent on a percent order is sent unchanged.
10160
+ * - A percent on a dollar order is refused. There is no conversion that keeps
10161
+ * the caller's intent without a live price this seam does not own.
10162
+ *
10163
+ * @param orderId - The order being replaced (for the refusal record).
10164
+ * @param order - The resting order as returned by the broker.
10165
+ * @param requested - Exactly one of a positive dollar distance or percent.
10166
+ * @returns The resolved `trail` value and its unit.
10167
+ * @throws {TrailUnitConversionRefusedError} When the unit is unknown, no
10168
+ * finite reference exists, the converted percent falls outside
10169
+ * [{@link MIN_CONVERTED_TRAIL_PERCENT}, {@link ALPACA_MAX_TRAIL_PERCENT}],
10170
+ * or a percent targets a dollar order.
10171
+ */
10172
+ function resolveReplaceTrail(orderId, order, requested) {
10173
+ const unit = readTrailUnit(order);
10174
+ if (unit === null) {
10175
+ throw new TrailUnitConversionRefusedError({
10176
+ orderId,
10177
+ reason: "unit_unknown",
10178
+ pct: null,
10179
+ ref: null,
10180
+ detail: `order carries trail_price=${String(order.trail_price)} trail_percent=${String(order.trail_percent)}; the replace unit cannot be determined`,
10181
+ });
10182
+ }
10183
+ if ("trailPercent" in requested) {
10184
+ if (unit === "price") {
10185
+ throw new TrailUnitConversionRefusedError({
10186
+ orderId,
10187
+ reason: "percent_on_price_order",
10188
+ pct: requested.trailPercent,
10189
+ ref: null,
10190
+ detail: `a ${requested.trailPercent}% trail sent to a dollar-trail order would be stored as $${requested.trailPercent}`,
10191
+ });
10192
+ }
10193
+ return { trail: requested.trailPercent.toString(), unit, referencePrice: null };
10194
+ }
10195
+ if (unit === "price") {
10196
+ return { trail: requested.trailPrice.toString(), unit, referencePrice: null };
10197
+ }
10198
+ const candidates = [positiveOrNull(order.hwm), positiveOrNull(order.stop_price)].filter((value) => value !== null);
10199
+ if (candidates.length === 0) {
10200
+ throw new TrailUnitConversionRefusedError({
10201
+ orderId,
10202
+ reason: "reference_unavailable",
10203
+ pct: null,
10204
+ ref: null,
10205
+ detail: `percent-trail order has no finite hwm (${String(order.hwm)}) or stop_price (${String(order.stop_price)}) to convert $${requested.trailPrice} against`,
10206
+ });
10207
+ }
10208
+ const ref = Math.max(...candidates);
10209
+ const pct = Math.floor((requested.trailPrice / ref) * PERCENT_PER_UNIT * PERCENT_DECIMALS_SCALE) /
10210
+ PERCENT_DECIMALS_SCALE;
10211
+ if (!Number.isFinite(pct) ||
10212
+ pct < MIN_CONVERTED_TRAIL_PERCENT ||
10213
+ pct > ALPACA_MAX_TRAIL_PERCENT) {
10214
+ throw new TrailUnitConversionRefusedError({
10215
+ orderId,
10216
+ reason: "converted_percent_out_of_range",
10217
+ pct,
10218
+ ref,
10219
+ detail: `$${requested.trailPrice} against ref ${ref} is ${pct}%, outside [${MIN_CONVERTED_TRAIL_PERCENT}, ${ALPACA_MAX_TRAIL_PERCENT}]`,
10220
+ });
10221
+ }
10222
+ return { trail: pct.toFixed(2), unit, referencePrice: ref };
10223
+ }
10224
+
10071
10225
  const limitPriceSlippagePercent100 = 0.1; // 0.1%
10072
10226
  /**
10073
10227
  * Alpaca's maximum page size for GET /orders — also our explicit default.
@@ -11017,12 +11171,18 @@ class AlpacaTradingAPI {
11017
11171
  return null;
11018
11172
  }
11019
11173
  const originalOrderId = trailingStopOrder.id;
11174
+ // Alpaca reads the replace `trail` in the resting order's unit. A percent
11175
+ // sent to a dollar-trail order would be stored as dollars, so it is refused
11176
+ // (typed, no replace sent) and the resting dollar trail keeps protecting.
11177
+ const resolvedTrail = resolveReplaceTrail(originalOrderId, trailingStopOrder, {
11178
+ trailPercent: trailPercent100,
11179
+ });
11020
11180
  this.log(`Updating trailing stop for ${symbol} from ${currentTrailPercent}% to ${trailPercent100}% (orderId=${originalOrderId})`, {
11021
11181
  symbol,
11022
11182
  });
11023
11183
  try {
11024
11184
  const updatedOrder = await this.makeRequest(`/orders/${trailingStopOrder.id}`, "PATCH", {
11025
- trail: trailPercent100.toString(),
11185
+ trail: resolvedTrail.trail,
11026
11186
  });
11027
11187
  // Log the replacement: Alpaca replaces orders on PATCH, so new ID is returned
11028
11188
  this.log(`Trailing stop updated for ${symbol}: newOrderId=${updatedOrder.id}, replaces=${updatedOrder.replaces || originalOrderId}`, { symbol });
@@ -63438,13 +63598,6 @@ var orderUtils$1 = /*#__PURE__*/Object.freeze({
63438
63598
  });
63439
63599
 
63440
63600
  const LOG_SOURCE$7 = "TrailingStops";
63441
- /**
63442
- * Alpaca's hard upper limit for `trail_percent` on trailing-stop orders.
63443
- * Submissions exceeding this value are rejected with HTTP 422 / code 42210000
63444
- * ("trail_percent must be <= 25"). See:
63445
- * https://docs.alpaca.markets/reference/postorder
63446
- */
63447
- const ALPACA_MAX_TRAIL_PERCENT = 25;
63448
63601
  /**
63449
63602
  * Internal logging helper with consistent source
63450
63603
  */
@@ -63573,23 +63726,42 @@ async function createTrailingStop(client, params) {
63573
63726
  }
63574
63727
  }
63575
63728
  /**
63576
- * Update an existing trailing stop order
63729
+ * Update the trail distance of an existing trailing stop order.
63730
+ *
63731
+ * ## Unit contract
63732
+ *
63733
+ * Alpaca's replace takes a single unitless `trail` field and reads it in the
63734
+ * unit of the ORIGINAL order; a replace cannot change the unit. This function
63735
+ * therefore reads the resting order first and resolves the request against its
63736
+ * unit ({@link resolveReplaceTrail}):
63577
63737
  *
63578
- * You can update the trail_percent or trail_price of an existing order.
63579
- * Note: Alpaca uses 'trail' parameter for replacements (works for both percent and price).
63738
+ * - `trailPrice` on a dollar-trail order is sent as dollars, unchanged.
63739
+ * - `trailPrice` on a percent-trail order is converted to a percent against
63740
+ * `max(hwm, stop_price)`, which is at or above the live price on both sides,
63741
+ * so the resulting stop is at or tighter than `live ∓ trailPrice`. The
63742
+ * percent is rounded down to hundredths (tighter).
63743
+ * - `trailPercent` on a percent-trail order is sent unchanged.
63744
+ * - `trailPercent` on a dollar-trail order is refused.
63745
+ *
63746
+ * A refusal throws {@link TrailUnitConversionRefusedError} and sends no
63747
+ * replace, so the resting stop keeps protecting. The unit is never guessed and
63748
+ * the conversion reference is never defaulted.
63580
63749
  *
63581
63750
  * @param client - AlpacaClient instance
63582
63751
  * @param orderId - The ID of the order to update
63583
- * @param updates - New trail parameters (specify one of trailPercent or trailPrice)
63584
- * @returns The updated order
63585
- * @throws {Error} If no update parameters provided or update fails
63752
+ * @param updates - New trail parameters (specify exactly one of trailPercent or trailPrice)
63753
+ * @returns The replacement order
63754
+ * @throws {Error} If no/both update parameters are given, or a value is not positive
63755
+ * @throws {TrailUnitConversionRefusedError} If the request cannot be expressed
63756
+ * in the resting order's unit
63757
+ * @throws {AlpacaApiError} If the order read or the replace fails at the broker
63586
63758
  *
63587
63759
  * @example
63588
63760
  * ```typescript
63589
- * // Tighten trailing stop to 1.5%
63761
+ * // Tighten a percent trailing stop to 1.5%
63590
63762
  * await updateTrailingStop(client, 'order-id-123', { trailPercent: 1.5 });
63591
63763
  *
63592
- * // Change to $3 trail
63764
+ * // Pin a $3 trail distance (converted when the order trails in percent)
63593
63765
  * await updateTrailingStop(client, 'order-id-123', { trailPrice: 3.00 });
63594
63766
  * ```
63595
63767
  */
@@ -63609,21 +63781,41 @@ async function updateTrailingStop(client, orderId, updates) {
63609
63781
  throw new Error("trailPrice must be greater than 0");
63610
63782
  }
63611
63783
  const sdk = client.getSDK();
63612
- const updateDescription = updates.trailPercent
63613
- ? `${updates.trailPercent}%`
63614
- : `$${updates.trailPrice?.toFixed(2)}`;
63784
+ const requested = updates.trailPercent !== undefined
63785
+ ? { trailPercent: updates.trailPercent }
63786
+ : { trailPrice: updates.trailPrice };
63787
+ const updateDescription = "trailPercent" in requested
63788
+ ? `${requested.trailPercent}%`
63789
+ : `$${requested.trailPrice.toFixed(2)}`;
63615
63790
  log$g(`Updating trailing stop ${orderId} to trail: ${updateDescription}`, {
63616
63791
  type: "info",
63617
63792
  });
63793
+ let resting;
63618
63794
  try {
63619
- const replaceParams = {};
63620
- // Alpaca's replaceOrder uses 'trail' for both percent and price updates
63621
- if (updates.trailPercent !== undefined) {
63622
- replaceParams.trail = updates.trailPercent.toString();
63623
- }
63624
- else if (updates.trailPrice !== undefined) {
63625
- replaceParams.trail = updates.trailPrice.toString();
63626
- }
63795
+ resting = (await sdk.getOrder(orderId));
63796
+ }
63797
+ catch (error) {
63798
+ const err = error;
63799
+ log$g(`Trailing stop update aborted for ${orderId}: order read failed: ${err.message}`, {
63800
+ type: "error",
63801
+ });
63802
+ throw enrichAlpacaError(new Error(`Failed to read trailing stop ${orderId} before update: ${err.message}`), error);
63803
+ }
63804
+ let resolved;
63805
+ try {
63806
+ resolved = resolveReplaceTrail(orderId, resting, requested);
63807
+ }
63808
+ catch (refusal) {
63809
+ log$g(`Trailing stop update refused for ${orderId}: ${refusal.message}`, {
63810
+ type: "warn",
63811
+ });
63812
+ throw refusal;
63813
+ }
63814
+ if (resolved.referencePrice !== null) {
63815
+ log$g(`Trailing stop ${orderId}: ${updateDescription} on percent order → ${resolved.trail}% (ref ${resolved.referencePrice})`, { type: "info" });
63816
+ }
63817
+ try {
63818
+ const replaceParams = { trail: resolved.trail };
63627
63819
  const order = await sdk.replaceOrder(orderId, replaceParams);
63628
63820
  log$g(`Trailing stop updated: orderId=${order.id}, new replacement created`, {
63629
63821
  type: "info",
@@ -72328,6 +72520,192 @@ const alpaca = {
72328
72520
  streams: streams$1,
72329
72521
  };
72330
72522
 
72523
+ /**
72524
+ * Rolling healthy-latency evidence per (provider, model), split by prompt size.
72525
+ *
72526
+ * The chain's timeouts and hedge points are only as good as its idea of how
72527
+ * long a healthy answer takes. A flat route budget encodes no such idea: it
72528
+ * treats a thirty-second wait on a model whose healthy answers arrive in four
72529
+ * the same as a thirty-second wait on one that needs twenty-five. This tracker
72530
+ * supplies the measured alternative.
72531
+ *
72532
+ * It is keyed by provider and model rather than by alias and role, because
72533
+ * health is a property of the model at its host: the same model reached as one
72534
+ * alias's primary and another alias's secondary is one population, and
72535
+ * splitting it would halve the evidence each side sees. Samples are split by
72536
+ * prompt size, because generation time grows with input and a model that
72537
+ * struggles on large prompts would otherwise have its large-prompt tail hidden
72538
+ * by a crowd of small, fast calls.
72539
+ *
72540
+ * Only healthy (answered) attempts are recorded. A timeout says the answer
72541
+ * took at least the budget, not how long it took, and folding it in would pull
72542
+ * the quantiles toward whatever budget happened to be configured.
72543
+ *
72544
+ * Unknown stays unknown: a cell with fewer than the minimum samples answers
72545
+ * `null`, and every consumer treats `null` as "no evidence", never as a value.
72546
+ *
72547
+ * @module llm/leg-latency-tracker
72548
+ */
72549
+ /** Characters per token used to size a prompt before the provider has counted it. */
72550
+ const CHARS_PER_TOKEN = 4;
72551
+ /** The bucket used when a prompt's size could not be estimated. */
72552
+ const UNKNOWN_BUCKET = "unknown";
72553
+ /**
72554
+ * Estimate a prompt's size in tokens from its serialised length.
72555
+ *
72556
+ * Only used to choose a size bucket, and the same estimate is applied when a
72557
+ * sample is recorded and when it is looked up, so its bias cancels. Content
72558
+ * that cannot be serialised yields `null` rather than a guessed size.
72559
+ *
72560
+ * @param parts The prompt, developer instruction and prior turns.
72561
+ * @returns The estimated token count, or null.
72562
+ */
72563
+ function estimatePromptTokens(parts) {
72564
+ let characters = 0;
72565
+ for (const part of parts) {
72566
+ if (part === undefined) {
72567
+ continue;
72568
+ }
72569
+ if (typeof part === "string") {
72570
+ characters += part.length;
72571
+ continue;
72572
+ }
72573
+ try {
72574
+ const serialised = JSON.stringify(part);
72575
+ if (typeof serialised !== "string") {
72576
+ return null;
72577
+ }
72578
+ characters += serialised.length;
72579
+ }
72580
+ catch {
72581
+ // A circular or otherwise unserialisable part has no knowable size; the
72582
+ // caller files its latency under the unknown bucket rather than a guess.
72583
+ return null;
72584
+ }
72585
+ }
72586
+ return Math.ceil(characters / CHARS_PER_TOKEN);
72587
+ }
72588
+ /**
72589
+ * The nearest-rank quantile of a sorted sample.
72590
+ *
72591
+ * @param sorted Ascending values; must be non-empty.
72592
+ * @param q The quantile, in (0, 1).
72593
+ * @returns The value at that rank.
72594
+ */
72595
+ function nearestRank(sorted, q) {
72596
+ const rank = Math.min(sorted.length, Math.max(1, Math.ceil(q * sorted.length)));
72597
+ return sorted[rank - 1];
72598
+ }
72599
+ /**
72600
+ * Per-process healthy-latency windows.
72601
+ */
72602
+ class LegLatencyTracker {
72603
+ cells = new Map();
72604
+ config;
72605
+ now;
72606
+ /**
72607
+ * @param config Window sizing and prompt-size buckets.
72608
+ * @param now Clock, injected so ageing is testable without waiting.
72609
+ */
72610
+ constructor(config, now = Date.now) {
72611
+ this.config = config;
72612
+ this.now = now;
72613
+ }
72614
+ /**
72615
+ * The size bucket a prompt falls in.
72616
+ *
72617
+ * @param promptTokens Estimated prompt tokens, or null when unknown.
72618
+ * @returns The bucket label.
72619
+ */
72620
+ bucketOf(promptTokens) {
72621
+ if (promptTokens === null || !Number.isFinite(promptTokens)) {
72622
+ return UNKNOWN_BUCKET;
72623
+ }
72624
+ const index = this.config.promptTokenBuckets.findIndex((edge) => promptTokens < edge);
72625
+ return String(index === -1 ? this.config.promptTokenBuckets.length : index);
72626
+ }
72627
+ /**
72628
+ * Record one healthy (answered) attempt.
72629
+ *
72630
+ * @param provider The provider that answered.
72631
+ * @param modelId The model it answered with.
72632
+ * @param promptTokens Estimated prompt tokens, or null.
72633
+ * @param durationMs How long the answer took.
72634
+ * @returns void
72635
+ */
72636
+ record(provider, modelId, promptTokens, durationMs) {
72637
+ if (!Number.isFinite(durationMs) || durationMs < 0) {
72638
+ return;
72639
+ }
72640
+ const key = this.cellKey(provider, modelId, promptTokens);
72641
+ const samples = this.fresh(key);
72642
+ samples.push({ atMs: this.now(), durationMs });
72643
+ while (samples.length > this.config.windowSize) {
72644
+ samples.shift();
72645
+ }
72646
+ this.cells.set(key, samples);
72647
+ }
72648
+ /**
72649
+ * A healthy-latency quantile, or null without enough evidence.
72650
+ *
72651
+ * @param provider The provider.
72652
+ * @param modelId The model.
72653
+ * @param promptTokens Estimated prompt tokens, or null.
72654
+ * @param q The quantile, in (0, 1).
72655
+ * @returns The quantile in milliseconds, or null.
72656
+ */
72657
+ quantile(provider, modelId, promptTokens, q) {
72658
+ const samples = this.fresh(this.cellKey(provider, modelId, promptTokens));
72659
+ if (samples.length < this.config.minSamples) {
72660
+ return null;
72661
+ }
72662
+ const sorted = samples.map((sample) => sample.durationMs).sort((a, b) => a - b);
72663
+ return nearestRank(sorted, q);
72664
+ }
72665
+ /**
72666
+ * How many fresh samples a cell holds.
72667
+ *
72668
+ * @param provider The provider.
72669
+ * @param modelId The model.
72670
+ * @param promptTokens Estimated prompt tokens, or null.
72671
+ * @returns The count.
72672
+ */
72673
+ sampleCount(provider, modelId, promptTokens) {
72674
+ return this.fresh(this.cellKey(provider, modelId, promptTokens)).length;
72675
+ }
72676
+ /**
72677
+ * Discard all evidence.
72678
+ *
72679
+ * @returns void
72680
+ */
72681
+ reset() {
72682
+ this.cells.clear();
72683
+ }
72684
+ /**
72685
+ * @param provider The provider.
72686
+ * @param modelId The model.
72687
+ * @param promptTokens Estimated prompt tokens, or null.
72688
+ * @returns The cell key.
72689
+ */
72690
+ cellKey(provider, modelId, promptTokens) {
72691
+ return `${provider}/${modelId}@${this.bucketOf(promptTokens)}`;
72692
+ }
72693
+ /**
72694
+ * A cell's samples with the stale ones removed.
72695
+ *
72696
+ * @param key The cell key.
72697
+ * @returns The fresh samples (the stored array, pruned in place).
72698
+ */
72699
+ fresh(key) {
72700
+ const samples = this.cells.get(key) ?? [];
72701
+ const oldest = this.now() - this.config.sampleMaxAgeMs;
72702
+ while (samples.length > 0 && samples[0].atMs < oldest) {
72703
+ samples.shift();
72704
+ }
72705
+ return samples;
72706
+ }
72707
+ }
72708
+
72331
72709
  /**
72332
72710
  * Per-route circuit breaker for the alias client.
72333
72711
  *
@@ -72357,6 +72735,17 @@ const alpaca = {
72357
72735
  * for the full `cooldown_ms`. Either way the route then admits a bounded number
72358
72736
  * of half-open probes, and one success closes it.
72359
72737
  *
72738
+ * A route can also be opened by LATENCY (when `latency_trip` is armed): a
72739
+ * provider whose answers arrive, but later than the latency class's objective
72740
+ * in most recent windows, is failing a hot path without ever producing an
72741
+ * error. A latency-opened breaker is not closed by an answer from an attempt
72742
+ * that started before it opened, because such an answer is exactly the slow
72743
+ * evidence that opened it.
72744
+ *
72745
+ * The half-open probe budget scales with how much traffic the route carried
72746
+ * before it opened (`probe_fraction`), so a route that served forty calls at
72747
+ * once is not re-tested by a single probe whose one slow answer decides it.
72748
+ *
72360
72749
  * The clock is injected. Breaker behaviour is entirely about elapsed time, and
72361
72750
  * a test that must sleep to observe a cooldown is a test nobody runs.
72362
72751
  *
@@ -72371,6 +72760,8 @@ function freshRecord() {
72371
72760
  openedAtMs: null,
72372
72761
  probesInFlight: 0,
72373
72762
  runHasHardFailure: false,
72763
+ openedByLatency: false,
72764
+ concurrencyAtOpen: 0,
72374
72765
  };
72375
72766
  }
72376
72767
  /**
@@ -72378,6 +72769,11 @@ function freshRecord() {
72378
72769
  */
72379
72770
  class CircuitBreakerRegistry {
72380
72771
  records = new Map();
72772
+ latency = new Map();
72773
+ /** Attempts currently in flight per route, for scaling the probe budget. */
72774
+ inFlight = new Map();
72775
+ /** Peak of {@link inFlight} since the route last opened. */
72776
+ peakInFlight = new Map();
72381
72777
  config;
72382
72778
  now;
72383
72779
  /**
@@ -72446,7 +72842,20 @@ class CircuitBreakerRegistry {
72446
72842
  return false;
72447
72843
  }
72448
72844
  const record = this.recordFor(routeKey);
72449
- return record.probesInFlight < this.config.half_open_probes;
72845
+ return record.probesInFlight < this.probeBudgetFor(record);
72846
+ }
72847
+ /**
72848
+ * How many half-open probes a route admits at once.
72849
+ *
72850
+ * @param record The route's record.
72851
+ * @returns The larger of the configured floor and the concurrency-scaled budget.
72852
+ */
72853
+ probeBudgetFor(record) {
72854
+ const fraction = this.config.probe_fraction;
72855
+ if (fraction === undefined || fraction <= 0) {
72856
+ return this.config.half_open_probes;
72857
+ }
72858
+ return Math.max(this.config.half_open_probes, Math.ceil(fraction * record.concurrencyAtOpen));
72450
72859
  }
72451
72860
  /**
72452
72861
  * Register that an attempt is starting, so half-open probes stay bounded.
@@ -72457,6 +72866,9 @@ class CircuitBreakerRegistry {
72457
72866
  * {@link onAttemptAbandoned}, or the slot is never returned.
72458
72867
  */
72459
72868
  onAttemptStart(routeKey) {
72869
+ const inFlight = (this.inFlight.get(routeKey) ?? 0) + 1;
72870
+ this.inFlight.set(routeKey, inFlight);
72871
+ this.peakInFlight.set(routeKey, Math.max(this.peakInFlight.get(routeKey) ?? 0, inFlight));
72460
72872
  if (this.stateOf(routeKey) === "half-open") {
72461
72873
  this.recordFor(routeKey).probesInFlight += 1;
72462
72874
  return true;
@@ -72483,6 +72895,17 @@ class CircuitBreakerRegistry {
72483
72895
  record.probesInFlight -= 1;
72484
72896
  }
72485
72897
  }
72898
+ /**
72899
+ * Mark an attempt started with {@link onAttemptStart} as finished, whatever
72900
+ * its outcome, so the in-flight count that scales the probe budget stays true.
72901
+ *
72902
+ * @param routeKey The route's stable key.
72903
+ * @returns void
72904
+ */
72905
+ onAttemptEnd(routeKey) {
72906
+ const inFlight = this.inFlight.get(routeKey) ?? 0;
72907
+ this.inFlight.set(routeKey, Math.max(0, inFlight - 1));
72908
+ }
72486
72909
  /**
72487
72910
  * Record a success, closing the breaker.
72488
72911
  *
@@ -72491,12 +72914,69 @@ class CircuitBreakerRegistry {
72491
72914
  * working call answers it; requiring several would keep a recovered provider
72492
72915
  * excluded while the chain paid for slower legs.
72493
72916
  *
72917
+ * The one exception is a breaker opened by latency: an answer from an
72918
+ * attempt that started before it opened is the slow evidence that opened it,
72919
+ * not evidence of recovery, and is ignored.
72920
+ *
72494
72921
  * @param routeKey The route's stable key.
72922
+ * @param startedAtMs When the answering attempt started, on the registry's clock.
72495
72923
  * @returns void
72496
72924
  */
72497
- onSuccess(routeKey) {
72925
+ onSuccess(routeKey, startedAtMs) {
72926
+ const record = this.records.get(routeKey);
72927
+ if (record !== undefined &&
72928
+ record.openedByLatency &&
72929
+ record.openedAtMs !== null &&
72930
+ startedAtMs !== undefined &&
72931
+ startedAtMs < record.openedAtMs) {
72932
+ return;
72933
+ }
72498
72934
  this.records.set(routeKey, freshRecord());
72499
72935
  }
72936
+ /**
72937
+ * Record how long an attempt that reached the provider took.
72938
+ *
72939
+ * Durations fill fixed-size windows; each full window is judged against the
72940
+ * latency class's objective at the configured quantile, and the breaker opens
72941
+ * when enough of the recent windows were over it. Does nothing unless the
72942
+ * latency trip is armed.
72943
+ *
72944
+ * @param routeKey The route's stable key.
72945
+ * @param durationMs The attempt's duration.
72946
+ * @param latencyClass The alias's latency class, which selects the objective.
72947
+ * @returns void
72948
+ */
72949
+ onLatencySample(routeKey, durationMs, latencyClass) {
72950
+ const trip = this.config.latency_trip;
72951
+ if (trip === undefined || !trip.enabled || latencyClass === undefined) {
72952
+ return;
72953
+ }
72954
+ if (!Number.isFinite(durationMs) || durationMs < 0) {
72955
+ return;
72956
+ }
72957
+ let state = this.latency.get(routeKey);
72958
+ if (state === undefined) {
72959
+ state = { samples: [], verdicts: [] };
72960
+ this.latency.set(routeKey, state);
72961
+ }
72962
+ state.samples.push(durationMs);
72963
+ if (state.samples.length < trip.window_size) {
72964
+ return;
72965
+ }
72966
+ const sorted = [...state.samples].sort((a, b) => a - b);
72967
+ state.samples = [];
72968
+ state.verdicts.push(nearestRank(sorted, trip.quantile) > trip.slo_ms[latencyClass]);
72969
+ while (state.verdicts.length > trip.of_windows) {
72970
+ state.verdicts.shift();
72971
+ }
72972
+ const over = state.verdicts.filter(Boolean).length;
72973
+ if (over >= trip.trip_windows && this.stateOf(routeKey) === "closed") {
72974
+ const record = this.recordFor(routeKey);
72975
+ this.open(routeKey, record);
72976
+ record.openedByLatency = true;
72977
+ state.verdicts = [];
72978
+ }
72979
+ }
72500
72980
  /**
72501
72981
  * Record a failure, opening the breaker once the threshold is reached.
72502
72982
  *
@@ -72518,9 +72998,22 @@ class CircuitBreakerRegistry {
72518
72998
  record.runHasHardFailure = true;
72519
72999
  }
72520
73000
  if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
72521
- record.openedAtMs = this.now();
73001
+ this.open(routeKey, record);
73002
+ record.openedByLatency = false;
72522
73003
  }
72523
73004
  }
73005
+ /**
73006
+ * Open a route, capturing the concurrency its probe budget scales with.
73007
+ *
73008
+ * @param routeKey The route's stable key.
73009
+ * @param record Its record.
73010
+ * @returns void
73011
+ */
73012
+ open(routeKey, record) {
73013
+ record.openedAtMs = this.now();
73014
+ record.concurrencyAtOpen = Math.max(record.concurrencyAtOpen, this.peakInFlight.get(routeKey) ?? 0);
73015
+ this.peakInFlight.set(routeKey, this.inFlight.get(routeKey) ?? 0);
73016
+ }
72524
73017
  /**
72525
73018
  * Inspect a route's breaker.
72526
73019
  *
@@ -72541,6 +73034,8 @@ class CircuitBreakerRegistry {
72541
73034
  ? "hard"
72542
73035
  : "capacity",
72543
73036
  cooldownMs: this.cooldownFor(record),
73037
+ openedByLatency: record.openedByLatency,
73038
+ probeBudget: this.probeBudgetFor(record),
72544
73039
  };
72545
73040
  }
72546
73041
  /**
@@ -72560,6 +73055,9 @@ class CircuitBreakerRegistry {
72560
73055
  */
72561
73056
  reset() {
72562
73057
  this.records.clear();
73058
+ this.latency.clear();
73059
+ this.inFlight.clear();
73060
+ this.peakInFlight.clear();
72563
73061
  }
72564
73062
  /**
72565
73063
  * @param routeKey The route's stable key.
@@ -73269,6 +73767,35 @@ async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
73269
73767
  release();
73270
73768
  }
73271
73769
  }
73770
+ /**
73771
+ * Whether a provider guard has room for a DUPLICATE attempt above a reserve.
73772
+ *
73773
+ * A duplicate is a hedge: a second request for an answer another attempt is
73774
+ * already fetching. It is only worth sending with capacity no first attempt
73775
+ * needs, so it is admitted only when nobody is queued behind either bound and
73776
+ * both the free concurrency (after the duplicate) and the rate tokens stay at
73777
+ * or above the reserved fraction of the guard's ceilings. A provider that is
73778
+ * already busy therefore never sees duplicates, which is when they would do
73779
+ * the most harm.
73780
+ *
73781
+ * @param provider The provider key.
73782
+ * @param modelId The model the duplicate would address.
73783
+ * @param reserveFraction Fraction of each ceiling kept free, in [0, 1).
73784
+ * @returns Whether the duplicate may start.
73785
+ */
73786
+ function hasDuplicateHeadroom(provider, modelId, reserveFraction) {
73787
+ const identity = guardIdentity(provider, modelId);
73788
+ const limits = limitsFor(provider, identity.modelId);
73789
+ const limiter = rateLimiterFor(identity);
73790
+ const gate = concurrencyGateFor(identity);
73791
+ if (limiter.getQueueLength() > 0 || gate.queueLength() > 0) {
73792
+ return false;
73793
+ }
73794
+ const freeAfter = limits.max_concurrent - gate.inFlightCount() - 1;
73795
+ const tokensAfter = limiter.getAvailableTokens() - TOKENS_PER_REQUEST;
73796
+ return (freeAfter >= Math.ceil(limits.max_concurrent * reserveFraction) &&
73797
+ tokensAfter >= Math.ceil(limits.requests_per_minute * reserveFraction));
73798
+ }
73272
73799
  /**
73273
73800
  * Inspect the guards currently in use.
73274
73801
  *
@@ -73411,6 +73938,703 @@ function parseStructuredContent(content, responseFormat, usage) {
73411
73938
  }
73412
73939
  }
73413
73940
 
73941
+ /**
73942
+ * One attempt against one leg: dispatch under a hard timeout, and the
73943
+ * classification of how it ended.
73944
+ *
73945
+ * Split from the chain walker so the same-model hedge runner and the walker
73946
+ * share exactly one definition of what a timeout, a capacity refusal, a
73947
+ * cancellation and a superseded attempt are. Two copies of that
73948
+ * classification would drift, and the breaker would then read the same event
73949
+ * differently depending on which code path produced it.
73950
+ *
73951
+ * @module llm/leg-attempt
73952
+ */
73953
+ /** Raised internally when an attempt exceeds its hard budget. */
73954
+ class LegTimeoutError extends Error {
73955
+ /**
73956
+ * @param routeKey The leg that timed out.
73957
+ * @param budgetMs Its budget in milliseconds.
73958
+ */
73959
+ constructor(routeKey, budgetMs) {
73960
+ super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
73961
+ this.name = "LegTimeoutError";
73962
+ }
73963
+ }
73964
+ /**
73965
+ * Raised internally when an attempt ran past its MEASURED timeout and was
73966
+ * replaced by another attempt on the same model.
73967
+ *
73968
+ * Not a verdict on the provider: the measured timeout is the chain's own
73969
+ * impatience, applied only because a same-model alternative could take over,
73970
+ * so the breaker learns nothing from it.
73971
+ */
73972
+ class AttemptSupersededError extends Error {
73973
+ /**
73974
+ * @param routeKey The attempt's leg.
73975
+ * @param afterMs How long it ran before it was replaced.
73976
+ */
73977
+ constructor(routeKey, afterMs) {
73978
+ super(`route ${routeKey} exceeded its measured ${afterMs} ms attempt timeout and was ` +
73979
+ "superseded by a same-model attempt");
73980
+ this.name = "AttemptSupersededError";
73981
+ }
73982
+ }
73983
+ /**
73984
+ * Raised internally on an attempt that was still running when another
73985
+ * attempt on the same model answered first. Not a verdict on the provider.
73986
+ */
73987
+ class HedgeLoserError extends Error {
73988
+ /**
73989
+ * @param routeKey The losing attempt's leg.
73990
+ */
73991
+ constructor(routeKey) {
73992
+ super(`route ${routeKey} was cancelled: a same-model attempt answered first`);
73993
+ this.name = "HedgeLoserError";
73994
+ }
73995
+ }
73996
+ /**
73997
+ * Start one attempt under a hard timeout, honouring the caller's own cancellation.
73998
+ *
73999
+ * The timer is always cleared and the abort listener always removed, including
74000
+ * on the success path. A long-lived process that leaked one timer per LLM call
74001
+ * would accumulate them at exactly the rate it does useful work.
74002
+ *
74003
+ * @param leg The leg to run.
74004
+ * @param params Normalised parameters for this leg.
74005
+ * @param request The call's request fields and cancellation.
74006
+ * @param budgetMs The attempt's hard budget.
74007
+ * @returns A handle on the attempt.
74008
+ */
74009
+ function startAttempt(leg, params, request, budgetMs) {
74010
+ const controller = new AbortController();
74011
+ let chainReason;
74012
+ let hardTimeout = false;
74013
+ const timer = setTimeout(() => {
74014
+ hardTimeout = true;
74015
+ controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
74016
+ }, budgetMs);
74017
+ const forwardAbort = () => {
74018
+ controller.abort(request.callerSignal?.reason);
74019
+ };
74020
+ if (request.callerSignal !== undefined) {
74021
+ if (request.callerSignal.aborted) {
74022
+ forwardAbort();
74023
+ }
74024
+ else {
74025
+ request.callerSignal.addEventListener("abort", forwardAbort, { once: true });
74026
+ }
74027
+ }
74028
+ const run = async () => {
74029
+ try {
74030
+ // The guards wrap the transport rather than the whole attempt, so the
74031
+ // hard timeout above still bounds the total wait: a caller queued behind
74032
+ // the rate limiter is spending its budget just as surely as one waiting
74033
+ // on the provider, and only one clock should govern both. The attempt's
74034
+ // own signal is handed to the guard as well, so an attempt whose budget
74035
+ // or caller is gone leaves the queue at once instead of holding its place.
74036
+ const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
74037
+ route: leg.route,
74038
+ content: request.content,
74039
+ responseFormat: request.responseFormat,
74040
+ params,
74041
+ developerPrompt: request.developerPrompt,
74042
+ context: request.context,
74043
+ signal: controller.signal,
74044
+ correlationId: request.correlationId,
74045
+ }), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
74046
+ assertToolChoiceHonoured(leg.route, params, response);
74047
+ return response;
74048
+ }
74049
+ finally {
74050
+ clearTimeout(timer);
74051
+ request.callerSignal?.removeEventListener("abort", forwardAbort);
74052
+ }
74053
+ };
74054
+ return {
74055
+ promise: run(),
74056
+ abort: (reason) => {
74057
+ if (chainReason === undefined && !controller.signal.aborted) {
74058
+ chainReason = reason;
74059
+ controller.abort(reason);
74060
+ }
74061
+ },
74062
+ abortReason: () => chainReason,
74063
+ timedOut: () => hardTimeout,
74064
+ };
74065
+ }
74066
+ /**
74067
+ * HTTP statuses a provider (or the gateway relaying it) uses to say it is full
74068
+ * rather than that the request or the route is wrong: request timeout, too
74069
+ * early, too many requests, service unavailable, and Anthropic's overloaded.
74070
+ */
74071
+ const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
74072
+ /**
74073
+ * Wording providers use for a capacity refusal when the status is lost on the
74074
+ * way (a relayed body, a client library's own error). DeepInfra's is
74075
+ * "Model busy, retry later"; Anthropic's is "Overloaded".
74076
+ */
74077
+ const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
74078
+ /**
74079
+ * Whether a failure is the provider saying it is full rather than broken.
74080
+ *
74081
+ * Read by shape rather than by class, because the same signal reaches the
74082
+ * chain from more than one transport and not every transport's error class is
74083
+ * importable here.
74084
+ *
74085
+ * @param error The thrown value.
74086
+ * @param reason Its message.
74087
+ * @returns Whether it is a capacity signal.
74088
+ */
74089
+ function isCapacitySignal(error, reason) {
74090
+ if (typeof error === "object" && error !== null) {
74091
+ const status = error.status;
74092
+ if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
74093
+ return true;
74094
+ }
74095
+ }
74096
+ return CAPACITY_WORDING.test(reason);
74097
+ }
74098
+ /**
74099
+ * Classify why a leg failed.
74100
+ *
74101
+ * The distinction matters to the breaker: a timeout and a 5xx are evidence the
74102
+ * provider is unhealthy, while the caller cancelling is not. Counting a
74103
+ * cancellation as a provider failure would let a burst of user-cancelled
74104
+ * requests open the breaker on a perfectly healthy route.
74105
+ *
74106
+ * Among failures that do count, a capacity signal (the provider said it is
74107
+ * busy, or the leg ran out its budget waiting on it) is told apart from a hard
74108
+ * failure so the breaker can re-admit a busy route sooner than a broken one. A
74109
+ * timeout is read as capacity: on a reachable provider it is what a full queue
74110
+ * looks like from outside, and a provider that is actually down still costs no
74111
+ * more than one probe per capacity cooldown.
74112
+ *
74113
+ * An attempt the chain itself cancelled — replaced after its measured
74114
+ * timeout, or beaten by a same-model attempt — is not a verdict on the
74115
+ * provider either, and is classified by the chain's reason rather than by
74116
+ * whatever the transport happened to throw on the way out.
74117
+ *
74118
+ * @param error The thrown value.
74119
+ * @param callerSignal The caller's cancellation signal, if any.
74120
+ * @returns The outcome and whether it counts against route health.
74121
+ */
74122
+ function classify(error, callerSignal) {
74123
+ if (callerSignal !== undefined && callerSignal.aborted) {
74124
+ return {
74125
+ outcome: "skipped",
74126
+ reason: "caller cancelled",
74127
+ countsAgainstHealth: false,
74128
+ failureKind: "hard",
74129
+ };
74130
+ }
74131
+ if (error instanceof AttemptSupersededError) {
74132
+ return {
74133
+ outcome: "timeout",
74134
+ reason: error.message,
74135
+ countsAgainstHealth: false,
74136
+ failureKind: "capacity",
74137
+ };
74138
+ }
74139
+ if (error instanceof HedgeLoserError) {
74140
+ return {
74141
+ outcome: "skipped",
74142
+ reason: error.message,
74143
+ countsAgainstHealth: false,
74144
+ failureKind: "capacity",
74145
+ };
74146
+ }
74147
+ if (error instanceof LegTimeoutError) {
74148
+ return {
74149
+ outcome: "timeout",
74150
+ reason: error.message,
74151
+ countsAgainstHealth: true,
74152
+ failureKind: "capacity",
74153
+ };
74154
+ }
74155
+ if (error instanceof UnsupportedCapabilityError) {
74156
+ return {
74157
+ outcome: "skipped",
74158
+ reason: error.message,
74159
+ countsAgainstHealth: false,
74160
+ failureKind: "hard",
74161
+ };
74162
+ }
74163
+ if (error instanceof ToolChoiceIgnoredError) {
74164
+ // The route answered; it broke a declared guarantee rather than failing to
74165
+ // be available, so its breaker is not charged for it.
74166
+ return {
74167
+ outcome: "error",
74168
+ reason: error.message,
74169
+ countsAgainstHealth: false,
74170
+ failureKind: "hard",
74171
+ };
74172
+ }
74173
+ // Self-inflicted pacing, not provider ill-health. Counting it would let the
74174
+ // client's own throttling open a breaker on a perfectly healthy provider and
74175
+ // permanently reroute traffic nobody chose to reroute.
74176
+ if (error instanceof RateGuardTimeoutError) {
74177
+ return {
74178
+ outcome: "skipped",
74179
+ reason: error.message,
74180
+ countsAgainstHealth: false,
74181
+ failureKind: "hard",
74182
+ };
74183
+ }
74184
+ const reason = error instanceof Error ? error.message : String(error);
74185
+ if (/abort/i.test(reason)) {
74186
+ return {
74187
+ outcome: "timeout",
74188
+ reason: `aborted: ${reason}`,
74189
+ countsAgainstHealth: true,
74190
+ failureKind: "capacity",
74191
+ };
74192
+ }
74193
+ if (error instanceof LlmResponseFormatError) {
74194
+ // The provider answered, badly. That is a route defect, not a full queue,
74195
+ // whatever words the unparseable content happens to contain.
74196
+ return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
74197
+ }
74198
+ return {
74199
+ outcome: "error",
74200
+ reason,
74201
+ countsAgainstHealth: true,
74202
+ failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
74203
+ };
74204
+ }
74205
+ /**
74206
+ * Whether the caller has stopped waiting.
74207
+ *
74208
+ * Read through a function rather than inline, because `AbortSignal.aborted` is
74209
+ * a live getter: it can flip to true while a leg is in flight, but a compiler
74210
+ * that narrowed it at the top of the loop would prove the later check
74211
+ * unreachable and invite its removal. The check is not redundant — it is the
74212
+ * only thing that stops the chain spending money on an answer nobody will read.
74213
+ *
74214
+ * @param signal The caller's signal, if any.
74215
+ * @returns Whether the call has been cancelled.
74216
+ */
74217
+ function isAborted(signal) {
74218
+ return signal !== undefined && signal.aborted;
74219
+ }
74220
+ /**
74221
+ * The usage a failed leg was billed for, when the leg reached an answer.
74222
+ *
74223
+ * A leg that failed after the provider answered — content that does not parse,
74224
+ * or prose where a tool call was mandatory — was still charged. A leg that
74225
+ * never answered (timeout, outage, skip) carries no usage, and none is invented.
74226
+ *
74227
+ * @param error The thrown value.
74228
+ * @returns The billed usage, or undefined when the leg never produced an answer.
74229
+ */
74230
+ function billedUsageOf(error) {
74231
+ if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
74232
+ return error.usage;
74233
+ }
74234
+ return undefined;
74235
+ }
74236
+
74237
+ /**
74238
+ * Same-model attempts for one leg: hedging, measured attempt timeouts, and
74239
+ * reserving deadline for a same-model alternative.
74240
+ *
74241
+ * A leg is a model. Before the chain gives up on it and reaches a DIFFERENT
74242
+ * model — which changes the answer's quality, not just its latency — it is
74243
+ * worth spending the leg's budget on every way of getting that same model to
74244
+ * answer: the same model at another provider (an "equivalent"), or a second
74245
+ * request to the same provider when that provider has capacity to spare (a
74246
+ * "duplicate"). This module runs those attempts as one group.
74247
+ *
74248
+ * Three mechanics, all bounded by the leg's budget and none of them selecting
74249
+ * a different model:
74250
+ *
74251
+ * - **Hedging.** Once an attempt has run past the model's healthy p90 (from
74252
+ * the latency tracker), the next same-model attempt starts beside it. The
74253
+ * first answer wins and every other attempt is cancelled through its own
74254
+ * abort signal. A hedge that loses, or an attempt cancelled because another
74255
+ * won, says nothing about the provider and never touches its breaker.
74256
+ *
74257
+ * - **Measured attempt timeout.** With latency evidence, an attempt that has
74258
+ * run past `k × p99` of healthy latency is replaced by a same-model
74259
+ * alternative instead of holding the budget to its end. It is replaced only
74260
+ * when an alternative exists: with none, cutting it short would only move
74261
+ * the call to a different model sooner, and it runs to the leg budget as
74262
+ * before.
74263
+ *
74264
+ * - **Deadline reservation.** When a same-model equivalent exists, no single
74265
+ * attempt holds more than `max_attempt_share` of the remaining budget before
74266
+ * the equivalent starts beside it, so a slow first attempt cannot consume the
74267
+ * whole deadline while the equivalent that could have answered never runs.
74268
+ *
74269
+ * With no latency evidence and no equivalent, none of the three can act and
74270
+ * the group is exactly one attempt with the leg's full budget: the serial
74271
+ * chain's behaviour, which a fresh process therefore starts from.
74272
+ *
74273
+ * @module llm/hedge
74274
+ */
74275
+ /**
74276
+ * Read the policy from the table's defaults.
74277
+ *
74278
+ * @param defaults The `hedging` defaults.
74279
+ * @returns The policy.
74280
+ */
74281
+ function sameModelPolicyFrom(defaults) {
74282
+ return {
74283
+ maxExtraAttempts: defaults.max_same_model_hedges,
74284
+ hedgeQuantile: defaults.hedge_quantile,
74285
+ timeoutQuantile: defaults.timeout_quantile,
74286
+ kTimeout: defaults.k_timeout,
74287
+ attemptTimeoutFloorMs: defaults.attempt_timeout_floor_ms,
74288
+ maxAttemptShare: defaults.max_attempt_share,
74289
+ duplicateReserve: defaults.duplicate_headroom_reserve,
74290
+ };
74291
+ }
74292
+ /**
74293
+ * The parameters of a prepared leg, when it can serve.
74294
+ *
74295
+ * @param leg The leg.
74296
+ * @returns Its parameters, or undefined when it cannot serve this request.
74297
+ */
74298
+ function paramsOf(leg) {
74299
+ return leg.params instanceof UnsupportedCapabilityError ? undefined : leg.params;
74300
+ }
74301
+ /**
74302
+ * Run every same-model attempt for one leg until one answers or the budget,
74303
+ * the alternatives, or the caller run out.
74304
+ *
74305
+ * The caller has already found the leg servable (breaker allows it, budget
74306
+ * positive). The returned promise settles only once every attempt it started
74307
+ * has been recorded, so the call's attempt record is complete when it returns.
74308
+ *
74309
+ * @param leg The leg, with its equivalents.
74310
+ * @param groupBudgetMs The leg's budget: its route budget cut to the deadline.
74311
+ * @param budgetIsDeadline Whether that budget was cut by the caller's deadline.
74312
+ * @param ctx The chain around the group.
74313
+ * @returns How the group ended.
74314
+ */
74315
+ function runSameModelGroup(leg, groupBudgetMs, budgetIsDeadline, ctx) {
74316
+ const { breakers, now, policy, tracker, promptTokens, request } = ctx;
74317
+ const endsAt = now() + groupBudgetMs;
74318
+ const equivalents = (leg.equivalents ?? []).filter((equivalent) => paramsOf(equivalent) !== undefined);
74319
+ const live = new Set();
74320
+ const billed = [];
74321
+ const timeoutCharged = new Set();
74322
+ let nextEquivalent = 0;
74323
+ let extraLaunched = 0;
74324
+ let settled = false;
74325
+ let finished = false;
74326
+ let deadlineBound = false;
74327
+ let hedgeTimer;
74328
+ let answer;
74329
+ return new Promise((resolve) => {
74330
+ /**
74331
+ * A healthy-latency quantile for a leg's model, when there is evidence.
74332
+ *
74333
+ * @param route The leg's route.
74334
+ * @param q The quantile.
74335
+ * @returns Milliseconds, or null.
74336
+ */
74337
+ const quantileOf = (route, q) => tracker === undefined ? null : tracker.quantile(route.providerName, route.modelId, promptTokens, q);
74338
+ /**
74339
+ * The next same-model attempt that could start now: an equivalent first,
74340
+ * then a duplicate, which needs latency evidence and spare provider capacity.
74341
+ *
74342
+ * @returns The candidate, or undefined.
74343
+ */
74344
+ const peek = () => {
74345
+ if (policy === undefined || extraLaunched >= policy.maxExtraAttempts) {
74346
+ return undefined;
74347
+ }
74348
+ for (let index = nextEquivalent; index < equivalents.length; index += 1) {
74349
+ const equivalent = equivalents[index];
74350
+ if (breakers.allows(equivalent.route.routeKey)) {
74351
+ return { leg: equivalent, kind: "equivalent", index };
74352
+ }
74353
+ }
74354
+ if (quantileOf(leg.route, policy.hedgeQuantile) !== null &&
74355
+ breakers.allows(leg.route.routeKey) &&
74356
+ ctx.admitDuplicate(leg.route, policy.duplicateReserve)) {
74357
+ return { leg, kind: "duplicate", index: -1 };
74358
+ }
74359
+ return undefined;
74360
+ };
74361
+ /** Resolve once nothing is left in flight. */
74362
+ const finishIfIdle = () => {
74363
+ if (finished || live.size > 0) {
74364
+ return;
74365
+ }
74366
+ finished = true;
74367
+ if (hedgeTimer !== undefined) {
74368
+ clearTimeout(hedgeTimer);
74369
+ }
74370
+ resolve({ answer, billed, deadlineBound });
74371
+ };
74372
+ /**
74373
+ * Arm the hedge for the most recent attempt: at the model's healthy p90,
74374
+ * or — when an equivalent is waiting — no later than the reserved share of
74375
+ * the remaining budget.
74376
+ *
74377
+ * @param latest The attempt just started.
74378
+ */
74379
+ const scheduleHedge = (latest) => {
74380
+ if (hedgeTimer !== undefined) {
74381
+ clearTimeout(hedgeTimer);
74382
+ hedgeTimer = undefined;
74383
+ }
74384
+ const candidate = peek();
74385
+ if (policy === undefined || candidate === undefined) {
74386
+ return;
74387
+ }
74388
+ const measured = quantileOf(latest.leg.route, policy.hedgeQuantile);
74389
+ const reserved = candidate.kind === "equivalent"
74390
+ ? policy.maxAttemptShare * (endsAt - now())
74391
+ : Number.POSITIVE_INFINITY;
74392
+ const delay = Math.min(measured ?? Number.POSITIVE_INFINITY, reserved);
74393
+ if (!Number.isFinite(delay)) {
74394
+ return;
74395
+ }
74396
+ hedgeTimer = setTimeout(() => {
74397
+ hedgeTimer = undefined;
74398
+ if (settled) {
74399
+ return;
74400
+ }
74401
+ const next = peek();
74402
+ if (next !== undefined) {
74403
+ launch(next.leg, next, true);
74404
+ }
74405
+ }, Math.max(0, delay));
74406
+ };
74407
+ /**
74408
+ * Start one attempt.
74409
+ *
74410
+ * @param target The leg to address.
74411
+ * @param candidate The candidate it came from, for a hedge.
74412
+ * @param hedged Whether this is a hedge rather than the leg's first attempt.
74413
+ * @returns Whether it started.
74414
+ */
74415
+ const launch = (target, candidate, hedged) => {
74416
+ const params = paramsOf(target);
74417
+ const budgetMs = endsAt - now();
74418
+ if (params === undefined || budgetMs <= 0 || isAborted(request.callerSignal)) {
74419
+ return false;
74420
+ }
74421
+ if (candidate !== undefined) {
74422
+ extraLaunched += 1;
74423
+ if (candidate.kind === "equivalent") {
74424
+ nextEquivalent = candidate.index + 1;
74425
+ }
74426
+ }
74427
+ const key = target.route.routeKey;
74428
+ const attempt = {
74429
+ leg: target,
74430
+ handle: startAttempt(target, params, request, budgetMs),
74431
+ startedAt: now(),
74432
+ budgetMs,
74433
+ holdsProbe: breakers.onAttemptStart(key),
74434
+ hedged,
74435
+ attemptIndex: ctx.nextAttemptIndex(),
74436
+ closed: false,
74437
+ };
74438
+ live.add(attempt);
74439
+ if (policy !== undefined) {
74440
+ const tail = quantileOf(target.route, policy.timeoutQuantile);
74441
+ if (tail !== null) {
74442
+ const measuredMs = Math.max(policy.attemptTimeoutFloorMs, policy.kTimeout * tail);
74443
+ if (measuredMs < budgetMs) {
74444
+ attempt.softTimer = setTimeout(() => supersede(attempt, measuredMs), measuredMs);
74445
+ }
74446
+ }
74447
+ }
74448
+ scheduleHedge(attempt);
74449
+ attempt.handle.promise.then((response) => onAnswer(attempt, response), (error) => onFailure(attempt, error));
74450
+ return true;
74451
+ };
74452
+ /**
74453
+ * An attempt ran past its measured timeout: replace it if a same-model
74454
+ * alternative can take over, otherwise leave it running to the leg budget.
74455
+ *
74456
+ * @param attempt The slow attempt.
74457
+ * @param afterMs Its measured timeout.
74458
+ */
74459
+ const supersede = (attempt, afterMs) => {
74460
+ attempt.softTimer = undefined;
74461
+ if (settled || !live.has(attempt)) {
74462
+ return;
74463
+ }
74464
+ const othersInFlight = live.size > 1;
74465
+ const candidate = othersInFlight ? undefined : peek();
74466
+ if (!othersInFlight && candidate === undefined) {
74467
+ return;
74468
+ }
74469
+ // Recorded now rather than when its rejection arrives, so a replacement
74470
+ // that answers first cannot find it still live and misfile it as a loser.
74471
+ const superseded = new AttemptSupersededError(attempt.leg.route.routeKey, afterMs);
74472
+ close(attempt, superseded);
74473
+ if (candidate !== undefined) {
74474
+ launch(candidate.leg, candidate, true);
74475
+ }
74476
+ finishIfIdle();
74477
+ };
74478
+ /**
74479
+ * Cancel an attempt the chain no longer wants and record it at once, with
74480
+ * no verdict on the provider.
74481
+ *
74482
+ * @param attempt The attempt.
74483
+ * @param reason Why it was cancelled.
74484
+ */
74485
+ const close = (attempt, reason) => {
74486
+ const fields = baseFields(attempt);
74487
+ attempt.handle.abort(reason);
74488
+ attempt.closed = true;
74489
+ retire(attempt);
74490
+ if (attempt.holdsProbe) {
74491
+ breakers.onAttemptAbandoned(attempt.leg.route.routeKey);
74492
+ }
74493
+ const failure = classify(reason, request.callerSignal);
74494
+ emit(attempt, { ...fields, outcome: failure.outcome, reason: failure.reason });
74495
+ };
74496
+ /**
74497
+ * Stop tracking an attempt and release its breaker bookkeeping.
74498
+ *
74499
+ * @param attempt The attempt.
74500
+ */
74501
+ const retire = (attempt) => {
74502
+ if (attempt.softTimer !== undefined) {
74503
+ clearTimeout(attempt.softTimer);
74504
+ attempt.softTimer = undefined;
74505
+ }
74506
+ live.delete(attempt);
74507
+ breakers.onAttemptEnd(attempt.leg.route.routeKey);
74508
+ };
74509
+ /**
74510
+ * Record one attempt through the chain.
74511
+ *
74512
+ * @param attempt The attempt.
74513
+ * @param fields What happened.
74514
+ * @param servedProvider The provider's own report of who served, if any.
74515
+ */
74516
+ const emit = (attempt, fields, servedProvider) => {
74517
+ ctx.record(attempt.leg.route, fields, { hedged: attempt.hedged, attemptIndex: attempt.attemptIndex }, servedProvider);
74518
+ };
74519
+ /**
74520
+ * Base fields shared by every record of an attempt.
74521
+ *
74522
+ * @param attempt The attempt.
74523
+ * @returns The identity and timing fields.
74524
+ */
74525
+ const baseFields = (attempt) => ({
74526
+ routeKey: attempt.leg.route.routeKey,
74527
+ role: attempt.leg.route.role,
74528
+ provider: attempt.leg.route.providerName,
74529
+ modelId: attempt.leg.route.modelId,
74530
+ durationMs: now() - attempt.startedAt,
74531
+ budgetMs: attempt.budgetMs,
74532
+ });
74533
+ const onAnswer = (attempt, response) => {
74534
+ if (attempt.closed) {
74535
+ return;
74536
+ }
74537
+ const { route } = attempt.leg;
74538
+ const fields = baseFields(attempt);
74539
+ retire(attempt);
74540
+ breakers.onSuccess(route.routeKey, attempt.startedAt);
74541
+ breakers.onLatencySample(route.routeKey, fields.durationMs, route.latencyClass);
74542
+ tracker?.record(route.providerName, route.modelId, promptTokens, fields.durationMs);
74543
+ billed.push(response.usage);
74544
+ if (settled) {
74545
+ // Answered in the same instant as the winner, before its cancellation
74546
+ // arrived. Its spend is real; its answer is not the one returned.
74547
+ emit(attempt, {
74548
+ ...fields,
74549
+ outcome: "skipped",
74550
+ reason: "answered after a same-model attempt had already won",
74551
+ servedModel: response.servedModel ?? null,
74552
+ usage: response.usage,
74553
+ }, response.servedProvider);
74554
+ finishIfIdle();
74555
+ return;
74556
+ }
74557
+ settled = true;
74558
+ answer = { response, route, hedged: attempt.hedged };
74559
+ emit(attempt, {
74560
+ ...fields,
74561
+ outcome: "ok",
74562
+ servedModel: response.servedModel ?? null,
74563
+ usage: response.usage,
74564
+ }, response.servedProvider);
74565
+ for (const loser of [...live]) {
74566
+ // Cancelled through its own signal and recorded now, so the answer is
74567
+ // returned without waiting for a loser still queued at the provider.
74568
+ close(loser, new HedgeLoserError(loser.leg.route.routeKey));
74569
+ }
74570
+ finishIfIdle();
74571
+ };
74572
+ const onFailure = (attempt, error) => {
74573
+ if (attempt.closed) {
74574
+ return;
74575
+ }
74576
+ const { route } = attempt.leg;
74577
+ const key = route.routeKey;
74578
+ const fields = baseFields(attempt);
74579
+ const hardTimeout = attempt.handle.timedOut();
74580
+ retire(attempt);
74581
+ const failure = classify(attempt.handle.abortReason() ?? error, request.callerSignal);
74582
+ let charged = failure.countsAgainstHealth;
74583
+ if (charged && hardTimeout) {
74584
+ // Every attempt of a leg shares the leg's end, so several can time out
74585
+ // together; the provider is charged once for the leg, as before hedging.
74586
+ charged = !timeoutCharged.has(key);
74587
+ timeoutCharged.add(key);
74588
+ if (budgetIsDeadline) {
74589
+ deadlineBound = true;
74590
+ }
74591
+ }
74592
+ if (charged) {
74593
+ breakers.onFailure(key, failure.failureKind);
74594
+ }
74595
+ else if (attempt.holdsProbe) {
74596
+ // No verdict on the route's health, but the probe slot this attempt
74597
+ // took must come back, or a half-open route admits no probe ever again.
74598
+ breakers.onAttemptAbandoned(key);
74599
+ }
74600
+ if (failure.countsAgainstHealth) {
74601
+ breakers.onLatencySample(key, fields.durationMs, route.latencyClass);
74602
+ }
74603
+ // A provider that answered — with unparseable content, or in prose where a
74604
+ // tool call was mandatory — still billed for the answer.
74605
+ const usage = billedUsageOf(error);
74606
+ if (usage !== undefined) {
74607
+ billed.push(usage);
74608
+ }
74609
+ const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
74610
+ emit(attempt, {
74611
+ ...fields,
74612
+ outcome: failure.outcome,
74613
+ reason: failure.reason,
74614
+ ...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
74615
+ ...(usage === undefined ? {} : { usage }),
74616
+ });
74617
+ if (settled || live.size > 0) {
74618
+ finishIfIdle();
74619
+ return;
74620
+ }
74621
+ if (!hardTimeout && !isAborted(request.callerSignal)) {
74622
+ // A failed attempt with time left moves to the same model at another
74623
+ // provider at once. A duplicate on the provider that just failed is not
74624
+ // started here: it would most likely fail the same way.
74625
+ const candidate = peek();
74626
+ if (candidate !== undefined && candidate.kind === "equivalent") {
74627
+ launch(candidate.leg, candidate, true);
74628
+ }
74629
+ }
74630
+ finishIfIdle();
74631
+ };
74632
+ if (!launch(leg, undefined, false)) {
74633
+ finishIfIdle();
74634
+ }
74635
+ });
74636
+ }
74637
+
73414
74638
  /**
73415
74639
  * Ordered execution of an alias's fallback chain (PD-3).
73416
74640
  *
@@ -73423,10 +74647,21 @@ function parseStructuredContent(content, responseFormat, usage) {
73423
74647
  * why the budget is enforced here — at the only place that knows both the
73424
74648
  * caller's deadline and how many legs are left to spend it on.
73425
74649
  *
74650
+ * Each leg is first run as a group of SAME-MODEL attempts (see `hedge.ts`):
74651
+ * hedged at the model's healthy p90, replaced after a measured timeout, and
74652
+ * reaching the same model at another provider before the leg is given up.
74653
+ * Only then does the walk move to the next leg — and a leg that serves a
74654
+ * different model than the configured one runs only when the caller's
74655
+ * cross-model policy allows it. With no latency evidence and no equivalent
74656
+ * configured, each group is a single attempt with the leg's budget, which is
74657
+ * the serial walk exactly.
74658
+ *
73426
74659
  * Nothing here ever substitutes a value for an outcome. When every leg is
73427
- * exhausted the caller gets a typed error naming each leg and why it failed,
73428
- * because a default returned in place of an answer is a wrong answer that
73429
- * nobody is told about.
74660
+ * exhausted the caller gets a typed error naming each leg and why it failed —
74661
+ * `LlmDeadlineExceededError` when the caller's deadline is what ran out,
74662
+ * `ChainExhaustedError` with reason `cross_model_denied` when policy stopped
74663
+ * the walk — because a default returned in place of an answer is a wrong
74664
+ * answer that nobody is told about.
73430
74665
  *
73431
74666
  * @module llm/fallback-chain
73432
74667
  */
@@ -73453,21 +74688,64 @@ class ChainExhaustedError extends Error {
73453
74688
  attempts;
73454
74689
  /** Usage spent across the failed attempts, so the spend is still accounted for. */
73455
74690
  totalUsage;
74691
+ /**
74692
+ * Why the chain ended. `cross_model_denied`: the configured model's attempts
74693
+ * were spent and policy forbade a different model. Callers map every reason
74694
+ * to no decision; the reason says which remedy applies.
74695
+ */
74696
+ reason;
73456
74697
  /**
73457
74698
  * @param alias The alias.
73458
74699
  * @param attempts The attempt record.
73459
74700
  * @param totalUsage Usage spent across all attempts.
74701
+ * @param reason Why the chain ended; defaults to plain exhaustion.
73460
74702
  */
73461
- constructor(alias, attempts, totalUsage) {
74703
+ constructor(alias, attempts, totalUsage, reason = "exhausted") {
73462
74704
  const detail = attempts
73463
74705
  .map((attempt) => `${attempt.role}(${attempt.provider}/${attempt.modelId}): ${attempt.outcome}` +
73464
74706
  (attempt.reason === undefined ? "" : ` — ${attempt.reason}`))
73465
74707
  .join("; ");
73466
- super(`LLM alias "${alias}" exhausted its fallback chain. Attempts: ${detail || "(no leg was servable)"}`);
74708
+ const why = reason === "cross_model_denied"
74709
+ ? " Different-model legs were denied by the caller's cross-model policy."
74710
+ : reason === "deadline_exceeded"
74711
+ ? " The caller's deadline ran out."
74712
+ : "";
74713
+ super(`LLM alias "${alias}" exhausted its fallback chain.${why} Attempts: ${detail || "(no leg was servable)"}`);
73467
74714
  this.name = "ChainExhaustedError";
73468
74715
  this.alias = alias;
73469
74716
  this.attempts = attempts;
73470
74717
  this.totalUsage = totalUsage;
74718
+ this.reason = reason;
74719
+ }
74720
+ }
74721
+ /**
74722
+ * Thrown when the caller's deadline ran out before any leg answered.
74723
+ *
74724
+ * A subclass of {@link ChainExhaustedError}, so a consumer that already treats
74725
+ * exhaustion as "no answer" keeps doing so, while one that needs to tell "the
74726
+ * models failed" from "we ran out of time" can match this class — the two call
74727
+ * for different remedies (a provider problem versus a budget problem), and
74728
+ * both map to no decision, never to a default.
74729
+ */
74730
+ class LlmDeadlineExceededError extends ChainExhaustedError {
74731
+ /** Discriminant for consumers that switch on shape rather than class. */
74732
+ kind = "deadline_exceeded";
74733
+ /** The whole-call budget the chain started with, in milliseconds. */
74734
+ deadlineMs;
74735
+ /** The model class of the last attempt dispatched, or null when none was. */
74736
+ lastModelClass;
74737
+ /**
74738
+ * @param alias The alias.
74739
+ * @param attempts The attempt record.
74740
+ * @param totalUsage Usage spent across all attempts.
74741
+ * @param deadlineMs The budget the chain started with.
74742
+ * @param lastModelClass The last dispatched attempt's model class.
74743
+ */
74744
+ constructor(alias, attempts, totalUsage, deadlineMs, lastModelClass) {
74745
+ super(alias, attempts, totalUsage, "deadline_exceeded");
74746
+ this.name = "LlmDeadlineExceededError";
74747
+ this.deadlineMs = deadlineMs;
74748
+ this.lastModelClass = lastModelClass;
73471
74749
  }
73472
74750
  }
73473
74751
  /**
@@ -73533,219 +74811,70 @@ function legBudgetMs(routeBudgetMs, deadlineAtMs, nowMs) {
73533
74811
  }
73534
74812
  return Math.min(routeBudgetMs, deadlineAtMs - nowMs);
73535
74813
  }
73536
- /** Raised internally when a leg exceeds its budget. */
73537
- class LegTimeoutError extends Error {
73538
- /**
73539
- * @param routeKey The leg that timed out.
73540
- * @param budgetMs Its budget in milliseconds.
73541
- */
73542
- constructor(routeKey, budgetMs) {
73543
- super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
73544
- this.name = "LegTimeoutError";
73545
- }
73546
- }
73547
74814
  /**
73548
- * Run one leg under a hard timeout, honouring the caller's own cancellation.
73549
- *
73550
- * The timer is always cleared and the abort listener always removed, including
73551
- * on the success path. A long-lived process that leaked one timer per LLM call
73552
- * would accumulate them at exactly the rate it does useful work.
74815
+ * The model a leg serves, independent of which provider hosts it.
73553
74816
  *
73554
- * @param leg The leg to run.
73555
- * @param params Normalised parameters for this leg.
73556
- * @param execution The call context.
73557
- * @param budgetMs The leg's budget: its route budget cut to the caller's deadline.
73558
- * @returns The provider's answer.
74817
+ * @param route The leg's route.
74818
+ * @returns Its model class.
73559
74819
  */
73560
- async function runLeg(leg, params, execution, budgetMs) {
73561
- const controller = new AbortController();
73562
- const timer = setTimeout(() => {
73563
- controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
73564
- }, budgetMs);
73565
- const forwardAbort = () => {
73566
- controller.abort(execution.callerSignal?.reason);
73567
- };
73568
- if (execution.callerSignal !== undefined) {
73569
- if (execution.callerSignal.aborted) {
73570
- forwardAbort();
73571
- }
73572
- else {
73573
- execution.callerSignal.addEventListener("abort", forwardAbort, { once: true });
73574
- }
73575
- }
73576
- try {
73577
- // The guards wrap the transport rather than the whole leg, so the per-leg
73578
- // timeout above still bounds the total wait: a caller queued behind the
73579
- // rate limiter is spending its budget just as surely as one waiting on the
73580
- // provider, and only one clock should govern both. The leg's own signal is
73581
- // handed to the guard as well, so a leg whose budget or caller is gone
73582
- // leaves the queue at once instead of holding its place in it.
73583
- const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
73584
- route: leg.route,
73585
- content: execution.content,
73586
- responseFormat: execution.responseFormat,
73587
- params,
73588
- developerPrompt: execution.developerPrompt,
73589
- context: execution.context,
73590
- signal: controller.signal,
73591
- correlationId: execution.correlationId,
73592
- }), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
73593
- assertToolChoiceHonoured(leg.route, params, response);
73594
- return response;
73595
- }
73596
- finally {
73597
- clearTimeout(timer);
73598
- execution.callerSignal?.removeEventListener("abort", forwardAbort);
73599
- }
74820
+ function modelClassOf(route) {
74821
+ return route.modelClass ?? route.modelId;
73600
74822
  }
73601
74823
  /**
73602
- * HTTP statuses a provider (or the gateway relaying it) uses to say it is full
73603
- * rather than that the request or the route is wrong: request timeout, too
73604
- * early, too many requests, service unavailable, and Anthropic's overloaded.
73605
- */
73606
- const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
73607
- /**
73608
- * Wording providers use for a capacity refusal when the status is lost on the
73609
- * way (a relayed body, a client library's own error). DeepInfra's is
73610
- * "Model busy, retry later"; Anthropic's is "Overloaded".
73611
- */
73612
- const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
73613
- /**
73614
- * Whether a failure is the provider saying it is full rather than broken.
74824
+ * Whether a provider-reported model names the model a route addressed.
73615
74825
  *
73616
- * Read by shape rather than by class, because the same signal reaches the
73617
- * chain from more than one transport and not every transport's error class is
73618
- * importable here.
74826
+ * Providers report with or without an organisation prefix and in their own
74827
+ * case, so the comparison is case-insensitive and accepts one side being a
74828
+ * `/`-suffix of the other.
73619
74829
  *
73620
- * @param error The thrown value.
73621
- * @param reason Its message.
73622
- * @returns Whether it is a capacity signal.
74830
+ * @param reported The provider's report.
74831
+ * @param expected The route's model id.
74832
+ * @returns Whether they name the same model.
73623
74833
  */
73624
- function isCapacitySignal(error, reason) {
73625
- if (typeof error === "object" && error !== null) {
73626
- const status = error.status;
73627
- if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
73628
- return true;
73629
- }
73630
- }
73631
- return CAPACITY_WORDING.test(reason);
74834
+ function isSameReportedModel(reported, expected) {
74835
+ const a = reported.trim().toLowerCase();
74836
+ const b = expected.trim().toLowerCase();
74837
+ return a === b || a.endsWith(`/${b}`) || b.endsWith(`/${a}`);
73632
74838
  }
73633
74839
  /**
73634
- * Classify why a leg failed.
74840
+ * How an attempt's model relates to the configured one.
73635
74841
  *
73636
- * The distinction matters to the breaker: a timeout and a 5xx are evidence the
73637
- * provider is unhealthy, while the caller cancelling is not. Counting a
73638
- * cancellation as a provider failure would let a burst of user-cancelled
73639
- * requests open the breaker on a perfectly healthy route.
73640
- *
73641
- * Among failures that do count, a capacity signal (the provider said it is
73642
- * busy, or the leg ran out its budget waiting on it) is told apart from a hard
73643
- * failure so the breaker can re-admit a busy route sooner than a broken one. A
73644
- * timeout is read as capacity: on a reachable provider it is what a full queue
73645
- * looks like from outside, and a provider that is actually down still costs no
73646
- * more than one probe per capacity cooldown.
74842
+ * A leg addressed to a different model class is `different` whatever it
74843
+ * reported. A leg addressed to the configured class is `same` only when it
74844
+ * answered and the provider named the expected model; `different` when the
74845
+ * provider named another; `unknown` when it answered without saying. An
74846
+ * attempt that never answered carries the relation of the model it was
74847
+ * addressed to.
73647
74848
  *
73648
- * @param error The thrown value.
73649
- * @param callerSignal The caller's cancellation signal, if any.
73650
- * @returns The outcome and whether it counts against route health.
74849
+ * @param route The attempt's route.
74850
+ * @param configuredClass The configured model class.
74851
+ * @param fields What the attempt recorded.
74852
+ * @returns The relation.
73651
74853
  */
73652
- function classify(error, callerSignal) {
73653
- if (callerSignal !== undefined && callerSignal.aborted) {
73654
- return {
73655
- outcome: "skipped",
73656
- reason: "caller cancelled",
73657
- countsAgainstHealth: false,
73658
- failureKind: "hard",
73659
- };
73660
- }
73661
- if (error instanceof LegTimeoutError) {
73662
- return {
73663
- outcome: "timeout",
73664
- reason: error.message,
73665
- countsAgainstHealth: true,
73666
- failureKind: "capacity",
73667
- };
73668
- }
73669
- if (error instanceof UnsupportedCapabilityError) {
73670
- return {
73671
- outcome: "skipped",
73672
- reason: error.message,
73673
- countsAgainstHealth: false,
73674
- failureKind: "hard",
73675
- };
73676
- }
73677
- if (error instanceof ToolChoiceIgnoredError) {
73678
- // The route answered; it broke a declared guarantee rather than failing to
73679
- // be available, so its breaker is not charged for it.
73680
- return {
73681
- outcome: "error",
73682
- reason: error.message,
73683
- countsAgainstHealth: false,
73684
- failureKind: "hard",
73685
- };
74854
+ function modelClassRelationOf(route, configuredClass, fields) {
74855
+ if (modelClassOf(route) !== configuredClass) {
74856
+ return "different";
73686
74857
  }
73687
- // Self-inflicted pacing, not provider ill-health. Counting it would let the
73688
- // client's own throttling open a breaker on a perfectly healthy provider and
73689
- // permanently reroute traffic nobody chose to reroute.
73690
- if (error instanceof RateGuardTimeoutError) {
73691
- return {
73692
- outcome: "skipped",
73693
- reason: error.message,
73694
- countsAgainstHealth: false,
73695
- failureKind: "hard",
73696
- };
74858
+ if (fields.outcome !== "ok") {
74859
+ return "same";
73697
74860
  }
73698
- const reason = error instanceof Error ? error.message : String(error);
73699
- if (/abort/i.test(reason)) {
73700
- return {
73701
- outcome: "timeout",
73702
- reason: `aborted: ${reason}`,
73703
- countsAgainstHealth: true,
73704
- failureKind: "capacity",
73705
- };
73706
- }
73707
- if (error instanceof LlmResponseFormatError) {
73708
- // The provider answered, badly. That is a route defect, not a full queue,
73709
- // whatever words the unparseable content happens to contain.
73710
- return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
74861
+ if (fields.servedModel === undefined || fields.servedModel === null) {
74862
+ return "unknown";
73711
74863
  }
73712
- return {
73713
- outcome: "error",
73714
- reason,
73715
- countsAgainstHealth: true,
73716
- failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
73717
- };
74864
+ return isSameReportedModel(fields.servedModel, route.modelId) ? "same" : "different";
73718
74865
  }
73719
74866
  /**
73720
- * Whether the caller has stopped waiting.
74867
+ * The prompt size the latency evidence is bucketed by.
73721
74868
  *
73722
- * Read through a function rather than inline, because `AbortSignal.aborted` is
73723
- * a live getter: it can flip to true while a leg is in flight, but a compiler
73724
- * that narrowed it at the top of the loop would prove the later check
73725
- * unreachable and invite its removal. The check is not redundant — it is the
73726
- * only thing that stops the chain spending money on an answer nobody will read.
73727
- *
73728
- * @param signal The caller's signal, if any.
73729
- * @returns Whether the call has been cancelled.
74869
+ * @param execution The call.
74870
+ * @returns Estimated prompt tokens, or null.
73730
74871
  */
73731
- function isAborted(signal) {
73732
- return signal !== undefined && signal.aborted;
73733
- }
73734
- /**
73735
- * The usage a failed leg was billed for, when the leg reached an answer.
73736
- *
73737
- * A leg that failed after the provider answered — content that does not parse,
73738
- * or prose where a tool call was mandatory — was still charged. A leg that
73739
- * never answered (timeout, outage, skip) carries no usage, and none is invented.
73740
- *
73741
- * @param error The thrown value.
73742
- * @returns The billed usage, or undefined when the leg never produced an answer.
73743
- */
73744
- function billedUsageOf(error) {
73745
- if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
73746
- return error.usage;
73747
- }
73748
- return undefined;
74872
+ function promptTokensOf(execution) {
74873
+ return estimatePromptTokens([
74874
+ execution.content,
74875
+ execution.developerPrompt,
74876
+ execution.context,
74877
+ ]);
73749
74878
  }
73750
74879
  /**
73751
74880
  * Walk a chain until a leg answers.
@@ -73753,12 +74882,58 @@ function billedUsageOf(error) {
73753
74882
  * @param alias The alias being served, for error attribution.
73754
74883
  * @param execution The call context.
73755
74884
  * @returns The first successful leg's answer, with the full attempt record.
74885
+ * @throws {LlmDeadlineExceededError} When the caller's deadline ran out first.
73756
74886
  * @throws {ChainExhaustedError} When no leg produced an answer.
73757
74887
  */
73758
74888
  async function executeChain(alias, execution) {
73759
74889
  const now = execution.now ?? Date.now;
74890
+ const startedAt = now();
73760
74891
  const attempts = [];
74892
+ const firstRoute = execution.legs[0]?.route;
74893
+ const configuredClass = execution.configuredModelClass ?? (firstRoute === undefined ? "" : modelClassOf(firstRoute));
74894
+ const policy = execution.crossModelPolicy ?? "allow_record";
74895
+ const promptTokens = execution.latency === undefined ? null : promptTokensOf(execution);
73761
74896
  let totalUsage = EMPTY_USAGE;
74897
+ let dispatchIndex = 0;
74898
+ let deadlineHit = false;
74899
+ let crossModelDenied = false;
74900
+ let lastModelClass = null;
74901
+ /**
74902
+ * Record one attempt with its provenance.
74903
+ *
74904
+ * @param route The attempt's route.
74905
+ * @param fields What happened.
74906
+ * @param dispatch Whether it was a hedge, and its dispatch index.
74907
+ * @param servedProvider The provider's own report of who served, if any.
74908
+ * @returns The record.
74909
+ */
74910
+ const record = (route, fields, dispatch, servedProvider) => {
74911
+ const full = {
74912
+ ...fields,
74913
+ servedProvider: servedProvider ?? route.providerName,
74914
+ modelClass: modelClassOf(route),
74915
+ modelClassRelation: modelClassRelationOf(route, configuredClass, fields),
74916
+ ...(dispatch === undefined
74917
+ ? {}
74918
+ : { hedged: dispatch.hedged, attemptIndex: dispatch.attemptIndex }),
74919
+ };
74920
+ attempts.push(full);
74921
+ execution.onAttempt?.(full);
74922
+ return full;
74923
+ };
74924
+ /**
74925
+ * The identity fields of a leg that was never dispatched.
74926
+ *
74927
+ * @param route The leg's route.
74928
+ * @returns The fields.
74929
+ */
74930
+ const undispatched = (route) => ({
74931
+ routeKey: route.routeKey,
74932
+ role: route.role,
74933
+ provider: route.providerName,
74934
+ modelId: route.modelId,
74935
+ durationMs: 0,
74936
+ });
73762
74937
  for (const leg of execution.legs) {
73763
74938
  const { route } = leg;
73764
74939
  if (isAborted(execution.callerSignal)) {
@@ -73766,108 +74941,80 @@ async function executeChain(alias, execution) {
73766
74941
  // spend money on an answer nobody will read.
73767
74942
  break;
73768
74943
  }
73769
- if (leg.params instanceof UnsupportedCapabilityError) {
73770
- const record = {
73771
- routeKey: route.routeKey,
73772
- role: route.role,
73773
- provider: route.providerName,
73774
- modelId: route.modelId,
74944
+ if (policy === "deny" && modelClassOf(route) !== configuredClass) {
74945
+ crossModelDenied = true;
74946
+ record(route, {
74947
+ ...undispatched(route),
73775
74948
  outcome: "skipped",
73776
- durationMs: 0,
73777
- reason: leg.params.message,
73778
- };
73779
- attempts.push(record);
73780
- execution.onAttempt?.(record);
74949
+ reason: `cross-model leg denied by policy: configured model is ${configuredClass}, this leg serves ${modelClassOf(route)}`,
74950
+ });
74951
+ continue;
74952
+ }
74953
+ if (leg.params instanceof UnsupportedCapabilityError) {
74954
+ record(route, { ...undispatched(route), outcome: "skipped", reason: leg.params.message });
73781
74955
  continue;
73782
74956
  }
73783
74957
  if (!execution.breakers.allows(route.routeKey)) {
73784
- const record = {
73785
- routeKey: route.routeKey,
73786
- role: route.role,
73787
- provider: route.providerName,
73788
- modelId: route.modelId,
74958
+ record(route, {
74959
+ ...undispatched(route),
73789
74960
  outcome: "breaker-open",
73790
- durationMs: 0,
73791
74961
  reason: `circuit breaker is ${execution.breakers.stateOf(route.routeKey)}`,
73792
- };
73793
- attempts.push(record);
73794
- execution.onAttempt?.(record);
74962
+ });
73795
74963
  continue;
73796
74964
  }
73797
74965
  const budgetMs = legBudgetMs(route.timeoutMs, execution.deadlineAtMs, now());
73798
74966
  if (budgetMs <= 0) {
73799
74967
  // The caller's deadline is spent. Dispatching now would start a call
73800
74968
  // that is cancelled the moment it begins, and charge nothing but noise.
73801
- const record = {
73802
- routeKey: route.routeKey,
73803
- role: route.role,
73804
- provider: route.providerName,
73805
- modelId: route.modelId,
74969
+ deadlineHit = true;
74970
+ record(route, {
74971
+ ...undispatched(route),
73806
74972
  outcome: "skipped",
73807
- durationMs: 0,
73808
74973
  reason: "caller deadline exhausted before this leg",
73809
- };
73810
- attempts.push(record);
73811
- execution.onAttempt?.(record);
74974
+ });
73812
74975
  continue;
73813
74976
  }
73814
- const startedAt = now();
73815
- const holdsProbe = execution.breakers.onAttemptStart(route.routeKey);
73816
- try {
73817
- const response = await runLeg(leg, leg.params, execution, budgetMs);
73818
- execution.breakers.onSuccess(route.routeKey);
73819
- totalUsage = sumUsage(totalUsage, response.usage);
73820
- const record = {
73821
- routeKey: route.routeKey,
73822
- role: route.role,
73823
- provider: route.providerName,
73824
- modelId: route.modelId,
73825
- outcome: "ok",
73826
- durationMs: now() - startedAt,
73827
- budgetMs,
73828
- servedModel: response.servedModel ?? null,
73829
- usage: response.usage,
73830
- };
73831
- attempts.push(record);
73832
- execution.onAttempt?.(record);
73833
- return { response, servedBy: route, attempts, totalUsage };
74977
+ lastModelClass = modelClassOf(route);
74978
+ const group = await runSameModelGroup(leg, budgetMs, budgetMs < route.timeoutMs, {
74979
+ request: execution,
74980
+ breakers: execution.breakers,
74981
+ now,
74982
+ policy: execution.hedging,
74983
+ tracker: execution.latency,
74984
+ promptTokens,
74985
+ admitDuplicate: execution.admitDuplicate ?? (() => false),
74986
+ record: (attemptRoute, fields, dispatch, servedProvider) => {
74987
+ record(attemptRoute, fields, dispatch, servedProvider);
74988
+ },
74989
+ nextAttemptIndex: () => {
74990
+ const index = dispatchIndex;
74991
+ dispatchIndex += 1;
74992
+ return index;
74993
+ },
74994
+ });
74995
+ for (const usage of group.billed) {
74996
+ totalUsage = sumUsage(totalUsage, usage);
73834
74997
  }
73835
- catch (error) {
73836
- const { outcome, reason, countsAgainstHealth, failureKind } = classify(error, execution.callerSignal);
73837
- if (countsAgainstHealth) {
73838
- execution.breakers.onFailure(route.routeKey, failureKind);
73839
- }
73840
- else if (holdsProbe) {
73841
- // No verdict on the route's health, but the probe slot this attempt
73842
- // took must come back, or a half-open route admits no probe ever again.
73843
- execution.breakers.onAttemptAbandoned(route.routeKey);
73844
- }
73845
- // A provider that answered — with unparseable content, or in prose where a
73846
- // tool call was mandatory — still billed for the answer; the spend belongs
73847
- // in the total whether or not a later leg serves.
73848
- const billed = billedUsageOf(error);
73849
- const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
73850
- totalUsage = sumUsage(totalUsage, billed);
73851
- const record = {
73852
- routeKey: route.routeKey,
73853
- role: route.role,
73854
- provider: route.providerName,
73855
- modelId: route.modelId,
73856
- outcome,
73857
- durationMs: now() - startedAt,
73858
- budgetMs,
73859
- reason,
73860
- ...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
73861
- ...(billed === undefined ? {} : { usage: billed }),
74998
+ deadlineHit = deadlineHit || group.deadlineBound;
74999
+ if (group.answer !== undefined) {
75000
+ const winner = attempts.find((attempt) => attempt.outcome === "ok" && attempt.routeKey === group.answer?.route.routeKey);
75001
+ return {
75002
+ response: group.answer.response,
75003
+ servedBy: group.answer.route,
75004
+ attempts,
75005
+ totalUsage,
75006
+ modelClassRelation: winner?.modelClassRelation ?? "unknown",
75007
+ hedged: group.answer.hedged,
73862
75008
  };
73863
- attempts.push(record);
73864
- execution.onAttempt?.(record);
73865
- if (outcome === "skipped" && isAborted(execution.callerSignal)) {
73866
- break;
73867
- }
73868
75009
  }
75010
+ if (isAborted(execution.callerSignal)) {
75011
+ break;
75012
+ }
75013
+ }
75014
+ if (deadlineHit && !isAborted(execution.callerSignal) && execution.deadlineAtMs !== undefined) {
75015
+ throw new LlmDeadlineExceededError(alias, attempts, totalUsage, execution.deadlineAtMs - startedAt, lastModelClass);
73869
75016
  }
73870
- throw new ChainExhaustedError(alias, attempts, totalUsage);
75017
+ throw new ChainExhaustedError(alias, attempts, totalUsage, crossModelDenied ? "cross_model_denied" : "exhausted");
73871
75018
  }
73872
75019
 
73873
75020
  var schema_version = 1;
@@ -73884,7 +75031,37 @@ var defaults = {
73884
75031
  failure_threshold: 5,
73885
75032
  cooldown_ms: 60000,
73886
75033
  capacity_cooldown_ms: 15000,
73887
- half_open_probes: 1
75034
+ half_open_probes: 1,
75035
+ probe_fraction: 0.1,
75036
+ latency_trip: {
75037
+ enabled: false,
75038
+ slo_ms: {
75039
+ "hot-path": 20000,
75040
+ background: 60000,
75041
+ batch: 240000
75042
+ },
75043
+ quantile: 0.9,
75044
+ window_size: 20,
75045
+ trip_windows: 3,
75046
+ of_windows: 5
75047
+ }
75048
+ },
75049
+ hedging: {
75050
+ max_same_model_hedges: 1,
75051
+ hedge_quantile: 0.9,
75052
+ timeout_quantile: 0.99,
75053
+ k_timeout: 3,
75054
+ attempt_timeout_floor_ms: 5000,
75055
+ max_attempt_share: 0.5,
75056
+ duplicate_headroom_reserve: 0.25,
75057
+ min_samples: 20,
75058
+ window_size: 200,
75059
+ sample_max_age_ms: 900000,
75060
+ prompt_token_buckets: [
75061
+ 2000,
75062
+ 8000,
75063
+ 32000
75064
+ ]
73888
75065
  }
73889
75066
  };
73890
75067
  var providers = {
@@ -74620,6 +75797,86 @@ const ISOLATED_SUFFIX = ".isolated";
74620
75797
  * caller change every other caller's routing.
74621
75798
  */
74622
75799
  const routeTable = rawTable;
75800
+ /** Separates a leg's route key from the provider of one of its equivalents. */
75801
+ const EQUIVALENT_SEPARATOR = "~";
75802
+ /** Gateway name segment for an equivalent leg. */
75803
+ const EQUIVALENT_NAMESPACE = "equivalent";
75804
+ /**
75805
+ * Whether a number lies in a closed or open range.
75806
+ *
75807
+ * @param value The value.
75808
+ * @param min Lower bound.
75809
+ * @param max Upper bound.
75810
+ * @param open Whether the bounds are excluded.
75811
+ * @returns Whether it is in range.
75812
+ */
75813
+ function inRange(value, min, max, open) {
75814
+ if (typeof value !== "number" || !Number.isFinite(value)) {
75815
+ return false;
75816
+ }
75817
+ return open ? value > min && value < max : value >= min && value <= max;
75818
+ }
75819
+ /**
75820
+ * Bounds violations in the tail-latency defaults (hedging, probe scaling and
75821
+ * the latency trip).
75822
+ *
75823
+ * These values are mechanics rather than routing choices, but a value outside
75824
+ * its bounds turns a mechanic into a routing change — a quantile of 1.0 makes
75825
+ * every hedge wait for the slowest answer ever seen, a zero floor lets a
75826
+ * measured timeout cut an attempt the instant it starts — so they are checked
75827
+ * as strictly as the routing itself.
75828
+ *
75829
+ * @param defaults The table's defaults.
75830
+ * @returns One message per violation; empty when every value is in bounds.
75831
+ */
75832
+ function tailLatencyViolations(defaults) {
75833
+ const violations = [];
75834
+ const check = (ok, message) => {
75835
+ if (!ok) {
75836
+ violations.push(message);
75837
+ }
75838
+ };
75839
+ const hedging = defaults.hedging;
75840
+ if (hedging !== undefined) {
75841
+ check(Number.isInteger(hedging.max_same_model_hedges) && inRange(hedging.max_same_model_hedges, 0, 3, false), "hedging.max_same_model_hedges must be an integer in [0, 3]");
75842
+ check(inRange(hedging.hedge_quantile, 0.5, 1, true), "hedging.hedge_quantile must be in (0.5, 1)");
75843
+ check(inRange(hedging.timeout_quantile, 0.5, 1, true), "hedging.timeout_quantile must be in (0.5, 1)");
75844
+ check(hedging.timeout_quantile >= hedging.hedge_quantile, "hedging.timeout_quantile must not be below hedging.hedge_quantile");
75845
+ check(inRange(hedging.k_timeout, 1, 10, false), "hedging.k_timeout must be in [1, 10]");
75846
+ check(inRange(hedging.attempt_timeout_floor_ms, 1000, Number.MAX_SAFE_INTEGER, false), "hedging.attempt_timeout_floor_ms must be at least 1000");
75847
+ check(inRange(hedging.max_attempt_share, 0.25, 1, false), "hedging.max_attempt_share must be in [0.25, 1]");
75848
+ check(inRange(hedging.duplicate_headroom_reserve, 0, 0.9, false), "hedging.duplicate_headroom_reserve must be in [0, 0.9]");
75849
+ check(Number.isInteger(hedging.min_samples) && inRange(hedging.min_samples, 5, 10_000, false), "hedging.min_samples must be an integer in [5, 10000]");
75850
+ check(Number.isInteger(hedging.window_size) && hedging.window_size >= hedging.min_samples, "hedging.window_size must be an integer no smaller than min_samples");
75851
+ check(inRange(hedging.sample_max_age_ms, 60_000, Number.MAX_SAFE_INTEGER, false), "hedging.sample_max_age_ms must be at least 60000");
75852
+ check(hedging.prompt_token_buckets.every((edge, index, edges) => Number.isFinite(edge) && edge > 0 && (index === 0 || edge > edges[index - 1])), "hedging.prompt_token_buckets must be positive and strictly ascending");
75853
+ }
75854
+ const breaker = defaults.circuit_breaker;
75855
+ if (breaker.probe_fraction !== undefined) {
75856
+ check(inRange(breaker.probe_fraction, 0, 0.5, false), "circuit_breaker.probe_fraction must be in [0, 0.5]");
75857
+ }
75858
+ const trip = breaker.latency_trip;
75859
+ if (trip !== undefined) {
75860
+ check(inRange(trip.quantile, 0.5, 1, true), "circuit_breaker.latency_trip.quantile must be in (0.5, 1)");
75861
+ check(Number.isInteger(trip.window_size) && trip.window_size >= 5, "circuit_breaker.latency_trip.window_size must be an integer of at least 5");
75862
+ check(Number.isInteger(trip.of_windows) &&
75863
+ Number.isInteger(trip.trip_windows) &&
75864
+ trip.trip_windows >= 1 &&
75865
+ trip.trip_windows <= trip.of_windows, "circuit_breaker.latency_trip needs integer 1 <= trip_windows <= of_windows");
75866
+ for (const latencyClass of Object.keys(defaults.request_timeout_ms)) {
75867
+ check(inRange(trip.slo_ms[latencyClass], 1000, defaults.request_timeout_ms[latencyClass], false), `circuit_breaker.latency_trip.slo_ms.${latencyClass} must be in [1000, its request timeout]`);
75868
+ }
75869
+ }
75870
+ return violations;
75871
+ }
75872
+ {
75873
+ // Checked when the table loads: it is bundled, so a violation can only
75874
+ // arrive in a release, and it must stop that release rather than run it.
75875
+ const violations = tailLatencyViolations(routeTable.defaults);
75876
+ if (violations.length > 0) {
75877
+ throw new Error(`alias route table has out-of-bounds tail-latency defaults: ${violations.join("; ")}`);
75878
+ }
75879
+ }
74623
75880
  /**
74624
75881
  * Every alias the table defines.
74625
75882
  *
@@ -74799,6 +76056,8 @@ function resolveChain(alias, options = {}) {
74799
76056
  });
74800
76057
  continue;
74801
76058
  }
76059
+ const routeKey = routeKeyFor(alias, isolated, route.role);
76060
+ const modelClass = route.model_class ?? admission.modelId;
74802
76061
  resolved.push({
74803
76062
  alias,
74804
76063
  isolated,
@@ -74808,13 +76067,67 @@ function resolveChain(alias, options = {}) {
74808
76067
  modelId: admission.modelId,
74809
76068
  lumicModel: route.lumic_model ?? null,
74810
76069
  params: route.params ?? {},
74811
- routeKey: routeKeyFor(alias, isolated, route.role),
76070
+ routeKey,
74812
76071
  timeoutMs,
74813
- retriesPerLeg: routeTable.defaults.retries_per_leg,
76072
+ modelClass,
76073
+ latencyClass: definition.latency_class,
76074
+ equivalents: resolveEquivalents(route, {
76075
+ alias,
76076
+ isolated,
76077
+ routeKey,
76078
+ timeoutMs,
76079
+ modelClass,
76080
+ latencyClass: definition.latency_class,
76081
+ }),
74814
76082
  });
74815
76083
  }
74816
76084
  return { alias, isolated, routes: resolved, exclusions };
74817
76085
  }
76086
+ /**
76087
+ * Resolve a leg's same-model equivalents that can serve today.
76088
+ *
76089
+ * An equivalent is admitted by the leg's own rules — a confirmed model id and a
76090
+ * live provider account — and inherits the leg's model class, because being the
76091
+ * same model is what makes it an equivalent. One that cannot serve is simply
76092
+ * absent: it is never a reason to skip the leg.
76093
+ *
76094
+ * @param route The authored leg.
76095
+ * @param leg The resolved leg's identity.
76096
+ * @param leg.alias The alias.
76097
+ * @param leg.isolated Whether this is the isolated variant.
76098
+ * @param leg.routeKey The leg's route key.
76099
+ * @param leg.timeoutMs The leg's budget.
76100
+ * @param leg.modelClass The leg's model class.
76101
+ * @param leg.latencyClass The alias's latency class.
76102
+ * @returns The servable equivalents, in table order.
76103
+ */
76104
+ function resolveEquivalents(route, leg) {
76105
+ const resolved = [];
76106
+ for (const equivalent of route.equivalents ?? []) {
76107
+ const provider = routeTable.providers[equivalent.provider];
76108
+ if (provider === undefined ||
76109
+ provider.account_status !== "live" ||
76110
+ equivalent.model_id_status !== "confirmed" ||
76111
+ equivalent.model_id === null) {
76112
+ continue;
76113
+ }
76114
+ resolved.push({
76115
+ alias: leg.alias,
76116
+ isolated: leg.isolated,
76117
+ role: route.role,
76118
+ providerName: equivalent.provider,
76119
+ provider,
76120
+ modelId: equivalent.model_id,
76121
+ lumicModel: equivalent.lumic_model ?? null,
76122
+ params: equivalent.params ?? route.params ?? {},
76123
+ routeKey: `${leg.routeKey}${EQUIVALENT_SEPARATOR}${equivalent.provider}`,
76124
+ timeoutMs: leg.timeoutMs,
76125
+ modelClass: leg.modelClass,
76126
+ latencyClass: leg.latencyClass,
76127
+ });
76128
+ }
76129
+ return resolved;
76130
+ }
74818
76131
  /**
74819
76132
  * The permanent closed-incumbent leg of an alias (PD-11).
74820
76133
  *
@@ -74843,6 +76156,12 @@ function closedIncumbentLeg(chain) {
74843
76156
  */
74844
76157
  function gatewayModelNameFor(route, chain) {
74845
76158
  const base = `${route.alias}${route.isolated ? ISOLATED_SUFFIX : ""}`;
76159
+ if (route.routeKey.includes(EQUIVALENT_SEPARATOR)) {
76160
+ // The same model at another provider is its own gateway deployment, named
76161
+ // so the gateway serves exactly that deployment and nothing it might fall
76162
+ // back to on its own.
76163
+ return `${base}.${EQUIVALENT_NAMESPACE}.${route.role}.${route.providerName}`;
76164
+ }
74846
76165
  return chain.routes[0]?.routeKey === route.routeKey
74847
76166
  ? base
74848
76167
  : `${base}.fallback.${route.role}`;
@@ -75130,8 +76449,11 @@ async function resolveDefaultDirectCaller() {
75130
76449
  * gateway by its own model name, so the gateway serves one deployment per leg
75131
76450
  * and needs no fallback of its own. A proxy-side fallback inside a leg would
75132
76451
  * spend the leg's budget on a model the chain did not choose and report the
75133
- * answer as the leg's; the served model is read from the response body so such
75134
- * a substitution stays visible while any remains configured.
76452
+ * answer as the leg's; the served model is read from the gateway's own
76453
+ * served-model header, else from the upstream body's `model`, so such a
76454
+ * substitution stays visible while any remains configured. A body `model` that
76455
+ * merely echoes the name the leg was addressed by is the gateway naming its
76456
+ * model GROUP, not the model that answered, and is reported as unknown.
75135
76457
  *
75136
76458
  * The gateway key is read from the environment by NAME at call time and never
75137
76459
  * stored, logged, or included in an error (PD-2). Reading it per call rather
@@ -75147,6 +76469,14 @@ const ERROR_BODY_EXCERPT = 400;
75147
76469
  const RESPONSE_COST_HEADER = "x-litellm-response-cost";
75148
76470
  /** Response header naming the proxy deployment that served the call. */
75149
76471
  const DEPLOYMENT_ID_HEADER = "x-litellm-model-id";
76472
+ /**
76473
+ * Response header in which the gateway reports the upstream model that
76474
+ * actually answered. Takes precedence over the body's `model`, which a proxy
76475
+ * may overwrite with the model-group name it was addressed by.
76476
+ */
76477
+ const SERVED_MODEL_HEADER = "x-adaptic-served-model";
76478
+ /** Response header in which the gateway reports the upstream provider that answered. */
76479
+ const SERVED_PROVIDER_HEADER = "x-adaptic-served-provider";
75150
76480
  /**
75151
76481
  * Thrown when the gateway itself is unreachable, as opposed to a provider
75152
76482
  * behind it failing.
@@ -75309,6 +76639,7 @@ function createGatewayTransport(config) {
75309
76639
  throw new GatewayResponseError(response.status, await response.text());
75310
76640
  }
75311
76641
  const payload = (await response.json());
76642
+ const addressedAs = body.model;
75312
76643
  const choices = payload.choices;
75313
76644
  const message = choices?.[0]?.message;
75314
76645
  // Usage is read before the content is interpreted. The provider billed for
@@ -75321,14 +76652,36 @@ function createGatewayTransport(config) {
75321
76652
  tool_calls: Array.isArray(message?.tool_calls)
75322
76653
  ? message.tool_calls
75323
76654
  : undefined,
75324
- // The model the provider says answered, which a proxy-side fallback can
75325
- // make differ from the leg's route model; unreported stays null.
75326
- servedModel: nonEmptyOrNull(payload.model),
76655
+ // The model that answered, which a proxy-side fallback can make differ
76656
+ // from the leg's route model; unreported (or only the group echo) stays null.
76657
+ servedModel: servedModelOf(response.headers, payload.model, addressedAs),
75327
76658
  servedDeploymentId: nonEmptyOrNull(response.headers?.get(DEPLOYMENT_ID_HEADER)),
76659
+ servedProvider: nonEmptyOrNull(response.headers?.get(SERVED_PROVIDER_HEADER)),
75328
76660
  };
75329
76661
  },
75330
76662
  };
75331
76663
  }
76664
+ /**
76665
+ * The model that answered, as far as the gateway says.
76666
+ *
76667
+ * The gateway's served-model header is authoritative when present. Otherwise
76668
+ * the body's `model` is used — unless it equals the name the leg was addressed
76669
+ * by, which is the proxy echoing its model group rather than naming the
76670
+ * upstream model, and would otherwise read as a confirmed same-model answer.
76671
+ *
76672
+ * @param headers The response headers.
76673
+ * @param bodyModel The body's `model` field.
76674
+ * @param addressedAs The gateway model name the leg was sent to.
76675
+ * @returns The served model, or null when the gateway did not say.
76676
+ */
76677
+ function servedModelOf(headers, bodyModel, addressedAs) {
76678
+ const fromHeader = nonEmptyOrNull(headers?.get(SERVED_MODEL_HEADER));
76679
+ if (fromHeader !== null) {
76680
+ return fromHeader;
76681
+ }
76682
+ const fromBody = nonEmptyOrNull(bodyModel);
76683
+ return fromBody === null || fromBody === addressedAs ? null : fromBody;
76684
+ }
75332
76685
  /**
75333
76686
  * Compose the message array for a request.
75334
76687
  *
@@ -75392,7 +76745,9 @@ function interpretContent(content, responseFormat, usage) {
75392
76745
  * Every call gets, in order: alias resolution against the canonical route
75393
76746
  * table, per-provider parameter normalisation, a hard per-leg timeout, a
75394
76747
  * per-route circuit breaker, an ordered fallback chain ending at the closed
75395
- * incumbent, and — where the caller supplies a validator — one schema-feedback
76748
+ * incumbent — each leg first hedged and failed over on the SAME model (see
76749
+ * `hedge.ts`), and a different-model leg reached only when the caller's
76750
+ * cross-model policy allows it — and, where the caller supplies a validator, one schema-feedback
75396
76751
  * retry ahead of the chain. The caller's `timeoutMs` is ONE deadline for the
75397
76752
  * whole call: each leg runs for its route budget or for what remains of that
75398
76753
  * deadline, whichever is shorter, so the chain is the single fallback owner and
@@ -75408,6 +76763,34 @@ const GATEWAY_BASE_URL_ENV = "LLM_GATEWAY_BASE_URL";
75408
76763
  const DEFAULT_GATEWAY_KEY_ENV = "LLM_GATEWAY_API_KEY";
75409
76764
  /** Process-wide breaker registry, so route health is shared across call sites. */
75410
76765
  let breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker);
76766
+ /**
76767
+ * Build the process-wide latency tracker from the table's hedging defaults.
76768
+ *
76769
+ * @param now Clock.
76770
+ * @returns The tracker, or undefined when the table configures no hedging.
76771
+ */
76772
+ function buildLatencyTracker(now) {
76773
+ const hedging = routeTable.defaults.hedging;
76774
+ if (hedging === undefined) {
76775
+ return undefined;
76776
+ }
76777
+ return new LegLatencyTracker({
76778
+ minSamples: hedging.min_samples,
76779
+ windowSize: hedging.window_size,
76780
+ sampleMaxAgeMs: hedging.sample_max_age_ms,
76781
+ promptTokenBuckets: hedging.prompt_token_buckets,
76782
+ }, now);
76783
+ }
76784
+ /**
76785
+ * Process-wide healthy-latency evidence, shared across call sites for the same
76786
+ * reason the breakers are: a model's health is one population however many
76787
+ * callers reach it.
76788
+ */
76789
+ let latencyTracker = buildLatencyTracker();
76790
+ /** Same-model hedging policy from the table, or undefined when none is configured. */
76791
+ const hedgingPolicy = routeTable.defaults.hedging === undefined
76792
+ ? undefined
76793
+ : sameModelPolicyFrom(routeTable.defaults.hedging);
75411
76794
  /** Active runtime wiring. */
75412
76795
  let config = {};
75413
76796
  /** Lazily built transports, rebuilt whenever configuration changes. */
@@ -75429,6 +76812,28 @@ function configureLlmClient(next) {
75429
76812
  gatewayTransport = null;
75430
76813
  directTransport = null;
75431
76814
  breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker, next.now);
76815
+ latencyTracker = buildLatencyTracker(next.now);
76816
+ }
76817
+ /**
76818
+ * Inspect the healthy-latency evidence the hedging controls read.
76819
+ *
76820
+ * @returns The live tracker, or undefined when the table configures no hedging.
76821
+ */
76822
+ function llmLatencyTracker() {
76823
+ return latencyTracker;
76824
+ }
76825
+ /**
76826
+ * Whether a duplicate attempt on the same provider may start.
76827
+ *
76828
+ * @param route The leg the duplicate would address.
76829
+ * @param reserveFraction Share of the provider's capacity kept free.
76830
+ * @returns Whether it may start.
76831
+ */
76832
+ function admitDuplicate(route, reserveFraction) {
76833
+ if (config.duplicateAdmission !== undefined) {
76834
+ return config.duplicateAdmission(route, reserveFraction);
76835
+ }
76836
+ return hasDuplicateHeadroom(route.providerName, route.modelId, reserveFraction);
75432
76837
  }
75433
76838
  /**
75434
76839
  * Inspect route health.
@@ -75501,10 +76906,11 @@ function directFor() {
75501
76906
  * @param options The caller's options.
75502
76907
  * @param responseFormat The requested response shape.
75503
76908
  * @param transport The transport to carry every leg.
76909
+ * @param admitEquivalent Which same-model equivalents this transport can reach.
75504
76910
  * @returns Prepared legs, in chain order.
75505
76911
  */
75506
- function prepareLegs(chain, options, responseFormat, transport) {
75507
- return chain.routes.map((route) => {
76912
+ function prepareLegs(chain, options, responseFormat, transport, admitEquivalent = () => true) {
76913
+ const prepare = (route) => {
75508
76914
  try {
75509
76915
  return {
75510
76916
  route,
@@ -75518,7 +76924,11 @@ function prepareLegs(chain, options, responseFormat, transport) {
75518
76924
  }
75519
76925
  throw error;
75520
76926
  }
75521
- });
76927
+ };
76928
+ return chain.routes.map((route) => ({
76929
+ ...prepare(route),
76930
+ equivalents: (route.equivalents ?? []).filter(admitEquivalent).map(prepare),
76931
+ }));
75522
76932
  }
75523
76933
  /**
75524
76934
  * Call a language model by semantic alias.
@@ -75562,6 +76972,26 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75562
76972
  }
75563
76973
  const gateway = gatewayFor();
75564
76974
  const attemptLog = [];
76975
+ // What every execution of this call shares, whichever transport carries it.
76976
+ // The configured model is the head of the full chain, so the degraded path —
76977
+ // whose legs are only the closed incumbents — still knows that its answer is
76978
+ // a different model from the one configured.
76979
+ const shared = {
76980
+ responseFormat,
76981
+ developerPrompt: options.developerPrompt,
76982
+ context: options.context,
76983
+ breakers,
76984
+ correlationId: options.correlationId,
76985
+ callerSignal: options.signal,
76986
+ deadlineAtMs,
76987
+ now: config.now,
76988
+ onAttempt: (record) => attemptLog.push(record),
76989
+ hedging: hedgingPolicy,
76990
+ latency: latencyTracker,
76991
+ admitDuplicate,
76992
+ crossModelPolicy: options.crossModelPolicy,
76993
+ configuredModelClass: modelClassOf(chain.routes[0]),
76994
+ };
75565
76995
  /**
75566
76996
  * Run the chain, falling back from the gateway to the degraded direct path
75567
76997
  * only when the gateway itself is unreachable.
@@ -75574,17 +77004,9 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75574
77004
  if (gateway !== null) {
75575
77005
  try {
75576
77006
  const outcome = await executeChain(options.alias, {
77007
+ ...shared,
75577
77008
  legs: prepareLegs(chain, options, responseFormat, gateway),
75578
77009
  content: boundContent,
75579
- responseFormat,
75580
- developerPrompt: options.developerPrompt,
75581
- context: options.context,
75582
- breakers,
75583
- correlationId: options.correlationId,
75584
- callerSignal: options.signal,
75585
- deadlineAtMs,
75586
- now: config.now,
75587
- onAttempt: (record) => attemptLog.push(record),
75588
77010
  });
75589
77011
  return { ...outcome, degraded: false };
75590
77012
  }
@@ -75609,17 +77031,11 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75609
77031
  });
75610
77032
  }
75611
77033
  const outcome = await executeChain(options.alias, {
75612
- legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct),
77034
+ ...shared,
77035
+ // The direct transport serves closed vendors only, so only a closed
77036
+ // equivalent can be reached on the degraded path.
77037
+ legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct, (route) => route.provider.tier === "closed"),
75613
77038
  content: boundContent,
75614
- responseFormat,
75615
- developerPrompt: options.developerPrompt,
75616
- context: options.context,
75617
- breakers,
75618
- correlationId: options.correlationId,
75619
- callerSignal: options.signal,
75620
- deadlineAtMs,
75621
- now: config.now,
75622
- onAttempt: (record) => attemptLog.push(record),
75623
77039
  });
75624
77040
  return { ...outcome, degraded: true };
75625
77041
  };
@@ -75634,6 +77050,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75634
77050
  attempts: attemptLog,
75635
77051
  degraded: outcome.degraded,
75636
77052
  totalUsage: outcome.totalUsage,
77053
+ modelClassRelation: outcome.modelClassRelation,
77054
+ hedged: outcome.hedged,
75637
77055
  };
75638
77056
  }
75639
77057
  // A validator is only meaningful against a text prompt, because the retry has
@@ -75651,6 +77069,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75651
77069
  servedBy: outcome.servedBy,
75652
77070
  degraded: outcome.degraded,
75653
77071
  totalUsage: outcome.totalUsage,
77072
+ modelClassRelation: outcome.modelClassRelation,
77073
+ hedged: outcome.hedged,
75654
77074
  };
75655
77075
  return outcome.response;
75656
77076
  },
@@ -75668,6 +77088,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75668
77088
  attempts: attemptLog,
75669
77089
  degraded: routing.degraded,
75670
77090
  totalUsage: validated.totalUsage,
77091
+ modelClassRelation: routing.modelClassRelation,
77092
+ hedged: routing.hedged,
75671
77093
  };
75672
77094
  }
75673
77095
  /**
@@ -81087,5 +82509,5 @@ const adaptic = {
81087
82509
  };
81088
82510
  const adptc = adaptic;
81089
82511
 
81090
- export { API_RETRY_CONFIGS, AVNewsArticleSchema, AVNewsResponseSchema, AdapticUtilsError, AlpacaAccountDetailsSchema, AlpacaApiError, AlpacaBarSchema, AlpacaClient, AlpacaCryptoBarsResponseSchema, AlpacaHistoricalBarsResponseSchema, AlpacaLatestBarsResponseSchema, AlpacaLatestQuotesResponseSchema, AlpacaLatestTradesResponseSchema, AlpacaMarketDataAPI, AlpacaNewsArticleSchema, AlpacaNewsResponseSchema, AlpacaOrderSchema, AlpacaOrdersArraySchema, AlpacaPortfolioHistoryResponseSchema, AlpacaPositionSchema, AlpacaPositionsArraySchema, AlpacaQuoteSchema, AlpacaTradeSchema, AlpacaTradingAPI, AlphaVantageError, AlphaVantageQuoteResponseSchema, AssetAllocationEngine, AuthenticationError, AutonomyMode, BTC_PAIRS, BarError, ChainExhaustedError, CircuitBreakerRegistry, CircuitOpenError, CryptoDataError, CryptoOrderError, DEFAULT_CACHE_OPTIONS, DEFAULT_RISK_FREE_RATE, DEFAULT_TIMEOUTS, DEFAULT_TRADING_POLICY, DataFormatError, DecisionMemoryOutcome, DecisionOutcome, DecisionRecordStatus, DirectTransportRefusedError, DuplicateClientOrderIdError, GatewayResponseError, GatewayUnreachableError, HttpClientError, HttpServerError, KEEP_ALIVE_DEFAULTS, LlmProvider, LlmResponseFormatError, MARKET_DATA_API, MassiveAggregatesResponseSchema, MassiveApiError, MassiveDailyOpenCloseSchema, MassiveErrorResponseSchema, MassiveGroupedDailyResponseSchema, MassiveLastTradeResponseSchema, MassiveTickerDetailsResponseSchema, MassiveTickerInfoSchema, MassiveTradeSchema as MassiveTradeZodSchema, MassiveTradesResponseSchema, NetworkError, NewsError, NoServableRouteError, OptionStrategyError, OptionsDataError, OverlaySeverity, OverlayStatus, OverlayType, PENDING_CANCEL_ERROR_CODE, PendingCancelError, QuoteError, RISK_FREE_RATE_TTL_MS, RateGuardTimeoutError, RateLimitError, RawMassivePriceDataSchema, SchemaRetryExhaustedError, StampedeProtectedCache, StreamProviderError, StreamTruncatedError, TRADING_API, TimeoutError, TokenBucketRateLimiter, ToolChoiceIgnoredError, TradeError, TrailingStopValidationError, USDC_PAIRS, USDT_PAIRS, USD_PAIRS, UnknownAliasError, UnsupportedBrokerError, UnsupportedCapabilityError, ValidationError, ValidationResponseError, WEBSOCKET_STREAMS, WebSocketError, account, adaptic, adptc, alpaca, analyzeBars, approximateImpliedVolatility, assertToolChoiceHonoured, atrNs as atr, availableStatistic, bracketOrders, buildOCCSymbol, buildOptionSymbol, buildRetryPrompt, buyCryptoNotional, buyToClose, buyToOpen, buyWithStopLoss, buyWithTrailingStop, calculateMoneyness, calculateOrderValue, calculatePeriodPerformance, calculatePutCallRatio, calculateTotalFilledValue, callLLMByAlias, callWithValidation, cancelAllCryptoOrders, cancelOCOOrder, cancelOTOOrder, cancelTrailingStop, cancelTrailingStopsForSymbol, checkTradingEligibility, clearClientCache, clock, closeAllOptionPositions, closeOptionPosition, closedIncumbentLeg, collectStream, configureLlmClient, createAlpacaClient, createAlpacaMarketDataAPI, createAlpacaTradingAPI, createBracketOrder, createBrokerClient, createButterflySpread, createClientFromEnv, createCoveredCall, createCryptoLimitOrder, createCryptoMarketOrder, createCryptoOrder, createCryptoStopLimitOrder, createCryptoStopOrder, createDirectTransport, createExecutorFromTradingAPI, createGatewayTransport, createIronCondor$1 as createIronCondor, createIronCondor as createIronCondorAdvanced, createMultiLegOptionOrder, createOCOOrder, createOTOOrder, createOptionOrder, createPortfolioTrailingStops, createProtectiveBracket, createStampedeProtectedCache, createStraddle$1 as createStraddle, createStraddle as createStraddleAdvanced, createStrangle$1 as createStrangle, createStrangle as createStrangleAdvanced, createStreamManager, createTimeoutSignal, createTrailingStop, createVerticalSpread$1 as createVerticalSpread, createVerticalSpread as createVerticalSpreadAdvanced, enrichAlpacaError, entryWithPercentStopLoss, exerciseOption, extractAlpacaBrokerError, extractGreeks, filterByExpiration, filterByStrike, filterByType, filterOrdersByDateRange, findATMOptions, findATMStrikes, findNearestExpiration, findOptionsByDelta, formatOrderForLog, formatOrderSummary, gatewayModelNameFor, generateOptimalAllocation, getAccountConfiguration, getAccountDetails, getAccountSummary, getAgentPoolStatus, getAllOrders, getAlpacaBrokerErrorCode, getAlpacaBrokerErrorDetail, getAlpacaCalendar, getAlpacaClock, getAverageDailyVolume, getBars, getBuyingPower, getCachedRiskFreeRateSync, getCachedRiskFreeRateSyncWithProvenance, getCrypto24HourChange, getCryptoBars, getCryptoDailyPrices, getCryptoPairsByQuote, getCryptoPrice, getCryptoSnapshots, getCryptoSpread, getCryptoStreamUrl, getCryptoTrades, getCurrentPrice, getCurrentPrices, getDailyPrices, getDailyReturns, getDaysToExpiration, getDefaultRiskProfile, getEquityCurve, getExpirationDates, getFilledOrders, getGroupedOptionChain, getHistoricalOptionsBars, getHistoricalTrades, getIntradayPrices, getLatestBars, getLatestCryptoQuotes, getLatestCryptoTrades, getLatestNews, getLatestOptionsQuotes, getLatestOptionsTrades, getLatestQuote, getLatestQuotes, getLatestTrade, getLatestTrades, getLogger, getMarginInfo, getNews, getNewsForSymbols, getOCOOrderStatus, getOTOOrderStatus, getOpenCryptoOrders, getOpenOrders$1 as getOpenOrdersQuery, getOpenTrailingStops, getOptionChain, getOptionContract, getOptionContracts, getOptionSpread, getOptionsChain, getOptionsSnapshots, getOptionsStreamUrl, getOptionsTradingLevel, getOrderHistory, getOrdersBySymbol, getPDTStatus, getPopularCryptoPairs, getPortfolioHistory, getPreviousClose, getPriceRange, getRiskFreeRate, getRiskFreeRateWithProvenance, getSpread, getSpreads, getStockStreamUrl, getStrikePrices, getSupportedCryptoPairs, getSymbolSentiment, getTimeout, getTradeVolume, getTradingApiUrl, getTradingWebSocketUrl, getTrailingStopHWM, groupOrdersByStatus, groupOrdersBySymbol, guardSnapshots, hasActiveTrailingStop, hasGoodLiquidity as hasOptionLiquidity, hasGoodLiquidity$1 as hasStockLiquidity, hasSufficientVolume, httpAgent, httpsAgent, isAlpacaBrokerCredentials, isAvailable, isContractTradable, isCryptoPair, isExpiringWithin, isMarginAccount, isOptionOrderCancelable, isOptionOrderTerminal, isOrderFillable, isOrderFilled, isOrderOpen, isOrderTerminal$1 as isOrderTerminalStatus, isPendingCancelRejection, isSupportedCryptoPair, isTransientNetworkError, legBudgetMs, index$1 as legacyApi, limitBuyWithTakeProfit, limitsFor, limitsInventory, listAliases, llmAliases, llmBreakers, normaliseAnthropicStream, normaliseOpenAiStream, normaliseParams, normaliseStream, ocoOrders, orderUtils, orderedRoutes, otoOrders, paginate, paginateAll, parseOCCSymbol, protectLongPosition, protectShortPosition, rateLimiters$1 as rateLimiters, resetLogger, resetProviderGuards, resetRiskFreeRateCache, resolveChain, resolveDefaultDirectCaller, riskNs as risk, rollOptionPosition, roundPriceForAlpaca$3 as roundPriceForAlpaca, roundPriceForAlpacaNumber, routeKeyFor, routeSupports, routeTable, safeValidateResponse, sampleCohort, searchNews, sellAllCrypto, sellCryptoNotional, sellToClose, sellToOpen, setLogger, setRiskFreeRate, shortWithStopLoss, sortOrdersByDate, strategyNs as strategy, sumUsage, index as tradingPolicy, trailingStops, unavailableStatistic, updateAccountConfiguration, updateTrailingStop, validateAlpacaCredentials, validateAlphaVantageApiKey, validateMassiveApiKey$1 as validateMassiveApiKey, validateMultiLegOrder, validateResponse, verifyFetchKeepAlive, volatilityNs as volatility, waitForOrderFill, withProviderGuards, withRetry, withTimeout };
82512
+ export { API_RETRY_CONFIGS, AVNewsArticleSchema, AVNewsResponseSchema, AdapticUtilsError, AlpacaAccountDetailsSchema, AlpacaApiError, AlpacaBarSchema, AlpacaClient, AlpacaCryptoBarsResponseSchema, AlpacaHistoricalBarsResponseSchema, AlpacaLatestBarsResponseSchema, AlpacaLatestQuotesResponseSchema, AlpacaLatestTradesResponseSchema, AlpacaMarketDataAPI, AlpacaNewsArticleSchema, AlpacaNewsResponseSchema, AlpacaOrderSchema, AlpacaOrdersArraySchema, AlpacaPortfolioHistoryResponseSchema, AlpacaPositionSchema, AlpacaPositionsArraySchema, AlpacaQuoteSchema, AlpacaTradeSchema, AlpacaTradingAPI, AlphaVantageError, AlphaVantageQuoteResponseSchema, AssetAllocationEngine, AuthenticationError, AutonomyMode, BTC_PAIRS, BarError, ChainExhaustedError, CircuitBreakerRegistry, CircuitOpenError, CryptoDataError, CryptoOrderError, DEFAULT_CACHE_OPTIONS, DEFAULT_RISK_FREE_RATE, DEFAULT_TIMEOUTS, DEFAULT_TRADING_POLICY, DataFormatError, DecisionMemoryOutcome, DecisionOutcome, DecisionRecordStatus, DirectTransportRefusedError, DuplicateClientOrderIdError, EQUIVALENT_SEPARATOR, GatewayResponseError, GatewayUnreachableError, HttpClientError, HttpServerError, KEEP_ALIVE_DEFAULTS, LegLatencyTracker, LlmDeadlineExceededError, LlmProvider, LlmResponseFormatError, MARKET_DATA_API, MIN_CONVERTED_TRAIL_PERCENT, MassiveAggregatesResponseSchema, MassiveApiError, MassiveDailyOpenCloseSchema, MassiveErrorResponseSchema, MassiveGroupedDailyResponseSchema, MassiveLastTradeResponseSchema, MassiveTickerDetailsResponseSchema, MassiveTickerInfoSchema, MassiveTradeSchema as MassiveTradeZodSchema, MassiveTradesResponseSchema, NetworkError, NewsError, NoServableRouteError, OptionStrategyError, OptionsDataError, OverlaySeverity, OverlayStatus, OverlayType, PENDING_CANCEL_ERROR_CODE, PendingCancelError, QuoteError, RISK_FREE_RATE_TTL_MS, RateGuardTimeoutError, RateLimitError, RawMassivePriceDataSchema, SERVED_MODEL_HEADER, SERVED_PROVIDER_HEADER, SchemaRetryExhaustedError, StampedeProtectedCache, StreamProviderError, StreamTruncatedError, TRADING_API, TimeoutError, TokenBucketRateLimiter, ToolChoiceIgnoredError, TradeError, TrailUnitConversionRefusedError, TrailingStopValidationError, USDC_PAIRS, USDT_PAIRS, USD_PAIRS, UnknownAliasError, UnsupportedBrokerError, UnsupportedCapabilityError, ValidationError, ValidationResponseError, WEBSOCKET_STREAMS, WebSocketError, account, adaptic, adptc, alpaca, analyzeBars, approximateImpliedVolatility, assertToolChoiceHonoured, atrNs as atr, availableStatistic, bracketOrders, buildOCCSymbol, buildOptionSymbol, buildRetryPrompt, buyCryptoNotional, buyToClose, buyToOpen, buyWithStopLoss, buyWithTrailingStop, calculateMoneyness, calculateOrderValue, calculatePeriodPerformance, calculatePutCallRatio, calculateTotalFilledValue, callLLMByAlias, callWithValidation, cancelAllCryptoOrders, cancelOCOOrder, cancelOTOOrder, cancelTrailingStop, cancelTrailingStopsForSymbol, checkTradingEligibility, clearClientCache, clock, closeAllOptionPositions, closeOptionPosition, closedIncumbentLeg, collectStream, configureLlmClient, createAlpacaClient, createAlpacaMarketDataAPI, createAlpacaTradingAPI, createBracketOrder, createBrokerClient, createButterflySpread, createClientFromEnv, createCoveredCall, createCryptoLimitOrder, createCryptoMarketOrder, createCryptoOrder, createCryptoStopLimitOrder, createCryptoStopOrder, createDirectTransport, createExecutorFromTradingAPI, createGatewayTransport, createIronCondor$1 as createIronCondor, createIronCondor as createIronCondorAdvanced, createMultiLegOptionOrder, createOCOOrder, createOTOOrder, createOptionOrder, createPortfolioTrailingStops, createProtectiveBracket, createStampedeProtectedCache, createStraddle$1 as createStraddle, createStraddle as createStraddleAdvanced, createStrangle$1 as createStrangle, createStrangle as createStrangleAdvanced, createStreamManager, createTimeoutSignal, createTrailingStop, createVerticalSpread$1 as createVerticalSpread, createVerticalSpread as createVerticalSpreadAdvanced, enrichAlpacaError, entryWithPercentStopLoss, estimatePromptTokens, exerciseOption, extractAlpacaBrokerError, extractGreeks, filterByExpiration, filterByStrike, filterByType, filterOrdersByDateRange, findATMOptions, findATMStrikes, findNearestExpiration, findOptionsByDelta, formatOrderForLog, formatOrderSummary, gatewayModelNameFor, generateOptimalAllocation, getAccountConfiguration, getAccountDetails, getAccountSummary, getAgentPoolStatus, getAllOrders, getAlpacaBrokerErrorCode, getAlpacaBrokerErrorDetail, getAlpacaCalendar, getAlpacaClock, getAverageDailyVolume, getBars, getBuyingPower, getCachedRiskFreeRateSync, getCachedRiskFreeRateSyncWithProvenance, getCrypto24HourChange, getCryptoBars, getCryptoDailyPrices, getCryptoPairsByQuote, getCryptoPrice, getCryptoSnapshots, getCryptoSpread, getCryptoStreamUrl, getCryptoTrades, getCurrentPrice, getCurrentPrices, getDailyPrices, getDailyReturns, getDaysToExpiration, getDefaultRiskProfile, getEquityCurve, getExpirationDates, getFilledOrders, getGroupedOptionChain, getHistoricalOptionsBars, getHistoricalTrades, getIntradayPrices, getLatestBars, getLatestCryptoQuotes, getLatestCryptoTrades, getLatestNews, getLatestOptionsQuotes, getLatestOptionsTrades, getLatestQuote, getLatestQuotes, getLatestTrade, getLatestTrades, getLogger, getMarginInfo, getNews, getNewsForSymbols, getOCOOrderStatus, getOTOOrderStatus, getOpenCryptoOrders, getOpenOrders$1 as getOpenOrdersQuery, getOpenTrailingStops, getOptionChain, getOptionContract, getOptionContracts, getOptionSpread, getOptionsChain, getOptionsSnapshots, getOptionsStreamUrl, getOptionsTradingLevel, getOrderHistory, getOrdersBySymbol, getPDTStatus, getPopularCryptoPairs, getPortfolioHistory, getPreviousClose, getPriceRange, getRiskFreeRate, getRiskFreeRateWithProvenance, getSpread, getSpreads, getStockStreamUrl, getStrikePrices, getSupportedCryptoPairs, getSymbolSentiment, getTimeout, getTradeVolume, getTradingApiUrl, getTradingWebSocketUrl, getTrailingStopHWM, groupOrdersByStatus, groupOrdersBySymbol, guardSnapshots, hasActiveTrailingStop, hasDuplicateHeadroom, hasGoodLiquidity as hasOptionLiquidity, hasGoodLiquidity$1 as hasStockLiquidity, hasSufficientVolume, httpAgent, httpsAgent, isAlpacaBrokerCredentials, isAvailable, isContractTradable, isCryptoPair, isExpiringWithin, isMarginAccount, isOptionOrderCancelable, isOptionOrderTerminal, isOrderFillable, isOrderFilled, isOrderOpen, isOrderTerminal$1 as isOrderTerminalStatus, isPendingCancelRejection, isSameReportedModel, isSupportedCryptoPair, isTransientNetworkError, legBudgetMs, index$1 as legacyApi, limitBuyWithTakeProfit, limitsFor, limitsInventory, listAliases, llmAliases, llmBreakers, llmLatencyTracker, modelClassOf, modelClassRelationOf, normaliseAnthropicStream, normaliseOpenAiStream, normaliseParams, normaliseStream, ocoOrders, orderUtils, orderedRoutes, otoOrders, paginate, paginateAll, parseOCCSymbol, protectLongPosition, protectShortPosition, rateLimiters$1 as rateLimiters, readTrailUnit, resetLogger, resetProviderGuards, resetRiskFreeRateCache, resolveChain, resolveDefaultDirectCaller, resolveReplaceTrail, riskNs as risk, rollOptionPosition, roundPriceForAlpaca$3 as roundPriceForAlpaca, roundPriceForAlpacaNumber, routeKeyFor, routeSupports, routeTable, safeValidateResponse, sampleCohort, searchNews, sellAllCrypto, sellCryptoNotional, sellToClose, sellToOpen, servedModelOf, setLogger, setRiskFreeRate, shortWithStopLoss, sortOrdersByDate, strategyNs as strategy, sumUsage, tailLatencyViolations, index as tradingPolicy, trailingStops, unavailableStatistic, updateAccountConfiguration, updateTrailingStop, validateAlpacaCredentials, validateAlphaVantageApiKey, validateMassiveApiKey$1 as validateMassiveApiKey, validateMultiLegOrder, validateResponse, verifyFetchKeepAlive, volatilityNs as volatility, waitForOrderFill, withProviderGuards, withRetry, withTimeout };
81091
82513
  //# sourceMappingURL=index.mjs.map