@adaptic/utils 0.0.1015 → 0.0.1017

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (155) hide show
  1. package/dist/index.cjs +2885 -25
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.mjs +2845 -25
  4. package/dist/index.mjs.map +1 -1
  5. package/dist/types/__tests__/llm/client/support/rejections.d.ts +21 -0
  6. package/dist/types/__tests__/llm/client/support/rejections.d.ts.map +1 -0
  7. package/dist/types/__tests__/llm/client/support/routes.d.ts +67 -0
  8. package/dist/types/__tests__/llm/client/support/routes.d.ts.map +1 -0
  9. package/dist/types/__tests__/llm/client/support/streams.d.ts +74 -0
  10. package/dist/types/__tests__/llm/client/support/streams.d.ts.map +1 -0
  11. package/dist/types/__tests__/llm/client/support/transports.d.ts +105 -0
  12. package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -0
  13. package/dist/types/index.d.ts +1 -0
  14. package/dist/types/index.d.ts.map +1 -1
  15. package/dist/types/llm/alias-client.d.ts +64 -0
  16. package/dist/types/llm/alias-client.d.ts.map +1 -0
  17. package/dist/types/llm/circuit-breaker.d.ts +124 -0
  18. package/dist/types/llm/circuit-breaker.d.ts.map +1 -0
  19. package/dist/types/llm/eval/comparators.d.ts +127 -0
  20. package/dist/types/llm/eval/comparators.d.ts.map +1 -0
  21. package/dist/types/llm/eval/coverage.d.ts +44 -0
  22. package/dist/types/llm/eval/coverage.d.ts.map +1 -0
  23. package/dist/types/llm/eval/golden-set.d.ts +47 -0
  24. package/dist/types/llm/eval/golden-set.d.ts.map +1 -0
  25. package/dist/types/llm/eval/index.d.ts +25 -0
  26. package/dist/types/llm/eval/index.d.ts.map +1 -0
  27. package/dist/types/llm/eval/json-shape.d.ts +74 -0
  28. package/dist/types/llm/eval/json-shape.d.ts.map +1 -0
  29. package/dist/types/llm/eval/judge.d.ts +131 -0
  30. package/dist/types/llm/eval/judge.d.ts.map +1 -0
  31. package/dist/types/llm/eval/metrics.d.ts +51 -0
  32. package/dist/types/llm/eval/metrics.d.ts.map +1 -0
  33. package/dist/types/llm/eval/run.d.ts +97 -0
  34. package/dist/types/llm/eval/run.d.ts.map +1 -0
  35. package/dist/types/llm/eval/types.d.ts +242 -0
  36. package/dist/types/llm/eval/types.d.ts.map +1 -0
  37. package/dist/types/llm/fallback-chain.d.ts +97 -0
  38. package/dist/types/llm/fallback-chain.d.ts.map +1 -0
  39. package/dist/types/llm/index.d.ts +30 -0
  40. package/dist/types/llm/index.d.ts.map +1 -0
  41. package/dist/types/llm/param-matrix.d.ts +65 -0
  42. package/dist/types/llm/param-matrix.d.ts.map +1 -0
  43. package/dist/types/llm/rate-guard.d.ts +119 -0
  44. package/dist/types/llm/rate-guard.d.ts.map +1 -0
  45. package/dist/types/llm/route-table.d.ts +187 -0
  46. package/dist/types/llm/route-table.d.ts.map +1 -0
  47. package/dist/types/llm/schema-retry.d.ts +80 -0
  48. package/dist/types/llm/schema-retry.d.ts.map +1 -0
  49. package/dist/types/llm/streaming.d.ts +93 -0
  50. package/dist/types/llm/streaming.d.ts.map +1 -0
  51. package/dist/types/llm/transports/direct.d.ts +92 -0
  52. package/dist/types/llm/transports/direct.d.ts.map +1 -0
  53. package/dist/types/llm/transports/gateway.d.ts +73 -0
  54. package/dist/types/llm/transports/gateway.d.ts.map +1 -0
  55. package/dist/types/llm/types.d.ts +292 -0
  56. package/dist/types/llm/types.d.ts.map +1 -0
  57. package/dist/types/schemas/alpaca-schemas.d.ts +6 -6
  58. package/dist/types/trading-policy/schemas/effective-policy.schema.d.ts +18 -18
  59. package/dist/types/trading-policy/schemas/model-prefs.schema.d.ts +24 -24
  60. package/dist/types/trading-policy/schemas/policy-mutation.schema.d.ts +36 -36
  61. package/package.json +1 -1
  62. package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts +0 -2
  63. package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts.map +0 -1
  64. package/dist/types/__tests__/alpaca-client-order-id.test.d.ts +0 -2
  65. package/dist/types/__tests__/alpaca-client-order-id.test.d.ts.map +0 -1
  66. package/dist/types/__tests__/alpaca-functions.test.d.ts +0 -2
  67. package/dist/types/__tests__/alpaca-functions.test.d.ts.map +0 -1
  68. package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts +0 -2
  69. package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts.map +0 -1
  70. package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts +0 -2
  71. package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts.map +0 -1
  72. package/dist/types/__tests__/alpaca-trading-api.test.d.ts +0 -2
  73. package/dist/types/__tests__/alpaca-trading-api.test.d.ts.map +0 -1
  74. package/dist/types/__tests__/api-endpoints.test.d.ts +0 -2
  75. package/dist/types/__tests__/api-endpoints.test.d.ts.map +0 -1
  76. package/dist/types/__tests__/asset-allocation.test.d.ts +0 -2
  77. package/dist/types/__tests__/asset-allocation.test.d.ts.map +0 -1
  78. package/dist/types/__tests__/atr.test.d.ts +0 -2
  79. package/dist/types/__tests__/atr.test.d.ts.map +0 -1
  80. package/dist/types/__tests__/auth-validator.test.d.ts +0 -2
  81. package/dist/types/__tests__/auth-validator.test.d.ts.map +0 -1
  82. package/dist/types/__tests__/broker-factory.test.d.ts +0 -2
  83. package/dist/types/__tests__/broker-factory.test.d.ts.map +0 -1
  84. package/dist/types/__tests__/broker-types.test.d.ts +0 -2
  85. package/dist/types/__tests__/broker-types.test.d.ts.map +0 -1
  86. package/dist/types/__tests__/cache.test.d.ts +0 -2
  87. package/dist/types/__tests__/cache.test.d.ts.map +0 -1
  88. package/dist/types/__tests__/errors.test.d.ts +0 -2
  89. package/dist/types/__tests__/errors.test.d.ts.map +0 -1
  90. package/dist/types/__tests__/financial-regression.test.d.ts +0 -2
  91. package/dist/types/__tests__/financial-regression.test.d.ts.map +0 -1
  92. package/dist/types/__tests__/format-tools.test.d.ts +0 -2
  93. package/dist/types/__tests__/format-tools.test.d.ts.map +0 -1
  94. package/dist/types/__tests__/http-keep-alive.test.d.ts +0 -2
  95. package/dist/types/__tests__/http-keep-alive.test.d.ts.map +0 -1
  96. package/dist/types/__tests__/http-timeout.test.d.ts +0 -2
  97. package/dist/types/__tests__/http-timeout.test.d.ts.map +0 -1
  98. package/dist/types/__tests__/index.test.d.ts +0 -2
  99. package/dist/types/__tests__/index.test.d.ts.map +0 -1
  100. package/dist/types/__tests__/legacy-auth.test.d.ts +0 -2
  101. package/dist/types/__tests__/legacy-auth.test.d.ts.map +0 -1
  102. package/dist/types/__tests__/logger.test.d.ts +0 -2
  103. package/dist/types/__tests__/logger.test.d.ts.map +0 -1
  104. package/dist/types/__tests__/logging.test.d.ts +0 -2
  105. package/dist/types/__tests__/logging.test.d.ts.map +0 -1
  106. package/dist/types/__tests__/market-time.test.d.ts +0 -2
  107. package/dist/types/__tests__/market-time.test.d.ts.map +0 -1
  108. package/dist/types/__tests__/massive.test.d.ts +0 -2
  109. package/dist/types/__tests__/massive.test.d.ts.map +0 -1
  110. package/dist/types/__tests__/metrics-calcs-direction.test.d.ts +0 -2
  111. package/dist/types/__tests__/metrics-calcs-direction.test.d.ts.map +0 -1
  112. package/dist/types/__tests__/misc-utils.test.d.ts +0 -2
  113. package/dist/types/__tests__/misc-utils.test.d.ts.map +0 -1
  114. package/dist/types/__tests__/paginator.test.d.ts +0 -2
  115. package/dist/types/__tests__/paginator.test.d.ts.map +0 -1
  116. package/dist/types/__tests__/performance-metrics-fees.test.d.ts +0 -2
  117. package/dist/types/__tests__/performance-metrics-fees.test.d.ts.map +0 -1
  118. package/dist/types/__tests__/performance-metrics.test.d.ts +0 -2
  119. package/dist/types/__tests__/performance-metrics.test.d.ts.map +0 -1
  120. package/dist/types/__tests__/price-utils-fees.test.d.ts +0 -2
  121. package/dist/types/__tests__/price-utils-fees.test.d.ts.map +0 -1
  122. package/dist/types/__tests__/price-utils.test.d.ts +0 -2
  123. package/dist/types/__tests__/price-utils.test.d.ts.map +0 -1
  124. package/dist/types/__tests__/property-based-financial.test.d.ts +0 -2
  125. package/dist/types/__tests__/property-based-financial.test.d.ts.map +0 -1
  126. package/dist/types/__tests__/protective-order-sides.test.d.ts +0 -2
  127. package/dist/types/__tests__/protective-order-sides.test.d.ts.map +0 -1
  128. package/dist/types/__tests__/rate-limiter.test.d.ts +0 -2
  129. package/dist/types/__tests__/rate-limiter.test.d.ts.map +0 -1
  130. package/dist/types/__tests__/retry-classification.test.d.ts +0 -2
  131. package/dist/types/__tests__/retry-classification.test.d.ts.map +0 -1
  132. package/dist/types/__tests__/retry.test.d.ts +0 -2
  133. package/dist/types/__tests__/retry.test.d.ts.map +0 -1
  134. package/dist/types/__tests__/risk-free-rate.test.d.ts +0 -2
  135. package/dist/types/__tests__/risk-free-rate.test.d.ts.map +0 -1
  136. package/dist/types/__tests__/risk-metrics.test.d.ts +0 -2
  137. package/dist/types/__tests__/risk-metrics.test.d.ts.map +0 -1
  138. package/dist/types/__tests__/schema-validation.test.d.ts +0 -2
  139. package/dist/types/__tests__/schema-validation.test.d.ts.map +0 -1
  140. package/dist/types/__tests__/stampede-load-timeout.test.d.ts +0 -2
  141. package/dist/types/__tests__/stampede-load-timeout.test.d.ts.map +0 -1
  142. package/dist/types/__tests__/strategy-metrics.test.d.ts +0 -2
  143. package/dist/types/__tests__/strategy-metrics.test.d.ts.map +0 -1
  144. package/dist/types/__tests__/technical-analysis-totality.test.d.ts +0 -2
  145. package/dist/types/__tests__/technical-analysis-totality.test.d.ts.map +0 -1
  146. package/dist/types/__tests__/technical-analysis.test.d.ts +0 -2
  147. package/dist/types/__tests__/technical-analysis.test.d.ts.map +0 -1
  148. package/dist/types/__tests__/time-utils.test.d.ts +0 -2
  149. package/dist/types/__tests__/time-utils.test.d.ts.map +0 -1
  150. package/dist/types/__tests__/trading-policy-schemas.test.d.ts +0 -2
  151. package/dist/types/__tests__/trading-policy-schemas.test.d.ts.map +0 -1
  152. package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts +0 -2
  153. package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts.map +0 -1
  154. package/dist/types/__tests__/volatility.test.d.ts +0 -2
  155. package/dist/types/__tests__/volatility.test.d.ts.map +0 -1
package/dist/index.cjs CHANGED
@@ -3290,7 +3290,7 @@ const MIN_WAKE_DELAY_MS = 1;
3290
3290
  * Number of whole tokens required to release a single queued request. The token
3291
3291
  * bucket consumes exactly one token per admitted request.
3292
3292
  */
3293
- const TOKENS_PER_REQUEST = 1;
3293
+ const TOKENS_PER_REQUEST$1 = 1;
3294
3294
  /**
3295
3295
  * Token bucket rate limiter implementation
3296
3296
  *
@@ -3362,8 +3362,8 @@ class TokenBucketRateLimiter {
3362
3362
  // Require a WHOLE token: refill() accrues fractionally, and admitting on
3363
3363
  // any positive fraction would release a full request per accrual tick,
3364
3364
  // driving the bucket negative and overrunning the configured rate.
3365
- if (this.tokens >= TOKENS_PER_REQUEST) {
3366
- this.tokens -= TOKENS_PER_REQUEST;
3365
+ if (this.tokens >= TOKENS_PER_REQUEST$1) {
3366
+ this.tokens -= TOKENS_PER_REQUEST$1;
3367
3367
  logger.debug(`Rate limit token acquired for ${this.config.label}`, {
3368
3368
  remainingTokens: this.tokens,
3369
3369
  queueLength: this.queue.length,
@@ -3408,7 +3408,7 @@ class TokenBucketRateLimiter {
3408
3408
  if (this.wakeTimer !== null || this.queue.length === 0) {
3409
3409
  return;
3410
3410
  }
3411
- const tokensNeeded = Math.max(0, TOKENS_PER_REQUEST - this.tokens);
3411
+ const tokensNeeded = Math.max(0, TOKENS_PER_REQUEST$1 - this.tokens);
3412
3412
  const deficitMs = Math.max(MIN_WAKE_DELAY_MS, Math.ceil((tokensNeeded / this.config.refillRate) * MS_PER_SECOND$1));
3413
3413
  const timer = setTimeout(() => {
3414
3414
  this.wakeTimer = null;
@@ -3460,8 +3460,8 @@ class TokenBucketRateLimiter {
3460
3460
  const logger = getLogger();
3461
3461
  try {
3462
3462
  // Whole-token admission — see the matching guard in acquire().
3463
- while (this.queue.length > 0 && this.tokens >= TOKENS_PER_REQUEST) {
3464
- this.tokens -= TOKENS_PER_REQUEST;
3463
+ while (this.queue.length > 0 && this.tokens >= TOKENS_PER_REQUEST$1) {
3464
+ this.tokens -= TOKENS_PER_REQUEST$1;
3465
3465
  const next = this.queue.shift();
3466
3466
  if (next) {
3467
3467
  clearTimeout(next.timeoutHandle);
@@ -3543,7 +3543,7 @@ class TokenBucketRateLimiter {
3543
3543
  * await rateLimiters.alphaVantage.acquire();
3544
3544
  * ```
3545
3545
  */
3546
- const rateLimiters = {
3546
+ const rateLimiters$1 = {
3547
3547
  /**
3548
3548
  * Alpaca API rate limiter
3549
3549
  *
@@ -4149,7 +4149,7 @@ class AlpacaMarketDataAPI extends require$$0$1.EventEmitter {
4149
4149
  // bucket so concurrent callers (bar fetches, quotes, options, snapshots)
4150
4150
  // can't overrun Alpaca's server-side rate limit. Prior to this, parallel
4151
4151
  // historical-bar fan-out produced ~125 server-side 429s per minute.
4152
- await rateLimiters.alpaca.acquire();
4152
+ await rateLimiters$1.alpaca.acquire();
4153
4153
  // Retry ONLY transient connection faults, and only on GET (every
4154
4154
  // market-data read here is idempotent). A non-2xx response is a real
4155
4155
  // answer from Alpaca and is never retried — that path still throws on
@@ -4195,7 +4195,7 @@ class AlpacaMarketDataAPI extends require$$0$1.EventEmitter {
4195
4195
  await new Promise((resolve) => setTimeout(resolve, delayMs));
4196
4196
  // Re-acquire the rate-limit token so a retry storm cannot overrun
4197
4197
  // Alpaca's server-side limit.
4198
- await rateLimiters.alpaca.acquire();
4198
+ await rateLimiters$1.alpaca.acquire();
4199
4199
  }
4200
4200
  }
4201
4201
  if (!response) {
@@ -9902,7 +9902,7 @@ const fetchTickerInfo = async (symbol, options) => {
9902
9902
  apiKey,
9903
9903
  });
9904
9904
  return massiveLimit(async () => {
9905
- await rateLimiters.massive.acquire();
9905
+ await rateLimiters$1.massive.acquire();
9906
9906
  try {
9907
9907
  const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
9908
9908
  const data = await response.json();
@@ -10015,7 +10015,7 @@ const fetchLastTradeImpl = async (symbol, options) => {
10015
10015
  order: "desc",
10016
10016
  });
10017
10017
  return massiveLimit(async () => {
10018
- await rateLimiters.massive.acquire();
10018
+ await rateLimiters$1.massive.acquire();
10019
10019
  try {
10020
10020
  const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
10021
10021
  const data = (await response.json());
@@ -10086,7 +10086,7 @@ const fetchLastQuote = async (symbol, options) => {
10086
10086
  order: "desc",
10087
10087
  });
10088
10088
  return massiveLimit(async () => {
10089
- await rateLimiters.massive.acquire();
10089
+ await rateLimiters$1.massive.acquire();
10090
10090
  try {
10091
10091
  const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
10092
10092
  const data = (await response.json());
@@ -10175,7 +10175,7 @@ const fetchPrices = async (params, options) => {
10175
10175
  let aggregatedStatus = "OK";
10176
10176
  while (nextUrl) {
10177
10177
  //getLogger().info(`Debug: Fetching ${nextUrl}`);
10178
- await rateLimiters.massive.acquire();
10178
+ await rateLimiters$1.massive.acquire();
10179
10179
  const response = await fetchWithRetry(nextUrl, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
10180
10180
  const data = await response.json();
10181
10181
  if (!MASSIVE_VALID_STATUSES.has(data.status)) {
@@ -10362,7 +10362,7 @@ const fetchGroupedDaily = async (date, options) => {
10362
10362
  include_otc: options?.includeOTC ? "true" : "false",
10363
10363
  });
10364
10364
  return massiveLimit(async () => {
10365
- await rateLimiters.massive.acquire();
10365
+ await rateLimiters$1.massive.acquire();
10366
10366
  try {
10367
10367
  const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
10368
10368
  const data = await response.json();
@@ -10456,7 +10456,7 @@ symbol, date = new Date(), options) => {
10456
10456
  adjusted: (options?.adjusted ?? true).toString(),
10457
10457
  });
10458
10458
  return massiveLimit(async () => {
10459
- await rateLimiters.massive.acquire();
10459
+ await rateLimiters$1.massive.acquire();
10460
10460
  try {
10461
10461
  const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
10462
10462
  const data = await response.json();
@@ -10530,7 +10530,7 @@ const fetchTrades = async (symbol, options) => {
10530
10530
  if (options?.sort)
10531
10531
  params.append("sort", options.sort);
10532
10532
  return massiveLimit(async () => {
10533
- await rateLimiters.massive.acquire();
10533
+ await rateLimiters$1.massive.acquire();
10534
10534
  const url = `${baseUrl}?${params.toString()}`;
10535
10535
  try {
10536
10536
  logIfDebug(`Fetching trades for ${symbol} from ${url}`);
@@ -10613,7 +10613,7 @@ const fetchIndicesAggregates = async (params, options) => {
10613
10613
  }
10614
10614
  url.search = queryParams.toString();
10615
10615
  return massiveIndicesLimit(async () => {
10616
- await rateLimiters.massive.acquire();
10616
+ await rateLimiters$1.massive.acquire();
10617
10617
  try {
10618
10618
  const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
10619
10619
  const data = await response.json();
@@ -10643,7 +10643,7 @@ const fetchIndicesPreviousClose = async (indicesTicker, options) => {
10643
10643
  queryParams.append("apiKey", apiKey);
10644
10644
  url.search = queryParams.toString();
10645
10645
  return massiveIndicesLimit(async () => {
10646
- await rateLimiters.massive.acquire();
10646
+ await rateLimiters$1.massive.acquire();
10647
10647
  try {
10648
10648
  const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
10649
10649
  const data = await response.json();
@@ -10674,7 +10674,7 @@ const fetchIndicesDailyOpenClose = async (indicesTicker, date, options) => {
10674
10674
  queryParams.append("apiKey", apiKey);
10675
10675
  url.search = queryParams.toString();
10676
10676
  return massiveIndicesLimit(async () => {
10677
- await rateLimiters.massive.acquire();
10677
+ await rateLimiters$1.massive.acquire();
10678
10678
  try {
10679
10679
  const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
10680
10680
  const data = await response.json();
@@ -10716,7 +10716,7 @@ const fetchIndicesSnapshot = async (params, options) => {
10716
10716
  }
10717
10717
  url.search = queryParams.toString();
10718
10718
  return massiveIndicesLimit(async () => {
10719
- await rateLimiters.massive.acquire();
10719
+ await rateLimiters$1.massive.acquire();
10720
10720
  try {
10721
10721
  const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
10722
10722
  const data = await response.json();
@@ -10765,7 +10765,7 @@ const fetchUniversalSnapshot = async (tickers, options) => {
10765
10765
  }
10766
10766
  url.search = queryParams.toString();
10767
10767
  return massiveIndicesLimit(async () => {
10768
- await rateLimiters.massive.acquire();
10768
+ await rateLimiters$1.massive.acquire();
10769
10769
  try {
10770
10770
  const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
10771
10771
  const data = await response.json();
@@ -52147,7 +52147,7 @@ class AlpacaClient {
52147
52147
  * @returns Result of the operation
52148
52148
  */
52149
52149
  async executeWithRateLimit(operation, label) {
52150
- await rateLimiters.alpaca.acquire();
52150
+ await rateLimiters$1.alpaca.acquire();
52151
52151
  return withRetry(operation, {
52152
52152
  maxRetries: 2,
52153
52153
  baseDelayMs: 1000,
@@ -52206,7 +52206,7 @@ class AlpacaClient {
52206
52206
  * @returns Response data
52207
52207
  */
52208
52208
  async makeRequest(endpoint, method = "GET", body) {
52209
- await rateLimiters.alpaca.acquire();
52209
+ await rateLimiters$1.alpaca.acquire();
52210
52210
  const url = `${this.apiBaseUrl}${endpoint}`;
52211
52211
  const options = {
52212
52212
  method,
@@ -62911,6 +62911,2826 @@ const alpaca = {
62911
62911
  streams: streams$1,
62912
62912
  };
62913
62913
 
62914
+ /**
62915
+ * Per-route circuit breaker for the alias client.
62916
+ *
62917
+ * A provider that has just failed five calls will almost certainly fail the
62918
+ * sixth, and every attempt spends latency budget the next leg of the chain
62919
+ * needs. The breaker converts that repeated discovery into a decision made
62920
+ * once: an unhealthy leg is skipped outright until it has had time to recover,
62921
+ * so a chain reaches its working leg quickly instead of paying the full
62922
+ * timeout of each dead one first.
62923
+ *
62924
+ * Breakers are keyed by chain POSITION, not by provider. Two consequences
62925
+ * follow, and both are intended. Reverting an alias to a different model at the
62926
+ * same position inherits that position's health rather than starting blind. And
62927
+ * an isolated route never shares a breaker with the shared one, so isolated
62928
+ * traffic can neither trip nor be tripped by traffic on the other side of the
62929
+ * PD-9 boundary.
62930
+ *
62931
+ * The clock is injected. Breaker behaviour is entirely about elapsed time, and
62932
+ * a test that must sleep to observe a cooldown is a test nobody runs.
62933
+ *
62934
+ * @module llm/circuit-breaker
62935
+ */
62936
+ /**
62937
+ * Tracks route health and decides whether a leg may be attempted.
62938
+ */
62939
+ class CircuitBreakerRegistry {
62940
+ records = new Map();
62941
+ config;
62942
+ now;
62943
+ /**
62944
+ * @param config Failure threshold, cooldown and half-open probe budget.
62945
+ * @param now Clock, injected so cooldowns are testable without waiting.
62946
+ */
62947
+ constructor(config, now = Date.now) {
62948
+ this.config = config;
62949
+ this.now = now;
62950
+ }
62951
+ /**
62952
+ * Current state of a route's breaker.
62953
+ *
62954
+ * The transition from open to half-open is computed from elapsed time at read
62955
+ * time rather than scheduled with a timer. A timer would keep the process
62956
+ * awake for every route that ever failed, and would drift whenever the
62957
+ * process was busy — which is exactly when a breaker matters most.
62958
+ *
62959
+ * @param routeKey The route's stable key.
62960
+ * @returns Its state.
62961
+ */
62962
+ stateOf(routeKey) {
62963
+ const record = this.records.get(routeKey);
62964
+ if (record === undefined || record.openedAtMs === null) {
62965
+ return "closed";
62966
+ }
62967
+ const elapsed = this.now() - record.openedAtMs;
62968
+ return elapsed >= this.config.cooldown_ms ? "half-open" : "open";
62969
+ }
62970
+ /**
62971
+ * Whether a route may be attempted now.
62972
+ *
62973
+ * A half-open route admits a bounded number of probes at once. Letting the
62974
+ * whole queue through the moment a cooldown expires would re-hammer a
62975
+ * provider that is still recovering, which is how a breaker turns into a
62976
+ * synchronised retry storm.
62977
+ *
62978
+ * @param routeKey The route's stable key.
62979
+ * @returns Whether an attempt is permitted.
62980
+ */
62981
+ allows(routeKey) {
62982
+ const state = this.stateOf(routeKey);
62983
+ if (state === "closed") {
62984
+ return true;
62985
+ }
62986
+ if (state === "open") {
62987
+ return false;
62988
+ }
62989
+ const record = this.recordFor(routeKey);
62990
+ return record.probesInFlight < this.config.half_open_probes;
62991
+ }
62992
+ /**
62993
+ * Register that an attempt is starting, so half-open probes stay bounded.
62994
+ *
62995
+ * @param routeKey The route's stable key.
62996
+ * @returns void
62997
+ */
62998
+ onAttemptStart(routeKey) {
62999
+ if (this.stateOf(routeKey) === "half-open") {
63000
+ this.recordFor(routeKey).probesInFlight += 1;
63001
+ }
63002
+ }
63003
+ /**
63004
+ * Record a success, closing the breaker.
63005
+ *
63006
+ * A single success closes it fully rather than decrementing the failure
63007
+ * count. The breaker's question is "is this route working now", and one
63008
+ * working call answers it; requiring several would keep a recovered provider
63009
+ * excluded while the chain paid for slower legs.
63010
+ *
63011
+ * @param routeKey The route's stable key.
63012
+ * @returns void
63013
+ */
63014
+ onSuccess(routeKey) {
63015
+ this.records.set(routeKey, {
63016
+ consecutiveFailures: 0,
63017
+ openedAtMs: null,
63018
+ probesInFlight: 0,
63019
+ });
63020
+ }
63021
+ /**
63022
+ * Record a failure, opening the breaker once the threshold is reached.
63023
+ *
63024
+ * A failure while half-open re-opens immediately without waiting to
63025
+ * re-accumulate the threshold: the probe was the test, and it failed.
63026
+ *
63027
+ * @param routeKey The route's stable key.
63028
+ * @returns void
63029
+ */
63030
+ onFailure(routeKey) {
63031
+ const wasHalfOpen = this.stateOf(routeKey) === "half-open";
63032
+ const record = this.recordFor(routeKey);
63033
+ record.probesInFlight = 0;
63034
+ record.consecutiveFailures += 1;
63035
+ if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
63036
+ record.openedAtMs = this.now();
63037
+ }
63038
+ }
63039
+ /**
63040
+ * Inspect a route's breaker.
63041
+ *
63042
+ * @param routeKey The route's stable key.
63043
+ * @returns A snapshot.
63044
+ */
63045
+ snapshot(routeKey) {
63046
+ const record = this.records.get(routeKey) ?? {
63047
+ consecutiveFailures: 0,
63048
+ openedAtMs: null,
63049
+ probesInFlight: 0,
63050
+ };
63051
+ return {
63052
+ routeKey,
63053
+ state: this.stateOf(routeKey),
63054
+ consecutiveFailures: record.consecutiveFailures,
63055
+ openedAtMs: record.openedAtMs,
63056
+ probesInFlight: record.probesInFlight,
63057
+ };
63058
+ }
63059
+ /**
63060
+ * Every route the registry has observed.
63061
+ *
63062
+ * @returns Snapshots, sorted by route key for stable output.
63063
+ */
63064
+ snapshotAll() {
63065
+ return [...this.records.keys()]
63066
+ .sort()
63067
+ .map((routeKey) => this.snapshot(routeKey));
63068
+ }
63069
+ /**
63070
+ * Discard all breaker state.
63071
+ *
63072
+ * @returns void
63073
+ */
63074
+ reset() {
63075
+ this.records.clear();
63076
+ }
63077
+ /**
63078
+ * @param routeKey The route's stable key.
63079
+ * @returns The mutable record, created on first use.
63080
+ */
63081
+ recordFor(routeKey) {
63082
+ let record = this.records.get(routeKey);
63083
+ if (record === undefined) {
63084
+ record = { consecutiveFailures: 0, openedAtMs: null, probesInFlight: 0 };
63085
+ this.records.set(routeKey, record);
63086
+ }
63087
+ return record;
63088
+ }
63089
+ }
63090
+
63091
+ /**
63092
+ * Parameter normalisation across providers.
63093
+ *
63094
+ * A caller passes one option bag and the chain may serve it from any of three
63095
+ * providers, so the same request has to be expressible to all of them. Vendors
63096
+ * disagree about more than spelling: some reject a sampling parameter they do
63097
+ * not support with a hard 400 rather than ignoring it, some name the output
63098
+ * cap differently, and reasoning models take an effort knob that non-reasoning
63099
+ * models refuse. Left unnormalised, a fallback would fail on the leg it fell
63100
+ * back to — turning the mechanism that exists to survive an outage into a
63101
+ * second way to fail.
63102
+ *
63103
+ * The rule throughout is that an unsupported parameter is OMITTED, never sent
63104
+ * with a default. Sending a default asserts a value the caller did not choose;
63105
+ * omitting it lets the provider apply its own, which is what "unsupported"
63106
+ * actually means.
63107
+ *
63108
+ * @module llm/param-matrix
63109
+ */
63110
+ /**
63111
+ * Providers that name the output cap `max_tokens` rather than
63112
+ * `max_completion_tokens`.
63113
+ *
63114
+ * The split is a wire-format fact about each API, so it is recorded as data
63115
+ * next to the translation that uses it rather than inferred from a version
63116
+ * number that will move.
63117
+ */
63118
+ const MAX_TOKENS_PARAM_BY_API_STYLE = {
63119
+ anthropic: "max_tokens",
63120
+ "openai-compatible": "max_completion_tokens",
63121
+ };
63122
+ /** The response-format key an OpenAI-compatible provider expects. */
63123
+ const RESPONSE_FORMAT_KEY = "response_format";
63124
+ /**
63125
+ * Normalise a caller's options into the exact parameter set one leg accepts.
63126
+ *
63127
+ * @param options The caller's options.
63128
+ * @param route The leg the request is being prepared for.
63129
+ * @param responseFormat The response shape the caller asked for.
63130
+ * @returns Parameters ready to send verbatim.
63131
+ */
63132
+ function normaliseParams(options, route, responseFormat) {
63133
+ const params = {};
63134
+ const capabilities = route.params;
63135
+ // Temperature. A model that accepts only its default value returns a hard
63136
+ // error for the parameter's mere presence, so support is checked before the
63137
+ // caller's preference is consulted at all.
63138
+ const temperature = options.temperature ?? capabilities.temperature ?? undefined;
63139
+ if (capabilities.supports_temperature !== false && typeof temperature === "number") {
63140
+ params.temperature = temperature;
63141
+ }
63142
+ // Output cap. The caller's value wins where present; the route's declared cap
63143
+ // is the ceiling, because exceeding it is a provider error rather than a
63144
+ // larger answer.
63145
+ const routeCap = capabilities.max_output_tokens ?? undefined;
63146
+ const requested = options.maxOutputTokens ?? routeCap;
63147
+ if (typeof requested === "number") {
63148
+ const capped = typeof routeCap === "number" ? Math.min(requested, routeCap) : requested;
63149
+ const key = MAX_TOKENS_PARAM_BY_API_STYLE[route.provider.api_style] ?? "max_tokens";
63150
+ params[key] = capped;
63151
+ }
63152
+ // Reasoning effort is meaningful only where the route declares it, and is
63153
+ // dropped elsewhere rather than translated into a temperature.
63154
+ const effort = options.reasoningEffort ?? capabilities.reasoning_effort ?? undefined;
63155
+ if (typeof effort === "string" && capabilities.reasoning_effort !== undefined) {
63156
+ params.reasoning_effort = effort;
63157
+ }
63158
+ const format = normaliseResponseFormat(responseFormat, route);
63159
+ if (format !== undefined) {
63160
+ params[RESPONSE_FORMAT_KEY] = format;
63161
+ }
63162
+ if (options.tools !== undefined && options.tools.length > 0) {
63163
+ if (capabilities.supports_tools === false) {
63164
+ throw new UnsupportedCapabilityError(route, "tools");
63165
+ }
63166
+ params.tools = [...options.tools];
63167
+ // Parallel tool calls are off by default: a decision path that fans out
63168
+ // tool calls concurrently reorders its own effects, and ordering is part
63169
+ // of the meaning of a sequence of trading actions.
63170
+ params.parallel_tool_calls = false;
63171
+ }
63172
+ if (options.metadata !== undefined) {
63173
+ params.metadata = { ...options.metadata };
63174
+ }
63175
+ return params;
63176
+ }
63177
+ /**
63178
+ * Translate the requested response shape into the parameter a route accepts.
63179
+ *
63180
+ * A strict JSON schema is a real capability, not a formatting preference: a
63181
+ * route that cannot enforce one would return prose where the caller's parser
63182
+ * expects an object. Rather than silently degrading to free JSON — which fails
63183
+ * later, further from the cause — a route without the capability refuses the
63184
+ * request so the chain advances to one that has it.
63185
+ *
63186
+ * @param responseFormat What the caller asked for.
63187
+ * @param route The leg being prepared.
63188
+ * @returns The provider-facing value, or undefined for plain text.
63189
+ */
63190
+ function normaliseResponseFormat(responseFormat, route) {
63191
+ if (responseFormat === "text") {
63192
+ return undefined;
63193
+ }
63194
+ if (responseFormat === "json") {
63195
+ return { type: "json_object" };
63196
+ }
63197
+ if (route.params.supports_json_schema === false) {
63198
+ throw new UnsupportedCapabilityError(route, "json_schema");
63199
+ }
63200
+ return {
63201
+ type: "json_schema",
63202
+ json_schema: {
63203
+ name: "structured_response",
63204
+ strict: true,
63205
+ schema: responseFormat.schema,
63206
+ },
63207
+ };
63208
+ }
63209
+ /**
63210
+ * Thrown when a leg cannot honour a capability the caller requires.
63211
+ *
63212
+ * A distinct type rather than a generic error, because the chain treats it
63213
+ * differently from a provider outage: the leg is not broken, it is simply the
63214
+ * wrong leg for this request, and no retry against it will help.
63215
+ */
63216
+ class UnsupportedCapabilityError extends Error {
63217
+ /** The leg that cannot serve the request. */
63218
+ routeKey;
63219
+ /** The capability it lacks. */
63220
+ capability;
63221
+ /**
63222
+ * @param route The leg.
63223
+ * @param capability The missing capability.
63224
+ */
63225
+ constructor(route, capability) {
63226
+ super(`route ${route.routeKey} (${route.providerName}/${route.modelId}) does not support ${capability}; ` +
63227
+ "the chain advances rather than degrading the request, so the caller's contract is never silently weakened");
63228
+ this.name = "UnsupportedCapabilityError";
63229
+ this.routeKey = route.routeKey;
63230
+ this.capability = capability;
63231
+ }
63232
+ }
63233
+ /**
63234
+ * Whether a leg can serve a request needing the given capabilities at all.
63235
+ *
63236
+ * Used to skip a leg before spending a network round trip on it. Checking
63237
+ * up front rather than reacting to the provider's rejection keeps a
63238
+ * capability mismatch from consuming the caller's latency budget.
63239
+ *
63240
+ * @param route The leg.
63241
+ * @param needs Capabilities the request requires.
63242
+ * @returns Whether the leg is a candidate.
63243
+ */
63244
+ function routeSupports(route, needs) {
63245
+ if (needs.tools === true && route.params.supports_tools === false) {
63246
+ return false;
63247
+ }
63248
+ if (needs.jsonSchema === true && route.params.supports_json_schema === false) {
63249
+ return false;
63250
+ }
63251
+ if (needs.vision === true && route.params.supports_vision !== true) {
63252
+ return false;
63253
+ }
63254
+ if (needs.cacheControl === true && route.params.supports_cache_control !== true) {
63255
+ return false;
63256
+ }
63257
+ return true;
63258
+ }
63259
+
63260
+ var defaults$1 = {
63261
+ _readme: "Applied to any provider whose published ceiling is unknown. Chosen to be comfortably below the slowest plausible published limit: being slower than necessary costs latency, while being faster than permitted costs 429s that the fallback chain will read as provider ill-health and use to open a circuit breaker. The asymmetry is what makes the conservative side the correct default.",
63262
+ basis: "conservative-default",
63263
+ requests_per_minute: 60,
63264
+ max_concurrent: 4,
63265
+ acquire_timeout_ms: 15000
63266
+ };
63267
+ var providers$1 = {
63268
+ anthropic: {
63269
+ basis: "conservative-default",
63270
+ requests_per_minute: 50,
63271
+ max_concurrent: 4,
63272
+ acquire_timeout_ms: 15000,
63273
+ source: null,
63274
+ note: "Anthropic publishes tier-dependent limits; the account's tier is not recorded here. Transcribe the real ceiling from the console at W3-07 and set basis to published."
63275
+ },
63276
+ openai: {
63277
+ basis: "conservative-default",
63278
+ requests_per_minute: 60,
63279
+ max_concurrent: 4,
63280
+ acquire_timeout_ms: 15000,
63281
+ source: null,
63282
+ note: "Tier-dependent. No alias routes here today; the entry exists so a revert to an OpenAI incumbent inherits a bounded client rather than an unbounded one."
63283
+ },
63284
+ deepseek: {
63285
+ basis: "conservative-default",
63286
+ requests_per_minute: 60,
63287
+ max_concurrent: 4,
63288
+ acquire_timeout_ms: 30000,
63289
+ source: null,
63290
+ note: "Serves llm.extract, a batch-class alias, so a longer acquire timeout is appropriate: a batch caller can afford to queue where a hot-path caller cannot."
63291
+ },
63292
+ deepinfra: {
63293
+ basis: "conservative-default",
63294
+ requests_per_minute: 60,
63295
+ max_concurrent: 4,
63296
+ acquire_timeout_ms: 15000,
63297
+ source: null,
63298
+ note: "Awaiting W3-02. Transcribe the published ceiling then."
63299
+ },
63300
+ fireworks: {
63301
+ basis: "conservative-default",
63302
+ requests_per_minute: 60,
63303
+ max_concurrent: 4,
63304
+ acquire_timeout_ms: 15000,
63305
+ source: null,
63306
+ note: "The backlog calls out recording Fireworks' published RPM ceiling specifically (W3-03 -> W4-04). Do that at onboarding."
63307
+ },
63308
+ zai: {
63309
+ basis: "conservative-default",
63310
+ requests_per_minute: 60,
63311
+ max_concurrent: 4,
63312
+ acquire_timeout_ms: 15000,
63313
+ source: null,
63314
+ note: "Awaiting W3-04."
63315
+ },
63316
+ groq: {
63317
+ basis: "conservative-default",
63318
+ requests_per_minute: 30,
63319
+ max_concurrent: 2,
63320
+ acquire_timeout_ms: 10000,
63321
+ source: null,
63322
+ note: "Declared primary of the hot-path llm.fast alias but blocked on open item OI-01. Held tighter than the default because a hot-path caller cannot afford to queue: if the limit binds, failing fast into the chain beats waiting."
63323
+ },
63324
+ openrouter: {
63325
+ basis: "conservative-default",
63326
+ requests_per_minute: 30,
63327
+ max_concurrent: 2,
63328
+ acquire_timeout_ms: 15000,
63329
+ source: null,
63330
+ note: "Aggregator backstop, reached by explicit route only. Held tight because its own limits are a function of whichever upstream it selects, which the client cannot observe."
63331
+ }
63332
+ };
63333
+ var limitsConfig = {
63334
+ defaults: defaults$1,
63335
+ providers: providers$1
63336
+ };
63337
+
63338
+ /**
63339
+ * Client-side rate and concurrency guards, per provider (W4-04).
63340
+ *
63341
+ * A provider's rate limit is enforced at the provider whether or not the client
63342
+ * respects it. The reason to respect it here is what a 429 means once it
63343
+ * arrives: to the fallback chain it is indistinguishable from provider
63344
+ * ill-health, so a client that over-drives a healthy provider will open that
63345
+ * provider's circuit breaker, fail over to a more expensive leg, and keep doing
63346
+ * so — converting a self-inflicted pacing problem into a permanent routing
63347
+ * change nobody chose. Pacing at the client is what keeps the breaker measuring
63348
+ * the provider rather than measuring us.
63349
+ *
63350
+ * Two distinct bounds are applied because they fail differently. The rate bound
63351
+ * (requests per minute) protects the provider's published ceiling. The
63352
+ * concurrency bound protects the caller: a hundred simultaneous in-flight
63353
+ * requests will each wait behind the other ninety-nine at the provider, so
63354
+ * every one of them blows its latency budget and the fan-out produces a hundred
63355
+ * timeouts instead of a queue.
63356
+ *
63357
+ * Limits live in `provider-limits.json`, not here. A rate limit discovered
63358
+ * during an incident should be correctable by config, not by a release.
63359
+ *
63360
+ * @module llm/rate-guard
63361
+ */
63362
+ /** Seconds in a minute, converting a published per-minute ceiling to a refill rate. */
63363
+ const SECONDS_PER_MINUTE = 60;
63364
+ /** One request consumes one token. */
63365
+ const TOKENS_PER_REQUEST = 1;
63366
+ const config$1 = limitsConfig;
63367
+ /**
63368
+ * Resolve the limits that apply to a provider.
63369
+ *
63370
+ * An unregistered provider falls back to the conservative defaults rather than
63371
+ * to no limit at all. Treating "unknown" as "unlimited" would make every newly
63372
+ * onboarded provider the one most likely to be over-driven, which is exactly
63373
+ * backwards: a new provider is the one whose real ceiling is least understood.
63374
+ *
63375
+ * @param provider The provider key.
63376
+ * @returns Its limits.
63377
+ */
63378
+ function limitsFor(provider) {
63379
+ return config$1.providers[provider] ?? config$1.defaults;
63380
+ }
63381
+ /** Every provider with a recorded limit, plus whether it is published or a default. */
63382
+ function limitsInventory() {
63383
+ return Object.keys(config$1.providers)
63384
+ .sort()
63385
+ .map((provider) => ({ provider, limits: config$1.providers[provider] }));
63386
+ }
63387
+ /**
63388
+ * Thrown when a caller could not acquire a slot within its budget.
63389
+ *
63390
+ * Distinguished from a provider failure so the chain does not count it against
63391
+ * route health: the provider was never asked, so nothing was learned about it.
63392
+ */
63393
+ class RateGuardTimeoutError extends Error {
63394
+ /** The provider whose guard could not admit the call. */
63395
+ provider;
63396
+ /** Which of the two bounds the caller waited on. */
63397
+ bound;
63398
+ /**
63399
+ * @param provider The provider.
63400
+ * @param bound Which bound was binding.
63401
+ * @param waitedMs How long the caller waited.
63402
+ */
63403
+ constructor(provider, bound, waitedMs) {
63404
+ super(`client-side ${bound} guard for provider "${provider}" did not admit the call within ${waitedMs} ms. ` +
63405
+ "The provider was never contacted, so this says nothing about its health.");
63406
+ this.name = "RateGuardTimeoutError";
63407
+ this.provider = provider;
63408
+ this.bound = bound;
63409
+ }
63410
+ }
63411
+ /**
63412
+ * A counting semaphore bounding simultaneous in-flight calls.
63413
+ *
63414
+ * Written here rather than pulled from a dependency because it is fifteen lines
63415
+ * and because the waiting behaviour matters: a waiter that times out must be
63416
+ * removed from the queue, or a burst of abandoned callers permanently consumes
63417
+ * the permits that later callers need.
63418
+ */
63419
+ class ConcurrencyGate {
63420
+ inFlight = 0;
63421
+ waiters = [];
63422
+ limit;
63423
+ provider;
63424
+ /**
63425
+ * @param provider The provider this gate guards.
63426
+ * @param limit Maximum simultaneous in-flight calls.
63427
+ */
63428
+ constructor(provider, limit) {
63429
+ this.provider = provider;
63430
+ this.limit = limit;
63431
+ }
63432
+ /**
63433
+ * Wait for a permit.
63434
+ *
63435
+ * @param timeoutMs How long the caller is willing to queue.
63436
+ * @returns A release function the caller must invoke exactly once.
63437
+ */
63438
+ async acquire(timeoutMs) {
63439
+ if (this.inFlight < this.limit) {
63440
+ this.inFlight += 1;
63441
+ return () => this.release();
63442
+ }
63443
+ await new Promise((resolve, reject) => {
63444
+ const timer = setTimeout(() => {
63445
+ const index = this.waiters.findIndex((waiter) => waiter.timer === timer);
63446
+ if (index !== -1) {
63447
+ this.waiters.splice(index, 1);
63448
+ }
63449
+ reject(new RateGuardTimeoutError(this.provider, "concurrency", timeoutMs));
63450
+ }, timeoutMs);
63451
+ this.waiters.push({ resolve, reject, timer });
63452
+ });
63453
+ this.inFlight += 1;
63454
+ return () => this.release();
63455
+ }
63456
+ /**
63457
+ * Return a permit and admit the next waiter.
63458
+ *
63459
+ * @returns void
63460
+ */
63461
+ release() {
63462
+ this.inFlight -= 1;
63463
+ const next = this.waiters.shift();
63464
+ if (next !== undefined) {
63465
+ clearTimeout(next.timer);
63466
+ next.resolve();
63467
+ }
63468
+ }
63469
+ /**
63470
+ * @returns How many calls are currently in flight.
63471
+ */
63472
+ inFlightCount() {
63473
+ return this.inFlight;
63474
+ }
63475
+ /**
63476
+ * @returns How many callers are queued.
63477
+ */
63478
+ queueLength() {
63479
+ return this.waiters.length;
63480
+ }
63481
+ }
63482
+ /** Per-provider guards, created on first use and shared process-wide. */
63483
+ const rateLimiters = new Map();
63484
+ const concurrencyGates = new Map();
63485
+ /**
63486
+ * The rate limiter for a provider.
63487
+ *
63488
+ * Shared process-wide rather than per-call-site, because the provider's ceiling
63489
+ * applies to the process as a whole. Per-call-site limiters would each stay
63490
+ * under the ceiling while their sum sailed past it.
63491
+ *
63492
+ * @param provider The provider key.
63493
+ * @returns Its limiter.
63494
+ */
63495
+ function rateLimiterFor(provider) {
63496
+ let limiter = rateLimiters.get(provider);
63497
+ if (limiter === undefined) {
63498
+ const limits = limitsFor(provider);
63499
+ limiter = new TokenBucketRateLimiter({
63500
+ maxTokens: limits.requests_per_minute,
63501
+ refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
63502
+ label: `llm:${provider}`,
63503
+ timeoutMs: limits.acquire_timeout_ms,
63504
+ });
63505
+ rateLimiters.set(provider, limiter);
63506
+ }
63507
+ return limiter;
63508
+ }
63509
+ /**
63510
+ * The concurrency gate for a provider.
63511
+ *
63512
+ * @param provider The provider key.
63513
+ * @returns Its gate.
63514
+ */
63515
+ function concurrencyGateFor(provider) {
63516
+ let gate = concurrencyGates.get(provider);
63517
+ if (gate === undefined) {
63518
+ gate = new ConcurrencyGate(provider, limitsFor(provider).max_concurrent);
63519
+ concurrencyGates.set(provider, gate);
63520
+ }
63521
+ return gate;
63522
+ }
63523
+ /**
63524
+ * Run a call under a provider's rate and concurrency guards.
63525
+ *
63526
+ * The concurrency permit is taken AFTER the rate token. Taking it first would
63527
+ * let callers hold scarce permits while idling in the rate queue, which
63528
+ * throttles the provider twice over and turns a pacing bound into a deadlock
63529
+ * shaped like slowness.
63530
+ *
63531
+ * `maxWaitMs` bounds how long a caller may queue. It exists because the queue
63532
+ * spends the SAME budget the call itself does: a caller that waits out its whole
63533
+ * deadline in a rate queue has failed just as completely as one that waited on
63534
+ * the provider, and worse, it never reached the fallback chain that could have
63535
+ * answered it. Passing the leg's own timeout keeps one clock governing the
63536
+ * whole attempt.
63537
+ *
63538
+ * @param provider The provider key.
63539
+ * @param call The work to run once admitted.
63540
+ * @param maxWaitMs Ceiling on queue time; the configured guard timeout applies when lower.
63541
+ * @returns The call's result.
63542
+ * @throws {RateGuardTimeoutError} When neither bound admitted the call in time.
63543
+ */
63544
+ async function withProviderGuards(provider, call, maxWaitMs) {
63545
+ const limits = limitsFor(provider);
63546
+ const waitBudgetMs = maxWaitMs === undefined
63547
+ ? limits.acquire_timeout_ms
63548
+ : Math.min(maxWaitMs, limits.acquire_timeout_ms);
63549
+ try {
63550
+ await rateLimiterFor(provider).acquire();
63551
+ }
63552
+ catch {
63553
+ throw new RateGuardTimeoutError(provider, "rate", waitBudgetMs);
63554
+ }
63555
+ const release = await concurrencyGateFor(provider).acquire(waitBudgetMs);
63556
+ try {
63557
+ return await call();
63558
+ }
63559
+ finally {
63560
+ // Released on every path. A permit leaked on the error path would shrink
63561
+ // the effective limit by one on each failure until nothing could run — and
63562
+ // failures cluster exactly when throughput matters most.
63563
+ release();
63564
+ }
63565
+ }
63566
+ /**
63567
+ * Inspect the guards currently in use.
63568
+ *
63569
+ * @returns A snapshot per provider that has been used, sorted by provider.
63570
+ */
63571
+ function guardSnapshots() {
63572
+ const providers = new Set([...rateLimiters.keys(), ...concurrencyGates.keys()]);
63573
+ return [...providers].sort().map((provider) => {
63574
+ const limits = limitsFor(provider);
63575
+ const limiter = rateLimiters.get(provider);
63576
+ const gate = concurrencyGates.get(provider);
63577
+ return {
63578
+ provider,
63579
+ basis: limits.basis,
63580
+ requestsPerMinute: limits.requests_per_minute,
63581
+ maxConcurrent: limits.max_concurrent,
63582
+ inFlight: gate?.inFlightCount() ?? 0,
63583
+ rateQueueLength: limiter?.getQueueLength() ?? 0,
63584
+ concurrencyQueueLength: gate?.queueLength() ?? 0,
63585
+ availableTokens: limiter?.getAvailableTokens() ?? limits.requests_per_minute * TOKENS_PER_REQUEST,
63586
+ };
63587
+ });
63588
+ }
63589
+ /**
63590
+ * Discard all guard state.
63591
+ *
63592
+ * Exists so a test can start from a known position; a shared process-wide
63593
+ * limiter is otherwise carried between tests and makes their order matter.
63594
+ *
63595
+ * @returns void
63596
+ */
63597
+ function resetProviderGuards() {
63598
+ for (const limiter of rateLimiters.values()) {
63599
+ limiter.reset();
63600
+ }
63601
+ rateLimiters.clear();
63602
+ concurrencyGates.clear();
63603
+ }
63604
+
63605
+ /**
63606
+ * Ordered execution of an alias's fallback chain (PD-3).
63607
+ *
63608
+ * Every call walks the chain primary -> secondary -> closed incumbent, and each
63609
+ * leg runs under a hard timeout and a circuit breaker. Those three controls are
63610
+ * one mechanism rather than three features: a chain without timeouts never
63611
+ * reaches its second leg, a chain without a breaker pays a dead provider's full
63612
+ * timeout on every call, and a timeout without a chain is just a slower
63613
+ * failure. Timeout cascades are this system's known brown-out mode, which is
63614
+ * why the budget is enforced here — at the only place that knows both the
63615
+ * caller's deadline and how many legs are left to spend it on.
63616
+ *
63617
+ * Nothing here ever substitutes a value for an outcome. When every leg is
63618
+ * exhausted the caller gets a typed error naming each leg and why it failed,
63619
+ * because a default returned in place of an answer is a wrong answer that
63620
+ * nobody is told about.
63621
+ *
63622
+ * @module llm/fallback-chain
63623
+ */
63624
+ /** Zero-valued usage, used as the identity when summing across attempts. */
63625
+ const EMPTY_USAGE = {
63626
+ prompt_tokens: 0,
63627
+ completion_tokens: 0,
63628
+ provider: "none",
63629
+ model: "none",
63630
+ cost: 0,
63631
+ };
63632
+ /**
63633
+ * Thrown when every leg of a chain has been tried and none produced an answer.
63634
+ *
63635
+ * Carries the full attempt record rather than only the last error. The last
63636
+ * error is usually the least informative one — the incumbent timing out says
63637
+ * nothing about why the two legs before it were skipped — and an operator
63638
+ * reading only that would go looking in the wrong place.
63639
+ */
63640
+ class ChainExhaustedError extends Error {
63641
+ /** The alias whose chain was exhausted. */
63642
+ alias;
63643
+ /** Every leg tried, in order, with its outcome. */
63644
+ attempts;
63645
+ /** Usage spent across the failed attempts, so the spend is still accounted for. */
63646
+ totalUsage;
63647
+ /**
63648
+ * @param alias The alias.
63649
+ * @param attempts The attempt record.
63650
+ * @param totalUsage Usage spent across all attempts.
63651
+ */
63652
+ constructor(alias, attempts, totalUsage) {
63653
+ const detail = attempts
63654
+ .map((attempt) => `${attempt.role}(${attempt.provider}/${attempt.modelId}): ${attempt.outcome}` +
63655
+ (attempt.reason === undefined ? "" : ` — ${attempt.reason}`))
63656
+ .join("; ");
63657
+ super(`LLM alias "${alias}" exhausted its fallback chain. Attempts: ${detail || "(no leg was servable)"}`);
63658
+ this.name = "ChainExhaustedError";
63659
+ this.alias = alias;
63660
+ this.attempts = attempts;
63661
+ this.totalUsage = totalUsage;
63662
+ }
63663
+ }
63664
+ /**
63665
+ * Add two usage records.
63666
+ *
63667
+ * Attribution keeps the LAST attempt's provider and model, because that is the
63668
+ * one that produced the answer the caller is holding, while the token counts
63669
+ * accumulate across every attempt. Charging only the successful attempt would
63670
+ * understate spend by exactly the amount the failures cost — which is the
63671
+ * amount a fallback chain is most likely to run up.
63672
+ *
63673
+ * @param a The running total.
63674
+ * @param b The attempt to add, if any.
63675
+ * @returns The combined usage.
63676
+ */
63677
+ function sumUsage(a, b) {
63678
+ if (b === undefined) {
63679
+ return a;
63680
+ }
63681
+ return {
63682
+ prompt_tokens: a.prompt_tokens + b.prompt_tokens,
63683
+ completion_tokens: a.completion_tokens + b.completion_tokens,
63684
+ reasoning_tokens: a.reasoning_tokens === undefined && b.reasoning_tokens === undefined
63685
+ ? undefined
63686
+ : (a.reasoning_tokens ?? 0) + (b.reasoning_tokens ?? 0),
63687
+ cached_tokens: a.cached_tokens === undefined && b.cached_tokens === undefined
63688
+ ? undefined
63689
+ : (a.cached_tokens ?? 0) + (b.cached_tokens ?? 0),
63690
+ provider: b.provider,
63691
+ model: b.model,
63692
+ cost: a.cost + b.cost,
63693
+ };
63694
+ }
63695
+ /** Raised internally when a leg exceeds its budget. */
63696
+ class LegTimeoutError extends Error {
63697
+ /**
63698
+ * @param routeKey The leg that timed out.
63699
+ * @param budgetMs Its budget in milliseconds.
63700
+ */
63701
+ constructor(routeKey, budgetMs) {
63702
+ super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
63703
+ this.name = "LegTimeoutError";
63704
+ }
63705
+ }
63706
+ /**
63707
+ * Run one leg under a hard timeout, honouring the caller's own cancellation.
63708
+ *
63709
+ * The timer is always cleared and the abort listener always removed, including
63710
+ * on the success path. A long-lived process that leaked one timer per LLM call
63711
+ * would accumulate them at exactly the rate it does useful work.
63712
+ *
63713
+ * @param leg The leg to run.
63714
+ * @param params Normalised parameters for this leg.
63715
+ * @param execution The call context.
63716
+ * @returns The provider's answer.
63717
+ */
63718
+ async function runLeg(leg, params, execution) {
63719
+ const controller = new AbortController();
63720
+ const budgetMs = leg.route.timeoutMs;
63721
+ const timer = setTimeout(() => {
63722
+ controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
63723
+ }, budgetMs);
63724
+ const forwardAbort = () => {
63725
+ controller.abort(execution.callerSignal?.reason);
63726
+ };
63727
+ if (execution.callerSignal !== undefined) {
63728
+ if (execution.callerSignal.aborted) {
63729
+ forwardAbort();
63730
+ }
63731
+ else {
63732
+ execution.callerSignal.addEventListener("abort", forwardAbort, { once: true });
63733
+ }
63734
+ }
63735
+ try {
63736
+ // The guards wrap the transport rather than the whole leg, so the per-leg
63737
+ // timeout above still bounds the total wait: a caller queued behind the
63738
+ // rate limiter is spending its budget just as surely as one waiting on the
63739
+ // provider, and only one clock should govern both.
63740
+ return await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
63741
+ route: leg.route,
63742
+ content: execution.content,
63743
+ responseFormat: execution.responseFormat,
63744
+ params,
63745
+ signal: controller.signal,
63746
+ correlationId: execution.correlationId,
63747
+ }), budgetMs);
63748
+ }
63749
+ finally {
63750
+ clearTimeout(timer);
63751
+ execution.callerSignal?.removeEventListener("abort", forwardAbort);
63752
+ }
63753
+ }
63754
+ /**
63755
+ * Classify why a leg failed.
63756
+ *
63757
+ * The distinction matters to the breaker: a timeout and a 5xx are evidence the
63758
+ * provider is unhealthy, while the caller cancelling is not. Counting a
63759
+ * cancellation as a provider failure would let a burst of user-cancelled
63760
+ * requests open the breaker on a perfectly healthy route.
63761
+ *
63762
+ * @param error The thrown value.
63763
+ * @param callerSignal The caller's cancellation signal, if any.
63764
+ * @returns The outcome and whether it counts against route health.
63765
+ */
63766
+ function classify(error, callerSignal) {
63767
+ if (callerSignal !== undefined && callerSignal.aborted) {
63768
+ return {
63769
+ outcome: "skipped",
63770
+ reason: "caller cancelled",
63771
+ countsAgainstHealth: false,
63772
+ };
63773
+ }
63774
+ if (error instanceof LegTimeoutError) {
63775
+ return { outcome: "timeout", reason: error.message, countsAgainstHealth: true };
63776
+ }
63777
+ if (error instanceof UnsupportedCapabilityError) {
63778
+ return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
63779
+ }
63780
+ if (error instanceof RateGuardTimeoutError) {
63781
+ // Self-inflicted pacing, not provider ill-health. Counting it would let the
63782
+ // client's own throttling open a breaker on a perfectly healthy provider
63783
+ // and permanently reroute traffic nobody chose to reroute.
63784
+ return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
63785
+ }
63786
+ const reason = error instanceof Error ? error.message : String(error);
63787
+ if (/abort/i.test(reason)) {
63788
+ return {
63789
+ outcome: "timeout",
63790
+ reason: `aborted: ${reason}`,
63791
+ countsAgainstHealth: true,
63792
+ };
63793
+ }
63794
+ return { outcome: "error", reason, countsAgainstHealth: true };
63795
+ }
63796
+ /**
63797
+ * Whether the caller has stopped waiting.
63798
+ *
63799
+ * Read through a function rather than inline, because `AbortSignal.aborted` is
63800
+ * a live getter: it can flip to true while a leg is in flight, but a compiler
63801
+ * that narrowed it at the top of the loop would prove the later check
63802
+ * unreachable and invite its removal. The check is not redundant — it is the
63803
+ * only thing that stops the chain spending money on an answer nobody will read.
63804
+ *
63805
+ * @param signal The caller's signal, if any.
63806
+ * @returns Whether the call has been cancelled.
63807
+ */
63808
+ function isAborted$1(signal) {
63809
+ return signal !== undefined && signal.aborted;
63810
+ }
63811
+ /**
63812
+ * Walk a chain until a leg answers.
63813
+ *
63814
+ * @param alias The alias being served, for error attribution.
63815
+ * @param execution The call context.
63816
+ * @returns The first successful leg's answer, with the full attempt record.
63817
+ * @throws {ChainExhaustedError} When no leg produced an answer.
63818
+ */
63819
+ async function executeChain(alias, execution) {
63820
+ const now = execution.now ?? Date.now;
63821
+ const attempts = [];
63822
+ let totalUsage = EMPTY_USAGE;
63823
+ for (const leg of execution.legs) {
63824
+ const { route } = leg;
63825
+ if (isAborted$1(execution.callerSignal)) {
63826
+ // The caller has stopped waiting. Continuing to walk the chain would
63827
+ // spend money on an answer nobody will read.
63828
+ break;
63829
+ }
63830
+ if (leg.params instanceof UnsupportedCapabilityError) {
63831
+ const record = {
63832
+ routeKey: route.routeKey,
63833
+ role: route.role,
63834
+ provider: route.providerName,
63835
+ modelId: route.modelId,
63836
+ outcome: "skipped",
63837
+ durationMs: 0,
63838
+ reason: leg.params.message,
63839
+ };
63840
+ attempts.push(record);
63841
+ execution.onAttempt?.(record);
63842
+ continue;
63843
+ }
63844
+ if (!execution.breakers.allows(route.routeKey)) {
63845
+ const record = {
63846
+ routeKey: route.routeKey,
63847
+ role: route.role,
63848
+ provider: route.providerName,
63849
+ modelId: route.modelId,
63850
+ outcome: "breaker-open",
63851
+ durationMs: 0,
63852
+ reason: `circuit breaker is ${execution.breakers.stateOf(route.routeKey)}`,
63853
+ };
63854
+ attempts.push(record);
63855
+ execution.onAttempt?.(record);
63856
+ continue;
63857
+ }
63858
+ const startedAt = now();
63859
+ execution.breakers.onAttemptStart(route.routeKey);
63860
+ try {
63861
+ const response = await runLeg(leg, leg.params, execution);
63862
+ execution.breakers.onSuccess(route.routeKey);
63863
+ totalUsage = sumUsage(totalUsage, response.usage);
63864
+ const record = {
63865
+ routeKey: route.routeKey,
63866
+ role: route.role,
63867
+ provider: route.providerName,
63868
+ modelId: route.modelId,
63869
+ outcome: "ok",
63870
+ durationMs: now() - startedAt,
63871
+ usage: response.usage,
63872
+ };
63873
+ attempts.push(record);
63874
+ execution.onAttempt?.(record);
63875
+ return { response, servedBy: route, attempts, totalUsage };
63876
+ }
63877
+ catch (error) {
63878
+ const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
63879
+ if (countsAgainstHealth) {
63880
+ execution.breakers.onFailure(route.routeKey);
63881
+ }
63882
+ const record = {
63883
+ routeKey: route.routeKey,
63884
+ role: route.role,
63885
+ provider: route.providerName,
63886
+ modelId: route.modelId,
63887
+ outcome,
63888
+ durationMs: now() - startedAt,
63889
+ reason,
63890
+ };
63891
+ attempts.push(record);
63892
+ execution.onAttempt?.(record);
63893
+ if (outcome === "skipped" && isAborted$1(execution.callerSignal)) {
63894
+ break;
63895
+ }
63896
+ }
63897
+ }
63898
+ throw new ChainExhaustedError(alias, attempts, totalUsage);
63899
+ }
63900
+
63901
+ var schema_version = 1;
63902
+ var policy_source = "docs/llm-provider-migration.md#3-routing-policy (v1.1)";
63903
+ var revised = "2026-09-11";
63904
+ var defaults = {
63905
+ request_timeout_ms: {
63906
+ "hot-path": 30000,
63907
+ background: 90000,
63908
+ batch: 300000
63909
+ },
63910
+ retries_per_leg: 1,
63911
+ circuit_breaker: {
63912
+ failure_threshold: 5,
63913
+ cooldown_ms: 60000,
63914
+ half_open_probes: 1
63915
+ }
63916
+ };
63917
+ var providers = {
63918
+ anthropic: {
63919
+ display_name: "Anthropic",
63920
+ tier: "closed",
63921
+ api_style: "anthropic",
63922
+ base_url: null,
63923
+ base_url_env: null,
63924
+ api_key_env: "ANTHROPIC_API_KEY",
63925
+ secret_path: "llm/anthropic/apiKey",
63926
+ account_status: "live",
63927
+ lumic_provider: "anthropic",
63928
+ published_rpm: null,
63929
+ published_tpm: null,
63930
+ limits_source: null,
63931
+ docs_url: "https://docs.claude.com/en/api/overview",
63932
+ notes: "Permanent designated tier. Retained, keyed, budgeted and continuously exercised as the revert target of every alias (PD-11)."
63933
+ },
63934
+ openai: {
63935
+ display_name: "OpenAI",
63936
+ tier: "closed",
63937
+ api_style: "openai-compatible",
63938
+ base_url: null,
63939
+ base_url_env: null,
63940
+ api_key_env: "OPENAI_API_KEY",
63941
+ secret_path: "llm/openai/apiKey",
63942
+ account_status: "live",
63943
+ lumic_provider: "openai",
63944
+ published_rpm: null,
63945
+ published_tpm: null,
63946
+ limits_source: null,
63947
+ docs_url: "https://platform.openai.com/docs/api-reference",
63948
+ notes: "Permanent designated tier. No alias routes to it today; it stays keyed and budgeted so a per-alias revert to an OpenAI incumbent is a config change (PD-11)."
63949
+ },
63950
+ deepinfra: {
63951
+ display_name: "DeepInfra",
63952
+ tier: "open",
63953
+ api_style: "openai-compatible",
63954
+ base_url: "https://api.deepinfra.com/v1/openai",
63955
+ base_url_env: "DEEPINFRA_BASE_URL",
63956
+ api_key_env: "DEEPINFRA_API_KEY",
63957
+ secret_path: "llm/deepinfra/apiKey",
63958
+ account_status: "live",
63959
+ lumic_provider: null,
63960
+ published_rpm: null,
63961
+ published_tpm: null,
63962
+ limits_source: null,
63963
+ docs_url: "https://deepinfra.com/docs",
63964
+ notes: "Sole open-weight host. The routing policy originally spread these models across Z.ai first-party, DeepInfra and Groq; DeepInfra's catalogue carries all of them, so consolidating removes three onboardings, three keys and three terms filings at the cost of roughly 28% on llm.reason's input price versus the first-party GLM anchor. Fewer credentials and fewer trust boundaries is worth more than that margin on one alias."
63965
+ },
63966
+ fireworks: {
63967
+ display_name: "Fireworks AI",
63968
+ tier: "open",
63969
+ api_style: "openai-compatible",
63970
+ base_url: "https://api.fireworks.ai/inference/v1",
63971
+ base_url_env: "FIREWORKS_BASE_URL",
63972
+ api_key_env: "FIREWORKS_API_KEY",
63973
+ secret_path: "llm/fireworks/apiKey",
63974
+ account_status: "not-in-scope",
63975
+ lumic_provider: null,
63976
+ published_rpm: null,
63977
+ published_tpm: null,
63978
+ limits_source: null,
63979
+ docs_url: "https://docs.fireworks.ai",
63980
+ notes: "Superseded by the DeepInfra consolidation: the models this host was chosen for are served from DeepInfra under one account. Retained in the registry so a future re-split needs a config change rather than a new onboarding."
63981
+ },
63982
+ zai: {
63983
+ display_name: "Z.ai",
63984
+ tier: "open",
63985
+ api_style: "openai-compatible",
63986
+ base_url: "https://api.z.ai/api/paas/v4",
63987
+ base_url_env: "ZAI_BASE_URL",
63988
+ api_key_env: "ZAI_API_KEY",
63989
+ secret_path: "llm/zai/apiKey",
63990
+ account_status: "not-in-scope",
63991
+ lumic_provider: null,
63992
+ published_rpm: null,
63993
+ published_tpm: null,
63994
+ limits_source: null,
63995
+ docs_url: "https://docs.z.ai",
63996
+ notes: "Superseded by the DeepInfra consolidation: the models this host was chosen for are served from DeepInfra under one account. Retained in the registry so a future re-split needs a config change rather than a new onboarding."
63997
+ },
63998
+ deepseek: {
63999
+ display_name: "DeepSeek",
64000
+ tier: "open",
64001
+ api_style: "openai-compatible",
64002
+ base_url: "https://api.deepseek.com",
64003
+ base_url_env: "DEEPSEEK_BASE_URL",
64004
+ api_key_env: "DEEPSEEK_API_KEY",
64005
+ secret_path: "llm/deepseek/apiKey",
64006
+ account_status: "not-in-scope",
64007
+ lumic_provider: "deepseek",
64008
+ published_rpm: null,
64009
+ published_tpm: null,
64010
+ limits_source: null,
64011
+ docs_url: "https://api-docs.deepseek.com",
64012
+ notes: "First-party DeepSeek API, no longer routed to. Its G1 filing found no commitment not to train on API inputs, no stated retention period, no sub-processor list, and storage under PRC jurisdiction — a materially weaker data posture than any other provider reviewed, and not one to send trading prompts through. The DeepSeek MODELS are still used and are a good fit; they are now served as open weights from DeepInfra under its Zero Data Retention commitment. The weights and the vendor's hosted API are separable, and only the API carried the exposure. Retained in the registry so the distinction stays visible rather than being rediscovered."
64013
+ },
64014
+ groq: {
64015
+ display_name: "Groq",
64016
+ tier: "open",
64017
+ api_style: "openai-compatible",
64018
+ base_url: "https://api.groq.com/openai/v1",
64019
+ base_url_env: "GROQ_BASE_URL",
64020
+ api_key_env: "GROQ_API_KEY",
64021
+ secret_path: "llm/groq/apiKey",
64022
+ account_status: "not-in-scope",
64023
+ lumic_provider: null,
64024
+ published_rpm: null,
64025
+ published_tpm: null,
64026
+ limits_source: null,
64027
+ docs_url: "https://console.groq.com/docs",
64028
+ notes: "Superseded by the DeepInfra consolidation: the models this host was chosen for are served from DeepInfra under one account. Retained in the registry so a future re-split needs a config change rather than a new onboarding."
64029
+ },
64030
+ openrouter: {
64031
+ display_name: "OpenRouter",
64032
+ tier: "aggregator",
64033
+ api_style: "openai-compatible",
64034
+ base_url: "https://openrouter.ai/api/v1",
64035
+ base_url_env: "OPENROUTER_BASE_URL",
64036
+ api_key_env: "OPENROUTER_API_KEY",
64037
+ secret_path: "llm/openrouter/apiKey",
64038
+ account_status: "pending-onboarding",
64039
+ lumic_provider: null,
64040
+ published_rpm: null,
64041
+ published_tpm: null,
64042
+ limits_source: null,
64043
+ docs_url: "https://openrouter.ai/docs",
64044
+ notes: "Aggregator backstop. Reaches an open-weight model when its dedicated host is down, without a new account. Not a chain leg by default: an aggregator hides which upstream served a call, which defeats per-provider attribution."
64045
+ }
64046
+ };
64047
+ var aliases = {
64048
+ "llm.reason": {
64049
+ workload: "Deep reasoning, audit loops, config tuning",
64050
+ latency_class: "background",
64051
+ criticality: "ops",
64052
+ isolation_capable: true,
64053
+ eval_gate: "free-text-judge",
64054
+ budget: {
64055
+ basis: "provisional-pre-baseline",
64056
+ monthly_usd: 750,
64057
+ alert_pct: [
64058
+ 50,
64059
+ 80,
64060
+ 100
64061
+ ]
64062
+ },
64063
+ routes: [
64064
+ {
64065
+ role: "primary",
64066
+ provider: "deepinfra",
64067
+ model_id: "deepseek-ai/DeepSeek-V4-Pro",
64068
+ model_id_status: "confirmed",
64069
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64070
+ model_family: "DeepSeek V4 Pro",
64071
+ lumic_model: "deepseek-ai/DeepSeek-V4-Pro",
64072
+ params: {
64073
+ temperature: null,
64074
+ max_output_tokens: null,
64075
+ supports_temperature: true,
64076
+ supports_json_schema: true,
64077
+ supports_tools: true,
64078
+ supports_cache_control: true,
64079
+ supports_vision: false,
64080
+ context_window: 1048576
64081
+ },
64082
+ price_per_mtok: {
64083
+ input: 1.3,
64084
+ output: 2.6,
64085
+ as_of: "2026-09-11",
64086
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64087
+ },
64088
+ notes: "1.6T MoE, 49B active, 1M context. Its model card cites advanced reasoning and long-running agent tasks, and it is the cheapest model in the flagship tier by a wide margin — a quarter of the incumbent's input price and a tenth of its output price."
64089
+ },
64090
+ {
64091
+ role: "secondary",
64092
+ provider: "deepinfra",
64093
+ model_id: "zai-org/GLM-5.3",
64094
+ model_id_status: "confirmed",
64095
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64096
+ model_family: "GLM-5.3",
64097
+ lumic_model: "zai-org/GLM-5.3",
64098
+ params: {
64099
+ temperature: null,
64100
+ max_output_tokens: null,
64101
+ supports_temperature: true,
64102
+ supports_json_schema: true,
64103
+ supports_tools: true,
64104
+ supports_cache_control: true,
64105
+ supports_vision: false,
64106
+ context_window: 1048576
64107
+ },
64108
+ price_per_mtok: {
64109
+ input: 1.2,
64110
+ output: 4,
64111
+ as_of: "2026-09-11",
64112
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64113
+ },
64114
+ notes: "A separate lineage from the primary rather than a smaller sibling, so a DeepSeek-family regression cannot take both legs of this chain at once."
64115
+ },
64116
+ {
64117
+ role: "closed_incumbent",
64118
+ provider: "anthropic",
64119
+ model_id: "claude-opus-4-7",
64120
+ model_id_status: "confirmed",
64121
+ model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
64122
+ model_family: "Opus-class",
64123
+ lumic_model: "claude-opus-4-7",
64124
+ params: {
64125
+ temperature: null,
64126
+ max_output_tokens: 128000,
64127
+ supports_temperature: true,
64128
+ supports_json_schema: true,
64129
+ supports_tools: true,
64130
+ supports_cache_control: true,
64131
+ supports_vision: true,
64132
+ context_window: 1000000
64133
+ },
64134
+ price_per_mtok: {
64135
+ input: 5,
64136
+ output: 25,
64137
+ as_of: "2026-09-10",
64138
+ source: "docs/llm-provider-migration.md#3 Opus-class anchor; matches lumic-utils/src/functions/llm-config.ts"
64139
+ }
64140
+ }
64141
+ ]
64142
+ },
64143
+ "llm.agentic": {
64144
+ workload: "Hardest long-horizon agentic work",
64145
+ latency_class: "background",
64146
+ criticality: "ops",
64147
+ isolation_capable: true,
64148
+ eval_gate: "tool-call",
64149
+ policy_note: "Section 3 names the primary as Fable-class. The strongest Anthropic model registered in the lumic model registry — the table this codebase actually routes on — is claude-opus-4-7, so that is the configured id. Registering a Fable-class model in @adaptic/lumic-utils would change live model selection for every consumer of the advanced tier and is therefore a separate, evidence-gated decision, recorded as open item OI-02 rather than smuggled in here.",
64150
+ budget: {
64151
+ basis: "provisional-pre-baseline",
64152
+ monthly_usd: 1500,
64153
+ alert_pct: [
64154
+ 50,
64155
+ 80,
64156
+ 100
64157
+ ]
64158
+ },
64159
+ routes: [
64160
+ {
64161
+ role: "primary",
64162
+ provider: "deepinfra",
64163
+ model_id: "moonshotai/Kimi-K3",
64164
+ model_id_status: "confirmed",
64165
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64166
+ model_family: "Kimi K3",
64167
+ lumic_model: "moonshotai/Kimi-K3",
64168
+ params: {
64169
+ temperature: null,
64170
+ max_output_tokens: null,
64171
+ supports_temperature: true,
64172
+ supports_json_schema: true,
64173
+ supports_tools: true,
64174
+ supports_cache_control: true,
64175
+ supports_vision: true,
64176
+ context_window: 1048576
64177
+ },
64178
+ price_per_mtok: {
64179
+ input: 2.85,
64180
+ output: 14.25,
64181
+ as_of: "2026-09-11",
64182
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64183
+ },
64184
+ notes: "2.8T open-weight, built explicitly for long-horizon agentic workflows: tool calling, repository navigation, and iterating on logs and test feedback. The most expensive open model selected, which is justified here because this is the hardest workload and the lowest volume."
64185
+ },
64186
+ {
64187
+ role: "secondary",
64188
+ provider: "deepinfra",
64189
+ model_id: "deepseek-ai/DeepSeek-V4-Pro",
64190
+ model_id_status: "confirmed",
64191
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64192
+ model_family: "DeepSeek V4 Pro",
64193
+ lumic_model: "deepseek-ai/DeepSeek-V4-Pro",
64194
+ params: {
64195
+ temperature: null,
64196
+ max_output_tokens: null,
64197
+ supports_temperature: true,
64198
+ supports_json_schema: true,
64199
+ supports_tools: true,
64200
+ supports_cache_control: true,
64201
+ supports_vision: false,
64202
+ context_window: 1048576
64203
+ },
64204
+ price_per_mtok: {
64205
+ input: 1.3,
64206
+ output: 2.6,
64207
+ as_of: "2026-09-11",
64208
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64209
+ },
64210
+ notes: "Also cites long-running agent tasks, at a fifth of the primary's output price."
64211
+ },
64212
+ {
64213
+ role: "closed_incumbent",
64214
+ provider: "anthropic",
64215
+ model_id: "claude-opus-4-7",
64216
+ model_id_status: "confirmed",
64217
+ model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
64218
+ model_family: "Fable-class (see policy_note)",
64219
+ lumic_model: "claude-opus-4-7",
64220
+ params: {
64221
+ temperature: null,
64222
+ max_output_tokens: 128000,
64223
+ supports_temperature: true,
64224
+ supports_json_schema: true,
64225
+ supports_tools: true,
64226
+ supports_cache_control: true,
64227
+ supports_vision: true,
64228
+ context_window: 1000000
64229
+ },
64230
+ price_per_mtok: {
64231
+ input: 10,
64232
+ output: 50,
64233
+ as_of: "2026-09-10",
64234
+ source: "docs/llm-provider-migration.md#3 Fable-class anchor"
64235
+ }
64236
+ }
64237
+ ]
64238
+ },
64239
+ "llm.fast": {
64240
+ workload: "Latency-critical paths",
64241
+ latency_class: "hot-path",
64242
+ criticality: "trading-adjacent",
64243
+ isolation_capable: false,
64244
+ eval_gate: "structured-output",
64245
+ policy_note: "Trading-adjacent and hot-path, so under Section 7 this alias is the LAST to promote: no hot-path alias reaches LIVE until every background alias has completed W5-03 and held steady state for a week.",
64246
+ budget: {
64247
+ basis: "provisional-pre-baseline",
64248
+ monthly_usd: 400,
64249
+ alert_pct: [
64250
+ 50,
64251
+ 80,
64252
+ 100
64253
+ ]
64254
+ },
64255
+ routes: [
64256
+ {
64257
+ role: "primary",
64258
+ provider: "deepinfra",
64259
+ model_id: "deepseek-ai/DeepSeek-V4-Flash",
64260
+ model_id_status: "confirmed",
64261
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64262
+ model_family: "DeepSeek V4 Flash",
64263
+ lumic_model: "deepseek-ai/DeepSeek-V4-Flash",
64264
+ params: {
64265
+ temperature: null,
64266
+ max_output_tokens: null,
64267
+ supports_temperature: true,
64268
+ supports_json_schema: true,
64269
+ supports_tools: true,
64270
+ supports_cache_control: true,
64271
+ supports_vision: false,
64272
+ context_window: 1048576
64273
+ },
64274
+ price_per_mtok: {
64275
+ input: 0.09,
64276
+ output: 0.18,
64277
+ as_of: "2026-09-11",
64278
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64279
+ },
64280
+ notes: "284B MoE, 13B active, 1M context. Measured median 1241 ms on a realistic trading prompt, and the only candidate that answered in directly usable form instead of narrating its reasoning first — which on a hot path is the difference between a parseable answer and a parsing problem."
64281
+ },
64282
+ {
64283
+ role: "secondary",
64284
+ provider: "deepinfra",
64285
+ model_id: "zai-org/GLM-5.3-Flash",
64286
+ model_id_status: "confirmed",
64287
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64288
+ model_family: "GLM-5.3 Flash",
64289
+ lumic_model: "zai-org/GLM-5.3-Flash",
64290
+ params: {
64291
+ temperature: null,
64292
+ max_output_tokens: null,
64293
+ supports_temperature: true,
64294
+ supports_json_schema: true,
64295
+ supports_tools: true,
64296
+ supports_cache_control: true,
64297
+ supports_vision: true,
64298
+ context_window: 1048576
64299
+ },
64300
+ price_per_mtok: {
64301
+ input: 0.15,
64302
+ output: 0.5,
64303
+ as_of: "2026-09-11",
64304
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64305
+ },
64306
+ notes: "A different lineage from the primary, with a 1M context and JSON mode. Chosen over the faster Nemotron 3 Nano, which measured the quickest median of any candidate but rejects response_format outright with HTTP 405 — and llm.fast is a structured-output alias, so a leg that cannot express the contract is not a fallback at all, only a slower way to fail."
64307
+ },
64308
+ {
64309
+ role: "closed_incumbent",
64310
+ provider: "anthropic",
64311
+ model_id: "claude-haiku-4-5",
64312
+ model_id_status: "confirmed",
64313
+ model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
64314
+ model_family: "Haiku-class",
64315
+ lumic_model: "claude-haiku-4-5",
64316
+ params: {
64317
+ temperature: null,
64318
+ max_output_tokens: 64000,
64319
+ supports_temperature: true,
64320
+ supports_json_schema: true,
64321
+ supports_tools: true,
64322
+ supports_cache_control: true,
64323
+ supports_vision: true,
64324
+ context_window: 200000
64325
+ },
64326
+ price_per_mtok: {
64327
+ input: 1,
64328
+ output: 5,
64329
+ as_of: "2026-09-10",
64330
+ source: "docs/llm-provider-migration.md#3 Haiku-class anchor"
64331
+ }
64332
+ }
64333
+ ]
64334
+ },
64335
+ "llm.extract": {
64336
+ workload: "High-volume extraction, classification, log analysis",
64337
+ latency_class: "batch",
64338
+ criticality: "internal",
64339
+ isolation_capable: true,
64340
+ eval_gate: "structured-output",
64341
+ budget: {
64342
+ basis: "provisional-pre-baseline",
64343
+ monthly_usd: 300,
64344
+ alert_pct: [
64345
+ 50,
64346
+ 80,
64347
+ 100
64348
+ ]
64349
+ },
64350
+ routes: [
64351
+ {
64352
+ role: "primary",
64353
+ provider: "deepinfra",
64354
+ model_id: "openai/gpt-oss-20b",
64355
+ model_id_status: "confirmed",
64356
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64357
+ model_family: "gpt-oss-20B",
64358
+ lumic_model: "openai/gpt-oss-20b",
64359
+ params: {
64360
+ temperature: null,
64361
+ max_output_tokens: null,
64362
+ supports_temperature: true,
64363
+ supports_json_schema: true,
64364
+ supports_tools: true,
64365
+ supports_cache_control: false,
64366
+ supports_vision: false,
64367
+ context_window: 131072
64368
+ },
64369
+ price_per_mtok: {
64370
+ input: 0.03,
64371
+ output: 0.14,
64372
+ as_of: "2026-09-11",
64373
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64374
+ },
64375
+ notes: "The cheapest tool-capable model in the catalogue at $0.03/$0.14, which is what a high-volume classification and log-analysis workload should be paying."
64376
+ },
64377
+ {
64378
+ role: "secondary",
64379
+ provider: "deepinfra",
64380
+ model_id: "inclusionAI/Ling-3.0-flash-Fin",
64381
+ model_id_status: "confirmed",
64382
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64383
+ model_family: "Ling 3.0 Flash Fin",
64384
+ lumic_model: "inclusionAI/Ling-3.0-flash-Fin",
64385
+ params: {
64386
+ temperature: null,
64387
+ max_output_tokens: null,
64388
+ supports_temperature: true,
64389
+ supports_json_schema: true,
64390
+ supports_tools: true,
64391
+ supports_cache_control: true,
64392
+ supports_vision: false,
64393
+ context_window: 262144
64394
+ },
64395
+ price_per_mtok: {
64396
+ input: 0.06,
64397
+ output: 0.18,
64398
+ as_of: "2026-09-11",
64399
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64400
+ },
64401
+ notes: "Finance-enhanced through continued training on financial data, with a 262k context. A deliberate second opinion on financial text rather than a generic fallback."
64402
+ },
64403
+ {
64404
+ role: "closed_incumbent",
64405
+ provider: "anthropic",
64406
+ model_id: "claude-haiku-4-5",
64407
+ model_id_status: "confirmed",
64408
+ model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
64409
+ model_family: "Haiku-class",
64410
+ lumic_model: "claude-haiku-4-5",
64411
+ params: {
64412
+ temperature: null,
64413
+ max_output_tokens: 64000,
64414
+ supports_temperature: true,
64415
+ supports_json_schema: true,
64416
+ supports_tools: true,
64417
+ supports_cache_control: true,
64418
+ supports_vision: true,
64419
+ context_window: 200000
64420
+ },
64421
+ price_per_mtok: {
64422
+ input: 1,
64423
+ output: 5,
64424
+ as_of: "2026-09-10",
64425
+ source: "docs/llm-provider-migration.md#3 Haiku-class anchor"
64426
+ }
64427
+ }
64428
+ ]
64429
+ },
64430
+ "llm.judge": {
64431
+ workload: "Eval judging only",
64432
+ latency_class: "batch",
64433
+ criticality: "internal",
64434
+ isolation_capable: false,
64435
+ pinned: true,
64436
+ eval_gate: "none-pinned-judge",
64437
+ policy_note: "PD-6: pinned and never auto-swapped, and deliberately carries no fallback. Moved off the closed incumbent because a judge running on a vendor we are migrating away from cannot impartially grade that migration — it would be scoring the challengers against its own family.",
64438
+ budget: {
64439
+ basis: "provisional-pre-baseline",
64440
+ monthly_usd: 150,
64441
+ alert_pct: [
64442
+ 50,
64443
+ 80,
64444
+ 100
64445
+ ]
64446
+ },
64447
+ routes: [
64448
+ {
64449
+ role: "primary",
64450
+ provider: "deepinfra",
64451
+ model_id: "deepseek-ai/DeepSeek-V4-Pro",
64452
+ model_id_status: "confirmed",
64453
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
64454
+ model_family: "DeepSeek V4 Pro",
64455
+ lumic_model: "deepseek-ai/DeepSeek-V4-Pro",
64456
+ params: {
64457
+ temperature: null,
64458
+ max_output_tokens: null,
64459
+ supports_temperature: true,
64460
+ supports_json_schema: true,
64461
+ supports_tools: true,
64462
+ supports_cache_control: true,
64463
+ supports_vision: false,
64464
+ context_window: 1048576
64465
+ },
64466
+ price_per_mtok: {
64467
+ input: 1.3,
64468
+ output: 2.6,
64469
+ as_of: "2026-09-11",
64470
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
64471
+ },
64472
+ notes: "Pinned with no fallback (PD-6). A judge that failed over would grade one model's output against another model's standard, corrupting every gate decided by it."
64473
+ }
64474
+ ]
64475
+ }
64476
+ };
64477
+ var open_items = [
64478
+ {
64479
+ id: "OI-02",
64480
+ question: "Section 3 names a Fable-class primary for llm.agentic, but the lumic model registry's strongest registered Anthropic model is claude-opus-4-7. Register a Fable-class model in @adaptic/lumic-utils, or accept opus-4-7 as the llm.agentic primary?",
64481
+ blocks: "llm.agentic primary fidelity to Section 3",
64482
+ owner: "CEO"
64483
+ },
64484
+ {
64485
+ id: "OI-03",
64486
+ question: "Every alias budget is provisional-pre-baseline because Section 9's baseline monthly spend is still FILL. Which caps apply once W1-03 produces the billing baseline?",
64487
+ blocks: "PD-7 layer 3 calibration; Section 9 savings target measurement",
64488
+ owner: "CEO"
64489
+ },
64490
+ {
64491
+ id: "OI-04",
64492
+ question: "llm.agentic's serving chain is one leg long because Section 3 makes its primary the closed incumbent and its only secondary shadow-eval-only. An Anthropic outage therefore leaves the alias with no configured fallback. Widen it (a second closed provider, or promoting the Kimi leg out of shadow), or accept the single leg?",
64493
+ blocks: "llm.agentic availability under a primary-provider outage",
64494
+ owner: "CEO"
64495
+ }
64496
+ ];
64497
+ var rawTable = {
64498
+ schema_version: schema_version,
64499
+ policy_source: policy_source,
64500
+ revised: revised,
64501
+ defaults: defaults,
64502
+ providers: providers,
64503
+ aliases: aliases,
64504
+ open_items: open_items
64505
+ };
64506
+
64507
+ /**
64508
+ * Typed access to the canonical alias route table.
64509
+ *
64510
+ * The table is imported rather than fetched so an alias always resolves, even
64511
+ * when the gateway is unreachable. A client that could only learn its routes
64512
+ * from the gateway would have no way to fall back when the gateway itself is
64513
+ * the thing that failed, which would make the mandatory fallback chain of PD-3
64514
+ * conditional on the very component it exists to survive.
64515
+ *
64516
+ * Resolution is deliberately conservative in three ways. An unknown alias is an
64517
+ * error rather than a default, because a typo silently served by whichever
64518
+ * model happened to be configured is worse than a loud failure. A route whose
64519
+ * vendor model id is unconfirmed is excluded, because a guessed id fails at the
64520
+ * first live call rather than at review. And a shadow-only route is never
64521
+ * served, because a candidate that can answer a live request has stopped being
64522
+ * a candidate.
64523
+ *
64524
+ * @module llm/route-table
64525
+ */
64526
+ /** Chain order, fixed by role rather than by authoring order in the file. */
64527
+ const ROLE_ORDER = ["primary", "secondary", "closed_incumbent"];
64528
+ /** Suffix naming an alias's research-isolated variant (PD-9). */
64529
+ const ISOLATED_SUFFIX = ".isolated";
64530
+ /**
64531
+ * The canonical route table.
64532
+ *
64533
+ * Exposed as a readonly view so a consumer can inspect routing without being
64534
+ * able to mutate it. A table that could be edited at runtime would let one
64535
+ * caller change every other caller's routing.
64536
+ */
64537
+ const routeTable = rawTable;
64538
+ /**
64539
+ * Every alias the table defines.
64540
+ *
64541
+ * @returns The alias names, sorted for stable iteration.
64542
+ */
64543
+ function listAliases() {
64544
+ return Object.keys(routeTable.aliases).sort();
64545
+ }
64546
+ /**
64547
+ * Look up an alias definition.
64548
+ *
64549
+ * @param alias The alias to resolve.
64550
+ * @returns Its definition.
64551
+ * @throws When the alias is not defined by the route table.
64552
+ */
64553
+ function aliasDefinition(alias) {
64554
+ const definition = routeTable.aliases[alias];
64555
+ if (definition === undefined) {
64556
+ throw new UnknownAliasError(alias, listAliases());
64557
+ }
64558
+ return definition;
64559
+ }
64560
+ /**
64561
+ * Look up a provider registry entry.
64562
+ *
64563
+ * @param name The provider key.
64564
+ * @returns Its registry entry.
64565
+ * @throws When the provider is not registered.
64566
+ */
64567
+ function providerEntry(name) {
64568
+ const provider = routeTable.providers[name];
64569
+ if (provider === undefined) {
64570
+ throw new Error(`route table names provider "${name}", which is not in the provider registry`);
64571
+ }
64572
+ return provider;
64573
+ }
64574
+ /** Thrown when a caller names an alias the route table does not define. */
64575
+ class UnknownAliasError extends Error {
64576
+ /** The alias that was requested. */
64577
+ alias;
64578
+ /** The aliases that do exist, so the message is actionable. */
64579
+ known;
64580
+ /**
64581
+ * @param alias The unrecognised alias.
64582
+ * @param known The aliases the table defines.
64583
+ */
64584
+ constructor(alias, known) {
64585
+ super(`unknown LLM alias "${alias}". Known aliases: ${known.join(", ")}. ` +
64586
+ "Application code names aliases only; adding one is a change to the route table.");
64587
+ this.name = "UnknownAliasError";
64588
+ this.alias = alias;
64589
+ this.known = known;
64590
+ }
64591
+ }
64592
+ /** Thrown when an alias exists but has no leg that can currently serve a caller. */
64593
+ class NoServableRouteError extends Error {
64594
+ /** The alias that could not be served. */
64595
+ alias;
64596
+ /** Why each of its legs was excluded, in chain order. */
64597
+ exclusions;
64598
+ /**
64599
+ * @param alias The alias.
64600
+ * @param exclusions Per-leg reasons, in chain order.
64601
+ */
64602
+ constructor(alias, exclusions) {
64603
+ super(`alias "${alias}" has no servable route. Legs excluded: ${exclusions.join("; ")}`);
64604
+ this.name = "NoServableRouteError";
64605
+ this.alias = alias;
64606
+ this.exclusions = exclusions;
64607
+ }
64608
+ }
64609
+ /**
64610
+ * Order an alias's routes into the chain the client walks.
64611
+ *
64612
+ * @param definition The alias definition.
64613
+ * @returns Routes ordered primary -> secondary -> closed incumbent.
64614
+ */
64615
+ function orderedRoutes(definition) {
64616
+ return [...definition.routes].sort((a, b) => ROLE_ORDER.indexOf(a.role) - ROLE_ORDER.indexOf(b.role));
64617
+ }
64618
+ /**
64619
+ * Stable identity for one leg, used to key its circuit breaker and its metrics.
64620
+ *
64621
+ * Keyed by alias, isolation and role rather than by provider and model, because
64622
+ * the breaker guards a position in a chain: reverting an alias to a different
64623
+ * model at the same position should inherit that position's health rather than
64624
+ * start blind, and the isolated variant must never share a breaker with the
64625
+ * shared one.
64626
+ *
64627
+ * @param alias The alias.
64628
+ * @param isolated Whether this is the isolated variant.
64629
+ * @param role The leg's role.
64630
+ * @returns The route key.
64631
+ */
64632
+ function routeKeyFor(alias, isolated, role) {
64633
+ return `${alias}${isolated ? ISOLATED_SUFFIX : ""}#${role}`;
64634
+ }
64635
+ /**
64636
+ * Whether one route may serve, and if not, why.
64637
+ *
64638
+ * Extracted from the chain resolver rather than left inline so the admission
64639
+ * rules can be exercised directly. Inline, the only way to prove an exclusion
64640
+ * rule works was to point it at the live table and hope the table still
64641
+ * contained something excludable — which made the proof a statement about how
64642
+ * incomplete the routing policy happened to be that week, and turned finishing
64643
+ * the policy into a test failure. A rule worth enforcing has to be provable on
64644
+ * a route constructed to violate it.
64645
+ *
64646
+ * Order matters: shadow-only is checked before model-id confirmation because a
64647
+ * shadow leg is held back by policy regardless of whether its id is confirmed,
64648
+ * and reporting it as "unconfirmed" would misdescribe a deliberate choice as an
64649
+ * unfinished one.
64650
+ *
64651
+ * The admitting branch carries the confirmed model id rather than leaving the
64652
+ * caller to re-read it. The id is the very thing admission validated, so
64653
+ * handing it back is what lets the caller use it without a non-null assertion
64654
+ * re-stating a check that already happened.
64655
+ *
64656
+ * @param route The authored route.
64657
+ * @param provider The provider entry the route names.
64658
+ * @returns Admission with the confirmed model id, or the reason for exclusion.
64659
+ */
64660
+ function routeAdmission(route, provider) {
64661
+ if (route.shadow_only === true) {
64662
+ return {
64663
+ admit: false,
64664
+ reason: "shadow-only: configured and scored, never served to a caller",
64665
+ };
64666
+ }
64667
+ if (route.model_id_status !== "confirmed" || route.model_id === null) {
64668
+ return {
64669
+ admit: false,
64670
+ reason: `model id unconfirmed (${route.model_family}); it is transcribed from the provider console at onboarding, never guessed`,
64671
+ };
64672
+ }
64673
+ if (provider.account_status !== "live") {
64674
+ return {
64675
+ admit: false,
64676
+ reason: `provider account status is "${provider.account_status}"`,
64677
+ };
64678
+ }
64679
+ return { admit: true, modelId: route.model_id };
64680
+ }
64681
+ /**
64682
+ * Resolve an alias to the ordered chain of legs that can serve it today.
64683
+ *
64684
+ * Exclusions are returned rather than discarded so an exhausted chain can say
64685
+ * why each leg was unavailable. "No route available" without that detail sends
64686
+ * an operator to read config; with it, the answer is in the error.
64687
+ *
64688
+ * @param alias The alias to resolve.
64689
+ * @param options Resolution options.
64690
+ * @param options.isolated Route through the isolated variant (PD-9).
64691
+ * @param options.timeoutMsOverride Per-leg budget override, never wider than the caller's own deadline.
64692
+ * @returns The resolved chain.
64693
+ * @throws {UnknownAliasError} When the alias is not defined.
64694
+ */
64695
+ function resolveChain(alias, options = {}) {
64696
+ const definition = aliasDefinition(alias);
64697
+ const isolated = options.isolated === true;
64698
+ if (isolated && !definition.isolation_capable) {
64699
+ throw new Error(`alias "${alias}" is not isolation-capable, so it has no isolated variant. ` +
64700
+ "PD-9 forbids serving isolated work from a shared route, so this fails rather than falling back.");
64701
+ }
64702
+ const timeoutMs = options.timeoutMsOverride ??
64703
+ routeTable.defaults.request_timeout_ms[definition.latency_class];
64704
+ const resolved = [];
64705
+ const exclusions = [];
64706
+ for (const route of orderedRoutes(definition)) {
64707
+ const provider = providerEntry(route.provider);
64708
+ const admission = routeAdmission(route, provider);
64709
+ if (!admission.admit) {
64710
+ exclusions.push({
64711
+ role: route.role,
64712
+ provider: route.provider,
64713
+ reason: admission.reason,
64714
+ });
64715
+ continue;
64716
+ }
64717
+ resolved.push({
64718
+ alias,
64719
+ isolated,
64720
+ role: route.role,
64721
+ providerName: route.provider,
64722
+ provider,
64723
+ modelId: admission.modelId,
64724
+ lumicModel: route.lumic_model ?? null,
64725
+ params: route.params ?? {},
64726
+ routeKey: routeKeyFor(alias, isolated, route.role),
64727
+ timeoutMs,
64728
+ retriesPerLeg: routeTable.defaults.retries_per_leg,
64729
+ });
64730
+ }
64731
+ return { alias, isolated, routes: resolved, exclusions };
64732
+ }
64733
+ /**
64734
+ * The permanent closed-incumbent leg of an alias (PD-11).
64735
+ *
64736
+ * Found by provider tier rather than by role, because an alias whose primary is
64737
+ * already a closed vendor has that primary as its revert target. Both shapes
64738
+ * therefore hold the same guarantee: there is always a configured route that
64739
+ * needs no new account and no code change to fall back to.
64740
+ *
64741
+ * @param chain A resolved chain.
64742
+ * @returns The closed leg, or undefined when none is currently servable.
64743
+ */
64744
+ function closedIncumbentLeg(chain) {
64745
+ return chain.routes.find((route) => route.provider.tier === "closed");
64746
+ }
64747
+ /**
64748
+ * The name the gateway knows a leg by.
64749
+ *
64750
+ * The head of a chain is addressed by the bare alias so a caller never learns a
64751
+ * leg name; every other leg carries the derived name the renderer gives it.
64752
+ * Keeping this derivation beside the resolver — rather than in the transport —
64753
+ * is what stops the client and the gateway config from disagreeing about a name.
64754
+ *
64755
+ * @param route The resolved leg.
64756
+ * @param chain The chain it belongs to.
64757
+ * @returns The gateway `model_name`.
64758
+ */
64759
+ function gatewayModelNameFor(route, chain) {
64760
+ const base = `${route.alias}${route.isolated ? ISOLATED_SUFFIX : ""}`;
64761
+ return chain.routes[0]?.routeKey === route.routeKey
64762
+ ? base
64763
+ : `${base}.fallback.${route.role}`;
64764
+ }
64765
+
64766
+ /**
64767
+ * Bounded validate-and-retry for structured output.
64768
+ *
64769
+ * A model asked for a schema-shaped answer sometimes returns something close
64770
+ * but wrong. Feeding the validator's own complaint back once fixes most of
64771
+ * those, because the model can see precisely what it got wrong. A second retry
64772
+ * almost never helps: if the model is still wrong after being told exactly what
64773
+ * was wrong, it is persistently wrong for this prompt, and further attempts
64774
+ * spend budget and latency to arrive at the same place.
64775
+ *
64776
+ * The retry lives ahead of the fallback chain rather than inside it. A payload
64777
+ * that fails validation is evidence about the PROMPT, not about the provider's
64778
+ * health, so it must not open a circuit breaker or advance to the next leg —
64779
+ * the next leg would receive the same prompt and be no likelier to satisfy it.
64780
+ *
64781
+ * Exhaustion throws. Returning a partial or defaulted object would hand the
64782
+ * caller a well-typed value that no model ever produced, and the resulting
64783
+ * decision would be made on invented data with nothing to show it.
64784
+ *
64785
+ * @module llm/schema-retry
64786
+ */
64787
+ /** Attempts allowed: the first, plus exactly one feedback retry. */
64788
+ const MAX_ATTEMPTS = 2;
64789
+ /** Characters of an invalid payload echoed back to the model. */
64790
+ const PAYLOAD_EXCERPT = 2000;
64791
+ /**
64792
+ * Thrown when both attempts failed validation.
64793
+ *
64794
+ * Carries both rejection reasons, because the pair is what distinguishes a
64795
+ * flaky answer from a prompt the model cannot satisfy: two different complaints
64796
+ * suggest instability, while the same complaint twice points at the prompt or
64797
+ * the schema.
64798
+ */
64799
+ class SchemaRetryExhaustedError extends Error {
64800
+ /** Why the first attempt was rejected. */
64801
+ firstReason;
64802
+ /** Why the retry was rejected. */
64803
+ secondReason;
64804
+ /** Usage spent across both attempts, so the spend is still accounted for. */
64805
+ totalUsage;
64806
+ /**
64807
+ * @param firstReason Validator's complaint about attempt one.
64808
+ * @param secondReason Validator's complaint about attempt two.
64809
+ * @param totalUsage Usage across both attempts.
64810
+ */
64811
+ constructor(firstReason, secondReason, totalUsage) {
64812
+ super("LLM structured output failed validation twice. " +
64813
+ `First: ${firstReason}. After feedback: ${secondReason}. ` +
64814
+ "No value is returned: a defaulted object would be data no model produced.");
64815
+ this.name = "SchemaRetryExhaustedError";
64816
+ this.firstReason = firstReason;
64817
+ this.secondReason = secondReason;
64818
+ this.totalUsage = totalUsage;
64819
+ }
64820
+ }
64821
+ /**
64822
+ * Build the retry prompt.
64823
+ *
64824
+ * The rejected payload is echoed back alongside the complaint, because a model
64825
+ * asked to "fix the error" without seeing what it produced will usually
64826
+ * regenerate from scratch and reproduce the same mistake.
64827
+ *
64828
+ * @param originalPrompt The prompt that produced the invalid payload.
64829
+ * @param rejected The payload that failed.
64830
+ * @param reason The validator's complaint.
64831
+ * @returns The retry prompt.
64832
+ */
64833
+ function buildRetryPrompt(originalPrompt, rejected, reason) {
64834
+ const excerpt = typeof rejected === "string"
64835
+ ? rejected
64836
+ : JSON.stringify(rejected, null, 2) ?? String(rejected);
64837
+ return [
64838
+ originalPrompt,
64839
+ "",
64840
+ "Your previous response was rejected by a schema validator.",
64841
+ "",
64842
+ "Previous response:",
64843
+ excerpt.slice(0, PAYLOAD_EXCERPT),
64844
+ "",
64845
+ `Validator rejection: ${reason}`,
64846
+ "",
64847
+ "Return a corrected response that satisfies the schema. Return only the corrected response.",
64848
+ ].join("\n");
64849
+ }
64850
+ /**
64851
+ * Run a call with one validator-feedback retry.
64852
+ *
64853
+ * @param options Retry options.
64854
+ * @param options.prompt The original prompt.
64855
+ * @param options.validate Validator applied to each attempt's payload.
64856
+ * @param options.call Executes one attempt with the given prompt.
64857
+ * @returns The validated value with combined usage.
64858
+ * @throws {SchemaRetryExhaustedError} When both attempts fail validation.
64859
+ */
64860
+ async function callWithValidation(options) {
64861
+ const first = await options.call(options.prompt);
64862
+ const firstOutcome = options.validate(first.response);
64863
+ if (firstOutcome.ok) {
64864
+ return {
64865
+ value: firstOutcome.value,
64866
+ attempts: 1,
64867
+ response: first,
64868
+ totalUsage: first.usage,
64869
+ };
64870
+ }
64871
+ const retryPrompt = buildRetryPrompt(options.prompt, first.response, firstOutcome.reason);
64872
+ const second = await options.call(retryPrompt);
64873
+ const combined = sumUsage(first.usage, second.usage);
64874
+ const secondOutcome = options.validate(second.response);
64875
+ if (secondOutcome.ok) {
64876
+ return {
64877
+ value: secondOutcome.value,
64878
+ attempts: MAX_ATTEMPTS,
64879
+ response: second,
64880
+ totalUsage: combined,
64881
+ };
64882
+ }
64883
+ throw new SchemaRetryExhaustedError(firstOutcome.reason, secondOutcome.reason, combined);
64884
+ }
64885
+
64886
+ /**
64887
+ * Degraded direct transport: the path used when the gateway itself is gone.
64888
+ *
64889
+ * Routing every call through one proxy concentrates a great deal of value — one
64890
+ * place to swap a model, one place to bound spend, one place to see cost. It
64891
+ * also concentrates risk: without this path, a gateway outage would take every
64892
+ * LLM call in the system down at once, which is a worse failure than any of the
64893
+ * provider outages the gateway exists to survive.
64894
+ *
64895
+ * Two constraints keep this a safety net rather than a second routing policy.
64896
+ * It serves CLOSED-tier legs only, so the degraded path can never be the thing
64897
+ * that silently promotes an open-weight model past its evaluation gates. And it
64898
+ * resolves the model from the same route table the gateway is rendered from, so
64899
+ * degraded traffic reaches the same model the gateway would have chosen.
64900
+ *
64901
+ * The provider SDK is reached through an injected caller, resolved lazily. A
64902
+ * static import would make `@adaptic/utils` load a provider SDK for every
64903
+ * consumer, including those that never make an LLM call, and would harden a
64904
+ * package cycle that is currently only a declaration.
64905
+ *
64906
+ * @module llm/transports/direct
64907
+ */
64908
+ /**
64909
+ * Thrown when the degraded path is asked to serve a leg it must not serve.
64910
+ *
64911
+ * Refusing loudly rather than serving the leg anyway is the point: the whole
64912
+ * value of restricting this path is lost if it quietly widens under pressure,
64913
+ * and pressure is exactly when it runs.
64914
+ */
64915
+ class DirectTransportRefusedError extends Error {
64916
+ /**
64917
+ * @param routeKey The leg that was refused.
64918
+ * @param reason Why it cannot be served directly.
64919
+ */
64920
+ constructor(routeKey, reason) {
64921
+ super(`the degraded direct transport refuses route ${routeKey}: ${reason}. ` +
64922
+ "It exists so a gateway outage does not stop every LLM call, not as a second routing policy.");
64923
+ this.name = "DirectTransportRefusedError";
64924
+ }
64925
+ }
64926
+ /**
64927
+ * Build the degraded direct transport.
64928
+ *
64929
+ * @param config Transport configuration.
64930
+ * @returns A transport that reaches a closed-tier provider without the gateway.
64931
+ */
64932
+ function createDirectTransport(config) {
64933
+ return {
64934
+ name: "direct",
64935
+ async execute(request) {
64936
+ const { route } = request;
64937
+ if (route.provider.tier !== "closed") {
64938
+ throw new DirectTransportRefusedError(route.routeKey, `provider "${route.providerName}" is ${route.provider.tier}-tier; only a closed incumbent may be served without the gateway, so a degraded call can never promote an open route past its evaluation gates`);
64939
+ }
64940
+ if (route.lumicModel === null) {
64941
+ throw new DirectTransportRefusedError(route.routeKey, "the route names no lumic_model, so there is no registered model to call directly");
64942
+ }
64943
+ const call = await config.resolveCaller();
64944
+ const result = await call(request.content, request.responseFormat, {
64945
+ ...request.params,
64946
+ model: route.lumicModel,
64947
+ signal: request.signal,
64948
+ timeout: route.timeoutMs,
64949
+ });
64950
+ return {
64951
+ response: result.response,
64952
+ usage: readUsage$1(result.usage, request),
64953
+ tool_calls: Array.isArray(result.tool_calls)
64954
+ ? result.tool_calls
64955
+ : undefined,
64956
+ };
64957
+ },
64958
+ };
64959
+ }
64960
+ /**
64961
+ * Normalise the incumbent client's usage shape.
64962
+ *
64963
+ * Missing counts stay zero rather than being estimated, for the same reason
64964
+ * they do on the gateway path: an invented token count flows straight into the
64965
+ * budget accounting the spend controls depend on.
64966
+ *
64967
+ * @param usage The incumbent client's usage, if any.
64968
+ * @param request The request it answers.
64969
+ * @returns The normalised usage record.
64970
+ */
64971
+ function readUsage$1(usage, request) {
64972
+ return {
64973
+ prompt_tokens: usage?.prompt_tokens ?? 0,
64974
+ completion_tokens: usage?.completion_tokens ?? 0,
64975
+ reasoning_tokens: usage?.reasoning_tokens,
64976
+ cached_tokens: usage?.cached_tokens,
64977
+ provider: usage?.provider ?? request.route.providerName,
64978
+ model: usage?.model ?? request.route.modelId,
64979
+ cost: usage?.cost ?? 0,
64980
+ };
64981
+ }
64982
+ /**
64983
+ * Package providing the default provider client, resolved at runtime.
64984
+ *
64985
+ * Assembled rather than written as a literal so the module specifier is opaque
64986
+ * to the compiler and the bundler. That is not a trick to dodge a type error:
64987
+ * this package is genuinely OPTIONAL. The stable lineage of `@adaptic/utils`
64988
+ * does not depend on `@adaptic/lumic-utils` at all, the transport is injectable
64989
+ * precisely so a consumer can supply its own, and the default exists only as a
64990
+ * convenience for consumers that already have it installed. A static specifier
64991
+ * would assert a dependency that does not exist and would harden the
64992
+ * utils/lumic-utils package cycle from a declaration into a build-time fact.
64993
+ */
64994
+ const DEFAULT_PROVIDER_CLIENT_PACKAGE = ["@adaptic", "lumic-utils"].join("/");
64995
+ /**
64996
+ * The default caller: the incumbent client in `@adaptic/lumic-utils`.
64997
+ *
64998
+ * Resolved with a dynamic import so this module has no load-time dependency on
64999
+ * that package. When it cannot be loaded the failure names the degraded path
65000
+ * explicitly, because "cannot find module" during an outage is otherwise a
65001
+ * confusing second mystery on top of the first — and a consumer that does not
65002
+ * ship that package is expected to register its own transport rather than to
65003
+ * discover this at the moment the gateway fails.
65004
+ *
65005
+ * @returns The provider-calling function.
65006
+ */
65007
+ async function resolveDefaultDirectCaller() {
65008
+ try {
65009
+ const lumic = (await import(
65010
+ /* @vite-ignore */ DEFAULT_PROVIDER_CLIENT_PACKAGE));
65011
+ const call = lumic.lumic?.llm?.call;
65012
+ if (typeof call !== "function") {
65013
+ throw new Error(`${DEFAULT_PROVIDER_CLIENT_PACKAGE} exposes no lumic.llm.call`);
65014
+ }
65015
+ return call;
65016
+ }
65017
+ catch (error) {
65018
+ throw new Error("the degraded direct transport could not load its provider client: " +
65019
+ `${error instanceof Error ? error.message : String(error)}. ` +
65020
+ "Register a direct transport explicitly via configureLlmClient() where the default is unavailable.");
65021
+ }
65022
+ }
65023
+
65024
+ /**
65025
+ * Gateway transport: the normal path for every LLM call.
65026
+ *
65027
+ * The client sends an alias to the LiteLLM proxy and the proxy resolves it. The
65028
+ * vendor model string therefore exists only in the gateway's configuration and
65029
+ * never in application code (PD-5), which is what makes a model swap a config
65030
+ * change rather than a deploy.
65031
+ *
65032
+ * The client still walks its own chain on top of the gateway's, and the
65033
+ * duplication is deliberate. The gateway's fallbacks cover a provider being
65034
+ * down; the client's cover the gateway being down. Only one of those two can
65035
+ * cover the other, so the outer chain is the one that must exist.
65036
+ *
65037
+ * The gateway key is read from the environment by NAME at call time and never
65038
+ * stored, logged, or included in an error (PD-2). Reading it per call rather
65039
+ * than caching it at import means a rotation takes effect without a restart.
65040
+ *
65041
+ * @module llm/transports/gateway
65042
+ */
65043
+ /** HTTP statuses that mean "try the next leg" rather than "this request is wrong". */
65044
+ const RETRYABLE_STATUSES = new Set([408, 409, 425, 429, 500, 502, 503, 504]);
65045
+ /** Maximum characters of an error body echoed into a message. */
65046
+ const ERROR_BODY_EXCERPT = 400;
65047
+ /**
65048
+ * Thrown when the gateway itself is unreachable, as opposed to a provider
65049
+ * behind it failing.
65050
+ *
65051
+ * The distinction is what licenses the degraded direct path: a 502 from a
65052
+ * provider means try the next leg, while a connection refused from the proxy
65053
+ * means the whole gateway is gone and the chain cannot be walked through it at
65054
+ * all.
65055
+ */
65056
+ class GatewayUnreachableError extends Error {
65057
+ /**
65058
+ * @param baseUrl The gateway that could not be reached.
65059
+ * @param cause The underlying transport error.
65060
+ */
65061
+ constructor(baseUrl, cause) {
65062
+ super(`LLM gateway at ${baseUrl} is unreachable: ${cause instanceof Error ? cause.message : String(cause)}`);
65063
+ this.name = "GatewayUnreachableError";
65064
+ }
65065
+ }
65066
+ /** Thrown when the gateway answered, but with a failure. */
65067
+ class GatewayResponseError extends Error {
65068
+ /** The HTTP status. */
65069
+ status;
65070
+ /** Whether advancing to the next leg could plausibly help. */
65071
+ retryable;
65072
+ /**
65073
+ * @param status The HTTP status.
65074
+ * @param body A bounded excerpt of the response body.
65075
+ */
65076
+ constructor(status, body) {
65077
+ super(`LLM gateway returned ${status}: ${body.slice(0, ERROR_BODY_EXCERPT)}`);
65078
+ this.name = "GatewayResponseError";
65079
+ this.status = status;
65080
+ this.retryable = RETRYABLE_STATUSES.has(status);
65081
+ }
65082
+ }
65083
+ /**
65084
+ * Read the gateway key from the environment by name.
65085
+ *
65086
+ * @param envVar The variable's NAME.
65087
+ * @returns The key.
65088
+ * @throws When the variable is unset, because an unauthenticated call would
65089
+ * reach the gateway as an anonymous caller and be rejected there anyway —
65090
+ * later, and with a less useful message.
65091
+ */
65092
+ function readGatewayKey(envVar) {
65093
+ const value = process.env[envVar];
65094
+ if (value === undefined || value.length === 0) {
65095
+ throw new Error(`${envVar} is unset, so the LLM gateway cannot be authenticated against. ` +
65096
+ "Provision it from the secrets manager; it is never read from a file or a default.");
65097
+ }
65098
+ return value;
65099
+ }
65100
+ /**
65101
+ * Extract usage from a chat-completion response.
65102
+ *
65103
+ * Absent counts stay zero rather than being estimated. A fabricated token count
65104
+ * would flow straight into the budget accounting that the spend controls are
65105
+ * built on, and a budget computed from invented numbers is worse than one that
65106
+ * knows it is missing a call.
65107
+ *
65108
+ * @param payload The parsed response body.
65109
+ * @param request The request it answers.
65110
+ * @returns The usage record.
65111
+ */
65112
+ function readUsage(payload, request) {
65113
+ const usage = (payload.usage ?? {});
65114
+ const details = (usage.prompt_tokens_details ?? {});
65115
+ const cached = details.cached_tokens;
65116
+ const reasoningDetails = (usage.completion_tokens_details ?? {});
65117
+ const reasoning = reasoningDetails.reasoning_tokens;
65118
+ return {
65119
+ prompt_tokens: typeof usage.prompt_tokens === "number" ? usage.prompt_tokens : 0,
65120
+ completion_tokens: typeof usage.completion_tokens === "number" ? usage.completion_tokens : 0,
65121
+ reasoning_tokens: typeof reasoning === "number" ? reasoning : undefined,
65122
+ cached_tokens: typeof cached === "number" ? cached : undefined,
65123
+ provider: request.route.providerName,
65124
+ model: request.route.modelId,
65125
+ cost: typeof usage.response_cost === "number" ? usage.response_cost : 0,
65126
+ };
65127
+ }
65128
+ /**
65129
+ * Build the gateway transport.
65130
+ *
65131
+ * @param config Transport configuration.
65132
+ * @returns A transport that executes one leg through the proxy.
65133
+ */
65134
+ function createGatewayTransport(config) {
65135
+ const doFetch = config.fetchImpl ?? fetch;
65136
+ return {
65137
+ name: "gateway",
65138
+ async execute(request) {
65139
+ const messages = buildMessages(request);
65140
+ const body = {
65141
+ model: config.modelNameFor(request),
65142
+ messages,
65143
+ ...request.params,
65144
+ };
65145
+ let response;
65146
+ try {
65147
+ response = await doFetch(`${config.baseUrl.replace(/\/$/, "")}/chat/completions`, {
65148
+ method: "POST",
65149
+ headers: {
65150
+ "content-type": "application/json",
65151
+ authorization: `Bearer ${readGatewayKey(config.apiKeyEnv)}`,
65152
+ ...(request.correlationId === undefined
65153
+ ? {}
65154
+ : { "x-correlation-id": request.correlationId }),
65155
+ },
65156
+ body: JSON.stringify(body),
65157
+ signal: request.signal,
65158
+ });
65159
+ }
65160
+ catch (error) {
65161
+ // A transport-level throw means the proxy was never reached. Abort is
65162
+ // re-thrown untouched so the chain's timeout classification stays
65163
+ // accurate rather than being masked as a gateway outage.
65164
+ if (request.signal.aborted) {
65165
+ throw error;
65166
+ }
65167
+ throw new GatewayUnreachableError(config.baseUrl, error);
65168
+ }
65169
+ if (!response.ok) {
65170
+ throw new GatewayResponseError(response.status, await response.text());
65171
+ }
65172
+ const payload = (await response.json());
65173
+ const choices = payload.choices;
65174
+ const message = choices?.[0]?.message;
65175
+ return {
65176
+ response: parseContent(message?.content, request.responseFormat),
65177
+ usage: readUsage(payload, request),
65178
+ tool_calls: Array.isArray(message?.tool_calls)
65179
+ ? message.tool_calls
65180
+ : undefined,
65181
+ };
65182
+ },
65183
+ };
65184
+ }
65185
+ /**
65186
+ * Compose the message array for a request.
65187
+ *
65188
+ * @param request The request.
65189
+ * @returns Chat messages.
65190
+ */
65191
+ function buildMessages(request) {
65192
+ const content = typeof request.content === "string" ? request.content : [...request.content];
65193
+ return [{ role: "user", content }];
65194
+ }
65195
+ /**
65196
+ * Interpret the model's content according to the requested format.
65197
+ *
65198
+ * A JSON format that does not parse is an error, not an empty object. Returning
65199
+ * a default here would hand the caller a well-typed value that means nothing,
65200
+ * and the failure would surface much later as a decision made on absent data.
65201
+ *
65202
+ * @param content The raw content.
65203
+ * @param responseFormat The format the caller asked for.
65204
+ * @returns The parsed value.
65205
+ */
65206
+ function parseContent(content, responseFormat) {
65207
+ const text = typeof content === "string" ? content : "";
65208
+ if (responseFormat === "text") {
65209
+ return text;
65210
+ }
65211
+ try {
65212
+ return JSON.parse(text);
65213
+ }
65214
+ catch (error) {
65215
+ throw new Error(`LLM returned content that is not valid JSON for a ${typeof responseFormat === "string" ? responseFormat : "json_schema"} request: ${error instanceof Error ? error.message : String(error)}`);
65216
+ }
65217
+ }
65218
+
65219
+ /**
65220
+ * The alias-resolving LLM client.
65221
+ *
65222
+ * This is the only supported way to reach a language model from application
65223
+ * code. Its signature mirrors the incumbent client's — content, response
65224
+ * format, options — so migrating a call site is replacing a model string with
65225
+ * an alias, and nothing else. That similarity is the point: a migration that
65226
+ * required rewriting call sites would be a migration that stalls half-done,
65227
+ * leaving some calls inside the timeout, breaker and fallback controls and
65228
+ * some outside them, which is worse than either end state.
65229
+ *
65230
+ * There is deliberately no way to name a model. PD-5 makes vendor strings a CI
65231
+ * failure in application code, and an option that accepted one would let a call
65232
+ * site opt out of the routing policy without anyone noticing.
65233
+ *
65234
+ * Every call gets, in order: alias resolution against the canonical route
65235
+ * table, per-provider parameter normalisation, a hard per-leg timeout, a
65236
+ * per-route circuit breaker, an ordered fallback chain ending at the closed
65237
+ * incumbent, and — where the caller supplies a validator — one schema-feedback
65238
+ * retry ahead of the chain. None of them is optional, because a control that a
65239
+ * caller can switch off is a control that will be off on the call that needed
65240
+ * it.
65241
+ *
65242
+ * @module llm/alias-client
65243
+ */
65244
+ /** Env var naming the gateway's base URL. */
65245
+ const GATEWAY_BASE_URL_ENV = "LLM_GATEWAY_BASE_URL";
65246
+ /** Env var NAME holding the gateway key. The key itself is never read here. */
65247
+ const DEFAULT_GATEWAY_KEY_ENV = "LLM_GATEWAY_API_KEY";
65248
+ /** Process-wide breaker registry, so route health is shared across call sites. */
65249
+ let breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker);
65250
+ /** Active runtime wiring. */
65251
+ let config = {};
65252
+ /** Lazily built transports, rebuilt whenever configuration changes. */
65253
+ let gatewayTransport = null;
65254
+ let directTransport = null;
65255
+ /**
65256
+ * Wire the client.
65257
+ *
65258
+ * Called once at process start. Transports are injectable so a consumer can
65259
+ * supply its own instrumented client, and so tests can exercise the chain
65260
+ * without a network — a fallback chain that could only be observed against live
65261
+ * providers would in practice never be observed at all.
65262
+ *
65263
+ * @param next Runtime wiring; unspecified fields fall back to the environment.
65264
+ * @returns void
65265
+ */
65266
+ function configureLlmClient(next) {
65267
+ config = { ...next };
65268
+ gatewayTransport = null;
65269
+ directTransport = null;
65270
+ breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker, next.now);
65271
+ }
65272
+ /**
65273
+ * Inspect route health.
65274
+ *
65275
+ * @returns The live breaker registry.
65276
+ */
65277
+ function llmBreakers() {
65278
+ return breakers;
65279
+ }
65280
+ /**
65281
+ * Resolve the gateway transport, building it on first use.
65282
+ *
65283
+ * @param chain The chain being served, used to derive each leg's gateway name.
65284
+ * @returns The transport, or null when no gateway is configured.
65285
+ */
65286
+ function gatewayFor(chain) {
65287
+ if (config.gatewayTransport !== undefined) {
65288
+ return config.gatewayTransport;
65289
+ }
65290
+ const baseUrl = config.gatewayBaseUrl ?? process.env[GATEWAY_BASE_URL_ENV];
65291
+ if (baseUrl === undefined || baseUrl.length === 0) {
65292
+ return null;
65293
+ }
65294
+ if (gatewayTransport === null) {
65295
+ gatewayTransport = createGatewayTransport({
65296
+ baseUrl,
65297
+ apiKeyEnv: config.gatewayApiKeyEnv ?? DEFAULT_GATEWAY_KEY_ENV,
65298
+ modelNameFor: (request) => gatewayModelNameFor(request.route, chain),
65299
+ });
65300
+ }
65301
+ return gatewayTransport;
65302
+ }
65303
+ /**
65304
+ * Resolve the degraded direct transport, building it on first use.
65305
+ *
65306
+ * @returns The transport.
65307
+ */
65308
+ function directFor() {
65309
+ if (config.directTransport !== undefined) {
65310
+ return config.directTransport;
65311
+ }
65312
+ if (directTransport === null) {
65313
+ directTransport = createDirectTransport({
65314
+ resolveCaller: resolveDefaultDirectCaller,
65315
+ });
65316
+ }
65317
+ return directTransport;
65318
+ }
65319
+ /**
65320
+ * Prepare each leg, normalising parameters up front.
65321
+ *
65322
+ * Normalisation happens before the walk rather than inside it so a capability
65323
+ * mismatch is known without spending a network round trip on it, and so a leg
65324
+ * that cannot serve the request is recorded as skipped rather than counted
65325
+ * against the provider's health.
65326
+ *
65327
+ * @param chain The resolved chain.
65328
+ * @param options The caller's options.
65329
+ * @param responseFormat The requested response shape.
65330
+ * @param transport The transport to carry every leg.
65331
+ * @returns Prepared legs, in chain order.
65332
+ */
65333
+ function prepareLegs(chain, options, responseFormat, transport) {
65334
+ return chain.routes.map((route) => {
65335
+ try {
65336
+ return {
65337
+ route,
65338
+ transport,
65339
+ params: normaliseParams(options, route, responseFormat),
65340
+ };
65341
+ }
65342
+ catch (error) {
65343
+ if (error instanceof UnsupportedCapabilityError) {
65344
+ return { route, transport, params: error };
65345
+ }
65346
+ throw error;
65347
+ }
65348
+ });
65349
+ }
65350
+ /**
65351
+ * Call a language model by semantic alias.
65352
+ *
65353
+ * @param content The prompt, or multi-part content for a vision-capable route.
65354
+ * @param responseFormat The response shape. Defaults to plain text.
65355
+ * @param options Call options; `alias` is required and there is no model option.
65356
+ * @returns The answer, with the routing decision and full attempt record attached.
65357
+ * @throws {UnknownAliasError} When the alias is not in the route table.
65358
+ * @throws {ChainExhaustedError} When no leg produced an answer.
65359
+ * @throws {SchemaRetryExhaustedError} When a validated payload failed twice.
65360
+ */
65361
+ async function callLLMByAlias(content, responseFormat = "text", options) {
65362
+ const chain = resolveChain(options.alias, {
65363
+ isolated: options.isolated,
65364
+ timeoutMsOverride: options.timeoutMs,
65365
+ });
65366
+ if (chain.routes.length === 0) {
65367
+ // Nothing is servable. The exclusions say why each leg was unavailable,
65368
+ // which is the difference between an operator reading config and an
65369
+ // operator reading the error.
65370
+ throw new ChainExhaustedError(options.alias, chain.exclusions.map((exclusion) => ({
65371
+ routeKey: `${options.alias}#${exclusion.role}`,
65372
+ role: exclusion.role,
65373
+ provider: exclusion.provider,
65374
+ modelId: "(unresolved)",
65375
+ outcome: "skipped",
65376
+ durationMs: 0,
65377
+ reason: exclusion.reason,
65378
+ })), {
65379
+ prompt_tokens: 0,
65380
+ completion_tokens: 0,
65381
+ provider: "none",
65382
+ model: "none",
65383
+ cost: 0,
65384
+ });
65385
+ }
65386
+ const gateway = gatewayFor(chain);
65387
+ const attemptLog = [];
65388
+ /**
65389
+ * Run the chain, falling back from the gateway to the degraded direct path
65390
+ * only when the gateway itself is unreachable.
65391
+ *
65392
+ * @param prompt The prompt for this attempt.
65393
+ * @returns The transport response and the routing facts about it.
65394
+ */
65395
+ const runOnce = async (prompt) => {
65396
+ const boundContent = prompt;
65397
+ if (gateway !== null) {
65398
+ try {
65399
+ const outcome = await executeChain(options.alias, {
65400
+ legs: prepareLegs(chain, options, responseFormat, gateway),
65401
+ content: boundContent,
65402
+ responseFormat,
65403
+ breakers,
65404
+ correlationId: options.correlationId,
65405
+ callerSignal: options.signal,
65406
+ now: config.now,
65407
+ onAttempt: (record) => attemptLog.push(record),
65408
+ });
65409
+ return { ...outcome, degraded: false };
65410
+ }
65411
+ catch (error) {
65412
+ if (!isGatewayOutage(error)) {
65413
+ throw error;
65414
+ }
65415
+ // The proxy is gone, not a provider behind it. Walking the chain again
65416
+ // through the gateway would repeat the same failure on every leg, so
65417
+ // the degraded path takes over — restricted to the closed incumbent.
65418
+ }
65419
+ }
65420
+ const direct = directFor();
65421
+ const closedLegs = chain.routes.filter((route) => route.provider.tier === "closed");
65422
+ if (closedLegs.length === 0) {
65423
+ throw new ChainExhaustedError(options.alias, attemptLog, {
65424
+ prompt_tokens: 0,
65425
+ completion_tokens: 0,
65426
+ provider: "none",
65427
+ model: "none",
65428
+ cost: 0,
65429
+ });
65430
+ }
65431
+ const outcome = await executeChain(options.alias, {
65432
+ legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct),
65433
+ content: boundContent,
65434
+ responseFormat,
65435
+ breakers,
65436
+ correlationId: options.correlationId,
65437
+ callerSignal: options.signal,
65438
+ now: config.now,
65439
+ onAttempt: (record) => attemptLog.push(record),
65440
+ });
65441
+ return { ...outcome, degraded: true };
65442
+ };
65443
+ if (options.validate === undefined) {
65444
+ const outcome = await runOnce(content);
65445
+ return {
65446
+ response: outcome.response.response,
65447
+ usage: outcome.response.usage,
65448
+ tool_calls: outcome.response.tool_calls,
65449
+ servedBy: outcome.servedBy,
65450
+ attempts: attemptLog,
65451
+ degraded: outcome.degraded,
65452
+ totalUsage: outcome.totalUsage,
65453
+ };
65454
+ }
65455
+ // A validator is only meaningful against a text prompt, because the retry has
65456
+ // to be able to append the validator's complaint to it.
65457
+ if (typeof content !== "string") {
65458
+ throw new Error("a validator requires a string prompt: the feedback retry appends the validator's rejection to the original prompt");
65459
+ }
65460
+ let lastRouting = null;
65461
+ const validated = await callWithValidation({
65462
+ prompt: content,
65463
+ validate: options.validate,
65464
+ call: async (prompt) => {
65465
+ const outcome = await runOnce(prompt);
65466
+ lastRouting = {
65467
+ servedBy: outcome.servedBy,
65468
+ degraded: outcome.degraded,
65469
+ totalUsage: outcome.totalUsage,
65470
+ };
65471
+ return outcome.response;
65472
+ },
65473
+ });
65474
+ if (lastRouting === null) {
65475
+ throw new Error("validated call completed without recording a routing decision");
65476
+ }
65477
+ const routing = lastRouting;
65478
+ return {
65479
+ response: validated.value,
65480
+ usage: validated.response.usage,
65481
+ tool_calls: validated.response.tool_calls,
65482
+ servedBy: routing.servedBy,
65483
+ attempts: attemptLog,
65484
+ degraded: routing.degraded,
65485
+ totalUsage: validated.totalUsage,
65486
+ };
65487
+ }
65488
+ /**
65489
+ * Whether an error means the gateway itself is gone.
65490
+ *
65491
+ * Only a transport-level failure to reach the proxy qualifies. A provider error
65492
+ * relayed BY the proxy is a normal chain event and must not trigger the
65493
+ * degraded path, or a single flaky provider would quietly move every call onto
65494
+ * the closed incumbent — a fallback the routing policy reserves for last.
65495
+ *
65496
+ * @param error The error to classify.
65497
+ * @returns Whether the gateway is unreachable.
65498
+ */
65499
+ function isGatewayOutage(error) {
65500
+ if (error instanceof GatewayUnreachableError) {
65501
+ return true;
65502
+ }
65503
+ if (error instanceof ChainExhaustedError) {
65504
+ return (error.attempts.length > 0 &&
65505
+ error.attempts.every((attempt) => attempt.reason === undefined
65506
+ ? false
65507
+ : attempt.reason.includes("is unreachable")));
65508
+ }
65509
+ return false;
65510
+ }
65511
+ /**
65512
+ * The aliases application code may name.
65513
+ *
65514
+ * @returns The alias names, sorted.
65515
+ */
65516
+ function llmAliases() {
65517
+ return Object.keys(routeTable.aliases).sort();
65518
+ }
65519
+
65520
+ /**
65521
+ * Streaming normalisation across provider wire formats.
65522
+ *
65523
+ * A caller that streams wants one thing — text as it arrives — but the wire
65524
+ * formats disagree about how to say it. OpenAI-compatible providers emit
65525
+ * server-sent events whose payload nests the increment under
65526
+ * `choices[0].delta.content` and end with a literal `[DONE]` sentinel;
65527
+ * Anthropic emits typed events where the increment is `delta.text` and the end
65528
+ * is an explicit `message_stop`. A consumer written against one shape breaks on
65529
+ * the other, which would make a fallback across providers fail precisely when
65530
+ * the fallback was needed.
65531
+ *
65532
+ * The one behaviour that matters more than the format is how a stream ENDS. A
65533
+ * stream cut short mid-answer looks exactly like a short answer: the consumer
65534
+ * has already received and probably already acted on the text. So a stream
65535
+ * that stops without its terminal event raises rather than returning what it
65536
+ * had — a truncated answer accepted as complete is a wrong answer that nothing
65537
+ * reports.
65538
+ *
65539
+ * @module llm/streaming
65540
+ */
65541
+ /** SSE payload that marks the end of an OpenAI-compatible stream. */
65542
+ const SSE_DONE_SENTINEL = "[DONE]";
65543
+ /** Prefix carrying the payload on an SSE line. */
65544
+ const SSE_DATA_PREFIX = "data:";
65545
+ /** Anthropic event type carrying a text increment. */
65546
+ const ANTHROPIC_DELTA_EVENT = "content_block_delta";
65547
+ /** Anthropic event type marking the end of a message. */
65548
+ const ANTHROPIC_STOP_EVENT = "message_stop";
65549
+ /** Anthropic event type carrying a mid-stream error. */
65550
+ const ANTHROPIC_ERROR_EVENT = "error";
65551
+ /**
65552
+ * Thrown when a stream ends without its terminal event.
65553
+ *
65554
+ * A distinct type because the caller's correct response differs from a normal
65555
+ * failure: the partial text exists and may be worth logging for diagnosis, but
65556
+ * it must never be treated as the answer.
65557
+ */
65558
+ class StreamTruncatedError extends Error {
65559
+ /** Text received before the stream stopped. Present for diagnosis only. */
65560
+ partialText;
65561
+ /**
65562
+ * @param partialText What had arrived when the stream stopped.
65563
+ * @param reason Why it stopped, when known.
65564
+ */
65565
+ constructor(partialText, reason) {
65566
+ super(`LLM stream ended without a terminal event after ${partialText.length} characters: ${reason}. ` +
65567
+ "The partial text is not returned as an answer: a truncated answer accepted as complete is a wrong answer nothing reports.");
65568
+ this.name = "StreamTruncatedError";
65569
+ this.partialText = partialText;
65570
+ }
65571
+ }
65572
+ /** Thrown when a provider reports an error inside an already-open stream. */
65573
+ class StreamProviderError extends Error {
65574
+ /** Text received before the error. */
65575
+ partialText;
65576
+ /**
65577
+ * @param partialText What had arrived when the error appeared.
65578
+ * @param detail The provider's message.
65579
+ */
65580
+ constructor(partialText, detail) {
65581
+ super(`LLM stream failed mid-response: ${detail}`);
65582
+ this.name = "StreamProviderError";
65583
+ this.partialText = partialText;
65584
+ }
65585
+ }
65586
+ /**
65587
+ * Split a byte stream into complete lines.
65588
+ *
65589
+ * Chunk boundaries fall wherever the network puts them, not on line breaks, so
65590
+ * a partial line is carried across chunks. Parsing each network chunk as if it
65591
+ * were whole would drop or corrupt every event unlucky enough to be split.
65592
+ *
65593
+ * @param source The response body stream.
65594
+ * @returns An async iterable of complete lines.
65595
+ */
65596
+ async function* toLines(source) {
65597
+ const decoder = new TextDecoder();
65598
+ let buffer = "";
65599
+ for await (const chunk of source) {
65600
+ buffer += decoder.decode(chunk, { stream: true });
65601
+ let newline = buffer.indexOf("\n");
65602
+ while (newline !== -1) {
65603
+ yield buffer.slice(0, newline).replace(/\r$/, "");
65604
+ buffer = buffer.slice(newline + 1);
65605
+ newline = buffer.indexOf("\n");
65606
+ }
65607
+ }
65608
+ buffer += decoder.decode();
65609
+ if (buffer.length > 0) {
65610
+ yield buffer;
65611
+ }
65612
+ }
65613
+ /**
65614
+ * Normalise an OpenAI-compatible SSE stream.
65615
+ *
65616
+ * @param source The response body stream.
65617
+ * @returns Normalised chunks.
65618
+ * @throws {StreamTruncatedError} When the stream ends without `[DONE]`.
65619
+ * @throws {StreamProviderError} When an error event appears mid-stream.
65620
+ */
65621
+ async function* normaliseOpenAiStream(source) {
65622
+ let text = "";
65623
+ let sawDone = false;
65624
+ for await (const line of toLines(source)) {
65625
+ if (!line.startsWith(SSE_DATA_PREFIX)) {
65626
+ continue;
65627
+ }
65628
+ const payload = line.slice(SSE_DATA_PREFIX.length).trim();
65629
+ if (payload === SSE_DONE_SENTINEL) {
65630
+ sawDone = true;
65631
+ break;
65632
+ }
65633
+ if (payload.length === 0) {
65634
+ continue;
65635
+ }
65636
+ let event;
65637
+ try {
65638
+ event = JSON.parse(payload);
65639
+ }
65640
+ catch (error) {
65641
+ throw new StreamProviderError(text, `unparseable event: ${error instanceof Error ? error.message : String(error)}`);
65642
+ }
65643
+ if (event.error !== undefined) {
65644
+ throw new StreamProviderError(text, JSON.stringify(event.error));
65645
+ }
65646
+ const choices = event.choices;
65647
+ const delta = choices?.[0]?.delta?.content;
65648
+ if (typeof delta === "string" && delta.length > 0) {
65649
+ text += delta;
65650
+ yield { delta, text };
65651
+ }
65652
+ }
65653
+ if (!sawDone) {
65654
+ throw new StreamTruncatedError(text, "no [DONE] sentinel was received");
65655
+ }
65656
+ }
65657
+ /**
65658
+ * Normalise an Anthropic event stream.
65659
+ *
65660
+ * @param source The response body stream.
65661
+ * @returns Normalised chunks.
65662
+ * @throws {StreamTruncatedError} When the stream ends without `message_stop`.
65663
+ * @throws {StreamProviderError} When an error event appears mid-stream.
65664
+ */
65665
+ async function* normaliseAnthropicStream(source) {
65666
+ let text = "";
65667
+ let sawStop = false;
65668
+ for await (const line of toLines(source)) {
65669
+ if (!line.startsWith(SSE_DATA_PREFIX)) {
65670
+ continue;
65671
+ }
65672
+ const payload = line.slice(SSE_DATA_PREFIX.length).trim();
65673
+ if (payload.length === 0) {
65674
+ continue;
65675
+ }
65676
+ let event;
65677
+ try {
65678
+ event = JSON.parse(payload);
65679
+ }
65680
+ catch (error) {
65681
+ throw new StreamProviderError(text, `unparseable event: ${error instanceof Error ? error.message : String(error)}`);
65682
+ }
65683
+ const type = event.type;
65684
+ if (type === ANTHROPIC_ERROR_EVENT) {
65685
+ throw new StreamProviderError(text, JSON.stringify(event.error ?? event));
65686
+ }
65687
+ if (type === ANTHROPIC_STOP_EVENT) {
65688
+ sawStop = true;
65689
+ break;
65690
+ }
65691
+ if (type === ANTHROPIC_DELTA_EVENT) {
65692
+ const delta = event.delta?.text;
65693
+ if (typeof delta === "string" && delta.length > 0) {
65694
+ text += delta;
65695
+ yield { delta, text };
65696
+ }
65697
+ }
65698
+ }
65699
+ if (!sawStop) {
65700
+ throw new StreamTruncatedError(text, "no message_stop event was received");
65701
+ }
65702
+ }
65703
+ /**
65704
+ * Normalise a stream according to the wire format its provider speaks.
65705
+ *
65706
+ * Selecting on the route's declared API style rather than sniffing the payload
65707
+ * keeps the decision with the route table, which is the thing that actually
65708
+ * knows which provider is answering.
65709
+ *
65710
+ * @param apiStyle The provider's wire format.
65711
+ * @param source The response body stream.
65712
+ * @returns Normalised chunks, identical in shape across providers.
65713
+ */
65714
+ function normaliseStream(apiStyle, source) {
65715
+ return apiStyle === "anthropic"
65716
+ ? normaliseAnthropicStream(source)
65717
+ : normaliseOpenAiStream(source);
65718
+ }
65719
+ /**
65720
+ * Drain a normalised stream into its complete text.
65721
+ *
65722
+ * @param stream A normalised stream.
65723
+ * @returns The full text.
65724
+ * @throws Whatever the stream raises; a truncated stream never resolves to text.
65725
+ */
65726
+ async function collectStream(stream) {
65727
+ let text = "";
65728
+ for await (const chunk of stream) {
65729
+ text = chunk.text;
65730
+ }
65731
+ return text;
65732
+ }
65733
+
62914
65734
  /**
62915
65735
  * @module LRUCache
62916
65736
  */
@@ -72035,7 +74855,7 @@ const adaptic = {
72035
74855
  },
72036
74856
  rateLimiter: {
72037
74857
  TokenBucketRateLimiter,
72038
- limiters: rateLimiters,
74858
+ limiters: rateLimiters$1,
72039
74859
  },
72040
74860
  };
72041
74861
  const adptc = adaptic;
@@ -72070,6 +74890,8 @@ exports.AssetAllocationEngine = AssetAllocationEngine;
72070
74890
  exports.AuthenticationError = AuthenticationError;
72071
74891
  exports.BTC_PAIRS = BTC_PAIRS;
72072
74892
  exports.BarError = BarError;
74893
+ exports.ChainExhaustedError = ChainExhaustedError;
74894
+ exports.CircuitBreakerRegistry = CircuitBreakerRegistry;
72073
74895
  exports.CircuitOpenError = CircuitOpenError;
72074
74896
  exports.CryptoDataError = CryptoDataError;
72075
74897
  exports.CryptoOrderError = CryptoOrderError;
@@ -72078,7 +74900,10 @@ exports.DEFAULT_RISK_FREE_RATE = DEFAULT_RISK_FREE_RATE;
72078
74900
  exports.DEFAULT_TIMEOUTS = DEFAULT_TIMEOUTS;
72079
74901
  exports.DEFAULT_TRADING_POLICY = DEFAULT_TRADING_POLICY;
72080
74902
  exports.DataFormatError = DataFormatError;
74903
+ exports.DirectTransportRefusedError = DirectTransportRefusedError;
72081
74904
  exports.DuplicateClientOrderIdError = DuplicateClientOrderIdError;
74905
+ exports.GatewayResponseError = GatewayResponseError;
74906
+ exports.GatewayUnreachableError = GatewayUnreachableError;
72082
74907
  exports.HttpClientError = HttpClientError;
72083
74908
  exports.HttpServerError = HttpServerError;
72084
74909
  exports.KEEP_ALIVE_DEFAULTS = KEEP_ALIVE_DEFAULTS;
@@ -72095,13 +74920,18 @@ exports.MassiveTradeZodSchema = MassiveTradeSchema;
72095
74920
  exports.MassiveTradesResponseSchema = MassiveTradesResponseSchema;
72096
74921
  exports.NetworkError = NetworkError;
72097
74922
  exports.NewsError = NewsError;
74923
+ exports.NoServableRouteError = NoServableRouteError;
72098
74924
  exports.OptionStrategyError = OptionStrategyError;
72099
74925
  exports.OptionsDataError = OptionsDataError;
72100
74926
  exports.QuoteError = QuoteError;
72101
74927
  exports.RISK_FREE_RATE_TTL_MS = RISK_FREE_RATE_TTL_MS;
74928
+ exports.RateGuardTimeoutError = RateGuardTimeoutError;
72102
74929
  exports.RateLimitError = RateLimitError;
72103
74930
  exports.RawMassivePriceDataSchema = RawMassivePriceDataSchema;
74931
+ exports.SchemaRetryExhaustedError = SchemaRetryExhaustedError;
72104
74932
  exports.StampedeProtectedCache = StampedeProtectedCache;
74933
+ exports.StreamProviderError = StreamProviderError;
74934
+ exports.StreamTruncatedError = StreamTruncatedError;
72105
74935
  exports.TRADING_API = TRADING_API;
72106
74936
  exports.TimeoutError = TimeoutError;
72107
74937
  exports.TokenBucketRateLimiter = TokenBucketRateLimiter;
@@ -72110,7 +74940,9 @@ exports.TrailingStopValidationError = TrailingStopValidationError;
72110
74940
  exports.USDC_PAIRS = USDC_PAIRS;
72111
74941
  exports.USDT_PAIRS = USDT_PAIRS;
72112
74942
  exports.USD_PAIRS = USD_PAIRS;
74943
+ exports.UnknownAliasError = UnknownAliasError;
72113
74944
  exports.UnsupportedBrokerError = UnsupportedBrokerError;
74945
+ exports.UnsupportedCapabilityError = UnsupportedCapabilityError;
72114
74946
  exports.ValidationError = ValidationError;
72115
74947
  exports.ValidationResponseError = ValidationResponseError;
72116
74948
  exports.WEBSOCKET_STREAMS = WEBSOCKET_STREAMS;
@@ -72125,6 +74957,7 @@ exports.atr = atrNs;
72125
74957
  exports.bracketOrders = bracketOrders;
72126
74958
  exports.buildOCCSymbol = buildOCCSymbol;
72127
74959
  exports.buildOptionSymbol = buildOptionSymbol;
74960
+ exports.buildRetryPrompt = buildRetryPrompt;
72128
74961
  exports.buyCryptoNotional = buyCryptoNotional;
72129
74962
  exports.buyToClose = buyToClose;
72130
74963
  exports.buyToOpen = buyToOpen;
@@ -72135,6 +74968,8 @@ exports.calculateOrderValue = calculateOrderValue;
72135
74968
  exports.calculatePeriodPerformance = calculatePeriodPerformance;
72136
74969
  exports.calculatePutCallRatio = calculatePutCallRatio;
72137
74970
  exports.calculateTotalFilledValue = calculateTotalFilledValue;
74971
+ exports.callLLMByAlias = callLLMByAlias;
74972
+ exports.callWithValidation = callWithValidation;
72138
74973
  exports.cancelAllCryptoOrders = cancelAllCryptoOrders;
72139
74974
  exports.cancelOCOOrder = cancelOCOOrder;
72140
74975
  exports.cancelOTOOrder = cancelOTOOrder;
@@ -72145,6 +74980,9 @@ exports.clearClientCache = clearClientCache;
72145
74980
  exports.clock = clock;
72146
74981
  exports.closeAllOptionPositions = closeAllOptionPositions;
72147
74982
  exports.closeOptionPosition = closeOptionPosition;
74983
+ exports.closedIncumbentLeg = closedIncumbentLeg;
74984
+ exports.collectStream = collectStream;
74985
+ exports.configureLlmClient = configureLlmClient;
72148
74986
  exports.createAlpacaClient = createAlpacaClient;
72149
74987
  exports.createAlpacaMarketDataAPI = createAlpacaMarketDataAPI;
72150
74988
  exports.createAlpacaTradingAPI = createAlpacaTradingAPI;
@@ -72158,7 +74996,9 @@ exports.createCryptoMarketOrder = createCryptoMarketOrder;
72158
74996
  exports.createCryptoOrder = createCryptoOrder;
72159
74997
  exports.createCryptoStopLimitOrder = createCryptoStopLimitOrder;
72160
74998
  exports.createCryptoStopOrder = createCryptoStopOrder;
74999
+ exports.createDirectTransport = createDirectTransport;
72161
75000
  exports.createExecutorFromTradingAPI = createExecutorFromTradingAPI;
75001
+ exports.createGatewayTransport = createGatewayTransport;
72162
75002
  exports.createIronCondor = createIronCondor$1;
72163
75003
  exports.createIronCondorAdvanced = createIronCondor;
72164
75004
  exports.createMultiLegOptionOrder = createMultiLegOptionOrder;
@@ -72192,6 +75032,7 @@ exports.findNearestExpiration = findNearestExpiration;
72192
75032
  exports.findOptionsByDelta = findOptionsByDelta;
72193
75033
  exports.formatOrderForLog = formatOrderForLog;
72194
75034
  exports.formatOrderSummary = formatOrderSummary;
75035
+ exports.gatewayModelNameFor = gatewayModelNameFor;
72195
75036
  exports.generateOptimalAllocation = generateOptimalAllocation;
72196
75037
  exports.getAccountConfiguration = getAccountConfiguration;
72197
75038
  exports.getAccountDetails = getAccountDetails;
@@ -72278,6 +75119,7 @@ exports.getTradingWebSocketUrl = getTradingWebSocketUrl;
72278
75119
  exports.getTrailingStopHWM = getTrailingStopHWM;
72279
75120
  exports.groupOrdersByStatus = groupOrdersByStatus;
72280
75121
  exports.groupOrdersBySymbol = groupOrdersBySymbol;
75122
+ exports.guardSnapshots = guardSnapshots;
72281
75123
  exports.hasActiveTrailingStop = hasActiveTrailingStop;
72282
75124
  exports.hasOptionLiquidity = hasGoodLiquidity;
72283
75125
  exports.hasStockLiquidity = hasGoodLiquidity$1;
@@ -72299,21 +75141,37 @@ exports.isSupportedCryptoPair = isSupportedCryptoPair;
72299
75141
  exports.isTransientNetworkError = isTransientNetworkError;
72300
75142
  exports.legacyApi = index$1;
72301
75143
  exports.limitBuyWithTakeProfit = limitBuyWithTakeProfit;
75144
+ exports.limitsFor = limitsFor;
75145
+ exports.limitsInventory = limitsInventory;
75146
+ exports.listAliases = listAliases;
75147
+ exports.llmAliases = llmAliases;
75148
+ exports.llmBreakers = llmBreakers;
75149
+ exports.normaliseAnthropicStream = normaliseAnthropicStream;
75150
+ exports.normaliseOpenAiStream = normaliseOpenAiStream;
75151
+ exports.normaliseParams = normaliseParams;
75152
+ exports.normaliseStream = normaliseStream;
72302
75153
  exports.ocoOrders = ocoOrders;
72303
75154
  exports.orderUtils = orderUtils;
75155
+ exports.orderedRoutes = orderedRoutes;
72304
75156
  exports.otoOrders = otoOrders;
72305
75157
  exports.paginate = paginate;
72306
75158
  exports.paginateAll = paginateAll;
72307
75159
  exports.parseOCCSymbol = parseOCCSymbol;
72308
75160
  exports.protectLongPosition = protectLongPosition;
72309
75161
  exports.protectShortPosition = protectShortPosition;
72310
- exports.rateLimiters = rateLimiters;
75162
+ exports.rateLimiters = rateLimiters$1;
72311
75163
  exports.resetLogger = resetLogger;
75164
+ exports.resetProviderGuards = resetProviderGuards;
72312
75165
  exports.resetRiskFreeRateCache = resetRiskFreeRateCache;
75166
+ exports.resolveChain = resolveChain;
75167
+ exports.resolveDefaultDirectCaller = resolveDefaultDirectCaller;
72313
75168
  exports.risk = riskNs;
72314
75169
  exports.rollOptionPosition = rollOptionPosition;
72315
75170
  exports.roundPriceForAlpaca = roundPriceForAlpaca$3;
72316
75171
  exports.roundPriceForAlpacaNumber = roundPriceForAlpacaNumber;
75172
+ exports.routeKeyFor = routeKeyFor;
75173
+ exports.routeSupports = routeSupports;
75174
+ exports.routeTable = routeTable;
72317
75175
  exports.safeValidateResponse = safeValidateResponse;
72318
75176
  exports.searchNews = searchNews;
72319
75177
  exports.sellAllCrypto = sellAllCrypto;
@@ -72325,6 +75183,7 @@ exports.setRiskFreeRate = setRiskFreeRate;
72325
75183
  exports.shortWithStopLoss = shortWithStopLoss;
72326
75184
  exports.sortOrdersByDate = sortOrdersByDate;
72327
75185
  exports.strategy = strategyNs;
75186
+ exports.sumUsage = sumUsage;
72328
75187
  exports.tradingPolicy = index;
72329
75188
  exports.trailingStops = trailingStops;
72330
75189
  exports.updateAccountConfiguration = updateAccountConfiguration;
@@ -72337,6 +75196,7 @@ exports.validateResponse = validateResponse;
72337
75196
  exports.verifyFetchKeepAlive = verifyFetchKeepAlive;
72338
75197
  exports.volatility = volatilityNs;
72339
75198
  exports.waitForOrderFill = waitForOrderFill;
75199
+ exports.withProviderGuards = withProviderGuards;
72340
75200
  exports.withRetry = withRetry;
72341
75201
  exports.withTimeout = withTimeout;
72342
75202
  //# sourceMappingURL=index.cjs.map