@adaptic/utils 0.0.1034 → 0.0.1035

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -7506,6 +7506,68 @@ class DuplicateClientOrderIdError extends AlpacaApiError {
7506
7506
  this.wasDerived = wasDerived;
7507
7507
  }
7508
7508
  }
7509
+ /** Stable `code` carried by {@link PendingCancelError}. */
7510
+ const PENDING_CANCEL_ERROR_CODE = "ORDER_PENDING_CANCEL";
7511
+ /**
7512
+ * Alpaca's numeric code for an unprocessable order request. It is shared by
7513
+ * many distinct reasons (pending cancel, not cancelable, bad replace), so the
7514
+ * reason text is what separates them.
7515
+ */
7516
+ const ALPACA_UNPROCESSABLE_BROKER_CODE = 42210000;
7517
+ /** HTTP status Alpaca answers an unprocessable order request on. */
7518
+ const HTTP_UNPROCESSABLE_STATUS = 422;
7519
+ /** Alpaca's reason text for an order it already holds in `pending_cancel`. */
7520
+ const PENDING_CANCEL_REASON = /pending[ _]cancel/i;
7521
+ /**
7522
+ * Alpaca refused a DELETE because it has already accepted a cancel for the
7523
+ * order and holds it in `pending_cancel` until the venue confirms (HTTP 422,
7524
+ * code 42210000, "order pending cancel").
7525
+ *
7526
+ * This is a distinct broker answer from "not cancelable" (the order already
7527
+ * filled or is otherwise terminal): the order is being torn down, so it is
7528
+ * NOT live protection, and a second DELETE only returns the same 422. A caller
7529
+ * must treat the order as dying and wait for terminal truth (a `canceled` or
7530
+ * `filled` trade update, or a GET), never re-cancel it and never count it as
7531
+ * kept.
7532
+ *
7533
+ * The `message` is kept byte-identical to the generic rewrite
7534
+ * ("Order X is not cancelable") so consumers that still match that text keep
7535
+ * their behaviour; new consumers branch on the type, on
7536
+ * {@link PENDING_CANCEL_ERROR_CODE}, or on {@link isPendingCancelRejection}.
7537
+ * Never retryable.
7538
+ */
7539
+ class PendingCancelError extends AlpacaApiError {
7540
+ orderId;
7541
+ constructor(message,
7542
+ /** The broker order id the DELETE targeted. */
7543
+ orderId, cause) {
7544
+ super(message, PENDING_CANCEL_ERROR_CODE, HTTP_UNPROCESSABLE_STATUS, cause, extractAlpacaBrokerError(cause));
7545
+ this.orderId = orderId;
7546
+ }
7547
+ }
7548
+ /**
7549
+ * Did Alpaca answer that the order is already pending cancel?
7550
+ *
7551
+ * True for a {@link PendingCancelError}, and for any thrown value whose broker
7552
+ * payload (on the value or its `cause` chain) is a 422 / 42210000 whose reason
7553
+ * names `pending cancel` / `pending_cancel`. A payload with no reason text is
7554
+ * never read as pending cancel: absence stays unknown.
7555
+ *
7556
+ * @param error - The thrown value.
7557
+ * @returns Whether the broker holds the order in `pending_cancel`.
7558
+ */
7559
+ function isPendingCancelRejection(error) {
7560
+ if (error instanceof PendingCancelError) {
7561
+ return true;
7562
+ }
7563
+ const detail = extractAlpacaBrokerError(error);
7564
+ if (detail === undefined || detail.brokerMessage === null) {
7565
+ return false;
7566
+ }
7567
+ const unprocessable = detail.statusCode === HTTP_UNPROCESSABLE_STATUS ||
7568
+ detail.brokerCode === ALPACA_UNPROCESSABLE_BROKER_CODE;
7569
+ return unprocessable && PENDING_CANCEL_REASON.test(detail.brokerMessage);
7570
+ }
7509
7571
  /** Max depth walked along the `error.cause` chain when locating a broker payload. */
7510
7572
  const MAX_BROKER_ERROR_CAUSE_DEPTH = 6;
7511
7573
  /**
@@ -11026,6 +11088,9 @@ class AlpacaTradingAPI {
11026
11088
  /**
11027
11089
  * Cancel a specific order by its ID
11028
11090
  * @param orderId The id of the order to cancel
11091
+ * @throws PendingCancelError if the broker already holds the order in
11092
+ * `pending_cancel` (422 / 42210000 "order pending cancel"): the order is
11093
+ * being torn down, so it is not live and must not be cancelled again
11029
11094
  * @throws Error if the order is not cancelable (status 422) or if the order doesn't exist
11030
11095
  * @returns Promise that resolves when the order is successfully canceled
11031
11096
  */
@@ -11036,6 +11101,12 @@ class AlpacaTradingAPI {
11036
11101
  this.log(`Successfully canceled order ${orderId}`);
11037
11102
  }
11038
11103
  catch (error) {
11104
+ // A cancel the broker already accepted: the order is dying, not kept.
11105
+ // Typed so callers stop reading it as "not cancelable, still live".
11106
+ if (isPendingCancelRejection(error)) {
11107
+ this.log(`Order ${orderId} is already pending cancel at the broker; not re-cancelling`, { type: "warn" });
11108
+ throw new PendingCancelError(`Order ${orderId} is not cancelable`, orderId, error);
11109
+ }
11039
11110
  // If the error is a 422, it means the order is not cancelable
11040
11111
  if (error instanceof Error && error.message.includes("422")) {
11041
11112
  this.log(`Order ${orderId} is not cancelable`, {
@@ -12315,6 +12386,9 @@ async function replaceOrder$1(auth, orderId, params) {
12315
12386
  * @param auth - The authentication details for Alpaca
12316
12387
  * @param orderId - The ID of the order to cancel
12317
12388
  * @returns Success status and optional message if order not found
12389
+ * @throws PendingCancelError if the broker already holds the order in
12390
+ * `pending_cancel` (the order is dying; never cancel it again). The message
12391
+ * is the same "Failed to cancel order: ..." text as any other refusal.
12318
12392
  */
12319
12393
  async function cancelOrder$1(auth, orderId) {
12320
12394
  try {
@@ -12334,7 +12408,13 @@ async function cancelOrder$1(auth, orderId) {
12334
12408
  return { success: false, message: `Order not found: ${orderId}` };
12335
12409
  }
12336
12410
  else {
12337
- throw alpacaHttpError(`Failed to cancel order: ${response.status} ${response.statusText} ${errorText}`, response.status, errorText);
12411
+ const httpError = alpacaHttpError(`Failed to cancel order: ${response.status} ${response.statusText} ${errorText}`, response.status, errorText);
12412
+ // A cancel the broker already accepted: the order is dying, not kept,
12413
+ // and a second DELETE only returns the same 422.
12414
+ if (isPendingCancelRejection(httpError)) {
12415
+ throw new PendingCancelError(httpError.message, orderId, httpError);
12416
+ }
12417
+ throw httpError;
12338
12418
  }
12339
12419
  }
12340
12420
  return { success: true };
@@ -68338,6 +68418,8 @@ async function getOrders(client, params = {}) {
68338
68418
  *
68339
68419
  * @param client - The AlpacaClient instance
68340
68420
  * @param orderId - The unique identifier of the order to cancel
68421
+ * @throws PendingCancelError if the broker already holds the order in
68422
+ * `pending_cancel` (the order is dying; never cancel it again)
68341
68423
  * @throws Error if order cannot be canceled (e.g., already filled or canceled)
68342
68424
  *
68343
68425
  * @example
@@ -68353,6 +68435,11 @@ async function cancelOrder(client, orderId) {
68353
68435
  }
68354
68436
  catch (error) {
68355
68437
  const errorMessage = error instanceof Error ? error.message : "Unknown error";
68438
+ // A cancel the broker already accepted: the order is dying, not kept.
68439
+ if (isPendingCancelRejection(error)) {
68440
+ log$6(`Order ${orderId} is already pending cancel at the broker; not re-cancelling`, { type: "warn" });
68441
+ throw new PendingCancelError(`Order ${orderId} is not cancelable`, orderId, error);
68442
+ }
68356
68443
  // Check for specific error conditions
68357
68444
  if (errorMessage.includes("422") ||
68358
68445
  errorMessage.includes("not cancelable")) {
@@ -72495,6 +72582,10 @@ const MAX_TOKENS_PARAM_BY_API_STYLE = {
72495
72582
  };
72496
72583
  /** The response-format key an OpenAI-compatible provider expects. */
72497
72584
  const RESPONSE_FORMAT_KEY = "response_format";
72585
+ /** The tool-choice key, in the OpenAI-compatible vocabulary the gateway speaks. */
72586
+ const TOOL_CHOICE_KEY = "tool_choice";
72587
+ /** The mandatory tool-choice value: the answer must be a tool call. */
72588
+ const TOOL_CHOICE_REQUIRED = "required";
72498
72589
  /**
72499
72590
  * Normalise a caller's options into the exact parameter set one leg accepts.
72500
72591
  *
@@ -72542,12 +72633,91 @@ function normaliseParams(options, route, responseFormat) {
72542
72633
  // tool calls concurrently reorders its own effects, and ordering is part
72543
72634
  // of the meaning of a sequence of trading actions.
72544
72635
  params.parallel_tool_calls = false;
72636
+ const toolChoice = normaliseToolChoice(options.toolChoice, route);
72637
+ if (toolChoice !== undefined) {
72638
+ params[TOOL_CHOICE_KEY] = toolChoice;
72639
+ }
72545
72640
  }
72546
72641
  if (options.metadata !== undefined) {
72547
72642
  params.metadata = { ...options.metadata };
72548
72643
  }
72549
72644
  return params;
72550
72645
  }
72646
+ /**
72647
+ * Translate the caller's tool-choice policy into the value one leg is sent.
72648
+ *
72649
+ * `"auto"` is every provider's default once tools are present, so it is never
72650
+ * sent. `"required"` is sent only to a route MEASURED to honour it: serving
72651
+ * stacks accept the parameter and silently answer in prose anyway, so sending
72652
+ * it to an unmeasured route would assert a constraint nobody checked. Omitting
72653
+ * it there leaves that leg exactly as it was, with the caller's own validation
72654
+ * as the guard.
72655
+ *
72656
+ * @param toolChoice The caller's policy, if any.
72657
+ * @param route The leg being prepared.
72658
+ * @returns The value to send, or undefined to omit the parameter.
72659
+ */
72660
+ function normaliseToolChoice(toolChoice, route) {
72661
+ if (toolChoice !== TOOL_CHOICE_REQUIRED) {
72662
+ return undefined;
72663
+ }
72664
+ return route.params.supports_tool_choice === true ? TOOL_CHOICE_REQUIRED : undefined;
72665
+ }
72666
+ /**
72667
+ * Fail a leg that was sent a mandatory tool choice and answered without a tool call.
72668
+ *
72669
+ * The parameter is sent only to a route declared to honour it, so an answer
72670
+ * with no tool call means the declaration no longer holds for this leg — a
72671
+ * serving-stack upgrade can drop the constraint without an error. Returning
72672
+ * that answer would hand a caller prose where it demanded an action; failing
72673
+ * the leg lets the chain advance and names the broken declaration.
72674
+ *
72675
+ * @param route The leg that answered.
72676
+ * @param params The parameters it was sent.
72677
+ * @param response The leg's answer: its tool calls, and the usage and serving
72678
+ * model the error carries so the leg's spend and attribution are not lost.
72679
+ * @throws {ToolChoiceIgnoredError} When a mandatory choice produced no tool call.
72680
+ */
72681
+ function assertToolChoiceHonoured(route, params, response) {
72682
+ if (params[TOOL_CHOICE_KEY] !== TOOL_CHOICE_REQUIRED) {
72683
+ return;
72684
+ }
72685
+ if (response.tool_calls !== undefined && response.tool_calls.length > 0) {
72686
+ return;
72687
+ }
72688
+ throw new ToolChoiceIgnoredError(route, response.usage, response.servedModel ?? null);
72689
+ }
72690
+ /**
72691
+ * Thrown when a route declared to honour a mandatory tool choice answered
72692
+ * without a tool call.
72693
+ *
72694
+ * Distinct from a provider outage: the route answered, but not in the form it
72695
+ * is declared to guarantee. The chain advances, and the route's health is not
72696
+ * charged, because the fault is in the declaration, not in availability. The
72697
+ * provider still billed for the answer, so the error carries the leg's usage:
72698
+ * the spend belongs in the chain's total whether or not a later leg serves.
72699
+ */
72700
+ class ToolChoiceIgnoredError extends Error {
72701
+ /** The leg that ignored the choice. */
72702
+ routeKey;
72703
+ /** What the provider billed for the answer that carried no tool call. */
72704
+ usage;
72705
+ /** The model the provider reports as having answered, or `null` when unreported. */
72706
+ servedModel;
72707
+ /**
72708
+ * @param route The leg.
72709
+ * @param usage What the provider billed for the answer.
72710
+ * @param servedModel The provider-reported serving model, or `null`.
72711
+ */
72712
+ constructor(route, usage, servedModel) {
72713
+ super(`route ${route.routeKey} (${route.providerName}/${route.modelId}) was sent tool_choice "required" ` +
72714
+ "and answered without a tool call; its supports_tool_choice declaration no longer holds");
72715
+ this.name = "ToolChoiceIgnoredError";
72716
+ this.routeKey = route.routeKey;
72717
+ this.usage = usage;
72718
+ this.servedModel = servedModel;
72719
+ }
72720
+ }
72551
72721
  /**
72552
72722
  * Translate the requested response shape into the parameter a route accepts.
72553
72723
  *
@@ -73249,6 +73419,9 @@ class ChainExhaustedError extends Error {
73249
73419
  * understate spend by exactly the amount the failures cost — which is the
73250
73420
  * amount a fallback chain is most likely to run up.
73251
73421
  *
73422
+ * A count one attempt did not report makes the total for that count unknown
73423
+ * (`null`): a sum that skipped it would present a partial figure as complete.
73424
+ *
73252
73425
  * @param a The running total.
73253
73426
  * @param b The attempt to add, if any.
73254
73427
  * @returns The combined usage.
@@ -73258,8 +73431,8 @@ function sumUsage(a, b) {
73258
73431
  return a;
73259
73432
  }
73260
73433
  return {
73261
- prompt_tokens: a.prompt_tokens + b.prompt_tokens,
73262
- completion_tokens: a.completion_tokens + b.completion_tokens,
73434
+ prompt_tokens: addKnown(a.prompt_tokens, b.prompt_tokens),
73435
+ completion_tokens: addKnown(a.completion_tokens, b.completion_tokens),
73263
73436
  reasoning_tokens: a.reasoning_tokens === undefined && b.reasoning_tokens === undefined
73264
73437
  ? undefined
73265
73438
  : (a.reasoning_tokens ?? 0) + (b.reasoning_tokens ?? 0),
@@ -73268,9 +73441,38 @@ function sumUsage(a, b) {
73268
73441
  : (a.cached_tokens ?? 0) + (b.cached_tokens ?? 0),
73269
73442
  provider: b.provider,
73270
73443
  model: b.model,
73271
- cost: a.cost + b.cost,
73444
+ cost: addKnown(a.cost, b.cost),
73272
73445
  };
73273
73446
  }
73447
+ /**
73448
+ * Add two measured quantities, either of which may be unreported.
73449
+ *
73450
+ * @param a The running total, or null when already unknown.
73451
+ * @param b The value to add, or null when unreported.
73452
+ * @returns The sum, or null when either side is unknown.
73453
+ */
73454
+ function addKnown(a, b) {
73455
+ return a === null || b === null ? null : a + b;
73456
+ }
73457
+ /**
73458
+ * The budget one leg may spend.
73459
+ *
73460
+ * The route's own budget, cut to what remains of the caller's deadline. The
73461
+ * route budget is never widened, so a caller with a long deadline still has a
73462
+ * slow leg abandoned in time for the next one to run; and the deadline is never
73463
+ * exceeded, so the last leg reached ends when the caller stops waiting.
73464
+ *
73465
+ * @param routeBudgetMs The leg's route budget.
73466
+ * @param deadlineAtMs The caller's deadline instant, if any.
73467
+ * @param nowMs The current instant on the same clock.
73468
+ * @returns The leg's budget in milliseconds; zero or less means none is left.
73469
+ */
73470
+ function legBudgetMs(routeBudgetMs, deadlineAtMs, nowMs) {
73471
+ if (deadlineAtMs === undefined) {
73472
+ return routeBudgetMs;
73473
+ }
73474
+ return Math.min(routeBudgetMs, deadlineAtMs - nowMs);
73475
+ }
73274
73476
  /** Raised internally when a leg exceeds its budget. */
73275
73477
  class LegTimeoutError extends Error {
73276
73478
  /**
@@ -73292,11 +73494,11 @@ class LegTimeoutError extends Error {
73292
73494
  * @param leg The leg to run.
73293
73495
  * @param params Normalised parameters for this leg.
73294
73496
  * @param execution The call context.
73497
+ * @param budgetMs The leg's budget: its route budget cut to the caller's deadline.
73295
73498
  * @returns The provider's answer.
73296
73499
  */
73297
- async function runLeg(leg, params, execution) {
73500
+ async function runLeg(leg, params, execution, budgetMs) {
73298
73501
  const controller = new AbortController();
73299
- const budgetMs = leg.route.timeoutMs;
73300
73502
  const timer = setTimeout(() => {
73301
73503
  controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
73302
73504
  }, budgetMs);
@@ -73318,7 +73520,7 @@ async function runLeg(leg, params, execution) {
73318
73520
  // provider, and only one clock should govern both. The leg's own signal is
73319
73521
  // handed to the guard as well, so a leg whose budget or caller is gone
73320
73522
  // leaves the queue at once instead of holding its place in it.
73321
- return await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
73523
+ const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
73322
73524
  route: leg.route,
73323
73525
  content: execution.content,
73324
73526
  responseFormat: execution.responseFormat,
@@ -73328,6 +73530,8 @@ async function runLeg(leg, params, execution) {
73328
73530
  signal: controller.signal,
73329
73531
  correlationId: execution.correlationId,
73330
73532
  }), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
73533
+ assertToolChoiceHonoured(leg.route, params, response);
73534
+ return response;
73331
73535
  }
73332
73536
  finally {
73333
73537
  clearTimeout(timer);
@@ -73360,6 +73564,11 @@ function classify(error, callerSignal) {
73360
73564
  if (error instanceof UnsupportedCapabilityError) {
73361
73565
  return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
73362
73566
  }
73567
+ if (error instanceof ToolChoiceIgnoredError) {
73568
+ // The route answered; it broke a declared guarantee rather than failing to
73569
+ // be available, so its breaker is not charged for it.
73570
+ return { outcome: "error", reason: error.message, countsAgainstHealth: false };
73571
+ }
73363
73572
  if (error instanceof RateGuardTimeoutError) {
73364
73573
  // Self-inflicted pacing, not provider ill-health. Counting it would let the
73365
73574
  // client's own throttling open a breaker on a perfectly healthy provider
@@ -73391,6 +73600,22 @@ function classify(error, callerSignal) {
73391
73600
  function isAborted(signal) {
73392
73601
  return signal !== undefined && signal.aborted;
73393
73602
  }
73603
+ /**
73604
+ * The usage a failed leg was billed for, when the leg reached an answer.
73605
+ *
73606
+ * A leg that failed after the provider answered — content that does not parse,
73607
+ * or prose where a tool call was mandatory — was still charged. A leg that
73608
+ * never answered (timeout, outage, skip) carries no usage, and none is invented.
73609
+ *
73610
+ * @param error The thrown value.
73611
+ * @returns The billed usage, or undefined when the leg never produced an answer.
73612
+ */
73613
+ function billedUsageOf(error) {
73614
+ if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
73615
+ return error.usage;
73616
+ }
73617
+ return undefined;
73618
+ }
73394
73619
  /**
73395
73620
  * Walk a chain until a leg answers.
73396
73621
  *
@@ -73438,10 +73663,27 @@ async function executeChain(alias, execution) {
73438
73663
  execution.onAttempt?.(record);
73439
73664
  continue;
73440
73665
  }
73666
+ const budgetMs = legBudgetMs(route.timeoutMs, execution.deadlineAtMs, now());
73667
+ if (budgetMs <= 0) {
73668
+ // The caller's deadline is spent. Dispatching now would start a call
73669
+ // that is cancelled the moment it begins, and charge nothing but noise.
73670
+ const record = {
73671
+ routeKey: route.routeKey,
73672
+ role: route.role,
73673
+ provider: route.providerName,
73674
+ modelId: route.modelId,
73675
+ outcome: "skipped",
73676
+ durationMs: 0,
73677
+ reason: "caller deadline exhausted before this leg",
73678
+ };
73679
+ attempts.push(record);
73680
+ execution.onAttempt?.(record);
73681
+ continue;
73682
+ }
73441
73683
  const startedAt = now();
73442
73684
  const holdsProbe = execution.breakers.onAttemptStart(route.routeKey);
73443
73685
  try {
73444
- const response = await runLeg(leg, leg.params, execution);
73686
+ const response = await runLeg(leg, leg.params, execution, budgetMs);
73445
73687
  execution.breakers.onSuccess(route.routeKey);
73446
73688
  totalUsage = sumUsage(totalUsage, response.usage);
73447
73689
  const record = {
@@ -73451,6 +73693,8 @@ async function executeChain(alias, execution) {
73451
73693
  modelId: route.modelId,
73452
73694
  outcome: "ok",
73453
73695
  durationMs: now() - startedAt,
73696
+ budgetMs,
73697
+ servedModel: response.servedModel ?? null,
73454
73698
  usage: response.usage,
73455
73699
  };
73456
73700
  attempts.push(record);
@@ -73467,9 +73711,11 @@ async function executeChain(alias, execution) {
73467
73711
  // took must come back, or a half-open route admits no probe ever again.
73468
73712
  execution.breakers.onAttemptAbandoned(route.routeKey);
73469
73713
  }
73470
- // A provider that answered with unparseable content still billed for the
73471
- // answer; the spend belongs in the total whether or not a later leg serves.
73472
- const billed = error instanceof LlmResponseFormatError ? error.usage : undefined;
73714
+ // A provider that answered — with unparseable content, or in prose where a
73715
+ // tool call was mandatory — still billed for the answer; the spend belongs
73716
+ // in the total whether or not a later leg serves.
73717
+ const billed = billedUsageOf(error);
73718
+ const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
73473
73719
  totalUsage = sumUsage(totalUsage, billed);
73474
73720
  const record = {
73475
73721
  routeKey: route.routeKey,
@@ -73478,7 +73724,9 @@ async function executeChain(alias, execution) {
73478
73724
  modelId: route.modelId,
73479
73725
  outcome,
73480
73726
  durationMs: now() - startedAt,
73727
+ budgetMs,
73481
73728
  reason,
73729
+ ...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
73482
73730
  ...(billed === undefined ? {} : { usage: billed }),
73483
73731
  };
73484
73732
  attempts.push(record);
@@ -73670,6 +73918,7 @@ var aliases = {
73670
73918
  supports_tools: true,
73671
73919
  supports_cache_control: true,
73672
73920
  supports_vision: false,
73921
+ supports_tool_choice: false,
73673
73922
  context_window: 1048576
73674
73923
  },
73675
73924
  price_per_mtok: {
@@ -73696,6 +73945,7 @@ var aliases = {
73696
73945
  supports_tools: true,
73697
73946
  supports_cache_control: true,
73698
73947
  supports_vision: false,
73948
+ supports_tool_choice: true,
73699
73949
  context_window: 1048576
73700
73950
  },
73701
73951
  price_per_mtok: {
@@ -73722,6 +73972,7 @@ var aliases = {
73722
73972
  supports_tools: true,
73723
73973
  supports_cache_control: true,
73724
73974
  supports_vision: true,
73975
+ supports_tool_choice: true,
73725
73976
  context_window: 1000000
73726
73977
  },
73727
73978
  price_per_mtok: {
@@ -73766,6 +74017,7 @@ var aliases = {
73766
74017
  supports_tools: true,
73767
74018
  supports_cache_control: true,
73768
74019
  supports_vision: true,
74020
+ supports_tool_choice: true,
73769
74021
  context_window: 1048576
73770
74022
  },
73771
74023
  price_per_mtok: {
@@ -73792,6 +74044,7 @@ var aliases = {
73792
74044
  supports_tools: true,
73793
74045
  supports_cache_control: true,
73794
74046
  supports_vision: false,
74047
+ supports_tool_choice: false,
73795
74048
  context_window: 1048576
73796
74049
  },
73797
74050
  price_per_mtok: {
@@ -73818,6 +74071,7 @@ var aliases = {
73818
74071
  supports_tools: true,
73819
74072
  supports_cache_control: true,
73820
74073
  supports_vision: true,
74074
+ supports_tool_choice: true,
73821
74075
  context_window: 1000000
73822
74076
  },
73823
74077
  price_per_mtok: {
@@ -73862,6 +74116,7 @@ var aliases = {
73862
74116
  supports_tools: true,
73863
74117
  supports_cache_control: true,
73864
74118
  supports_vision: false,
74119
+ supports_tool_choice: false,
73865
74120
  context_window: 1048576
73866
74121
  },
73867
74122
  price_per_mtok: {
@@ -73888,6 +74143,7 @@ var aliases = {
73888
74143
  supports_tools: true,
73889
74144
  supports_cache_control: false,
73890
74145
  supports_vision: false,
74146
+ supports_tool_choice: true,
73891
74147
  context_window: 131072
73892
74148
  },
73893
74149
  price_per_mtok: {
@@ -73914,6 +74170,7 @@ var aliases = {
73914
74170
  supports_tools: true,
73915
74171
  supports_cache_control: true,
73916
74172
  supports_vision: true,
74173
+ supports_tool_choice: true,
73917
74174
  context_window: 200000
73918
74175
  },
73919
74176
  price_per_mtok: {
@@ -73957,6 +74214,7 @@ var aliases = {
73957
74214
  supports_tools: true,
73958
74215
  supports_cache_control: false,
73959
74216
  supports_vision: false,
74217
+ supports_tool_choice: true,
73960
74218
  context_window: 131072
73961
74219
  },
73962
74220
  price_per_mtok: {
@@ -73983,6 +74241,7 @@ var aliases = {
73983
74241
  supports_tools: true,
73984
74242
  supports_cache_control: true,
73985
74243
  supports_vision: false,
74244
+ supports_tool_choice: true,
73986
74245
  context_window: 262144
73987
74246
  },
73988
74247
  price_per_mtok: {
@@ -74009,6 +74268,7 @@ var aliases = {
74009
74268
  supports_tools: true,
74010
74269
  supports_cache_control: true,
74011
74270
  supports_vision: true,
74271
+ supports_tool_choice: true,
74012
74272
  context_window: 200000
74013
74273
  },
74014
74274
  price_per_mtok: {
@@ -74054,6 +74314,7 @@ var aliases = {
74054
74314
  supports_tools: true,
74055
74315
  supports_cache_control: true,
74056
74316
  supports_vision: false,
74317
+ supports_tool_choice: false,
74057
74318
  context_window: 1048576
74058
74319
  },
74059
74320
  price_per_mtok: {
@@ -74065,6 +74326,105 @@ var aliases = {
74065
74326
  notes: "Pinned with no fallback (PD-6). A judge that failed over would grade one model's output against another model's standard, corrupting every gate decided by it."
74066
74327
  }
74067
74328
  ]
74329
+ },
74330
+ "llm.decide": {
74331
+ workload: "Per-symbol trading decision: a multi-turn tool-calling read that ends in exactly one decision tool call",
74332
+ latency_class: "hot-path",
74333
+ criticality: "trading-adjacent",
74334
+ isolation_capable: false,
74335
+ eval_gate: "tool-call",
74336
+ policy_note: "The dedicated decision alias. The per-symbol decision call is a trading decision, not a latency-critical extraction, so it gets its own chain instead of borrowing llm.fast's. It is evaluated in SHADOW beside the live llm.fast path (schema-valid rate, latency, served model, action agreement) and serves live traffic only through the engine's governed routing flag after that evidence. Leg budgets are the hot-path class's today; budgets set from each model's measured p95 are a separate, evidence-gated change. Its monthly budget bounds the shadow's own spend and is added to the gateway total, so the shadow cannot consume the live aliases' budget.",
74337
+ budget: {
74338
+ basis: "provisional-pre-baseline",
74339
+ monthly_usd: 600,
74340
+ alert_pct: [
74341
+ 50,
74342
+ 80,
74343
+ 100
74344
+ ]
74345
+ },
74346
+ routes: [
74347
+ {
74348
+ role: "primary",
74349
+ provider: "deepinfra",
74350
+ model_id: "deepseek-ai/DeepSeek-V4-Pro",
74351
+ model_id_status: "confirmed",
74352
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
74353
+ model_family: "DeepSeek V4 Pro",
74354
+ lumic_model: "deepseek-ai/DeepSeek-V4-Pro",
74355
+ params: {
74356
+ temperature: null,
74357
+ max_output_tokens: null,
74358
+ supports_temperature: true,
74359
+ supports_json_schema: true,
74360
+ supports_tools: true,
74361
+ supports_cache_control: true,
74362
+ supports_vision: false,
74363
+ supports_tool_choice: false,
74364
+ context_window: 1048576
74365
+ },
74366
+ price_per_mtok: {
74367
+ input: 1.3,
74368
+ output: 2.6,
74369
+ as_of: "2026-09-11",
74370
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
74371
+ },
74372
+ notes: "The advanced tier the per-symbol decision prompt was written for: a long, multi-turn, tool-calling read of about 21k prompt tokens, where the hot-path alias's small models answer faster but lose the schema contract more often. It ignores a mandatory tool_choice (measured), so the decision contract's own validation stays the guard on this leg."
74373
+ },
74374
+ {
74375
+ role: "secondary",
74376
+ provider: "deepinfra",
74377
+ model_id: "deepseek-ai/DeepSeek-V4-Flash",
74378
+ model_id_status: "confirmed",
74379
+ model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
74380
+ model_family: "DeepSeek V4 Flash",
74381
+ lumic_model: "deepseek-ai/DeepSeek-V4-Flash",
74382
+ params: {
74383
+ temperature: null,
74384
+ max_output_tokens: null,
74385
+ supports_temperature: true,
74386
+ supports_json_schema: true,
74387
+ supports_tools: true,
74388
+ supports_cache_control: true,
74389
+ supports_vision: false,
74390
+ supports_tool_choice: false,
74391
+ context_window: 1048576
74392
+ },
74393
+ price_per_mtok: {
74394
+ input: 0.09,
74395
+ output: 0.18,
74396
+ as_of: "2026-09-11",
74397
+ source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
74398
+ },
74399
+ notes: "The current decision model, kept as the secondary so a slow or failing primary falls back to the model the decision path already runs on."
74400
+ },
74401
+ {
74402
+ role: "closed_incumbent",
74403
+ provider: "anthropic",
74404
+ model_id: "claude-haiku-4-5",
74405
+ model_id_status: "confirmed",
74406
+ model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
74407
+ model_family: "Haiku-class",
74408
+ lumic_model: "claude-haiku-4-5",
74409
+ params: {
74410
+ temperature: null,
74411
+ max_output_tokens: 64000,
74412
+ supports_temperature: true,
74413
+ supports_json_schema: true,
74414
+ supports_tools: true,
74415
+ supports_cache_control: true,
74416
+ supports_vision: true,
74417
+ supports_tool_choice: true,
74418
+ context_window: 200000
74419
+ },
74420
+ price_per_mtok: {
74421
+ input: 1,
74422
+ output: 5,
74423
+ as_of: "2026-09-10",
74424
+ source: "docs/llm-provider-migration.md#3 Haiku-class anchor"
74425
+ }
74426
+ }
74427
+ ]
74068
74428
  }
74069
74429
  };
74070
74430
  var open_items = [
@@ -74554,6 +74914,10 @@ function createDirectTransport(config) {
74554
74914
  tool_calls: Array.isArray(result.tool_calls)
74555
74915
  ? result.tool_calls
74556
74916
  : undefined,
74917
+ // The incumbent client reports the model it resolved; unreported stays null.
74918
+ servedModel: typeof result.usage?.model === "string" && result.usage.model !== ""
74919
+ ? result.usage.model
74920
+ : null,
74557
74921
  };
74558
74922
  },
74559
74923
  };
@@ -74561,9 +74925,9 @@ function createDirectTransport(config) {
74561
74925
  /**
74562
74926
  * Normalise the incumbent client's usage shape.
74563
74927
  *
74564
- * Missing counts stay zero rather than being estimated, for the same reason
74565
- * they do on the gateway path: an invented token count flows straight into the
74566
- * budget accounting the spend controls depend on.
74928
+ * A missing count is `null` rather than zero or an estimate, for the same
74929
+ * reason as on the gateway path: an invented token count — zero included —
74930
+ * flows straight into the budget accounting the spend controls depend on.
74567
74931
  *
74568
74932
  * @param usage The incumbent client's usage, if any.
74569
74933
  * @param request The request it answers.
@@ -74571,13 +74935,13 @@ function createDirectTransport(config) {
74571
74935
  */
74572
74936
  function readUsage$1(usage, request) {
74573
74937
  return {
74574
- prompt_tokens: usage?.prompt_tokens ?? 0,
74575
- completion_tokens: usage?.completion_tokens ?? 0,
74938
+ prompt_tokens: usage?.prompt_tokens ?? null,
74939
+ completion_tokens: usage?.completion_tokens ?? null,
74576
74940
  reasoning_tokens: usage?.reasoning_tokens,
74577
74941
  cached_tokens: usage?.cached_tokens,
74578
74942
  provider: usage?.provider ?? request.route.providerName,
74579
74943
  model: usage?.model ?? request.route.modelId,
74580
- cost: usage?.cost ?? 0,
74944
+ cost: usage?.cost ?? null,
74581
74945
  };
74582
74946
  }
74583
74947
  /**
@@ -74630,10 +74994,12 @@ async function resolveDefaultDirectCaller() {
74630
74994
  * never in application code (PD-5), which is what makes a model swap a config
74631
74995
  * change rather than a deploy.
74632
74996
  *
74633
- * The client still walks its own chain on top of the gateway's, and the
74634
- * duplication is deliberate. The gateway's fallbacks cover a provider being
74635
- * down; the client's cover the gateway being down. Only one of those two can
74636
- * cover the other, so the outer chain is the one that must exist.
74997
+ * The client's chain is the single fallback owner: every leg is addressed to the
74998
+ * gateway by its own model name, so the gateway serves one deployment per leg
74999
+ * and needs no fallback of its own. A proxy-side fallback inside a leg would
75000
+ * spend the leg's budget on a model the chain did not choose and report the
75001
+ * answer as the leg's; the served model is read from the response body so such
75002
+ * a substitution stays visible while any remains configured.
74637
75003
  *
74638
75004
  * The gateway key is read from the environment by NAME at call time and never
74639
75005
  * stored, logged, or included in an error (PD-2). Reading it per call rather
@@ -74645,6 +75011,10 @@ async function resolveDefaultDirectCaller() {
74645
75011
  const RETRYABLE_STATUSES = new Set([408, 409, 425, 429, 500, 502, 503, 504]);
74646
75012
  /** Maximum characters of an error body echoed into a message. */
74647
75013
  const ERROR_BODY_EXCERPT = 400;
75014
+ /** Response header in which the LiteLLM proxy reports the call's cost in USD. */
75015
+ const RESPONSE_COST_HEADER = "x-litellm-response-cost";
75016
+ /** Response header naming the proxy deployment that served the call. */
75017
+ const DEPLOYMENT_ID_HEADER = "x-litellm-model-id";
74648
75018
  /**
74649
75019
  * Thrown when the gateway itself is unreachable, as opposed to a provider
74650
75020
  * behind it failing.
@@ -74701,31 +75071,67 @@ function readGatewayKey(envVar) {
74701
75071
  /**
74702
75072
  * Extract usage from a chat-completion response.
74703
75073
  *
74704
- * Absent counts stay zero rather than being estimated. A fabricated token count
74705
- * would flow straight into the budget accounting that the spend controls are
74706
- * built on, and a budget computed from invented numbers is worse than one that
74707
- * knows it is missing a call.
75074
+ * A count the provider did not report is `null`, never zero and never an
75075
+ * estimate. A fabricated count — zero included — would flow straight into the
75076
+ * budget accounting the spend controls are built on, where a zero reads as a
75077
+ * free call and can never trip a limit.
75078
+ *
75079
+ * Cost is read from the body's `usage.response_cost` or, failing that, from the
75080
+ * proxy's cost header, which is where the LiteLLM proxy reports it.
74708
75081
  *
74709
75082
  * @param payload The parsed response body.
74710
75083
  * @param request The request it answers.
75084
+ * @param headers The response headers.
74711
75085
  * @returns The usage record.
74712
75086
  */
74713
- function readUsage(payload, request) {
75087
+ function readUsage(payload, request, headers) {
74714
75088
  const usage = (payload.usage ?? {});
74715
75089
  const details = (usage.prompt_tokens_details ?? {});
74716
75090
  const cached = details.cached_tokens;
74717
75091
  const reasoningDetails = (usage.completion_tokens_details ?? {});
74718
75092
  const reasoning = reasoningDetails.reasoning_tokens;
74719
75093
  return {
74720
- prompt_tokens: typeof usage.prompt_tokens === "number" ? usage.prompt_tokens : 0,
74721
- completion_tokens: typeof usage.completion_tokens === "number" ? usage.completion_tokens : 0,
75094
+ prompt_tokens: finiteOrNull(usage.prompt_tokens),
75095
+ completion_tokens: finiteOrNull(usage.completion_tokens),
74722
75096
  reasoning_tokens: typeof reasoning === "number" ? reasoning : undefined,
74723
75097
  cached_tokens: typeof cached === "number" ? cached : undefined,
74724
75098
  provider: request.route.providerName,
74725
75099
  model: request.route.modelId,
74726
- cost: typeof usage.response_cost === "number" ? usage.response_cost : 0,
75100
+ cost: finiteOrNull(usage.response_cost) ?? headerNumber(headers, RESPONSE_COST_HEADER),
74727
75101
  };
74728
75102
  }
75103
+ /**
75104
+ * A reported number, or null when it was not reported as a finite number.
75105
+ *
75106
+ * @param value The raw value.
75107
+ * @returns The number, or null.
75108
+ */
75109
+ function finiteOrNull(value) {
75110
+ return typeof value === "number" && Number.isFinite(value) ? value : null;
75111
+ }
75112
+ /**
75113
+ * A numeric response header, or null when absent or not a finite number.
75114
+ *
75115
+ * @param headers The response headers, if the transport exposes them.
75116
+ * @param name The header name.
75117
+ * @returns The number, or null.
75118
+ */
75119
+ function headerNumber(headers, name) {
75120
+ const raw = headers?.get(name);
75121
+ if (raw === undefined || raw === null || raw.trim() === "") {
75122
+ return null;
75123
+ }
75124
+ return finiteOrNull(Number(raw));
75125
+ }
75126
+ /**
75127
+ * A non-empty string, or null.
75128
+ *
75129
+ * @param value The raw value.
75130
+ * @returns The string, or null.
75131
+ */
75132
+ function nonEmptyOrNull(value) {
75133
+ return typeof value === "string" && value.trim() !== "" ? value : null;
75134
+ }
74729
75135
  /**
74730
75136
  * Build the gateway transport.
74731
75137
  *
@@ -74776,13 +75182,17 @@ function createGatewayTransport(config) {
74776
75182
  // Usage is read before the content is interpreted. The provider billed for
74777
75183
  // this answer whether or not it parses, and a parse failure that dropped
74778
75184
  // the count would report the attempt as free.
74779
- const usage = readUsage(payload, request);
75185
+ const usage = readUsage(payload, request, response.headers);
74780
75186
  return {
74781
75187
  response: interpretContent(message?.content, request.responseFormat, usage),
74782
75188
  usage,
74783
75189
  tool_calls: Array.isArray(message?.tool_calls)
74784
75190
  ? message.tool_calls
74785
75191
  : undefined,
75192
+ // The model the provider says answered, which a proxy-side fallback can
75193
+ // make differ from the leg's route model; unreported stays null.
75194
+ servedModel: nonEmptyOrNull(payload.model),
75195
+ servedDeploymentId: nonEmptyOrNull(response.headers?.get(DEPLOYMENT_ID_HEADER)),
74786
75196
  };
74787
75197
  },
74788
75198
  };
@@ -74851,7 +75261,10 @@ function interpretContent(content, responseFormat, usage) {
74851
75261
  * table, per-provider parameter normalisation, a hard per-leg timeout, a
74852
75262
  * per-route circuit breaker, an ordered fallback chain ending at the closed
74853
75263
  * incumbent, and — where the caller supplies a validator — one schema-feedback
74854
- * retry ahead of the chain. None of them is optional, because a control that a
75264
+ * retry ahead of the chain. The caller's `timeoutMs` is ONE deadline for the
75265
+ * whole call: each leg runs for its route budget or for what remains of that
75266
+ * deadline, whichever is shorter, so the chain is the single fallback owner and
75267
+ * a slow primary cannot spend the time its fallbacks need. None of them is optional, because a control that a
74855
75268
  * caller can switch off is a control that will be off on the call that needed
74856
75269
  * it.
74857
75270
  *
@@ -74987,10 +75400,14 @@ function prepareLegs(chain, options, responseFormat, transport) {
74987
75400
  * @throws {SchemaRetryExhaustedError} When a validated payload failed twice.
74988
75401
  */
74989
75402
  async function callLLMByAlias(content, responseFormat = "text", options) {
74990
- const chain = resolveChain(options.alias, {
74991
- isolated: options.isolated,
74992
- timeoutMsOverride: options.timeoutMs,
74993
- });
75403
+ // Legs keep their route budgets; the caller's timeout is the deadline they
75404
+ // share, fixed once here so a validation retry and the degraded direct path
75405
+ // spend what is left of it rather than starting a fresh one.
75406
+ const chain = resolveChain(options.alias, { isolated: options.isolated });
75407
+ const clock = config.now ?? Date.now;
75408
+ const deadlineAtMs = typeof options.timeoutMs === "number" && Number.isFinite(options.timeoutMs)
75409
+ ? clock() + options.timeoutMs
75410
+ : undefined;
74994
75411
  if (chain.routes.length === 0) {
74995
75412
  // Nothing is servable. The exclusions say why each leg was unavailable,
74996
75413
  // which is the difference between an operator reading config and an
@@ -75033,6 +75450,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75033
75450
  breakers,
75034
75451
  correlationId: options.correlationId,
75035
75452
  callerSignal: options.signal,
75453
+ deadlineAtMs,
75036
75454
  now: config.now,
75037
75455
  onAttempt: (record) => attemptLog.push(record),
75038
75456
  });
@@ -75067,6 +75485,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75067
75485
  breakers,
75068
75486
  correlationId: options.correlationId,
75069
75487
  callerSignal: options.signal,
75488
+ deadlineAtMs,
75070
75489
  now: config.now,
75071
75490
  onAttempt: (record) => attemptLog.push(record),
75072
75491
  });
@@ -75079,6 +75498,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75079
75498
  usage: outcome.response.usage,
75080
75499
  tool_calls: outcome.response.tool_calls,
75081
75500
  servedBy: outcome.servedBy,
75501
+ servedModel: outcome.response.servedModel ?? null,
75082
75502
  attempts: attemptLog,
75083
75503
  degraded: outcome.degraded,
75084
75504
  totalUsage: outcome.totalUsage,
@@ -75112,6 +75532,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
75112
75532
  usage: validated.response.usage,
75113
75533
  tool_calls: validated.response.tool_calls,
75114
75534
  servedBy: routing.servedBy,
75535
+ servedModel: validated.response.servedModel ?? null,
75115
75536
  attempts: attemptLog,
75116
75537
  degraded: routing.degraded,
75117
75538
  totalUsage: validated.totalUsage,
@@ -80598,6 +81019,8 @@ exports.NewsError = NewsError;
80598
81019
  exports.NoServableRouteError = NoServableRouteError;
80599
81020
  exports.OptionStrategyError = OptionStrategyError;
80600
81021
  exports.OptionsDataError = OptionsDataError;
81022
+ exports.PENDING_CANCEL_ERROR_CODE = PENDING_CANCEL_ERROR_CODE;
81023
+ exports.PendingCancelError = PendingCancelError;
80601
81024
  exports.QuoteError = QuoteError;
80602
81025
  exports.RISK_FREE_RATE_TTL_MS = RISK_FREE_RATE_TTL_MS;
80603
81026
  exports.RateGuardTimeoutError = RateGuardTimeoutError;
@@ -80610,6 +81033,7 @@ exports.StreamTruncatedError = StreamTruncatedError;
80610
81033
  exports.TRADING_API = TRADING_API;
80611
81034
  exports.TimeoutError = TimeoutError;
80612
81035
  exports.TokenBucketRateLimiter = TokenBucketRateLimiter;
81036
+ exports.ToolChoiceIgnoredError = ToolChoiceIgnoredError;
80613
81037
  exports.TradeError = TradeError;
80614
81038
  exports.TrailingStopValidationError = TrailingStopValidationError;
80615
81039
  exports.USDC_PAIRS = USDC_PAIRS;
@@ -80628,6 +81052,7 @@ exports.adptc = adptc;
80628
81052
  exports.alpaca = alpaca;
80629
81053
  exports.analyzeBars = analyzeBars;
80630
81054
  exports.approximateImpliedVolatility = approximateImpliedVolatility;
81055
+ exports.assertToolChoiceHonoured = assertToolChoiceHonoured;
80631
81056
  exports.atr = atrNs;
80632
81057
  exports.availableStatistic = availableStatistic;
80633
81058
  exports.bracketOrders = bracketOrders;
@@ -80814,8 +81239,10 @@ exports.isOrderFillable = isOrderFillable;
80814
81239
  exports.isOrderFilled = isOrderFilled;
80815
81240
  exports.isOrderOpen = isOrderOpen;
80816
81241
  exports.isOrderTerminalStatus = isOrderTerminal$1;
81242
+ exports.isPendingCancelRejection = isPendingCancelRejection;
80817
81243
  exports.isSupportedCryptoPair = isSupportedCryptoPair;
80818
81244
  exports.isTransientNetworkError = isTransientNetworkError;
81245
+ exports.legBudgetMs = legBudgetMs;
80819
81246
  exports.legacyApi = index$1;
80820
81247
  exports.limitBuyWithTakeProfit = limitBuyWithTakeProfit;
80821
81248
  exports.limitsFor = limitsFor;