@adaptic/utils 0.0.1034 → 0.0.1036
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +462 -35
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +457 -36
- package/dist/index.mjs.map +1 -1
- package/dist/types/alpaca/legacy/orders.d.ts +3 -0
- package/dist/types/alpaca/legacy/orders.d.ts.map +1 -1
- package/dist/types/alpaca/trading/orders.d.ts +2 -0
- package/dist/types/alpaca/trading/orders.d.ts.map +1 -1
- package/dist/types/alpaca-trading-api.d.ts +3 -0
- package/dist/types/alpaca-trading-api.d.ts.map +1 -1
- package/dist/types/errors/index.d.ts +39 -0
- package/dist/types/errors/index.d.ts.map +1 -1
- package/dist/types/index.d.ts +1 -1
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/llm/alias-client.d.ts +4 -1
- package/dist/types/llm/alias-client.d.ts.map +1 -1
- package/dist/types/llm/fallback-chain.d.ts +24 -0
- package/dist/types/llm/fallback-chain.d.ts.map +1 -1
- package/dist/types/llm/index.d.ts +3 -3
- package/dist/types/llm/index.d.ts.map +1 -1
- package/dist/types/llm/param-matrix.d.ts +43 -1
- package/dist/types/llm/param-matrix.d.ts.map +1 -1
- package/dist/types/llm/transports/direct.d.ts.map +1 -1
- package/dist/types/llm/transports/gateway.d.ts +6 -4
- package/dist/types/llm/transports/gateway.d.ts.map +1 -1
- package/dist/types/llm/types.d.ts +67 -6
- package/dist/types/llm/types.d.ts.map +1 -1
- package/package.json +2 -2
package/dist/index.cjs
CHANGED
|
@@ -7506,6 +7506,68 @@ class DuplicateClientOrderIdError extends AlpacaApiError {
|
|
|
7506
7506
|
this.wasDerived = wasDerived;
|
|
7507
7507
|
}
|
|
7508
7508
|
}
|
|
7509
|
+
/** Stable `code` carried by {@link PendingCancelError}. */
|
|
7510
|
+
const PENDING_CANCEL_ERROR_CODE = "ORDER_PENDING_CANCEL";
|
|
7511
|
+
/**
|
|
7512
|
+
* Alpaca's numeric code for an unprocessable order request. It is shared by
|
|
7513
|
+
* many distinct reasons (pending cancel, not cancelable, bad replace), so the
|
|
7514
|
+
* reason text is what separates them.
|
|
7515
|
+
*/
|
|
7516
|
+
const ALPACA_UNPROCESSABLE_BROKER_CODE = 42210000;
|
|
7517
|
+
/** HTTP status Alpaca answers an unprocessable order request on. */
|
|
7518
|
+
const HTTP_UNPROCESSABLE_STATUS = 422;
|
|
7519
|
+
/** Alpaca's reason text for an order it already holds in `pending_cancel`. */
|
|
7520
|
+
const PENDING_CANCEL_REASON = /pending[ _]cancel/i;
|
|
7521
|
+
/**
|
|
7522
|
+
* Alpaca refused a DELETE because it has already accepted a cancel for the
|
|
7523
|
+
* order and holds it in `pending_cancel` until the venue confirms (HTTP 422,
|
|
7524
|
+
* code 42210000, "order pending cancel").
|
|
7525
|
+
*
|
|
7526
|
+
* This is a distinct broker answer from "not cancelable" (the order already
|
|
7527
|
+
* filled or is otherwise terminal): the order is being torn down, so it is
|
|
7528
|
+
* NOT live protection, and a second DELETE only returns the same 422. A caller
|
|
7529
|
+
* must treat the order as dying and wait for terminal truth (a `canceled` or
|
|
7530
|
+
* `filled` trade update, or a GET), never re-cancel it and never count it as
|
|
7531
|
+
* kept.
|
|
7532
|
+
*
|
|
7533
|
+
* The `message` is kept byte-identical to the generic rewrite
|
|
7534
|
+
* ("Order X is not cancelable") so consumers that still match that text keep
|
|
7535
|
+
* their behaviour; new consumers branch on the type, on
|
|
7536
|
+
* {@link PENDING_CANCEL_ERROR_CODE}, or on {@link isPendingCancelRejection}.
|
|
7537
|
+
* Never retryable.
|
|
7538
|
+
*/
|
|
7539
|
+
class PendingCancelError extends AlpacaApiError {
|
|
7540
|
+
orderId;
|
|
7541
|
+
constructor(message,
|
|
7542
|
+
/** The broker order id the DELETE targeted. */
|
|
7543
|
+
orderId, cause) {
|
|
7544
|
+
super(message, PENDING_CANCEL_ERROR_CODE, HTTP_UNPROCESSABLE_STATUS, cause, extractAlpacaBrokerError(cause));
|
|
7545
|
+
this.orderId = orderId;
|
|
7546
|
+
}
|
|
7547
|
+
}
|
|
7548
|
+
/**
|
|
7549
|
+
* Did Alpaca answer that the order is already pending cancel?
|
|
7550
|
+
*
|
|
7551
|
+
* True for a {@link PendingCancelError}, and for any thrown value whose broker
|
|
7552
|
+
* payload (on the value or its `cause` chain) is a 422 / 42210000 whose reason
|
|
7553
|
+
* names `pending cancel` / `pending_cancel`. A payload with no reason text is
|
|
7554
|
+
* never read as pending cancel: absence stays unknown.
|
|
7555
|
+
*
|
|
7556
|
+
* @param error - The thrown value.
|
|
7557
|
+
* @returns Whether the broker holds the order in `pending_cancel`.
|
|
7558
|
+
*/
|
|
7559
|
+
function isPendingCancelRejection(error) {
|
|
7560
|
+
if (error instanceof PendingCancelError) {
|
|
7561
|
+
return true;
|
|
7562
|
+
}
|
|
7563
|
+
const detail = extractAlpacaBrokerError(error);
|
|
7564
|
+
if (detail === undefined || detail.brokerMessage === null) {
|
|
7565
|
+
return false;
|
|
7566
|
+
}
|
|
7567
|
+
const unprocessable = detail.statusCode === HTTP_UNPROCESSABLE_STATUS ||
|
|
7568
|
+
detail.brokerCode === ALPACA_UNPROCESSABLE_BROKER_CODE;
|
|
7569
|
+
return unprocessable && PENDING_CANCEL_REASON.test(detail.brokerMessage);
|
|
7570
|
+
}
|
|
7509
7571
|
/** Max depth walked along the `error.cause` chain when locating a broker payload. */
|
|
7510
7572
|
const MAX_BROKER_ERROR_CAUSE_DEPTH = 6;
|
|
7511
7573
|
/**
|
|
@@ -11026,6 +11088,9 @@ class AlpacaTradingAPI {
|
|
|
11026
11088
|
/**
|
|
11027
11089
|
* Cancel a specific order by its ID
|
|
11028
11090
|
* @param orderId The id of the order to cancel
|
|
11091
|
+
* @throws PendingCancelError if the broker already holds the order in
|
|
11092
|
+
* `pending_cancel` (422 / 42210000 "order pending cancel"): the order is
|
|
11093
|
+
* being torn down, so it is not live and must not be cancelled again
|
|
11029
11094
|
* @throws Error if the order is not cancelable (status 422) or if the order doesn't exist
|
|
11030
11095
|
* @returns Promise that resolves when the order is successfully canceled
|
|
11031
11096
|
*/
|
|
@@ -11036,6 +11101,12 @@ class AlpacaTradingAPI {
|
|
|
11036
11101
|
this.log(`Successfully canceled order ${orderId}`);
|
|
11037
11102
|
}
|
|
11038
11103
|
catch (error) {
|
|
11104
|
+
// A cancel the broker already accepted: the order is dying, not kept.
|
|
11105
|
+
// Typed so callers stop reading it as "not cancelable, still live".
|
|
11106
|
+
if (isPendingCancelRejection(error)) {
|
|
11107
|
+
this.log(`Order ${orderId} is already pending cancel at the broker; not re-cancelling`, { type: "warn" });
|
|
11108
|
+
throw new PendingCancelError(`Order ${orderId} is not cancelable`, orderId, error);
|
|
11109
|
+
}
|
|
11039
11110
|
// If the error is a 422, it means the order is not cancelable
|
|
11040
11111
|
if (error instanceof Error && error.message.includes("422")) {
|
|
11041
11112
|
this.log(`Order ${orderId} is not cancelable`, {
|
|
@@ -12315,6 +12386,9 @@ async function replaceOrder$1(auth, orderId, params) {
|
|
|
12315
12386
|
* @param auth - The authentication details for Alpaca
|
|
12316
12387
|
* @param orderId - The ID of the order to cancel
|
|
12317
12388
|
* @returns Success status and optional message if order not found
|
|
12389
|
+
* @throws PendingCancelError if the broker already holds the order in
|
|
12390
|
+
* `pending_cancel` (the order is dying; never cancel it again). The message
|
|
12391
|
+
* is the same "Failed to cancel order: ..." text as any other refusal.
|
|
12318
12392
|
*/
|
|
12319
12393
|
async function cancelOrder$1(auth, orderId) {
|
|
12320
12394
|
try {
|
|
@@ -12334,7 +12408,13 @@ async function cancelOrder$1(auth, orderId) {
|
|
|
12334
12408
|
return { success: false, message: `Order not found: ${orderId}` };
|
|
12335
12409
|
}
|
|
12336
12410
|
else {
|
|
12337
|
-
|
|
12411
|
+
const httpError = alpacaHttpError(`Failed to cancel order: ${response.status} ${response.statusText} ${errorText}`, response.status, errorText);
|
|
12412
|
+
// A cancel the broker already accepted: the order is dying, not kept,
|
|
12413
|
+
// and a second DELETE only returns the same 422.
|
|
12414
|
+
if (isPendingCancelRejection(httpError)) {
|
|
12415
|
+
throw new PendingCancelError(httpError.message, orderId, httpError);
|
|
12416
|
+
}
|
|
12417
|
+
throw httpError;
|
|
12338
12418
|
}
|
|
12339
12419
|
}
|
|
12340
12420
|
return { success: true };
|
|
@@ -68338,6 +68418,8 @@ async function getOrders(client, params = {}) {
|
|
|
68338
68418
|
*
|
|
68339
68419
|
* @param client - The AlpacaClient instance
|
|
68340
68420
|
* @param orderId - The unique identifier of the order to cancel
|
|
68421
|
+
* @throws PendingCancelError if the broker already holds the order in
|
|
68422
|
+
* `pending_cancel` (the order is dying; never cancel it again)
|
|
68341
68423
|
* @throws Error if order cannot be canceled (e.g., already filled or canceled)
|
|
68342
68424
|
*
|
|
68343
68425
|
* @example
|
|
@@ -68353,6 +68435,11 @@ async function cancelOrder(client, orderId) {
|
|
|
68353
68435
|
}
|
|
68354
68436
|
catch (error) {
|
|
68355
68437
|
const errorMessage = error instanceof Error ? error.message : "Unknown error";
|
|
68438
|
+
// A cancel the broker already accepted: the order is dying, not kept.
|
|
68439
|
+
if (isPendingCancelRejection(error)) {
|
|
68440
|
+
log$6(`Order ${orderId} is already pending cancel at the broker; not re-cancelling`, { type: "warn" });
|
|
68441
|
+
throw new PendingCancelError(`Order ${orderId} is not cancelable`, orderId, error);
|
|
68442
|
+
}
|
|
68356
68443
|
// Check for specific error conditions
|
|
68357
68444
|
if (errorMessage.includes("422") ||
|
|
68358
68445
|
errorMessage.includes("not cancelable")) {
|
|
@@ -72495,6 +72582,10 @@ const MAX_TOKENS_PARAM_BY_API_STYLE = {
|
|
|
72495
72582
|
};
|
|
72496
72583
|
/** The response-format key an OpenAI-compatible provider expects. */
|
|
72497
72584
|
const RESPONSE_FORMAT_KEY = "response_format";
|
|
72585
|
+
/** The tool-choice key, in the OpenAI-compatible vocabulary the gateway speaks. */
|
|
72586
|
+
const TOOL_CHOICE_KEY = "tool_choice";
|
|
72587
|
+
/** The mandatory tool-choice value: the answer must be a tool call. */
|
|
72588
|
+
const TOOL_CHOICE_REQUIRED = "required";
|
|
72498
72589
|
/**
|
|
72499
72590
|
* Normalise a caller's options into the exact parameter set one leg accepts.
|
|
72500
72591
|
*
|
|
@@ -72542,12 +72633,91 @@ function normaliseParams(options, route, responseFormat) {
|
|
|
72542
72633
|
// tool calls concurrently reorders its own effects, and ordering is part
|
|
72543
72634
|
// of the meaning of a sequence of trading actions.
|
|
72544
72635
|
params.parallel_tool_calls = false;
|
|
72636
|
+
const toolChoice = normaliseToolChoice(options.toolChoice, route);
|
|
72637
|
+
if (toolChoice !== undefined) {
|
|
72638
|
+
params[TOOL_CHOICE_KEY] = toolChoice;
|
|
72639
|
+
}
|
|
72545
72640
|
}
|
|
72546
72641
|
if (options.metadata !== undefined) {
|
|
72547
72642
|
params.metadata = { ...options.metadata };
|
|
72548
72643
|
}
|
|
72549
72644
|
return params;
|
|
72550
72645
|
}
|
|
72646
|
+
/**
|
|
72647
|
+
* Translate the caller's tool-choice policy into the value one leg is sent.
|
|
72648
|
+
*
|
|
72649
|
+
* `"auto"` is every provider's default once tools are present, so it is never
|
|
72650
|
+
* sent. `"required"` is sent only to a route MEASURED to honour it: serving
|
|
72651
|
+
* stacks accept the parameter and silently answer in prose anyway, so sending
|
|
72652
|
+
* it to an unmeasured route would assert a constraint nobody checked. Omitting
|
|
72653
|
+
* it there leaves that leg exactly as it was, with the caller's own validation
|
|
72654
|
+
* as the guard.
|
|
72655
|
+
*
|
|
72656
|
+
* @param toolChoice The caller's policy, if any.
|
|
72657
|
+
* @param route The leg being prepared.
|
|
72658
|
+
* @returns The value to send, or undefined to omit the parameter.
|
|
72659
|
+
*/
|
|
72660
|
+
function normaliseToolChoice(toolChoice, route) {
|
|
72661
|
+
if (toolChoice !== TOOL_CHOICE_REQUIRED) {
|
|
72662
|
+
return undefined;
|
|
72663
|
+
}
|
|
72664
|
+
return route.params.supports_tool_choice === true ? TOOL_CHOICE_REQUIRED : undefined;
|
|
72665
|
+
}
|
|
72666
|
+
/**
|
|
72667
|
+
* Fail a leg that was sent a mandatory tool choice and answered without a tool call.
|
|
72668
|
+
*
|
|
72669
|
+
* The parameter is sent only to a route declared to honour it, so an answer
|
|
72670
|
+
* with no tool call means the declaration no longer holds for this leg — a
|
|
72671
|
+
* serving-stack upgrade can drop the constraint without an error. Returning
|
|
72672
|
+
* that answer would hand a caller prose where it demanded an action; failing
|
|
72673
|
+
* the leg lets the chain advance and names the broken declaration.
|
|
72674
|
+
*
|
|
72675
|
+
* @param route The leg that answered.
|
|
72676
|
+
* @param params The parameters it was sent.
|
|
72677
|
+
* @param response The leg's answer: its tool calls, and the usage and serving
|
|
72678
|
+
* model the error carries so the leg's spend and attribution are not lost.
|
|
72679
|
+
* @throws {ToolChoiceIgnoredError} When a mandatory choice produced no tool call.
|
|
72680
|
+
*/
|
|
72681
|
+
function assertToolChoiceHonoured(route, params, response) {
|
|
72682
|
+
if (params[TOOL_CHOICE_KEY] !== TOOL_CHOICE_REQUIRED) {
|
|
72683
|
+
return;
|
|
72684
|
+
}
|
|
72685
|
+
if (response.tool_calls !== undefined && response.tool_calls.length > 0) {
|
|
72686
|
+
return;
|
|
72687
|
+
}
|
|
72688
|
+
throw new ToolChoiceIgnoredError(route, response.usage, response.servedModel ?? null);
|
|
72689
|
+
}
|
|
72690
|
+
/**
|
|
72691
|
+
* Thrown when a route declared to honour a mandatory tool choice answered
|
|
72692
|
+
* without a tool call.
|
|
72693
|
+
*
|
|
72694
|
+
* Distinct from a provider outage: the route answered, but not in the form it
|
|
72695
|
+
* is declared to guarantee. The chain advances, and the route's health is not
|
|
72696
|
+
* charged, because the fault is in the declaration, not in availability. The
|
|
72697
|
+
* provider still billed for the answer, so the error carries the leg's usage:
|
|
72698
|
+
* the spend belongs in the chain's total whether or not a later leg serves.
|
|
72699
|
+
*/
|
|
72700
|
+
class ToolChoiceIgnoredError extends Error {
|
|
72701
|
+
/** The leg that ignored the choice. */
|
|
72702
|
+
routeKey;
|
|
72703
|
+
/** What the provider billed for the answer that carried no tool call. */
|
|
72704
|
+
usage;
|
|
72705
|
+
/** The model the provider reports as having answered, or `null` when unreported. */
|
|
72706
|
+
servedModel;
|
|
72707
|
+
/**
|
|
72708
|
+
* @param route The leg.
|
|
72709
|
+
* @param usage What the provider billed for the answer.
|
|
72710
|
+
* @param servedModel The provider-reported serving model, or `null`.
|
|
72711
|
+
*/
|
|
72712
|
+
constructor(route, usage, servedModel) {
|
|
72713
|
+
super(`route ${route.routeKey} (${route.providerName}/${route.modelId}) was sent tool_choice "required" ` +
|
|
72714
|
+
"and answered without a tool call; its supports_tool_choice declaration no longer holds");
|
|
72715
|
+
this.name = "ToolChoiceIgnoredError";
|
|
72716
|
+
this.routeKey = route.routeKey;
|
|
72717
|
+
this.usage = usage;
|
|
72718
|
+
this.servedModel = servedModel;
|
|
72719
|
+
}
|
|
72720
|
+
}
|
|
72551
72721
|
/**
|
|
72552
72722
|
* Translate the requested response shape into the parameter a route accepts.
|
|
72553
72723
|
*
|
|
@@ -73249,6 +73419,9 @@ class ChainExhaustedError extends Error {
|
|
|
73249
73419
|
* understate spend by exactly the amount the failures cost — which is the
|
|
73250
73420
|
* amount a fallback chain is most likely to run up.
|
|
73251
73421
|
*
|
|
73422
|
+
* A count one attempt did not report makes the total for that count unknown
|
|
73423
|
+
* (`null`): a sum that skipped it would present a partial figure as complete.
|
|
73424
|
+
*
|
|
73252
73425
|
* @param a The running total.
|
|
73253
73426
|
* @param b The attempt to add, if any.
|
|
73254
73427
|
* @returns The combined usage.
|
|
@@ -73258,8 +73431,8 @@ function sumUsage(a, b) {
|
|
|
73258
73431
|
return a;
|
|
73259
73432
|
}
|
|
73260
73433
|
return {
|
|
73261
|
-
prompt_tokens: a.prompt_tokens
|
|
73262
|
-
completion_tokens: a.completion_tokens
|
|
73434
|
+
prompt_tokens: addKnown(a.prompt_tokens, b.prompt_tokens),
|
|
73435
|
+
completion_tokens: addKnown(a.completion_tokens, b.completion_tokens),
|
|
73263
73436
|
reasoning_tokens: a.reasoning_tokens === undefined && b.reasoning_tokens === undefined
|
|
73264
73437
|
? undefined
|
|
73265
73438
|
: (a.reasoning_tokens ?? 0) + (b.reasoning_tokens ?? 0),
|
|
@@ -73268,9 +73441,38 @@ function sumUsage(a, b) {
|
|
|
73268
73441
|
: (a.cached_tokens ?? 0) + (b.cached_tokens ?? 0),
|
|
73269
73442
|
provider: b.provider,
|
|
73270
73443
|
model: b.model,
|
|
73271
|
-
cost: a.cost
|
|
73444
|
+
cost: addKnown(a.cost, b.cost),
|
|
73272
73445
|
};
|
|
73273
73446
|
}
|
|
73447
|
+
/**
|
|
73448
|
+
* Add two measured quantities, either of which may be unreported.
|
|
73449
|
+
*
|
|
73450
|
+
* @param a The running total, or null when already unknown.
|
|
73451
|
+
* @param b The value to add, or null when unreported.
|
|
73452
|
+
* @returns The sum, or null when either side is unknown.
|
|
73453
|
+
*/
|
|
73454
|
+
function addKnown(a, b) {
|
|
73455
|
+
return a === null || b === null ? null : a + b;
|
|
73456
|
+
}
|
|
73457
|
+
/**
|
|
73458
|
+
* The budget one leg may spend.
|
|
73459
|
+
*
|
|
73460
|
+
* The route's own budget, cut to what remains of the caller's deadline. The
|
|
73461
|
+
* route budget is never widened, so a caller with a long deadline still has a
|
|
73462
|
+
* slow leg abandoned in time for the next one to run; and the deadline is never
|
|
73463
|
+
* exceeded, so the last leg reached ends when the caller stops waiting.
|
|
73464
|
+
*
|
|
73465
|
+
* @param routeBudgetMs The leg's route budget.
|
|
73466
|
+
* @param deadlineAtMs The caller's deadline instant, if any.
|
|
73467
|
+
* @param nowMs The current instant on the same clock.
|
|
73468
|
+
* @returns The leg's budget in milliseconds; zero or less means none is left.
|
|
73469
|
+
*/
|
|
73470
|
+
function legBudgetMs(routeBudgetMs, deadlineAtMs, nowMs) {
|
|
73471
|
+
if (deadlineAtMs === undefined) {
|
|
73472
|
+
return routeBudgetMs;
|
|
73473
|
+
}
|
|
73474
|
+
return Math.min(routeBudgetMs, deadlineAtMs - nowMs);
|
|
73475
|
+
}
|
|
73274
73476
|
/** Raised internally when a leg exceeds its budget. */
|
|
73275
73477
|
class LegTimeoutError extends Error {
|
|
73276
73478
|
/**
|
|
@@ -73292,11 +73494,11 @@ class LegTimeoutError extends Error {
|
|
|
73292
73494
|
* @param leg The leg to run.
|
|
73293
73495
|
* @param params Normalised parameters for this leg.
|
|
73294
73496
|
* @param execution The call context.
|
|
73497
|
+
* @param budgetMs The leg's budget: its route budget cut to the caller's deadline.
|
|
73295
73498
|
* @returns The provider's answer.
|
|
73296
73499
|
*/
|
|
73297
|
-
async function runLeg(leg, params, execution) {
|
|
73500
|
+
async function runLeg(leg, params, execution, budgetMs) {
|
|
73298
73501
|
const controller = new AbortController();
|
|
73299
|
-
const budgetMs = leg.route.timeoutMs;
|
|
73300
73502
|
const timer = setTimeout(() => {
|
|
73301
73503
|
controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
|
|
73302
73504
|
}, budgetMs);
|
|
@@ -73318,7 +73520,7 @@ async function runLeg(leg, params, execution) {
|
|
|
73318
73520
|
// provider, and only one clock should govern both. The leg's own signal is
|
|
73319
73521
|
// handed to the guard as well, so a leg whose budget or caller is gone
|
|
73320
73522
|
// leaves the queue at once instead of holding its place in it.
|
|
73321
|
-
|
|
73523
|
+
const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
|
|
73322
73524
|
route: leg.route,
|
|
73323
73525
|
content: execution.content,
|
|
73324
73526
|
responseFormat: execution.responseFormat,
|
|
@@ -73328,6 +73530,8 @@ async function runLeg(leg, params, execution) {
|
|
|
73328
73530
|
signal: controller.signal,
|
|
73329
73531
|
correlationId: execution.correlationId,
|
|
73330
73532
|
}), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
|
|
73533
|
+
assertToolChoiceHonoured(leg.route, params, response);
|
|
73534
|
+
return response;
|
|
73331
73535
|
}
|
|
73332
73536
|
finally {
|
|
73333
73537
|
clearTimeout(timer);
|
|
@@ -73360,6 +73564,11 @@ function classify(error, callerSignal) {
|
|
|
73360
73564
|
if (error instanceof UnsupportedCapabilityError) {
|
|
73361
73565
|
return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
|
|
73362
73566
|
}
|
|
73567
|
+
if (error instanceof ToolChoiceIgnoredError) {
|
|
73568
|
+
// The route answered; it broke a declared guarantee rather than failing to
|
|
73569
|
+
// be available, so its breaker is not charged for it.
|
|
73570
|
+
return { outcome: "error", reason: error.message, countsAgainstHealth: false };
|
|
73571
|
+
}
|
|
73363
73572
|
if (error instanceof RateGuardTimeoutError) {
|
|
73364
73573
|
// Self-inflicted pacing, not provider ill-health. Counting it would let the
|
|
73365
73574
|
// client's own throttling open a breaker on a perfectly healthy provider
|
|
@@ -73391,6 +73600,22 @@ function classify(error, callerSignal) {
|
|
|
73391
73600
|
function isAborted(signal) {
|
|
73392
73601
|
return signal !== undefined && signal.aborted;
|
|
73393
73602
|
}
|
|
73603
|
+
/**
|
|
73604
|
+
* The usage a failed leg was billed for, when the leg reached an answer.
|
|
73605
|
+
*
|
|
73606
|
+
* A leg that failed after the provider answered — content that does not parse,
|
|
73607
|
+
* or prose where a tool call was mandatory — was still charged. A leg that
|
|
73608
|
+
* never answered (timeout, outage, skip) carries no usage, and none is invented.
|
|
73609
|
+
*
|
|
73610
|
+
* @param error The thrown value.
|
|
73611
|
+
* @returns The billed usage, or undefined when the leg never produced an answer.
|
|
73612
|
+
*/
|
|
73613
|
+
function billedUsageOf(error) {
|
|
73614
|
+
if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
|
|
73615
|
+
return error.usage;
|
|
73616
|
+
}
|
|
73617
|
+
return undefined;
|
|
73618
|
+
}
|
|
73394
73619
|
/**
|
|
73395
73620
|
* Walk a chain until a leg answers.
|
|
73396
73621
|
*
|
|
@@ -73438,10 +73663,27 @@ async function executeChain(alias, execution) {
|
|
|
73438
73663
|
execution.onAttempt?.(record);
|
|
73439
73664
|
continue;
|
|
73440
73665
|
}
|
|
73666
|
+
const budgetMs = legBudgetMs(route.timeoutMs, execution.deadlineAtMs, now());
|
|
73667
|
+
if (budgetMs <= 0) {
|
|
73668
|
+
// The caller's deadline is spent. Dispatching now would start a call
|
|
73669
|
+
// that is cancelled the moment it begins, and charge nothing but noise.
|
|
73670
|
+
const record = {
|
|
73671
|
+
routeKey: route.routeKey,
|
|
73672
|
+
role: route.role,
|
|
73673
|
+
provider: route.providerName,
|
|
73674
|
+
modelId: route.modelId,
|
|
73675
|
+
outcome: "skipped",
|
|
73676
|
+
durationMs: 0,
|
|
73677
|
+
reason: "caller deadline exhausted before this leg",
|
|
73678
|
+
};
|
|
73679
|
+
attempts.push(record);
|
|
73680
|
+
execution.onAttempt?.(record);
|
|
73681
|
+
continue;
|
|
73682
|
+
}
|
|
73441
73683
|
const startedAt = now();
|
|
73442
73684
|
const holdsProbe = execution.breakers.onAttemptStart(route.routeKey);
|
|
73443
73685
|
try {
|
|
73444
|
-
const response = await runLeg(leg, leg.params, execution);
|
|
73686
|
+
const response = await runLeg(leg, leg.params, execution, budgetMs);
|
|
73445
73687
|
execution.breakers.onSuccess(route.routeKey);
|
|
73446
73688
|
totalUsage = sumUsage(totalUsage, response.usage);
|
|
73447
73689
|
const record = {
|
|
@@ -73451,6 +73693,8 @@ async function executeChain(alias, execution) {
|
|
|
73451
73693
|
modelId: route.modelId,
|
|
73452
73694
|
outcome: "ok",
|
|
73453
73695
|
durationMs: now() - startedAt,
|
|
73696
|
+
budgetMs,
|
|
73697
|
+
servedModel: response.servedModel ?? null,
|
|
73454
73698
|
usage: response.usage,
|
|
73455
73699
|
};
|
|
73456
73700
|
attempts.push(record);
|
|
@@ -73467,9 +73711,11 @@ async function executeChain(alias, execution) {
|
|
|
73467
73711
|
// took must come back, or a half-open route admits no probe ever again.
|
|
73468
73712
|
execution.breakers.onAttemptAbandoned(route.routeKey);
|
|
73469
73713
|
}
|
|
73470
|
-
// A provider that answered with unparseable content
|
|
73471
|
-
//
|
|
73472
|
-
|
|
73714
|
+
// A provider that answered — with unparseable content, or in prose where a
|
|
73715
|
+
// tool call was mandatory — still billed for the answer; the spend belongs
|
|
73716
|
+
// in the total whether or not a later leg serves.
|
|
73717
|
+
const billed = billedUsageOf(error);
|
|
73718
|
+
const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
|
|
73473
73719
|
totalUsage = sumUsage(totalUsage, billed);
|
|
73474
73720
|
const record = {
|
|
73475
73721
|
routeKey: route.routeKey,
|
|
@@ -73478,7 +73724,9 @@ async function executeChain(alias, execution) {
|
|
|
73478
73724
|
modelId: route.modelId,
|
|
73479
73725
|
outcome,
|
|
73480
73726
|
durationMs: now() - startedAt,
|
|
73727
|
+
budgetMs,
|
|
73481
73728
|
reason,
|
|
73729
|
+
...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
|
|
73482
73730
|
...(billed === undefined ? {} : { usage: billed }),
|
|
73483
73731
|
};
|
|
73484
73732
|
attempts.push(record);
|
|
@@ -73670,6 +73918,7 @@ var aliases = {
|
|
|
73670
73918
|
supports_tools: true,
|
|
73671
73919
|
supports_cache_control: true,
|
|
73672
73920
|
supports_vision: false,
|
|
73921
|
+
supports_tool_choice: false,
|
|
73673
73922
|
context_window: 1048576
|
|
73674
73923
|
},
|
|
73675
73924
|
price_per_mtok: {
|
|
@@ -73696,6 +73945,7 @@ var aliases = {
|
|
|
73696
73945
|
supports_tools: true,
|
|
73697
73946
|
supports_cache_control: true,
|
|
73698
73947
|
supports_vision: false,
|
|
73948
|
+
supports_tool_choice: true,
|
|
73699
73949
|
context_window: 1048576
|
|
73700
73950
|
},
|
|
73701
73951
|
price_per_mtok: {
|
|
@@ -73722,6 +73972,7 @@ var aliases = {
|
|
|
73722
73972
|
supports_tools: true,
|
|
73723
73973
|
supports_cache_control: true,
|
|
73724
73974
|
supports_vision: true,
|
|
73975
|
+
supports_tool_choice: true,
|
|
73725
73976
|
context_window: 1000000
|
|
73726
73977
|
},
|
|
73727
73978
|
price_per_mtok: {
|
|
@@ -73766,6 +74017,7 @@ var aliases = {
|
|
|
73766
74017
|
supports_tools: true,
|
|
73767
74018
|
supports_cache_control: true,
|
|
73768
74019
|
supports_vision: true,
|
|
74020
|
+
supports_tool_choice: true,
|
|
73769
74021
|
context_window: 1048576
|
|
73770
74022
|
},
|
|
73771
74023
|
price_per_mtok: {
|
|
@@ -73792,6 +74044,7 @@ var aliases = {
|
|
|
73792
74044
|
supports_tools: true,
|
|
73793
74045
|
supports_cache_control: true,
|
|
73794
74046
|
supports_vision: false,
|
|
74047
|
+
supports_tool_choice: false,
|
|
73795
74048
|
context_window: 1048576
|
|
73796
74049
|
},
|
|
73797
74050
|
price_per_mtok: {
|
|
@@ -73818,6 +74071,7 @@ var aliases = {
|
|
|
73818
74071
|
supports_tools: true,
|
|
73819
74072
|
supports_cache_control: true,
|
|
73820
74073
|
supports_vision: true,
|
|
74074
|
+
supports_tool_choice: true,
|
|
73821
74075
|
context_window: 1000000
|
|
73822
74076
|
},
|
|
73823
74077
|
price_per_mtok: {
|
|
@@ -73862,6 +74116,7 @@ var aliases = {
|
|
|
73862
74116
|
supports_tools: true,
|
|
73863
74117
|
supports_cache_control: true,
|
|
73864
74118
|
supports_vision: false,
|
|
74119
|
+
supports_tool_choice: false,
|
|
73865
74120
|
context_window: 1048576
|
|
73866
74121
|
},
|
|
73867
74122
|
price_per_mtok: {
|
|
@@ -73888,6 +74143,7 @@ var aliases = {
|
|
|
73888
74143
|
supports_tools: true,
|
|
73889
74144
|
supports_cache_control: false,
|
|
73890
74145
|
supports_vision: false,
|
|
74146
|
+
supports_tool_choice: true,
|
|
73891
74147
|
context_window: 131072
|
|
73892
74148
|
},
|
|
73893
74149
|
price_per_mtok: {
|
|
@@ -73914,6 +74170,7 @@ var aliases = {
|
|
|
73914
74170
|
supports_tools: true,
|
|
73915
74171
|
supports_cache_control: true,
|
|
73916
74172
|
supports_vision: true,
|
|
74173
|
+
supports_tool_choice: true,
|
|
73917
74174
|
context_window: 200000
|
|
73918
74175
|
},
|
|
73919
74176
|
price_per_mtok: {
|
|
@@ -73957,6 +74214,7 @@ var aliases = {
|
|
|
73957
74214
|
supports_tools: true,
|
|
73958
74215
|
supports_cache_control: false,
|
|
73959
74216
|
supports_vision: false,
|
|
74217
|
+
supports_tool_choice: true,
|
|
73960
74218
|
context_window: 131072
|
|
73961
74219
|
},
|
|
73962
74220
|
price_per_mtok: {
|
|
@@ -73983,6 +74241,7 @@ var aliases = {
|
|
|
73983
74241
|
supports_tools: true,
|
|
73984
74242
|
supports_cache_control: true,
|
|
73985
74243
|
supports_vision: false,
|
|
74244
|
+
supports_tool_choice: true,
|
|
73986
74245
|
context_window: 262144
|
|
73987
74246
|
},
|
|
73988
74247
|
price_per_mtok: {
|
|
@@ -74009,6 +74268,7 @@ var aliases = {
|
|
|
74009
74268
|
supports_tools: true,
|
|
74010
74269
|
supports_cache_control: true,
|
|
74011
74270
|
supports_vision: true,
|
|
74271
|
+
supports_tool_choice: true,
|
|
74012
74272
|
context_window: 200000
|
|
74013
74273
|
},
|
|
74014
74274
|
price_per_mtok: {
|
|
@@ -74054,6 +74314,7 @@ var aliases = {
|
|
|
74054
74314
|
supports_tools: true,
|
|
74055
74315
|
supports_cache_control: true,
|
|
74056
74316
|
supports_vision: false,
|
|
74317
|
+
supports_tool_choice: false,
|
|
74057
74318
|
context_window: 1048576
|
|
74058
74319
|
},
|
|
74059
74320
|
price_per_mtok: {
|
|
@@ -74065,6 +74326,105 @@ var aliases = {
|
|
|
74065
74326
|
notes: "Pinned with no fallback (PD-6). A judge that failed over would grade one model's output against another model's standard, corrupting every gate decided by it."
|
|
74066
74327
|
}
|
|
74067
74328
|
]
|
|
74329
|
+
},
|
|
74330
|
+
"llm.decide": {
|
|
74331
|
+
workload: "Per-symbol trading decision: a multi-turn tool-calling read that ends in exactly one decision tool call",
|
|
74332
|
+
latency_class: "hot-path",
|
|
74333
|
+
criticality: "trading-adjacent",
|
|
74334
|
+
isolation_capable: false,
|
|
74335
|
+
eval_gate: "tool-call",
|
|
74336
|
+
policy_note: "The dedicated decision alias. The per-symbol decision call is a trading decision, not a latency-critical extraction, so it gets its own chain instead of borrowing llm.fast's. It is evaluated in SHADOW beside the live llm.fast path (schema-valid rate, latency, served model, action agreement) and serves live traffic only through the engine's governed routing flag after that evidence. Leg budgets are the hot-path class's today; budgets set from each model's measured p95 are a separate, evidence-gated change. Its monthly budget bounds the shadow's own spend and is added to the gateway total, so the shadow cannot consume the live aliases' budget.",
|
|
74337
|
+
budget: {
|
|
74338
|
+
basis: "provisional-pre-baseline",
|
|
74339
|
+
monthly_usd: 600,
|
|
74340
|
+
alert_pct: [
|
|
74341
|
+
50,
|
|
74342
|
+
80,
|
|
74343
|
+
100
|
|
74344
|
+
]
|
|
74345
|
+
},
|
|
74346
|
+
routes: [
|
|
74347
|
+
{
|
|
74348
|
+
role: "primary",
|
|
74349
|
+
provider: "deepinfra",
|
|
74350
|
+
model_id: "deepseek-ai/DeepSeek-V4-Pro",
|
|
74351
|
+
model_id_status: "confirmed",
|
|
74352
|
+
model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
|
|
74353
|
+
model_family: "DeepSeek V4 Pro",
|
|
74354
|
+
lumic_model: "deepseek-ai/DeepSeek-V4-Pro",
|
|
74355
|
+
params: {
|
|
74356
|
+
temperature: null,
|
|
74357
|
+
max_output_tokens: null,
|
|
74358
|
+
supports_temperature: true,
|
|
74359
|
+
supports_json_schema: true,
|
|
74360
|
+
supports_tools: true,
|
|
74361
|
+
supports_cache_control: true,
|
|
74362
|
+
supports_vision: false,
|
|
74363
|
+
supports_tool_choice: false,
|
|
74364
|
+
context_window: 1048576
|
|
74365
|
+
},
|
|
74366
|
+
price_per_mtok: {
|
|
74367
|
+
input: 1.3,
|
|
74368
|
+
output: 2.6,
|
|
74369
|
+
as_of: "2026-09-11",
|
|
74370
|
+
source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
|
|
74371
|
+
},
|
|
74372
|
+
notes: "The advanced tier the per-symbol decision prompt was written for: a long, multi-turn, tool-calling read of about 21k prompt tokens, where the hot-path alias's small models answer faster but lose the schema contract more often. It ignores a mandatory tool_choice (measured), so the decision contract's own validation stays the guard on this leg."
|
|
74373
|
+
},
|
|
74374
|
+
{
|
|
74375
|
+
role: "secondary",
|
|
74376
|
+
provider: "deepinfra",
|
|
74377
|
+
model_id: "deepseek-ai/DeepSeek-V4-Flash",
|
|
74378
|
+
model_id_status: "confirmed",
|
|
74379
|
+
model_id_source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account",
|
|
74380
|
+
model_family: "DeepSeek V4 Flash",
|
|
74381
|
+
lumic_model: "deepseek-ai/DeepSeek-V4-Flash",
|
|
74382
|
+
params: {
|
|
74383
|
+
temperature: null,
|
|
74384
|
+
max_output_tokens: null,
|
|
74385
|
+
supports_temperature: true,
|
|
74386
|
+
supports_json_schema: true,
|
|
74387
|
+
supports_tools: true,
|
|
74388
|
+
supports_cache_control: true,
|
|
74389
|
+
supports_vision: false,
|
|
74390
|
+
supports_tool_choice: false,
|
|
74391
|
+
context_window: 1048576
|
|
74392
|
+
},
|
|
74393
|
+
price_per_mtok: {
|
|
74394
|
+
input: 0.09,
|
|
74395
|
+
output: 0.18,
|
|
74396
|
+
as_of: "2026-09-11",
|
|
74397
|
+
source: "DeepInfra /v1/openai/models catalogue queried against the live account 2026-09-11; tool-calling and latency verified by direct call on the same account"
|
|
74398
|
+
},
|
|
74399
|
+
notes: "The current decision model, kept as the secondary so a slow or failing primary falls back to the model the decision path already runs on."
|
|
74400
|
+
},
|
|
74401
|
+
{
|
|
74402
|
+
role: "closed_incumbent",
|
|
74403
|
+
provider: "anthropic",
|
|
74404
|
+
model_id: "claude-haiku-4-5",
|
|
74405
|
+
model_id_status: "confirmed",
|
|
74406
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
74407
|
+
model_family: "Haiku-class",
|
|
74408
|
+
lumic_model: "claude-haiku-4-5",
|
|
74409
|
+
params: {
|
|
74410
|
+
temperature: null,
|
|
74411
|
+
max_output_tokens: 64000,
|
|
74412
|
+
supports_temperature: true,
|
|
74413
|
+
supports_json_schema: true,
|
|
74414
|
+
supports_tools: true,
|
|
74415
|
+
supports_cache_control: true,
|
|
74416
|
+
supports_vision: true,
|
|
74417
|
+
supports_tool_choice: true,
|
|
74418
|
+
context_window: 200000
|
|
74419
|
+
},
|
|
74420
|
+
price_per_mtok: {
|
|
74421
|
+
input: 1,
|
|
74422
|
+
output: 5,
|
|
74423
|
+
as_of: "2026-09-10",
|
|
74424
|
+
source: "docs/llm-provider-migration.md#3 Haiku-class anchor"
|
|
74425
|
+
}
|
|
74426
|
+
}
|
|
74427
|
+
]
|
|
74068
74428
|
}
|
|
74069
74429
|
};
|
|
74070
74430
|
var open_items = [
|
|
@@ -74554,6 +74914,10 @@ function createDirectTransport(config) {
|
|
|
74554
74914
|
tool_calls: Array.isArray(result.tool_calls)
|
|
74555
74915
|
? result.tool_calls
|
|
74556
74916
|
: undefined,
|
|
74917
|
+
// The incumbent client reports the model it resolved; unreported stays null.
|
|
74918
|
+
servedModel: typeof result.usage?.model === "string" && result.usage.model !== ""
|
|
74919
|
+
? result.usage.model
|
|
74920
|
+
: null,
|
|
74557
74921
|
};
|
|
74558
74922
|
},
|
|
74559
74923
|
};
|
|
@@ -74561,9 +74925,9 @@ function createDirectTransport(config) {
|
|
|
74561
74925
|
/**
|
|
74562
74926
|
* Normalise the incumbent client's usage shape.
|
|
74563
74927
|
*
|
|
74564
|
-
*
|
|
74565
|
-
*
|
|
74566
|
-
* budget accounting the spend controls depend on.
|
|
74928
|
+
* A missing count is `null` rather than zero or an estimate, for the same
|
|
74929
|
+
* reason as on the gateway path: an invented token count — zero included —
|
|
74930
|
+
* flows straight into the budget accounting the spend controls depend on.
|
|
74567
74931
|
*
|
|
74568
74932
|
* @param usage The incumbent client's usage, if any.
|
|
74569
74933
|
* @param request The request it answers.
|
|
@@ -74571,13 +74935,13 @@ function createDirectTransport(config) {
|
|
|
74571
74935
|
*/
|
|
74572
74936
|
function readUsage$1(usage, request) {
|
|
74573
74937
|
return {
|
|
74574
|
-
prompt_tokens: usage?.prompt_tokens ??
|
|
74575
|
-
completion_tokens: usage?.completion_tokens ??
|
|
74938
|
+
prompt_tokens: usage?.prompt_tokens ?? null,
|
|
74939
|
+
completion_tokens: usage?.completion_tokens ?? null,
|
|
74576
74940
|
reasoning_tokens: usage?.reasoning_tokens,
|
|
74577
74941
|
cached_tokens: usage?.cached_tokens,
|
|
74578
74942
|
provider: usage?.provider ?? request.route.providerName,
|
|
74579
74943
|
model: usage?.model ?? request.route.modelId,
|
|
74580
|
-
cost: usage?.cost ??
|
|
74944
|
+
cost: usage?.cost ?? null,
|
|
74581
74945
|
};
|
|
74582
74946
|
}
|
|
74583
74947
|
/**
|
|
@@ -74630,10 +74994,12 @@ async function resolveDefaultDirectCaller() {
|
|
|
74630
74994
|
* never in application code (PD-5), which is what makes a model swap a config
|
|
74631
74995
|
* change rather than a deploy.
|
|
74632
74996
|
*
|
|
74633
|
-
* The client
|
|
74634
|
-
*
|
|
74635
|
-
*
|
|
74636
|
-
*
|
|
74997
|
+
* The client's chain is the single fallback owner: every leg is addressed to the
|
|
74998
|
+
* gateway by its own model name, so the gateway serves one deployment per leg
|
|
74999
|
+
* and needs no fallback of its own. A proxy-side fallback inside a leg would
|
|
75000
|
+
* spend the leg's budget on a model the chain did not choose and report the
|
|
75001
|
+
* answer as the leg's; the served model is read from the response body so such
|
|
75002
|
+
* a substitution stays visible while any remains configured.
|
|
74637
75003
|
*
|
|
74638
75004
|
* The gateway key is read from the environment by NAME at call time and never
|
|
74639
75005
|
* stored, logged, or included in an error (PD-2). Reading it per call rather
|
|
@@ -74645,6 +75011,10 @@ async function resolveDefaultDirectCaller() {
|
|
|
74645
75011
|
const RETRYABLE_STATUSES = new Set([408, 409, 425, 429, 500, 502, 503, 504]);
|
|
74646
75012
|
/** Maximum characters of an error body echoed into a message. */
|
|
74647
75013
|
const ERROR_BODY_EXCERPT = 400;
|
|
75014
|
+
/** Response header in which the LiteLLM proxy reports the call's cost in USD. */
|
|
75015
|
+
const RESPONSE_COST_HEADER = "x-litellm-response-cost";
|
|
75016
|
+
/** Response header naming the proxy deployment that served the call. */
|
|
75017
|
+
const DEPLOYMENT_ID_HEADER = "x-litellm-model-id";
|
|
74648
75018
|
/**
|
|
74649
75019
|
* Thrown when the gateway itself is unreachable, as opposed to a provider
|
|
74650
75020
|
* behind it failing.
|
|
@@ -74701,31 +75071,67 @@ function readGatewayKey(envVar) {
|
|
|
74701
75071
|
/**
|
|
74702
75072
|
* Extract usage from a chat-completion response.
|
|
74703
75073
|
*
|
|
74704
|
-
*
|
|
74705
|
-
*
|
|
74706
|
-
*
|
|
74707
|
-
*
|
|
75074
|
+
* A count the provider did not report is `null`, never zero and never an
|
|
75075
|
+
* estimate. A fabricated count — zero included — would flow straight into the
|
|
75076
|
+
* budget accounting the spend controls are built on, where a zero reads as a
|
|
75077
|
+
* free call and can never trip a limit.
|
|
75078
|
+
*
|
|
75079
|
+
* Cost is read from the body's `usage.response_cost` or, failing that, from the
|
|
75080
|
+
* proxy's cost header, which is where the LiteLLM proxy reports it.
|
|
74708
75081
|
*
|
|
74709
75082
|
* @param payload The parsed response body.
|
|
74710
75083
|
* @param request The request it answers.
|
|
75084
|
+
* @param headers The response headers.
|
|
74711
75085
|
* @returns The usage record.
|
|
74712
75086
|
*/
|
|
74713
|
-
function readUsage(payload, request) {
|
|
75087
|
+
function readUsage(payload, request, headers) {
|
|
74714
75088
|
const usage = (payload.usage ?? {});
|
|
74715
75089
|
const details = (usage.prompt_tokens_details ?? {});
|
|
74716
75090
|
const cached = details.cached_tokens;
|
|
74717
75091
|
const reasoningDetails = (usage.completion_tokens_details ?? {});
|
|
74718
75092
|
const reasoning = reasoningDetails.reasoning_tokens;
|
|
74719
75093
|
return {
|
|
74720
|
-
prompt_tokens:
|
|
74721
|
-
completion_tokens:
|
|
75094
|
+
prompt_tokens: finiteOrNull(usage.prompt_tokens),
|
|
75095
|
+
completion_tokens: finiteOrNull(usage.completion_tokens),
|
|
74722
75096
|
reasoning_tokens: typeof reasoning === "number" ? reasoning : undefined,
|
|
74723
75097
|
cached_tokens: typeof cached === "number" ? cached : undefined,
|
|
74724
75098
|
provider: request.route.providerName,
|
|
74725
75099
|
model: request.route.modelId,
|
|
74726
|
-
cost:
|
|
75100
|
+
cost: finiteOrNull(usage.response_cost) ?? headerNumber(headers, RESPONSE_COST_HEADER),
|
|
74727
75101
|
};
|
|
74728
75102
|
}
|
|
75103
|
+
/**
|
|
75104
|
+
* A reported number, or null when it was not reported as a finite number.
|
|
75105
|
+
*
|
|
75106
|
+
* @param value The raw value.
|
|
75107
|
+
* @returns The number, or null.
|
|
75108
|
+
*/
|
|
75109
|
+
function finiteOrNull(value) {
|
|
75110
|
+
return typeof value === "number" && Number.isFinite(value) ? value : null;
|
|
75111
|
+
}
|
|
75112
|
+
/**
|
|
75113
|
+
* A numeric response header, or null when absent or not a finite number.
|
|
75114
|
+
*
|
|
75115
|
+
* @param headers The response headers, if the transport exposes them.
|
|
75116
|
+
* @param name The header name.
|
|
75117
|
+
* @returns The number, or null.
|
|
75118
|
+
*/
|
|
75119
|
+
function headerNumber(headers, name) {
|
|
75120
|
+
const raw = headers?.get(name);
|
|
75121
|
+
if (raw === undefined || raw === null || raw.trim() === "") {
|
|
75122
|
+
return null;
|
|
75123
|
+
}
|
|
75124
|
+
return finiteOrNull(Number(raw));
|
|
75125
|
+
}
|
|
75126
|
+
/**
|
|
75127
|
+
* A non-empty string, or null.
|
|
75128
|
+
*
|
|
75129
|
+
* @param value The raw value.
|
|
75130
|
+
* @returns The string, or null.
|
|
75131
|
+
*/
|
|
75132
|
+
function nonEmptyOrNull(value) {
|
|
75133
|
+
return typeof value === "string" && value.trim() !== "" ? value : null;
|
|
75134
|
+
}
|
|
74729
75135
|
/**
|
|
74730
75136
|
* Build the gateway transport.
|
|
74731
75137
|
*
|
|
@@ -74776,13 +75182,17 @@ function createGatewayTransport(config) {
|
|
|
74776
75182
|
// Usage is read before the content is interpreted. The provider billed for
|
|
74777
75183
|
// this answer whether or not it parses, and a parse failure that dropped
|
|
74778
75184
|
// the count would report the attempt as free.
|
|
74779
|
-
const usage = readUsage(payload, request);
|
|
75185
|
+
const usage = readUsage(payload, request, response.headers);
|
|
74780
75186
|
return {
|
|
74781
75187
|
response: interpretContent(message?.content, request.responseFormat, usage),
|
|
74782
75188
|
usage,
|
|
74783
75189
|
tool_calls: Array.isArray(message?.tool_calls)
|
|
74784
75190
|
? message.tool_calls
|
|
74785
75191
|
: undefined,
|
|
75192
|
+
// The model the provider says answered, which a proxy-side fallback can
|
|
75193
|
+
// make differ from the leg's route model; unreported stays null.
|
|
75194
|
+
servedModel: nonEmptyOrNull(payload.model),
|
|
75195
|
+
servedDeploymentId: nonEmptyOrNull(response.headers?.get(DEPLOYMENT_ID_HEADER)),
|
|
74786
75196
|
};
|
|
74787
75197
|
},
|
|
74788
75198
|
};
|
|
@@ -74851,7 +75261,10 @@ function interpretContent(content, responseFormat, usage) {
|
|
|
74851
75261
|
* table, per-provider parameter normalisation, a hard per-leg timeout, a
|
|
74852
75262
|
* per-route circuit breaker, an ordered fallback chain ending at the closed
|
|
74853
75263
|
* incumbent, and — where the caller supplies a validator — one schema-feedback
|
|
74854
|
-
* retry ahead of the chain.
|
|
75264
|
+
* retry ahead of the chain. The caller's `timeoutMs` is ONE deadline for the
|
|
75265
|
+
* whole call: each leg runs for its route budget or for what remains of that
|
|
75266
|
+
* deadline, whichever is shorter, so the chain is the single fallback owner and
|
|
75267
|
+
* a slow primary cannot spend the time its fallbacks need. None of them is optional, because a control that a
|
|
74855
75268
|
* caller can switch off is a control that will be off on the call that needed
|
|
74856
75269
|
* it.
|
|
74857
75270
|
*
|
|
@@ -74987,10 +75400,14 @@ function prepareLegs(chain, options, responseFormat, transport) {
|
|
|
74987
75400
|
* @throws {SchemaRetryExhaustedError} When a validated payload failed twice.
|
|
74988
75401
|
*/
|
|
74989
75402
|
async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
74990
|
-
|
|
74991
|
-
|
|
74992
|
-
|
|
74993
|
-
});
|
|
75403
|
+
// Legs keep their route budgets; the caller's timeout is the deadline they
|
|
75404
|
+
// share, fixed once here so a validation retry and the degraded direct path
|
|
75405
|
+
// spend what is left of it rather than starting a fresh one.
|
|
75406
|
+
const chain = resolveChain(options.alias, { isolated: options.isolated });
|
|
75407
|
+
const clock = config.now ?? Date.now;
|
|
75408
|
+
const deadlineAtMs = typeof options.timeoutMs === "number" && Number.isFinite(options.timeoutMs)
|
|
75409
|
+
? clock() + options.timeoutMs
|
|
75410
|
+
: undefined;
|
|
74994
75411
|
if (chain.routes.length === 0) {
|
|
74995
75412
|
// Nothing is servable. The exclusions say why each leg was unavailable,
|
|
74996
75413
|
// which is the difference between an operator reading config and an
|
|
@@ -75033,6 +75450,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75033
75450
|
breakers,
|
|
75034
75451
|
correlationId: options.correlationId,
|
|
75035
75452
|
callerSignal: options.signal,
|
|
75453
|
+
deadlineAtMs,
|
|
75036
75454
|
now: config.now,
|
|
75037
75455
|
onAttempt: (record) => attemptLog.push(record),
|
|
75038
75456
|
});
|
|
@@ -75067,6 +75485,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75067
75485
|
breakers,
|
|
75068
75486
|
correlationId: options.correlationId,
|
|
75069
75487
|
callerSignal: options.signal,
|
|
75488
|
+
deadlineAtMs,
|
|
75070
75489
|
now: config.now,
|
|
75071
75490
|
onAttempt: (record) => attemptLog.push(record),
|
|
75072
75491
|
});
|
|
@@ -75079,6 +75498,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75079
75498
|
usage: outcome.response.usage,
|
|
75080
75499
|
tool_calls: outcome.response.tool_calls,
|
|
75081
75500
|
servedBy: outcome.servedBy,
|
|
75501
|
+
servedModel: outcome.response.servedModel ?? null,
|
|
75082
75502
|
attempts: attemptLog,
|
|
75083
75503
|
degraded: outcome.degraded,
|
|
75084
75504
|
totalUsage: outcome.totalUsage,
|
|
@@ -75112,6 +75532,7 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75112
75532
|
usage: validated.response.usage,
|
|
75113
75533
|
tool_calls: validated.response.tool_calls,
|
|
75114
75534
|
servedBy: routing.servedBy,
|
|
75535
|
+
servedModel: validated.response.servedModel ?? null,
|
|
75115
75536
|
attempts: attemptLog,
|
|
75116
75537
|
degraded: routing.degraded,
|
|
75117
75538
|
totalUsage: validated.totalUsage,
|
|
@@ -80598,6 +81019,8 @@ exports.NewsError = NewsError;
|
|
|
80598
81019
|
exports.NoServableRouteError = NoServableRouteError;
|
|
80599
81020
|
exports.OptionStrategyError = OptionStrategyError;
|
|
80600
81021
|
exports.OptionsDataError = OptionsDataError;
|
|
81022
|
+
exports.PENDING_CANCEL_ERROR_CODE = PENDING_CANCEL_ERROR_CODE;
|
|
81023
|
+
exports.PendingCancelError = PendingCancelError;
|
|
80601
81024
|
exports.QuoteError = QuoteError;
|
|
80602
81025
|
exports.RISK_FREE_RATE_TTL_MS = RISK_FREE_RATE_TTL_MS;
|
|
80603
81026
|
exports.RateGuardTimeoutError = RateGuardTimeoutError;
|
|
@@ -80610,6 +81033,7 @@ exports.StreamTruncatedError = StreamTruncatedError;
|
|
|
80610
81033
|
exports.TRADING_API = TRADING_API;
|
|
80611
81034
|
exports.TimeoutError = TimeoutError;
|
|
80612
81035
|
exports.TokenBucketRateLimiter = TokenBucketRateLimiter;
|
|
81036
|
+
exports.ToolChoiceIgnoredError = ToolChoiceIgnoredError;
|
|
80613
81037
|
exports.TradeError = TradeError;
|
|
80614
81038
|
exports.TrailingStopValidationError = TrailingStopValidationError;
|
|
80615
81039
|
exports.USDC_PAIRS = USDC_PAIRS;
|
|
@@ -80628,6 +81052,7 @@ exports.adptc = adptc;
|
|
|
80628
81052
|
exports.alpaca = alpaca;
|
|
80629
81053
|
exports.analyzeBars = analyzeBars;
|
|
80630
81054
|
exports.approximateImpliedVolatility = approximateImpliedVolatility;
|
|
81055
|
+
exports.assertToolChoiceHonoured = assertToolChoiceHonoured;
|
|
80631
81056
|
exports.atr = atrNs;
|
|
80632
81057
|
exports.availableStatistic = availableStatistic;
|
|
80633
81058
|
exports.bracketOrders = bracketOrders;
|
|
@@ -80814,8 +81239,10 @@ exports.isOrderFillable = isOrderFillable;
|
|
|
80814
81239
|
exports.isOrderFilled = isOrderFilled;
|
|
80815
81240
|
exports.isOrderOpen = isOrderOpen;
|
|
80816
81241
|
exports.isOrderTerminalStatus = isOrderTerminal$1;
|
|
81242
|
+
exports.isPendingCancelRejection = isPendingCancelRejection;
|
|
80817
81243
|
exports.isSupportedCryptoPair = isSupportedCryptoPair;
|
|
80818
81244
|
exports.isTransientNetworkError = isTransientNetworkError;
|
|
81245
|
+
exports.legBudgetMs = legBudgetMs;
|
|
80819
81246
|
exports.legacyApi = index$1;
|
|
80820
81247
|
exports.limitBuyWithTakeProfit = limitBuyWithTakeProfit;
|
|
80821
81248
|
exports.limitsFor = limitsFor;
|