@adaptic/utils 0.0.1032 → 0.0.1034

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -62508,13 +62508,18 @@ const DEFAULT_PAGINATION_DELAY_MS = 300;
62508
62508
  */
62509
62509
  const MAX_ORDERS_PER_REQUEST = 500;
62510
62510
  /**
62511
- * Order statuses that are considered "open"
62511
+ * Order statuses that are considered "open".
62512
+ *
62513
+ * `held` is included because a held conditional leg (a bracket's stop-loss,
62514
+ * say) is a working order resting at the broker, returned by Alpaca's own
62515
+ * `status=open` listing and cancelable like any other open order.
62512
62516
  */
62513
62517
  const OPEN_ORDER_STATUSES = [
62514
62518
  "new",
62515
62519
  "accepted",
62516
62520
  "pending_new",
62517
62521
  "accepted_for_bidding",
62522
+ "held",
62518
62523
  "partially_filled",
62519
62524
  ];
62520
62525
  /**
@@ -62522,13 +62527,15 @@ const OPEN_ORDER_STATUSES = [
62522
62527
  */
62523
62528
  const FILLED_ORDER_STATUSES = ["filled"];
62524
62529
  /**
62525
- * Order statuses that can still potentially be filled
62530
+ * Order statuses that can still potentially be filled. A `held` leg fills once
62531
+ * its parent fills or its trigger is met.
62526
62532
  */
62527
62533
  const FILLABLE_ORDER_STATUSES = [
62528
62534
  "new",
62529
62535
  "accepted",
62530
62536
  "pending_new",
62531
62537
  "accepted_for_bidding",
62538
+ "held",
62532
62539
  "partially_filled",
62533
62540
  ];
62534
62541
  /**
@@ -72336,11 +72343,35 @@ class CircuitBreakerRegistry {
72336
72343
  * Register that an attempt is starting, so half-open probes stay bounded.
72337
72344
  *
72338
72345
  * @param routeKey The route's stable key.
72339
- * @returns void
72346
+ * @returns Whether the attempt took a half-open probe slot. A caller holding
72347
+ * one must end the attempt with {@link onSuccess}, {@link onFailure} or
72348
+ * {@link onAttemptAbandoned}, or the slot is never returned.
72340
72349
  */
72341
72350
  onAttemptStart(routeKey) {
72342
72351
  if (this.stateOf(routeKey) === "half-open") {
72343
72352
  this.recordFor(routeKey).probesInFlight += 1;
72353
+ return true;
72354
+ }
72355
+ return false;
72356
+ }
72357
+ /**
72358
+ * Return a half-open probe slot whose attempt ended without a verdict.
72359
+ *
72360
+ * A probe that never tested the provider — refused by the client's own
72361
+ * pacing guard, cancelled by its caller, or found to be the wrong leg for the
72362
+ * request — says nothing about whether the route has recovered, so neither a
72363
+ * success nor a failure is recorded. The slot must still come back. Without
72364
+ * it the half-open route admits no further probe, no probe can ever close or
72365
+ * re-open the breaker, and the route stays excluded for the life of the
72366
+ * process while its traffic is quietly served by the next leg.
72367
+ *
72368
+ * @param routeKey The route's stable key.
72369
+ * @returns void
72370
+ */
72371
+ onAttemptAbandoned(routeKey) {
72372
+ const record = this.records.get(routeKey);
72373
+ if (record !== undefined && record.probesInFlight > 0) {
72374
+ record.probesInFlight -= 1;
72344
72375
  }
72345
72376
  }
72346
72377
  /**
@@ -72611,11 +72642,13 @@ var defaults$1 = {
72611
72642
  var providers$1 = {
72612
72643
  anthropic: {
72613
72644
  basis: "conservative-default",
72645
+ scope: "model",
72646
+ scope_source: "https://platform.claude.com/docs/en/api/rate-limits",
72614
72647
  requests_per_minute: 120,
72615
72648
  max_concurrent: 12,
72616
72649
  acquire_timeout_ms: 15000,
72617
72650
  source: null,
72618
- note: "Raised 2026-09-15 after the 4-concurrent ceiling was measured starving the live equity decision path: the alias chain reported 'exhausted its fallback chain' with every leg skipped by this guard (deepinfra primary+secondary and the anthropic incumbent), 173 of 473 signal-coordination calls failed (36.6%), and decisions were lost outright. Corroborating evidence at the time: ZERO 429s observed on any provider, the engine's own global fan-out gate permits 100 concurrent with 17 active, and 877 signal-analysis calls had been admitted to that gate. Concurrency and RPM are raised TOGETHER because they bind in series - lifting max_concurrent alone would only move the bottleneck to the token bucket. STILL conservative-default, NOT published: no provider console was read for these numbers, so they remain a deliberate under-estimate of an unknown ceiling. Transcribe the real tier limits (W3-07) and set basis to published."
72651
+ note: "The unit is per model: the scope_source states 'Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously', measured as requests, input tokens and output tokens per minute for each model class, with no concurrency ceiling. Each model therefore gets its own guard, so traffic on one model class (an Opus incumbent serving background aliases) cannot refuse calls to another (the Haiku incumbent of the hot-path alias). The NUMBERS stay conservative-default: the organisation's usage tier sets the real ceilings and is read only from the Claude Console rate-limits page, which has not been transcribed. The lowest standard tier listed at the source allows 1,000 requests per minute per model; organisations with limited history can start on an evaluation tier below that, which is why these values are held well under it. Transcribe the tier's limits (W3-07) and set basis to published."
72619
72652
  },
72620
72653
  openai: {
72621
72654
  basis: "conservative-default",
@@ -72634,12 +72667,14 @@ var providers$1 = {
72634
72667
  note: "Raised 2026-09-15 after the 4-concurrent ceiling was measured starving the live equity decision path: the alias chain reported 'exhausted its fallback chain' with every leg skipped by this guard (deepinfra primary+secondary and the anthropic incumbent), 173 of 473 signal-coordination calls failed (36.6%), and decisions were lost outright. Corroborating evidence at the time: ZERO 429s observed on any provider, the engine's own global fan-out gate permits 100 concurrent with 17 active, and 877 signal-analysis calls had been admitted to that gate. Concurrency and RPM are raised TOGETHER because they bind in series - lifting max_concurrent alone would only move the bottleneck to the token bucket. STILL conservative-default, NOT published: no provider console was read for these numbers, so they remain a deliberate under-estimate of an unknown ceiling. Transcribe the real tier limits (W3-07) and set basis to published."
72635
72668
  },
72636
72669
  deepinfra: {
72637
- basis: "conservative-default",
72670
+ basis: "published",
72671
+ scope: "model",
72638
72672
  requests_per_minute: 240,
72639
- max_concurrent: 24,
72673
+ requests_per_minute_basis: "conservative-default",
72674
+ max_concurrent: 200,
72640
72675
  acquire_timeout_ms: 15000,
72641
- source: null,
72642
- note: "Raised 2026-09-15 after the 4-concurrent ceiling was measured starving the live equity decision path: the alias chain reported 'exhausted its fallback chain' with every leg skipped by this guard (deepinfra primary+secondary and the anthropic incumbent), 173 of 473 signal-coordination calls failed (36.6%), and decisions were lost outright. Corroborating evidence at the time: ZERO 429s observed on any provider, the engine's own global fan-out gate permits 100 concurrent with 17 active, and 877 signal-analysis calls had been admitted to that gate. Concurrency and RPM are raised TOGETHER because they bind in series - lifting max_concurrent alone would only move the bottleneck to the token bucket. STILL conservative-default, NOT published: no provider console was read for these numbers, so they remain a deliberate under-estimate of an unknown ceiling. Transcribe the real tier limits (W3-07) and set basis to published."
72676
+ source: "https://docs.deepinfra.com/account/rate-limits",
72677
+ note: "Transcribed 2026-09-23 from the source, which states 'Every account has a default limit of 200 concurrent requests per model', that two models queried simultaneously allow 400 in total (200 per model), and that 'The rate limit is on concurrent requests, not per-minute volume.' Concurrency is therefore the bound DeepInfra enforces, and it is enforced per MODEL, so each model gets its own guard at 200. One guard shared by the whole account enforced a ceiling DeepInfra does not impose, and because an alias's primary and secondary are both served from this account, it refused the secondary exactly when the primary's queue was full. DeepInfra publishes no per-minute ceiling, so requests_per_minute is the client's own pacing backstop (requests_per_minute_basis: conservative-default), keyed per model like the bound DeepInfra does enforce. Exceeding the ceiling returns HTTP 429, and a very busy model can return 429 below it. The ceiling belongs to the account and this guard to one process, so every process calling the same model through the same account shares the 200."
72643
72678
  },
72644
72679
  fireworks: {
72645
72680
  basis: "conservative-default",
@@ -72689,7 +72724,10 @@ var limitsConfig = {
72689
72724
  * provider's circuit breaker, fail over to a more expensive leg, and keep doing
72690
72725
  * so — converting a self-inflicted pacing problem into a permanent routing
72691
72726
  * change nobody chose. Pacing at the client is what keeps the breaker measuring
72692
- * the provider rather than measuring us.
72727
+ * the provider rather than measuring us. The same reasoning bounds the guard
72728
+ * from the other side: a client held far BELOW the provider's ceiling refuses
72729
+ * calls the provider would have served, and the chain answers those refusals by
72730
+ * failing over — the same unchosen routing change, arrived at by under-driving.
72693
72731
  *
72694
72732
  * Two distinct bounds are applied because they fail differently. The rate bound
72695
72733
  * (requests per minute) protects the provider's published ceiling. The
@@ -72698,8 +72736,18 @@ var limitsConfig = {
72698
72736
  * every one of them blows its latency budget and the fan-out produces a hundred
72699
72737
  * timeouts instead of a queue.
72700
72738
  *
72701
- * Limits live in `provider-limits.json`, not here. A rate limit discovered
72702
- * during an incident should be correctable by config, not by a release.
72739
+ * Each guard is keyed by the unit its provider enforces limits in. A provider
72740
+ * that publishes its ceilings per model gets one independent guard per model.
72741
+ * Sharing one guard across its models would enforce a ceiling the provider does
72742
+ * not impose, and — when a chain's primary and secondary are served by the same
72743
+ * provider — would refuse the secondary at exactly the moment the primary's
72744
+ * queue is full, so the fallback that exists for that moment is never reached.
72745
+ *
72746
+ * Limits live in `provider-limits.json` rather than in code, each beside the
72747
+ * source it was transcribed from, so a published ceiling and a conservative
72748
+ * guess can never be mistaken for one another in review. The file is bundled
72749
+ * at build time: changing a limit is a release of this package, not a runtime
72750
+ * switch.
72703
72751
  *
72704
72752
  * @module llm/rate-guard
72705
72753
  */
@@ -72739,64 +72787,141 @@ class RateGuardTimeoutError extends Error {
72739
72787
  provider;
72740
72788
  /** Which of the two bounds the caller waited on. */
72741
72789
  bound;
72790
+ /** The model whose guard refused the call, when the provider's limits apply per model. */
72791
+ modelId;
72792
+ /** Whether the caller stopped waiting before the guard's own wait budget ran out. */
72793
+ abandoned;
72742
72794
  /**
72743
72795
  * @param provider The provider.
72744
72796
  * @param bound Which bound was binding.
72745
- * @param waitedMs How long the caller waited.
72797
+ * @param waitedMs How long the caller was prepared to wait.
72798
+ * @param detail The model, and whether the caller left before the budget ran out.
72746
72799
  */
72747
- constructor(provider, bound, waitedMs) {
72748
- super(`client-side ${bound} guard for provider "${provider}" did not admit the call within ${waitedMs} ms. ` +
72800
+ constructor(provider, bound, waitedMs, detail = {}) {
72801
+ const guard = detail.modelId === undefined
72802
+ ? `client-side ${bound} guard for provider "${provider}"`
72803
+ : `client-side ${bound} guard for provider "${provider}", model "${detail.modelId}",`;
72804
+ const outcome = detail.abandoned === true
72805
+ ? `was left by its caller before it could admit the call (wait budget ${waitedMs} ms)`
72806
+ : `did not admit the call within ${waitedMs} ms`;
72807
+ super(`${guard} ${outcome}. ` +
72749
72808
  "The provider was never contacted, so this says nothing about its health.");
72750
72809
  this.name = "RateGuardTimeoutError";
72751
72810
  this.provider = provider;
72752
72811
  this.bound = bound;
72812
+ this.modelId = detail.modelId;
72813
+ this.abandoned = detail.abandoned === true;
72753
72814
  }
72754
72815
  }
72816
+ /**
72817
+ * The guard a call is held by.
72818
+ *
72819
+ * A call to a per-model provider that names no model shares one provider-wide
72820
+ * guard held at the per-model ceiling. That is never looser than the limit of
72821
+ * any single model it might reach, so the fallback errs toward pacing.
72822
+ *
72823
+ * @param provider The provider key.
72824
+ * @param modelId The model the call is addressed to, if known.
72825
+ * @returns The guard's identity.
72826
+ */
72827
+ function guardIdentity(provider, modelId) {
72828
+ const perModel = limitsFor(provider).scope === "model" && modelId !== undefined && modelId.length > 0;
72829
+ return perModel
72830
+ ? { key: `${provider}/${modelId}`, provider, modelId }
72831
+ : { key: provider, provider, modelId: undefined };
72832
+ }
72755
72833
  /**
72756
72834
  * A counting semaphore bounding simultaneous in-flight calls.
72757
72835
  *
72758
- * Written here rather than pulled from a dependency because it is fifteen lines
72759
- * and because the waiting behaviour matters: a waiter that times out must be
72760
- * removed from the queue, or a burst of abandoned callers permanently consumes
72761
- * the permits that later callers need.
72836
+ * Written here rather than pulled from a dependency because the waiting
72837
+ * behaviour is the point. A waiter that times out must be removed from the
72838
+ * queue, or a burst of abandoned callers permanently consumes the permits that
72839
+ * later callers need. And a waiter whose caller has stopped waiting must leave
72840
+ * at once: left queued, it holds its caller until the wait budget expires and
72841
+ * is then handed a permit it can only waste.
72762
72842
  */
72763
72843
  class ConcurrencyGate {
72764
72844
  inFlight = 0;
72765
72845
  waiters = [];
72766
72846
  limit;
72767
- provider;
72847
+ identity;
72768
72848
  /**
72769
- * @param provider The provider this gate guards.
72849
+ * @param identity The guard this gate implements.
72770
72850
  * @param limit Maximum simultaneous in-flight calls.
72771
72851
  */
72772
- constructor(provider, limit) {
72773
- this.provider = provider;
72852
+ constructor(identity, limit) {
72853
+ this.identity = identity;
72774
72854
  this.limit = limit;
72775
72855
  }
72776
72856
  /**
72777
72857
  * Wait for a permit.
72778
72858
  *
72779
72859
  * @param timeoutMs How long the caller is willing to queue.
72860
+ * @param signal The caller's cancellation; firing it takes the caller out of the queue.
72780
72861
  * @returns A release function the caller must invoke exactly once.
72862
+ * @throws {RateGuardTimeoutError} When no permit was granted in time, or the caller stopped waiting.
72781
72863
  */
72782
- async acquire(timeoutMs) {
72864
+ async acquire(timeoutMs, signal) {
72865
+ if (signal?.aborted === true) {
72866
+ // Nobody is waiting for this answer. Taking a permit for it would spend
72867
+ // capacity a live caller needs on a call that can only be torn down.
72868
+ throw this.refusal(timeoutMs, true);
72869
+ }
72783
72870
  if (this.inFlight < this.limit) {
72784
72871
  this.inFlight += 1;
72785
72872
  return () => this.release();
72786
72873
  }
72787
72874
  await new Promise((resolve, reject) => {
72788
- const timer = setTimeout(() => {
72789
- const index = this.waiters.findIndex((waiter) => waiter.timer === timer);
72790
- if (index !== -1) {
72791
- this.waiters.splice(index, 1);
72875
+ /**
72876
+ * Take this waiter out of the queue and refuse it. A waiter that `release`
72877
+ * has already admitted is no longer queued; it now holds a permit, which
72878
+ * its call returns, so there is nothing to undo here.
72879
+ *
72880
+ * @param abandoned Whether the caller left before the wait budget ran out.
72881
+ * @returns void
72882
+ */
72883
+ const leave = (abandoned) => {
72884
+ const index = this.waiters.indexOf(waiter);
72885
+ if (index === -1) {
72886
+ return;
72792
72887
  }
72793
- reject(new RateGuardTimeoutError(this.provider, "concurrency", timeoutMs));
72888
+ this.waiters.splice(index, 1);
72889
+ clearTimeout(timer);
72890
+ signal?.removeEventListener("abort", onAbort);
72891
+ reject(this.refusal(timeoutMs, abandoned));
72892
+ };
72893
+ const onAbort = () => {
72894
+ leave(true);
72895
+ };
72896
+ const timer = setTimeout(() => {
72897
+ leave(false);
72794
72898
  }, timeoutMs);
72795
- this.waiters.push({ resolve, reject, timer });
72899
+ const waiter = {
72900
+ admit: () => {
72901
+ clearTimeout(timer);
72902
+ signal?.removeEventListener("abort", onAbort);
72903
+ resolve();
72904
+ },
72905
+ };
72906
+ this.waiters.push(waiter);
72907
+ signal?.addEventListener("abort", onAbort, { once: true });
72796
72908
  });
72797
72909
  this.inFlight += 1;
72798
72910
  return () => this.release();
72799
72911
  }
72912
+ /**
72913
+ * Build the refusal for a caller this gate did not admit.
72914
+ *
72915
+ * @param timeoutMs The wait budget the caller had.
72916
+ * @param abandoned Whether the caller left before the budget ran out.
72917
+ * @returns The error to raise.
72918
+ */
72919
+ refusal(timeoutMs, abandoned) {
72920
+ return new RateGuardTimeoutError(this.identity.provider, "concurrency", timeoutMs, {
72921
+ modelId: this.identity.modelId,
72922
+ abandoned,
72923
+ });
72924
+ }
72800
72925
  /**
72801
72926
  * Return a permit and admit the next waiter.
72802
72927
  *
@@ -72806,8 +72931,7 @@ class ConcurrencyGate {
72806
72931
  this.inFlight -= 1;
72807
72932
  const next = this.waiters.shift();
72808
72933
  if (next !== undefined) {
72809
- clearTimeout(next.timer);
72810
- next.resolve();
72934
+ next.admit();
72811
72935
  }
72812
72936
  }
72813
72937
  /**
@@ -72823,44 +72947,47 @@ class ConcurrencyGate {
72823
72947
  return this.waiters.length;
72824
72948
  }
72825
72949
  }
72826
- /** Per-provider guards, created on first use and shared process-wide. */
72950
+ /** Guards, created on first use and shared process-wide, keyed by {@link GuardIdentity.key}. */
72827
72951
  const rateLimiters = new Map();
72828
72952
  const concurrencyGates = new Map();
72953
+ const guardIdentities = new Map();
72829
72954
  /**
72830
- * The rate limiter for a provider.
72955
+ * The rate limiter for a guard.
72831
72956
  *
72832
72957
  * Shared process-wide rather than per-call-site, because the provider's ceiling
72833
72958
  * applies to the process as a whole. Per-call-site limiters would each stay
72834
72959
  * under the ceiling while their sum sailed past it.
72835
72960
  *
72836
- * @param provider The provider key.
72961
+ * @param identity The guard.
72837
72962
  * @returns Its limiter.
72838
72963
  */
72839
- function rateLimiterFor(provider) {
72840
- let limiter = rateLimiters.get(provider);
72964
+ function rateLimiterFor(identity) {
72965
+ let limiter = rateLimiters.get(identity.key);
72841
72966
  if (limiter === undefined) {
72842
- const limits = limitsFor(provider);
72967
+ const limits = limitsFor(identity.provider);
72843
72968
  limiter = new TokenBucketRateLimiter({
72844
72969
  maxTokens: limits.requests_per_minute,
72845
72970
  refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
72846
- label: `llm:${provider}`,
72971
+ label: `llm:${identity.key}`,
72847
72972
  timeoutMs: limits.acquire_timeout_ms,
72848
72973
  });
72849
- rateLimiters.set(provider, limiter);
72974
+ rateLimiters.set(identity.key, limiter);
72975
+ guardIdentities.set(identity.key, identity);
72850
72976
  }
72851
72977
  return limiter;
72852
72978
  }
72853
72979
  /**
72854
- * The concurrency gate for a provider.
72980
+ * The concurrency gate for a guard.
72855
72981
  *
72856
- * @param provider The provider key.
72982
+ * @param identity The guard.
72857
72983
  * @returns Its gate.
72858
72984
  */
72859
- function concurrencyGateFor(provider) {
72860
- let gate = concurrencyGates.get(provider);
72985
+ function concurrencyGateFor(identity) {
72986
+ let gate = concurrencyGates.get(identity.key);
72861
72987
  if (gate === undefined) {
72862
- gate = new ConcurrencyGate(provider, limitsFor(provider).max_concurrent);
72863
- concurrencyGates.set(provider, gate);
72988
+ gate = new ConcurrencyGate(identity, limitsFor(identity.provider).max_concurrent);
72989
+ concurrencyGates.set(identity.key, gate);
72990
+ guardIdentities.set(identity.key, identity);
72864
72991
  }
72865
72992
  return gate;
72866
72993
  }
@@ -72882,21 +73009,26 @@ function concurrencyGateFor(provider) {
72882
73009
  * @param provider The provider key.
72883
73010
  * @param call The work to run once admitted.
72884
73011
  * @param maxWaitMs Ceiling on queue time; the configured guard timeout applies when lower.
73012
+ * @param scope The model the call addresses, and the caller's cancellation.
72885
73013
  * @returns The call's result.
72886
- * @throws {RateGuardTimeoutError} When neither bound admitted the call in time.
73014
+ * @throws {RateGuardTimeoutError} When neither bound admitted the call in time,
73015
+ * or the caller stopped waiting first.
72887
73016
  */
72888
- async function withProviderGuards(provider, call, maxWaitMs) {
73017
+ async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
72889
73018
  const limits = limitsFor(provider);
73019
+ const identity = guardIdentity(provider, scope.modelId);
72890
73020
  const waitBudgetMs = maxWaitMs === undefined
72891
73021
  ? limits.acquire_timeout_ms
72892
73022
  : Math.min(maxWaitMs, limits.acquire_timeout_ms);
72893
73023
  try {
72894
- await rateLimiterFor(provider).acquire();
73024
+ await rateLimiterFor(identity).acquire();
72895
73025
  }
72896
73026
  catch {
72897
- throw new RateGuardTimeoutError(provider, "rate", waitBudgetMs);
73027
+ throw new RateGuardTimeoutError(provider, "rate", waitBudgetMs, {
73028
+ modelId: identity.modelId,
73029
+ });
72898
73030
  }
72899
- const release = await concurrencyGateFor(provider).acquire(waitBudgetMs);
73031
+ const release = await concurrencyGateFor(identity).acquire(waitBudgetMs, scope.signal);
72900
73032
  try {
72901
73033
  return await call();
72902
73034
  }
@@ -72910,16 +73042,20 @@ async function withProviderGuards(provider, call, maxWaitMs) {
72910
73042
  /**
72911
73043
  * Inspect the guards currently in use.
72912
73044
  *
72913
- * @returns A snapshot per provider that has been used, sorted by provider.
73045
+ * @returns A snapshot per guard that has been used, sorted by guard key.
72914
73046
  */
72915
73047
  function guardSnapshots() {
72916
- const providers = new Set([...rateLimiters.keys(), ...concurrencyGates.keys()]);
72917
- return [...providers].sort().map((provider) => {
72918
- const limits = limitsFor(provider);
72919
- const limiter = rateLimiters.get(provider);
72920
- const gate = concurrencyGates.get(provider);
73048
+ return [...guardIdentities.values()]
73049
+ .sort((a, b) => (a.key < b.key ? -1 : a.key > b.key ? 1 : 0))
73050
+ .map((identity) => {
73051
+ const limits = limitsFor(identity.provider);
73052
+ const limiter = rateLimiters.get(identity.key);
73053
+ const gate = concurrencyGates.get(identity.key);
72921
73054
  return {
72922
- provider,
73055
+ key: identity.key,
73056
+ provider: identity.provider,
73057
+ modelId: identity.modelId,
73058
+ scope: limits.scope ?? "provider",
72923
73059
  basis: limits.basis,
72924
73060
  requestsPerMinute: limits.requests_per_minute,
72925
73061
  maxConcurrent: limits.max_concurrent,
@@ -72944,6 +73080,105 @@ function resetProviderGuards() {
72944
73080
  }
72945
73081
  rateLimiters.clear();
72946
73082
  concurrencyGates.clear();
73083
+ guardIdentities.clear();
73084
+ }
73085
+
73086
+ /**
73087
+ * Interpretation of a model's answer to a structured (JSON) request.
73088
+ *
73089
+ * A JSON request is a promise about the answer's SHAPE, and providers keep it
73090
+ * in different ways. An OpenAI-compatible host given `json_object` constrains
73091
+ * its decoder, so its answer is bare JSON. A provider with no schema-less JSON
73092
+ * mode — Anthropic, reached through the gateway, supports structured output
73093
+ * only against a caller-supplied schema — receives nothing but the prompt's
73094
+ * instructions for a `json` request, and a model following them commonly
73095
+ * returns the object inside one markdown code fence. The fence is presentation,
73096
+ * not content: the object inside it is the answer the model gave.
73097
+ *
73098
+ * So exactly ONE enclosing fence is removed before parsing, and nothing else is
73099
+ * forgiven. Prose before or after the fence, two fenced blocks, a fence that
73100
+ * never closes (a truncated answer), and a fence declaring another language all
73101
+ * still fail. Each of those is an answer whose meaning a parser would have to
73102
+ * guess, and a guessed object is a decision made on data no model produced.
73103
+ *
73104
+ * Content that is not fenced is parsed exactly as it always was: JSON cannot
73105
+ * begin with a backtick, so every answer that parsed before this unwrapping
73106
+ * existed takes the same path and yields the same value.
73107
+ *
73108
+ * @module llm/structured-content
73109
+ */
73110
+ /**
73111
+ * One markdown fence enclosing the whole answer: an opening line of three
73112
+ * backticks, optionally labelled `json`, then the body, then three closing
73113
+ * backticks, with nothing but whitespace outside them. The body is anchored at
73114
+ * both ends, so an answer holding two fenced blocks captures the text between
73115
+ * them and fails to parse instead of yielding either block.
73116
+ */
73117
+ const SINGLE_ENCLOSING_JSON_FENCE = /^\s*```(?:json)?[ \t]*\r?\n([\s\S]*?)\r?\n?[ \t]*```\s*$/i;
73118
+ /**
73119
+ * Thrown when a provider answered a structured request with content that does
73120
+ * not parse.
73121
+ *
73122
+ * Carries the usage the provider billed for that answer. The tokens were spent
73123
+ * whether or not the content parsed, and a chain that dropped them would report
73124
+ * a failed attempt as free — understating spend by exactly the calls that went
73125
+ * wrong.
73126
+ */
73127
+ class LlmResponseFormatError extends Error {
73128
+ /** The format the caller asked for. */
73129
+ responseFormat;
73130
+ /** What the provider billed for the answer that did not parse. */
73131
+ usage;
73132
+ /** Whether the answer sat inside one enclosing fence that was removed before parsing. */
73133
+ fenced;
73134
+ /**
73135
+ * @param responseFormat The format the caller asked for.
73136
+ * @param usage What the provider billed for the answer.
73137
+ * @param fenced Whether one enclosing fence was removed before parsing.
73138
+ * @param cause The parser's own complaint.
73139
+ */
73140
+ constructor(responseFormat, usage, fenced, cause) {
73141
+ super(`LLM returned content that is not valid JSON for a ${responseFormat} request` +
73142
+ (fenced ? " (inside one enclosing markdown fence)" : "") +
73143
+ `: ${cause instanceof Error ? cause.message : String(cause)}`);
73144
+ this.name = "LlmResponseFormatError";
73145
+ this.responseFormat = responseFormat;
73146
+ this.usage = usage;
73147
+ this.fenced = fenced;
73148
+ }
73149
+ }
73150
+ /**
73151
+ * The body of the one markdown fence that encloses an answer, if exactly one does.
73152
+ *
73153
+ * @param text The model's answer.
73154
+ * @returns The fenced body, or null when the answer is not wholly one fenced block.
73155
+ */
73156
+ function unwrapSingleJsonFence(text) {
73157
+ const match = SINGLE_ENCLOSING_JSON_FENCE.exec(text);
73158
+ return match === null ? null : match[1];
73159
+ }
73160
+ /**
73161
+ * Parse a model's answer to a structured request.
73162
+ *
73163
+ * A JSON format that does not parse is an error, not an empty object. Returning
73164
+ * a default here would hand the caller a well-typed value that means nothing,
73165
+ * and the failure would surface much later as a decision made on absent data.
73166
+ *
73167
+ * @param content The raw content of the model's message.
73168
+ * @param responseFormat The structured format the caller asked for.
73169
+ * @param usage What the provider billed for this answer, carried on failure.
73170
+ * @returns The parsed value.
73171
+ * @throws {LlmResponseFormatError} When the content is not JSON, fenced or not.
73172
+ */
73173
+ function parseStructuredContent(content, responseFormat, usage) {
73174
+ const text = typeof content === "string" ? content : "";
73175
+ const fencedBody = unwrapSingleJsonFence(text);
73176
+ try {
73177
+ return JSON.parse(fencedBody ?? text);
73178
+ }
73179
+ catch (error) {
73180
+ throw new LlmResponseFormatError(typeof responseFormat === "string" ? responseFormat : "json_schema", usage, fencedBody !== null, error);
73181
+ }
72947
73182
  }
72948
73183
 
72949
73184
  /**
@@ -73080,7 +73315,9 @@ async function runLeg(leg, params, execution) {
73080
73315
  // The guards wrap the transport rather than the whole leg, so the per-leg
73081
73316
  // timeout above still bounds the total wait: a caller queued behind the
73082
73317
  // rate limiter is spending its budget just as surely as one waiting on the
73083
- // provider, and only one clock should govern both.
73318
+ // provider, and only one clock should govern both. The leg's own signal is
73319
+ // handed to the guard as well, so a leg whose budget or caller is gone
73320
+ // leaves the queue at once instead of holding its place in it.
73084
73321
  return await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
73085
73322
  route: leg.route,
73086
73323
  content: execution.content,
@@ -73090,7 +73327,7 @@ async function runLeg(leg, params, execution) {
73090
73327
  context: execution.context,
73091
73328
  signal: controller.signal,
73092
73329
  correlationId: execution.correlationId,
73093
- }), budgetMs);
73330
+ }), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
73094
73331
  }
73095
73332
  finally {
73096
73333
  clearTimeout(timer);
@@ -73202,7 +73439,7 @@ async function executeChain(alias, execution) {
73202
73439
  continue;
73203
73440
  }
73204
73441
  const startedAt = now();
73205
- execution.breakers.onAttemptStart(route.routeKey);
73442
+ const holdsProbe = execution.breakers.onAttemptStart(route.routeKey);
73206
73443
  try {
73207
73444
  const response = await runLeg(leg, leg.params, execution);
73208
73445
  execution.breakers.onSuccess(route.routeKey);
@@ -73225,6 +73462,15 @@ async function executeChain(alias, execution) {
73225
73462
  if (countsAgainstHealth) {
73226
73463
  execution.breakers.onFailure(route.routeKey);
73227
73464
  }
73465
+ else if (holdsProbe) {
73466
+ // No verdict on the route's health, but the probe slot this attempt
73467
+ // took must come back, or a half-open route admits no probe ever again.
73468
+ execution.breakers.onAttemptAbandoned(route.routeKey);
73469
+ }
73470
+ // A provider that answered with unparseable content still billed for the
73471
+ // answer; the spend belongs in the total whether or not a later leg serves.
73472
+ const billed = error instanceof LlmResponseFormatError ? error.usage : undefined;
73473
+ totalUsage = sumUsage(totalUsage, billed);
73228
73474
  const record = {
73229
73475
  routeKey: route.routeKey,
73230
73476
  role: route.role,
@@ -73233,6 +73479,7 @@ async function executeChain(alias, execution) {
73233
73479
  outcome,
73234
73480
  durationMs: now() - startedAt,
73235
73481
  reason,
73482
+ ...(billed === undefined ? {} : { usage: billed }),
73236
73483
  };
73237
73484
  attempts.push(record);
73238
73485
  execution.onAttempt?.(record);
@@ -74526,9 +74773,13 @@ function createGatewayTransport(config) {
74526
74773
  const payload = (await response.json());
74527
74774
  const choices = payload.choices;
74528
74775
  const message = choices?.[0]?.message;
74776
+ // Usage is read before the content is interpreted. The provider billed for
74777
+ // this answer whether or not it parses, and a parse failure that dropped
74778
+ // the count would report the attempt as free.
74779
+ const usage = readUsage(payload, request);
74529
74780
  return {
74530
- response: parseContent(message?.content, request.responseFormat),
74531
- usage: readUsage(payload, request),
74781
+ response: interpretContent(message?.content, request.responseFormat, usage),
74782
+ usage,
74532
74783
  tool_calls: Array.isArray(message?.tool_calls)
74533
74784
  ? message.tool_calls
74534
74785
  : undefined,
@@ -74563,25 +74814,22 @@ function buildMessages(request) {
74563
74814
  /**
74564
74815
  * Interpret the model's content according to the requested format.
74565
74816
  *
74566
- * A JSON format that does not parse is an error, not an empty object. Returning
74567
- * a default here would hand the caller a well-typed value that means nothing,
74568
- * and the failure would surface much later as a decision made on absent data.
74817
+ * Text is returned as sent. A structured format is parsed under the strict
74818
+ * single-fence rule of {@link parseStructuredContent}; a structured answer that
74819
+ * does not parse is an error carrying what the provider billed for it, never an
74820
+ * empty object.
74569
74821
  *
74570
74822
  * @param content The raw content.
74571
74823
  * @param responseFormat The format the caller asked for.
74572
- * @returns The parsed value.
74824
+ * @param usage What the provider billed for this answer.
74825
+ * @returns The interpreted value.
74826
+ * @throws {LlmResponseFormatError} When a structured answer does not parse.
74573
74827
  */
74574
- function parseContent(content, responseFormat) {
74575
- const text = typeof content === "string" ? content : "";
74828
+ function interpretContent(content, responseFormat, usage) {
74576
74829
  if (responseFormat === "text") {
74577
- return text;
74578
- }
74579
- try {
74580
- return JSON.parse(text);
74581
- }
74582
- catch (error) {
74583
- throw new Error(`LLM returned content that is not valid JSON for a ${typeof responseFormat === "string" ? responseFormat : "json_schema"} request: ${error instanceof Error ? error.message : String(error)}`);
74830
+ return (typeof content === "string" ? content : "");
74584
74831
  }
74832
+ return parseStructuredContent(content, responseFormat, usage);
74585
74833
  }
74586
74834
 
74587
74835
  /**
@@ -78676,6 +78924,7 @@ const OrderStatusSchema = enumType([
78676
78924
  "accepted",
78677
78925
  "pending_new",
78678
78926
  "accepted_for_bidding",
78927
+ "held",
78679
78928
  "stopped",
78680
78929
  "rejected",
78681
78930
  "suspended",
@@ -80332,6 +80581,7 @@ exports.GatewayUnreachableError = GatewayUnreachableError;
80332
80581
  exports.HttpClientError = HttpClientError;
80333
80582
  exports.HttpServerError = HttpServerError;
80334
80583
  exports.KEEP_ALIVE_DEFAULTS = KEEP_ALIVE_DEFAULTS;
80584
+ exports.LlmResponseFormatError = LlmResponseFormatError;
80335
80585
  exports.MARKET_DATA_API = MARKET_DATA_API;
80336
80586
  exports.MassiveAggregatesResponseSchema = MassiveAggregatesResponseSchema;
80337
80587
  exports.MassiveApiError = MassiveApiError;