@oh-my-pi/pi-ai 18.2.1 → 18.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3783,7 +3783,7 @@ export class AuthStorage {
3783
3783
  ingestUsageHeaders(
3784
3784
  provider: Provider,
3785
3785
  headers: Record<string, string>,
3786
- options?: { sessionId?: string; baseUrl?: string },
3786
+ options?: { sessionId?: string; baseUrl?: string; responseStatus?: number },
3787
3787
  ): boolean {
3788
3788
  if (this.#fetchUsageReportsOverride) return false;
3789
3789
  const parseHeaders = this.#resolveUsageProvider(provider)?.parseRateLimitHeaders;
@@ -3796,7 +3796,7 @@ export class AuthStorage {
3796
3796
  this.#buildUsageRequestForOauth(provider, credential, options?.baseUrl),
3797
3797
  );
3798
3798
  const now = Date.now();
3799
- const parsedReport = parseHeaders(headers, now);
3799
+ const parsedReport = parseHeaders(headers, now, { responseStatus: options?.responseStatus });
3800
3800
  if (!parsedReport) return false;
3801
3801
  // Throttled to one ingest per interval — except when a window reads
3802
3802
  // exhausted: persist that snapshot immediately. A full-backed cache can
package/src/index.ts CHANGED
@@ -8,6 +8,7 @@ export * from "./auth-storage";
8
8
  export * from "./error/rate-limit";
9
9
  export * from "./oneshot-retry";
10
10
  export * from "./provider-details";
11
+ export * from "./provider-session-state";
11
12
  export * from "./providers/anthropic";
12
13
  export * from "./providers/anthropic-client";
13
14
  export * from "./providers/azure-openai-responses";
@@ -0,0 +1,56 @@
1
+ /**
2
+ * Credential-rotation handling for a retained `providerSessionState` map.
3
+ *
4
+ * A host that keeps one provider-session map per logical conversation (the
5
+ * auth-gateway's server-owned store, an in-process omp session) can outlive the
6
+ * credential that filled it: `AuthStorage.markUsageLimitReached` and the
7
+ * auth-retry resolver both switch a session to a sibling account mid-flight.
8
+ * Most of what a provider learns is a property of the *endpoint*, so rebuilding
9
+ * the whole map on a switch would re-pay every rejected round-trip the map
10
+ * exists to avoid. A minority is a property of the *account*, and keeping that
11
+ * across a switch is a bug.
12
+ *
13
+ * Audit of what the retained records hold, per provider:
14
+ *
15
+ * - **Anthropic** — `fastModeDisabled` is account-scoped: the rejection reads
16
+ * "this model does not support fast mode for your account", i.e. a plan
17
+ * entitlement, so a switch to an entitled sibling must re-probe. Its
18
+ * siblings are endpoint-scoped and stay: `strictToolsDisabled`
19
+ * (grammar-too-large 400 for the model's tool schema),
20
+ * `replayUnsignedThinkingDisabled` / `thinkingReplayDisabled` (the endpoint
21
+ * is a signing proxy), `prefixDroppedThinkingBlocks` (blocks the API itself
22
+ * dropped), `controlStates` (per-conversation control baselines).
23
+ * - **OpenAI Responses** — the `previous_response_id` chain baselines are
24
+ * account-scoped: a stored response belongs to the account that created it.
25
+ * Strict-tools / reasoning-effort fallbacks, replay warmup and the chaining
26
+ * circuit breaker are endpoint-scoped and stay.
27
+ * - **OpenAI Completions** — strict-tools and reasoning-effort fallbacks only;
28
+ * both endpoint-scoped. Nothing to reset.
29
+ * - **Codex** — already sub-keys its WebSocket sessions by account id AND
30
+ * bearer (`getCodexWebSocketSessionKey`), so a switch naturally lands on a
31
+ * fresh transport session while the old one stays reachable for teardown.
32
+ * Resetting from the outside would close a socket a retry may still be on.
33
+ * - **Antigravity** — `lastGoodEndpoint` is endpoint-scoped; the agent /
34
+ * conversation ids are conversation-scoped. Neither depends on the account.
35
+ * - **GitLab Duo** — the active workflow is account-bound, but it is a live
36
+ * server-side workflow plus socket, and the switch happens *inside* the
37
+ * request that may still be resuming it. Tearing it down here would abort the
38
+ * very turn that rotated; it stays on its existing session-close path.
39
+ */
40
+
41
+ import { clearAnthropicFastModeFallback } from "./providers/anthropic";
42
+ import { resetOpenAIResponsesAccountScopedState } from "./providers/openai-responses";
43
+ import type { ProviderSessionState } from "./types";
44
+
45
+ /**
46
+ * Reset the account-dependent lessons in `states`, keeping everything a
47
+ * provider learned about the endpoint. Call when a retained map is about to be
48
+ * reused for a session whose credential now resolves to a different account.
49
+ */
50
+ export function resetAccountScopedProviderSessionState(states: Map<string, ProviderSessionState>): void {
51
+ if (states.size === 0) return;
52
+ // Fast mode is the account-scoped half of the Anthropic record; the helper
53
+ // the `/fast on` re-arm path already uses clears exactly that flag.
54
+ clearAnthropicFastModeFallback(states);
55
+ resetOpenAIResponsesAccountScopedState(states);
56
+ }
@@ -5,6 +5,9 @@
5
5
  * SigV4 signing and decodes the `application/vnd.amazon.eventstream` response.
6
6
  * No `@aws-sdk/*`, no `@smithy/*`, no `proxy-agent`. Proxies are honored via
7
7
  * Bun's native `HTTPS_PROXY` support.
8
+ *
9
+ * A `models.yml` `baseUrl` is the request origin verbatim (VPC endpoint, gateway, …);
10
+ * only AWS's own regional host is re-pointed at the resolved region. SigV4 unaffected.
8
11
  */
9
12
 
10
13
  import type { Effort } from "@oh-my-pi/pi-catalog/effort";
@@ -145,6 +148,13 @@ const INFERENCE_PROFILE_GEO_DEFAULT_REGION: Record<string, string> = {
145
148
  jp: "ap-northeast-1",
146
149
  };
147
150
 
151
+ /**
152
+ * AWS's own regional host, which every bundled catalog entry carries as a required
153
+ * placeholder `baseUrl` — no routing info, so its region segment is re-derived.
154
+ * FIPS, VPC-endpoint and gateway hosts don't match and are used as configured.
155
+ */
156
+ const AWS_REGIONAL_BEDROCK_HOST = /^bedrock-runtime\.[a-z0-9-]+\.amazonaws\.com$/;
157
+
148
158
  /** Geo prefix of a cross-region inference-profile id, e.g. `eu.anthropic.…` → `eu`. */
149
159
  function inferenceProfileGeo(modelId: string): string | undefined {
150
160
  const dot = modelId.indexOf(".");
@@ -453,9 +463,15 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
453
463
  // raw dump so the inspector shows exactly what was sent.
454
464
  commandInput = { ...commandInput, requestMetadata: sanitizeRequestMetadata(commandInput.requestMetadata) };
455
465
 
456
- const host = `bedrock-runtime.${region}.amazonaws.com`;
457
- const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`;
458
- const urlPath = `/model/${encodeURIComponent(model.id)}/converse-stream`;
466
+ // `baseUrl` is the origin verbatim, path prefix (and query, for gateways
467
+ // that authenticate via a query parameter) included, so a gateway mounted
468
+ // under a path works. AWS's own host is re-pointed: the catalog can't know the region.
469
+ const base = new URL(model.baseUrl || `https://bedrock-runtime.${region}.amazonaws.com`);
470
+ if (AWS_REGIONAL_BEDROCK_HOST.test(base.host)) base.host = `bedrock-runtime.${region}.amazonaws.com`;
471
+ const host = base.host;
472
+ const urlPath = `${base.pathname.replace(/\/+$/, "")}/model/${encodeURIComponent(model.id)}/converse-stream`;
473
+ const query = base.search.slice(1) || undefined;
474
+ const url = `${base.origin}${urlPath}${base.search}`;
459
475
  rawRequestDump = {
460
476
  provider: model.provider,
461
477
  api: output.api,
@@ -527,6 +543,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
527
543
  method: "POST",
528
544
  host,
529
545
  path: urlPath,
546
+ query,
530
547
  body,
531
548
  region,
532
549
  service: "bedrock",
@@ -19,6 +19,7 @@ import type {
19
19
  ToolResultMessage,
20
20
  UserMessage,
21
21
  } from "../types";
22
+ import { isCursorExecResolved } from "../utils/block-symbols";
22
23
  import {
23
24
  type AnthropicAssistantContentBlock,
24
25
  type AnthropicMessage,
@@ -505,14 +506,33 @@ function randomFallback(): string {
505
506
  return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}`;
506
507
  }
507
508
 
508
- function mapStopReasonOut(reason: StopReason): "end_turn" | "max_tokens" | "tool_use" {
509
+ /**
510
+ * True for a `toolCall` block the client is expected to execute.
511
+ *
512
+ * Cursor's exec channel stamps {@link kCursorExecResolved} on calls it already
513
+ * ran server-side — `todo`, `web_fetch`, `connect_scm`, a native it declined —
514
+ * and those are not handoffs: the client never declared the tool, has no
515
+ * implementation to run, and repeating one would reapply a side effect the
516
+ * server already committed. Only unresolved calls are real external handoffs.
517
+ */
518
+ function isClientToolUse(content: AssistantMessage["content"][number]): content is ToolCall {
519
+ return content.type === "toolCall" && !isCursorExecResolved(content);
520
+ }
521
+
522
+ function mapStopReasonOut(reason: StopReason, hasToolUse: boolean): "end_turn" | "max_tokens" | "tool_use" {
509
523
  switch (reason) {
510
524
  case "length":
511
525
  return "max_tokens";
512
526
  case "toolUse":
513
527
  return "tool_use";
514
528
  default:
515
- return "end_turn";
529
+ // A provider whose protocol has no separate tool-use stop — Cursor
530
+ // ends the turn with `stop` when it hands a client-declared tool
531
+ // back for the caller to execute — still owes the client
532
+ // `tool_use`, or the canonical Anthropic loop (run tools while
533
+ // `stop_reason === "tool_use"`) never runs the tool it asked for.
534
+ // The OpenAI chat wire maps the same case to `tool_calls`.
535
+ return hasToolUse ? "tool_use" : "end_turn";
516
536
  }
517
537
  }
518
538
 
@@ -536,6 +556,9 @@ function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>
536
556
  blocks.push(c.block);
537
557
  break;
538
558
  case "toolCall":
559
+ // Cursor already executed this one; the client must not run it
560
+ // again and cannot answer it. See `isClientToolUse`.
561
+ if (!isClientToolUse(c)) break;
539
562
  blocks.push({ type: "tool_use", id: c.id, name: c.name, input: c.arguments ?? {} });
540
563
  break;
541
564
  }
@@ -568,7 +591,7 @@ export function encodeResponse(message: AssistantMessage, requestedModelId: stri
568
591
  role: "assistant",
569
592
  model: requestedModelId,
570
593
  content: encodeContentBlocks(message),
571
- stop_reason: mapStopReasonOut(message.stopReason),
594
+ stop_reason: mapStopReasonOut(message.stopReason, message.content.some(isClientToolUse)),
572
595
  // TODO: surface the matched stop sequence once pi-ai's
573
596
  // `AssistantMessage.stopReason` carries the matched string. Intentionally
574
597
  // `null` for now (Anthropic schema allows it).
@@ -633,6 +656,23 @@ export function encodeStream(
633
656
  const messageId = newMessageId();
634
657
  let started = false;
635
658
  const open = new Map<number, OpenBlock>();
659
+ // Cursor's exec channel hands back calls it already ran server-side;
660
+ // `isClientToolUse` keeps them off the wire. Anthropic clients (the
661
+ // official SDK included) append every `content_block_start` to their
662
+ // snapshot and then address deltas by `index`, so a hole in the
663
+ // numbering misroutes each later delta. Shift emitted indices down by
664
+ // the number of blocks suppressed before them — identity while
665
+ // nothing is suppressed. A suppressed call always closes the
666
+ // preceding text/thinking block before it opens, so no block that is
667
+ // still open is ever renumbered.
668
+ const suppressed = new Set<number>();
669
+ const wireIndex = (contentIndex: number): number => {
670
+ let shift = 0;
671
+ for (const index of suppressed) {
672
+ if (index < contentIndex) shift++;
673
+ }
674
+ return contentIndex - shift;
675
+ };
636
676
 
637
677
  const ensureStart = (partial: AssistantMessage | undefined) => {
638
678
  if (started) return;
@@ -663,10 +703,11 @@ export function encodeStream(
663
703
  const emitServerToolBlocksBefore = (message: AssistantMessage, beforeIndex: number) => {
664
704
  const limit = Math.min(beforeIndex, message.content.length);
665
705
  while (nextContentIndexToInspect < limit) {
666
- const index = nextContentIndexToInspect++;
667
- const content = message.content[index];
706
+ const contentIndex = nextContentIndexToInspect++;
707
+ const content = message.content[contentIndex];
668
708
  if (content?.type !== "anthropicServerTool") continue;
669
709
  ensureStart(message);
710
+ const index = wireIndex(contentIndex);
670
711
  controller.enqueue(
671
712
  sseFrame("content_block_start", {
672
713
  type: "content_block_start",
@@ -678,10 +719,11 @@ export function encodeStream(
678
719
  }
679
720
  };
680
721
 
681
- const closeBlock = (index: number) => {
682
- if (!open.has(index)) return;
683
- controller.enqueue(sseFrame("content_block_stop", { type: "content_block_stop", index }));
684
- open.delete(index);
722
+ const closeBlock = (contentIndex: number) => {
723
+ const block = open.get(contentIndex);
724
+ if (!block) return;
725
+ controller.enqueue(sseFrame("content_block_stop", { type: "content_block_stop", index: block.index }));
726
+ open.delete(contentIndex);
685
727
  };
686
728
 
687
729
  pingTimer = setInterval(() => {
@@ -711,11 +753,12 @@ export function encodeStream(
711
753
  case "text_start": {
712
754
  emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
713
755
  ensureStart(ev.partial);
714
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "text" });
756
+ const index = wireIndex(ev.contentIndex);
757
+ open.set(ev.contentIndex, { index, kind: "text" });
715
758
  controller.enqueue(
716
759
  sseFrame("content_block_start", {
717
760
  type: "content_block_start",
718
- index: ev.contentIndex,
761
+ index,
719
762
  content_block: { type: "text", text: "" },
720
763
  }),
721
764
  );
@@ -725,7 +768,7 @@ export function encodeStream(
725
768
  controller.enqueue(
726
769
  sseFrame("content_block_delta", {
727
770
  type: "content_block_delta",
728
- index: ev.contentIndex,
771
+ index: wireIndex(ev.contentIndex),
729
772
  delta: { type: "text_delta", text: ev.delta },
730
773
  }),
731
774
  );
@@ -736,11 +779,12 @@ export function encodeStream(
736
779
  case "thinking_start": {
737
780
  emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
738
781
  ensureStart(ev.partial);
739
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
782
+ const index = wireIndex(ev.contentIndex);
783
+ open.set(ev.contentIndex, { index, kind: "thinking" });
740
784
  controller.enqueue(
741
785
  sseFrame("content_block_start", {
742
786
  type: "content_block_start",
743
- index: ev.contentIndex,
787
+ index,
744
788
  content_block: { type: "thinking", thinking: "" },
745
789
  }),
746
790
  );
@@ -750,7 +794,7 @@ export function encodeStream(
750
794
  controller.enqueue(
751
795
  sseFrame("content_block_delta", {
752
796
  type: "content_block_delta",
753
- index: ev.contentIndex,
797
+ index: wireIndex(ev.contentIndex),
754
798
  delta: { type: "thinking_delta", thinking: ev.delta },
755
799
  }),
756
800
  );
@@ -761,7 +805,7 @@ export function encodeStream(
761
805
  controller.enqueue(
762
806
  sseFrame("content_block_delta", {
763
807
  type: "content_block_delta",
764
- index: ev.contentIndex,
808
+ index: wireIndex(ev.contentIndex),
765
809
  delta: { type: "signature_delta", signature: c.thinkingSignature },
766
810
  }),
767
811
  );
@@ -773,11 +817,21 @@ export function encodeStream(
773
817
  emitServerToolBlocksBefore(ev.partial, ev.contentIndex);
774
818
  ensureStart(ev.partial);
775
819
  const tc = ev.partial.content[ev.contentIndex] as ToolCall | undefined;
776
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "tool_use" });
820
+ if (tc && !isClientToolUse(tc)) {
821
+ // Cursor's exec channel already ran this call and
822
+ // buffered its result. Streaming it would invite the
823
+ // client to repeat a committed side effect and answer
824
+ // a tool it never declared, so drop the whole block —
825
+ // start, deltas and stop — from the wire.
826
+ suppressed.add(ev.contentIndex);
827
+ break;
828
+ }
829
+ const index = wireIndex(ev.contentIndex);
830
+ open.set(ev.contentIndex, { index, kind: "tool_use" });
777
831
  controller.enqueue(
778
832
  sseFrame("content_block_start", {
779
833
  type: "content_block_start",
780
- index: ev.contentIndex,
834
+ index,
781
835
  content_block: {
782
836
  type: "tool_use",
783
837
  id: tc?.id ?? "",
@@ -789,10 +843,11 @@ export function encodeStream(
789
843
  break;
790
844
  }
791
845
  case "toolcall_delta":
846
+ if (suppressed.has(ev.contentIndex)) break;
792
847
  controller.enqueue(
793
848
  sseFrame("content_block_delta", {
794
849
  type: "content_block_delta",
795
- index: ev.contentIndex,
850
+ index: wireIndex(ev.contentIndex),
796
851
  delta: { type: "input_json_delta", partial_json: ev.delta },
797
852
  }),
798
853
  );
@@ -809,7 +864,12 @@ export function encodeStream(
809
864
  // TODO: surface matched stop sequence once pi-ai
810
865
  // propagates it on the `done` event.
811
866
  delta: {
812
- stop_reason: mapStopReasonOut(ev.reason),
867
+ // A call Cursor resolved after it opened (an MCP
868
+ // frame answered by a local handler) already
869
+ // streamed; the client still must not be told to
870
+ // run it, so it does not terminate the turn with
871
+ // `tool_use` either.
872
+ stop_reason: mapStopReasonOut(ev.reason, ev.message.content.some(isClientToolUse)),
813
873
  stop_sequence: null,
814
874
  },
815
875
  ...(bindingControlsRequested
@@ -17,7 +17,7 @@ const HEADER_MODEL_FIELD = 6;
17
17
  const MAX_MODEL_ID_LENGTH = 128;
18
18
  const MODEL_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._:/-]*$/;
19
19
 
20
- /** Returns the first length-delimited field `field` in `message`, or undefined. */
20
+ /** Finds a length-delimited field, rejecting tags and lengths that exceed uint32. */
21
21
  function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array | undefined {
22
22
  let offset = 0;
23
23
  while (offset < message.length) {
@@ -27,6 +27,7 @@ function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array |
27
27
  do {
28
28
  if (offset >= message.length) return undefined;
29
29
  byte = message[offset++];
30
+ if (shift === 28 && byte > 0x0f) return undefined;
30
31
  tag |= (byte & 0x7f) << shift;
31
32
  shift += 7;
32
33
  } while (byte & 0x80);
@@ -49,10 +50,12 @@ function lengthDelimitedField(message: Uint8Array, field: number): Uint8Array |
49
50
  do {
50
51
  if (offset >= message.length) return undefined;
51
52
  byte = message[offset++];
53
+ if (shift === 28 && byte > 0x0f) return undefined;
52
54
  length |= (byte & 0x7f) << shift;
53
55
  shift += 7;
54
56
  } while (byte & 0x80);
55
- if (offset + length > message.length) return undefined;
57
+ length >>>= 0;
58
+ if (length > message.length - offset) return undefined;
56
59
  if (fieldNumber === field) return message.subarray(offset, offset + length);
57
60
  offset += length;
58
61
  break;
@@ -137,18 +137,29 @@ function encodeRfc3986(str: string): string {
137
137
  return encodeURIComponent(str).replace(/[!'()*]/g, c => `%${c.charCodeAt(0).toString(16).toUpperCase()}`);
138
138
  }
139
139
 
140
- function canonicalQuery(query: string | undefined): string {
140
+ /**
141
+ * AWS's canonical-request spec encodes each name/value first, THEN sorts by
142
+ * the encoded form ("Sort the encoded parameter names by character code" —
143
+ * https://docs.aws.amazon.com/IAM/latest/UserGuide/create-canonical-request.html).
144
+ * Sorting the decoded form instead gives the wrong order whenever encoding
145
+ * changes a character's relative position — e.g. raw key `%7B` (decodes to
146
+ * `{`, 0x7B) vs `x` (0x78): decoded, `x` < `{`; encoded, `%` (0x25) < `x`, so
147
+ * `%7B` sorts first. A gateway that validates SigV4 (or AWS itself) computes
148
+ * the signature over ITS OWN canonicalization and rejects ours if the two
149
+ * disagree on order.
150
+ */
151
+ export function canonicalQuery(query: string | undefined): string {
141
152
  if (!query) return "";
142
153
  const pairs: Array<[string, string]> = [];
143
154
  for (const part of query.split("&")) {
144
155
  if (!part) continue;
145
156
  const eq = part.indexOf("=");
146
- const k = eq === -1 ? part : part.slice(0, eq);
147
- const v = eq === -1 ? "" : part.slice(eq + 1);
148
- pairs.push([decodeURIComponent(k), decodeURIComponent(v)]);
157
+ const rawKey = eq === -1 ? part : part.slice(0, eq);
158
+ const rawValue = eq === -1 ? "" : part.slice(eq + 1);
159
+ pairs.push([encodeRfc3986(decodeURIComponent(rawKey)), encodeRfc3986(decodeURIComponent(rawValue))]);
149
160
  }
150
161
  pairs.sort((a, b) => (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0));
151
- return pairs.map(([k, v]) => `${encodeRfc3986(k)}=${encodeRfc3986(v)}`).join("&");
162
+ return pairs.map(([k, v]) => `${k}=${v}`).join("&");
152
163
  }
153
164
 
154
165
  export interface SignedHeaders {
@@ -1,6 +1,7 @@
1
1
  import { createHash } from "node:crypto";
2
2
  import * as fs from "node:fs/promises";
3
3
  import http2 from "node:http2";
4
+ import { isCursorMaxModeWireId } from "@oh-my-pi/pi-catalog/compat/collapse";
4
5
  import { classifyModel, collapseVariantId } from "@oh-my-pi/pi-catalog/compat/taxonomy";
5
6
  import type {
6
7
  ConversationStep,
@@ -5234,6 +5235,46 @@ function extractImages(content: (TextContent | ImageContent)[]) {
5234
5235
  );
5235
5236
  }
5236
5237
 
5238
+ /**
5239
+ * Resolve `max_mode` for the wire id a request actually routes to.
5240
+ *
5241
+ * `GetUsableModels` marks max-mode models per raw row and discovery copies that
5242
+ * onto `cursorMaxMode`, so on a row that puts its own id on the wire the marker
5243
+ * is the authority — Cursor serves the whole Opus `-fast` lane in max mode
5244
+ * (`claude-opus-4-8-high-fast` included) and leaves reasoning tiers such as
5245
+ * `claude-4.6-opus-max` out of it, neither of which the wire slug can tell.
5246
+ *
5247
+ * Collapsing a family ORs the members' markers onto the logical row, so there
5248
+ * `cursorMaxMode: true` only means *some* tier needs max mode; sending it for
5249
+ * every tier is the refused `-low` request of issue #9478. The members' own
5250
+ * markers survive per wire id in `cursorMaxModeRoutes`, so the routed id is
5251
+ * looked up there first.
5252
+ *
5253
+ * A row's own wire id still owns its marker even when it has effort routing
5254
+ * (for example a bare/thinking pair). Logical-only bundled rows and routes
5255
+ * discovery never advertised have no per-id marker; only those use the suffix.
5256
+ * A collapsed row whose `true` no route's suffix can explain keeps it for every
5257
+ * route: the marker came from a member the suffix rule cannot see.
5258
+ */
5259
+ function resolveCursorMaxMode(model: Model<"cursor-agent">, wireModelId: string): boolean {
5260
+ const discovered = model.cursorMaxModeRoutes?.[wireModelId];
5261
+ if (discovered !== undefined) return discovered;
5262
+ const routing = model.thinking?.effortRouting;
5263
+ if (routing === undefined || wireModelId === model.id) {
5264
+ return model.cursorMaxMode ?? isCursorMaxModeWireId(wireModelId);
5265
+ }
5266
+ let routesOwnId = routing.off === model.id;
5267
+ let hasInferredMaxRoute = typeof routing.off === "string" && isCursorMaxModeWireId(routing.off);
5268
+ for (const effort of THINKING_EFFORTS) {
5269
+ const target = routing[effort];
5270
+ if (target === model.id) routesOwnId = true;
5271
+ if (typeof target === "string" && isCursorMaxModeWireId(target)) hasInferredMaxRoute = true;
5272
+ }
5273
+ if (routesOwnId) return model.cursorMaxMode ?? isCursorMaxModeWireId(wireModelId);
5274
+ if (model.cursorMaxMode === true && !hasInferredMaxRoute) return true;
5275
+ return isCursorMaxModeWireId(wireModelId);
5276
+ }
5277
+
5237
5278
  /**
5238
5279
  * Resolve the Cursor Run wire model id and its parameter list.
5239
5280
  *
@@ -5261,9 +5302,11 @@ function resolveCursorWireModel(
5261
5302
  ): {
5262
5303
  modelId: string;
5263
5304
  parameters: RequestedModel_ModelParameterbytes[];
5305
+ maxMode: boolean;
5264
5306
  } {
5265
5307
  const wireModelId = requestModelId ?? model.requestModelId ?? model.id;
5266
- if (wireMode === "discovered") return { modelId: wireModelId, parameters: [] };
5308
+ const maxMode = resolveCursorMaxMode(model, wireModelId);
5309
+ if (wireMode === "discovered") return { modelId: wireModelId, parameters: [], maxMode };
5267
5310
  // `collapseVariantId` keeps the lane in the logical id (`-high-fast` →
5268
5311
  // base `-fast`) and decodes the KDL effort (`-none` → `off`).
5269
5312
  const collapsed = collapseVariantId("cursor", wireModelId);
@@ -5271,11 +5314,12 @@ function resolveCursorWireModel(
5271
5314
  const base = effort !== undefined ? collapsed.logicalId : undefined;
5272
5315
  if (effort !== undefined && base && classifyModel("cursor", base).class === "openai") {
5273
5316
  if (effort === "off") {
5274
- return { modelId: base, parameters: [] };
5317
+ return { modelId: base, parameters: [], maxMode };
5275
5318
  }
5276
5319
  if ((THINKING_EFFORTS as readonly string[]).includes(effort)) {
5277
5320
  return {
5278
5321
  modelId: base,
5322
+ maxMode,
5279
5323
  parameters: [
5280
5324
  create(RequestedModel_ModelParameterbytesSchema, { id: "reasoning", value: collapsed.effort }),
5281
5325
  ],
@@ -5289,9 +5333,10 @@ function resolveCursorWireModel(
5289
5333
  return {
5290
5334
  modelId: wireModelId,
5291
5335
  parameters: [create(RequestedModel_ModelParameterbytesSchema, { id: "fast", value: "false" })],
5336
+ maxMode,
5292
5337
  };
5293
5338
  }
5294
- return { modelId: wireModelId, parameters: [] };
5339
+ return { modelId: wireModelId, parameters: [], maxMode };
5295
5340
  }
5296
5341
 
5297
5342
  async function buildGrpcRequestForWireMode(
@@ -5401,12 +5446,11 @@ async function buildGrpcRequestForWireMode(
5401
5446
  turns,
5402
5447
  });
5403
5448
 
5404
- const { modelId: wireModelId, parameters: wireParameters } = resolveCursorWireModel(
5405
- model,
5406
- options?.wireModelId,
5407
- wireMode,
5408
- );
5409
- const cursorMaxMode = model.cursorMaxMode === true;
5449
+ const {
5450
+ modelId: wireModelId,
5451
+ parameters: wireParameters,
5452
+ maxMode: cursorMaxMode,
5453
+ } = resolveCursorWireModel(model, options?.wireModelId, wireMode);
5410
5454
  const modelDetails = create(ModelDetailsSchema, {
5411
5455
  modelId: wireModelId,
5412
5456
  displayModelId: model.id,
@@ -1277,6 +1277,12 @@ const streamOpenAICompletionsOnce = (
1277
1277
 
1278
1278
  if (choice?.delta?.tool_calls && choice.delta.tool_calls.length > 0) {
1279
1279
  const toolCalls = choice.delta.tool_calls;
1280
+ // Pure tool-call responses never emit a text/thinking delta, so
1281
+ // without this stamp TTFT stays undefined for every turn that
1282
+ // begins with a structured call (measured: all toolUse-stop
1283
+ // rows on OpenAI-compatible gateways) and the usage row's
1284
+ // TTFT/tok/s figures silently degrade.
1285
+ if (!firstTokenTime) firstTokenTime = performance.now();
1280
1286
  for (let toolCallOffset = 0; toolCallOffset < toolCalls.length; toolCallOffset++) {
1281
1287
  const toolCall = toolCalls[toolCallOffset]!;
1282
1288
  const streamIndex = typeof toolCall.index === "number" ? toolCall.index : undefined;
@@ -295,6 +295,33 @@ function resetOpenAIResponsesChainState(state: OpenAIResponsesChainState): void
295
295
  state.lastPromptCacheBreakpointPolicy = undefined;
296
296
  }
297
297
 
298
+ /**
299
+ * Drop the account-bound half of every retained `openai-responses` record in
300
+ * `states`: the stateful `previous_response_id` chain baselines.
301
+ *
302
+ * Chaining stores the turn server-side under the account that created it, so a
303
+ * baseline minted by one credential is dead weight the moment the session is
304
+ * switched to a sibling account — the next delta request answers
305
+ * `Previous response not found` and burns a turn re-learning that. Everything
306
+ * else this record holds describes the *deployment*, not the account
307
+ * (strict-tools demotion, reasoning-effort fallback, native-history-replay
308
+ * warmup, the chaining circuit breaker), and is deliberately preserved:
309
+ * re-learning an endpoint's limits on every credential switch is the cost this
310
+ * state exists to avoid.
311
+ */
312
+ export function resetOpenAIResponsesAccountScopedState(states: Map<string, ProviderSessionState>): void {
313
+ for (const [key, value] of states) {
314
+ if (!key.startsWith(OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX)) continue;
315
+ const state = value as OpenAIResponsesProviderSessionState;
316
+ for (const chain of state.chains.values()) {
317
+ resetOpenAIResponsesChainState(chain);
318
+ // The stale-failure counter tallies the previous account's 404s; a
319
+ // fresh account must not inherit a tripped circuit breaker.
320
+ chain.staleFailures = 0;
321
+ }
322
+ }
323
+ }
324
+
298
325
  interface OpenAIResponsesChainedParams {
299
326
  params: OpenAIResponsesSamplingParams;
300
327
  /** Set iff the params carry previous_response_id (delta request). */