@oh-my-pi/pi-ai 18.4.2 → 18.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/CHANGELOG.md +30 -41
  2. package/dist/types/auth-broker/protocol.d.ts +12 -0
  3. package/dist/types/dialect/rendering.d.ts +4 -0
  4. package/dist/types/images/openai-hosted.d.ts +1 -1
  5. package/dist/types/images/shared.d.ts +5 -2
  6. package/dist/types/providers/anthropic-wire.d.ts +9 -1
  7. package/dist/types/providers/anthropic.d.ts +17 -0
  8. package/dist/types/providers/aws-sigv4.d.ts +5 -0
  9. package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
  10. package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
  11. package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
  12. package/dist/types/providers/openai-chat-wire.d.ts +2 -2
  13. package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
  14. package/dist/types/providers/openai-responses-wire.d.ts +2 -2
  15. package/dist/types/providers/xai-base-url.d.ts +17 -0
  16. package/dist/types/types.d.ts +14 -3
  17. package/dist/types/usage/commandcode.d.ts +4 -0
  18. package/dist/types/usage/shared.d.ts +13 -1
  19. package/dist/types/utils/event-stream.d.ts +7 -0
  20. package/dist/types/utils/openai-http.d.ts +7 -1
  21. package/package.json +6 -6
  22. package/src/auth-broker/client.ts +1 -12
  23. package/src/auth-broker/protocol.ts +32 -0
  24. package/src/auth-broker/remote-store.ts +4 -40
  25. package/src/auth-broker/server.ts +1 -20
  26. package/src/auth-broker/snapshot-cache.ts +1 -9
  27. package/src/dialect/anthropic.ts +3 -25
  28. package/src/dialect/minimax.ts +3 -24
  29. package/src/dialect/rendering.ts +18 -0
  30. package/src/dialect/thinking.ts +24 -6
  31. package/src/dialect/xml.ts +3 -19
  32. package/src/images/openai-hosted.ts +5 -2
  33. package/src/images/openai-images.ts +10 -4
  34. package/src/images/shared.ts +9 -4
  35. package/src/providers/amazon-bedrock.ts +61 -22
  36. package/src/providers/anthropic-compaction.ts +10 -1
  37. package/src/providers/anthropic-wire.ts +12 -1
  38. package/src/providers/anthropic.ts +58 -15
  39. package/src/providers/aws-eventstream.ts +4 -3
  40. package/src/providers/aws-sigv4.ts +1 -1
  41. package/src/providers/azure-openai-responses.ts +3 -4
  42. package/src/providers/bedrock-anthropic.ts +30 -0
  43. package/src/providers/bedrock-request-metadata.ts +6 -0
  44. package/src/providers/connect-error-detail.ts +1 -5
  45. package/src/providers/cursor/interaction-query.ts +4 -2
  46. package/src/providers/cursor.ts +30 -26
  47. package/src/providers/google-gemini-cli.ts +9 -2
  48. package/src/providers/google-shared.ts +16 -6
  49. package/src/providers/openai-chat-wire.ts +2 -2
  50. package/src/providers/openai-codex/request-transformer.ts +1 -1
  51. package/src/providers/openai-codex-responses.ts +27 -16
  52. package/src/providers/openai-completions.ts +51 -36
  53. package/src/providers/openai-responses-wire.ts +2 -2
  54. package/src/providers/openai-responses.ts +3 -4
  55. package/src/providers/openai-shared.ts +130 -18
  56. package/src/providers/xai-base-url.ts +32 -0
  57. package/src/registry/engine/api-key.ts +8 -3
  58. package/src/types.ts +35 -4
  59. package/src/usage/claude.ts +4 -11
  60. package/src/usage/cline-pass.ts +2 -14
  61. package/src/usage/commandcode.ts +209 -0
  62. package/src/usage/cursor.ts +9 -1
  63. package/src/usage/openai-codex.ts +3 -5
  64. package/src/usage/registry.ts +3 -0
  65. package/src/usage/shared.ts +28 -1
  66. package/src/usage/synthetic.ts +4 -40
  67. package/src/usage/umans.ts +8 -36
  68. package/src/usage/zai.ts +10 -38
  69. package/src/utils/event-stream.ts +38 -2
  70. package/src/utils/http-inspector.ts +4 -8
  71. package/src/utils/openai-http.ts +10 -3
  72. package/src/utils/schema/json-schema-validator.ts +23 -26
  73. package/src/utils/schema/meta-validator.ts +4 -7
  74. package/src/utils/schema/wire.ts +17 -21
@@ -4,7 +4,13 @@ import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
4
4
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
5
5
  import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
6
6
  import { clinePassClientHeaders } from "@oh-my-pi/pi-catalog/wire/cline-pass";
7
- import { $env, logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
7
+ import {
8
+ $env,
9
+ logger,
10
+ parseStreamingJson,
11
+ parseStreamingJsonThrottled,
12
+ type ServerSentEvent,
13
+ } from "@oh-my-pi/pi-utils";
8
14
  import { renderDemotedThinking } from "../dialect/demotion";
9
15
  import * as AIError from "../error";
10
16
  import { getKimiCommonHeaders } from "../registry/oauth/kimi";
@@ -16,7 +22,6 @@ import type {
16
22
  MessageAttribution,
17
23
  Model,
18
24
  ProviderSessionState,
19
- RawSseEvent,
20
25
  ServiceTier,
21
26
  StopReason,
22
27
  StreamFunction,
@@ -800,28 +805,29 @@ const streamOpenAICompletionsOnce = (
800
805
  // Track the OpenAI `[DONE]` sentinel independently of `onSseEvent`: it is
801
806
  // the streaming protocol's terminal signal, so a stream that ends with it
802
807
  // completed by server agreement even when no `finish_reason` chunk arrived.
808
+ // It arrives through `onDoneSentinel`, so the diagnostic observer below
809
+ // stays unset (and raw wire-line capture off) when nobody listens.
803
810
  let sawDoneSentinel = false;
804
- const rawSseObserver = (event: RawSseEvent) => {
805
- if (event.data === "[DONE]") sawDoneSentinel = true;
806
- if (onSseEvent) {
807
- if (!event.event && event.data && event.data !== "[DONE]") {
808
- try {
809
- const parsed = JSON.parse(event.data);
810
- const resolvedEvent =
811
- typeof parsed.type === "string"
812
- ? parsed.type
813
- : typeof parsed.object === "string"
814
- ? parsed.object
815
- : null;
816
- if (resolvedEvent) {
817
- event.event = resolvedEvent;
818
- event.raw = [`event: ${resolvedEvent}`, ...event.raw];
819
- }
820
- } catch {}
811
+ const rawSseObserver = onSseEvent
812
+ ? (event: ServerSentEvent) => {
813
+ if (!event.event && event.data && event.data !== "[DONE]") {
814
+ try {
815
+ const parsed = JSON.parse(event.data);
816
+ const resolvedEvent =
817
+ typeof parsed.type === "string"
818
+ ? parsed.type
819
+ : typeof parsed.object === "string"
820
+ ? parsed.object
821
+ : null;
822
+ if (resolvedEvent) {
823
+ event.event = resolvedEvent;
824
+ event.raw = [`event: ${resolvedEvent}`, ...event.raw];
825
+ }
826
+ } catch {}
827
+ }
828
+ onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
821
829
  }
822
- onSseEvent(event, model);
823
- }
824
- };
830
+ : undefined;
825
831
  // Assigned once the block helpers exist (they are scoped to the `try`);
826
832
  // the catch handler uses it to close open blocks before emitting the
827
833
  // terminal error so both exit paths obey the same block lifecycle.
@@ -931,6 +937,9 @@ const streamOpenAICompletionsOnce = (
931
937
  // bounds every attempt and backoff sleep — retries cannot
932
938
  // extend the deadline.
933
939
  onSseEvent: rawSseObserver,
940
+ onDoneSentinel: () => {
941
+ sawDoneSentinel = true;
942
+ },
934
943
  });
935
944
  // Disarm the first-event watchdog as soon as headers arrive — a slow
936
945
  // onResponse callback must not abort an already-connected stream.
@@ -1030,9 +1039,19 @@ const streamOpenAICompletionsOnce = (
1030
1039
  };
1031
1040
  let currentBlock: OpenAIStreamBlock | undefined;
1032
1041
  let messageThoughtSignature: GeminiMessageThoughtSignature | undefined;
1042
+ // Content blocks are append-only for the lifetime of the stream, so each
1043
+ // block's index is stable once pushed. Map block → index to keep the
1044
+ // per-delta contentIndex lookup O(1): a linear `indexOf` per delta turns a
1045
+ // long turn (many blocks × many deltas) quadratic, as openai-shared's
1046
+ // Responses decoder documents (issue #10605).
1047
+ const contentIndexByBlock = new Map<OpenAIStreamBlock, number>();
1048
+ const pushContentBlock = (block: OpenAIStreamBlock): void => {
1049
+ contentIndexByBlock.set(block, output.content.length);
1050
+ output.content.push(block);
1051
+ };
1033
1052
  const blockIndex = (block: OpenAIStreamBlock | undefined): number => {
1034
1053
  if (!block) return Math.max(0, output.content.length - 1);
1035
- return output.content.indexOf(block);
1054
+ return contentIndexByBlock.get(block) ?? output.content.indexOf(block);
1036
1055
  };
1037
1056
  const finishToolCallBlock = (block: ToolCallStreamBlock): void => {
1038
1057
  if (block.partialArgs === undefined) return;
@@ -1087,11 +1106,7 @@ const streamOpenAICompletionsOnce = (
1087
1106
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1088
1107
  finishPendingToolCallBlocks();
1089
1108
  };
1090
- const appendText = (
1091
- message: AssistantMessage,
1092
- eventStream: AssistantMessageEventStream,
1093
- text: string,
1094
- ): void => {
1109
+ const appendText = (text: string): void => {
1095
1110
  if (currentBlock?.type !== "text") {
1096
1111
  // Leave toolCall blocks pending across text transitions: chunks after
1097
1112
  // the first typically carry only `index`, so a finished (de-registered)
@@ -1099,15 +1114,15 @@ const streamOpenAICompletionsOnce = (
1099
1114
  // resume. The stream-end sweep finalizes pending calls.
1100
1115
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1101
1116
  currentBlock = { type: "text", text: "" };
1102
- message.content.push(currentBlock);
1103
- eventStream.push({ type: "text_start", contentIndex: blockIndex(currentBlock), partial: message });
1117
+ pushContentBlock(currentBlock);
1118
+ stream.push({ type: "text_start", contentIndex: blockIndex(currentBlock), partial: output });
1104
1119
  }
1105
1120
  currentBlock.text += text;
1106
- eventStream.push({
1121
+ stream.push({
1107
1122
  type: "text_delta",
1108
1123
  contentIndex: blockIndex(currentBlock),
1109
1124
  delta: text,
1110
- partial: message,
1125
+ partial: output,
1111
1126
  });
1112
1127
  };
1113
1128
  const openThinkingBlock = (signature?: string): ThinkingContent => {
@@ -1116,7 +1131,7 @@ const streamOpenAICompletionsOnce = (
1116
1131
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1117
1132
  const block: ThinkingContent = { type: "thinking", thinking: "", thinkingSignature: signature };
1118
1133
  currentBlock = block;
1119
- output.content.push(block);
1134
+ pushContentBlock(block);
1120
1135
  stream.push({ type: "thinking_start", contentIndex: blockIndex(block), partial: output });
1121
1136
  return block;
1122
1137
  };
@@ -1177,7 +1192,7 @@ const streamOpenAICompletionsOnce = (
1177
1192
  const appendTextDelta = (text: string): void => {
1178
1193
  if (!text) return;
1179
1194
  if (!firstTokenTime) firstTokenTime = performance.now();
1180
- appendText(output, stream, text);
1195
+ appendText(text);
1181
1196
  };
1182
1197
  // Tracks the last full cumulative reasoning snapshot per signature (the
1183
1198
  // reasoning field name) so dedup survives block transitions. Required
@@ -1249,7 +1264,7 @@ const streamOpenAICompletionsOnce = (
1249
1264
  };
1250
1265
  block.arguments = parseStreamingJson(call.arguments);
1251
1266
  currentBlock = block;
1252
- output.content.push(block);
1267
+ pushContentBlock(block);
1253
1268
  stream.push({ type: "toolcall_start", contentIndex: blockIndex(block), partial: output });
1254
1269
  stream.push({
1255
1270
  type: "toolcall_delta",
@@ -1473,7 +1488,7 @@ const streamOpenAICompletionsOnce = (
1473
1488
  if (streamIndex !== undefined) toolCallBlockByIndex.set(streamIndex, block);
1474
1489
  pendingToolCallBlocks.push(block);
1475
1490
  currentBlock = block;
1476
- output.content.push(block);
1491
+ pushContentBlock(block);
1477
1492
  stream.push({
1478
1493
  type: "toolcall_start",
1479
1494
  contentIndex: blockIndex(block),
@@ -791,7 +791,7 @@ export interface Response {
791
791
  * When this parameter is set, the response body will include the `service_tier`
792
792
  * utilized.
793
793
  */
794
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
794
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
795
795
  /**
796
796
  * The status of the response generation. One of `completed`, `failed`,
797
797
  * `in_progress`, `cancelled`, `queued`, or `incomplete`.
@@ -5986,7 +5986,7 @@ export interface ResponseCreateParamsBase {
5986
5986
  * When this parameter is set, the response body will include the `service_tier`
5987
5987
  * utilized.
5988
5988
  */
5989
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
5989
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
5990
5990
  /**
5991
5991
  * Whether to store the generated model response for later retrieval via API.
5992
5992
  */
@@ -1,5 +1,5 @@
1
1
  import { scheduler } from "node:timers/promises";
2
- import { $flag, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
2
+ import { $flag, logger, type ServerSentEvent, structuredCloneJSON } from "@oh-my-pi/pi-utils";
3
3
  import * as AIError from "../error";
4
4
  import { getEnvApiKey } from "../stream";
5
5
  import type {
@@ -9,7 +9,6 @@ import type {
9
9
  Model,
10
10
  OpenAICompat,
11
11
  ProviderSessionState,
12
- RawSseEvent,
13
12
  ServiceTier,
14
13
  StreamFunction,
15
14
  StreamOptions,
@@ -455,7 +454,7 @@ const streamOpenAIResponsesOnce = (
455
454
  const { requestAbortController, requestSignal } = abortTracker;
456
455
  const onSseEvent = options?.onSseEvent;
457
456
  const rawSseObserver = onSseEvent
458
- ? (event: RawSseEvent) => {
457
+ ? (event: ServerSentEvent) => {
459
458
  if (!event.event && event.data && event.data !== "[DONE]") {
460
459
  try {
461
460
  const parsed = JSON.parse(event.data);
@@ -471,7 +470,7 @@ const streamOpenAIResponsesOnce = (
471
470
  }
472
471
  } catch {}
473
472
  }
474
- onSseEvent(event, model);
473
+ onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
475
474
  }
476
475
  : undefined;
477
476
 
@@ -61,6 +61,7 @@ import {
61
61
  type Usage,
62
62
  } from "../types";
63
63
  import { resolveCopilotRequestIdentity } from "./github-copilot-headers";
64
+ import { resolveXaiBaseUrl } from "./xai-base-url";
64
65
 
65
66
  export type { OpenAIPromptCacheOptions } from "../types";
66
67
 
@@ -256,6 +257,9 @@ export function resolveOpenAIRequestSetup(
256
257
  baseUrl = sakanaBaseUrl;
257
258
  }
258
259
  }
260
+ if (model.provider === "xai" || model.provider === "xai-oauth") {
261
+ baseUrl = resolveXaiBaseUrl(model.provider, baseUrl, rawApiKey);
262
+ }
259
263
  if (model.provider === "github-copilot") {
260
264
  const copilotApiKey = parseGitHubCopilotApiKey(rawApiKey);
261
265
  apiKey = copilotApiKey.accessToken;
@@ -1667,20 +1671,58 @@ function classifyResponsesBatchItem(item: object): ResponsesBatchItemKind {
1667
1671
  * strict validator. See #8789.
1668
1672
  */
1669
1673
  export function hoistInterleavedResponsesToolBatchMessages<T extends object>(items: readonly T[]): T[] {
1670
- const moved = new Set<number>();
1674
+ const callIdOf = (item: T): string | undefined =>
1675
+ "call_id" in item && typeof item.call_id === "string" ? item.call_id : undefined;
1676
+ // Does a call with `callId` precede `index` within the same contiguous batch body?
1677
+ const hasEarlierBatchCall = (index: number, callId: string): boolean => {
1678
+ for (let probe = index - 1; probe >= 0; probe--) {
1679
+ const kind = classifyResponsesBatchItem(items[probe]);
1680
+ if (kind === "other") return false;
1681
+ if (kind === "call" && callIdOf(items[probe]) === callId) return true;
1682
+ }
1683
+ return false;
1684
+ };
1685
+ const bucketOf = new Map<number, number>();
1671
1686
  const insertBefore = new Map<number, number[]>();
1672
1687
  for (let index = 0; index < items.length; index++) {
1673
1688
  if (classifyResponsesBatchItem(items[index]) !== "output") continue;
1674
1689
  // Only anchor on the first output of a run.
1675
1690
  if (index > 0 && classifyResponsesBatchItem(items[index - 1]) === "output") continue;
1691
+ // Calls the batch still owns further back: the anchor run's outputs, plus
1692
+ // any earlier output crossed on the way.
1693
+ const pending = new Set<string>();
1694
+ for (let probe = index; probe < items.length; probe++) {
1695
+ if (classifyResponsesBatchItem(items[probe]) !== "output") break;
1696
+ const callId = callIdOf(items[probe]);
1697
+ if (callId) pending.add(callId);
1698
+ }
1676
1699
  // Walk back over the batch body (calls interleaved with assistant messages).
1700
+ // An earlier output is crossed only when it and a call the batch still owns
1701
+ // both pair with calls further back — i.e. the output belongs to this same
1702
+ // interrupted batch (#13083). Otherwise it closes a completed prior round,
1703
+ // whose trailing messages stay put.
1677
1704
  let start = index;
1678
1705
  let sawCall = false;
1679
1706
  const messageIndexes: number[] = [];
1680
1707
  while (start > 0) {
1681
- const kind = classifyResponsesBatchItem(items[start - 1]);
1708
+ const item = items[start - 1];
1709
+ const kind = classifyResponsesBatchItem(item);
1682
1710
  if (kind === "call") {
1683
1711
  sawCall = true;
1712
+ const callId = callIdOf(item);
1713
+ if (callId) pending.delete(callId);
1714
+ } else if (kind === "output") {
1715
+ const callId = callIdOf(item);
1716
+ if (!callId || !hasEarlierBatchCall(start - 1, callId)) break;
1717
+ let ownsEarlierCall = false;
1718
+ for (const owned of pending) {
1719
+ if (hasEarlierBatchCall(start - 1, owned)) {
1720
+ ownsEarlierCall = true;
1721
+ break;
1722
+ }
1723
+ }
1724
+ if (!ownsEarlierCall) break;
1725
+ pending.add(callId);
1684
1726
  } else if (kind === "assistant-message") {
1685
1727
  messageIndexes.push(start - 1);
1686
1728
  } else {
@@ -1693,17 +1735,25 @@ export function hoistInterleavedResponsesToolBatchMessages<T extends object>(ite
1693
1735
  messageIndexes.reverse();
1694
1736
  const target = insertBefore.get(start) ?? [];
1695
1737
  for (const messageIndex of messageIndexes) {
1696
- moved.add(messageIndex);
1738
+ // A wider batch can re-collect a message an earlier anchor already
1739
+ // scheduled; move it rather than emitting it twice.
1740
+ const previousStart = bucketOf.get(messageIndex);
1741
+ if (previousStart !== undefined) {
1742
+ const previous = insertBefore.get(previousStart);
1743
+ const slot = previous?.indexOf(messageIndex) ?? -1;
1744
+ if (previous && slot >= 0) previous.splice(slot, 1);
1745
+ }
1746
+ bucketOf.set(messageIndex, start);
1697
1747
  target.push(messageIndex);
1698
1748
  }
1699
1749
  insertBefore.set(start, target);
1700
1750
  }
1701
- if (moved.size === 0) return items.slice();
1751
+ if (bucketOf.size === 0) return items.slice();
1702
1752
  const result: T[] = [];
1703
1753
  for (let index = 0; index < items.length; index++) {
1704
1754
  const pending = insertBefore.get(index);
1705
1755
  if (pending) for (const messageIndex of pending) result.push(items[messageIndex]);
1706
- if (moved.has(index)) continue;
1756
+ if (bucketOf.has(index)) continue;
1707
1757
  result.push(items[index]);
1708
1758
  }
1709
1759
  return result;
@@ -2070,11 +2120,16 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
2070
2120
  },
2071
2121
  );
2072
2122
  const sanitizedHistoryItems = rawSanitizedHistoryItems
2073
- ? adaptResponsesReplayItemsForModel(
2074
- rawSanitizedHistoryItems,
2075
- supportsCustomToolCalls,
2076
- customToolWireNameMap,
2077
- options.model.supportsComputerUse === true,
2123
+ ? ensureRequiredResponsesReasoningReplay(
2124
+ adaptResponsesReplayItemsForModel(
2125
+ rawSanitizedHistoryItems,
2126
+ supportsCustomToolCalls,
2127
+ customToolWireNameMap,
2128
+ options.model.supportsComputerUse === true,
2129
+ ),
2130
+ assistantMsg.stopReason,
2131
+ options.requiresReasoningReplayForAllTurns ?? false,
2132
+ options.requiresReasoningReplayForToolCalls ?? false,
2078
2133
  )
2079
2134
  : undefined;
2080
2135
  if (nativeReplayEnabled && sanitizedHistoryItems) {
@@ -2175,6 +2230,70 @@ function parseResponseReasoningReplayItem(signature: string | undefined): Respon
2175
2230
  */
2176
2231
  export const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
2177
2232
 
2233
+ function createSyntheticResponsesReasoningItem(
2234
+ text = SYNTHETIC_REASONING_REPLAY_PLACEHOLDER,
2235
+ id?: string,
2236
+ ): ResponseReasoningItem {
2237
+ const item = {
2238
+ type: "reasoning",
2239
+ ...(id ? { id } : {}),
2240
+ summary: [],
2241
+ content: [{ type: "reasoning_text", text }],
2242
+ } satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
2243
+ // The vendored SDK type marks `id` required; the wire accepts its absence.
2244
+ return item as ResponseReasoningItem;
2245
+ }
2246
+
2247
+ function isResponsesAssistantTurnBoundary(item: ResponseInput[number]): boolean {
2248
+ if (responsesToolOutputKind(item.type) !== undefined) return true;
2249
+ if (item.type === "compaction") return true;
2250
+ return "role" in item && item.role !== "assistant";
2251
+ }
2252
+
2253
+ function ensureRequiredResponsesReasoningReplay(
2254
+ items: ResponseInput,
2255
+ stopReason: AssistantMessage["stopReason"],
2256
+ requiresAllTurns: boolean,
2257
+ requiresToolCalls: boolean,
2258
+ ): ResponseInput {
2259
+ if (stopReason === "error" || (!requiresAllTurns && !requiresToolCalls)) return items;
2260
+
2261
+ const insertBefore: number[] = [];
2262
+ let turnStart = 0;
2263
+ for (let index = 0; index <= items.length; index++) {
2264
+ if (index < items.length && !isResponsesAssistantTurnBoundary(items[index])) continue;
2265
+
2266
+ let hasContent = false;
2267
+ let hasReasoning = false;
2268
+ let hasToolCall = false;
2269
+ for (let turnIndex = turnStart; turnIndex < index; turnIndex++) {
2270
+ const item = items[turnIndex];
2271
+ if (item.type === "reasoning") {
2272
+ hasReasoning = true;
2273
+ continue;
2274
+ }
2275
+ hasContent = true;
2276
+ if (classifyResponsesBatchItem(item) === "call") hasToolCall = true;
2277
+ }
2278
+ if (hasContent && !hasReasoning && (requiresAllTurns || (requiresToolCalls && hasToolCall))) {
2279
+ insertBefore.push(turnStart);
2280
+ }
2281
+ turnStart = index + 1;
2282
+ }
2283
+ if (insertBefore.length === 0) return items;
2284
+
2285
+ const repaired: ResponseInput = [];
2286
+ let insertionIndex = 0;
2287
+ for (let index = 0; index < items.length; index++) {
2288
+ if (insertBefore[insertionIndex] === index) {
2289
+ repaired.push(createSyntheticResponsesReasoningItem());
2290
+ insertionIndex++;
2291
+ }
2292
+ repaired.push(items[index]);
2293
+ }
2294
+ return repaired;
2295
+ }
2296
+
2178
2297
  export function convertResponsesAssistantMessage<TApi extends Api>(
2179
2298
  assistantMsg: AssistantMessage,
2180
2299
  model: Model<TApi>,
@@ -2345,14 +2464,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
2345
2464
  const carriedReasoningText = carriedReasoningTexts.join("\n");
2346
2465
  const reasoningText =
2347
2466
  carriedReasoningText.length > 0 ? carriedReasoningText : SYNTHETIC_REASONING_REPLAY_PLACEHOLDER;
2348
- const reasoningItem = {
2349
- type: "reasoning",
2350
- ...(synthesizedReasoningItemId ? { id: synthesizedReasoningItemId } : {}),
2351
- summary: [],
2352
- content: [{ type: "reasoning_text", text: reasoningText }],
2353
- } satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
2354
- // The vendored SDK type marks `id` required; the wire accepts its absence.
2355
- outputItems.unshift(reasoningItem as ResponseReasoningItem);
2467
+ outputItems.unshift(createSyntheticResponsesReasoningItem(reasoningText, synthesizedReasoningItemId));
2356
2468
  }
2357
2469
 
2358
2470
  return outputItems;
@@ -0,0 +1,32 @@
1
+ import { $env } from "@oh-my-pi/pi-utils";
2
+ import { parseXAIAccessTokenPayload } from "../registry/oauth/xai-oauth";
3
+
4
+ /** Bundled xAI API endpoint for the `xai` and `xai-oauth` providers. */
5
+ export const XAI_DEFAULT_BASE_URL = "https://api.x.ai/v1";
6
+
7
+ /**
8
+ * Resolve the base URL for an xAI request (`xai` / `xai-oauth` chat, image
9
+ * generation, web search, and HTTP tools).
10
+ *
11
+ * `XAI_BASE_URL` redirects traffic that targets the bundled default endpoint
12
+ * (or has no base URL); trailing slashes on the override are stripped. A
13
+ * custom `baseUrl` (models.yml, provider config) always wins.
14
+ *
15
+ * The override never receives official xAI OAuth credentials: an `xai-oauth`
16
+ * request whose bearer is an xAI OAuth access token (a JWT, whether stored,
17
+ * from `XAI_OAUTH_TOKEN`, or unknown because no bearer was supplied) stays on
18
+ * the bundled endpoint. API keys, including command-backed ones, follow the
19
+ * override.
20
+ */
21
+ export function resolveXaiBaseUrl(
22
+ provider: string,
23
+ baseUrl: string | undefined,
24
+ bearer: string | undefined,
25
+ ): string | undefined {
26
+ if (baseUrl && baseUrl.replace(/\/+$/, "") !== XAI_DEFAULT_BASE_URL) return baseUrl;
27
+ if (provider === "xai-oauth" && (bearer === undefined || parseXAIAccessTokenPayload(bearer) !== null)) {
28
+ return baseUrl;
29
+ }
30
+ const override = $env.XAI_BASE_URL?.trim().replace(/\/+$/, "");
31
+ return override || baseUrl;
32
+ }
@@ -96,9 +96,14 @@ export function createApiKeyLogin(
96
96
  try {
97
97
  await runValidation(rule.validate, label, trimmed, options);
98
98
  } catch (error) {
99
- // An optional probe only rejects on a real auth failure (401/403);
100
- // any other validation-endpoint failure trusts the supplied key.
101
- if (!rule.validate.optional || AIError.is(AIError.classify(error), AIError.Flag.AuthFailed)) {
99
+ // An optional probe only rejects on a real auth failure (401/403, or
100
+ // only 401 with `trustForbidden`); any other validation-endpoint
101
+ // failure trusts the supplied key.
102
+ const trustsForbidden = rule.validate.trustForbidden && AIError.status(error) === 403;
103
+ if (
104
+ !rule.validate.optional ||
105
+ (AIError.is(AIError.classify(error), AIError.Flag.AuthFailed) && !trustsForbidden)
106
+ ) {
102
107
  throw error;
103
108
  }
104
109
  options.onProgress?.(`Skipping ${label} validation endpoint; continuing with provided API key.`);
package/src/types.ts CHANGED
@@ -127,7 +127,9 @@ export type CacheRetention = "none" | "short" | "long";
127
127
  * values providers consume on the wire:
128
128
  *
129
129
  * - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field
130
- * (`flex`/`scale`/`priority`).
130
+ * (`flex`/`scale`/`priority`/`ultrafast`). `ultrafast` is a separate
131
+ * low-latency serving path: sent to the OpenAI API as-is (preview access is
132
+ * per project), and to Codex only for models whose discovery advertises it.
131
133
  * - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier`
132
134
  * field (`flex`/`priority`).
133
135
  * - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for
@@ -139,7 +141,7 @@ export type CacheRetention = "none" | "short" | "long";
139
141
  * Per-family scoping is expressed by {@link ServiceTierByFamily}, not by
140
142
  * scoped sentinel values — see {@link serviceTierFamily}.
141
143
  */
142
- export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority";
144
+ export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast";
143
145
 
144
146
  /** Provider families that expose an independent service-tier knob. */
145
147
  export type ServiceTierFamily = "openai" | "anthropic" | "google";
@@ -152,7 +154,7 @@ export type ServiceTierFamily = "openai" | "anthropic" | "google";
152
154
  */
153
155
  export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
154
156
 
155
- type ServiceTierModel = Pick<Model, "provider" | "api" | "identity">;
157
+ type ServiceTierModel = Pick<Model, "provider" | "api" | "identity"> & Partial<Pick<Model, "serviceTiers">>;
156
158
  // The service-tier matrix below intentionally stays in TypeScript rather than
157
159
  // the KDL compat tree: `shouldSendServiceTier` accepts bare provider strings
158
160
  // (agent telemetry, google-shared header placement) and the stats parser
@@ -228,6 +230,15 @@ export function resolveModelServiceTier(
228
230
  * Vertex) and OpenRouter accept `flex`/`priority`; Fireworks Serverless
229
231
  * realizes only its Priority serving path. Anthropic is absent because it
230
232
  * realizes `priority` via `speed: "fast"`.
233
+ *
234
+ * Codex-backend models (`openai-codex-responses`): `ultrafast` is sent only
235
+ * when the model's discovered `service_tiers` lists it. `priority`/`scale`
236
+ * are dropped only when that list is non-empty and omits them (codex-rs
237
+ * `service_tier_for_request`); an empty or missing list counts as "not
238
+ * reported" — accounts whose `/models` lists no tiers keep `/fast` — so the
239
+ * provider-level answer stands. `flex` and `default` are never gated.
240
+ * First-party OpenAI takes `ultrafast` as-is. A bare provider string cannot
241
+ * carry the list, so it answers for the provider alone.
231
242
  */
232
243
  export function shouldSendServiceTier(
233
244
  serviceTier: ServiceTier | null | undefined,
@@ -235,6 +246,19 @@ export function shouldSendServiceTier(
235
246
  ): boolean {
236
247
  if (!serviceTier || serviceTier === "auto") return false;
237
248
  const provider = typeof target === "string" ? target : target?.provider;
249
+ if (
250
+ typeof target !== "string" &&
251
+ target?.api === "openai-codex-responses" &&
252
+ serviceTier !== "flex" &&
253
+ serviceTier !== "default"
254
+ ) {
255
+ const advertised = target.serviceTiers;
256
+ if (serviceTier === "ultrafast") return advertised?.includes(serviceTier) === true;
257
+ if (advertised !== undefined && advertised.length > 0) return advertised.includes(serviceTier);
258
+ }
259
+ if (serviceTier === "ultrafast") {
260
+ return provider === "openai" || (typeof target === "string" && provider === "openai-codex");
261
+ }
238
262
  if (provider === "openai" || provider === "openai-codex") return true;
239
263
  if (provider === "openrouter") {
240
264
  return serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority";
@@ -311,7 +335,14 @@ export function coerceServiceTierByFamily(value: unknown): ServiceTierByFamily |
311
335
  const out: ServiceTierByFamily = {};
312
336
  for (const family of ["openai", "anthropic", "google"] as const) {
313
337
  const tier = src[family];
314
- if (tier === "auto" || tier === "default" || tier === "flex" || tier === "scale" || tier === "priority") {
338
+ if (
339
+ tier === "auto" ||
340
+ tier === "default" ||
341
+ tier === "flex" ||
342
+ tier === "scale" ||
343
+ tier === "priority" ||
344
+ tier === "ultrafast"
345
+ ) {
315
346
  out[family] = tier;
316
347
  }
317
348
  }
@@ -12,13 +12,12 @@ import {
12
12
  type UsageLimit,
13
13
  type UsageProvider,
14
14
  type UsageReport,
15
- type UsageStatus,
16
15
  type UsageWindow,
17
16
  } from "../usage";
18
17
  import { isRecord } from "../utils";
19
18
  import { buildClaudeOAuthHeaders, claudeOAuthBaseUrls } from "./claude-api";
20
19
  import { listClaudeResetCredits, parseClaudeResetCreditsFromUsagePayload } from "./claude-reset";
21
- import { HOUR_MS, parseIsoTimestamp, WEEK_MS } from "./shared";
20
+ import { HOUR_MS, parseIsoTimestamp, usageStatus, WEEK_MS } from "./shared";
22
21
 
23
22
  const MAX_ATTEMPTS = 3;
24
23
  const BASE_RETRY_DELAY_MS = 500;
@@ -416,13 +415,6 @@ function buildUsageAmount(utilization: number | undefined): UsageAmount | undefi
416
415
  };
417
416
  }
418
417
 
419
- function buildUsageStatus(usedFraction: number | undefined): UsageStatus | undefined {
420
- if (usedFraction === undefined) return undefined;
421
- if (usedFraction >= 1) return "exhausted";
422
- if (usedFraction >= 0.9) return "warning";
423
- return "ok";
424
- }
425
-
426
418
  function parseDollarAmount(
427
419
  amountMinor: unknown,
428
420
  exponent: unknown,
@@ -500,12 +492,13 @@ function buildClaudeExtraUsageLimit(payload: ClaudeUsageResponse): UsageLimit |
500
492
 
501
493
  const amount = buildExtraUsageAmount(parsed.used, parsed.limit);
502
494
  if (!amount) return null;
495
+ // A defined limit makes `buildExtraUsageAmount` set `usedFraction`.
503
496
  const status =
504
497
  parsed.limit === undefined
505
498
  ? undefined
506
499
  : parsed.used >= parsed.limit
507
500
  ? "exhausted"
508
- : (buildUsageStatus(amount.usedFraction) ?? "ok");
501
+ : usageStatus(amount.usedFraction);
509
502
  return {
510
503
  id: "anthropic:extra",
511
504
  label: "Claude Extra Usage",
@@ -549,7 +542,7 @@ function buildUsageLimit(args: {
549
542
  },
550
543
  window,
551
544
  amount,
552
- status: buildUsageStatus(amount.usedFraction),
545
+ status: amount.usedFraction === undefined ? undefined : usageStatus(amount.usedFraction),
553
546
  };
554
547
  }
555
548
 
@@ -1,14 +1,8 @@
1
1
  import { CLINEPASS_API_BASE_URL, clinePassClientHeaders } from "@oh-my-pi/pi-catalog/wire/cline-pass";
2
2
  import { ProviderHttpError } from "../error";
3
- import type {
4
- UsageFetchContext,
5
- UsageFetchParams,
6
- UsageLimit,
7
- UsageProvider,
8
- UsageReport,
9
- UsageStatus,
10
- } from "../usage";
3
+ import type { UsageFetchContext, UsageFetchParams, UsageLimit, UsageProvider, UsageReport } from "../usage";
11
4
  import { isRecord } from "../utils";
5
+ import { usageStatus } from "./shared";
12
6
 
13
7
  const PROVIDER = "cline-pass";
14
8
  const DEFAULT_BASE_URL = CLINEPASS_API_BASE_URL;
@@ -30,12 +24,6 @@ function parseResetTime(value: unknown): number | undefined {
30
24
  return Number.isFinite(timestamp) ? timestamp : undefined;
31
25
  }
32
26
 
33
- function usageStatus(usedFraction: number): UsageStatus {
34
- if (usedFraction >= 1) return "exhausted";
35
- if (usedFraction >= 0.9) return "warning";
36
- return "ok";
37
- }
38
-
39
27
  function parseLimit(raw: unknown, provider: UsageFetchParams["provider"]): UsageLimit | null {
40
28
  if (!isRecord(raw) || typeof raw.type !== "string" || !(raw.type in WINDOW_CONFIG)) return null;
41
29
  if (typeof raw.percentUsed !== "number" || !Number.isFinite(raw.percentUsed)) return null;