@oh-my-pi/pi-ai 18.4.2 → 18.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -41
- package/dist/types/auth-broker/protocol.d.ts +12 -0
- package/dist/types/dialect/rendering.d.ts +4 -0
- package/dist/types/images/openai-hosted.d.ts +1 -1
- package/dist/types/images/shared.d.ts +5 -2
- package/dist/types/providers/anthropic-wire.d.ts +9 -1
- package/dist/types/providers/anthropic.d.ts +17 -0
- package/dist/types/providers/aws-sigv4.d.ts +5 -0
- package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
- package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
- package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
- package/dist/types/providers/openai-chat-wire.d.ts +2 -2
- package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
- package/dist/types/providers/openai-responses-wire.d.ts +2 -2
- package/dist/types/providers/xai-base-url.d.ts +17 -0
- package/dist/types/types.d.ts +14 -3
- package/dist/types/usage/commandcode.d.ts +4 -0
- package/dist/types/usage/shared.d.ts +13 -1
- package/dist/types/utils/event-stream.d.ts +7 -0
- package/dist/types/utils/openai-http.d.ts +7 -1
- package/package.json +6 -6
- package/src/auth-broker/client.ts +1 -12
- package/src/auth-broker/protocol.ts +32 -0
- package/src/auth-broker/remote-store.ts +4 -40
- package/src/auth-broker/server.ts +1 -20
- package/src/auth-broker/snapshot-cache.ts +1 -9
- package/src/dialect/anthropic.ts +3 -25
- package/src/dialect/minimax.ts +3 -24
- package/src/dialect/rendering.ts +18 -0
- package/src/dialect/thinking.ts +24 -6
- package/src/dialect/xml.ts +3 -19
- package/src/images/openai-hosted.ts +5 -2
- package/src/images/openai-images.ts +10 -4
- package/src/images/shared.ts +9 -4
- package/src/providers/amazon-bedrock.ts +61 -22
- package/src/providers/anthropic-compaction.ts +10 -1
- package/src/providers/anthropic-wire.ts +12 -1
- package/src/providers/anthropic.ts +58 -15
- package/src/providers/aws-eventstream.ts +4 -3
- package/src/providers/aws-sigv4.ts +1 -1
- package/src/providers/azure-openai-responses.ts +3 -4
- package/src/providers/bedrock-anthropic.ts +30 -0
- package/src/providers/bedrock-request-metadata.ts +6 -0
- package/src/providers/connect-error-detail.ts +1 -5
- package/src/providers/cursor/interaction-query.ts +4 -2
- package/src/providers/cursor.ts +30 -26
- package/src/providers/google-gemini-cli.ts +9 -2
- package/src/providers/google-shared.ts +16 -6
- package/src/providers/openai-chat-wire.ts +2 -2
- package/src/providers/openai-codex/request-transformer.ts +1 -1
- package/src/providers/openai-codex-responses.ts +27 -16
- package/src/providers/openai-completions.ts +51 -36
- package/src/providers/openai-responses-wire.ts +2 -2
- package/src/providers/openai-responses.ts +3 -4
- package/src/providers/openai-shared.ts +130 -18
- package/src/providers/xai-base-url.ts +32 -0
- package/src/registry/engine/api-key.ts +8 -3
- package/src/types.ts +35 -4
- package/src/usage/claude.ts +4 -11
- package/src/usage/cline-pass.ts +2 -14
- package/src/usage/commandcode.ts +209 -0
- package/src/usage/cursor.ts +9 -1
- package/src/usage/openai-codex.ts +3 -5
- package/src/usage/registry.ts +3 -0
- package/src/usage/shared.ts +28 -1
- package/src/usage/synthetic.ts +4 -40
- package/src/usage/umans.ts +8 -36
- package/src/usage/zai.ts +10 -38
- package/src/utils/event-stream.ts +38 -2
- package/src/utils/http-inspector.ts +4 -8
- package/src/utils/openai-http.ts +10 -3
- package/src/utils/schema/json-schema-validator.ts +23 -26
- package/src/utils/schema/meta-validator.ts +4 -7
- package/src/utils/schema/wire.ts +17 -21
|
@@ -4,7 +4,13 @@ import { resolveWireModelId } from "@oh-my-pi/pi-catalog/model-thinking";
|
|
|
4
4
|
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
|
5
5
|
import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types";
|
|
6
6
|
import { clinePassClientHeaders } from "@oh-my-pi/pi-catalog/wire/cline-pass";
|
|
7
|
-
import {
|
|
7
|
+
import {
|
|
8
|
+
$env,
|
|
9
|
+
logger,
|
|
10
|
+
parseStreamingJson,
|
|
11
|
+
parseStreamingJsonThrottled,
|
|
12
|
+
type ServerSentEvent,
|
|
13
|
+
} from "@oh-my-pi/pi-utils";
|
|
8
14
|
import { renderDemotedThinking } from "../dialect/demotion";
|
|
9
15
|
import * as AIError from "../error";
|
|
10
16
|
import { getKimiCommonHeaders } from "../registry/oauth/kimi";
|
|
@@ -16,7 +22,6 @@ import type {
|
|
|
16
22
|
MessageAttribution,
|
|
17
23
|
Model,
|
|
18
24
|
ProviderSessionState,
|
|
19
|
-
RawSseEvent,
|
|
20
25
|
ServiceTier,
|
|
21
26
|
StopReason,
|
|
22
27
|
StreamFunction,
|
|
@@ -800,28 +805,29 @@ const streamOpenAICompletionsOnce = (
|
|
|
800
805
|
// Track the OpenAI `[DONE]` sentinel independently of `onSseEvent`: it is
|
|
801
806
|
// the streaming protocol's terminal signal, so a stream that ends with it
|
|
802
807
|
// completed by server agreement even when no `finish_reason` chunk arrived.
|
|
808
|
+
// It arrives through `onDoneSentinel`, so the diagnostic observer below
|
|
809
|
+
// stays unset (and raw wire-line capture off) when nobody listens.
|
|
803
810
|
let sawDoneSentinel = false;
|
|
804
|
-
const rawSseObserver =
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
}
|
|
820
|
-
}
|
|
811
|
+
const rawSseObserver = onSseEvent
|
|
812
|
+
? (event: ServerSentEvent) => {
|
|
813
|
+
if (!event.event && event.data && event.data !== "[DONE]") {
|
|
814
|
+
try {
|
|
815
|
+
const parsed = JSON.parse(event.data);
|
|
816
|
+
const resolvedEvent =
|
|
817
|
+
typeof parsed.type === "string"
|
|
818
|
+
? parsed.type
|
|
819
|
+
: typeof parsed.object === "string"
|
|
820
|
+
? parsed.object
|
|
821
|
+
: null;
|
|
822
|
+
if (resolvedEvent) {
|
|
823
|
+
event.event = resolvedEvent;
|
|
824
|
+
event.raw = [`event: ${resolvedEvent}`, ...event.raw];
|
|
825
|
+
}
|
|
826
|
+
} catch {}
|
|
827
|
+
}
|
|
828
|
+
onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
|
|
821
829
|
}
|
|
822
|
-
|
|
823
|
-
}
|
|
824
|
-
};
|
|
830
|
+
: undefined;
|
|
825
831
|
// Assigned once the block helpers exist (they are scoped to the `try`);
|
|
826
832
|
// the catch handler uses it to close open blocks before emitting the
|
|
827
833
|
// terminal error so both exit paths obey the same block lifecycle.
|
|
@@ -931,6 +937,9 @@ const streamOpenAICompletionsOnce = (
|
|
|
931
937
|
// bounds every attempt and backoff sleep — retries cannot
|
|
932
938
|
// extend the deadline.
|
|
933
939
|
onSseEvent: rawSseObserver,
|
|
940
|
+
onDoneSentinel: () => {
|
|
941
|
+
sawDoneSentinel = true;
|
|
942
|
+
},
|
|
934
943
|
});
|
|
935
944
|
// Disarm the first-event watchdog as soon as headers arrive — a slow
|
|
936
945
|
// onResponse callback must not abort an already-connected stream.
|
|
@@ -1030,9 +1039,19 @@ const streamOpenAICompletionsOnce = (
|
|
|
1030
1039
|
};
|
|
1031
1040
|
let currentBlock: OpenAIStreamBlock | undefined;
|
|
1032
1041
|
let messageThoughtSignature: GeminiMessageThoughtSignature | undefined;
|
|
1042
|
+
// Content blocks are append-only for the lifetime of the stream, so each
|
|
1043
|
+
// block's index is stable once pushed. Map block → index to keep the
|
|
1044
|
+
// per-delta contentIndex lookup O(1): a linear `indexOf` per delta turns a
|
|
1045
|
+
// long turn (many blocks × many deltas) quadratic, as openai-shared's
|
|
1046
|
+
// Responses decoder documents (issue #10605).
|
|
1047
|
+
const contentIndexByBlock = new Map<OpenAIStreamBlock, number>();
|
|
1048
|
+
const pushContentBlock = (block: OpenAIStreamBlock): void => {
|
|
1049
|
+
contentIndexByBlock.set(block, output.content.length);
|
|
1050
|
+
output.content.push(block);
|
|
1051
|
+
};
|
|
1033
1052
|
const blockIndex = (block: OpenAIStreamBlock | undefined): number => {
|
|
1034
1053
|
if (!block) return Math.max(0, output.content.length - 1);
|
|
1035
|
-
return output.content.indexOf(block);
|
|
1054
|
+
return contentIndexByBlock.get(block) ?? output.content.indexOf(block);
|
|
1036
1055
|
};
|
|
1037
1056
|
const finishToolCallBlock = (block: ToolCallStreamBlock): void => {
|
|
1038
1057
|
if (block.partialArgs === undefined) return;
|
|
@@ -1087,11 +1106,7 @@ const streamOpenAICompletionsOnce = (
|
|
|
1087
1106
|
if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
|
|
1088
1107
|
finishPendingToolCallBlocks();
|
|
1089
1108
|
};
|
|
1090
|
-
const appendText = (
|
|
1091
|
-
message: AssistantMessage,
|
|
1092
|
-
eventStream: AssistantMessageEventStream,
|
|
1093
|
-
text: string,
|
|
1094
|
-
): void => {
|
|
1109
|
+
const appendText = (text: string): void => {
|
|
1095
1110
|
if (currentBlock?.type !== "text") {
|
|
1096
1111
|
// Leave toolCall blocks pending across text transitions: chunks after
|
|
1097
1112
|
// the first typically carry only `index`, so a finished (de-registered)
|
|
@@ -1099,15 +1114,15 @@ const streamOpenAICompletionsOnce = (
|
|
|
1099
1114
|
// resume. The stream-end sweep finalizes pending calls.
|
|
1100
1115
|
if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
|
|
1101
1116
|
currentBlock = { type: "text", text: "" };
|
|
1102
|
-
|
|
1103
|
-
|
|
1117
|
+
pushContentBlock(currentBlock);
|
|
1118
|
+
stream.push({ type: "text_start", contentIndex: blockIndex(currentBlock), partial: output });
|
|
1104
1119
|
}
|
|
1105
1120
|
currentBlock.text += text;
|
|
1106
|
-
|
|
1121
|
+
stream.push({
|
|
1107
1122
|
type: "text_delta",
|
|
1108
1123
|
contentIndex: blockIndex(currentBlock),
|
|
1109
1124
|
delta: text,
|
|
1110
|
-
partial:
|
|
1125
|
+
partial: output,
|
|
1111
1126
|
});
|
|
1112
1127
|
};
|
|
1113
1128
|
const openThinkingBlock = (signature?: string): ThinkingContent => {
|
|
@@ -1116,7 +1131,7 @@ const streamOpenAICompletionsOnce = (
|
|
|
1116
1131
|
if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
|
|
1117
1132
|
const block: ThinkingContent = { type: "thinking", thinking: "", thinkingSignature: signature };
|
|
1118
1133
|
currentBlock = block;
|
|
1119
|
-
|
|
1134
|
+
pushContentBlock(block);
|
|
1120
1135
|
stream.push({ type: "thinking_start", contentIndex: blockIndex(block), partial: output });
|
|
1121
1136
|
return block;
|
|
1122
1137
|
};
|
|
@@ -1177,7 +1192,7 @@ const streamOpenAICompletionsOnce = (
|
|
|
1177
1192
|
const appendTextDelta = (text: string): void => {
|
|
1178
1193
|
if (!text) return;
|
|
1179
1194
|
if (!firstTokenTime) firstTokenTime = performance.now();
|
|
1180
|
-
appendText(
|
|
1195
|
+
appendText(text);
|
|
1181
1196
|
};
|
|
1182
1197
|
// Tracks the last full cumulative reasoning snapshot per signature (the
|
|
1183
1198
|
// reasoning field name) so dedup survives block transitions. Required
|
|
@@ -1249,7 +1264,7 @@ const streamOpenAICompletionsOnce = (
|
|
|
1249
1264
|
};
|
|
1250
1265
|
block.arguments = parseStreamingJson(call.arguments);
|
|
1251
1266
|
currentBlock = block;
|
|
1252
|
-
|
|
1267
|
+
pushContentBlock(block);
|
|
1253
1268
|
stream.push({ type: "toolcall_start", contentIndex: blockIndex(block), partial: output });
|
|
1254
1269
|
stream.push({
|
|
1255
1270
|
type: "toolcall_delta",
|
|
@@ -1473,7 +1488,7 @@ const streamOpenAICompletionsOnce = (
|
|
|
1473
1488
|
if (streamIndex !== undefined) toolCallBlockByIndex.set(streamIndex, block);
|
|
1474
1489
|
pendingToolCallBlocks.push(block);
|
|
1475
1490
|
currentBlock = block;
|
|
1476
|
-
|
|
1491
|
+
pushContentBlock(block);
|
|
1477
1492
|
stream.push({
|
|
1478
1493
|
type: "toolcall_start",
|
|
1479
1494
|
contentIndex: blockIndex(block),
|
|
@@ -791,7 +791,7 @@ export interface Response {
|
|
|
791
791
|
* When this parameter is set, the response body will include the `service_tier`
|
|
792
792
|
* utilized.
|
|
793
793
|
*/
|
|
794
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
794
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
795
795
|
/**
|
|
796
796
|
* The status of the response generation. One of `completed`, `failed`,
|
|
797
797
|
* `in_progress`, `cancelled`, `queued`, or `incomplete`.
|
|
@@ -5986,7 +5986,7 @@ export interface ResponseCreateParamsBase {
|
|
|
5986
5986
|
* When this parameter is set, the response body will include the `service_tier`
|
|
5987
5987
|
* utilized.
|
|
5988
5988
|
*/
|
|
5989
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
5989
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
5990
5990
|
/**
|
|
5991
5991
|
* Whether to store the generated model response for later retrieval via API.
|
|
5992
5992
|
*/
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { scheduler } from "node:timers/promises";
|
|
2
|
-
import { $flag, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
2
|
+
import { $flag, logger, type ServerSentEvent, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
|
3
3
|
import * as AIError from "../error";
|
|
4
4
|
import { getEnvApiKey } from "../stream";
|
|
5
5
|
import type {
|
|
@@ -9,7 +9,6 @@ import type {
|
|
|
9
9
|
Model,
|
|
10
10
|
OpenAICompat,
|
|
11
11
|
ProviderSessionState,
|
|
12
|
-
RawSseEvent,
|
|
13
12
|
ServiceTier,
|
|
14
13
|
StreamFunction,
|
|
15
14
|
StreamOptions,
|
|
@@ -455,7 +454,7 @@ const streamOpenAIResponsesOnce = (
|
|
|
455
454
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
456
455
|
const onSseEvent = options?.onSseEvent;
|
|
457
456
|
const rawSseObserver = onSseEvent
|
|
458
|
-
? (event:
|
|
457
|
+
? (event: ServerSentEvent) => {
|
|
459
458
|
if (!event.event && event.data && event.data !== "[DONE]") {
|
|
460
459
|
try {
|
|
461
460
|
const parsed = JSON.parse(event.data);
|
|
@@ -471,7 +470,7 @@ const streamOpenAIResponsesOnce = (
|
|
|
471
470
|
}
|
|
472
471
|
} catch {}
|
|
473
472
|
}
|
|
474
|
-
onSseEvent(event, model);
|
|
473
|
+
onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
|
|
475
474
|
}
|
|
476
475
|
: undefined;
|
|
477
476
|
|
|
@@ -61,6 +61,7 @@ import {
|
|
|
61
61
|
type Usage,
|
|
62
62
|
} from "../types";
|
|
63
63
|
import { resolveCopilotRequestIdentity } from "./github-copilot-headers";
|
|
64
|
+
import { resolveXaiBaseUrl } from "./xai-base-url";
|
|
64
65
|
|
|
65
66
|
export type { OpenAIPromptCacheOptions } from "../types";
|
|
66
67
|
|
|
@@ -256,6 +257,9 @@ export function resolveOpenAIRequestSetup(
|
|
|
256
257
|
baseUrl = sakanaBaseUrl;
|
|
257
258
|
}
|
|
258
259
|
}
|
|
260
|
+
if (model.provider === "xai" || model.provider === "xai-oauth") {
|
|
261
|
+
baseUrl = resolveXaiBaseUrl(model.provider, baseUrl, rawApiKey);
|
|
262
|
+
}
|
|
259
263
|
if (model.provider === "github-copilot") {
|
|
260
264
|
const copilotApiKey = parseGitHubCopilotApiKey(rawApiKey);
|
|
261
265
|
apiKey = copilotApiKey.accessToken;
|
|
@@ -1667,20 +1671,58 @@ function classifyResponsesBatchItem(item: object): ResponsesBatchItemKind {
|
|
|
1667
1671
|
* strict validator. See #8789.
|
|
1668
1672
|
*/
|
|
1669
1673
|
export function hoistInterleavedResponsesToolBatchMessages<T extends object>(items: readonly T[]): T[] {
|
|
1670
|
-
const
|
|
1674
|
+
const callIdOf = (item: T): string | undefined =>
|
|
1675
|
+
"call_id" in item && typeof item.call_id === "string" ? item.call_id : undefined;
|
|
1676
|
+
// Does a call with `callId` precede `index` within the same contiguous batch body?
|
|
1677
|
+
const hasEarlierBatchCall = (index: number, callId: string): boolean => {
|
|
1678
|
+
for (let probe = index - 1; probe >= 0; probe--) {
|
|
1679
|
+
const kind = classifyResponsesBatchItem(items[probe]);
|
|
1680
|
+
if (kind === "other") return false;
|
|
1681
|
+
if (kind === "call" && callIdOf(items[probe]) === callId) return true;
|
|
1682
|
+
}
|
|
1683
|
+
return false;
|
|
1684
|
+
};
|
|
1685
|
+
const bucketOf = new Map<number, number>();
|
|
1671
1686
|
const insertBefore = new Map<number, number[]>();
|
|
1672
1687
|
for (let index = 0; index < items.length; index++) {
|
|
1673
1688
|
if (classifyResponsesBatchItem(items[index]) !== "output") continue;
|
|
1674
1689
|
// Only anchor on the first output of a run.
|
|
1675
1690
|
if (index > 0 && classifyResponsesBatchItem(items[index - 1]) === "output") continue;
|
|
1691
|
+
// Calls the batch still owns further back: the anchor run's outputs, plus
|
|
1692
|
+
// any earlier output crossed on the way.
|
|
1693
|
+
const pending = new Set<string>();
|
|
1694
|
+
for (let probe = index; probe < items.length; probe++) {
|
|
1695
|
+
if (classifyResponsesBatchItem(items[probe]) !== "output") break;
|
|
1696
|
+
const callId = callIdOf(items[probe]);
|
|
1697
|
+
if (callId) pending.add(callId);
|
|
1698
|
+
}
|
|
1676
1699
|
// Walk back over the batch body (calls interleaved with assistant messages).
|
|
1700
|
+
// An earlier output is crossed only when it and a call the batch still owns
|
|
1701
|
+
// both pair with calls further back — i.e. the output belongs to this same
|
|
1702
|
+
// interrupted batch (#13083). Otherwise it closes a completed prior round,
|
|
1703
|
+
// whose trailing messages stay put.
|
|
1677
1704
|
let start = index;
|
|
1678
1705
|
let sawCall = false;
|
|
1679
1706
|
const messageIndexes: number[] = [];
|
|
1680
1707
|
while (start > 0) {
|
|
1681
|
-
const
|
|
1708
|
+
const item = items[start - 1];
|
|
1709
|
+
const kind = classifyResponsesBatchItem(item);
|
|
1682
1710
|
if (kind === "call") {
|
|
1683
1711
|
sawCall = true;
|
|
1712
|
+
const callId = callIdOf(item);
|
|
1713
|
+
if (callId) pending.delete(callId);
|
|
1714
|
+
} else if (kind === "output") {
|
|
1715
|
+
const callId = callIdOf(item);
|
|
1716
|
+
if (!callId || !hasEarlierBatchCall(start - 1, callId)) break;
|
|
1717
|
+
let ownsEarlierCall = false;
|
|
1718
|
+
for (const owned of pending) {
|
|
1719
|
+
if (hasEarlierBatchCall(start - 1, owned)) {
|
|
1720
|
+
ownsEarlierCall = true;
|
|
1721
|
+
break;
|
|
1722
|
+
}
|
|
1723
|
+
}
|
|
1724
|
+
if (!ownsEarlierCall) break;
|
|
1725
|
+
pending.add(callId);
|
|
1684
1726
|
} else if (kind === "assistant-message") {
|
|
1685
1727
|
messageIndexes.push(start - 1);
|
|
1686
1728
|
} else {
|
|
@@ -1693,17 +1735,25 @@ export function hoistInterleavedResponsesToolBatchMessages<T extends object>(ite
|
|
|
1693
1735
|
messageIndexes.reverse();
|
|
1694
1736
|
const target = insertBefore.get(start) ?? [];
|
|
1695
1737
|
for (const messageIndex of messageIndexes) {
|
|
1696
|
-
|
|
1738
|
+
// A wider batch can re-collect a message an earlier anchor already
|
|
1739
|
+
// scheduled; move it rather than emitting it twice.
|
|
1740
|
+
const previousStart = bucketOf.get(messageIndex);
|
|
1741
|
+
if (previousStart !== undefined) {
|
|
1742
|
+
const previous = insertBefore.get(previousStart);
|
|
1743
|
+
const slot = previous?.indexOf(messageIndex) ?? -1;
|
|
1744
|
+
if (previous && slot >= 0) previous.splice(slot, 1);
|
|
1745
|
+
}
|
|
1746
|
+
bucketOf.set(messageIndex, start);
|
|
1697
1747
|
target.push(messageIndex);
|
|
1698
1748
|
}
|
|
1699
1749
|
insertBefore.set(start, target);
|
|
1700
1750
|
}
|
|
1701
|
-
if (
|
|
1751
|
+
if (bucketOf.size === 0) return items.slice();
|
|
1702
1752
|
const result: T[] = [];
|
|
1703
1753
|
for (let index = 0; index < items.length; index++) {
|
|
1704
1754
|
const pending = insertBefore.get(index);
|
|
1705
1755
|
if (pending) for (const messageIndex of pending) result.push(items[messageIndex]);
|
|
1706
|
-
if (
|
|
1756
|
+
if (bucketOf.has(index)) continue;
|
|
1707
1757
|
result.push(items[index]);
|
|
1708
1758
|
}
|
|
1709
1759
|
return result;
|
|
@@ -2070,11 +2120,16 @@ export function buildResponsesInput<TApi extends Api>(options: BuildResponsesInp
|
|
|
2070
2120
|
},
|
|
2071
2121
|
);
|
|
2072
2122
|
const sanitizedHistoryItems = rawSanitizedHistoryItems
|
|
2073
|
-
?
|
|
2074
|
-
|
|
2075
|
-
|
|
2076
|
-
|
|
2077
|
-
|
|
2123
|
+
? ensureRequiredResponsesReasoningReplay(
|
|
2124
|
+
adaptResponsesReplayItemsForModel(
|
|
2125
|
+
rawSanitizedHistoryItems,
|
|
2126
|
+
supportsCustomToolCalls,
|
|
2127
|
+
customToolWireNameMap,
|
|
2128
|
+
options.model.supportsComputerUse === true,
|
|
2129
|
+
),
|
|
2130
|
+
assistantMsg.stopReason,
|
|
2131
|
+
options.requiresReasoningReplayForAllTurns ?? false,
|
|
2132
|
+
options.requiresReasoningReplayForToolCalls ?? false,
|
|
2078
2133
|
)
|
|
2079
2134
|
: undefined;
|
|
2080
2135
|
if (nativeReplayEnabled && sanitizedHistoryItems) {
|
|
@@ -2175,6 +2230,70 @@ function parseResponseReasoningReplayItem(signature: string | undefined): Respon
|
|
|
2175
2230
|
*/
|
|
2176
2231
|
export const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
|
|
2177
2232
|
|
|
2233
|
+
function createSyntheticResponsesReasoningItem(
|
|
2234
|
+
text = SYNTHETIC_REASONING_REPLAY_PLACEHOLDER,
|
|
2235
|
+
id?: string,
|
|
2236
|
+
): ResponseReasoningItem {
|
|
2237
|
+
const item = {
|
|
2238
|
+
type: "reasoning",
|
|
2239
|
+
...(id ? { id } : {}),
|
|
2240
|
+
summary: [],
|
|
2241
|
+
content: [{ type: "reasoning_text", text }],
|
|
2242
|
+
} satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
|
|
2243
|
+
// The vendored SDK type marks `id` required; the wire accepts its absence.
|
|
2244
|
+
return item as ResponseReasoningItem;
|
|
2245
|
+
}
|
|
2246
|
+
|
|
2247
|
+
function isResponsesAssistantTurnBoundary(item: ResponseInput[number]): boolean {
|
|
2248
|
+
if (responsesToolOutputKind(item.type) !== undefined) return true;
|
|
2249
|
+
if (item.type === "compaction") return true;
|
|
2250
|
+
return "role" in item && item.role !== "assistant";
|
|
2251
|
+
}
|
|
2252
|
+
|
|
2253
|
+
function ensureRequiredResponsesReasoningReplay(
|
|
2254
|
+
items: ResponseInput,
|
|
2255
|
+
stopReason: AssistantMessage["stopReason"],
|
|
2256
|
+
requiresAllTurns: boolean,
|
|
2257
|
+
requiresToolCalls: boolean,
|
|
2258
|
+
): ResponseInput {
|
|
2259
|
+
if (stopReason === "error" || (!requiresAllTurns && !requiresToolCalls)) return items;
|
|
2260
|
+
|
|
2261
|
+
const insertBefore: number[] = [];
|
|
2262
|
+
let turnStart = 0;
|
|
2263
|
+
for (let index = 0; index <= items.length; index++) {
|
|
2264
|
+
if (index < items.length && !isResponsesAssistantTurnBoundary(items[index])) continue;
|
|
2265
|
+
|
|
2266
|
+
let hasContent = false;
|
|
2267
|
+
let hasReasoning = false;
|
|
2268
|
+
let hasToolCall = false;
|
|
2269
|
+
for (let turnIndex = turnStart; turnIndex < index; turnIndex++) {
|
|
2270
|
+
const item = items[turnIndex];
|
|
2271
|
+
if (item.type === "reasoning") {
|
|
2272
|
+
hasReasoning = true;
|
|
2273
|
+
continue;
|
|
2274
|
+
}
|
|
2275
|
+
hasContent = true;
|
|
2276
|
+
if (classifyResponsesBatchItem(item) === "call") hasToolCall = true;
|
|
2277
|
+
}
|
|
2278
|
+
if (hasContent && !hasReasoning && (requiresAllTurns || (requiresToolCalls && hasToolCall))) {
|
|
2279
|
+
insertBefore.push(turnStart);
|
|
2280
|
+
}
|
|
2281
|
+
turnStart = index + 1;
|
|
2282
|
+
}
|
|
2283
|
+
if (insertBefore.length === 0) return items;
|
|
2284
|
+
|
|
2285
|
+
const repaired: ResponseInput = [];
|
|
2286
|
+
let insertionIndex = 0;
|
|
2287
|
+
for (let index = 0; index < items.length; index++) {
|
|
2288
|
+
if (insertBefore[insertionIndex] === index) {
|
|
2289
|
+
repaired.push(createSyntheticResponsesReasoningItem());
|
|
2290
|
+
insertionIndex++;
|
|
2291
|
+
}
|
|
2292
|
+
repaired.push(items[index]);
|
|
2293
|
+
}
|
|
2294
|
+
return repaired;
|
|
2295
|
+
}
|
|
2296
|
+
|
|
2178
2297
|
export function convertResponsesAssistantMessage<TApi extends Api>(
|
|
2179
2298
|
assistantMsg: AssistantMessage,
|
|
2180
2299
|
model: Model<TApi>,
|
|
@@ -2345,14 +2464,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
|
|
2345
2464
|
const carriedReasoningText = carriedReasoningTexts.join("\n");
|
|
2346
2465
|
const reasoningText =
|
|
2347
2466
|
carriedReasoningText.length > 0 ? carriedReasoningText : SYNTHETIC_REASONING_REPLAY_PLACEHOLDER;
|
|
2348
|
-
|
|
2349
|
-
type: "reasoning",
|
|
2350
|
-
...(synthesizedReasoningItemId ? { id: synthesizedReasoningItemId } : {}),
|
|
2351
|
-
summary: [],
|
|
2352
|
-
content: [{ type: "reasoning_text", text: reasoningText }],
|
|
2353
|
-
} satisfies Omit<ResponseReasoningItem, "id"> & Partial<Pick<ResponseReasoningItem, "id">>;
|
|
2354
|
-
// The vendored SDK type marks `id` required; the wire accepts its absence.
|
|
2355
|
-
outputItems.unshift(reasoningItem as ResponseReasoningItem);
|
|
2467
|
+
outputItems.unshift(createSyntheticResponsesReasoningItem(reasoningText, synthesizedReasoningItemId));
|
|
2356
2468
|
}
|
|
2357
2469
|
|
|
2358
2470
|
return outputItems;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { $env } from "@oh-my-pi/pi-utils";
|
|
2
|
+
import { parseXAIAccessTokenPayload } from "../registry/oauth/xai-oauth";
|
|
3
|
+
|
|
4
|
+
/** Bundled xAI API endpoint for the `xai` and `xai-oauth` providers. */
|
|
5
|
+
export const XAI_DEFAULT_BASE_URL = "https://api.x.ai/v1";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Resolve the base URL for an xAI request (`xai` / `xai-oauth` chat, image
|
|
9
|
+
* generation, web search, and HTTP tools).
|
|
10
|
+
*
|
|
11
|
+
* `XAI_BASE_URL` redirects traffic that targets the bundled default endpoint
|
|
12
|
+
* (or has no base URL); trailing slashes on the override are stripped. A
|
|
13
|
+
* custom `baseUrl` (models.yml, provider config) always wins.
|
|
14
|
+
*
|
|
15
|
+
* The override never receives official xAI OAuth credentials: an `xai-oauth`
|
|
16
|
+
* request whose bearer is an xAI OAuth access token (a JWT, whether stored,
|
|
17
|
+
* from `XAI_OAUTH_TOKEN`, or unknown because no bearer was supplied) stays on
|
|
18
|
+
* the bundled endpoint. API keys, including command-backed ones, follow the
|
|
19
|
+
* override.
|
|
20
|
+
*/
|
|
21
|
+
export function resolveXaiBaseUrl(
|
|
22
|
+
provider: string,
|
|
23
|
+
baseUrl: string | undefined,
|
|
24
|
+
bearer: string | undefined,
|
|
25
|
+
): string | undefined {
|
|
26
|
+
if (baseUrl && baseUrl.replace(/\/+$/, "") !== XAI_DEFAULT_BASE_URL) return baseUrl;
|
|
27
|
+
if (provider === "xai-oauth" && (bearer === undefined || parseXAIAccessTokenPayload(bearer) !== null)) {
|
|
28
|
+
return baseUrl;
|
|
29
|
+
}
|
|
30
|
+
const override = $env.XAI_BASE_URL?.trim().replace(/\/+$/, "");
|
|
31
|
+
return override || baseUrl;
|
|
32
|
+
}
|
|
@@ -96,9 +96,14 @@ export function createApiKeyLogin(
|
|
|
96
96
|
try {
|
|
97
97
|
await runValidation(rule.validate, label, trimmed, options);
|
|
98
98
|
} catch (error) {
|
|
99
|
-
// An optional probe only rejects on a real auth failure (401/403
|
|
100
|
-
// any other validation-endpoint
|
|
101
|
-
|
|
99
|
+
// An optional probe only rejects on a real auth failure (401/403, or
|
|
100
|
+
// only 401 with `trustForbidden`); any other validation-endpoint
|
|
101
|
+
// failure trusts the supplied key.
|
|
102
|
+
const trustsForbidden = rule.validate.trustForbidden && AIError.status(error) === 403;
|
|
103
|
+
if (
|
|
104
|
+
!rule.validate.optional ||
|
|
105
|
+
(AIError.is(AIError.classify(error), AIError.Flag.AuthFailed) && !trustsForbidden)
|
|
106
|
+
) {
|
|
102
107
|
throw error;
|
|
103
108
|
}
|
|
104
109
|
options.onProgress?.(`Skipping ${label} validation endpoint; continuing with provided API key.`);
|
package/src/types.ts
CHANGED
|
@@ -127,7 +127,9 @@ export type CacheRetention = "none" | "short" | "long";
|
|
|
127
127
|
* values providers consume on the wire:
|
|
128
128
|
*
|
|
129
129
|
* - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field
|
|
130
|
-
* (`flex`/`scale`/`priority`).
|
|
130
|
+
* (`flex`/`scale`/`priority`/`ultrafast`). `ultrafast` is a separate
|
|
131
|
+
* low-latency serving path: sent to the OpenAI API as-is (preview access is
|
|
132
|
+
* per project), and to Codex only for models whose discovery advertises it.
|
|
131
133
|
* - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier`
|
|
132
134
|
* field (`flex`/`priority`).
|
|
133
135
|
* - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for
|
|
@@ -139,7 +141,7 @@ export type CacheRetention = "none" | "short" | "long";
|
|
|
139
141
|
* Per-family scoping is expressed by {@link ServiceTierByFamily}, not by
|
|
140
142
|
* scoped sentinel values — see {@link serviceTierFamily}.
|
|
141
143
|
*/
|
|
142
|
-
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority";
|
|
144
|
+
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast";
|
|
143
145
|
|
|
144
146
|
/** Provider families that expose an independent service-tier knob. */
|
|
145
147
|
export type ServiceTierFamily = "openai" | "anthropic" | "google";
|
|
@@ -152,7 +154,7 @@ export type ServiceTierFamily = "openai" | "anthropic" | "google";
|
|
|
152
154
|
*/
|
|
153
155
|
export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
|
|
154
156
|
|
|
155
|
-
type ServiceTierModel = Pick<Model, "provider" | "api" | "identity"
|
|
157
|
+
type ServiceTierModel = Pick<Model, "provider" | "api" | "identity"> & Partial<Pick<Model, "serviceTiers">>;
|
|
156
158
|
// The service-tier matrix below intentionally stays in TypeScript rather than
|
|
157
159
|
// the KDL compat tree: `shouldSendServiceTier` accepts bare provider strings
|
|
158
160
|
// (agent telemetry, google-shared header placement) and the stats parser
|
|
@@ -228,6 +230,15 @@ export function resolveModelServiceTier(
|
|
|
228
230
|
* Vertex) and OpenRouter accept `flex`/`priority`; Fireworks Serverless
|
|
229
231
|
* realizes only its Priority serving path. Anthropic is absent because it
|
|
230
232
|
* realizes `priority` via `speed: "fast"`.
|
|
233
|
+
*
|
|
234
|
+
* Codex-backend models (`openai-codex-responses`): `ultrafast` is sent only
|
|
235
|
+
* when the model's discovered `service_tiers` lists it. `priority`/`scale`
|
|
236
|
+
* are dropped only when that list is non-empty and omits them (codex-rs
|
|
237
|
+
* `service_tier_for_request`); an empty or missing list counts as "not
|
|
238
|
+
* reported" — accounts whose `/models` lists no tiers keep `/fast` — so the
|
|
239
|
+
* provider-level answer stands. `flex` and `default` are never gated.
|
|
240
|
+
* First-party OpenAI takes `ultrafast` as-is. A bare provider string cannot
|
|
241
|
+
* carry the list, so it answers for the provider alone.
|
|
231
242
|
*/
|
|
232
243
|
export function shouldSendServiceTier(
|
|
233
244
|
serviceTier: ServiceTier | null | undefined,
|
|
@@ -235,6 +246,19 @@ export function shouldSendServiceTier(
|
|
|
235
246
|
): boolean {
|
|
236
247
|
if (!serviceTier || serviceTier === "auto") return false;
|
|
237
248
|
const provider = typeof target === "string" ? target : target?.provider;
|
|
249
|
+
if (
|
|
250
|
+
typeof target !== "string" &&
|
|
251
|
+
target?.api === "openai-codex-responses" &&
|
|
252
|
+
serviceTier !== "flex" &&
|
|
253
|
+
serviceTier !== "default"
|
|
254
|
+
) {
|
|
255
|
+
const advertised = target.serviceTiers;
|
|
256
|
+
if (serviceTier === "ultrafast") return advertised?.includes(serviceTier) === true;
|
|
257
|
+
if (advertised !== undefined && advertised.length > 0) return advertised.includes(serviceTier);
|
|
258
|
+
}
|
|
259
|
+
if (serviceTier === "ultrafast") {
|
|
260
|
+
return provider === "openai" || (typeof target === "string" && provider === "openai-codex");
|
|
261
|
+
}
|
|
238
262
|
if (provider === "openai" || provider === "openai-codex") return true;
|
|
239
263
|
if (provider === "openrouter") {
|
|
240
264
|
return serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority";
|
|
@@ -311,7 +335,14 @@ export function coerceServiceTierByFamily(value: unknown): ServiceTierByFamily |
|
|
|
311
335
|
const out: ServiceTierByFamily = {};
|
|
312
336
|
for (const family of ["openai", "anthropic", "google"] as const) {
|
|
313
337
|
const tier = src[family];
|
|
314
|
-
if (
|
|
338
|
+
if (
|
|
339
|
+
tier === "auto" ||
|
|
340
|
+
tier === "default" ||
|
|
341
|
+
tier === "flex" ||
|
|
342
|
+
tier === "scale" ||
|
|
343
|
+
tier === "priority" ||
|
|
344
|
+
tier === "ultrafast"
|
|
345
|
+
) {
|
|
315
346
|
out[family] = tier;
|
|
316
347
|
}
|
|
317
348
|
}
|
package/src/usage/claude.ts
CHANGED
|
@@ -12,13 +12,12 @@ import {
|
|
|
12
12
|
type UsageLimit,
|
|
13
13
|
type UsageProvider,
|
|
14
14
|
type UsageReport,
|
|
15
|
-
type UsageStatus,
|
|
16
15
|
type UsageWindow,
|
|
17
16
|
} from "../usage";
|
|
18
17
|
import { isRecord } from "../utils";
|
|
19
18
|
import { buildClaudeOAuthHeaders, claudeOAuthBaseUrls } from "./claude-api";
|
|
20
19
|
import { listClaudeResetCredits, parseClaudeResetCreditsFromUsagePayload } from "./claude-reset";
|
|
21
|
-
import { HOUR_MS, parseIsoTimestamp, WEEK_MS } from "./shared";
|
|
20
|
+
import { HOUR_MS, parseIsoTimestamp, usageStatus, WEEK_MS } from "./shared";
|
|
22
21
|
|
|
23
22
|
const MAX_ATTEMPTS = 3;
|
|
24
23
|
const BASE_RETRY_DELAY_MS = 500;
|
|
@@ -416,13 +415,6 @@ function buildUsageAmount(utilization: number | undefined): UsageAmount | undefi
|
|
|
416
415
|
};
|
|
417
416
|
}
|
|
418
417
|
|
|
419
|
-
function buildUsageStatus(usedFraction: number | undefined): UsageStatus | undefined {
|
|
420
|
-
if (usedFraction === undefined) return undefined;
|
|
421
|
-
if (usedFraction >= 1) return "exhausted";
|
|
422
|
-
if (usedFraction >= 0.9) return "warning";
|
|
423
|
-
return "ok";
|
|
424
|
-
}
|
|
425
|
-
|
|
426
418
|
function parseDollarAmount(
|
|
427
419
|
amountMinor: unknown,
|
|
428
420
|
exponent: unknown,
|
|
@@ -500,12 +492,13 @@ function buildClaudeExtraUsageLimit(payload: ClaudeUsageResponse): UsageLimit |
|
|
|
500
492
|
|
|
501
493
|
const amount = buildExtraUsageAmount(parsed.used, parsed.limit);
|
|
502
494
|
if (!amount) return null;
|
|
495
|
+
// A defined limit makes `buildExtraUsageAmount` set `usedFraction`.
|
|
503
496
|
const status =
|
|
504
497
|
parsed.limit === undefined
|
|
505
498
|
? undefined
|
|
506
499
|
: parsed.used >= parsed.limit
|
|
507
500
|
? "exhausted"
|
|
508
|
-
: (
|
|
501
|
+
: usageStatus(amount.usedFraction);
|
|
509
502
|
return {
|
|
510
503
|
id: "anthropic:extra",
|
|
511
504
|
label: "Claude Extra Usage",
|
|
@@ -549,7 +542,7 @@ function buildUsageLimit(args: {
|
|
|
549
542
|
},
|
|
550
543
|
window,
|
|
551
544
|
amount,
|
|
552
|
-
status:
|
|
545
|
+
status: amount.usedFraction === undefined ? undefined : usageStatus(amount.usedFraction),
|
|
553
546
|
};
|
|
554
547
|
}
|
|
555
548
|
|
package/src/usage/cline-pass.ts
CHANGED
|
@@ -1,14 +1,8 @@
|
|
|
1
1
|
import { CLINEPASS_API_BASE_URL, clinePassClientHeaders } from "@oh-my-pi/pi-catalog/wire/cline-pass";
|
|
2
2
|
import { ProviderHttpError } from "../error";
|
|
3
|
-
import type {
|
|
4
|
-
UsageFetchContext,
|
|
5
|
-
UsageFetchParams,
|
|
6
|
-
UsageLimit,
|
|
7
|
-
UsageProvider,
|
|
8
|
-
UsageReport,
|
|
9
|
-
UsageStatus,
|
|
10
|
-
} from "../usage";
|
|
3
|
+
import type { UsageFetchContext, UsageFetchParams, UsageLimit, UsageProvider, UsageReport } from "../usage";
|
|
11
4
|
import { isRecord } from "../utils";
|
|
5
|
+
import { usageStatus } from "./shared";
|
|
12
6
|
|
|
13
7
|
const PROVIDER = "cline-pass";
|
|
14
8
|
const DEFAULT_BASE_URL = CLINEPASS_API_BASE_URL;
|
|
@@ -30,12 +24,6 @@ function parseResetTime(value: unknown): number | undefined {
|
|
|
30
24
|
return Number.isFinite(timestamp) ? timestamp : undefined;
|
|
31
25
|
}
|
|
32
26
|
|
|
33
|
-
function usageStatus(usedFraction: number): UsageStatus {
|
|
34
|
-
if (usedFraction >= 1) return "exhausted";
|
|
35
|
-
if (usedFraction >= 0.9) return "warning";
|
|
36
|
-
return "ok";
|
|
37
|
-
}
|
|
38
|
-
|
|
39
27
|
function parseLimit(raw: unknown, provider: UsageFetchParams["provider"]): UsageLimit | null {
|
|
40
28
|
if (!isRecord(raw) || typeof raw.type !== "string" || !(raw.type in WINDOW_CONFIG)) return null;
|
|
41
29
|
if (typeof raw.percentUsed !== "number" || !Number.isFinite(raw.percentUsed)) return null;
|