@gajae-code/ai 0.12.7 → 0.12.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/dist/types/auth-storage.d.ts +33 -6
  3. package/dist/types/index.d.ts +2 -1
  4. package/dist/types/model-manager.d.ts +2 -0
  5. package/dist/types/model-pricing.d.ts +3 -0
  6. package/dist/types/provider-models/special.d.ts +1 -0
  7. package/dist/types/providers/composer-discipline.d.ts +29 -23
  8. package/dist/types/providers/openai-opencodex-responses.d.ts +10 -0
  9. package/dist/types/providers/openai-responses-shared.d.ts +1 -0
  10. package/dist/types/providers/register-builtins.d.ts +2 -2
  11. package/dist/types/types.d.ts +15 -7
  12. package/dist/types/utils/fallback-transport.d.ts +23 -0
  13. package/dist/types/utils/oauth/anthropic.d.ts +21 -2
  14. package/dist/types/utils/oauth/callback-server.d.ts +7 -0
  15. package/dist/types/utils/oauth/types.d.ts +12 -1
  16. package/package.json +2 -2
  17. package/src/auth-gateway/server.ts +6 -0
  18. package/src/auth-storage.ts +325 -39
  19. package/src/index.ts +2 -0
  20. package/src/model-manager.ts +9 -3
  21. package/src/model-pricing.ts +68 -0
  22. package/src/model-thinking.ts +23 -1
  23. package/src/models.json +122 -24
  24. package/src/models.ts +7 -4
  25. package/src/prompts/composer-bash-policy-recovery.md +1 -0
  26. package/src/prompts/cursor-composer-bash-policy-recovery.md +1 -0
  27. package/src/prompts/cursor-composer-edit-discipline.md +7 -0
  28. package/src/provider-models/descriptors.ts +7 -1
  29. package/src/provider-models/special.ts +8 -0
  30. package/src/providers/composer-discipline.ts +54 -0
  31. package/src/providers/cursor.ts +2 -2
  32. package/src/providers/openai-codex/response-handler.ts +24 -2
  33. package/src/providers/openai-codex-responses.ts +5 -1
  34. package/src/providers/openai-completions.ts +109 -23
  35. package/src/providers/openai-opencodex-responses.ts +173 -0
  36. package/src/providers/openai-responses-shared.ts +14 -3
  37. package/src/providers/openai-responses.ts +91 -13
  38. package/src/providers/register-builtins.ts +4 -4
  39. package/src/stream.ts +61 -6
  40. package/src/types.ts +17 -6
  41. package/src/utils/discovery/openai-compatible.ts +18 -2
  42. package/src/utils/fallback-transport.ts +79 -6
  43. package/src/utils/http-inspector.ts +1 -0
  44. package/src/utils/idle-iterator.ts +2 -0
  45. package/src/utils/oauth/anthropic.ts +41 -8
  46. package/src/utils/oauth/callback-server.ts +64 -16
  47. package/src/utils/oauth/index.ts +5 -0
  48. package/src/utils/oauth/types.ts +13 -0
@@ -1,5 +1,5 @@
1
1
  import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
2
- import OpenAI from "openai";
2
+ import OpenAI, { APIConnectionTimeoutError } from "openai";
3
3
  import type {
4
4
  ChatCompletionAssistantMessageParam,
5
5
  ChatCompletionChunk,
@@ -47,6 +47,7 @@ import {
47
47
  rewriteCopilotError,
48
48
  } from "../utils/http-inspector";
49
49
  import {
50
+ FirstEventTimeoutError,
50
51
  getOpenAIStreamIdleTimeoutMs,
51
52
  getProviderFirstEventTimeoutFallbackMs,
52
53
  getStreamFirstEventTimeoutMs,
@@ -119,6 +120,69 @@ export function resolveOpenAICompletionsBaseUrlForTest(
119
120
  ): string {
120
121
  return resolveOpenAIProviderBaseUrl(baseUrl, authCredentialType);
121
122
  }
123
+ function appendUrlPath(baseUrl: string | undefined, path: string): string | undefined {
124
+ if (!baseUrl) return undefined;
125
+ const normalizedPath = path.replace(/^\/+/g, "");
126
+ try {
127
+ const parsed = new URL(baseUrl);
128
+ parsed.pathname = `${parsed.pathname.replace(/\/+$/g, "")}/${normalizedPath}`;
129
+ return parsed.toString();
130
+ } catch {
131
+ return `${baseUrl.replace(/\/+$/g, "")}/${normalizedPath}`;
132
+ }
133
+ }
134
+
135
+ type OpenAICompletionsQuery = string;
136
+
137
+ function splitBaseUrlQuery(baseUrl: string | undefined): {
138
+ baseUrl: string | undefined;
139
+ query?: OpenAICompletionsQuery;
140
+ } {
141
+ if (!baseUrl) return { baseUrl };
142
+ try {
143
+ const parsed = new URL(baseUrl);
144
+ if (!parsed.search) return { baseUrl };
145
+ const queryStart = baseUrl.indexOf("?");
146
+ const fragmentStart = baseUrl.indexOf("#", queryStart);
147
+ const query = baseUrl.slice(queryStart + 1, fragmentStart === -1 ? undefined : fragmentStart);
148
+ if (!query) return { baseUrl };
149
+ parsed.search = "";
150
+ return {
151
+ baseUrl: parsed.toString(),
152
+ query,
153
+ };
154
+ } catch {
155
+ return { baseUrl };
156
+ }
157
+ }
158
+
159
+ function hasQueryParameter(query: OpenAICompletionsQuery | undefined, name: string): boolean {
160
+ return query ? new URLSearchParams(query).has(name) : false;
161
+ }
162
+
163
+ function appendRawQuery(url: string, query: OpenAICompletionsQuery | undefined): string {
164
+ if (!query) return url;
165
+ const fragmentStart = url.indexOf("#");
166
+ const beforeFragment = fragmentStart === -1 ? url : url.slice(0, fragmentStart);
167
+ const fragment = fragmentStart === -1 ? "" : url.slice(fragmentStart);
168
+ return `${beforeFragment}${beforeFragment.includes("?") ? "&" : "?"}${query}${fragment}`;
169
+ }
170
+
171
+ function buildRequestUrl(
172
+ baseUrl: string | undefined,
173
+ path: string,
174
+ query?: OpenAICompletionsQuery,
175
+ ): string | undefined {
176
+ const url = appendUrlPath(baseUrl, path);
177
+ return url ? appendRawQuery(url, query) : undefined;
178
+ }
179
+
180
+ function appendQueryToRequest(input: string | URL | Request, query?: OpenAICompletionsQuery): string | URL | Request {
181
+ if (!query) return input;
182
+ const url = appendRawQuery(input instanceof Request ? input.url : String(input), query);
183
+ if (input instanceof Request) return new Request(url, input as unknown as RequestInit);
184
+ return url;
185
+ }
122
186
 
123
187
  /**
124
188
  * Normalize tool call ID for Mistral.
@@ -437,8 +501,6 @@ function getTrailingPartialDeepseekToken(text: string): string {
437
501
  return tail;
438
502
  }
439
503
 
440
- const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
441
-
442
504
  const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
443
505
  "OpenAI completions stream timed out while waiting for the first event";
444
506
 
@@ -452,6 +514,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
452
514
  (async () => {
453
515
  const startTime = Date.now();
454
516
  let firstTokenTime: number | undefined;
517
+ let streamConnected = false;
455
518
  let getCapturedErrorResponse: (() => CapturedHttpErrorResponse | undefined) | undefined;
456
519
 
457
520
  const output: AssistantMessage = createInitialResponsesAssistantMessage(model.api, model.provider, model.id);
@@ -466,6 +529,8 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
466
529
  client,
467
530
  copilotPremiumRequests,
468
531
  baseUrl,
532
+ requestBaseUrl,
533
+ requestQuery,
469
534
  requestHeaders,
470
535
  getCapturedErrorResponse: captureErrorResponse,
471
536
  clearCapturedErrorResponse,
@@ -511,7 +576,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
511
576
  api: output.api,
512
577
  model: model.id,
513
578
  method: "POST",
514
- url: `${baseUrl}/chat/completions`,
579
+ url: buildRequestUrl(requestBaseUrl, "chat/completions", requestQuery),
515
580
  headers: requestHeaders,
516
581
  body: params,
517
582
  };
@@ -575,10 +640,8 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
575
640
  openaiStream = await createCompletionsStream("none");
576
641
  }
577
642
  }
578
- const firstEventFallbackMs =
579
- model.provider === "alibaba-token-plan"
580
- ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
581
- : getProviderFirstEventTimeoutFallbackMs(model.provider);
643
+ streamConnected = true;
644
+ const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
582
645
  const firstEventTimeoutMs =
583
646
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
584
647
  if (premiumRequestsTotal !== undefined) {
@@ -984,20 +1047,25 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
984
1047
  } catch (error) {
985
1048
  for (const block of output.content) delete (block as any).index;
986
1049
  const localAbortReason = abortTracker.getLocalAbortReason();
1050
+ const normalizedError =
1051
+ !streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
1052
+ ? new FirstEventTimeoutError(OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE)
1053
+ : error;
987
1054
  const capturedErrorResponse = getCapturedErrorResponse?.();
988
1055
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
989
1056
  output.errorStatus =
990
- extractHttpStatusFromError(localAbortReason ?? error) ??
1057
+ extractHttpStatusFromError(localAbortReason ?? normalizedError) ??
991
1058
  (localAbortReason ? undefined : capturedErrorResponse?.status);
992
1059
  output.transportFailure = localAbortReason
993
1060
  ? transportFailureFacts(localAbortReason)
994
- : transportFailureFacts(error, capturedErrorResponse);
1061
+ : transportFailureFacts(normalizedError, capturedErrorResponse);
995
1062
  output.errorMessage =
996
- localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse));
1063
+ localAbortReason?.message ??
1064
+ (await finalizeErrorMessage(normalizedError, rawRequestDump, capturedErrorResponse));
997
1065
  // Some providers via OpenRouter include extra details here.
998
- const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
1066
+ const rawMetadata = (normalizedError as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
999
1067
  if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
1000
- output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
1068
+ output.errorMessage = rewriteCopilotError(output.errorMessage, normalizedError, model.provider);
1001
1069
  if (hasContentFilterSafetyCode(capturedErrorResponse)) {
1002
1070
  output.errorKind = "provider_safety_stop";
1003
1071
  }
@@ -1029,6 +1097,8 @@ async function createClient(
1029
1097
  client: OpenAI;
1030
1098
  copilotPremiumRequests: number | undefined;
1031
1099
  baseUrl: string | undefined;
1100
+ requestBaseUrl: string | undefined;
1101
+ requestQuery: OpenAICompletionsQuery | undefined;
1032
1102
  requestHeaders: Record<string, string>;
1033
1103
  getCapturedErrorResponse: () => CapturedHttpErrorResponse | undefined;
1034
1104
  clearCapturedErrorResponse: () => void;
@@ -1103,19 +1173,29 @@ async function createClient(
1103
1173
  }
1104
1174
  // Azure OpenAI requires /deployments/{id}/chat/completions?api-version=YYYY-MM-DD.
1105
1175
  // The generic openai-completions path adds neither, producing silent 404s.
1106
- let azureDefaultQuery: Record<string, string> | undefined;
1176
+ let azureQuery: OpenAICompletionsQuery | undefined;
1107
1177
  if (baseUrl?.includes(".openai.azure.com")) {
1108
- const apiVersion = $env.AZURE_OPENAI_API_VERSION || "2024-10-21";
1109
1178
  if (!baseUrl.includes("/deployments/")) {
1110
- baseUrl = `${baseUrl}/deployments/${model.id}`;
1179
+ baseUrl = appendUrlPath(baseUrl, `deployments/${model.id}`) ?? baseUrl;
1111
1180
  }
1112
- azureDefaultQuery = { "api-version": apiVersion };
1113
1181
  }
1182
+ const { baseUrl: clientBaseUrl, query: endpointQuery } = splitBaseUrlQuery(baseUrl);
1183
+ if (baseUrl?.includes(".openai.azure.com") && !hasQueryParameter(endpointQuery, "api-version")) {
1184
+ azureQuery = new URLSearchParams({
1185
+ "api-version": $env.AZURE_OPENAI_API_VERSION || "2024-10-21",
1186
+ }).toString();
1187
+ }
1188
+ const endpointRequestQuery = endpointQuery;
1189
+ const requestQuery =
1190
+ [endpointRequestQuery, azureQuery].filter((query): query is string => query !== undefined).join("&") || undefined;
1114
1191
  let capturedErrorResponse: CapturedHttpErrorResponse | undefined;
1115
1192
  const baseFetch = fetchOverride ?? fetch;
1116
1193
  const wrappedFetch = Object.assign(
1117
1194
  async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
1118
- const response = await baseFetch(input, init);
1195
+ const response = await baseFetch(
1196
+ appendQueryToRequest(appendQueryToRequest(input, endpointRequestQuery), azureQuery),
1197
+ init,
1198
+ );
1119
1199
  if (response.ok) {
1120
1200
  capturedErrorResponse = undefined;
1121
1201
  return response;
@@ -1158,29 +1238,35 @@ async function createClient(
1158
1238
  // in the IIFE.
1159
1239
  // A caller may raise `StreamOptions.streamFirstEventTimeoutMs` for a slow-
1160
1240
  // before-headers provider; respect it so the SDK doesn't give up before the
1161
- // wrapping watchdog arms. An explicit `0` disables the first-event watchdog,
1241
+ // wrapping watchdog arms. Provider-specific fallbacks apply only when the
1242
+ // caller does not pin a value, so an explicit nonzero override must beat that
1243
+ // fallback even when it is shorter. An explicit `0` disables the watchdog,
1162
1244
  // and the SDK treats `timeout: 0` as an immediate timeout, so do not pass a
1163
1245
  // request timeout in that case.
1164
- const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs());
1246
+ const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
1247
+ const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
1165
1248
  const sdkTimeoutMs =
1166
1249
  streamFirstEventTimeoutOverride === 0
1167
1250
  ? undefined
1168
1251
  : streamFirstEventTimeoutOverride !== undefined
1169
- ? Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
1252
+ ? providerFirstEventFallbackMs !== undefined
1253
+ ? streamFirstEventTimeoutOverride
1254
+ : Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
1170
1255
  : envSdkTimeoutMs;
1171
1256
  return {
1172
1257
  client: new OpenAI({
1173
1258
  apiKey,
1174
- baseURL: baseUrl,
1259
+ baseURL: clientBaseUrl,
1175
1260
  dangerouslyAllowBrowser: true,
1176
1261
  maxRetries: resolveRetryBudget(requestMaxRetries, 5),
1177
1262
  defaultHeaders: headers,
1178
- defaultQuery: azureDefaultQuery,
1179
1263
  fetch: debugFetch,
1180
1264
  ...(sdkTimeoutMs !== undefined ? { timeout: sdkTimeoutMs } : {}),
1181
1265
  }),
1182
1266
  copilotPremiumRequests,
1183
1267
  baseUrl,
1268
+ requestBaseUrl: clientBaseUrl,
1269
+ requestQuery,
1184
1270
  requestHeaders: headers,
1185
1271
  getCapturedErrorResponse: () => capturedErrorResponse,
1186
1272
  clearCapturedErrorResponse: () => {
@@ -0,0 +1,173 @@
1
+ import * as fs from "node:fs/promises";
2
+ import * as net from "node:net";
3
+ import * as os from "node:os";
4
+ import * as path from "node:path";
5
+
6
+ import type { Model } from "../types";
7
+
8
+ export const OPENCODEX_DEFAULT_PORT = 10100;
9
+ export const OPENCODEX_PROBE_TIMEOUT_MS = 750;
10
+ export const OPENCODEX_MODEL_CACHE_TTL_MS = 5 * 60 * 1000;
11
+
12
+ interface RuntimePortFile {
13
+ hostname?: unknown;
14
+ host?: unknown;
15
+ port?: unknown;
16
+ }
17
+
18
+ interface HealthPayload {
19
+ ok?: unknown;
20
+ pid?: unknown;
21
+ port?: unknown;
22
+ version?: unknown;
23
+ }
24
+
25
+ interface CatalogRow {
26
+ id?: unknown;
27
+ model?: unknown;
28
+ name?: unknown;
29
+ displayName?: unknown;
30
+ contextWindow?: unknown;
31
+ maxTokens?: unknown;
32
+ reasoning?: unknown;
33
+ input?: unknown;
34
+ }
35
+
36
+ export interface OpenCodexEndpoint {
37
+ baseUrl: string;
38
+ }
39
+
40
+ function timeoutSignal(signal?: AbortSignal): AbortSignal {
41
+ return signal
42
+ ? AbortSignal.any([signal, AbortSignal.timeout(OPENCODEX_PROBE_TIMEOUT_MS)])
43
+ : AbortSignal.timeout(OPENCODEX_PROBE_TIMEOUT_MS);
44
+ }
45
+
46
+ function normalizeEndpoint(hostname: string, port: number): string | undefined {
47
+ if (!Number.isInteger(port) || port < 1 || port > 65535) return undefined;
48
+ const host = normalizeLoopbackHost(hostname);
49
+ if (!host) return undefined;
50
+ return `http://${formatEndpointHost(host)}:${port}`;
51
+ }
52
+
53
+ function normalizeLoopbackHost(hostname: string): string | undefined {
54
+ const host = hostname.trim().toLowerCase();
55
+ if (net.isIP(host) === 4 && host.startsWith("127.")) return host;
56
+ if (host === "::1") return host;
57
+ return undefined;
58
+ }
59
+
60
+ function formatEndpointHost(host: string): string {
61
+ return host.includes(":") ? `[${host}]` : host;
62
+ }
63
+
64
+ function healthPort(endpoint: string): number {
65
+ return Number(new URL(endpoint).port);
66
+ }
67
+
68
+ async function readRuntimeEndpoint(): Promise<string | undefined> {
69
+ const home = process.env.OPENCODEX_HOME?.trim() || path.join(os.homedir(), ".opencodex");
70
+ try {
71
+ const raw = JSON.parse(await fs.readFile(path.join(home, "runtime-port.json"), "utf8")) as RuntimePortFile;
72
+ const hostname =
73
+ typeof raw.hostname === "string" ? raw.hostname : typeof raw.host === "string" ? raw.host : "127.0.0.1";
74
+ const port = typeof raw.port === "number" ? raw.port : typeof raw.port === "string" ? Number(raw.port) : NaN;
75
+ return normalizeEndpoint(hostname, port);
76
+ } catch {
77
+ return undefined;
78
+ }
79
+ }
80
+
81
+ function candidateEndpoints(runtimeEndpoint: string | undefined): string[] {
82
+ const candidates = runtimeEndpoint ? [runtimeEndpoint] : [];
83
+ const fallback = normalizeEndpoint("127.0.0.1", OPENCODEX_DEFAULT_PORT);
84
+ if (fallback && !candidates.includes(fallback)) candidates.push(fallback);
85
+ return candidates;
86
+ }
87
+
88
+ async function fetchJson(url: string, signal?: AbortSignal): Promise<unknown> {
89
+ const response = await fetch(url, {
90
+ headers: { Accept: "application/json" },
91
+ redirect: "error",
92
+ signal: timeoutSignal(signal),
93
+ });
94
+ if (!response.ok) return undefined;
95
+ return response.json();
96
+ }
97
+
98
+ function isOpenCodexHealth(payload: unknown, expectedPort: number): boolean {
99
+ if (!payload || typeof payload !== "object" || Array.isArray(payload)) return false;
100
+ const health = payload as HealthPayload;
101
+ return health.ok === true && health.version === "opencodex" && health.port === expectedPort;
102
+ }
103
+
104
+ export async function resolveOpenCodexEndpoint(signal?: AbortSignal): Promise<OpenCodexEndpoint | undefined> {
105
+ const runtimeEndpoint = await readRuntimeEndpoint();
106
+ for (const candidate of candidateEndpoints(runtimeEndpoint)) {
107
+ try {
108
+ const health = await fetchJson(`${candidate}/healthz`, signal);
109
+ if (isOpenCodexHealth(health, healthPort(candidate))) return { baseUrl: candidate };
110
+ } catch {
111
+ // An unavailable or foreign listener is a normal provider absence.
112
+ }
113
+ }
114
+ return undefined;
115
+ }
116
+
117
+ function asPositiveNumber(value: unknown, fallback: number): number {
118
+ return typeof value === "number" && Number.isFinite(value) && value > 0 ? value : fallback;
119
+ }
120
+
121
+ function normalizeCatalogPayload(payload: unknown): CatalogRow[] {
122
+ if (Array.isArray(payload)) return payload as CatalogRow[];
123
+ if (payload && typeof payload === "object" && Array.isArray((payload as { models?: unknown }).models)) {
124
+ return (payload as { models: CatalogRow[] }).models;
125
+ }
126
+ return [];
127
+ }
128
+
129
+ function normalizeModel(row: CatalogRow, endpoint: OpenCodexEndpoint): Model<"openai-responses"> | undefined {
130
+ const rawId = typeof row.id === "string" ? row.id.trim() : typeof row.model === "string" ? row.model.trim() : "";
131
+ if (!rawId || rawId.includes("\n")) return undefined;
132
+ const publicId = `opencodex/${rawId}`;
133
+ const input =
134
+ Array.isArray(row.input) && row.input.every(value => value === "text" || value === "image")
135
+ ? row.input
136
+ : ["text"];
137
+ return {
138
+ id: publicId,
139
+ wireModelId: rawId,
140
+ name: typeof row.displayName === "string" ? row.displayName : typeof row.name === "string" ? row.name : rawId,
141
+ api: "openai-responses",
142
+ provider: "opencodex",
143
+ baseUrl: `${endpoint.baseUrl}/v1`,
144
+ reasoning: row.reasoning !== false,
145
+ input,
146
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
147
+ contextWindow: asPositiveNumber(row.contextWindow, 128_000),
148
+ maxTokens: asPositiveNumber(row.maxTokens, 16_384),
149
+ };
150
+ }
151
+
152
+ export async function fetchOpenCodexModels(): Promise<readonly Model<"openai-responses">[] | null> {
153
+ const endpoint = await resolveOpenCodexEndpoint();
154
+ if (!endpoint) return null;
155
+ try {
156
+ const rows = normalizeCatalogPayload(await fetchJson(`${endpoint.baseUrl}/api/models`));
157
+ const models = rows
158
+ .map(row => normalizeModel(row, endpoint))
159
+ .filter((model): model is Model<"openai-responses"> => model !== undefined);
160
+ return models.length > 0 ? models : null;
161
+ } catch {
162
+ return null;
163
+ }
164
+ }
165
+
166
+ export async function checkOpenCodexStatus(onProgress?: (message: string) => void): Promise<void> {
167
+ const endpoint = await resolveOpenCodexEndpoint();
168
+ if (endpoint) {
169
+ onProgress?.(`OpenCodex is available at ${endpoint.baseUrl}`);
170
+ return;
171
+ }
172
+ onProgress?.("OpenCodex is unavailable; no identity-checked local proxy was found.");
173
+ }
@@ -960,20 +960,31 @@ export function populateResponsesUsageFromResponse(
960
960
  input_tokens?: number | null;
961
961
  output_tokens?: number | null;
962
962
  total_tokens?: number | null;
963
- input_tokens_details?: { cached_tokens?: number | null } | null;
963
+ input_tokens_details?: {
964
+ cached_tokens?: number | null;
965
+ cache_write_tokens?: number | null;
966
+ } | null;
964
967
  output_tokens_details?: { reasoning_tokens?: number | null } | null;
965
968
  }
966
969
  | null
967
970
  | undefined,
968
971
  ): void {
969
972
  if (!usage) return;
973
+ const inputTokens = usage.input_tokens || 0;
970
974
  const cachedTokens = usage.input_tokens_details?.cached_tokens || 0;
975
+ const reportedCacheWrite = usage.input_tokens_details?.cache_write_tokens || 0;
976
+ const cacheWriteTokens =
977
+ Number.isSafeInteger(reportedCacheWrite) &&
978
+ reportedCacheWrite >= 0 &&
979
+ cachedTokens + reportedCacheWrite <= inputTokens
980
+ ? reportedCacheWrite
981
+ : 0;
971
982
  const reasoningTokens = usage.output_tokens_details?.reasoning_tokens || 0;
972
983
  output.usage = {
973
- input: (usage.input_tokens || 0) - cachedTokens,
984
+ input: Math.max(0, inputTokens - cachedTokens - cacheWriteTokens),
974
985
  output: usage.output_tokens || 0,
975
986
  cacheRead: cachedTokens,
976
- cacheWrite: 0,
987
+ cacheWrite: cacheWriteTokens,
977
988
  totalTokens: usage.total_tokens || 0,
978
989
  ...(reasoningTokens > 0 ? { reasoningTokens } : {}),
979
990
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
@@ -1,5 +1,5 @@
1
1
  import { $credentialEnv, extractHttpStatusFromError, logger, structuredCloneJSON } from "@gajae-code/utils";
2
- import OpenAI from "openai";
2
+ import OpenAI, { APIConnectionTimeoutError } from "openai";
3
3
  import type {
4
4
  Tool as OpenAITool,
5
5
  ResponseCreateParamsStreaming,
@@ -38,7 +38,9 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
38
38
  import { transportFailureFacts } from "../utils/fallback-transport";
39
39
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
40
40
  import {
41
+ FirstEventTimeoutError,
41
42
  getOpenAIStreamIdleTimeoutMs,
43
+ getProviderFirstEventTimeoutFallbackMs,
42
44
  getStreamFirstEventTimeoutMs,
43
45
  iterateWithIdleTimeout,
44
46
  } from "../utils/idle-iterator";
@@ -124,7 +126,6 @@ export interface OpenAIResponsesOptions extends StreamOptions {
124
126
  }
125
127
 
126
128
  const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
127
- const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
128
129
  const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
129
130
  "OpenAI responses stream timed out while waiting for the first event";
130
131
  const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
@@ -172,6 +173,62 @@ export function resolveOpenAIProviderBaseUrlForTest(
172
173
  return resolveOpenAIProviderBaseUrl(baseUrl, authCredentialType);
173
174
  }
174
175
 
176
+ function appendUrlPath(baseUrl: string | undefined, path: string): string | undefined {
177
+ if (!baseUrl) return undefined;
178
+ const normalizedPath = path.replace(/^\/+/g, "");
179
+ try {
180
+ const parsed = new URL(baseUrl);
181
+ parsed.pathname = `${parsed.pathname.replace(/\/+$/g, "")}/${normalizedPath}`;
182
+ return parsed.toString();
183
+ } catch {
184
+ return `${baseUrl.replace(/\/+$/g, "")}/${normalizedPath}`;
185
+ }
186
+ }
187
+
188
+ type OpenAIResponsesQuery = string;
189
+
190
+ function splitBaseUrlQuery(baseUrl: string | undefined): {
191
+ baseUrl: string | undefined;
192
+ query?: OpenAIResponsesQuery;
193
+ } {
194
+ if (!baseUrl) return { baseUrl };
195
+ try {
196
+ const parsed = new URL(baseUrl);
197
+ if (!parsed.search) return { baseUrl };
198
+ const queryStart = baseUrl.indexOf("?");
199
+ const fragmentStart = baseUrl.indexOf("#", queryStart);
200
+ const query = baseUrl.slice(queryStart + 1, fragmentStart === -1 ? undefined : fragmentStart);
201
+ if (!query) return { baseUrl };
202
+ parsed.search = "";
203
+ return {
204
+ baseUrl: parsed.toString(),
205
+ query,
206
+ };
207
+ } catch {
208
+ return { baseUrl };
209
+ }
210
+ }
211
+
212
+ function appendRawQuery(url: string, query: OpenAIResponsesQuery | undefined): string {
213
+ if (!query) return url;
214
+ const fragmentStart = url.indexOf("#");
215
+ const beforeFragment = fragmentStart === -1 ? url : url.slice(0, fragmentStart);
216
+ const fragment = fragmentStart === -1 ? "" : url.slice(fragmentStart);
217
+ return `${beforeFragment}${beforeFragment.includes("?") ? "&" : "?"}${query}${fragment}`;
218
+ }
219
+
220
+ function buildRequestUrl(baseUrl: string | undefined, path: string, query?: OpenAIResponsesQuery): string | undefined {
221
+ const url = appendUrlPath(baseUrl, path);
222
+ return url ? appendRawQuery(url, query) : undefined;
223
+ }
224
+
225
+ function appendQueryToRequest(input: string | URL | Request, query?: OpenAIResponsesQuery): string | URL | Request {
226
+ if (!query) return input;
227
+ const url = appendRawQuery(input instanceof Request ? input.url : String(input), query);
228
+ if (input instanceof Request) return new Request(url, input as unknown as RequestInit);
229
+ return url;
230
+ }
231
+
175
232
  const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
176
233
  "response.created",
177
234
  "response.output_item.added",
@@ -258,6 +315,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
258
315
  (async () => {
259
316
  const startTime = Date.now();
260
317
  let firstTokenTime: number | undefined;
318
+ let streamConnected = false;
261
319
 
262
320
  const output: AssistantMessage = createInitialResponsesAssistantMessage(
263
321
  "openai-responses",
@@ -272,7 +330,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
272
330
  // Keep request headers and prompt-cache routing on the same session-derived value.
273
331
  const cacheSessionId = getOpenAIResponsesCacheSessionId(options);
274
332
  const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
275
- const { client, copilotPremiumRequests, baseUrl } = createClient(
333
+ const { client, copilotPremiumRequests, baseUrl, requestBaseUrl, requestQuery } = createClient(
276
334
  model,
277
335
  context,
278
336
  apiKey,
@@ -296,7 +354,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
296
354
  api: output.api,
297
355
  model: model.id,
298
356
  method: "POST",
299
- url: `${baseUrl}/responses`,
357
+ url: buildRequestUrl(requestBaseUrl, "responses", requestQuery),
300
358
  body: params,
301
359
  };
302
360
  const openaiStream = await callWithCopilotModelRetry(
@@ -336,8 +394,8 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
336
394
  await notifyProviderResponse(options, response, model, request_id);
337
395
  return data;
338
396
  });
339
- const firstEventFallbackMs =
340
- model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
397
+ streamConnected = true;
398
+ const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
341
399
  const firstEventTimeoutMs =
342
400
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
343
401
  if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
@@ -391,11 +449,16 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
391
449
  } catch (error) {
392
450
  for (const block of output.content) delete (block as { index?: number }).index;
393
451
  const localAbortReason = abortTracker.getLocalAbortReason();
452
+ const normalizedError =
453
+ !streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
454
+ ? new FirstEventTimeoutError(OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE)
455
+ : error;
394
456
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
395
- output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
396
- output.transportFailure = transportFailureFacts(localAbortReason ?? error);
397
- output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
398
- output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
457
+ output.errorStatus = extractHttpStatusFromError(localAbortReason ?? normalizedError);
458
+ output.transportFailure = transportFailureFacts(localAbortReason ?? normalizedError);
459
+ output.errorMessage =
460
+ localAbortReason?.message ?? (await finalizeErrorMessage(normalizedError, rawRequestDump));
461
+ output.errorMessage = rewriteCopilotError(output.errorMessage, normalizedError, model.provider);
399
462
  // Explicitly mark the poisoned-history rejection so the shared
400
463
  // `invalid_prompt` contract is present even when the SDK error surfaces
401
464
  // only a message (no structured code). This keeps the responses
@@ -439,6 +502,8 @@ function createClient(
439
502
  client: OpenAI;
440
503
  copilotPremiumRequests: number | undefined;
441
504
  baseUrl: string | undefined;
505
+ requestBaseUrl: string | undefined;
506
+ requestQuery: OpenAIResponsesQuery | undefined;
442
507
  } {
443
508
  if (!apiKey) {
444
509
  apiKey = $credentialEnv("OPENAI_API_KEY");
@@ -489,8 +554,15 @@ function createClient(
489
554
  headers.session_id ??= sessionId;
490
555
  headers["x-client-request-id"] ??= sessionId;
491
556
  }
557
+ const { baseUrl: clientBaseUrl, query: endpointQuery } = splitBaseUrlQuery(baseUrl);
492
558
  const baseFetch = fetchOverride ?? fetch;
493
- const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
559
+ const queryFetch = Object.assign(
560
+ async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
561
+ return baseFetch(appendQueryToRequest(input, endpointQuery), init);
562
+ },
563
+ baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
564
+ );
565
+ const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(queryFetch, maxRetryDelayMs);
494
566
  const transformedFetch = wrapFetchForOpenAIRequestTransform(
495
567
  boundedFetch,
496
568
  model.requestTransform,
@@ -499,7 +571,7 @@ function createClient(
499
571
  return {
500
572
  client: new OpenAI({
501
573
  apiKey,
502
- baseURL: baseUrl,
574
+ baseURL: clientBaseUrl,
503
575
  dangerouslyAllowBrowser: true,
504
576
  maxRetries: resolveRetryBudget(requestMaxRetries, 5),
505
577
  defaultHeaders: headers,
@@ -509,6 +581,8 @@ function createClient(
509
581
  }),
510
582
  copilotPremiumRequests,
511
583
  baseUrl,
584
+ requestBaseUrl: clientBaseUrl,
585
+ requestQuery: endpointQuery,
512
586
  };
513
587
  }
514
588
 
@@ -764,7 +838,7 @@ function isForcedOpenAIResponsesToolChoice(choice: unknown): boolean {
764
838
  /** @internal Exported for tests. */
765
839
  export function convertTools(tools: Tool[], strictMode: boolean, model: Model<"openai-responses">): OpenAITool[] {
766
840
  const allowFreeform = supportsFreeformApplyPatch(model);
767
- return tools.map(tool => {
841
+ const payloads = tools.map(tool => {
768
842
  if (allowFreeform && tool.customFormat) {
769
843
  return {
770
844
  type: "custom",
@@ -792,4 +866,8 @@ export function convertTools(tools: Tool[], strictMode: boolean, model: Model<"o
792
866
  ...(effectiveStrict && { strict: true }),
793
867
  } as OpenAITool;
794
868
  });
869
+ // Tool definitions bypass the `input`/`instructions` sanitizers, so a
870
+ // leaked Harmony marker in an MCP/skill tool description or schema string
871
+ // rejects every gpt-5.x request (`Request blocked`).
872
+ return neutralizeResponsesInputControlTokens(payloads);
795
873
  }