@gajae-code/ai 0.12.8 → 0.12.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/dist/types/auth-storage.d.ts +25 -2
  3. package/dist/types/index.d.ts +1 -1
  4. package/dist/types/model-pricing.d.ts +3 -0
  5. package/dist/types/providers/composer-discipline.d.ts +29 -23
  6. package/dist/types/providers/openai-responses-shared.d.ts +1 -0
  7. package/dist/types/providers/register-builtins.d.ts +2 -2
  8. package/dist/types/types.d.ts +14 -6
  9. package/dist/types/utils/fallback-transport.d.ts +23 -0
  10. package/dist/types/utils/oauth/anthropic.d.ts +21 -2
  11. package/dist/types/utils/oauth/callback-server.d.ts +7 -0
  12. package/dist/types/utils/oauth/types.d.ts +11 -0
  13. package/package.json +2 -2
  14. package/src/auth-gateway/server.ts +6 -0
  15. package/src/auth-storage.ts +46 -5
  16. package/src/index.ts +1 -0
  17. package/src/model-manager.ts +4 -1
  18. package/src/model-pricing.ts +68 -0
  19. package/src/model-thinking.ts +23 -1
  20. package/src/models.json +122 -24
  21. package/src/models.ts +7 -4
  22. package/src/prompts/composer-bash-policy-recovery.md +1 -0
  23. package/src/prompts/cursor-composer-bash-policy-recovery.md +1 -0
  24. package/src/prompts/cursor-composer-edit-discipline.md +7 -0
  25. package/src/providers/composer-discipline.ts +54 -0
  26. package/src/providers/cursor.ts +2 -2
  27. package/src/providers/openai-completions.ts +23 -15
  28. package/src/providers/openai-responses-shared.ts +14 -3
  29. package/src/providers/openai-responses.ts +15 -8
  30. package/src/providers/register-builtins.ts +4 -4
  31. package/src/stream.ts +60 -5
  32. package/src/types.ts +16 -6
  33. package/src/utils/fallback-transport.ts +79 -6
  34. package/src/utils/idle-iterator.ts +2 -0
  35. package/src/utils/oauth/anthropic.ts +41 -8
  36. package/src/utils/oauth/callback-server.ts +64 -16
  37. package/src/utils/oauth/types.ts +12 -0
package/src/models.json CHANGED
@@ -1,5 +1,40 @@
1
1
  {
2
2
  "alibaba-token-plan": {
3
+ "deepseek-v4-flash-0731": {
4
+ "id": "deepseek-v4-flash-0731",
5
+ "name": "DeepSeek V4 Flash 0731",
6
+ "api": "openai-completions",
7
+ "provider": "alibaba-token-plan",
8
+ "baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
9
+ "reasoning": true,
10
+ "input": [
11
+ "text"
12
+ ],
13
+ "cost": {
14
+ "input": 0,
15
+ "output": 0,
16
+ "cacheRead": 0,
17
+ "cacheWrite": 0
18
+ },
19
+ "contextWindow": 1000000,
20
+ "maxTokens": 384000,
21
+ "compat": {
22
+ "supportsDeveloperRole": false,
23
+ "supportsReasoningEffort": true,
24
+ "reasoningContentField": "reasoning_content",
25
+ "requiresReasoningContentForToolCalls": true
26
+ },
27
+ "thinking": {
28
+ "mode": "effort",
29
+ "minLevel": "low",
30
+ "maxLevel": "max",
31
+ "levels": [
32
+ "low",
33
+ "high",
34
+ "max"
35
+ ]
36
+ }
37
+ },
3
38
  "deepseek-v4-pro": {
4
39
  "id": "deepseek-v4-pro",
5
40
  "name": "DeepSeek V4 Pro",
@@ -58436,7 +58471,16 @@
58436
58471
  "minLevel": "low",
58437
58472
  "maxLevel": "max"
58438
58473
  },
58439
- "applyPatchToolType": "freeform"
58474
+ "applyPatchToolType": "freeform",
58475
+ "longContextPricing": {
58476
+ "threshold": 272000,
58477
+ "cost": {
58478
+ "input": 10,
58479
+ "output": 45,
58480
+ "cacheRead": 1,
58481
+ "cacheWrite": 12.5
58482
+ }
58483
+ }
58440
58484
  },
58441
58485
  "gpt-5.6-luna": {
58442
58486
  "id": "gpt-5.6-luna",
@@ -58450,10 +58494,10 @@
58450
58494
  "image"
58451
58495
  ],
58452
58496
  "cost": {
58453
- "input": 1,
58454
- "output": 6,
58455
- "cacheRead": 0.1,
58456
- "cacheWrite": 1.25
58497
+ "input": 0.2,
58498
+ "output": 1.2,
58499
+ "cacheRead": 0.02,
58500
+ "cacheWrite": 0.25
58457
58501
  },
58458
58502
  "contextWindow": 1050000,
58459
58503
  "maxTokens": 128000,
@@ -58462,7 +58506,16 @@
58462
58506
  "minLevel": "low",
58463
58507
  "maxLevel": "max"
58464
58508
  },
58465
- "applyPatchToolType": "freeform"
58509
+ "applyPatchToolType": "freeform",
58510
+ "longContextPricing": {
58511
+ "threshold": 272000,
58512
+ "cost": {
58513
+ "input": 0.4,
58514
+ "output": 1.8,
58515
+ "cacheRead": 0.04,
58516
+ "cacheWrite": 0.5
58517
+ }
58518
+ }
58466
58519
  },
58467
58520
  "gpt-5.6-sol": {
58468
58521
  "id": "gpt-5.6-sol",
@@ -58488,7 +58541,16 @@
58488
58541
  "minLevel": "low",
58489
58542
  "maxLevel": "max"
58490
58543
  },
58491
- "applyPatchToolType": "freeform"
58544
+ "applyPatchToolType": "freeform",
58545
+ "longContextPricing": {
58546
+ "threshold": 272000,
58547
+ "cost": {
58548
+ "input": 10,
58549
+ "output": 45,
58550
+ "cacheRead": 1,
58551
+ "cacheWrite": 12.5
58552
+ }
58553
+ }
58492
58554
  },
58493
58555
  "gpt-5.6-terra": {
58494
58556
  "id": "gpt-5.6-terra",
@@ -58502,10 +58564,10 @@
58502
58564
  "image"
58503
58565
  ],
58504
58566
  "cost": {
58505
- "input": 2.5,
58506
- "output": 15,
58507
- "cacheRead": 0.25,
58508
- "cacheWrite": 3.125
58567
+ "input": 2,
58568
+ "output": 12,
58569
+ "cacheRead": 0.2,
58570
+ "cacheWrite": 2.5
58509
58571
  },
58510
58572
  "contextWindow": 1050000,
58511
58573
  "maxTokens": 128000,
@@ -58514,7 +58576,16 @@
58514
58576
  "minLevel": "low",
58515
58577
  "maxLevel": "max"
58516
58578
  },
58517
- "applyPatchToolType": "freeform"
58579
+ "applyPatchToolType": "freeform",
58580
+ "longContextPricing": {
58581
+ "threshold": 272000,
58582
+ "cost": {
58583
+ "input": 4,
58584
+ "output": 18,
58585
+ "cacheRead": 0.4,
58586
+ "cacheWrite": 5
58587
+ }
58588
+ }
58518
58589
  },
58519
58590
  "gpt-image-2": {
58520
58591
  "id": "gpt-image-2",
@@ -59224,10 +59295,10 @@
59224
59295
  "image"
59225
59296
  ],
59226
59297
  "cost": {
59227
- "input": 1,
59228
- "output": 6,
59229
- "cacheRead": 0.1,
59230
- "cacheWrite": 1.25
59298
+ "input": 0.2,
59299
+ "output": 1.2,
59300
+ "cacheRead": 0.02,
59301
+ "cacheWrite": 0.25
59231
59302
  },
59232
59303
  "contextWindow": 272000,
59233
59304
  "maxTokens": 128000,
@@ -59238,7 +59309,16 @@
59238
59309
  "minLevel": "low",
59239
59310
  "maxLevel": "max"
59240
59311
  },
59241
- "applyPatchToolType": "freeform"
59312
+ "applyPatchToolType": "freeform",
59313
+ "longContextPricing": {
59314
+ "threshold": 272000,
59315
+ "cost": {
59316
+ "input": 0.4,
59317
+ "output": 1.8,
59318
+ "cacheRead": 0.04,
59319
+ "cacheWrite": 0.5
59320
+ }
59321
+ }
59242
59322
  },
59243
59323
  "gpt-5.6-sol": {
59244
59324
  "id": "gpt-5.6-sol",
@@ -59266,7 +59346,16 @@
59266
59346
  "minLevel": "low",
59267
59347
  "maxLevel": "max"
59268
59348
  },
59269
- "applyPatchToolType": "freeform"
59349
+ "applyPatchToolType": "freeform",
59350
+ "longContextPricing": {
59351
+ "threshold": 272000,
59352
+ "cost": {
59353
+ "input": 10,
59354
+ "output": 45,
59355
+ "cacheRead": 1,
59356
+ "cacheWrite": 12.5
59357
+ }
59358
+ }
59270
59359
  },
59271
59360
  "gpt-5.6-terra": {
59272
59361
  "id": "gpt-5.6-terra",
@@ -59280,10 +59369,10 @@
59280
59369
  "image"
59281
59370
  ],
59282
59371
  "cost": {
59283
- "input": 2.5,
59284
- "output": 15,
59285
- "cacheRead": 0.25,
59286
- "cacheWrite": 3.125
59372
+ "input": 2,
59373
+ "output": 12,
59374
+ "cacheRead": 0.2,
59375
+ "cacheWrite": 2.5
59287
59376
  },
59288
59377
  "contextWindow": 272000,
59289
59378
  "maxTokens": 128000,
@@ -59294,7 +59383,16 @@
59294
59383
  "minLevel": "low",
59295
59384
  "maxLevel": "max"
59296
59385
  },
59297
- "applyPatchToolType": "freeform"
59386
+ "applyPatchToolType": "freeform",
59387
+ "longContextPricing": {
59388
+ "threshold": 272000,
59389
+ "cost": {
59390
+ "input": 4,
59391
+ "output": 18,
59392
+ "cacheRead": 0.4,
59393
+ "cacheWrite": 5
59394
+ }
59395
+ }
59298
59396
  },
59299
59397
  "gpt-image-2": {
59300
59398
  "id": "gpt-image-2",
@@ -85511,4 +85609,4 @@
85511
85609
  }
85512
85610
  }
85513
85611
  }
85514
- }
85612
+ }
package/src/models.ts CHANGED
@@ -1,4 +1,5 @@
1
1
  import { readFileSync } from "node:fs";
2
+ import { getOpenAIModelCost } from "./model-pricing";
2
3
  import { isRetiredModelKey } from "./model-retirements";
3
4
  import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
4
5
  // `with { type: "file" }` is embedded by `bun build --compile` and resolves to
@@ -92,10 +93,12 @@ export function getBundledModels(provider: GeneratedProvider): Model<Api>[] {
92
93
  }
93
94
 
94
95
  export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage): Usage["cost"] {
95
- usage.cost.input = (model.cost.input / 1000000) * usage.input;
96
- usage.cost.output = (model.cost.output / 1000000) * usage.output;
97
- usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * usage.cacheRead;
98
- usage.cost.cacheWrite = (model.cost.cacheWrite / 1000000) * usage.cacheWrite;
96
+ const inputTokens = usage.input + usage.cacheRead + usage.cacheWrite;
97
+ const pricing = getOpenAIModelCost(model, inputTokens) ?? model.cost;
98
+ usage.cost.input = (pricing.input / 1000000) * usage.input;
99
+ usage.cost.output = (pricing.output / 1000000) * usage.output;
100
+ usage.cost.cacheRead = (pricing.cacheRead / 1000000) * usage.cacheRead;
101
+ usage.cost.cacheWrite = (pricing.cacheWrite / 1000000) * usage.cacheWrite;
99
102
  usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite;
100
103
  return usage.cost;
101
104
  }
@@ -0,0 +1 @@
1
+ A Composer bash policy block interrupted a shell attempt. This is not a terminal condition: continue the same task now. Do not retry repository file I/O in bash. Use the dedicated find, search, read, and edit tools for repository work, then continue with the next safe step.
@@ -0,0 +1 @@
1
+ A Composer bash policy block interrupted a shell attempt. This is not a terminal condition: continue the same task now. Do not retry repository file I/O in shell. Use Cursor-native read for files or directories, grep for search or globs, write for changes, and delete only when deletion is required; then continue with the next safe step.
@@ -0,0 +1,7 @@
1
+ File-editing discipline for this Cursor Composer harness (this OVERRIDES contrary habits from your training):
2
+
3
+ - Inspect repository files ONLY with Cursor-native read and grep: use read for file bodies or directories, and grep for content search or glob discovery. NEVER inspect repository files through shell commands (ls, find, fd, cat, sed, awk, grep, rg, head, tail, less, more) or scripts.
4
+ - Modify files ONLY with Cursor-native write, or delete only when deletion is required. NEVER mutate files through shell redirection, tee, sed -i, perl -pi, inline python/node/bun scripts, or other out-of-band writes.
5
+ - Re-read a file after any write before relying on its contents again. Do not fabricate line anchors, paths, tool names, or tool-call arguments.
6
+ - Tool-call arguments must be the exact schema object requested by the native tool. Do not include Markdown, commentary, analysis text, or invented fields inside tool arguments.
7
+ - Use shell only for terminal operations such as tests, builds, package scripts, and git commands. A shell command string must contain only the command itself; NEVER interleave reasoning or commentary into command strings or heredocs.
@@ -1,3 +1,9 @@
1
+ import composerBashPolicyRecoveryPrompt from "../prompts/composer-bash-policy-recovery.md" with { type: "text" };
2
+ import cursorComposerBashPolicyRecoveryPrompt from "../prompts/cursor-composer-bash-policy-recovery.md" with {
3
+ type: "text",
4
+ };
5
+ import cursorComposerEditDisciplinePrompt from "../prompts/cursor-composer-edit-discipline.md" with { type: "text" };
6
+
1
7
  /**
2
8
  * Anchor/edit discipline for composer-harness models (xai grok-composer-*,
3
9
  * cursor composer-*).
@@ -30,6 +36,47 @@ export function isComposerHarnessModel(modelId: string): boolean {
30
36
  return COMPOSER_MODEL_ID_PATTERN.test(modelId);
31
37
  }
32
38
 
39
+ /** Stable text contract for a local shell rejection caused by Composer file-I/O discipline. */
40
+ export const COMPOSER_BASH_POLICY_ERROR_PREFIX = "Composer bash policy blocked repository file I/O.";
41
+ export const COMPOSER_BASH_POLICY_ERROR_CODE = "composer-bash-policy:repository-file-io";
42
+
43
+ export type ComposerBashPolicyToolSurface = "generic" | "cursor";
44
+
45
+ /**
46
+ * Format the model-visible policy rejection with a stable marker and the tool
47
+ * vocabulary the model actually receives on this provider surface.
48
+ */
49
+ export function formatComposerBashPolicyError(surface: ComposerBashPolicyToolSurface = "generic"): string {
50
+ const recovery =
51
+ surface === "cursor"
52
+ ? "Continue the same task with Cursor-native read, grep, write, or delete tools; do not retry repository file I/O through shell."
53
+ : "Continue the same task with find, search, read, and edit tools; do not retry repository file I/O through bash.";
54
+ return `${COMPOSER_BASH_POLICY_ERROR_PREFIX} [${COMPOSER_BASH_POLICY_ERROR_CODE}] Recovery required: ${recovery}`;
55
+ }
56
+
57
+ /**
58
+ * Matches both the structured current error and the original prefix so a
59
+ * resumed session can recover after an upgrade without string-version skew.
60
+ */
61
+ export function isComposerBashPolicyBlockedError(text: string): boolean {
62
+ return text.includes(COMPOSER_BASH_POLICY_ERROR_PREFIX);
63
+ }
64
+
65
+ /**
66
+ * Matches only errors emitted directly by the current policy implementation.
67
+ * Live recovery must use this strict form so failed shell output that merely
68
+ * quotes a policy error cannot masquerade as the policy gate itself.
69
+ */
70
+ export function isCurrentComposerBashPolicyBlockedError(text: string): boolean {
71
+ return text === formatComposerBashPolicyError("generic") || text === formatComposerBashPolicyError("cursor");
72
+ }
73
+
74
+ /** One bounded, tool-enabled retry instruction for generic Composer agent loops. */
75
+ export const COMPOSER_BASH_POLICY_RECOVERY_PROMPT = composerBashPolicyRecoveryPrompt;
76
+
77
+ /** One bounded, tool-enabled retry instruction for Cursor's native remote tool surface. */
78
+ export const CURSOR_COMPOSER_BASH_POLICY_RECOVERY_PROMPT = cursorComposerBashPolicyRecoveryPrompt;
79
+
33
80
  export const COMPOSER_EDIT_DISCIPLINE_PROMPT = `File-editing discipline for this Composer harness (this OVERRIDES contrary habits from your training):
34
81
 
35
82
  - Discover file names ONLY with the find tool; search file contents ONLY with the search tool; read file bodies or line ranges ONLY with the read tool. NEVER inspect repository files through shell commands (ls, find, fd, cat, sed, awk, grep, rg, head, tail, less, more) or scripts — that output carries no hashline anchors and bypasses the agent's safety limits.
@@ -39,3 +86,10 @@ export const COMPOSER_EDIT_DISCIPLINE_PROMPT = `File-editing discipline for this
39
86
  - If an edit is rejected with "anchors do not match", the rejection message prints the current lines WITH fresh anchors. Retry using exactly those printed anchors.
40
87
  - Tool-call arguments must be the exact JSON/schema object requested by the tool. Do not include Markdown, commentary, analysis text, or invented fields inside tool arguments.
41
88
  - Use bash only for terminal operations such as tests, builds, package scripts, and git commands. A shell command string must contain only the command itself; NEVER interleave reasoning or commentary into command strings or heredocs.`;
89
+
90
+ /**
91
+ * Cursor executes a different native tool vocabulary from the generic agent
92
+ * loop. Keep this prompt separate so Composer is never told to call `edit`,
93
+ * `find`, or `search` when those names are unavailable remotely.
94
+ */
95
+ export const CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT = cursorComposerEditDisciplinePrompt;
@@ -30,7 +30,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
30
30
  import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
32
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
33
- import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
33
+ import { CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
34
34
  import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
35
35
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
36
36
  import {
@@ -2329,7 +2329,7 @@ export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | u
2329
2329
  // Composer-harness models need anchor/edit discipline pinned ahead of any
2330
2330
  // host/default prompt (see composer-discipline.ts for the observed failure modes).
2331
2331
  if (modelId !== undefined && isComposerHarnessModel(modelId)) {
2332
- jsons.unshift(JSON.stringify({ role: "system", content: COMPOSER_EDIT_DISCIPLINE_PROMPT }));
2332
+ jsons.unshift(JSON.stringify({ role: "system", content: CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT }));
2333
2333
  }
2334
2334
  return jsons;
2335
2335
  }
@@ -1,5 +1,5 @@
1
1
  import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
2
- import OpenAI from "openai";
2
+ import OpenAI, { APIConnectionTimeoutError } from "openai";
3
3
  import type {
4
4
  ChatCompletionAssistantMessageParam,
5
5
  ChatCompletionChunk,
@@ -47,6 +47,7 @@ import {
47
47
  rewriteCopilotError,
48
48
  } from "../utils/http-inspector";
49
49
  import {
50
+ FirstEventTimeoutError,
50
51
  getOpenAIStreamIdleTimeoutMs,
51
52
  getProviderFirstEventTimeoutFallbackMs,
52
53
  getStreamFirstEventTimeoutMs,
@@ -500,8 +501,6 @@ function getTrailingPartialDeepseekToken(text: string): string {
500
501
  return tail;
501
502
  }
502
503
 
503
- const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
504
-
505
504
  const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
506
505
  "OpenAI completions stream timed out while waiting for the first event";
507
506
 
@@ -515,6 +514,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
515
514
  (async () => {
516
515
  const startTime = Date.now();
517
516
  let firstTokenTime: number | undefined;
517
+ let streamConnected = false;
518
518
  let getCapturedErrorResponse: (() => CapturedHttpErrorResponse | undefined) | undefined;
519
519
 
520
520
  const output: AssistantMessage = createInitialResponsesAssistantMessage(model.api, model.provider, model.id);
@@ -640,10 +640,8 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
640
640
  openaiStream = await createCompletionsStream("none");
641
641
  }
642
642
  }
643
- const firstEventFallbackMs =
644
- model.provider === "alibaba-token-plan"
645
- ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
646
- : getProviderFirstEventTimeoutFallbackMs(model.provider);
643
+ streamConnected = true;
644
+ const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
647
645
  const firstEventTimeoutMs =
648
646
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
649
647
  if (premiumRequestsTotal !== undefined) {
@@ -1049,20 +1047,25 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
1049
1047
  } catch (error) {
1050
1048
  for (const block of output.content) delete (block as any).index;
1051
1049
  const localAbortReason = abortTracker.getLocalAbortReason();
1050
+ const normalizedError =
1051
+ !streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
1052
+ ? new FirstEventTimeoutError(OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE)
1053
+ : error;
1052
1054
  const capturedErrorResponse = getCapturedErrorResponse?.();
1053
1055
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
1054
1056
  output.errorStatus =
1055
- extractHttpStatusFromError(localAbortReason ?? error) ??
1057
+ extractHttpStatusFromError(localAbortReason ?? normalizedError) ??
1056
1058
  (localAbortReason ? undefined : capturedErrorResponse?.status);
1057
1059
  output.transportFailure = localAbortReason
1058
1060
  ? transportFailureFacts(localAbortReason)
1059
- : transportFailureFacts(error, capturedErrorResponse);
1061
+ : transportFailureFacts(normalizedError, capturedErrorResponse);
1060
1062
  output.errorMessage =
1061
- localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse));
1063
+ localAbortReason?.message ??
1064
+ (await finalizeErrorMessage(normalizedError, rawRequestDump, capturedErrorResponse));
1062
1065
  // Some providers via OpenRouter include extra details here.
1063
- const rawMetadata = (error as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
1066
+ const rawMetadata = (normalizedError as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
1064
1067
  if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
1065
- output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
1068
+ output.errorMessage = rewriteCopilotError(output.errorMessage, normalizedError, model.provider);
1066
1069
  if (hasContentFilterSafetyCode(capturedErrorResponse)) {
1067
1070
  output.errorKind = "provider_safety_stop";
1068
1071
  }
@@ -1235,15 +1238,20 @@ async function createClient(
1235
1238
  // in the IIFE.
1236
1239
  // A caller may raise `StreamOptions.streamFirstEventTimeoutMs` for a slow-
1237
1240
  // before-headers provider; respect it so the SDK doesn't give up before the
1238
- // wrapping watchdog arms. An explicit `0` disables the first-event watchdog,
1241
+ // wrapping watchdog arms. Provider-specific fallbacks apply only when the
1242
+ // caller does not pin a value, so an explicit nonzero override must beat that
1243
+ // fallback even when it is shorter. An explicit `0` disables the watchdog,
1239
1244
  // and the SDK treats `timeout: 0` as an immediate timeout, so do not pass a
1240
1245
  // request timeout in that case.
1241
- const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs());
1246
+ const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
1247
+ const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
1242
1248
  const sdkTimeoutMs =
1243
1249
  streamFirstEventTimeoutOverride === 0
1244
1250
  ? undefined
1245
1251
  : streamFirstEventTimeoutOverride !== undefined
1246
- ? Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
1252
+ ? providerFirstEventFallbackMs !== undefined
1253
+ ? streamFirstEventTimeoutOverride
1254
+ : Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
1247
1255
  : envSdkTimeoutMs;
1248
1256
  return {
1249
1257
  client: new OpenAI({
@@ -960,20 +960,31 @@ export function populateResponsesUsageFromResponse(
960
960
  input_tokens?: number | null;
961
961
  output_tokens?: number | null;
962
962
  total_tokens?: number | null;
963
- input_tokens_details?: { cached_tokens?: number | null } | null;
963
+ input_tokens_details?: {
964
+ cached_tokens?: number | null;
965
+ cache_write_tokens?: number | null;
966
+ } | null;
964
967
  output_tokens_details?: { reasoning_tokens?: number | null } | null;
965
968
  }
966
969
  | null
967
970
  | undefined,
968
971
  ): void {
969
972
  if (!usage) return;
973
+ const inputTokens = usage.input_tokens || 0;
970
974
  const cachedTokens = usage.input_tokens_details?.cached_tokens || 0;
975
+ const reportedCacheWrite = usage.input_tokens_details?.cache_write_tokens || 0;
976
+ const cacheWriteTokens =
977
+ Number.isSafeInteger(reportedCacheWrite) &&
978
+ reportedCacheWrite >= 0 &&
979
+ cachedTokens + reportedCacheWrite <= inputTokens
980
+ ? reportedCacheWrite
981
+ : 0;
971
982
  const reasoningTokens = usage.output_tokens_details?.reasoning_tokens || 0;
972
983
  output.usage = {
973
- input: (usage.input_tokens || 0) - cachedTokens,
984
+ input: Math.max(0, inputTokens - cachedTokens - cacheWriteTokens),
974
985
  output: usage.output_tokens || 0,
975
986
  cacheRead: cachedTokens,
976
- cacheWrite: 0,
987
+ cacheWrite: cacheWriteTokens,
977
988
  totalTokens: usage.total_tokens || 0,
978
989
  ...(reasoningTokens > 0 ? { reasoningTokens } : {}),
979
990
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
@@ -1,5 +1,5 @@
1
1
  import { $credentialEnv, extractHttpStatusFromError, logger, structuredCloneJSON } from "@gajae-code/utils";
2
- import OpenAI from "openai";
2
+ import OpenAI, { APIConnectionTimeoutError } from "openai";
3
3
  import type {
4
4
  Tool as OpenAITool,
5
5
  ResponseCreateParamsStreaming,
@@ -38,7 +38,9 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
38
38
  import { transportFailureFacts } from "../utils/fallback-transport";
39
39
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
40
40
  import {
41
+ FirstEventTimeoutError,
41
42
  getOpenAIStreamIdleTimeoutMs,
43
+ getProviderFirstEventTimeoutFallbackMs,
42
44
  getStreamFirstEventTimeoutMs,
43
45
  iterateWithIdleTimeout,
44
46
  } from "../utils/idle-iterator";
@@ -124,7 +126,6 @@ export interface OpenAIResponsesOptions extends StreamOptions {
124
126
  }
125
127
 
126
128
  const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
127
- const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
128
129
  const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
129
130
  "OpenAI responses stream timed out while waiting for the first event";
130
131
  const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
@@ -314,6 +315,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
314
315
  (async () => {
315
316
  const startTime = Date.now();
316
317
  let firstTokenTime: number | undefined;
318
+ let streamConnected = false;
317
319
 
318
320
  const output: AssistantMessage = createInitialResponsesAssistantMessage(
319
321
  "openai-responses",
@@ -392,8 +394,8 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
392
394
  await notifyProviderResponse(options, response, model, request_id);
393
395
  return data;
394
396
  });
395
- const firstEventFallbackMs =
396
- model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
397
+ streamConnected = true;
398
+ const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
397
399
  const firstEventTimeoutMs =
398
400
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
399
401
  if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
@@ -447,11 +449,16 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
447
449
  } catch (error) {
448
450
  for (const block of output.content) delete (block as { index?: number }).index;
449
451
  const localAbortReason = abortTracker.getLocalAbortReason();
452
+ const normalizedError =
453
+ !streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
454
+ ? new FirstEventTimeoutError(OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE)
455
+ : error;
450
456
  output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
451
- output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
452
- output.transportFailure = transportFailureFacts(localAbortReason ?? error);
453
- output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
454
- output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
457
+ output.errorStatus = extractHttpStatusFromError(localAbortReason ?? normalizedError);
458
+ output.transportFailure = transportFailureFacts(localAbortReason ?? normalizedError);
459
+ output.errorMessage =
460
+ localAbortReason?.message ?? (await finalizeErrorMessage(normalizedError, rawRequestDump));
461
+ output.errorMessage = rewriteCopilotError(output.errorMessage, normalizedError, model.provider);
455
462
  // Explicitly mark the poisoned-history rejection so the shared
456
463
  // `invalid_prompt` contract is present even when the SDK error surfaces
457
464
  // only a message (no structured code). This keeps the responses
@@ -25,6 +25,7 @@ import { AssistantMessageEventStream as EventStreamImpl } from "../utils/event-s
25
25
  import { transportFailureFacts } from "../utils/fallback-transport";
26
26
  import {
27
27
  FirstEventTimeoutError,
28
+ getProviderFirstEventTimeoutFallbackMs,
28
29
  getStreamFirstEventTimeoutMs,
29
30
  getStreamIdleTimeoutMs,
30
31
  iterateWithIdleTimeout,
@@ -197,13 +198,12 @@ interface LazyStreamLimits {
197
198
  const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
198
199
  defaultFirstEventTimeoutMs: 300_000,
199
200
  };
200
- const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
201
201
 
202
202
  /**
203
203
  * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
204
204
  * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
205
- * otherwise providers known to have slow first events get a five-minute floor
206
- * matching their inner provider-level override. Returns `undefined` for
205
+ * otherwise providers known to have slow first events use the same centralized
206
+ * fallback as their inner provider-level watchdog. Returns `undefined` for
207
207
  * providers that should use the shared default.
208
208
  */
209
209
  export function resolveLazyStreamFirstEventFallbackMs(
@@ -211,7 +211,7 @@ export function resolveLazyStreamFirstEventFallbackMs(
211
211
  configuredFallbackMs?: number,
212
212
  ): number | undefined {
213
213
  if (configuredFallbackMs !== undefined) return configuredFallbackMs;
214
- return SLOW_FIRST_EVENT_PROVIDERS.has(provider) ? 300_000 : undefined;
214
+ return getProviderFirstEventTimeoutFallbackMs(provider);
215
215
  }
216
216
 
217
217
  function forwardStream<TApi extends Api>(