@gajae-code/ai 0.12.8 → 0.12.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/dist/types/auth-storage.d.ts +25 -2
- package/dist/types/index.d.ts +1 -1
- package/dist/types/model-pricing.d.ts +3 -0
- package/dist/types/providers/composer-discipline.d.ts +29 -23
- package/dist/types/providers/openai-responses-shared.d.ts +1 -0
- package/dist/types/providers/register-builtins.d.ts +2 -2
- package/dist/types/types.d.ts +14 -6
- package/dist/types/utils/fallback-transport.d.ts +23 -0
- package/dist/types/utils/oauth/anthropic.d.ts +21 -2
- package/dist/types/utils/oauth/callback-server.d.ts +7 -0
- package/dist/types/utils/oauth/types.d.ts +11 -0
- package/package.json +2 -2
- package/src/auth-gateway/server.ts +6 -0
- package/src/auth-storage.ts +46 -5
- package/src/index.ts +1 -0
- package/src/model-manager.ts +4 -1
- package/src/model-pricing.ts +68 -0
- package/src/model-thinking.ts +23 -1
- package/src/models.json +122 -24
- package/src/models.ts +7 -4
- package/src/prompts/composer-bash-policy-recovery.md +1 -0
- package/src/prompts/cursor-composer-bash-policy-recovery.md +1 -0
- package/src/prompts/cursor-composer-edit-discipline.md +7 -0
- package/src/providers/composer-discipline.ts +54 -0
- package/src/providers/cursor.ts +2 -2
- package/src/providers/openai-completions.ts +23 -15
- package/src/providers/openai-responses-shared.ts +14 -3
- package/src/providers/openai-responses.ts +15 -8
- package/src/providers/register-builtins.ts +4 -4
- package/src/stream.ts +60 -5
- package/src/types.ts +16 -6
- package/src/utils/fallback-transport.ts +79 -6
- package/src/utils/idle-iterator.ts +2 -0
- package/src/utils/oauth/anthropic.ts +41 -8
- package/src/utils/oauth/callback-server.ts +64 -16
- package/src/utils/oauth/types.ts +12 -0
package/src/models.json
CHANGED
|
@@ -1,5 +1,40 @@
|
|
|
1
1
|
{
|
|
2
2
|
"alibaba-token-plan": {
|
|
3
|
+
"deepseek-v4-flash-0731": {
|
|
4
|
+
"id": "deepseek-v4-flash-0731",
|
|
5
|
+
"name": "DeepSeek V4 Flash 0731",
|
|
6
|
+
"api": "openai-completions",
|
|
7
|
+
"provider": "alibaba-token-plan",
|
|
8
|
+
"baseUrl": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
|
|
9
|
+
"reasoning": true,
|
|
10
|
+
"input": [
|
|
11
|
+
"text"
|
|
12
|
+
],
|
|
13
|
+
"cost": {
|
|
14
|
+
"input": 0,
|
|
15
|
+
"output": 0,
|
|
16
|
+
"cacheRead": 0,
|
|
17
|
+
"cacheWrite": 0
|
|
18
|
+
},
|
|
19
|
+
"contextWindow": 1000000,
|
|
20
|
+
"maxTokens": 384000,
|
|
21
|
+
"compat": {
|
|
22
|
+
"supportsDeveloperRole": false,
|
|
23
|
+
"supportsReasoningEffort": true,
|
|
24
|
+
"reasoningContentField": "reasoning_content",
|
|
25
|
+
"requiresReasoningContentForToolCalls": true
|
|
26
|
+
},
|
|
27
|
+
"thinking": {
|
|
28
|
+
"mode": "effort",
|
|
29
|
+
"minLevel": "low",
|
|
30
|
+
"maxLevel": "max",
|
|
31
|
+
"levels": [
|
|
32
|
+
"low",
|
|
33
|
+
"high",
|
|
34
|
+
"max"
|
|
35
|
+
]
|
|
36
|
+
}
|
|
37
|
+
},
|
|
3
38
|
"deepseek-v4-pro": {
|
|
4
39
|
"id": "deepseek-v4-pro",
|
|
5
40
|
"name": "DeepSeek V4 Pro",
|
|
@@ -58436,7 +58471,16 @@
|
|
|
58436
58471
|
"minLevel": "low",
|
|
58437
58472
|
"maxLevel": "max"
|
|
58438
58473
|
},
|
|
58439
|
-
"applyPatchToolType": "freeform"
|
|
58474
|
+
"applyPatchToolType": "freeform",
|
|
58475
|
+
"longContextPricing": {
|
|
58476
|
+
"threshold": 272000,
|
|
58477
|
+
"cost": {
|
|
58478
|
+
"input": 10,
|
|
58479
|
+
"output": 45,
|
|
58480
|
+
"cacheRead": 1,
|
|
58481
|
+
"cacheWrite": 12.5
|
|
58482
|
+
}
|
|
58483
|
+
}
|
|
58440
58484
|
},
|
|
58441
58485
|
"gpt-5.6-luna": {
|
|
58442
58486
|
"id": "gpt-5.6-luna",
|
|
@@ -58450,10 +58494,10 @@
|
|
|
58450
58494
|
"image"
|
|
58451
58495
|
],
|
|
58452
58496
|
"cost": {
|
|
58453
|
-
"input":
|
|
58454
|
-
"output":
|
|
58455
|
-
"cacheRead": 0.
|
|
58456
|
-
"cacheWrite":
|
|
58497
|
+
"input": 0.2,
|
|
58498
|
+
"output": 1.2,
|
|
58499
|
+
"cacheRead": 0.02,
|
|
58500
|
+
"cacheWrite": 0.25
|
|
58457
58501
|
},
|
|
58458
58502
|
"contextWindow": 1050000,
|
|
58459
58503
|
"maxTokens": 128000,
|
|
@@ -58462,7 +58506,16 @@
|
|
|
58462
58506
|
"minLevel": "low",
|
|
58463
58507
|
"maxLevel": "max"
|
|
58464
58508
|
},
|
|
58465
|
-
"applyPatchToolType": "freeform"
|
|
58509
|
+
"applyPatchToolType": "freeform",
|
|
58510
|
+
"longContextPricing": {
|
|
58511
|
+
"threshold": 272000,
|
|
58512
|
+
"cost": {
|
|
58513
|
+
"input": 0.4,
|
|
58514
|
+
"output": 1.8,
|
|
58515
|
+
"cacheRead": 0.04,
|
|
58516
|
+
"cacheWrite": 0.5
|
|
58517
|
+
}
|
|
58518
|
+
}
|
|
58466
58519
|
},
|
|
58467
58520
|
"gpt-5.6-sol": {
|
|
58468
58521
|
"id": "gpt-5.6-sol",
|
|
@@ -58488,7 +58541,16 @@
|
|
|
58488
58541
|
"minLevel": "low",
|
|
58489
58542
|
"maxLevel": "max"
|
|
58490
58543
|
},
|
|
58491
|
-
"applyPatchToolType": "freeform"
|
|
58544
|
+
"applyPatchToolType": "freeform",
|
|
58545
|
+
"longContextPricing": {
|
|
58546
|
+
"threshold": 272000,
|
|
58547
|
+
"cost": {
|
|
58548
|
+
"input": 10,
|
|
58549
|
+
"output": 45,
|
|
58550
|
+
"cacheRead": 1,
|
|
58551
|
+
"cacheWrite": 12.5
|
|
58552
|
+
}
|
|
58553
|
+
}
|
|
58492
58554
|
},
|
|
58493
58555
|
"gpt-5.6-terra": {
|
|
58494
58556
|
"id": "gpt-5.6-terra",
|
|
@@ -58502,10 +58564,10 @@
|
|
|
58502
58564
|
"image"
|
|
58503
58565
|
],
|
|
58504
58566
|
"cost": {
|
|
58505
|
-
"input": 2
|
|
58506
|
-
"output":
|
|
58507
|
-
"cacheRead": 0.
|
|
58508
|
-
"cacheWrite":
|
|
58567
|
+
"input": 2,
|
|
58568
|
+
"output": 12,
|
|
58569
|
+
"cacheRead": 0.2,
|
|
58570
|
+
"cacheWrite": 2.5
|
|
58509
58571
|
},
|
|
58510
58572
|
"contextWindow": 1050000,
|
|
58511
58573
|
"maxTokens": 128000,
|
|
@@ -58514,7 +58576,16 @@
|
|
|
58514
58576
|
"minLevel": "low",
|
|
58515
58577
|
"maxLevel": "max"
|
|
58516
58578
|
},
|
|
58517
|
-
"applyPatchToolType": "freeform"
|
|
58579
|
+
"applyPatchToolType": "freeform",
|
|
58580
|
+
"longContextPricing": {
|
|
58581
|
+
"threshold": 272000,
|
|
58582
|
+
"cost": {
|
|
58583
|
+
"input": 4,
|
|
58584
|
+
"output": 18,
|
|
58585
|
+
"cacheRead": 0.4,
|
|
58586
|
+
"cacheWrite": 5
|
|
58587
|
+
}
|
|
58588
|
+
}
|
|
58518
58589
|
},
|
|
58519
58590
|
"gpt-image-2": {
|
|
58520
58591
|
"id": "gpt-image-2",
|
|
@@ -59224,10 +59295,10 @@
|
|
|
59224
59295
|
"image"
|
|
59225
59296
|
],
|
|
59226
59297
|
"cost": {
|
|
59227
|
-
"input":
|
|
59228
|
-
"output":
|
|
59229
|
-
"cacheRead": 0.
|
|
59230
|
-
"cacheWrite":
|
|
59298
|
+
"input": 0.2,
|
|
59299
|
+
"output": 1.2,
|
|
59300
|
+
"cacheRead": 0.02,
|
|
59301
|
+
"cacheWrite": 0.25
|
|
59231
59302
|
},
|
|
59232
59303
|
"contextWindow": 272000,
|
|
59233
59304
|
"maxTokens": 128000,
|
|
@@ -59238,7 +59309,16 @@
|
|
|
59238
59309
|
"minLevel": "low",
|
|
59239
59310
|
"maxLevel": "max"
|
|
59240
59311
|
},
|
|
59241
|
-
"applyPatchToolType": "freeform"
|
|
59312
|
+
"applyPatchToolType": "freeform",
|
|
59313
|
+
"longContextPricing": {
|
|
59314
|
+
"threshold": 272000,
|
|
59315
|
+
"cost": {
|
|
59316
|
+
"input": 0.4,
|
|
59317
|
+
"output": 1.8,
|
|
59318
|
+
"cacheRead": 0.04,
|
|
59319
|
+
"cacheWrite": 0.5
|
|
59320
|
+
}
|
|
59321
|
+
}
|
|
59242
59322
|
},
|
|
59243
59323
|
"gpt-5.6-sol": {
|
|
59244
59324
|
"id": "gpt-5.6-sol",
|
|
@@ -59266,7 +59346,16 @@
|
|
|
59266
59346
|
"minLevel": "low",
|
|
59267
59347
|
"maxLevel": "max"
|
|
59268
59348
|
},
|
|
59269
|
-
"applyPatchToolType": "freeform"
|
|
59349
|
+
"applyPatchToolType": "freeform",
|
|
59350
|
+
"longContextPricing": {
|
|
59351
|
+
"threshold": 272000,
|
|
59352
|
+
"cost": {
|
|
59353
|
+
"input": 10,
|
|
59354
|
+
"output": 45,
|
|
59355
|
+
"cacheRead": 1,
|
|
59356
|
+
"cacheWrite": 12.5
|
|
59357
|
+
}
|
|
59358
|
+
}
|
|
59270
59359
|
},
|
|
59271
59360
|
"gpt-5.6-terra": {
|
|
59272
59361
|
"id": "gpt-5.6-terra",
|
|
@@ -59280,10 +59369,10 @@
|
|
|
59280
59369
|
"image"
|
|
59281
59370
|
],
|
|
59282
59371
|
"cost": {
|
|
59283
|
-
"input": 2
|
|
59284
|
-
"output":
|
|
59285
|
-
"cacheRead": 0.
|
|
59286
|
-
"cacheWrite":
|
|
59372
|
+
"input": 2,
|
|
59373
|
+
"output": 12,
|
|
59374
|
+
"cacheRead": 0.2,
|
|
59375
|
+
"cacheWrite": 2.5
|
|
59287
59376
|
},
|
|
59288
59377
|
"contextWindow": 272000,
|
|
59289
59378
|
"maxTokens": 128000,
|
|
@@ -59294,7 +59383,16 @@
|
|
|
59294
59383
|
"minLevel": "low",
|
|
59295
59384
|
"maxLevel": "max"
|
|
59296
59385
|
},
|
|
59297
|
-
"applyPatchToolType": "freeform"
|
|
59386
|
+
"applyPatchToolType": "freeform",
|
|
59387
|
+
"longContextPricing": {
|
|
59388
|
+
"threshold": 272000,
|
|
59389
|
+
"cost": {
|
|
59390
|
+
"input": 4,
|
|
59391
|
+
"output": 18,
|
|
59392
|
+
"cacheRead": 0.4,
|
|
59393
|
+
"cacheWrite": 5
|
|
59394
|
+
}
|
|
59395
|
+
}
|
|
59298
59396
|
},
|
|
59299
59397
|
"gpt-image-2": {
|
|
59300
59398
|
"id": "gpt-image-2",
|
|
@@ -85511,4 +85609,4 @@
|
|
|
85511
85609
|
}
|
|
85512
85610
|
}
|
|
85513
85611
|
}
|
|
85514
|
-
}
|
|
85612
|
+
}
|
package/src/models.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { readFileSync } from "node:fs";
|
|
2
|
+
import { getOpenAIModelCost } from "./model-pricing";
|
|
2
3
|
import { isRetiredModelKey } from "./model-retirements";
|
|
3
4
|
import { applyGeneratedModelPolicies, enrichModelThinking } from "./model-thinking";
|
|
4
5
|
// `with { type: "file" }` is embedded by `bun build --compile` and resolves to
|
|
@@ -92,10 +93,12 @@ export function getBundledModels(provider: GeneratedProvider): Model<Api>[] {
|
|
|
92
93
|
}
|
|
93
94
|
|
|
94
95
|
export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage): Usage["cost"] {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
usage.cost.
|
|
98
|
-
usage.cost.
|
|
96
|
+
const inputTokens = usage.input + usage.cacheRead + usage.cacheWrite;
|
|
97
|
+
const pricing = getOpenAIModelCost(model, inputTokens) ?? model.cost;
|
|
98
|
+
usage.cost.input = (pricing.input / 1000000) * usage.input;
|
|
99
|
+
usage.cost.output = (pricing.output / 1000000) * usage.output;
|
|
100
|
+
usage.cost.cacheRead = (pricing.cacheRead / 1000000) * usage.cacheRead;
|
|
101
|
+
usage.cost.cacheWrite = (pricing.cacheWrite / 1000000) * usage.cacheWrite;
|
|
99
102
|
usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite;
|
|
100
103
|
return usage.cost;
|
|
101
104
|
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
A Composer bash policy block interrupted a shell attempt. This is not a terminal condition: continue the same task now. Do not retry repository file I/O in bash. Use the dedicated find, search, read, and edit tools for repository work, then continue with the next safe step.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
A Composer bash policy block interrupted a shell attempt. This is not a terminal condition: continue the same task now. Do not retry repository file I/O in shell. Use Cursor-native read for files or directories, grep for search or globs, write for changes, and delete only when deletion is required; then continue with the next safe step.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
File-editing discipline for this Cursor Composer harness (this OVERRIDES contrary habits from your training):
|
|
2
|
+
|
|
3
|
+
- Inspect repository files ONLY with Cursor-native read and grep: use read for file bodies or directories, and grep for content search or glob discovery. NEVER inspect repository files through shell commands (ls, find, fd, cat, sed, awk, grep, rg, head, tail, less, more) or scripts.
|
|
4
|
+
- Modify files ONLY with Cursor-native write, or delete only when deletion is required. NEVER mutate files through shell redirection, tee, sed -i, perl -pi, inline python/node/bun scripts, or other out-of-band writes.
|
|
5
|
+
- Re-read a file after any write before relying on its contents again. Do not fabricate line anchors, paths, tool names, or tool-call arguments.
|
|
6
|
+
- Tool-call arguments must be the exact schema object requested by the native tool. Do not include Markdown, commentary, analysis text, or invented fields inside tool arguments.
|
|
7
|
+
- Use shell only for terminal operations such as tests, builds, package scripts, and git commands. A shell command string must contain only the command itself; NEVER interleave reasoning or commentary into command strings or heredocs.
|
|
@@ -1,3 +1,9 @@
|
|
|
1
|
+
import composerBashPolicyRecoveryPrompt from "../prompts/composer-bash-policy-recovery.md" with { type: "text" };
|
|
2
|
+
import cursorComposerBashPolicyRecoveryPrompt from "../prompts/cursor-composer-bash-policy-recovery.md" with {
|
|
3
|
+
type: "text",
|
|
4
|
+
};
|
|
5
|
+
import cursorComposerEditDisciplinePrompt from "../prompts/cursor-composer-edit-discipline.md" with { type: "text" };
|
|
6
|
+
|
|
1
7
|
/**
|
|
2
8
|
* Anchor/edit discipline for composer-harness models (xai grok-composer-*,
|
|
3
9
|
* cursor composer-*).
|
|
@@ -30,6 +36,47 @@ export function isComposerHarnessModel(modelId: string): boolean {
|
|
|
30
36
|
return COMPOSER_MODEL_ID_PATTERN.test(modelId);
|
|
31
37
|
}
|
|
32
38
|
|
|
39
|
+
/** Stable text contract for a local shell rejection caused by Composer file-I/O discipline. */
|
|
40
|
+
export const COMPOSER_BASH_POLICY_ERROR_PREFIX = "Composer bash policy blocked repository file I/O.";
|
|
41
|
+
export const COMPOSER_BASH_POLICY_ERROR_CODE = "composer-bash-policy:repository-file-io";
|
|
42
|
+
|
|
43
|
+
export type ComposerBashPolicyToolSurface = "generic" | "cursor";
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Format the model-visible policy rejection with a stable marker and the tool
|
|
47
|
+
* vocabulary the model actually receives on this provider surface.
|
|
48
|
+
*/
|
|
49
|
+
export function formatComposerBashPolicyError(surface: ComposerBashPolicyToolSurface = "generic"): string {
|
|
50
|
+
const recovery =
|
|
51
|
+
surface === "cursor"
|
|
52
|
+
? "Continue the same task with Cursor-native read, grep, write, or delete tools; do not retry repository file I/O through shell."
|
|
53
|
+
: "Continue the same task with find, search, read, and edit tools; do not retry repository file I/O through bash.";
|
|
54
|
+
return `${COMPOSER_BASH_POLICY_ERROR_PREFIX} [${COMPOSER_BASH_POLICY_ERROR_CODE}] Recovery required: ${recovery}`;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Matches both the structured current error and the original prefix so a
|
|
59
|
+
* resumed session can recover after an upgrade without string-version skew.
|
|
60
|
+
*/
|
|
61
|
+
export function isComposerBashPolicyBlockedError(text: string): boolean {
|
|
62
|
+
return text.includes(COMPOSER_BASH_POLICY_ERROR_PREFIX);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Matches only errors emitted directly by the current policy implementation.
|
|
67
|
+
* Live recovery must use this strict form so failed shell output that merely
|
|
68
|
+
* quotes a policy error cannot masquerade as the policy gate itself.
|
|
69
|
+
*/
|
|
70
|
+
export function isCurrentComposerBashPolicyBlockedError(text: string): boolean {
|
|
71
|
+
return text === formatComposerBashPolicyError("generic") || text === formatComposerBashPolicyError("cursor");
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** One bounded, tool-enabled retry instruction for generic Composer agent loops. */
|
|
75
|
+
export const COMPOSER_BASH_POLICY_RECOVERY_PROMPT = composerBashPolicyRecoveryPrompt;
|
|
76
|
+
|
|
77
|
+
/** One bounded, tool-enabled retry instruction for Cursor's native remote tool surface. */
|
|
78
|
+
export const CURSOR_COMPOSER_BASH_POLICY_RECOVERY_PROMPT = cursorComposerBashPolicyRecoveryPrompt;
|
|
79
|
+
|
|
33
80
|
export const COMPOSER_EDIT_DISCIPLINE_PROMPT = `File-editing discipline for this Composer harness (this OVERRIDES contrary habits from your training):
|
|
34
81
|
|
|
35
82
|
- Discover file names ONLY with the find tool; search file contents ONLY with the search tool; read file bodies or line ranges ONLY with the read tool. NEVER inspect repository files through shell commands (ls, find, fd, cat, sed, awk, grep, rg, head, tail, less, more) or scripts — that output carries no hashline anchors and bypasses the agent's safety limits.
|
|
@@ -39,3 +86,10 @@ export const COMPOSER_EDIT_DISCIPLINE_PROMPT = `File-editing discipline for this
|
|
|
39
86
|
- If an edit is rejected with "anchors do not match", the rejection message prints the current lines WITH fresh anchors. Retry using exactly those printed anchors.
|
|
40
87
|
- Tool-call arguments must be the exact JSON/schema object requested by the tool. Do not include Markdown, commentary, analysis text, or invented fields inside tool arguments.
|
|
41
88
|
- Use bash only for terminal operations such as tests, builds, package scripts, and git commands. A shell command string must contain only the command itself; NEVER interleave reasoning or commentary into command strings or heredocs.`;
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Cursor executes a different native tool vocabulary from the generic agent
|
|
92
|
+
* loop. Keep this prompt separate so Composer is never told to call `edit`,
|
|
93
|
+
* `find`, or `search` when those names are unavailable remotely.
|
|
94
|
+
*/
|
|
95
|
+
export const CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT = cursorComposerEditDisciplinePrompt;
|
package/src/providers/cursor.ts
CHANGED
|
@@ -30,7 +30,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
30
30
|
import { parseStreamingJson } from "../utils/json-parse";
|
|
31
31
|
import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
|
|
32
32
|
import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
|
|
33
|
-
import {
|
|
33
|
+
import { CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
34
34
|
import { CURSOR_CLIENT_VERSION } from "./cursor/client-version";
|
|
35
35
|
import type { McpToolDefinition } from "./cursor/gen/agent_pb";
|
|
36
36
|
import {
|
|
@@ -2329,7 +2329,7 @@ export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | u
|
|
|
2329
2329
|
// Composer-harness models need anchor/edit discipline pinned ahead of any
|
|
2330
2330
|
// host/default prompt (see composer-discipline.ts for the observed failure modes).
|
|
2331
2331
|
if (modelId !== undefined && isComposerHarnessModel(modelId)) {
|
|
2332
|
-
jsons.unshift(JSON.stringify({ role: "system", content:
|
|
2332
|
+
jsons.unshift(JSON.stringify({ role: "system", content: CURSOR_COMPOSER_EDIT_DISCIPLINE_PROMPT }));
|
|
2333
2333
|
}
|
|
2334
2334
|
return jsons;
|
|
2335
2335
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
|
|
2
|
-
import OpenAI from "openai";
|
|
2
|
+
import OpenAI, { APIConnectionTimeoutError } from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
ChatCompletionAssistantMessageParam,
|
|
5
5
|
ChatCompletionChunk,
|
|
@@ -47,6 +47,7 @@ import {
|
|
|
47
47
|
rewriteCopilotError,
|
|
48
48
|
} from "../utils/http-inspector";
|
|
49
49
|
import {
|
|
50
|
+
FirstEventTimeoutError,
|
|
50
51
|
getOpenAIStreamIdleTimeoutMs,
|
|
51
52
|
getProviderFirstEventTimeoutFallbackMs,
|
|
52
53
|
getStreamFirstEventTimeoutMs,
|
|
@@ -500,8 +501,6 @@ function getTrailingPartialDeepseekToken(text: string): string {
|
|
|
500
501
|
return tail;
|
|
501
502
|
}
|
|
502
503
|
|
|
503
|
-
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
504
|
-
|
|
505
504
|
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
506
505
|
"OpenAI completions stream timed out while waiting for the first event";
|
|
507
506
|
|
|
@@ -515,6 +514,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
515
514
|
(async () => {
|
|
516
515
|
const startTime = Date.now();
|
|
517
516
|
let firstTokenTime: number | undefined;
|
|
517
|
+
let streamConnected = false;
|
|
518
518
|
let getCapturedErrorResponse: (() => CapturedHttpErrorResponse | undefined) | undefined;
|
|
519
519
|
|
|
520
520
|
const output: AssistantMessage = createInitialResponsesAssistantMessage(model.api, model.provider, model.id);
|
|
@@ -640,10 +640,8 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
640
640
|
openaiStream = await createCompletionsStream("none");
|
|
641
641
|
}
|
|
642
642
|
}
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
|
|
646
|
-
: getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
643
|
+
streamConnected = true;
|
|
644
|
+
const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
647
645
|
const firstEventTimeoutMs =
|
|
648
646
|
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
|
|
649
647
|
if (premiumRequestsTotal !== undefined) {
|
|
@@ -1049,20 +1047,25 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
1049
1047
|
} catch (error) {
|
|
1050
1048
|
for (const block of output.content) delete (block as any).index;
|
|
1051
1049
|
const localAbortReason = abortTracker.getLocalAbortReason();
|
|
1050
|
+
const normalizedError =
|
|
1051
|
+
!streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
|
|
1052
|
+
? new FirstEventTimeoutError(OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE)
|
|
1053
|
+
: error;
|
|
1052
1054
|
const capturedErrorResponse = getCapturedErrorResponse?.();
|
|
1053
1055
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
1054
1056
|
output.errorStatus =
|
|
1055
|
-
extractHttpStatusFromError(localAbortReason ??
|
|
1057
|
+
extractHttpStatusFromError(localAbortReason ?? normalizedError) ??
|
|
1056
1058
|
(localAbortReason ? undefined : capturedErrorResponse?.status);
|
|
1057
1059
|
output.transportFailure = localAbortReason
|
|
1058
1060
|
? transportFailureFacts(localAbortReason)
|
|
1059
|
-
: transportFailureFacts(
|
|
1061
|
+
: transportFailureFacts(normalizedError, capturedErrorResponse);
|
|
1060
1062
|
output.errorMessage =
|
|
1061
|
-
localAbortReason?.message ??
|
|
1063
|
+
localAbortReason?.message ??
|
|
1064
|
+
(await finalizeErrorMessage(normalizedError, rawRequestDump, capturedErrorResponse));
|
|
1062
1065
|
// Some providers via OpenRouter include extra details here.
|
|
1063
|
-
const rawMetadata = (
|
|
1066
|
+
const rawMetadata = (normalizedError as { error?: { metadata?: { raw?: string } } })?.error?.metadata?.raw;
|
|
1064
1067
|
if (rawMetadata) output.errorMessage += `\n${rawMetadata}`;
|
|
1065
|
-
output.errorMessage = rewriteCopilotError(output.errorMessage,
|
|
1068
|
+
output.errorMessage = rewriteCopilotError(output.errorMessage, normalizedError, model.provider);
|
|
1066
1069
|
if (hasContentFilterSafetyCode(capturedErrorResponse)) {
|
|
1067
1070
|
output.errorKind = "provider_safety_stop";
|
|
1068
1071
|
}
|
|
@@ -1235,15 +1238,20 @@ async function createClient(
|
|
|
1235
1238
|
// in the IIFE.
|
|
1236
1239
|
// A caller may raise `StreamOptions.streamFirstEventTimeoutMs` for a slow-
|
|
1237
1240
|
// before-headers provider; respect it so the SDK doesn't give up before the
|
|
1238
|
-
// wrapping watchdog arms.
|
|
1241
|
+
// wrapping watchdog arms. Provider-specific fallbacks apply only when the
|
|
1242
|
+
// caller does not pin a value, so an explicit nonzero override must beat that
|
|
1243
|
+
// fallback even when it is shorter. An explicit `0` disables the watchdog,
|
|
1239
1244
|
// and the SDK treats `timeout: 0` as an immediate timeout, so do not pass a
|
|
1240
1245
|
// request timeout in that case.
|
|
1241
|
-
const
|
|
1246
|
+
const providerFirstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
1247
|
+
const envSdkTimeoutMs = getStreamFirstEventTimeoutMs(getOpenAIStreamIdleTimeoutMs(), providerFirstEventFallbackMs);
|
|
1242
1248
|
const sdkTimeoutMs =
|
|
1243
1249
|
streamFirstEventTimeoutOverride === 0
|
|
1244
1250
|
? undefined
|
|
1245
1251
|
: streamFirstEventTimeoutOverride !== undefined
|
|
1246
|
-
?
|
|
1252
|
+
? providerFirstEventFallbackMs !== undefined
|
|
1253
|
+
? streamFirstEventTimeoutOverride
|
|
1254
|
+
: Math.max(envSdkTimeoutMs ?? 0, streamFirstEventTimeoutOverride)
|
|
1247
1255
|
: envSdkTimeoutMs;
|
|
1248
1256
|
return {
|
|
1249
1257
|
client: new OpenAI({
|
|
@@ -960,20 +960,31 @@ export function populateResponsesUsageFromResponse(
|
|
|
960
960
|
input_tokens?: number | null;
|
|
961
961
|
output_tokens?: number | null;
|
|
962
962
|
total_tokens?: number | null;
|
|
963
|
-
input_tokens_details?: {
|
|
963
|
+
input_tokens_details?: {
|
|
964
|
+
cached_tokens?: number | null;
|
|
965
|
+
cache_write_tokens?: number | null;
|
|
966
|
+
} | null;
|
|
964
967
|
output_tokens_details?: { reasoning_tokens?: number | null } | null;
|
|
965
968
|
}
|
|
966
969
|
| null
|
|
967
970
|
| undefined,
|
|
968
971
|
): void {
|
|
969
972
|
if (!usage) return;
|
|
973
|
+
const inputTokens = usage.input_tokens || 0;
|
|
970
974
|
const cachedTokens = usage.input_tokens_details?.cached_tokens || 0;
|
|
975
|
+
const reportedCacheWrite = usage.input_tokens_details?.cache_write_tokens || 0;
|
|
976
|
+
const cacheWriteTokens =
|
|
977
|
+
Number.isSafeInteger(reportedCacheWrite) &&
|
|
978
|
+
reportedCacheWrite >= 0 &&
|
|
979
|
+
cachedTokens + reportedCacheWrite <= inputTokens
|
|
980
|
+
? reportedCacheWrite
|
|
981
|
+
: 0;
|
|
971
982
|
const reasoningTokens = usage.output_tokens_details?.reasoning_tokens || 0;
|
|
972
983
|
output.usage = {
|
|
973
|
-
input: (
|
|
984
|
+
input: Math.max(0, inputTokens - cachedTokens - cacheWriteTokens),
|
|
974
985
|
output: usage.output_tokens || 0,
|
|
975
986
|
cacheRead: cachedTokens,
|
|
976
|
-
cacheWrite:
|
|
987
|
+
cacheWrite: cacheWriteTokens,
|
|
977
988
|
totalTokens: usage.total_tokens || 0,
|
|
978
989
|
...(reasoningTokens > 0 ? { reasoningTokens } : {}),
|
|
979
990
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { $credentialEnv, extractHttpStatusFromError, logger, structuredCloneJSON } from "@gajae-code/utils";
|
|
2
|
-
import OpenAI from "openai";
|
|
2
|
+
import OpenAI, { APIConnectionTimeoutError } from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
Tool as OpenAITool,
|
|
5
5
|
ResponseCreateParamsStreaming,
|
|
@@ -38,7 +38,9 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
38
38
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
39
39
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
40
40
|
import {
|
|
41
|
+
FirstEventTimeoutError,
|
|
41
42
|
getOpenAIStreamIdleTimeoutMs,
|
|
43
|
+
getProviderFirstEventTimeoutFallbackMs,
|
|
42
44
|
getStreamFirstEventTimeoutMs,
|
|
43
45
|
iterateWithIdleTimeout,
|
|
44
46
|
} from "../utils/idle-iterator";
|
|
@@ -124,7 +126,6 @@ export interface OpenAIResponsesOptions extends StreamOptions {
|
|
|
124
126
|
}
|
|
125
127
|
|
|
126
128
|
const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
|
|
127
|
-
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
128
129
|
const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
129
130
|
"OpenAI responses stream timed out while waiting for the first event";
|
|
130
131
|
const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
|
|
@@ -314,6 +315,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
314
315
|
(async () => {
|
|
315
316
|
const startTime = Date.now();
|
|
316
317
|
let firstTokenTime: number | undefined;
|
|
318
|
+
let streamConnected = false;
|
|
317
319
|
|
|
318
320
|
const output: AssistantMessage = createInitialResponsesAssistantMessage(
|
|
319
321
|
"openai-responses",
|
|
@@ -392,8 +394,8 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
392
394
|
await notifyProviderResponse(options, response, model, request_id);
|
|
393
395
|
return data;
|
|
394
396
|
});
|
|
395
|
-
|
|
396
|
-
|
|
397
|
+
streamConnected = true;
|
|
398
|
+
const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
397
399
|
const firstEventTimeoutMs =
|
|
398
400
|
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs);
|
|
399
401
|
if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
|
|
@@ -447,11 +449,16 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
447
449
|
} catch (error) {
|
|
448
450
|
for (const block of output.content) delete (block as { index?: number }).index;
|
|
449
451
|
const localAbortReason = abortTracker.getLocalAbortReason();
|
|
452
|
+
const normalizedError =
|
|
453
|
+
!streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
|
|
454
|
+
? new FirstEventTimeoutError(OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE)
|
|
455
|
+
: error;
|
|
450
456
|
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
451
|
-
output.errorStatus = extractHttpStatusFromError(localAbortReason ??
|
|
452
|
-
output.transportFailure = transportFailureFacts(localAbortReason ??
|
|
453
|
-
output.errorMessage =
|
|
454
|
-
|
|
457
|
+
output.errorStatus = extractHttpStatusFromError(localAbortReason ?? normalizedError);
|
|
458
|
+
output.transportFailure = transportFailureFacts(localAbortReason ?? normalizedError);
|
|
459
|
+
output.errorMessage =
|
|
460
|
+
localAbortReason?.message ?? (await finalizeErrorMessage(normalizedError, rawRequestDump));
|
|
461
|
+
output.errorMessage = rewriteCopilotError(output.errorMessage, normalizedError, model.provider);
|
|
455
462
|
// Explicitly mark the poisoned-history rejection so the shared
|
|
456
463
|
// `invalid_prompt` contract is present even when the SDK error surfaces
|
|
457
464
|
// only a message (no structured code). This keeps the responses
|
|
@@ -25,6 +25,7 @@ import { AssistantMessageEventStream as EventStreamImpl } from "../utils/event-s
|
|
|
25
25
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
26
26
|
import {
|
|
27
27
|
FirstEventTimeoutError,
|
|
28
|
+
getProviderFirstEventTimeoutFallbackMs,
|
|
28
29
|
getStreamFirstEventTimeoutMs,
|
|
29
30
|
getStreamIdleTimeoutMs,
|
|
30
31
|
iterateWithIdleTimeout,
|
|
@@ -197,13 +198,12 @@ interface LazyStreamLimits {
|
|
|
197
198
|
const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
|
|
198
199
|
defaultFirstEventTimeoutMs: 300_000,
|
|
199
200
|
};
|
|
200
|
-
const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
|
|
201
201
|
|
|
202
202
|
/**
|
|
203
203
|
* Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
|
|
204
204
|
* A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
|
|
205
|
-
* otherwise providers known to have slow first events
|
|
206
|
-
*
|
|
205
|
+
* otherwise providers known to have slow first events use the same centralized
|
|
206
|
+
* fallback as their inner provider-level watchdog. Returns `undefined` for
|
|
207
207
|
* providers that should use the shared default.
|
|
208
208
|
*/
|
|
209
209
|
export function resolveLazyStreamFirstEventFallbackMs(
|
|
@@ -211,7 +211,7 @@ export function resolveLazyStreamFirstEventFallbackMs(
|
|
|
211
211
|
configuredFallbackMs?: number,
|
|
212
212
|
): number | undefined {
|
|
213
213
|
if (configuredFallbackMs !== undefined) return configuredFallbackMs;
|
|
214
|
-
return
|
|
214
|
+
return getProviderFirstEventTimeoutFallbackMs(provider);
|
|
215
215
|
}
|
|
216
216
|
|
|
217
217
|
function forwardStream<TApi extends Api>(
|