@gajae-code/ai 0.4.1 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/types/model-thinking.d.ts +3 -1
- package/dist/types/types.d.ts +8 -0
- package/dist/types/utils/http-inspector.d.ts +4 -0
- package/package.json +2 -2
- package/src/model-thinking.ts +18 -3
- package/src/models.json +14 -11
- package/src/types.ts +8 -0
- package/src/utils/http-inspector.ts +45 -2
- package/src/utils/idle-iterator.ts +12 -5
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.4.2] - 2026-06-09
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Treated `gpt-5.5` as a 400K-context model wherever context caps / auto-promote thresholds are resolved, so a ~272K session is no longer considered over-cap and no longer demotes to `gpt-5.4`. Pinned the OpenAI Codex `gpt-5.5` context window to 400K and removed its `gpt-5.4` promotion target (the smaller window made it a demotion) ([#428](https://github.com/Yeachan-Heo/gajae-code/issues/428)).
|
|
10
|
+
|
|
5
11
|
## [0.4.0] - 2026-06-06
|
|
6
12
|
|
|
7
13
|
### Added
|
|
@@ -40,7 +40,9 @@ export declare function applyGeneratedModelPolicies(models: ApiModel<Api>[]): vo
|
|
|
40
40
|
* When a model's context is exhausted, the agent can promote to a sibling
|
|
41
41
|
* model with a larger context window on the same provider:
|
|
42
42
|
* - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
|
|
43
|
-
*
|
|
43
|
+
*
|
|
44
|
+
* `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
|
|
45
|
+
* (which has a smaller window), so it has no promotion target.
|
|
44
46
|
*/
|
|
45
47
|
export declare function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void;
|
|
46
48
|
/**
|
package/dist/types/types.d.ts
CHANGED
|
@@ -717,6 +717,14 @@ export interface Model<TApi extends Api = any> {
|
|
|
717
717
|
baseUrl: string;
|
|
718
718
|
reasoning: boolean;
|
|
719
719
|
input: ("text" | "image")[];
|
|
720
|
+
/**
|
|
721
|
+
* Output modalities the model can produce. Defaults to text-only when
|
|
722
|
+
* unset. A model that lists `"image"` advertises image-generation support
|
|
723
|
+
* (e.g. an OpenAI-compatible `gpt-image` model behind a proxy), which the
|
|
724
|
+
* `generate_image` tool uses to route requests without first-party
|
|
725
|
+
* provider/id heuristics.
|
|
726
|
+
*/
|
|
727
|
+
output?: ("text" | "image")[];
|
|
720
728
|
cost: {
|
|
721
729
|
input: number;
|
|
722
730
|
output: number;
|
|
@@ -13,6 +13,10 @@ export type CapturedHttpErrorResponse = {
|
|
|
13
13
|
bodyText?: string;
|
|
14
14
|
bodyJson?: unknown;
|
|
15
15
|
};
|
|
16
|
+
/** Whether `message` (from a 400 response) signals an unavailable/unknown model. */
|
|
17
|
+
export declare function isModelUnavailableError(message: string, error: unknown): boolean;
|
|
18
|
+
/** Actionable guidance for selecting an available model/provider. */
|
|
19
|
+
export declare function formatModelUnavailableGuidance(dump: RawHttpRequestDump | undefined): string;
|
|
16
20
|
export declare function appendRawHttpRequestDumpFor400(message: string, error: unknown, dump: RawHttpRequestDump | undefined): Promise<string>;
|
|
17
21
|
export declare function finalizeErrorMessage(error: unknown, rawRequestDump: RawHttpRequestDump | undefined, capturedErrorResponse?: CapturedHttpErrorResponse): Promise<string>;
|
|
18
22
|
export declare function withHttpStatus(error: unknown, status: number): Error;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.3",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gaebal-gajae.dev",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
45
45
|
"@bufbuild/protobuf": "^2.12.0",
|
|
46
|
-
"@gajae-code/utils": "0.4.
|
|
46
|
+
"@gajae-code/utils": "0.4.3",
|
|
47
47
|
"openai": "^6.36.0",
|
|
48
48
|
"partial-json": "^0.1.7",
|
|
49
49
|
"zod": "4.4.3"
|
package/src/model-thinking.ts
CHANGED
|
@@ -195,7 +195,9 @@ export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
|
|
|
195
195
|
* When a model's context is exhausted, the agent can promote to a sibling
|
|
196
196
|
* model with a larger context window on the same provider:
|
|
197
197
|
* - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
|
|
198
|
-
*
|
|
198
|
+
*
|
|
199
|
+
* `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
|
|
200
|
+
* (which has a smaller window), so it has no promotion target.
|
|
199
201
|
*/
|
|
200
202
|
export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
|
|
201
203
|
for (const candidate of models) {
|
|
@@ -204,8 +206,6 @@ export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
|
|
|
204
206
|
let targetId: string | undefined;
|
|
205
207
|
if (parsedCandidate.variant === "codex-spark") {
|
|
206
208
|
targetId = "gpt-5.5";
|
|
207
|
-
} else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
|
|
208
|
-
targetId = "gpt-5.4";
|
|
209
209
|
} else {
|
|
210
210
|
continue;
|
|
211
211
|
}
|
|
@@ -430,7 +430,22 @@ function inferGeneratedApplyPatchToolType(
|
|
|
430
430
|
return undefined;
|
|
431
431
|
}
|
|
432
432
|
|
|
433
|
+
function applyGpt55ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel): boolean {
|
|
434
|
+
// gpt-5.5 is a 400K-context model. OpenAI code backend discovery can omit the
|
|
435
|
+
// context window, falling back to the 272K default, which incorrectly trips
|
|
436
|
+
// context-cap / auto-promote thresholds (a ~272K session would look over-cap
|
|
437
|
+
// and demote to gpt-5.4). Pin gpt-5.5 to its true 400K window.
|
|
438
|
+
if (parsedModel.variant === "base" && semverEqual(parsedModel.version, "5.5")) {
|
|
439
|
+
model.contextWindow = 400000;
|
|
440
|
+
return true;
|
|
441
|
+
}
|
|
442
|
+
return false;
|
|
443
|
+
}
|
|
444
|
+
|
|
433
445
|
function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel): void {
|
|
446
|
+
if (applyGpt55ContextWindow(model, parsedModel)) {
|
|
447
|
+
return;
|
|
448
|
+
}
|
|
434
449
|
// OpenAI code backend models: 400K figure includes output budget; input window is 272K.
|
|
435
450
|
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
|
|
436
451
|
model.contextWindow = 272000;
|
package/src/models.json
CHANGED
|
@@ -9853,7 +9853,8 @@
|
|
|
9853
9853
|
"baseUrl": "https://api.kilo.ai/api/gateway",
|
|
9854
9854
|
"reasoning": false,
|
|
9855
9855
|
"input": [
|
|
9856
|
-
"text"
|
|
9856
|
+
"text",
|
|
9857
|
+
"image"
|
|
9857
9858
|
],
|
|
9858
9859
|
"cost": {
|
|
9859
9860
|
"input": 0,
|
|
@@ -9872,7 +9873,8 @@
|
|
|
9872
9873
|
"baseUrl": "https://api.kilo.ai/api/gateway",
|
|
9873
9874
|
"reasoning": false,
|
|
9874
9875
|
"input": [
|
|
9875
|
-
"text"
|
|
9876
|
+
"text",
|
|
9877
|
+
"image"
|
|
9876
9878
|
],
|
|
9877
9879
|
"cost": {
|
|
9878
9880
|
"input": 0,
|
|
@@ -16768,7 +16770,8 @@
|
|
|
16768
16770
|
"baseUrl": "https://api.kilo.ai/api/gateway",
|
|
16769
16771
|
"reasoning": false,
|
|
16770
16772
|
"input": [
|
|
16771
|
-
"text"
|
|
16773
|
+
"text",
|
|
16774
|
+
"image"
|
|
16772
16775
|
],
|
|
16773
16776
|
"cost": {
|
|
16774
16777
|
"input": 0,
|
|
@@ -34660,7 +34663,7 @@
|
|
|
34660
34663
|
},
|
|
34661
34664
|
"minimax-m3": {
|
|
34662
34665
|
"id": "minimax-m3",
|
|
34663
|
-
"name": "MiniMax
|
|
34666
|
+
"name": "MiniMax-M3",
|
|
34664
34667
|
"api": "anthropic-messages",
|
|
34665
34668
|
"provider": "minimax",
|
|
34666
34669
|
"baseUrl": "https://api.minimax.io/anthropic",
|
|
@@ -34855,7 +34858,7 @@
|
|
|
34855
34858
|
},
|
|
34856
34859
|
"minimax-m3": {
|
|
34857
34860
|
"id": "minimax-m3",
|
|
34858
|
-
"name": "MiniMax
|
|
34861
|
+
"name": "MiniMax-M3",
|
|
34859
34862
|
"api": "anthropic-messages",
|
|
34860
34863
|
"provider": "minimax-cn",
|
|
34861
34864
|
"baseUrl": "https://api.minimaxi.com/anthropic",
|
|
@@ -35122,7 +35125,7 @@
|
|
|
35122
35125
|
},
|
|
35123
35126
|
"minimax-m3": {
|
|
35124
35127
|
"id": "minimax-m3",
|
|
35125
|
-
"name": "MiniMax
|
|
35128
|
+
"name": "MiniMax-M3",
|
|
35126
35129
|
"api": "openai-completions",
|
|
35127
35130
|
"provider": "minimax-code",
|
|
35128
35131
|
"baseUrl": "https://api.minimax.io/v1",
|
|
@@ -35395,7 +35398,7 @@
|
|
|
35395
35398
|
},
|
|
35396
35399
|
"minimax-m3": {
|
|
35397
35400
|
"id": "minimax-m3",
|
|
35398
|
-
"name": "MiniMax
|
|
35401
|
+
"name": "MiniMax-M3",
|
|
35399
35402
|
"api": "openai-completions",
|
|
35400
35403
|
"provider": "minimax-code-cn",
|
|
35401
35404
|
"baseUrl": "https://api.minimaxi.com/v1",
|
|
@@ -53264,7 +53267,7 @@
|
|
|
53264
53267
|
"cacheRead": 0.5,
|
|
53265
53268
|
"cacheWrite": 0
|
|
53266
53269
|
},
|
|
53267
|
-
"contextWindow":
|
|
53270
|
+
"contextWindow": 400000,
|
|
53268
53271
|
"maxTokens": 128000,
|
|
53269
53272
|
"preferWebsockets": true,
|
|
53270
53273
|
"priority": 9,
|
|
@@ -53274,8 +53277,7 @@
|
|
|
53274
53277
|
"maxLevel": "xhigh",
|
|
53275
53278
|
"defaultLevel": "xhigh"
|
|
53276
53279
|
},
|
|
53277
|
-
"applyPatchToolType": "freeform"
|
|
53278
|
-
"contextPromotionTarget": "openai-codex/gpt-5.4"
|
|
53280
|
+
"applyPatchToolType": "freeform"
|
|
53279
53281
|
}
|
|
53280
53282
|
},
|
|
53281
53283
|
"opencode": {
|
|
@@ -63602,7 +63604,8 @@
|
|
|
63602
63604
|
"baseUrl": "https://api.venice.ai/api/v1",
|
|
63603
63605
|
"reasoning": false,
|
|
63604
63606
|
"input": [
|
|
63605
|
-
"text"
|
|
63607
|
+
"text",
|
|
63608
|
+
"image"
|
|
63606
63609
|
],
|
|
63607
63610
|
"cost": {
|
|
63608
63611
|
"input": 0,
|
package/src/types.ts
CHANGED
|
@@ -854,6 +854,14 @@ export interface Model<TApi extends Api = any> {
|
|
|
854
854
|
baseUrl: string;
|
|
855
855
|
reasoning: boolean;
|
|
856
856
|
input: ("text" | "image")[];
|
|
857
|
+
/**
|
|
858
|
+
* Output modalities the model can produce. Defaults to text-only when
|
|
859
|
+
* unset. A model that lists `"image"` advertises image-generation support
|
|
860
|
+
* (e.g. an OpenAI-compatible `gpt-image` model behind a proxy), which the
|
|
861
|
+
* `generate_image` tool uses to route requests without first-party
|
|
862
|
+
* provider/id heuristics.
|
|
863
|
+
*/
|
|
864
|
+
output?: ("text" | "image")[];
|
|
857
865
|
cost: {
|
|
858
866
|
input: number; // $/million tokens
|
|
859
867
|
output: number; // $/million tokens
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import * as path from "node:path";
|
|
2
|
-
import { extractHttpStatusFromError, getLogsDir } from "@gajae-code/utils";
|
|
2
|
+
import { APP_NAME, extractHttpStatusFromError, getLogsDir } from "@gajae-code/utils";
|
|
3
3
|
import { isCopilotTransientModelError } from "./retry.js";
|
|
4
4
|
import { formatErrorMessageWithRetryAfter } from "./retry-after.js";
|
|
5
5
|
|
|
@@ -26,6 +26,46 @@ type ErrorWithStatus = {
|
|
|
26
26
|
|
|
27
27
|
const SENSITIVE_HEADERS = ["authorization", "x-api-key", "api-key", "cookie", "set-cookie", "proxy-authorization"];
|
|
28
28
|
|
|
29
|
+
/**
|
|
30
|
+
* Privacy note appended next to a saved raw HTTP request dump. The dump is
|
|
31
|
+
* sanitized (secrets/thinking redacted) but can still contain prompt content
|
|
32
|
+
* and request metadata, so we explicitly discourage pasting it into public
|
|
33
|
+
* channels (issue #438).
|
|
34
|
+
*/
|
|
35
|
+
const RAW_HTTP_REQUEST_PRIVACY_NOTE =
|
|
36
|
+
"note: this local file is for your own debugging and may contain prompt content or request metadata — review it before sharing and do not paste its contents into public channels (issues, Discord, etc.).";
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Patterns that indicate the configured model is unavailable on the provider
|
|
40
|
+
* (e.g. OpenAI's "The requested model '...' does not exist."). Used to surface
|
|
41
|
+
* actionable model/provider guidance instead of only a raw 400 + log path.
|
|
42
|
+
*/
|
|
43
|
+
const MODEL_UNAVAILABLE_PATTERNS: readonly RegExp[] = [
|
|
44
|
+
/\bmodel\b[^\n]*\bdoes not exist\b/i,
|
|
45
|
+
/\bdoes not exist\b[^\n]*\bmodel\b/i,
|
|
46
|
+
/\bmodel\b[^\n]*\b(not found|unavailable|not supported|no access|does not have access)\b/i,
|
|
47
|
+
/\b(unknown|unsupported|invalid)\s+model\b/i,
|
|
48
|
+
];
|
|
49
|
+
|
|
50
|
+
/** Whether `message` (from a 400 response) signals an unavailable/unknown model. */
|
|
51
|
+
export function isModelUnavailableError(message: string, error: unknown): boolean {
|
|
52
|
+
if (extractHttpStatusFromError(error) !== 400) return false;
|
|
53
|
+
return MODEL_UNAVAILABLE_PATTERNS.some(pattern => pattern.test(message));
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** Actionable guidance for selecting an available model/provider. */
|
|
57
|
+
export function formatModelUnavailableGuidance(dump: RawHttpRequestDump | undefined): string {
|
|
58
|
+
const modelPart = dump?.model ? ` '${dump.model}'` : "";
|
|
59
|
+
const providerPart = dump?.provider ? ` on provider '${dump.provider}'` : "";
|
|
60
|
+
return [
|
|
61
|
+
`The configured model${modelPart}${providerPart} is not available on this account/provider.`,
|
|
62
|
+
"Pick an available model before retrying:",
|
|
63
|
+
` • List available models: ${APP_NAME} --list-models`,
|
|
64
|
+
` • Run with a model: ${APP_NAME} --model <model>`,
|
|
65
|
+
` • Configure a provider: ${APP_NAME} setup provider`,
|
|
66
|
+
].join("\n");
|
|
67
|
+
}
|
|
68
|
+
|
|
29
69
|
export async function appendRawHttpRequestDumpFor400(
|
|
30
70
|
message: string,
|
|
31
71
|
error: unknown,
|
|
@@ -41,7 +81,7 @@ export async function appendRawHttpRequestDumpFor400(
|
|
|
41
81
|
|
|
42
82
|
try {
|
|
43
83
|
await Bun.write(filePath, `${JSON.stringify(sanitizedDump, null, 2)}\n`);
|
|
44
|
-
return `${message}\nraw-http-request=${filePath}`;
|
|
84
|
+
return `${message}\nraw-http-request=${filePath}\n${RAW_HTTP_REQUEST_PRIVACY_NOTE}`;
|
|
45
85
|
} catch (writeError) {
|
|
46
86
|
const writeMessage = writeError instanceof Error ? writeError.message : String(writeError);
|
|
47
87
|
return `${message}\nraw-http-request-save-failed=${writeMessage}`;
|
|
@@ -62,6 +102,9 @@ export async function finalizeErrorMessage(
|
|
|
62
102
|
message = `${message}\n${capturedMessage}`;
|
|
63
103
|
}
|
|
64
104
|
}
|
|
105
|
+
if (isModelUnavailableError(message, error)) {
|
|
106
|
+
message = `${message}\n\n${formatModelUnavailableGuidance(rawRequestDump)}`;
|
|
107
|
+
}
|
|
65
108
|
return appendRawHttpRequestDumpFor400(message, error, rawRequestDump);
|
|
66
109
|
}
|
|
67
110
|
|
|
@@ -157,10 +157,14 @@ export async function* iterateWithIdleTimeout<T>(
|
|
|
157
157
|
activeTimeoutMs = firstItemTimeoutMs;
|
|
158
158
|
} else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) {
|
|
159
159
|
activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt);
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
160
|
+
// The idle deadline may already have elapsed because the *consumer*
|
|
161
|
+
// was slow, not because the provider stalled — and the next item may
|
|
162
|
+
// already be buffered and ready to deliver. Clamp to 0 instead of
|
|
163
|
+
// throwing eagerly so the next() race below still gets a chance to
|
|
164
|
+
// win (it settles on a microtask, ahead of the 0ms timer). Only a
|
|
165
|
+
// genuinely hung iterator loses that race and surfaces as a stall.
|
|
166
|
+
if (activeTimeoutMs < 0) {
|
|
167
|
+
activeTimeoutMs = 0;
|
|
164
168
|
}
|
|
165
169
|
}
|
|
166
170
|
|
|
@@ -177,7 +181,10 @@ export async function* iterateWithIdleTimeout<T>(
|
|
|
177
181
|
|
|
178
182
|
let timer: NodeJS.Timeout | undefined;
|
|
179
183
|
let resolveTimeout: ((value: { kind: "timeout" }) => void) | undefined;
|
|
180
|
-
const enforceTimeout =
|
|
184
|
+
const enforceTimeout =
|
|
185
|
+
!noTimeoutEnforced &&
|
|
186
|
+
activeTimeoutMs !== undefined &&
|
|
187
|
+
(awaitingFirstItem ? activeTimeoutMs > 0 : activeTimeoutMs >= 0);
|
|
181
188
|
if (enforceTimeout) {
|
|
182
189
|
const { promise, resolve } = Promise.withResolvers<{ kind: "timeout" }>();
|
|
183
190
|
resolveTimeout = resolve;
|