@gajae-code/ai 0.4.1 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,12 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.4.2] - 2026-06-09
6
+
7
+ ### Fixed
8
+
9
+ - Treated `gpt-5.5` as a 400K-context model wherever context caps / auto-promote thresholds are resolved, so a ~272K session is no longer considered over-cap and no longer demotes to `gpt-5.4`. Pinned the OpenAI Codex `gpt-5.5` context window to 400K and removed its `gpt-5.4` promotion target (the smaller window made it a demotion) ([#428](https://github.com/Yeachan-Heo/gajae-code/issues/428)).
10
+
5
11
  ## [0.4.0] - 2026-06-06
6
12
 
7
13
  ### Added
@@ -40,7 +40,9 @@ export declare function applyGeneratedModelPolicies(models: ApiModel<Api>[]): vo
40
40
  * When a model's context is exhausted, the agent can promote to a sibling
41
41
  * model with a larger context window on the same provider:
42
42
  * - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
43
- * - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
43
+ *
44
+ * `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
45
+ * (which has a smaller window), so it has no promotion target.
44
46
  */
45
47
  export declare function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void;
46
48
  /**
@@ -717,6 +717,14 @@ export interface Model<TApi extends Api = any> {
717
717
  baseUrl: string;
718
718
  reasoning: boolean;
719
719
  input: ("text" | "image")[];
720
+ /**
721
+ * Output modalities the model can produce. Defaults to text-only when
722
+ * unset. A model that lists `"image"` advertises image-generation support
723
+ * (e.g. an OpenAI-compatible `gpt-image` model behind a proxy), which the
724
+ * `generate_image` tool uses to route requests without first-party
725
+ * provider/id heuristics.
726
+ */
727
+ output?: ("text" | "image")[];
720
728
  cost: {
721
729
  input: number;
722
730
  output: number;
@@ -13,6 +13,10 @@ export type CapturedHttpErrorResponse = {
13
13
  bodyText?: string;
14
14
  bodyJson?: unknown;
15
15
  };
16
+ /** Whether `message` (from a 400 response) signals an unavailable/unknown model. */
17
+ export declare function isModelUnavailableError(message: string, error: unknown): boolean;
18
+ /** Actionable guidance for selecting an available model/provider. */
19
+ export declare function formatModelUnavailableGuidance(dump: RawHttpRequestDump | undefined): string;
16
20
  export declare function appendRawHttpRequestDumpFor400(message: string, error: unknown, dump: RawHttpRequestDump | undefined): Promise<string>;
17
21
  export declare function finalizeErrorMessage(error: unknown, rawRequestDump: RawHttpRequestDump | undefined, capturedErrorResponse?: CapturedHttpErrorResponse): Promise<string>;
18
22
  export declare function withHttpStatus(error: unknown, status: number): Error;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.4.1",
4
+ "version": "0.4.3",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@gajae-code/utils": "0.4.1",
46
+ "@gajae-code/utils": "0.4.3",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -195,7 +195,9 @@ export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
195
195
  * When a model's context is exhausted, the agent can promote to a sibling
196
196
  * model with a larger context window on the same provider:
197
197
  * - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
198
- * - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
198
+ *
199
+ * `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
200
+ * (which has a smaller window), so it has no promotion target.
199
201
  */
200
202
  export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
201
203
  for (const candidate of models) {
@@ -204,8 +206,6 @@ export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
204
206
  let targetId: string | undefined;
205
207
  if (parsedCandidate.variant === "codex-spark") {
206
208
  targetId = "gpt-5.5";
207
- } else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
208
- targetId = "gpt-5.4";
209
209
  } else {
210
210
  continue;
211
211
  }
@@ -430,7 +430,22 @@ function inferGeneratedApplyPatchToolType(
430
430
  return undefined;
431
431
  }
432
432
 
433
+ function applyGpt55ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel): boolean {
434
+ // gpt-5.5 is a 400K-context model. OpenAI code backend discovery can omit the
435
+ // context window, falling back to the 272K default, which incorrectly trips
436
+ // context-cap / auto-promote thresholds (a ~272K session would look over-cap
437
+ // and demote to gpt-5.4). Pin gpt-5.5 to its true 400K window.
438
+ if (parsedModel.variant === "base" && semverEqual(parsedModel.version, "5.5")) {
439
+ model.contextWindow = 400000;
440
+ return true;
441
+ }
442
+ return false;
443
+ }
444
+
433
445
  function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel): void {
446
+ if (applyGpt55ContextWindow(model, parsedModel)) {
447
+ return;
448
+ }
434
449
  // OpenAI code backend models: 400K figure includes output budget; input window is 272K.
435
450
  if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
436
451
  model.contextWindow = 272000;
package/src/models.json CHANGED
@@ -9853,7 +9853,8 @@
9853
9853
  "baseUrl": "https://api.kilo.ai/api/gateway",
9854
9854
  "reasoning": false,
9855
9855
  "input": [
9856
- "text"
9856
+ "text",
9857
+ "image"
9857
9858
  ],
9858
9859
  "cost": {
9859
9860
  "input": 0,
@@ -9872,7 +9873,8 @@
9872
9873
  "baseUrl": "https://api.kilo.ai/api/gateway",
9873
9874
  "reasoning": false,
9874
9875
  "input": [
9875
- "text"
9876
+ "text",
9877
+ "image"
9876
9878
  ],
9877
9879
  "cost": {
9878
9880
  "input": 0,
@@ -16768,7 +16770,8 @@
16768
16770
  "baseUrl": "https://api.kilo.ai/api/gateway",
16769
16771
  "reasoning": false,
16770
16772
  "input": [
16771
- "text"
16773
+ "text",
16774
+ "image"
16772
16775
  ],
16773
16776
  "cost": {
16774
16777
  "input": 0,
@@ -34660,7 +34663,7 @@
34660
34663
  },
34661
34664
  "minimax-m3": {
34662
34665
  "id": "minimax-m3",
34663
- "name": "MiniMax M3",
34666
+ "name": "MiniMax-M3",
34664
34667
  "api": "anthropic-messages",
34665
34668
  "provider": "minimax",
34666
34669
  "baseUrl": "https://api.minimax.io/anthropic",
@@ -34855,7 +34858,7 @@
34855
34858
  },
34856
34859
  "minimax-m3": {
34857
34860
  "id": "minimax-m3",
34858
- "name": "MiniMax M3",
34861
+ "name": "MiniMax-M3",
34859
34862
  "api": "anthropic-messages",
34860
34863
  "provider": "minimax-cn",
34861
34864
  "baseUrl": "https://api.minimaxi.com/anthropic",
@@ -35122,7 +35125,7 @@
35122
35125
  },
35123
35126
  "minimax-m3": {
35124
35127
  "id": "minimax-m3",
35125
- "name": "MiniMax M3",
35128
+ "name": "MiniMax-M3",
35126
35129
  "api": "openai-completions",
35127
35130
  "provider": "minimax-code",
35128
35131
  "baseUrl": "https://api.minimax.io/v1",
@@ -35395,7 +35398,7 @@
35395
35398
  },
35396
35399
  "minimax-m3": {
35397
35400
  "id": "minimax-m3",
35398
- "name": "MiniMax M3",
35401
+ "name": "MiniMax-M3",
35399
35402
  "api": "openai-completions",
35400
35403
  "provider": "minimax-code-cn",
35401
35404
  "baseUrl": "https://api.minimaxi.com/v1",
@@ -53264,7 +53267,7 @@
53264
53267
  "cacheRead": 0.5,
53265
53268
  "cacheWrite": 0
53266
53269
  },
53267
- "contextWindow": 272000,
53270
+ "contextWindow": 400000,
53268
53271
  "maxTokens": 128000,
53269
53272
  "preferWebsockets": true,
53270
53273
  "priority": 9,
@@ -53274,8 +53277,7 @@
53274
53277
  "maxLevel": "xhigh",
53275
53278
  "defaultLevel": "xhigh"
53276
53279
  },
53277
- "applyPatchToolType": "freeform",
53278
- "contextPromotionTarget": "openai-codex/gpt-5.4"
53280
+ "applyPatchToolType": "freeform"
53279
53281
  }
53280
53282
  },
53281
53283
  "opencode": {
@@ -63602,7 +63604,8 @@
63602
63604
  "baseUrl": "https://api.venice.ai/api/v1",
63603
63605
  "reasoning": false,
63604
63606
  "input": [
63605
- "text"
63607
+ "text",
63608
+ "image"
63606
63609
  ],
63607
63610
  "cost": {
63608
63611
  "input": 0,
package/src/types.ts CHANGED
@@ -854,6 +854,14 @@ export interface Model<TApi extends Api = any> {
854
854
  baseUrl: string;
855
855
  reasoning: boolean;
856
856
  input: ("text" | "image")[];
857
+ /**
858
+ * Output modalities the model can produce. Defaults to text-only when
859
+ * unset. A model that lists `"image"` advertises image-generation support
860
+ * (e.g. an OpenAI-compatible `gpt-image` model behind a proxy), which the
861
+ * `generate_image` tool uses to route requests without first-party
862
+ * provider/id heuristics.
863
+ */
864
+ output?: ("text" | "image")[];
857
865
  cost: {
858
866
  input: number; // $/million tokens
859
867
  output: number; // $/million tokens
@@ -1,5 +1,5 @@
1
1
  import * as path from "node:path";
2
- import { extractHttpStatusFromError, getLogsDir } from "@gajae-code/utils";
2
+ import { APP_NAME, extractHttpStatusFromError, getLogsDir } from "@gajae-code/utils";
3
3
  import { isCopilotTransientModelError } from "./retry.js";
4
4
  import { formatErrorMessageWithRetryAfter } from "./retry-after.js";
5
5
 
@@ -26,6 +26,46 @@ type ErrorWithStatus = {
26
26
 
27
27
  const SENSITIVE_HEADERS = ["authorization", "x-api-key", "api-key", "cookie", "set-cookie", "proxy-authorization"];
28
28
 
29
+ /**
30
+ * Privacy note appended next to a saved raw HTTP request dump. The dump is
31
+ * sanitized (secrets/thinking redacted) but can still contain prompt content
32
+ * and request metadata, so we explicitly discourage pasting it into public
33
+ * channels (issue #438).
34
+ */
35
+ const RAW_HTTP_REQUEST_PRIVACY_NOTE =
36
+ "note: this local file is for your own debugging and may contain prompt content or request metadata — review it before sharing and do not paste its contents into public channels (issues, Discord, etc.).";
37
+
38
+ /**
39
+ * Patterns that indicate the configured model is unavailable on the provider
40
+ * (e.g. OpenAI's "The requested model '...' does not exist."). Used to surface
41
+ * actionable model/provider guidance instead of only a raw 400 + log path.
42
+ */
43
+ const MODEL_UNAVAILABLE_PATTERNS: readonly RegExp[] = [
44
+ /\bmodel\b[^\n]*\bdoes not exist\b/i,
45
+ /\bdoes not exist\b[^\n]*\bmodel\b/i,
46
+ /\bmodel\b[^\n]*\b(not found|unavailable|not supported|no access|does not have access)\b/i,
47
+ /\b(unknown|unsupported|invalid)\s+model\b/i,
48
+ ];
49
+
50
+ /** Whether `message` (from a 400 response) signals an unavailable/unknown model. */
51
+ export function isModelUnavailableError(message: string, error: unknown): boolean {
52
+ if (extractHttpStatusFromError(error) !== 400) return false;
53
+ return MODEL_UNAVAILABLE_PATTERNS.some(pattern => pattern.test(message));
54
+ }
55
+
56
+ /** Actionable guidance for selecting an available model/provider. */
57
+ export function formatModelUnavailableGuidance(dump: RawHttpRequestDump | undefined): string {
58
+ const modelPart = dump?.model ? ` '${dump.model}'` : "";
59
+ const providerPart = dump?.provider ? ` on provider '${dump.provider}'` : "";
60
+ return [
61
+ `The configured model${modelPart}${providerPart} is not available on this account/provider.`,
62
+ "Pick an available model before retrying:",
63
+ ` • List available models: ${APP_NAME} --list-models`,
64
+ ` • Run with a model: ${APP_NAME} --model <model>`,
65
+ ` • Configure a provider: ${APP_NAME} setup provider`,
66
+ ].join("\n");
67
+ }
68
+
29
69
  export async function appendRawHttpRequestDumpFor400(
30
70
  message: string,
31
71
  error: unknown,
@@ -41,7 +81,7 @@ export async function appendRawHttpRequestDumpFor400(
41
81
 
42
82
  try {
43
83
  await Bun.write(filePath, `${JSON.stringify(sanitizedDump, null, 2)}\n`);
44
- return `${message}\nraw-http-request=${filePath}`;
84
+ return `${message}\nraw-http-request=${filePath}\n${RAW_HTTP_REQUEST_PRIVACY_NOTE}`;
45
85
  } catch (writeError) {
46
86
  const writeMessage = writeError instanceof Error ? writeError.message : String(writeError);
47
87
  return `${message}\nraw-http-request-save-failed=${writeMessage}`;
@@ -62,6 +102,9 @@ export async function finalizeErrorMessage(
62
102
  message = `${message}\n${capturedMessage}`;
63
103
  }
64
104
  }
105
+ if (isModelUnavailableError(message, error)) {
106
+ message = `${message}\n\n${formatModelUnavailableGuidance(rawRequestDump)}`;
107
+ }
65
108
  return appendRawHttpRequestDumpFor400(message, error, rawRequestDump);
66
109
  }
67
110
 
@@ -157,10 +157,14 @@ export async function* iterateWithIdleTimeout<T>(
157
157
  activeTimeoutMs = firstItemTimeoutMs;
158
158
  } else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) {
159
159
  activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt);
160
- if (activeTimeoutMs <= 0) {
161
- options.onIdle?.();
162
- closeIterator();
163
- throw new Error(options.errorMessage);
160
+ // The idle deadline may already have elapsed because the *consumer*
161
+ // was slow, not because the provider stalled — and the next item may
162
+ // already be buffered and ready to deliver. Clamp to 0 instead of
163
+ // throwing eagerly so the next() race below still gets a chance to
164
+ // win (it settles on a microtask, ahead of the 0ms timer). Only a
165
+ // genuinely hung iterator loses that race and surfaces as a stall.
166
+ if (activeTimeoutMs < 0) {
167
+ activeTimeoutMs = 0;
164
168
  }
165
169
  }
166
170
 
@@ -177,7 +181,10 @@ export async function* iterateWithIdleTimeout<T>(
177
181
 
178
182
  let timer: NodeJS.Timeout | undefined;
179
183
  let resolveTimeout: ((value: { kind: "timeout" }) => void) | undefined;
180
- const enforceTimeout = !noTimeoutEnforced && activeTimeoutMs !== undefined && activeTimeoutMs > 0;
184
+ const enforceTimeout =
185
+ !noTimeoutEnforced &&
186
+ activeTimeoutMs !== undefined &&
187
+ (awaitingFirstItem ? activeTimeoutMs > 0 : activeTimeoutMs >= 0);
181
188
  if (enforceTimeout) {
182
189
  const { promise, resolve } = Promise.withResolvers<{ kind: "timeout" }>();
183
190
  resolveTimeout = resolve;