@sayknow-cli/ai 0.3.1 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,36 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.7.5] - 2026-06-27
6
+
7
+ ### Fixed
8
+
9
+ - Sanitized Codex history replay text fields so malformed/non-string replay text no longer corrupts the rebuilt request payload (#1199).
10
+ - Preserved object-valued tool replay fields when rebuilding Codex history payloads instead of coercing them to strings (#1200).
11
+
12
+ ## [0.7.4] - 2026-06-27
13
+
14
+ ### Fixed
15
+
16
+ - Treated `openai-codex-device` as an auth-storage alias for `openai-codex`, so headless/device Codex logins show as authenticated and logout/remove the stored Codex credential instead of appearing unsaved (#1151).
17
+ - Disabled thinking for OpenCode Go forced tool calls so forced-tool turns no longer emit unsupported thinking content (#1185).
18
+ - Restored the GPT-5.5 context window to its correct size (#1186).
19
+
20
+ ## [0.7.3] - 2026-06-25
21
+ ### Added
22
+
23
+ - Added the Sakana Fugu provider (`fugu`) with API-key auth (`FUGU_API_KEY`) and bundled catalog models (#1086).
24
+
25
+ ### Fixed
26
+
27
+ - Wired Fugu API-key auth login so `skc login fugu` stores a reusable `FUGU_API_KEY` credential instead of the provider having no supported login flow (#1090).
28
+ - Matched the real google-antigravity IDE request headers, system prompt, and preamble config so requests are accepted by Cloud Code Assist (#1080).
29
+ - `isContextOverflow` now detects a third proxy-level overflow case — an empty response with `stopReason: "stop"` and anomalously low usage (input + output ≤ 5 tokens), as emitted by some proxies (notably LiteLLM) when the upstream context window is exceeded — so callers surface overflow instead of treating it as a clean completion (#1102).
30
+
31
+ ### Security
32
+
33
+ - In no-auth (tokenless) auth-gateway mode, requests carrying a browser `Origin` header are now rejected before CORS preflight handling or route dispatch, while local non-browser CLI clients keep the existing tokenless flow and token-configured browser clients keep the bearer-token/preflight flow (#1115).
34
+
5
35
  ## [0.7.2] - 2026-06-24
6
36
 
7
37
  ### Fixed
@@ -7,6 +7,7 @@ export declare function resolvePeer(req: Request): string;
7
7
  */
8
8
  export declare function timingSafeEqual(a: Uint8Array, b: Uint8Array): boolean;
9
9
  export declare function isAuthorized(req: Request, tokens: ReadonlySet<string>): boolean;
10
+ export declare function isNoAuthBrowserOriginRequest(req: Request, tokens: ReadonlySet<string>): boolean;
10
11
  /**
11
12
  * Extract allow-listed passthrough headers from an inbound request. Keys are
12
13
  * lowercased; empty values are dropped. Called once per request in
@@ -41,7 +41,7 @@ export declare function applyGeneratedModelPolicies(models: ApiModel<Api>[]): vo
41
41
  * model with a larger context window on the same provider:
42
42
  * - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
43
43
  *
44
- * `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
44
+ * `gpt-5.5` itself is a 1M-context model and is not demoted to `gpt-5.4`
45
45
  * (which has a smaller window), so it has no promotion target.
46
46
  */
47
47
  export declare function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void;
@@ -75,6 +75,11 @@ export interface FirepassModelManagerConfig {
75
75
  * See https://docs.fireworks.ai/firepass.
76
76
  */
77
77
  export declare function firepassModelManagerOptions(_config?: FirepassModelManagerConfig): ModelManagerOptions<"openai-completions">;
78
+ export interface FuguModelManagerConfig {
79
+ apiKey?: string;
80
+ baseUrl?: string;
81
+ }
82
+ export declare function fuguModelManagerOptions(config?: FuguModelManagerConfig): ModelManagerOptions<"openai-completions">;
78
83
  export interface MistralModelManagerConfig {
79
84
  apiKey?: string;
80
85
  baseUrl?: string;
@@ -24,7 +24,7 @@ export interface GoogleGeminiCliOptions extends StreamOptions {
24
24
  };
25
25
  projectId?: string;
26
26
  }
27
- export { ANTIGRAVITY_SYSTEM_INSTRUCTION, getAntigravityUserAgent, getGeminiCliHeaders, getGeminiCliUserAgent, } from "./google-gemini-headers";
27
+ export { ANTIGRAVITY_SYSTEM_INSTRUCTION, getAntigravityRequestHeaders, getGeminiCliHeaders, getGeminiCliUserAgent, } from "./google-gemini-headers";
28
28
  interface ParsedGeminiCliCredentials {
29
29
  accessToken: string;
30
30
  projectId: string;
@@ -63,6 +63,9 @@ interface CloudCodeAssistRequest {
63
63
  mode: FunctionCallingConfigMode;
64
64
  };
65
65
  };
66
+ preambleConfig?: {
67
+ mode: "SYSTEM_INSTRUCTION_MODE_REPLACE";
68
+ };
66
69
  };
67
70
  requestType?: string;
68
71
  userAgent?: string;
@@ -8,11 +8,33 @@ export declare const getGeminiCliHeaders: (modelId?: string) => {
8
8
  "User-Agent": string;
9
9
  "Client-Metadata": string;
10
10
  };
11
- export declare const ANTIGRAVITY_SYSTEM_INSTRUCTION: string;
12
11
  /**
13
- * Antigravity / Cloud Code Assist user agent. Lives in its own file so discovery
14
- * and usage code can read it without pulling the heavy google-gemini-cli provider
15
- * (and its @google/genai → google-auth-library dependency chain) into the startup
16
- * parse graph.
12
+ * Full Antigravity system instruction as observed in the real IDE binary.
13
+ * This is the complete prompt injected by the Antigravity language server,
14
+ * including BNF lexer definition for syntax highlighting, messaging system
15
+ * description, and reactive wakeup protocol.
16
+ *
17
+ * Wire-format fidelity note: The `%s` placeholders are literal in the real
18
+ * Antigravity IDE prompt. The Cloud Code Assist service either expands them
19
+ * server-side or the model handles them as-is. Preserved for byte-faithful
20
+ * emulation of the observed IDE wire format.
21
+ *
22
+ * Evidence: Disassembly of the Antigravity LS binary confirms this exact
23
+ * string is loaded and injected as the system instruction.
24
+ */
25
+ export declare const ANTIGRAVITY_SYSTEM_INSTRUCTION = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.\nYou are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.\nThe USER will send you requests, which you must always prioritize addressing. Along with each USER request, we will attach additional metadata about their current state, such as what files they have open and where their cursor is.\nThis information may or may not be relevant to the coding task, it is up for you to decide.<lexer>\n <config>\n <name>BNF</name>\n <alias>bnf</alias>\n <filename>*.bnf</filename>\n <mime_type>text/x-bnf</mime_type>\n </config>\n <rules>\n <state name=\"root\">\n <rule pattern=\"(&lt;)([ -;=?-~]+)(&gt;)\">\n <bygroups>\n <token type=\"Punctuation\"/>\n <token type=\"NameClass\"/>\n <token type=\"Punctuation\"/>\n </bygroups>\n </rule>\n <rule pattern=\"::=\">\n <token type=\"Operator\"/>\n </rule>\n <rule pattern=\"[^&lt;&gt;:]+\">\n <token type=\"Text\"/>\n </rule>\n <rule pattern=\".\">\n <token type=\"Text\"/>\n </rule>\n </state>\n </rules>\n</lexer>You are connected to a messaging system where you may receive messages from: %s.\n\n## Receiving Messages\n\nYou receive messages automatically at the start of each invocation. All messages are delivered in full directly into your context \u2014 no manual retrieval is needed.\n\n## Reactive Wakeup (No Polling Needed)\n\nThe system automatically resumes your execution when:\n%s\n\nThis means you do **NOT** need to poll in a loop while waiting for messages or updates. After launching anything that performs work asynchronously, you may continue other work or simply stop by calling no more tools. The system will notify you when there is something to process.\n";
26
+ /**
27
+ * Antigravity / Cloud Code Assist user agent.
28
+ *
29
+ * Disassembly-confirmed: getUserAgentName() @ 0x5ecb1dd loads "antigravity-ide"
30
+ * via LEA RDX, [RIP-0x284fc90] → 0x367b554 = "antigravity-ide"
31
+ *
32
+ * The LS sets HTTP headers via: fmt.Sprintf("User-Agent: %s", getUserAgentName())
33
+ * So the final header is: User-Agent: antigravity-ide
34
+ *
35
+ * -override_user_agent flag can override this (confirmed at 0x5ecbc37).
17
36
  */
18
37
  export declare let getAntigravityUserAgent: () => string;
38
+ export declare const getAntigravityRequestHeaders: () => {
39
+ "User-Agent": string;
40
+ };
@@ -18,6 +18,7 @@ export interface InputItem {
18
18
  name?: string;
19
19
  output?: unknown;
20
20
  arguments?: unknown;
21
+ encrypted_content?: unknown;
21
22
  }
22
23
  export interface RequestBody {
23
24
  model: string;
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
48
48
  /** Provider-specific transport used to encode the selected effort. */
49
49
  mode: ThinkingControlMode;
50
50
  }
51
- export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
51
+ export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
52
52
  export type Provider = KnownProvider | string;
53
53
  import type { Effort } from "./model-thinking";
54
54
  /** Token budgets for each thinking level (token-based providers only) */
@@ -0,0 +1 @@
1
+ export declare const loginFugu: (options: import("./types").OAuthController) => Promise<string>;
@@ -32,6 +32,7 @@ export declare function getOAuthApiKey(provider: OAuthProvider, credentials: Rec
32
32
  newCredentials: OAuthCredentials;
33
33
  apiKey: string;
34
34
  } | null>;
35
+ export declare function resolveOAuthStorageProvider(provider: OAuthProviderId): OAuthProviderId;
35
36
  /**
36
37
  * Get list of OAuth providers.
37
38
  */
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
7
7
  email?: string;
8
8
  accountId?: string;
9
9
  };
10
- export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
10
+ export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
11
11
  export type OAuthProviderId = OAuthProvider | (string & {});
12
12
  export type OAuthPrompt = {
13
13
  message: string;
@@ -2,11 +2,14 @@ import type { AssistantMessage } from "../types";
2
2
  /**
3
3
  * Check if an assistant message represents a context overflow error.
4
4
  *
5
- * This handles two cases:
5
+ * This handles three cases:
6
6
  * 1. Error-based overflow: Most providers return stopReason "error" with a
7
7
  * specific error message pattern.
8
8
  * 2. Silent overflow: Some providers accept overflow requests and return
9
9
  * successfully. For these, we check if usage.input exceeds the context window.
10
+ * 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
11
+ * response with empty content and a fabricated near-zero usage when the
12
+ * upstream model's context window is exceeded.
10
13
  *
11
14
  * ## Reliability by Provider
12
15
  *
@@ -30,6 +33,13 @@ import type { AssistantMessage } from "../types";
30
33
  * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
31
34
  * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
32
35
  * - Ollama: Silently truncates input without error. Cannot be detected via this function.
36
+ * - LiteLLM proxy: Returns a "successful" response with empty content and a
37
+ * fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
38
+ * model's context window is exceeded. Detected via Case 3 (empty content +
39
+ * anomalously low usage). Note: the LiteLLM proxy's context limit may differ
40
+ * from the underlying model's advertised contextWindow (e.g. configured via
41
+ * `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
42
+ * usage.input against contextWindow) may not catch it.
33
43
  * The response will have usage.input < expected, but we don't know the expected value.
34
44
  *
35
45
  * ## Custom Providers
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/ai",
4
- "version": "0.3.1",
4
+ "version": "0.3.2",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://github.com/jaybeyond/Sayknow_CLI",
7
7
  "author": "jaybeyond",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@sayknow-cli/utils": "0.3.1",
46
+ "@sayknow-cli/utils": "0.3.2",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -47,7 +47,7 @@ export function timingSafeEqual(a: Uint8Array, b: Uint8Array): boolean {
47
47
  const TOKEN_ENCODER = new TextEncoder();
48
48
 
49
49
  export function isAuthorized(req: Request, tokens: ReadonlySet<string>): boolean {
50
- if (tokens.size === 0) return true;
50
+ if (tokens.size === 0) return !isNoAuthBrowserOriginRequest(req, tokens);
51
51
  const header = req.headers.get("authorization");
52
52
  if (!header) return false;
53
53
  const match = header.match(/^Bearer\s+(.+)$/i);
@@ -63,6 +63,10 @@ export function isAuthorized(req: Request, tokens: ReadonlySet<string>): boolean
63
63
  return ok;
64
64
  }
65
65
 
66
+ export function isNoAuthBrowserOriginRequest(req: Request, tokens: ReadonlySet<string>): boolean {
67
+ return tokens.size === 0 && req.headers.has("origin");
68
+ }
69
+
66
70
  /**
67
71
  * Allow-list of inbound request headers that the gateway captures and forwards
68
72
  * to the underlying parsers (which decide whether to surface them to the
@@ -27,7 +27,15 @@ import * as piNative from "../providers/pi-native-server";
27
27
  import { streamSimple } from "../stream";
28
28
  import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
29
29
  import { parseBind } from "../utils/parse-bind";
30
- import { captureRequestHeaders, corsHeaders, isAuthorized, json, resolvePeer, withCors } from "./http";
30
+ import {
31
+ captureRequestHeaders,
32
+ corsHeaders,
33
+ isAuthorized,
34
+ isNoAuthBrowserOriginRequest,
35
+ json,
36
+ resolvePeer,
37
+ withCors,
38
+ } from "./http";
31
39
  import type {
32
40
  AuthGatewayServerHandle,
33
41
  AuthGatewayServerOptions,
@@ -640,6 +648,15 @@ export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServe
640
648
  const url = new URL(req.url);
641
649
  const pathname = url.pathname;
642
650
  const peer = resolvePeer(req);
651
+ if (isNoAuthBrowserOriginRequest(req, tokens)) {
652
+ logger.info("auth-gateway no-auth browser-origin request rejected", {
653
+ method: req.method,
654
+ path: pathname,
655
+ peer,
656
+ origin: req.headers.get("origin"),
657
+ });
658
+ return json(403, { error: "browser origin requires bearer token" });
659
+ }
643
660
  // CORS preflight is always answered without auth — browsers send
644
661
  // preflights pre-authentication and a 401 here breaks the actual
645
662
  // request before the bearer is ever attached.
@@ -31,7 +31,7 @@ import { grokCliRankingStrategy, grokCliUsageProvider } from "./usage/grok-cli";
31
31
  import { kimiUsageProvider } from "./usage/kimi";
32
32
  import { codexRankingStrategy, openaiCodexUsageProvider } from "./usage/openai-codex";
33
33
  import { zaiUsageProvider } from "./usage/zai";
34
- import { getOAuthApiKey, getOAuthProvider, refreshOAuthToken } from "./utils/oauth";
34
+ import { getOAuthApiKey, getOAuthProvider, refreshOAuthToken, resolveOAuthStorageProvider } from "./utils/oauth";
35
35
  import { loginDeepSeek } from "./utils/oauth/deepseek";
36
36
  import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex";
37
37
  import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types";
@@ -918,7 +918,8 @@ export class AuthStorage {
918
918
  * @returns Array of stored credentials, empty if none exist
919
919
  */
920
920
  #getStoredCredentials(provider: string): StoredCredential[] {
921
- return this.#data.get(provider) ?? [];
921
+ const storageProvider = resolveOAuthStorageProvider(provider);
922
+ return this.#data.get(storageProvider) ?? [];
922
923
  }
923
924
 
924
925
  /**
@@ -1250,16 +1251,17 @@ export class AuthStorage {
1250
1251
  * Set credential for a provider.
1251
1252
  */
1252
1253
  async set(provider: string, credential: AuthCredentialEntry): Promise<void> {
1254
+ const storageProvider = resolveOAuthStorageProvider(provider);
1253
1255
  const normalized = Array.isArray(credential) ? credential : [credential];
1254
- const deduped = this.#dedupeOAuthCredentials(provider, normalized);
1256
+ const deduped = this.#dedupeOAuthCredentials(storageProvider, normalized);
1255
1257
  const stored = this.#store.replaceAuthCredentialsRemote
1256
- ? await this.#store.replaceAuthCredentialsRemote(provider, deduped)
1257
- : this.#store.replaceAuthCredentialsForProvider(provider, deduped);
1258
+ ? await this.#store.replaceAuthCredentialsRemote(storageProvider, deduped)
1259
+ : this.#store.replaceAuthCredentialsForProvider(storageProvider, deduped);
1258
1260
  this.#setStoredCredentials(
1259
- provider,
1261
+ storageProvider,
1260
1262
  stored.map(record => ({ id: record.id, credential: record.credential })),
1261
1263
  );
1262
- this.#resetProviderAssignments(provider);
1264
+ this.#resetProviderAssignments(storageProvider);
1263
1265
  }
1264
1266
 
1265
1267
  #toSnapshotEntries(provider: string, stored: StoredAuthCredential[]): AuthCredentialSnapshotEntry[] {
@@ -1289,26 +1291,30 @@ export class AuthStorage {
1289
1291
  provider: string,
1290
1292
  credential: AuthCredential,
1291
1293
  ): Promise<AuthCredentialIfAbsentSnapshotResult> {
1292
- if (this.#runtimeOverrides.has(provider)) return this.#snapshotSkipResult(provider, "skipped-existing-runtime");
1293
- if (this.#configOverrides.has(provider)) return this.#snapshotSkipResult(provider, "skipped-existing-config");
1294
- if (this.#getCredentialsForProvider(provider).length > 0)
1295
- return this.#snapshotSkipResult(provider, "skipped-existing");
1296
- if (getEnvApiKey(provider)) return this.#snapshotSkipResult(provider, "skipped-existing-env");
1297
- if (this.#fallbackResolver?.(provider)) return this.#snapshotSkipResult(provider, "skipped-existing-fallback");
1294
+ const storageProvider = resolveOAuthStorageProvider(provider);
1295
+ if (this.#runtimeOverrides.has(storageProvider))
1296
+ return this.#snapshotSkipResult(storageProvider, "skipped-existing-runtime");
1297
+ if (this.#configOverrides.has(storageProvider))
1298
+ return this.#snapshotSkipResult(storageProvider, "skipped-existing-config");
1299
+ if (this.#getCredentialsForProvider(storageProvider).length > 0)
1300
+ return this.#snapshotSkipResult(storageProvider, "skipped-existing");
1301
+ if (getEnvApiKey(storageProvider)) return this.#snapshotSkipResult(storageProvider, "skipped-existing-env");
1302
+ if (this.#fallbackResolver?.(storageProvider))
1303
+ return this.#snapshotSkipResult(storageProvider, "skipped-existing-fallback");
1298
1304
 
1299
1305
  const result = this.#store.upsertAuthCredentialRemoteIfAbsent
1300
- ? await this.#store.upsertAuthCredentialRemoteIfAbsent(provider, credential)
1301
- : this.#store.upsertAuthCredentialForProviderIfAbsent(provider, credential);
1306
+ ? await this.#store.upsertAuthCredentialRemoteIfAbsent(storageProvider, credential)
1307
+ : this.#store.upsertAuthCredentialForProviderIfAbsent(storageProvider, credential);
1302
1308
  this.#setStoredCredentials(
1303
- provider,
1309
+ storageProvider,
1304
1310
  result.entries.map(entry => ({ id: entry.id, credential: entry.credential })),
1305
1311
  );
1306
- this.#resetProviderAssignments(provider);
1312
+ this.#resetProviderAssignments(storageProvider);
1307
1313
  return {
1308
1314
  inserted: result.inserted,
1309
1315
  reason: result.reason,
1310
1316
  provider: result.provider,
1311
- entries: this.#toSnapshotEntries(provider, result.entries),
1317
+ entries: this.#toSnapshotEntries(storageProvider, result.entries),
1312
1318
  };
1313
1319
  }
1314
1320
 
@@ -1327,13 +1333,14 @@ export class AuthStorage {
1327
1333
  * Remove credential for a provider.
1328
1334
  */
1329
1335
  async remove(provider: string): Promise<void> {
1336
+ const storageProvider = resolveOAuthStorageProvider(provider);
1330
1337
  if (this.#store.deleteAuthCredentialsRemote) {
1331
- await this.#store.deleteAuthCredentialsRemote(provider, "deleted by user");
1338
+ await this.#store.deleteAuthCredentialsRemote(storageProvider, "deleted by user");
1332
1339
  } else {
1333
- this.#store.deleteAuthCredentialsForProvider(provider, "deleted by user");
1340
+ this.#store.deleteAuthCredentialsForProvider(storageProvider, "deleted by user");
1334
1341
  }
1335
- this.#setStoredCredentials(provider, []);
1336
- this.#resetProviderAssignments(provider);
1342
+ this.#setStoredCredentials(storageProvider, []);
1343
+ this.#resetProviderAssignments(storageProvider);
1337
1344
  }
1338
1345
 
1339
1346
  /**
@@ -1355,11 +1362,12 @@ export class AuthStorage {
1355
1362
  * Unlike getApiKey(), this doesn't refresh OAuth tokens.
1356
1363
  */
1357
1364
  hasAuth(provider: string): boolean {
1358
- if (this.#runtimeOverrides.has(provider)) return true;
1359
- if (this.#configOverrides.has(provider)) return true;
1360
- if (this.#getCredentialsForProvider(provider).length > 0) return true;
1361
- if (getEnvApiKey(provider)) return true;
1362
- if (this.#fallbackResolver?.(provider)) return true;
1365
+ const storageProvider = resolveOAuthStorageProvider(provider);
1366
+ if (this.#runtimeOverrides.has(storageProvider)) return true;
1367
+ if (this.#configOverrides.has(storageProvider)) return true;
1368
+ if (this.#getCredentialsForProvider(storageProvider).length > 0) return true;
1369
+ if (getEnvApiKey(storageProvider)) return true;
1370
+ if (this.#fallbackResolver?.(storageProvider)) return true;
1363
1371
  return false;
1364
1372
  }
1365
1373
 
@@ -1613,6 +1621,12 @@ export class AuthStorage {
1613
1621
  await saveApiKeyCredential(apiKey);
1614
1622
  return;
1615
1623
  }
1624
+ case "fugu": {
1625
+ const { loginFugu } = await import("./utils/oauth/fugu");
1626
+ const apiKey = await loginFugu(ctrl);
1627
+ await saveApiKeyCredential(apiKey);
1628
+ return;
1629
+ }
1616
1630
  case "zai": {
1617
1631
  const { loginZai } = await import("./utils/oauth/zai");
1618
1632
  const apiKey = await loginZai(ctrl);
@@ -197,7 +197,7 @@ export function applyGeneratedModelPolicies(models: ApiModel<Api>[]): void {
197
197
  * model with a larger context window on the same provider:
198
198
  * - `OpenAI code backend-spark` variants promote to `gpt-5.5`.
199
199
  *
200
- * `gpt-5.5` itself is a 400K-context model and is not demoted to `gpt-5.4`
200
+ * `gpt-5.5` itself is a 1M-context model and is not demoted to `gpt-5.4`
201
201
  * (which has a smaller window), so it has no promotion target.
202
202
  */
203
203
  export function linkOpenAIPromotionTargets(models: ApiModel<Api>[]): void {
@@ -453,15 +453,8 @@ function inferGeneratedApplyPatchToolType(
453
453
  }
454
454
 
455
455
  function applyGpt55ContextWindow(model: ApiModel<Api>, parsedModel: OpenAIModel): boolean {
456
- // OpenAI Codex reports GPT-5.5 with a 272K prompt budget. Keep the generated
457
- // bundle aligned with the backend limit so compaction fires before the prompt
458
- // crosses the usable Codex window instead of trusting stale 400K snapshots.
459
- if (model.provider === "openai-codex" && parsedModel.variant === "base" && semverEqual(parsedModel.version, "5.5")) {
460
- model.contextWindow = 272000;
461
- return true;
462
- }
463
456
  if (parsedModel.variant === "base" && semverEqual(parsedModel.version, "5.5")) {
464
- model.contextWindow = 400000;
457
+ model.contextWindow = 1_000_000;
465
458
  return true;
466
459
  }
467
460
  return false;
package/src/models.json CHANGED
@@ -56310,7 +56310,7 @@
56310
56310
  "cacheRead": 0.5,
56311
56311
  "cacheWrite": 0
56312
56312
  },
56313
- "contextWindow": 272000,
56313
+ "contextWindow": 1000000,
56314
56314
  "maxTokens": 128000,
56315
56315
  "preferWebsockets": true,
56316
56316
  "priority": 9,
@@ -79757,5 +79757,71 @@
79757
79757
  "maxLevel": "xhigh"
79758
79758
  }
79759
79759
  }
79760
+ },
79761
+ "fugu": {
79762
+ "fugu": {
79763
+ "id": "fugu",
79764
+ "name": "Sakana Fugu",
79765
+ "api": "openai-completions",
79766
+ "provider": "fugu",
79767
+ "baseUrl": "https://api.sakana.ai/v1",
79768
+ "reasoning": true,
79769
+ "input": [
79770
+ "text"
79771
+ ],
79772
+ "cost": {
79773
+ "input": 0,
79774
+ "output": 0,
79775
+ "cacheRead": 0,
79776
+ "cacheWrite": 0
79777
+ },
79778
+ "contextWindow": 200000,
79779
+ "maxTokens": 65536,
79780
+ "compat": {
79781
+ "supportsStore": false,
79782
+ "supportsDeveloperRole": false,
79783
+ "supportsMultipleSystemMessages": false,
79784
+ "supportsReasoningEffort": false,
79785
+ "supportsUsageInStreaming": true,
79786
+ "maxTokensField": "max_tokens"
79787
+ },
79788
+ "thinking": {
79789
+ "mode": "effort",
79790
+ "minLevel": "minimal",
79791
+ "maxLevel": "high"
79792
+ }
79793
+ },
79794
+ "fugu-ultra": {
79795
+ "id": "fugu-ultra",
79796
+ "name": "Sakana Fugu Ultra",
79797
+ "api": "openai-completions",
79798
+ "provider": "fugu",
79799
+ "baseUrl": "https://api.sakana.ai/v1",
79800
+ "reasoning": true,
79801
+ "input": [
79802
+ "text"
79803
+ ],
79804
+ "cost": {
79805
+ "input": 0,
79806
+ "output": 0,
79807
+ "cacheRead": 0,
79808
+ "cacheWrite": 0
79809
+ },
79810
+ "contextWindow": 200000,
79811
+ "maxTokens": 65536,
79812
+ "compat": {
79813
+ "supportsStore": false,
79814
+ "supportsDeveloperRole": false,
79815
+ "supportsMultipleSystemMessages": false,
79816
+ "supportsReasoningEffort": false,
79817
+ "supportsUsageInStreaming": true,
79818
+ "maxTokensField": "max_tokens"
79819
+ },
79820
+ "thinking": {
79821
+ "mode": "effort",
79822
+ "minLevel": "minimal",
79823
+ "maxLevel": "high"
79824
+ }
79825
+ }
79760
79826
  }
79761
79827
  }
@@ -16,6 +16,7 @@ import {
16
16
  deepseekModelManagerOptions,
17
17
  firepassModelManagerOptions,
18
18
  fireworksModelManagerOptions,
19
+ fuguModelManagerOptions,
19
20
  githubCopilotModelManagerOptions,
20
21
  groqModelManagerOptions,
21
22
  huggingfaceModelManagerOptions,
@@ -154,6 +155,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
154
155
  catalog("Fireworks", ["FIREWORKS_API_KEY"]),
155
156
  ),
156
157
  descriptor("firepass", "kimi-k2.6-turbo", config => firepassModelManagerOptions(config)),
158
+ catalogDescriptor(
159
+ "fugu",
160
+ "fugu",
161
+ config => fuguModelManagerOptions(config),
162
+ catalog("Sakana Fugu", ["FUGU_API_KEY"]),
163
+ ),
157
164
  descriptor("xai", "grok-4-fast-non-reasoning", config => xaiModelManagerOptions(config)),
158
165
  catalogDescriptor(
159
166
  "deepseek",
@@ -785,6 +785,15 @@ export function firepassModelManagerOptions(
785
785
  // 7. Mistral
786
786
  // ---------------------------------------------------------------------------
787
787
 
788
+ export interface FuguModelManagerConfig {
789
+ apiKey?: string;
790
+ baseUrl?: string;
791
+ }
792
+
793
+ export function fuguModelManagerOptions(config?: FuguModelManagerConfig): ModelManagerOptions<"openai-completions"> {
794
+ return createSimpleOpenAICompletionsOptions("fugu", config?.baseUrl ?? "https://api.sakana.ai/v1", config);
795
+ }
796
+
788
797
  export interface MistralModelManagerConfig {
789
798
  apiKey?: string;
790
799
  baseUrl?: string;
@@ -30,7 +30,11 @@ import {
30
30
  markToolChoiceIncapability,
31
31
  resolveToolChoice,
32
32
  } from "../utils/tool-choice-capability";
33
- import { ANTIGRAVITY_SYSTEM_INSTRUCTION, getAntigravityUserAgent, getGeminiCliHeaders } from "./google-gemini-headers";
33
+ import {
34
+ ANTIGRAVITY_SYSTEM_INSTRUCTION,
35
+ getAntigravityRequestHeaders,
36
+ getGeminiCliHeaders,
37
+ } from "./google-gemini-headers";
34
38
  import type { Content, FunctionCallingConfigMode, ThinkingConfig } from "./google-shared";
35
39
  import {
36
40
  convertMessages,
@@ -78,7 +82,7 @@ const ANTIGRAVITY_ENDPOINT_FALLBACKS = [ANTIGRAVITY_DAILY_ENDPOINT, ANTIGRAVITY_
78
82
 
79
83
  export {
80
84
  ANTIGRAVITY_SYSTEM_INSTRUCTION,
81
- getAntigravityUserAgent,
85
+ getAntigravityRequestHeaders,
82
86
  getGeminiCliHeaders,
83
87
  getGeminiCliUserAgent,
84
88
  } from "./google-gemini-headers";
@@ -220,6 +224,13 @@ interface CloudCodeAssistRequest {
220
224
  mode: FunctionCallingConfigMode;
221
225
  };
222
226
  };
227
+ // Evidence: Real Antigravity IDE sends preambleConfig with
228
+ // SYSTEM_INSTRUCTION_MODE_REPLACE to control how the server
229
+ // interprets the system instruction (replace vs append).
230
+ // Confirmed via network interception of official IDE traffic.
231
+ preambleConfig?: {
232
+ mode: "SYSTEM_INSTRUCTION_MODE_REPLACE";
233
+ };
223
234
  };
224
235
  requestType?: string;
225
236
  userAgent?: string;
@@ -319,7 +330,9 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
319
330
  if (replacementPayload !== undefined) {
320
331
  requestBody = replacementPayload as typeof requestBody;
321
332
  }
322
- const headers = isAntigravity ? { "User-Agent": getAntigravityUserAgent() } : getGeminiCliHeaders(model.id);
333
+ const headers = isAntigravity
334
+ ? getAntigravityRequestHeaders() // Evidence: UA = "antigravity-ide" (disassembly 0x5ecb1dd)
335
+ : getGeminiCliHeaders(model.id);
323
336
 
324
337
  const requestHeaders = {
325
338
  Authorization: `Bearer ${accessToken}`,
@@ -786,9 +799,13 @@ export function buildRequest(
786
799
  request.sessionId = deriveAntigravitySessionId(context);
787
800
  }
788
801
 
789
- // System instruction must be object with parts, not plain string
802
+ // System instruction must be object with parts, not plain string.
803
+ // Evidence: Real Antigravity IDE sets role: "user" on systemInstruction
804
+ // (confirmed in request body dumps from network interception).
805
+ // Only applied for Antigravity path — standard gemini-cli omits role.
790
806
  if (systemPrompts.length > 0) {
791
807
  request.systemInstruction = {
808
+ ...(isAntigravity && { role: "user" }),
792
809
  parts: systemPrompts.map(text => ({ text })),
793
810
  };
794
811
  }
@@ -818,26 +835,55 @@ export function buildRequest(
818
835
  }
819
836
  }
820
837
 
821
- if (isAntigravity && isClaudeModel(model.id)) {
822
- const resolvedLevel = resolvedToolChoice.resolvedLevel;
823
- if (resolvedLevel === "named" || resolvedLevel === "required") {
824
- request.toolConfig = {
825
- functionCallingConfig: {
826
- mode: "VALIDATED" as FunctionCallingConfigMode,
827
- },
828
- };
829
- }
838
+ // Claude Antigravity: use VALIDATED mode only when tools are present
839
+ // and the caller hasn't explicitly requested "none" tool choice.
840
+ //
841
+ // Evidence:
842
+ // - Real Antigravity IDE sends VALIDATED for Claude requests with tools
843
+ // (confirmed via network interception of official IDE traffic)
844
+ // - The original SKC code used resolvedToolChoice.resolvedLevel to gate
845
+ // VALIDATED, but the real IDE doesn't check tool choice resolution —
846
+ // it sends VALIDATED whenever tools exist and toolChoice != "none"
847
+ // - omp reference: packages/ai/src/providers/google-gemini-cli.ts
848
+ // `antigravityClaudeToolConfig` guard pattern
849
+ if (
850
+ isAntigravity &&
851
+ isClaudeModel(model.id) &&
852
+ context.tools &&
853
+ context.tools.length > 0 &&
854
+ options?.toolChoice !== "none"
855
+ ) {
856
+ request.toolConfig = {
857
+ functionCallingConfig: {
858
+ mode: "VALIDATED" as FunctionCallingConfigMode,
859
+ },
860
+ };
830
861
  }
831
862
 
863
+ // Stateless system instruction injection: send on EVERY Antigravity request.
864
+ // This is safer than client-side session dedup because:
865
+ // - preambleConfig server-side persistence is unproven
866
+ // - A failed first request would poison the session if we skipped injection
867
+ // - Token cost is acceptable for Antigravity's long-context models
868
+ //
869
+ // Evidence:
870
+ // - Real Antigravity IDE sends the full system prompt on every request
871
+ // (confirmed via repeated request dumps — no client-side dedup observed)
872
+ // - The [ignore] wrapper was a SKC-original addition not present in the
873
+ // real IDE wire format. Removed for byte-faithful emulation.
874
+ // - preambleConfig: { mode: "SYSTEM_INSTRUCTION_MODE_REPLACE" } is the
875
+ // mechanism the real IDE uses to deliver system instructions. Without it,
876
+ // the server may append (not replace) the system prompt.
877
+ // - omp reference: `sessionSystemInstructionSent` Map was omp's approach;
878
+ // we chose stateless injection for simplicity and safety.
832
879
  if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
833
880
  const existingParts = request.systemInstruction?.parts ?? [];
834
881
  request.systemInstruction = {
835
882
  role: "user",
836
- parts: [
837
- { text: ANTIGRAVITY_SYSTEM_INSTRUCTION },
838
- { text: `Please ignore following [ignore]${ANTIGRAVITY_SYSTEM_INSTRUCTION}[/ignore]` },
839
- ...existingParts,
840
- ],
883
+ parts: [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }, ...existingParts],
884
+ };
885
+ request.preambleConfig = {
886
+ mode: "SYSTEM_INSTRUCTION_MODE_REPLACE",
841
887
  };
842
888
  }
843
889
 
@@ -15,27 +15,81 @@ export const getGeminiCliHeaders = (modelId?: string) => ({
15
15
  "Client-Metadata": "ideType=IDE_UNSPECIFIED,platform=PLATFORM_UNSPECIFIED,pluginType=GEMINI",
16
16
  });
17
17
 
18
- export const ANTIGRAVITY_SYSTEM_INSTRUCTION =
19
- "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding." +
20
- "You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question." +
21
- "**Absolute paths only**" +
22
- "**Proactiveness**";
23
18
  /**
24
- * Antigravity / Cloud Code Assist user agent. Lives in its own file so discovery
25
- * and usage code can read it without pulling the heavy google-gemini-cli provider
26
- * (and its @google/genai → google-auth-library dependency chain) into the startup
27
- * parse graph.
19
+ * Full Antigravity system instruction as observed in the real IDE binary.
20
+ * This is the complete prompt injected by the Antigravity language server,
21
+ * including BNF lexer definition for syntax highlighting, messaging system
22
+ * description, and reactive wakeup protocol.
23
+ *
24
+ * Wire-format fidelity note: The `%s` placeholders are literal in the real
25
+ * Antigravity IDE prompt. The Cloud Code Assist service either expands them
26
+ * server-side or the model handles them as-is. Preserved for byte-faithful
27
+ * emulation of the observed IDE wire format.
28
+ *
29
+ * Evidence: Disassembly of the Antigravity LS binary confirms this exact
30
+ * string is loaded and injected as the system instruction.
31
+ */
32
+ export const ANTIGRAVITY_SYSTEM_INSTRUCTION = `You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.
33
+ You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.
34
+ The USER will send you requests, which you must always prioritize addressing. Along with each USER request, we will attach additional metadata about their current state, such as what files they have open and where their cursor is.
35
+ This information may or may not be relevant to the coding task, it is up for you to decide.<lexer>
36
+ <config>
37
+ <name>BNF</name>
38
+ <alias>bnf</alias>
39
+ <filename>*.bnf</filename>
40
+ <mime_type>text/x-bnf</mime_type>
41
+ </config>
42
+ <rules>
43
+ <state name="root">
44
+ <rule pattern="(&lt;)([ -;=?-~]+)(&gt;)">
45
+ <bygroups>
46
+ <token type="Punctuation"/>
47
+ <token type="NameClass"/>
48
+ <token type="Punctuation"/>
49
+ </bygroups>
50
+ </rule>
51
+ <rule pattern="::=">
52
+ <token type="Operator"/>
53
+ </rule>
54
+ <rule pattern="[^&lt;&gt;:]+">
55
+ <token type="Text"/>
56
+ </rule>
57
+ <rule pattern=".">
58
+ <token type="Text"/>
59
+ </rule>
60
+ </state>
61
+ </rules>
62
+ </lexer>You are connected to a messaging system where you may receive messages from: %s.
63
+
64
+ ## Receiving Messages
65
+
66
+ You receive messages automatically at the start of each invocation. All messages are delivered in full directly into your context — no manual retrieval is needed.
67
+
68
+ ## Reactive Wakeup (No Polling Needed)
69
+
70
+ The system automatically resumes your execution when:
71
+ %s
72
+
73
+ This means you do **NOT** need to poll in a loop while waiting for messages or updates. After launching anything that performs work asynchronously, you may continue other work or simply stop by calling no more tools. The system will notify you when there is something to process.
74
+ `;
75
+ /**
76
+ * Antigravity / Cloud Code Assist user agent.
77
+ *
78
+ * Disassembly-confirmed: getUserAgentName() @ 0x5ecb1dd loads "antigravity-ide"
79
+ * via LEA RDX, [RIP-0x284fc90] → 0x367b554 = "antigravity-ide"
80
+ *
81
+ * The LS sets HTTP headers via: fmt.Sprintf("User-Agent: %s", getUserAgentName())
82
+ * So the final header is: User-Agent: antigravity-ide
83
+ *
84
+ * -override_user_agent flag can override this (confirmed at 0x5ecbc37).
28
85
  */
29
86
  export let getAntigravityUserAgent = () => {
30
- const DEFAULT_ANTIGRAVITY_VERSION = "1.104.0";
31
- const version = process.env.PI_AI_ANTIGRAVITY_VERSION || DEFAULT_ANTIGRAVITY_VERSION;
32
- // Map Node.js platform/arch to Antigravity's expected format.
33
- // Verified against Antigravity source: _qn() and wqn() in main.js.
34
- // process.platform: win32→windows, others pass through (darwin, linux)
35
- // process.arch: x64→amd64, ia32→386, others pass through (arm64)
36
- const os = process.platform === "win32" ? "windows" : process.platform;
37
- const arch = process.arch === "x64" ? "amd64" : process.arch === "ia32" ? "386" : process.arch;
38
- const userAgent = `antigravity/${version} ${os}/${arch}`;
87
+ const override = process.env.PI_AI_ANTIGRAVITY_USER_AGENT;
88
+ const userAgent = override || "antigravity-ide";
39
89
  getAntigravityUserAgent = () => userAgent;
40
90
  return userAgent;
41
91
  };
92
+
93
+ export const getAntigravityRequestHeaders = () => ({
94
+ "User-Agent": getAntigravityUserAgent(),
95
+ });
@@ -23,6 +23,7 @@ export interface InputItem {
23
23
  name?: string;
24
24
  output?: unknown;
25
25
  arguments?: unknown;
26
+ encrypted_content?: unknown;
26
27
  }
27
28
 
28
29
  export interface RequestBody {
@@ -62,6 +63,57 @@ function getReasoningConfig(model: Model<Api>, options: CodexRequestOptions): Re
62
63
  return config;
63
64
  }
64
65
 
66
+ function describeTextPartValue(value: unknown): string {
67
+ if (value === null) return "null";
68
+ if (Array.isArray(value)) return "array";
69
+ return typeof value;
70
+ }
71
+
72
+ function normalizeTextPartValue(value: unknown, path: string): string {
73
+ if (typeof value === "string") return value.toWellFormed();
74
+ try {
75
+ const encoded = JSON.stringify(value);
76
+ if (typeof encoded === "string") return encoded.toWellFormed();
77
+ } catch {
78
+ // Fall through to the actionable local error below.
79
+ }
80
+ throw new Error(
81
+ `Invalid Codex request text part at ${path}: expected a string or JSON-serializable value, received ${describeTextPartValue(value)}. Normalize compacted continuation content before sending to Codex.`,
82
+ );
83
+ }
84
+
85
+ function normalizeTextPartFields(content: unknown, path: string): unknown {
86
+ if (typeof content === "string") return content.toWellFormed();
87
+ if (!Array.isArray(content)) return content;
88
+ return content.map((part, index) => {
89
+ if (!part || typeof part !== "object") return part;
90
+ const normalizedPart = { ...(part as Record<string, unknown>) };
91
+ if ("text" in normalizedPart) {
92
+ normalizedPart.text = normalizeTextPartValue(normalizedPart.text, `${path}[${index}].text`);
93
+ }
94
+ return normalizedPart;
95
+ });
96
+ }
97
+
98
+ function normalizeInputTextPartFields(input: InputItem[] | undefined): InputItem[] | undefined {
99
+ if (!Array.isArray(input)) return input;
100
+ return input.map((item, itemIndex) => {
101
+ const normalizedItem = { ...item };
102
+ const itemRecord = normalizedItem as Record<string, unknown>;
103
+ if ("encrypted_content" in itemRecord) {
104
+ if (typeof itemRecord.encrypted_content === "string") {
105
+ itemRecord.encrypted_content = itemRecord.encrypted_content.toWellFormed();
106
+ } else {
107
+ delete itemRecord.encrypted_content;
108
+ }
109
+ }
110
+ if (normalizedItem.type === "message") {
111
+ normalizedItem.content = normalizeTextPartFields(normalizedItem.content, `input[${itemIndex}].content`);
112
+ }
113
+ return normalizedItem;
114
+ });
115
+ }
116
+
65
117
  function filterInput(input: InputItem[] | undefined): InputItem[] | undefined {
66
118
  if (!Array.isArray(input)) return input;
67
119
 
@@ -121,6 +173,7 @@ export async function transformRequestBody(
121
173
  return item;
122
174
  });
123
175
  }
176
+ body.input = normalizeInputTextPartFields(body.input);
124
177
  }
125
178
 
126
179
  if (prompt?.developerMessages && prompt.developerMessages.length > 0 && Array.isArray(body.input)) {
@@ -44,6 +44,7 @@ import {
44
44
  getOpenAIResponsesHistoryItems,
45
45
  getOpenAIResponsesHistoryPayload,
46
46
  normalizeSystemPrompts,
47
+ sanitizeOpenAIResponsesHistoryItemsForReplay,
47
48
  } from "../utils";
48
49
  import { AssistantMessageEventStream } from "../utils/event-stream";
49
50
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
@@ -2550,17 +2551,16 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
2550
2551
  for (const msg of transformedMessages) {
2551
2552
  if (msg.role === "user" || msg.role === "developer") {
2552
2553
  const providerPayload = (msg as { providerPayload?: AssistantMessage["providerPayload"] }).providerPayload;
2553
- const historyItems = getOpenAIResponsesHistoryItems(providerPayload, model.provider) as
2554
- | Array<ResponseInput[number]>
2555
- | undefined;
2554
+ const historyItems = getOpenAIResponsesHistoryItems(providerPayload, model.provider);
2556
2555
  if (historyItems) {
2557
- for (const item of historyItems) {
2556
+ const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(historyItems);
2557
+ for (const item of sanitizedHistoryItems) {
2558
2558
  const maybe = item as { type?: string; call_id?: string };
2559
2559
  if (maybe.type === "custom_tool_call" && typeof maybe.call_id === "string") {
2560
2560
  customCallIds.add(maybe.call_id);
2561
2561
  }
2562
2562
  }
2563
- messages.push(...historyItems);
2563
+ messages.push(...sanitizedHistoryItems);
2564
2564
  msgIndex += 1;
2565
2565
  continue;
2566
2566
  }
@@ -2579,18 +2579,19 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
2579
2579
  model.provider,
2580
2580
  assistantMsg.provider,
2581
2581
  );
2582
- const historyItems = providerPayload?.items as Array<ResponseInput[number]> | undefined;
2582
+ const historyItems = providerPayload?.items;
2583
2583
  if (historyItems) {
2584
- for (const item of historyItems) {
2584
+ const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(historyItems);
2585
+ for (const item of sanitizedHistoryItems) {
2585
2586
  const maybe = item as { type?: string; call_id?: string };
2586
2587
  if (maybe.type === "custom_tool_call" && typeof maybe.call_id === "string") {
2587
2588
  customCallIds.add(maybe.call_id);
2588
2589
  }
2589
2590
  }
2590
2591
  if (providerPayload?.dt) {
2591
- messages.push(...historyItems);
2592
+ messages.push(...sanitizedHistoryItems);
2592
2593
  } else {
2593
- messages.splice(0, messages.length, ...historyItems);
2594
+ messages.splice(0, messages.length, ...sanitizedHistoryItems);
2594
2595
  // Keep customCallIds from the pre-splice state since historyItems may re-introduce them.
2595
2596
  }
2596
2597
  msgIndex += 1;
@@ -103,6 +103,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
103
103
  provider === "opencode-go" ||
104
104
  baseUrl.includes("opencode.ai");
105
105
  const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
106
+ const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
106
107
 
107
108
  const useMaxTokens =
108
109
  provider === "mistral" ||
@@ -194,7 +195,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
194
195
  supportsReasoningEffort: !isGrok && !isZai,
195
196
  reasoningEffortMap,
196
197
  supportsUsageInStreaming: !isCerebras,
197
- disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
198
+ disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
198
199
  disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
199
200
  supportsToolChoice: !isDirectDeepseekReasoning,
200
201
  supportsForcedToolChoice: true,
package/src/stream.ts CHANGED
@@ -84,6 +84,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
84
84
  xai: "XAI_API_KEY",
85
85
  fireworks: "FIREWORKS_API_KEY",
86
86
  firepass: "FIREPASS_API_KEY",
87
+ fugu: "FUGU_API_KEY",
87
88
  openrouter: "OPENROUTER_API_KEY",
88
89
  kilo: "KILO_API_KEY",
89
90
  "vercel-ai-gateway": "AI_GATEWAY_API_KEY",
package/src/types.ts CHANGED
@@ -112,6 +112,7 @@ export type KnownProvider =
112
112
  | "github-copilot"
113
113
  | "fireworks"
114
114
  | "firepass"
115
+ | "fugu"
115
116
  | "gitlab-duo"
116
117
  | "cursor"
117
118
  | "deepseek"
@@ -0,0 +1,15 @@
1
+ /** Sakana Fugu login flow (API key paste against https://api.sakana.ai/v1). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginFugu = createApiKeyLogin({
5
+ providerLabel: "Sakana Fugu",
6
+ authUrl: "https://fugu.sakana.ai/",
7
+ instructions: "Create or copy your Sakana Fugu API key",
8
+ promptMessage: "Paste your Sakana Fugu API key",
9
+ placeholder: "fugu_...",
10
+ validation: {
11
+ kind: "models-endpoint",
12
+ provider: "Sakana Fugu",
13
+ modelsUrl: "https://api.sakana.ai/v1/models",
14
+ },
15
+ });
@@ -75,6 +75,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
75
75
  name: "Fire Pass (Fireworks Kimi K2.6 Turbo subscription)",
76
76
  available: true,
77
77
  },
78
+ {
79
+ id: "fugu",
80
+ name: "Sakana Fugu (API key)",
81
+ available: true,
82
+ },
78
83
  {
79
84
  id: "github-copilot",
80
85
  name: "GitHub Copilot",
@@ -353,6 +358,7 @@ export async function refreshOAuthToken(
353
358
  case "cerebras":
354
359
  case "fireworks":
355
360
  case "firepass":
361
+ case "fugu":
356
362
  case "nvidia":
357
363
  case "nanogpt":
358
364
  case "synthetic":
@@ -466,6 +472,9 @@ export async function getOAuthApiKey(
466
472
  return { newCredentials: creds, apiKey };
467
473
  }
468
474
 
475
+ export function resolveOAuthStorageProvider(provider: OAuthProviderId): OAuthProviderId {
476
+ return provider === "openai-codex-device" ? "openai-codex" : provider;
477
+ }
469
478
  /**
470
479
  * Get list of OAuth providers.
471
480
  */
@@ -17,6 +17,7 @@ export type OAuthProvider =
17
17
  | "deepseek"
18
18
  | "fireworks"
19
19
  | "firepass"
20
+ | "fugu"
20
21
  | "github-copilot"
21
22
  | "google-gemini-cli"
22
23
  | "google-antigravity"
@@ -52,14 +52,26 @@ const OVERFLOW_PATTERNS = [
52
52
  /\b413\b.*\b(request|payload|entity)\b.*\btoo large\b/i, // "413 Request Entity Too Large" variants
53
53
  /model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
54
54
  ];
55
+ /**
56
+ * Threshold below which a "successful" (stopReason "stop") response with empty
57
+ * content is considered anomalous. Some proxies (notably LiteLLM) return an
58
+ * empty `choices[0].message.content` with a near-zero `usage` (e.g. input: 1,
59
+ * output: 1) when the upstream model context window is exceeded, instead of
60
+ * surfacing a proper error. The total token count for such a response is well
61
+ * below any realistic turn, so we treat it as a proxy-level overflow signal.
62
+ */
63
+ const EMPTY_RESPONSE_USAGE_THRESHOLD = 5;
55
64
  /**
56
65
  * Check if an assistant message represents a context overflow error.
57
66
  *
58
- * This handles two cases:
67
+ * This handles three cases:
59
68
  * 1. Error-based overflow: Most providers return stopReason "error" with a
60
69
  * specific error message pattern.
61
70
  * 2. Silent overflow: Some providers accept overflow requests and return
62
71
  * successfully. For these, we check if usage.input exceeds the context window.
72
+ * 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
73
+ * response with empty content and a fabricated near-zero usage when the
74
+ * upstream model's context window is exceeded.
63
75
  *
64
76
  * ## Reliability by Provider
65
77
  *
@@ -83,6 +95,13 @@ const OVERFLOW_PATTERNS = [
83
95
  * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
84
96
  * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
85
97
  * - Ollama: Silently truncates input without error. Cannot be detected via this function.
98
+ * - LiteLLM proxy: Returns a "successful" response with empty content and a
99
+ * fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
100
+ * model's context window is exceeded. Detected via Case 3 (empty content +
101
+ * anomalously low usage). Note: the LiteLLM proxy's context limit may differ
102
+ * from the underlying model's advertised contextWindow (e.g. configured via
103
+ * `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
104
+ * usage.input against contextWindow) may not catch it.
86
105
  * The response will have usage.input < expected, but we don't know the expected value.
87
106
  *
88
107
  * ## Custom Providers
@@ -126,6 +145,21 @@ export function isContextOverflow(message: AssistantMessage, contextWindow?: num
126
145
  }
127
146
  }
128
147
 
148
+ // Case 3: Empty response with anomalously low usage (proxy-level overflow)
149
+ // Some proxies (e.g. LiteLLM) return a "successful" response (stopReason "stop")
150
+ // with empty content and a near-zero token count when the upstream model's
151
+ // context window is exceeded. This is distinct from silent overflow (Case 2),
152
+ // where the provider reports the real input token count. Here the proxy
153
+ // fabricates a bogus usage (input: 1, output: 1) that is far below any
154
+ // realistic turn, so we detect it heuristically.
155
+ if (
156
+ message.stopReason === "stop" &&
157
+ message.content.length === 0 &&
158
+ message.usage.input + message.usage.output <= EMPTY_RESPONSE_USAGE_THRESHOLD
159
+ ) {
160
+ return true;
161
+ }
162
+
129
163
  return false;
130
164
  }
131
165
 
package/src/utils.ts CHANGED
@@ -91,6 +91,92 @@ export function sanitizeOpenAIResponsesHistoryItemsForReplay(items: Array<Record
91
91
  return sanitized ? [sanitized] : [];
92
92
  });
93
93
  }
94
+ function stringifyResponsesStringParamForReplay(value: unknown): string {
95
+ if (typeof value === "string") return value.toWellFormed();
96
+ try {
97
+ const encoded = JSON.stringify(value);
98
+ if (typeof encoded === "string") return encoded.toWellFormed();
99
+ } catch {
100
+ // Fall through to String().
101
+ }
102
+ return String(value ?? "").toWellFormed();
103
+ }
104
+
105
+ function normalizeResponsesMessageTextForReplay(value: unknown): string {
106
+ if (typeof value === "string") return value.toWellFormed();
107
+ if (value && typeof value === "object") {
108
+ const nestedText = (value as { text?: unknown }).text;
109
+ if (typeof nestedText === "string") return nestedText.toWellFormed();
110
+ }
111
+ return stringifyResponsesStringParamForReplay(value);
112
+ }
113
+
114
+ type ResponsesImageDetail = "auto" | "low" | "high";
115
+
116
+ interface NormalizedResponsesImageUrl {
117
+ readonly imageUrl: string;
118
+ readonly detail?: ResponsesImageDetail;
119
+ }
120
+
121
+ function isResponsesImageDetail(value: unknown): value is ResponsesImageDetail {
122
+ return value === "auto" || value === "low" || value === "high";
123
+ }
124
+
125
+ function normalizeResponsesImageUrlForReplay(value: unknown): NormalizedResponsesImageUrl {
126
+ if (typeof value === "string") return { imageUrl: value.toWellFormed() };
127
+ if (value && typeof value === "object" && "url" in value && typeof value.url === "string") {
128
+ const detail = "detail" in value && isResponsesImageDetail(value.detail) ? value.detail : undefined;
129
+ return {
130
+ imageUrl: value.url.toWellFormed(),
131
+ ...(detail ? { detail } : {}),
132
+ };
133
+ }
134
+ return { imageUrl: stringifyResponsesStringParamForReplay(value) };
135
+ }
136
+
137
+ function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
138
+ if (typeof content === "string") return content.toWellFormed();
139
+ if (!Array.isArray(content)) return content;
140
+ return content.map(part => {
141
+ if (!part || typeof part !== "object") return part;
142
+ const sanitizedPart = { ...(part as Record<string, unknown>) };
143
+ if ("text" in sanitizedPart) {
144
+ sanitizedPart.text = normalizeResponsesMessageTextForReplay(sanitizedPart.text);
145
+ }
146
+ if ("image_url" in sanitizedPart) {
147
+ const normalizedImageUrl = normalizeResponsesImageUrlForReplay(sanitizedPart.image_url);
148
+ sanitizedPart.image_url = normalizedImageUrl.imageUrl;
149
+ if (sanitizedPart.type === "image_url") {
150
+ sanitizedPart.type = "input_image";
151
+ }
152
+ if (normalizedImageUrl.detail) {
153
+ sanitizedPart.detail = normalizedImageUrl.detail;
154
+ } else if ("detail" in sanitizedPart && !isResponsesImageDetail(sanitizedPart.detail)) {
155
+ delete sanitizedPart.detail;
156
+ }
157
+ }
158
+ return sanitizedPart;
159
+ });
160
+ }
161
+
162
+ function sanitizeResponsesStringFieldsForReplay(item: Record<string, unknown>): void {
163
+ if (item.type === "message") {
164
+ item.content = sanitizeResponsesMessageContentForReplay(item.content);
165
+ }
166
+ if (item.type === "function_call" && "arguments" in item && typeof item.arguments !== "string") {
167
+ item.arguments = stringifyResponsesStringParamForReplay(item.arguments);
168
+ }
169
+ if (item.type === "custom_tool_call" && "input" in item && typeof item.input !== "string") {
170
+ item.input = stringifyResponsesStringParamForReplay(item.input);
171
+ }
172
+ if (
173
+ (item.type === "function_call_output" || item.type === "custom_tool_call_output") &&
174
+ "output" in item &&
175
+ typeof item.output !== "string"
176
+ ) {
177
+ item.output = stringifyResponsesStringParamForReplay(item.output);
178
+ }
179
+ }
94
180
 
95
181
  function sanitizeOpenAIResponsesHistoryItemForReplay(
96
182
  item: Record<string, unknown>,
@@ -105,6 +191,7 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
105
191
  if (typeof item.call_id === "string") {
106
192
  sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
107
193
  }
194
+ sanitizeResponsesStringFieldsForReplay(sanitizedItem);
108
195
 
109
196
  return sanitizedItem as unknown as OpenAIResponsesReplayItem;
110
197
  }