@gajae-code/ai 0.7.1 → 0.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/dist/types/auth-gateway/http.d.ts +1 -0
- package/dist/types/provider-models/openai-compat.d.ts +5 -0
- package/dist/types/providers/google-gemini-cli.d.ts +4 -1
- package/dist/types/providers/google-gemini-headers.d.ts +27 -5
- package/dist/types/providers/mock.d.ts +2 -0
- package/dist/types/providers/openai-responses-shared.d.ts +16 -1
- package/dist/types/types.d.ts +9 -1
- package/dist/types/utils/json-parse.d.ts +8 -0
- package/dist/types/utils/oauth/fugu.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/dist/types/utils/overflow.d.ts +11 -1
- package/package.json +2 -2
- package/src/auth-gateway/http.ts +5 -1
- package/src/auth-gateway/server.ts +18 -1
- package/src/auth-storage.ts +6 -0
- package/src/models.json +66 -0
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +9 -0
- package/src/providers/google-gemini-cli.ts +64 -18
- package/src/providers/google-gemini-headers.ts +72 -18
- package/src/providers/mock.ts +3 -0
- package/src/providers/openai-codex-responses.ts +14 -2
- package/src/providers/openai-completions.ts +12 -1
- package/src/providers/openai-responses-shared.ts +43 -1
- package/src/stream.ts +1 -0
- package/src/types.ts +9 -0
- package/src/utils/json-parse.ts +18 -0
- package/src/utils/oauth/fugu.ts +15 -0
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/types.ts +1 -0
- package/src/utils/overflow.ts +35 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,27 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.7.3] - 2026-06-25
|
|
6
|
+
### Added
|
|
7
|
+
|
|
8
|
+
- Added the Sakana Fugu provider (`fugu`) with API-key auth (`FUGU_API_KEY`) and bundled catalog models (#1086).
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
|
|
12
|
+
- Wired Fugu API-key auth login so `gjc login fugu` stores a reusable `FUGU_API_KEY` credential instead of the provider having no supported login flow (#1090).
|
|
13
|
+
- Matched the real google-antigravity IDE request headers, system prompt, and preamble config so requests are accepted by Cloud Code Assist (#1080).
|
|
14
|
+
- `isContextOverflow` now detects a third proxy-level overflow case — an empty response with `stopReason: "stop"` and anomalously low usage (input + output ≤ 5 tokens), as emitted by some proxies (notably LiteLLM) when the upstream context window is exceeded — so callers surface overflow instead of treating it as a clean completion (#1102).
|
|
15
|
+
|
|
16
|
+
### Security
|
|
17
|
+
|
|
18
|
+
- In no-auth (tokenless) auth-gateway mode, requests carrying a browser `Origin` header are now rejected before CORS preflight handling or route dispatch, while local non-browser CLI clients keep the existing tokenless flow and token-configured browser clients keep the bearer-token/preflight flow (#1115).
|
|
19
|
+
|
|
20
|
+
## [0.7.2] - 2026-06-24
|
|
21
|
+
|
|
22
|
+
### Fixed
|
|
23
|
+
|
|
24
|
+
- Reject truncated or incomplete streamed tool calls instead of executing them with partial arguments, so a cut-off tool-call payload fails fast rather than running against a mismatched schema.
|
|
25
|
+
|
|
5
26
|
## [0.7.1] - 2026-06-23
|
|
6
27
|
|
|
7
28
|
### Changed
|
|
@@ -7,6 +7,7 @@ export declare function resolvePeer(req: Request): string;
|
|
|
7
7
|
*/
|
|
8
8
|
export declare function timingSafeEqual(a: Uint8Array, b: Uint8Array): boolean;
|
|
9
9
|
export declare function isAuthorized(req: Request, tokens: ReadonlySet<string>): boolean;
|
|
10
|
+
export declare function isNoAuthBrowserOriginRequest(req: Request, tokens: ReadonlySet<string>): boolean;
|
|
10
11
|
/**
|
|
11
12
|
* Extract allow-listed passthrough headers from an inbound request. Keys are
|
|
12
13
|
* lowercased; empty values are dropped. Called once per request in
|
|
@@ -75,6 +75,11 @@ export interface FirepassModelManagerConfig {
|
|
|
75
75
|
* See https://docs.fireworks.ai/firepass.
|
|
76
76
|
*/
|
|
77
77
|
export declare function firepassModelManagerOptions(_config?: FirepassModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
78
|
+
export interface FuguModelManagerConfig {
|
|
79
|
+
apiKey?: string;
|
|
80
|
+
baseUrl?: string;
|
|
81
|
+
}
|
|
82
|
+
export declare function fuguModelManagerOptions(config?: FuguModelManagerConfig): ModelManagerOptions<"openai-completions">;
|
|
78
83
|
export interface MistralModelManagerConfig {
|
|
79
84
|
apiKey?: string;
|
|
80
85
|
baseUrl?: string;
|
|
@@ -24,7 +24,7 @@ export interface GoogleGeminiCliOptions extends StreamOptions {
|
|
|
24
24
|
};
|
|
25
25
|
projectId?: string;
|
|
26
26
|
}
|
|
27
|
-
export { ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
|
27
|
+
export { ANTIGRAVITY_SYSTEM_INSTRUCTION, getAntigravityRequestHeaders, getGeminiCliHeaders, getGeminiCliUserAgent, } from "./google-gemini-headers";
|
|
28
28
|
interface ParsedGeminiCliCredentials {
|
|
29
29
|
accessToken: string;
|
|
30
30
|
projectId: string;
|
|
@@ -63,6 +63,9 @@ interface CloudCodeAssistRequest {
|
|
|
63
63
|
mode: FunctionCallingConfigMode;
|
|
64
64
|
};
|
|
65
65
|
};
|
|
66
|
+
preambleConfig?: {
|
|
67
|
+
mode: "SYSTEM_INSTRUCTION_MODE_REPLACE";
|
|
68
|
+
};
|
|
66
69
|
};
|
|
67
70
|
requestType?: string;
|
|
68
71
|
userAgent?: string;
|
|
@@ -8,11 +8,33 @@ export declare const getGeminiCliHeaders: (modelId?: string) => {
|
|
|
8
8
|
"User-Agent": string;
|
|
9
9
|
"Client-Metadata": string;
|
|
10
10
|
};
|
|
11
|
-
export declare const ANTIGRAVITY_SYSTEM_INSTRUCTION: string;
|
|
12
11
|
/**
|
|
13
|
-
* Antigravity
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
12
|
+
* Full Antigravity system instruction as observed in the real IDE binary.
|
|
13
|
+
* This is the complete prompt injected by the Antigravity language server,
|
|
14
|
+
* including BNF lexer definition for syntax highlighting, messaging system
|
|
15
|
+
* description, and reactive wakeup protocol.
|
|
16
|
+
*
|
|
17
|
+
* Wire-format fidelity note: The `%s` placeholders are literal in the real
|
|
18
|
+
* Antigravity IDE prompt. The Cloud Code Assist service either expands them
|
|
19
|
+
* server-side or the model handles them as-is. Preserved for byte-faithful
|
|
20
|
+
* emulation of the observed IDE wire format.
|
|
21
|
+
*
|
|
22
|
+
* Evidence: Disassembly of the Antigravity LS binary confirms this exact
|
|
23
|
+
* string is loaded and injected as the system instruction.
|
|
24
|
+
*/
|
|
25
|
+
export declare const ANTIGRAVITY_SYSTEM_INSTRUCTION = "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.\nYou are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.\nThe USER will send you requests, which you must always prioritize addressing. Along with each USER request, we will attach additional metadata about their current state, such as what files they have open and where their cursor is.\nThis information may or may not be relevant to the coding task, it is up for you to decide.<lexer>\n <config>\n <name>BNF</name>\n <alias>bnf</alias>\n <filename>*.bnf</filename>\n <mime_type>text/x-bnf</mime_type>\n </config>\n <rules>\n <state name=\"root\">\n <rule pattern=\"(<)([ -;=?-~]+)(>)\">\n <bygroups>\n <token type=\"Punctuation\"/>\n <token type=\"NameClass\"/>\n <token type=\"Punctuation\"/>\n </bygroups>\n </rule>\n <rule pattern=\"::=\">\n <token type=\"Operator\"/>\n </rule>\n <rule pattern=\"[^<>:]+\">\n <token type=\"Text\"/>\n </rule>\n <rule pattern=\".\">\n <token type=\"Text\"/>\n </rule>\n </state>\n </rules>\n</lexer>You are connected to a messaging system where you may receive messages from: %s.\n\n## Receiving Messages\n\nYou receive messages automatically at the start of each invocation. All messages are delivered in full directly into your context \u2014 no manual retrieval is needed.\n\n## Reactive Wakeup (No Polling Needed)\n\nThe system automatically resumes your execution when:\n%s\n\nThis means you do **NOT** need to poll in a loop while waiting for messages or updates. After launching anything that performs work asynchronously, you may continue other work or simply stop by calling no more tools. The system will notify you when there is something to process.\n";
|
|
26
|
+
/**
|
|
27
|
+
* Antigravity / Cloud Code Assist user agent.
|
|
28
|
+
*
|
|
29
|
+
* Disassembly-confirmed: getUserAgentName() @ 0x5ecb1dd loads "antigravity-ide"
|
|
30
|
+
* via LEA RDX, [RIP-0x284fc90] → 0x367b554 = "antigravity-ide"
|
|
31
|
+
*
|
|
32
|
+
* The LS sets HTTP headers via: fmt.Sprintf("User-Agent: %s", getUserAgentName())
|
|
33
|
+
* So the final header is: User-Agent: antigravity-ide
|
|
34
|
+
*
|
|
35
|
+
* -override_user_agent flag can override this (confirmed at 0x5ecbc37).
|
|
17
36
|
*/
|
|
18
37
|
export declare let getAntigravityUserAgent: () => string;
|
|
38
|
+
export declare const getAntigravityRequestHeaders: () => {
|
|
39
|
+
"User-Agent": string;
|
|
40
|
+
};
|
|
@@ -60,6 +60,8 @@ export type MockContent = string | {
|
|
|
60
60
|
name: string;
|
|
61
61
|
/** Object form is preferred; strings are passed through verbatim. */
|
|
62
62
|
arguments: Record<string, unknown> | string;
|
|
63
|
+
/** Simulate a provider-flagged truncated call (cut off mid-arguments). */
|
|
64
|
+
incompleteArguments?: boolean;
|
|
63
65
|
};
|
|
64
66
|
/** One scripted response. */
|
|
65
67
|
export interface MockResponse {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type OpenAI from "openai";
|
|
2
2
|
import type { ResponseInput, ResponseInputContent, ResponseOutputItem } from "openai/resources/responses/responses";
|
|
3
|
-
import { type Api, type AssistantMessage, type ImageContent, type Model, type ServiceTier, type StopReason, type StreamOptions, type TextContent, type TextSignatureV1, type ToolResultMessage } from "../types";
|
|
3
|
+
import { type Api, type AssistantMessage, type ImageContent, type Model, type ServiceTier, type StopReason, type StreamOptions, type TextContent, type TextSignatureV1, type ToolCall, type ToolResultMessage } from "../types";
|
|
4
4
|
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
5
5
|
export declare function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string;
|
|
6
6
|
export declare function parseTextSignature(signature: string | undefined): {
|
|
@@ -44,6 +44,21 @@ export interface ProcessResponsesStreamOptions {
|
|
|
44
44
|
onOutputItemDone?: (item: ResponseOutputItem) => void;
|
|
45
45
|
}
|
|
46
46
|
export declare function processResponsesStream<TApi extends Api>(openaiStream: AsyncIterable<OpenAI.Responses.ResponseStreamEvent>, output: AssistantMessage, stream: AssistantMessageEventStream, model: Model<TApi>, options?: ProcessResponsesStreamOptions): Promise<void>;
|
|
47
|
+
/**
|
|
48
|
+
* Mark tool-call blocks left incomplete by a length-truncated response so the
|
|
49
|
+
* agent loop rejects them instead of executing a best-effort partial parse.
|
|
50
|
+
*
|
|
51
|
+
* The universal signal is finalization: a call that never received its terminal
|
|
52
|
+
* `output_item.done` (passed in via `isFinalized`) was cut off mid-arguments.
|
|
53
|
+
* This covers both JSON function calls and raw-input custom tools without
|
|
54
|
+
* mis-flagging a *completed* custom tool whose raw input is not valid JSON. As a
|
|
55
|
+
* defensive secondary, a finalized JSON function call whose buffered arguments
|
|
56
|
+
* still don't parse (e.g. a misbehaving relay) is flagged too. No-op unless the
|
|
57
|
+
* turn stopped for length.
|
|
58
|
+
*
|
|
59
|
+
* Shared by both Responses providers (`openai-responses`, `openai-codex-responses`).
|
|
60
|
+
*/
|
|
61
|
+
export declare function flagTruncatedToolCalls(output: AssistantMessage, stopReason: StopReason, isFinalized: (block: ToolCall) => boolean): void;
|
|
47
62
|
export declare function mapOpenAIResponsesStopReason(status: OpenAI.Responses.ResponseStatus | undefined): StopReason;
|
|
48
63
|
/** Initial empty `AssistantMessage` that streaming providers accumulate into. */
|
|
49
64
|
export declare function createInitialResponsesAssistantMessage(api: Api, provider: string, modelId: string): AssistantMessage;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -48,7 +48,7 @@ export interface ThinkingConfig {
|
|
|
48
48
|
/** Provider-specific transport used to encode the selected effort. */
|
|
49
49
|
mode: ThinkingControlMode;
|
|
50
50
|
}
|
|
51
|
-
export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
51
|
+
export type KnownProvider = "alibaba-coding-plan" | "amazon-bedrock" | "azure-openai" | "anthropic" | "google" | "google-gemini-cli" | "google-antigravity" | "google-vertex" | "openai" | "openai-codex" | "kimi-code" | "minimax-code" | "minimax-code-cn" | "github-copilot" | "fireworks" | "firepass" | "fugu" | "gitlab-duo" | "cursor" | "deepseek" | "xai" | "groq" | "cerebras" | "openrouter" | "kilo" | "vercel-ai-gateway" | "zai" | "glm-zcode" | "mistral" | "minimax" | "opencode-go" | "opencode-zen" | "synthetic" | "cloudflare-ai-gateway" | "huggingface" | "litellm" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "qianfan" | "qwen-portal" | "together" | "venice" | "vllm" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "lm-studio";
|
|
52
52
|
export type Provider = KnownProvider | string;
|
|
53
53
|
import type { Effort } from "./model-thinking";
|
|
54
54
|
/** Token budgets for each thinking level (token-based providers only) */
|
|
@@ -338,6 +338,14 @@ export interface ToolCall {
|
|
|
338
338
|
* JSON function tools.
|
|
339
339
|
*/
|
|
340
340
|
customWireName?: string;
|
|
341
|
+
/**
|
|
342
|
+
* Set when the provider detected the argument JSON was truncated — the model
|
|
343
|
+
* hit its output-token limit (or the response was otherwise cut short) before
|
|
344
|
+
* emitting a complete arguments object. The `arguments` field then holds a
|
|
345
|
+
* best-effort partial parse and must not be executed as-is; the agent loop
|
|
346
|
+
* rejects the call with a retryable error instead.
|
|
347
|
+
*/
|
|
348
|
+
incompleteArguments?: boolean;
|
|
341
349
|
}
|
|
342
350
|
export interface Usage {
|
|
343
351
|
/** Non-cached input tokens (matches the bucket the provider bills as new input). */
|
|
@@ -8,3 +8,11 @@ export declare function parseJsonWithRepair<T>(json: string): T;
|
|
|
8
8
|
* @returns Parsed object or empty object if parsing fails
|
|
9
9
|
*/
|
|
10
10
|
export declare function parseStreamingJson<T = Record<string, unknown>>(partialJson: string | undefined): T;
|
|
11
|
+
/**
|
|
12
|
+
* Whether a string is a complete, well-formed JSON document (strict parse, no
|
|
13
|
+
* repair). Used to distinguish a tool-call argument blob that finished cleanly
|
|
14
|
+
* from one that was cut off mid-stream (truncation). An empty / whitespace-only
|
|
15
|
+
* string is treated as complete: a tool invoked with no arguments legitimately
|
|
16
|
+
* streams an empty buffer and must not be flagged as truncated.
|
|
17
|
+
*/
|
|
18
|
+
export declare function isCompleteJson(text: string | undefined): boolean;
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare const loginFugu: (options: import("./types").OAuthController) => Promise<string>;
|
|
@@ -7,7 +7,7 @@ export type OAuthCredentials = {
|
|
|
7
7
|
email?: string;
|
|
8
8
|
accountId?: string;
|
|
9
9
|
};
|
|
10
|
-
export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
10
|
+
export type OAuthProvider = "alibaba-coding-plan" | "anthropic" | "cerebras" | "cloudflare-ai-gateway" | "cursor" | "deepseek" | "fireworks" | "firepass" | "fugu" | "github-copilot" | "google-gemini-cli" | "google-antigravity" | "gitlab-duo" | "huggingface" | "kimi-code" | "kilo" | "kagi" | "litellm" | "lm-studio" | "minimax-code" | "minimax-code-cn" | "moonshot" | "nvidia" | "nanogpt" | "ollama" | "ollama-cloud" | "openai-codex" | "openai-codex-device" | "opencode-go" | "opencode-zen" | "parallel" | "perplexity" | "qianfan" | "qwen-portal" | "synthetic" | "tavily" | "together" | "venice" | "vercel-ai-gateway" | "vllm" | "xai" | "glm-zcode" | "xiaomi" | "xiaomi-token-plan-sgp" | "xiaomi-token-plan-ams" | "xiaomi-token-plan-cn" | "zenmux" | "zai";
|
|
11
11
|
export type OAuthProviderId = OAuthProvider | (string & {});
|
|
12
12
|
export type OAuthPrompt = {
|
|
13
13
|
message: string;
|
|
@@ -2,11 +2,14 @@ import type { AssistantMessage } from "../types";
|
|
|
2
2
|
/**
|
|
3
3
|
* Check if an assistant message represents a context overflow error.
|
|
4
4
|
*
|
|
5
|
-
* This handles
|
|
5
|
+
* This handles three cases:
|
|
6
6
|
* 1. Error-based overflow: Most providers return stopReason "error" with a
|
|
7
7
|
* specific error message pattern.
|
|
8
8
|
* 2. Silent overflow: Some providers accept overflow requests and return
|
|
9
9
|
* successfully. For these, we check if usage.input exceeds the context window.
|
|
10
|
+
* 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
|
|
11
|
+
* response with empty content and a fabricated near-zero usage when the
|
|
12
|
+
* upstream model's context window is exceeded.
|
|
10
13
|
*
|
|
11
14
|
* ## Reliability by Provider
|
|
12
15
|
*
|
|
@@ -30,6 +33,13 @@ import type { AssistantMessage } from "../types";
|
|
|
30
33
|
* - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
|
|
31
34
|
* sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
|
|
32
35
|
* - Ollama: Silently truncates input without error. Cannot be detected via this function.
|
|
36
|
+
* - LiteLLM proxy: Returns a "successful" response with empty content and a
|
|
37
|
+
* fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
|
|
38
|
+
* model's context window is exceeded. Detected via Case 3 (empty content +
|
|
39
|
+
* anomalously low usage). Note: the LiteLLM proxy's context limit may differ
|
|
40
|
+
* from the underlying model's advertised contextWindow (e.g. configured via
|
|
41
|
+
* `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
|
|
42
|
+
* usage.input against contextWindow) may not catch it.
|
|
33
43
|
* The response will have usage.input < expected, but we don't know the expected value.
|
|
34
44
|
*
|
|
35
45
|
* ## Custom Providers
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.7.
|
|
4
|
+
"version": "0.7.3",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gaebal-gajae.dev",
|
|
7
7
|
"author": "Yeachan-Heo",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
45
45
|
"@bufbuild/protobuf": "^2.12.0",
|
|
46
|
-
"@gajae-code/utils": "0.7.
|
|
46
|
+
"@gajae-code/utils": "0.7.3",
|
|
47
47
|
"openai": "^6.36.0",
|
|
48
48
|
"partial-json": "^0.1.7",
|
|
49
49
|
"zod": "4.4.3"
|
package/src/auth-gateway/http.ts
CHANGED
|
@@ -47,7 +47,7 @@ export function timingSafeEqual(a: Uint8Array, b: Uint8Array): boolean {
|
|
|
47
47
|
const TOKEN_ENCODER = new TextEncoder();
|
|
48
48
|
|
|
49
49
|
export function isAuthorized(req: Request, tokens: ReadonlySet<string>): boolean {
|
|
50
|
-
if (tokens.size === 0) return
|
|
50
|
+
if (tokens.size === 0) return !isNoAuthBrowserOriginRequest(req, tokens);
|
|
51
51
|
const header = req.headers.get("authorization");
|
|
52
52
|
if (!header) return false;
|
|
53
53
|
const match = header.match(/^Bearer\s+(.+)$/i);
|
|
@@ -63,6 +63,10 @@ export function isAuthorized(req: Request, tokens: ReadonlySet<string>): boolean
|
|
|
63
63
|
return ok;
|
|
64
64
|
}
|
|
65
65
|
|
|
66
|
+
export function isNoAuthBrowserOriginRequest(req: Request, tokens: ReadonlySet<string>): boolean {
|
|
67
|
+
return tokens.size === 0 && req.headers.has("origin");
|
|
68
|
+
}
|
|
69
|
+
|
|
66
70
|
/**
|
|
67
71
|
* Allow-list of inbound request headers that the gateway captures and forwards
|
|
68
72
|
* to the underlying parsers (which decide whether to surface them to the
|
|
@@ -27,7 +27,15 @@ import * as piNative from "../providers/pi-native-server";
|
|
|
27
27
|
import { streamSimple } from "../stream";
|
|
28
28
|
import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
|
|
29
29
|
import { parseBind } from "../utils/parse-bind";
|
|
30
|
-
import {
|
|
30
|
+
import {
|
|
31
|
+
captureRequestHeaders,
|
|
32
|
+
corsHeaders,
|
|
33
|
+
isAuthorized,
|
|
34
|
+
isNoAuthBrowserOriginRequest,
|
|
35
|
+
json,
|
|
36
|
+
resolvePeer,
|
|
37
|
+
withCors,
|
|
38
|
+
} from "./http";
|
|
31
39
|
import type {
|
|
32
40
|
AuthGatewayServerHandle,
|
|
33
41
|
AuthGatewayServerOptions,
|
|
@@ -640,6 +648,15 @@ export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServe
|
|
|
640
648
|
const url = new URL(req.url);
|
|
641
649
|
const pathname = url.pathname;
|
|
642
650
|
const peer = resolvePeer(req);
|
|
651
|
+
if (isNoAuthBrowserOriginRequest(req, tokens)) {
|
|
652
|
+
logger.info("auth-gateway no-auth browser-origin request rejected", {
|
|
653
|
+
method: req.method,
|
|
654
|
+
path: pathname,
|
|
655
|
+
peer,
|
|
656
|
+
origin: req.headers.get("origin"),
|
|
657
|
+
});
|
|
658
|
+
return json(403, { error: "browser origin requires bearer token" });
|
|
659
|
+
}
|
|
643
660
|
// CORS preflight is always answered without auth — browsers send
|
|
644
661
|
// preflights pre-authentication and a 401 here breaks the actual
|
|
645
662
|
// request before the bearer is ever attached.
|
package/src/auth-storage.ts
CHANGED
|
@@ -1613,6 +1613,12 @@ export class AuthStorage {
|
|
|
1613
1613
|
await saveApiKeyCredential(apiKey);
|
|
1614
1614
|
return;
|
|
1615
1615
|
}
|
|
1616
|
+
case "fugu": {
|
|
1617
|
+
const { loginFugu } = await import("./utils/oauth/fugu");
|
|
1618
|
+
const apiKey = await loginFugu(ctrl);
|
|
1619
|
+
await saveApiKeyCredential(apiKey);
|
|
1620
|
+
return;
|
|
1621
|
+
}
|
|
1616
1622
|
case "zai": {
|
|
1617
1623
|
const { loginZai } = await import("./utils/oauth/zai");
|
|
1618
1624
|
const apiKey = await loginZai(ctrl);
|
package/src/models.json
CHANGED
|
@@ -79757,5 +79757,71 @@
|
|
|
79757
79757
|
"maxLevel": "xhigh"
|
|
79758
79758
|
}
|
|
79759
79759
|
}
|
|
79760
|
+
},
|
|
79761
|
+
"fugu": {
|
|
79762
|
+
"fugu": {
|
|
79763
|
+
"id": "fugu",
|
|
79764
|
+
"name": "Sakana Fugu",
|
|
79765
|
+
"api": "openai-completions",
|
|
79766
|
+
"provider": "fugu",
|
|
79767
|
+
"baseUrl": "https://api.sakana.ai/v1",
|
|
79768
|
+
"reasoning": true,
|
|
79769
|
+
"input": [
|
|
79770
|
+
"text"
|
|
79771
|
+
],
|
|
79772
|
+
"cost": {
|
|
79773
|
+
"input": 0,
|
|
79774
|
+
"output": 0,
|
|
79775
|
+
"cacheRead": 0,
|
|
79776
|
+
"cacheWrite": 0
|
|
79777
|
+
},
|
|
79778
|
+
"contextWindow": 200000,
|
|
79779
|
+
"maxTokens": 65536,
|
|
79780
|
+
"compat": {
|
|
79781
|
+
"supportsStore": false,
|
|
79782
|
+
"supportsDeveloperRole": false,
|
|
79783
|
+
"supportsMultipleSystemMessages": false,
|
|
79784
|
+
"supportsReasoningEffort": false,
|
|
79785
|
+
"supportsUsageInStreaming": true,
|
|
79786
|
+
"maxTokensField": "max_tokens"
|
|
79787
|
+
},
|
|
79788
|
+
"thinking": {
|
|
79789
|
+
"mode": "effort",
|
|
79790
|
+
"minLevel": "minimal",
|
|
79791
|
+
"maxLevel": "high"
|
|
79792
|
+
}
|
|
79793
|
+
},
|
|
79794
|
+
"fugu-ultra": {
|
|
79795
|
+
"id": "fugu-ultra",
|
|
79796
|
+
"name": "Sakana Fugu Ultra",
|
|
79797
|
+
"api": "openai-completions",
|
|
79798
|
+
"provider": "fugu",
|
|
79799
|
+
"baseUrl": "https://api.sakana.ai/v1",
|
|
79800
|
+
"reasoning": true,
|
|
79801
|
+
"input": [
|
|
79802
|
+
"text"
|
|
79803
|
+
],
|
|
79804
|
+
"cost": {
|
|
79805
|
+
"input": 0,
|
|
79806
|
+
"output": 0,
|
|
79807
|
+
"cacheRead": 0,
|
|
79808
|
+
"cacheWrite": 0
|
|
79809
|
+
},
|
|
79810
|
+
"contextWindow": 200000,
|
|
79811
|
+
"maxTokens": 65536,
|
|
79812
|
+
"compat": {
|
|
79813
|
+
"supportsStore": false,
|
|
79814
|
+
"supportsDeveloperRole": false,
|
|
79815
|
+
"supportsMultipleSystemMessages": false,
|
|
79816
|
+
"supportsReasoningEffort": false,
|
|
79817
|
+
"supportsUsageInStreaming": true,
|
|
79818
|
+
"maxTokensField": "max_tokens"
|
|
79819
|
+
},
|
|
79820
|
+
"thinking": {
|
|
79821
|
+
"mode": "effort",
|
|
79822
|
+
"minLevel": "minimal",
|
|
79823
|
+
"maxLevel": "high"
|
|
79824
|
+
}
|
|
79825
|
+
}
|
|
79760
79826
|
}
|
|
79761
79827
|
}
|
|
@@ -16,6 +16,7 @@ import {
|
|
|
16
16
|
deepseekModelManagerOptions,
|
|
17
17
|
firepassModelManagerOptions,
|
|
18
18
|
fireworksModelManagerOptions,
|
|
19
|
+
fuguModelManagerOptions,
|
|
19
20
|
githubCopilotModelManagerOptions,
|
|
20
21
|
groqModelManagerOptions,
|
|
21
22
|
huggingfaceModelManagerOptions,
|
|
@@ -154,6 +155,12 @@ export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
|
154
155
|
catalog("Fireworks", ["FIREWORKS_API_KEY"]),
|
|
155
156
|
),
|
|
156
157
|
descriptor("firepass", "kimi-k2.6-turbo", config => firepassModelManagerOptions(config)),
|
|
158
|
+
catalogDescriptor(
|
|
159
|
+
"fugu",
|
|
160
|
+
"fugu",
|
|
161
|
+
config => fuguModelManagerOptions(config),
|
|
162
|
+
catalog("Sakana Fugu", ["FUGU_API_KEY"]),
|
|
163
|
+
),
|
|
157
164
|
descriptor("xai", "grok-4-fast-non-reasoning", config => xaiModelManagerOptions(config)),
|
|
158
165
|
catalogDescriptor(
|
|
159
166
|
"deepseek",
|
|
@@ -785,6 +785,15 @@ export function firepassModelManagerOptions(
|
|
|
785
785
|
// 7. Mistral
|
|
786
786
|
// ---------------------------------------------------------------------------
|
|
787
787
|
|
|
788
|
+
export interface FuguModelManagerConfig {
|
|
789
|
+
apiKey?: string;
|
|
790
|
+
baseUrl?: string;
|
|
791
|
+
}
|
|
792
|
+
|
|
793
|
+
export function fuguModelManagerOptions(config?: FuguModelManagerConfig): ModelManagerOptions<"openai-completions"> {
|
|
794
|
+
return createSimpleOpenAICompletionsOptions("fugu", config?.baseUrl ?? "https://api.sakana.ai/v1", config);
|
|
795
|
+
}
|
|
796
|
+
|
|
788
797
|
export interface MistralModelManagerConfig {
|
|
789
798
|
apiKey?: string;
|
|
790
799
|
baseUrl?: string;
|
|
@@ -30,7 +30,11 @@ import {
|
|
|
30
30
|
markToolChoiceIncapability,
|
|
31
31
|
resolveToolChoice,
|
|
32
32
|
} from "../utils/tool-choice-capability";
|
|
33
|
-
import {
|
|
33
|
+
import {
|
|
34
|
+
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
|
35
|
+
getAntigravityRequestHeaders,
|
|
36
|
+
getGeminiCliHeaders,
|
|
37
|
+
} from "./google-gemini-headers";
|
|
34
38
|
import type { Content, FunctionCallingConfigMode, ThinkingConfig } from "./google-shared";
|
|
35
39
|
import {
|
|
36
40
|
convertMessages,
|
|
@@ -78,7 +82,7 @@ const ANTIGRAVITY_ENDPOINT_FALLBACKS = [ANTIGRAVITY_DAILY_ENDPOINT, ANTIGRAVITY_
|
|
|
78
82
|
|
|
79
83
|
export {
|
|
80
84
|
ANTIGRAVITY_SYSTEM_INSTRUCTION,
|
|
81
|
-
|
|
85
|
+
getAntigravityRequestHeaders,
|
|
82
86
|
getGeminiCliHeaders,
|
|
83
87
|
getGeminiCliUserAgent,
|
|
84
88
|
} from "./google-gemini-headers";
|
|
@@ -220,6 +224,13 @@ interface CloudCodeAssistRequest {
|
|
|
220
224
|
mode: FunctionCallingConfigMode;
|
|
221
225
|
};
|
|
222
226
|
};
|
|
227
|
+
// Evidence: Real Antigravity IDE sends preambleConfig with
|
|
228
|
+
// SYSTEM_INSTRUCTION_MODE_REPLACE to control how the server
|
|
229
|
+
// interprets the system instruction (replace vs append).
|
|
230
|
+
// Confirmed via network interception of official IDE traffic.
|
|
231
|
+
preambleConfig?: {
|
|
232
|
+
mode: "SYSTEM_INSTRUCTION_MODE_REPLACE";
|
|
233
|
+
};
|
|
223
234
|
};
|
|
224
235
|
requestType?: string;
|
|
225
236
|
userAgent?: string;
|
|
@@ -319,7 +330,9 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
319
330
|
if (replacementPayload !== undefined) {
|
|
320
331
|
requestBody = replacementPayload as typeof requestBody;
|
|
321
332
|
}
|
|
322
|
-
const headers = isAntigravity
|
|
333
|
+
const headers = isAntigravity
|
|
334
|
+
? getAntigravityRequestHeaders() // Evidence: UA = "antigravity-ide" (disassembly 0x5ecb1dd)
|
|
335
|
+
: getGeminiCliHeaders(model.id);
|
|
323
336
|
|
|
324
337
|
const requestHeaders = {
|
|
325
338
|
Authorization: `Bearer ${accessToken}`,
|
|
@@ -786,9 +799,13 @@ export function buildRequest(
|
|
|
786
799
|
request.sessionId = deriveAntigravitySessionId(context);
|
|
787
800
|
}
|
|
788
801
|
|
|
789
|
-
// System instruction must be object with parts, not plain string
|
|
802
|
+
// System instruction must be object with parts, not plain string.
|
|
803
|
+
// Evidence: Real Antigravity IDE sets role: "user" on systemInstruction
|
|
804
|
+
// (confirmed in request body dumps from network interception).
|
|
805
|
+
// Only applied for Antigravity path — standard gemini-cli omits role.
|
|
790
806
|
if (systemPrompts.length > 0) {
|
|
791
807
|
request.systemInstruction = {
|
|
808
|
+
...(isAntigravity && { role: "user" }),
|
|
792
809
|
parts: systemPrompts.map(text => ({ text })),
|
|
793
810
|
};
|
|
794
811
|
}
|
|
@@ -818,26 +835,55 @@ export function buildRequest(
|
|
|
818
835
|
}
|
|
819
836
|
}
|
|
820
837
|
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
838
|
+
// Claude Antigravity: use VALIDATED mode only when tools are present
|
|
839
|
+
// and the caller hasn't explicitly requested "none" tool choice.
|
|
840
|
+
//
|
|
841
|
+
// Evidence:
|
|
842
|
+
// - Real Antigravity IDE sends VALIDATED for Claude requests with tools
|
|
843
|
+
// (confirmed via network interception of official IDE traffic)
|
|
844
|
+
// - The original GJC code used resolvedToolChoice.resolvedLevel to gate
|
|
845
|
+
// VALIDATED, but the real IDE doesn't check tool choice resolution —
|
|
846
|
+
// it sends VALIDATED whenever tools exist and toolChoice != "none"
|
|
847
|
+
// - omp reference: packages/ai/src/providers/google-gemini-cli.ts
|
|
848
|
+
// `antigravityClaudeToolConfig` guard pattern
|
|
849
|
+
if (
|
|
850
|
+
isAntigravity &&
|
|
851
|
+
isClaudeModel(model.id) &&
|
|
852
|
+
context.tools &&
|
|
853
|
+
context.tools.length > 0 &&
|
|
854
|
+
options?.toolChoice !== "none"
|
|
855
|
+
) {
|
|
856
|
+
request.toolConfig = {
|
|
857
|
+
functionCallingConfig: {
|
|
858
|
+
mode: "VALIDATED" as FunctionCallingConfigMode,
|
|
859
|
+
},
|
|
860
|
+
};
|
|
830
861
|
}
|
|
831
862
|
|
|
863
|
+
// Stateless system instruction injection: send on EVERY Antigravity request.
|
|
864
|
+
// This is safer than client-side session dedup because:
|
|
865
|
+
// - preambleConfig server-side persistence is unproven
|
|
866
|
+
// - A failed first request would poison the session if we skipped injection
|
|
867
|
+
// - Token cost is acceptable for Antigravity's long-context models
|
|
868
|
+
//
|
|
869
|
+
// Evidence:
|
|
870
|
+
// - Real Antigravity IDE sends the full system prompt on every request
|
|
871
|
+
// (confirmed via repeated request dumps — no client-side dedup observed)
|
|
872
|
+
// - The [ignore] wrapper was a GJC-original addition not present in the
|
|
873
|
+
// real IDE wire format. Removed for byte-faithful emulation.
|
|
874
|
+
// - preambleConfig: { mode: "SYSTEM_INSTRUCTION_MODE_REPLACE" } is the
|
|
875
|
+
// mechanism the real IDE uses to deliver system instructions. Without it,
|
|
876
|
+
// the server may append (not replace) the system prompt.
|
|
877
|
+
// - omp reference: `sessionSystemInstructionSent` Map was omp's approach;
|
|
878
|
+
// we chose stateless injection for simplicity and safety.
|
|
832
879
|
if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
|
|
833
880
|
const existingParts = request.systemInstruction?.parts ?? [];
|
|
834
881
|
request.systemInstruction = {
|
|
835
882
|
role: "user",
|
|
836
|
-
parts: [
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
],
|
|
883
|
+
parts: [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }, ...existingParts],
|
|
884
|
+
};
|
|
885
|
+
request.preambleConfig = {
|
|
886
|
+
mode: "SYSTEM_INSTRUCTION_MODE_REPLACE",
|
|
841
887
|
};
|
|
842
888
|
}
|
|
843
889
|
|
|
@@ -15,27 +15,81 @@ export const getGeminiCliHeaders = (modelId?: string) => ({
|
|
|
15
15
|
"Client-Metadata": "ideType=IDE_UNSPECIFIED,platform=PLATFORM_UNSPECIFIED,pluginType=GEMINI",
|
|
16
16
|
});
|
|
17
17
|
|
|
18
|
-
export const ANTIGRAVITY_SYSTEM_INSTRUCTION =
|
|
19
|
-
"You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding." +
|
|
20
|
-
"You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question." +
|
|
21
|
-
"**Absolute paths only**" +
|
|
22
|
-
"**Proactiveness**";
|
|
23
18
|
/**
|
|
24
|
-
* Antigravity
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
19
|
+
* Full Antigravity system instruction as observed in the real IDE binary.
|
|
20
|
+
* This is the complete prompt injected by the Antigravity language server,
|
|
21
|
+
* including BNF lexer definition for syntax highlighting, messaging system
|
|
22
|
+
* description, and reactive wakeup protocol.
|
|
23
|
+
*
|
|
24
|
+
* Wire-format fidelity note: The `%s` placeholders are literal in the real
|
|
25
|
+
* Antigravity IDE prompt. The Cloud Code Assist service either expands them
|
|
26
|
+
* server-side or the model handles them as-is. Preserved for byte-faithful
|
|
27
|
+
* emulation of the observed IDE wire format.
|
|
28
|
+
*
|
|
29
|
+
* Evidence: Disassembly of the Antigravity LS binary confirms this exact
|
|
30
|
+
* string is loaded and injected as the system instruction.
|
|
31
|
+
*/
|
|
32
|
+
export const ANTIGRAVITY_SYSTEM_INSTRUCTION = `You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.
|
|
33
|
+
You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.
|
|
34
|
+
The USER will send you requests, which you must always prioritize addressing. Along with each USER request, we will attach additional metadata about their current state, such as what files they have open and where their cursor is.
|
|
35
|
+
This information may or may not be relevant to the coding task, it is up for you to decide.<lexer>
|
|
36
|
+
<config>
|
|
37
|
+
<name>BNF</name>
|
|
38
|
+
<alias>bnf</alias>
|
|
39
|
+
<filename>*.bnf</filename>
|
|
40
|
+
<mime_type>text/x-bnf</mime_type>
|
|
41
|
+
</config>
|
|
42
|
+
<rules>
|
|
43
|
+
<state name="root">
|
|
44
|
+
<rule pattern="(<)([ -;=?-~]+)(>)">
|
|
45
|
+
<bygroups>
|
|
46
|
+
<token type="Punctuation"/>
|
|
47
|
+
<token type="NameClass"/>
|
|
48
|
+
<token type="Punctuation"/>
|
|
49
|
+
</bygroups>
|
|
50
|
+
</rule>
|
|
51
|
+
<rule pattern="::=">
|
|
52
|
+
<token type="Operator"/>
|
|
53
|
+
</rule>
|
|
54
|
+
<rule pattern="[^<>:]+">
|
|
55
|
+
<token type="Text"/>
|
|
56
|
+
</rule>
|
|
57
|
+
<rule pattern=".">
|
|
58
|
+
<token type="Text"/>
|
|
59
|
+
</rule>
|
|
60
|
+
</state>
|
|
61
|
+
</rules>
|
|
62
|
+
</lexer>You are connected to a messaging system where you may receive messages from: %s.
|
|
63
|
+
|
|
64
|
+
## Receiving Messages
|
|
65
|
+
|
|
66
|
+
You receive messages automatically at the start of each invocation. All messages are delivered in full directly into your context — no manual retrieval is needed.
|
|
67
|
+
|
|
68
|
+
## Reactive Wakeup (No Polling Needed)
|
|
69
|
+
|
|
70
|
+
The system automatically resumes your execution when:
|
|
71
|
+
%s
|
|
72
|
+
|
|
73
|
+
This means you do **NOT** need to poll in a loop while waiting for messages or updates. After launching anything that performs work asynchronously, you may continue other work or simply stop by calling no more tools. The system will notify you when there is something to process.
|
|
74
|
+
`;
|
|
75
|
+
/**
|
|
76
|
+
* Antigravity / Cloud Code Assist user agent.
|
|
77
|
+
*
|
|
78
|
+
* Disassembly-confirmed: getUserAgentName() @ 0x5ecb1dd loads "antigravity-ide"
|
|
79
|
+
* via LEA RDX, [RIP-0x284fc90] → 0x367b554 = "antigravity-ide"
|
|
80
|
+
*
|
|
81
|
+
* The LS sets HTTP headers via: fmt.Sprintf("User-Agent: %s", getUserAgentName())
|
|
82
|
+
* So the final header is: User-Agent: antigravity-ide
|
|
83
|
+
*
|
|
84
|
+
* -override_user_agent flag can override this (confirmed at 0x5ecbc37).
|
|
28
85
|
*/
|
|
29
86
|
export let getAntigravityUserAgent = () => {
|
|
30
|
-
const
|
|
31
|
-
const
|
|
32
|
-
// Map Node.js platform/arch to Antigravity's expected format.
|
|
33
|
-
// Verified against Antigravity source: _qn() and wqn() in main.js.
|
|
34
|
-
// process.platform: win32→windows, others pass through (darwin, linux)
|
|
35
|
-
// process.arch: x64→amd64, ia32→386, others pass through (arm64)
|
|
36
|
-
const os = process.platform === "win32" ? "windows" : process.platform;
|
|
37
|
-
const arch = process.arch === "x64" ? "amd64" : process.arch === "ia32" ? "386" : process.arch;
|
|
38
|
-
const userAgent = `antigravity/${version} ${os}/${arch}`;
|
|
87
|
+
const override = process.env.PI_AI_ANTIGRAVITY_USER_AGENT;
|
|
88
|
+
const userAgent = override || "antigravity-ide";
|
|
39
89
|
getAntigravityUserAgent = () => userAgent;
|
|
40
90
|
return userAgent;
|
|
41
91
|
};
|
|
92
|
+
|
|
93
|
+
export const getAntigravityRequestHeaders = () => ({
|
|
94
|
+
"User-Agent": getAntigravityUserAgent(),
|
|
95
|
+
});
|
package/src/providers/mock.ts
CHANGED
|
@@ -73,6 +73,8 @@ export type MockContent =
|
|
|
73
73
|
name: string;
|
|
74
74
|
/** Object form is preferred; strings are passed through verbatim. */
|
|
75
75
|
arguments: Record<string, unknown> | string;
|
|
76
|
+
/** Simulate a provider-flagged truncated call (cut off mid-arguments). */
|
|
77
|
+
incompleteArguments?: boolean;
|
|
76
78
|
};
|
|
77
79
|
|
|
78
80
|
/** One scripted response. */
|
|
@@ -416,6 +418,7 @@ function normalizeContent(input: MockContent, state: MockModel): TextContent | T
|
|
|
416
418
|
id: input.id ?? generateToolCallId(state),
|
|
417
419
|
name: input.name,
|
|
418
420
|
arguments: typeof input.arguments === "string" ? input.arguments : { ...input.arguments },
|
|
421
|
+
...(input.incompleteArguments ? { incompleteArguments: true } : {}),
|
|
419
422
|
} as ToolCall;
|
|
420
423
|
}
|
|
421
424
|
return input;
|
|
@@ -78,6 +78,7 @@ import {
|
|
|
78
78
|
convertResponsesInputContent,
|
|
79
79
|
encodeResponsesToolCallId,
|
|
80
80
|
encodeTextSignatureV1,
|
|
81
|
+
flagTruncatedToolCalls,
|
|
81
82
|
mapOpenAIResponsesStopReason,
|
|
82
83
|
populateResponsesUsageFromResponse,
|
|
83
84
|
} from "./openai-responses-shared";
|
|
@@ -254,6 +255,8 @@ interface CodexStreamRuntime {
|
|
|
254
255
|
providerRetryAttempt: number;
|
|
255
256
|
sawTerminalEvent: boolean;
|
|
256
257
|
canSafelyReplayWebsocketOverSse: boolean;
|
|
258
|
+
/** Ids of tool calls that received their terminal `output_item.done`. */
|
|
259
|
+
finalizedToolCallIds: Set<string>;
|
|
257
260
|
}
|
|
258
261
|
|
|
259
262
|
interface CodexStreamProcessingContext {
|
|
@@ -910,6 +913,7 @@ function createCodexStreamRuntime(initial: {
|
|
|
910
913
|
providerRetryAttempt: 0,
|
|
911
914
|
sawTerminalEvent: false,
|
|
912
915
|
canSafelyReplayWebsocketOverSse: true,
|
|
916
|
+
finalizedToolCallIds: new Set<string>(),
|
|
913
917
|
};
|
|
914
918
|
}
|
|
915
919
|
|
|
@@ -1267,9 +1271,11 @@ function handleOutputItemDone(
|
|
|
1267
1271
|
}
|
|
1268
1272
|
|
|
1269
1273
|
if (item.type === "function_call") {
|
|
1274
|
+
const id = encodeResponsesToolCallId(item.call_id, item.id);
|
|
1275
|
+
runtime.finalizedToolCallIds.add(id);
|
|
1270
1276
|
const toolCall: ToolCall = {
|
|
1271
1277
|
type: "toolCall",
|
|
1272
|
-
id
|
|
1278
|
+
id,
|
|
1273
1279
|
name: item.name,
|
|
1274
1280
|
arguments: parseStreamingJson(item.arguments || "{}"),
|
|
1275
1281
|
};
|
|
@@ -1279,13 +1285,15 @@ function handleOutputItemDone(
|
|
|
1279
1285
|
}
|
|
1280
1286
|
|
|
1281
1287
|
if (item.type === "custom_tool_call") {
|
|
1288
|
+
const id = encodeResponsesToolCallId(item.call_id, item.id);
|
|
1289
|
+
runtime.finalizedToolCallIds.add(id);
|
|
1282
1290
|
const rawInput =
|
|
1283
1291
|
runtime.currentBlock?.type === "toolCall" && runtime.currentBlock.partialJson
|
|
1284
1292
|
? runtime.currentBlock.partialJson
|
|
1285
1293
|
: (item.input ?? "");
|
|
1286
1294
|
const toolCall: ToolCall = {
|
|
1287
1295
|
type: "toolCall",
|
|
1288
|
-
id
|
|
1296
|
+
id,
|
|
1289
1297
|
name: item.name,
|
|
1290
1298
|
arguments: { input: rawInput },
|
|
1291
1299
|
customWireName: item.name,
|
|
@@ -1349,6 +1357,10 @@ function handleResponseCompleted(
|
|
|
1349
1357
|
calculateCost(model, output.usage);
|
|
1350
1358
|
applyCodexServiceTierPricing(model, output.usage, response?.service_tier, runtime.requestBodyForState.service_tier);
|
|
1351
1359
|
output.stopReason = mapOpenAIResponsesStopReason(response?.status as OpenAI.Responses.ResponseStatus | undefined);
|
|
1360
|
+
// A response cut short for length may have stopped mid-tool-call. Flag any
|
|
1361
|
+
// call that never received its `output_item.done` so the agent loop rejects
|
|
1362
|
+
// the truncated arguments instead of executing a best-effort partial parse.
|
|
1363
|
+
flagTruncatedToolCalls(output, output.stopReason, block => runtime.finalizedToolCallIds.has(block.id));
|
|
1352
1364
|
if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
|
|
1353
1365
|
output.stopReason = "toolUse";
|
|
1354
1366
|
}
|
|
@@ -51,7 +51,7 @@ import {
|
|
|
51
51
|
getStreamFirstEventTimeoutMs,
|
|
52
52
|
iterateWithIdleTimeout,
|
|
53
53
|
} from "../utils/idle-iterator";
|
|
54
|
-
import { parseStreamingJson } from "../utils/json-parse";
|
|
54
|
+
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
55
55
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
56
56
|
import { getKimiCommonHeaders } from "../utils/oauth/kimi";
|
|
57
57
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
@@ -889,6 +889,17 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
889
889
|
}
|
|
890
890
|
}
|
|
891
891
|
|
|
892
|
+
// A turn cut short for length may have stopped mid-tool-call. The open
|
|
893
|
+
// block's `partialArgs` would otherwise be repaired into a plausible-
|
|
894
|
+
// but-wrong object by `finishCurrentBlock`; flag it first so the agent
|
|
895
|
+
// loop rejects the truncated call instead of executing it.
|
|
896
|
+
if (output.stopReason === "length" && currentBlock?.type === "toolCall") {
|
|
897
|
+
const partial = (currentBlock as { partialArgs?: string }).partialArgs;
|
|
898
|
+
if (partial !== undefined && !isCompleteJson(partial)) {
|
|
899
|
+
currentBlock.incompleteArguments = true;
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
|
|
892
903
|
finishCurrentBlock(currentBlock);
|
|
893
904
|
|
|
894
905
|
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
|
@@ -30,7 +30,7 @@ import {
|
|
|
30
30
|
} from "../types";
|
|
31
31
|
import { normalizeResponsesToolCallId } from "../utils";
|
|
32
32
|
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
33
|
-
import { parseStreamingJson } from "../utils/json-parse";
|
|
33
|
+
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
34
34
|
import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
|
|
35
35
|
|
|
36
36
|
export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
|
|
@@ -710,6 +710,13 @@ export async function processResponsesStream<TApi extends Api>(
|
|
|
710
710
|
: "Unknown error (no error details in response)";
|
|
711
711
|
throw new Error(message);
|
|
712
712
|
}
|
|
713
|
+
// A response cut short for length (`incomplete`) may have stopped
|
|
714
|
+
// mid-tool-call. Any tool-call item still tracked in `items` never
|
|
715
|
+
// received its terminal `output_item.done`, so it was cut off; flag it
|
|
716
|
+
// (along with any finalized-but-unparseable JSON call) so the agent loop
|
|
717
|
+
// rejects it instead of executing repaired/partial arguments.
|
|
718
|
+
const openBlocks = new Set<unknown>(Array.from(items.values(), entry => entry.block));
|
|
719
|
+
flagTruncatedToolCalls(output, output.stopReason, block => !openBlocks.has(block));
|
|
713
720
|
if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
|
|
714
721
|
output.stopReason = "toolUse";
|
|
715
722
|
}
|
|
@@ -728,6 +735,41 @@ export async function processResponsesStream<TApi extends Api>(
|
|
|
728
735
|
}
|
|
729
736
|
}
|
|
730
737
|
|
|
738
|
+
/**
|
|
739
|
+
* Mark tool-call blocks left incomplete by a length-truncated response so the
|
|
740
|
+
* agent loop rejects them instead of executing a best-effort partial parse.
|
|
741
|
+
*
|
|
742
|
+
* The universal signal is finalization: a call that never received its terminal
|
|
743
|
+
* `output_item.done` (passed in via `isFinalized`) was cut off mid-arguments.
|
|
744
|
+
* This covers both JSON function calls and raw-input custom tools without
|
|
745
|
+
* mis-flagging a *completed* custom tool whose raw input is not valid JSON. As a
|
|
746
|
+
* defensive secondary, a finalized JSON function call whose buffered arguments
|
|
747
|
+
* still don't parse (e.g. a misbehaving relay) is flagged too. No-op unless the
|
|
748
|
+
* turn stopped for length.
|
|
749
|
+
*
|
|
750
|
+
* Shared by both Responses providers (`openai-responses`, `openai-codex-responses`).
|
|
751
|
+
*/
|
|
752
|
+
export function flagTruncatedToolCalls(
|
|
753
|
+
output: AssistantMessage,
|
|
754
|
+
stopReason: StopReason,
|
|
755
|
+
isFinalized: (block: ToolCall) => boolean,
|
|
756
|
+
): void {
|
|
757
|
+
if (stopReason !== "length") return;
|
|
758
|
+
for (const block of output.content) {
|
|
759
|
+
if (block.type !== "toolCall") continue;
|
|
760
|
+
if (!isFinalized(block)) {
|
|
761
|
+
block.incompleteArguments = true;
|
|
762
|
+
continue;
|
|
763
|
+
}
|
|
764
|
+
// Finalized: custom tools carry raw (non-JSON) input and are complete once
|
|
765
|
+
// finalized; only JSON function calls get the parse double-check.
|
|
766
|
+
if (!block.customWireName) {
|
|
767
|
+
const partial = (block as { partialJson?: string }).partialJson;
|
|
768
|
+
if (partial !== undefined && !isCompleteJson(partial)) block.incompleteArguments = true;
|
|
769
|
+
}
|
|
770
|
+
}
|
|
771
|
+
}
|
|
772
|
+
|
|
731
773
|
export function mapOpenAIResponsesStopReason(status: OpenAI.Responses.ResponseStatus | undefined): StopReason {
|
|
732
774
|
if (!status) return "stop";
|
|
733
775
|
switch (status) {
|
package/src/stream.ts
CHANGED
|
@@ -84,6 +84,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
84
84
|
xai: "XAI_API_KEY",
|
|
85
85
|
fireworks: "FIREWORKS_API_KEY",
|
|
86
86
|
firepass: "FIREPASS_API_KEY",
|
|
87
|
+
fugu: "FUGU_API_KEY",
|
|
87
88
|
openrouter: "OPENROUTER_API_KEY",
|
|
88
89
|
kilo: "KILO_API_KEY",
|
|
89
90
|
"vercel-ai-gateway": "AI_GATEWAY_API_KEY",
|
package/src/types.ts
CHANGED
|
@@ -112,6 +112,7 @@ export type KnownProvider =
|
|
|
112
112
|
| "github-copilot"
|
|
113
113
|
| "fireworks"
|
|
114
114
|
| "firepass"
|
|
115
|
+
| "fugu"
|
|
115
116
|
| "gitlab-duo"
|
|
116
117
|
| "cursor"
|
|
117
118
|
| "deepseek"
|
|
@@ -490,6 +491,14 @@ export interface ToolCall {
|
|
|
490
491
|
* JSON function tools.
|
|
491
492
|
*/
|
|
492
493
|
customWireName?: string;
|
|
494
|
+
/**
|
|
495
|
+
* Set when the provider detected the argument JSON was truncated — the model
|
|
496
|
+
* hit its output-token limit (or the response was otherwise cut short) before
|
|
497
|
+
* emitting a complete arguments object. The `arguments` field then holds a
|
|
498
|
+
* best-effort partial parse and must not be executed as-is; the agent loop
|
|
499
|
+
* rejects the call with a retryable error instead.
|
|
500
|
+
*/
|
|
501
|
+
incompleteArguments?: boolean;
|
|
493
502
|
}
|
|
494
503
|
|
|
495
504
|
export interface Usage {
|
package/src/utils/json-parse.ts
CHANGED
|
@@ -146,3 +146,21 @@ export function parseStreamingJson<T = Record<string, unknown>>(partialJson: str
|
|
|
146
146
|
}
|
|
147
147
|
}
|
|
148
148
|
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Whether a string is a complete, well-formed JSON document (strict parse, no
|
|
152
|
+
* repair). Used to distinguish a tool-call argument blob that finished cleanly
|
|
153
|
+
* from one that was cut off mid-stream (truncation). An empty / whitespace-only
|
|
154
|
+
* string is treated as complete: a tool invoked with no arguments legitimately
|
|
155
|
+
* streams an empty buffer and must not be flagged as truncated.
|
|
156
|
+
*/
|
|
157
|
+
export function isCompleteJson(text: string | undefined): boolean {
|
|
158
|
+
const trimmed = text?.trim();
|
|
159
|
+
if (!trimmed) return true;
|
|
160
|
+
try {
|
|
161
|
+
JSON.parse(trimmed);
|
|
162
|
+
return true;
|
|
163
|
+
} catch {
|
|
164
|
+
return false;
|
|
165
|
+
}
|
|
166
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/** Sakana Fugu login flow (API key paste against https://api.sakana.ai/v1). */
|
|
2
|
+
import { createApiKeyLogin } from "./api-key-login";
|
|
3
|
+
|
|
4
|
+
export const loginFugu = createApiKeyLogin({
|
|
5
|
+
providerLabel: "Sakana Fugu",
|
|
6
|
+
authUrl: "https://fugu.sakana.ai/",
|
|
7
|
+
instructions: "Create or copy your Sakana Fugu API key",
|
|
8
|
+
promptMessage: "Paste your Sakana Fugu API key",
|
|
9
|
+
placeholder: "fugu_...",
|
|
10
|
+
validation: {
|
|
11
|
+
kind: "models-endpoint",
|
|
12
|
+
provider: "Sakana Fugu",
|
|
13
|
+
modelsUrl: "https://api.sakana.ai/v1/models",
|
|
14
|
+
},
|
|
15
|
+
});
|
package/src/utils/oauth/index.ts
CHANGED
|
@@ -75,6 +75,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
|
|
|
75
75
|
name: "Fire Pass (Fireworks Kimi K2.6 Turbo subscription)",
|
|
76
76
|
available: true,
|
|
77
77
|
},
|
|
78
|
+
{
|
|
79
|
+
id: "fugu",
|
|
80
|
+
name: "Sakana Fugu (API key)",
|
|
81
|
+
available: true,
|
|
82
|
+
},
|
|
78
83
|
{
|
|
79
84
|
id: "github-copilot",
|
|
80
85
|
name: "GitHub Copilot",
|
|
@@ -353,6 +358,7 @@ export async function refreshOAuthToken(
|
|
|
353
358
|
case "cerebras":
|
|
354
359
|
case "fireworks":
|
|
355
360
|
case "firepass":
|
|
361
|
+
case "fugu":
|
|
356
362
|
case "nvidia":
|
|
357
363
|
case "nanogpt":
|
|
358
364
|
case "synthetic":
|
package/src/utils/oauth/types.ts
CHANGED
package/src/utils/overflow.ts
CHANGED
|
@@ -52,14 +52,26 @@ const OVERFLOW_PATTERNS = [
|
|
|
52
52
|
/\b413\b.*\b(request|payload|entity)\b.*\btoo large\b/i, // "413 Request Entity Too Large" variants
|
|
53
53
|
/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text
|
|
54
54
|
];
|
|
55
|
+
/**
|
|
56
|
+
* Threshold below which a "successful" (stopReason "stop") response with empty
|
|
57
|
+
* content is considered anomalous. Some proxies (notably LiteLLM) return an
|
|
58
|
+
* empty `choices[0].message.content` with a near-zero `usage` (e.g. input: 1,
|
|
59
|
+
* output: 1) when the upstream model context window is exceeded, instead of
|
|
60
|
+
* surfacing a proper error. The total token count for such a response is well
|
|
61
|
+
* below any realistic turn, so we treat it as a proxy-level overflow signal.
|
|
62
|
+
*/
|
|
63
|
+
const EMPTY_RESPONSE_USAGE_THRESHOLD = 5;
|
|
55
64
|
/**
|
|
56
65
|
* Check if an assistant message represents a context overflow error.
|
|
57
66
|
*
|
|
58
|
-
* This handles
|
|
67
|
+
* This handles three cases:
|
|
59
68
|
* 1. Error-based overflow: Most providers return stopReason "error" with a
|
|
60
69
|
* specific error message pattern.
|
|
61
70
|
* 2. Silent overflow: Some providers accept overflow requests and return
|
|
62
71
|
* successfully. For these, we check if usage.input exceeds the context window.
|
|
72
|
+
* 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
|
|
73
|
+
* response with empty content and a fabricated near-zero usage when the
|
|
74
|
+
* upstream model's context window is exceeded.
|
|
63
75
|
*
|
|
64
76
|
* ## Reliability by Provider
|
|
65
77
|
*
|
|
@@ -83,6 +95,13 @@ const OVERFLOW_PATTERNS = [
|
|
|
83
95
|
* - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
|
|
84
96
|
* sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
|
|
85
97
|
* - Ollama: Silently truncates input without error. Cannot be detected via this function.
|
|
98
|
+
* - LiteLLM proxy: Returns a "successful" response with empty content and a
|
|
99
|
+
* fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
|
|
100
|
+
* model's context window is exceeded. Detected via Case 3 (empty content +
|
|
101
|
+
* anomalously low usage). Note: the LiteLLM proxy's context limit may differ
|
|
102
|
+
* from the underlying model's advertised contextWindow (e.g. configured via
|
|
103
|
+
* `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
|
|
104
|
+
* usage.input against contextWindow) may not catch it.
|
|
86
105
|
* The response will have usage.input < expected, but we don't know the expected value.
|
|
87
106
|
*
|
|
88
107
|
* ## Custom Providers
|
|
@@ -126,6 +145,21 @@ export function isContextOverflow(message: AssistantMessage, contextWindow?: num
|
|
|
126
145
|
}
|
|
127
146
|
}
|
|
128
147
|
|
|
148
|
+
// Case 3: Empty response with anomalously low usage (proxy-level overflow)
|
|
149
|
+
// Some proxies (e.g. LiteLLM) return a "successful" response (stopReason "stop")
|
|
150
|
+
// with empty content and a near-zero token count when the upstream model's
|
|
151
|
+
// context window is exceeded. This is distinct from silent overflow (Case 2),
|
|
152
|
+
// where the provider reports the real input token count. Here the proxy
|
|
153
|
+
// fabricates a bogus usage (input: 1, output: 1) that is far below any
|
|
154
|
+
// realistic turn, so we detect it heuristically.
|
|
155
|
+
if (
|
|
156
|
+
message.stopReason === "stop" &&
|
|
157
|
+
message.content.length === 0 &&
|
|
158
|
+
message.usage.input + message.usage.output <= EMPTY_RESPONSE_USAGE_THRESHOLD
|
|
159
|
+
) {
|
|
160
|
+
return true;
|
|
161
|
+
}
|
|
162
|
+
|
|
129
163
|
return false;
|
|
130
164
|
}
|
|
131
165
|
|