@sayknow-cli/ai 0.3.0 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +36 -0
  2. package/dist/types/auth-gateway/http.d.ts +1 -0
  3. package/dist/types/model-thinking.d.ts +1 -1
  4. package/dist/types/provider-models/openai-compat.d.ts +5 -0
  5. package/dist/types/providers/google-gemini-cli.d.ts +4 -1
  6. package/dist/types/providers/google-gemini-headers.d.ts +27 -5
  7. package/dist/types/providers/mock.d.ts +2 -0
  8. package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -0
  9. package/dist/types/providers/openai-responses-shared.d.ts +16 -1
  10. package/dist/types/types.d.ts +9 -1
  11. package/dist/types/utils/json-parse.d.ts +8 -0
  12. package/dist/types/utils/oauth/fugu.d.ts +1 -0
  13. package/dist/types/utils/oauth/index.d.ts +1 -0
  14. package/dist/types/utils/oauth/types.d.ts +1 -1
  15. package/dist/types/utils/overflow.d.ts +11 -1
  16. package/package.json +2 -2
  17. package/src/auth-gateway/http.ts +5 -1
  18. package/src/auth-gateway/server.ts +18 -1
  19. package/src/auth-storage.ts +41 -27
  20. package/src/model-thinking.ts +2 -9
  21. package/src/models.json +67 -1
  22. package/src/provider-models/descriptors.ts +7 -0
  23. package/src/provider-models/openai-compat.ts +9 -0
  24. package/src/providers/google-gemini-cli.ts +64 -18
  25. package/src/providers/google-gemini-headers.ts +72 -18
  26. package/src/providers/mock.ts +3 -0
  27. package/src/providers/openai-codex/request-transformer.ts +53 -0
  28. package/src/providers/openai-codex-responses.ts +24 -11
  29. package/src/providers/openai-completions-compat.ts +2 -1
  30. package/src/providers/openai-completions.ts +12 -1
  31. package/src/providers/openai-responses-shared.ts +43 -1
  32. package/src/stream.ts +1 -0
  33. package/src/types.ts +9 -0
  34. package/src/utils/json-parse.ts +18 -0
  35. package/src/utils/oauth/fugu.ts +15 -0
  36. package/src/utils/oauth/index.ts +9 -0
  37. package/src/utils/oauth/types.ts +1 -0
  38. package/src/utils/overflow.ts +35 -1
  39. package/src/utils.ts +87 -0
@@ -30,7 +30,11 @@ import {
30
30
  markToolChoiceIncapability,
31
31
  resolveToolChoice,
32
32
  } from "../utils/tool-choice-capability";
33
- import { ANTIGRAVITY_SYSTEM_INSTRUCTION, getAntigravityUserAgent, getGeminiCliHeaders } from "./google-gemini-headers";
33
+ import {
34
+ ANTIGRAVITY_SYSTEM_INSTRUCTION,
35
+ getAntigravityRequestHeaders,
36
+ getGeminiCliHeaders,
37
+ } from "./google-gemini-headers";
34
38
  import type { Content, FunctionCallingConfigMode, ThinkingConfig } from "./google-shared";
35
39
  import {
36
40
  convertMessages,
@@ -78,7 +82,7 @@ const ANTIGRAVITY_ENDPOINT_FALLBACKS = [ANTIGRAVITY_DAILY_ENDPOINT, ANTIGRAVITY_
78
82
 
79
83
  export {
80
84
  ANTIGRAVITY_SYSTEM_INSTRUCTION,
81
- getAntigravityUserAgent,
85
+ getAntigravityRequestHeaders,
82
86
  getGeminiCliHeaders,
83
87
  getGeminiCliUserAgent,
84
88
  } from "./google-gemini-headers";
@@ -220,6 +224,13 @@ interface CloudCodeAssistRequest {
220
224
  mode: FunctionCallingConfigMode;
221
225
  };
222
226
  };
227
+ // Evidence: Real Antigravity IDE sends preambleConfig with
228
+ // SYSTEM_INSTRUCTION_MODE_REPLACE to control how the server
229
+ // interprets the system instruction (replace vs append).
230
+ // Confirmed via network interception of official IDE traffic.
231
+ preambleConfig?: {
232
+ mode: "SYSTEM_INSTRUCTION_MODE_REPLACE";
233
+ };
223
234
  };
224
235
  requestType?: string;
225
236
  userAgent?: string;
@@ -319,7 +330,9 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
319
330
  if (replacementPayload !== undefined) {
320
331
  requestBody = replacementPayload as typeof requestBody;
321
332
  }
322
- const headers = isAntigravity ? { "User-Agent": getAntigravityUserAgent() } : getGeminiCliHeaders(model.id);
333
+ const headers = isAntigravity
334
+ ? getAntigravityRequestHeaders() // Evidence: UA = "antigravity-ide" (disassembly 0x5ecb1dd)
335
+ : getGeminiCliHeaders(model.id);
323
336
 
324
337
  const requestHeaders = {
325
338
  Authorization: `Bearer ${accessToken}`,
@@ -786,9 +799,13 @@ export function buildRequest(
786
799
  request.sessionId = deriveAntigravitySessionId(context);
787
800
  }
788
801
 
789
- // System instruction must be object with parts, not plain string
802
+ // System instruction must be object with parts, not plain string.
803
+ // Evidence: Real Antigravity IDE sets role: "user" on systemInstruction
804
+ // (confirmed in request body dumps from network interception).
805
+ // Only applied for Antigravity path — standard gemini-cli omits role.
790
806
  if (systemPrompts.length > 0) {
791
807
  request.systemInstruction = {
808
+ ...(isAntigravity && { role: "user" }),
792
809
  parts: systemPrompts.map(text => ({ text })),
793
810
  };
794
811
  }
@@ -818,26 +835,55 @@ export function buildRequest(
818
835
  }
819
836
  }
820
837
 
821
- if (isAntigravity && isClaudeModel(model.id)) {
822
- const resolvedLevel = resolvedToolChoice.resolvedLevel;
823
- if (resolvedLevel === "named" || resolvedLevel === "required") {
824
- request.toolConfig = {
825
- functionCallingConfig: {
826
- mode: "VALIDATED" as FunctionCallingConfigMode,
827
- },
828
- };
829
- }
838
+ // Claude Antigravity: use VALIDATED mode only when tools are present
839
+ // and the caller hasn't explicitly requested "none" tool choice.
840
+ //
841
+ // Evidence:
842
+ // - Real Antigravity IDE sends VALIDATED for Claude requests with tools
843
+ // (confirmed via network interception of official IDE traffic)
844
+ // - The original SKC code used resolvedToolChoice.resolvedLevel to gate
845
+ // VALIDATED, but the real IDE doesn't check tool choice resolution —
846
+ // it sends VALIDATED whenever tools exist and toolChoice != "none"
847
+ // - omp reference: packages/ai/src/providers/google-gemini-cli.ts
848
+ // `antigravityClaudeToolConfig` guard pattern
849
+ if (
850
+ isAntigravity &&
851
+ isClaudeModel(model.id) &&
852
+ context.tools &&
853
+ context.tools.length > 0 &&
854
+ options?.toolChoice !== "none"
855
+ ) {
856
+ request.toolConfig = {
857
+ functionCallingConfig: {
858
+ mode: "VALIDATED" as FunctionCallingConfigMode,
859
+ },
860
+ };
830
861
  }
831
862
 
863
+ // Stateless system instruction injection: send on EVERY Antigravity request.
864
+ // This is safer than client-side session dedup because:
865
+ // - preambleConfig server-side persistence is unproven
866
+ // - A failed first request would poison the session if we skipped injection
867
+ // - Token cost is acceptable for Antigravity's long-context models
868
+ //
869
+ // Evidence:
870
+ // - Real Antigravity IDE sends the full system prompt on every request
871
+ // (confirmed via repeated request dumps — no client-side dedup observed)
872
+ // - The [ignore] wrapper was a SKC-original addition not present in the
873
+ // real IDE wire format. Removed for byte-faithful emulation.
874
+ // - preambleConfig: { mode: "SYSTEM_INSTRUCTION_MODE_REPLACE" } is the
875
+ // mechanism the real IDE uses to deliver system instructions. Without it,
876
+ // the server may append (not replace) the system prompt.
877
+ // - omp reference: `sessionSystemInstructionSent` Map was omp's approach;
878
+ // we chose stateless injection for simplicity and safety.
832
879
  if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
833
880
  const existingParts = request.systemInstruction?.parts ?? [];
834
881
  request.systemInstruction = {
835
882
  role: "user",
836
- parts: [
837
- { text: ANTIGRAVITY_SYSTEM_INSTRUCTION },
838
- { text: `Please ignore following [ignore]${ANTIGRAVITY_SYSTEM_INSTRUCTION}[/ignore]` },
839
- ...existingParts,
840
- ],
883
+ parts: [{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION }, ...existingParts],
884
+ };
885
+ request.preambleConfig = {
886
+ mode: "SYSTEM_INSTRUCTION_MODE_REPLACE",
841
887
  };
842
888
  }
843
889
 
@@ -15,27 +15,81 @@ export const getGeminiCliHeaders = (modelId?: string) => ({
15
15
  "Client-Metadata": "ideType=IDE_UNSPECIFIED,platform=PLATFORM_UNSPECIFIED,pluginType=GEMINI",
16
16
  });
17
17
 
18
- export const ANTIGRAVITY_SYSTEM_INSTRUCTION =
19
- "You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding." +
20
- "You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question." +
21
- "**Absolute paths only**" +
22
- "**Proactiveness**";
23
18
  /**
24
- * Antigravity / Cloud Code Assist user agent. Lives in its own file so discovery
25
- * and usage code can read it without pulling the heavy google-gemini-cli provider
26
- * (and its @google/genai → google-auth-library dependency chain) into the startup
27
- * parse graph.
19
+ * Full Antigravity system instruction as observed in the real IDE binary.
20
+ * This is the complete prompt injected by the Antigravity language server,
21
+ * including BNF lexer definition for syntax highlighting, messaging system
22
+ * description, and reactive wakeup protocol.
23
+ *
24
+ * Wire-format fidelity note: The `%s` placeholders are literal in the real
25
+ * Antigravity IDE prompt. The Cloud Code Assist service either expands them
26
+ * server-side or the model handles them as-is. Preserved for byte-faithful
27
+ * emulation of the observed IDE wire format.
28
+ *
29
+ * Evidence: Disassembly of the Antigravity LS binary confirms this exact
30
+ * string is loaded and injected as the system instruction.
31
+ */
32
+ export const ANTIGRAVITY_SYSTEM_INSTRUCTION = `You are Antigravity, a powerful agentic AI coding assistant designed by the Google Deepmind team working on Advanced Agentic Coding.
33
+ You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question.
34
+ The USER will send you requests, which you must always prioritize addressing. Along with each USER request, we will attach additional metadata about their current state, such as what files they have open and where their cursor is.
35
+ This information may or may not be relevant to the coding task, it is up for you to decide.<lexer>
36
+ <config>
37
+ <name>BNF</name>
38
+ <alias>bnf</alias>
39
+ <filename>*.bnf</filename>
40
+ <mime_type>text/x-bnf</mime_type>
41
+ </config>
42
+ <rules>
43
+ <state name="root">
44
+ <rule pattern="(&lt;)([ -;=?-~]+)(&gt;)">
45
+ <bygroups>
46
+ <token type="Punctuation"/>
47
+ <token type="NameClass"/>
48
+ <token type="Punctuation"/>
49
+ </bygroups>
50
+ </rule>
51
+ <rule pattern="::=">
52
+ <token type="Operator"/>
53
+ </rule>
54
+ <rule pattern="[^&lt;&gt;:]+">
55
+ <token type="Text"/>
56
+ </rule>
57
+ <rule pattern=".">
58
+ <token type="Text"/>
59
+ </rule>
60
+ </state>
61
+ </rules>
62
+ </lexer>You are connected to a messaging system where you may receive messages from: %s.
63
+
64
+ ## Receiving Messages
65
+
66
+ You receive messages automatically at the start of each invocation. All messages are delivered in full directly into your context — no manual retrieval is needed.
67
+
68
+ ## Reactive Wakeup (No Polling Needed)
69
+
70
+ The system automatically resumes your execution when:
71
+ %s
72
+
73
+ This means you do **NOT** need to poll in a loop while waiting for messages or updates. After launching anything that performs work asynchronously, you may continue other work or simply stop by calling no more tools. The system will notify you when there is something to process.
74
+ `;
75
+ /**
76
+ * Antigravity / Cloud Code Assist user agent.
77
+ *
78
+ * Disassembly-confirmed: getUserAgentName() @ 0x5ecb1dd loads "antigravity-ide"
79
+ * via LEA RDX, [RIP-0x284fc90] → 0x367b554 = "antigravity-ide"
80
+ *
81
+ * The LS sets HTTP headers via: fmt.Sprintf("User-Agent: %s", getUserAgentName())
82
+ * So the final header is: User-Agent: antigravity-ide
83
+ *
84
+ * -override_user_agent flag can override this (confirmed at 0x5ecbc37).
28
85
  */
29
86
  export let getAntigravityUserAgent = () => {
30
- const DEFAULT_ANTIGRAVITY_VERSION = "1.104.0";
31
- const version = process.env.PI_AI_ANTIGRAVITY_VERSION || DEFAULT_ANTIGRAVITY_VERSION;
32
- // Map Node.js platform/arch to Antigravity's expected format.
33
- // Verified against Antigravity source: _qn() and wqn() in main.js.
34
- // process.platform: win32→windows, others pass through (darwin, linux)
35
- // process.arch: x64→amd64, ia32→386, others pass through (arm64)
36
- const os = process.platform === "win32" ? "windows" : process.platform;
37
- const arch = process.arch === "x64" ? "amd64" : process.arch === "ia32" ? "386" : process.arch;
38
- const userAgent = `antigravity/${version} ${os}/${arch}`;
87
+ const override = process.env.PI_AI_ANTIGRAVITY_USER_AGENT;
88
+ const userAgent = override || "antigravity-ide";
39
89
  getAntigravityUserAgent = () => userAgent;
40
90
  return userAgent;
41
91
  };
92
+
93
+ export const getAntigravityRequestHeaders = () => ({
94
+ "User-Agent": getAntigravityUserAgent(),
95
+ });
@@ -73,6 +73,8 @@ export type MockContent =
73
73
  name: string;
74
74
  /** Object form is preferred; strings are passed through verbatim. */
75
75
  arguments: Record<string, unknown> | string;
76
+ /** Simulate a provider-flagged truncated call (cut off mid-arguments). */
77
+ incompleteArguments?: boolean;
76
78
  };
77
79
 
78
80
  /** One scripted response. */
@@ -416,6 +418,7 @@ function normalizeContent(input: MockContent, state: MockModel): TextContent | T
416
418
  id: input.id ?? generateToolCallId(state),
417
419
  name: input.name,
418
420
  arguments: typeof input.arguments === "string" ? input.arguments : { ...input.arguments },
421
+ ...(input.incompleteArguments ? { incompleteArguments: true } : {}),
419
422
  } as ToolCall;
420
423
  }
421
424
  return input;
@@ -23,6 +23,7 @@ export interface InputItem {
23
23
  name?: string;
24
24
  output?: unknown;
25
25
  arguments?: unknown;
26
+ encrypted_content?: unknown;
26
27
  }
27
28
 
28
29
  export interface RequestBody {
@@ -62,6 +63,57 @@ function getReasoningConfig(model: Model<Api>, options: CodexRequestOptions): Re
62
63
  return config;
63
64
  }
64
65
 
66
+ function describeTextPartValue(value: unknown): string {
67
+ if (value === null) return "null";
68
+ if (Array.isArray(value)) return "array";
69
+ return typeof value;
70
+ }
71
+
72
+ function normalizeTextPartValue(value: unknown, path: string): string {
73
+ if (typeof value === "string") return value.toWellFormed();
74
+ try {
75
+ const encoded = JSON.stringify(value);
76
+ if (typeof encoded === "string") return encoded.toWellFormed();
77
+ } catch {
78
+ // Fall through to the actionable local error below.
79
+ }
80
+ throw new Error(
81
+ `Invalid Codex request text part at ${path}: expected a string or JSON-serializable value, received ${describeTextPartValue(value)}. Normalize compacted continuation content before sending to Codex.`,
82
+ );
83
+ }
84
+
85
+ function normalizeTextPartFields(content: unknown, path: string): unknown {
86
+ if (typeof content === "string") return content.toWellFormed();
87
+ if (!Array.isArray(content)) return content;
88
+ return content.map((part, index) => {
89
+ if (!part || typeof part !== "object") return part;
90
+ const normalizedPart = { ...(part as Record<string, unknown>) };
91
+ if ("text" in normalizedPart) {
92
+ normalizedPart.text = normalizeTextPartValue(normalizedPart.text, `${path}[${index}].text`);
93
+ }
94
+ return normalizedPart;
95
+ });
96
+ }
97
+
98
+ function normalizeInputTextPartFields(input: InputItem[] | undefined): InputItem[] | undefined {
99
+ if (!Array.isArray(input)) return input;
100
+ return input.map((item, itemIndex) => {
101
+ const normalizedItem = { ...item };
102
+ const itemRecord = normalizedItem as Record<string, unknown>;
103
+ if ("encrypted_content" in itemRecord) {
104
+ if (typeof itemRecord.encrypted_content === "string") {
105
+ itemRecord.encrypted_content = itemRecord.encrypted_content.toWellFormed();
106
+ } else {
107
+ delete itemRecord.encrypted_content;
108
+ }
109
+ }
110
+ if (normalizedItem.type === "message") {
111
+ normalizedItem.content = normalizeTextPartFields(normalizedItem.content, `input[${itemIndex}].content`);
112
+ }
113
+ return normalizedItem;
114
+ });
115
+ }
116
+
65
117
  function filterInput(input: InputItem[] | undefined): InputItem[] | undefined {
66
118
  if (!Array.isArray(input)) return input;
67
119
 
@@ -121,6 +173,7 @@ export async function transformRequestBody(
121
173
  return item;
122
174
  });
123
175
  }
176
+ body.input = normalizeInputTextPartFields(body.input);
124
177
  }
125
178
 
126
179
  if (prompt?.developerMessages && prompt.developerMessages.length > 0 && Array.isArray(body.input)) {
@@ -44,6 +44,7 @@ import {
44
44
  getOpenAIResponsesHistoryItems,
45
45
  getOpenAIResponsesHistoryPayload,
46
46
  normalizeSystemPrompts,
47
+ sanitizeOpenAIResponsesHistoryItemsForReplay,
47
48
  } from "../utils";
48
49
  import { AssistantMessageEventStream } from "../utils/event-stream";
49
50
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
@@ -78,6 +79,7 @@ import {
78
79
  convertResponsesInputContent,
79
80
  encodeResponsesToolCallId,
80
81
  encodeTextSignatureV1,
82
+ flagTruncatedToolCalls,
81
83
  mapOpenAIResponsesStopReason,
82
84
  populateResponsesUsageFromResponse,
83
85
  } from "./openai-responses-shared";
@@ -254,6 +256,8 @@ interface CodexStreamRuntime {
254
256
  providerRetryAttempt: number;
255
257
  sawTerminalEvent: boolean;
256
258
  canSafelyReplayWebsocketOverSse: boolean;
259
+ /** Ids of tool calls that received their terminal `output_item.done`. */
260
+ finalizedToolCallIds: Set<string>;
257
261
  }
258
262
 
259
263
  interface CodexStreamProcessingContext {
@@ -910,6 +914,7 @@ function createCodexStreamRuntime(initial: {
910
914
  providerRetryAttempt: 0,
911
915
  sawTerminalEvent: false,
912
916
  canSafelyReplayWebsocketOverSse: true,
917
+ finalizedToolCallIds: new Set<string>(),
913
918
  };
914
919
  }
915
920
 
@@ -1267,9 +1272,11 @@ function handleOutputItemDone(
1267
1272
  }
1268
1273
 
1269
1274
  if (item.type === "function_call") {
1275
+ const id = encodeResponsesToolCallId(item.call_id, item.id);
1276
+ runtime.finalizedToolCallIds.add(id);
1270
1277
  const toolCall: ToolCall = {
1271
1278
  type: "toolCall",
1272
- id: encodeResponsesToolCallId(item.call_id, item.id),
1279
+ id,
1273
1280
  name: item.name,
1274
1281
  arguments: parseStreamingJson(item.arguments || "{}"),
1275
1282
  };
@@ -1279,13 +1286,15 @@ function handleOutputItemDone(
1279
1286
  }
1280
1287
 
1281
1288
  if (item.type === "custom_tool_call") {
1289
+ const id = encodeResponsesToolCallId(item.call_id, item.id);
1290
+ runtime.finalizedToolCallIds.add(id);
1282
1291
  const rawInput =
1283
1292
  runtime.currentBlock?.type === "toolCall" && runtime.currentBlock.partialJson
1284
1293
  ? runtime.currentBlock.partialJson
1285
1294
  : (item.input ?? "");
1286
1295
  const toolCall: ToolCall = {
1287
1296
  type: "toolCall",
1288
- id: encodeResponsesToolCallId(item.call_id, item.id),
1297
+ id,
1289
1298
  name: item.name,
1290
1299
  arguments: { input: rawInput },
1291
1300
  customWireName: item.name,
@@ -1349,6 +1358,10 @@ function handleResponseCompleted(
1349
1358
  calculateCost(model, output.usage);
1350
1359
  applyCodexServiceTierPricing(model, output.usage, response?.service_tier, runtime.requestBodyForState.service_tier);
1351
1360
  output.stopReason = mapOpenAIResponsesStopReason(response?.status as OpenAI.Responses.ResponseStatus | undefined);
1361
+ // A response cut short for length may have stopped mid-tool-call. Flag any
1362
+ // call that never received its `output_item.done` so the agent loop rejects
1363
+ // the truncated arguments instead of executing a best-effort partial parse.
1364
+ flagTruncatedToolCalls(output, output.stopReason, block => runtime.finalizedToolCallIds.has(block.id));
1352
1365
  if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
1353
1366
  output.stopReason = "toolUse";
1354
1367
  }
@@ -2538,17 +2551,16 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
2538
2551
  for (const msg of transformedMessages) {
2539
2552
  if (msg.role === "user" || msg.role === "developer") {
2540
2553
  const providerPayload = (msg as { providerPayload?: AssistantMessage["providerPayload"] }).providerPayload;
2541
- const historyItems = getOpenAIResponsesHistoryItems(providerPayload, model.provider) as
2542
- | Array<ResponseInput[number]>
2543
- | undefined;
2554
+ const historyItems = getOpenAIResponsesHistoryItems(providerPayload, model.provider);
2544
2555
  if (historyItems) {
2545
- for (const item of historyItems) {
2556
+ const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(historyItems);
2557
+ for (const item of sanitizedHistoryItems) {
2546
2558
  const maybe = item as { type?: string; call_id?: string };
2547
2559
  if (maybe.type === "custom_tool_call" && typeof maybe.call_id === "string") {
2548
2560
  customCallIds.add(maybe.call_id);
2549
2561
  }
2550
2562
  }
2551
- messages.push(...historyItems);
2563
+ messages.push(...sanitizedHistoryItems);
2552
2564
  msgIndex += 1;
2553
2565
  continue;
2554
2566
  }
@@ -2567,18 +2579,19 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
2567
2579
  model.provider,
2568
2580
  assistantMsg.provider,
2569
2581
  );
2570
- const historyItems = providerPayload?.items as Array<ResponseInput[number]> | undefined;
2582
+ const historyItems = providerPayload?.items;
2571
2583
  if (historyItems) {
2572
- for (const item of historyItems) {
2584
+ const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(historyItems);
2585
+ for (const item of sanitizedHistoryItems) {
2573
2586
  const maybe = item as { type?: string; call_id?: string };
2574
2587
  if (maybe.type === "custom_tool_call" && typeof maybe.call_id === "string") {
2575
2588
  customCallIds.add(maybe.call_id);
2576
2589
  }
2577
2590
  }
2578
2591
  if (providerPayload?.dt) {
2579
- messages.push(...historyItems);
2592
+ messages.push(...sanitizedHistoryItems);
2580
2593
  } else {
2581
- messages.splice(0, messages.length, ...historyItems);
2594
+ messages.splice(0, messages.length, ...sanitizedHistoryItems);
2582
2595
  // Keep customCallIds from the pre-splice state since historyItems may re-introduce them.
2583
2596
  }
2584
2597
  msgIndex += 1;
@@ -103,6 +103,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
103
103
  provider === "opencode-go" ||
104
104
  baseUrl.includes("opencode.ai");
105
105
  const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
106
+ const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
106
107
 
107
108
  const useMaxTokens =
108
109
  provider === "mistral" ||
@@ -194,7 +195,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
194
195
  supportsReasoningEffort: !isGrok && !isZai,
195
196
  reasoningEffortMap,
196
197
  supportsUsageInStreaming: !isCerebras,
197
- disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
198
+ disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
198
199
  disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
199
200
  supportsToolChoice: !isDirectDeepseekReasoning,
200
201
  supportsForcedToolChoice: true,
@@ -51,7 +51,7 @@ import {
51
51
  getStreamFirstEventTimeoutMs,
52
52
  iterateWithIdleTimeout,
53
53
  } from "../utils/idle-iterator";
54
- import { parseStreamingJson } from "../utils/json-parse";
54
+ import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
55
55
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
56
56
  import { getKimiCommonHeaders } from "../utils/oauth/kimi";
57
57
  import { notifyProviderResponse } from "../utils/provider-response";
@@ -889,6 +889,17 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
889
889
  }
890
890
  }
891
891
 
892
+ // A turn cut short for length may have stopped mid-tool-call. The open
893
+ // block's `partialArgs` would otherwise be repaired into a plausible-
894
+ // but-wrong object by `finishCurrentBlock`; flag it first so the agent
895
+ // loop rejects the truncated call instead of executing it.
896
+ if (output.stopReason === "length" && currentBlock?.type === "toolCall") {
897
+ const partial = (currentBlock as { partialArgs?: string }).partialArgs;
898
+ if (partial !== undefined && !isCompleteJson(partial)) {
899
+ currentBlock.incompleteArguments = true;
900
+ }
901
+ }
902
+
892
903
  finishCurrentBlock(currentBlock);
893
904
 
894
905
  const firstEventTimeoutError = abortTracker.getLocalAbortReason();
@@ -30,7 +30,7 @@ import {
30
30
  } from "../types";
31
31
  import { normalizeResponsesToolCallId } from "../utils";
32
32
  import type { AssistantMessageEventStream } from "../utils/event-stream";
33
- import { parseStreamingJson } from "../utils/json-parse";
33
+ import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
34
34
  import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
35
35
 
36
36
  export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
@@ -710,6 +710,13 @@ export async function processResponsesStream<TApi extends Api>(
710
710
  : "Unknown error (no error details in response)";
711
711
  throw new Error(message);
712
712
  }
713
+ // A response cut short for length (`incomplete`) may have stopped
714
+ // mid-tool-call. Any tool-call item still tracked in `items` never
715
+ // received its terminal `output_item.done`, so it was cut off; flag it
716
+ // (along with any finalized-but-unparseable JSON call) so the agent loop
717
+ // rejects it instead of executing repaired/partial arguments.
718
+ const openBlocks = new Set<unknown>(Array.from(items.values(), entry => entry.block));
719
+ flagTruncatedToolCalls(output, output.stopReason, block => !openBlocks.has(block));
713
720
  if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
714
721
  output.stopReason = "toolUse";
715
722
  }
@@ -728,6 +735,41 @@ export async function processResponsesStream<TApi extends Api>(
728
735
  }
729
736
  }
730
737
 
738
+ /**
739
+ * Mark tool-call blocks left incomplete by a length-truncated response so the
740
+ * agent loop rejects them instead of executing a best-effort partial parse.
741
+ *
742
+ * The universal signal is finalization: a call that never received its terminal
743
+ * `output_item.done` (passed in via `isFinalized`) was cut off mid-arguments.
744
+ * This covers both JSON function calls and raw-input custom tools without
745
+ * mis-flagging a *completed* custom tool whose raw input is not valid JSON. As a
746
+ * defensive secondary, a finalized JSON function call whose buffered arguments
747
+ * still don't parse (e.g. a misbehaving relay) is flagged too. No-op unless the
748
+ * turn stopped for length.
749
+ *
750
+ * Shared by both Responses providers (`openai-responses`, `openai-codex-responses`).
751
+ */
752
+ export function flagTruncatedToolCalls(
753
+ output: AssistantMessage,
754
+ stopReason: StopReason,
755
+ isFinalized: (block: ToolCall) => boolean,
756
+ ): void {
757
+ if (stopReason !== "length") return;
758
+ for (const block of output.content) {
759
+ if (block.type !== "toolCall") continue;
760
+ if (!isFinalized(block)) {
761
+ block.incompleteArguments = true;
762
+ continue;
763
+ }
764
+ // Finalized: custom tools carry raw (non-JSON) input and are complete once
765
+ // finalized; only JSON function calls get the parse double-check.
766
+ if (!block.customWireName) {
767
+ const partial = (block as { partialJson?: string }).partialJson;
768
+ if (partial !== undefined && !isCompleteJson(partial)) block.incompleteArguments = true;
769
+ }
770
+ }
771
+ }
772
+
731
773
  export function mapOpenAIResponsesStopReason(status: OpenAI.Responses.ResponseStatus | undefined): StopReason {
732
774
  if (!status) return "stop";
733
775
  switch (status) {
package/src/stream.ts CHANGED
@@ -84,6 +84,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
84
84
  xai: "XAI_API_KEY",
85
85
  fireworks: "FIREWORKS_API_KEY",
86
86
  firepass: "FIREPASS_API_KEY",
87
+ fugu: "FUGU_API_KEY",
87
88
  openrouter: "OPENROUTER_API_KEY",
88
89
  kilo: "KILO_API_KEY",
89
90
  "vercel-ai-gateway": "AI_GATEWAY_API_KEY",
package/src/types.ts CHANGED
@@ -112,6 +112,7 @@ export type KnownProvider =
112
112
  | "github-copilot"
113
113
  | "fireworks"
114
114
  | "firepass"
115
+ | "fugu"
115
116
  | "gitlab-duo"
116
117
  | "cursor"
117
118
  | "deepseek"
@@ -490,6 +491,14 @@ export interface ToolCall {
490
491
  * JSON function tools.
491
492
  */
492
493
  customWireName?: string;
494
+ /**
495
+ * Set when the provider detected the argument JSON was truncated — the model
496
+ * hit its output-token limit (or the response was otherwise cut short) before
497
+ * emitting a complete arguments object. The `arguments` field then holds a
498
+ * best-effort partial parse and must not be executed as-is; the agent loop
499
+ * rejects the call with a retryable error instead.
500
+ */
501
+ incompleteArguments?: boolean;
493
502
  }
494
503
 
495
504
  export interface Usage {
@@ -146,3 +146,21 @@ export function parseStreamingJson<T = Record<string, unknown>>(partialJson: str
146
146
  }
147
147
  }
148
148
  }
149
+
150
+ /**
151
+ * Whether a string is a complete, well-formed JSON document (strict parse, no
152
+ * repair). Used to distinguish a tool-call argument blob that finished cleanly
153
+ * from one that was cut off mid-stream (truncation). An empty / whitespace-only
154
+ * string is treated as complete: a tool invoked with no arguments legitimately
155
+ * streams an empty buffer and must not be flagged as truncated.
156
+ */
157
+ export function isCompleteJson(text: string | undefined): boolean {
158
+ const trimmed = text?.trim();
159
+ if (!trimmed) return true;
160
+ try {
161
+ JSON.parse(trimmed);
162
+ return true;
163
+ } catch {
164
+ return false;
165
+ }
166
+ }
@@ -0,0 +1,15 @@
1
+ /** Sakana Fugu login flow (API key paste against https://api.sakana.ai/v1). */
2
+ import { createApiKeyLogin } from "./api-key-login";
3
+
4
+ export const loginFugu = createApiKeyLogin({
5
+ providerLabel: "Sakana Fugu",
6
+ authUrl: "https://fugu.sakana.ai/",
7
+ instructions: "Create or copy your Sakana Fugu API key",
8
+ promptMessage: "Paste your Sakana Fugu API key",
9
+ placeholder: "fugu_...",
10
+ validation: {
11
+ kind: "models-endpoint",
12
+ provider: "Sakana Fugu",
13
+ modelsUrl: "https://api.sakana.ai/v1/models",
14
+ },
15
+ });
@@ -75,6 +75,11 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
75
75
  name: "Fire Pass (Fireworks Kimi K2.6 Turbo subscription)",
76
76
  available: true,
77
77
  },
78
+ {
79
+ id: "fugu",
80
+ name: "Sakana Fugu (API key)",
81
+ available: true,
82
+ },
78
83
  {
79
84
  id: "github-copilot",
80
85
  name: "GitHub Copilot",
@@ -353,6 +358,7 @@ export async function refreshOAuthToken(
353
358
  case "cerebras":
354
359
  case "fireworks":
355
360
  case "firepass":
361
+ case "fugu":
356
362
  case "nvidia":
357
363
  case "nanogpt":
358
364
  case "synthetic":
@@ -466,6 +472,9 @@ export async function getOAuthApiKey(
466
472
  return { newCredentials: creds, apiKey };
467
473
  }
468
474
 
475
+ export function resolveOAuthStorageProvider(provider: OAuthProviderId): OAuthProviderId {
476
+ return provider === "openai-codex-device" ? "openai-codex" : provider;
477
+ }
469
478
  /**
470
479
  * Get list of OAuth providers.
471
480
  */
@@ -17,6 +17,7 @@ export type OAuthProvider =
17
17
  | "deepseek"
18
18
  | "fireworks"
19
19
  | "firepass"
20
+ | "fugu"
20
21
  | "github-copilot"
21
22
  | "google-gemini-cli"
22
23
  | "google-antigravity"