@sayknow-cli/ai 0.3.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,12 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.7.2] - 2026-06-24
6
+
7
+ ### Fixed
8
+
9
+ - Reject truncated or incomplete streamed tool calls instead of executing them with partial arguments, so a cut-off tool-call payload fails fast rather than running against a mismatched schema.
10
+
5
11
  ## [0.7.1] - 2026-06-23
6
12
 
7
13
  ### Changed
@@ -60,6 +60,8 @@ export type MockContent = string | {
60
60
  name: string;
61
61
  /** Object form is preferred; strings are passed through verbatim. */
62
62
  arguments: Record<string, unknown> | string;
63
+ /** Simulate a provider-flagged truncated call (cut off mid-arguments). */
64
+ incompleteArguments?: boolean;
63
65
  };
64
66
  /** One scripted response. */
65
67
  export interface MockResponse {
@@ -1,6 +1,6 @@
1
1
  import type OpenAI from "openai";
2
2
  import type { ResponseInput, ResponseInputContent, ResponseOutputItem } from "openai/resources/responses/responses";
3
- import { type Api, type AssistantMessage, type ImageContent, type Model, type ServiceTier, type StopReason, type StreamOptions, type TextContent, type TextSignatureV1, type ToolResultMessage } from "../types";
3
+ import { type Api, type AssistantMessage, type ImageContent, type Model, type ServiceTier, type StopReason, type StreamOptions, type TextContent, type TextSignatureV1, type ToolCall, type ToolResultMessage } from "../types";
4
4
  import type { AssistantMessageEventStream } from "../utils/event-stream";
5
5
  export declare function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string;
6
6
  export declare function parseTextSignature(signature: string | undefined): {
@@ -44,6 +44,21 @@ export interface ProcessResponsesStreamOptions {
44
44
  onOutputItemDone?: (item: ResponseOutputItem) => void;
45
45
  }
46
46
  export declare function processResponsesStream<TApi extends Api>(openaiStream: AsyncIterable<OpenAI.Responses.ResponseStreamEvent>, output: AssistantMessage, stream: AssistantMessageEventStream, model: Model<TApi>, options?: ProcessResponsesStreamOptions): Promise<void>;
47
+ /**
48
+ * Mark tool-call blocks left incomplete by a length-truncated response so the
49
+ * agent loop rejects them instead of executing a best-effort partial parse.
50
+ *
51
+ * The universal signal is finalization: a call that never received its terminal
52
+ * `output_item.done` (passed in via `isFinalized`) was cut off mid-arguments.
53
+ * This covers both JSON function calls and raw-input custom tools without
54
+ * mis-flagging a *completed* custom tool whose raw input is not valid JSON. As a
55
+ * defensive secondary, a finalized JSON function call whose buffered arguments
56
+ * still don't parse (e.g. a misbehaving relay) is flagged too. No-op unless the
57
+ * turn stopped for length.
58
+ *
59
+ * Shared by both Responses providers (`openai-responses`, `openai-codex-responses`).
60
+ */
61
+ export declare function flagTruncatedToolCalls(output: AssistantMessage, stopReason: StopReason, isFinalized: (block: ToolCall) => boolean): void;
47
62
  export declare function mapOpenAIResponsesStopReason(status: OpenAI.Responses.ResponseStatus | undefined): StopReason;
48
63
  /** Initial empty `AssistantMessage` that streaming providers accumulate into. */
49
64
  export declare function createInitialResponsesAssistantMessage(api: Api, provider: string, modelId: string): AssistantMessage;
@@ -338,6 +338,14 @@ export interface ToolCall {
338
338
  * JSON function tools.
339
339
  */
340
340
  customWireName?: string;
341
+ /**
342
+ * Set when the provider detected the argument JSON was truncated — the model
343
+ * hit its output-token limit (or the response was otherwise cut short) before
344
+ * emitting a complete arguments object. The `arguments` field then holds a
345
+ * best-effort partial parse and must not be executed as-is; the agent loop
346
+ * rejects the call with a retryable error instead.
347
+ */
348
+ incompleteArguments?: boolean;
341
349
  }
342
350
  export interface Usage {
343
351
  /** Non-cached input tokens (matches the bucket the provider bills as new input). */
@@ -8,3 +8,11 @@ export declare function parseJsonWithRepair<T>(json: string): T;
8
8
  * @returns Parsed object or empty object if parsing fails
9
9
  */
10
10
  export declare function parseStreamingJson<T = Record<string, unknown>>(partialJson: string | undefined): T;
11
+ /**
12
+ * Whether a string is a complete, well-formed JSON document (strict parse, no
13
+ * repair). Used to distinguish a tool-call argument blob that finished cleanly
14
+ * from one that was cut off mid-stream (truncation). An empty / whitespace-only
15
+ * string is treated as complete: a tool invoked with no arguments legitimately
16
+ * streams an empty buffer and must not be flagged as truncated.
17
+ */
18
+ export declare function isCompleteJson(text: string | undefined): boolean;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@sayknow-cli/ai",
4
- "version": "0.3.0",
4
+ "version": "0.3.1",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://github.com/jaybeyond/Sayknow_CLI",
7
7
  "author": "jaybeyond",
@@ -43,7 +43,7 @@
43
43
  "dependencies": {
44
44
  "@anthropic-ai/sdk": "^0.94.0",
45
45
  "@bufbuild/protobuf": "^2.12.0",
46
- "@sayknow-cli/utils": "0.3.0",
46
+ "@sayknow-cli/utils": "0.3.1",
47
47
  "openai": "^6.36.0",
48
48
  "partial-json": "^0.1.7",
49
49
  "zod": "4.4.3"
@@ -73,6 +73,8 @@ export type MockContent =
73
73
  name: string;
74
74
  /** Object form is preferred; strings are passed through verbatim. */
75
75
  arguments: Record<string, unknown> | string;
76
+ /** Simulate a provider-flagged truncated call (cut off mid-arguments). */
77
+ incompleteArguments?: boolean;
76
78
  };
77
79
 
78
80
  /** One scripted response. */
@@ -416,6 +418,7 @@ function normalizeContent(input: MockContent, state: MockModel): TextContent | T
416
418
  id: input.id ?? generateToolCallId(state),
417
419
  name: input.name,
418
420
  arguments: typeof input.arguments === "string" ? input.arguments : { ...input.arguments },
421
+ ...(input.incompleteArguments ? { incompleteArguments: true } : {}),
419
422
  } as ToolCall;
420
423
  }
421
424
  return input;
@@ -78,6 +78,7 @@ import {
78
78
  convertResponsesInputContent,
79
79
  encodeResponsesToolCallId,
80
80
  encodeTextSignatureV1,
81
+ flagTruncatedToolCalls,
81
82
  mapOpenAIResponsesStopReason,
82
83
  populateResponsesUsageFromResponse,
83
84
  } from "./openai-responses-shared";
@@ -254,6 +255,8 @@ interface CodexStreamRuntime {
254
255
  providerRetryAttempt: number;
255
256
  sawTerminalEvent: boolean;
256
257
  canSafelyReplayWebsocketOverSse: boolean;
258
+ /** Ids of tool calls that received their terminal `output_item.done`. */
259
+ finalizedToolCallIds: Set<string>;
257
260
  }
258
261
 
259
262
  interface CodexStreamProcessingContext {
@@ -910,6 +913,7 @@ function createCodexStreamRuntime(initial: {
910
913
  providerRetryAttempt: 0,
911
914
  sawTerminalEvent: false,
912
915
  canSafelyReplayWebsocketOverSse: true,
916
+ finalizedToolCallIds: new Set<string>(),
913
917
  };
914
918
  }
915
919
 
@@ -1267,9 +1271,11 @@ function handleOutputItemDone(
1267
1271
  }
1268
1272
 
1269
1273
  if (item.type === "function_call") {
1274
+ const id = encodeResponsesToolCallId(item.call_id, item.id);
1275
+ runtime.finalizedToolCallIds.add(id);
1270
1276
  const toolCall: ToolCall = {
1271
1277
  type: "toolCall",
1272
- id: encodeResponsesToolCallId(item.call_id, item.id),
1278
+ id,
1273
1279
  name: item.name,
1274
1280
  arguments: parseStreamingJson(item.arguments || "{}"),
1275
1281
  };
@@ -1279,13 +1285,15 @@ function handleOutputItemDone(
1279
1285
  }
1280
1286
 
1281
1287
  if (item.type === "custom_tool_call") {
1288
+ const id = encodeResponsesToolCallId(item.call_id, item.id);
1289
+ runtime.finalizedToolCallIds.add(id);
1282
1290
  const rawInput =
1283
1291
  runtime.currentBlock?.type === "toolCall" && runtime.currentBlock.partialJson
1284
1292
  ? runtime.currentBlock.partialJson
1285
1293
  : (item.input ?? "");
1286
1294
  const toolCall: ToolCall = {
1287
1295
  type: "toolCall",
1288
- id: encodeResponsesToolCallId(item.call_id, item.id),
1296
+ id,
1289
1297
  name: item.name,
1290
1298
  arguments: { input: rawInput },
1291
1299
  customWireName: item.name,
@@ -1349,6 +1357,10 @@ function handleResponseCompleted(
1349
1357
  calculateCost(model, output.usage);
1350
1358
  applyCodexServiceTierPricing(model, output.usage, response?.service_tier, runtime.requestBodyForState.service_tier);
1351
1359
  output.stopReason = mapOpenAIResponsesStopReason(response?.status as OpenAI.Responses.ResponseStatus | undefined);
1360
+ // A response cut short for length may have stopped mid-tool-call. Flag any
1361
+ // call that never received its `output_item.done` so the agent loop rejects
1362
+ // the truncated arguments instead of executing a best-effort partial parse.
1363
+ flagTruncatedToolCalls(output, output.stopReason, block => runtime.finalizedToolCallIds.has(block.id));
1352
1364
  if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
1353
1365
  output.stopReason = "toolUse";
1354
1366
  }
@@ -51,7 +51,7 @@ import {
51
51
  getStreamFirstEventTimeoutMs,
52
52
  iterateWithIdleTimeout,
53
53
  } from "../utils/idle-iterator";
54
- import { parseStreamingJson } from "../utils/json-parse";
54
+ import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
55
55
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
56
56
  import { getKimiCommonHeaders } from "../utils/oauth/kimi";
57
57
  import { notifyProviderResponse } from "../utils/provider-response";
@@ -889,6 +889,17 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
889
889
  }
890
890
  }
891
891
 
892
+ // A turn cut short for length may have stopped mid-tool-call. The open
893
+ // block's `partialArgs` would otherwise be repaired into a plausible-
894
+ // but-wrong object by `finishCurrentBlock`; flag it first so the agent
895
+ // loop rejects the truncated call instead of executing it.
896
+ if (output.stopReason === "length" && currentBlock?.type === "toolCall") {
897
+ const partial = (currentBlock as { partialArgs?: string }).partialArgs;
898
+ if (partial !== undefined && !isCompleteJson(partial)) {
899
+ currentBlock.incompleteArguments = true;
900
+ }
901
+ }
902
+
892
903
  finishCurrentBlock(currentBlock);
893
904
 
894
905
  const firstEventTimeoutError = abortTracker.getLocalAbortReason();
@@ -30,7 +30,7 @@ import {
30
30
  } from "../types";
31
31
  import { normalizeResponsesToolCallId } from "../utils";
32
32
  import type { AssistantMessageEventStream } from "../utils/event-stream";
33
- import { parseStreamingJson } from "../utils/json-parse";
33
+ import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
34
34
  import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
35
35
 
36
36
  export function encodeTextSignatureV1(id: string, phase?: TextSignatureV1["phase"]): string {
@@ -710,6 +710,13 @@ export async function processResponsesStream<TApi extends Api>(
710
710
  : "Unknown error (no error details in response)";
711
711
  throw new Error(message);
712
712
  }
713
+ // A response cut short for length (`incomplete`) may have stopped
714
+ // mid-tool-call. Any tool-call item still tracked in `items` never
715
+ // received its terminal `output_item.done`, so it was cut off; flag it
716
+ // (along with any finalized-but-unparseable JSON call) so the agent loop
717
+ // rejects it instead of executing repaired/partial arguments.
718
+ const openBlocks = new Set<unknown>(Array.from(items.values(), entry => entry.block));
719
+ flagTruncatedToolCalls(output, output.stopReason, block => !openBlocks.has(block));
713
720
  if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
714
721
  output.stopReason = "toolUse";
715
722
  }
@@ -728,6 +735,41 @@ export async function processResponsesStream<TApi extends Api>(
728
735
  }
729
736
  }
730
737
 
738
+ /**
739
+ * Mark tool-call blocks left incomplete by a length-truncated response so the
740
+ * agent loop rejects them instead of executing a best-effort partial parse.
741
+ *
742
+ * The universal signal is finalization: a call that never received its terminal
743
+ * `output_item.done` (passed in via `isFinalized`) was cut off mid-arguments.
744
+ * This covers both JSON function calls and raw-input custom tools without
745
+ * mis-flagging a *completed* custom tool whose raw input is not valid JSON. As a
746
+ * defensive secondary, a finalized JSON function call whose buffered arguments
747
+ * still don't parse (e.g. a misbehaving relay) is flagged too. No-op unless the
748
+ * turn stopped for length.
749
+ *
750
+ * Shared by both Responses providers (`openai-responses`, `openai-codex-responses`).
751
+ */
752
+ export function flagTruncatedToolCalls(
753
+ output: AssistantMessage,
754
+ stopReason: StopReason,
755
+ isFinalized: (block: ToolCall) => boolean,
756
+ ): void {
757
+ if (stopReason !== "length") return;
758
+ for (const block of output.content) {
759
+ if (block.type !== "toolCall") continue;
760
+ if (!isFinalized(block)) {
761
+ block.incompleteArguments = true;
762
+ continue;
763
+ }
764
+ // Finalized: custom tools carry raw (non-JSON) input and are complete once
765
+ // finalized; only JSON function calls get the parse double-check.
766
+ if (!block.customWireName) {
767
+ const partial = (block as { partialJson?: string }).partialJson;
768
+ if (partial !== undefined && !isCompleteJson(partial)) block.incompleteArguments = true;
769
+ }
770
+ }
771
+ }
772
+
731
773
  export function mapOpenAIResponsesStopReason(status: OpenAI.Responses.ResponseStatus | undefined): StopReason {
732
774
  if (!status) return "stop";
733
775
  switch (status) {
package/src/types.ts CHANGED
@@ -490,6 +490,14 @@ export interface ToolCall {
490
490
  * JSON function tools.
491
491
  */
492
492
  customWireName?: string;
493
+ /**
494
+ * Set when the provider detected the argument JSON was truncated — the model
495
+ * hit its output-token limit (or the response was otherwise cut short) before
496
+ * emitting a complete arguments object. The `arguments` field then holds a
497
+ * best-effort partial parse and must not be executed as-is; the agent loop
498
+ * rejects the call with a retryable error instead.
499
+ */
500
+ incompleteArguments?: boolean;
493
501
  }
494
502
 
495
503
  export interface Usage {
@@ -146,3 +146,21 @@ export function parseStreamingJson<T = Record<string, unknown>>(partialJson: str
146
146
  }
147
147
  }
148
148
  }
149
+
150
+ /**
151
+ * Whether a string is a complete, well-formed JSON document (strict parse, no
152
+ * repair). Used to distinguish a tool-call argument blob that finished cleanly
153
+ * from one that was cut off mid-stream (truncation). An empty / whitespace-only
154
+ * string is treated as complete: a tool invoked with no arguments legitimately
155
+ * streams an empty buffer and must not be flagged as truncated.
156
+ */
157
+ export function isCompleteJson(text: string | undefined): boolean {
158
+ const trimmed = text?.trim();
159
+ if (!trimmed) return true;
160
+ try {
161
+ JSON.parse(trimmed);
162
+ return true;
163
+ } catch {
164
+ return false;
165
+ }
166
+ }