@gajae-code/ai 0.13.2 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/CHANGELOG.md +61 -2
  2. package/dist/types/auth-broker/client.d.ts +9 -1
  3. package/dist/types/auth-broker/redact.d.ts +7 -0
  4. package/dist/types/auth-broker/remote-store.d.ts +50 -9
  5. package/dist/types/auth-broker/types.d.ts +14 -0
  6. package/dist/types/auth-broker/wire-schemas.d.ts +25 -0
  7. package/dist/types/auth-storage.d.ts +200 -6
  8. package/dist/types/core.d.ts +1 -0
  9. package/dist/types/model-cache.d.ts +4 -1
  10. package/dist/types/model-manager.d.ts +11 -0
  11. package/dist/types/provider-models/openai-compat.d.ts +5 -0
  12. package/dist/types/provider-models/special.d.ts +3 -0
  13. package/dist/types/providers/anthropic.d.ts +31 -0
  14. package/dist/types/providers/cursor.d.ts +9 -1
  15. package/dist/types/providers/kiro-codewhisperer.d.ts +8 -0
  16. package/dist/types/providers/mock.d.ts +8 -0
  17. package/dist/types/providers/register-builtins.d.ts +1 -0
  18. package/dist/types/providers/transform-messages.d.ts +18 -0
  19. package/dist/types/types.d.ts +34 -8
  20. package/dist/types/usage/grok-cli.d.ts +5 -0
  21. package/dist/types/usage.d.ts +6 -0
  22. package/dist/types/utils/discovery/openai-compatible.d.ts +5 -0
  23. package/dist/types/utils/event-stream.d.ts +4 -2
  24. package/dist/types/utils/fallback-transport.d.ts +10 -0
  25. package/dist/types/utils/http-inspector.d.ts +1 -0
  26. package/dist/types/utils/idle-iterator.d.ts +13 -1
  27. package/dist/types/utils/json-parse.d.ts +19 -0
  28. package/dist/types/utils/oauth/callback-server.d.ts +13 -0
  29. package/dist/types/utils/oauth/kiro.d.ts +71 -0
  30. package/dist/types/utils/oauth/types.d.ts +1 -1
  31. package/dist/types/utils/parse-bind.d.ts +8 -5
  32. package/dist/types/utils/tool-call-healing.d.ts +7 -0
  33. package/dist/types/utils/tool-choice-capability.d.ts +11 -0
  34. package/package.json +3 -2
  35. package/src/auth-broker/client.ts +30 -0
  36. package/src/auth-broker/redact.ts +15 -0
  37. package/src/auth-broker/refresher.ts +4 -2
  38. package/src/auth-broker/remote-store.ts +693 -70
  39. package/src/auth-broker/server.ts +57 -12
  40. package/src/auth-broker/types.ts +16 -0
  41. package/src/auth-broker/wire-schemas.ts +21 -0
  42. package/src/auth-gateway/server.ts +84 -19
  43. package/src/auth-storage.ts +985 -41
  44. package/src/core.ts +1 -0
  45. package/src/model-cache.ts +23 -4
  46. package/src/model-manager.ts +70 -11
  47. package/src/model-thinking.ts +45 -1
  48. package/src/models.json +9604 -1932
  49. package/src/openai-completions-compat.ts +2 -1
  50. package/src/provider-models/descriptors.ts +7 -1
  51. package/src/provider-models/openai-compat.ts +52 -28
  52. package/src/provider-models/special.ts +12 -0
  53. package/src/providers/amazon-bedrock.ts +2 -1
  54. package/src/providers/anthropic.ts +831 -27
  55. package/src/providers/cursor.ts +83 -3
  56. package/src/providers/kiro-codewhisperer.ts +572 -0
  57. package/src/providers/mock.ts +15 -2
  58. package/src/providers/ollama.ts +9 -2
  59. package/src/providers/openai-codex-responses.ts +16 -9
  60. package/src/providers/openai-completions.ts +6 -1
  61. package/src/providers/openai-responses-shared.ts +180 -18
  62. package/src/providers/register-builtins.ts +24 -2
  63. package/src/providers/transform-messages.ts +64 -1
  64. package/src/stream.ts +25 -2
  65. package/src/types.ts +36 -7
  66. package/src/usage/grok-cli.ts +86 -1
  67. package/src/usage.ts +7 -0
  68. package/src/utils/discovery/openai-compatible.ts +89 -4
  69. package/src/utils/event-stream.ts +11 -2
  70. package/src/utils/fallback-transport.ts +44 -2
  71. package/src/utils/http-inspector.ts +1 -0
  72. package/src/utils/idle-iterator.ts +29 -6
  73. package/src/utils/json-parse.ts +80 -0
  74. package/src/utils/oauth/callback-server.ts +31 -1
  75. package/src/utils/oauth/index.ts +14 -1
  76. package/src/utils/oauth/kiro.ts +448 -0
  77. package/src/utils/oauth/synthetic.ts +2 -3
  78. package/src/utils/oauth/types.ts +1 -0
  79. package/src/utils/parse-bind.ts +27 -0
  80. package/src/utils/tool-call-healing.ts +13 -2
  81. package/src/utils/tool-choice-capability.ts +386 -6
@@ -75,8 +75,12 @@ export type MockContent =
75
75
  arguments: Record<string, unknown> | string;
76
76
  /** Simulate a provider-flagged truncated call (cut off mid-arguments). */
77
77
  incompleteArguments?: boolean;
78
+ /** Typed reason matching `ToolCall.incompleteArgumentsReason`. Defaults to `"truncated"`. */
79
+ incompleteArgumentsReason?: "truncated" | "malformed" | "conflicting" | "ambiguous";
80
+ /** Simulate a provider-flagged `\uXXXX`-escaped non-ASCII argument payload. */
81
+ escapedNonAsciiArguments?: boolean;
82
+ thoughtSignature?: string;
78
83
  };
79
-
80
84
  /** One scripted response. */
81
85
  export interface MockResponse {
82
86
  /** Content blocks to emit, in order. Strings become text blocks. */
@@ -87,6 +91,9 @@ export interface MockResponse {
87
91
  usage?: Partial<Omit<Usage, "cost">> & { cost?: Partial<Usage["cost"]> };
88
92
  /** Pre-set responseId. */
89
93
  responseId?: string;
94
+ /** Optional provider metadata copied onto the final assistant message. */
95
+ disabledFeatures?: string[];
96
+ providerPayload?: AssistantMessage["providerPayload"];
90
97
  /** Optional typed provider failure metadata for retry/fallback tests. */
91
98
  transportFailure?: AssistantMessage["transportFailure"];
92
99
  /** If set, the stream emits a terminal error event instead of completing. */
@@ -365,6 +372,8 @@ async function runMock(
365
372
  provider: model.provider,
366
373
  model: model.id,
367
374
  responseId: response.responseId,
375
+ disabledFeatures: response.disabledFeatures,
376
+ providerPayload: response.providerPayload,
368
377
  transportFailure: response.transportFailure,
369
378
  usage: emptyUsage(),
370
379
  stopReason: "stop",
@@ -422,7 +431,11 @@ function normalizeContent(input: MockContent, state: MockModel): TextContent | T
422
431
  id: input.id ?? generateToolCallId(state),
423
432
  name: input.name,
424
433
  arguments: typeof input.arguments === "string" ? input.arguments : { ...input.arguments },
425
- ...(input.incompleteArguments ? { incompleteArguments: true } : {}),
434
+ ...(input.incompleteArguments
435
+ ? { incompleteArguments: true, incompleteArgumentsReason: input.incompleteArgumentsReason ?? "truncated" }
436
+ : {}),
437
+ ...(input.escapedNonAsciiArguments ? { escapedNonAsciiArguments: true } : {}),
438
+ ...(input.thoughtSignature ? { thoughtSignature: input.thoughtSignature } : {}),
426
439
  } as ToolCall;
427
440
  }
428
441
  return input;
@@ -18,7 +18,7 @@ import { normalizeSystemPrompts } from "../utils";
18
18
  import { AssistantMessageEventStream } from "../utils/event-stream";
19
19
  import { transportFailureFacts } from "../utils/fallback-transport";
20
20
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
21
- import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
21
+ import { findUnnecessaryUnicodeEscape, isCompleteJson, parseStreamingJson } from "../utils/json-parse";
22
22
  import { resolveRetryBudget } from "../utils/retry-budget";
23
23
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
24
24
  import {
@@ -361,6 +361,9 @@ function endToolCallBlock(stream: AssistantMessageEventStream, output: Assistant
361
361
  if (toolCall.partialJson !== undefined) {
362
362
  if (toolCall.partialJson.trim()) {
363
363
  toolCall.arguments = parseStreamingJson<Record<string, unknown>>(toolCall.partialJson);
364
+ if (findUnnecessaryUnicodeEscape(toolCall.partialJson)) {
365
+ toolCall.escapedNonAsciiArguments = true;
366
+ }
364
367
  }
365
368
  delete toolCall.partialJson;
366
369
  }
@@ -550,6 +553,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
550
553
  name,
551
554
  arguments: parseStreamingJson<Record<string, unknown>>(partialJson),
552
555
  partialJson,
556
+ ...(findUnnecessaryUnicodeEscape(partialJson) ? { escapedNonAsciiArguments: true } : {}),
553
557
  };
554
558
  if (unverifiableArguments) unverifiableArgumentToolCallIds.add(toolCall.id);
555
559
  output.content.push(toolCall);
@@ -591,7 +595,10 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
591
595
  for (const block of output.content) {
592
596
  if (block.type !== "toolCall") continue;
593
597
  const partialJson = (block as InternalToolCallBlock).partialJson;
594
- if (partialJson !== undefined && !isCompleteJson(partialJson)) block.incompleteArguments = true;
598
+ if (partialJson !== undefined && !isCompleteJson(partialJson)) {
599
+ block.incompleteArguments = true;
600
+ block.incompleteArgumentsReason = "truncated";
601
+ }
595
602
  }
596
603
  }
597
604
  for (const index of activeToolIndices) {
@@ -57,7 +57,7 @@ import {
57
57
  getStreamFirstEventTimeoutMs,
58
58
  iterateWithIdleTimeout,
59
59
  } from "../utils/idle-iterator";
60
- import { parseStreamingJson } from "../utils/json-parse";
60
+ import { findUnnecessaryUnicodeEscape, parseStreamingJson } from "../utils/json-parse";
61
61
  import { resolveRetryBudget } from "../utils/retry-budget";
62
62
  import {
63
63
  adaptSchemaForStrict,
@@ -1276,6 +1276,7 @@ function handleToolCallArgumentsDone(
1276
1276
  if (typeof args === "string") {
1277
1277
  currentBlock.partialJson = args;
1278
1278
  currentBlock.arguments = parseStreamingJson(currentBlock.partialJson);
1279
+ if (findUnnecessaryUnicodeEscape(args)) currentBlock.escapedNonAsciiArguments = true;
1279
1280
  }
1280
1281
  }
1281
1282
 
@@ -1391,6 +1392,7 @@ function handleOutputItemDone(
1391
1392
  id,
1392
1393
  name: codexToolCanonicalName(item.name),
1393
1394
  arguments: parseStreamingJson(item.arguments || "{}"),
1395
+ ...(findUnnecessaryUnicodeEscape(item.arguments || "") ? { escapedNonAsciiArguments: true } : {}),
1394
1396
  };
1395
1397
  runtime.canSafelyReplayWebsocketOverSse = false;
1396
1398
  stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output });
@@ -1853,24 +1855,29 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
1853
1855
  context: Context,
1854
1856
  options?: OpenAICodexResponsesOptions,
1855
1857
  ): AssistantMessageEventStream => {
1856
- const stream = new AssistantMessageEventStream();
1858
+ const consumerAbortController = new AbortController();
1859
+ const stream = new AssistantMessageEventStream(() => consumerAbortController.abort());
1860
+ const signal = options?.signal
1861
+ ? AbortSignal.any([options.signal, consumerAbortController.signal])
1862
+ : consumerAbortController.signal;
1863
+ const streamOptions = { ...options, signal };
1857
1864
 
1858
1865
  (async () => {
1859
1866
  const startTime = Date.now();
1860
1867
  const output = createAssistantOutput(model);
1861
- const requestSetup = createRequestSetup(options);
1868
+ const requestSetup = createRequestSetup(streamOptions);
1862
1869
  let processingContext: CodexStreamProcessingContext | undefined;
1863
1870
 
1864
1871
  try {
1865
- const requestContext = await buildCodexRequestContext(model, context, options, output);
1872
+ const requestContext = await buildCodexRequestContext(model, context, streamOptions, output);
1866
1873
  let initialTransport: CodexInitialTransport;
1867
1874
  try {
1868
- initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
1875
+ initialTransport = await openInitialCodexEventStream(model, streamOptions, requestSetup, requestContext);
1869
1876
  } catch (error) {
1870
- if (options?.fallbackManaged) throw error;
1877
+ if (streamOptions.fallbackManaged) throw error;
1871
1878
  initialTransport = await retryCodexInitialTransportWithoutToolChoice(
1872
1879
  model,
1873
- options,
1880
+ streamOptions,
1874
1881
  requestSetup,
1875
1882
  requestContext,
1876
1883
  stream,
@@ -1889,7 +1896,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
1889
1896
  model,
1890
1897
  output,
1891
1898
  stream,
1892
- options,
1899
+ options: streamOptions,
1893
1900
  requestSetup,
1894
1901
  requestContext,
1895
1902
  startTime,
@@ -1907,7 +1914,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
1907
1914
  model,
1908
1915
  output,
1909
1916
  stream,
1910
- options,
1917
+ options: streamOptions,
1911
1918
  requestSetup,
1912
1919
  requestContext: {
1913
1920
  apiKey: "",
@@ -54,7 +54,7 @@ import {
54
54
  iterateWithIdleTimeout,
55
55
  resolveOpenAISdkRequestTimeoutMs,
56
56
  } from "../utils/idle-iterator";
57
- import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
57
+ import { findUnnecessaryUnicodeEscape, isCompleteJson, parseStreamingJson } from "../utils/json-parse";
58
58
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
59
59
  import { getKimiCommonHeaders } from "../utils/oauth/kimi";
60
60
  import { notifyProviderResponse } from "../utils/provider-response";
@@ -676,6 +676,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
676
676
  return;
677
677
  }
678
678
  block.arguments = parseStreamingJson(block.partialArgs);
679
+ if (findUnnecessaryUnicodeEscape(block.partialArgs ?? "")) block.escapedNonAsciiArguments = true;
679
680
  delete (block as { partialArgs?: string }).partialArgs;
680
681
  stream.push({ type: "toolcall_end", contentIndex, toolCall: block, partial: output });
681
682
  };
@@ -804,6 +805,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
804
805
  partialArgs: call.arguments,
805
806
  };
806
807
  block.arguments = parseStreamingJson(call.arguments);
808
+ // The healer already normalized `call.arguments`, decoding any escapes away,
809
+ // so the signal has to come from its pre-round-trip sample of the raw payload.
810
+ if (call.escapedNonAsciiArguments) block.escapedNonAsciiArguments = true;
807
811
  currentBlock = block;
808
812
  output.content.push(block);
809
813
  stream.push({ type: "toolcall_start", contentIndex: blockIndex(block), partial: output });
@@ -1020,6 +1024,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
1020
1024
  const partial = (currentBlock as { partialArgs?: string }).partialArgs;
1021
1025
  if (partial !== undefined && !isCompleteJson(partial)) {
1022
1026
  currentBlock.incompleteArguments = true;
1027
+ currentBlock.incompleteArgumentsReason = "truncated";
1023
1028
  }
1024
1029
  }
1025
1030
 
@@ -30,7 +30,8 @@ import {
30
30
  } from "../types";
31
31
  import { normalizeResponsesToolCallId, sanitizeJsonStrings } from "../utils";
32
32
  import type { AssistantMessageEventStream } from "../utils/event-stream";
33
- import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
33
+ import { findUnnecessaryUnicodeEscape, isCompleteJson, parseStreamingJson } from "../utils/json-parse";
34
+ import { areJsonValuesEqual } from "../utils/schema";
34
35
  import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
35
36
 
36
37
  const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
@@ -405,6 +406,27 @@ export async function processResponsesStream<TApi extends Api>(
405
406
  summaryBuffer: string;
406
407
  rawBuffer: string;
407
408
  summaryStarted: boolean;
409
+ /**
410
+ * Raw `arguments` carried by the item's `response.output_item.added` snapshot.
411
+ * Kept out of the streaming buffer (a relay may put a `{}` placeholder here)
412
+ * but retained as the lowest-precedence source for relays that supply the
413
+ * real payload only in that snapshot.
414
+ */
415
+ addedArguments: string;
416
+ /**
417
+ * Set when this entry's tool identity is ambiguous (a duplicate `call_id`,
418
+ * an `id`/`call_id` namespace collision, or any other shape where a delta
419
+ * cannot be unambiguously attributed). The entry is finalized as
420
+ * `incompleteArguments` so the agent loop rejects it instead of executing
421
+ * possibly-misattributed arguments.
422
+ */
423
+ ambiguousIdentity: boolean;
424
+ /**
425
+ * Whether this entry has already been finalized by a terminal
426
+ * `response.output_item.done`. A duplicate terminal event for the same item
427
+ * must not emit a second `toolcall_end`/`text_end`/`thinking_end`.
428
+ */
429
+ finalized: boolean;
408
430
  }
409
431
  // Per-item argument buffer keyed on stable item identity. Multiple tool-call
410
432
  // items can stream interleaved argument deltas in one response, so a single
@@ -412,6 +434,7 @@ export async function processResponsesStream<TApi extends Api>(
412
434
  const items = new Map<string, ItemEntry>();
413
435
  let lastKey: string | null = null;
414
436
  const idKey = (id: string) => `id:${id}`;
437
+ const callKey = (id: string) => `call:${id}`;
415
438
  const idxKey = (n: number) => `idx:${n}`;
416
439
  const hasIndex = (n: number | undefined): n is number => typeof n === "number" && Number.isFinite(n);
417
440
  const resolveEntry = (
@@ -428,7 +451,18 @@ export async function processResponsesStream<TApi extends Api>(
428
451
  ): ItemEntry | undefined => {
429
452
  if (itemId) {
430
453
  const byId = items.get(idKey(itemId));
454
+ const byCallId = items.get(callKey(itemId));
455
+ // Ambiguous identity: `item_id` matches one entry as its canonical id and
456
+ // a *different* entry as its `call_id` (an id/call_id namespace collision).
457
+ // Picking either silently mis-attributes the payload, so mark both
458
+ // ambiguous and drop the delta instead of resolving.
459
+ if (byId && byCallId && byId !== byCallId) {
460
+ byId.ambiguousIdentity = true;
461
+ byCallId.ambiguousIdentity = true;
462
+ return undefined;
463
+ }
431
464
  if (byId) return byId;
465
+ if (byCallId) return byCallId;
432
466
  }
433
467
  if (hasIndex(outputIndex)) {
434
468
  const byIdx = items.get(idxKey(outputIndex));
@@ -448,23 +482,62 @@ export async function processResponsesStream<TApi extends Api>(
448
482
  summaryBuffer: "",
449
483
  rawBuffer: "",
450
484
  summaryStarted: false,
485
+ addedArguments: item.type === "function_call" ? (item.arguments ?? "") : "",
486
+ ambiguousIdentity: false,
487
+ finalized: false,
451
488
  };
452
489
  // Primary key prefers the stable item id; if the wire omits it, fall back to
453
490
  // the positional index. A synthetic key keeps the entry addressable as lastKey
454
491
  // for continuation-style non-tool events even when neither is present.
455
492
  const key = item.id ? idKey(item.id) : hasIndex(outputIndex) ? idxKey(outputIndex) : `seq:${items.size}`;
456
493
  items.set(key, entry);
457
- if (item.id && hasIndex(outputIndex)) items.set(idxKey(outputIndex), entry);
494
+ // Index alias: only claim it when no other entry already holds it. Two items
495
+ // sharing one `output_index` (a relay defect) must not have the second steal
496
+ // the alias and drop the first's index-routed deltas; each stays addressable
497
+ // by its own stable id/call_id, and the index keeps resolving to the first
498
+ // occupant rather than silently reassigning.
499
+ if (hasIndex(outputIndex)) {
500
+ const idxK = idxKey(outputIndex);
501
+ if (!items.has(idxK)) items.set(idxK, entry);
502
+ }
503
+ if ((item.type === "function_call" || item.type === "custom_tool_call") && item.call_id) {
504
+ const callK = callKey(item.call_id);
505
+ const existing = items.get(callK);
506
+ // Duplicate `call_id` in one response: two distinct items claim the same
507
+ // alias. Fail closed for both — neither's arguments can be trusted to
508
+ // belong to the right call once their deltas and terminals are aliased.
509
+ if (existing && existing !== entry) {
510
+ existing.ambiguousIdentity = true;
511
+ entry.ambiguousIdentity = true;
512
+ } else if (!existing) {
513
+ items.set(callK, entry);
514
+ }
515
+ }
516
+ // Detect an id/call_id collision at registration too: a new item whose id
517
+ // equals another item's call_id (or vice versa) makes id-based resolution
518
+ // ambiguous for any delta keyed on that shared string.
519
+ if (item.id) {
520
+ const callAliasOfOther = items.get(callKey(item.id));
521
+ if (callAliasOfOther && callAliasOfOther !== entry) {
522
+ callAliasOfOther.ambiguousIdentity = true;
523
+ entry.ambiguousIdentity = true;
524
+ }
525
+ }
458
526
  lastKey = key;
459
527
  return entry;
460
528
  };
461
- const dropEntry = (itemId: string | undefined, outputIndex: number | undefined): void => {
462
- const key = itemId ? idKey(itemId) : hasIndex(outputIndex) ? idxKey(outputIndex) : null;
463
- if (key) {
529
+ const dropEntry = (itemId: string | undefined, outputIndex: number | undefined, callId?: string): void => {
530
+ const entry =
531
+ (itemId ? (items.get(idKey(itemId)) ?? items.get(callKey(itemId))) : undefined) ??
532
+ (callId ? items.get(callKey(callId)) : undefined) ??
533
+ (hasIndex(outputIndex) ? items.get(idxKey(outputIndex)) : undefined);
534
+ if (!entry) return;
535
+ entry.finalized = true;
536
+ for (const [key, candidate] of items) {
537
+ if (candidate !== entry) continue;
464
538
  items.delete(key);
465
539
  if (lastKey === key) lastKey = null;
466
540
  }
467
- if (itemId && hasIndex(outputIndex)) items.delete(idxKey(outputIndex));
468
541
  };
469
542
  let sawFirstToken = false;
470
543
 
@@ -492,7 +565,7 @@ export async function processResponsesStream<TApi extends Api>(
492
565
  id: encodeResponsesToolCallId(item.call_id, item.id),
493
566
  name: item.name,
494
567
  arguments: {},
495
- partialJson: item.arguments || "",
568
+ partialJson: "",
496
569
  };
497
570
  const entry = registerEntry(item, block, outputIndex);
498
571
  stream.push({ type: "toolcall_start", contentIndex: entry.blockContentIndex, partial: output });
@@ -649,7 +722,21 @@ export async function processResponsesStream<TApi extends Api>(
649
722
  } else if (event.type === "response.output_item.done") {
650
723
  const item = structuredCloneJSON(event.item);
651
724
  options?.onOutputItemDone?.(item);
652
- const entry = resolveEntry(item.id, event.output_index, "never");
725
+ // A tool item may be registered under its call id alone (relays that omit
726
+ // item ids in `added`) and then introduce an item id in the terminal event,
727
+ // so both identities are tried before the positional fallback.
728
+ const isToolItem = item.type === "function_call" || item.type === "custom_tool_call";
729
+ const entry =
730
+ resolveEntry(item.id, event.output_index, "never") ??
731
+ (isToolItem && item.call_id ? resolveEntry(item.call_id, event.output_index, "never") : undefined);
732
+ // A duplicate terminal event for an item already finalized (dropped) must
733
+ // not emit a second end event. After finalization the entry is gone from
734
+ // the map, so a second `output_item.done` for the same tool item resolves
735
+ // to no live entry — skip it rather than re-emitting.
736
+ // An orphan terminal event (no preceding `output_item.added`, so no live
737
+ // entry) for a tool item must not synthesize a phantom block at a stale
738
+ // content index. Only finalize tool items that resolved to a live entry.
739
+ if (isToolItem && !entry) continue;
653
740
  if (item.type === "reasoning") {
654
741
  // Prefer the streamed summary buffer only when it carries real text. When it
655
742
  // holds only synthetic separators (e.g. a part.done arrived before/without any
@@ -732,26 +819,77 @@ export async function processResponsesStream<TApi extends Api>(
732
819
  });
733
820
  dropEntry(item.id, event.output_index);
734
821
  } else if (item.type === "function_call") {
735
- // Finalize onto the same block object stored in output.content, reading
736
- // the matching entry's buffered partialJson first and only then the done
737
- // item's arguments — never an adjacent item's buffer.
738
- const args =
739
- entry?.block.type === "toolCall" && entry.block.partialJson
740
- ? parseStreamingJson(entry.block.partialJson)
741
- : parseStreamingJson(item.arguments || "{}");
822
+ // The terminal item is canonical. Some compatible Responses relays put an
823
+ // empty placeholder in output_item.added and only provide real arguments
824
+ // here. When streamed arguments also exist, require agreement rather than
825
+ // silently choosing one source — but compare the decoded payloads, since a
826
+ // relay that re-serializes the terminal item (different key spacing or
827
+ // escaping) is not a disagreement about what the model asked for.
828
+ const streamedArguments = entry?.block.type === "toolCall" ? entry.block.partialJson : "";
829
+ const finalArguments = item.arguments ?? "";
830
+ const hasStreamedArguments = streamedArguments.length > 0;
831
+ const hasFinalArguments = finalArguments.length > 0;
832
+ const conflictingArgumentSources =
833
+ hasStreamedArguments &&
834
+ hasFinalArguments &&
835
+ streamedArguments !== finalArguments &&
836
+ !isEquivalentJsonPayload(streamedArguments, finalArguments);
837
+ // Source precedence: terminal, then streamed deltas, then the `added`
838
+ // snapshot. The last one only matters for relays that never emit deltas
839
+ // and leave the terminal `arguments` empty; without it their real payload
840
+ // would silently degrade to `{}`.
841
+ const rawArguments = hasFinalArguments
842
+ ? finalArguments
843
+ : hasStreamedArguments
844
+ ? streamedArguments
845
+ : (entry?.addedArguments ?? "");
846
+ const decodedArguments =
847
+ conflictingArgumentSources || !isCompleteJson(rawArguments)
848
+ ? undefined
849
+ : parseStreamingJson(rawArguments);
850
+ // Function-call arguments must decode to a JSON object; `null`, arrays and
851
+ // scalars cannot be dispatched against a tool schema, so they fail closed
852
+ // instead of reaching validation as a non-record value. An ambiguous
853
+ // tool-call identity (duplicate call_id, id/call_id collision) also fails
854
+ // closed: attribution of the streamed/terminal payload is unsafe.
855
+ const ambiguousIdentity = entry?.ambiguousIdentity ?? false;
856
+ const incompleteArguments = ambiguousIdentity || !isJsonRecord(decodedArguments);
857
+ const args = incompleteArguments ? {} : (decodedArguments as Record<string, unknown>);
858
+ // Typed reason lets the agent loop give accurate recovery guidance instead
859
+ // of always suggesting "split the work" (truncation-only) for a malformed
860
+ // or conflicting terminal payload, or an ambiguous identity.
861
+ const incompleteArgumentsReason: "malformed" | "conflicting" | "ambiguous" | undefined = incompleteArguments
862
+ ? ambiguousIdentity
863
+ ? "ambiguous"
864
+ : conflictingArgumentSources
865
+ ? "conflicting"
866
+ : "malformed"
867
+ : undefined;
868
+ const escapedNonAscii = findUnnecessaryUnicodeEscape(rawArguments) !== undefined;
742
869
  const toolCall: ToolCall = {
743
870
  type: "toolCall",
744
871
  id: encodeResponsesToolCallId(item.call_id, item.id),
745
872
  name: item.name,
746
873
  arguments: args,
874
+ ...(incompleteArguments ? { incompleteArguments: true, incompleteArgumentsReason } : {}),
875
+ ...(escapedNonAscii ? { escapedNonAsciiArguments: true } : {}),
747
876
  };
748
877
  if (entry?.block.type === "toolCall") {
749
878
  entry.block.id = toolCall.id;
750
879
  entry.block.name = toolCall.name;
751
880
  entry.block.arguments = args;
881
+ if (escapedNonAscii) entry.block.escapedNonAsciiArguments = true;
882
+ else delete entry.block.escapedNonAsciiArguments;
883
+ if (incompleteArguments) {
884
+ entry.block.incompleteArguments = true;
885
+ entry.block.incompleteArgumentsReason = incompleteArgumentsReason;
886
+ } else {
887
+ delete entry.block.incompleteArguments;
888
+ delete entry.block.incompleteArgumentsReason;
889
+ }
752
890
  }
753
891
  const contentIndex = entry?.blockContentIndex ?? output.content.length - 1;
754
- dropEntry(item.id, event.output_index);
892
+ dropEntry(item.id, event.output_index, item.call_id);
755
893
  stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
756
894
  } else if (item.type === "custom_tool_call") {
757
895
  const rawInput =
@@ -771,7 +909,7 @@ export async function processResponsesStream<TApi extends Api>(
771
909
  entry.block.arguments = { input: rawInput };
772
910
  }
773
911
  const contentIndex = entry?.blockContentIndex ?? output.content.length - 1;
774
- dropEntry(item.id, event.output_index);
912
+ dropEntry(item.id, event.output_index, item.call_id);
775
913
  stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
776
914
  }
777
915
  } else if (event.type === "response.completed") {
@@ -820,6 +958,26 @@ export async function processResponsesStream<TApi extends Api>(
820
958
  }
821
959
  }
822
960
 
961
+ /**
962
+ * Whether two raw JSON argument strings decode to the same value. Used to tell a
963
+ * relay's re-serialization of the same tool arguments apart from a genuine
964
+ * disagreement between the streamed and terminal payloads; anything that does
965
+ * not decode cleanly on both sides is treated as a disagreement (fail closed).
966
+ */
967
+ /** Whether a decoded JSON value is a plain object usable as tool-call arguments. */
968
+ function isJsonRecord(value: unknown): value is Record<string, unknown> {
969
+ return typeof value === "object" && value !== null && !Array.isArray(value);
970
+ }
971
+
972
+ function isEquivalentJsonPayload(left: string, right: string): boolean {
973
+ if (!isCompleteJson(left) || !isCompleteJson(right)) return false;
974
+ try {
975
+ return areJsonValuesEqual(JSON.parse(left), JSON.parse(right));
976
+ } catch {
977
+ return false;
978
+ }
979
+ }
980
+
823
981
  /**
824
982
  * Mark tool-call blocks left incomplete by a length-truncated response so the
825
983
  * agent loop rejects them instead of executing a best-effort partial parse.
@@ -844,13 +1002,17 @@ export function flagTruncatedToolCalls(
844
1002
  if (block.type !== "toolCall") continue;
845
1003
  if (!isFinalized(block)) {
846
1004
  block.incompleteArguments = true;
1005
+ block.incompleteArgumentsReason = "truncated";
847
1006
  continue;
848
1007
  }
849
1008
  // Finalized: custom tools carry raw (non-JSON) input and are complete once
850
1009
  // finalized; only JSON function calls get the parse double-check.
851
1010
  if (!block.customWireName) {
852
1011
  const partial = (block as { partialJson?: string }).partialJson;
853
- if (partial !== undefined && !isCompleteJson(partial)) block.incompleteArguments = true;
1012
+ if (partial !== undefined && !isCompleteJson(partial)) {
1013
+ block.incompleteArguments = true;
1014
+ block.incompleteArgumentsReason = "truncated";
1015
+ }
854
1016
  }
855
1017
  }
856
1018
  }
@@ -37,6 +37,7 @@ import type { CursorOptions } from "./cursor";
37
37
  import type { GoogleOptions } from "./google";
38
38
  import type { GoogleGeminiCliOptions } from "./google-gemini-cli";
39
39
  import type { GoogleVertexOptions } from "./google-vertex";
40
+ import type { KiroCodeWhispererOptions } from "./kiro-codewhisperer";
40
41
  import type { OllamaChatOptions } from "./ollama";
41
42
  import type { OpenAICodexResponsesOptions } from "./openai-codex-responses";
42
43
  import type { OpenAICompletionsOptions } from "./openai-completions";
@@ -153,6 +154,14 @@ interface BedrockProviderModule {
153
154
  ) => AssistantMessageEventStream;
154
155
  }
155
156
 
157
+ interface KiroCodeWhispererProviderModule {
158
+ streamKiroCodeWhisperer: (
159
+ model: Model<"kiro-codewhisperer-stream">,
160
+ context: Context,
161
+ options: KiroCodeWhispererOptions,
162
+ ) => AssistantMessageEventStream;
163
+ }
164
+
156
165
  // ---------------------------------------------------------------------------
157
166
  // Module-level lazy promise caches
158
167
  // ---------------------------------------------------------------------------
@@ -168,6 +177,7 @@ let openAIResponsesProviderModulePromise: Promise<LazyProviderModule<"openai-res
168
177
  let ollamaProviderModulePromise: Promise<LazyProviderModule<"ollama-chat">> | undefined;
169
178
  let cursorProviderModulePromise: Promise<LazyProviderModule<"cursor-agent">> | undefined;
170
179
  let bedrockProviderModuleOverride: LazyProviderModule<"bedrock-converse-stream"> | undefined;
180
+ let kiroCodeWhispererProviderModulePromise: Promise<LazyProviderModule<"kiro-codewhisperer-stream">> | undefined;
171
181
  let bedrockProviderModulePromise: Promise<LazyProviderModule<"bedrock-converse-stream">> | undefined;
172
182
 
173
183
  export function setBedrockProviderModule(module: BedrockProviderModule): void {
@@ -329,12 +339,15 @@ function createLazyStream<TApi extends Api>(
329
339
  limits?: LazyStreamLimits,
330
340
  ): (model: Model<TApi>, context: Context, options: OptionsForApi<TApi>) => EventStreamImpl {
331
341
  return (model, context, options) => {
332
- const outer = new EventStreamImpl();
342
+ let abortTracker: AbortSourceTracker | undefined;
343
+ const outer = new EventStreamImpl(() =>
344
+ abortTracker?.abortLocally(new Error("Provider stream consumer stopped before completion")),
345
+ );
333
346
  const streamOptions = (options ?? {}) as OptionsForApi<TApi>;
334
347
 
335
348
  loadModule()
336
349
  .then(module => {
337
- const abortTracker = createAbortSourceTracker(streamOptions.signal);
350
+ abortTracker = createAbortSourceTracker(streamOptions.signal);
338
351
  const providerOptions = { ...streamOptions, signal: abortTracker.requestSignal } as OptionsForApi<TApi>;
339
352
  const inner = module.stream(model, context, providerOptions);
340
353
  forwardStream(outer, inner, model, streamOptions, abortTracker, limits);
@@ -443,6 +456,13 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
443
456
  });
444
457
  return bedrockProviderModulePromise;
445
458
  }
459
+ function loadKiroCodeWhispererProviderModule(): Promise<LazyProviderModule<"kiro-codewhisperer-stream">> {
460
+ kiroCodeWhispererProviderModulePromise ||= Promise.resolve().then(() => {
461
+ const provider = require("./kiro-codewhisperer") as KiroCodeWhispererProviderModule;
462
+ return { stream: provider.streamKiroCodeWhisperer };
463
+ });
464
+ return kiroCodeWhispererProviderModulePromise;
465
+ }
446
466
 
447
467
  /**
448
468
  * Lazy provider descriptors used by core consumers that need to inspect or
@@ -459,6 +479,7 @@ export const PROVIDER_RUNTIME_DESCRIPTORS: readonly ProviderRuntimeDescriptor<Ap
459
479
  { api: "openai-responses", load: loadOpenAIResponsesProviderModule },
460
480
  { api: "ollama-chat", load: loadOllamaProviderModule },
461
481
  { api: "cursor-agent", load: loadCursorProviderModule },
482
+ { api: "kiro-codewhisperer-stream", load: loadKiroCodeWhispererProviderModule },
462
483
  { api: "bedrock-converse-stream", load: loadBedrockProviderModule },
463
484
  ] as readonly ErasedProviderRuntimeDescriptor[];
464
485
 
@@ -507,3 +528,4 @@ export const streamCursor = createLazyStream(loadCursorProviderModule);
507
528
  export const streamOllama = createLazyStream(loadOllamaProviderModule);
508
529
 
509
530
  export const streamBedrock = createLazyStream(loadBedrockProviderModule);
531
+ export const streamKiroCodeWhisperer = createLazyStream(loadKiroCodeWhispererProviderModule);
@@ -27,6 +27,64 @@ const enum ToolCallStatus {
27
27
  * - Injects synthetic "aborted" tool results
28
28
  * - Adds a <turn-aborted> guidance marker for the model
29
29
  */
30
+ /**
31
+ * Detect directly adjacent private thinking blocks inside one assistant message's
32
+ * content. `thinking` and `redacted_thinking` are one adjacency class: the
33
+ * Anthropic wire contract rejects a replayed assistant turn where two such blocks
34
+ * sit next to each other with no intervening `tool_use`/`text` block (#4416).
35
+ *
36
+ * This is a pure, allocation-free predicate used by defense-in-depth diagnostics
37
+ * (issue #4443): the write-time transcript assertion (coding-agent persistence)
38
+ * and the stream-assembler SSE diagnostic (anthropic stream completion). It never
39
+ * inspects block payloads — only the block-type sequence — so it cannot leak
40
+ * thinking text, signatures, or credentials.
41
+ *
42
+ * Blocks separated by any non-private block (`tool_use`, `text`, …) are ordinary
43
+ * interleaved-thinking shape and return `false`.
44
+ */
45
+ export function hasAdjacentPrivateThinkingBlocks(content: { type: string }[]): boolean {
46
+ let previousWasPrivate = false;
47
+ for (const block of content) {
48
+ const isPrivate = block.type === "thinking" || block.type === "redactedThinking";
49
+ if (isPrivate && previousWasPrivate) return true;
50
+ previousWasPrivate = isPrivate;
51
+ }
52
+ return false;
53
+ }
54
+
55
+ /**
56
+ * Collapse a run of directly adjacent `thinking` blocks inside one assistant message down
57
+ * to its first block.
58
+ *
59
+ * Anthropic accepts a replayed assistant turn carrying a single thinking block, and accepts
60
+ * thinking blocks separated by a `tool_use` (ordinary interleaved-thinking shape), but
61
+ * rejects two directly adjacent `thinking` blocks with
62
+ * `messages.N.content.M: thinking or redacted_thinking blocks in the latest assistant
63
+ * message cannot be modified`, citing the *second* block of the pair. Because the offending
64
+ * message keeps its index as history grows, a single such turn makes every later request in
65
+ * that session fail, and the mutation repair - scoped to the latest assistant message -
66
+ * can never reach it (#4416).
67
+ *
68
+ * `redactedThinking` is not folded in this phase, but the final send-boundary
69
+ * collapse in `convertAnthropicMessages` (#4425) treats `thinking` and
70
+ * `redacted_thinking` as one adjacency class per the API contract.
71
+ */
72
+ function collapseAdjacentThinking<T extends { type: string }>(content: T[]): T[] {
73
+ let previousWasThinking = false;
74
+ let dropped = false;
75
+ const collapsed: T[] = [];
76
+ for (const block of content) {
77
+ const thinking = block.type === "thinking";
78
+ if (thinking && previousWasThinking) {
79
+ dropped = true;
80
+ continue;
81
+ }
82
+ previousWasThinking = thinking;
83
+ collapsed.push(block);
84
+ }
85
+ return dropped ? collapsed : content;
86
+ }
87
+
30
88
  export function transformMessages<TApi extends Api>(
31
89
  messages: Message[],
32
90
  model: Model<TApi>,
@@ -160,9 +218,14 @@ export function transformMessages<TApi extends Api>(
160
218
  return block;
161
219
  });
162
220
 
221
+ // Only the Anthropic wire shape rejects adjacent private blocks; other targets
222
+ // either degrade reasoning to text above or carry their own encoding rules.
223
+ const replayableContent =
224
+ model.api === "anthropic-messages" ? collapseAdjacentThinking(transformedContent) : transformedContent;
225
+
163
226
  return {
164
227
  ...assistantMsg,
165
- content: transformedContent,
228
+ content: replayableContent,
166
229
  };
167
230
  }
168
231
  return msg;