@sayknow-cli/ai 0.2.3 → 0.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,7 +27,7 @@ import {
27
27
  iterateWithIdleTimeout,
28
28
  } from "../utils/idle-iterator";
29
29
  import { resolveRetryBudget } from "../utils/retry-budget";
30
- import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
30
+ import { flattenToolRootCombinators, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
31
31
  import { wrapFetchForSseDebug } from "../utils/sse-debug";
32
32
  import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
33
33
  import {
@@ -373,7 +373,7 @@ function convertTools(tools: Tool[]): OpenAITool[] {
373
373
  type: "function",
374
374
  name: tool.name,
375
375
  description: tool.description || "",
376
- parameters: sanitizeSchemaForOpenAIResponses(toolWireSchema(tool)),
376
+ parameters: sanitizeSchemaForOpenAIResponses(flattenToolRootCombinators(toolWireSchema(tool))),
377
377
  strict: false,
378
378
  }));
379
379
  }
@@ -29,7 +29,7 @@ import { normalizeSystemPrompts } from "../utils";
29
29
  import { AssistantMessageEventStream } from "../utils/event-stream";
30
30
  import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
- import { toolWireSchema } from "../utils/schema/wire";
32
+ import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
33
33
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
34
34
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
35
35
  import {
@@ -2183,7 +2183,7 @@ function buildMcpToolDefinitions(tools: Tool[] | undefined): McpToolDefinition[]
2183
2183
  }
2184
2184
 
2185
2185
  return advertisedTools.map(tool => {
2186
- const jsonSchema = toolWireSchema(tool);
2186
+ const jsonSchema = flattenToolRootCombinators(toolWireSchema(tool));
2187
2187
  const schemaValue: JsonValue =
2188
2188
  jsonSchema && typeof jsonSchema === "object"
2189
2189
  ? (jsonSchema as JsonValue)
@@ -19,7 +19,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
19
19
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
20
20
  import { parseStreamingJson } from "../utils/json-parse";
21
21
  import { resolveRetryBudget } from "../utils/retry-budget";
22
- import { toolWireSchema } from "../utils/schema/wire";
22
+ import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
23
23
  import {
24
24
  isForcedToolChoiceUnsupportedError,
25
25
  markToolChoiceIncapability,
@@ -253,7 +253,7 @@ function convertTools(tools: Tool[] | undefined): OllamaFunctionTool[] | undefin
253
253
  function: {
254
254
  name: tool.name,
255
255
  description: tool.description,
256
- parameters: toolWireSchema(tool),
256
+ parameters: flattenToolRootCombinators(toolWireSchema(tool)),
257
257
  },
258
258
  }));
259
259
  }
@@ -50,7 +50,13 @@ import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-ins
50
50
  import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
51
51
  import { parseStreamingJson } from "../utils/json-parse";
52
52
  import { resolveRetryBudget } from "../utils/retry-budget";
53
- import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
53
+ import {
54
+ adaptSchemaForStrict,
55
+ flattenToolRootCombinators,
56
+ NO_STRICT,
57
+ sanitizeSchemaForOpenAIResponses,
58
+ toolWireSchema,
59
+ } from "../utils/schema";
54
60
  import {
55
61
  isForcedToolChoiceUnsupportedError,
56
62
  markToolChoiceIncapability,
@@ -98,6 +104,14 @@ const CODEX_WEBSOCKET_RETRY_BUDGET = CODEX_MAX_RETRIES;
98
104
  const CODEX_WEBSOCKET_TRANSPORT_ERROR_PREFIX = "Codex websocket transport error";
99
105
  const CODEX_PREVIOUS_RESPONSE_STALE_CODES = new Set(["previous_response_not_found", "codex_previous_response_stale"]);
100
106
  const CODEX_RETRYABLE_EVENT_CODES = new Set(["model_error", "server_error", "internal_error"]);
107
+ const CODEX_NON_RETRYABLE_EVENT_CODES = new Set([
108
+ "invalid_function_parameters",
109
+ "invalid_request_error",
110
+ "invalid_schema",
111
+ "invalid_tool_schema",
112
+ ]);
113
+ const CODEX_NON_RETRYABLE_EVENT_MESSAGE =
114
+ /invalid[_ -]function[_ -]parameters|invalid schema for function|invalid[_ -]tool[_ -]schema|schema must have type ["']?object["']?/i;
101
115
  const CODEX_RETRYABLE_EVENT_MESSAGE =
102
116
  /processing your request|retry your request|temporar(?:y|ily)|overloaded|service.?unavailable|internal error|server error/i;
103
117
  const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses";
@@ -2658,7 +2672,7 @@ export function convertOpenAICodexResponsesTools(
2658
2672
  };
2659
2673
  }
2660
2674
  const strict = !!(!NO_STRICT && tool.strict);
2661
- const baseParameters = sanitizeSchemaForOpenAIResponses(toolWireSchema(tool));
2675
+ const baseParameters = sanitizeSchemaForOpenAIResponses(flattenToolRootCombinators(toolWireSchema(tool)));
2662
2676
  const { schema: parameters, strict: effectiveStrict } = adaptSchemaForStrict(baseParameters, strict);
2663
2677
  return {
2664
2678
  type: "function",
@@ -2703,11 +2717,17 @@ class CodexProviderStreamError extends Error {
2703
2717
  }
2704
2718
 
2705
2719
  function isRetryableCodexFailureEvent(rawEvent: Record<string, unknown>): boolean {
2706
- const code = getCodexEventErrorCode(rawEvent);
2707
- if (code && CODEX_RETRYABLE_EVENT_CODES.has(code.toLowerCase())) {
2720
+ const code = getCodexEventErrorCode(rawEvent).toLowerCase();
2721
+ const message = getCodexEventErrorMessage(rawEvent);
2722
+ if (
2723
+ (code && CODEX_NON_RETRYABLE_EVENT_CODES.has(code)) ||
2724
+ (!!message && CODEX_NON_RETRYABLE_EVENT_MESSAGE.test(message))
2725
+ ) {
2726
+ return false;
2727
+ }
2728
+ if (code && CODEX_RETRYABLE_EVENT_CODES.has(code)) {
2708
2729
  return true;
2709
2730
  }
2710
- const message = getCodexEventErrorMessage(rawEvent);
2711
2731
  return !!message && CODEX_RETRYABLE_EVENT_MESSAGE.test(message);
2712
2732
  }
2713
2733
 
@@ -189,6 +189,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
189
189
  return {
190
190
  supportsStore: !isNonStandard,
191
191
  supportsDeveloperRole: !isNonStandard,
192
+ sendSessionHeaders: false,
192
193
  supportsMultipleSystemMessages: supportsMultipleSystemMessagesDefault,
193
194
  supportsReasoningEffort: !isGrok && !isZai,
194
195
  reasoningEffortMap,
@@ -254,6 +255,7 @@ export function resolveOpenAICompat(
254
255
  return {
255
256
  supportsStore: model.compat.supportsStore ?? detected.supportsStore,
256
257
  supportsDeveloperRole: model.compat.supportsDeveloperRole ?? detected.supportsDeveloperRole,
258
+ sendSessionHeaders: model.compat.sendSessionHeaders ?? detected.sendSessionHeaders,
257
259
  supportsMultipleSystemMessages:
258
260
  model.compat.supportsMultipleSystemMessages ?? detected.supportsMultipleSystemMessages,
259
261
  supportsReasoningEffort: model.compat.supportsReasoningEffort ?? detected.supportsReasoningEffort,
@@ -57,7 +57,7 @@ import { getKimiCommonHeaders } from "../utils/oauth/kimi";
57
57
  import { notifyProviderResponse } from "../utils/provider-response";
58
58
  import { callWithCopilotModelRetry } from "../utils/retry";
59
59
  import { resolveRetryBudget } from "../utils/retry-budget";
60
- import { adaptSchemaForStrict, NO_STRICT, toolWireSchema } from "../utils/schema";
60
+ import { adaptSchemaForStrict, flattenToolRootCombinators, NO_STRICT, toolWireSchema } from "../utils/schema";
61
61
  import { wrapFetchForSseDebug } from "../utils/sse-debug";
62
62
  import { type HealedToolCall, modelMayLeakKimiToolCalls, ToolCallHealer } from "../utils/tool-call-healing";
63
63
  import { isForcedToolChoice, mapToOpenAICompletionsToolChoice } from "../utils/tool-choice";
@@ -453,6 +453,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
453
453
  options?.streamFirstEventTimeoutMs,
454
454
  options?.authCredentialType,
455
455
  options?.requestMaxRetries,
456
+ options?.sessionId,
456
457
  );
457
458
  const premiumRequestsTotal = copilotPremiumRequests;
458
459
  getCapturedErrorResponse = captureErrorResponse;
@@ -943,6 +944,7 @@ async function createClient(
943
944
  streamFirstEventTimeoutOverride?: number,
944
945
  authCredentialType?: OpenAICompletionsOptions["authCredentialType"],
945
946
  requestMaxRetries?: number,
947
+ sessionId?: string,
946
948
  ): Promise<{
947
949
  client: OpenAI;
948
950
  copilotPremiumRequests: number | undefined;
@@ -981,6 +983,14 @@ async function createClient(
981
983
  headers["X-OpenRouter-Cache-TTL"] = "3600";
982
984
  }
983
985
  Object.assign(headers, extraHeaders);
986
+ if (sessionId && resolveOpenAICompat(model).sendSessionHeaders) {
987
+ // Forward the agent session id as vendor-neutral session-identity headers so
988
+ // OpenAI-compatible proxies/relays can do session-affinity routing and reuse a
989
+ // server-side prompt cache. Opt-in via `compat.sendSessionHeaders`; never
990
+ // overwrite a header the caller already set (model.headers / requestTransform).
991
+ headers.session_id ??= sessionId;
992
+ headers["x-session-id"] ??= sessionId;
993
+ }
984
994
  if (model.provider === "kimi-code") {
985
995
  headers = { ...getKimiCommonHeaders(), ...headers };
986
996
  }
@@ -1832,7 +1842,7 @@ function convertTools(
1832
1842
  ): BuiltOpenAICompletionTools {
1833
1843
  const adaptedTools = tools.map(tool => {
1834
1844
  const strict = !NO_STRICT && compat.supportsStrictMode !== false && tool.strict !== false;
1835
- const baseParameters = toolWireSchema(tool);
1845
+ const baseParameters = flattenToolRootCombinators(toolWireSchema(tool));
1836
1846
  const adapted = adaptSchemaForStrict(baseParameters, strict);
1837
1847
  return {
1838
1848
  tool,
@@ -370,15 +370,66 @@ export async function processResponsesStream<TApi extends Api>(
370
370
  model: Model<TApi>,
371
371
  options?: ProcessResponsesStreamOptions,
372
372
  ): Promise<void> {
373
- let currentItem:
374
- | ResponseReasoningItem
375
- | ResponseOutputMessage
376
- | ResponseFunctionToolCall
377
- | ResponseCustomToolCall
378
- | null = null;
379
- let currentBlock: ThinkingContent | TextContent | (ToolCall & { partialJson: string }) | null = null;
380
- const blocks = output.content;
381
- const blockIndex = () => blocks.length - 1;
373
+ type StreamItem = ResponseReasoningItem | ResponseOutputMessage | ResponseFunctionToolCall | ResponseCustomToolCall;
374
+ type StreamBlock = ThinkingContent | TextContent | (ToolCall & { partialJson: string });
375
+ interface ItemEntry {
376
+ item: StreamItem;
377
+ block: StreamBlock;
378
+ blockContentIndex: number;
379
+ }
380
+ // Per-item argument buffer keyed on stable item identity. Multiple tool-call
381
+ // items can stream interleaved argument deltas in one response, so a single
382
+ // most-recent slot would mis-attribute deltas to the wrong item.
383
+ const items = new Map<string, ItemEntry>();
384
+ let lastKey: string | null = null;
385
+ const idKey = (id: string) => `id:${id}`;
386
+ const idxKey = (n: number) => `idx:${n}`;
387
+ const hasIndex = (n: number | undefined): n is number => typeof n === "number" && Number.isFinite(n);
388
+ const resolveEntry = (
389
+ itemId: string | undefined,
390
+ outputIndex: number | undefined,
391
+ // Fallback to the most-recently-added entry (`lastKey`) when the event
392
+ // cannot be resolved by identity:
393
+ // - "never": tool ghost events with an explicit but unmatched key are ignored.
394
+ // - "no-key": only when BOTH item_id and a finite output_index are absent —
395
+ // the legacy single continuation-style tool delta/done shape.
396
+ // - "always": continuation-style non-tool events (reasoning/text), which may
397
+ // legitimately omit identity and target the open block.
398
+ fallback: "never" | "no-key" | "always",
399
+ ): ItemEntry | undefined => {
400
+ if (itemId) {
401
+ const byId = items.get(idKey(itemId));
402
+ if (byId) return byId;
403
+ }
404
+ if (hasIndex(outputIndex)) {
405
+ const byIdx = items.get(idxKey(outputIndex));
406
+ if (byIdx) return byIdx;
407
+ }
408
+ const hasExplicitKey = !!itemId || hasIndex(outputIndex);
409
+ const allowLastKey = fallback === "always" || (fallback === "no-key" && !hasExplicitKey);
410
+ if (allowLastKey && lastKey) return items.get(lastKey);
411
+ return undefined;
412
+ };
413
+ const registerEntry = (item: StreamItem, block: StreamBlock, outputIndex: number | undefined): ItemEntry => {
414
+ output.content.push(block);
415
+ const entry: ItemEntry = { item, block, blockContentIndex: output.content.length - 1 };
416
+ // Primary key prefers the stable item id; if the wire omits it, fall back to
417
+ // the positional index. A synthetic key keeps the entry addressable as lastKey
418
+ // for continuation-style non-tool events even when neither is present.
419
+ const key = item.id ? idKey(item.id) : hasIndex(outputIndex) ? idxKey(outputIndex) : `seq:${items.size}`;
420
+ items.set(key, entry);
421
+ if (item.id && hasIndex(outputIndex)) items.set(idxKey(outputIndex), entry);
422
+ lastKey = key;
423
+ return entry;
424
+ };
425
+ const dropEntry = (itemId: string | undefined, outputIndex: number | undefined): void => {
426
+ const key = itemId ? idKey(itemId) : hasIndex(outputIndex) ? idxKey(outputIndex) : null;
427
+ if (key) {
428
+ items.delete(key);
429
+ if (lastKey === key) lastKey = null;
430
+ }
431
+ if (itemId && hasIndex(outputIndex)) items.delete(idxKey(outputIndex));
432
+ };
382
433
  let sawFirstToken = false;
383
434
 
384
435
  for await (const event of openaiStream) {
@@ -390,30 +441,27 @@ export async function processResponsesStream<TApi extends Api>(
390
441
  options?.onFirstToken?.();
391
442
  }
392
443
  const item = event.item;
444
+ const outputIndex = event.output_index;
393
445
  if (item.type === "reasoning") {
394
- currentItem = item;
395
- currentBlock = { type: "thinking", thinking: "", itemId: item.id };
396
- output.content.push(currentBlock);
397
- stream.push({ type: "thinking_start", contentIndex: blockIndex(), partial: output });
446
+ const block: ThinkingContent = { type: "thinking", thinking: "", itemId: item.id };
447
+ const entry = registerEntry(item, block, outputIndex);
448
+ stream.push({ type: "thinking_start", contentIndex: entry.blockContentIndex, partial: output });
398
449
  } else if (item.type === "message") {
399
- currentItem = item;
400
- currentBlock = { type: "text", text: "" };
401
- output.content.push(currentBlock);
402
- stream.push({ type: "text_start", contentIndex: blockIndex(), partial: output });
450
+ const block: TextContent = { type: "text", text: "" };
451
+ const entry = registerEntry(item, block, outputIndex);
452
+ stream.push({ type: "text_start", contentIndex: entry.blockContentIndex, partial: output });
403
453
  } else if (item.type === "function_call") {
404
- currentItem = item;
405
- currentBlock = {
454
+ const block: ToolCall & { partialJson: string } = {
406
455
  type: "toolCall",
407
456
  id: encodeResponsesToolCallId(item.call_id, item.id),
408
457
  name: item.name,
409
458
  arguments: {},
410
459
  partialJson: item.arguments || "",
411
460
  };
412
- output.content.push(currentBlock);
413
- stream.push({ type: "toolcall_start", contentIndex: blockIndex(), partial: output });
461
+ const entry = registerEntry(item, block, outputIndex);
462
+ stream.push({ type: "toolcall_start", contentIndex: entry.blockContentIndex, partial: output });
414
463
  } else if (item.type === "custom_tool_call") {
415
- currentItem = item;
416
- currentBlock = {
464
+ const block: ToolCall & { partialJson: string } = {
417
465
  type: "toolCall",
418
466
  id: encodeResponsesToolCallId(item.call_id, item.id),
419
467
  // Preserve the raw wire name (e.g. `apply_patch`). The agent-loop
@@ -427,39 +475,42 @@ export async function processResponsesStream<TApi extends Api>(
427
475
  // accumulation buffer so later code that inspects the field still works.
428
476
  partialJson: item.input ?? "",
429
477
  };
430
- output.content.push(currentBlock);
431
- stream.push({ type: "toolcall_start", contentIndex: blockIndex(), partial: output });
478
+ const entry = registerEntry(item, block, outputIndex);
479
+ stream.push({ type: "toolcall_start", contentIndex: entry.blockContentIndex, partial: output });
432
480
  }
433
481
  } else if (event.type === "response.reasoning_summary_part.added") {
434
- if (currentItem?.type === "reasoning") {
435
- currentItem.summary = currentItem.summary || [];
436
- currentItem.summary.push(event.part);
482
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
483
+ if (entry?.item.type === "reasoning") {
484
+ entry.item.summary = entry.item.summary || [];
485
+ entry.item.summary.push(event.part);
437
486
  }
438
487
  } else if (event.type === "response.reasoning_summary_text.delta") {
439
- if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") {
440
- currentItem.summary = currentItem.summary || [];
441
- const lastPart = currentItem.summary[currentItem.summary.length - 1];
488
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
489
+ if (entry?.item.type === "reasoning" && entry.block.type === "thinking") {
490
+ entry.item.summary = entry.item.summary || [];
491
+ const lastPart = entry.item.summary[entry.item.summary.length - 1];
442
492
  if (lastPart) {
443
- currentBlock.thinking += event.delta;
493
+ entry.block.thinking += event.delta;
444
494
  lastPart.text += event.delta;
445
495
  stream.push({
446
496
  type: "thinking_delta",
447
- contentIndex: blockIndex(),
497
+ contentIndex: entry.blockContentIndex,
448
498
  delta: event.delta,
449
499
  partial: output,
450
500
  });
451
501
  }
452
502
  }
453
503
  } else if (event.type === "response.reasoning_summary_part.done") {
454
- if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") {
455
- currentItem.summary = currentItem.summary || [];
456
- const lastPart = currentItem.summary[currentItem.summary.length - 1];
504
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
505
+ if (entry?.item.type === "reasoning" && entry.block.type === "thinking") {
506
+ entry.item.summary = entry.item.summary || [];
507
+ const lastPart = entry.item.summary[entry.item.summary.length - 1];
457
508
  if (lastPart) {
458
- currentBlock.thinking += "\n\n";
509
+ entry.block.thinking += "\n\n";
459
510
  lastPart.text += "\n\n";
460
511
  stream.push({
461
512
  type: "thinking_delta",
462
- contentIndex: blockIndex(),
513
+ contentIndex: entry.blockContentIndex,
463
514
  delta: "\n\n",
464
515
  partial: output,
465
516
  });
@@ -468,85 +519,94 @@ export async function processResponsesStream<TApi extends Api>(
468
519
  } else if (event.type === "response.reasoning_text.delta") {
469
520
  // Raw reasoning text delta from local providers that stream thinking
470
521
  // directly rather than via the OpenAI summary tracking protocol.
471
- if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") {
472
- currentBlock.thinking += event.delta;
522
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
523
+ if (entry?.item.type === "reasoning" && entry.block.type === "thinking") {
524
+ entry.block.thinking += event.delta;
473
525
  stream.push({
474
526
  type: "thinking_delta",
475
- contentIndex: blockIndex(),
527
+ contentIndex: entry.blockContentIndex,
476
528
  delta: event.delta,
477
529
  partial: output,
478
530
  });
479
531
  }
480
532
  } else if (event.type === "response.content_part.added") {
481
- if (currentItem?.type === "message") {
482
- currentItem.content = currentItem.content || [];
533
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
534
+ if (entry?.item.type === "message") {
535
+ entry.item.content = entry.item.content || [];
483
536
  if (event.part.type === "output_text" || event.part.type === "refusal") {
484
- currentItem.content.push(event.part);
537
+ entry.item.content.push(event.part);
485
538
  }
486
539
  }
487
540
  } else if (event.type === "response.output_text.delta") {
488
- if (currentItem?.type === "message" && currentBlock?.type === "text") {
489
- const lastPart = currentItem.content?.[currentItem.content.length - 1];
541
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
542
+ if (entry?.item.type === "message" && entry.block.type === "text") {
543
+ const lastPart = entry.item.content?.[entry.item.content.length - 1];
490
544
  if (lastPart?.type === "output_text") {
491
- currentBlock.text += event.delta;
545
+ entry.block.text += event.delta;
492
546
  lastPart.text += event.delta;
493
547
  stream.push({
494
548
  type: "text_delta",
495
- contentIndex: blockIndex(),
549
+ contentIndex: entry.blockContentIndex,
496
550
  delta: event.delta,
497
551
  partial: output,
498
552
  });
499
553
  }
500
554
  }
501
555
  } else if (event.type === "response.refusal.delta") {
502
- if (currentItem?.type === "message" && currentBlock?.type === "text") {
503
- const lastPart = currentItem.content?.[currentItem.content.length - 1];
556
+ const entry = resolveEntry(event.item_id, event.output_index, "always");
557
+ if (entry?.item.type === "message" && entry.block.type === "text") {
558
+ const lastPart = entry.item.content?.[entry.item.content.length - 1];
504
559
  if (lastPart?.type === "refusal") {
505
- currentBlock.text += event.delta;
560
+ entry.block.text += event.delta;
506
561
  lastPart.refusal += event.delta;
507
562
  stream.push({
508
563
  type: "text_delta",
509
- contentIndex: blockIndex(),
564
+ contentIndex: entry.blockContentIndex,
510
565
  delta: event.delta,
511
566
  partial: output,
512
567
  });
513
568
  }
514
569
  }
515
570
  } else if (event.type === "response.function_call_arguments.delta") {
516
- if (currentItem?.type === "function_call" && currentBlock?.type === "toolCall") {
517
- currentBlock.partialJson += event.delta;
518
- currentBlock.arguments = parseStreamingJson(currentBlock.partialJson);
571
+ const entry = resolveEntry(event.item_id, event.output_index, "no-key");
572
+ if (entry?.item.type === "function_call" && entry.block.type === "toolCall") {
573
+ entry.block.partialJson += event.delta;
574
+ entry.block.arguments = parseStreamingJson(entry.block.partialJson);
519
575
  stream.push({
520
576
  type: "toolcall_delta",
521
- contentIndex: blockIndex(),
577
+ contentIndex: entry.blockContentIndex,
522
578
  delta: event.delta,
523
579
  partial: output,
524
580
  });
525
581
  }
526
582
  } else if (event.type === "response.function_call_arguments.done") {
527
- if (currentItem?.type === "function_call" && currentBlock?.type === "toolCall") {
528
- currentBlock.partialJson = event.arguments;
529
- currentBlock.arguments = parseStreamingJson(currentBlock.partialJson);
583
+ const entry = resolveEntry(event.item_id, event.output_index, "no-key");
584
+ if (entry?.item.type === "function_call" && entry.block.type === "toolCall") {
585
+ entry.block.partialJson = event.arguments;
586
+ entry.block.arguments = parseStreamingJson(entry.block.partialJson);
530
587
  }
531
588
  } else if (event.type === "response.custom_tool_call_input.delta") {
532
- if (currentItem?.type === "custom_tool_call" && currentBlock?.type === "toolCall") {
533
- currentBlock.partialJson += event.delta;
534
- currentBlock.arguments = { input: currentBlock.partialJson };
589
+ const entry = resolveEntry(event.item_id, event.output_index, "no-key");
590
+ if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") {
591
+ entry.block.partialJson += event.delta;
592
+ entry.block.arguments = { input: entry.block.partialJson };
535
593
  stream.push({
536
594
  type: "toolcall_delta",
537
- contentIndex: blockIndex(),
595
+ contentIndex: entry.blockContentIndex,
538
596
  delta: event.delta,
539
597
  partial: output,
540
598
  });
541
599
  }
542
600
  } else if (event.type === "response.custom_tool_call_input.done") {
543
- if (currentItem?.type === "custom_tool_call" && currentBlock?.type === "toolCall") {
544
- currentBlock.partialJson = event.input;
545
- currentBlock.arguments = { input: event.input };
601
+ const entry = resolveEntry(event.item_id, event.output_index, "no-key");
602
+ if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") {
603
+ entry.block.partialJson = event.input;
604
+ entry.block.arguments = { input: event.input };
546
605
  }
547
606
  } else if (event.type === "response.output_item.done") {
548
607
  const item = structuredCloneJSON(event.item);
549
608
  options?.onOutputItemDone?.(item);
609
+ const entry = resolveEntry(item.id, event.output_index, "never");
550
610
  if (item.type === "reasoning") {
551
611
  const thinking =
552
612
  item.summary?.length > 0
@@ -554,13 +614,17 @@ export async function processResponsesStream<TApi extends Api>(
554
614
  : item.content?.[0]?.type === "reasoning_text"
555
615
  ? (item.content[0].text ?? "")
556
616
  : "";
557
- const reasoningBlock = output.content.find(
558
- b => b.type === "thinking" && (b as ThinkingContent).itemId === item.id,
559
- ) as ThinkingContent | undefined;
617
+ const reasoningBlock =
618
+ entry?.block.type === "thinking"
619
+ ? entry.block
620
+ : (output.content.find(b => b.type === "thinking" && (b as ThinkingContent).itemId === item.id) as
621
+ | ThinkingContent
622
+ | undefined);
560
623
  if (reasoningBlock) {
561
624
  reasoningBlock.thinking = thinking;
562
625
  reasoningBlock.thinkingSignature = JSON.stringify(item);
563
- const reasoningBlockIndex = output.content.indexOf(reasoningBlock);
626
+ const reasoningBlockIndex =
627
+ entry?.block === reasoningBlock ? entry.blockContentIndex : output.content.indexOf(reasoningBlock);
564
628
  stream.push({
565
629
  type: "thinking_end",
566
630
  contentIndex: reasoningBlockIndex,
@@ -568,23 +632,27 @@ export async function processResponsesStream<TApi extends Api>(
568
632
  partial: output,
569
633
  });
570
634
  }
571
- if ((currentBlock as ThinkingContent | null)?.itemId === item.id) currentBlock = null;
572
- } else if (item.type === "message" && currentBlock?.type === "text") {
573
- currentBlock.text = item.content
635
+ dropEntry(item.id, event.output_index);
636
+ } else if (item.type === "message" && entry?.block.type === "text") {
637
+ const block = entry.block;
638
+ block.text = item.content
574
639
  .map(part => (part.type === "output_text" ? (part.text ?? "") : (part.refusal ?? "")))
575
640
  .join("");
576
- currentBlock.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined);
641
+ block.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined);
577
642
  stream.push({
578
643
  type: "text_end",
579
- contentIndex: blockIndex(),
580
- content: currentBlock.text,
644
+ contentIndex: entry.blockContentIndex,
645
+ content: block.text,
581
646
  partial: output,
582
647
  });
583
- currentBlock = null;
648
+ dropEntry(item.id, event.output_index);
584
649
  } else if (item.type === "function_call") {
650
+ // Finalize onto the same block object stored in output.content, reading
651
+ // the matching entry's buffered partialJson first and only then the done
652
+ // item's arguments — never an adjacent item's buffer.
585
653
  const args =
586
- currentBlock?.type === "toolCall" && currentBlock.partialJson
587
- ? parseStreamingJson(currentBlock.partialJson)
654
+ entry?.block.type === "toolCall" && entry.block.partialJson
655
+ ? parseStreamingJson(entry.block.partialJson)
588
656
  : parseStreamingJson(item.arguments || "{}");
589
657
  const toolCall: ToolCall = {
590
658
  type: "toolCall",
@@ -592,12 +660,18 @@ export async function processResponsesStream<TApi extends Api>(
592
660
  name: item.name,
593
661
  arguments: args,
594
662
  };
595
- currentBlock = null;
596
- stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output });
663
+ if (entry?.block.type === "toolCall") {
664
+ entry.block.id = toolCall.id;
665
+ entry.block.name = toolCall.name;
666
+ entry.block.arguments = args;
667
+ }
668
+ const contentIndex = entry?.blockContentIndex ?? output.content.length - 1;
669
+ dropEntry(item.id, event.output_index);
670
+ stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
597
671
  } else if (item.type === "custom_tool_call") {
598
672
  const rawInput =
599
- currentBlock?.type === "toolCall" && currentBlock.partialJson
600
- ? currentBlock.partialJson
673
+ entry?.block.type === "toolCall" && entry.block.partialJson
674
+ ? entry.block.partialJson
601
675
  : (item.input ?? "");
602
676
  const toolCall: ToolCall = {
603
677
  type: "toolCall",
@@ -606,8 +680,14 @@ export async function processResponsesStream<TApi extends Api>(
606
680
  arguments: { input: rawInput },
607
681
  customWireName: item.name,
608
682
  };
609
- currentBlock = null;
610
- stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output });
683
+ if (entry?.block.type === "toolCall") {
684
+ entry.block.id = toolCall.id;
685
+ entry.block.name = toolCall.name;
686
+ entry.block.arguments = { input: rawInput };
687
+ }
688
+ const contentIndex = entry?.blockContentIndex ?? output.content.length - 1;
689
+ dropEntry(item.id, event.output_index);
690
+ stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
611
691
  }
612
692
  } else if (event.type === "response.completed") {
613
693
  const response = event.response;
@@ -50,7 +50,13 @@ import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
50
50
  import { notifyProviderResponse } from "../utils/provider-response";
51
51
  import { callWithCopilotModelRetry } from "../utils/retry";
52
52
  import { resolveRetryBudget } from "../utils/retry-budget";
53
- import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
53
+ import {
54
+ adaptSchemaForStrict,
55
+ flattenToolRootCombinators,
56
+ NO_STRICT,
57
+ sanitizeSchemaForOpenAIResponses,
58
+ toolWireSchema,
59
+ } from "../utils/schema";
54
60
  import { wrapFetchForSseDebug } from "../utils/sse-debug";
55
61
  import { mapToOpenAIResponsesToolChoice, type OpenAIResponsesToolChoice } from "../utils/tool-choice";
56
62
  import {
@@ -724,7 +730,7 @@ export function convertTools(tools: Tool[], strictMode: boolean, model: Model<"o
724
730
  } as unknown as OpenAITool;
725
731
  }
726
732
  const strict = !NO_STRICT && strictMode && tool.strict !== false;
727
- const baseParameters = toolWireSchema(tool);
733
+ const baseParameters = flattenToolRootCombinators(toolWireSchema(tool));
728
734
  const responseParameters = sanitizeSchemaForOpenAIResponses(baseParameters);
729
735
  const { schema: parameters, strict: effectiveStrict } = adaptSchemaForStrict(responseParameters, strict);
730
736
  return {
package/src/types.ts CHANGED
@@ -742,6 +742,17 @@ export interface OpenAICompat extends ToolChoiceCompat {
742
742
  supportsStore?: boolean;
743
743
  /** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */
744
744
  supportsDeveloperRole?: boolean;
745
+ /**
746
+ * Whether to forward the agent session id as vendor-neutral session-identity
747
+ * headers (`session_id`, `x-session-id`) on every chat-completions request.
748
+ * Off by default. Opt in for OpenAI-compatible proxies/relays that route on
749
+ * session affinity or reuse a server-side prompt cache keyed by session.
750
+ * First-party OpenAI does not need this (it has its own gated injection in
751
+ * the openai-responses provider). Headers are only added when a non-empty
752
+ * session id is available and are never allowed to overwrite a header the
753
+ * caller already set via `headers`/`requestTransform`.
754
+ */
755
+ sendSessionHeaders?: boolean;
745
756
  /**
746
757
  * Whether the provider's chat-completions endpoint accepts multiple
747
758
  * leading `system`/`developer` messages. When false, ordered system
@@ -7,6 +7,7 @@ export * from "./fields";
7
7
  export * from "./json-schema-validator";
8
8
  export * from "./meta-validator";
9
9
  export * from "./normalize";
10
+ export * from "./root-combinator";
10
11
  export * from "./spill";
11
12
  export * from "./types";
12
13
  export * from "./wire";