@gajae-code/ai 0.11.0 → 0.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,6 +14,7 @@ import type {
14
14
  ResolvedServiceTier,
15
15
  StopReason,
16
16
  TextContent,
17
+ ThinkingContent,
17
18
  Tool,
18
19
  ToolCall,
19
20
  ToolResultMessage,
@@ -416,6 +417,29 @@ function buildUsage(message: AssistantMessage): Record<string, unknown> {
416
417
  return usage;
417
418
  }
418
419
 
420
+ function isResponsesFamilyApi(api: AssistantMessage["api"]): boolean {
421
+ return api === "openai-responses" || api === "openai-codex-responses";
422
+ }
423
+
424
+ function safeThinkingText(content: ThinkingContent, api: AssistantMessage["api"]): string | undefined {
425
+ if (isResponsesFamilyApi(api) && content.provenance === undefined) return undefined;
426
+ if (content.provenance === "raw") return undefined;
427
+ if (content.provenance === "mixed") return content.summaryText;
428
+ if (content.provenance === "summary") return content.summaryText ?? content.thinking;
429
+ return content.thinking;
430
+ }
431
+
432
+ function hasRawOrMixedThinking(partial: AssistantMessage, contentIndex: number): boolean {
433
+ const content = partial.content[contentIndex];
434
+ return content?.type === "thinking" && (content.provenance === "raw" || content.provenance === "mixed");
435
+ }
436
+
437
+ /** Responses-family reasoning is untrusted until output_item.done assigns provenance. */
438
+ function hasUnfinalizedResponsesThinking(partial: AssistantMessage, contentIndex: number): boolean {
439
+ const content = partial.content[contentIndex];
440
+ return content?.type !== "thinking" || (isResponsesFamilyApi(partial.api) && content.provenance === undefined);
441
+ }
442
+
419
443
  function flattenAssistant(message: AssistantMessage): {
420
444
  text: string;
421
445
  reasoning: string;
@@ -429,9 +453,11 @@ function flattenAssistant(message: AssistantMessage): {
429
453
  case "text":
430
454
  text += part.text;
431
455
  break;
432
- case "thinking":
433
- reasoning += part.thinking;
456
+ case "thinking": {
457
+ const thinking = safeThinkingText(part, message.api);
458
+ if (thinking !== undefined) reasoning += thinking;
434
459
  break;
460
+ }
435
461
  case "redactedThinking":
436
462
  // Opaque blob — surface verbatim on the reasoning channel so the
437
463
  // concatenation round-trips through clients that just echo it.
@@ -521,6 +547,18 @@ export function encodeStream(
521
547
  let nextToolIndex = 0;
522
548
  let hasToolCalls = false;
523
549
  let finishReason: string = "stop";
550
+ // contentIndexes that already streamed a reasoning summary delta, so a
551
+ // final-only reasoning_summary_end does not duplicate streamed summary text.
552
+ const summaryDeltaSeen = new Set<number>();
553
+ // Responses assigns reasoning provenance only at output_item.done. Keep its
554
+ // pre-classification bytes out of this public compatibility stream.
555
+ const pendingThinkingDeltas = new Map<number, string[]>();
556
+
557
+ const writeThinkingDelta = (thinking: string) => {
558
+ // DeepSeek-style / o-series reasoning channel. Clients that don't
559
+ // understand it ignore the unknown delta key.
560
+ if (thinking.length > 0) writeSse(controller, baseChunk({ reasoning_content: thinking }, null));
561
+ };
524
562
 
525
563
  try {
526
564
  // Initial role chunk.
@@ -534,14 +572,60 @@ export function encodeStream(
534
572
  }
535
573
  break;
536
574
 
537
- case "thinking_delta":
538
- // DeepSeek-style / o-series reasoning channel. Clients that don't
539
- // understand it ignore the unknown delta key.
575
+ case "thinking_delta": {
576
+ if (hasRawOrMixedThinking(event.partial, event.contentIndex)) break;
577
+ if (hasUnfinalizedResponsesThinking(event.partial, event.contentIndex)) {
578
+ const deltas = pendingThinkingDeltas.get(event.contentIndex) ?? [];
579
+ deltas.push(event.delta);
580
+ pendingThinkingDeltas.set(event.contentIndex, deltas);
581
+ break;
582
+ }
583
+ writeThinkingDelta(event.delta);
584
+ break;
585
+ }
586
+ case "thinking_start":
587
+ if (
588
+ !hasRawOrMixedThinking(event.partial, event.contentIndex) &&
589
+ hasUnfinalizedResponsesThinking(event.partial, event.contentIndex)
590
+ ) {
591
+ pendingThinkingDeltas.set(event.contentIndex, []);
592
+ }
593
+ break;
594
+ case "thinking_end": {
595
+ const pending = pendingThinkingDeltas.get(event.contentIndex);
596
+ pendingThinkingDeltas.delete(event.contentIndex);
597
+ if (
598
+ hasRawOrMixedThinking(event.partial, event.contentIndex) ||
599
+ hasUnfinalizedResponsesThinking(event.partial, event.contentIndex)
600
+ )
601
+ break;
602
+ if (pending) for (const delta of pending) writeThinkingDelta(delta);
603
+ break;
604
+ }
605
+ case "reasoning_summary_start":
606
+ // Chat format has no explicit reasoning open frame.
607
+ break;
608
+
609
+ case "reasoning_summary_delta":
610
+ // Provider-displayable summary reasoning surfaces on the same
611
+ // reasoning_content channel as raw thinking for this legacy format.
540
612
  if (event.delta.length > 0) {
613
+ // Only a non-whitespace delta counts as a delivered summary; a bare
614
+ // separator ("\n\n") must not suppress a later final-only end content.
615
+ if (event.delta.trim().length > 0) summaryDeltaSeen.add(event.contentIndex);
541
616
  writeSse(controller, baseChunk({ reasoning_content: event.delta }, null));
542
617
  }
543
618
  break;
544
619
 
620
+ case "reasoning_summary_end":
621
+ // Final-only summary: text arrives only on the end event with no prior
622
+ // deltas, so surface it now (skip when deltas already streamed to avoid
623
+ // duplicating the summary).
624
+ if (event.content.length > 0 && !summaryDeltaSeen.has(event.contentIndex)) {
625
+ writeSse(controller, baseChunk({ reasoning_content: event.content }, null));
626
+ }
627
+ break;
628
+
545
629
  case "toolcall_start": {
546
630
  hasToolCalls = true;
547
631
  const idx = nextToolIndex++;
@@ -578,6 +662,7 @@ export function encodeStream(
578
662
  }
579
663
 
580
664
  case "done":
665
+ pendingThinkingDeltas.clear();
581
666
  finishReason =
582
667
  event.reason === "toolUse"
583
668
  ? "tool_calls"
@@ -593,6 +678,7 @@ export function encodeStream(
593
678
  return;
594
679
 
595
680
  case "error": {
681
+ pendingThinkingDeltas.clear();
596
682
  const msg = event.error.errorMessage ?? "stream error";
597
683
  writeSse(controller, { error: { message: msg, type: "upstream_error" } });
598
684
  controller.close();
@@ -607,10 +693,12 @@ export function encodeStream(
607
693
  }
608
694
 
609
695
  // Stream ended without a terminal `done` (defensive). Close gracefully.
696
+ pendingThinkingDeltas.clear();
610
697
  writeSse(controller, baseChunk({}, hasToolCalls ? "tool_calls" : "stop"));
611
698
  controller.enqueue(encoder.encode("data: [DONE]\n\n"));
612
699
  controller.close();
613
700
  } catch (err) {
701
+ pendingThinkingDeltas.clear();
614
702
  const msg = err instanceof Error ? err.message : String(err);
615
703
  writeSse(controller, { error: { message: msg, type: "upstream_error" } });
616
704
  controller.close();
@@ -113,9 +113,15 @@ const CODEX_NON_RETRYABLE_EVENT_CODES = new Set([
113
113
  "invalid_request_error",
114
114
  "invalid_schema",
115
115
  "invalid_tool_schema",
116
+ // A poisoned-history rejection (`Request blocked (code=invalid_prompt)`) is a
117
+ // deterministic content fault, not a transient upstream failure: retrying the
118
+ // same request re-sends the same offending item and re-triggers the block, so
119
+ // classify it as explicitly non-retryable instead of relying on omission
120
+ // (the request-boundary sanitizer, not a provider retry, is the recovery path).
121
+ "invalid_prompt",
116
122
  ]);
117
123
  const CODEX_NON_RETRYABLE_EVENT_MESSAGE =
118
- /invalid[_ -]function[_ -]parameters|invalid schema for function|invalid[_ -]tool[_ -]schema|schema must have type ["']?object["']?/i;
124
+ /invalid[_ -]function[_ -]parameters|invalid schema for function|invalid[_ -]tool[_ -]schema|schema must have type ["']?object["']?|request blocked[^\n]*invalid[_ -]prompt|code=invalid[_ -]prompt/i;
119
125
  const CODEX_RETRYABLE_EVENT_MESSAGE =
120
126
  /processing your request|retry your request|temporar(?:y|ily)|overloaded|service.?unavailable|internal error|server error/i;
121
127
  const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses";
@@ -133,6 +139,7 @@ const CODEX_PROGRESS_EVENT_TYPES = new Set([
133
139
  "response.reasoning_summary_part.added",
134
140
  "response.reasoning_summary_text.delta",
135
141
  "response.reasoning_summary_part.done",
142
+ "response.reasoning_text.delta",
136
143
  "response.content_part.added",
137
144
  "response.output_text.delta",
138
145
  "response.refusal.delta",
@@ -155,8 +162,8 @@ function isCodexStreamProgressEvent(event: unknown): boolean {
155
162
  }
156
163
  type CodexTransport = "sse" | "websocket";
157
164
  type CodexEventItem = ResponseReasoningItem | ResponseOutputMessage | ResponseFunctionToolCall | ResponseCustomToolCall;
158
- type CodexOutputBlock = ThinkingContent | TextContent | (ToolCall & { partialJson: string });
159
-
165
+ type CodexThinkingBlock = ThinkingContent & { summaryBuffer: string; rawBuffer: string; summaryStarted: boolean };
166
+ type CodexOutputBlock = CodexThinkingBlock | TextContent | (ToolCall & { partialJson: string });
160
167
  export interface OpenAICodexWebSocketDebugStats {
161
168
  fullContextRequests: number;
162
169
  deltaRequests: number;
@@ -995,6 +1002,11 @@ function handleCodexStreamEvent(args: {
995
1002
  return firstTokenTime;
996
1003
  }
997
1004
 
1005
+ if (eventType === "response.reasoning_text.delta") {
1006
+ handleReasoningTextDelta(runtime.currentItem, runtime.currentBlock, rawEvent, stream, output, blockIndex);
1007
+ return firstTokenTime;
1008
+ }
1009
+
998
1010
  if (eventType === "response.content_part.added") {
999
1011
  handleContentPartAdded(runtime.currentItem, rawEvent);
1000
1012
  return firstTokenTime;
@@ -1069,7 +1081,7 @@ function handleCodexStreamEvent(args: {
1069
1081
 
1070
1082
  function createOutputBlockForItem(item: CodexEventItem): CodexOutputBlock | null {
1071
1083
  if (item.type === "reasoning") {
1072
- return { type: "thinking", thinking: "" };
1084
+ return { type: "thinking", thinking: "", summaryBuffer: "", rawBuffer: "", summaryStarted: false };
1073
1085
  }
1074
1086
  if (item.type === "message") {
1075
1087
  return { type: "text", text: "" };
@@ -1120,13 +1132,18 @@ function handleReasoningSummaryTextDelta(
1120
1132
  blockIndex: () => number,
1121
1133
  ): void {
1122
1134
  if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
1135
+ if (!currentBlock.summaryStarted) {
1136
+ currentBlock.summaryStarted = true;
1137
+ stream.push({ type: "reasoning_summary_start", contentIndex: blockIndex(), partial: output });
1138
+ }
1123
1139
  currentItem.summary = currentItem.summary || [];
1124
1140
  const lastPart = currentItem.summary[currentItem.summary.length - 1];
1125
1141
  if (!lastPart) return;
1126
1142
  const delta = (rawEvent as { delta?: string }).delta || "";
1127
1143
  currentBlock.thinking += delta;
1144
+ currentBlock.summaryBuffer += delta;
1128
1145
  lastPart.text += delta;
1129
- stream.push({ type: "thinking_delta", contentIndex: blockIndex(), delta, partial: output });
1146
+ stream.push({ type: "reasoning_summary_delta", contentIndex: blockIndex(), delta, partial: output });
1130
1147
  }
1131
1148
 
1132
1149
  function handleReasoningSummaryPartDone(
@@ -1141,8 +1158,24 @@ function handleReasoningSummaryPartDone(
1141
1158
  const lastPart = currentItem.summary[currentItem.summary.length - 1];
1142
1159
  if (!lastPart) return;
1143
1160
  currentBlock.thinking += "\n\n";
1161
+ currentBlock.summaryBuffer += "\n\n";
1144
1162
  lastPart.text += "\n\n";
1145
- stream.push({ type: "thinking_delta", contentIndex: blockIndex(), delta: "\n\n", partial: output });
1163
+ stream.push({ type: "reasoning_summary_delta", contentIndex: blockIndex(), delta: "\n\n", partial: output });
1164
+ }
1165
+
1166
+ function handleReasoningTextDelta(
1167
+ currentItem: CodexEventItem | null,
1168
+ currentBlock: CodexOutputBlock | null,
1169
+ rawEvent: Record<string, unknown>,
1170
+ stream: AssistantMessageEventStream,
1171
+ output: AssistantMessage,
1172
+ blockIndex: () => number,
1173
+ ): void {
1174
+ if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
1175
+ const delta = (rawEvent as { delta?: string }).delta || "";
1176
+ currentBlock.thinking += delta;
1177
+ currentBlock.rawBuffer += delta;
1178
+ stream.push({ type: "thinking_delta", contentIndex: blockIndex(), delta, partial: output });
1146
1179
  }
1147
1180
 
1148
1181
  function handleContentPartAdded(currentItem: CodexEventItem | null, rawEvent: Record<string, unknown>): void {
@@ -1245,14 +1278,50 @@ function handleOutputItemDone(
1245
1278
  runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
1246
1279
 
1247
1280
  if (item.type === "reasoning" && runtime.currentBlock?.type === "thinking") {
1248
- runtime.currentBlock.thinking = item.summary?.map(summary => summary.text).join("\n\n") || "";
1249
- runtime.currentBlock.thinkingSignature = JSON.stringify(item);
1250
- stream.push({
1251
- type: "thinking_end",
1252
- contentIndex: blockIndex(),
1253
- content: runtime.currentBlock.thinking,
1254
- partial: output,
1255
- });
1281
+ const block = runtime.currentBlock;
1282
+ // Prefer the streamed summary buffer only when it carries real text; a
1283
+ // part.done before/without any summary_text delta leaves only separators, so
1284
+ // fall back to the canonical item.summary from output_item.done (matches the
1285
+ // shared Responses decoder).
1286
+ const bufferSummary = block.summaryBuffer ?? "";
1287
+ const itemSummary = item.summary?.map(summary => summary.text).join("\n\n") ?? "";
1288
+ const summaryText = bufferSummary.trim() ? bufferSummary : itemSummary;
1289
+ const rawText = block.rawBuffer;
1290
+ const mutable = block as { provenance?: "summary" | "raw" | "mixed"; summaryText?: string; rawText?: string };
1291
+ if (mutable.provenance === undefined) {
1292
+ if (mutable.summaryText === undefined && summaryText) mutable.summaryText = summaryText;
1293
+ if (mutable.rawText === undefined && rawText) mutable.rawText = rawText;
1294
+ mutable.provenance = summaryText && rawText ? "mixed" : summaryText ? "summary" : rawText ? "raw" : undefined;
1295
+ }
1296
+ // Finalized display string must exclude raw CoT when a summary exists (parity
1297
+ // with openai-responses-shared). Derive from STORED write-once provenance fields
1298
+ // so a later/duplicate raw-only finalization cannot overwrite a summary/mixed
1299
+ // block's safe display with raw CoT; raw-only stays raw.
1300
+ {
1301
+ const effSummary = mutable.summaryText ?? summaryText;
1302
+ const effRaw = mutable.rawText ?? rawText;
1303
+ block.thinking = mutable.provenance === "raw" ? effRaw : effSummary || effRaw;
1304
+ }
1305
+ block.thinkingSignature = JSON.stringify(item);
1306
+ delete (block as { summaryBuffer?: string }).summaryBuffer;
1307
+ delete (block as { rawBuffer?: string }).rawBuffer;
1308
+ const wasSummaryStarted = block.summaryStarted;
1309
+ delete (block as { summaryStarted?: boolean }).summaryStarted;
1310
+ if (summaryText) {
1311
+ // Emit a summary start first when none was streamed (part.added/done or
1312
+ // canonical done-item summary with no summary_text delta), so consumers that
1313
+ // open a summary on start don't receive an orphaned reasoning_summary_end.
1314
+ if (!wasSummaryStarted) {
1315
+ stream.push({ type: "reasoning_summary_start", contentIndex: blockIndex(), partial: output });
1316
+ }
1317
+ stream.push({
1318
+ type: "reasoning_summary_end",
1319
+ contentIndex: blockIndex(),
1320
+ content: summaryText,
1321
+ partial: output,
1322
+ });
1323
+ }
1324
+ stream.push({ type: "thinking_end", contentIndex: blockIndex(), content: block.thinking, partial: output });
1256
1325
  runtime.currentBlock = null;
1257
1326
  return;
1258
1327
  }
@@ -2650,8 +2719,11 @@ function normalizeInputMessageContent(
2650
2719
  return convertResponsesInputContent(content, model.input.includes("image")) ?? [];
2651
2720
  }
2652
2721
 
2653
- /** @internal Exported for tests. */
2654
- export { convertMessages as convertCodexResponsesMessages };
2722
+ /** @internal Exported for tests. `classifyCodexFailureEventRetryable` is the retry classification of a Codex failure event. */
2723
+ export {
2724
+ convertMessages as convertCodexResponsesMessages,
2725
+ isRetryableCodexFailureEvent as classifyCodexFailureEventRetryable,
2726
+ };
2655
2727
 
2656
2728
  /**
2657
2729
  * Whether this OpenAI code backend-backend model should get the custom-tool grammar
@@ -95,12 +95,20 @@ let warnedReasoningSummaryLevel = false;
95
95
 
96
96
  // ─── inbound parser helpers ─────────────────────────────────────────────────
97
97
 
98
- function extractReasoningTextFromItem(item: OpenAIResponsesReasoningItem): string {
99
- // Prefer `summary[]` — mirrors real OpenAI and the openai-responses provider
100
- // which writes the surfaced reasoning summary into `summary[].text`.
101
- const fromSummary = (item.summary ?? []).map(c => c.text).join("");
102
- if (fromSummary) return fromSummary;
103
- return (item.content ?? []).map(c => c.text).join("");
98
+ function reasoningContentFromItem(
99
+ item: OpenAIResponsesReasoningItem,
100
+ ): Pick<ThinkingContent, "thinking" | "provenance" | "summaryText" | "rawText"> {
101
+ // `summary[]` is provider-displayable; `content[]` is raw reasoning. Keep
102
+ // these channels distinct so a Responses gateway round-trip cannot relabel
103
+ // raw CoT as a summary merely because summary text is absent.
104
+ const summaryText = (item.summary ?? []).map(part => part.text).join("");
105
+ const rawText = (item.content ?? []).map(part => part.text).join("");
106
+ if (summaryText && rawText) {
107
+ return { thinking: summaryText, provenance: "mixed", summaryText, rawText };
108
+ }
109
+ if (summaryText) return { thinking: summaryText, provenance: "summary", summaryText };
110
+ if (rawText) return { thinking: rawText, provenance: "raw", rawText };
111
+ return { thinking: "" };
104
112
  }
105
113
 
106
114
  type InputBlockUnion =
@@ -335,10 +343,10 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
335
343
  }
336
344
  if (effectiveType === "reasoning") {
337
345
  const reasoning = item as OpenAIResponsesReasoningItem;
338
- const text = extractReasoningTextFromItem(reasoning);
346
+ const content = reasoningContentFromItem(reasoning);
339
347
  const thinking: ThinkingContent = {
340
348
  type: "thinking",
341
- thinking: text,
349
+ ...content,
342
350
  thinkingSignature: JSON.stringify(reasoning),
343
351
  ...(reasoning.id ? { itemId: reasoning.id } : {}),
344
352
  };
@@ -538,6 +546,41 @@ function responseStatusForStopReason(message: AssistantMessage): ResponseStatus
538
546
  return "completed";
539
547
  }
540
548
 
549
+ /**
550
+ * Privacy boundary for the public Responses envelope: a reasoning item's
551
+ * `summary_text` must carry ONLY provider-displayable summary text, never raw
552
+ * chain-of-thought. A summary is published ONLY for blocks explicitly marked
553
+ * `provenance: "summary" | "mixed"` (the #2304 provenance path), sourced from
554
+ * `summaryText`. Raw-provenance AND unmarked blocks are omitted: unmarked
555
+ * `thinking` can be raw CoT from providers that stream unmarked reasoning (e.g.
556
+ * openai-completions / ollama) which the auth gateway re-encodes into the
557
+ * Responses wire format, so falling open to `part.thinking` would leak raw CoT.
558
+ */
559
+ function envelopeSummaryText(part: ThinkingContent): string | undefined {
560
+ if (part.provenance === "summary" || part.provenance === "mixed") return part.summaryText;
561
+ return undefined;
562
+ }
563
+
564
+ function envelopeSummaryParts(part: ThinkingContent): Array<{ type: "summary_text"; text: string }> {
565
+ const text = envelopeSummaryText(part);
566
+ return text ? [{ type: "summary_text", text }] : [];
567
+ }
568
+
569
+ function normalizeSummaryParts(value: unknown): Array<{ type: "summary_text"; text: string }> {
570
+ // A serialized signature's `summary` is, by the Responses protocol, provider-
571
+ // displayable summary text (raw reasoning lives in content[]/encrypted_content,
572
+ // which is stripped). Coerce to the canonical shape, keeping only well-formed
573
+ // summary_text entries. This is NOT the unsafe `part.thinking` fallback.
574
+ if (!Array.isArray(value)) return [];
575
+ const out: Array<{ type: "summary_text"; text: string }> = [];
576
+ for (const entry of value) {
577
+ if (isObj(entry) && entry.type === "summary_text" && typeof entry.text === "string") {
578
+ out.push({ type: "summary_text", text: entry.text });
579
+ }
580
+ }
581
+ return out;
582
+ }
583
+
541
584
  function buildReasoningItem(part: ThinkingContent): ReasoningOutputItem {
542
585
  const baseId = part.itemId ?? makeReasoningId();
543
586
  if (part.thinkingSignature) {
@@ -548,10 +591,15 @@ function buildReasoningItem(part: ThinkingContent): ReasoningOutputItem {
548
591
  // Preserve any extra fields (encrypted_content, …) the original carried,
549
592
  // but normalize the summary into the canonical `{type, text}[]` shape.
550
593
  const merged: Record<string, unknown> = { ...sigParsed, type: "reasoning", id };
551
- merged.summary = [{ type: "summary_text", text: part.thinking }];
552
- // `content[]` is the encrypted/raw side-channel; leave whatever was
553
- // already there. If absent, omit — real OpenAI only emits `content[]`
554
- // when `include=['reasoning.encrypted_content']` is set.
594
+ merged.summary =
595
+ part.provenance === "summary" || part.provenance === "mixed"
596
+ ? envelopeSummaryParts(part)
597
+ : normalizeSummaryParts(sigParsed.summary);
598
+ // Strip any `content[]` (raw `reasoning_text`) the serialized signature
599
+ // carried: raw chain-of-thought must never surface in the public final
600
+ // envelope (#2304 CoT boundary). Opaque top-level `encrypted_content`
601
+ // (when present) is a separate field and is preserved by the spread above.
602
+ delete merged.content;
555
603
  return merged as ReasoningOutputItem;
556
604
  }
557
605
  } catch {
@@ -561,7 +609,7 @@ function buildReasoningItem(part: ThinkingContent): ReasoningOutputItem {
561
609
  return {
562
610
  type: "reasoning",
563
611
  id: baseId,
564
- summary: [{ type: "summary_text", text: part.thinking }],
612
+ summary: envelopeSummaryParts(part),
565
613
  };
566
614
  }
567
615
 
@@ -701,7 +749,8 @@ interface OpenReasoning {
701
749
  kind: "reasoning";
702
750
  itemId: string;
703
751
  outputIndex: number;
704
- reasoningText: string;
752
+ summaryText: string;
753
+ summaryPartText: string;
705
754
  }
706
755
  interface OpenFunctionCall {
707
756
  kind: "function_call";
@@ -781,16 +830,13 @@ export function encodeStream(
781
830
  summary: [] as Array<{ type: "summary_text"; text: string }>,
782
831
  };
783
832
  emit("response.output_item.added", { output_index: outputIndex, item });
784
- // Open the summary part. Real OpenAI streams summary text in the
785
- // canonical `reasoning_summary_*` lifecycle; pi-ai's own decoder
786
- // reads `summary[].text` from the eventual `output_item.done`.
787
- emit("response.reasoning_summary_part.added", {
788
- item_id: itemId,
789
- output_index: outputIndex,
790
- summary_index: 0,
791
- part: { type: "summary_text", text: "" },
792
- });
793
- const next: OpenReasoning = { kind: "reasoning", itemId, outputIndex, reasoningText: "" };
833
+ const next: OpenReasoning = {
834
+ kind: "reasoning",
835
+ itemId,
836
+ outputIndex,
837
+ summaryText: "",
838
+ summaryPartText: "",
839
+ };
794
840
  state.open = next;
795
841
  return next;
796
842
  };
@@ -856,18 +902,21 @@ export function encodeStream(
856
902
  content: state.open.content,
857
903
  });
858
904
  } else if (state.open.kind === "reasoning") {
859
- const summary = [{ type: "summary_text" as const, text: state.open.reasoningText ?? "" }];
860
- const item = {
905
+ const summary = state.open.summaryText
906
+ ? [{ type: "summary_text" as const, text: state.open.summaryText }]
907
+ : [];
908
+ // Final reasoning envelope carries the displayable summary ONLY. Raw
909
+ // chain-of-thought is streamed live via response.reasoning_text.delta
910
+ // (the internal raw channel) and is deliberately NOT persisted into the
911
+ // terminal item's content[] — the public final envelope must never carry
912
+ // raw CoT (#2304 CoT boundary).
913
+ const item: ReasoningOutputItem = {
861
914
  type: "reasoning",
862
915
  id: state.open.itemId,
863
916
  summary,
864
917
  };
865
918
  emit("response.output_item.done", { output_index: state.open.outputIndex, item });
866
- finishedItems.push({
867
- type: "reasoning",
868
- id: state.open.itemId,
869
- summary,
870
- });
919
+ finishedItems.push(item);
871
920
  } else {
872
921
  const text = state.open.argsText ?? "";
873
922
  if (state.open.customWireName) {
@@ -1006,9 +1055,34 @@ export function encodeStream(
1006
1055
  break;
1007
1056
  }
1008
1057
  case "thinking_delta": {
1058
+ // Raw reasoning is private. The public Responses gateway emits only
1059
+ // provider-displayable reasoning_summary_* events.
1060
+ break;
1061
+ }
1062
+ case "thinking_end": {
1063
+ if (state.open?.kind !== "reasoning") break;
1064
+ // Raw reasoning is intentionally omitted from every public gateway
1065
+ // frame. Only reasoning_summary_* events populate the terminal item.
1066
+ closeOpen();
1067
+ break;
1068
+ }
1069
+ case "reasoning_summary_start": {
1009
1070
  if (state.open?.kind !== "reasoning") break;
1010
1071
  const cur: OpenReasoning = state.open;
1011
- cur.reasoningText += ev.delta;
1072
+ cur.summaryPartText = "";
1073
+ emit("response.reasoning_summary_part.added", {
1074
+ item_id: cur.itemId,
1075
+ output_index: cur.outputIndex,
1076
+ summary_index: 0,
1077
+ part: { type: "summary_text", text: "" },
1078
+ });
1079
+ break;
1080
+ }
1081
+ case "reasoning_summary_delta": {
1082
+ if (state.open?.kind !== "reasoning") break;
1083
+ const cur: OpenReasoning = state.open;
1084
+ cur.summaryPartText += ev.delta;
1085
+ cur.summaryText += ev.delta;
1012
1086
  emit("response.reasoning_summary_text.delta", {
1013
1087
  item_id: cur.itemId,
1014
1088
  output_index: cur.outputIndex,
@@ -1017,11 +1091,13 @@ export function encodeStream(
1017
1091
  });
1018
1092
  break;
1019
1093
  }
1020
- case "thinking_end": {
1094
+ case "reasoning_summary_end": {
1021
1095
  if (state.open?.kind !== "reasoning") break;
1022
1096
  const cur: OpenReasoning = state.open;
1023
- const text = ev.content ?? cur.reasoningText;
1024
- cur.reasoningText = text;
1097
+ const text = ev.content ?? cur.summaryPartText;
1098
+ // A separator-only accumulated summary (e.g. a part.done "\n\n" before any
1099
+ // real text) is treated as empty so the real end content wins.
1100
+ if (!cur.summaryText.trim()) cur.summaryText = text;
1025
1101
  emit("response.reasoning_summary_text.done", {
1026
1102
  item_id: cur.itemId,
1027
1103
  output_index: cur.outputIndex,
@@ -1034,7 +1110,6 @@ export function encodeStream(
1034
1110
  summary_index: 0,
1035
1111
  part: { type: "summary_text", text },
1036
1112
  });
1037
- closeOpen();
1038
1113
  break;
1039
1114
  }
1040
1115
  case "toolcall_start": {