@gajae-code/ai 0.11.1 → 0.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,12 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.11.2] - 2026-07-19
6
+
7
+ ### Fixed
8
+
9
+ - `transportFailureFacts` now reduces transport headers to a plain record containing only the retained retry signals (`retry-after`, `retry-after-ms`). Providers attach these facts to error `AssistantMessage`s, and the previous shape carried the live fetch/SDK `Headers` instance — which is not structured-cloneable (`structuredClone` throws `DataCloneError`, "The object can not be cloned." under Bun) and not JSON-serializable (persisted as `{}` in session files, silently dropping the retry hint). Under a managed model fallback chain, snapshotting such an error message replaced the real provider failure with the local clone error and exhausted the whole chain. Normalization is idempotent (re-running facts on facts is structurally stable; errors carrying only unretained headers with no status/code now yield no facts instead of an empty facts object), Retry-After classification (`classifyFallbackTrigger`) is unchanged, and arbitrary response headers no longer reach persisted facts.
10
+
5
11
  ## [0.11.0] - 2026-07-15
6
12
  ### Added
7
13
 
@@ -410,6 +410,8 @@ export declare class AuthStorage {
410
410
  * Remove a runtime API key override.
411
411
  */
412
412
  removeRuntimeApiKey(provider: string): void;
413
+ /** Whether a provider is currently authenticated by a runtime API-key override. */
414
+ hasRuntimeApiKey(provider: string): boolean;
413
415
  /**
414
416
  * Register a per-provider API key sourced from user configuration
415
417
  * (e.g. `models.yml` `providers.<name>.apiKey`). Higher priority than
@@ -30,6 +30,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
30
30
  modelsDev?: ModelsDevFallback<TApi, TModelsDevPayload>;
31
31
  /** Clock override for deterministic tests. */
32
32
  now?: () => number;
33
+ /** Optional guard that must permit cache publication. Default: writes are permitted. */
34
+ canPublishCache?: () => boolean;
33
35
  }
34
36
  /**
35
37
  * Resolution result.
@@ -1,5 +1,5 @@
1
1
  import Anthropic, { type ClientOptions as AnthropicSdkClientOptions } from "@anthropic-ai/sdk";
2
- import type { MessageParam } from "@anthropic-ai/sdk/resources/messages";
2
+ import type { MessageCreateParamsStreaming, MessageParam } from "@anthropic-ai/sdk/resources/messages";
3
3
  import type { FetchImpl, Message, Model, ProviderSessionState, ServiceTier, SimpleStreamOptions, StreamFunction, StreamOptions, Usage } from "../types";
4
4
  export type AnthropicHeaderOptions = {
5
5
  apiKey: string;
@@ -181,6 +181,7 @@ type SystemBlockOptions = {
181
181
  export declare function buildAnthropicSystemBlocks(systemPrompt: readonly string[] | undefined, options?: SystemBlockOptions): AnthropicSystemBlock[] | undefined;
182
182
  export declare function normalizeExtraBetas(betas?: string[] | string): string[];
183
183
  export declare function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): AnthropicClientOptionsResult;
184
+ export declare function normalizeCacheControlTtlOrdering(params: MessageCreateParamsStreaming): void;
184
185
  export declare function convertAnthropicMessages(messages: Message[], model: Model<"anthropic-messages">, isOAuthToken: boolean, options?: {
185
186
  repairLatestAssistantThinking?: boolean;
186
187
  }): MessageParam[];
@@ -10,13 +10,12 @@
10
10
  * translations impose on first-class pi-ai fields (service tier, cache
11
11
  * markers, thinking budgets, tool-choice variants, …).
12
12
  *
13
- * The streaming wire is {@link AssistantMessageEvent} serialized verbatim and
14
- * SSE-framed. Same type pi-ai already produces internally; the client feeds
15
- * each parsed event straight into `AssistantMessageEventStream.push()` with
16
- * no translation. Including `partial: AssistantMessage` on every delta is
17
- * O(N²) in turn length on the wire — acceptable for the loopback / sidecar
18
- * topology this transport is designed for; provider latency dominates the
19
- * actual cost.
13
+ * The streaming wire is {@link AssistantMessageEvent} serialized as SSE. Public
14
+ * projections omit private raw reasoning and serialized Responses reasoning
15
+ * signatures while preserving provider-displayable summaries and genuine opaque
16
+ * signatures. Including `partial: AssistantMessage` on every delta is O(N²) in
17
+ * turn length on the wire — acceptable for the loopback / sidecar topology this
18
+ * transport is designed for; provider latency dominates the actual cost.
20
19
  *
21
20
  * Endpoint contract:
22
21
  * POST /v1/pi/stream
@@ -45,16 +44,9 @@ export interface PiNativeParsedRequest {
45
44
  */
46
45
  export declare function parseRequest(body: unknown, _headers?: Headers): PiNativeParsedRequest;
47
46
  /**
48
- * Ship every {@link AssistantMessageEvent} verbatim, SSE-framed.
49
- *
50
- * No per-event re-shaping: the pi-native client is pi-ai itself, so the
51
- * canonical event type IS the wire type. Including the rolling
52
- * `partial: AssistantMessage` on every delta is quadratic in turn length
53
- * on the wire, but for the loopback / sidecar topology this transport
54
- * targets (containerized GJC → host gateway) the bandwidth cost is negligible
55
- * compared to provider latency —
56
- * and the client gets to feed the events straight into its existing
57
- * `AssistantMessageEventStream.push()` plumbing with zero translation.
47
+ * Ship only public-safe {@link AssistantMessageEvent} projections. Unknown
48
+ * thinking blocks remain buffered until their terminal partial establishes that
49
+ * the provider-native block is safe; raw and mixed blocks never reach SSE.
58
50
  */
59
51
  export declare function encodeStream(events: AssistantMessageEventStream): ReadableStream<Uint8Array>;
60
52
  /**
@@ -319,6 +319,9 @@ export interface ThinkingContent {
319
319
  thinking: string;
320
320
  thinkingSignature?: string;
321
321
  itemId?: string;
322
+ readonly provenance?: "summary" | "raw" | "mixed";
323
+ readonly summaryText?: string;
324
+ readonly rawText?: string;
322
325
  }
323
326
  export interface RedactedThinkingContent {
324
327
  type: "redactedThinking";
@@ -542,6 +545,16 @@ export interface Tool<TParameters extends TSchema = TSchema> {
542
545
  * calls route correctly. Absent for regular JSON function tools.
543
546
  */
544
547
  customWireName?: string;
548
+ /**
549
+ * Optional safe projection for tool arguments or results. Extensions use this
550
+ * only for explicitly opt-in, display-safe summaries.
551
+ */
552
+ safeSummary?: (kind: "args" | "result", value: unknown) => string | undefined;
553
+ /** Allowlisted argument/result field names for a safe fallback summary. */
554
+ safeSummaryFields?: {
555
+ args?: string[];
556
+ result?: string[];
557
+ };
545
558
  }
546
559
  export interface Context {
547
560
  systemPrompt?: string[];
@@ -580,6 +593,20 @@ export type AssistantMessageEvent = {
580
593
  contentIndex: number;
581
594
  content: string;
582
595
  partial: AssistantMessage;
596
+ } | {
597
+ type: "reasoning_summary_start";
598
+ contentIndex: number;
599
+ partial: AssistantMessage;
600
+ } | {
601
+ type: "reasoning_summary_delta";
602
+ contentIndex: number;
603
+ delta: string;
604
+ partial: AssistantMessage;
605
+ } | {
606
+ type: "reasoning_summary_end";
607
+ contentIndex: number;
608
+ content: string;
609
+ partial: AssistantMessage;
583
610
  } | {
584
611
  type: "toolcall_start";
585
612
  contentIndex: number;
@@ -731,6 +758,12 @@ export interface AnthropicCompat extends ToolChoiceCompat {
731
758
  supportsForcedToolChoice?: boolean;
732
759
  /** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */
733
760
  supportsLongCacheRetention?: boolean;
761
+ /**
762
+ * Prompt-cache transport accepted by this Anthropic-compatible endpoint.
763
+ * Canonical Anthropic defaults to `"automatic"`; noncanonical endpoints default
764
+ * to `"none"` and must explicitly opt into generated `"explicit"` markers.
765
+ */
766
+ promptCacheMode?: "none" | "explicit" | "automatic";
734
767
  }
735
768
  /**
736
769
  * OpenRouter provider routing preferences.
@@ -1,4 +1,4 @@
1
- export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "other";
1
+ export type FallbackTriggerClass = "rate_limit" | "quota" | "auth" | "server" | "unknown" | "other";
2
2
  export interface FallbackTrigger {
3
3
  class: FallbackTriggerClass;
4
4
  retryAfterMs?: number;
@@ -7,12 +7,23 @@ export type TransportHeaders = Headers | Record<string, string | undefined>;
7
7
  /**
8
8
  * Structured facts from an upstream HTTP or transport failure. Retry decisions
9
9
  * must use these facts rather than provider- or application-owned error text.
10
+ *
11
+ * `headers` is always a plain record limited to the retained retry-signal
12
+ * entries: facts travel on persisted `AssistantMessage`s and through
13
+ * `structuredClone` snapshots (managed fallback attempt staging), so they must
14
+ * never carry a live `Headers` instance — cloning one throws `DataCloneError`
15
+ * ("The object can not be cloned.") and masks the real provider failure.
10
16
  */
11
17
  export interface TransportFailureFacts {
12
18
  kind: "transport";
13
19
  status?: number;
20
+ /** Canonical provider error code used for fallback classification. */
14
21
  providerCode?: string;
15
- headers?: TransportHeaders;
22
+ /** Anthropic's typed `error.type`, preserved separately at the transport boundary. */
23
+ anthropicErrorType?: string;
24
+ /** OpenAI's typed `error.code`, preserved separately at the transport boundary. */
25
+ openaiErrorCode?: string;
26
+ headers?: Record<string, string>;
16
27
  }
17
28
  /** Opaque per-invocation marker required by managed fallback transport calls. */
18
29
  export interface FallbackAttemptToken {
@@ -1,61 +1,11 @@
1
1
  import type { AssistantMessage } from "../types";
2
+ import type { TransportFailureFacts } from "./fallback-transport";
3
+ export declare function classifyContextOverflow(message: AssistantMessage, transportFailure?: TransportFailureFacts, contextWindow?: number): boolean;
2
4
  /**
3
5
  * Check if an assistant message represents a context overflow error.
4
6
  *
5
- * This handles three cases:
6
- * 1. Error-based overflow: Most providers return stopReason "error" with a
7
- * specific error message pattern.
8
- * 2. Silent overflow: Some providers accept overflow requests and return
9
- * successfully. For these, we check if usage.input exceeds the context window.
10
- * 3. Proxy-level overflow: Some proxies (e.g. LiteLLM) return a "successful"
11
- * response with empty content and a fabricated near-zero usage when the
12
- * upstream model's context window is exceeded.
13
- *
14
- * ## Reliability by Provider
15
- *
16
- * **Reliable detection (returns error with detectable message):**
17
- * - Anthropic: "prompt is too long: X tokens > Y maximum"
18
- * - OpenAI (Completions & Responses): "exceeds the context window"
19
- * - Google Gemini: "input token count exceeds the maximum"
20
- * - xAI (Grok): "maximum prompt length is X but request contains Y"
21
- * - Groq: "reduce the length of the messages"
22
- * - Cerebras: 400/413 status code (no body)
23
- * - Mistral: 400/413 status code (no body)
24
- * - HTTP 413 payload/entity-too-large variants
25
- * - OpenRouter (all backends): "maximum context length is X tokens"
26
- * - llama.cpp: "exceeds the available context size"
27
- * - LM Studio: "greater than the context length"
28
- * - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
29
- * - Anthropic 413: "request_too_large" (request body exceeds size limit)
30
- * - HTTP 413: "Payload Too Large" / "Request Entity Too Large"
31
- *
32
- * **Unreliable detection:**
33
- * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
34
- * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.
35
- * - Ollama: Silently truncates input without error. Cannot be detected via this function.
36
- * - LiteLLM proxy: Returns a "successful" response with empty content and a
37
- * fabricated near-zero usage (e.g. input: 1, output: 1) when the upstream
38
- * model's context window is exceeded. Detected via Case 3 (empty content +
39
- * anomalously low usage). Note: the LiteLLM proxy's context limit may differ
40
- * from the underlying model's advertised contextWindow (e.g. configured via
41
- * `model_info.max_tokens` in LiteLLM's config.yaml), so Case 2 (which compares
42
- * usage.input against contextWindow) may not catch it.
43
- * The response will have usage.input < expected, but we don't know the expected value.
44
- *
45
- * ## Custom Providers
46
- *
47
- * If you've added custom models via settings.json, this function may not detect
48
- * overflow errors from those providers. To add support:
49
- *
50
- * 1. Send a request that exceeds the model's context window
51
- * 2. Check the errorMessage in the response
52
- * 3. Create a regex pattern that matches the error
53
- * 4. The pattern should be added to OVERFLOW_PATTERNS in this file, or
54
- * check the errorMessage yourself before calling this function
55
- *
56
- * @param message - The assistant message to check
57
- * @param contextWindow - Optional context window size for detecting silent overflow (z.ai)
58
- * @returns true if the message indicates a context overflow
7
+ * Callers with normalized transport facts should use {@link classifyContextOverflow}
8
+ * so typed provider codes take precedence over error prose.
59
9
  */
60
10
  export declare function isContextOverflow(message: AssistantMessage, contextWindow?: number): boolean;
61
11
  /**
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.11.1",
4
+ "version": "0.11.2",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.11.1",
43
+ "@gajae-code/utils": "0.11.2",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -775,7 +775,9 @@ export class AuthStorage {
775
775
  */
776
776
  static async create(dbPath: string, options: AuthStorageOptions = {}): Promise<AuthStorage> {
777
777
  const store = await SqliteAuthCredentialStore.open(dbPath);
778
- return new AuthStorage(store, options);
778
+ const storage = new AuthStorage(store, options);
779
+ await storage.reload();
780
+ return storage;
779
781
  }
780
782
 
781
783
  /**
@@ -854,6 +856,7 @@ export class AuthStorage {
854
856
  */
855
857
  setRuntimeApiKey(provider: string, apiKey: string): void {
856
858
  this.#runtimeOverrides.set(provider, apiKey);
859
+ this.#bumpGeneration("set-runtime-api-key");
857
860
  }
858
861
 
859
862
  /**
@@ -877,7 +880,12 @@ export class AuthStorage {
877
880
  * Remove a runtime API key override.
878
881
  */
879
882
  removeRuntimeApiKey(provider: string): void {
880
- this.#runtimeOverrides.delete(provider);
883
+ if (this.#runtimeOverrides.delete(provider)) this.#bumpGeneration("remove-runtime-api-key");
884
+ }
885
+
886
+ /** Whether a provider is currently authenticated by a runtime API-key override. */
887
+ hasRuntimeApiKey(provider: string): boolean {
888
+ return Boolean(this.#runtimeOverrides.get(provider));
881
889
  }
882
890
 
883
891
  /**
@@ -892,13 +900,14 @@ export class AuthStorage {
892
900
  */
893
901
  setConfigApiKey(provider: string, apiKey: string): void {
894
902
  this.#configOverrides.set(provider, apiKey);
903
+ this.#bumpGeneration("set-config-api-key");
895
904
  }
896
905
 
897
906
  /**
898
907
  * Remove a single config-sourced API key override.
899
908
  */
900
909
  removeConfigApiKey(provider: string): void {
901
- this.#configOverrides.delete(provider);
910
+ if (this.#configOverrides.delete(provider)) this.#bumpGeneration("remove-config-api-key");
902
911
  }
903
912
 
904
913
  /**
@@ -906,7 +915,9 @@ export class AuthStorage {
906
915
  * re-parsing `models.yml` so removed entries actually disappear.
907
916
  */
908
917
  clearConfigApiKeys(): void {
918
+ if (this.#configOverrides.size === 0) return;
909
919
  this.#configOverrides.clear();
920
+ this.#bumpGeneration("clear-config-api-keys");
910
921
  }
911
922
 
912
923
  /**
@@ -41,6 +41,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
41
41
  modelsDev?: ModelsDevFallback<TApi, TModelsDevPayload>;
42
42
  /** Clock override for deterministic tests. */
43
43
  now?: () => number;
44
+ /** Optional guard that must permit cache publication. Default: writes are permitted. */
45
+ canPublishCache?: () => boolean;
44
46
  }
45
47
 
46
48
  /**
@@ -148,7 +150,9 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
148
150
  return { models: cachedModels, stale: false };
149
151
  }
150
152
  const repairedModels = mergeDynamicModels(staticModels, cachedModels);
151
- writeModelCache(options.providerId, now(), repairedModels, true, staticFingerprint, dbPath);
153
+ if (options.canPublishCache?.() ?? true) {
154
+ writeModelCache(options.providerId, now(), repairedModels, true, staticFingerprint, dbPath);
155
+ }
152
156
  return { models: repairedModels, stale: false };
153
157
  }
154
158
 
@@ -169,24 +173,28 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
169
173
  const snapshotModels = applyFinalCodexGpt56ContextCap(
170
174
  mergeDynamicModels(mergeModelSources(staticModels, modelsDevModels), dynamicModels),
171
175
  );
172
- writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
176
+ if (options.canPublishCache?.() ?? true) {
177
+ writeModelCache(options.providerId, now(), snapshotModels, true, staticFingerprint, dbPath);
178
+ }
173
179
  } else {
174
180
  // Dynamic fetch failed — update cache with a non-authoritative snapshot so
175
181
  // stale state remains visible while retry backoff still applies.
176
182
  const latestCache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
177
- writeModelCache(
178
- options.providerId,
179
- now(),
180
- applyFinalCodexGpt56ContextCap(
181
- mergeDynamicModels(
182
- mergeModelSources(staticModels, modelsDevModels),
183
- normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
183
+ if (options.canPublishCache?.() ?? true) {
184
+ writeModelCache(
185
+ options.providerId,
186
+ now(),
187
+ applyFinalCodexGpt56ContextCap(
188
+ mergeDynamicModels(
189
+ mergeModelSources(staticModels, modelsDevModels),
190
+ normalizeModelList<TApi>(latestCache?.models ?? cache?.models ?? []),
191
+ ),
184
192
  ),
185
- ),
186
- false,
187
- staticFingerprint,
188
- dbPath,
189
- );
193
+ false,
194
+ staticFingerprint,
195
+ dbPath,
196
+ );
197
+ }
190
198
  }
191
199
  }
192
200
  return {
@@ -417,6 +417,62 @@ function mapStopReasonOut(reason: StopReason): "end_turn" | "max_tokens" | "tool
417
417
  }
418
418
  }
419
419
 
420
+ /**
421
+ * True when `signature` is one of OUR serialized OpenAI Responses reasoning-item
422
+ * envelopes (produced by the Responses/Codex decoders as
423
+ * `JSON.stringify(reasoningItem)` with `type: "reasoning"`). Such an envelope is
424
+ * NOT a valid Anthropic thinking signature and may embed raw chain-of-thought in
425
+ * `content[]`, so it must never be forwarded on Anthropic egress. Genuine opaque
426
+ * Anthropic signatures and any non-envelope string return false (fail safe).
427
+ */
428
+ function isSerializedResponsesReasoningItem(signature: string): boolean {
429
+ try {
430
+ const parsed: unknown = JSON.parse(signature);
431
+ return (
432
+ typeof parsed === "object" &&
433
+ parsed !== null &&
434
+ !Array.isArray(parsed) &&
435
+ (parsed as { type?: unknown }).type === "reasoning"
436
+ );
437
+ } catch {
438
+ return false;
439
+ }
440
+ }
441
+
442
+ /**
443
+ * A thinking-block signature safe to emit on Anthropic egress: cross-protocol
444
+ * serialized Responses reasoning envelopes are omitted (they can carry raw CoT);
445
+ * genuine opaque Anthropic signatures round-trip unchanged.
446
+ */
447
+ function safeThinkingSignature(signature: string | undefined): string | undefined {
448
+ if (!signature) return undefined;
449
+ if (isSerializedResponsesReasoningItem(signature)) return undefined;
450
+ return signature;
451
+ }
452
+
453
+ function isResponsesFamilyApi(api: AssistantMessage["api"]): boolean {
454
+ return api === "openai-responses" || api === "openai-codex-responses";
455
+ }
456
+
457
+ function safeThinkingText(content: ThinkingContent, api: AssistantMessage["api"]): string | undefined {
458
+ if (isResponsesFamilyApi(api) && content.provenance === undefined) return undefined;
459
+ if (content.provenance === "raw") return undefined;
460
+ if (content.provenance === "mixed") return content.summaryText;
461
+ if (content.provenance === "summary") return content.summaryText ?? content.thinking;
462
+ return content.thinking;
463
+ }
464
+
465
+ function hasRawOrMixedThinking(partial: AssistantMessage, contentIndex: number): boolean {
466
+ const content = partial.content[contentIndex];
467
+ return content?.type === "thinking" && (content.provenance === "raw" || content.provenance === "mixed");
468
+ }
469
+
470
+ /** Responses-family reasoning is untrusted until output_item.done assigns provenance. */
471
+ function hasUnfinalizedResponsesThinking(partial: AssistantMessage, contentIndex: number): boolean {
472
+ const content = partial.content[contentIndex];
473
+ return content?.type !== "thinking" || (isResponsesFamilyApi(partial.api) && content.provenance === undefined);
474
+ }
475
+
420
476
  function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>[] {
421
477
  const blocks: Record<string, unknown>[] = [];
422
478
  for (const c of message.content) {
@@ -425,8 +481,11 @@ function encodeContentBlocks(message: AssistantMessage): Record<string, unknown>
425
481
  blocks.push({ type: "text", text: c.text });
426
482
  break;
427
483
  case "thinking": {
428
- const b: Record<string, unknown> = { type: "thinking", thinking: c.thinking };
429
- if (c.thinkingSignature) b.signature = c.thinkingSignature;
484
+ const thinking = safeThinkingText(c, message.api);
485
+ if (thinking === undefined) break;
486
+ const b: Record<string, unknown> = { type: "thinking", thinking };
487
+ const sig = safeThinkingSignature(c.thinkingSignature);
488
+ if (sig) b.signature = sig;
430
489
  blocks.push(b);
431
490
  break;
432
491
  }
@@ -495,6 +554,12 @@ export function encodeStream(
495
554
  const messageId = newMessageId();
496
555
  let started = false;
497
556
  const open = new Map<number, OpenBlock>();
557
+ // contentIndexes that already streamed a reasoning summary delta, so a
558
+ // final-only reasoning_summary_end does not duplicate streamed summary text.
559
+ const summaryDeltaSeen = new Set<number>();
560
+ // Responses assigns reasoning provenance only at output_item.done. Keep its
561
+ // pre-classification bytes out of this public compatibility stream.
562
+ const pendingThinkingDeltas = new Map<number, string[]>();
498
563
 
499
564
  const ensureStart = (partial: AssistantMessage) => {
500
565
  if (started) return;
@@ -524,6 +589,30 @@ export function encodeStream(
524
589
  open.delete(index);
525
590
  };
526
591
 
592
+ const openThinking = (partial: AssistantMessage, index: number) => {
593
+ if (open.has(index)) return;
594
+ ensureStart(partial);
595
+ open.set(index, { index, kind: "thinking" });
596
+ controller.enqueue(
597
+ sseFrame("content_block_start", {
598
+ type: "content_block_start",
599
+ index,
600
+ content_block: { type: "thinking", thinking: "" },
601
+ }),
602
+ );
603
+ };
604
+
605
+ const writeThinkingDelta = (index: number, thinking: string) => {
606
+ if (thinking.length === 0) return;
607
+ controller.enqueue(
608
+ sseFrame("content_block_delta", {
609
+ type: "content_block_delta",
610
+ index,
611
+ delta: { type: "thinking_delta", thinking },
612
+ }),
613
+ );
614
+ };
615
+
527
616
  try {
528
617
  for await (const ev of events) {
529
618
  switch (ev.type) {
@@ -555,18 +644,73 @@ export function encodeStream(
555
644
  closeBlock(ev.contentIndex);
556
645
  break;
557
646
  case "thinking_start": {
558
- ensureStart(ev.partial);
559
- open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
560
- controller.enqueue(
561
- sseFrame("content_block_start", {
562
- type: "content_block_start",
563
- index: ev.contentIndex,
564
- content_block: { type: "thinking", thinking: "" },
565
- }),
566
- );
647
+ if (hasRawOrMixedThinking(ev.partial, ev.contentIndex)) break;
648
+ if (hasUnfinalizedResponsesThinking(ev.partial, ev.contentIndex)) {
649
+ pendingThinkingDeltas.set(ev.contentIndex, []);
650
+ break;
651
+ }
652
+ openThinking(ev.partial, ev.contentIndex);
653
+ break;
654
+ }
655
+ case "thinking_delta": {
656
+ if (hasRawOrMixedThinking(ev.partial, ev.contentIndex)) break;
657
+ if (hasUnfinalizedResponsesThinking(ev.partial, ev.contentIndex)) {
658
+ const deltas = pendingThinkingDeltas.get(ev.contentIndex) ?? [];
659
+ deltas.push(ev.delta);
660
+ pendingThinkingDeltas.set(ev.contentIndex, deltas);
661
+ break;
662
+ }
663
+ writeThinkingDelta(ev.contentIndex, ev.delta);
664
+ break;
665
+ }
666
+ case "thinking_end": {
667
+ const unfinalized = hasUnfinalizedResponsesThinking(ev.partial, ev.contentIndex);
668
+ const rawOrMixed = hasRawOrMixedThinking(ev.partial, ev.contentIndex);
669
+ const pending = pendingThinkingDeltas.get(ev.contentIndex);
670
+ pendingThinkingDeltas.delete(ev.contentIndex);
671
+ if (rawOrMixed || unfinalized) {
672
+ closeBlock(ev.contentIndex);
673
+ break;
674
+ }
675
+ if (pending) {
676
+ openThinking(ev.partial, ev.contentIndex);
677
+ for (const delta of pending) writeThinkingDelta(ev.contentIndex, delta);
678
+ }
679
+ const c = ev.partial.content[ev.contentIndex];
680
+ const sig = c?.type === "thinking" ? safeThinkingSignature(c.thinkingSignature) : undefined;
681
+ if (sig) {
682
+ controller.enqueue(
683
+ sseFrame("content_block_delta", {
684
+ type: "content_block_delta",
685
+ index: ev.contentIndex,
686
+ delta: { type: "signature_delta", signature: sig },
687
+ }),
688
+ );
689
+ }
690
+ closeBlock(ev.contentIndex);
691
+ break;
692
+ }
693
+ case "reasoning_summary_start": {
694
+ // Provider-displayable summary reasoning surfaces as this format's
695
+ // native thinking channel. Open the thinking block if a thinking_start
696
+ // did not already (summary-only streams emit no thinking_start).
697
+ if (!open.has(ev.contentIndex)) {
698
+ ensureStart(ev.partial);
699
+ open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
700
+ controller.enqueue(
701
+ sseFrame("content_block_start", {
702
+ type: "content_block_start",
703
+ index: ev.contentIndex,
704
+ content_block: { type: "thinking", thinking: "" },
705
+ }),
706
+ );
707
+ }
567
708
  break;
568
709
  }
569
- case "thinking_delta":
710
+ case "reasoning_summary_delta":
711
+ // Only a non-whitespace delta counts as a delivered summary; a bare
712
+ // separator ("\n\n") must not suppress a later final-only end content.
713
+ if (ev.delta.trim().length > 0) summaryDeltaSeen.add(ev.contentIndex);
570
714
  controller.enqueue(
571
715
  sseFrame("content_block_delta", {
572
716
  type: "content_block_delta",
@@ -575,18 +719,31 @@ export function encodeStream(
575
719
  }),
576
720
  );
577
721
  break;
578
- case "thinking_end": {
579
- const c = ev.partial.content[ev.contentIndex];
580
- if (c?.type === "thinking" && c.thinkingSignature) {
722
+ case "reasoning_summary_end": {
723
+ // Final-only summary: text arrives only on the end event with no prior
724
+ // deltas. Ensure the thinking block is open, then surface the content as
725
+ // a thinking_delta (skip when deltas already streamed to avoid dup). The
726
+ // thinking block is closed by the subsequent thinking_end.
727
+ if (ev.content.length > 0 && !summaryDeltaSeen.has(ev.contentIndex)) {
728
+ if (!open.has(ev.contentIndex)) {
729
+ ensureStart(ev.partial);
730
+ open.set(ev.contentIndex, { index: ev.contentIndex, kind: "thinking" });
731
+ controller.enqueue(
732
+ sseFrame("content_block_start", {
733
+ type: "content_block_start",
734
+ index: ev.contentIndex,
735
+ content_block: { type: "thinking", thinking: "" },
736
+ }),
737
+ );
738
+ }
581
739
  controller.enqueue(
582
740
  sseFrame("content_block_delta", {
583
741
  type: "content_block_delta",
584
742
  index: ev.contentIndex,
585
- delta: { type: "signature_delta", signature: c.thinkingSignature },
743
+ delta: { type: "thinking_delta", thinking: ev.content },
586
744
  }),
587
745
  );
588
746
  }
589
- closeBlock(ev.contentIndex);
590
747
  break;
591
748
  }
592
749
  case "toolcall_start": {
@@ -621,6 +778,7 @@ export function encodeStream(
621
778
  break;
622
779
  case "done": {
623
780
  for (const idx of [...open.keys()]) closeBlock(idx);
781
+ pendingThinkingDeltas.clear();
624
782
  controller.enqueue(
625
783
  sseFrame("message_delta", {
626
784
  type: "message_delta",
@@ -636,6 +794,7 @@ export function encodeStream(
636
794
  }
637
795
  case "error": {
638
796
  const msg = ev.error.errorMessage ?? "stream error";
797
+ pendingThinkingDeltas.clear();
639
798
  controller.enqueue(
640
799
  sseFrame("error", { type: "error", error: { type: "api_error", message: msg } }),
641
800
  );
@@ -645,10 +804,12 @@ export function encodeStream(
645
804
  }
646
805
  }
647
806
  // stream ended without explicit done; close gracefully
807
+ pendingThinkingDeltas.clear();
648
808
  for (const idx of [...open.keys()]) closeBlock(idx);
649
809
  controller.enqueue(sseFrame("message_stop", { type: "message_stop" }));
650
810
  controller.close();
651
811
  } catch (err) {
812
+ pendingThinkingDeltas.clear();
652
813
  controller.enqueue(
653
814
  sseFrame("error", {
654
815
  type: "error",