@prestyj/ai 5.9.0 → 5.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -75,19 +77,31 @@ interface RawContent {
75
77
  data: Record<string, unknown>;
76
78
  }
77
79
  type ContentPart = TextContent | ThinkingContent | ImageContent | VideoContent | ToolCall | ServerToolCall | ServerToolResult | RawContent;
78
- interface SystemMessage {
80
+ type MessageProvenanceSource = "human" | "agent" | "runtime";
81
+ type MessageProvenanceKind = "prompt" | "steering" | "notification" | "completion_gate" | "review_follow_up" | "continuation" | "model_switch" | "automation" | "compaction_summary" | "compaction_ack";
82
+ type MessageProvenanceVisibility = "transcript" | "hidden" | "summary";
83
+ /** Internal message metadata. `stream()` removes it before provider dispatch. */
84
+ interface MessageProvenance {
85
+ source: MessageProvenanceSource;
86
+ kind: MessageProvenanceKind;
87
+ visibility: MessageProvenanceVisibility;
88
+ }
89
+ interface MessageMetadata {
90
+ provenance?: MessageProvenance;
91
+ }
92
+ interface SystemMessage extends MessageMetadata {
79
93
  role: "system";
80
94
  content: string;
81
95
  }
82
- interface UserMessage {
96
+ interface UserMessage extends MessageMetadata {
83
97
  role: "user";
84
98
  content: string | (TextContent | ImageContent | VideoContent)[];
85
99
  }
86
- interface AssistantMessage {
100
+ interface AssistantMessage extends MessageMetadata {
87
101
  role: "assistant";
88
102
  content: string | ContentPart[];
89
103
  }
90
- interface ToolResultMessage {
104
+ interface ToolResultMessage extends MessageMetadata {
91
105
  role: "tool";
92
106
  content: ToolResult[];
93
107
  }
@@ -159,7 +173,10 @@ interface StreamResponse {
159
173
  }
160
174
  interface Usage {
161
175
  inputTokens: number;
176
+ /** Total billed output tokens, including reasoning tokens when the provider reports them separately. */
162
177
  outputTokens: number;
178
+ /** Reasoning/thinking-token subset of outputTokens. */
179
+ reasoningTokens?: number;
163
180
  cacheRead?: number;
164
181
  cacheWrite?: number;
165
182
  serverToolUse?: {
@@ -293,6 +310,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
310
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
311
  }
295
312
 
313
+ /**
314
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
315
+ * the same model name served by two machines stays distinct in the registry.
316
+ * The server only knows the raw id, so strip the routing prefix here — at the
317
+ * one place that talks to the wire. Counterpart to gg-core's
318
+ * `formatLocalModelId`/`parseLocalModelId`.
319
+ */
320
+ declare function localWireModelId(id: string): string;
296
321
  /**
297
322
  * Unified streaming entry point. Returns a StreamResult that is both
298
323
  * an async iterable (for streaming events) and thenable (await for
@@ -480,6 +505,21 @@ declare function redactText(text: string, options?: RedactionOptions): string;
480
505
  */
481
506
  declare function redactValue<T>(value: T, options?: RedactionOptions): T;
482
507
 
508
+ /** True when the string contains at least one unpaired surrogate. */
509
+ declare function hasLoneSurrogate(text: string): boolean;
510
+ /** Replace unpaired surrogates with U+FFFD; returns the input when already valid. */
511
+ declare function toWellFormedText(text: string): string;
512
+ /** `text.slice(0, chars)` that never cuts an astral character in half. */
513
+ declare function sliceHead(text: string, chars: number): string;
514
+ /** `text.slice(-chars)` that never cuts an astral character in half. */
515
+ declare function sliceTail(text: string, chars: number): string;
516
+ /**
517
+ * Strip unpaired surrogates from everything headed for the wire. Returns the
518
+ * same array (and same message objects) when the history is already valid, so
519
+ * the clean path stays allocation-free.
520
+ */
521
+ declare function sanitizeMessagesForWire(messages: Message[]): Message[];
522
+
483
523
  /**
484
524
  * Provider-level diagnostic hook. Mirrors the pattern used by gg-agent's
485
525
  * setStreamDiagnostic — the host app wires a callback (typically writing to
@@ -507,6 +547,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
507
547
  provider?: string;
508
548
  thinking?: boolean;
509
549
  supportsImages?: boolean;
550
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
551
+ reasoningField?: string;
510
552
  }): OpenAI.ChatCompletionMessageParam[];
511
553
 
512
554
  /**
@@ -600,4 +642,4 @@ interface PalsuProviderConfig {
600
642
  */
601
643
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
602
644
 
603
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
645
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type MessageProvenance, type MessageProvenanceKind, type MessageProvenanceSource, type MessageProvenanceVisibility, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, hasLoneSurrogate, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, sanitizeMessagesForWire, setProviderDiagnostic, sliceHead, sliceTail, stream, toAnthropicMessages, toOpenAIMessages, toWellFormedText };
package/dist/index.d.ts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -75,19 +77,31 @@ interface RawContent {
75
77
  data: Record<string, unknown>;
76
78
  }
77
79
  type ContentPart = TextContent | ThinkingContent | ImageContent | VideoContent | ToolCall | ServerToolCall | ServerToolResult | RawContent;
78
- interface SystemMessage {
80
+ type MessageProvenanceSource = "human" | "agent" | "runtime";
81
+ type MessageProvenanceKind = "prompt" | "steering" | "notification" | "completion_gate" | "review_follow_up" | "continuation" | "model_switch" | "automation" | "compaction_summary" | "compaction_ack";
82
+ type MessageProvenanceVisibility = "transcript" | "hidden" | "summary";
83
+ /** Internal message metadata. `stream()` removes it before provider dispatch. */
84
+ interface MessageProvenance {
85
+ source: MessageProvenanceSource;
86
+ kind: MessageProvenanceKind;
87
+ visibility: MessageProvenanceVisibility;
88
+ }
89
+ interface MessageMetadata {
90
+ provenance?: MessageProvenance;
91
+ }
92
+ interface SystemMessage extends MessageMetadata {
79
93
  role: "system";
80
94
  content: string;
81
95
  }
82
- interface UserMessage {
96
+ interface UserMessage extends MessageMetadata {
83
97
  role: "user";
84
98
  content: string | (TextContent | ImageContent | VideoContent)[];
85
99
  }
86
- interface AssistantMessage {
100
+ interface AssistantMessage extends MessageMetadata {
87
101
  role: "assistant";
88
102
  content: string | ContentPart[];
89
103
  }
90
- interface ToolResultMessage {
104
+ interface ToolResultMessage extends MessageMetadata {
91
105
  role: "tool";
92
106
  content: ToolResult[];
93
107
  }
@@ -159,7 +173,10 @@ interface StreamResponse {
159
173
  }
160
174
  interface Usage {
161
175
  inputTokens: number;
176
+ /** Total billed output tokens, including reasoning tokens when the provider reports them separately. */
162
177
  outputTokens: number;
178
+ /** Reasoning/thinking-token subset of outputTokens. */
179
+ reasoningTokens?: number;
163
180
  cacheRead?: number;
164
181
  cacheWrite?: number;
165
182
  serverToolUse?: {
@@ -293,6 +310,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
310
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
311
  }
295
312
 
313
+ /**
314
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
315
+ * the same model name served by two machines stays distinct in the registry.
316
+ * The server only knows the raw id, so strip the routing prefix here — at the
317
+ * one place that talks to the wire. Counterpart to gg-core's
318
+ * `formatLocalModelId`/`parseLocalModelId`.
319
+ */
320
+ declare function localWireModelId(id: string): string;
296
321
  /**
297
322
  * Unified streaming entry point. Returns a StreamResult that is both
298
323
  * an async iterable (for streaming events) and thenable (await for
@@ -480,6 +505,21 @@ declare function redactText(text: string, options?: RedactionOptions): string;
480
505
  */
481
506
  declare function redactValue<T>(value: T, options?: RedactionOptions): T;
482
507
 
508
+ /** True when the string contains at least one unpaired surrogate. */
509
+ declare function hasLoneSurrogate(text: string): boolean;
510
+ /** Replace unpaired surrogates with U+FFFD; returns the input when already valid. */
511
+ declare function toWellFormedText(text: string): string;
512
+ /** `text.slice(0, chars)` that never cuts an astral character in half. */
513
+ declare function sliceHead(text: string, chars: number): string;
514
+ /** `text.slice(-chars)` that never cuts an astral character in half. */
515
+ declare function sliceTail(text: string, chars: number): string;
516
+ /**
517
+ * Strip unpaired surrogates from everything headed for the wire. Returns the
518
+ * same array (and same message objects) when the history is already valid, so
519
+ * the clean path stays allocation-free.
520
+ */
521
+ declare function sanitizeMessagesForWire(messages: Message[]): Message[];
522
+
483
523
  /**
484
524
  * Provider-level diagnostic hook. Mirrors the pattern used by gg-agent's
485
525
  * setStreamDiagnostic — the host app wires a callback (typically writing to
@@ -507,6 +547,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
507
547
  provider?: string;
508
548
  thinking?: boolean;
509
549
  supportsImages?: boolean;
550
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
551
+ reasoningField?: string;
510
552
  }): OpenAI.ChatCompletionMessageParam[];
511
553
 
512
554
  /**
@@ -600,4 +642,4 @@ interface PalsuProviderConfig {
600
642
  */
601
643
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
602
644
 
603
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
645
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type MessageProvenance, type MessageProvenanceKind, type MessageProvenanceSource, type MessageProvenanceVisibility, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, hasLoneSurrogate, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, sanitizeMessagesForWire, setProviderDiagnostic, sliceHead, sliceTail, stream, toAnthropicMessages, toOpenAIMessages, toWellFormedText };
package/dist/index.js CHANGED
@@ -501,6 +501,39 @@ function normalizeRootForAnthropic(schema) {
501
501
  return out;
502
502
  }
503
503
 
504
+ // src/providers/reasoning-field.ts
505
+ var REASONING_FIELD_ALIASES = [
506
+ "reasoning_content",
507
+ "reasoning",
508
+ "reasoning_text"
509
+ ];
510
+ var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
511
+ function readReasoning(obj) {
512
+ if (!obj) return void 0;
513
+ for (const field of REASONING_FIELD_ALIASES) {
514
+ const value = obj[field];
515
+ if (typeof value === "string" && value) return { field, text: value };
516
+ }
517
+ return void 0;
518
+ }
519
+ function reasoningFieldKey(provider, baseUrl, model) {
520
+ return `${provider}|${baseUrl ?? ""}|${model}`;
521
+ }
522
+ var MAX_REMEMBERED_ENDPOINTS = 64;
523
+ var detectedFields = /* @__PURE__ */ new Map();
524
+ function rememberReasoningField(key, field) {
525
+ if (detectedFields.get(key) === field) return;
526
+ detectedFields.set(key, field);
527
+ while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
528
+ const oldest = detectedFields.keys().next();
529
+ if (oldest.done) break;
530
+ detectedFields.delete(oldest.value);
531
+ }
532
+ }
533
+ function getReasoningField(key) {
534
+ return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
535
+ }
536
+
504
537
  // src/providers/transform.ts
505
538
  function hasValidThinkingSignature(part) {
506
539
  return typeof part.signature === "string" && part.signature.trim().length > 0;
@@ -885,12 +918,12 @@ function toAnthropicToolChoice(choice) {
885
918
  return { type: "tool", name: choice.name };
886
919
  }
887
920
  function isAdaptiveThinkingModel(model) {
888
- return /opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
921
+ return /opus-5|opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
889
922
  }
890
923
  function toAnthropicThinking(level, maxTokens, model) {
891
924
  if (isAdaptiveThinkingModel(model)) {
892
925
  let effort = level;
893
- if (effort === "xhigh" && !/opus-4-8|opus-4-7/.test(model)) {
926
+ if (effort === "xhigh" && !/opus-5|opus-4-8|opus-4-7/.test(model)) {
894
927
  effort = "high";
895
928
  }
896
929
  return {
@@ -921,6 +954,7 @@ function remapToolCallId(id, idMap) {
921
954
  return mapped;
922
955
  }
923
956
  function toOpenAIMessages(messages, options) {
957
+ const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
924
958
  const out = [];
925
959
  const idMap = /* @__PURE__ */ new Map();
926
960
  const mergeToolResultText = options?.provider === "glm";
@@ -987,9 +1021,9 @@ function toOpenAIMessages(messages, options) {
987
1021
  ...hasToolCalls ? { tool_calls: toolCalls } : {}
988
1022
  };
989
1023
  if (thinkingParts) {
990
- assistantMsg.reasoning_content = thinkingParts;
1024
+ assistantMsg[reasoningField] = thinkingParts;
991
1025
  } else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
992
- assistantMsg.reasoning_content = " ";
1026
+ assistantMsg[reasoningField] = " ";
993
1027
  }
994
1028
  out.push(assistantMsg);
995
1029
  continue;
@@ -1067,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
1067
1101
  if (choice === "required") return "required";
1068
1102
  return { type: "function", function: { name: choice.name } };
1069
1103
  }
1104
+ function toLocalReasoningEffort(level) {
1105
+ if (level === "max" || level === "ultra" || level === "xhigh") return "max";
1106
+ return level;
1107
+ }
1070
1108
  function toOpenAIReasoningEffort(level, model) {
1071
1109
  const effort = level === "max" || level === "ultra" ? "xhigh" : level;
1072
1110
  if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
@@ -1863,7 +1901,9 @@ function streamOpenAI(options) {
1863
1901
  async function* runStream2(options) {
1864
1902
  const providerName = options.provider ?? "openai";
1865
1903
  const useStreaming = options.streaming !== false;
1904
+ const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
1866
1905
  const client = createClient2(options);
1906
+ const isLocal = options.provider === "local";
1867
1907
  const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1868
1908
  const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1869
1909
  const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
@@ -1886,7 +1926,8 @@ async function* runStream2(options) {
1886
1926
  // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1887
1927
  // official CLI: reasoning is preserved only while thinking is enabled).
1888
1928
  thinking: isKimiK27 || !!options.thinking,
1889
- supportsImages: options.supportsImages
1929
+ supportsImages: options.supportsImages,
1930
+ reasoningField: getReasoningField(endpointKey)
1890
1931
  });
1891
1932
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
1892
1933
  const effectiveTemp = options.temperature ?? defaultTemp;
@@ -1898,7 +1939,7 @@ async function* runStream2(options) {
1898
1939
  ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1899
1940
  ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1900
1941
  ...options.stop ? { stop: options.stop } : {},
1901
- ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1942
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1902
1943
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1903
1944
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1904
1945
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1912,6 +1953,11 @@ async function* runStream2(options) {
1912
1953
  paramsAny.prompt_cache_retention = "24h";
1913
1954
  }
1914
1955
  }
1956
+ if (isLocal && options.thinking) {
1957
+ params.reasoning_effort = toLocalReasoningEffort(
1958
+ options.thinking
1959
+ );
1960
+ }
1915
1961
  if (options.provider === "openai" && options.serviceTier) {
1916
1962
  params.service_tier = options.serviceTier;
1917
1963
  }
@@ -1948,8 +1994,8 @@ async function* runStream2(options) {
1948
1994
  const completion = await client.chat.completions.create(params, {
1949
1995
  signal: options.signal ?? void 0
1950
1996
  });
1951
- yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
1952
- return completionToResponse(completion);
1997
+ yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
1998
+ return completionToResponse(completion, endpointKey);
1953
1999
  } catch (err) {
1954
2000
  throw toError2(err, providerName);
1955
2001
  }
@@ -1984,11 +2030,12 @@ async function* runStream2(options) {
1984
2030
  finishReason = choice.finish_reason;
1985
2031
  }
1986
2032
  const delta = choice.delta;
1987
- const reasoningContent = delta.reasoning_content;
1988
- if (typeof reasoningContent === "string" && reasoningContent) {
1989
- thinkingAccum += reasoningContent;
2033
+ const reasoning = readReasoning(delta);
2034
+ if (reasoning) {
2035
+ rememberReasoningField(endpointKey, reasoning.field);
2036
+ thinkingAccum += reasoning.text;
1990
2037
  if (options.thinking) {
1991
- yield { type: "thinking_delta", text: reasoningContent };
2038
+ yield { type: "thinking_delta", text: reasoning.text };
1992
2039
  }
1993
2040
  }
1994
2041
  if (delta.content) {
@@ -2073,16 +2120,17 @@ async function* runStream2(options) {
2073
2120
  yield { type: "done", stopReason };
2074
2121
  return response;
2075
2122
  }
2076
- function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2123
+ function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
2077
2124
  const choice = completion.choices?.[0];
2078
2125
  if (!choice) {
2079
2126
  yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
2080
2127
  return;
2081
2128
  }
2082
2129
  const msg = choice.message;
2083
- const reasoning = msg.reasoning_content;
2084
- if (typeof reasoning === "string" && reasoning && thinkingEnabled) {
2085
- yield { type: "thinking_delta", text: reasoning };
2130
+ const reasoning = readReasoning(msg);
2131
+ if (reasoning) {
2132
+ rememberReasoningField(endpointKey, reasoning.field);
2133
+ if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
2086
2134
  }
2087
2135
  if (typeof msg.content === "string" && msg.content) {
2088
2136
  yield { type: "text_delta", text: msg.content };
@@ -2110,15 +2158,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2110
2158
  }
2111
2159
  yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
2112
2160
  }
2113
- function completionToResponse(completion) {
2161
+ function completionToResponse(completion, endpointKey) {
2114
2162
  const choice = completion.choices?.[0];
2115
2163
  const contentParts = [];
2116
2164
  let textAccum = "";
2117
2165
  if (choice) {
2118
2166
  const msg = choice.message;
2119
- const reasoning = msg.reasoning_content;
2120
- if (typeof reasoning === "string" && reasoning) {
2121
- contentParts.push({ type: "thinking", text: reasoning });
2167
+ const reasoning = readReasoning(msg);
2168
+ if (reasoning) {
2169
+ rememberReasoningField(endpointKey, reasoning.field);
2170
+ contentParts.push({ type: "thinking", text: reasoning.text });
2122
2171
  }
2123
2172
  if (typeof msg.content === "string" && msg.content) {
2124
2173
  textAccum = msg.content;
@@ -3270,14 +3319,16 @@ async function* runStream4(options) {
3270
3319
  let thinkingAccum = "";
3271
3320
  let stopReason = "end_turn";
3272
3321
  let inputTokens = 0;
3273
- let outputTokens = 0;
3322
+ let candidateTokens = 0;
3323
+ let reasoningTokens = 0;
3274
3324
  let cacheRead = 0;
3275
3325
  let toolIndex = 0;
3276
3326
  const handleResponse = function* (chunk) {
3277
3327
  const usage = usageFromResponse(chunk);
3278
3328
  if (usage) {
3279
3329
  inputTokens = usage.promptTokenCount ?? inputTokens;
3280
- outputTokens = usage.candidatesTokenCount ?? outputTokens;
3330
+ candidateTokens = usage.candidatesTokenCount ?? candidateTokens;
3331
+ reasoningTokens = usage.thoughtsTokenCount ?? reasoningTokens;
3281
3332
  cacheRead = usage.cachedContentTokenCount ?? cacheRead;
3282
3333
  }
3283
3334
  const reason = finishReasonFromResponse(chunk);
@@ -3333,6 +3384,7 @@ async function* runStream4(options) {
3333
3384
  }
3334
3385
  if (pendingToolCalls.length > 0) stopReason = "tool_use";
3335
3386
  const adjustedInputTokens = Math.max(0, inputTokens - cacheRead);
3387
+ const outputTokens = candidateTokens + reasoningTokens;
3336
3388
  const streamResponse = {
3337
3389
  message: {
3338
3390
  role: "assistant",
@@ -3342,6 +3394,7 @@ async function* runStream4(options) {
3342
3394
  usage: {
3343
3395
  inputTokens: adjustedInputTokens,
3344
3396
  outputTokens,
3397
+ ...reasoningTokens > 0 ? { reasoningTokens } : {},
3345
3398
  ...cacheRead > 0 ? { cacheRead } : {}
3346
3399
  }
3347
3400
  };
@@ -3390,6 +3443,140 @@ var ProviderRegistryImpl = class {
3390
3443
  };
3391
3444
  var providerRegistry = new ProviderRegistryImpl();
3392
3445
 
3446
+ // src/utils/well-formed.ts
3447
+ var LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/;
3448
+ var LONE_SURROGATE_GLOBAL = new RegExp(LONE_SURROGATE, "g");
3449
+ var REPLACEMENT = "\uFFFD";
3450
+ function hasLoneSurrogate(text) {
3451
+ const isWellFormed = text.isWellFormed;
3452
+ if (typeof isWellFormed === "function") return !isWellFormed.call(text);
3453
+ return LONE_SURROGATE.test(text);
3454
+ }
3455
+ function toWellFormedText(text) {
3456
+ if (!hasLoneSurrogate(text)) return text;
3457
+ const toWellFormed = text.toWellFormed;
3458
+ if (typeof toWellFormed === "function") return toWellFormed.call(text);
3459
+ return text.replace(LONE_SURROGATE_GLOBAL, REPLACEMENT);
3460
+ }
3461
+ function isHighSurrogate(code) {
3462
+ return code !== void 0 && code >= 55296 && code <= 56319;
3463
+ }
3464
+ function isLowSurrogate(code) {
3465
+ return code !== void 0 && code >= 56320 && code <= 57343;
3466
+ }
3467
+ function sliceHead(text, chars) {
3468
+ if (chars <= 0) return "";
3469
+ if (chars >= text.length) return text;
3470
+ const end = isHighSurrogate(text.charCodeAt(chars - 1)) ? chars - 1 : chars;
3471
+ return text.slice(0, end);
3472
+ }
3473
+ function sliceTail(text, chars) {
3474
+ if (chars <= 0) return "";
3475
+ if (chars >= text.length) return text;
3476
+ const start = text.length - chars;
3477
+ return text.slice(isLowSurrogate(text.charCodeAt(start)) ? start + 1 : start);
3478
+ }
3479
+ function sanitizeJsonValue(value) {
3480
+ if (typeof value === "string") return toWellFormedText(value);
3481
+ if (Array.isArray(value)) {
3482
+ let changed = false;
3483
+ const next = value.map((item) => {
3484
+ const sanitized = sanitizeJsonValue(item);
3485
+ if (sanitized !== item) changed = true;
3486
+ return sanitized;
3487
+ });
3488
+ return changed ? next : value;
3489
+ }
3490
+ if (value !== null && typeof value === "object") {
3491
+ let changed = false;
3492
+ const next = {};
3493
+ for (const [key, item] of Object.entries(value)) {
3494
+ const sanitizedKey = toWellFormedText(key);
3495
+ const sanitized = sanitizeJsonValue(item);
3496
+ if (sanitizedKey !== key || sanitized !== item) changed = true;
3497
+ next[sanitizedKey] = sanitized;
3498
+ }
3499
+ return changed ? next : value;
3500
+ }
3501
+ return value;
3502
+ }
3503
+ function sanitizeRecord(value) {
3504
+ return sanitizeJsonValue(value);
3505
+ }
3506
+ function sanitizePart(part) {
3507
+ switch (part.type) {
3508
+ case "text":
3509
+ case "thinking": {
3510
+ const text = toWellFormedText(part.text);
3511
+ return text === part.text ? part : { ...part, text };
3512
+ }
3513
+ case "tool_call": {
3514
+ const args = sanitizeRecord(part.args);
3515
+ return args === part.args ? part : { ...part, args };
3516
+ }
3517
+ case "server_tool_call": {
3518
+ const input = sanitizeJsonValue(part.input);
3519
+ return input === part.input ? part : { ...part, input };
3520
+ }
3521
+ case "server_tool_result": {
3522
+ const data = sanitizeJsonValue(part.data);
3523
+ return data === part.data ? part : { ...part, data };
3524
+ }
3525
+ case "raw": {
3526
+ const data = sanitizeRecord(part.data);
3527
+ return data === part.data ? part : { ...part, data };
3528
+ }
3529
+ default:
3530
+ return part;
3531
+ }
3532
+ }
3533
+ function sanitizeParts(parts) {
3534
+ let changed = false;
3535
+ const next = parts.map((part) => {
3536
+ const sanitized = sanitizePart(part);
3537
+ if (sanitized !== part) changed = true;
3538
+ return sanitized;
3539
+ });
3540
+ return changed ? next : parts;
3541
+ }
3542
+ function sanitizeToolResultContent(content) {
3543
+ if (typeof content === "string") return toWellFormedText(content);
3544
+ return sanitizeParts(content);
3545
+ }
3546
+ function sanitizeToolResults(results) {
3547
+ let changed = false;
3548
+ const next = results.map((result) => {
3549
+ const content = sanitizeToolResultContent(result.content);
3550
+ if (content === result.content) return result;
3551
+ changed = true;
3552
+ return { ...result, content };
3553
+ });
3554
+ return changed ? next : results;
3555
+ }
3556
+ function sanitizeMessage(message) {
3557
+ if (message.role === "tool") {
3558
+ const content2 = sanitizeToolResults(message.content);
3559
+ return content2 === message.content ? message : { ...message, content: content2 };
3560
+ }
3561
+ if (typeof message.content === "string") {
3562
+ const content2 = toWellFormedText(message.content);
3563
+ return content2 === message.content ? message : { ...message, content: content2 };
3564
+ }
3565
+ const content = sanitizeParts(message.content);
3566
+ return content === message.content ? message : { ...message, content };
3567
+ }
3568
+ function sanitizeMessagesForWire(messages) {
3569
+ let sanitized;
3570
+ for (let index = 0; index < messages.length; index++) {
3571
+ const message = messages[index];
3572
+ const next = sanitizeMessage(message);
3573
+ if (next === message) continue;
3574
+ sanitized ??= messages.slice();
3575
+ sanitized[index] = next;
3576
+ }
3577
+ return sanitized ?? messages;
3578
+ }
3579
+
3393
3580
  // src/stream.ts
3394
3581
  var GLM_CODING_BASE_URL = "https://api.z.ai/api/coding/paas/v4";
3395
3582
  var KIMI_CODE_USER_AGENT = `kimi-code-cli/${process.env.KIMI_CODE_VERSION ?? "1.0.11"}`;
@@ -3479,6 +3666,28 @@ providerRegistry.register("minimax", {
3479
3666
  serverTools: void 0
3480
3667
  })
3481
3668
  });
3669
+ function localWireModelId(id) {
3670
+ const match = /^local\/[^/]+\/(.+)$/.exec(id);
3671
+ return match?.[1] ?? id;
3672
+ }
3673
+ providerRegistry.register("local", {
3674
+ // Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
3675
+ // vLLM). There is no default endpoint: the baseUrl comes from the endpoint
3676
+ // credential the discovery layer wrote, so a missing one is a wiring bug, not
3677
+ // something to paper over with a guess at someone else's port.
3678
+ stream: (options) => {
3679
+ if (!options.baseUrl) {
3680
+ throw new EZCoderAIError(
3681
+ "Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
3682
+ );
3683
+ }
3684
+ return streamOpenAI({
3685
+ ...options,
3686
+ model: localWireModelId(options.model),
3687
+ webSearch: false
3688
+ });
3689
+ }
3690
+ });
3482
3691
  function stream(options) {
3483
3692
  const entry = providerRegistry.get(options.provider);
3484
3693
  if (!entry) {
@@ -3489,13 +3698,25 @@ function stream(options) {
3489
3698
  if (options.supportsVideo !== true && messagesContainVideo(options.messages)) {
3490
3699
  throw new VideoUnsupportedError();
3491
3700
  }
3701
+ const wireMessages = stripMessageProvenance(options.messages);
3492
3702
  const messages = clampProviderContextImages(
3493
- options.messages,
3703
+ sanitizeMessagesForWire(wireMessages),
3494
3704
  options.provider,
3495
3705
  options.supportsImages
3496
3706
  );
3497
3707
  return entry.stream(messages === options.messages ? options : { ...options, messages });
3498
3708
  }
3709
+ function stripMessageProvenance(messages) {
3710
+ let stripped;
3711
+ for (let index = 0; index < messages.length; index++) {
3712
+ const message = messages[index];
3713
+ if (!message.provenance) continue;
3714
+ stripped ??= messages.slice();
3715
+ const { provenance: _provenance, ...wireMessage } = message;
3716
+ stripped[index] = wireMessage;
3717
+ }
3718
+ return stripped ?? messages;
3719
+ }
3499
3720
  function messagesContainVideo(messages) {
3500
3721
  for (const msg of messages) {
3501
3722
  if (typeof msg.content === "string" || !Array.isArray(msg.content)) continue;
@@ -3880,8 +4101,10 @@ export {
3880
4101
  environmentSecrets,
3881
4102
  formatError,
3882
4103
  formatErrorForDisplay,
4104
+ hasLoneSurrogate,
3883
4105
  isHardBillingMessage,
3884
4106
  isUsageLimitError,
4107
+ localWireModelId,
3885
4108
  palsuAssistantMessage,
3886
4109
  palsuText,
3887
4110
  palsuThinking,
@@ -3891,9 +4114,13 @@ export {
3891
4114
  redactText,
3892
4115
  redactValue,
3893
4116
  registerPalsuProvider,
4117
+ sanitizeMessagesForWire,
3894
4118
  setProviderDiagnostic,
4119
+ sliceHead,
4120
+ sliceTail,
3895
4121
  stream,
3896
4122
  toAnthropicMessages,
3897
- toOpenAIMessages
4123
+ toOpenAIMessages,
4124
+ toWellFormedText
3898
4125
  };
3899
4126
  //# sourceMappingURL=index.js.map