@prestyj/ai 5.8.0 → 5.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -41,6 +43,21 @@ interface ToolResult {
41
43
  toolCallId: string;
42
44
  content: ToolResultContent;
43
45
  isError?: boolean;
46
+ /**
47
+ * Set when the agent loop trimmed `content` to fit a per-result or per-turn
48
+ * budget. The provider (model input) and the persistent transcript both see
49
+ * the trimmed `content`, but the live `tool_call_end` event carried the FULL
50
+ * preview — so this marker makes that divergence explicit and reconcilable.
51
+ * Internal metadata only: it is never serialized onto the provider wire.
52
+ */
53
+ capped?: {
54
+ /** Length of the original, untrimmed string content. */
55
+ originalChars: number;
56
+ /** Length of the trimmed content actually sent to the model. */
57
+ keptChars: number;
58
+ /** Which budget triggered the trim. */
59
+ scope: "per-result" | "per-turn";
60
+ };
44
61
  }
45
62
  interface ServerToolCall {
46
63
  type: "server_tool_call";
@@ -278,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
278
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
279
296
  }
280
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
281
306
  /**
282
307
  * Unified streaming entry point. Returns a StreamResult that is both
283
308
  * an async iterable (for streaming events) and thenable (await for
@@ -492,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
492
517
  provider?: string;
493
518
  thinking?: boolean;
494
519
  supportsImages?: boolean;
520
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
521
+ reasoningField?: string;
495
522
  }): OpenAI.ChatCompletionMessageParam[];
496
523
 
497
524
  /**
@@ -585,4 +612,4 @@ interface PalsuProviderConfig {
585
612
  */
586
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
587
614
 
588
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.d.ts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -41,6 +43,21 @@ interface ToolResult {
41
43
  toolCallId: string;
42
44
  content: ToolResultContent;
43
45
  isError?: boolean;
46
+ /**
47
+ * Set when the agent loop trimmed `content` to fit a per-result or per-turn
48
+ * budget. The provider (model input) and the persistent transcript both see
49
+ * the trimmed `content`, but the live `tool_call_end` event carried the FULL
50
+ * preview — so this marker makes that divergence explicit and reconcilable.
51
+ * Internal metadata only: it is never serialized onto the provider wire.
52
+ */
53
+ capped?: {
54
+ /** Length of the original, untrimmed string content. */
55
+ originalChars: number;
56
+ /** Length of the trimmed content actually sent to the model. */
57
+ keptChars: number;
58
+ /** Which budget triggered the trim. */
59
+ scope: "per-result" | "per-turn";
60
+ };
44
61
  }
45
62
  interface ServerToolCall {
46
63
  type: "server_tool_call";
@@ -278,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
278
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
279
296
  }
280
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
281
306
  /**
282
307
  * Unified streaming entry point. Returns a StreamResult that is both
283
308
  * an async iterable (for streaming events) and thenable (await for
@@ -492,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
492
517
  provider?: string;
493
518
  thinking?: boolean;
494
519
  supportsImages?: boolean;
520
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
521
+ reasoningField?: string;
495
522
  }): OpenAI.ChatCompletionMessageParam[];
496
523
 
497
524
  /**
@@ -585,4 +612,4 @@ interface PalsuProviderConfig {
585
612
  */
586
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
587
614
 
588
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.js CHANGED
@@ -264,6 +264,9 @@ function providerGuidance(provider, message, statusCode) {
264
264
  if (lower.includes("context_length_exceeded") || lower.includes("prompt is too long")) {
265
265
  return `Context window for this ${name} model is full. Compact the conversation to shrink history, or start a new session.`;
266
266
  }
267
+ if (lower.includes("many-image request") || lower.includes("image dimensions") && lower.includes("max allowed size")) {
268
+ return `An image in conversation history exceeds ${name}'s many-image limit. Restart EZ Coder so restored images are resized, then retry; if it persists, start a new session.`;
269
+ }
267
270
  if (statusCode === 413 || lower.includes("request_too_large") || lower.includes("request exceeds the maximum size")) {
268
271
  return `The request to ${name} is too large. Compact the conversation to shrink history, or start a new session.`;
269
272
  }
@@ -498,6 +501,39 @@ function normalizeRootForAnthropic(schema) {
498
501
  return out;
499
502
  }
500
503
 
504
+ // src/providers/reasoning-field.ts
505
+ var REASONING_FIELD_ALIASES = [
506
+ "reasoning_content",
507
+ "reasoning",
508
+ "reasoning_text"
509
+ ];
510
+ var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
511
+ function readReasoning(obj) {
512
+ if (!obj) return void 0;
513
+ for (const field of REASONING_FIELD_ALIASES) {
514
+ const value = obj[field];
515
+ if (typeof value === "string" && value) return { field, text: value };
516
+ }
517
+ return void 0;
518
+ }
519
+ function reasoningFieldKey(provider, baseUrl, model) {
520
+ return `${provider}|${baseUrl ?? ""}|${model}`;
521
+ }
522
+ var MAX_REMEMBERED_ENDPOINTS = 64;
523
+ var detectedFields = /* @__PURE__ */ new Map();
524
+ function rememberReasoningField(key, field) {
525
+ if (detectedFields.get(key) === field) return;
526
+ detectedFields.set(key, field);
527
+ while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
528
+ const oldest = detectedFields.keys().next();
529
+ if (oldest.done) break;
530
+ detectedFields.delete(oldest.value);
531
+ }
532
+ }
533
+ function getReasoningField(key) {
534
+ return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
535
+ }
536
+
501
537
  // src/providers/transform.ts
502
538
  function hasValidThinkingSignature(part) {
503
539
  return typeof part.signature === "string" && part.signature.trim().length > 0;
@@ -760,9 +796,14 @@ function toAnthropicMessages(messages, cacheControl) {
760
796
  continue;
761
797
  }
762
798
  if (msg.role === "user") {
799
+ if (typeof msg.content === "string") {
800
+ if (msg.content === "") continue;
801
+ } else if (!msg.content.some((p) => !(p.type === "text" && p.text === ""))) {
802
+ continue;
803
+ }
763
804
  out.push({
764
805
  role: "user",
765
- content: typeof msg.content === "string" ? msg.content : msg.content.map((part) => {
806
+ content: typeof msg.content === "string" ? msg.content : msg.content.filter((part) => !(part.type === "text" && part.text === "")).map((part) => {
766
807
  if (part.type === "text") return { type: "text", text: part.text };
767
808
  if (part.type === "video") {
768
809
  return {
@@ -787,6 +828,7 @@ function toAnthropicMessages(messages, cacheControl) {
787
828
  continue;
788
829
  }
789
830
  if (msg.role === "assistant") {
831
+ if (typeof msg.content === "string" && msg.content === "") continue;
790
832
  const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
791
833
  if (Array.isArray(content) && content.length === 0) continue;
792
834
  out.push({ role: "assistant", content });
@@ -876,12 +918,12 @@ function toAnthropicToolChoice(choice) {
876
918
  return { type: "tool", name: choice.name };
877
919
  }
878
920
  function isAdaptiveThinkingModel(model) {
879
- return /opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
921
+ return /opus-5|opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
880
922
  }
881
923
  function toAnthropicThinking(level, maxTokens, model) {
882
924
  if (isAdaptiveThinkingModel(model)) {
883
925
  let effort = level;
884
- if (effort === "xhigh" && !/opus-4-8|opus-4-7/.test(model)) {
926
+ if (effort === "xhigh" && !/opus-5|opus-4-8|opus-4-7/.test(model)) {
885
927
  effort = "high";
886
928
  }
887
929
  return {
@@ -907,11 +949,12 @@ function remapToolCallId(id, idMap) {
907
949
  if (!id.startsWith("toolu_")) return id;
908
950
  const existing = idMap.get(id);
909
951
  if (existing) return existing;
910
- const mapped = `call_${id.slice(5)}`;
952
+ const mapped = `call_${id.slice(6)}`;
911
953
  idMap.set(id, mapped);
912
954
  return mapped;
913
955
  }
914
956
  function toOpenAIMessages(messages, options) {
957
+ const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
915
958
  const out = [];
916
959
  const idMap = /* @__PURE__ */ new Map();
917
960
  const mergeToolResultText = options?.provider === "glm";
@@ -978,9 +1021,9 @@ function toOpenAIMessages(messages, options) {
978
1021
  ...hasToolCalls ? { tool_calls: toolCalls } : {}
979
1022
  };
980
1023
  if (thinkingParts) {
981
- assistantMsg.reasoning_content = thinkingParts;
1024
+ assistantMsg[reasoningField] = thinkingParts;
982
1025
  } else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
983
- assistantMsg.reasoning_content = " ";
1026
+ assistantMsg[reasoningField] = " ";
984
1027
  }
985
1028
  out.push(assistantMsg);
986
1029
  continue;
@@ -1058,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
1058
1101
  if (choice === "required") return "required";
1059
1102
  return { type: "function", function: { name: choice.name } };
1060
1103
  }
1104
+ function toLocalReasoningEffort(level) {
1105
+ if (level === "max" || level === "ultra" || level === "xhigh") return "max";
1106
+ return level;
1107
+ }
1061
1108
  function toOpenAIReasoningEffort(level, model) {
1062
1109
  const effort = level === "max" || level === "ultra" ? "xhigh" : level;
1063
1110
  if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
@@ -1518,6 +1565,12 @@ async function* runStream(options) {
1518
1565
  statusCode: 504
1519
1566
  });
1520
1567
  }
1568
+ if (stopReason === null) {
1569
+ throw new ProviderError("anthropic", "Stream ended before completion (no stop_reason).", {
1570
+ statusCode: 504,
1571
+ cause: { partialContent: contentParts, outputTokens }
1572
+ });
1573
+ }
1521
1574
  const normalizedStop = normalizeAnthropicStopReason(stopReason);
1522
1575
  const response = {
1523
1576
  message: {
@@ -1784,6 +1837,17 @@ function getEnvironment() {
1784
1837
  }
1785
1838
 
1786
1839
  // src/providers/openai.ts
1840
+ function toKimiK3Effort(level) {
1841
+ switch (level) {
1842
+ case "low":
1843
+ return "low";
1844
+ case "medium":
1845
+ case "high":
1846
+ return "high";
1847
+ default:
1848
+ return "max";
1849
+ }
1850
+ }
1787
1851
  function extractOpenAIUsage(usage) {
1788
1852
  let cacheRead = 0;
1789
1853
  let cacheWrite = 0;
@@ -1837,9 +1901,12 @@ function streamOpenAI(options) {
1837
1901
  async function* runStream2(options) {
1838
1902
  const providerName = options.provider ?? "openai";
1839
1903
  const useStreaming = options.streaming !== false;
1904
+ const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
1840
1905
  const client = createClient2(options);
1906
+ const isLocal = options.provider === "local";
1841
1907
  const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1842
1908
  const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1909
+ const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
1843
1910
  const isKimiK27 = options.provider === "moonshot" && options.model.startsWith("kimi-k2.7-code");
1844
1911
  const hasFixedKimiSampling = isKimiK3 || isKimiK27;
1845
1912
  const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" && !isKimiK3 && !isKimiK27 || options.provider === "xiaomi";
@@ -1854,10 +1921,13 @@ async function* runStream2(options) {
1854
1921
  }
1855
1922
  const messages = toOpenAIMessages(downgradedMessages, {
1856
1923
  provider: options.provider,
1857
- // K3 and K2.7 preserve reasoning even when the user hides thinking in the
1858
- // UI; keep assistant tool-call history wire-valid in that display mode.
1859
- thinking: isKimiK3 || isKimiK27 || !!options.thinking,
1860
- supportsImages: options.supportsImages
1924
+ // K2.7 preserves reasoning even when the user hides thinking in the UI;
1925
+ // keep assistant tool-call history wire-valid in that display mode. A
1926
+ // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1927
+ // official CLI: reasoning is preserved only while thinking is enabled).
1928
+ thinking: isKimiK27 || !!options.thinking,
1929
+ supportsImages: options.supportsImages,
1930
+ reasoningField: getReasoningField(endpointKey)
1861
1931
  });
1862
1932
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
1863
1933
  const effectiveTemp = options.temperature ?? defaultTemp;
@@ -1869,7 +1939,7 @@ async function* runStream2(options) {
1869
1939
  ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1870
1940
  ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1871
1941
  ...options.stop ? { stop: options.stop } : {},
1872
- ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1942
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1873
1943
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1874
1944
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1875
1945
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1883,15 +1953,22 @@ async function* runStream2(options) {
1883
1953
  paramsAny.prompt_cache_retention = "24h";
1884
1954
  }
1885
1955
  }
1956
+ if (isLocal && options.thinking) {
1957
+ params.reasoning_effort = toLocalReasoningEffort(
1958
+ options.thinking
1959
+ );
1960
+ }
1886
1961
  if (options.provider === "openai" && options.serviceTier) {
1887
1962
  params.service_tier = options.serviceTier;
1888
1963
  }
1889
1964
  if (isKimiK3) {
1890
1965
  const paramsAny = params;
1891
1966
  if (isManagedKimiK3) {
1892
- paramsAny.thinking = { type: "enabled", effort: "max", keep: "all" };
1967
+ paramsAny.thinking = k3Effort ? { type: "enabled", effort: k3Effort, keep: "all" } : { type: "disabled" };
1968
+ } else if (k3Effort) {
1969
+ paramsAny.reasoning_effort = k3Effort;
1893
1970
  } else {
1894
- paramsAny.reasoning_effort = "max";
1971
+ paramsAny.thinking = { type: "disabled" };
1895
1972
  }
1896
1973
  }
1897
1974
  if (usesThinkingParam) {
@@ -1917,8 +1994,8 @@ async function* runStream2(options) {
1917
1994
  const completion = await client.chat.completions.create(params, {
1918
1995
  signal: options.signal ?? void 0
1919
1996
  });
1920
- yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
1921
- return completionToResponse(completion);
1997
+ yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
1998
+ return completionToResponse(completion, endpointKey);
1922
1999
  } catch (err) {
1923
2000
  throw toError2(err, providerName);
1924
2001
  }
@@ -1953,11 +2030,12 @@ async function* runStream2(options) {
1953
2030
  finishReason = choice.finish_reason;
1954
2031
  }
1955
2032
  const delta = choice.delta;
1956
- const reasoningContent = delta.reasoning_content;
1957
- if (typeof reasoningContent === "string" && reasoningContent) {
1958
- thinkingAccum += reasoningContent;
2033
+ const reasoning = readReasoning(delta);
2034
+ if (reasoning) {
2035
+ rememberReasoningField(endpointKey, reasoning.field);
2036
+ thinkingAccum += reasoning.text;
1959
2037
  if (options.thinking) {
1960
- yield { type: "thinking_delta", text: reasoningContent };
2038
+ yield { type: "thinking_delta", text: reasoning.text };
1961
2039
  }
1962
2040
  }
1963
2041
  if (delta.content) {
@@ -1997,6 +2075,12 @@ async function* runStream2(options) {
1997
2075
  statusCode: 504
1998
2076
  });
1999
2077
  }
2078
+ if (finishReason === null) {
2079
+ throw new ProviderError(providerName, "Stream ended before completion (no finish_reason).", {
2080
+ statusCode: 504,
2081
+ cause: { partialText: textAccum, outputTokens }
2082
+ });
2083
+ }
2000
2084
  if (thinkingAccum) {
2001
2085
  contentParts.push({ type: "thinking", text: thinkingAccum });
2002
2086
  }
@@ -2036,16 +2120,17 @@ async function* runStream2(options) {
2036
2120
  yield { type: "done", stopReason };
2037
2121
  return response;
2038
2122
  }
2039
- function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2123
+ function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
2040
2124
  const choice = completion.choices?.[0];
2041
2125
  if (!choice) {
2042
2126
  yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
2043
2127
  return;
2044
2128
  }
2045
2129
  const msg = choice.message;
2046
- const reasoning = msg.reasoning_content;
2047
- if (typeof reasoning === "string" && reasoning && thinkingEnabled) {
2048
- yield { type: "thinking_delta", text: reasoning };
2130
+ const reasoning = readReasoning(msg);
2131
+ if (reasoning) {
2132
+ rememberReasoningField(endpointKey, reasoning.field);
2133
+ if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
2049
2134
  }
2050
2135
  if (typeof msg.content === "string" && msg.content) {
2051
2136
  yield { type: "text_delta", text: msg.content };
@@ -2073,15 +2158,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2073
2158
  }
2074
2159
  yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
2075
2160
  }
2076
- function completionToResponse(completion) {
2161
+ function completionToResponse(completion, endpointKey) {
2077
2162
  const choice = completion.choices?.[0];
2078
2163
  const contentParts = [];
2079
2164
  let textAccum = "";
2080
2165
  if (choice) {
2081
2166
  const msg = choice.message;
2082
- const reasoning = msg.reasoning_content;
2083
- if (typeof reasoning === "string" && reasoning) {
2084
- contentParts.push({ type: "thinking", text: reasoning });
2167
+ const reasoning = readReasoning(msg);
2168
+ if (reasoning) {
2169
+ rememberReasoningField(endpointKey, reasoning.field);
2170
+ contentParts.push({ type: "thinking", text: reasoning.text });
2085
2171
  }
2086
2172
  if (typeof msg.content === "string" && msg.content) {
2087
2173
  textAccum = msg.content;
@@ -3442,6 +3528,28 @@ providerRegistry.register("minimax", {
3442
3528
  serverTools: void 0
3443
3529
  })
3444
3530
  });
3531
+ function localWireModelId(id) {
3532
+ const match = /^local\/[^/]+\/(.+)$/.exec(id);
3533
+ return match?.[1] ?? id;
3534
+ }
3535
+ providerRegistry.register("local", {
3536
+ // Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
3537
+ // vLLM). There is no default endpoint: the baseUrl comes from the endpoint
3538
+ // credential the discovery layer wrote, so a missing one is a wiring bug, not
3539
+ // something to paper over with a guess at someone else's port.
3540
+ stream: (options) => {
3541
+ if (!options.baseUrl) {
3542
+ throw new EZCoderAIError(
3543
+ "Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
3544
+ );
3545
+ }
3546
+ return streamOpenAI({
3547
+ ...options,
3548
+ model: localWireModelId(options.model),
3549
+ webSearch: false
3550
+ });
3551
+ }
3552
+ });
3445
3553
  function stream(options) {
3446
3554
  const entry = providerRegistry.get(options.provider);
3447
3555
  if (!entry) {
@@ -3845,6 +3953,7 @@ export {
3845
3953
  formatErrorForDisplay,
3846
3954
  isHardBillingMessage,
3847
3955
  isUsageLimitError,
3956
+ localWireModelId,
3848
3957
  palsuAssistantMessage,
3849
3958
  palsuText,
3850
3959
  palsuThinking,