@prestyj/ai 5.8.0 → 5.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -40,7 +40,7 @@ Tool parameters are Zod schemas. Converted to JSON Schema at the provider bounda
40
40
 
41
41
  | Provider | Models | Notes |
42
42
  |---|---|---|
43
- | `anthropic` | Claude Opus 4.8, Sonnet 5, Haiku 4.5 | Extended thinking, prompt caching, server-side compaction |
43
+ | `anthropic` | Claude Opus 5, Sonnet 5, Haiku 4.5 | Extended thinking, prompt caching, server-side compaction |
44
44
  | `openai` | GPT-4.1, o3, o4-mini | Supports OAuth (codex endpoint) and API keys |
45
45
  | `glm` | GLM-5.1, GLM-4.7 | Z.AI platform, OpenAI-compatible |
46
46
  | `moonshot` | Kimi K3, Kimi K2.7 Code | Moonshot platform, OpenAI-compatible |
package/dist/index.cjs CHANGED
@@ -42,6 +42,7 @@ __export(index_exports, {
42
42
  formatErrorForDisplay: () => formatErrorForDisplay,
43
43
  isHardBillingMessage: () => isHardBillingMessage,
44
44
  isUsageLimitError: () => isUsageLimitError,
45
+ localWireModelId: () => localWireModelId,
45
46
  palsuAssistantMessage: () => palsuAssistantMessage,
46
47
  palsuText: () => palsuText,
47
48
  palsuThinking: () => palsuThinking,
@@ -324,6 +325,9 @@ function providerGuidance(provider, message, statusCode) {
324
325
  if (lower.includes("context_length_exceeded") || lower.includes("prompt is too long")) {
325
326
  return `Context window for this ${name} model is full. Compact the conversation to shrink history, or start a new session.`;
326
327
  }
328
+ if (lower.includes("many-image request") || lower.includes("image dimensions") && lower.includes("max allowed size")) {
329
+ return `An image in conversation history exceeds ${name}'s many-image limit. Restart EZ Coder so restored images are resized, then retry; if it persists, start a new session.`;
330
+ }
327
331
  if (statusCode === 413 || lower.includes("request_too_large") || lower.includes("request exceeds the maximum size")) {
328
332
  return `The request to ${name} is too large. Compact the conversation to shrink history, or start a new session.`;
329
333
  }
@@ -558,6 +562,39 @@ function normalizeRootForAnthropic(schema) {
558
562
  return out;
559
563
  }
560
564
 
565
+ // src/providers/reasoning-field.ts
566
+ var REASONING_FIELD_ALIASES = [
567
+ "reasoning_content",
568
+ "reasoning",
569
+ "reasoning_text"
570
+ ];
571
+ var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
572
+ function readReasoning(obj) {
573
+ if (!obj) return void 0;
574
+ for (const field of REASONING_FIELD_ALIASES) {
575
+ const value = obj[field];
576
+ if (typeof value === "string" && value) return { field, text: value };
577
+ }
578
+ return void 0;
579
+ }
580
+ function reasoningFieldKey(provider, baseUrl, model) {
581
+ return `${provider}|${baseUrl ?? ""}|${model}`;
582
+ }
583
+ var MAX_REMEMBERED_ENDPOINTS = 64;
584
+ var detectedFields = /* @__PURE__ */ new Map();
585
+ function rememberReasoningField(key, field) {
586
+ if (detectedFields.get(key) === field) return;
587
+ detectedFields.set(key, field);
588
+ while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
589
+ const oldest = detectedFields.keys().next();
590
+ if (oldest.done) break;
591
+ detectedFields.delete(oldest.value);
592
+ }
593
+ }
594
+ function getReasoningField(key) {
595
+ return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
596
+ }
597
+
561
598
  // src/providers/transform.ts
562
599
  function hasValidThinkingSignature(part) {
563
600
  return typeof part.signature === "string" && part.signature.trim().length > 0;
@@ -820,9 +857,14 @@ function toAnthropicMessages(messages, cacheControl) {
820
857
  continue;
821
858
  }
822
859
  if (msg.role === "user") {
860
+ if (typeof msg.content === "string") {
861
+ if (msg.content === "") continue;
862
+ } else if (!msg.content.some((p) => !(p.type === "text" && p.text === ""))) {
863
+ continue;
864
+ }
823
865
  out.push({
824
866
  role: "user",
825
- content: typeof msg.content === "string" ? msg.content : msg.content.map((part) => {
867
+ content: typeof msg.content === "string" ? msg.content : msg.content.filter((part) => !(part.type === "text" && part.text === "")).map((part) => {
826
868
  if (part.type === "text") return { type: "text", text: part.text };
827
869
  if (part.type === "video") {
828
870
  return {
@@ -847,6 +889,7 @@ function toAnthropicMessages(messages, cacheControl) {
847
889
  continue;
848
890
  }
849
891
  if (msg.role === "assistant") {
892
+ if (typeof msg.content === "string" && msg.content === "") continue;
850
893
  const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
851
894
  if (Array.isArray(content) && content.length === 0) continue;
852
895
  out.push({ role: "assistant", content });
@@ -936,12 +979,12 @@ function toAnthropicToolChoice(choice) {
936
979
  return { type: "tool", name: choice.name };
937
980
  }
938
981
  function isAdaptiveThinkingModel(model) {
939
- return /opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
982
+ return /opus-5|opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
940
983
  }
941
984
  function toAnthropicThinking(level, maxTokens, model) {
942
985
  if (isAdaptiveThinkingModel(model)) {
943
986
  let effort = level;
944
- if (effort === "xhigh" && !/opus-4-8|opus-4-7/.test(model)) {
987
+ if (effort === "xhigh" && !/opus-5|opus-4-8|opus-4-7/.test(model)) {
945
988
  effort = "high";
946
989
  }
947
990
  return {
@@ -967,11 +1010,12 @@ function remapToolCallId(id, idMap) {
967
1010
  if (!id.startsWith("toolu_")) return id;
968
1011
  const existing = idMap.get(id);
969
1012
  if (existing) return existing;
970
- const mapped = `call_${id.slice(5)}`;
1013
+ const mapped = `call_${id.slice(6)}`;
971
1014
  idMap.set(id, mapped);
972
1015
  return mapped;
973
1016
  }
974
1017
  function toOpenAIMessages(messages, options) {
1018
+ const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
975
1019
  const out = [];
976
1020
  const idMap = /* @__PURE__ */ new Map();
977
1021
  const mergeToolResultText = options?.provider === "glm";
@@ -1038,9 +1082,9 @@ function toOpenAIMessages(messages, options) {
1038
1082
  ...hasToolCalls ? { tool_calls: toolCalls } : {}
1039
1083
  };
1040
1084
  if (thinkingParts) {
1041
- assistantMsg.reasoning_content = thinkingParts;
1085
+ assistantMsg[reasoningField] = thinkingParts;
1042
1086
  } else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
1043
- assistantMsg.reasoning_content = " ";
1087
+ assistantMsg[reasoningField] = " ";
1044
1088
  }
1045
1089
  out.push(assistantMsg);
1046
1090
  continue;
@@ -1118,6 +1162,10 @@ function toOpenAIToolChoice(choice) {
1118
1162
  if (choice === "required") return "required";
1119
1163
  return { type: "function", function: { name: choice.name } };
1120
1164
  }
1165
+ function toLocalReasoningEffort(level) {
1166
+ if (level === "max" || level === "ultra" || level === "xhigh") return "max";
1167
+ return level;
1168
+ }
1121
1169
  function toOpenAIReasoningEffort(level, model) {
1122
1170
  const effort = level === "max" || level === "ultra" ? "xhigh" : level;
1123
1171
  if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
@@ -1578,6 +1626,12 @@ async function* runStream(options) {
1578
1626
  statusCode: 504
1579
1627
  });
1580
1628
  }
1629
+ if (stopReason === null) {
1630
+ throw new ProviderError("anthropic", "Stream ended before completion (no stop_reason).", {
1631
+ statusCode: 504,
1632
+ cause: { partialContent: contentParts, outputTokens }
1633
+ });
1634
+ }
1581
1635
  const normalizedStop = normalizeAnthropicStopReason(stopReason);
1582
1636
  const response = {
1583
1637
  message: {
@@ -1844,6 +1898,17 @@ function getEnvironment() {
1844
1898
  }
1845
1899
 
1846
1900
  // src/providers/openai.ts
1901
+ function toKimiK3Effort(level) {
1902
+ switch (level) {
1903
+ case "low":
1904
+ return "low";
1905
+ case "medium":
1906
+ case "high":
1907
+ return "high";
1908
+ default:
1909
+ return "max";
1910
+ }
1911
+ }
1847
1912
  function extractOpenAIUsage(usage) {
1848
1913
  let cacheRead = 0;
1849
1914
  let cacheWrite = 0;
@@ -1897,9 +1962,12 @@ function streamOpenAI(options) {
1897
1962
  async function* runStream2(options) {
1898
1963
  const providerName = options.provider ?? "openai";
1899
1964
  const useStreaming = options.streaming !== false;
1965
+ const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
1900
1966
  const client = createClient2(options);
1967
+ const isLocal = options.provider === "local";
1901
1968
  const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1902
1969
  const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1970
+ const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
1903
1971
  const isKimiK27 = options.provider === "moonshot" && options.model.startsWith("kimi-k2.7-code");
1904
1972
  const hasFixedKimiSampling = isKimiK3 || isKimiK27;
1905
1973
  const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" && !isKimiK3 && !isKimiK27 || options.provider === "xiaomi";
@@ -1914,10 +1982,13 @@ async function* runStream2(options) {
1914
1982
  }
1915
1983
  const messages = toOpenAIMessages(downgradedMessages, {
1916
1984
  provider: options.provider,
1917
- // K3 and K2.7 preserve reasoning even when the user hides thinking in the
1918
- // UI; keep assistant tool-call history wire-valid in that display mode.
1919
- thinking: isKimiK3 || isKimiK27 || !!options.thinking,
1920
- supportsImages: options.supportsImages
1985
+ // K2.7 preserves reasoning even when the user hides thinking in the UI;
1986
+ // keep assistant tool-call history wire-valid in that display mode. A
1987
+ // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1988
+ // official CLI: reasoning is preserved only while thinking is enabled).
1989
+ thinking: isKimiK27 || !!options.thinking,
1990
+ supportsImages: options.supportsImages,
1991
+ reasoningField: getReasoningField(endpointKey)
1921
1992
  });
1922
1993
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
1923
1994
  const effectiveTemp = options.temperature ?? defaultTemp;
@@ -1929,7 +2000,7 @@ async function* runStream2(options) {
1929
2000
  ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1930
2001
  ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1931
2002
  ...options.stop ? { stop: options.stop } : {},
1932
- ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
2003
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1933
2004
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1934
2005
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1935
2006
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1943,15 +2014,22 @@ async function* runStream2(options) {
1943
2014
  paramsAny.prompt_cache_retention = "24h";
1944
2015
  }
1945
2016
  }
2017
+ if (isLocal && options.thinking) {
2018
+ params.reasoning_effort = toLocalReasoningEffort(
2019
+ options.thinking
2020
+ );
2021
+ }
1946
2022
  if (options.provider === "openai" && options.serviceTier) {
1947
2023
  params.service_tier = options.serviceTier;
1948
2024
  }
1949
2025
  if (isKimiK3) {
1950
2026
  const paramsAny = params;
1951
2027
  if (isManagedKimiK3) {
1952
- paramsAny.thinking = { type: "enabled", effort: "max", keep: "all" };
2028
+ paramsAny.thinking = k3Effort ? { type: "enabled", effort: k3Effort, keep: "all" } : { type: "disabled" };
2029
+ } else if (k3Effort) {
2030
+ paramsAny.reasoning_effort = k3Effort;
1953
2031
  } else {
1954
- paramsAny.reasoning_effort = "max";
2032
+ paramsAny.thinking = { type: "disabled" };
1955
2033
  }
1956
2034
  }
1957
2035
  if (usesThinkingParam) {
@@ -1977,8 +2055,8 @@ async function* runStream2(options) {
1977
2055
  const completion = await client.chat.completions.create(params, {
1978
2056
  signal: options.signal ?? void 0
1979
2057
  });
1980
- yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
1981
- return completionToResponse(completion);
2058
+ yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
2059
+ return completionToResponse(completion, endpointKey);
1982
2060
  } catch (err) {
1983
2061
  throw toError2(err, providerName);
1984
2062
  }
@@ -2013,11 +2091,12 @@ async function* runStream2(options) {
2013
2091
  finishReason = choice.finish_reason;
2014
2092
  }
2015
2093
  const delta = choice.delta;
2016
- const reasoningContent = delta.reasoning_content;
2017
- if (typeof reasoningContent === "string" && reasoningContent) {
2018
- thinkingAccum += reasoningContent;
2094
+ const reasoning = readReasoning(delta);
2095
+ if (reasoning) {
2096
+ rememberReasoningField(endpointKey, reasoning.field);
2097
+ thinkingAccum += reasoning.text;
2019
2098
  if (options.thinking) {
2020
- yield { type: "thinking_delta", text: reasoningContent };
2099
+ yield { type: "thinking_delta", text: reasoning.text };
2021
2100
  }
2022
2101
  }
2023
2102
  if (delta.content) {
@@ -2057,6 +2136,12 @@ async function* runStream2(options) {
2057
2136
  statusCode: 504
2058
2137
  });
2059
2138
  }
2139
+ if (finishReason === null) {
2140
+ throw new ProviderError(providerName, "Stream ended before completion (no finish_reason).", {
2141
+ statusCode: 504,
2142
+ cause: { partialText: textAccum, outputTokens }
2143
+ });
2144
+ }
2060
2145
  if (thinkingAccum) {
2061
2146
  contentParts.push({ type: "thinking", text: thinkingAccum });
2062
2147
  }
@@ -2096,16 +2181,17 @@ async function* runStream2(options) {
2096
2181
  yield { type: "done", stopReason };
2097
2182
  return response;
2098
2183
  }
2099
- function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2184
+ function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
2100
2185
  const choice = completion.choices?.[0];
2101
2186
  if (!choice) {
2102
2187
  yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
2103
2188
  return;
2104
2189
  }
2105
2190
  const msg = choice.message;
2106
- const reasoning = msg.reasoning_content;
2107
- if (typeof reasoning === "string" && reasoning && thinkingEnabled) {
2108
- yield { type: "thinking_delta", text: reasoning };
2191
+ const reasoning = readReasoning(msg);
2192
+ if (reasoning) {
2193
+ rememberReasoningField(endpointKey, reasoning.field);
2194
+ if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
2109
2195
  }
2110
2196
  if (typeof msg.content === "string" && msg.content) {
2111
2197
  yield { type: "text_delta", text: msg.content };
@@ -2133,15 +2219,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2133
2219
  }
2134
2220
  yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
2135
2221
  }
2136
- function completionToResponse(completion) {
2222
+ function completionToResponse(completion, endpointKey) {
2137
2223
  const choice = completion.choices?.[0];
2138
2224
  const contentParts = [];
2139
2225
  let textAccum = "";
2140
2226
  if (choice) {
2141
2227
  const msg = choice.message;
2142
- const reasoning = msg.reasoning_content;
2143
- if (typeof reasoning === "string" && reasoning) {
2144
- contentParts.push({ type: "thinking", text: reasoning });
2228
+ const reasoning = readReasoning(msg);
2229
+ if (reasoning) {
2230
+ rememberReasoningField(endpointKey, reasoning.field);
2231
+ contentParts.push({ type: "thinking", text: reasoning.text });
2145
2232
  }
2146
2233
  if (typeof msg.content === "string" && msg.content) {
2147
2234
  textAccum = msg.content;
@@ -3502,6 +3589,28 @@ providerRegistry.register("minimax", {
3502
3589
  serverTools: void 0
3503
3590
  })
3504
3591
  });
3592
+ function localWireModelId(id) {
3593
+ const match = /^local\/[^/]+\/(.+)$/.exec(id);
3594
+ return match?.[1] ?? id;
3595
+ }
3596
+ providerRegistry.register("local", {
3597
+ // Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
3598
+ // vLLM). There is no default endpoint: the baseUrl comes from the endpoint
3599
+ // credential the discovery layer wrote, so a missing one is a wiring bug, not
3600
+ // something to paper over with a guess at someone else's port.
3601
+ stream: (options) => {
3602
+ if (!options.baseUrl) {
3603
+ throw new EZCoderAIError(
3604
+ "Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
3605
+ );
3606
+ }
3607
+ return streamOpenAI({
3608
+ ...options,
3609
+ model: localWireModelId(options.model),
3610
+ webSearch: false
3611
+ });
3612
+ }
3613
+ });
3505
3614
  function stream(options) {
3506
3615
  const entry = providerRegistry.get(options.provider);
3507
3616
  if (!entry) {
@@ -3906,6 +4015,7 @@ function registerPalsuProvider(config) {
3906
4015
  formatErrorForDisplay,
3907
4016
  isHardBillingMessage,
3908
4017
  isUsageLimitError,
4018
+ localWireModelId,
3909
4019
  palsuAssistantMessage,
3910
4020
  palsuText,
3911
4021
  palsuThinking,