@prestyj/ai 5.9.0 → 5.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
296
  }
295
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
296
306
  /**
297
307
  * Unified streaming entry point. Returns a StreamResult that is both
298
308
  * an async iterable (for streaming events) and thenable (await for
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
507
517
  provider?: string;
508
518
  thinking?: boolean;
509
519
  supportsImages?: boolean;
520
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
521
+ reasoningField?: string;
510
522
  }): OpenAI.ChatCompletionMessageParam[];
511
523
 
512
524
  /**
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
600
612
  */
601
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
602
614
 
603
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.d.ts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
296
  }
295
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
296
306
  /**
297
307
  * Unified streaming entry point. Returns a StreamResult that is both
298
308
  * an async iterable (for streaming events) and thenable (await for
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
507
517
  provider?: string;
508
518
  thinking?: boolean;
509
519
  supportsImages?: boolean;
520
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
521
+ reasoningField?: string;
510
522
  }): OpenAI.ChatCompletionMessageParam[];
511
523
 
512
524
  /**
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
600
612
  */
601
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
602
614
 
603
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.js CHANGED
@@ -501,6 +501,39 @@ function normalizeRootForAnthropic(schema) {
501
501
  return out;
502
502
  }
503
503
 
504
+ // src/providers/reasoning-field.ts
505
+ var REASONING_FIELD_ALIASES = [
506
+ "reasoning_content",
507
+ "reasoning",
508
+ "reasoning_text"
509
+ ];
510
+ var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
511
+ function readReasoning(obj) {
512
+ if (!obj) return void 0;
513
+ for (const field of REASONING_FIELD_ALIASES) {
514
+ const value = obj[field];
515
+ if (typeof value === "string" && value) return { field, text: value };
516
+ }
517
+ return void 0;
518
+ }
519
+ function reasoningFieldKey(provider, baseUrl, model) {
520
+ return `${provider}|${baseUrl ?? ""}|${model}`;
521
+ }
522
+ var MAX_REMEMBERED_ENDPOINTS = 64;
523
+ var detectedFields = /* @__PURE__ */ new Map();
524
+ function rememberReasoningField(key, field) {
525
+ if (detectedFields.get(key) === field) return;
526
+ detectedFields.set(key, field);
527
+ while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
528
+ const oldest = detectedFields.keys().next();
529
+ if (oldest.done) break;
530
+ detectedFields.delete(oldest.value);
531
+ }
532
+ }
533
+ function getReasoningField(key) {
534
+ return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
535
+ }
536
+
504
537
  // src/providers/transform.ts
505
538
  function hasValidThinkingSignature(part) {
506
539
  return typeof part.signature === "string" && part.signature.trim().length > 0;
@@ -885,12 +918,12 @@ function toAnthropicToolChoice(choice) {
885
918
  return { type: "tool", name: choice.name };
886
919
  }
887
920
  function isAdaptiveThinkingModel(model) {
888
- return /opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
921
+ return /opus-5|opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
889
922
  }
890
923
  function toAnthropicThinking(level, maxTokens, model) {
891
924
  if (isAdaptiveThinkingModel(model)) {
892
925
  let effort = level;
893
- if (effort === "xhigh" && !/opus-4-8|opus-4-7/.test(model)) {
926
+ if (effort === "xhigh" && !/opus-5|opus-4-8|opus-4-7/.test(model)) {
894
927
  effort = "high";
895
928
  }
896
929
  return {
@@ -921,6 +954,7 @@ function remapToolCallId(id, idMap) {
921
954
  return mapped;
922
955
  }
923
956
  function toOpenAIMessages(messages, options) {
957
+ const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
924
958
  const out = [];
925
959
  const idMap = /* @__PURE__ */ new Map();
926
960
  const mergeToolResultText = options?.provider === "glm";
@@ -987,9 +1021,9 @@ function toOpenAIMessages(messages, options) {
987
1021
  ...hasToolCalls ? { tool_calls: toolCalls } : {}
988
1022
  };
989
1023
  if (thinkingParts) {
990
- assistantMsg.reasoning_content = thinkingParts;
1024
+ assistantMsg[reasoningField] = thinkingParts;
991
1025
  } else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
992
- assistantMsg.reasoning_content = " ";
1026
+ assistantMsg[reasoningField] = " ";
993
1027
  }
994
1028
  out.push(assistantMsg);
995
1029
  continue;
@@ -1067,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
1067
1101
  if (choice === "required") return "required";
1068
1102
  return { type: "function", function: { name: choice.name } };
1069
1103
  }
1104
+ function toLocalReasoningEffort(level) {
1105
+ if (level === "max" || level === "ultra" || level === "xhigh") return "max";
1106
+ return level;
1107
+ }
1070
1108
  function toOpenAIReasoningEffort(level, model) {
1071
1109
  const effort = level === "max" || level === "ultra" ? "xhigh" : level;
1072
1110
  if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
@@ -1863,7 +1901,9 @@ function streamOpenAI(options) {
1863
1901
  async function* runStream2(options) {
1864
1902
  const providerName = options.provider ?? "openai";
1865
1903
  const useStreaming = options.streaming !== false;
1904
+ const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
1866
1905
  const client = createClient2(options);
1906
+ const isLocal = options.provider === "local";
1867
1907
  const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1868
1908
  const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1869
1909
  const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
@@ -1886,7 +1926,8 @@ async function* runStream2(options) {
1886
1926
  // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1887
1927
  // official CLI: reasoning is preserved only while thinking is enabled).
1888
1928
  thinking: isKimiK27 || !!options.thinking,
1889
- supportsImages: options.supportsImages
1929
+ supportsImages: options.supportsImages,
1930
+ reasoningField: getReasoningField(endpointKey)
1890
1931
  });
1891
1932
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
1892
1933
  const effectiveTemp = options.temperature ?? defaultTemp;
@@ -1898,7 +1939,7 @@ async function* runStream2(options) {
1898
1939
  ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1899
1940
  ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1900
1941
  ...options.stop ? { stop: options.stop } : {},
1901
- ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1942
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1902
1943
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1903
1944
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1904
1945
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1912,6 +1953,11 @@ async function* runStream2(options) {
1912
1953
  paramsAny.prompt_cache_retention = "24h";
1913
1954
  }
1914
1955
  }
1956
+ if (isLocal && options.thinking) {
1957
+ params.reasoning_effort = toLocalReasoningEffort(
1958
+ options.thinking
1959
+ );
1960
+ }
1915
1961
  if (options.provider === "openai" && options.serviceTier) {
1916
1962
  params.service_tier = options.serviceTier;
1917
1963
  }
@@ -1948,8 +1994,8 @@ async function* runStream2(options) {
1948
1994
  const completion = await client.chat.completions.create(params, {
1949
1995
  signal: options.signal ?? void 0
1950
1996
  });
1951
- yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
1952
- return completionToResponse(completion);
1997
+ yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
1998
+ return completionToResponse(completion, endpointKey);
1953
1999
  } catch (err) {
1954
2000
  throw toError2(err, providerName);
1955
2001
  }
@@ -1984,11 +2030,12 @@ async function* runStream2(options) {
1984
2030
  finishReason = choice.finish_reason;
1985
2031
  }
1986
2032
  const delta = choice.delta;
1987
- const reasoningContent = delta.reasoning_content;
1988
- if (typeof reasoningContent === "string" && reasoningContent) {
1989
- thinkingAccum += reasoningContent;
2033
+ const reasoning = readReasoning(delta);
2034
+ if (reasoning) {
2035
+ rememberReasoningField(endpointKey, reasoning.field);
2036
+ thinkingAccum += reasoning.text;
1990
2037
  if (options.thinking) {
1991
- yield { type: "thinking_delta", text: reasoningContent };
2038
+ yield { type: "thinking_delta", text: reasoning.text };
1992
2039
  }
1993
2040
  }
1994
2041
  if (delta.content) {
@@ -2073,16 +2120,17 @@ async function* runStream2(options) {
2073
2120
  yield { type: "done", stopReason };
2074
2121
  return response;
2075
2122
  }
2076
- function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2123
+ function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
2077
2124
  const choice = completion.choices?.[0];
2078
2125
  if (!choice) {
2079
2126
  yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
2080
2127
  return;
2081
2128
  }
2082
2129
  const msg = choice.message;
2083
- const reasoning = msg.reasoning_content;
2084
- if (typeof reasoning === "string" && reasoning && thinkingEnabled) {
2085
- yield { type: "thinking_delta", text: reasoning };
2130
+ const reasoning = readReasoning(msg);
2131
+ if (reasoning) {
2132
+ rememberReasoningField(endpointKey, reasoning.field);
2133
+ if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
2086
2134
  }
2087
2135
  if (typeof msg.content === "string" && msg.content) {
2088
2136
  yield { type: "text_delta", text: msg.content };
@@ -2110,15 +2158,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2110
2158
  }
2111
2159
  yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
2112
2160
  }
2113
- function completionToResponse(completion) {
2161
+ function completionToResponse(completion, endpointKey) {
2114
2162
  const choice = completion.choices?.[0];
2115
2163
  const contentParts = [];
2116
2164
  let textAccum = "";
2117
2165
  if (choice) {
2118
2166
  const msg = choice.message;
2119
- const reasoning = msg.reasoning_content;
2120
- if (typeof reasoning === "string" && reasoning) {
2121
- contentParts.push({ type: "thinking", text: reasoning });
2167
+ const reasoning = readReasoning(msg);
2168
+ if (reasoning) {
2169
+ rememberReasoningField(endpointKey, reasoning.field);
2170
+ contentParts.push({ type: "thinking", text: reasoning.text });
2122
2171
  }
2123
2172
  if (typeof msg.content === "string" && msg.content) {
2124
2173
  textAccum = msg.content;
@@ -3479,6 +3528,28 @@ providerRegistry.register("minimax", {
3479
3528
  serverTools: void 0
3480
3529
  })
3481
3530
  });
3531
+ function localWireModelId(id) {
3532
+ const match = /^local\/[^/]+\/(.+)$/.exec(id);
3533
+ return match?.[1] ?? id;
3534
+ }
3535
+ providerRegistry.register("local", {
3536
+ // Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
3537
+ // vLLM). There is no default endpoint: the baseUrl comes from the endpoint
3538
+ // credential the discovery layer wrote, so a missing one is a wiring bug, not
3539
+ // something to paper over with a guess at someone else's port.
3540
+ stream: (options) => {
3541
+ if (!options.baseUrl) {
3542
+ throw new EZCoderAIError(
3543
+ "Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
3544
+ );
3545
+ }
3546
+ return streamOpenAI({
3547
+ ...options,
3548
+ model: localWireModelId(options.model),
3549
+ webSearch: false
3550
+ });
3551
+ }
3552
+ });
3482
3553
  function stream(options) {
3483
3554
  const entry = providerRegistry.get(options.provider);
3484
3555
  if (!entry) {
@@ -3882,6 +3953,7 @@ export {
3882
3953
  formatErrorForDisplay,
3883
3954
  isHardBillingMessage,
3884
3955
  isUsageLimitError,
3956
+ localWireModelId,
3885
3957
  palsuAssistantMessage,
3886
3958
  palsuText,
3887
3959
  palsuThinking,