@prestyj/ai 5.9.0 → 5.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.cjs +93 -20
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +14 -2
- package/dist/index.d.ts +14 -2
- package/dist/index.js +92 -20
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.cts
CHANGED
|
@@ -2,7 +2,9 @@ import { z } from 'zod';
|
|
|
2
2
|
import Anthropic from '@anthropic-ai/sdk';
|
|
3
3
|
import OpenAI from 'openai';
|
|
4
4
|
|
|
5
|
-
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
5
|
+
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
6
|
+
/** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
|
|
7
|
+
| "local";
|
|
6
8
|
type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
|
|
7
9
|
type CacheRetention = "none" | "short" | "long";
|
|
8
10
|
interface TextContent {
|
|
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
|
|
|
293
295
|
then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
|
|
294
296
|
}
|
|
295
297
|
|
|
298
|
+
/**
|
|
299
|
+
* Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
|
|
300
|
+
* the same model name served by two machines stays distinct in the registry.
|
|
301
|
+
* The server only knows the raw id, so strip the routing prefix here — at the
|
|
302
|
+
* one place that talks to the wire. Counterpart to gg-core's
|
|
303
|
+
* `formatLocalModelId`/`parseLocalModelId`.
|
|
304
|
+
*/
|
|
305
|
+
declare function localWireModelId(id: string): string;
|
|
296
306
|
/**
|
|
297
307
|
* Unified streaming entry point. Returns a StreamResult that is both
|
|
298
308
|
* an async iterable (for streaming events) and thenable (await for
|
|
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
|
|
|
507
517
|
provider?: string;
|
|
508
518
|
thinking?: boolean;
|
|
509
519
|
supportsImages?: boolean;
|
|
520
|
+
/** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
|
|
521
|
+
reasoningField?: string;
|
|
510
522
|
}): OpenAI.ChatCompletionMessageParam[];
|
|
511
523
|
|
|
512
524
|
/**
|
|
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
|
|
|
600
612
|
*/
|
|
601
613
|
declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
|
|
602
614
|
|
|
603
|
-
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
|
615
|
+
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
package/dist/index.d.ts
CHANGED
|
@@ -2,7 +2,9 @@ import { z } from 'zod';
|
|
|
2
2
|
import Anthropic from '@anthropic-ai/sdk';
|
|
3
3
|
import OpenAI from 'openai';
|
|
4
4
|
|
|
5
|
-
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
5
|
+
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
6
|
+
/** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
|
|
7
|
+
| "local";
|
|
6
8
|
type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
|
|
7
9
|
type CacheRetention = "none" | "short" | "long";
|
|
8
10
|
interface TextContent {
|
|
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
|
|
|
293
295
|
then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
|
|
294
296
|
}
|
|
295
297
|
|
|
298
|
+
/**
|
|
299
|
+
* Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
|
|
300
|
+
* the same model name served by two machines stays distinct in the registry.
|
|
301
|
+
* The server only knows the raw id, so strip the routing prefix here — at the
|
|
302
|
+
* one place that talks to the wire. Counterpart to gg-core's
|
|
303
|
+
* `formatLocalModelId`/`parseLocalModelId`.
|
|
304
|
+
*/
|
|
305
|
+
declare function localWireModelId(id: string): string;
|
|
296
306
|
/**
|
|
297
307
|
* Unified streaming entry point. Returns a StreamResult that is both
|
|
298
308
|
* an async iterable (for streaming events) and thenable (await for
|
|
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
|
|
|
507
517
|
provider?: string;
|
|
508
518
|
thinking?: boolean;
|
|
509
519
|
supportsImages?: boolean;
|
|
520
|
+
/** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
|
|
521
|
+
reasoningField?: string;
|
|
510
522
|
}): OpenAI.ChatCompletionMessageParam[];
|
|
511
523
|
|
|
512
524
|
/**
|
|
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
|
|
|
600
612
|
*/
|
|
601
613
|
declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
|
|
602
614
|
|
|
603
|
-
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
|
615
|
+
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, EZCoderAIError, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
package/dist/index.js
CHANGED
|
@@ -501,6 +501,39 @@ function normalizeRootForAnthropic(schema) {
|
|
|
501
501
|
return out;
|
|
502
502
|
}
|
|
503
503
|
|
|
504
|
+
// src/providers/reasoning-field.ts
|
|
505
|
+
var REASONING_FIELD_ALIASES = [
|
|
506
|
+
"reasoning_content",
|
|
507
|
+
"reasoning",
|
|
508
|
+
"reasoning_text"
|
|
509
|
+
];
|
|
510
|
+
var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
|
|
511
|
+
function readReasoning(obj) {
|
|
512
|
+
if (!obj) return void 0;
|
|
513
|
+
for (const field of REASONING_FIELD_ALIASES) {
|
|
514
|
+
const value = obj[field];
|
|
515
|
+
if (typeof value === "string" && value) return { field, text: value };
|
|
516
|
+
}
|
|
517
|
+
return void 0;
|
|
518
|
+
}
|
|
519
|
+
function reasoningFieldKey(provider, baseUrl, model) {
|
|
520
|
+
return `${provider}|${baseUrl ?? ""}|${model}`;
|
|
521
|
+
}
|
|
522
|
+
var MAX_REMEMBERED_ENDPOINTS = 64;
|
|
523
|
+
var detectedFields = /* @__PURE__ */ new Map();
|
|
524
|
+
function rememberReasoningField(key, field) {
|
|
525
|
+
if (detectedFields.get(key) === field) return;
|
|
526
|
+
detectedFields.set(key, field);
|
|
527
|
+
while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
|
|
528
|
+
const oldest = detectedFields.keys().next();
|
|
529
|
+
if (oldest.done) break;
|
|
530
|
+
detectedFields.delete(oldest.value);
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
function getReasoningField(key) {
|
|
534
|
+
return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
|
|
535
|
+
}
|
|
536
|
+
|
|
504
537
|
// src/providers/transform.ts
|
|
505
538
|
function hasValidThinkingSignature(part) {
|
|
506
539
|
return typeof part.signature === "string" && part.signature.trim().length > 0;
|
|
@@ -885,12 +918,12 @@ function toAnthropicToolChoice(choice) {
|
|
|
885
918
|
return { type: "tool", name: choice.name };
|
|
886
919
|
}
|
|
887
920
|
function isAdaptiveThinkingModel(model) {
|
|
888
|
-
return /opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
|
|
921
|
+
return /opus-5|opus-4[-.]8|opus-4[-.]7|opus-4[-.]6|sonnet-5|fable-5|mythos-5/.test(model);
|
|
889
922
|
}
|
|
890
923
|
function toAnthropicThinking(level, maxTokens, model) {
|
|
891
924
|
if (isAdaptiveThinkingModel(model)) {
|
|
892
925
|
let effort = level;
|
|
893
|
-
if (effort === "xhigh" && !/opus-4-8|opus-4-7/.test(model)) {
|
|
926
|
+
if (effort === "xhigh" && !/opus-5|opus-4-8|opus-4-7/.test(model)) {
|
|
894
927
|
effort = "high";
|
|
895
928
|
}
|
|
896
929
|
return {
|
|
@@ -921,6 +954,7 @@ function remapToolCallId(id, idMap) {
|
|
|
921
954
|
return mapped;
|
|
922
955
|
}
|
|
923
956
|
function toOpenAIMessages(messages, options) {
|
|
957
|
+
const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
|
|
924
958
|
const out = [];
|
|
925
959
|
const idMap = /* @__PURE__ */ new Map();
|
|
926
960
|
const mergeToolResultText = options?.provider === "glm";
|
|
@@ -987,9 +1021,9 @@ function toOpenAIMessages(messages, options) {
|
|
|
987
1021
|
...hasToolCalls ? { tool_calls: toolCalls } : {}
|
|
988
1022
|
};
|
|
989
1023
|
if (thinkingParts) {
|
|
990
|
-
assistantMsg
|
|
1024
|
+
assistantMsg[reasoningField] = thinkingParts;
|
|
991
1025
|
} else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
|
|
992
|
-
assistantMsg
|
|
1026
|
+
assistantMsg[reasoningField] = " ";
|
|
993
1027
|
}
|
|
994
1028
|
out.push(assistantMsg);
|
|
995
1029
|
continue;
|
|
@@ -1067,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
|
|
|
1067
1101
|
if (choice === "required") return "required";
|
|
1068
1102
|
return { type: "function", function: { name: choice.name } };
|
|
1069
1103
|
}
|
|
1104
|
+
function toLocalReasoningEffort(level) {
|
|
1105
|
+
if (level === "max" || level === "ultra" || level === "xhigh") return "max";
|
|
1106
|
+
return level;
|
|
1107
|
+
}
|
|
1070
1108
|
function toOpenAIReasoningEffort(level, model) {
|
|
1071
1109
|
const effort = level === "max" || level === "ultra" ? "xhigh" : level;
|
|
1072
1110
|
if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
|
|
@@ -1863,7 +1901,9 @@ function streamOpenAI(options) {
|
|
|
1863
1901
|
async function* runStream2(options) {
|
|
1864
1902
|
const providerName = options.provider ?? "openai";
|
|
1865
1903
|
const useStreaming = options.streaming !== false;
|
|
1904
|
+
const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
|
|
1866
1905
|
const client = createClient2(options);
|
|
1906
|
+
const isLocal = options.provider === "local";
|
|
1867
1907
|
const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
|
|
1868
1908
|
const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
|
|
1869
1909
|
const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
|
|
@@ -1886,7 +1926,8 @@ async function* runStream2(options) {
|
|
|
1886
1926
|
// disabled K3 must NOT carry placeholder reasoning_content (mirrors the
|
|
1887
1927
|
// official CLI: reasoning is preserved only while thinking is enabled).
|
|
1888
1928
|
thinking: isKimiK27 || !!options.thinking,
|
|
1889
|
-
supportsImages: options.supportsImages
|
|
1929
|
+
supportsImages: options.supportsImages,
|
|
1930
|
+
reasoningField: getReasoningField(endpointKey)
|
|
1890
1931
|
});
|
|
1891
1932
|
const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
|
|
1892
1933
|
const effectiveTemp = options.temperature ?? defaultTemp;
|
|
@@ -1898,7 +1939,7 @@ async function* runStream2(options) {
|
|
|
1898
1939
|
...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
|
|
1899
1940
|
...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
|
|
1900
1941
|
...options.stop ? { stop: options.stop } : {},
|
|
1901
|
-
...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
|
|
1942
|
+
...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
|
|
1902
1943
|
...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
|
|
1903
1944
|
...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
|
|
1904
1945
|
...useStreaming ? { stream_options: { include_usage: true } } : {}
|
|
@@ -1912,6 +1953,11 @@ async function* runStream2(options) {
|
|
|
1912
1953
|
paramsAny.prompt_cache_retention = "24h";
|
|
1913
1954
|
}
|
|
1914
1955
|
}
|
|
1956
|
+
if (isLocal && options.thinking) {
|
|
1957
|
+
params.reasoning_effort = toLocalReasoningEffort(
|
|
1958
|
+
options.thinking
|
|
1959
|
+
);
|
|
1960
|
+
}
|
|
1915
1961
|
if (options.provider === "openai" && options.serviceTier) {
|
|
1916
1962
|
params.service_tier = options.serviceTier;
|
|
1917
1963
|
}
|
|
@@ -1948,8 +1994,8 @@ async function* runStream2(options) {
|
|
|
1948
1994
|
const completion = await client.chat.completions.create(params, {
|
|
1949
1995
|
signal: options.signal ?? void 0
|
|
1950
1996
|
});
|
|
1951
|
-
yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
|
|
1952
|
-
return completionToResponse(completion);
|
|
1997
|
+
yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
|
|
1998
|
+
return completionToResponse(completion, endpointKey);
|
|
1953
1999
|
} catch (err) {
|
|
1954
2000
|
throw toError2(err, providerName);
|
|
1955
2001
|
}
|
|
@@ -1984,11 +2030,12 @@ async function* runStream2(options) {
|
|
|
1984
2030
|
finishReason = choice.finish_reason;
|
|
1985
2031
|
}
|
|
1986
2032
|
const delta = choice.delta;
|
|
1987
|
-
const
|
|
1988
|
-
if (
|
|
1989
|
-
|
|
2033
|
+
const reasoning = readReasoning(delta);
|
|
2034
|
+
if (reasoning) {
|
|
2035
|
+
rememberReasoningField(endpointKey, reasoning.field);
|
|
2036
|
+
thinkingAccum += reasoning.text;
|
|
1990
2037
|
if (options.thinking) {
|
|
1991
|
-
yield { type: "thinking_delta", text:
|
|
2038
|
+
yield { type: "thinking_delta", text: reasoning.text };
|
|
1992
2039
|
}
|
|
1993
2040
|
}
|
|
1994
2041
|
if (delta.content) {
|
|
@@ -2073,16 +2120,17 @@ async function* runStream2(options) {
|
|
|
2073
2120
|
yield { type: "done", stopReason };
|
|
2074
2121
|
return response;
|
|
2075
2122
|
}
|
|
2076
|
-
function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
|
|
2123
|
+
function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
|
|
2077
2124
|
const choice = completion.choices?.[0];
|
|
2078
2125
|
if (!choice) {
|
|
2079
2126
|
yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
|
|
2080
2127
|
return;
|
|
2081
2128
|
}
|
|
2082
2129
|
const msg = choice.message;
|
|
2083
|
-
const reasoning = msg
|
|
2084
|
-
if (
|
|
2085
|
-
|
|
2130
|
+
const reasoning = readReasoning(msg);
|
|
2131
|
+
if (reasoning) {
|
|
2132
|
+
rememberReasoningField(endpointKey, reasoning.field);
|
|
2133
|
+
if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
|
|
2086
2134
|
}
|
|
2087
2135
|
if (typeof msg.content === "string" && msg.content) {
|
|
2088
2136
|
yield { type: "text_delta", text: msg.content };
|
|
@@ -2110,15 +2158,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
|
|
|
2110
2158
|
}
|
|
2111
2159
|
yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
|
|
2112
2160
|
}
|
|
2113
|
-
function completionToResponse(completion) {
|
|
2161
|
+
function completionToResponse(completion, endpointKey) {
|
|
2114
2162
|
const choice = completion.choices?.[0];
|
|
2115
2163
|
const contentParts = [];
|
|
2116
2164
|
let textAccum = "";
|
|
2117
2165
|
if (choice) {
|
|
2118
2166
|
const msg = choice.message;
|
|
2119
|
-
const reasoning = msg
|
|
2120
|
-
if (
|
|
2121
|
-
|
|
2167
|
+
const reasoning = readReasoning(msg);
|
|
2168
|
+
if (reasoning) {
|
|
2169
|
+
rememberReasoningField(endpointKey, reasoning.field);
|
|
2170
|
+
contentParts.push({ type: "thinking", text: reasoning.text });
|
|
2122
2171
|
}
|
|
2123
2172
|
if (typeof msg.content === "string" && msg.content) {
|
|
2124
2173
|
textAccum = msg.content;
|
|
@@ -3479,6 +3528,28 @@ providerRegistry.register("minimax", {
|
|
|
3479
3528
|
serverTools: void 0
|
|
3480
3529
|
})
|
|
3481
3530
|
});
|
|
3531
|
+
function localWireModelId(id) {
|
|
3532
|
+
const match = /^local\/[^/]+\/(.+)$/.exec(id);
|
|
3533
|
+
return match?.[1] ?? id;
|
|
3534
|
+
}
|
|
3535
|
+
providerRegistry.register("local", {
|
|
3536
|
+
// Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
|
|
3537
|
+
// vLLM). There is no default endpoint: the baseUrl comes from the endpoint
|
|
3538
|
+
// credential the discovery layer wrote, so a missing one is a wiring bug, not
|
|
3539
|
+
// something to paper over with a guess at someone else's port.
|
|
3540
|
+
stream: (options) => {
|
|
3541
|
+
if (!options.baseUrl) {
|
|
3542
|
+
throw new EZCoderAIError(
|
|
3543
|
+
"Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
|
|
3544
|
+
);
|
|
3545
|
+
}
|
|
3546
|
+
return streamOpenAI({
|
|
3547
|
+
...options,
|
|
3548
|
+
model: localWireModelId(options.model),
|
|
3549
|
+
webSearch: false
|
|
3550
|
+
});
|
|
3551
|
+
}
|
|
3552
|
+
});
|
|
3482
3553
|
function stream(options) {
|
|
3483
3554
|
const entry = providerRegistry.get(options.provider);
|
|
3484
3555
|
if (!entry) {
|
|
@@ -3882,6 +3953,7 @@ export {
|
|
|
3882
3953
|
formatErrorForDisplay,
|
|
3883
3954
|
isHardBillingMessage,
|
|
3884
3955
|
isUsageLimitError,
|
|
3956
|
+
localWireModelId,
|
|
3885
3957
|
palsuAssistantMessage,
|
|
3886
3958
|
palsuText,
|
|
3887
3959
|
palsuThinking,
|