@kenkaiiii/gg-ai 5.23.3 → 5.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +91 -18
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +14 -2
- package/dist/index.d.ts +14 -2
- package/dist/index.js +90 -18
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.cts
CHANGED
|
@@ -2,7 +2,9 @@ import { z } from 'zod';
|
|
|
2
2
|
import Anthropic from '@anthropic-ai/sdk';
|
|
3
3
|
import OpenAI from 'openai';
|
|
4
4
|
|
|
5
|
-
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
5
|
+
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
6
|
+
/** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
|
|
7
|
+
| "local";
|
|
6
8
|
type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
|
|
7
9
|
type CacheRetention = "none" | "short" | "long";
|
|
8
10
|
interface TextContent {
|
|
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
|
|
|
293
295
|
then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
|
|
294
296
|
}
|
|
295
297
|
|
|
298
|
+
/**
|
|
299
|
+
* Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
|
|
300
|
+
* the same model name served by two machines stays distinct in the registry.
|
|
301
|
+
* The server only knows the raw id, so strip the routing prefix here — at the
|
|
302
|
+
* one place that talks to the wire. Counterpart to gg-core's
|
|
303
|
+
* `formatLocalModelId`/`parseLocalModelId`.
|
|
304
|
+
*/
|
|
305
|
+
declare function localWireModelId(id: string): string;
|
|
296
306
|
/**
|
|
297
307
|
* Unified streaming entry point. Returns a StreamResult that is both
|
|
298
308
|
* an async iterable (for streaming events) and thenable (await for
|
|
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
|
|
|
507
517
|
provider?: string;
|
|
508
518
|
thinking?: boolean;
|
|
509
519
|
supportsImages?: boolean;
|
|
520
|
+
/** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
|
|
521
|
+
reasoningField?: string;
|
|
510
522
|
}): OpenAI.ChatCompletionMessageParam[];
|
|
511
523
|
|
|
512
524
|
/**
|
|
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
|
|
|
600
612
|
*/
|
|
601
613
|
declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
|
|
602
614
|
|
|
603
|
-
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
|
615
|
+
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
package/dist/index.d.ts
CHANGED
|
@@ -2,7 +2,9 @@ import { z } from 'zod';
|
|
|
2
2
|
import Anthropic from '@anthropic-ai/sdk';
|
|
3
3
|
import OpenAI from 'openai';
|
|
4
4
|
|
|
5
|
-
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
5
|
+
type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
|
|
6
|
+
/** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
|
|
7
|
+
| "local";
|
|
6
8
|
type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
|
|
7
9
|
type CacheRetention = "none" | "short" | "long";
|
|
8
10
|
interface TextContent {
|
|
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
|
|
|
293
295
|
then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
|
|
294
296
|
}
|
|
295
297
|
|
|
298
|
+
/**
|
|
299
|
+
* Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
|
|
300
|
+
* the same model name served by two machines stays distinct in the registry.
|
|
301
|
+
* The server only knows the raw id, so strip the routing prefix here — at the
|
|
302
|
+
* one place that talks to the wire. Counterpart to gg-core's
|
|
303
|
+
* `formatLocalModelId`/`parseLocalModelId`.
|
|
304
|
+
*/
|
|
305
|
+
declare function localWireModelId(id: string): string;
|
|
296
306
|
/**
|
|
297
307
|
* Unified streaming entry point. Returns a StreamResult that is both
|
|
298
308
|
* an async iterable (for streaming events) and thenable (await for
|
|
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
|
|
|
507
517
|
provider?: string;
|
|
508
518
|
thinking?: boolean;
|
|
509
519
|
supportsImages?: boolean;
|
|
520
|
+
/** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
|
|
521
|
+
reasoningField?: string;
|
|
510
522
|
}): OpenAI.ChatCompletionMessageParam[];
|
|
511
523
|
|
|
512
524
|
/**
|
|
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
|
|
|
600
612
|
*/
|
|
601
613
|
declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
|
|
602
614
|
|
|
603
|
-
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
|
615
|
+
export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
|
package/dist/index.js
CHANGED
|
@@ -501,6 +501,39 @@ function normalizeRootForAnthropic(schema) {
|
|
|
501
501
|
return out;
|
|
502
502
|
}
|
|
503
503
|
|
|
504
|
+
// src/providers/reasoning-field.ts
|
|
505
|
+
var REASONING_FIELD_ALIASES = [
|
|
506
|
+
"reasoning_content",
|
|
507
|
+
"reasoning",
|
|
508
|
+
"reasoning_text"
|
|
509
|
+
];
|
|
510
|
+
var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
|
|
511
|
+
function readReasoning(obj) {
|
|
512
|
+
if (!obj) return void 0;
|
|
513
|
+
for (const field of REASONING_FIELD_ALIASES) {
|
|
514
|
+
const value = obj[field];
|
|
515
|
+
if (typeof value === "string" && value) return { field, text: value };
|
|
516
|
+
}
|
|
517
|
+
return void 0;
|
|
518
|
+
}
|
|
519
|
+
function reasoningFieldKey(provider, baseUrl, model) {
|
|
520
|
+
return `${provider}|${baseUrl ?? ""}|${model}`;
|
|
521
|
+
}
|
|
522
|
+
var MAX_REMEMBERED_ENDPOINTS = 64;
|
|
523
|
+
var detectedFields = /* @__PURE__ */ new Map();
|
|
524
|
+
function rememberReasoningField(key, field) {
|
|
525
|
+
if (detectedFields.get(key) === field) return;
|
|
526
|
+
detectedFields.set(key, field);
|
|
527
|
+
while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
|
|
528
|
+
const oldest = detectedFields.keys().next();
|
|
529
|
+
if (oldest.done) break;
|
|
530
|
+
detectedFields.delete(oldest.value);
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
function getReasoningField(key) {
|
|
534
|
+
return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
|
|
535
|
+
}
|
|
536
|
+
|
|
504
537
|
// src/providers/transform.ts
|
|
505
538
|
function hasValidThinkingSignature(part) {
|
|
506
539
|
return typeof part.signature === "string" && part.signature.trim().length > 0;
|
|
@@ -921,6 +954,7 @@ function remapToolCallId(id, idMap) {
|
|
|
921
954
|
return mapped;
|
|
922
955
|
}
|
|
923
956
|
function toOpenAIMessages(messages, options) {
|
|
957
|
+
const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
|
|
924
958
|
const out = [];
|
|
925
959
|
const idMap = /* @__PURE__ */ new Map();
|
|
926
960
|
const mergeToolResultText = options?.provider === "glm";
|
|
@@ -987,9 +1021,9 @@ function toOpenAIMessages(messages, options) {
|
|
|
987
1021
|
...hasToolCalls ? { tool_calls: toolCalls } : {}
|
|
988
1022
|
};
|
|
989
1023
|
if (thinkingParts) {
|
|
990
|
-
assistantMsg
|
|
1024
|
+
assistantMsg[reasoningField] = thinkingParts;
|
|
991
1025
|
} else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
|
|
992
|
-
assistantMsg
|
|
1026
|
+
assistantMsg[reasoningField] = " ";
|
|
993
1027
|
}
|
|
994
1028
|
out.push(assistantMsg);
|
|
995
1029
|
continue;
|
|
@@ -1067,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
|
|
|
1067
1101
|
if (choice === "required") return "required";
|
|
1068
1102
|
return { type: "function", function: { name: choice.name } };
|
|
1069
1103
|
}
|
|
1104
|
+
function toLocalReasoningEffort(level) {
|
|
1105
|
+
if (level === "max" || level === "ultra" || level === "xhigh") return "max";
|
|
1106
|
+
return level;
|
|
1107
|
+
}
|
|
1070
1108
|
function toOpenAIReasoningEffort(level, model) {
|
|
1071
1109
|
const effort = level === "max" || level === "ultra" ? "xhigh" : level;
|
|
1072
1110
|
if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
|
|
@@ -1862,7 +1900,9 @@ function streamOpenAI(options) {
|
|
|
1862
1900
|
async function* runStream2(options) {
|
|
1863
1901
|
const providerName = options.provider ?? "openai";
|
|
1864
1902
|
const useStreaming = options.streaming !== false;
|
|
1903
|
+
const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
|
|
1865
1904
|
const client = createClient2(options);
|
|
1905
|
+
const isLocal = options.provider === "local";
|
|
1866
1906
|
const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
|
|
1867
1907
|
const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
|
|
1868
1908
|
const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
|
|
@@ -1885,7 +1925,8 @@ async function* runStream2(options) {
|
|
|
1885
1925
|
// disabled K3 must NOT carry placeholder reasoning_content (mirrors the
|
|
1886
1926
|
// official CLI: reasoning is preserved only while thinking is enabled).
|
|
1887
1927
|
thinking: isKimiK27 || !!options.thinking,
|
|
1888
|
-
supportsImages: options.supportsImages
|
|
1928
|
+
supportsImages: options.supportsImages,
|
|
1929
|
+
reasoningField: getReasoningField(endpointKey)
|
|
1889
1930
|
});
|
|
1890
1931
|
const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
|
|
1891
1932
|
const effectiveTemp = options.temperature ?? defaultTemp;
|
|
@@ -1897,7 +1938,7 @@ async function* runStream2(options) {
|
|
|
1897
1938
|
...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
|
|
1898
1939
|
...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
|
|
1899
1940
|
...options.stop ? { stop: options.stop } : {},
|
|
1900
|
-
...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
|
|
1941
|
+
...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
|
|
1901
1942
|
...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
|
|
1902
1943
|
...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
|
|
1903
1944
|
...useStreaming ? { stream_options: { include_usage: true } } : {}
|
|
@@ -1911,6 +1952,11 @@ async function* runStream2(options) {
|
|
|
1911
1952
|
paramsAny.prompt_cache_retention = "24h";
|
|
1912
1953
|
}
|
|
1913
1954
|
}
|
|
1955
|
+
if (isLocal && options.thinking) {
|
|
1956
|
+
params.reasoning_effort = toLocalReasoningEffort(
|
|
1957
|
+
options.thinking
|
|
1958
|
+
);
|
|
1959
|
+
}
|
|
1914
1960
|
if (options.provider === "openai" && options.serviceTier) {
|
|
1915
1961
|
params.service_tier = options.serviceTier;
|
|
1916
1962
|
}
|
|
@@ -1947,8 +1993,8 @@ async function* runStream2(options) {
|
|
|
1947
1993
|
const completion = await client.chat.completions.create(params, {
|
|
1948
1994
|
signal: options.signal ?? void 0
|
|
1949
1995
|
});
|
|
1950
|
-
yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
|
|
1951
|
-
return completionToResponse(completion);
|
|
1996
|
+
yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
|
|
1997
|
+
return completionToResponse(completion, endpointKey);
|
|
1952
1998
|
} catch (err) {
|
|
1953
1999
|
throw toError2(err, providerName);
|
|
1954
2000
|
}
|
|
@@ -1983,11 +2029,12 @@ async function* runStream2(options) {
|
|
|
1983
2029
|
finishReason = choice.finish_reason;
|
|
1984
2030
|
}
|
|
1985
2031
|
const delta = choice.delta;
|
|
1986
|
-
const
|
|
1987
|
-
if (
|
|
1988
|
-
|
|
2032
|
+
const reasoning = readReasoning(delta);
|
|
2033
|
+
if (reasoning) {
|
|
2034
|
+
rememberReasoningField(endpointKey, reasoning.field);
|
|
2035
|
+
thinkingAccum += reasoning.text;
|
|
1989
2036
|
if (options.thinking) {
|
|
1990
|
-
yield { type: "thinking_delta", text:
|
|
2037
|
+
yield { type: "thinking_delta", text: reasoning.text };
|
|
1991
2038
|
}
|
|
1992
2039
|
}
|
|
1993
2040
|
if (delta.content) {
|
|
@@ -2072,16 +2119,17 @@ async function* runStream2(options) {
|
|
|
2072
2119
|
yield { type: "done", stopReason };
|
|
2073
2120
|
return response;
|
|
2074
2121
|
}
|
|
2075
|
-
function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
|
|
2122
|
+
function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
|
|
2076
2123
|
const choice = completion.choices?.[0];
|
|
2077
2124
|
if (!choice) {
|
|
2078
2125
|
yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
|
|
2079
2126
|
return;
|
|
2080
2127
|
}
|
|
2081
2128
|
const msg = choice.message;
|
|
2082
|
-
const reasoning = msg
|
|
2083
|
-
if (
|
|
2084
|
-
|
|
2129
|
+
const reasoning = readReasoning(msg);
|
|
2130
|
+
if (reasoning) {
|
|
2131
|
+
rememberReasoningField(endpointKey, reasoning.field);
|
|
2132
|
+
if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
|
|
2085
2133
|
}
|
|
2086
2134
|
if (typeof msg.content === "string" && msg.content) {
|
|
2087
2135
|
yield { type: "text_delta", text: msg.content };
|
|
@@ -2109,15 +2157,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
|
|
|
2109
2157
|
}
|
|
2110
2158
|
yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
|
|
2111
2159
|
}
|
|
2112
|
-
function completionToResponse(completion) {
|
|
2160
|
+
function completionToResponse(completion, endpointKey) {
|
|
2113
2161
|
const choice = completion.choices?.[0];
|
|
2114
2162
|
const contentParts = [];
|
|
2115
2163
|
let textAccum = "";
|
|
2116
2164
|
if (choice) {
|
|
2117
2165
|
const msg = choice.message;
|
|
2118
|
-
const reasoning = msg
|
|
2119
|
-
if (
|
|
2120
|
-
|
|
2166
|
+
const reasoning = readReasoning(msg);
|
|
2167
|
+
if (reasoning) {
|
|
2168
|
+
rememberReasoningField(endpointKey, reasoning.field);
|
|
2169
|
+
contentParts.push({ type: "thinking", text: reasoning.text });
|
|
2121
2170
|
}
|
|
2122
2171
|
if (typeof msg.content === "string" && msg.content) {
|
|
2123
2172
|
textAccum = msg.content;
|
|
@@ -3469,6 +3518,28 @@ providerRegistry.register("minimax", {
|
|
|
3469
3518
|
serverTools: void 0
|
|
3470
3519
|
})
|
|
3471
3520
|
});
|
|
3521
|
+
function localWireModelId(id) {
|
|
3522
|
+
const match = /^local\/[^/]+\/(.+)$/.exec(id);
|
|
3523
|
+
return match?.[1] ?? id;
|
|
3524
|
+
}
|
|
3525
|
+
providerRegistry.register("local", {
|
|
3526
|
+
// Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
|
|
3527
|
+
// vLLM). There is no default endpoint: the baseUrl comes from the endpoint
|
|
3528
|
+
// credential the discovery layer wrote, so a missing one is a wiring bug, not
|
|
3529
|
+
// something to paper over with a guess at someone else's port.
|
|
3530
|
+
stream: (options) => {
|
|
3531
|
+
if (!options.baseUrl) {
|
|
3532
|
+
throw new GGAIError(
|
|
3533
|
+
"Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
|
|
3534
|
+
);
|
|
3535
|
+
}
|
|
3536
|
+
return streamOpenAI({
|
|
3537
|
+
...options,
|
|
3538
|
+
model: localWireModelId(options.model),
|
|
3539
|
+
webSearch: false
|
|
3540
|
+
});
|
|
3541
|
+
}
|
|
3542
|
+
});
|
|
3472
3543
|
function stream(options) {
|
|
3473
3544
|
const entry = providerRegistry.get(options.provider);
|
|
3474
3545
|
if (!entry) {
|
|
@@ -3872,6 +3943,7 @@ export {
|
|
|
3872
3943
|
formatErrorForDisplay,
|
|
3873
3944
|
isHardBillingMessage,
|
|
3874
3945
|
isUsageLimitError,
|
|
3946
|
+
localWireModelId,
|
|
3875
3947
|
palsuAssistantMessage,
|
|
3876
3948
|
palsuText,
|
|
3877
3949
|
palsuThinking,
|