@kenkaiiii/gg-ai 5.23.3 → 5.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
296
  }
295
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
296
306
  /**
297
307
  * Unified streaming entry point. Returns a StreamResult that is both
298
308
  * an async iterable (for streaming events) and thenable (await for
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
507
517
  provider?: string;
508
518
  thinking?: boolean;
509
519
  supportsImages?: boolean;
520
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
521
+ reasoningField?: string;
510
522
  }): OpenAI.ChatCompletionMessageParam[];
511
523
 
512
524
  /**
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
600
612
  */
601
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
602
614
 
603
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.d.ts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
296
  }
295
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
296
306
  /**
297
307
  * Unified streaming entry point. Returns a StreamResult that is both
298
308
  * an async iterable (for streaming events) and thenable (await for
@@ -507,6 +517,8 @@ declare function toOpenAIMessages(messages: Message[], options?: {
507
517
  provider?: string;
508
518
  thinking?: boolean;
509
519
  supportsImages?: boolean;
520
+ /** Wire name for reasoning on assistant messages. Defaults to `reasoning_content`. */
521
+ reasoningField?: string;
510
522
  }): OpenAI.ChatCompletionMessageParam[];
511
523
 
512
524
  /**
@@ -600,4 +612,4 @@ interface PalsuProviderConfig {
600
612
  */
601
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
602
614
 
603
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.js CHANGED
@@ -501,6 +501,39 @@ function normalizeRootForAnthropic(schema) {
501
501
  return out;
502
502
  }
503
503
 
504
+ // src/providers/reasoning-field.ts
505
+ var REASONING_FIELD_ALIASES = [
506
+ "reasoning_content",
507
+ "reasoning",
508
+ "reasoning_text"
509
+ ];
510
+ var DEFAULT_REASONING_FIELD = REASONING_FIELD_ALIASES[0];
511
+ function readReasoning(obj) {
512
+ if (!obj) return void 0;
513
+ for (const field of REASONING_FIELD_ALIASES) {
514
+ const value = obj[field];
515
+ if (typeof value === "string" && value) return { field, text: value };
516
+ }
517
+ return void 0;
518
+ }
519
+ function reasoningFieldKey(provider, baseUrl, model) {
520
+ return `${provider}|${baseUrl ?? ""}|${model}`;
521
+ }
522
+ var MAX_REMEMBERED_ENDPOINTS = 64;
523
+ var detectedFields = /* @__PURE__ */ new Map();
524
+ function rememberReasoningField(key, field) {
525
+ if (detectedFields.get(key) === field) return;
526
+ detectedFields.set(key, field);
527
+ while (detectedFields.size > MAX_REMEMBERED_ENDPOINTS) {
528
+ const oldest = detectedFields.keys().next();
529
+ if (oldest.done) break;
530
+ detectedFields.delete(oldest.value);
531
+ }
532
+ }
533
+ function getReasoningField(key) {
534
+ return detectedFields.get(key) ?? DEFAULT_REASONING_FIELD;
535
+ }
536
+
504
537
  // src/providers/transform.ts
505
538
  function hasValidThinkingSignature(part) {
506
539
  return typeof part.signature === "string" && part.signature.trim().length > 0;
@@ -921,6 +954,7 @@ function remapToolCallId(id, idMap) {
921
954
  return mapped;
922
955
  }
923
956
  function toOpenAIMessages(messages, options) {
957
+ const reasoningField = options?.reasoningField || DEFAULT_REASONING_FIELD;
924
958
  const out = [];
925
959
  const idMap = /* @__PURE__ */ new Map();
926
960
  const mergeToolResultText = options?.provider === "glm";
@@ -987,9 +1021,9 @@ function toOpenAIMessages(messages, options) {
987
1021
  ...hasToolCalls ? { tool_calls: toolCalls } : {}
988
1022
  };
989
1023
  if (thinkingParts) {
990
- assistantMsg.reasoning_content = thinkingParts;
1024
+ assistantMsg[reasoningField] = thinkingParts;
991
1025
  } else if (options?.thinking && hasToolCalls && options.provider !== "glm") {
992
- assistantMsg.reasoning_content = " ";
1026
+ assistantMsg[reasoningField] = " ";
993
1027
  }
994
1028
  out.push(assistantMsg);
995
1029
  continue;
@@ -1067,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
1067
1101
  if (choice === "required") return "required";
1068
1102
  return { type: "function", function: { name: choice.name } };
1069
1103
  }
1104
+ function toLocalReasoningEffort(level) {
1105
+ if (level === "max" || level === "ultra" || level === "xhigh") return "max";
1106
+ return level;
1107
+ }
1070
1108
  function toOpenAIReasoningEffort(level, model) {
1071
1109
  const effort = level === "max" || level === "ultra" ? "xhigh" : level;
1072
1110
  if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
@@ -1862,7 +1900,9 @@ function streamOpenAI(options) {
1862
1900
  async function* runStream2(options) {
1863
1901
  const providerName = options.provider ?? "openai";
1864
1902
  const useStreaming = options.streaming !== false;
1903
+ const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
1865
1904
  const client = createClient2(options);
1905
+ const isLocal = options.provider === "local";
1866
1906
  const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1867
1907
  const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1868
1908
  const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
@@ -1885,7 +1925,8 @@ async function* runStream2(options) {
1885
1925
  // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1886
1926
  // official CLI: reasoning is preserved only while thinking is enabled).
1887
1927
  thinking: isKimiK27 || !!options.thinking,
1888
- supportsImages: options.supportsImages
1928
+ supportsImages: options.supportsImages,
1929
+ reasoningField: getReasoningField(endpointKey)
1889
1930
  });
1890
1931
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
1891
1932
  const effectiveTemp = options.temperature ?? defaultTemp;
@@ -1897,7 +1938,7 @@ async function* runStream2(options) {
1897
1938
  ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1898
1939
  ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1899
1940
  ...options.stop ? { stop: options.stop } : {},
1900
- ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1941
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1901
1942
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1902
1943
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1903
1944
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1911,6 +1952,11 @@ async function* runStream2(options) {
1911
1952
  paramsAny.prompt_cache_retention = "24h";
1912
1953
  }
1913
1954
  }
1955
+ if (isLocal && options.thinking) {
1956
+ params.reasoning_effort = toLocalReasoningEffort(
1957
+ options.thinking
1958
+ );
1959
+ }
1914
1960
  if (options.provider === "openai" && options.serviceTier) {
1915
1961
  params.service_tier = options.serviceTier;
1916
1962
  }
@@ -1947,8 +1993,8 @@ async function* runStream2(options) {
1947
1993
  const completion = await client.chat.completions.create(params, {
1948
1994
  signal: options.signal ?? void 0
1949
1995
  });
1950
- yield* synthesizeEventsFromCompletion(completion, !!options.thinking);
1951
- return completionToResponse(completion);
1996
+ yield* synthesizeEventsFromCompletion(completion, !!options.thinking, endpointKey);
1997
+ return completionToResponse(completion, endpointKey);
1952
1998
  } catch (err) {
1953
1999
  throw toError2(err, providerName);
1954
2000
  }
@@ -1983,11 +2029,12 @@ async function* runStream2(options) {
1983
2029
  finishReason = choice.finish_reason;
1984
2030
  }
1985
2031
  const delta = choice.delta;
1986
- const reasoningContent = delta.reasoning_content;
1987
- if (typeof reasoningContent === "string" && reasoningContent) {
1988
- thinkingAccum += reasoningContent;
2032
+ const reasoning = readReasoning(delta);
2033
+ if (reasoning) {
2034
+ rememberReasoningField(endpointKey, reasoning.field);
2035
+ thinkingAccum += reasoning.text;
1989
2036
  if (options.thinking) {
1990
- yield { type: "thinking_delta", text: reasoningContent };
2037
+ yield { type: "thinking_delta", text: reasoning.text };
1991
2038
  }
1992
2039
  }
1993
2040
  if (delta.content) {
@@ -2072,16 +2119,17 @@ async function* runStream2(options) {
2072
2119
  yield { type: "done", stopReason };
2073
2120
  return response;
2074
2121
  }
2075
- function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2122
+ function* synthesizeEventsFromCompletion(completion, thinkingEnabled, endpointKey) {
2076
2123
  const choice = completion.choices?.[0];
2077
2124
  if (!choice) {
2078
2125
  yield { type: "done", stopReason: normalizeOpenAIStopReason(null) };
2079
2126
  return;
2080
2127
  }
2081
2128
  const msg = choice.message;
2082
- const reasoning = msg.reasoning_content;
2083
- if (typeof reasoning === "string" && reasoning && thinkingEnabled) {
2084
- yield { type: "thinking_delta", text: reasoning };
2129
+ const reasoning = readReasoning(msg);
2130
+ if (reasoning) {
2131
+ rememberReasoningField(endpointKey, reasoning.field);
2132
+ if (thinkingEnabled) yield { type: "thinking_delta", text: reasoning.text };
2085
2133
  }
2086
2134
  if (typeof msg.content === "string" && msg.content) {
2087
2135
  yield { type: "text_delta", text: msg.content };
@@ -2109,15 +2157,16 @@ function* synthesizeEventsFromCompletion(completion, thinkingEnabled) {
2109
2157
  }
2110
2158
  yield { type: "done", stopReason: normalizeOpenAIStopReason(choice.finish_reason ?? null) };
2111
2159
  }
2112
- function completionToResponse(completion) {
2160
+ function completionToResponse(completion, endpointKey) {
2113
2161
  const choice = completion.choices?.[0];
2114
2162
  const contentParts = [];
2115
2163
  let textAccum = "";
2116
2164
  if (choice) {
2117
2165
  const msg = choice.message;
2118
- const reasoning = msg.reasoning_content;
2119
- if (typeof reasoning === "string" && reasoning) {
2120
- contentParts.push({ type: "thinking", text: reasoning });
2166
+ const reasoning = readReasoning(msg);
2167
+ if (reasoning) {
2168
+ rememberReasoningField(endpointKey, reasoning.field);
2169
+ contentParts.push({ type: "thinking", text: reasoning.text });
2121
2170
  }
2122
2171
  if (typeof msg.content === "string" && msg.content) {
2123
2172
  textAccum = msg.content;
@@ -3469,6 +3518,28 @@ providerRegistry.register("minimax", {
3469
3518
  serverTools: void 0
3470
3519
  })
3471
3520
  });
3521
+ function localWireModelId(id) {
3522
+ const match = /^local\/[^/]+\/(.+)$/.exec(id);
3523
+ return match?.[1] ?? id;
3524
+ }
3525
+ providerRegistry.register("local", {
3526
+ // Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
3527
+ // vLLM). There is no default endpoint: the baseUrl comes from the endpoint
3528
+ // credential the discovery layer wrote, so a missing one is a wiring bug, not
3529
+ // something to paper over with a guess at someone else's port.
3530
+ stream: (options) => {
3531
+ if (!options.baseUrl) {
3532
+ throw new GGAIError(
3533
+ "Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
3534
+ );
3535
+ }
3536
+ return streamOpenAI({
3537
+ ...options,
3538
+ model: localWireModelId(options.model),
3539
+ webSearch: false
3540
+ });
3541
+ }
3542
+ });
3472
3543
  function stream(options) {
3473
3544
  const entry = providerRegistry.get(options.provider);
3474
3545
  if (!entry) {
@@ -3872,6 +3943,7 @@ export {
3872
3943
  formatErrorForDisplay,
3873
3944
  isHardBillingMessage,
3874
3945
  isUsageLimitError,
3946
+ localWireModelId,
3875
3947
  palsuAssistantMessage,
3876
3948
  palsuText,
3877
3949
  palsuThinking,