@kenkaiiii/gg-ai 5.24.0 → 5.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
296
  }
295
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
296
306
  /**
297
307
  * Unified streaming entry point. Returns a StreamResult that is both
298
308
  * an async iterable (for streaming events) and thenable (await for
@@ -602,4 +612,4 @@ interface PalsuProviderConfig {
602
612
  */
603
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
604
614
 
605
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.d.ts CHANGED
@@ -2,7 +2,9 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
+ | "local";
6
8
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
9
  type CacheRetention = "none" | "short" | "long";
8
10
  interface TextContent {
@@ -293,6 +295,14 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
293
295
  then<TResult1 = StreamResponse, TResult2 = never>(onfulfilled?: ((value: StreamResponse) => TResult1 | PromiseLike<TResult1>) | null, onrejected?: ((reason: unknown) => TResult2 | PromiseLike<TResult2>) | null): Promise<TResult1 | TResult2>;
294
296
  }
295
297
 
298
+ /**
299
+ * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
300
+ * the same model name served by two machines stays distinct in the registry.
301
+ * The server only knows the raw id, so strip the routing prefix here — at the
302
+ * one place that talks to the wire. Counterpart to gg-core's
303
+ * `formatLocalModelId`/`parseLocalModelId`.
304
+ */
305
+ declare function localWireModelId(id: string): string;
296
306
  /**
297
307
  * Unified streaming entry point. Returns a StreamResult that is both
298
308
  * an async iterable (for streaming events) and thenable (await for
@@ -602,4 +612,4 @@ interface PalsuProviderConfig {
602
612
  */
603
613
  declare function registerPalsuProvider(config?: PalsuProviderConfig): PalsuProviderHandle;
604
614
 
605
- export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
615
+ export { type AssistantMessage, type CacheRetention, type ContentPart, type DoneEvent, type ErrorEvent, type ErrorSource, EventStream, type FormattedError, GGAIError, type ImageContent, type Message, type PalsuModelConfig, type PalsuModelHandle, type PalsuProviderConfig, type PalsuProviderHandle, type PalsuProviderState, type PalsuResponse, type PalsuResponseFactory, type Provider, type ProviderDiagnosticFn, type ProviderEntry, ProviderError, type ProviderStreamFn, REDACTED as REDACTION_MARKER, type RawContent, type RedactionOptions, type ServerToolCall, type ServerToolCallEvent, type ServerToolDefinition, type ServerToolResult, type ServerToolResultEvent, type StopReason, type StreamEvent, type StreamOptions, type StreamResponse, StreamResult, type SystemMessage, type TextContent, type TextDeltaEvent, type ThinkingContent, type ThinkingDeltaEvent, type ThinkingLevel, type Tool, type ToolCall, type ToolCallDeltaEvent, type ToolCallDoneEvent, type ToolChoice, type ToolResult, type ToolResultContent, type ToolResultMessage, type Usage, type UserMessage, type VideoContent, clampProviderContextImages, classifyProviderError, environmentSecrets, formatError, formatErrorForDisplay, isHardBillingMessage, isUsageLimitError, localWireModelId, palsuAssistantMessage, palsuText, palsuThinking, palsuToolCall, prewarmAnthropicCache, providerRegistry, redactText, redactValue, registerPalsuProvider, setProviderDiagnostic, stream, toAnthropicMessages, toOpenAIMessages };
package/dist/index.js CHANGED
@@ -1101,6 +1101,10 @@ function toOpenAIToolChoice(choice) {
1101
1101
  if (choice === "required") return "required";
1102
1102
  return { type: "function", function: { name: choice.name } };
1103
1103
  }
1104
+ function toLocalReasoningEffort(level) {
1105
+ if (level === "max" || level === "ultra" || level === "xhigh") return "max";
1106
+ return level;
1107
+ }
1104
1108
  function toOpenAIReasoningEffort(level, model) {
1105
1109
  const effort = level === "max" || level === "ultra" ? "xhigh" : level;
1106
1110
  if (model.startsWith("fugu") && (effort === "low" || effort === "medium")) {
@@ -1898,6 +1902,7 @@ async function* runStream2(options) {
1898
1902
  const useStreaming = options.streaming !== false;
1899
1903
  const endpointKey = reasoningFieldKey(providerName, options.baseUrl, options.model);
1900
1904
  const client = createClient2(options);
1905
+ const isLocal = options.provider === "local";
1901
1906
  const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1902
1907
  const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1903
1908
  const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
@@ -1933,7 +1938,7 @@ async function* runStream2(options) {
1933
1938
  ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1934
1939
  ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1935
1940
  ...options.stop ? { stop: options.stop } : {},
1936
- ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1941
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 && !isLocal ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1937
1942
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1938
1943
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1939
1944
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1947,6 +1952,11 @@ async function* runStream2(options) {
1947
1952
  paramsAny.prompt_cache_retention = "24h";
1948
1953
  }
1949
1954
  }
1955
+ if (isLocal && options.thinking) {
1956
+ params.reasoning_effort = toLocalReasoningEffort(
1957
+ options.thinking
1958
+ );
1959
+ }
1950
1960
  if (options.provider === "openai" && options.serviceTier) {
1951
1961
  params.service_tier = options.serviceTier;
1952
1962
  }
@@ -3508,6 +3518,28 @@ providerRegistry.register("minimax", {
3508
3518
  serverTools: void 0
3509
3519
  })
3510
3520
  });
3521
+ function localWireModelId(id) {
3522
+ const match = /^local\/[^/]+\/(.+)$/.exec(id);
3523
+ return match?.[1] ?? id;
3524
+ }
3525
+ providerRegistry.register("local", {
3526
+ // Locally hosted OpenAI-compatible servers (Ollama, LM Studio, llama.cpp,
3527
+ // vLLM). There is no default endpoint: the baseUrl comes from the endpoint
3528
+ // credential the discovery layer wrote, so a missing one is a wiring bug, not
3529
+ // something to paper over with a guess at someone else's port.
3530
+ stream: (options) => {
3531
+ if (!options.baseUrl) {
3532
+ throw new GGAIError(
3533
+ "Local provider requires a baseUrl (e.g. http://127.0.0.1:11434/v1). No local endpoint was resolved for this model \u2014 re-scan for local models."
3534
+ );
3535
+ }
3536
+ return streamOpenAI({
3537
+ ...options,
3538
+ model: localWireModelId(options.model),
3539
+ webSearch: false
3540
+ });
3541
+ }
3542
+ });
3511
3543
  function stream(options) {
3512
3544
  const entry = providerRegistry.get(options.provider);
3513
3545
  if (!entry) {
@@ -3911,6 +3943,7 @@ export {
3911
3943
  formatErrorForDisplay,
3912
3944
  isHardBillingMessage,
3913
3945
  isUsageLimitError,
3946
+ localWireModelId,
3914
3947
  palsuAssistantMessage,
3915
3948
  palsuText,
3916
3949
  palsuThinking,