@prestyj/ai 5.7.0 → 5.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,7 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
6
6
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
7
  type CacheRetention = "none" | "short" | "long";
8
8
  interface TextContent {
package/dist/index.d.ts CHANGED
@@ -2,7 +2,7 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
6
6
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
7
  type CacheRetention = "none" | "short" | "long";
8
8
  interface TextContent {
package/dist/index.js CHANGED
@@ -58,12 +58,14 @@ var PROVIDER_DISPLAY = {
58
58
  deepseek: "DeepSeek",
59
59
  openrouter: "OpenRouter",
60
60
  sakana: "Sakana",
61
+ xai: "xAI (Grok)",
61
62
  xiaomi: "Xiaomi (MiMo)",
62
63
  minimax: "MiniMax"
63
64
  };
64
65
  var PROVIDER_STATUS_URL = {
65
66
  openai: "status.openai.com",
66
- anthropic: "status.anthropic.com"
67
+ anthropic: "status.anthropic.com",
68
+ xai: "status.x.ai"
67
69
  };
68
70
  function providerDisplayName(provider) {
69
71
  return PROVIDER_DISPLAY[provider] ?? provider;
@@ -247,6 +249,9 @@ function providerGuidance(provider, message, statusCode) {
247
249
  if (statusCode === 503 || lower.includes("service unavailable")) {
248
250
  return `${name} is temporarily unavailable. Retry shortly \u2014 not a EZ Coder issue.`;
249
251
  }
252
+ if (statusCode === 507 || lower.includes("exceeded request buffer limit while retrying upstream")) {
253
+ return `${name}'s proxy could not retry this large request. EZ Coder already retried automatically \u2014 compact the conversation, then retry.`;
254
+ }
250
255
  if (statusCode === 500 || lower.includes("server_error") || lower.includes("500") && lower.includes("internal server error")) {
251
256
  return status ? `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, check ${status}.` : `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, try a different model via the model selector.`;
252
257
  }
@@ -1833,7 +1838,11 @@ async function* runStream2(options) {
1833
1838
  const providerName = options.provider ?? "openai";
1834
1839
  const useStreaming = options.streaming !== false;
1835
1840
  const client = createClient2(options);
1836
- const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" || options.provider === "xiaomi";
1841
+ const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1842
+ const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1843
+ const isKimiK27 = options.provider === "moonshot" && options.model.startsWith("kimi-k2.7-code");
1844
+ const hasFixedKimiSampling = isKimiK3 || isKimiK27;
1845
+ const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" && !isKimiK3 && !isKimiK27 || options.provider === "xiaomi";
1837
1846
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1838
1847
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1839
1848
  if (options.provider === "moonshot") {
@@ -1845,7 +1854,9 @@ async function* runStream2(options) {
1845
1854
  }
1846
1855
  const messages = toOpenAIMessages(downgradedMessages, {
1847
1856
  provider: options.provider,
1848
- thinking: !!options.thinking,
1857
+ // K3 and K2.7 preserve reasoning even when the user hides thinking in the
1858
+ // UI; keep assistant tool-call history wire-valid in that display mode.
1859
+ thinking: isKimiK3 || isKimiK27 || !!options.thinking,
1849
1860
  supportsImages: options.supportsImages
1850
1861
  });
1851
1862
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
@@ -1855,10 +1866,10 @@ async function* runStream2(options) {
1855
1866
  messages,
1856
1867
  stream: useStreaming,
1857
1868
  ...options.maxTokens ? { max_completion_tokens: options.maxTokens } : {},
1858
- ...effectiveTemp != null && !options.thinking ? { temperature: effectiveTemp } : {},
1859
- ...options.topP != null ? { top_p: options.topP } : {},
1869
+ ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1870
+ ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1860
1871
  ...options.stop ? { stop: options.stop } : {},
1861
- ...options.thinking && !usesThinkingParam ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1872
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1862
1873
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1863
1874
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1864
1875
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1868,13 +1879,21 @@ async function* runStream2(options) {
1868
1879
  paramsAny.prompt_cache_key = normalizePromptCacheKey(options.promptCacheKey ?? "ezcoder");
1869
1880
  if (options.provider === "openai" && options.model.startsWith("gpt-5.6")) {
1870
1881
  paramsAny.prompt_cache_options = { mode: "implicit", ttl: "30m" };
1871
- } else if ((options.cacheRetention ?? "short") === "long") {
1882
+ } else if (!isKimiK3 && (options.cacheRetention ?? "short") === "long") {
1872
1883
  paramsAny.prompt_cache_retention = "24h";
1873
1884
  }
1874
1885
  }
1875
1886
  if (options.provider === "openai" && options.serviceTier) {
1876
1887
  params.service_tier = options.serviceTier;
1877
1888
  }
1889
+ if (isKimiK3) {
1890
+ const paramsAny = params;
1891
+ if (isManagedKimiK3) {
1892
+ paramsAny.thinking = { type: "enabled", effort: "max", keep: "all" };
1893
+ } else {
1894
+ paramsAny.reasoning_effort = "max";
1895
+ }
1896
+ }
1878
1897
  if (usesThinkingParam) {
1879
1898
  if (options.thinking) {
1880
1899
  params.thinking = { type: "enabled" };
@@ -2170,6 +2189,7 @@ function toError2(err, provider = "openai") {
2170
2189
 
2171
2190
  // src/providers/openai-codex.ts
2172
2191
  import os from "os";
2192
+ import * as zstd from "@bokuweb/zstd-wasm";
2173
2193
 
2174
2194
  // src/utils/sse.ts
2175
2195
  function parseSseBuffer(buffer) {
@@ -2225,6 +2245,50 @@ function extractRequestIdFromMessage(message) {
2225
2245
  // src/providers/openai-codex.ts
2226
2246
  var DEFAULT_BASE_URL = "https://chatgpt.com/backend-api";
2227
2247
  var CODEX_CLIENT_VERSION = "0.144.1";
2248
+ var CODEX_REQUEST_COMPRESSION_MIN_BYTES = 16 * 1024;
2249
+ var zstdInitPromise;
2250
+ async function encodeCodexRequest(body) {
2251
+ const json = JSON.stringify(body);
2252
+ const raw = new TextEncoder().encode(json);
2253
+ if (raw.byteLength < CODEX_REQUEST_COMPRESSION_MIN_BYTES) {
2254
+ return {
2255
+ body: json,
2256
+ compressed: false,
2257
+ rawBytes: raw.byteLength,
2258
+ encodedBytes: raw.byteLength
2259
+ };
2260
+ }
2261
+ try {
2262
+ zstdInitPromise ??= zstd.init();
2263
+ await zstdInitPromise;
2264
+ const compressed = Uint8Array.from(zstd.compress(raw));
2265
+ if (compressed.byteLength >= raw.byteLength) {
2266
+ return {
2267
+ body: json,
2268
+ compressed: false,
2269
+ rawBytes: raw.byteLength,
2270
+ encodedBytes: raw.byteLength
2271
+ };
2272
+ }
2273
+ return {
2274
+ body: compressed,
2275
+ compressed: true,
2276
+ rawBytes: raw.byteLength,
2277
+ encodedBytes: compressed.byteLength
2278
+ };
2279
+ } catch (error) {
2280
+ providerDiag("codex_request_compression_failed", {
2281
+ error: error instanceof Error ? error.message : String(error),
2282
+ rawBytes: raw.byteLength
2283
+ });
2284
+ return {
2285
+ body: json,
2286
+ compressed: false,
2287
+ rawBytes: raw.byteLength,
2288
+ encodedBytes: raw.byteLength
2289
+ };
2290
+ }
2291
+ }
2228
2292
  function usesResponsesLite(model) {
2229
2293
  return model.startsWith("gpt-5.6-");
2230
2294
  }
@@ -2305,10 +2369,17 @@ async function* runStream3(options) {
2305
2369
  headers["session_id"] = transportSessionId;
2306
2370
  headers["x-client-request-id"] = transportSessionId;
2307
2371
  }
2372
+ const encodedRequest = await encodeCodexRequest(body);
2373
+ if (encodedRequest.compressed) headers["Content-Encoding"] = "zstd";
2374
+ providerDiag("codex_request_body", {
2375
+ rawBytes: encodedRequest.rawBytes,
2376
+ encodedBytes: encodedRequest.encodedBytes,
2377
+ compressed: encodedRequest.compressed
2378
+ });
2308
2379
  const response = await fetch(url, {
2309
2380
  method: "POST",
2310
2381
  headers,
2311
- body: JSON.stringify(body),
2382
+ body: encodedRequest.body,
2312
2383
  signal: options.signal
2313
2384
  });
2314
2385
  if (!response.ok) {
@@ -3347,6 +3418,18 @@ providerRegistry.register("sakana", {
3347
3418
  baseUrl: options.baseUrl ?? "https://api.sakana.ai/v1"
3348
3419
  })
3349
3420
  });
3421
+ providerRegistry.register("xai", {
3422
+ // xAI's public API (console.x.ai key) is OpenAI-compatible — ride the Chat
3423
+ // Completions transport like Moonshot/DeepSeek. Grok reasoning models take
3424
+ // top-level `reasoning_effort` (low/medium/high), which the shared thinking
3425
+ // path already sends. xAI's OAuth path exists but only via the Grok CLI's
3426
+ // private Responses proxy (cli-chat-proxy.grok.com) with reverse-engineered
3427
+ // attribution headers and account-tier gating — intentionally not wired.
3428
+ stream: (options) => streamOpenAI({
3429
+ ...options,
3430
+ baseUrl: options.baseUrl ?? "https://api.x.ai/v1"
3431
+ })
3432
+ });
3350
3433
  providerRegistry.register("minimax", {
3351
3434
  stream: (options) => streamAnthropic({
3352
3435
  ...options,