@prestyj/ai 5.7.0 → 5.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -43,7 +43,7 @@ Tool parameters are Zod schemas. Converted to JSON Schema at the provider bounda
43
43
  | `anthropic` | Claude Opus 4.8, Sonnet 5, Haiku 4.5 | Extended thinking, prompt caching, server-side compaction |
44
44
  | `openai` | GPT-4.1, o3, o4-mini | Supports OAuth (codex endpoint) and API keys |
45
45
  | `glm` | GLM-5.1, GLM-4.7 | Z.AI platform, OpenAI-compatible |
46
- | `moonshot` | Kimi K2.7 | Moonshot platform, OpenAI-compatible |
46
+ | `moonshot` | Kimi K3, Kimi K2.7 Code | Moonshot platform, OpenAI-compatible |
47
47
 
48
48
  ---
49
49
 
package/dist/index.cjs CHANGED
@@ -118,12 +118,14 @@ var PROVIDER_DISPLAY = {
118
118
  deepseek: "DeepSeek",
119
119
  openrouter: "OpenRouter",
120
120
  sakana: "Sakana",
121
+ xai: "xAI (Grok)",
121
122
  xiaomi: "Xiaomi (MiMo)",
122
123
  minimax: "MiniMax"
123
124
  };
124
125
  var PROVIDER_STATUS_URL = {
125
126
  openai: "status.openai.com",
126
- anthropic: "status.anthropic.com"
127
+ anthropic: "status.anthropic.com",
128
+ xai: "status.x.ai"
127
129
  };
128
130
  function providerDisplayName(provider) {
129
131
  return PROVIDER_DISPLAY[provider] ?? provider;
@@ -307,6 +309,9 @@ function providerGuidance(provider, message, statusCode) {
307
309
  if (statusCode === 503 || lower.includes("service unavailable")) {
308
310
  return `${name} is temporarily unavailable. Retry shortly \u2014 not a EZ Coder issue.`;
309
311
  }
312
+ if (statusCode === 507 || lower.includes("exceeded request buffer limit while retrying upstream")) {
313
+ return `${name}'s proxy could not retry this large request. EZ Coder already retried automatically \u2014 compact the conversation, then retry.`;
314
+ }
310
315
  if (statusCode === 500 || lower.includes("server_error") || lower.includes("500") && lower.includes("internal server error")) {
311
316
  return status ? `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, check ${status}.` : `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, try a different model via the model selector.`;
312
317
  }
@@ -1893,7 +1898,11 @@ async function* runStream2(options) {
1893
1898
  const providerName = options.provider ?? "openai";
1894
1899
  const useStreaming = options.streaming !== false;
1895
1900
  const client = createClient2(options);
1896
- const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" || options.provider === "xiaomi";
1901
+ const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1902
+ const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1903
+ const isKimiK27 = options.provider === "moonshot" && options.model.startsWith("kimi-k2.7-code");
1904
+ const hasFixedKimiSampling = isKimiK3 || isKimiK27;
1905
+ const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" && !isKimiK3 && !isKimiK27 || options.provider === "xiaomi";
1897
1906
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1898
1907
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1899
1908
  if (options.provider === "moonshot") {
@@ -1905,7 +1914,9 @@ async function* runStream2(options) {
1905
1914
  }
1906
1915
  const messages = toOpenAIMessages(downgradedMessages, {
1907
1916
  provider: options.provider,
1908
- thinking: !!options.thinking,
1917
+ // K3 and K2.7 preserve reasoning even when the user hides thinking in the
1918
+ // UI; keep assistant tool-call history wire-valid in that display mode.
1919
+ thinking: isKimiK3 || isKimiK27 || !!options.thinking,
1909
1920
  supportsImages: options.supportsImages
1910
1921
  });
1911
1922
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
@@ -1915,10 +1926,10 @@ async function* runStream2(options) {
1915
1926
  messages,
1916
1927
  stream: useStreaming,
1917
1928
  ...options.maxTokens ? { max_completion_tokens: options.maxTokens } : {},
1918
- ...effectiveTemp != null && !options.thinking ? { temperature: effectiveTemp } : {},
1919
- ...options.topP != null ? { top_p: options.topP } : {},
1929
+ ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1930
+ ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1920
1931
  ...options.stop ? { stop: options.stop } : {},
1921
- ...options.thinking && !usesThinkingParam ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1932
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1922
1933
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1923
1934
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1924
1935
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1928,13 +1939,21 @@ async function* runStream2(options) {
1928
1939
  paramsAny.prompt_cache_key = normalizePromptCacheKey(options.promptCacheKey ?? "ezcoder");
1929
1940
  if (options.provider === "openai" && options.model.startsWith("gpt-5.6")) {
1930
1941
  paramsAny.prompt_cache_options = { mode: "implicit", ttl: "30m" };
1931
- } else if ((options.cacheRetention ?? "short") === "long") {
1942
+ } else if (!isKimiK3 && (options.cacheRetention ?? "short") === "long") {
1932
1943
  paramsAny.prompt_cache_retention = "24h";
1933
1944
  }
1934
1945
  }
1935
1946
  if (options.provider === "openai" && options.serviceTier) {
1936
1947
  params.service_tier = options.serviceTier;
1937
1948
  }
1949
+ if (isKimiK3) {
1950
+ const paramsAny = params;
1951
+ if (isManagedKimiK3) {
1952
+ paramsAny.thinking = { type: "enabled", effort: "max", keep: "all" };
1953
+ } else {
1954
+ paramsAny.reasoning_effort = "max";
1955
+ }
1956
+ }
1938
1957
  if (usesThinkingParam) {
1939
1958
  if (options.thinking) {
1940
1959
  params.thinking = { type: "enabled" };
@@ -2230,6 +2249,7 @@ function toError2(err, provider = "openai") {
2230
2249
 
2231
2250
  // src/providers/openai-codex.ts
2232
2251
  var import_node_os = __toESM(require("os"), 1);
2252
+ var zstd = __toESM(require("@bokuweb/zstd-wasm"), 1);
2233
2253
 
2234
2254
  // src/utils/sse.ts
2235
2255
  function parseSseBuffer(buffer) {
@@ -2285,6 +2305,50 @@ function extractRequestIdFromMessage(message) {
2285
2305
  // src/providers/openai-codex.ts
2286
2306
  var DEFAULT_BASE_URL = "https://chatgpt.com/backend-api";
2287
2307
  var CODEX_CLIENT_VERSION = "0.144.1";
2308
+ var CODEX_REQUEST_COMPRESSION_MIN_BYTES = 16 * 1024;
2309
+ var zstdInitPromise;
2310
+ async function encodeCodexRequest(body) {
2311
+ const json = JSON.stringify(body);
2312
+ const raw = new TextEncoder().encode(json);
2313
+ if (raw.byteLength < CODEX_REQUEST_COMPRESSION_MIN_BYTES) {
2314
+ return {
2315
+ body: json,
2316
+ compressed: false,
2317
+ rawBytes: raw.byteLength,
2318
+ encodedBytes: raw.byteLength
2319
+ };
2320
+ }
2321
+ try {
2322
+ zstdInitPromise ??= zstd.init();
2323
+ await zstdInitPromise;
2324
+ const compressed = Uint8Array.from(zstd.compress(raw));
2325
+ if (compressed.byteLength >= raw.byteLength) {
2326
+ return {
2327
+ body: json,
2328
+ compressed: false,
2329
+ rawBytes: raw.byteLength,
2330
+ encodedBytes: raw.byteLength
2331
+ };
2332
+ }
2333
+ return {
2334
+ body: compressed,
2335
+ compressed: true,
2336
+ rawBytes: raw.byteLength,
2337
+ encodedBytes: compressed.byteLength
2338
+ };
2339
+ } catch (error) {
2340
+ providerDiag("codex_request_compression_failed", {
2341
+ error: error instanceof Error ? error.message : String(error),
2342
+ rawBytes: raw.byteLength
2343
+ });
2344
+ return {
2345
+ body: json,
2346
+ compressed: false,
2347
+ rawBytes: raw.byteLength,
2348
+ encodedBytes: raw.byteLength
2349
+ };
2350
+ }
2351
+ }
2288
2352
  function usesResponsesLite(model) {
2289
2353
  return model.startsWith("gpt-5.6-");
2290
2354
  }
@@ -2365,10 +2429,17 @@ async function* runStream3(options) {
2365
2429
  headers["session_id"] = transportSessionId;
2366
2430
  headers["x-client-request-id"] = transportSessionId;
2367
2431
  }
2432
+ const encodedRequest = await encodeCodexRequest(body);
2433
+ if (encodedRequest.compressed) headers["Content-Encoding"] = "zstd";
2434
+ providerDiag("codex_request_body", {
2435
+ rawBytes: encodedRequest.rawBytes,
2436
+ encodedBytes: encodedRequest.encodedBytes,
2437
+ compressed: encodedRequest.compressed
2438
+ });
2368
2439
  const response = await fetch(url, {
2369
2440
  method: "POST",
2370
2441
  headers,
2371
- body: JSON.stringify(body),
2442
+ body: encodedRequest.body,
2372
2443
  signal: options.signal
2373
2444
  });
2374
2445
  if (!response.ok) {
@@ -3407,6 +3478,18 @@ providerRegistry.register("sakana", {
3407
3478
  baseUrl: options.baseUrl ?? "https://api.sakana.ai/v1"
3408
3479
  })
3409
3480
  });
3481
+ providerRegistry.register("xai", {
3482
+ // xAI's public API (console.x.ai key) is OpenAI-compatible — ride the Chat
3483
+ // Completions transport like Moonshot/DeepSeek. Grok reasoning models take
3484
+ // top-level `reasoning_effort` (low/medium/high), which the shared thinking
3485
+ // path already sends. xAI's OAuth path exists but only via the Grok CLI's
3486
+ // private Responses proxy (cli-chat-proxy.grok.com) with reverse-engineered
3487
+ // attribution headers and account-tier gating — intentionally not wired.
3488
+ stream: (options) => streamOpenAI({
3489
+ ...options,
3490
+ baseUrl: options.baseUrl ?? "https://api.x.ai/v1"
3491
+ })
3492
+ });
3410
3493
  providerRegistry.register("minimax", {
3411
3494
  stream: (options) => streamAnthropic({
3412
3495
  ...options,