@prestyj/ai 5.7.0 → 5.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -43,7 +43,7 @@ Tool parameters are Zod schemas. Converted to JSON Schema at the provider bounda
43
43
  | `anthropic` | Claude Opus 4.8, Sonnet 5, Haiku 4.5 | Extended thinking, prompt caching, server-side compaction |
44
44
  | `openai` | GPT-4.1, o3, o4-mini | Supports OAuth (codex endpoint) and API keys |
45
45
  | `glm` | GLM-5.1, GLM-4.7 | Z.AI platform, OpenAI-compatible |
46
- | `moonshot` | Kimi K2.7 | Moonshot platform, OpenAI-compatible |
46
+ | `moonshot` | Kimi K3, Kimi K2.7 Code | Moonshot platform, OpenAI-compatible |
47
47
 
48
48
  ---
49
49
 
package/dist/index.cjs CHANGED
@@ -118,12 +118,14 @@ var PROVIDER_DISPLAY = {
118
118
  deepseek: "DeepSeek",
119
119
  openrouter: "OpenRouter",
120
120
  sakana: "Sakana",
121
+ xai: "xAI (Grok)",
121
122
  xiaomi: "Xiaomi (MiMo)",
122
123
  minimax: "MiniMax"
123
124
  };
124
125
  var PROVIDER_STATUS_URL = {
125
126
  openai: "status.openai.com",
126
- anthropic: "status.anthropic.com"
127
+ anthropic: "status.anthropic.com",
128
+ xai: "status.x.ai"
127
129
  };
128
130
  function providerDisplayName(provider) {
129
131
  return PROVIDER_DISPLAY[provider] ?? provider;
@@ -307,6 +309,9 @@ function providerGuidance(provider, message, statusCode) {
307
309
  if (statusCode === 503 || lower.includes("service unavailable")) {
308
310
  return `${name} is temporarily unavailable. Retry shortly \u2014 not a EZ Coder issue.`;
309
311
  }
312
+ if (statusCode === 507 || lower.includes("exceeded request buffer limit while retrying upstream")) {
313
+ return `${name}'s proxy could not retry this large request. EZ Coder already retried automatically \u2014 compact the conversation, then retry.`;
314
+ }
310
315
  if (statusCode === 500 || lower.includes("server_error") || lower.includes("500") && lower.includes("internal server error")) {
311
316
  return status ? `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, check ${status}.` : `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, try a different model via the model selector.`;
312
317
  }
@@ -319,6 +324,9 @@ function providerGuidance(provider, message, statusCode) {
319
324
  if (lower.includes("context_length_exceeded") || lower.includes("prompt is too long")) {
320
325
  return `Context window for this ${name} model is full. Compact the conversation to shrink history, or start a new session.`;
321
326
  }
327
+ if (lower.includes("many-image request") || lower.includes("image dimensions") && lower.includes("max allowed size")) {
328
+ return `An image in conversation history exceeds ${name}'s many-image limit. Restart EZ Coder so restored images are resized, then retry; if it persists, start a new session.`;
329
+ }
322
330
  if (statusCode === 413 || lower.includes("request_too_large") || lower.includes("request exceeds the maximum size")) {
323
331
  return `The request to ${name} is too large. Compact the conversation to shrink history, or start a new session.`;
324
332
  }
@@ -815,9 +823,14 @@ function toAnthropicMessages(messages, cacheControl) {
815
823
  continue;
816
824
  }
817
825
  if (msg.role === "user") {
826
+ if (typeof msg.content === "string") {
827
+ if (msg.content === "") continue;
828
+ } else if (!msg.content.some((p) => !(p.type === "text" && p.text === ""))) {
829
+ continue;
830
+ }
818
831
  out.push({
819
832
  role: "user",
820
- content: typeof msg.content === "string" ? msg.content : msg.content.map((part) => {
833
+ content: typeof msg.content === "string" ? msg.content : msg.content.filter((part) => !(part.type === "text" && part.text === "")).map((part) => {
821
834
  if (part.type === "text") return { type: "text", text: part.text };
822
835
  if (part.type === "video") {
823
836
  return {
@@ -842,6 +855,7 @@ function toAnthropicMessages(messages, cacheControl) {
842
855
  continue;
843
856
  }
844
857
  if (msg.role === "assistant") {
858
+ if (typeof msg.content === "string" && msg.content === "") continue;
845
859
  const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
846
860
  if (Array.isArray(content) && content.length === 0) continue;
847
861
  out.push({ role: "assistant", content });
@@ -962,7 +976,7 @@ function remapToolCallId(id, idMap) {
962
976
  if (!id.startsWith("toolu_")) return id;
963
977
  const existing = idMap.get(id);
964
978
  if (existing) return existing;
965
- const mapped = `call_${id.slice(5)}`;
979
+ const mapped = `call_${id.slice(6)}`;
966
980
  idMap.set(id, mapped);
967
981
  return mapped;
968
982
  }
@@ -1573,6 +1587,12 @@ async function* runStream(options) {
1573
1587
  statusCode: 504
1574
1588
  });
1575
1589
  }
1590
+ if (stopReason === null) {
1591
+ throw new ProviderError("anthropic", "Stream ended before completion (no stop_reason).", {
1592
+ statusCode: 504,
1593
+ cause: { partialContent: contentParts, outputTokens }
1594
+ });
1595
+ }
1576
1596
  const normalizedStop = normalizeAnthropicStopReason(stopReason);
1577
1597
  const response = {
1578
1598
  message: {
@@ -1839,6 +1859,17 @@ function getEnvironment() {
1839
1859
  }
1840
1860
 
1841
1861
  // src/providers/openai.ts
1862
+ function toKimiK3Effort(level) {
1863
+ switch (level) {
1864
+ case "low":
1865
+ return "low";
1866
+ case "medium":
1867
+ case "high":
1868
+ return "high";
1869
+ default:
1870
+ return "max";
1871
+ }
1872
+ }
1842
1873
  function extractOpenAIUsage(usage) {
1843
1874
  let cacheRead = 0;
1844
1875
  let cacheWrite = 0;
@@ -1893,7 +1924,12 @@ async function* runStream2(options) {
1893
1924
  const providerName = options.provider ?? "openai";
1894
1925
  const useStreaming = options.streaming !== false;
1895
1926
  const client = createClient2(options);
1896
- const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" || options.provider === "xiaomi";
1927
+ const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1928
+ const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1929
+ const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
1930
+ const isKimiK27 = options.provider === "moonshot" && options.model.startsWith("kimi-k2.7-code");
1931
+ const hasFixedKimiSampling = isKimiK3 || isKimiK27;
1932
+ const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" && !isKimiK3 && !isKimiK27 || options.provider === "xiaomi";
1897
1933
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1898
1934
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1899
1935
  if (options.provider === "moonshot") {
@@ -1905,7 +1941,11 @@ async function* runStream2(options) {
1905
1941
  }
1906
1942
  const messages = toOpenAIMessages(downgradedMessages, {
1907
1943
  provider: options.provider,
1908
- thinking: !!options.thinking,
1944
+ // K2.7 preserves reasoning even when the user hides thinking in the UI;
1945
+ // keep assistant tool-call history wire-valid in that display mode. A
1946
+ // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1947
+ // official CLI: reasoning is preserved only while thinking is enabled).
1948
+ thinking: isKimiK27 || !!options.thinking,
1909
1949
  supportsImages: options.supportsImages
1910
1950
  });
1911
1951
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
@@ -1915,10 +1955,10 @@ async function* runStream2(options) {
1915
1955
  messages,
1916
1956
  stream: useStreaming,
1917
1957
  ...options.maxTokens ? { max_completion_tokens: options.maxTokens } : {},
1918
- ...effectiveTemp != null && !options.thinking ? { temperature: effectiveTemp } : {},
1919
- ...options.topP != null ? { top_p: options.topP } : {},
1958
+ ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1959
+ ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1920
1960
  ...options.stop ? { stop: options.stop } : {},
1921
- ...options.thinking && !usesThinkingParam ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1961
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1922
1962
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1923
1963
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1924
1964
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1928,13 +1968,23 @@ async function* runStream2(options) {
1928
1968
  paramsAny.prompt_cache_key = normalizePromptCacheKey(options.promptCacheKey ?? "ezcoder");
1929
1969
  if (options.provider === "openai" && options.model.startsWith("gpt-5.6")) {
1930
1970
  paramsAny.prompt_cache_options = { mode: "implicit", ttl: "30m" };
1931
- } else if ((options.cacheRetention ?? "short") === "long") {
1971
+ } else if (!isKimiK3 && (options.cacheRetention ?? "short") === "long") {
1932
1972
  paramsAny.prompt_cache_retention = "24h";
1933
1973
  }
1934
1974
  }
1935
1975
  if (options.provider === "openai" && options.serviceTier) {
1936
1976
  params.service_tier = options.serviceTier;
1937
1977
  }
1978
+ if (isKimiK3) {
1979
+ const paramsAny = params;
1980
+ if (isManagedKimiK3) {
1981
+ paramsAny.thinking = k3Effort ? { type: "enabled", effort: k3Effort, keep: "all" } : { type: "disabled" };
1982
+ } else if (k3Effort) {
1983
+ paramsAny.reasoning_effort = k3Effort;
1984
+ } else {
1985
+ paramsAny.thinking = { type: "disabled" };
1986
+ }
1987
+ }
1938
1988
  if (usesThinkingParam) {
1939
1989
  if (options.thinking) {
1940
1990
  params.thinking = { type: "enabled" };
@@ -2038,6 +2088,12 @@ async function* runStream2(options) {
2038
2088
  statusCode: 504
2039
2089
  });
2040
2090
  }
2091
+ if (finishReason === null) {
2092
+ throw new ProviderError(providerName, "Stream ended before completion (no finish_reason).", {
2093
+ statusCode: 504,
2094
+ cause: { partialText: textAccum, outputTokens }
2095
+ });
2096
+ }
2041
2097
  if (thinkingAccum) {
2042
2098
  contentParts.push({ type: "thinking", text: thinkingAccum });
2043
2099
  }
@@ -2230,6 +2286,7 @@ function toError2(err, provider = "openai") {
2230
2286
 
2231
2287
  // src/providers/openai-codex.ts
2232
2288
  var import_node_os = __toESM(require("os"), 1);
2289
+ var zstd = __toESM(require("@bokuweb/zstd-wasm"), 1);
2233
2290
 
2234
2291
  // src/utils/sse.ts
2235
2292
  function parseSseBuffer(buffer) {
@@ -2285,6 +2342,50 @@ function extractRequestIdFromMessage(message) {
2285
2342
  // src/providers/openai-codex.ts
2286
2343
  var DEFAULT_BASE_URL = "https://chatgpt.com/backend-api";
2287
2344
  var CODEX_CLIENT_VERSION = "0.144.1";
2345
+ var CODEX_REQUEST_COMPRESSION_MIN_BYTES = 16 * 1024;
2346
+ var zstdInitPromise;
2347
+ async function encodeCodexRequest(body) {
2348
+ const json = JSON.stringify(body);
2349
+ const raw = new TextEncoder().encode(json);
2350
+ if (raw.byteLength < CODEX_REQUEST_COMPRESSION_MIN_BYTES) {
2351
+ return {
2352
+ body: json,
2353
+ compressed: false,
2354
+ rawBytes: raw.byteLength,
2355
+ encodedBytes: raw.byteLength
2356
+ };
2357
+ }
2358
+ try {
2359
+ zstdInitPromise ??= zstd.init();
2360
+ await zstdInitPromise;
2361
+ const compressed = Uint8Array.from(zstd.compress(raw));
2362
+ if (compressed.byteLength >= raw.byteLength) {
2363
+ return {
2364
+ body: json,
2365
+ compressed: false,
2366
+ rawBytes: raw.byteLength,
2367
+ encodedBytes: raw.byteLength
2368
+ };
2369
+ }
2370
+ return {
2371
+ body: compressed,
2372
+ compressed: true,
2373
+ rawBytes: raw.byteLength,
2374
+ encodedBytes: compressed.byteLength
2375
+ };
2376
+ } catch (error) {
2377
+ providerDiag("codex_request_compression_failed", {
2378
+ error: error instanceof Error ? error.message : String(error),
2379
+ rawBytes: raw.byteLength
2380
+ });
2381
+ return {
2382
+ body: json,
2383
+ compressed: false,
2384
+ rawBytes: raw.byteLength,
2385
+ encodedBytes: raw.byteLength
2386
+ };
2387
+ }
2388
+ }
2288
2389
  function usesResponsesLite(model) {
2289
2390
  return model.startsWith("gpt-5.6-");
2290
2391
  }
@@ -2365,10 +2466,17 @@ async function* runStream3(options) {
2365
2466
  headers["session_id"] = transportSessionId;
2366
2467
  headers["x-client-request-id"] = transportSessionId;
2367
2468
  }
2469
+ const encodedRequest = await encodeCodexRequest(body);
2470
+ if (encodedRequest.compressed) headers["Content-Encoding"] = "zstd";
2471
+ providerDiag("codex_request_body", {
2472
+ rawBytes: encodedRequest.rawBytes,
2473
+ encodedBytes: encodedRequest.encodedBytes,
2474
+ compressed: encodedRequest.compressed
2475
+ });
2368
2476
  const response = await fetch(url, {
2369
2477
  method: "POST",
2370
2478
  headers,
2371
- body: JSON.stringify(body),
2479
+ body: encodedRequest.body,
2372
2480
  signal: options.signal
2373
2481
  });
2374
2482
  if (!response.ok) {
@@ -3407,6 +3515,18 @@ providerRegistry.register("sakana", {
3407
3515
  baseUrl: options.baseUrl ?? "https://api.sakana.ai/v1"
3408
3516
  })
3409
3517
  });
3518
+ providerRegistry.register("xai", {
3519
+ // xAI's public API (console.x.ai key) is OpenAI-compatible — ride the Chat
3520
+ // Completions transport like Moonshot/DeepSeek. Grok reasoning models take
3521
+ // top-level `reasoning_effort` (low/medium/high), which the shared thinking
3522
+ // path already sends. xAI's OAuth path exists but only via the Grok CLI's
3523
+ // private Responses proxy (cli-chat-proxy.grok.com) with reverse-engineered
3524
+ // attribution headers and account-tier gating — intentionally not wired.
3525
+ stream: (options) => streamOpenAI({
3526
+ ...options,
3527
+ baseUrl: options.baseUrl ?? "https://api.x.ai/v1"
3528
+ })
3529
+ });
3410
3530
  providerRegistry.register("minimax", {
3411
3531
  stream: (options) => streamAnthropic({
3412
3532
  ...options,