@prestyj/ai 5.7.0 → 5.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -2,7 +2,7 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
6
6
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
7
  type CacheRetention = "none" | "short" | "long";
8
8
  interface TextContent {
@@ -41,6 +41,21 @@ interface ToolResult {
41
41
  toolCallId: string;
42
42
  content: ToolResultContent;
43
43
  isError?: boolean;
44
+ /**
45
+ * Set when the agent loop trimmed `content` to fit a per-result or per-turn
46
+ * budget. The provider (model input) and the persistent transcript both see
47
+ * the trimmed `content`, but the live `tool_call_end` event carried the FULL
48
+ * preview — so this marker makes that divergence explicit and reconcilable.
49
+ * Internal metadata only: it is never serialized onto the provider wire.
50
+ */
51
+ capped?: {
52
+ /** Length of the original, untrimmed string content. */
53
+ originalChars: number;
54
+ /** Length of the trimmed content actually sent to the model. */
55
+ keptChars: number;
56
+ /** Which budget triggered the trim. */
57
+ scope: "per-result" | "per-turn";
58
+ };
44
59
  }
45
60
  interface ServerToolCall {
46
61
  type: "server_tool_call";
package/dist/index.d.ts CHANGED
@@ -2,7 +2,7 @@ import { z } from 'zod';
2
2
  import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
- type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "palsu";
5
+ type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu";
6
6
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
7
7
  type CacheRetention = "none" | "short" | "long";
8
8
  interface TextContent {
@@ -41,6 +41,21 @@ interface ToolResult {
41
41
  toolCallId: string;
42
42
  content: ToolResultContent;
43
43
  isError?: boolean;
44
+ /**
45
+ * Set when the agent loop trimmed `content` to fit a per-result or per-turn
46
+ * budget. The provider (model input) and the persistent transcript both see
47
+ * the trimmed `content`, but the live `tool_call_end` event carried the FULL
48
+ * preview — so this marker makes that divergence explicit and reconcilable.
49
+ * Internal metadata only: it is never serialized onto the provider wire.
50
+ */
51
+ capped?: {
52
+ /** Length of the original, untrimmed string content. */
53
+ originalChars: number;
54
+ /** Length of the trimmed content actually sent to the model. */
55
+ keptChars: number;
56
+ /** Which budget triggered the trim. */
57
+ scope: "per-result" | "per-turn";
58
+ };
44
59
  }
45
60
  interface ServerToolCall {
46
61
  type: "server_tool_call";
package/dist/index.js CHANGED
@@ -58,12 +58,14 @@ var PROVIDER_DISPLAY = {
58
58
  deepseek: "DeepSeek",
59
59
  openrouter: "OpenRouter",
60
60
  sakana: "Sakana",
61
+ xai: "xAI (Grok)",
61
62
  xiaomi: "Xiaomi (MiMo)",
62
63
  minimax: "MiniMax"
63
64
  };
64
65
  var PROVIDER_STATUS_URL = {
65
66
  openai: "status.openai.com",
66
- anthropic: "status.anthropic.com"
67
+ anthropic: "status.anthropic.com",
68
+ xai: "status.x.ai"
67
69
  };
68
70
  function providerDisplayName(provider) {
69
71
  return PROVIDER_DISPLAY[provider] ?? provider;
@@ -247,6 +249,9 @@ function providerGuidance(provider, message, statusCode) {
247
249
  if (statusCode === 503 || lower.includes("service unavailable")) {
248
250
  return `${name} is temporarily unavailable. Retry shortly \u2014 not a EZ Coder issue.`;
249
251
  }
252
+ if (statusCode === 507 || lower.includes("exceeded request buffer limit while retrying upstream")) {
253
+ return `${name}'s proxy could not retry this large request. EZ Coder already retried automatically \u2014 compact the conversation, then retry.`;
254
+ }
250
255
  if (statusCode === 500 || lower.includes("server_error") || lower.includes("500") && lower.includes("internal server error")) {
251
256
  return status ? `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, check ${status}.` : `This is an error from ${name}, not EZ Coder. Retry \u2014 if it keeps happening, try a different model via the model selector.`;
252
257
  }
@@ -259,6 +264,9 @@ function providerGuidance(provider, message, statusCode) {
259
264
  if (lower.includes("context_length_exceeded") || lower.includes("prompt is too long")) {
260
265
  return `Context window for this ${name} model is full. Compact the conversation to shrink history, or start a new session.`;
261
266
  }
267
+ if (lower.includes("many-image request") || lower.includes("image dimensions") && lower.includes("max allowed size")) {
268
+ return `An image in conversation history exceeds ${name}'s many-image limit. Restart EZ Coder so restored images are resized, then retry; if it persists, start a new session.`;
269
+ }
262
270
  if (statusCode === 413 || lower.includes("request_too_large") || lower.includes("request exceeds the maximum size")) {
263
271
  return `The request to ${name} is too large. Compact the conversation to shrink history, or start a new session.`;
264
272
  }
@@ -755,9 +763,14 @@ function toAnthropicMessages(messages, cacheControl) {
755
763
  continue;
756
764
  }
757
765
  if (msg.role === "user") {
766
+ if (typeof msg.content === "string") {
767
+ if (msg.content === "") continue;
768
+ } else if (!msg.content.some((p) => !(p.type === "text" && p.text === ""))) {
769
+ continue;
770
+ }
758
771
  out.push({
759
772
  role: "user",
760
- content: typeof msg.content === "string" ? msg.content : msg.content.map((part) => {
773
+ content: typeof msg.content === "string" ? msg.content : msg.content.filter((part) => !(part.type === "text" && part.text === "")).map((part) => {
761
774
  if (part.type === "text") return { type: "text", text: part.text };
762
775
  if (part.type === "video") {
763
776
  return {
@@ -782,6 +795,7 @@ function toAnthropicMessages(messages, cacheControl) {
782
795
  continue;
783
796
  }
784
797
  if (msg.role === "assistant") {
798
+ if (typeof msg.content === "string" && msg.content === "") continue;
785
799
  const content = typeof msg.content === "string" ? msg.content : toAnthropicAssistantContent(msg.content, msgIdx > trajectoryStartIdx, idMap);
786
800
  if (Array.isArray(content) && content.length === 0) continue;
787
801
  out.push({ role: "assistant", content });
@@ -902,7 +916,7 @@ function remapToolCallId(id, idMap) {
902
916
  if (!id.startsWith("toolu_")) return id;
903
917
  const existing = idMap.get(id);
904
918
  if (existing) return existing;
905
- const mapped = `call_${id.slice(5)}`;
919
+ const mapped = `call_${id.slice(6)}`;
906
920
  idMap.set(id, mapped);
907
921
  return mapped;
908
922
  }
@@ -1513,6 +1527,12 @@ async function* runStream(options) {
1513
1527
  statusCode: 504
1514
1528
  });
1515
1529
  }
1530
+ if (stopReason === null) {
1531
+ throw new ProviderError("anthropic", "Stream ended before completion (no stop_reason).", {
1532
+ statusCode: 504,
1533
+ cause: { partialContent: contentParts, outputTokens }
1534
+ });
1535
+ }
1516
1536
  const normalizedStop = normalizeAnthropicStopReason(stopReason);
1517
1537
  const response = {
1518
1538
  message: {
@@ -1779,6 +1799,17 @@ function getEnvironment() {
1779
1799
  }
1780
1800
 
1781
1801
  // src/providers/openai.ts
1802
+ function toKimiK3Effort(level) {
1803
+ switch (level) {
1804
+ case "low":
1805
+ return "low";
1806
+ case "medium":
1807
+ case "high":
1808
+ return "high";
1809
+ default:
1810
+ return "max";
1811
+ }
1812
+ }
1782
1813
  function extractOpenAIUsage(usage) {
1783
1814
  let cacheRead = 0;
1784
1815
  let cacheWrite = 0;
@@ -1833,7 +1864,12 @@ async function* runStream2(options) {
1833
1864
  const providerName = options.provider ?? "openai";
1834
1865
  const useStreaming = options.streaming !== false;
1835
1866
  const client = createClient2(options);
1836
- const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" || options.provider === "xiaomi";
1867
+ const isKimiK3 = options.provider === "moonshot" && options.model === "kimi-k3";
1868
+ const isManagedKimiK3 = isKimiK3 && options.baseUrl?.replace(/\/+$/, "").endsWith("/coding/v1") === true;
1869
+ const k3Effort = options.thinking ? toKimiK3Effort(options.thinking) : void 0;
1870
+ const isKimiK27 = options.provider === "moonshot" && options.model.startsWith("kimi-k2.7-code");
1871
+ const hasFixedKimiSampling = isKimiK3 || isKimiK27;
1872
+ const usesThinkingParam = options.provider === "glm" || options.provider === "moonshot" && !isKimiK3 && !isKimiK27 || options.provider === "xiaomi";
1837
1873
  const downgradedImages = downgradeUnsupportedImages(options.messages, options.supportsImages);
1838
1874
  const downgradedMessages = downgradeUnsupportedVideos(downgradedImages, options.supportsVideo);
1839
1875
  if (options.provider === "moonshot") {
@@ -1845,7 +1881,11 @@ async function* runStream2(options) {
1845
1881
  }
1846
1882
  const messages = toOpenAIMessages(downgradedMessages, {
1847
1883
  provider: options.provider,
1848
- thinking: !!options.thinking,
1884
+ // K2.7 preserves reasoning even when the user hides thinking in the UI;
1885
+ // keep assistant tool-call history wire-valid in that display mode. A
1886
+ // disabled K3 must NOT carry placeholder reasoning_content (mirrors the
1887
+ // official CLI: reasoning is preserved only while thinking is enabled).
1888
+ thinking: isKimiK27 || !!options.thinking,
1849
1889
  supportsImages: options.supportsImages
1850
1890
  });
1851
1891
  const defaultTemp = options.provider === "glm" ? 0.6 : void 0;
@@ -1855,10 +1895,10 @@ async function* runStream2(options) {
1855
1895
  messages,
1856
1896
  stream: useStreaming,
1857
1897
  ...options.maxTokens ? { max_completion_tokens: options.maxTokens } : {},
1858
- ...effectiveTemp != null && !options.thinking ? { temperature: effectiveTemp } : {},
1859
- ...options.topP != null ? { top_p: options.topP } : {},
1898
+ ...effectiveTemp != null && !options.thinking && !hasFixedKimiSampling ? { temperature: effectiveTemp } : {},
1899
+ ...options.topP != null && !hasFixedKimiSampling ? { top_p: options.topP } : {},
1860
1900
  ...options.stop ? { stop: options.stop } : {},
1861
- ...options.thinking && !usesThinkingParam ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1901
+ ...options.thinking && !usesThinkingParam && !isKimiK3 && !isKimiK27 ? { reasoning_effort: toOpenAIReasoningEffort(options.thinking, options.model) } : {},
1862
1902
  ...options.tools?.length ? { tools: toOpenAITools(options.tools) } : {},
1863
1903
  ...options.toolChoice && options.tools?.length ? { tool_choice: toOpenAIToolChoice(options.toolChoice) } : {},
1864
1904
  ...useStreaming ? { stream_options: { include_usage: true } } : {}
@@ -1868,13 +1908,23 @@ async function* runStream2(options) {
1868
1908
  paramsAny.prompt_cache_key = normalizePromptCacheKey(options.promptCacheKey ?? "ezcoder");
1869
1909
  if (options.provider === "openai" && options.model.startsWith("gpt-5.6")) {
1870
1910
  paramsAny.prompt_cache_options = { mode: "implicit", ttl: "30m" };
1871
- } else if ((options.cacheRetention ?? "short") === "long") {
1911
+ } else if (!isKimiK3 && (options.cacheRetention ?? "short") === "long") {
1872
1912
  paramsAny.prompt_cache_retention = "24h";
1873
1913
  }
1874
1914
  }
1875
1915
  if (options.provider === "openai" && options.serviceTier) {
1876
1916
  params.service_tier = options.serviceTier;
1877
1917
  }
1918
+ if (isKimiK3) {
1919
+ const paramsAny = params;
1920
+ if (isManagedKimiK3) {
1921
+ paramsAny.thinking = k3Effort ? { type: "enabled", effort: k3Effort, keep: "all" } : { type: "disabled" };
1922
+ } else if (k3Effort) {
1923
+ paramsAny.reasoning_effort = k3Effort;
1924
+ } else {
1925
+ paramsAny.thinking = { type: "disabled" };
1926
+ }
1927
+ }
1878
1928
  if (usesThinkingParam) {
1879
1929
  if (options.thinking) {
1880
1930
  params.thinking = { type: "enabled" };
@@ -1978,6 +2028,12 @@ async function* runStream2(options) {
1978
2028
  statusCode: 504
1979
2029
  });
1980
2030
  }
2031
+ if (finishReason === null) {
2032
+ throw new ProviderError(providerName, "Stream ended before completion (no finish_reason).", {
2033
+ statusCode: 504,
2034
+ cause: { partialText: textAccum, outputTokens }
2035
+ });
2036
+ }
1981
2037
  if (thinkingAccum) {
1982
2038
  contentParts.push({ type: "thinking", text: thinkingAccum });
1983
2039
  }
@@ -2170,6 +2226,7 @@ function toError2(err, provider = "openai") {
2170
2226
 
2171
2227
  // src/providers/openai-codex.ts
2172
2228
  import os from "os";
2229
+ import * as zstd from "@bokuweb/zstd-wasm";
2173
2230
 
2174
2231
  // src/utils/sse.ts
2175
2232
  function parseSseBuffer(buffer) {
@@ -2225,6 +2282,50 @@ function extractRequestIdFromMessage(message) {
2225
2282
  // src/providers/openai-codex.ts
2226
2283
  var DEFAULT_BASE_URL = "https://chatgpt.com/backend-api";
2227
2284
  var CODEX_CLIENT_VERSION = "0.144.1";
2285
+ var CODEX_REQUEST_COMPRESSION_MIN_BYTES = 16 * 1024;
2286
+ var zstdInitPromise;
2287
+ async function encodeCodexRequest(body) {
2288
+ const json = JSON.stringify(body);
2289
+ const raw = new TextEncoder().encode(json);
2290
+ if (raw.byteLength < CODEX_REQUEST_COMPRESSION_MIN_BYTES) {
2291
+ return {
2292
+ body: json,
2293
+ compressed: false,
2294
+ rawBytes: raw.byteLength,
2295
+ encodedBytes: raw.byteLength
2296
+ };
2297
+ }
2298
+ try {
2299
+ zstdInitPromise ??= zstd.init();
2300
+ await zstdInitPromise;
2301
+ const compressed = Uint8Array.from(zstd.compress(raw));
2302
+ if (compressed.byteLength >= raw.byteLength) {
2303
+ return {
2304
+ body: json,
2305
+ compressed: false,
2306
+ rawBytes: raw.byteLength,
2307
+ encodedBytes: raw.byteLength
2308
+ };
2309
+ }
2310
+ return {
2311
+ body: compressed,
2312
+ compressed: true,
2313
+ rawBytes: raw.byteLength,
2314
+ encodedBytes: compressed.byteLength
2315
+ };
2316
+ } catch (error) {
2317
+ providerDiag("codex_request_compression_failed", {
2318
+ error: error instanceof Error ? error.message : String(error),
2319
+ rawBytes: raw.byteLength
2320
+ });
2321
+ return {
2322
+ body: json,
2323
+ compressed: false,
2324
+ rawBytes: raw.byteLength,
2325
+ encodedBytes: raw.byteLength
2326
+ };
2327
+ }
2328
+ }
2228
2329
  function usesResponsesLite(model) {
2229
2330
  return model.startsWith("gpt-5.6-");
2230
2331
  }
@@ -2305,10 +2406,17 @@ async function* runStream3(options) {
2305
2406
  headers["session_id"] = transportSessionId;
2306
2407
  headers["x-client-request-id"] = transportSessionId;
2307
2408
  }
2409
+ const encodedRequest = await encodeCodexRequest(body);
2410
+ if (encodedRequest.compressed) headers["Content-Encoding"] = "zstd";
2411
+ providerDiag("codex_request_body", {
2412
+ rawBytes: encodedRequest.rawBytes,
2413
+ encodedBytes: encodedRequest.encodedBytes,
2414
+ compressed: encodedRequest.compressed
2415
+ });
2308
2416
  const response = await fetch(url, {
2309
2417
  method: "POST",
2310
2418
  headers,
2311
- body: JSON.stringify(body),
2419
+ body: encodedRequest.body,
2312
2420
  signal: options.signal
2313
2421
  });
2314
2422
  if (!response.ok) {
@@ -3347,6 +3455,18 @@ providerRegistry.register("sakana", {
3347
3455
  baseUrl: options.baseUrl ?? "https://api.sakana.ai/v1"
3348
3456
  })
3349
3457
  });
3458
+ providerRegistry.register("xai", {
3459
+ // xAI's public API (console.x.ai key) is OpenAI-compatible — ride the Chat
3460
+ // Completions transport like Moonshot/DeepSeek. Grok reasoning models take
3461
+ // top-level `reasoning_effort` (low/medium/high), which the shared thinking
3462
+ // path already sends. xAI's OAuth path exists but only via the Grok CLI's
3463
+ // private Responses proxy (cli-chat-proxy.grok.com) with reverse-engineered
3464
+ // attribution headers and account-tier gating — intentionally not wired.
3465
+ stream: (options) => streamOpenAI({
3466
+ ...options,
3467
+ baseUrl: options.baseUrl ?? "https://api.x.ai/v1"
3468
+ })
3469
+ });
3350
3470
  providerRegistry.register("minimax", {
3351
3471
  stream: (options) => streamAnthropic({
3352
3472
  ...options,