@prestyj/ai 5.11.0 → 5.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -314,7 +314,7 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
314
314
  * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
315
315
  * the same model name served by two machines stays distinct in the registry.
316
316
  * The server only knows the raw id, so strip the routing prefix here — at the
317
- * one place that talks to the wire. Counterpart to gg-core's
317
+ * one place that talks to the wire. Counterpart to @prestyj/core's
318
318
  * `formatLocalModelId`/`parseLocalModelId`.
319
319
  */
320
320
  declare function localWireModelId(id: string): string;
@@ -383,7 +383,7 @@ declare class ProviderRegistryImpl {
383
383
  declare const providerRegistry: ProviderRegistryImpl;
384
384
 
385
385
  /**
386
- * Error model for gg-ai and downstream consumers.
386
+ * Error model for @prestyj/ai and downstream consumers.
387
387
  *
388
388
  * Every error users see should answer one question: "is this me or them?"
389
389
  * That answer drives whether they retry, switch model, log in, or report a
@@ -450,7 +450,7 @@ declare class ProviderError extends EZCoderAIError {
450
450
  * transient per-minute throttle)? These don't clear with a quick retry — the
451
451
  * user has to wait for the window to reset — so callers must surface them as a
452
452
  * hard stop, not silently retry for minutes. Detected from the canonical
453
- * "usage limit reached" message gg-ai stamps onto the ProviderError.
453
+ * "usage limit reached" message @prestyj/ai stamps onto the ProviderError.
454
454
  */
455
455
  declare function isUsageLimitError(err: unknown): boolean;
456
456
  /**
@@ -521,7 +521,7 @@ declare function sliceTail(text: string, chars: number): string;
521
521
  declare function sanitizeMessagesForWire(messages: Message[]): Message[];
522
522
 
523
523
  /**
524
- * Provider-level diagnostic hook. Mirrors the pattern used by gg-agent's
524
+ * Provider-level diagnostic hook. Mirrors the pattern used by @prestyj/agent's
525
525
  * setStreamDiagnostic — the host app wires a callback (typically writing to
526
526
  * a debug log) and providers call `providerDiag(...)` to record interesting
527
527
  * lifecycle events (e.g. raw SSE event types and timings).
package/dist/index.d.ts CHANGED
@@ -314,7 +314,7 @@ declare class StreamResult implements AsyncIterable<StreamEvent> {
314
314
  * Local model ids are namespaced by endpoint (`local/<endpointId>/<rawId>`) so
315
315
  * the same model name served by two machines stays distinct in the registry.
316
316
  * The server only knows the raw id, so strip the routing prefix here — at the
317
- * one place that talks to the wire. Counterpart to gg-core's
317
+ * one place that talks to the wire. Counterpart to @prestyj/core's
318
318
  * `formatLocalModelId`/`parseLocalModelId`.
319
319
  */
320
320
  declare function localWireModelId(id: string): string;
@@ -383,7 +383,7 @@ declare class ProviderRegistryImpl {
383
383
  declare const providerRegistry: ProviderRegistryImpl;
384
384
 
385
385
  /**
386
- * Error model for gg-ai and downstream consumers.
386
+ * Error model for @prestyj/ai and downstream consumers.
387
387
  *
388
388
  * Every error users see should answer one question: "is this me or them?"
389
389
  * That answer drives whether they retry, switch model, log in, or report a
@@ -450,7 +450,7 @@ declare class ProviderError extends EZCoderAIError {
450
450
  * transient per-minute throttle)? These don't clear with a quick retry — the
451
451
  * user has to wait for the window to reset — so callers must surface them as a
452
452
  * hard stop, not silently retry for minutes. Detected from the canonical
453
- * "usage limit reached" message gg-ai stamps onto the ProviderError.
453
+ * "usage limit reached" message @prestyj/ai stamps onto the ProviderError.
454
454
  */
455
455
  declare function isUsageLimitError(err: unknown): boolean;
456
456
  /**
@@ -521,7 +521,7 @@ declare function sliceTail(text: string, chars: number): string;
521
521
  declare function sanitizeMessagesForWire(messages: Message[]): Message[];
522
522
 
523
523
  /**
524
- * Provider-level diagnostic hook. Mirrors the pattern used by gg-agent's
524
+ * Provider-level diagnostic hook. Mirrors the pattern used by @prestyj/agent's
525
525
  * setStreamDiagnostic — the host app wires a callback (typically writing to
526
526
  * a debug log) and providers call `providerDiag(...)` to record interesting
527
527
  * lifecycle events (e.g. raw SSE event types and timings).
package/dist/index.js CHANGED
@@ -1160,7 +1160,7 @@ function parseToolArguments(argsJson) {
1160
1160
  var NON_STREAMING_TIMEOUT_MS = 60 * 60 * 1e3;
1161
1161
  var anthropicClientCache = /* @__PURE__ */ new Map();
1162
1162
  function fineGrainedToolStreamingEnabled() {
1163
- const raw = process.env.GG_FINE_GRAINED_TOOL_STREAMING ?? process.env.CLAUDE_CODE_ENABLE_FINE_GRAINED_TOOL_STREAMING;
1163
+ const raw = process.env.EZ_FINE_GRAINED_TOOL_STREAMING ?? process.env.CLAUDE_CODE_ENABLE_FINE_GRAINED_TOOL_STREAMING;
1164
1164
  if (!raw) return false;
1165
1165
  const v = raw.trim().toLowerCase();
1166
1166
  return v === "1" || v === "true" || v === "yes" || v === "on";
@@ -2025,7 +2025,15 @@ async function* runStream2(options) {
2025
2025
  if (chunk.usage) {
2026
2026
  ({ inputTokens, outputTokens, cacheRead, cacheWrite } = extractOpenAIUsage(chunk.usage));
2027
2027
  }
2028
- if (!choice) continue;
2028
+ if (!choice) {
2029
+ const gatewayError = classifyChoicelessFrame(chunk);
2030
+ if (gatewayError) {
2031
+ throw new ProviderError(providerName, gatewayError.message, {
2032
+ statusCode: gatewayError.statusCode
2033
+ });
2034
+ }
2035
+ continue;
2036
+ }
2029
2037
  if (choice.finish_reason) {
2030
2038
  finishReason = choice.finish_reason;
2031
2039
  }
@@ -2209,6 +2217,31 @@ function completionToResponse(completion, endpointKey) {
2209
2217
  }
2210
2218
  };
2211
2219
  }
2220
+ function classifyChoicelessFrame(frame) {
2221
+ if (!frame || typeof frame !== "object" || Array.isArray(frame)) return null;
2222
+ const rec = frame;
2223
+ if (Array.isArray(rec.choices)) return null;
2224
+ const statusOf = (value) => {
2225
+ const n = typeof value === "string" ? Number(value) : value;
2226
+ return typeof n === "number" && Number.isFinite(n) && n >= 400 && n <= 599 ? n : void 0;
2227
+ };
2228
+ const statusCode = statusOf(rec.status) ?? statusOf(rec.statusCode) ?? statusOf(rec.code);
2229
+ const typeIsError = typeof rec.type === "string" && rec.type.toLowerCase() === "error";
2230
+ let detailText;
2231
+ const detail = rec.detail;
2232
+ if (typeof detail === "string" && detail.trim()) {
2233
+ detailText = detail.trim();
2234
+ } else if (Array.isArray(detail)) {
2235
+ const parts = detail.map(
2236
+ (d) => d && typeof d === "object" && typeof d.msg === "string" ? d.msg : typeof d === "string" ? d : ""
2237
+ ).filter(Boolean);
2238
+ if (parts.length) detailText = parts.join("; ");
2239
+ }
2240
+ if (statusCode === void 0 && !typeIsError && !detailText) return null;
2241
+ const rawMessage = (typeof rec.message === "string" && rec.message.trim() ? rec.message.trim() : void 0) ?? detailText ?? (typeof rec.error === "string" && rec.error.trim() ? rec.error.trim() : void 0) ?? (statusCode !== void 0 ? `Gateway returned status ${statusCode}.` : "Gateway error.");
2242
+ const message = rawMessage.slice(0, 500);
2243
+ return { message, statusCode };
2244
+ }
2212
2245
  function classifyOpenAICompatLimit(args) {
2213
2246
  const { status, code, type, message } = args;
2214
2247
  const codeType = `${code ?? ""} ${type ?? ""}`.toLowerCase();
@@ -2220,6 +2253,7 @@ function classifyOpenAICompatLimit(args) {
2220
2253
  return null;
2221
2254
  }
2222
2255
  function toError2(err, provider = "openai") {
2256
+ if (err instanceof ProviderError) return err;
2223
2257
  if (err instanceof OpenAI.APIError) {
2224
2258
  const body = err.error;
2225
2259
  const bodyMessage = typeof body?.message === "string" && body.message.trim() ? body.message.trim() : void 0;
@@ -3580,6 +3614,8 @@ function sanitizeMessagesForWire(messages) {
3580
3614
  // src/stream.ts
3581
3615
  var GLM_CODING_BASE_URL = "https://api.z.ai/api/coding/paas/v4";
3582
3616
  var KIMI_CODE_USER_AGENT = `kimi-code-cli/${process.env.KIMI_CODE_VERSION ?? "1.0.11"}`;
3617
+ var GROK_CLI_PROXY_HOST = "cli-chat-proxy.grok.com";
3618
+ var GROK_CLI_VERSION = process.env.GROK_CLI_VERSION ?? "0.2.101";
3583
3619
  providerRegistry.register("anthropic", {
3584
3620
  stream: (options) => streamAnthropic(options)
3585
3621
  });
@@ -3646,13 +3682,25 @@ providerRegistry.register("xai", {
3646
3682
  // xAI's public API (console.x.ai key) is OpenAI-compatible — ride the Chat
3647
3683
  // Completions transport like Moonshot/DeepSeek. Grok reasoning models take
3648
3684
  // top-level `reasoning_effort` (low/medium/high), which the shared thinking
3649
- // path already sends. xAI's OAuth path exists but only via the Grok CLI's
3650
- // private Responses proxy (cli-chat-proxy.grok.com) with reverse-engineered
3651
- // attribution headers and account-tier gating — intentionally not wired.
3652
- stream: (options) => streamOpenAI({
3653
- ...options,
3654
- baseUrl: options.baseUrl ?? "https://api.x.ai/v1"
3655
- })
3685
+ // path already sends.
3686
+ //
3687
+ // Subscription OAuth (SuperGrok / X Premium) routes to the Grok CLI chat proxy
3688
+ // instead, which speaks the same Chat Completions wire but gates on Grok-CLI
3689
+ // client identity. Inject those headers centrally here — exactly as the Kimi
3690
+ // endpoint above — so EVERY stream (agent loop, compaction, title-gen,
3691
+ // sub-agents) is accepted rather than depending on each call site to thread
3692
+ // headers. Caller-provided headers still win on collision.
3693
+ stream: (options) => {
3694
+ const baseUrl = options.baseUrl ?? "https://api.x.ai/v1";
3695
+ const defaultHeaders = baseUrl.includes(GROK_CLI_PROXY_HOST) ? {
3696
+ "X-XAI-Token-Auth": "xai-grok-cli",
3697
+ "x-grok-client-version": GROK_CLI_VERSION,
3698
+ "x-grok-client-identifier": "ezcoder",
3699
+ "x-grok-model-override": options.model,
3700
+ ...options.defaultHeaders
3701
+ } : options.defaultHeaders;
3702
+ return streamOpenAI({ ...options, baseUrl, defaultHeaders });
3703
+ }
3656
3704
  });
3657
3705
  providerRegistry.register("minimax", {
3658
3706
  stream: (options) => streamAnthropic({