@prestyj/ai 5.13.0 → 5.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -3,6 +3,8 @@ import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
5
  type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Hugging Face Inference Providers router (OpenAI-compatible). */
7
+ | "huggingface"
6
8
  /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
9
  | "local";
8
10
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
package/dist/index.d.ts CHANGED
@@ -3,6 +3,8 @@ import Anthropic from '@anthropic-ai/sdk';
3
3
  import OpenAI from 'openai';
4
4
 
5
5
  type Provider = "anthropic" | "xiaomi" | "openai" | "gemini" | "glm" | "moonshot" | "minimax" | "deepseek" | "openrouter" | "sakana" | "xai" | "palsu"
6
+ /** Hugging Face Inference Providers router (OpenAI-compatible). */
7
+ | "huggingface"
6
8
  /** Locally hosted OpenAI-compatible server (Ollama, LM Studio, llama.cpp, vLLM). */
7
9
  | "local";
8
10
  type ThinkingLevel = "low" | "medium" | "high" | "xhigh" | "max" | "ultra";
package/dist/index.js CHANGED
@@ -59,6 +59,7 @@ var PROVIDER_DISPLAY = {
59
59
  openrouter: "OpenRouter",
60
60
  sakana: "Sakana",
61
61
  xai: "xAI (Grok)",
62
+ huggingface: "Hugging Face",
62
63
  xiaomi: "Xiaomi (MiMo)",
63
64
  minimax: "MiniMax"
64
65
  };
@@ -126,7 +127,7 @@ function formatError(err) {
126
127
  provider: err.provider,
127
128
  statusCode: err.statusCode,
128
129
  ...err.requestId ? { requestId: err.requestId } : {},
129
- guidance: "Request access via your Anthropic account team (see platform.claude.com/docs/en/about-claude/models/overview), or switch to Claude Fable 5 via the model selector \u2014 same underlying model, generally available."
130
+ guidance: "Request access via your Anthropic account team (see platform.claude.com/docs/en/about-claude/models/overview), or switch to Claude Fable 5.1 via the model selector \u2014 same underlying model, generally available."
130
131
  };
131
132
  }
132
133
  if (isUsageLimitError(err)) {
@@ -1554,7 +1555,15 @@ async function* runStream(options) {
1554
1555
  yield keepalive;
1555
1556
  break;
1556
1557
  }
1557
- // message_stop — loop exits naturally
1558
+ // message_stop — loop exits naturally.
1559
+ //
1560
+ // Deliberately NOT breaking early here. Breaking makes the SDK iterator
1561
+ // run `if (!done) controller.abort()` in its `finally`
1562
+ // (core/streaming.js:97), which tears the connection down instead of
1563
+ // returning it to the keep-alive pool — every turn would then pay a
1564
+ // fresh TLS handshake. Draining to the end is what every other Anthropic
1565
+ // client does, and the stall it guards against is handled by the agent
1566
+ // loop's idle timeout.
1558
1567
  default:
1559
1568
  yield keepalive;
1560
1569
  break;
@@ -2964,6 +2973,7 @@ var CODE_ASSIST_SUPPORTED_MODELS = /* @__PURE__ */ new Set([
2964
2973
  "gemini-3.5-flash",
2965
2974
  "gemini-3-flash",
2966
2975
  "gemini-3.1-flash-lite",
2976
+ "gemini-3.7-flash",
2967
2977
  "gemini-2.5-pro",
2968
2978
  "gemini-2.5-flash",
2969
2979
  "gemma-4-31b-it",
@@ -2986,13 +2996,14 @@ var ACCOUNT_GATED_MODELS = /* @__PURE__ */ new Set([
2986
2996
  "gemini-3-flash",
2987
2997
  "gemini-3.5-flash",
2988
2998
  "gemini-3.1-pro-preview",
2989
- "gemini-3.1-pro-preview-customtools"
2999
+ "gemini-3.1-pro-preview-customtools",
3000
+ "gemini-3.7-flash"
2990
3001
  ]);
2991
3002
  function accountGatedMessage(model) {
2992
3003
  return `Your Google account isn't entitled to "${model}" over Gemini Code Assist OAuth, so the API reports it as not found. This is an account-access limit, not a ezcoder bug.`;
2993
3004
  }
2994
3005
  function accountGatedHint() {
2995
- return `Newer Gemini models (3.5 Flash, 3.1 Pro Preview) are available only to Code Assist Standard/Enterprise accounts with preview/GA access enabled by a cloud admin \u2014 free/personal accounts usually can't call them. Switch to Gemini 3.1 Flash Lite (it works on this account) with /model, or sign in with a Code Assist Standard/Enterprise account that has preview access.`;
3006
+ return `Newer Gemini models (3.7 Flash, 3.5 Flash, 3.1 Pro Preview) are available only to Code Assist Standard/Enterprise accounts with preview/GA access enabled by a cloud admin \u2014 free/personal accounts usually can't call them. Switch to Gemini 3.1 Flash Lite (it works on this account) with /model, or sign in with a Code Assist Standard/Enterprise account that has preview access.`;
2996
3007
  }
2997
3008
  function formatErrorMessage(status, body, model) {
2998
3009
  if (status === 404 && !CODE_ASSIST_SUPPORTED_MODELS.has(model)) {
@@ -3676,6 +3687,18 @@ providerRegistry.register("openrouter", {
3676
3687
  baseUrl: options.baseUrl ?? "https://openrouter.ai/api/v1"
3677
3688
  })
3678
3689
  });
3690
+ providerRegistry.register("huggingface", {
3691
+ // Hugging Face Inference Providers router — one HF token (hf.co/settings/tokens,
3692
+ // "Make calls to Inference Providers" permission) routes to whichever hosted
3693
+ // backend serves each open model. Chat Completions-compatible; model ids are
3694
+ // Hub repo paths ("Qwen/Qwen3-Coder-480B-A35B-Instruct"), optionally with an
3695
+ // ":auto"/":fastest"/":cheapest" provider-selection suffix. Billing follows
3696
+ // each backend's per-token rates on the HF account (small free tier).
3697
+ stream: (options) => streamOpenAI({
3698
+ ...options,
3699
+ baseUrl: options.baseUrl ?? "https://router.huggingface.co/v1"
3700
+ })
3701
+ });
3679
3702
  providerRegistry.register("sakana", {
3680
3703
  // Sakana Fugu is a multi-agent system exposed as a standard LLM through the
3681
3704
  // OpenAI-compatible Sakana API. We ride the Chat Completions transport (the