@prestyj/core 5.13.0 → 5.16.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -1605,6 +1605,7 @@ var STATIC_API_KEY_PROVIDERS = /* @__PURE__ */ new Set([
1605
1605
  "minimax",
1606
1606
  "deepseek",
1607
1607
  "openrouter",
1608
+ "huggingface",
1608
1609
  "sakana",
1609
1610
  "xai",
1610
1611
  // Local endpoints: a fixed (usually placeholder) key, never refreshable.
@@ -2040,8 +2041,13 @@ var MODELS = [
2040
2041
  // Project Glasswing (limited, invitation-only) model unavailable to most
2041
2042
  // users. Re-enable once it's generally available.
2042
2043
  {
2043
- id: "claude-fable-5",
2044
- name: "Claude Fable 5",
2044
+ // Released 2026-09-01 — replaces Fable 5 at the same $10/$50 MTok (cache
2045
+ // reads drop to $0.25). Always-on adaptive thinking steered by effort;
2046
+ // forced tool use is rejected with a 400, which @prestyj/ai never sends on the
2047
+ // Anthropic path. Fable 5 is retired here — a session that still has it
2048
+ // saved falls back to the provider default on next start.
2049
+ id: "claude-fable-5-1",
2050
+ name: "Claude Fable 5.1",
2045
2051
  provider: "anthropic",
2046
2052
  contextWindow: 1e6,
2047
2053
  maxOutputTokens: 128e3,
@@ -2053,7 +2059,7 @@ var MODELS = [
2053
2059
  },
2054
2060
  // {
2055
2061
  // // Mythos-class model offered through Project Glasswing (limited
2056
- // // availability, invitation-only). Same underlying model as Fable 5 with
2062
+ // // availability, invitation-only). Same underlying model as Fable 5.1 with
2057
2063
  // // some safeguards lifted; kept here so approved accounts can select it.
2058
2064
  // id: "claude-mythos-5",
2059
2065
  // name: "Claude Mythos 5",
@@ -2204,11 +2210,29 @@ var MODELS = [
2204
2210
  maxThinkingLevel: "xhigh"
2205
2211
  },
2206
2212
  // ── xAI (Grok) ─────────────────────────────────────────
2207
- // Grok 4.5 (released 2026-07-08) is xAI's flagship for coding, agentic
2208
- // tasks, and knowledge work — 500K context, text+image input, configurable
2209
- // `reasoning_effort` (low/medium/high, server default high; reasoning can't
2210
- // be fully disabled). Served over the OpenAI-compatible API at
2211
- // https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
2213
+ // Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
2214
+ // and knowledge work, with a focus on long-running agents — 500K context,
2215
+ // text+image input, and a `reasoning_effort` ladder that adds a new `xhigh`
2216
+ // top rung (low/medium/high default/xhigh; reasoning still can't be fully
2217
+ // disabled). $2/$6 per MTok under 200K prompt tokens ($4/$12 at or above),
2218
+ // and it's the default model of the Grok Build coding agent. xAI advertises "no text output limit"; we keep the same
2219
+ // 131K practical cap as 4.5 for budget predictability and input headroom.
2220
+ {
2221
+ id: "grok-4.6",
2222
+ name: "Grok 4.6",
2223
+ provider: "xai",
2224
+ contextWindow: 5e5,
2225
+ maxOutputTokens: 131072,
2226
+ supportsThinking: true,
2227
+ supportsImages: true,
2228
+ supportsVideo: false,
2229
+ costTier: "medium",
2230
+ maxThinkingLevel: "xhigh"
2231
+ },
2232
+ // Grok 4.5 (released 2026-07-08) — superseded by 4.6 but retained as an explicit option. 500K context, text+image input,
2233
+ // configurable `reasoning_effort` (low/medium/high, server default high;
2234
+ // reasoning can't be fully disabled). Served over the OpenAI-compatible API
2235
+ // at https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
2212
2236
  // official max-output cap for 4.5; 131K matches the Grok Responses ceiling
2213
2237
  // third-party integrations use.
2214
2238
  {
@@ -2223,7 +2247,7 @@ var MODELS = [
2223
2247
  costTier: "medium",
2224
2248
  maxThinkingLevel: "high"
2225
2249
  },
2226
- // ── Gemini ─────────────────────────────────────────────
2250
+ // ── Gemini ─────────────────────────────────────────
2227
2251
  {
2228
2252
  id: "gemini-3.1-flash-lite",
2229
2253
  name: "Gemini 3.1 Flash Lite",
@@ -2237,6 +2261,28 @@ var MODELS = [
2237
2261
  costTier: "low",
2238
2262
  maxThinkingLevel: "high"
2239
2263
  },
2264
+ {
2265
+ // Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
2266
+ // coding, agents, and multi-step execution; GA-stable on the Gemini API as
2267
+ // `gemini-3.7-flash`. 1M context, 64K output, thinking low/medium/high.
2268
+ // Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
2269
+ // hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
2270
+ // free/personal accounts 404 (entitlement-gated) while Code Assist
2271
+ // Standard/Enterprise accounts get it. Listed SECOND, after flash-lite:
2272
+ // getFastModel picks the first low-tier entry, and flash-lite is the one
2273
+ // that works on every account.
2274
+ id: "gemini-3.7-flash",
2275
+ name: "Gemini 3.7 Flash",
2276
+ provider: "gemini",
2277
+ contextWindow: 1048576,
2278
+ maxOutputTokens: 65536,
2279
+ supportsThinking: true,
2280
+ supportsImages: true,
2281
+ supportsVideo: true,
2282
+ maxVideoBytes: 20 * 1024 * 1024,
2283
+ costTier: "low",
2284
+ maxThinkingLevel: "high"
2285
+ },
2240
2286
  {
2241
2287
  // Wire name `gemini-3-flash` — the Code Assist (OAuth) backend rejects the
2242
2288
  // display string `gemini-3.5-flash` with a 404, so gemini-cli keeps this
@@ -2304,13 +2350,12 @@ var MODELS = [
2304
2350
  maxThinkingLevel: "high"
2305
2351
  },
2306
2352
  // ── Z.AI (GLM) ─────────────────────────────────────────
2307
- // GLM-5.3 is the only GLM entry: it supersedes 5.2 (same GLM-5 base, all
2308
- // gains from post-training) and the coding endpoint already answers
2309
- // `glm-5.2` requests as glm-5.3, so the older ids were menu clutter that
2310
- // routed to strictly worse coding for the same plan quota.
2311
- // Released 2026-08-14; live on the coding endpoint (verified), while the
2312
- // standard paas API is still "coming soon". `max` is both the ceiling and
2313
- // Z.AI's own default — the rungs below it live in thinking-level.ts.
2353
+ // Two GLM entries, both live on the coding endpoint (verified against its
2354
+ // /models list). The pre-5.3 ids stay retired: they routed to strictly worse
2355
+ // coding for the same plan quota, and the endpoint already answers `glm-5.2`
2356
+ // requests as glm-5.3.
2357
+ // `max` is both the ceiling and Z.AI's own default — the rungs below it live
2358
+ // in thinking-level.ts.
2314
2359
  {
2315
2360
  id: "glm-5.3",
2316
2361
  name: "GLM-5.3",
@@ -2323,6 +2368,30 @@ var MODELS = [
2323
2368
  costTier: "medium",
2324
2369
  maxThinkingLevel: "max"
2325
2370
  },
2371
+ // GLM-5.3-Flash (released 2026-08-26): 320B-A18B natively multimodal sibling
2372
+ // at ~1/20th of 5.3's API price with 3× the coding-plan quota, so it is the
2373
+ // provider's `low` tier — scout sub-agents and compaction summaries route
2374
+ // here instead of paying 5.3 rates.
2375
+ // Images are native on the coding endpoint (verified: base64 data URL in an
2376
+ // `image_url` block answers correctly), which also means GLM image
2377
+ // attachments go inline for this model rather than through the zai_vision MCP
2378
+ // detour that `supportsImages: false` triggers.
2379
+ // Video/file input is documented but unverified on this transport, so it
2380
+ // stays off until measured. Thinking cannot be disabled server-side (Z.AI
2381
+ // maps a `disabled` toggle to the `low` rung and answers 200), and unlike
2382
+ // 5.3 it accepts any reasoning_effort string without a 400.
2383
+ {
2384
+ id: "glm-5.3-flash",
2385
+ name: "GLM-5.3-Flash",
2386
+ provider: "glm",
2387
+ contextWindow: 1e6,
2388
+ maxOutputTokens: 131072,
2389
+ supportsThinking: true,
2390
+ supportsImages: true,
2391
+ supportsVideo: false,
2392
+ costTier: "low",
2393
+ maxThinkingLevel: "max"
2394
+ },
2326
2395
  // ── MiniMax ────────────────────────────────────────────
2327
2396
  {
2328
2397
  id: "MiniMax-M3",
@@ -2389,15 +2458,21 @@ var MODELS = [
2389
2458
  },
2390
2459
  // ── DeepSeek ───────────────────────────────────────────
2391
2460
  {
2461
+ // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
2462
+ // first STABLE V4 Pro — supersedes the April preview; calling name
2463
+ // unchanged, same 1.6T/49B MoE). 1M context, 384K (393,216) max output,
2464
+ // text-only, reasoning ladder low/high plus Think Max — mapped from our
2465
+ // `xhigh`. ~$0.43/$0.87 per MTok on DeepSeek's own API, so a mid-tier
2466
+ // price band rather than the preview's top band.
2392
2467
  id: "deepseek-v4-pro",
2393
2468
  name: "DeepSeek V4 Pro",
2394
2469
  provider: "deepseek",
2395
2470
  contextWindow: 1048576,
2396
- maxOutputTokens: 384e3,
2471
+ maxOutputTokens: 393216,
2397
2472
  supportsThinking: true,
2398
2473
  supportsImages: false,
2399
2474
  supportsVideo: false,
2400
- costTier: "high",
2475
+ costTier: "medium",
2401
2476
  // DeepSeek V4 maps `xhigh` → its internal `max` tier.
2402
2477
  maxThinkingLevel: "xhigh"
2403
2478
  },
@@ -2425,6 +2500,45 @@ var MODELS = [
2425
2500
  supportsVideo: false,
2426
2501
  costTier: "medium",
2427
2502
  maxThinkingLevel: "high"
2503
+ },
2504
+ // ── Hugging Face (Inference Providers router) ────────
2505
+ // One HF token (hf.co/settings/tokens, "Make calls to Inference Providers"
2506
+ // permission) routes to whichever hosted backend serves each open model;
2507
+ // billing follows each backend's rates on the HF account (small free tier).
2508
+ // Model ids are Hub repo paths, so they intentionally contain a slash — the
2509
+ // same shape local/ vLLM ids already use (`local/vllm/Qwen/Qwen3-32B`).
2510
+ {
2511
+ // Qwen's open flagship for agentic coding — tool-calling native, non-thinking
2512
+ // (the Coder line dropped the <think> block). 262K native context (1M needs
2513
+ // YaRN, which the router doesn't apply), 131K max output. :auto suffix lets
2514
+ // HF pick the backend with capacity; we keep the bare repo id so the picker
2515
+ // matches what GET /v1/models reports.
2516
+ id: "Qwen/Qwen3-Coder-480B-A35B-Instruct",
2517
+ name: "Qwen3 Coder 480B",
2518
+ provider: "huggingface",
2519
+ contextWindow: 262144,
2520
+ maxOutputTokens: 131072,
2521
+ supportsThinking: false,
2522
+ supportsImages: false,
2523
+ supportsVideo: false,
2524
+ costTier: "medium",
2525
+ maxThinkingLevel: "low"
2526
+ },
2527
+ {
2528
+ // OpenAI's open-weight 120B MoE (5.1B active) — general-purpose, tool-calling
2529
+ // native, adjustable reasoning effort (low/medium/high, default medium) over
2530
+ // the router's Chat Completions API. Cheap enough to be the low-tier sibling
2531
+ // for summaries and fast sub-agents.
2532
+ id: "openai/gpt-oss-120b",
2533
+ name: "GPT-OSS 120B",
2534
+ provider: "huggingface",
2535
+ contextWindow: 131072,
2536
+ maxOutputTokens: 65536,
2537
+ supportsThinking: true,
2538
+ supportsImages: false,
2539
+ supportsVideo: false,
2540
+ costTier: "low",
2541
+ maxThinkingLevel: "high"
2428
2542
  }
2429
2543
  ];
2430
2544
  var runtimeModels = /* @__PURE__ */ new Map();
@@ -2470,9 +2584,11 @@ function getDefaultModel(provider) {
2470
2584
  if (provider === "moonshot") return MODELS.find((m) => m.id === "kimi-k3");
2471
2585
  if (provider === "minimax") return MODELS.find((m) => m.id === "MiniMax-M3");
2472
2586
  if (provider === "deepseek") return MODELS.find((m) => m.id === "deepseek-v4-pro");
2587
+ if (provider === "huggingface")
2588
+ return MODELS.find((m) => m.id === "Qwen/Qwen3-Coder-480B-A35B-Instruct");
2473
2589
  if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.6-plus");
2474
2590
  if (provider === "sakana") return MODELS.find((m) => m.id === "fugu");
2475
- if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.5");
2591
+ if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.6");
2476
2592
  if (provider === "local") {
2477
2593
  return getModelsForProvider("local")[0] ?? PLACEHOLDER_LOCAL_MODEL;
2478
2594
  }
@@ -2516,7 +2632,7 @@ function getSummaryModel(provider, currentModelId) {
2516
2632
  if (provider === "anthropic") {
2517
2633
  return MODELS.find((m) => m.id === "claude-sonnet-5");
2518
2634
  }
2519
- if (provider === "openai" || provider === "glm" || provider === "deepseek") {
2635
+ if (provider === "openai" || provider === "glm" || provider === "deepseek" || provider === "huggingface") {
2520
2636
  const low = getModelsForProvider(provider).find((m) => m.costTier === "low");
2521
2637
  if (low) return low;
2522
2638
  }
@@ -2538,7 +2654,7 @@ var OPENAI_GPT_56_THINKING_LEVELS = [
2538
2654
  "ultra"
2539
2655
  ];
2540
2656
  var SAKANA_THINKING_LEVELS = ["high", "xhigh"];
2541
- var XAI_THINKING_LEVELS = ["low", "medium", "high"];
2657
+ var XAI_THINKING_LEVELS = ["low", "medium", "high", "xhigh"];
2542
2658
  var ANTHROPIC_XHIGH_THINKING_LEVELS = [
2543
2659
  "low",
2544
2660
  "medium",
@@ -2633,10 +2749,7 @@ function getNextThinkingLevel(provider, model, current) {
2633
2749
 
2634
2750
  // src/local-models.ts
2635
2751
  var DEFAULT_LOCAL_ENDPOINTS = [
2636
- { id: "ollama", label: "Ollama", baseUrl: "http://127.0.0.1:11434/v1", kind: "ollama" },
2637
- { id: "lmstudio", label: "LM Studio", baseUrl: "http://127.0.0.1:1234/v1", kind: "lmstudio" },
2638
- { id: "llamacpp", label: "llama.cpp", baseUrl: "http://127.0.0.1:8080/v1", kind: "llamacpp" },
2639
- { id: "vllm", label: "vLLM", baseUrl: "http://127.0.0.1:8000/v1", kind: "vllm" }
2752
+ { id: "ollama", label: "Ollama", baseUrl: "http://127.0.0.1:11434/v1", kind: "ollama" }
2640
2753
  ];
2641
2754
  var FALLBACK_CONTEXT_WINDOW = 8192;
2642
2755
  var LOCAL_API_KEY_PLACEHOLDER = "local";