@prestyj/core 5.13.0 → 5.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1464,6 +1464,7 @@ var STATIC_API_KEY_PROVIDERS = /* @__PURE__ */ new Set([
1464
1464
  "minimax",
1465
1465
  "deepseek",
1466
1466
  "openrouter",
1467
+ "huggingface",
1467
1468
  "sakana",
1468
1469
  "xai",
1469
1470
  // Local endpoints: a fixed (usually placeholder) key, never refreshable.
@@ -1899,8 +1900,13 @@ var MODELS = [
1899
1900
  // Project Glasswing (limited, invitation-only) model unavailable to most
1900
1901
  // users. Re-enable once it's generally available.
1901
1902
  {
1902
- id: "claude-fable-5",
1903
- name: "Claude Fable 5",
1903
+ // Released 2026-09-01 — replaces Fable 5 at the same $10/$50 MTok (cache
1904
+ // reads drop to $0.25). Always-on adaptive thinking steered by effort;
1905
+ // forced tool use is rejected with a 400, which @prestyj/ai never sends on the
1906
+ // Anthropic path. Fable 5 is retired here — a session that still has it
1907
+ // saved falls back to the provider default on next start.
1908
+ id: "claude-fable-5-1",
1909
+ name: "Claude Fable 5.1",
1904
1910
  provider: "anthropic",
1905
1911
  contextWindow: 1e6,
1906
1912
  maxOutputTokens: 128e3,
@@ -1912,7 +1918,7 @@ var MODELS = [
1912
1918
  },
1913
1919
  // {
1914
1920
  // // Mythos-class model offered through Project Glasswing (limited
1915
- // // availability, invitation-only). Same underlying model as Fable 5 with
1921
+ // // availability, invitation-only). Same underlying model as Fable 5.1 with
1916
1922
  // // some safeguards lifted; kept here so approved accounts can select it.
1917
1923
  // id: "claude-mythos-5",
1918
1924
  // name: "Claude Mythos 5",
@@ -2063,11 +2069,29 @@ var MODELS = [
2063
2069
  maxThinkingLevel: "xhigh"
2064
2070
  },
2065
2071
  // ── xAI (Grok) ─────────────────────────────────────────
2066
- // Grok 4.5 (released 2026-07-08) is xAI's flagship for coding, agentic
2067
- // tasks, and knowledge work — 500K context, text+image input, configurable
2068
- // `reasoning_effort` (low/medium/high, server default high; reasoning can't
2069
- // be fully disabled). Served over the OpenAI-compatible API at
2070
- // https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
2072
+ // Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
2073
+ // and knowledge work, with a focus on long-running agents — 500K context,
2074
+ // text+image input, and a `reasoning_effort` ladder that adds a new `xhigh`
2075
+ // top rung (low/medium/high default/xhigh; reasoning still can't be fully
2076
+ // disabled). $2/$6 per MTok under 200K prompt tokens ($4/$12 at or above),
2077
+ // and it's the default model of the Grok Build coding agent. xAI advertises "no text output limit"; we keep the same
2078
+ // 131K practical cap as 4.5 for budget predictability and input headroom.
2079
+ {
2080
+ id: "grok-4.6",
2081
+ name: "Grok 4.6",
2082
+ provider: "xai",
2083
+ contextWindow: 5e5,
2084
+ maxOutputTokens: 131072,
2085
+ supportsThinking: true,
2086
+ supportsImages: true,
2087
+ supportsVideo: false,
2088
+ costTier: "medium",
2089
+ maxThinkingLevel: "xhigh"
2090
+ },
2091
+ // Grok 4.5 (released 2026-07-08) — superseded by 4.6 but retained as an explicit option. 500K context, text+image input,
2092
+ // configurable `reasoning_effort` (low/medium/high, server default high;
2093
+ // reasoning can't be fully disabled). Served over the OpenAI-compatible API
2094
+ // at https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
2071
2095
  // official max-output cap for 4.5; 131K matches the Grok Responses ceiling
2072
2096
  // third-party integrations use.
2073
2097
  {
@@ -2082,7 +2106,7 @@ var MODELS = [
2082
2106
  costTier: "medium",
2083
2107
  maxThinkingLevel: "high"
2084
2108
  },
2085
- // ── Gemini ─────────────────────────────────────────────
2109
+ // ── Gemini ─────────────────────────────────────────
2086
2110
  {
2087
2111
  id: "gemini-3.1-flash-lite",
2088
2112
  name: "Gemini 3.1 Flash Lite",
@@ -2096,6 +2120,28 @@ var MODELS = [
2096
2120
  costTier: "low",
2097
2121
  maxThinkingLevel: "high"
2098
2122
  },
2123
+ {
2124
+ // Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
2125
+ // coding, agents, and multi-step execution; GA-stable on the Gemini API as
2126
+ // `gemini-3.7-flash`. 1M context, 64K output, thinking low/medium/high.
2127
+ // Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
2128
+ // hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
2129
+ // free/personal accounts 404 (entitlement-gated) while Code Assist
2130
+ // Standard/Enterprise accounts get it. Listed SECOND, after flash-lite:
2131
+ // getFastModel picks the first low-tier entry, and flash-lite is the one
2132
+ // that works on every account.
2133
+ id: "gemini-3.7-flash",
2134
+ name: "Gemini 3.7 Flash",
2135
+ provider: "gemini",
2136
+ contextWindow: 1048576,
2137
+ maxOutputTokens: 65536,
2138
+ supportsThinking: true,
2139
+ supportsImages: true,
2140
+ supportsVideo: true,
2141
+ maxVideoBytes: 20 * 1024 * 1024,
2142
+ costTier: "low",
2143
+ maxThinkingLevel: "high"
2144
+ },
2099
2145
  {
2100
2146
  // Wire name `gemini-3-flash` — the Code Assist (OAuth) backend rejects the
2101
2147
  // display string `gemini-3.5-flash` with a 404, so gemini-cli keeps this
@@ -2163,13 +2209,12 @@ var MODELS = [
2163
2209
  maxThinkingLevel: "high"
2164
2210
  },
2165
2211
  // ── Z.AI (GLM) ─────────────────────────────────────────
2166
- // GLM-5.3 is the only GLM entry: it supersedes 5.2 (same GLM-5 base, all
2167
- // gains from post-training) and the coding endpoint already answers
2168
- // `glm-5.2` requests as glm-5.3, so the older ids were menu clutter that
2169
- // routed to strictly worse coding for the same plan quota.
2170
- // Released 2026-08-14; live on the coding endpoint (verified), while the
2171
- // standard paas API is still "coming soon". `max` is both the ceiling and
2172
- // Z.AI's own default — the rungs below it live in thinking-level.ts.
2212
+ // Two GLM entries, both live on the coding endpoint (verified against its
2213
+ // /models list). The pre-5.3 ids stay retired: they routed to strictly worse
2214
+ // coding for the same plan quota, and the endpoint already answers `glm-5.2`
2215
+ // requests as glm-5.3.
2216
+ // `max` is both the ceiling and Z.AI's own default — the rungs below it live
2217
+ // in thinking-level.ts.
2173
2218
  {
2174
2219
  id: "glm-5.3",
2175
2220
  name: "GLM-5.3",
@@ -2182,6 +2227,30 @@ var MODELS = [
2182
2227
  costTier: "medium",
2183
2228
  maxThinkingLevel: "max"
2184
2229
  },
2230
+ // GLM-5.3-Flash (released 2026-08-26): 320B-A18B natively multimodal sibling
2231
+ // at ~1/20th of 5.3's API price with 3× the coding-plan quota, so it is the
2232
+ // provider's `low` tier — scout sub-agents and compaction summaries route
2233
+ // here instead of paying 5.3 rates.
2234
+ // Images are native on the coding endpoint (verified: base64 data URL in an
2235
+ // `image_url` block answers correctly), which also means GLM image
2236
+ // attachments go inline for this model rather than through the zai_vision MCP
2237
+ // detour that `supportsImages: false` triggers.
2238
+ // Video/file input is documented but unverified on this transport, so it
2239
+ // stays off until measured. Thinking cannot be disabled server-side (Z.AI
2240
+ // maps a `disabled` toggle to the `low` rung and answers 200), and unlike
2241
+ // 5.3 it accepts any reasoning_effort string without a 400.
2242
+ {
2243
+ id: "glm-5.3-flash",
2244
+ name: "GLM-5.3-Flash",
2245
+ provider: "glm",
2246
+ contextWindow: 1e6,
2247
+ maxOutputTokens: 131072,
2248
+ supportsThinking: true,
2249
+ supportsImages: true,
2250
+ supportsVideo: false,
2251
+ costTier: "low",
2252
+ maxThinkingLevel: "max"
2253
+ },
2185
2254
  // ── MiniMax ────────────────────────────────────────────
2186
2255
  {
2187
2256
  id: "MiniMax-M3",
@@ -2248,15 +2317,21 @@ var MODELS = [
2248
2317
  },
2249
2318
  // ── DeepSeek ───────────────────────────────────────────
2250
2319
  {
2320
+ // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
2321
+ // first STABLE V4 Pro — supersedes the April preview; calling name
2322
+ // unchanged, same 1.6T/49B MoE). 1M context, 384K (393,216) max output,
2323
+ // text-only, reasoning ladder low/high plus Think Max — mapped from our
2324
+ // `xhigh`. ~$0.43/$0.87 per MTok on DeepSeek's own API, so a mid-tier
2325
+ // price band rather than the preview's top band.
2251
2326
  id: "deepseek-v4-pro",
2252
2327
  name: "DeepSeek V4 Pro",
2253
2328
  provider: "deepseek",
2254
2329
  contextWindow: 1048576,
2255
- maxOutputTokens: 384e3,
2330
+ maxOutputTokens: 393216,
2256
2331
  supportsThinking: true,
2257
2332
  supportsImages: false,
2258
2333
  supportsVideo: false,
2259
- costTier: "high",
2334
+ costTier: "medium",
2260
2335
  // DeepSeek V4 maps `xhigh` → its internal `max` tier.
2261
2336
  maxThinkingLevel: "xhigh"
2262
2337
  },
@@ -2284,6 +2359,45 @@ var MODELS = [
2284
2359
  supportsVideo: false,
2285
2360
  costTier: "medium",
2286
2361
  maxThinkingLevel: "high"
2362
+ },
2363
+ // ── Hugging Face (Inference Providers router) ────────
2364
+ // One HF token (hf.co/settings/tokens, "Make calls to Inference Providers"
2365
+ // permission) routes to whichever hosted backend serves each open model;
2366
+ // billing follows each backend's rates on the HF account (small free tier).
2367
+ // Model ids are Hub repo paths, so they intentionally contain a slash — the
2368
+ // same shape local/ vLLM ids already use (`local/vllm/Qwen/Qwen3-32B`).
2369
+ {
2370
+ // Qwen's open flagship for agentic coding — tool-calling native, non-thinking
2371
+ // (the Coder line dropped the <think> block). 262K native context (1M needs
2372
+ // YaRN, which the router doesn't apply), 131K max output. :auto suffix lets
2373
+ // HF pick the backend with capacity; we keep the bare repo id so the picker
2374
+ // matches what GET /v1/models reports.
2375
+ id: "Qwen/Qwen3-Coder-480B-A35B-Instruct",
2376
+ name: "Qwen3 Coder 480B",
2377
+ provider: "huggingface",
2378
+ contextWindow: 262144,
2379
+ maxOutputTokens: 131072,
2380
+ supportsThinking: false,
2381
+ supportsImages: false,
2382
+ supportsVideo: false,
2383
+ costTier: "medium",
2384
+ maxThinkingLevel: "low"
2385
+ },
2386
+ {
2387
+ // OpenAI's open-weight 120B MoE (5.1B active) — general-purpose, tool-calling
2388
+ // native, adjustable reasoning effort (low/medium/high, default medium) over
2389
+ // the router's Chat Completions API. Cheap enough to be the low-tier sibling
2390
+ // for summaries and fast sub-agents.
2391
+ id: "openai/gpt-oss-120b",
2392
+ name: "GPT-OSS 120B",
2393
+ provider: "huggingface",
2394
+ contextWindow: 131072,
2395
+ maxOutputTokens: 65536,
2396
+ supportsThinking: true,
2397
+ supportsImages: false,
2398
+ supportsVideo: false,
2399
+ costTier: "low",
2400
+ maxThinkingLevel: "high"
2287
2401
  }
2288
2402
  ];
2289
2403
  var runtimeModels = /* @__PURE__ */ new Map();
@@ -2329,9 +2443,11 @@ function getDefaultModel(provider) {
2329
2443
  if (provider === "moonshot") return MODELS.find((m) => m.id === "kimi-k3");
2330
2444
  if (provider === "minimax") return MODELS.find((m) => m.id === "MiniMax-M3");
2331
2445
  if (provider === "deepseek") return MODELS.find((m) => m.id === "deepseek-v4-pro");
2446
+ if (provider === "huggingface")
2447
+ return MODELS.find((m) => m.id === "Qwen/Qwen3-Coder-480B-A35B-Instruct");
2332
2448
  if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.6-plus");
2333
2449
  if (provider === "sakana") return MODELS.find((m) => m.id === "fugu");
2334
- if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.5");
2450
+ if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.6");
2335
2451
  if (provider === "local") {
2336
2452
  return getModelsForProvider("local")[0] ?? PLACEHOLDER_LOCAL_MODEL;
2337
2453
  }
@@ -2375,7 +2491,7 @@ function getSummaryModel(provider, currentModelId) {
2375
2491
  if (provider === "anthropic") {
2376
2492
  return MODELS.find((m) => m.id === "claude-sonnet-5");
2377
2493
  }
2378
- if (provider === "openai" || provider === "glm" || provider === "deepseek") {
2494
+ if (provider === "openai" || provider === "glm" || provider === "deepseek" || provider === "huggingface") {
2379
2495
  const low = getModelsForProvider(provider).find((m) => m.costTier === "low");
2380
2496
  if (low) return low;
2381
2497
  }
@@ -2443,4 +2559,4 @@ export {
2443
2559
  getSummaryModel,
2444
2560
  getFastModel
2445
2561
  };
2446
- //# sourceMappingURL=chunk-S6QBRJ4D.js.map
2562
+ //# sourceMappingURL=chunk-OUE2GRO6.js.map