@prestyj/core 5.13.0 → 5.16.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -132,8 +132,13 @@ var MODELS = [
132
132
  // Project Glasswing (limited, invitation-only) model unavailable to most
133
133
  // users. Re-enable once it's generally available.
134
134
  {
135
- id: "claude-fable-5",
136
- name: "Claude Fable 5",
135
+ // Released 2026-09-01 — replaces Fable 5 at the same $10/$50 MTok (cache
136
+ // reads drop to $0.25). Always-on adaptive thinking steered by effort;
137
+ // forced tool use is rejected with a 400, which @prestyj/ai never sends on the
138
+ // Anthropic path. Fable 5 is retired here — a session that still has it
139
+ // saved falls back to the provider default on next start.
140
+ id: "claude-fable-5-1",
141
+ name: "Claude Fable 5.1",
137
142
  provider: "anthropic",
138
143
  contextWindow: 1e6,
139
144
  maxOutputTokens: 128e3,
@@ -145,7 +150,7 @@ var MODELS = [
145
150
  },
146
151
  // {
147
152
  // // Mythos-class model offered through Project Glasswing (limited
148
- // // availability, invitation-only). Same underlying model as Fable 5 with
153
+ // // availability, invitation-only). Same underlying model as Fable 5.1 with
149
154
  // // some safeguards lifted; kept here so approved accounts can select it.
150
155
  // id: "claude-mythos-5",
151
156
  // name: "Claude Mythos 5",
@@ -296,11 +301,29 @@ var MODELS = [
296
301
  maxThinkingLevel: "xhigh"
297
302
  },
298
303
  // ── xAI (Grok) ─────────────────────────────────────────
299
- // Grok 4.5 (released 2026-07-08) is xAI's flagship for coding, agentic
300
- // tasks, and knowledge work — 500K context, text+image input, configurable
301
- // `reasoning_effort` (low/medium/high, server default high; reasoning can't
302
- // be fully disabled). Served over the OpenAI-compatible API at
303
- // https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
304
+ // Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
305
+ // and knowledge work, with a focus on long-running agents — 500K context,
306
+ // text+image input, and a `reasoning_effort` ladder that adds a new `xhigh`
307
+ // top rung (low/medium/high default/xhigh; reasoning still can't be fully
308
+ // disabled). $2/$6 per MTok under 200K prompt tokens ($4/$12 at or above),
309
+ // and it's the default model of the Grok Build coding agent. xAI advertises "no text output limit"; we keep the same
310
+ // 131K practical cap as 4.5 for budget predictability and input headroom.
311
+ {
312
+ id: "grok-4.6",
313
+ name: "Grok 4.6",
314
+ provider: "xai",
315
+ contextWindow: 5e5,
316
+ maxOutputTokens: 131072,
317
+ supportsThinking: true,
318
+ supportsImages: true,
319
+ supportsVideo: false,
320
+ costTier: "medium",
321
+ maxThinkingLevel: "xhigh"
322
+ },
323
+ // Grok 4.5 (released 2026-07-08) — superseded by 4.6 but retained as an explicit option. 500K context, text+image input,
324
+ // configurable `reasoning_effort` (low/medium/high, server default high;
325
+ // reasoning can't be fully disabled). Served over the OpenAI-compatible API
326
+ // at https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
304
327
  // official max-output cap for 4.5; 131K matches the Grok Responses ceiling
305
328
  // third-party integrations use.
306
329
  {
@@ -315,7 +338,7 @@ var MODELS = [
315
338
  costTier: "medium",
316
339
  maxThinkingLevel: "high"
317
340
  },
318
- // ── Gemini ─────────────────────────────────────────────
341
+ // ── Gemini ─────────────────────────────────────────
319
342
  {
320
343
  id: "gemini-3.1-flash-lite",
321
344
  name: "Gemini 3.1 Flash Lite",
@@ -329,6 +352,28 @@ var MODELS = [
329
352
  costTier: "low",
330
353
  maxThinkingLevel: "high"
331
354
  },
355
+ {
356
+ // Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
357
+ // coding, agents, and multi-step execution; GA-stable on the Gemini API as
358
+ // `gemini-3.7-flash`. 1M context, 64K output, thinking low/medium/high.
359
+ // Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
360
+ // hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
361
+ // free/personal accounts 404 (entitlement-gated) while Code Assist
362
+ // Standard/Enterprise accounts get it. Listed SECOND, after flash-lite:
363
+ // getFastModel picks the first low-tier entry, and flash-lite is the one
364
+ // that works on every account.
365
+ id: "gemini-3.7-flash",
366
+ name: "Gemini 3.7 Flash",
367
+ provider: "gemini",
368
+ contextWindow: 1048576,
369
+ maxOutputTokens: 65536,
370
+ supportsThinking: true,
371
+ supportsImages: true,
372
+ supportsVideo: true,
373
+ maxVideoBytes: 20 * 1024 * 1024,
374
+ costTier: "low",
375
+ maxThinkingLevel: "high"
376
+ },
332
377
  {
333
378
  // Wire name `gemini-3-flash` — the Code Assist (OAuth) backend rejects the
334
379
  // display string `gemini-3.5-flash` with a 404, so gemini-cli keeps this
@@ -396,13 +441,12 @@ var MODELS = [
396
441
  maxThinkingLevel: "high"
397
442
  },
398
443
  // ── Z.AI (GLM) ─────────────────────────────────────────
399
- // GLM-5.3 is the only GLM entry: it supersedes 5.2 (same GLM-5 base, all
400
- // gains from post-training) and the coding endpoint already answers
401
- // `glm-5.2` requests as glm-5.3, so the older ids were menu clutter that
402
- // routed to strictly worse coding for the same plan quota.
403
- // Released 2026-08-14; live on the coding endpoint (verified), while the
404
- // standard paas API is still "coming soon". `max` is both the ceiling and
405
- // Z.AI's own default — the rungs below it live in thinking-level.ts.
444
+ // Two GLM entries, both live on the coding endpoint (verified against its
445
+ // /models list). The pre-5.3 ids stay retired: they routed to strictly worse
446
+ // coding for the same plan quota, and the endpoint already answers `glm-5.2`
447
+ // requests as glm-5.3.
448
+ // `max` is both the ceiling and Z.AI's own default — the rungs below it live
449
+ // in thinking-level.ts.
406
450
  {
407
451
  id: "glm-5.3",
408
452
  name: "GLM-5.3",
@@ -415,6 +459,30 @@ var MODELS = [
415
459
  costTier: "medium",
416
460
  maxThinkingLevel: "max"
417
461
  },
462
+ // GLM-5.3-Flash (released 2026-08-26): 320B-A18B natively multimodal sibling
463
+ // at ~1/20th of 5.3's API price with 3× the coding-plan quota, so it is the
464
+ // provider's `low` tier — scout sub-agents and compaction summaries route
465
+ // here instead of paying 5.3 rates.
466
+ // Images are native on the coding endpoint (verified: base64 data URL in an
467
+ // `image_url` block answers correctly), which also means GLM image
468
+ // attachments go inline for this model rather than through the zai_vision MCP
469
+ // detour that `supportsImages: false` triggers.
470
+ // Video/file input is documented but unverified on this transport, so it
471
+ // stays off until measured. Thinking cannot be disabled server-side (Z.AI
472
+ // maps a `disabled` toggle to the `low` rung and answers 200), and unlike
473
+ // 5.3 it accepts any reasoning_effort string without a 400.
474
+ {
475
+ id: "glm-5.3-flash",
476
+ name: "GLM-5.3-Flash",
477
+ provider: "glm",
478
+ contextWindow: 1e6,
479
+ maxOutputTokens: 131072,
480
+ supportsThinking: true,
481
+ supportsImages: true,
482
+ supportsVideo: false,
483
+ costTier: "low",
484
+ maxThinkingLevel: "max"
485
+ },
418
486
  // ── MiniMax ────────────────────────────────────────────
419
487
  {
420
488
  id: "MiniMax-M3",
@@ -481,15 +549,21 @@ var MODELS = [
481
549
  },
482
550
  // ── DeepSeek ───────────────────────────────────────────
483
551
  {
552
+ // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
553
+ // first STABLE V4 Pro — supersedes the April preview; calling name
554
+ // unchanged, same 1.6T/49B MoE). 1M context, 384K (393,216) max output,
555
+ // text-only, reasoning ladder low/high plus Think Max — mapped from our
556
+ // `xhigh`. ~$0.43/$0.87 per MTok on DeepSeek's own API, so a mid-tier
557
+ // price band rather than the preview's top band.
484
558
  id: "deepseek-v4-pro",
485
559
  name: "DeepSeek V4 Pro",
486
560
  provider: "deepseek",
487
561
  contextWindow: 1048576,
488
- maxOutputTokens: 384e3,
562
+ maxOutputTokens: 393216,
489
563
  supportsThinking: true,
490
564
  supportsImages: false,
491
565
  supportsVideo: false,
492
- costTier: "high",
566
+ costTier: "medium",
493
567
  // DeepSeek V4 maps `xhigh` → its internal `max` tier.
494
568
  maxThinkingLevel: "xhigh"
495
569
  },
@@ -517,6 +591,45 @@ var MODELS = [
517
591
  supportsVideo: false,
518
592
  costTier: "medium",
519
593
  maxThinkingLevel: "high"
594
+ },
595
+ // ── Hugging Face (Inference Providers router) ────────
596
+ // One HF token (hf.co/settings/tokens, "Make calls to Inference Providers"
597
+ // permission) routes to whichever hosted backend serves each open model;
598
+ // billing follows each backend's rates on the HF account (small free tier).
599
+ // Model ids are Hub repo paths, so they intentionally contain a slash — the
600
+ // same shape local/ vLLM ids already use (`local/vllm/Qwen/Qwen3-32B`).
601
+ {
602
+ // Qwen's open flagship for agentic coding — tool-calling native, non-thinking
603
+ // (the Coder line dropped the <think> block). 262K native context (1M needs
604
+ // YaRN, which the router doesn't apply), 131K max output. :auto suffix lets
605
+ // HF pick the backend with capacity; we keep the bare repo id so the picker
606
+ // matches what GET /v1/models reports.
607
+ id: "Qwen/Qwen3-Coder-480B-A35B-Instruct",
608
+ name: "Qwen3 Coder 480B",
609
+ provider: "huggingface",
610
+ contextWindow: 262144,
611
+ maxOutputTokens: 131072,
612
+ supportsThinking: false,
613
+ supportsImages: false,
614
+ supportsVideo: false,
615
+ costTier: "medium",
616
+ maxThinkingLevel: "low"
617
+ },
618
+ {
619
+ // OpenAI's open-weight 120B MoE (5.1B active) — general-purpose, tool-calling
620
+ // native, adjustable reasoning effort (low/medium/high, default medium) over
621
+ // the router's Chat Completions API. Cheap enough to be the low-tier sibling
622
+ // for summaries and fast sub-agents.
623
+ id: "openai/gpt-oss-120b",
624
+ name: "GPT-OSS 120B",
625
+ provider: "huggingface",
626
+ contextWindow: 131072,
627
+ maxOutputTokens: 65536,
628
+ supportsThinking: true,
629
+ supportsImages: false,
630
+ supportsVideo: false,
631
+ costTier: "low",
632
+ maxThinkingLevel: "high"
520
633
  }
521
634
  ];
522
635
  var runtimeModels = /* @__PURE__ */ new Map();
@@ -562,9 +675,11 @@ function getDefaultModel(provider) {
562
675
  if (provider === "moonshot") return MODELS.find((m) => m.id === "kimi-k3");
563
676
  if (provider === "minimax") return MODELS.find((m) => m.id === "MiniMax-M3");
564
677
  if (provider === "deepseek") return MODELS.find((m) => m.id === "deepseek-v4-pro");
678
+ if (provider === "huggingface")
679
+ return MODELS.find((m) => m.id === "Qwen/Qwen3-Coder-480B-A35B-Instruct");
565
680
  if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.6-plus");
566
681
  if (provider === "sakana") return MODELS.find((m) => m.id === "fugu");
567
- if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.5");
682
+ if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.6");
568
683
  if (provider === "local") {
569
684
  return getModelsForProvider("local")[0] ?? PLACEHOLDER_LOCAL_MODEL;
570
685
  }
@@ -608,7 +723,7 @@ function getSummaryModel(provider, currentModelId) {
608
723
  if (provider === "anthropic") {
609
724
  return MODELS.find((m) => m.id === "claude-sonnet-5");
610
725
  }
611
- if (provider === "openai" || provider === "glm" || provider === "deepseek") {
726
+ if (provider === "openai" || provider === "glm" || provider === "deepseek" || provider === "huggingface") {
612
727
  const low = getModelsForProvider(provider).find((m) => m.costTier === "low");
613
728
  if (low) return low;
614
729
  }