@prestyj/core 5.27.0 → 5.28.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -123,6 +123,7 @@ var import_promises2 = __toESM(require("fs/promises"), 1);
123
123
  var import_promises3 = require("timers/promises");
124
124
 
125
125
  // src/auth-storage.ts
126
+ var MOONSHOT_OAUTH_KEY = "moonshot-oauth";
126
127
  var XIAOMI_CREDITS_KEY = "xiaomi-credits";
127
128
  var LOCAL_CREDENTIAL_LIFETIME_MS = 100 * 365 * 24 * 60 * 60 * 1e3;
128
129
  var USAGE_EXHAUSTED_DEFAULT_MS = 15 * 60 * 1e3;
@@ -208,8 +209,10 @@ var MODELS = [
208
209
  maxThinkingLevel: "max"
209
210
  },
210
211
  {
211
- id: "claude-sonnet-5",
212
- name: "Claude Sonnet 5",
212
+ // Released 2026-09-28 — replaces Sonnet 5 at $2/$10 MTok, with the same
213
+ // 1M context / 128K output and adaptive thinking, now including xhigh.
214
+ id: "claude-sonnet-5-5",
215
+ name: "Claude Sonnet 5.5",
213
216
  provider: "anthropic",
214
217
  contextWindow: 1e6,
215
218
  maxOutputTokens: 128e3,
@@ -234,7 +237,7 @@ var MODELS = [
234
237
  // ── OpenAI (Codex) ─────────────────────────────────────
235
238
  {
236
239
  // GPT-6 Astra — "Our most capable model for complex, demanding work."
237
- // (Codex catalog priority 1, listed for every ChatGPT plan, requires a
240
+ // (Codex catalog priority 2, listed for every ChatGPT plan, requires a
238
241
  // Codex client >= 0.153.0 — see CODEX_CLIENT_VERSION). Same split as 5.6:
239
242
  // 1.05M on the public Responses API, 272K on the ChatGPT OAuth route
240
243
  // (openai/codex models.json, `gpt-6-astra`). Reasoning ladder low → medium
@@ -258,32 +261,36 @@ var MODELS = [
258
261
  },
259
262
  // GPT-6 Sol + Luna — released 2026-09-22 below Astra, replacing the whole
260
263
  // GPT-5.6 family (Sol/Terra/Luna; there is no GPT-6 Terra — OpenAI's Codex
261
- // catalog upgrades 5.6 Terra to 6 Sol). Both need a Codex client >= 0.155.0
262
- // on the ChatGPT OAuth route. Same window split as Astra: 1.05M on the public
263
- // Responses API, 272K on the Codex route; 128K output, text+image input,
264
- // freeform apply_patch, responses-lite transport. The 5.6 ids are retired —
265
- // a saved session on one falls back to the provider default on next start.
266
- {
267
- // Sol — "Workhorse model for coding and everyday work." (Codex priority 2,
268
- // default medium). $2/$10 MTok. Ladder low → medium → high → xhigh → max →
269
- // ultra; ultra is the Codex orchestration preset (max effort on the wire +
270
- // proactive local subagent delegation).
271
- id: "gpt-6-sol",
272
- name: "GPT-6 Sol",
264
+ // catalog upgrades 5.6 Terra to 6 Sol). GPT-6.1 Sol (2026-09-29) then replaced
265
+ // GPT-6 Sol. Same window split as Astra: 1.05M on the public Responses API,
266
+ // 272K on the Codex route; 128K output, text+image input, freeform
267
+ // apply_patch, responses-lite transport. The GPT-5.6 ids and `gpt-6-sol` are
268
+ // retired — a saved session on one falls back to the provider default on
269
+ // next start.
270
+ {
271
+ // GPT-6.1 Sol — "Latest workhorse model for coding and everyday work."
272
+ // (Codex priority 1, default low). The catalog says Codex client >= 0.153.0,
273
+ // but the ChatGPT backend only serves it from 0.159.0 (see
274
+ // CODEX_CLIENT_VERSION). $2/$10 MTok, cached input $0.10. Ladder low →
275
+ // medium → high → xhigh → max → ultra; ultra is the Codex orchestration
276
+ // preset (max effort on the wire + proactive local subagent delegation).
277
+ id: "gpt-6.1-sol",
278
+ name: "GPT-6.1 Sol",
273
279
  provider: "openai",
274
280
  contextWindow: 105e4,
275
281
  codexContextWindow: 272e3,
276
282
  maxOutputTokens: 128e3,
277
283
  supportsThinking: true,
278
- defaultThinkingLevel: "medium",
284
+ defaultThinkingLevel: "low",
279
285
  supportsImages: true,
280
286
  supportsVideo: false,
281
287
  costTier: "medium",
282
288
  maxThinkingLevel: "ultra"
283
289
  },
284
290
  {
285
- // Luna — "Fast and affordable model for easier tasks." (Codex priority 3,
286
- // default medium). $0.10/$0.50 MTok. Reasoning tops out at `max`.
291
+ // Luna — "Fast and affordable model for easier tasks." (Codex priority 4,
292
+ // default medium, needs a Codex client >= 0.155.0). $0.10/$0.50 MTok.
293
+ // Reasoning tops out at `max`.
287
294
  id: "gpt-6-luna",
288
295
  name: "GPT-6 Luna",
289
296
  provider: "openai",
@@ -299,10 +306,12 @@ var MODELS = [
299
306
  },
300
307
  // ── Sakana (Fugu) ──────────────────────────────────────
301
308
  // Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
302
- // OpenAI-compatible Sakana API (https://api.sakana.ai/v1). Both models take
303
- // text + image input. Plain Fugu stops at xhigh; Ultra v1.1 also supports max.
304
- // `fugu` routes across all providers; `fugu-ultra` is
305
- // the heavier tier (may need larger client timeouts on complex tasks).
309
+ // OpenAI-compatible Sakana API (https://api.sakana.ai/v1). All three take
310
+ // text + image input (verified against the live /models list, 2026-09-28).
311
+ // `fugu` balances latency and quality; `fugu-max` (v1.0, 2026-09-11) is the
312
+ // cost tier over the largest open-weight pool ($2/$6 per 1M); `fugu-ultra`
313
+ // is the heavier quality tier (may need larger client timeouts). Plain Fugu
314
+ // and Fugu Max stop at xhigh — Sakana documents max as the same effort there.
306
315
  {
307
316
  id: "fugu",
308
317
  name: "Fugu",
@@ -315,6 +324,18 @@ var MODELS = [
315
324
  costTier: "medium",
316
325
  maxThinkingLevel: "xhigh"
317
326
  },
327
+ {
328
+ id: "fugu-max",
329
+ name: "Fugu Max",
330
+ provider: "sakana",
331
+ contextWindow: 1e6,
332
+ maxOutputTokens: 128e3,
333
+ supportsThinking: true,
334
+ supportsImages: true,
335
+ supportsVideo: false,
336
+ costTier: "medium",
337
+ maxThinkingLevel: "xhigh"
338
+ },
318
339
  {
319
340
  id: "fugu-ultra",
320
341
  name: "Fugu Ultra",
@@ -325,7 +346,7 @@ var MODELS = [
325
346
  supportsImages: true,
326
347
  supportsVideo: false,
327
348
  costTier: "high",
328
- // The rolling alias now serves v1.1, which adds a distinct max effort.
349
+ // The rolling alias now serves v2.0 (2026-09-11), which keeps max effort.
329
350
  maxThinkingLevel: "max"
330
351
  },
331
352
  // ── xAI (Grok) ─────────────────────────────────────────
@@ -469,7 +490,27 @@ var MODELS = [
469
490
  costTier: "high",
470
491
  maxThinkingLevel: "max"
471
492
  },
472
- // Retain the cheaper dedicated coding model as an explicit alternative.
493
+ // K2.8 Preview (2026-09-11) is served only on the Kimi For Coding OAuth
494
+ // endpoint, under its rolling `kimi-for-coding` id (live /models, 2026-09-28:
495
+ // display_name "K2.8 Preview", 1M context, image + video input, efforts
496
+ // low/high/max default max). The public API-key endpoint does not serve it,
497
+ // so it resolves from the Kimi sign-in credential only.
498
+ {
499
+ id: "kimi-for-coding",
500
+ name: "Kimi K2.8 Preview",
501
+ provider: "moonshot",
502
+ contextWindow: 1048576,
503
+ maxOutputTokens: 131072,
504
+ supportsThinking: true,
505
+ supportsImages: true,
506
+ supportsVideo: true,
507
+ maxVideoBytes: 100 * 1024 * 1024,
508
+ costTier: "medium",
509
+ maxThinkingLevel: "max",
510
+ authStorageKeys: [MOONSHOT_OAUTH_KEY]
511
+ },
512
+ // K2.7 Code is requested by its pinned id (not the `kimi-for-coding` alias
513
+ // that moved to K2.8), so it stays the real K2.7 on both endpoints.
473
514
  {
474
515
  id: "kimi-k2.7-code",
475
516
  name: "Kimi K2.7 Code",
@@ -616,10 +657,14 @@ var MODELS = [
616
657
  authStorageKeys: [XIAOMI_CREDITS_KEY]
617
658
  },
618
659
  // ── DeepSeek ───────────────────────────────────────────
660
+ // The live /models list (2026-09-28) serves exactly `deepseek-flash` and
661
+ // `deepseek-v4-pro`. V4 Flash and V4 Flash Vision Exp are retired; their old
662
+ // ids only temporarily route to V4.1 Flash, so they are retired here too.
619
663
  {
620
664
  // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
621
665
  // first STABLE V4 Pro — supersedes the April preview; calling name
622
666
  // unchanged, same 1.6T/49B MoE). 1M context, text-only, low/high/max effort.
667
+ // DeepSeek reversed its planned 2026-09-14 retirement, so it stays served.
623
668
  // Docs abbreviate output as 384K; use the same conservative 384,000-token
624
669
  // application cap across V4 models rather than mixing decimal/binary units.
625
670
  id: "deepseek-v4-pro",
@@ -634,21 +679,10 @@ var MODELS = [
634
679
  maxThinkingLevel: "max"
635
680
  },
636
681
  {
637
- id: "deepseek-v4-flash",
638
- name: "DeepSeek V4 Flash",
639
- provider: "deepseek",
640
- contextWindow: 1048576,
641
- maxOutputTokens: 384e3,
642
- supportsThinking: true,
643
- supportsImages: false,
644
- supportsVideo: false,
645
- costTier: "low",
646
- maxThinkingLevel: "max"
647
- },
648
- // Opt-in experimental vision sibling; never replaces the stable summary model.
649
- {
650
- id: "deepseek-v4-flash-vision-exp",
651
- name: "DeepSeek V4 Flash Vision (Experimental)",
682
+ // `deepseek-flash` is the rolling alias for the latest Flash — currently
683
+ // V4.1 Flash (2026-09-10): native image input, 1M context, 384K output.
684
+ id: "deepseek-flash",
685
+ name: "DeepSeek V4.1 Flash",
652
686
  provider: "deepseek",
653
687
  contextWindow: 1048576,
654
688
  maxOutputTokens: 384e3,
@@ -660,11 +694,13 @@ var MODELS = [
660
694
  },
661
695
  // ── OpenRouter ─────────────────────────────────────────
662
696
  {
663
- id: "qwen/qwen3.6-plus",
664
- name: "Qwen3.6-Plus",
697
+ // Qwen3.8 Max — Alibaba's flagship (live /endpoints, 2026-09-28): 1M
698
+ // context, 131,072 output, text + image + video input, reasoning on.
699
+ id: "qwen/qwen3.8-max",
700
+ name: "Qwen3.8 Max",
665
701
  provider: "openrouter",
666
702
  contextWindow: 1e6,
667
- maxOutputTokens: 65536,
703
+ maxOutputTokens: 131072,
668
704
  supportsThinking: true,
669
705
  supportsImages: true,
670
706
  supportsVideo: true,
@@ -750,7 +786,7 @@ function getVideoByteLimit(modelId) {
750
786
  }
751
787
  function getDefaultModel(provider) {
752
788
  if (provider === "xiaomi") return MODELS.find((m) => m.id === "mimo-v2.6-pro");
753
- if (provider === "openai") return MODELS.find((m) => m.id === "gpt-6-sol");
789
+ if (provider === "openai") return MODELS.find((m) => m.id === "gpt-6.1-sol");
754
790
  if (provider === "gemini") return MODELS.find((m) => m.id === "gemini-3.1-flash-lite");
755
791
  if (provider === "glm") return MODELS.find((m) => m.id === "glm-5.3");
756
792
  if (provider === "moonshot") return MODELS.find((m) => m.id === "kimi-k3");
@@ -758,13 +794,13 @@ function getDefaultModel(provider) {
758
794
  if (provider === "deepseek") return MODELS.find((m) => m.id === "deepseek-v4-pro");
759
795
  if (provider === "huggingface")
760
796
  return MODELS.find((m) => m.id === "Qwen/Qwen3-Coder-480B-A35B-Instruct");
761
- if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.6-plus");
797
+ if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.8-max");
762
798
  if (provider === "sakana") return MODELS.find((m) => m.id === "fugu");
763
799
  if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.7");
764
800
  if (provider === "local") {
765
801
  return getModelsForProvider("local")[0] ?? PLACEHOLDER_LOCAL_MODEL;
766
802
  }
767
- return MODELS.find((m) => m.id === "claude-sonnet-5");
803
+ return MODELS.find((m) => m.id === "claude-sonnet-5-5");
768
804
  }
769
805
  var PLACEHOLDER_LOCAL_MODEL = {
770
806
  id: "local/none/none",
@@ -802,7 +838,7 @@ function getDefaultThinkingLevel(modelId, options) {
802
838
  }
803
839
  function getSummaryModel(provider, currentModelId) {
804
840
  if (provider === "anthropic") {
805
- return MODELS.find((m) => m.id === "claude-sonnet-5");
841
+ return MODELS.find((m) => m.id === "claude-sonnet-5-5");
806
842
  }
807
843
  if (provider === "openai" || provider === "glm" || provider === "deepseek" || provider === "huggingface") {
808
844
  const low = getModelsForProvider(provider).find((m) => m.costTier === "low");