@prestyj/core 5.26.0 → 5.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -123,6 +123,7 @@ var import_promises2 = __toESM(require("fs/promises"), 1);
123
123
  var import_promises3 = require("timers/promises");
124
124
 
125
125
  // src/auth-storage.ts
126
+ var MOONSHOT_OAUTH_KEY = "moonshot-oauth";
126
127
  var XIAOMI_CREDITS_KEY = "xiaomi-credits";
127
128
  var LOCAL_CREDENTIAL_LIFETIME_MS = 100 * 365 * 24 * 60 * 60 * 1e3;
128
129
  var USAGE_EXHAUSTED_DEFAULT_MS = 15 * 60 * 1e3;
@@ -208,8 +209,10 @@ var MODELS = [
208
209
  maxThinkingLevel: "max"
209
210
  },
210
211
  {
211
- id: "claude-sonnet-5",
212
- name: "Claude Sonnet 5",
212
+ // Released 2026-09-28 — replaces Sonnet 5 at $2/$10 MTok, with the same
213
+ // 1M context / 128K output and adaptive thinking, now including xhigh.
214
+ id: "claude-sonnet-5-5",
215
+ name: "Claude Sonnet 5.5",
213
216
  provider: "anthropic",
214
217
  contextWindow: 1e6,
215
218
  maxOutputTokens: 128e3,
@@ -256,11 +259,18 @@ var MODELS = [
256
259
  costTier: "high",
257
260
  maxThinkingLevel: "ultra"
258
261
  },
259
- {
260
- // GPT-6 Sol — "Workhorse model for coding and everyday work." (Codex
261
- // catalog priority 2, default medium, requires Codex client >= 0.155.0).
262
- // Launched Sep 22 2026 at $2/$10 per 1M tokens — half of GPT-5.6 Sol.
263
- // Same 1.05M public / 272K Codex split and low → ultra ladder as Astra.
262
+ // GPT-6 Sol + Luna — released 2026-09-22 below Astra, replacing the whole
263
+ // GPT-5.6 family (Sol/Terra/Luna; there is no GPT-6 Terra — OpenAI's Codex
264
+ // catalog upgrades 5.6 Terra to 6 Sol). Both need a Codex client >= 0.155.0
265
+ // on the ChatGPT OAuth route. Same window split as Astra: 1.05M on the public
266
+ // Responses API, 272K on the Codex route; 128K output, text+image input,
267
+ // freeform apply_patch, responses-lite transport. The 5.6 ids are retired —
268
+ // a saved session on one falls back to the provider default on next start.
269
+ {
270
+ // Sol — "Workhorse model for coding and everyday work." (Codex priority 2,
271
+ // default medium). $2/$10 MTok. Ladder low → medium → high → xhigh → max →
272
+ // ultra; ultra is the Codex orchestration preset (max effort on the wire +
273
+ // proactive local subagent delegation).
264
274
  id: "gpt-6-sol",
265
275
  name: "GPT-6 Sol",
266
276
  provider: "openai",
@@ -275,10 +285,8 @@ var MODELS = [
275
285
  maxThinkingLevel: "ultra"
276
286
  },
277
287
  {
278
- // GPT-6 Luna — "Fast and affordable model for easier tasks." (Codex
279
- // catalog priority 3, default medium, client >= 0.155.0). $0.10/$0.50 per
280
- // 1M tokens. Reasoning tops out at `max` (no ultra preset). Listed ahead of
281
- // GPT-5.6 Luna so getFastModel picks it as the OpenAI fast tier.
288
+ // Luna — "Fast and affordable model for easier tasks." (Codex priority 3,
289
+ // default medium). $0.10/$0.50 MTok. Reasoning tops out at `max`.
282
290
  id: "gpt-6-luna",
283
291
  name: "GPT-6 Luna",
284
292
  provider: "openai",
@@ -292,71 +300,29 @@ var MODELS = [
292
300
  costTier: "low",
293
301
  maxThinkingLevel: "max"
294
302
  },
295
- // GPT-5.6 family — three agentic coding tiers launched July 2026. The public
296
- // Responses API advertises a 1.05M context window; OpenAI's Codex product
297
- // catalog advertises 272K on the ChatGPT OAuth route (corrected from the
298
- // initially advertised 372K — openai/codex PR #33972, Jul 18 2026 hotfix). All three take
299
- // text+image input, freeform apply_patch, text+image web search, and parallel
300
- // tool calls.
301
- {
302
- // Sol — "Latest frontier agentic coding model." (priority 1, default low).
303
- // Reasoning ladder: low → medium → high → xhigh → max → ultra. Ultra is a
304
- // Codex orchestration preset: the request uses max effort while the local
305
- // runtime proactively delegates suitable independent work to subagents.
306
- id: "gpt-5.6-sol",
307
- name: "GPT-5.6 Sol",
308
- provider: "openai",
309
- contextWindow: 105e4,
310
- codexContextWindow: 272e3,
311
- maxOutputTokens: 128e3,
312
- supportsThinking: true,
313
- defaultThinkingLevel: "low",
314
- supportsImages: true,
315
- supportsVideo: false,
316
- costTier: "high",
317
- maxThinkingLevel: "ultra"
318
- },
303
+ // ── Sakana (Fugu) ──────────────────────────────────────
304
+ // Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
305
+ // OpenAI-compatible Sakana API (https://api.sakana.ai/v1). All three take
306
+ // text + image input (verified against the live /models list, 2026-09-28).
307
+ // `fugu` balances latency and quality; `fugu-max` (v1.0, 2026-09-11) is the
308
+ // cost tier over the largest open-weight pool ($2/$6 per 1M); `fugu-ultra`
309
+ // is the heavier quality tier (may need larger client timeouts). Plain Fugu
310
+ // and Fugu Max stop at xhigh — Sakana documents max as the same effort there.
319
311
  {
320
- // Terra — "Balanced agentic coding model for everyday work." (priority 2,
321
- // default medium).
322
- id: "gpt-5.6-terra",
323
- name: "GPT-5.6 Terra",
324
- provider: "openai",
325
- contextWindow: 105e4,
326
- codexContextWindow: 272e3,
312
+ id: "fugu",
313
+ name: "Fugu",
314
+ provider: "sakana",
315
+ contextWindow: 1e6,
327
316
  maxOutputTokens: 128e3,
328
317
  supportsThinking: true,
329
- defaultThinkingLevel: "medium",
330
318
  supportsImages: true,
331
319
  supportsVideo: false,
332
320
  costTier: "medium",
333
- maxThinkingLevel: "ultra"
334
- },
335
- {
336
- // Luna — "Fast and affordable agentic coding model." (priority 3, default
337
- // medium). Reasoning tops out at `max`.
338
- id: "gpt-5.6-luna",
339
- name: "GPT-5.6 Luna",
340
- provider: "openai",
341
- contextWindow: 105e4,
342
- codexContextWindow: 272e3,
343
- maxOutputTokens: 128e3,
344
- supportsThinking: true,
345
- defaultThinkingLevel: "medium",
346
- supportsImages: true,
347
- supportsVideo: false,
348
- costTier: "low",
349
- maxThinkingLevel: "max"
321
+ maxThinkingLevel: "xhigh"
350
322
  },
351
- // ── Sakana (Fugu) ──────────────────────────────────────
352
- // Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
353
- // OpenAI-compatible Sakana API (https://api.sakana.ai/v1). Both models take
354
- // text + image input. Plain Fugu stops at xhigh; Ultra v1.1 also supports max.
355
- // `fugu` routes across all providers; `fugu-ultra` is
356
- // the heavier tier (may need larger client timeouts on complex tasks).
357
323
  {
358
- id: "fugu",
359
- name: "Fugu",
324
+ id: "fugu-max",
325
+ name: "Fugu Max",
360
326
  provider: "sakana",
361
327
  contextWindow: 1e6,
362
328
  maxOutputTokens: 128e3,
@@ -376,7 +342,7 @@ var MODELS = [
376
342
  supportsImages: true,
377
343
  supportsVideo: false,
378
344
  costTier: "high",
379
- // The rolling alias now serves v1.1, which adds a distinct max effort.
345
+ // The rolling alias now serves v2.0 (2026-09-11), which keeps max effort.
380
346
  maxThinkingLevel: "max"
381
347
  },
382
348
  // ── xAI (Grok) ─────────────────────────────────────────
@@ -520,7 +486,27 @@ var MODELS = [
520
486
  costTier: "high",
521
487
  maxThinkingLevel: "max"
522
488
  },
523
- // Retain the cheaper dedicated coding model as an explicit alternative.
489
+ // K2.8 Preview (2026-09-11) is served only on the Kimi For Coding OAuth
490
+ // endpoint, under its rolling `kimi-for-coding` id (live /models, 2026-09-28:
491
+ // display_name "K2.8 Preview", 1M context, image + video input, efforts
492
+ // low/high/max default max). The public API-key endpoint does not serve it,
493
+ // so it resolves from the Kimi sign-in credential only.
494
+ {
495
+ id: "kimi-for-coding",
496
+ name: "Kimi K2.8 Preview",
497
+ provider: "moonshot",
498
+ contextWindow: 1048576,
499
+ maxOutputTokens: 131072,
500
+ supportsThinking: true,
501
+ supportsImages: true,
502
+ supportsVideo: true,
503
+ maxVideoBytes: 100 * 1024 * 1024,
504
+ costTier: "medium",
505
+ maxThinkingLevel: "max",
506
+ authStorageKeys: [MOONSHOT_OAUTH_KEY]
507
+ },
508
+ // K2.7 Code is requested by its pinned id (not the `kimi-for-coding` alias
509
+ // that moved to K2.8), so it stays the real K2.7 on both endpoints.
524
510
  {
525
511
  id: "kimi-k2.7-code",
526
512
  name: "Kimi K2.7 Code",
@@ -667,10 +653,14 @@ var MODELS = [
667
653
  authStorageKeys: [XIAOMI_CREDITS_KEY]
668
654
  },
669
655
  // ── DeepSeek ───────────────────────────────────────────
656
+ // The live /models list (2026-09-28) serves exactly `deepseek-flash` and
657
+ // `deepseek-v4-pro`. V4 Flash and V4 Flash Vision Exp are retired; their old
658
+ // ids only temporarily route to V4.1 Flash, so they are retired here too.
670
659
  {
671
660
  // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
672
661
  // first STABLE V4 Pro — supersedes the April preview; calling name
673
662
  // unchanged, same 1.6T/49B MoE). 1M context, text-only, low/high/max effort.
663
+ // DeepSeek reversed its planned 2026-09-14 retirement, so it stays served.
674
664
  // Docs abbreviate output as 384K; use the same conservative 384,000-token
675
665
  // application cap across V4 models rather than mixing decimal/binary units.
676
666
  id: "deepseek-v4-pro",
@@ -685,21 +675,10 @@ var MODELS = [
685
675
  maxThinkingLevel: "max"
686
676
  },
687
677
  {
688
- id: "deepseek-v4-flash",
689
- name: "DeepSeek V4 Flash",
690
- provider: "deepseek",
691
- contextWindow: 1048576,
692
- maxOutputTokens: 384e3,
693
- supportsThinking: true,
694
- supportsImages: false,
695
- supportsVideo: false,
696
- costTier: "low",
697
- maxThinkingLevel: "max"
698
- },
699
- // Opt-in experimental vision sibling; never replaces the stable summary model.
700
- {
701
- id: "deepseek-v4-flash-vision-exp",
702
- name: "DeepSeek V4 Flash Vision (Experimental)",
678
+ // `deepseek-flash` is the rolling alias for the latest Flash — currently
679
+ // V4.1 Flash (2026-09-10): native image input, 1M context, 384K output.
680
+ id: "deepseek-flash",
681
+ name: "DeepSeek V4.1 Flash",
703
682
  provider: "deepseek",
704
683
  contextWindow: 1048576,
705
684
  maxOutputTokens: 384e3,
@@ -711,11 +690,13 @@ var MODELS = [
711
690
  },
712
691
  // ── OpenRouter ─────────────────────────────────────────
713
692
  {
714
- id: "qwen/qwen3.6-plus",
715
- name: "Qwen3.6-Plus",
693
+ // Qwen3.8 Max — Alibaba's flagship (live /endpoints, 2026-09-28): 1M
694
+ // context, 131,072 output, text + image + video input, reasoning on.
695
+ id: "qwen/qwen3.8-max",
696
+ name: "Qwen3.8 Max",
716
697
  provider: "openrouter",
717
698
  contextWindow: 1e6,
718
- maxOutputTokens: 65536,
699
+ maxOutputTokens: 131072,
719
700
  supportsThinking: true,
720
701
  supportsImages: true,
721
702
  supportsVideo: true,
@@ -809,13 +790,13 @@ function getDefaultModel(provider) {
809
790
  if (provider === "deepseek") return MODELS.find((m) => m.id === "deepseek-v4-pro");
810
791
  if (provider === "huggingface")
811
792
  return MODELS.find((m) => m.id === "Qwen/Qwen3-Coder-480B-A35B-Instruct");
812
- if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.6-plus");
793
+ if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.8-max");
813
794
  if (provider === "sakana") return MODELS.find((m) => m.id === "fugu");
814
795
  if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.7");
815
796
  if (provider === "local") {
816
797
  return getModelsForProvider("local")[0] ?? PLACEHOLDER_LOCAL_MODEL;
817
798
  }
818
- return MODELS.find((m) => m.id === "claude-sonnet-5");
799
+ return MODELS.find((m) => m.id === "claude-sonnet-5-5");
819
800
  }
820
801
  var PLACEHOLDER_LOCAL_MODEL = {
821
802
  id: "local/none/none",
@@ -853,7 +834,7 @@ function getDefaultThinkingLevel(modelId, options) {
853
834
  }
854
835
  function getSummaryModel(provider, currentModelId) {
855
836
  if (provider === "anthropic") {
856
- return MODELS.find((m) => m.id === "claude-sonnet-5");
837
+ return MODELS.find((m) => m.id === "claude-sonnet-5-5");
857
838
  }
858
839
  if (provider === "openai" || provider === "glm" || provider === "deepseek" || provider === "huggingface") {
859
840
  const low = getModelsForProvider(provider).find((m) => m.costTier === "low");