@prestyj/core 5.16.1 → 5.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -1616,7 +1616,8 @@ var AuthStorage = class {
1616
1616
  filePath;
1617
1617
  loaded = false;
1618
1618
  /**
1619
- * mtime+size of the file as of the cached snapshot (`size: -1` = no file).
1619
+ * inode+mtime+size of the cached file (`size: -1` = no file). The inode
1620
+ * detects atomic replacements with equal size within one filesystem clock tick.
1620
1621
  * auth.json is shared: the desktop app writes API keys and disconnects
1621
1622
  * NATIVELY (so they work with no daemon running), and every window/process has
1622
1623
  * its own AuthStorage. A load-once cache therefore goes stale — the sidecar
@@ -1625,6 +1626,7 @@ var AuthStorage = class {
1625
1626
  */
1626
1627
  snapshotMtimeMs = 0;
1627
1628
  snapshotSize = -1;
1629
+ snapshotIno = 0;
1628
1630
  /** Per-provider lock to serialize concurrent refresh calls. */
1629
1631
  refreshLocks = /* @__PURE__ */ new Map();
1630
1632
  constructor(filePath) {
@@ -1783,7 +1785,7 @@ var AuthStorage = class {
1783
1785
  let changed;
1784
1786
  try {
1785
1787
  const stat = await import_promises4.default.stat(this.filePath);
1786
- changed = stat.mtimeMs !== this.snapshotMtimeMs || stat.size !== this.snapshotSize;
1788
+ changed = stat.ino !== this.snapshotIno || stat.mtimeMs !== this.snapshotMtimeMs || stat.size !== this.snapshotSize;
1787
1789
  } catch {
1788
1790
  changed = this.snapshotSize !== -1;
1789
1791
  }
@@ -1795,9 +1797,11 @@ var AuthStorage = class {
1795
1797
  const stat = await import_promises4.default.stat(this.filePath);
1796
1798
  this.snapshotMtimeMs = stat.mtimeMs;
1797
1799
  this.snapshotSize = stat.size;
1800
+ this.snapshotIno = stat.ino;
1798
1801
  } catch {
1799
1802
  this.snapshotMtimeMs = 0;
1800
1803
  this.snapshotSize = -1;
1804
+ this.snapshotIno = 0;
1801
1805
  }
1802
1806
  }
1803
1807
  /**
@@ -2113,6 +2117,29 @@ var MODELS = [
2113
2117
  maxThinkingLevel: "high"
2114
2118
  },
2115
2119
  // ── OpenAI (Codex) ─────────────────────────────────────
2120
+ {
2121
+ // GPT-6 Astra — "Our most capable model for complex, demanding work."
2122
+ // (Codex catalog priority 1, listed for every ChatGPT plan, requires a
2123
+ // Codex client >= 0.153.0 — see CODEX_CLIENT_VERSION). Same split as 5.6:
2124
+ // 1.05M on the public Responses API, 272K on the ChatGPT OAuth route
2125
+ // (openai/codex models.json, `gpt-6-astra`). Reasoning ladder low → medium
2126
+ // → high → xhigh → max → ultra; `ultra` is the Codex orchestration preset
2127
+ // (multi_agent v2) and is Codex-only — the public API tops out at `max`.
2128
+ // Note: through a plain API key OpenAI requires the Responses API for tool
2129
+ // calling on Astra, so the Chat Completions path is text-only; the OAuth
2130
+ // Codex route is the supported way to use it as an agent.
2131
+ id: "gpt-6-astra",
2132
+ name: "GPT-6 Astra",
2133
+ provider: "openai",
2134
+ contextWindow: 105e4,
2135
+ codexContextWindow: 272e3,
2136
+ maxOutputTokens: 128e3,
2137
+ supportsThinking: true,
2138
+ supportsImages: true,
2139
+ supportsVideo: false,
2140
+ costTier: "high",
2141
+ maxThinkingLevel: "ultra"
2142
+ },
2116
2143
  // GPT-5.6 family — three agentic coding tiers launched July 2026. The public
2117
2144
  // Responses API advertises a 1.05M context window; OpenAI's Codex product
2118
2145
  // catalog advertises 272K on the ChatGPT OAuth route (corrected from the
@@ -2166,24 +2193,11 @@ var MODELS = [
2166
2193
  costTier: "low",
2167
2194
  maxThinkingLevel: "max"
2168
2195
  },
2169
- {
2170
- id: "gpt-5.5",
2171
- name: "GPT-5.5",
2172
- provider: "openai",
2173
- contextWindow: 105e4,
2174
- codexContextWindow: 272e3,
2175
- maxOutputTokens: 128e3,
2176
- supportsThinking: true,
2177
- supportsImages: true,
2178
- supportsVideo: false,
2179
- costTier: "high",
2180
- maxThinkingLevel: "xhigh"
2181
- },
2182
2196
  // ── Sakana (Fugu) ──────────────────────────────────────
2183
2197
  // Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
2184
2198
  // OpenAI-compatible Sakana API (https://api.sakana.ai/v1). Both models take
2185
- // text + image input and only accept "high"/"xhigh" reasoning effort, so the
2186
- // top tier is `xhigh`. `fugu` routes across all providers; `fugu-ultra` is
2199
+ // text + image input. Plain Fugu stops at xhigh; Ultra v1.1 also supports max.
2200
+ // `fugu` routes across all providers; `fugu-ultra` is
2187
2201
  // the heavier tier (may need larger client timeouts on complex tasks).
2188
2202
  {
2189
2203
  id: "fugu",
@@ -2207,7 +2221,8 @@ var MODELS = [
2207
2221
  supportsImages: true,
2208
2222
  supportsVideo: false,
2209
2223
  costTier: "high",
2210
- maxThinkingLevel: "xhigh"
2224
+ // The rolling alias now serves v1.1, which adds a distinct max effort.
2225
+ maxThinkingLevel: "max"
2211
2226
  },
2212
2227
  // ── xAI (Grok) ─────────────────────────────────────────
2213
2228
  // Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
@@ -2261,6 +2276,34 @@ var MODELS = [
2261
2276
  costTier: "low",
2262
2277
  maxThinkingLevel: "high"
2263
2278
  },
2279
+ // Keep 3.1 Flash Lite first for the working OAuth default and fast-model routing.
2280
+ // New GA models are opt-in; Code Assist access varies by account.
2281
+ {
2282
+ id: "gemini-3.8-flash",
2283
+ name: "Gemini 3.8 Flash",
2284
+ provider: "gemini",
2285
+ contextWindow: 1048576,
2286
+ maxOutputTokens: 65536,
2287
+ supportsThinking: true,
2288
+ supportsImages: true,
2289
+ supportsVideo: true,
2290
+ maxVideoBytes: 20 * 1024 * 1024,
2291
+ costTier: "low",
2292
+ maxThinkingLevel: "high"
2293
+ },
2294
+ {
2295
+ id: "gemini-3.5-flash-lite",
2296
+ name: "Gemini 3.5 Flash Lite",
2297
+ provider: "gemini",
2298
+ contextWindow: 1048576,
2299
+ maxOutputTokens: 65536,
2300
+ supportsThinking: true,
2301
+ supportsImages: true,
2302
+ supportsVideo: true,
2303
+ maxVideoBytes: 20 * 1024 * 1024,
2304
+ costTier: "low",
2305
+ maxThinkingLevel: "high"
2306
+ },
2264
2307
  {
2265
2308
  // Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
2266
2309
  // coding, agents, and multi-step execution; GA-stable on the Gemini API as
@@ -2268,7 +2311,7 @@ var MODELS = [
2268
2311
  // Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
2269
2312
  // hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
2270
2313
  // free/personal accounts 404 (entitlement-gated) while Code Assist
2271
- // Standard/Enterprise accounts get it. Listed SECOND, after flash-lite:
2314
+ // Standard/Enterprise accounts get it. Kept after the working flash-lite:
2272
2315
  // getFastModel picks the first low-tier entry, and flash-lite is the one
2273
2316
  // that works on every account.
2274
2317
  id: "gemini-3.7-flash",
@@ -2460,21 +2503,19 @@ var MODELS = [
2460
2503
  {
2461
2504
  // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
2462
2505
  // first STABLE V4 Pro — supersedes the April preview; calling name
2463
- // unchanged, same 1.6T/49B MoE). 1M context, 384K (393,216) max output,
2464
- // text-only, reasoning ladder low/high plus Think Max — mapped from our
2465
- // `xhigh`. ~$0.43/$0.87 per MTok on DeepSeek's own API, so a mid-tier
2466
- // price band rather than the preview's top band.
2506
+ // unchanged, same 1.6T/49B MoE). 1M context, text-only, low/high/max effort.
2507
+ // Docs abbreviate output as 384K; use the same conservative 384,000-token
2508
+ // application cap across V4 models rather than mixing decimal/binary units.
2467
2509
  id: "deepseek-v4-pro",
2468
2510
  name: "DeepSeek V4 Pro",
2469
2511
  provider: "deepseek",
2470
2512
  contextWindow: 1048576,
2471
- maxOutputTokens: 393216,
2513
+ maxOutputTokens: 384e3,
2472
2514
  supportsThinking: true,
2473
2515
  supportsImages: false,
2474
2516
  supportsVideo: false,
2475
2517
  costTier: "medium",
2476
- // DeepSeek V4 maps `xhigh` → its internal `max` tier.
2477
- maxThinkingLevel: "xhigh"
2518
+ maxThinkingLevel: "max"
2478
2519
  },
2479
2520
  {
2480
2521
  id: "deepseek-v4-flash",
@@ -2486,7 +2527,20 @@ var MODELS = [
2486
2527
  supportsImages: false,
2487
2528
  supportsVideo: false,
2488
2529
  costTier: "low",
2489
- maxThinkingLevel: "xhigh"
2530
+ maxThinkingLevel: "max"
2531
+ },
2532
+ // Opt-in experimental vision sibling; never replaces the stable summary model.
2533
+ {
2534
+ id: "deepseek-v4-flash-vision-exp",
2535
+ name: "DeepSeek V4 Flash Vision (Experimental)",
2536
+ provider: "deepseek",
2537
+ contextWindow: 1048576,
2538
+ maxOutputTokens: 384e3,
2539
+ supportsThinking: true,
2540
+ supportsImages: true,
2541
+ supportsVideo: false,
2542
+ costTier: "low",
2543
+ maxThinkingLevel: "max"
2490
2544
  },
2491
2545
  // ── OpenRouter ─────────────────────────────────────────
2492
2546
  {
@@ -2496,8 +2550,10 @@ var MODELS = [
2496
2550
  contextWindow: 1e6,
2497
2551
  maxOutputTokens: 65536,
2498
2552
  supportsThinking: true,
2499
- supportsImages: false,
2500
- supportsVideo: false,
2553
+ supportsImages: true,
2554
+ supportsVideo: true,
2555
+ // Practical inline-payload cap, not an asserted provider maximum.
2556
+ maxVideoBytes: 20 * 1024 * 1024,
2501
2557
  costTier: "medium",
2502
2558
  maxThinkingLevel: "high"
2503
2559
  },
@@ -2653,7 +2709,8 @@ var OPENAI_GPT_56_THINKING_LEVELS = [
2653
2709
  "max",
2654
2710
  "ultra"
2655
2711
  ];
2656
- var SAKANA_THINKING_LEVELS = ["high", "xhigh"];
2712
+ var SAKANA_THINKING_LEVELS = ["high", "xhigh", "max"];
2713
+ var DEEPSEEK_THINKING_LEVELS = ["low", "high", "max"];
2657
2714
  var XAI_THINKING_LEVELS = ["low", "medium", "high", "xhigh"];
2658
2715
  var ANTHROPIC_XHIGH_THINKING_LEVELS = [
2659
2716
  "low",
@@ -2717,13 +2774,14 @@ function getSupportedThinkingLevels(provider, model) {
2717
2774
  return XAI_THINKING_LEVELS.slice(0, maxIndex2 + 1);
2718
2775
  }
2719
2776
  if (isMoonshotK3Model(provider, model)) return MOONSHOT_K3_THINKING_LEVELS;
2777
+ if (provider === "deepseek") return DEEPSEEK_THINKING_LEVELS;
2720
2778
  if (isGlmModel(provider)) {
2721
2779
  const maxIndex2 = GLM_THINKING_LEVELS.indexOf(maxLevel);
2722
2780
  if (maxIndex2 === -1) return GLM_THINKING_LEVELS;
2723
2781
  return GLM_THINKING_LEVELS.slice(0, maxIndex2 + 1);
2724
2782
  }
2725
2783
  if (!isOpenAIGptModel(provider, model)) return [maxLevel];
2726
- const levels = model.startsWith("gpt-5.6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
2784
+ const levels = model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
2727
2785
  const maxIndex = levels.indexOf(maxLevel);
2728
2786
  if (maxIndex === -1) return ["medium"];
2729
2787
  return levels.slice(0, maxIndex + 1);
@@ -2733,7 +2791,7 @@ function isThinkingLevelSupported(provider, model, level) {
2733
2791
  }
2734
2792
  function getNextThinkingLevel(provider, model, current) {
2735
2793
  const supportedLevels = getSupportedThinkingLevels(provider, model);
2736
- const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || // Local servers take a real effort level, not just on/off: Ollama accepts
2794
+ const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || provider === "deepseek" || // Local servers take a real effort level, not just on/off: Ollama accepts
2737
2795
  // low/medium/high on `reasoning_effort` (verified against 0.32) and the
2738
2796
  // other OpenAI-compatible servers use the same three. A model that can't
2739
2797
  // reason at all already has no supported levels, so it never gets here.
@@ -2751,7 +2809,7 @@ function getNextThinkingLevel(provider, model, current) {
2751
2809
  var DEFAULT_LOCAL_ENDPOINTS = [
2752
2810
  { id: "ollama", label: "Ollama", baseUrl: "http://127.0.0.1:11434/v1", kind: "ollama" }
2753
2811
  ];
2754
- var FALLBACK_CONTEXT_WINDOW = 8192;
2812
+ var FALLBACK_CONTEXT_WINDOW = 4096;
2755
2813
  var LOCAL_API_KEY_PLACEHOLDER = "local";
2756
2814
  var DEFAULT_PROBE_TIMEOUT_MS = 1200;
2757
2815
  var ENRICH_CONCURRENCY = 6;
@@ -2844,7 +2902,7 @@ async function enrich(endpoint, entries, options) {
2844
2902
  return entries.map((entry) => genericModel(endpoint, entry));
2845
2903
  }
2846
2904
  function genericModel(endpoint, entry) {
2847
- const declared = typeof entry.max_model_len === "number" ? entry.max_model_len : void 0;
2905
+ const declared = contextLength(entry.max_model_len);
2848
2906
  return {
2849
2907
  rawId: entry.id,
2850
2908
  endpointId: endpoint.id,
@@ -2857,16 +2915,23 @@ function genericModel(endpoint, entry) {
2857
2915
  }
2858
2916
  async function enrichOllama(endpoint, entries, options) {
2859
2917
  const showUrl = `${endpointRoot(endpoint.baseUrl)}/api/show`;
2918
+ const running = await fetchJson(
2919
+ `${endpointRoot(endpoint.baseUrl)}/api/ps`,
2920
+ endpoint,
2921
+ options
2922
+ );
2923
+ const runningModels = Array.isArray(running?.models) ? running.models : [];
2860
2924
  const enriched = await mapLimited(entries, ENRICH_CONCURRENCY, async (entry) => {
2861
2925
  const show = await fetchJson(showUrl, endpoint, {
2862
2926
  ...options,
2863
2927
  method: "POST",
2864
2928
  body: { model: entry.id }
2865
2929
  });
2866
- if (!show) return genericModel(endpoint, entry);
2867
- const caps = show.capabilities ?? [];
2930
+ const caps = Array.isArray(show?.capabilities) ? show.capabilities : [];
2868
2931
  if (caps.includes("embedding") && !caps.includes("completion")) return void 0;
2869
- const ctx = ollamaContextLength(show.model_info);
2932
+ const ctx = contextLength(
2933
+ runningModels.find((model) => model?.name === entry.id || model?.model === entry.id)?.context_length
2934
+ );
2870
2935
  return {
2871
2936
  rawId: entry.id,
2872
2937
  endpointId: endpoint.id,
@@ -2874,38 +2939,61 @@ async function enrichOllama(endpoint, entries, options) {
2874
2939
  contextWindowKnown: ctx !== void 0,
2875
2940
  // Ollama reports capabilities honestly, so trust it here rather than
2876
2941
  // using the optimistic generic default.
2877
- supportsTools: caps.includes("tools"),
2942
+ supportsTools: show ? caps.includes("tools") : true,
2878
2943
  supportsImages: caps.includes("vision"),
2879
2944
  supportsThinking: caps.includes("thinking")
2880
2945
  };
2881
2946
  });
2882
2947
  return enriched.filter((model) => model !== void 0);
2883
2948
  }
2884
- function ollamaContextLength(info) {
2885
- if (!info) return void 0;
2886
- for (const [key, value] of Object.entries(info)) {
2887
- if (key.endsWith(".context_length") && typeof value === "number" && value > 0) return value;
2888
- }
2889
- return void 0;
2949
+ function contextLength(value) {
2950
+ return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : void 0;
2890
2951
  }
2891
2952
  async function enrichLmStudio(endpoint, entries, options) {
2892
- const detail = await fetchJson(
2893
- `${endpointRoot(endpoint.baseUrl)}/api/v0/models`,
2894
- endpoint,
2895
- options
2896
- );
2897
- if (!detail?.data) return entries.map((entry) => genericModel(endpoint, entry));
2898
- const byId = new Map(detail.data.filter((m) => m.id).map((m) => [m.id, m]));
2953
+ const root = endpointRoot(endpoint.baseUrl);
2954
+ const detail = await fetchJson(`${root}/api/v1/models`, endpoint, options);
2955
+ const byId = /* @__PURE__ */ new Map();
2956
+ if (Array.isArray(detail?.models)) {
2957
+ for (const model of detail.models) {
2958
+ if (typeof model?.key !== "string") continue;
2959
+ const instances = Array.isArray(model.loaded_instances) ? model.loaded_instances : [];
2960
+ const contexts = instances.map((instance) => contextLength(instance?.config?.context_length));
2961
+ const ctx = contexts.length && contexts.every((value) => value !== void 0) ? contexts.reduce((min, value) => Math.min(min, value)) : void 0;
2962
+ const info = {
2963
+ id: model.key,
2964
+ type: model.type === "llm" && model.capabilities?.vision === true ? "vlm" : model.type,
2965
+ state: instances.length ? "loaded" : "not-loaded",
2966
+ loaded_context_length: ctx
2967
+ };
2968
+ byId.set(model.key, info);
2969
+ for (const instance of instances) {
2970
+ if (typeof instance?.id === "string" && instance.id !== model.key) {
2971
+ byId.set(instance.id, {
2972
+ ...info,
2973
+ id: instance.id,
2974
+ loaded_context_length: contextLength(instance.config?.context_length)
2975
+ });
2976
+ }
2977
+ }
2978
+ }
2979
+ } else {
2980
+ const legacy = await fetchJson(`${root}/api/v0/models`, endpoint, options);
2981
+ if (Array.isArray(legacy?.data)) {
2982
+ for (const model of legacy.data) {
2983
+ if (typeof model?.id === "string") byId.set(model.id, model);
2984
+ }
2985
+ }
2986
+ }
2899
2987
  const models = [];
2900
2988
  for (const entry of entries) {
2901
2989
  const info = byId.get(entry.id);
2902
2990
  if (info && info.type !== "llm" && info.type !== "vlm") continue;
2903
- const ctx = info?.max_context_length;
2991
+ const ctx = info?.state === "loaded" ? contextLength(info.loaded_context_length) : void 0;
2904
2992
  models.push({
2905
2993
  rawId: entry.id,
2906
2994
  endpointId: endpoint.id,
2907
- contextWindow: typeof ctx === "number" && ctx > 0 ? ctx : FALLBACK_CONTEXT_WINDOW,
2908
- contextWindowKnown: typeof ctx === "number" && ctx > 0,
2995
+ contextWindow: ctx ?? FALLBACK_CONTEXT_WINDOW,
2996
+ contextWindowKnown: ctx !== void 0,
2909
2997
  // LM Studio doesn't report tool support; it gates per-model at request time.
2910
2998
  supportsTools: true,
2911
2999
  supportsImages: info?.type === "vlm",
@@ -2921,8 +3009,8 @@ async function enrichLlamaCpp(endpoint, entries, options) {
2921
3009
  endpoint,
2922
3010
  options
2923
3011
  );
2924
- const nCtx = props?.default_generation_settings?.n_ctx;
2925
- const known = typeof nCtx === "number" && nCtx > 0;
3012
+ const nCtx = contextLength(props?.default_generation_settings?.n_ctx);
3013
+ const known = nCtx !== void 0;
2926
3014
  return entries.map((entry) => ({
2927
3015
  ...genericModel(endpoint, entry),
2928
3016
  contextWindow: known ? nCtx : FALLBACK_CONTEXT_WINDOW,