@prestyj/core 5.16.0 → 5.16.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -2113,6 +2113,29 @@ var MODELS = [
2113
2113
  maxThinkingLevel: "high"
2114
2114
  },
2115
2115
  // ── OpenAI (Codex) ─────────────────────────────────────
2116
+ {
2117
+ // GPT-6 Astra — "Our most capable model for complex, demanding work."
2118
+ // (Codex catalog priority 1, listed for every ChatGPT plan, requires a
2119
+ // Codex client >= 0.153.0 — see CODEX_CLIENT_VERSION). Same split as 5.6:
2120
+ // 1.05M on the public Responses API, 272K on the ChatGPT OAuth route
2121
+ // (openai/codex models.json, `gpt-6-astra`). Reasoning ladder low → medium
2122
+ // → high → xhigh → max → ultra; `ultra` is the Codex orchestration preset
2123
+ // (multi_agent v2) and is Codex-only — the public API tops out at `max`.
2124
+ // Note: through a plain API key OpenAI requires the Responses API for tool
2125
+ // calling on Astra, so the Chat Completions path is text-only; the OAuth
2126
+ // Codex route is the supported way to use it as an agent.
2127
+ id: "gpt-6-astra",
2128
+ name: "GPT-6 Astra",
2129
+ provider: "openai",
2130
+ contextWindow: 105e4,
2131
+ codexContextWindow: 272e3,
2132
+ maxOutputTokens: 128e3,
2133
+ supportsThinking: true,
2134
+ supportsImages: true,
2135
+ supportsVideo: false,
2136
+ costTier: "high",
2137
+ maxThinkingLevel: "ultra"
2138
+ },
2116
2139
  // GPT-5.6 family — three agentic coding tiers launched July 2026. The public
2117
2140
  // Responses API advertises a 1.05M context window; OpenAI's Codex product
2118
2141
  // catalog advertises 272K on the ChatGPT OAuth route (corrected from the
@@ -2166,24 +2189,11 @@ var MODELS = [
2166
2189
  costTier: "low",
2167
2190
  maxThinkingLevel: "max"
2168
2191
  },
2169
- {
2170
- id: "gpt-5.5",
2171
- name: "GPT-5.5",
2172
- provider: "openai",
2173
- contextWindow: 105e4,
2174
- codexContextWindow: 272e3,
2175
- maxOutputTokens: 128e3,
2176
- supportsThinking: true,
2177
- supportsImages: true,
2178
- supportsVideo: false,
2179
- costTier: "high",
2180
- maxThinkingLevel: "xhigh"
2181
- },
2182
2192
  // ── Sakana (Fugu) ──────────────────────────────────────
2183
2193
  // Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
2184
2194
  // OpenAI-compatible Sakana API (https://api.sakana.ai/v1). Both models take
2185
- // text + image input and only accept "high"/"xhigh" reasoning effort, so the
2186
- // top tier is `xhigh`. `fugu` routes across all providers; `fugu-ultra` is
2195
+ // text + image input. Plain Fugu stops at xhigh; Ultra v1.1 also supports max.
2196
+ // `fugu` routes across all providers; `fugu-ultra` is
2187
2197
  // the heavier tier (may need larger client timeouts on complex tasks).
2188
2198
  {
2189
2199
  id: "fugu",
@@ -2207,7 +2217,8 @@ var MODELS = [
2207
2217
  supportsImages: true,
2208
2218
  supportsVideo: false,
2209
2219
  costTier: "high",
2210
- maxThinkingLevel: "xhigh"
2220
+ // The rolling alias now serves v1.1, which adds a distinct max effort.
2221
+ maxThinkingLevel: "max"
2211
2222
  },
2212
2223
  // ── xAI (Grok) ─────────────────────────────────────────
2213
2224
  // Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
@@ -2261,6 +2272,34 @@ var MODELS = [
2261
2272
  costTier: "low",
2262
2273
  maxThinkingLevel: "high"
2263
2274
  },
2275
+ // Keep 3.1 Flash Lite first for the working OAuth default and fast-model routing.
2276
+ // New GA models are opt-in; Code Assist access varies by account.
2277
+ {
2278
+ id: "gemini-3.8-flash",
2279
+ name: "Gemini 3.8 Flash",
2280
+ provider: "gemini",
2281
+ contextWindow: 1048576,
2282
+ maxOutputTokens: 65536,
2283
+ supportsThinking: true,
2284
+ supportsImages: true,
2285
+ supportsVideo: true,
2286
+ maxVideoBytes: 20 * 1024 * 1024,
2287
+ costTier: "low",
2288
+ maxThinkingLevel: "high"
2289
+ },
2290
+ {
2291
+ id: "gemini-3.5-flash-lite",
2292
+ name: "Gemini 3.5 Flash Lite",
2293
+ provider: "gemini",
2294
+ contextWindow: 1048576,
2295
+ maxOutputTokens: 65536,
2296
+ supportsThinking: true,
2297
+ supportsImages: true,
2298
+ supportsVideo: true,
2299
+ maxVideoBytes: 20 * 1024 * 1024,
2300
+ costTier: "low",
2301
+ maxThinkingLevel: "high"
2302
+ },
2264
2303
  {
2265
2304
  // Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
2266
2305
  // coding, agents, and multi-step execution; GA-stable on the Gemini API as
@@ -2268,7 +2307,7 @@ var MODELS = [
2268
2307
  // Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
2269
2308
  // hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
2270
2309
  // free/personal accounts 404 (entitlement-gated) while Code Assist
2271
- // Standard/Enterprise accounts get it. Listed SECOND, after flash-lite:
2310
+ // Standard/Enterprise accounts get it. Kept after the working flash-lite:
2272
2311
  // getFastModel picks the first low-tier entry, and flash-lite is the one
2273
2312
  // that works on every account.
2274
2313
  id: "gemini-3.7-flash",
@@ -2460,21 +2499,19 @@ var MODELS = [
2460
2499
  {
2461
2500
  // `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
2462
2501
  // first STABLE V4 Pro — supersedes the April preview; calling name
2463
- // unchanged, same 1.6T/49B MoE). 1M context, 384K (393,216) max output,
2464
- // text-only, reasoning ladder low/high plus Think Max — mapped from our
2465
- // `xhigh`. ~$0.43/$0.87 per MTok on DeepSeek's own API, so a mid-tier
2466
- // price band rather than the preview's top band.
2502
+ // unchanged, same 1.6T/49B MoE). 1M context, text-only, low/high/max effort.
2503
+ // Docs abbreviate output as 384K; use the same conservative 384,000-token
2504
+ // application cap across V4 models rather than mixing decimal/binary units.
2467
2505
  id: "deepseek-v4-pro",
2468
2506
  name: "DeepSeek V4 Pro",
2469
2507
  provider: "deepseek",
2470
2508
  contextWindow: 1048576,
2471
- maxOutputTokens: 393216,
2509
+ maxOutputTokens: 384e3,
2472
2510
  supportsThinking: true,
2473
2511
  supportsImages: false,
2474
2512
  supportsVideo: false,
2475
2513
  costTier: "medium",
2476
- // DeepSeek V4 maps `xhigh` → its internal `max` tier.
2477
- maxThinkingLevel: "xhigh"
2514
+ maxThinkingLevel: "max"
2478
2515
  },
2479
2516
  {
2480
2517
  id: "deepseek-v4-flash",
@@ -2486,7 +2523,20 @@ var MODELS = [
2486
2523
  supportsImages: false,
2487
2524
  supportsVideo: false,
2488
2525
  costTier: "low",
2489
- maxThinkingLevel: "xhigh"
2526
+ maxThinkingLevel: "max"
2527
+ },
2528
+ // Opt-in experimental vision sibling; never replaces the stable summary model.
2529
+ {
2530
+ id: "deepseek-v4-flash-vision-exp",
2531
+ name: "DeepSeek V4 Flash Vision (Experimental)",
2532
+ provider: "deepseek",
2533
+ contextWindow: 1048576,
2534
+ maxOutputTokens: 384e3,
2535
+ supportsThinking: true,
2536
+ supportsImages: true,
2537
+ supportsVideo: false,
2538
+ costTier: "low",
2539
+ maxThinkingLevel: "max"
2490
2540
  },
2491
2541
  // ── OpenRouter ─────────────────────────────────────────
2492
2542
  {
@@ -2496,8 +2546,10 @@ var MODELS = [
2496
2546
  contextWindow: 1e6,
2497
2547
  maxOutputTokens: 65536,
2498
2548
  supportsThinking: true,
2499
- supportsImages: false,
2500
- supportsVideo: false,
2549
+ supportsImages: true,
2550
+ supportsVideo: true,
2551
+ // Practical inline-payload cap, not an asserted provider maximum.
2552
+ maxVideoBytes: 20 * 1024 * 1024,
2501
2553
  costTier: "medium",
2502
2554
  maxThinkingLevel: "high"
2503
2555
  },
@@ -2653,7 +2705,8 @@ var OPENAI_GPT_56_THINKING_LEVELS = [
2653
2705
  "max",
2654
2706
  "ultra"
2655
2707
  ];
2656
- var SAKANA_THINKING_LEVELS = ["high", "xhigh"];
2708
+ var SAKANA_THINKING_LEVELS = ["high", "xhigh", "max"];
2709
+ var DEEPSEEK_THINKING_LEVELS = ["low", "high", "max"];
2657
2710
  var XAI_THINKING_LEVELS = ["low", "medium", "high", "xhigh"];
2658
2711
  var ANTHROPIC_XHIGH_THINKING_LEVELS = [
2659
2712
  "low",
@@ -2717,13 +2770,14 @@ function getSupportedThinkingLevels(provider, model) {
2717
2770
  return XAI_THINKING_LEVELS.slice(0, maxIndex2 + 1);
2718
2771
  }
2719
2772
  if (isMoonshotK3Model(provider, model)) return MOONSHOT_K3_THINKING_LEVELS;
2773
+ if (provider === "deepseek") return DEEPSEEK_THINKING_LEVELS;
2720
2774
  if (isGlmModel(provider)) {
2721
2775
  const maxIndex2 = GLM_THINKING_LEVELS.indexOf(maxLevel);
2722
2776
  if (maxIndex2 === -1) return GLM_THINKING_LEVELS;
2723
2777
  return GLM_THINKING_LEVELS.slice(0, maxIndex2 + 1);
2724
2778
  }
2725
2779
  if (!isOpenAIGptModel(provider, model)) return [maxLevel];
2726
- const levels = model.startsWith("gpt-5.6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
2780
+ const levels = model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
2727
2781
  const maxIndex = levels.indexOf(maxLevel);
2728
2782
  if (maxIndex === -1) return ["medium"];
2729
2783
  return levels.slice(0, maxIndex + 1);
@@ -2733,7 +2787,7 @@ function isThinkingLevelSupported(provider, model, level) {
2733
2787
  }
2734
2788
  function getNextThinkingLevel(provider, model, current) {
2735
2789
  const supportedLevels = getSupportedThinkingLevels(provider, model);
2736
- const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || // Local servers take a real effort level, not just on/off: Ollama accepts
2790
+ const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || provider === "deepseek" || // Local servers take a real effort level, not just on/off: Ollama accepts
2737
2791
  // low/medium/high on `reasoning_effort` (verified against 0.32) and the
2738
2792
  // other OpenAI-compatible servers use the same three. A model that can't
2739
2793
  // reason at all already has no supported levels, so it never gets here.
@@ -2751,7 +2805,7 @@ function getNextThinkingLevel(provider, model, current) {
2751
2805
  var DEFAULT_LOCAL_ENDPOINTS = [
2752
2806
  { id: "ollama", label: "Ollama", baseUrl: "http://127.0.0.1:11434/v1", kind: "ollama" }
2753
2807
  ];
2754
- var FALLBACK_CONTEXT_WINDOW = 8192;
2808
+ var FALLBACK_CONTEXT_WINDOW = 4096;
2755
2809
  var LOCAL_API_KEY_PLACEHOLDER = "local";
2756
2810
  var DEFAULT_PROBE_TIMEOUT_MS = 1200;
2757
2811
  var ENRICH_CONCURRENCY = 6;
@@ -2844,7 +2898,7 @@ async function enrich(endpoint, entries, options) {
2844
2898
  return entries.map((entry) => genericModel(endpoint, entry));
2845
2899
  }
2846
2900
  function genericModel(endpoint, entry) {
2847
- const declared = typeof entry.max_model_len === "number" ? entry.max_model_len : void 0;
2901
+ const declared = contextLength(entry.max_model_len);
2848
2902
  return {
2849
2903
  rawId: entry.id,
2850
2904
  endpointId: endpoint.id,
@@ -2857,16 +2911,23 @@ function genericModel(endpoint, entry) {
2857
2911
  }
2858
2912
  async function enrichOllama(endpoint, entries, options) {
2859
2913
  const showUrl = `${endpointRoot(endpoint.baseUrl)}/api/show`;
2914
+ const running = await fetchJson(
2915
+ `${endpointRoot(endpoint.baseUrl)}/api/ps`,
2916
+ endpoint,
2917
+ options
2918
+ );
2919
+ const runningModels = Array.isArray(running?.models) ? running.models : [];
2860
2920
  const enriched = await mapLimited(entries, ENRICH_CONCURRENCY, async (entry) => {
2861
2921
  const show = await fetchJson(showUrl, endpoint, {
2862
2922
  ...options,
2863
2923
  method: "POST",
2864
2924
  body: { model: entry.id }
2865
2925
  });
2866
- if (!show) return genericModel(endpoint, entry);
2867
- const caps = show.capabilities ?? [];
2926
+ const caps = Array.isArray(show?.capabilities) ? show.capabilities : [];
2868
2927
  if (caps.includes("embedding") && !caps.includes("completion")) return void 0;
2869
- const ctx = ollamaContextLength(show.model_info);
2928
+ const ctx = contextLength(
2929
+ runningModels.find((model) => model?.name === entry.id || model?.model === entry.id)?.context_length
2930
+ );
2870
2931
  return {
2871
2932
  rawId: entry.id,
2872
2933
  endpointId: endpoint.id,
@@ -2874,38 +2935,61 @@ async function enrichOllama(endpoint, entries, options) {
2874
2935
  contextWindowKnown: ctx !== void 0,
2875
2936
  // Ollama reports capabilities honestly, so trust it here rather than
2876
2937
  // using the optimistic generic default.
2877
- supportsTools: caps.includes("tools"),
2938
+ supportsTools: show ? caps.includes("tools") : true,
2878
2939
  supportsImages: caps.includes("vision"),
2879
2940
  supportsThinking: caps.includes("thinking")
2880
2941
  };
2881
2942
  });
2882
2943
  return enriched.filter((model) => model !== void 0);
2883
2944
  }
2884
- function ollamaContextLength(info) {
2885
- if (!info) return void 0;
2886
- for (const [key, value] of Object.entries(info)) {
2887
- if (key.endsWith(".context_length") && typeof value === "number" && value > 0) return value;
2888
- }
2889
- return void 0;
2945
+ function contextLength(value) {
2946
+ return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : void 0;
2890
2947
  }
2891
2948
  async function enrichLmStudio(endpoint, entries, options) {
2892
- const detail = await fetchJson(
2893
- `${endpointRoot(endpoint.baseUrl)}/api/v0/models`,
2894
- endpoint,
2895
- options
2896
- );
2897
- if (!detail?.data) return entries.map((entry) => genericModel(endpoint, entry));
2898
- const byId = new Map(detail.data.filter((m) => m.id).map((m) => [m.id, m]));
2949
+ const root = endpointRoot(endpoint.baseUrl);
2950
+ const detail = await fetchJson(`${root}/api/v1/models`, endpoint, options);
2951
+ const byId = /* @__PURE__ */ new Map();
2952
+ if (Array.isArray(detail?.models)) {
2953
+ for (const model of detail.models) {
2954
+ if (typeof model?.key !== "string") continue;
2955
+ const instances = Array.isArray(model.loaded_instances) ? model.loaded_instances : [];
2956
+ const contexts = instances.map((instance) => contextLength(instance?.config?.context_length));
2957
+ const ctx = contexts.length && contexts.every((value) => value !== void 0) ? contexts.reduce((min, value) => Math.min(min, value)) : void 0;
2958
+ const info = {
2959
+ id: model.key,
2960
+ type: model.type === "llm" && model.capabilities?.vision === true ? "vlm" : model.type,
2961
+ state: instances.length ? "loaded" : "not-loaded",
2962
+ loaded_context_length: ctx
2963
+ };
2964
+ byId.set(model.key, info);
2965
+ for (const instance of instances) {
2966
+ if (typeof instance?.id === "string" && instance.id !== model.key) {
2967
+ byId.set(instance.id, {
2968
+ ...info,
2969
+ id: instance.id,
2970
+ loaded_context_length: contextLength(instance.config?.context_length)
2971
+ });
2972
+ }
2973
+ }
2974
+ }
2975
+ } else {
2976
+ const legacy = await fetchJson(`${root}/api/v0/models`, endpoint, options);
2977
+ if (Array.isArray(legacy?.data)) {
2978
+ for (const model of legacy.data) {
2979
+ if (typeof model?.id === "string") byId.set(model.id, model);
2980
+ }
2981
+ }
2982
+ }
2899
2983
  const models = [];
2900
2984
  for (const entry of entries) {
2901
2985
  const info = byId.get(entry.id);
2902
2986
  if (info && info.type !== "llm" && info.type !== "vlm") continue;
2903
- const ctx = info?.max_context_length;
2987
+ const ctx = info?.state === "loaded" ? contextLength(info.loaded_context_length) : void 0;
2904
2988
  models.push({
2905
2989
  rawId: entry.id,
2906
2990
  endpointId: endpoint.id,
2907
- contextWindow: typeof ctx === "number" && ctx > 0 ? ctx : FALLBACK_CONTEXT_WINDOW,
2908
- contextWindowKnown: typeof ctx === "number" && ctx > 0,
2991
+ contextWindow: ctx ?? FALLBACK_CONTEXT_WINDOW,
2992
+ contextWindowKnown: ctx !== void 0,
2909
2993
  // LM Studio doesn't report tool support; it gates per-model at request time.
2910
2994
  supportsTools: true,
2911
2995
  supportsImages: info?.type === "vlm",
@@ -2921,8 +3005,8 @@ async function enrichLlamaCpp(endpoint, entries, options) {
2921
3005
  endpoint,
2922
3006
  options
2923
3007
  );
2924
- const nCtx = props?.default_generation_settings?.n_ctx;
2925
- const known = typeof nCtx === "number" && nCtx > 0;
3008
+ const nCtx = contextLength(props?.default_generation_settings?.n_ctx);
3009
+ const known = nCtx !== void 0;
2926
3010
  return entries.map((entry) => ({
2927
3011
  ...genericModel(endpoint, entry),
2928
3012
  contextWindow: known ? nCtx : FALLBACK_CONTEXT_WINDOW,