@prestyj/core 5.16.1 → 5.16.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-OUE2GRO6.js → chunk-3MQNCB44.js} +80 -28
- package/dist/{chunk-OUE2GRO6.js.map → chunk-3MQNCB44.js.map} +1 -1
- package/dist/index.cjs +138 -54
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +3 -3
- package/dist/index.d.ts +3 -3
- package/dist/index.js +60 -28
- package/dist/index.js.map +1 -1
- package/dist/model-registry.cjs +79 -27
- package/dist/model-registry.cjs.map +1 -1
- package/dist/model-registry.d.cts +2 -3
- package/dist/model-registry.d.ts +2 -3
- package/dist/model-registry.js +1 -1
- package/package.json +2 -2
package/dist/index.cjs
CHANGED
|
@@ -2113,6 +2113,29 @@ var MODELS = [
|
|
|
2113
2113
|
maxThinkingLevel: "high"
|
|
2114
2114
|
},
|
|
2115
2115
|
// ── OpenAI (Codex) ─────────────────────────────────────
|
|
2116
|
+
{
|
|
2117
|
+
// GPT-6 Astra — "Our most capable model for complex, demanding work."
|
|
2118
|
+
// (Codex catalog priority 1, listed for every ChatGPT plan, requires a
|
|
2119
|
+
// Codex client >= 0.153.0 — see CODEX_CLIENT_VERSION). Same split as 5.6:
|
|
2120
|
+
// 1.05M on the public Responses API, 272K on the ChatGPT OAuth route
|
|
2121
|
+
// (openai/codex models.json, `gpt-6-astra`). Reasoning ladder low → medium
|
|
2122
|
+
// → high → xhigh → max → ultra; `ultra` is the Codex orchestration preset
|
|
2123
|
+
// (multi_agent v2) and is Codex-only — the public API tops out at `max`.
|
|
2124
|
+
// Note: through a plain API key OpenAI requires the Responses API for tool
|
|
2125
|
+
// calling on Astra, so the Chat Completions path is text-only; the OAuth
|
|
2126
|
+
// Codex route is the supported way to use it as an agent.
|
|
2127
|
+
id: "gpt-6-astra",
|
|
2128
|
+
name: "GPT-6 Astra",
|
|
2129
|
+
provider: "openai",
|
|
2130
|
+
contextWindow: 105e4,
|
|
2131
|
+
codexContextWindow: 272e3,
|
|
2132
|
+
maxOutputTokens: 128e3,
|
|
2133
|
+
supportsThinking: true,
|
|
2134
|
+
supportsImages: true,
|
|
2135
|
+
supportsVideo: false,
|
|
2136
|
+
costTier: "high",
|
|
2137
|
+
maxThinkingLevel: "ultra"
|
|
2138
|
+
},
|
|
2116
2139
|
// GPT-5.6 family — three agentic coding tiers launched July 2026. The public
|
|
2117
2140
|
// Responses API advertises a 1.05M context window; OpenAI's Codex product
|
|
2118
2141
|
// catalog advertises 272K on the ChatGPT OAuth route (corrected from the
|
|
@@ -2166,24 +2189,11 @@ var MODELS = [
|
|
|
2166
2189
|
costTier: "low",
|
|
2167
2190
|
maxThinkingLevel: "max"
|
|
2168
2191
|
},
|
|
2169
|
-
{
|
|
2170
|
-
id: "gpt-5.5",
|
|
2171
|
-
name: "GPT-5.5",
|
|
2172
|
-
provider: "openai",
|
|
2173
|
-
contextWindow: 105e4,
|
|
2174
|
-
codexContextWindow: 272e3,
|
|
2175
|
-
maxOutputTokens: 128e3,
|
|
2176
|
-
supportsThinking: true,
|
|
2177
|
-
supportsImages: true,
|
|
2178
|
-
supportsVideo: false,
|
|
2179
|
-
costTier: "high",
|
|
2180
|
-
maxThinkingLevel: "xhigh"
|
|
2181
|
-
},
|
|
2182
2192
|
// ── Sakana (Fugu) ──────────────────────────────────────
|
|
2183
2193
|
// Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
|
|
2184
2194
|
// OpenAI-compatible Sakana API (https://api.sakana.ai/v1). Both models take
|
|
2185
|
-
// text + image input
|
|
2186
|
-
//
|
|
2195
|
+
// text + image input. Plain Fugu stops at xhigh; Ultra v1.1 also supports max.
|
|
2196
|
+
// `fugu` routes across all providers; `fugu-ultra` is
|
|
2187
2197
|
// the heavier tier (may need larger client timeouts on complex tasks).
|
|
2188
2198
|
{
|
|
2189
2199
|
id: "fugu",
|
|
@@ -2207,7 +2217,8 @@ var MODELS = [
|
|
|
2207
2217
|
supportsImages: true,
|
|
2208
2218
|
supportsVideo: false,
|
|
2209
2219
|
costTier: "high",
|
|
2210
|
-
|
|
2220
|
+
// The rolling alias now serves v1.1, which adds a distinct max effort.
|
|
2221
|
+
maxThinkingLevel: "max"
|
|
2211
2222
|
},
|
|
2212
2223
|
// ── xAI (Grok) ─────────────────────────────────────────
|
|
2213
2224
|
// Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
|
|
@@ -2261,6 +2272,34 @@ var MODELS = [
|
|
|
2261
2272
|
costTier: "low",
|
|
2262
2273
|
maxThinkingLevel: "high"
|
|
2263
2274
|
},
|
|
2275
|
+
// Keep 3.1 Flash Lite first for the working OAuth default and fast-model routing.
|
|
2276
|
+
// New GA models are opt-in; Code Assist access varies by account.
|
|
2277
|
+
{
|
|
2278
|
+
id: "gemini-3.8-flash",
|
|
2279
|
+
name: "Gemini 3.8 Flash",
|
|
2280
|
+
provider: "gemini",
|
|
2281
|
+
contextWindow: 1048576,
|
|
2282
|
+
maxOutputTokens: 65536,
|
|
2283
|
+
supportsThinking: true,
|
|
2284
|
+
supportsImages: true,
|
|
2285
|
+
supportsVideo: true,
|
|
2286
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2287
|
+
costTier: "low",
|
|
2288
|
+
maxThinkingLevel: "high"
|
|
2289
|
+
},
|
|
2290
|
+
{
|
|
2291
|
+
id: "gemini-3.5-flash-lite",
|
|
2292
|
+
name: "Gemini 3.5 Flash Lite",
|
|
2293
|
+
provider: "gemini",
|
|
2294
|
+
contextWindow: 1048576,
|
|
2295
|
+
maxOutputTokens: 65536,
|
|
2296
|
+
supportsThinking: true,
|
|
2297
|
+
supportsImages: true,
|
|
2298
|
+
supportsVideo: true,
|
|
2299
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2300
|
+
costTier: "low",
|
|
2301
|
+
maxThinkingLevel: "high"
|
|
2302
|
+
},
|
|
2264
2303
|
{
|
|
2265
2304
|
// Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
|
|
2266
2305
|
// coding, agents, and multi-step execution; GA-stable on the Gemini API as
|
|
@@ -2268,7 +2307,7 @@ var MODELS = [
|
|
|
2268
2307
|
// Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
|
|
2269
2308
|
// hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
|
|
2270
2309
|
// free/personal accounts 404 (entitlement-gated) while Code Assist
|
|
2271
|
-
// Standard/Enterprise accounts get it.
|
|
2310
|
+
// Standard/Enterprise accounts get it. Kept after the working flash-lite:
|
|
2272
2311
|
// getFastModel picks the first low-tier entry, and flash-lite is the one
|
|
2273
2312
|
// that works on every account.
|
|
2274
2313
|
id: "gemini-3.7-flash",
|
|
@@ -2460,21 +2499,19 @@ var MODELS = [
|
|
|
2460
2499
|
{
|
|
2461
2500
|
// `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
|
|
2462
2501
|
// first STABLE V4 Pro — supersedes the April preview; calling name
|
|
2463
|
-
// unchanged, same 1.6T/49B MoE). 1M context,
|
|
2464
|
-
//
|
|
2465
|
-
//
|
|
2466
|
-
// price band rather than the preview's top band.
|
|
2502
|
+
// unchanged, same 1.6T/49B MoE). 1M context, text-only, low/high/max effort.
|
|
2503
|
+
// Docs abbreviate output as 384K; use the same conservative 384,000-token
|
|
2504
|
+
// application cap across V4 models rather than mixing decimal/binary units.
|
|
2467
2505
|
id: "deepseek-v4-pro",
|
|
2468
2506
|
name: "DeepSeek V4 Pro",
|
|
2469
2507
|
provider: "deepseek",
|
|
2470
2508
|
contextWindow: 1048576,
|
|
2471
|
-
maxOutputTokens:
|
|
2509
|
+
maxOutputTokens: 384e3,
|
|
2472
2510
|
supportsThinking: true,
|
|
2473
2511
|
supportsImages: false,
|
|
2474
2512
|
supportsVideo: false,
|
|
2475
2513
|
costTier: "medium",
|
|
2476
|
-
|
|
2477
|
-
maxThinkingLevel: "xhigh"
|
|
2514
|
+
maxThinkingLevel: "max"
|
|
2478
2515
|
},
|
|
2479
2516
|
{
|
|
2480
2517
|
id: "deepseek-v4-flash",
|
|
@@ -2486,7 +2523,20 @@ var MODELS = [
|
|
|
2486
2523
|
supportsImages: false,
|
|
2487
2524
|
supportsVideo: false,
|
|
2488
2525
|
costTier: "low",
|
|
2489
|
-
maxThinkingLevel: "
|
|
2526
|
+
maxThinkingLevel: "max"
|
|
2527
|
+
},
|
|
2528
|
+
// Opt-in experimental vision sibling; never replaces the stable summary model.
|
|
2529
|
+
{
|
|
2530
|
+
id: "deepseek-v4-flash-vision-exp",
|
|
2531
|
+
name: "DeepSeek V4 Flash Vision (Experimental)",
|
|
2532
|
+
provider: "deepseek",
|
|
2533
|
+
contextWindow: 1048576,
|
|
2534
|
+
maxOutputTokens: 384e3,
|
|
2535
|
+
supportsThinking: true,
|
|
2536
|
+
supportsImages: true,
|
|
2537
|
+
supportsVideo: false,
|
|
2538
|
+
costTier: "low",
|
|
2539
|
+
maxThinkingLevel: "max"
|
|
2490
2540
|
},
|
|
2491
2541
|
// ── OpenRouter ─────────────────────────────────────────
|
|
2492
2542
|
{
|
|
@@ -2496,8 +2546,10 @@ var MODELS = [
|
|
|
2496
2546
|
contextWindow: 1e6,
|
|
2497
2547
|
maxOutputTokens: 65536,
|
|
2498
2548
|
supportsThinking: true,
|
|
2499
|
-
supportsImages:
|
|
2500
|
-
supportsVideo:
|
|
2549
|
+
supportsImages: true,
|
|
2550
|
+
supportsVideo: true,
|
|
2551
|
+
// Practical inline-payload cap, not an asserted provider maximum.
|
|
2552
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2501
2553
|
costTier: "medium",
|
|
2502
2554
|
maxThinkingLevel: "high"
|
|
2503
2555
|
},
|
|
@@ -2653,7 +2705,8 @@ var OPENAI_GPT_56_THINKING_LEVELS = [
|
|
|
2653
2705
|
"max",
|
|
2654
2706
|
"ultra"
|
|
2655
2707
|
];
|
|
2656
|
-
var SAKANA_THINKING_LEVELS = ["high", "xhigh"];
|
|
2708
|
+
var SAKANA_THINKING_LEVELS = ["high", "xhigh", "max"];
|
|
2709
|
+
var DEEPSEEK_THINKING_LEVELS = ["low", "high", "max"];
|
|
2657
2710
|
var XAI_THINKING_LEVELS = ["low", "medium", "high", "xhigh"];
|
|
2658
2711
|
var ANTHROPIC_XHIGH_THINKING_LEVELS = [
|
|
2659
2712
|
"low",
|
|
@@ -2717,13 +2770,14 @@ function getSupportedThinkingLevels(provider, model) {
|
|
|
2717
2770
|
return XAI_THINKING_LEVELS.slice(0, maxIndex2 + 1);
|
|
2718
2771
|
}
|
|
2719
2772
|
if (isMoonshotK3Model(provider, model)) return MOONSHOT_K3_THINKING_LEVELS;
|
|
2773
|
+
if (provider === "deepseek") return DEEPSEEK_THINKING_LEVELS;
|
|
2720
2774
|
if (isGlmModel(provider)) {
|
|
2721
2775
|
const maxIndex2 = GLM_THINKING_LEVELS.indexOf(maxLevel);
|
|
2722
2776
|
if (maxIndex2 === -1) return GLM_THINKING_LEVELS;
|
|
2723
2777
|
return GLM_THINKING_LEVELS.slice(0, maxIndex2 + 1);
|
|
2724
2778
|
}
|
|
2725
2779
|
if (!isOpenAIGptModel(provider, model)) return [maxLevel];
|
|
2726
|
-
const levels = model.startsWith("gpt-5.6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
|
|
2780
|
+
const levels = model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
|
|
2727
2781
|
const maxIndex = levels.indexOf(maxLevel);
|
|
2728
2782
|
if (maxIndex === -1) return ["medium"];
|
|
2729
2783
|
return levels.slice(0, maxIndex + 1);
|
|
@@ -2733,7 +2787,7 @@ function isThinkingLevelSupported(provider, model, level) {
|
|
|
2733
2787
|
}
|
|
2734
2788
|
function getNextThinkingLevel(provider, model, current) {
|
|
2735
2789
|
const supportedLevels = getSupportedThinkingLevels(provider, model);
|
|
2736
|
-
const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || // Local servers take a real effort level, not just on/off: Ollama accepts
|
|
2790
|
+
const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || provider === "deepseek" || // Local servers take a real effort level, not just on/off: Ollama accepts
|
|
2737
2791
|
// low/medium/high on `reasoning_effort` (verified against 0.32) and the
|
|
2738
2792
|
// other OpenAI-compatible servers use the same three. A model that can't
|
|
2739
2793
|
// reason at all already has no supported levels, so it never gets here.
|
|
@@ -2751,7 +2805,7 @@ function getNextThinkingLevel(provider, model, current) {
|
|
|
2751
2805
|
var DEFAULT_LOCAL_ENDPOINTS = [
|
|
2752
2806
|
{ id: "ollama", label: "Ollama", baseUrl: "http://127.0.0.1:11434/v1", kind: "ollama" }
|
|
2753
2807
|
];
|
|
2754
|
-
var FALLBACK_CONTEXT_WINDOW =
|
|
2808
|
+
var FALLBACK_CONTEXT_WINDOW = 4096;
|
|
2755
2809
|
var LOCAL_API_KEY_PLACEHOLDER = "local";
|
|
2756
2810
|
var DEFAULT_PROBE_TIMEOUT_MS = 1200;
|
|
2757
2811
|
var ENRICH_CONCURRENCY = 6;
|
|
@@ -2844,7 +2898,7 @@ async function enrich(endpoint, entries, options) {
|
|
|
2844
2898
|
return entries.map((entry) => genericModel(endpoint, entry));
|
|
2845
2899
|
}
|
|
2846
2900
|
function genericModel(endpoint, entry) {
|
|
2847
|
-
const declared =
|
|
2901
|
+
const declared = contextLength(entry.max_model_len);
|
|
2848
2902
|
return {
|
|
2849
2903
|
rawId: entry.id,
|
|
2850
2904
|
endpointId: endpoint.id,
|
|
@@ -2857,16 +2911,23 @@ function genericModel(endpoint, entry) {
|
|
|
2857
2911
|
}
|
|
2858
2912
|
async function enrichOllama(endpoint, entries, options) {
|
|
2859
2913
|
const showUrl = `${endpointRoot(endpoint.baseUrl)}/api/show`;
|
|
2914
|
+
const running = await fetchJson(
|
|
2915
|
+
`${endpointRoot(endpoint.baseUrl)}/api/ps`,
|
|
2916
|
+
endpoint,
|
|
2917
|
+
options
|
|
2918
|
+
);
|
|
2919
|
+
const runningModels = Array.isArray(running?.models) ? running.models : [];
|
|
2860
2920
|
const enriched = await mapLimited(entries, ENRICH_CONCURRENCY, async (entry) => {
|
|
2861
2921
|
const show = await fetchJson(showUrl, endpoint, {
|
|
2862
2922
|
...options,
|
|
2863
2923
|
method: "POST",
|
|
2864
2924
|
body: { model: entry.id }
|
|
2865
2925
|
});
|
|
2866
|
-
|
|
2867
|
-
const caps = show.capabilities ?? [];
|
|
2926
|
+
const caps = Array.isArray(show?.capabilities) ? show.capabilities : [];
|
|
2868
2927
|
if (caps.includes("embedding") && !caps.includes("completion")) return void 0;
|
|
2869
|
-
const ctx =
|
|
2928
|
+
const ctx = contextLength(
|
|
2929
|
+
runningModels.find((model) => model?.name === entry.id || model?.model === entry.id)?.context_length
|
|
2930
|
+
);
|
|
2870
2931
|
return {
|
|
2871
2932
|
rawId: entry.id,
|
|
2872
2933
|
endpointId: endpoint.id,
|
|
@@ -2874,38 +2935,61 @@ async function enrichOllama(endpoint, entries, options) {
|
|
|
2874
2935
|
contextWindowKnown: ctx !== void 0,
|
|
2875
2936
|
// Ollama reports capabilities honestly, so trust it here rather than
|
|
2876
2937
|
// using the optimistic generic default.
|
|
2877
|
-
supportsTools: caps.includes("tools"),
|
|
2938
|
+
supportsTools: show ? caps.includes("tools") : true,
|
|
2878
2939
|
supportsImages: caps.includes("vision"),
|
|
2879
2940
|
supportsThinking: caps.includes("thinking")
|
|
2880
2941
|
};
|
|
2881
2942
|
});
|
|
2882
2943
|
return enriched.filter((model) => model !== void 0);
|
|
2883
2944
|
}
|
|
2884
|
-
function
|
|
2885
|
-
|
|
2886
|
-
for (const [key, value] of Object.entries(info)) {
|
|
2887
|
-
if (key.endsWith(".context_length") && typeof value === "number" && value > 0) return value;
|
|
2888
|
-
}
|
|
2889
|
-
return void 0;
|
|
2945
|
+
function contextLength(value) {
|
|
2946
|
+
return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : void 0;
|
|
2890
2947
|
}
|
|
2891
2948
|
async function enrichLmStudio(endpoint, entries, options) {
|
|
2892
|
-
const
|
|
2893
|
-
|
|
2894
|
-
|
|
2895
|
-
|
|
2896
|
-
|
|
2897
|
-
|
|
2898
|
-
|
|
2949
|
+
const root = endpointRoot(endpoint.baseUrl);
|
|
2950
|
+
const detail = await fetchJson(`${root}/api/v1/models`, endpoint, options);
|
|
2951
|
+
const byId = /* @__PURE__ */ new Map();
|
|
2952
|
+
if (Array.isArray(detail?.models)) {
|
|
2953
|
+
for (const model of detail.models) {
|
|
2954
|
+
if (typeof model?.key !== "string") continue;
|
|
2955
|
+
const instances = Array.isArray(model.loaded_instances) ? model.loaded_instances : [];
|
|
2956
|
+
const contexts = instances.map((instance) => contextLength(instance?.config?.context_length));
|
|
2957
|
+
const ctx = contexts.length && contexts.every((value) => value !== void 0) ? contexts.reduce((min, value) => Math.min(min, value)) : void 0;
|
|
2958
|
+
const info = {
|
|
2959
|
+
id: model.key,
|
|
2960
|
+
type: model.type === "llm" && model.capabilities?.vision === true ? "vlm" : model.type,
|
|
2961
|
+
state: instances.length ? "loaded" : "not-loaded",
|
|
2962
|
+
loaded_context_length: ctx
|
|
2963
|
+
};
|
|
2964
|
+
byId.set(model.key, info);
|
|
2965
|
+
for (const instance of instances) {
|
|
2966
|
+
if (typeof instance?.id === "string" && instance.id !== model.key) {
|
|
2967
|
+
byId.set(instance.id, {
|
|
2968
|
+
...info,
|
|
2969
|
+
id: instance.id,
|
|
2970
|
+
loaded_context_length: contextLength(instance.config?.context_length)
|
|
2971
|
+
});
|
|
2972
|
+
}
|
|
2973
|
+
}
|
|
2974
|
+
}
|
|
2975
|
+
} else {
|
|
2976
|
+
const legacy = await fetchJson(`${root}/api/v0/models`, endpoint, options);
|
|
2977
|
+
if (Array.isArray(legacy?.data)) {
|
|
2978
|
+
for (const model of legacy.data) {
|
|
2979
|
+
if (typeof model?.id === "string") byId.set(model.id, model);
|
|
2980
|
+
}
|
|
2981
|
+
}
|
|
2982
|
+
}
|
|
2899
2983
|
const models = [];
|
|
2900
2984
|
for (const entry of entries) {
|
|
2901
2985
|
const info = byId.get(entry.id);
|
|
2902
2986
|
if (info && info.type !== "llm" && info.type !== "vlm") continue;
|
|
2903
|
-
const ctx = info?.
|
|
2987
|
+
const ctx = info?.state === "loaded" ? contextLength(info.loaded_context_length) : void 0;
|
|
2904
2988
|
models.push({
|
|
2905
2989
|
rawId: entry.id,
|
|
2906
2990
|
endpointId: endpoint.id,
|
|
2907
|
-
contextWindow:
|
|
2908
|
-
contextWindowKnown:
|
|
2991
|
+
contextWindow: ctx ?? FALLBACK_CONTEXT_WINDOW,
|
|
2992
|
+
contextWindowKnown: ctx !== void 0,
|
|
2909
2993
|
// LM Studio doesn't report tool support; it gates per-model at request time.
|
|
2910
2994
|
supportsTools: true,
|
|
2911
2995
|
supportsImages: info?.type === "vlm",
|
|
@@ -2921,8 +3005,8 @@ async function enrichLlamaCpp(endpoint, entries, options) {
|
|
|
2921
3005
|
endpoint,
|
|
2922
3006
|
options
|
|
2923
3007
|
);
|
|
2924
|
-
const nCtx = props?.default_generation_settings?.n_ctx;
|
|
2925
|
-
const known =
|
|
3008
|
+
const nCtx = contextLength(props?.default_generation_settings?.n_ctx);
|
|
3009
|
+
const known = nCtx !== void 0;
|
|
2926
3010
|
return entries.map((entry) => ({
|
|
2927
3011
|
...genericModel(endpoint, entry),
|
|
2928
3012
|
contextWindow: known ? nCtx : FALLBACK_CONTEXT_WINDOW,
|