@prestyj/core 5.12.0 → 5.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-234BODO4.js → chunk-OUE2GRO6.js} +134 -49
- package/dist/chunk-OUE2GRO6.js.map +1 -0
- package/dist/index.cjs +145 -54
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +3 -2
- package/dist/index.d.ts +3 -2
- package/dist/index.js +13 -7
- package/dist/index.js.map +1 -1
- package/dist/model-registry.cjs +132 -48
- package/dist/model-registry.cjs.map +1 -1
- package/dist/model-registry.d.cts +8 -7
- package/dist/model-registry.d.ts +8 -7
- package/dist/model-registry.js +1 -1
- package/package.json +2 -2
- package/dist/chunk-234BODO4.js.map +0 -1
|
@@ -1464,6 +1464,7 @@ var STATIC_API_KEY_PROVIDERS = /* @__PURE__ */ new Set([
|
|
|
1464
1464
|
"minimax",
|
|
1465
1465
|
"deepseek",
|
|
1466
1466
|
"openrouter",
|
|
1467
|
+
"huggingface",
|
|
1467
1468
|
"sakana",
|
|
1468
1469
|
"xai",
|
|
1469
1470
|
// Local endpoints: a fixed (usually placeholder) key, never refreshable.
|
|
@@ -1899,8 +1900,13 @@ var MODELS = [
|
|
|
1899
1900
|
// Project Glasswing (limited, invitation-only) model unavailable to most
|
|
1900
1901
|
// users. Re-enable once it's generally available.
|
|
1901
1902
|
{
|
|
1902
|
-
|
|
1903
|
-
|
|
1903
|
+
// Released 2026-09-01 — replaces Fable 5 at the same $10/$50 MTok (cache
|
|
1904
|
+
// reads drop to $0.25). Always-on adaptive thinking steered by effort;
|
|
1905
|
+
// forced tool use is rejected with a 400, which @prestyj/ai never sends on the
|
|
1906
|
+
// Anthropic path. Fable 5 is retired here — a session that still has it
|
|
1907
|
+
// saved falls back to the provider default on next start.
|
|
1908
|
+
id: "claude-fable-5-1",
|
|
1909
|
+
name: "Claude Fable 5.1",
|
|
1904
1910
|
provider: "anthropic",
|
|
1905
1911
|
contextWindow: 1e6,
|
|
1906
1912
|
maxOutputTokens: 128e3,
|
|
@@ -1912,7 +1918,7 @@ var MODELS = [
|
|
|
1912
1918
|
},
|
|
1913
1919
|
// {
|
|
1914
1920
|
// // Mythos-class model offered through Project Glasswing (limited
|
|
1915
|
-
// // availability, invitation-only). Same underlying model as Fable 5 with
|
|
1921
|
+
// // availability, invitation-only). Same underlying model as Fable 5.1 with
|
|
1916
1922
|
// // some safeguards lifted; kept here so approved accounts can select it.
|
|
1917
1923
|
// id: "claude-mythos-5",
|
|
1918
1924
|
// name: "Claude Mythos 5",
|
|
@@ -2063,11 +2069,29 @@ var MODELS = [
|
|
|
2063
2069
|
maxThinkingLevel: "xhigh"
|
|
2064
2070
|
},
|
|
2065
2071
|
// ── xAI (Grok) ─────────────────────────────────────────
|
|
2066
|
-
// Grok 4.
|
|
2067
|
-
//
|
|
2068
|
-
// `reasoning_effort`
|
|
2069
|
-
//
|
|
2070
|
-
//
|
|
2072
|
+
// Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
|
|
2073
|
+
// and knowledge work, with a focus on long-running agents — 500K context,
|
|
2074
|
+
// text+image input, and a `reasoning_effort` ladder that adds a new `xhigh`
|
|
2075
|
+
// top rung (low/medium/high default/xhigh; reasoning still can't be fully
|
|
2076
|
+
// disabled). $2/$6 per MTok under 200K prompt tokens ($4/$12 at or above),
|
|
2077
|
+
// and it's the default model of the Grok Build coding agent. xAI advertises "no text output limit"; we keep the same
|
|
2078
|
+
// 131K practical cap as 4.5 for budget predictability and input headroom.
|
|
2079
|
+
{
|
|
2080
|
+
id: "grok-4.6",
|
|
2081
|
+
name: "Grok 4.6",
|
|
2082
|
+
provider: "xai",
|
|
2083
|
+
contextWindow: 5e5,
|
|
2084
|
+
maxOutputTokens: 131072,
|
|
2085
|
+
supportsThinking: true,
|
|
2086
|
+
supportsImages: true,
|
|
2087
|
+
supportsVideo: false,
|
|
2088
|
+
costTier: "medium",
|
|
2089
|
+
maxThinkingLevel: "xhigh"
|
|
2090
|
+
},
|
|
2091
|
+
// Grok 4.5 (released 2026-07-08) — superseded by 4.6 but retained as an explicit option. 500K context, text+image input,
|
|
2092
|
+
// configurable `reasoning_effort` (low/medium/high, server default high;
|
|
2093
|
+
// reasoning can't be fully disabled). Served over the OpenAI-compatible API
|
|
2094
|
+
// at https://api.x.ai/v1 (API key from console.x.ai). xAI hasn't published an
|
|
2071
2095
|
// official max-output cap for 4.5; 131K matches the Grok Responses ceiling
|
|
2072
2096
|
// third-party integrations use.
|
|
2073
2097
|
{
|
|
@@ -2082,7 +2106,7 @@ var MODELS = [
|
|
|
2082
2106
|
costTier: "medium",
|
|
2083
2107
|
maxThinkingLevel: "high"
|
|
2084
2108
|
},
|
|
2085
|
-
// ── Gemini
|
|
2109
|
+
// ── Gemini ─────────────────────────────────────────
|
|
2086
2110
|
{
|
|
2087
2111
|
id: "gemini-3.1-flash-lite",
|
|
2088
2112
|
name: "Gemini 3.1 Flash Lite",
|
|
@@ -2096,6 +2120,28 @@ var MODELS = [
|
|
|
2096
2120
|
costTier: "low",
|
|
2097
2121
|
maxThinkingLevel: "high"
|
|
2098
2122
|
},
|
|
2123
|
+
{
|
|
2124
|
+
// Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
|
|
2125
|
+
// coding, agents, and multi-step execution; GA-stable on the Gemini API as
|
|
2126
|
+
// `gemini-3.7-flash`. 1M context, 64K output, thinking low/medium/high.
|
|
2127
|
+
// Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
|
|
2128
|
+
// hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
|
|
2129
|
+
// free/personal accounts 404 (entitlement-gated) while Code Assist
|
|
2130
|
+
// Standard/Enterprise accounts get it. Listed SECOND, after flash-lite:
|
|
2131
|
+
// getFastModel picks the first low-tier entry, and flash-lite is the one
|
|
2132
|
+
// that works on every account.
|
|
2133
|
+
id: "gemini-3.7-flash",
|
|
2134
|
+
name: "Gemini 3.7 Flash",
|
|
2135
|
+
provider: "gemini",
|
|
2136
|
+
contextWindow: 1048576,
|
|
2137
|
+
maxOutputTokens: 65536,
|
|
2138
|
+
supportsThinking: true,
|
|
2139
|
+
supportsImages: true,
|
|
2140
|
+
supportsVideo: true,
|
|
2141
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2142
|
+
costTier: "low",
|
|
2143
|
+
maxThinkingLevel: "high"
|
|
2144
|
+
},
|
|
2099
2145
|
{
|
|
2100
2146
|
// Wire name `gemini-3-flash` — the Code Assist (OAuth) backend rejects the
|
|
2101
2147
|
// display string `gemini-3.5-flash` with a 404, so gemini-cli keeps this
|
|
@@ -2163,11 +2209,15 @@ var MODELS = [
|
|
|
2163
2209
|
maxThinkingLevel: "high"
|
|
2164
2210
|
},
|
|
2165
2211
|
// ── Z.AI (GLM) ─────────────────────────────────────────
|
|
2166
|
-
// GLM
|
|
2167
|
-
//
|
|
2212
|
+
// Two GLM entries, both live on the coding endpoint (verified against its
|
|
2213
|
+
// /models list). The pre-5.3 ids stay retired: they routed to strictly worse
|
|
2214
|
+
// coding for the same plan quota, and the endpoint already answers `glm-5.2`
|
|
2215
|
+
// requests as glm-5.3.
|
|
2216
|
+
// `max` is both the ceiling and Z.AI's own default — the rungs below it live
|
|
2217
|
+
// in thinking-level.ts.
|
|
2168
2218
|
{
|
|
2169
|
-
id: "glm-5.
|
|
2170
|
-
name: "GLM-5.
|
|
2219
|
+
id: "glm-5.3",
|
|
2220
|
+
name: "GLM-5.3",
|
|
2171
2221
|
provider: "glm",
|
|
2172
2222
|
contextWindow: 1e6,
|
|
2173
2223
|
maxOutputTokens: 131072,
|
|
@@ -2175,43 +2225,31 @@ var MODELS = [
|
|
|
2175
2225
|
supportsImages: false,
|
|
2176
2226
|
supportsVideo: false,
|
|
2177
2227
|
costTier: "medium",
|
|
2178
|
-
maxThinkingLevel: "
|
|
2179
|
-
},
|
|
2180
|
-
{
|
|
2181
|
-
id: "glm-5.1",
|
|
2182
|
-
name: "GLM-5.1",
|
|
2183
|
-
provider: "glm",
|
|
2184
|
-
contextWindow: 204800,
|
|
2185
|
-
maxOutputTokens: 131072,
|
|
2186
|
-
supportsThinking: true,
|
|
2187
|
-
supportsImages: false,
|
|
2188
|
-
supportsVideo: false,
|
|
2189
|
-
costTier: "medium",
|
|
2190
|
-
maxThinkingLevel: "high"
|
|
2191
|
-
},
|
|
2192
|
-
{
|
|
2193
|
-
id: "glm-4.7",
|
|
2194
|
-
name: "GLM-4.7",
|
|
2195
|
-
provider: "glm",
|
|
2196
|
-
contextWindow: 2e5,
|
|
2197
|
-
maxOutputTokens: 131072,
|
|
2198
|
-
supportsThinking: true,
|
|
2199
|
-
supportsImages: false,
|
|
2200
|
-
supportsVideo: false,
|
|
2201
|
-
costTier: "low",
|
|
2202
|
-
maxThinkingLevel: "high"
|
|
2228
|
+
maxThinkingLevel: "max"
|
|
2203
2229
|
},
|
|
2230
|
+
// GLM-5.3-Flash (released 2026-08-26): 320B-A18B natively multimodal sibling
|
|
2231
|
+
// at ~1/20th of 5.3's API price with 3× the coding-plan quota, so it is the
|
|
2232
|
+
// provider's `low` tier — scout sub-agents and compaction summaries route
|
|
2233
|
+
// here instead of paying 5.3 rates.
|
|
2234
|
+
// Images are native on the coding endpoint (verified: base64 data URL in an
|
|
2235
|
+
// `image_url` block answers correctly), which also means GLM image
|
|
2236
|
+
// attachments go inline for this model rather than through the zai_vision MCP
|
|
2237
|
+
// detour that `supportsImages: false` triggers.
|
|
2238
|
+
// Video/file input is documented but unverified on this transport, so it
|
|
2239
|
+
// stays off until measured. Thinking cannot be disabled server-side (Z.AI
|
|
2240
|
+
// maps a `disabled` toggle to the `low` rung and answers 200), and unlike
|
|
2241
|
+
// 5.3 it accepts any reasoning_effort string without a 400.
|
|
2204
2242
|
{
|
|
2205
|
-
id: "glm-
|
|
2206
|
-
name: "GLM-
|
|
2243
|
+
id: "glm-5.3-flash",
|
|
2244
|
+
name: "GLM-5.3-Flash",
|
|
2207
2245
|
provider: "glm",
|
|
2208
|
-
contextWindow:
|
|
2246
|
+
contextWindow: 1e6,
|
|
2209
2247
|
maxOutputTokens: 131072,
|
|
2210
2248
|
supportsThinking: true,
|
|
2211
|
-
supportsImages:
|
|
2249
|
+
supportsImages: true,
|
|
2212
2250
|
supportsVideo: false,
|
|
2213
2251
|
costTier: "low",
|
|
2214
|
-
maxThinkingLevel: "
|
|
2252
|
+
maxThinkingLevel: "max"
|
|
2215
2253
|
},
|
|
2216
2254
|
// ── MiniMax ────────────────────────────────────────────
|
|
2217
2255
|
{
|
|
@@ -2279,15 +2317,21 @@ var MODELS = [
|
|
|
2279
2317
|
},
|
|
2280
2318
|
// ── DeepSeek ───────────────────────────────────────────
|
|
2281
2319
|
{
|
|
2320
|
+
// `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
|
|
2321
|
+
// first STABLE V4 Pro — supersedes the April preview; calling name
|
|
2322
|
+
// unchanged, same 1.6T/49B MoE). 1M context, 384K (393,216) max output,
|
|
2323
|
+
// text-only, reasoning ladder low/high plus Think Max — mapped from our
|
|
2324
|
+
// `xhigh`. ~$0.43/$0.87 per MTok on DeepSeek's own API, so a mid-tier
|
|
2325
|
+
// price band rather than the preview's top band.
|
|
2282
2326
|
id: "deepseek-v4-pro",
|
|
2283
2327
|
name: "DeepSeek V4 Pro",
|
|
2284
2328
|
provider: "deepseek",
|
|
2285
2329
|
contextWindow: 1048576,
|
|
2286
|
-
maxOutputTokens:
|
|
2330
|
+
maxOutputTokens: 393216,
|
|
2287
2331
|
supportsThinking: true,
|
|
2288
2332
|
supportsImages: false,
|
|
2289
2333
|
supportsVideo: false,
|
|
2290
|
-
costTier: "
|
|
2334
|
+
costTier: "medium",
|
|
2291
2335
|
// DeepSeek V4 maps `xhigh` → its internal `max` tier.
|
|
2292
2336
|
maxThinkingLevel: "xhigh"
|
|
2293
2337
|
},
|
|
@@ -2315,6 +2359,45 @@ var MODELS = [
|
|
|
2315
2359
|
supportsVideo: false,
|
|
2316
2360
|
costTier: "medium",
|
|
2317
2361
|
maxThinkingLevel: "high"
|
|
2362
|
+
},
|
|
2363
|
+
// ── Hugging Face (Inference Providers router) ────────
|
|
2364
|
+
// One HF token (hf.co/settings/tokens, "Make calls to Inference Providers"
|
|
2365
|
+
// permission) routes to whichever hosted backend serves each open model;
|
|
2366
|
+
// billing follows each backend's rates on the HF account (small free tier).
|
|
2367
|
+
// Model ids are Hub repo paths, so they intentionally contain a slash — the
|
|
2368
|
+
// same shape local/ vLLM ids already use (`local/vllm/Qwen/Qwen3-32B`).
|
|
2369
|
+
{
|
|
2370
|
+
// Qwen's open flagship for agentic coding — tool-calling native, non-thinking
|
|
2371
|
+
// (the Coder line dropped the <think> block). 262K native context (1M needs
|
|
2372
|
+
// YaRN, which the router doesn't apply), 131K max output. :auto suffix lets
|
|
2373
|
+
// HF pick the backend with capacity; we keep the bare repo id so the picker
|
|
2374
|
+
// matches what GET /v1/models reports.
|
|
2375
|
+
id: "Qwen/Qwen3-Coder-480B-A35B-Instruct",
|
|
2376
|
+
name: "Qwen3 Coder 480B",
|
|
2377
|
+
provider: "huggingface",
|
|
2378
|
+
contextWindow: 262144,
|
|
2379
|
+
maxOutputTokens: 131072,
|
|
2380
|
+
supportsThinking: false,
|
|
2381
|
+
supportsImages: false,
|
|
2382
|
+
supportsVideo: false,
|
|
2383
|
+
costTier: "medium",
|
|
2384
|
+
maxThinkingLevel: "low"
|
|
2385
|
+
},
|
|
2386
|
+
{
|
|
2387
|
+
// OpenAI's open-weight 120B MoE (5.1B active) — general-purpose, tool-calling
|
|
2388
|
+
// native, adjustable reasoning effort (low/medium/high, default medium) over
|
|
2389
|
+
// the router's Chat Completions API. Cheap enough to be the low-tier sibling
|
|
2390
|
+
// for summaries and fast sub-agents.
|
|
2391
|
+
id: "openai/gpt-oss-120b",
|
|
2392
|
+
name: "GPT-OSS 120B",
|
|
2393
|
+
provider: "huggingface",
|
|
2394
|
+
contextWindow: 131072,
|
|
2395
|
+
maxOutputTokens: 65536,
|
|
2396
|
+
supportsThinking: true,
|
|
2397
|
+
supportsImages: false,
|
|
2398
|
+
supportsVideo: false,
|
|
2399
|
+
costTier: "low",
|
|
2400
|
+
maxThinkingLevel: "high"
|
|
2318
2401
|
}
|
|
2319
2402
|
];
|
|
2320
2403
|
var runtimeModels = /* @__PURE__ */ new Map();
|
|
@@ -2356,13 +2439,15 @@ function getDefaultModel(provider) {
|
|
|
2356
2439
|
if (provider === "xiaomi") return MODELS.find((m) => m.id === "mimo-v2.5-pro");
|
|
2357
2440
|
if (provider === "openai") return MODELS.find((m) => m.id === "gpt-5.6-sol");
|
|
2358
2441
|
if (provider === "gemini") return MODELS.find((m) => m.id === "gemini-3.1-flash-lite");
|
|
2359
|
-
if (provider === "glm") return MODELS.find((m) => m.id === "glm-5.
|
|
2442
|
+
if (provider === "glm") return MODELS.find((m) => m.id === "glm-5.3");
|
|
2360
2443
|
if (provider === "moonshot") return MODELS.find((m) => m.id === "kimi-k3");
|
|
2361
2444
|
if (provider === "minimax") return MODELS.find((m) => m.id === "MiniMax-M3");
|
|
2362
2445
|
if (provider === "deepseek") return MODELS.find((m) => m.id === "deepseek-v4-pro");
|
|
2446
|
+
if (provider === "huggingface")
|
|
2447
|
+
return MODELS.find((m) => m.id === "Qwen/Qwen3-Coder-480B-A35B-Instruct");
|
|
2363
2448
|
if (provider === "openrouter") return MODELS.find((m) => m.id === "qwen/qwen3.6-plus");
|
|
2364
2449
|
if (provider === "sakana") return MODELS.find((m) => m.id === "fugu");
|
|
2365
|
-
if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.
|
|
2450
|
+
if (provider === "xai") return MODELS.find((m) => m.id === "grok-4.6");
|
|
2366
2451
|
if (provider === "local") {
|
|
2367
2452
|
return getModelsForProvider("local")[0] ?? PLACEHOLDER_LOCAL_MODEL;
|
|
2368
2453
|
}
|
|
@@ -2406,7 +2491,7 @@ function getSummaryModel(provider, currentModelId) {
|
|
|
2406
2491
|
if (provider === "anthropic") {
|
|
2407
2492
|
return MODELS.find((m) => m.id === "claude-sonnet-5");
|
|
2408
2493
|
}
|
|
2409
|
-
if (provider === "openai" || provider === "glm" || provider === "deepseek") {
|
|
2494
|
+
if (provider === "openai" || provider === "glm" || provider === "deepseek" || provider === "huggingface") {
|
|
2410
2495
|
const low = getModelsForProvider(provider).find((m) => m.costTier === "low");
|
|
2411
2496
|
if (low) return low;
|
|
2412
2497
|
}
|
|
@@ -2474,4 +2559,4 @@ export {
|
|
|
2474
2559
|
getSummaryModel,
|
|
2475
2560
|
getFastModel
|
|
2476
2561
|
};
|
|
2477
|
-
//# sourceMappingURL=chunk-
|
|
2562
|
+
//# sourceMappingURL=chunk-OUE2GRO6.js.map
|