@prestyj/core 5.16.1 → 5.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-OUE2GRO6.js → chunk-LQNMFHK5.js} +86 -30
- package/dist/chunk-LQNMFHK5.js.map +1 -0
- package/dist/index.cjs +144 -56
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +6 -4
- package/dist/index.d.ts +6 -4
- package/dist/index.js +60 -28
- package/dist/index.js.map +1 -1
- package/dist/model-registry.cjs +79 -27
- package/dist/model-registry.cjs.map +1 -1
- package/dist/model-registry.d.cts +2 -3
- package/dist/model-registry.d.ts +2 -3
- package/dist/model-registry.js +1 -1
- package/package.json +2 -2
- package/dist/chunk-OUE2GRO6.js.map +0 -1
package/dist/index.cjs
CHANGED
|
@@ -1616,7 +1616,8 @@ var AuthStorage = class {
|
|
|
1616
1616
|
filePath;
|
|
1617
1617
|
loaded = false;
|
|
1618
1618
|
/**
|
|
1619
|
-
* mtime+size of the
|
|
1619
|
+
* inode+mtime+size of the cached file (`size: -1` = no file). The inode
|
|
1620
|
+
* detects atomic replacements with equal size within one filesystem clock tick.
|
|
1620
1621
|
* auth.json is shared: the desktop app writes API keys and disconnects
|
|
1621
1622
|
* NATIVELY (so they work with no daemon running), and every window/process has
|
|
1622
1623
|
* its own AuthStorage. A load-once cache therefore goes stale — the sidecar
|
|
@@ -1625,6 +1626,7 @@ var AuthStorage = class {
|
|
|
1625
1626
|
*/
|
|
1626
1627
|
snapshotMtimeMs = 0;
|
|
1627
1628
|
snapshotSize = -1;
|
|
1629
|
+
snapshotIno = 0;
|
|
1628
1630
|
/** Per-provider lock to serialize concurrent refresh calls. */
|
|
1629
1631
|
refreshLocks = /* @__PURE__ */ new Map();
|
|
1630
1632
|
constructor(filePath) {
|
|
@@ -1783,7 +1785,7 @@ var AuthStorage = class {
|
|
|
1783
1785
|
let changed;
|
|
1784
1786
|
try {
|
|
1785
1787
|
const stat = await import_promises4.default.stat(this.filePath);
|
|
1786
|
-
changed = stat.mtimeMs !== this.snapshotMtimeMs || stat.size !== this.snapshotSize;
|
|
1788
|
+
changed = stat.ino !== this.snapshotIno || stat.mtimeMs !== this.snapshotMtimeMs || stat.size !== this.snapshotSize;
|
|
1787
1789
|
} catch {
|
|
1788
1790
|
changed = this.snapshotSize !== -1;
|
|
1789
1791
|
}
|
|
@@ -1795,9 +1797,11 @@ var AuthStorage = class {
|
|
|
1795
1797
|
const stat = await import_promises4.default.stat(this.filePath);
|
|
1796
1798
|
this.snapshotMtimeMs = stat.mtimeMs;
|
|
1797
1799
|
this.snapshotSize = stat.size;
|
|
1800
|
+
this.snapshotIno = stat.ino;
|
|
1798
1801
|
} catch {
|
|
1799
1802
|
this.snapshotMtimeMs = 0;
|
|
1800
1803
|
this.snapshotSize = -1;
|
|
1804
|
+
this.snapshotIno = 0;
|
|
1801
1805
|
}
|
|
1802
1806
|
}
|
|
1803
1807
|
/**
|
|
@@ -2113,6 +2117,29 @@ var MODELS = [
|
|
|
2113
2117
|
maxThinkingLevel: "high"
|
|
2114
2118
|
},
|
|
2115
2119
|
// ── OpenAI (Codex) ─────────────────────────────────────
|
|
2120
|
+
{
|
|
2121
|
+
// GPT-6 Astra — "Our most capable model for complex, demanding work."
|
|
2122
|
+
// (Codex catalog priority 1, listed for every ChatGPT plan, requires a
|
|
2123
|
+
// Codex client >= 0.153.0 — see CODEX_CLIENT_VERSION). Same split as 5.6:
|
|
2124
|
+
// 1.05M on the public Responses API, 272K on the ChatGPT OAuth route
|
|
2125
|
+
// (openai/codex models.json, `gpt-6-astra`). Reasoning ladder low → medium
|
|
2126
|
+
// → high → xhigh → max → ultra; `ultra` is the Codex orchestration preset
|
|
2127
|
+
// (multi_agent v2) and is Codex-only — the public API tops out at `max`.
|
|
2128
|
+
// Note: through a plain API key OpenAI requires the Responses API for tool
|
|
2129
|
+
// calling on Astra, so the Chat Completions path is text-only; the OAuth
|
|
2130
|
+
// Codex route is the supported way to use it as an agent.
|
|
2131
|
+
id: "gpt-6-astra",
|
|
2132
|
+
name: "GPT-6 Astra",
|
|
2133
|
+
provider: "openai",
|
|
2134
|
+
contextWindow: 105e4,
|
|
2135
|
+
codexContextWindow: 272e3,
|
|
2136
|
+
maxOutputTokens: 128e3,
|
|
2137
|
+
supportsThinking: true,
|
|
2138
|
+
supportsImages: true,
|
|
2139
|
+
supportsVideo: false,
|
|
2140
|
+
costTier: "high",
|
|
2141
|
+
maxThinkingLevel: "ultra"
|
|
2142
|
+
},
|
|
2116
2143
|
// GPT-5.6 family — three agentic coding tiers launched July 2026. The public
|
|
2117
2144
|
// Responses API advertises a 1.05M context window; OpenAI's Codex product
|
|
2118
2145
|
// catalog advertises 272K on the ChatGPT OAuth route (corrected from the
|
|
@@ -2166,24 +2193,11 @@ var MODELS = [
|
|
|
2166
2193
|
costTier: "low",
|
|
2167
2194
|
maxThinkingLevel: "max"
|
|
2168
2195
|
},
|
|
2169
|
-
{
|
|
2170
|
-
id: "gpt-5.5",
|
|
2171
|
-
name: "GPT-5.5",
|
|
2172
|
-
provider: "openai",
|
|
2173
|
-
contextWindow: 105e4,
|
|
2174
|
-
codexContextWindow: 272e3,
|
|
2175
|
-
maxOutputTokens: 128e3,
|
|
2176
|
-
supportsThinking: true,
|
|
2177
|
-
supportsImages: true,
|
|
2178
|
-
supportsVideo: false,
|
|
2179
|
-
costTier: "high",
|
|
2180
|
-
maxThinkingLevel: "xhigh"
|
|
2181
|
-
},
|
|
2182
2196
|
// ── Sakana (Fugu) ──────────────────────────────────────
|
|
2183
2197
|
// Sakana Fugu is a multi-agent system surfaced as a standard LLM via the
|
|
2184
2198
|
// OpenAI-compatible Sakana API (https://api.sakana.ai/v1). Both models take
|
|
2185
|
-
// text + image input
|
|
2186
|
-
//
|
|
2199
|
+
// text + image input. Plain Fugu stops at xhigh; Ultra v1.1 also supports max.
|
|
2200
|
+
// `fugu` routes across all providers; `fugu-ultra` is
|
|
2187
2201
|
// the heavier tier (may need larger client timeouts on complex tasks).
|
|
2188
2202
|
{
|
|
2189
2203
|
id: "fugu",
|
|
@@ -2207,7 +2221,8 @@ var MODELS = [
|
|
|
2207
2221
|
supportsImages: true,
|
|
2208
2222
|
supportsVideo: false,
|
|
2209
2223
|
costTier: "high",
|
|
2210
|
-
|
|
2224
|
+
// The rolling alias now serves v1.1, which adds a distinct max effort.
|
|
2225
|
+
maxThinkingLevel: "max"
|
|
2211
2226
|
},
|
|
2212
2227
|
// ── xAI (Grok) ─────────────────────────────────────────
|
|
2213
2228
|
// Grok 4.6 (released 2026-08-12) is xAI's flagship for coding, agentic tasks,
|
|
@@ -2261,6 +2276,34 @@ var MODELS = [
|
|
|
2261
2276
|
costTier: "low",
|
|
2262
2277
|
maxThinkingLevel: "high"
|
|
2263
2278
|
},
|
|
2279
|
+
// Keep 3.1 Flash Lite first for the working OAuth default and fast-model routing.
|
|
2280
|
+
// New GA models are opt-in; Code Assist access varies by account.
|
|
2281
|
+
{
|
|
2282
|
+
id: "gemini-3.8-flash",
|
|
2283
|
+
name: "Gemini 3.8 Flash",
|
|
2284
|
+
provider: "gemini",
|
|
2285
|
+
contextWindow: 1048576,
|
|
2286
|
+
maxOutputTokens: 65536,
|
|
2287
|
+
supportsThinking: true,
|
|
2288
|
+
supportsImages: true,
|
|
2289
|
+
supportsVideo: true,
|
|
2290
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2291
|
+
costTier: "low",
|
|
2292
|
+
maxThinkingLevel: "high"
|
|
2293
|
+
},
|
|
2294
|
+
{
|
|
2295
|
+
id: "gemini-3.5-flash-lite",
|
|
2296
|
+
name: "Gemini 3.5 Flash Lite",
|
|
2297
|
+
provider: "gemini",
|
|
2298
|
+
contextWindow: 1048576,
|
|
2299
|
+
maxOutputTokens: 65536,
|
|
2300
|
+
supportsThinking: true,
|
|
2301
|
+
supportsImages: true,
|
|
2302
|
+
supportsVideo: true,
|
|
2303
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2304
|
+
costTier: "low",
|
|
2305
|
+
maxThinkingLevel: "high"
|
|
2306
|
+
},
|
|
2264
2307
|
{
|
|
2265
2308
|
// Gemini 3.7 Flash (released 2026-08-13) — Google's most capable Flash for
|
|
2266
2309
|
// coding, agents, and multi-step execution; GA-stable on the Gemini API as
|
|
@@ -2268,7 +2311,7 @@ var MODELS = [
|
|
|
2268
2311
|
// Sent over our Code Assist (OAuth) transport ahead of gemini-cli — upstream
|
|
2269
2312
|
// hasn't listed 3.7 yet (google-gemini/gemini-cli#28802, still open) — so
|
|
2270
2313
|
// free/personal accounts 404 (entitlement-gated) while Code Assist
|
|
2271
|
-
// Standard/Enterprise accounts get it.
|
|
2314
|
+
// Standard/Enterprise accounts get it. Kept after the working flash-lite:
|
|
2272
2315
|
// getFastModel picks the first low-tier entry, and flash-lite is the one
|
|
2273
2316
|
// that works on every account.
|
|
2274
2317
|
id: "gemini-3.7-flash",
|
|
@@ -2460,21 +2503,19 @@ var MODELS = [
|
|
|
2460
2503
|
{
|
|
2461
2504
|
// `deepseek-v4-pro` now serves DeepSeek-V4-Pro-0813 (released 2026-08-13,
|
|
2462
2505
|
// first STABLE V4 Pro — supersedes the April preview; calling name
|
|
2463
|
-
// unchanged, same 1.6T/49B MoE). 1M context,
|
|
2464
|
-
//
|
|
2465
|
-
//
|
|
2466
|
-
// price band rather than the preview's top band.
|
|
2506
|
+
// unchanged, same 1.6T/49B MoE). 1M context, text-only, low/high/max effort.
|
|
2507
|
+
// Docs abbreviate output as 384K; use the same conservative 384,000-token
|
|
2508
|
+
// application cap across V4 models rather than mixing decimal/binary units.
|
|
2467
2509
|
id: "deepseek-v4-pro",
|
|
2468
2510
|
name: "DeepSeek V4 Pro",
|
|
2469
2511
|
provider: "deepseek",
|
|
2470
2512
|
contextWindow: 1048576,
|
|
2471
|
-
maxOutputTokens:
|
|
2513
|
+
maxOutputTokens: 384e3,
|
|
2472
2514
|
supportsThinking: true,
|
|
2473
2515
|
supportsImages: false,
|
|
2474
2516
|
supportsVideo: false,
|
|
2475
2517
|
costTier: "medium",
|
|
2476
|
-
|
|
2477
|
-
maxThinkingLevel: "xhigh"
|
|
2518
|
+
maxThinkingLevel: "max"
|
|
2478
2519
|
},
|
|
2479
2520
|
{
|
|
2480
2521
|
id: "deepseek-v4-flash",
|
|
@@ -2486,7 +2527,20 @@ var MODELS = [
|
|
|
2486
2527
|
supportsImages: false,
|
|
2487
2528
|
supportsVideo: false,
|
|
2488
2529
|
costTier: "low",
|
|
2489
|
-
maxThinkingLevel: "
|
|
2530
|
+
maxThinkingLevel: "max"
|
|
2531
|
+
},
|
|
2532
|
+
// Opt-in experimental vision sibling; never replaces the stable summary model.
|
|
2533
|
+
{
|
|
2534
|
+
id: "deepseek-v4-flash-vision-exp",
|
|
2535
|
+
name: "DeepSeek V4 Flash Vision (Experimental)",
|
|
2536
|
+
provider: "deepseek",
|
|
2537
|
+
contextWindow: 1048576,
|
|
2538
|
+
maxOutputTokens: 384e3,
|
|
2539
|
+
supportsThinking: true,
|
|
2540
|
+
supportsImages: true,
|
|
2541
|
+
supportsVideo: false,
|
|
2542
|
+
costTier: "low",
|
|
2543
|
+
maxThinkingLevel: "max"
|
|
2490
2544
|
},
|
|
2491
2545
|
// ── OpenRouter ─────────────────────────────────────────
|
|
2492
2546
|
{
|
|
@@ -2496,8 +2550,10 @@ var MODELS = [
|
|
|
2496
2550
|
contextWindow: 1e6,
|
|
2497
2551
|
maxOutputTokens: 65536,
|
|
2498
2552
|
supportsThinking: true,
|
|
2499
|
-
supportsImages:
|
|
2500
|
-
supportsVideo:
|
|
2553
|
+
supportsImages: true,
|
|
2554
|
+
supportsVideo: true,
|
|
2555
|
+
// Practical inline-payload cap, not an asserted provider maximum.
|
|
2556
|
+
maxVideoBytes: 20 * 1024 * 1024,
|
|
2501
2557
|
costTier: "medium",
|
|
2502
2558
|
maxThinkingLevel: "high"
|
|
2503
2559
|
},
|
|
@@ -2653,7 +2709,8 @@ var OPENAI_GPT_56_THINKING_LEVELS = [
|
|
|
2653
2709
|
"max",
|
|
2654
2710
|
"ultra"
|
|
2655
2711
|
];
|
|
2656
|
-
var SAKANA_THINKING_LEVELS = ["high", "xhigh"];
|
|
2712
|
+
var SAKANA_THINKING_LEVELS = ["high", "xhigh", "max"];
|
|
2713
|
+
var DEEPSEEK_THINKING_LEVELS = ["low", "high", "max"];
|
|
2657
2714
|
var XAI_THINKING_LEVELS = ["low", "medium", "high", "xhigh"];
|
|
2658
2715
|
var ANTHROPIC_XHIGH_THINKING_LEVELS = [
|
|
2659
2716
|
"low",
|
|
@@ -2717,13 +2774,14 @@ function getSupportedThinkingLevels(provider, model) {
|
|
|
2717
2774
|
return XAI_THINKING_LEVELS.slice(0, maxIndex2 + 1);
|
|
2718
2775
|
}
|
|
2719
2776
|
if (isMoonshotK3Model(provider, model)) return MOONSHOT_K3_THINKING_LEVELS;
|
|
2777
|
+
if (provider === "deepseek") return DEEPSEEK_THINKING_LEVELS;
|
|
2720
2778
|
if (isGlmModel(provider)) {
|
|
2721
2779
|
const maxIndex2 = GLM_THINKING_LEVELS.indexOf(maxLevel);
|
|
2722
2780
|
if (maxIndex2 === -1) return GLM_THINKING_LEVELS;
|
|
2723
2781
|
return GLM_THINKING_LEVELS.slice(0, maxIndex2 + 1);
|
|
2724
2782
|
}
|
|
2725
2783
|
if (!isOpenAIGptModel(provider, model)) return [maxLevel];
|
|
2726
|
-
const levels = model.startsWith("gpt-5.6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
|
|
2784
|
+
const levels = model.startsWith("gpt-5.6-") || model.startsWith("gpt-6-") ? OPENAI_GPT_56_THINKING_LEVELS : OPENAI_GPT_THINKING_LEVELS;
|
|
2727
2785
|
const maxIndex = levels.indexOf(maxLevel);
|
|
2728
2786
|
if (maxIndex === -1) return ["medium"];
|
|
2729
2787
|
return levels.slice(0, maxIndex + 1);
|
|
@@ -2733,7 +2791,7 @@ function isThinkingLevelSupported(provider, model, level) {
|
|
|
2733
2791
|
}
|
|
2734
2792
|
function getNextThinkingLevel(provider, model, current) {
|
|
2735
2793
|
const supportedLevels = getSupportedThinkingLevels(provider, model);
|
|
2736
|
-
const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || // Local servers take a real effort level, not just on/off: Ollama accepts
|
|
2794
|
+
const shouldCycleLevels = isOpenAIGptModel(provider, model) || isAnthropicAdaptiveModel(provider, model) || isSakanaModel(provider) || isXaiModel(provider) || isMoonshotK3Model(provider, model) || isGlmModel(provider) || provider === "deepseek" || // Local servers take a real effort level, not just on/off: Ollama accepts
|
|
2737
2795
|
// low/medium/high on `reasoning_effort` (verified against 0.32) and the
|
|
2738
2796
|
// other OpenAI-compatible servers use the same three. A model that can't
|
|
2739
2797
|
// reason at all already has no supported levels, so it never gets here.
|
|
@@ -2751,7 +2809,7 @@ function getNextThinkingLevel(provider, model, current) {
|
|
|
2751
2809
|
var DEFAULT_LOCAL_ENDPOINTS = [
|
|
2752
2810
|
{ id: "ollama", label: "Ollama", baseUrl: "http://127.0.0.1:11434/v1", kind: "ollama" }
|
|
2753
2811
|
];
|
|
2754
|
-
var FALLBACK_CONTEXT_WINDOW =
|
|
2812
|
+
var FALLBACK_CONTEXT_WINDOW = 4096;
|
|
2755
2813
|
var LOCAL_API_KEY_PLACEHOLDER = "local";
|
|
2756
2814
|
var DEFAULT_PROBE_TIMEOUT_MS = 1200;
|
|
2757
2815
|
var ENRICH_CONCURRENCY = 6;
|
|
@@ -2844,7 +2902,7 @@ async function enrich(endpoint, entries, options) {
|
|
|
2844
2902
|
return entries.map((entry) => genericModel(endpoint, entry));
|
|
2845
2903
|
}
|
|
2846
2904
|
function genericModel(endpoint, entry) {
|
|
2847
|
-
const declared =
|
|
2905
|
+
const declared = contextLength(entry.max_model_len);
|
|
2848
2906
|
return {
|
|
2849
2907
|
rawId: entry.id,
|
|
2850
2908
|
endpointId: endpoint.id,
|
|
@@ -2857,16 +2915,23 @@ function genericModel(endpoint, entry) {
|
|
|
2857
2915
|
}
|
|
2858
2916
|
async function enrichOllama(endpoint, entries, options) {
|
|
2859
2917
|
const showUrl = `${endpointRoot(endpoint.baseUrl)}/api/show`;
|
|
2918
|
+
const running = await fetchJson(
|
|
2919
|
+
`${endpointRoot(endpoint.baseUrl)}/api/ps`,
|
|
2920
|
+
endpoint,
|
|
2921
|
+
options
|
|
2922
|
+
);
|
|
2923
|
+
const runningModels = Array.isArray(running?.models) ? running.models : [];
|
|
2860
2924
|
const enriched = await mapLimited(entries, ENRICH_CONCURRENCY, async (entry) => {
|
|
2861
2925
|
const show = await fetchJson(showUrl, endpoint, {
|
|
2862
2926
|
...options,
|
|
2863
2927
|
method: "POST",
|
|
2864
2928
|
body: { model: entry.id }
|
|
2865
2929
|
});
|
|
2866
|
-
|
|
2867
|
-
const caps = show.capabilities ?? [];
|
|
2930
|
+
const caps = Array.isArray(show?.capabilities) ? show.capabilities : [];
|
|
2868
2931
|
if (caps.includes("embedding") && !caps.includes("completion")) return void 0;
|
|
2869
|
-
const ctx =
|
|
2932
|
+
const ctx = contextLength(
|
|
2933
|
+
runningModels.find((model) => model?.name === entry.id || model?.model === entry.id)?.context_length
|
|
2934
|
+
);
|
|
2870
2935
|
return {
|
|
2871
2936
|
rawId: entry.id,
|
|
2872
2937
|
endpointId: endpoint.id,
|
|
@@ -2874,38 +2939,61 @@ async function enrichOllama(endpoint, entries, options) {
|
|
|
2874
2939
|
contextWindowKnown: ctx !== void 0,
|
|
2875
2940
|
// Ollama reports capabilities honestly, so trust it here rather than
|
|
2876
2941
|
// using the optimistic generic default.
|
|
2877
|
-
supportsTools: caps.includes("tools"),
|
|
2942
|
+
supportsTools: show ? caps.includes("tools") : true,
|
|
2878
2943
|
supportsImages: caps.includes("vision"),
|
|
2879
2944
|
supportsThinking: caps.includes("thinking")
|
|
2880
2945
|
};
|
|
2881
2946
|
});
|
|
2882
2947
|
return enriched.filter((model) => model !== void 0);
|
|
2883
2948
|
}
|
|
2884
|
-
function
|
|
2885
|
-
|
|
2886
|
-
for (const [key, value] of Object.entries(info)) {
|
|
2887
|
-
if (key.endsWith(".context_length") && typeof value === "number" && value > 0) return value;
|
|
2888
|
-
}
|
|
2889
|
-
return void 0;
|
|
2949
|
+
function contextLength(value) {
|
|
2950
|
+
return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : void 0;
|
|
2890
2951
|
}
|
|
2891
2952
|
async function enrichLmStudio(endpoint, entries, options) {
|
|
2892
|
-
const
|
|
2893
|
-
|
|
2894
|
-
|
|
2895
|
-
|
|
2896
|
-
|
|
2897
|
-
|
|
2898
|
-
|
|
2953
|
+
const root = endpointRoot(endpoint.baseUrl);
|
|
2954
|
+
const detail = await fetchJson(`${root}/api/v1/models`, endpoint, options);
|
|
2955
|
+
const byId = /* @__PURE__ */ new Map();
|
|
2956
|
+
if (Array.isArray(detail?.models)) {
|
|
2957
|
+
for (const model of detail.models) {
|
|
2958
|
+
if (typeof model?.key !== "string") continue;
|
|
2959
|
+
const instances = Array.isArray(model.loaded_instances) ? model.loaded_instances : [];
|
|
2960
|
+
const contexts = instances.map((instance) => contextLength(instance?.config?.context_length));
|
|
2961
|
+
const ctx = contexts.length && contexts.every((value) => value !== void 0) ? contexts.reduce((min, value) => Math.min(min, value)) : void 0;
|
|
2962
|
+
const info = {
|
|
2963
|
+
id: model.key,
|
|
2964
|
+
type: model.type === "llm" && model.capabilities?.vision === true ? "vlm" : model.type,
|
|
2965
|
+
state: instances.length ? "loaded" : "not-loaded",
|
|
2966
|
+
loaded_context_length: ctx
|
|
2967
|
+
};
|
|
2968
|
+
byId.set(model.key, info);
|
|
2969
|
+
for (const instance of instances) {
|
|
2970
|
+
if (typeof instance?.id === "string" && instance.id !== model.key) {
|
|
2971
|
+
byId.set(instance.id, {
|
|
2972
|
+
...info,
|
|
2973
|
+
id: instance.id,
|
|
2974
|
+
loaded_context_length: contextLength(instance.config?.context_length)
|
|
2975
|
+
});
|
|
2976
|
+
}
|
|
2977
|
+
}
|
|
2978
|
+
}
|
|
2979
|
+
} else {
|
|
2980
|
+
const legacy = await fetchJson(`${root}/api/v0/models`, endpoint, options);
|
|
2981
|
+
if (Array.isArray(legacy?.data)) {
|
|
2982
|
+
for (const model of legacy.data) {
|
|
2983
|
+
if (typeof model?.id === "string") byId.set(model.id, model);
|
|
2984
|
+
}
|
|
2985
|
+
}
|
|
2986
|
+
}
|
|
2899
2987
|
const models = [];
|
|
2900
2988
|
for (const entry of entries) {
|
|
2901
2989
|
const info = byId.get(entry.id);
|
|
2902
2990
|
if (info && info.type !== "llm" && info.type !== "vlm") continue;
|
|
2903
|
-
const ctx = info?.
|
|
2991
|
+
const ctx = info?.state === "loaded" ? contextLength(info.loaded_context_length) : void 0;
|
|
2904
2992
|
models.push({
|
|
2905
2993
|
rawId: entry.id,
|
|
2906
2994
|
endpointId: endpoint.id,
|
|
2907
|
-
contextWindow:
|
|
2908
|
-
contextWindowKnown:
|
|
2995
|
+
contextWindow: ctx ?? FALLBACK_CONTEXT_WINDOW,
|
|
2996
|
+
contextWindowKnown: ctx !== void 0,
|
|
2909
2997
|
// LM Studio doesn't report tool support; it gates per-model at request time.
|
|
2910
2998
|
supportsTools: true,
|
|
2911
2999
|
supportsImages: info?.type === "vlm",
|
|
@@ -2921,8 +3009,8 @@ async function enrichLlamaCpp(endpoint, entries, options) {
|
|
|
2921
3009
|
endpoint,
|
|
2922
3010
|
options
|
|
2923
3011
|
);
|
|
2924
|
-
const nCtx = props?.default_generation_settings?.n_ctx;
|
|
2925
|
-
const known =
|
|
3012
|
+
const nCtx = contextLength(props?.default_generation_settings?.n_ctx);
|
|
3013
|
+
const known = nCtx !== void 0;
|
|
2926
3014
|
return entries.map((entry) => ({
|
|
2927
3015
|
...genericModel(endpoint, entry),
|
|
2928
3016
|
contextWindow: known ? nCtx : FALLBACK_CONTEXT_WINDOW,
|