failsafe-llm-model-resolver 1.0.2 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.d.ts +15 -1
- package/index.js +78 -1
- package/package.json +1 -1
package/index.d.ts
CHANGED
|
@@ -54,7 +54,21 @@ export function getNextFreeModel(apiKey: string): Promise<string>;
|
|
|
54
54
|
/** Reset the free-model round-robin index (call at the start of a batch). */
|
|
55
55
|
export function resetFreeModelIndex(): void;
|
|
56
56
|
|
|
57
|
-
/**
|
|
57
|
+
/**
|
|
58
|
+
* Record that a model failed the REAL call (generation/completion), not just
|
|
59
|
+
* resolution. Every subsequent resolveModel() call for this provider, in
|
|
60
|
+
* this process, will skip this id in favour of the next-best live candidate.
|
|
61
|
+
* Deliberately in-memory/process-lifetime only - see the design comment
|
|
62
|
+
* above _knownBad in index.js before adding persistence or a TTL here.
|
|
63
|
+
* Not applicable to 'openrouter' (its free-tier round-robin already moves
|
|
64
|
+
* on from a failing model via getNextFreeModel()'s own retry loop).
|
|
65
|
+
*/
|
|
66
|
+
export function markModelBad(
|
|
67
|
+
provider: Exclude<Provider, 'openrouter'>,
|
|
68
|
+
modelId: string
|
|
69
|
+
): void;
|
|
70
|
+
|
|
71
|
+
/** Clear all in-memory caches, including known-bad models (mainly for tests). */
|
|
58
72
|
export function clearCache(): void;
|
|
59
73
|
|
|
60
74
|
export const DEFAULT_FALLBACKS: Readonly<Record<Provider, string>>;
|
package/index.js
CHANGED
|
@@ -31,6 +31,49 @@ const _cache = {
|
|
|
31
31
|
};
|
|
32
32
|
let _orFreeIdx = 0; // round-robin index for OpenRouter free models
|
|
33
33
|
|
|
34
|
+
// ---------------------------------------------------------------------------
|
|
35
|
+
// "Known-bad" models — a model can be genuinely listed by a provider's own
|
|
36
|
+
// live /models endpoint and STILL fail the actual generation call (observed
|
|
37
|
+
// in production: Gemini returning 404 "no longer available to new users"
|
|
38
|
+
// for a model its own listing still showed - an account-tier mismatch the
|
|
39
|
+
// listing endpoint doesn't expose). This is a different fact from "does this
|
|
40
|
+
// model exist" (_cache above) and needs its own cache: a model can be valid
|
|
41
|
+
// AND bad at the same time.
|
|
42
|
+
//
|
|
43
|
+
// DESIGN NOTE FOR ANYONE FORKING OR EXTENDING THIS FILE, READ BEFORE ADDING
|
|
44
|
+
// PERSISTENCE OR A TTL HERE:
|
|
45
|
+
//
|
|
46
|
+
// This cache is deliberately in-memory only, scoped to the current process,
|
|
47
|
+
// with NO disk persistence and NO time-based expiry. That is not a missing
|
|
48
|
+
// feature - it's the fix for the exact bug class this whole package exists
|
|
49
|
+
// to prevent. If you persist "known-bad" to disk with no expiry, you
|
|
50
|
+
// recreate a hardcoded-stale-model bug with extra steps: the day the
|
|
51
|
+
// provider fixes the model, or the caller's account tier changes, a
|
|
52
|
+
// persisted bad-list keeps avoiding a model that now works, silently,
|
|
53
|
+
// forever, with no signal telling you it's wrong. A TTL "fixes" that at the
|
|
54
|
+
// cost of inventing its own new failure surface (wrong duration, clock
|
|
55
|
+
// skew, timezone bugs) to get right.
|
|
56
|
+
//
|
|
57
|
+
// Tying this cache to the process lifetime needs none of that. It goes away
|
|
58
|
+
// automatically the moment the process exits - the next invocation always
|
|
59
|
+
// re-validates a previously-bad model exactly once before falling through
|
|
60
|
+
// again if it's still broken. For a long batch job (thousands of calls, one
|
|
61
|
+
// process, hours), this is precisely where the win matters: discover the
|
|
62
|
+
// bad model once, skip it for the rest of the run. For a short one-off CLI
|
|
63
|
+
// call, the cost of "one wasted call on rediscovery" is trivial. Either way,
|
|
64
|
+
// correctness comes from where the boundary naturally sits, not from logic
|
|
65
|
+
// you have to write and can get wrong.
|
|
66
|
+
//
|
|
67
|
+
// If you genuinely need cross-process memory of a bad model, the correct
|
|
68
|
+
// place for that is the CALLING application's own operational monitoring
|
|
69
|
+
// (it already knows it's making repeated calls to the same provider), not
|
|
70
|
+
// this package pretending to be a persistent state store.
|
|
71
|
+
const _knownBad = {
|
|
72
|
+
xai: new Set(),
|
|
73
|
+
anthropic: new Set(),
|
|
74
|
+
gemini: new Set(),
|
|
75
|
+
};
|
|
76
|
+
|
|
34
77
|
// ---------------------------------------------------------------------------
|
|
35
78
|
// Defaults / env fallbacks
|
|
36
79
|
// ---------------------------------------------------------------------------
|
|
@@ -214,7 +257,11 @@ async function resolveModel(provider, apiKey, options = {}) {
|
|
|
214
257
|
if (p === 'gemini') list = await fetchGeminiModels(apiKey);
|
|
215
258
|
_cache[p] = list;
|
|
216
259
|
}
|
|
217
|
-
|
|
260
|
+
// Skip anything this process has already learned fails the real call,
|
|
261
|
+
// even though it's genuinely present in the live list (see _knownBad
|
|
262
|
+
// above for why "listed" and "actually works" are different facts).
|
|
263
|
+
const usable = (list || []).filter(id => !_knownBad[p].has(id));
|
|
264
|
+
id = pickPreferred(usable, PREFERENCE[p]);
|
|
218
265
|
}
|
|
219
266
|
|
|
220
267
|
if (id) return { id, source: 'live' };
|
|
@@ -271,11 +318,40 @@ function resetFreeModelIndex() {
|
|
|
271
318
|
/**
|
|
272
319
|
* Clear all in-memory caches (mainly for tests).
|
|
273
320
|
*/
|
|
321
|
+
/**
|
|
322
|
+
* Record that a model failed the REAL call (not just resolution) - e.g. a
|
|
323
|
+
* 404/permission error from the provider's generateContent/chat-completions
|
|
324
|
+
* endpoint. Every subsequent resolveModel() call for this provider, in this
|
|
325
|
+
* process, will skip this id and pick the next-best live candidate instead.
|
|
326
|
+
*
|
|
327
|
+
* Typical caller pattern (no separate retry-loop function needed - this is
|
|
328
|
+
* the whole point of folding known-bad awareness into resolveModel itself):
|
|
329
|
+
*
|
|
330
|
+
* let resolved = await resolveModel('gemini', key, { fallback: PINNED });
|
|
331
|
+
* try {
|
|
332
|
+
* return await actuallyCallTheApi(resolved.id);
|
|
333
|
+
* } catch (e) {
|
|
334
|
+
* markModelBad('gemini', resolved.id);
|
|
335
|
+
* resolved = await resolveModel('gemini', key, { fallback: PINNED });
|
|
336
|
+
* return await actuallyCallTheApi(resolved.id); // now skips the bad one
|
|
337
|
+
* }
|
|
338
|
+
*
|
|
339
|
+
* Not exposed for 'openrouter' - its free-tier round-robin already moves on
|
|
340
|
+
* from a failing model via getNextFreeModel()'s own retry loop.
|
|
341
|
+
*/
|
|
342
|
+
function markModelBad(provider, modelId) {
|
|
343
|
+
const p = String(provider || '').toLowerCase();
|
|
344
|
+
if (_knownBad[p] && modelId) _knownBad[p].add(modelId);
|
|
345
|
+
}
|
|
346
|
+
|
|
274
347
|
function clearCache() {
|
|
275
348
|
_cache.xai = null;
|
|
276
349
|
_cache.anthropic = null;
|
|
277
350
|
_cache.gemini = null;
|
|
278
351
|
_cache.openrouter = null;
|
|
352
|
+
_knownBad.xai.clear();
|
|
353
|
+
_knownBad.anthropic.clear();
|
|
354
|
+
_knownBad.gemini.clear();
|
|
279
355
|
_orFreeIdx = 0;
|
|
280
356
|
}
|
|
281
357
|
|
|
@@ -287,6 +363,7 @@ module.exports = {
|
|
|
287
363
|
fetchFreeModels,
|
|
288
364
|
getNextFreeModel,
|
|
289
365
|
resetFreeModelIndex,
|
|
366
|
+
markModelBad,
|
|
290
367
|
clearCache,
|
|
291
368
|
// exposed for advanced callers / tests
|
|
292
369
|
DEFAULT_FALLBACKS,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "failsafe-llm-model-resolver",
|
|
3
|
-
"version": "1.0
|
|
3
|
+
"version": "1.1.0",
|
|
4
4
|
"description": "Self-healing, failsafe resolver for the current frontier model across xAI, Anthropic, Gemini, and OpenRouter. Live /models fetch + preference rules + env fallbacks so a network blip never kills a batch job.",
|
|
5
5
|
"main": "index.js",
|
|
6
6
|
"types": "index.d.ts",
|