failsafe-llm-model-resolver 1.0.1 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/index.d.ts +15 -1
  2. package/index.js +87 -2
  3. package/package.json +1 -1
package/index.d.ts CHANGED
@@ -54,7 +54,21 @@ export function getNextFreeModel(apiKey: string): Promise<string>;
54
54
  /** Reset the free-model round-robin index (call at the start of a batch). */
55
55
  export function resetFreeModelIndex(): void;
56
56
 
57
- /** Clear all in-memory caches (mainly for tests). */
57
+ /**
58
+ * Record that a model failed the REAL call (generation/completion), not just
59
+ * resolution. Every subsequent resolveModel() call for this provider, in
60
+ * this process, will skip this id in favour of the next-best live candidate.
61
+ * Deliberately in-memory/process-lifetime only - see the design comment
62
+ * above _knownBad in index.js before adding persistence or a TTL here.
63
+ * Not applicable to 'openrouter' (its free-tier round-robin already moves
64
+ * on from a failing model via getNextFreeModel()'s own retry loop).
65
+ */
66
+ export function markModelBad(
67
+ provider: Exclude<Provider, 'openrouter'>,
68
+ modelId: string
69
+ ): void;
70
+
71
+ /** Clear all in-memory caches, including known-bad models (mainly for tests). */
58
72
  export function clearCache(): void;
59
73
 
60
74
  export const DEFAULT_FALLBACKS: Readonly<Record<Provider, string>>;
package/index.js CHANGED
@@ -31,6 +31,49 @@ const _cache = {
31
31
  };
32
32
  let _orFreeIdx = 0; // round-robin index for OpenRouter free models
33
33
 
34
+ // ---------------------------------------------------------------------------
35
+ // "Known-bad" models — a model can be genuinely listed by a provider's own
36
+ // live /models endpoint and STILL fail the actual generation call (observed
37
+ // in production: Gemini returning 404 "no longer available to new users"
38
+ // for a model its own listing still showed - an account-tier mismatch the
39
+ // listing endpoint doesn't expose). This is a different fact from "does this
40
+ // model exist" (_cache above) and needs its own cache: a model can be valid
41
+ // AND bad at the same time.
42
+ //
43
+ // DESIGN NOTE FOR ANYONE FORKING OR EXTENDING THIS FILE, READ BEFORE ADDING
44
+ // PERSISTENCE OR A TTL HERE:
45
+ //
46
+ // This cache is deliberately in-memory only, scoped to the current process,
47
+ // with NO disk persistence and NO time-based expiry. That is not a missing
48
+ // feature - it's the fix for the exact bug class this whole package exists
49
+ // to prevent. If you persist "known-bad" to disk with no expiry, you
50
+ // recreate a hardcoded-stale-model bug with extra steps: the day the
51
+ // provider fixes the model, or the caller's account tier changes, a
52
+ // persisted bad-list keeps avoiding a model that now works, silently,
53
+ // forever, with no signal telling you it's wrong. A TTL "fixes" that at the
54
+ // cost of inventing its own new failure surface (wrong duration, clock
55
+ // skew, timezone bugs) to get right.
56
+ //
57
+ // Tying this cache to the process lifetime needs none of that. It goes away
58
+ // automatically the moment the process exits - the next invocation always
59
+ // re-validates a previously-bad model exactly once before falling through
60
+ // again if it's still broken. For a long batch job (thousands of calls, one
61
+ // process, hours), this is precisely where the win matters: discover the
62
+ // bad model once, skip it for the rest of the run. For a short one-off CLI
63
+ // call, the cost of "one wasted call on rediscovery" is trivial. Either way,
64
+ // correctness comes from where the boundary naturally sits, not from logic
65
+ // you have to write and can get wrong.
66
+ //
67
+ // If you genuinely need cross-process memory of a bad model, the correct
68
+ // place for that is the CALLING application's own operational monitoring
69
+ // (it already knows it's making repeated calls to the same provider), not
70
+ // this package pretending to be a persistent state store.
71
+ const _knownBad = {
72
+ xai: new Set(),
73
+ anthropic: new Set(),
74
+ gemini: new Set(),
75
+ };
76
+
34
77
  // ---------------------------------------------------------------------------
35
78
  // Defaults / env fallbacks
36
79
  // ---------------------------------------------------------------------------
@@ -46,7 +89,15 @@ const PREFERENCE = {
46
89
  // Prefer flagship grok-*, skip fast / mini / code-fast variants when possible
47
90
  xai: {
48
91
  include: /^grok-/i,
49
- exclude: /fast|mini|lite|code-fast/i,
92
+ // "non-reasoning" is intentionally its own term here, not "reasoning" -
93
+ // xAI ships paired variants like grok-4.20-0309-reasoning and
94
+ // grok-4.20-0309-non-reasoning. Excluding bare /reasoning/ would match
95
+ // BOTH strings (non-reasoning contains "reasoning" as a substring) and
96
+ // throw away the good one along with the bad one. non-?reasoning also
97
+ // catches an unhyphenated "nonreasoning" form if xAI ever ships one.
98
+ // Tagging/classification work benefits from the reasoning variant, so
99
+ // steer away from the faster-but-shallower non-reasoning one by default.
100
+ exclude: /fast|mini|lite|code-fast|non-?reasoning/i,
50
101
  },
51
102
  // Prefer sonnet/opus over haiku; newest first already gives us the latest
52
103
  anthropic: {
@@ -206,7 +257,11 @@ async function resolveModel(provider, apiKey, options = {}) {
206
257
  if (p === 'gemini') list = await fetchGeminiModels(apiKey);
207
258
  _cache[p] = list;
208
259
  }
209
- id = pickPreferred(list, PREFERENCE[p]);
260
+ // Skip anything this process has already learned fails the real call,
261
+ // even though it's genuinely present in the live list (see _knownBad
262
+ // above for why "listed" and "actually works" are different facts).
263
+ const usable = (list || []).filter(id => !_knownBad[p].has(id));
264
+ id = pickPreferred(usable, PREFERENCE[p]);
210
265
  }
211
266
 
212
267
  if (id) return { id, source: 'live' };
@@ -263,11 +318,40 @@ function resetFreeModelIndex() {
263
318
  /**
264
319
  * Clear all in-memory caches (mainly for tests).
265
320
  */
321
+ /**
322
+ * Record that a model failed the REAL call (not just resolution) - e.g. a
323
+ * 404/permission error from the provider's generateContent/chat-completions
324
+ * endpoint. Every subsequent resolveModel() call for this provider, in this
325
+ * process, will skip this id and pick the next-best live candidate instead.
326
+ *
327
+ * Typical caller pattern (no separate retry-loop function needed - this is
328
+ * the whole point of folding known-bad awareness into resolveModel itself):
329
+ *
330
+ * let resolved = await resolveModel('gemini', key, { fallback: PINNED });
331
+ * try {
332
+ * return await actuallyCallTheApi(resolved.id);
333
+ * } catch (e) {
334
+ * markModelBad('gemini', resolved.id);
335
+ * resolved = await resolveModel('gemini', key, { fallback: PINNED });
336
+ * return await actuallyCallTheApi(resolved.id); // now skips the bad one
337
+ * }
338
+ *
339
+ * Not exposed for 'openrouter' - its free-tier round-robin already moves on
340
+ * from a failing model via getNextFreeModel()'s own retry loop.
341
+ */
342
+ function markModelBad(provider, modelId) {
343
+ const p = String(provider || '').toLowerCase();
344
+ if (_knownBad[p] && modelId) _knownBad[p].add(modelId);
345
+ }
346
+
266
347
  function clearCache() {
267
348
  _cache.xai = null;
268
349
  _cache.anthropic = null;
269
350
  _cache.gemini = null;
270
351
  _cache.openrouter = null;
352
+ _knownBad.xai.clear();
353
+ _knownBad.anthropic.clear();
354
+ _knownBad.gemini.clear();
271
355
  _orFreeIdx = 0;
272
356
  }
273
357
 
@@ -279,6 +363,7 @@ module.exports = {
279
363
  fetchFreeModels,
280
364
  getNextFreeModel,
281
365
  resetFreeModelIndex,
366
+ markModelBad,
282
367
  clearCache,
283
368
  // exposed for advanced callers / tests
284
369
  DEFAULT_FALLBACKS,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "failsafe-llm-model-resolver",
3
- "version": "1.0.1",
3
+ "version": "1.1.0",
4
4
  "description": "Self-healing, failsafe resolver for the current frontier model across xAI, Anthropic, Gemini, and OpenRouter. Live /models fetch + preference rules + env fallbacks so a network blip never kills a batch job.",
5
5
  "main": "index.js",
6
6
  "types": "index.d.ts",