@juspay/neurolink 12.47.4 → 12.47.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -125,7 +125,7 @@ export async function createCopilotCliReader() {
125
125
  errors.push({
126
126
  cliId: CLI_ID,
127
127
  filePath: dbPath,
128
- message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
128
+ message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
129
129
  });
130
130
  return { cliId: CLI_ID, totals, filesScanned: 0, errors };
131
131
  }
@@ -375,7 +375,7 @@ export async function createCursorReader() {
375
375
  errors.push({
376
376
  cliId: CLI_ID,
377
377
  filePath: chatsRoot(),
378
- message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
378
+ message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
379
379
  });
380
380
  return { cliId: CLI_ID, totals, filesScanned: 0, errors };
381
381
  }
@@ -318,7 +318,7 @@ export async function createHermesReader() {
318
318
  errors.push({
319
319
  cliId: CLI_ID,
320
320
  filePath: hermesHome(),
321
- message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
321
+ message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
322
322
  });
323
323
  return { cliId: CLI_ID, totals, filesScanned: 0, errors };
324
324
  }
@@ -80,10 +80,11 @@ export async function createOpenCodeReader() {
80
80
  const errors = [];
81
81
  const models = new Set();
82
82
  const dbPath = databasePath();
83
- // `node:sqlite` is built in from Node 22 but still flagged experimental,
84
- // so it can be absent or change shape. Imported lazily and behind a
85
- // try/catch: a runtime without it must degrade to a reported failure for
86
- // this one reader, not take down a scan of all the others.
83
+ // `node:sqlite` arrived in Node 22.5.0 behind `--experimental-sqlite`, was
84
+ // unflagged in 22.13.0 and is still marked experimental, so it can be
85
+ // absent or change shape. Imported lazily and behind a try/catch: a
86
+ // runtime without it must degrade to a reported failure for this one
87
+ // reader, not take down a scan of all the others.
87
88
  let DatabaseSync;
88
89
  try {
89
90
  const sqlite = await import("node:sqlite");
@@ -102,7 +103,7 @@ export async function createOpenCodeReader() {
102
103
  errors.push({
103
104
  cliId: CLI_ID,
104
105
  filePath: dbPath,
105
- message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
106
+ message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
106
107
  });
107
108
  return { cliId: CLI_ID, totals, filesScanned: 0, errors };
108
109
  }
@@ -129,7 +129,7 @@ export declare class NeuroLink {
129
129
  * request dedup — BZ-664's actual goal — is untouched: the first
130
130
  * occurrence in a request may still be served from cache. Request-scoped
131
131
  * like _disableToolCacheForCurrentRequest above (assigned a fresh Set at
132
- * request start so the router's save/restore-by-reference pattern works).
132
+ * request start so `preservingTurnState`'s save/restore-by-reference works).
133
133
  */
134
134
  private _toolCacheKeysServedThisRequest;
135
135
  /** True only while a generate()/stream() turn is executing — the
@@ -1094,6 +1094,18 @@ export declare class NeuroLink {
1094
1094
  */
1095
1095
  private streamWithIterationFallback;
1096
1096
  private executeStreamRequest;
1097
+ /**
1098
+ * Runs an internal call that re-enters the public generate() (the tool-routing
1099
+ * router, the classifier router) without ending the outer turn's tool-cache
1100
+ * state. generate()'s own `finally` resets these fields, so the outer turn
1101
+ * would otherwise lose its repeat-call cache bypass for every later tool
1102
+ * call. Restored by reference: the nested call assigns a new Set rather than
1103
+ * mutating the outer one.
1104
+ *
1105
+ * Covers one turn re-entering itself. Two concurrent turns on one instance
1106
+ * still share these fields.
1107
+ */
1108
+ private preservingTurnState;
1097
1109
  /**
1098
1110
  * Pre-call tool routing for both stream() and generate() turns: runs the
1099
1111
  * router LLM once per turn and appends the unpicked servers' registered tool
package/dist/neurolink.js CHANGED
@@ -433,7 +433,7 @@ export class NeuroLink {
433
433
  * request dedup — BZ-664's actual goal — is untouched: the first
434
434
  * occurrence in a request may still be served from cache. Request-scoped
435
435
  * like _disableToolCacheForCurrentRequest above (assigned a fresh Set at
436
- * request start so the router's save/restore-by-reference pattern works).
436
+ * request start so `preservingTurnState`'s save/restore-by-reference works).
437
437
  */
438
438
  _toolCacheKeysServedThisRequest = new Set();
439
439
  /** True only while a generate()/stream() turn is executing — the
@@ -994,9 +994,9 @@ export class NeuroLink {
994
994
  // generate() (marked so it never recursively re-routes). Fails open.
995
995
  this.classifierRouter = config?.classifierRouter?.enabled
996
996
  ? new ClassifierRouter(config.classifierRouter, {
997
- generate: (genOptions) => this.generate({
997
+ generate: (genOptions) => this.preservingTurnState(() => this.generate({
998
998
  ...genOptions,
999
- }),
999
+ })),
1000
1000
  // Fail-open by construction: tryDecide returns null rather than
1001
1001
  // throwing, so an absent or broken decision provider leaves routing
1002
1002
  // exactly as it was.
@@ -7265,6 +7265,30 @@ Current user's request: ${currentInput}`;
7265
7265
  throw error;
7266
7266
  }
7267
7267
  }
7268
+ /**
7269
+ * Runs an internal call that re-enters the public generate() (the tool-routing
7270
+ * router, the classifier router) without ending the outer turn's tool-cache
7271
+ * state. generate()'s own `finally` resets these fields, so the outer turn
7272
+ * would otherwise lose its repeat-call cache bypass for every later tool
7273
+ * call. Restored by reference: the nested call assigns a new Set rather than
7274
+ * mutating the outer one.
7275
+ *
7276
+ * Covers one turn re-entering itself. Two concurrent turns on one instance
7277
+ * still share these fields.
7278
+ */
7279
+ async preservingTurnState(run) {
7280
+ const disableToolCache = this._disableToolCacheForCurrentRequest;
7281
+ const keysServed = this._toolCacheKeysServedThisRequest;
7282
+ const turnActive = this._generationTurnActive;
7283
+ try {
7284
+ return await run();
7285
+ }
7286
+ finally {
7287
+ this._disableToolCacheForCurrentRequest = disableToolCache;
7288
+ this._toolCacheKeysServedThisRequest = keysServed;
7289
+ this._generationTurnActive = turnActive;
7290
+ }
7291
+ }
7268
7292
  /**
7269
7293
  * Pre-call tool routing for both stream() and generate() turns: runs the
7270
7294
  * router LLM once per turn and appends the unpicked servers' registered tool
@@ -7420,104 +7444,92 @@ Current user's request: ${currentInput}`;
7420
7444
  return;
7421
7445
  }
7422
7446
  }
7423
- // The router call below re-enters the public generate(), whose finally
7424
- // block resets _disableToolCacheForCurrentRequest to false. That flag is
7425
- // turn-scoped (set at the top of this turn) and read by the main tool
7426
- // execution path that runs after routing, so save it before the router
7427
- // call and restore it afterward to keep the turn's cache setting intact.
7428
- const cacheDisabledForCurrentRequest = this._disableToolCacheForCurrentRequest;
7429
7447
  let routedExcludeTools;
7430
7448
  let resolvedDecision;
7431
- try {
7432
- // Intercept the decision so we can store it in the cache.
7433
- const captureDecision = (decision) => {
7434
- resolvedDecision = decision;
7435
- emitDecision(decision);
7436
- };
7437
- // --- ITEM B: build the embedFn for the L2 embedding fast-path ---
7438
- // The vector cache is persisted at the NeuroLink instance level so
7439
- // tool embedding vectors are computed once and reused across turns
7440
- // (Finding 1 fix). It is cleared by setToolRoutingServers() when the
7441
- // catalog changes so stale vectors are never used after an update.
7442
- let routingEmbedFn;
7443
- const embeddingCfg = routingConfig.embedding;
7444
- if (embeddingCfg?.enabled === true) {
7445
- try {
7446
- // Resolve the embedding provider: use the explicitly configured one
7447
- // if present, otherwise fall back to the stream/generate call's
7448
- // provider. The factory call is wrapped in try/catch so a provider
7449
- // that doesn't support embedMany (it throws at call time, not
7450
- // construction time) fails open when routingEmbedFn is invoked.
7451
- const embProviderName = embeddingCfg.provider ??
7452
- (options.provider && options.provider !== "auto"
7453
- ? options.provider
7454
- : undefined) ??
7455
- routingConfig.routerModel?.provider;
7456
- if (embProviderName) {
7457
- const embProvider = await AIProviderFactory.createProvider(embProviderName, embeddingCfg.model, true, this, undefined, this.resolveCredentials(options.credentials));
7458
- // Bind embedMany with the configured model (may be undefined —
7459
- // the provider uses its default embedding model in that case).
7460
- routingEmbedFn = (texts) => withTimeout(embProvider.embedMany(texts, embeddingCfg.model), embeddingCfg.timeoutMs ?? 10000);
7461
- // Lazy-init the persistent vector cache for this instance.
7462
- // Subsequent turns reuse the same Map so text→vector lookups
7463
- // already populated from earlier turns are served from memory.
7464
- if (!this.toolRoutingVectorCache) {
7465
- this.toolRoutingVectorCache = new Map();
7466
- }
7449
+ // Intercept the decision so we can store it in the cache.
7450
+ const captureDecision = (decision) => {
7451
+ resolvedDecision = decision;
7452
+ emitDecision(decision);
7453
+ };
7454
+ // --- ITEM B: build the embedFn for the L2 embedding fast-path ---
7455
+ // The vector cache is persisted at the NeuroLink instance level so
7456
+ // tool embedding vectors are computed once and reused across turns
7457
+ // (Finding 1 fix). It is cleared by setToolRoutingServers() when the
7458
+ // catalog changes so stale vectors are never used after an update.
7459
+ let routingEmbedFn;
7460
+ const embeddingCfg = routingConfig.embedding;
7461
+ if (embeddingCfg?.enabled === true) {
7462
+ try {
7463
+ // Resolve the embedding provider: use the explicitly configured one
7464
+ // if present, otherwise fall back to the stream/generate call's
7465
+ // provider. The factory call is wrapped in try/catch so a provider
7466
+ // that doesn't support embedMany (it throws at call time, not
7467
+ // construction time) fails open when routingEmbedFn is invoked.
7468
+ const embProviderName = embeddingCfg.provider ??
7469
+ (options.provider && options.provider !== "auto"
7470
+ ? options.provider
7471
+ : undefined) ??
7472
+ routingConfig.routerModel?.provider;
7473
+ if (embProviderName) {
7474
+ const embProvider = await AIProviderFactory.createProvider(embProviderName, embeddingCfg.model, true, this, undefined, this.resolveCredentials(options.credentials));
7475
+ // Bind embedMany with the configured model (may be undefined —
7476
+ // the provider uses its default embedding model in that case).
7477
+ routingEmbedFn = (texts) => withTimeout(embProvider.embedMany(texts, embeddingCfg.model), embeddingCfg.timeoutMs ?? 10000);
7478
+ // Lazy-init the persistent vector cache for this instance.
7479
+ // Subsequent turns reuse the same Map so text→vector lookups
7480
+ // already populated from earlier turns are served from memory.
7481
+ if (!this.toolRoutingVectorCache) {
7482
+ this.toolRoutingVectorCache = new Map();
7467
7483
  }
7468
7484
  }
7469
- catch (embSetupError) {
7470
- logger.debug("[ToolRouting] Embedding provider setup failed, L2 path disabled for this turn", {
7471
- error: embSetupError instanceof Error
7472
- ? embSetupError.message
7473
- : String(embSetupError),
7474
- });
7475
- // routingEmbedFn remains undefined — fast-path is skipped.
7476
- }
7477
7485
  }
7478
- routedExcludeTools = await resolveToolRoutingExclusions({
7479
- catalog,
7480
- alwaysIncludeServerIds: routingConfig.alwaysIncludeServerIds ?? [],
7481
- userQuery: routingQuery,
7482
- routerPromptPrefix: routingConfig.routerPromptPrefix,
7483
- routerModel: {
7484
- provider: routingConfig.routerModel?.provider ??
7485
- options.provider,
7486
- model: routingConfig.routerModel?.model ?? options.model,
7487
- region: routingConfig.routerModel?.region ?? options.region,
7488
- temperature: routingConfig.routerModel?.temperature,
7489
- },
7490
- timeoutMs: routingConfig.timeoutMs ?? DEFAULT_TOOL_ROUTING_TIMEOUT_MS,
7491
- // Forward the abort signal so a cancelled turn aborts the router
7492
- // call promptly instead of waiting out the routing timeout.
7493
- generateFn: (generateOptions) => this.generate({
7494
- ...generateOptions,
7495
- abortSignal: options.abortSignal,
7496
- }),
7497
- // Calibrated per-server routing when a decision provider is
7498
- // configured. tryDecide returns null without one, so the resolver
7499
- // falls straight through to the generative router as before.
7500
- decideFn: (decisionOptions) => this.tryDecide({
7501
- ...decisionOptions,
7502
- signal: options.abortSignal,
7503
- }),
7504
- decisionMinDropConfidence: routingConfig.minDropConfidence,
7505
- emitDecision: captureDecision,
7506
- // L2 / ITEM D — only populated when embedding is configured.
7507
- embedFn: routingEmbedFn,
7508
- embeddingConfig: embeddingCfg,
7509
- granularity: routingConfig.granularity ?? "server",
7510
- // Pass the persistent vector cache so tool embeddings are reused
7511
- // across turns (Finding 1).
7512
- embeddingVectorCache: routingEmbedFn !== undefined
7513
- ? this.toolRoutingVectorCache
7514
- : undefined,
7515
- });
7516
- }
7517
- finally {
7518
- this._disableToolCacheForCurrentRequest =
7519
- cacheDisabledForCurrentRequest;
7520
- }
7486
+ catch (embSetupError) {
7487
+ logger.debug("[ToolRouting] Embedding provider setup failed, L2 path disabled for this turn", {
7488
+ error: embSetupError instanceof Error
7489
+ ? embSetupError.message
7490
+ : String(embSetupError),
7491
+ });
7492
+ // routingEmbedFn remains undefined — fast-path is skipped.
7493
+ }
7494
+ }
7495
+ routedExcludeTools = await resolveToolRoutingExclusions({
7496
+ catalog,
7497
+ alwaysIncludeServerIds: routingConfig.alwaysIncludeServerIds ?? [],
7498
+ userQuery: routingQuery,
7499
+ routerPromptPrefix: routingConfig.routerPromptPrefix,
7500
+ routerModel: {
7501
+ provider: routingConfig.routerModel?.provider ??
7502
+ options.provider,
7503
+ model: routingConfig.routerModel?.model ?? options.model,
7504
+ region: routingConfig.routerModel?.region ?? options.region,
7505
+ temperature: routingConfig.routerModel?.temperature,
7506
+ },
7507
+ timeoutMs: routingConfig.timeoutMs ?? DEFAULT_TOOL_ROUTING_TIMEOUT_MS,
7508
+ // Forward the abort signal so a cancelled turn aborts the router
7509
+ // call promptly instead of waiting out the routing timeout.
7510
+ generateFn: (generateOptions) => this.preservingTurnState(() => this.generate({
7511
+ ...generateOptions,
7512
+ abortSignal: options.abortSignal,
7513
+ })),
7514
+ // Calibrated per-server routing when a decision provider is
7515
+ // configured. tryDecide returns null without one, so the resolver
7516
+ // falls straight through to the generative router as before.
7517
+ decideFn: (decisionOptions) => this.tryDecide({
7518
+ ...decisionOptions,
7519
+ signal: options.abortSignal,
7520
+ }),
7521
+ decisionMinDropConfidence: routingConfig.minDropConfidence,
7522
+ emitDecision: captureDecision,
7523
+ // L2 / ITEM D — only populated when embedding is configured.
7524
+ embedFn: routingEmbedFn,
7525
+ embeddingConfig: embeddingCfg,
7526
+ granularity: routingConfig.granularity ?? "server",
7527
+ // Pass the persistent vector cache so tool embeddings are reused
7528
+ // across turns (Finding 1).
7529
+ embeddingVectorCache: routingEmbedFn !== undefined
7530
+ ? this.toolRoutingVectorCache
7531
+ : undefined,
7532
+ });
7521
7533
  // Aborted during the router call — skip applying now-stale exclusions;
7522
7534
  // the main generation path enforces the abort itself.
7523
7535
  if (options.abortSignal?.aborted) {
@@ -7986,6 +7998,20 @@ Current user's request: ${currentInput}`;
7986
7998
  yield* incrementalFallback;
7987
7999
  }
7988
8000
  ttsResolver?.(streamedTTSResult);
8001
+ // `streamState.finishReason` starts as a "stop" placeholder, before a
8002
+ // single chunk exists. A provider that records how the turn really
8003
+ // ended in metadata (a token-limit cut, a content filter, a step cap
8004
+ // that left tool calls pending) has that adopted here, once the drain
8005
+ // is done. Only the three unified values count: a raw vendor string
8006
+ // or "stop" leaves the graded reason alone, and a fallback's own
8007
+ // value is never overwritten.
8008
+ const drainedFinishReason = mcpStreamOutcome.metadata?.finishReason;
8009
+ if (!metadata.fallbackAttempted &&
8010
+ (drainedFinishReason === "length" ||
8011
+ drainedFinishReason === "tool-calls" ||
8012
+ drainedFinishReason === "content-filter")) {
8013
+ streamState.finishReason = drainedFinishReason;
8014
+ }
7989
8015
  resolvedUsage = mcpStreamOutcome.usage;
7990
8016
  if (!resolvedUsage && mcpStreamOutcome.analytics) {
7991
8017
  try {
@@ -8176,8 +8202,19 @@ Current user's request: ${currentInput}`;
8176
8202
  }
8177
8203
  })();
8178
8204
  const streamResult = await this.processStreamResult(processedStream, enhancedOptions, factoryResult);
8179
- streamResult.finishReason =
8205
+ streamState.finishReason =
8180
8206
  streamState.finishReason || streamResult.finishReason;
8207
+ // Live, like toolCalls/toolResults below: the reason a stream ends with is
8208
+ // only known after it drains, so a plain copy would stay the creation-time
8209
+ // placeholder for every truncated or capped turn.
8210
+ Object.defineProperty(streamResult, "finishReason", {
8211
+ enumerable: true,
8212
+ configurable: true,
8213
+ get: () => streamState.finishReason,
8214
+ set: (value) => {
8215
+ streamState.finishReason = value;
8216
+ },
8217
+ });
8181
8218
  // #1819 / E1b: a top-level cross-provider fallback (handleStreamFallback,
8182
8219
  // inside `processedStream` above) only reassigns
8183
8220
  // `streamState.toolCalls`/`toolResults` once the caller starts draining
@@ -9207,13 +9244,18 @@ Current user's request: ${currentInput}`;
9207
9244
  if (toolResultsDescriptor) {
9208
9245
  Object.defineProperty(response, "toolResults", toolResultsDescriptor);
9209
9246
  }
9247
+ const finishReasonDescriptor = Object.getOwnPropertyDescriptor(streamResult, "finishReason");
9248
+ if (finishReasonDescriptor) {
9249
+ Object.defineProperty(response, "finishReason", finishReasonDescriptor);
9250
+ }
9210
9251
  if (!source) {
9211
9252
  return response;
9212
9253
  }
9213
9254
  // NeuroLink grades the turn's terminal state itself (a normalized
9214
9255
  // finishReason, stopReason for aborts and time limits), so those fields
9215
- // stay plain values here. Copying the provider's getter for them would
9216
- // report the raw vendor reason, e.g. Anthropic's end_turn, instead.
9256
+ // keep NeuroLink's own value here (finishReason through the descriptor
9257
+ // copied above). Copying the provider's getter for them would report the
9258
+ // raw vendor reason, e.g. Anthropic's end_turn, instead.
9217
9259
  //
9218
9260
  // toolCalls/toolResults are excluded for a different reason: `source`
9219
9261
  // (the PRIMARY provider's own result) may itself define them as live
@@ -99,6 +99,28 @@ const resolveAgainstSchema = (text, schema) => {
99
99
  const scalar = recoverScalarRoot(text, schema);
100
100
  return scalar.kind === "accepted" ? scalar.value : undefined;
101
101
  };
102
+ /**
103
+ * The one place a finish reason becomes the unified spelling. Accepts the wire's
104
+ * `finish_reason` (`tool_calls`, `content_filter`) and the unified spelling a
105
+ * middleware's own finish part already carries (`tool-calls`, `content-filter`),
106
+ * since both reach the stream path. Anything else, including `stop`, `error` and
107
+ * an absent value, reads as `stop`.
108
+ */
109
+ const toUnifiedFinishReason = (reason) => {
110
+ switch (reason) {
111
+ case "length":
112
+ return "length";
113
+ case "tool_calls":
114
+ case "function_call":
115
+ case "tool-calls":
116
+ return "tool-calls";
117
+ case "content_filter":
118
+ case "content-filter":
119
+ return "content-filter";
120
+ default:
121
+ return "stop";
122
+ }
123
+ };
102
124
  // Pull one native chunk at a time and forward cancellation to its iterator.
103
125
  const chunksToV3Stream = (source, completion, cancel) => {
104
126
  const iterator = source[Symbol.asyncIterator]();
@@ -774,13 +796,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
774
796
  });
775
797
  }
776
798
  const rawFinish = choice?.finish_reason;
777
- const unified = rawFinish === "length"
778
- ? "length"
779
- : rawFinish === "tool_calls" || rawFinish === "function_call"
780
- ? "tool-calls"
781
- : rawFinish === "content_filter"
782
- ? "content-filter"
783
- : "stop";
799
+ const unified = toUnifiedFinishReason(rawFinish);
784
800
  return {
785
801
  content,
786
802
  finishReason: { unified, raw: rawFinish ?? "stop" },
@@ -1839,7 +1855,8 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
1839
1855
  if (!loopPromise) {
1840
1856
  resolveFinish("stop");
1841
1857
  }
1842
- streamMetadata.rawFinishReason = await finishPromise;
1858
+ const rawFinishReason = await finishPromise;
1859
+ streamMetadata.rawFinishReason = rawFinishReason;
1843
1860
  // Structured output for a `stream({ schema })` turn. Runs HERE —
1844
1861
  // after the stream is fully drained — so a tool-free re-ask (when the
1845
1862
  // streamed answer isn't already schema-valid) never reaches
@@ -1884,9 +1901,11 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
1884
1901
  // Only reached once the structured-output re-ask above (when there
1885
1902
  // was one) has resolved without throwing. A caller abort during that
1886
1903
  // re-ask rejects `resolveStreamStructuredData` and skips this line
1887
- // entirely, so `metadata.finishReason` never claims "stop" for a
1904
+ // entirely, so `metadata.finishReason` never claims a finish for a
1888
1905
  // turn that actually failed after the wire stream itself finished.
1889
- streamMetadata.finishReason = "stop";
1906
+ // The reason is the turn's own (a token-limit cut reads "length"), not
1907
+ // a blanket "stop" for every stream that drained without throwing.
1908
+ streamMetadata.finishReason = toUnifiedFinishReason(rawFinishReason);
1890
1909
  // No-output path: stream completed normally but yielded zero text.
1891
1910
  // Build an enriched sentinel + stamp the active OTel span so
1892
1911
  // Pipeline B (ContextEnricher) surfaces a WARNING-level Langfuse
@@ -241,6 +241,10 @@ function kebabToCamelRoutingKey(kebabKey) {
241
241
  * *because* one of the spellings was explicitly `null` — never merely
242
242
  * because both were absent — so callers can warn once per null key without
243
243
  * warning on ordinary omission.
244
+ *
245
+ * `account-allowlist` is the one key whose `null` `validateProxyConfig`
246
+ * rejects before this runs, so for it the `null` branch is only reachable
247
+ * through a direct `parseRoutingConfig` call.
244
248
  */
245
249
  function readLegacyRoutingKey(routing, kebabKey) {
246
250
  const camelKey = kebabToCamelRoutingKey(kebabKey);
@@ -313,6 +317,13 @@ export function validateProxyConfig(config) {
313
317
  }
314
318
  });
315
319
  }
320
+ // A `null` allowlist must not read as "unset": a reload that loses a
321
+ // restriction this way would silently open the proxy to every account.
322
+ // Checked per spelling, before the legacy read coalesces them.
323
+ if (routing["account-allowlist"] === null ||
324
+ routing[kebabToCamelRoutingKey("account-allowlist")] === null) {
325
+ errors.push("routing.account-allowlist must be an array of non-empty strings (null is not allowed; omit the key to remove the restriction)");
326
+ }
316
327
  const rawAccountAllowlist = readLegacyRoutingKey(routing, "account-allowlist").value;
317
328
  if (rawAccountAllowlist !== undefined) {
318
329
  if (!Array.isArray(rawAccountAllowlist)) {
@@ -1,7 +1,10 @@
1
1
  /**
2
- * Shared provider-error classification. Every provider's
3
- * `formatProviderError(error)` delegates here instead of hand-rolling its
2
+ * Shared provider-error classification. Migrated providers'
3
+ * `formatProviderError(error)` delegate here instead of hand-rolling their
4
4
  * own TimeoutError-check → .includes()-chain → `new XError(...)` ladder.
5
+ * Not yet migrated (they do not call it): Google AI Studio, SageMaker, the
6
+ * media and embedding providers (Ideogram, Recraft, Stability, Replicate,
7
+ * Jina, Voyage) and the System One decision provider.
5
8
  *
6
9
  * `classifyProviderError` picks the Error subclass + message; it does NOT
7
10
  * stamp statusCode/isRetryable/retryAfterMs onto the result — that
@@ -1,7 +1,10 @@
1
1
  /**
2
- * Shared provider-error classification. Every provider's
3
- * `formatProviderError(error)` delegates here instead of hand-rolling its
2
+ * Shared provider-error classification. Migrated providers'
3
+ * `formatProviderError(error)` delegate here instead of hand-rolling their
4
4
  * own TimeoutError-check → .includes()-chain → `new XError(...)` ladder.
5
+ * Not yet migrated (they do not call it): Google AI Studio, SageMaker, the
6
+ * media and embedding providers (Ideogram, Recraft, Stability, Replicate,
7
+ * Jina, Voyage) and the System One decision provider.
5
8
  *
6
9
  * `classifyProviderError` picks the Error subclass + message; it does NOT
7
10
  * stamp statusCode/isRetryable/retryAfterMs onto the result — that
@@ -116,7 +119,14 @@ export function classifyProviderError(error, rules, provider, modelName) {
116
119
  // A 404 alone is a route answer (a wrong base URL gives the same reply): it only
117
120
  // means "missing model" when the text names a model or deployment as absent.
118
121
  // The gap is bounded rather than "no dot" because real model ids contain dots.
119
- const MODEL_404_TEXT = /model[_ ]?not[_ ]?found|unknown model|no such model|invalid model|unsupported model|\b(?:model|deployment)\b.{0,120}\b(?:does not exist|not found|unavailable|not (?:available|supported))\b|\b(?:does not exist|not found)\b.{0,120}\b(?:model|deployment)\b|unable to access.{0,60}\bmodel\b/i;
122
+ // "model" or "deployment" followed by a route noun ("model gateway route") is a
123
+ // modifier of that route, not the subject of the 404, in either word order, and
124
+ // that holds for the named phrases ("unknown model gateway route") as well.
125
+ const NOT_A_ROUTE_MODIFIER = "(?![\\s-]+(?:route|gateway|endpoint|url|path|proxy|server|service|api|host)s?(?![\\w-]))";
126
+ const MODEL_WORD = `\\bmodel\\b${NOT_A_ROUTE_MODIFIER}`;
127
+ const MODEL_OR_DEPLOYMENT = `\\b(?:model|deployment)\\b${NOT_A_ROUTE_MODIFIER}`;
128
+ const NAMED_MODEL_ERROR = `(?:unknown|no such|invalid|unsupported) model${NOT_A_ROUTE_MODIFIER}`;
129
+ const MODEL_404_TEXT = new RegExp(`model[_ ]?not[_ ]?found|${NAMED_MODEL_ERROR}|${MODEL_OR_DEPLOYMENT}.{0,120}\\b(?:does not exist|not found|unavailable|not (?:available|supported))\\b|\\b(?:does not exist|not found)\\b.{0,120}${MODEL_OR_DEPLOYMENT}|unable to access.{0,60}${MODEL_WORD}`, "i");
120
130
  /**
121
131
  * Generic fallback rule table covering the five categories every
122
132
  * OpenAI-compatible provider already hand-rolled near-identically: