@juspay/neurolink 12.47.2 → 12.47.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -366,7 +366,7 @@ export function runAgenticLoop(adapter, initialConversation, options) {
366
366
  // The caller's span, when it passes one. withProviderRetry writes
367
367
  // gen_ai.provider.total_attempts here, so a loop that threaded a
368
368
  // span before it moved onto this engine keeps emitting it.
369
- options.span, `${adapter.providerLabel}.step`);
369
+ options.span, `${adapter.providerLabel}.step`, undefined, internalAbort.signal);
370
370
  }
371
371
  catch (err) {
372
372
  throw err instanceof PostEmissionStepError ? err.cause : err;
@@ -445,7 +445,10 @@ export class VideoProcessor extends BaseFileProcessor {
445
445
  metadata = this.buildMetadata(probeResult.data, buffer.length);
446
446
  }
447
447
  }
448
- if (!metadata) {
448
+ // mediabunny reports a duration of 0 for a clip it can open but not
449
+ // time, and ffprobe's "N/A" parses to NaN: neither is a missing
450
+ // result, so both take the ffmpeg fallback.
451
+ if (!metadata || !(metadata.duration > 0)) {
449
452
  // ffmpeg-static ships ffmpeg only, so a host that relies on it has
450
453
  // no ffprobe. Without a duration no frame timestamps can be chosen.
451
454
  const ffmpegProbe = await this.probeVideoWithFfmpeg(tempVideoPath);
@@ -456,7 +459,9 @@ export class VideoProcessor extends BaseFileProcessor {
456
459
  // Nothing downstream reports this: an empty duration selects no
457
460
  // frames, the request still succeeds, and the model is simply
458
461
  // told nothing about the video.
459
- logger.warn(`[NEUROLINK] No metadata could be read for ${filename} (mediabunny, ffprobe and ffmpeg all failed), so no keyframes will be extracted: ${ffmpegProbe.error}`);
462
+ logger.warn(metadata
463
+ ? `[NEUROLINK] No positive duration could be read for ${filename} (the first reader gave none and ffmpeg could not supply one), so frame times cannot be chosen: ${ffmpegProbe.error}`
464
+ : `[NEUROLINK] No metadata could be read for ${filename} (mediabunny, ffprobe and ffmpeg all failed), so no keyframes will be extracted: ${ffmpegProbe.error}`);
460
465
  }
461
466
  }
462
467
  if (!metadata) {
@@ -242,7 +242,7 @@ export class AmazonSageMakerProvider extends BaseProvider {
242
242
  ...(options.toolTimeoutMs !== undefined
243
243
  ? { toolTimeoutMs: options.toolTimeoutMs }
244
244
  : {}),
245
- runStep: (call) => withProviderRetry(call, undefined, "sagemaker generate").catch((err) => {
245
+ runStep: (call) => withProviderRetry(call, undefined, "sagemaker generate", undefined, options.abortSignal).catch((err) => {
246
246
  throw this.handleProviderError(err);
247
247
  }),
248
248
  }, toolExecutionSummaries);
@@ -11,6 +11,15 @@ export declare class AnthropicProvider extends BaseProvider {
11
11
  private readonly authMethod;
12
12
  private readonly subscriptionTier;
13
13
  private readonly enableBetaFeatures;
14
+ /**
15
+ * Where requests go when not to `api.anthropic.com`: the credentials'
16
+ * `baseURL`, else the config's, else `ANTHROPIC_BASE_URL` — normalized once
17
+ * (no trailing slash, no `/vN` suffix: the SDK appends `/v1` itself).
18
+ * Undefined means the vendor's own endpoint; every "proxy in use" decision
19
+ * reads this, never the environment directly, so a per-instance gateway is
20
+ * honoured the same way the environment one always was.
21
+ */
22
+ private readonly baseURL;
14
23
  private oauthToken;
15
24
  private lastResponseMetadata;
16
25
  private usageInfo;
@@ -25,6 +34,7 @@ export declare class AnthropicProvider extends BaseProvider {
25
34
  constructor(modelName?: string, sdk?: unknown, config?: AnthropicProviderConfig, credentials?: {
26
35
  apiKey?: string;
27
36
  oauthToken?: string;
37
+ baseURL?: string;
28
38
  });
29
39
  /**
30
40
  * Get authentication headers based on current auth method and configuration.
@@ -58,6 +58,31 @@ const getAnthropicApiKey = () => {
58
58
  const getDefaultAnthropicModel = () => {
59
59
  return getProviderModel("ANTHROPIC_MODEL", AnthropicModels.CLAUDE_SONNET_4_6);
60
60
  };
61
+ /**
62
+ * The official Anthropic SDK builds `${baseURL}/v1/messages` itself, so a
63
+ * version-suffixed base URL — the form the previous @ai-sdk/anthropic
64
+ * implementation REQUIRED (`https://api.anthropic.com/v1`) — would double up
65
+ * as `/v1/v1/messages`. Normalize the inverse way: strip trailing slashes and
66
+ * a trailing `/vN` segment, so both historical forms keep working whether the
67
+ * URL comes from the credentials, the config or `ANTHROPIC_BASE_URL`. Blank
68
+ * means "the vendor's endpoint".
69
+ */
70
+ function normalizeAnthropicBaseURL(raw) {
71
+ const value = raw?.trim();
72
+ if (!value) {
73
+ return undefined;
74
+ }
75
+ const trimmed = value.replace(/\/+$/, "");
76
+ const stripped = trimmed.replace(/\/v\d+$/, "");
77
+ if (stripped !== trimmed) {
78
+ logger.debug("[AnthropicProvider] Stripping the version suffix from the base URL — " +
79
+ "the official Anthropic SDK appends /v1 to the base URL itself.", {
80
+ baseURL: redactUrlCredentials(value),
81
+ rewrittenTo: redactUrlCredentials(stripped),
82
+ });
83
+ }
84
+ return stripped;
85
+ }
61
86
  const streamTracer = trace.getTracer("neurolink.provider.anthropic");
62
87
  /**
63
88
  * Get OAuth token from stored credentials file or environment.
@@ -476,6 +501,15 @@ export class AnthropicProvider extends BaseProvider {
476
501
  authMethod;
477
502
  subscriptionTier;
478
503
  enableBetaFeatures;
504
+ /**
505
+ * Where requests go when not to `api.anthropic.com`: the credentials'
506
+ * `baseURL`, else the config's, else `ANTHROPIC_BASE_URL` — normalized once
507
+ * (no trailing slash, no `/vN` suffix: the SDK appends `/v1` itself).
508
+ * Undefined means the vendor's own endpoint; every "proxy in use" decision
509
+ * reads this, never the environment directly, so a per-instance gateway is
510
+ * honoured the same way the environment one always was.
511
+ */
512
+ baseURL;
479
513
  oauthToken;
480
514
  lastResponseMetadata = null;
481
515
  usageInfo = null;
@@ -511,12 +545,16 @@ export class AnthropicProvider extends BaseProvider {
511
545
  const subscriptionTier = config?.subscriptionTier ??
512
546
  (authMethod === "oauth" ? detectSubscriptionTier(oauthToken) : "api");
513
547
  const targetModel = modelName || getDefaultAnthropicModel();
548
+ // Resolved before super(): the tier check below needs it, and the
549
+ // credentials win over the config, which wins over the environment —
550
+ // the same precedence `apiKey` has.
551
+ const baseURL = normalizeAnthropicBaseURL(credentials?.baseURL ?? config?.baseURL ?? process.env.ANTHROPIC_BASE_URL);
514
552
  // Determine effective model based on tier access.
515
- // Skip tier validation when a proxy is in use (ANTHROPIC_BASE_URL is set)
516
- // — the proxy handles model access and auth, so the SDK should pass
517
- // the requested model through without downgrading.
553
+ // Skip tier validation when a proxy is in use (a base URL is set) — the
554
+ // proxy handles model access and auth, so the SDK should pass the
555
+ // requested model through without downgrading.
518
556
  let effectiveModel = targetModel;
519
- const usingProxy = !!process.env.ANTHROPIC_BASE_URL;
557
+ const usingProxy = baseURL !== undefined;
520
558
  if (!usingProxy &&
521
559
  subscriptionTier !== "api" &&
522
560
  !isModelAvailableForTier(targetModel, subscriptionTier)) {
@@ -533,6 +571,7 @@ export class AnthropicProvider extends BaseProvider {
533
571
  // Store computed values
534
572
  this.oauthToken = oauthToken;
535
573
  this.subscriptionTier = subscriptionTier;
574
+ this.baseURL = baseURL;
536
575
  // Use the auth method already resolved above (before tier computation)
537
576
  this.authMethod = authMethod;
538
577
  // Build headers based on auth method and subscription tier
@@ -570,6 +609,9 @@ export class AnthropicProvider extends BaseProvider {
570
609
  // The claude-code-20250219 beta header triggers "credential only for Claude Code" error
571
610
  client = new Anthropic({
572
611
  apiKey: "oauth-authenticated", // Placeholder, actual auth is in fetch wrapper
612
+ // The same gateway the API-key branch honours: without it, OAuth
613
+ // credentials that also name a base URL would still reach the vendor.
614
+ ...(this.baseURL && { baseURL: this.baseURL }),
573
615
  // Note: No headers passed - fetch wrapper sets oauth-2025-04-20 beta header
574
616
  // Limit capture wraps the OAuth fetch so subscription quota headers
575
617
  // (anthropic-ratelimit-unified-*) are recorded on every request —
@@ -599,33 +641,10 @@ export class AnthropicProvider extends BaseProvider {
599
641
  else {
600
642
  // Traditional API key authentication
601
643
  const apiKeyToUse = credentials?.apiKey ?? config?.apiKey ?? getAnthropicApiKey();
602
- // The official Anthropic SDK builds `${baseURL}/v1/messages` itself, so
603
- // a version-suffixed base URL — the form the previous @ai-sdk/anthropic
604
- // implementation REQUIRED (`https://api.anthropic.com/v1`) — would
605
- // double up as `/v1/v1/messages`. Normalize the inverse way now: strip
606
- // a trailing `/vN` segment when present so both historical forms of
607
- // ANTHROPIC_BASE_URL keep working.
608
- const normalizedBaseURL = (() => {
609
- const raw = process.env.ANTHROPIC_BASE_URL;
610
- if (!raw) {
611
- return undefined;
612
- }
613
- const trimmed = raw.replace(/\/+$/, "");
614
- const stripped = trimmed.replace(/\/v\d+$/, "");
615
- if (stripped !== trimmed) {
616
- logger.debug("[AnthropicProvider] Stripping the version suffix from " +
617
- "ANTHROPIC_BASE_URL — the official Anthropic SDK appends /v1 " +
618
- "to the base URL itself.", {
619
- baseURL: redactUrlCredentials(raw),
620
- rewrittenTo: redactUrlCredentials(stripped),
621
- });
622
- }
623
- return stripped;
624
- })();
625
644
  client = new Anthropic({
626
645
  apiKey: apiKeyToUse,
627
646
  defaultHeaders: headers,
628
- ...(normalizedBaseURL && { baseURL: normalizedBaseURL }),
647
+ ...(this.baseURL && { baseURL: this.baseURL }),
629
648
  // Same capture as the OAuth branch: works for direct API-key traffic
630
649
  // (legacy requests/tokens counters) and for the NeuroLink Claude proxy
631
650
  // (verbatim unified quota plus x-neurolink-* account/pool state).
@@ -676,10 +695,11 @@ export class AnthropicProvider extends BaseProvider {
676
695
  */
677
696
  getAuthHeaders() {
678
697
  const headers = {};
679
- // When routing through proxy (ANTHROPIC_BASE_URL set), use the full
680
- // OAuth beta set so the proxy forwards them upstream. Without these,
681
- // Anthropic treats the request with tighter non-subscription rate limits.
682
- const usingProxy = !!process.env.ANTHROPIC_BASE_URL;
698
+ // When routing through a proxy (a base URL is set — per instance or
699
+ // ANTHROPIC_BASE_URL), use the full OAuth beta set so the proxy forwards
700
+ // them upstream. Without these, Anthropic treats the request with tighter
701
+ // non-subscription rate limits.
702
+ const usingProxy = this.baseURL !== undefined;
683
703
  if (this.enableBetaFeatures) {
684
704
  if (usingProxy) {
685
705
  // The 1M-context beta requires a plan upgrade on most accounts;
@@ -705,9 +725,9 @@ export class AnthropicProvider extends BaseProvider {
705
725
  }
706
726
  }
707
727
  if (usingProxy) {
708
- // WAFs in front of ANTHROPIC_BASE_URL proxies commonly block the bare
709
- // SDK UA ("Anthropic/JS x.y.z"); send the claude-cli UA the OAuth path
710
- // already uses. Direct-to-Anthropic traffic keeps the honest SDK UA.
728
+ // WAFs in front of Anthropic proxies commonly block the bare SDK UA
729
+ // ("Anthropic/JS x.y.z"); send the claude-cli UA the OAuth path already
730
+ // uses. Direct-to-Anthropic traffic keeps the honest SDK UA.
711
731
  headers["User-Agent"] = CLAUDE_CLI_USER_AGENT;
712
732
  }
713
733
  // Add subscription-specific headers if applicable
@@ -736,8 +756,8 @@ export class AnthropicProvider extends BaseProvider {
736
756
  // Proxy mode: bypass tier validation entirely — the proxy handles model
737
757
  // access. Log at debug level so users can tell why an unknown model name
738
758
  // "validated" when their proxy may not actually expose it.
739
- if (process.env.ANTHROPIC_BASE_URL) {
740
- logger.debug("[validateModelAccess] Bypassing tier check (ANTHROPIC_BASE_URL set — proxy enforces access)", { model });
759
+ if (this.baseURL !== undefined) {
760
+ logger.debug("[validateModelAccess] Bypassing tier check (a base URL is set — the proxy enforces access)", { model });
741
761
  return true;
742
762
  }
743
763
  // API tier has access to all models
@@ -1603,7 +1623,7 @@ export class AnthropicProvider extends BaseProvider {
1603
1623
  ...(options.toolTimeoutMs !== undefined
1604
1624
  ? { toolTimeoutMs: options.toolTimeoutMs }
1605
1625
  : {}),
1606
- runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, "anthropic generate").catch((err) => {
1626
+ runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, "anthropic generate", undefined, options.abortSignal).catch((err) => {
1607
1627
  throw this.handleProviderError(err);
1608
1628
  }),
1609
1629
  }, toolExecutionSummaries);
@@ -270,7 +270,9 @@ export class NvidiaNimProvider extends OpenAIChatCompletionsProvider {
270
270
  message: "NVIDIA NIM rate limit exceeded",
271
271
  },
272
272
  {
273
- match: (ctx) => /404|model_not_found/.test(ctx.message),
273
+ // NIM answers most of its roster with a 404 whose text ("Function …
274
+ // not found for account …") names no model, so the status decides.
275
+ match: (ctx) => ctx.statusCode === 404 || /404|model_not_found/.test(ctx.message),
274
276
  errorClass: InvalidModelError,
275
277
  message: () => `NVIDIA NIM model '${this.modelName}' not available. Browse the catalog at https://build.nvidia.com/models`,
276
278
  },
@@ -989,7 +989,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
989
989
  // supplied around each call: without them a 429 surfaces as a raw
990
990
  // upstream string instead of a RateLimitError, and a throttle is
991
991
  // never retried.
992
- runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, `${this.providerName} generate`).catch((err) => {
992
+ runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, `${this.providerName} generate`, undefined, options.abortSignal).catch((err) => {
993
993
  throw this.handleProviderError(err);
994
994
  }),
995
995
  }, toolExecutionSummaries);
@@ -2188,7 +2188,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
2188
2188
  };
2189
2189
  let res;
2190
2190
  try {
2191
- res = await withProviderRetry(doFetch, trace.getActiveSpan() ?? undefined, `${this.providerName} stream`);
2191
+ res = await withProviderRetry(doFetch, trace.getActiveSpan() ?? undefined, `${this.providerName} stream`, undefined, args.abortSignal);
2192
2192
  }
2193
2193
  catch (err) {
2194
2194
  // The one-shot 400 context-overflow fallback lives outside
@@ -272,9 +272,9 @@ export type ModelRoutingOptions = {
272
272
  /**
273
273
  * A single model's metadata inside a provider's manifest. This is the one
274
274
  * canonical shape every model-metadata consumer (context windows, pricing,
275
- * MODEL_REGISTRY, vision capability, output-token ceilings) is intended to
276
- * migrate onto — this PR is purely additive and does not yet move any
277
- * consumer over.
275
+ * MODEL_REGISTRY, vision capability, output-token ceilings) reads from —
276
+ * contextWindows.ts, pricing.ts, modelRegistry.ts, providerImageAdapter.ts and
277
+ * core/constants.ts.
278
278
  *
279
279
  * `pricingPerMTok` is optional by design: a model with no verified price
280
280
  * (e.g. a just-announced model pricing.ts hasn't priced yet) must not report
@@ -311,9 +311,10 @@ export type ProviderModelManifestEntry = {
311
311
  * forward verbatim for the ids that already had a MODEL_REGISTRY entry
312
312
  * before this migration. Absent for every id that never had one — those
313
313
  * get performance/useCases/category derived mechanically instead (see
314
- * Task 9's buildModelRegistryFromManifests). Never populate this for a
315
- * genuinely new model: mechanical derivation is the correct default, and
316
- * a fabricated "curated" value would be worse than an honestly-derived one.
314
+ * buildManifestDerivedEntries in src/lib/models/modelRegistry.ts). Never
315
+ * populate this for a genuinely new model: mechanical derivation is the
316
+ * correct default, and a fabricated "curated" value would be worse than an
317
+ * honestly-derived one.
317
318
  */
318
319
  curated?: {
319
320
  performance?: ModelPerformance;
@@ -105,9 +105,16 @@ export type NeurolinkCredentials = {
105
105
  apiKey?: string;
106
106
  baseURL?: string;
107
107
  };
108
+ /**
109
+ * Anthropic. `baseURL` points the official SDK client at a gateway or proxy
110
+ * instead of `api.anthropic.com` (with or without a trailing `/v1`); it
111
+ * takes precedence over `ANTHROPIC_BASE_URL`, exactly as `apiKey` does over
112
+ * `ANTHROPIC_API_KEY`.
113
+ */
108
114
  anthropic?: {
109
115
  apiKey?: string;
110
116
  oauthToken?: string;
117
+ baseURL?: string;
111
118
  };
112
119
  googleAiStudio?: {
113
120
  apiKey?: string;
@@ -2401,7 +2408,14 @@ export type ProviderDescriptor = {
2401
2408
  * checks to its own server (TypeSafe). See {@link DecisionLimits}.
2402
2409
  */
2403
2410
  decisionLimits?: DecisionLimits;
2404
- /** Ascending priority (1 = tried first) in the auto-select fallback chain used by getBestProvider(). Undefined = not part of the auto-select chain. */
2411
+ /**
2412
+ * Ascending priority (1 = tried first) in the auto-select fallback chain
2413
+ * used by getBestProvider(). Undefined = not part of the auto-select chain.
2414
+ *
2415
+ * Priorities favor local and self-hosted deployments first to avoid an
2416
+ * external dependency during fallback, then cloud providers according to
2417
+ * reliability, feature set and model coverage.
2418
+ */
2405
2419
  autoSelectPriority?: number;
2406
2420
  /** Format-validation regex sourced from providerConfig.ts's API_KEY_FORMATS, when one exists for this provider. */
2407
2421
  apiKeyFormatPattern?: RegExp;
@@ -21,10 +21,13 @@ export declare function classifyProviderError(error: unknown, rules: ProviderErr
21
21
  /**
22
22
  * Generic fallback rule table covering the five categories every
23
23
  * OpenAI-compatible provider already hand-rolled near-identically:
24
- * auth (401), rate limit (429), model-not-found (404), network/connection
25
- * errors, and 5xx server errors. Providers with a provider-specific auth
26
- * message (naming the exact env var) prepend one override rule and spread
27
- * this table after it — see errorClassifier usage in any migrated
24
+ * auth (401), rate limit (429), model-not-found, network/connection
25
+ * errors, and 5xx server errors. Model-not-found is a 404 whose text names
26
+ * the model or deployment as missing, or the old "model not found" message
27
+ * text at any status; any other 404 stays a plain `ProviderError` carrying
28
+ * the status and the vendor's message. Providers with a provider-specific
29
+ * auth message (naming the exact env var) prepend one override rule and
30
+ * spread this table after it — see errorClassifier usage in any migrated
28
31
  * provider's formatProviderError for the pattern.
29
32
  */
30
33
  export declare const DEFAULT_ERROR_RULES: ProviderErrorRule[];
@@ -113,13 +113,20 @@ export function classifyProviderError(error, rules, provider, modelName) {
113
113
  const message = typeof rule.message === "function" ? rule.message(ctx) : rule.message;
114
114
  return new rule.errorClass(message, provider);
115
115
  }
116
+ // A 404 alone is a route answer (a wrong base URL gives the same reply): it only
117
+ // means "missing model" when the text names a model or deployment as absent.
118
+ // The gap is bounded rather than "no dot" because real model ids contain dots.
119
+ const MODEL_404_TEXT = /model[_ ]?not[_ ]?found|unknown model|no such model|invalid model|unsupported model|\b(?:model|deployment)\b.{0,120}\b(?:does not exist|not found|unavailable|not (?:available|supported))\b|\b(?:does not exist|not found)\b.{0,120}\b(?:model|deployment)\b|unable to access.{0,60}\bmodel\b/i;
116
120
  /**
117
121
  * Generic fallback rule table covering the five categories every
118
122
  * OpenAI-compatible provider already hand-rolled near-identically:
119
- * auth (401), rate limit (429), model-not-found (404), network/connection
120
- * errors, and 5xx server errors. Providers with a provider-specific auth
121
- * message (naming the exact env var) prepend one override rule and spread
122
- * this table after it — see errorClassifier usage in any migrated
123
+ * auth (401), rate limit (429), model-not-found, network/connection
124
+ * errors, and 5xx server errors. Model-not-found is a 404 whose text names
125
+ * the model or deployment as missing, or the old "model not found" message
126
+ * text at any status; any other 404 stays a plain `ProviderError` carrying
127
+ * the status and the vendor's message. Providers with a provider-specific
128
+ * auth message (naming the exact env var) prepend one override rule and
129
+ * spread this table after it — see errorClassifier usage in any migrated
123
130
  * provider's formatProviderError for the pattern.
124
131
  */
125
132
  export const DEFAULT_ERROR_RULES = [
@@ -135,13 +142,18 @@ export const DEFAULT_ERROR_RULES = [
135
142
  message: (ctx) => `${ctx.provider} rate limit exceeded. Please try again later.`,
136
143
  },
137
144
  {
138
- match: (ctx) => ctx.statusCode === 404 ||
139
- /model_not_found|model not found/i.test(ctx.message),
145
+ match: (ctx) => /model_not_found|model not found/i.test(ctx.message) ||
146
+ (ctx.statusCode === 404 && MODEL_404_TEXT.test(ctx.message)),
140
147
  errorClass: InvalidModelError,
141
148
  message: (ctx) => ctx.modelName
142
149
  ? `${ctx.provider} model '${ctx.modelName}' not found.`
143
150
  : `${ctx.provider} model not found.`,
144
151
  },
152
+ {
153
+ match: (ctx) => ctx.statusCode === 404,
154
+ errorClass: ProviderError,
155
+ message: (ctx) => `${ctx.provider} returned HTTP 404: ${ctx.message}`,
156
+ },
145
157
  {
146
158
  // Message regex covers providers/SDKs that surface a code as text
147
159
  // (e.g. AWS SDK wrapping "ECONNRESET" into its own message). errorCode
@@ -2566,7 +2566,14 @@ mediaOptions = {}) {
2566
2566
  });
2567
2567
  try {
2568
2568
  const extracted = await parser.getText();
2569
- const pdfText = (extracted?.text ?? "").trim();
2569
+ // pdf-parse appends a "-- n of N --" marker after every page, so
2570
+ // extracted.text is never empty; a scan is detected from the
2571
+ // per-page text instead.
2572
+ const pageTexts = extracted?.pages?.map((page) => page.text) ?? [
2573
+ extracted?.text ?? "",
2574
+ ];
2575
+ const hasTextLayer = pageTexts.some((text) => text.trim().length > 0);
2576
+ const pdfText = hasTextLayer ? (extracted?.text ?? "").trim() : "";
2570
2577
  if (pdfText.length > 0) {
2571
2578
  content.push({
2572
2579
  type: "text",
@@ -2574,6 +2581,13 @@ mediaOptions = {}) {
2574
2581
  });
2575
2582
  logger.info(`[PDF→Text] ✅ Extracted text for non-vision provider ${provider}: ${name} (${pdfText.length} chars)`);
2576
2583
  }
2584
+ else if (providerCanSeeImages) {
2585
+ content.push({
2586
+ type: "text",
2587
+ text: `\n[Attached PDF: ${name} — no extractable text layer (likely a scanned document); page images are attached below.]`,
2588
+ });
2589
+ logger.warn(`[PDF→Text] ${name} has no text layer; page images follow`);
2590
+ }
2577
2591
  else {
2578
2592
  content.push({
2579
2593
  type: "text",
@@ -15,6 +15,13 @@ import type { CatalogProviderName, ModelChoice } from "../types/index.js";
15
15
  *
16
16
  * AUTO is also excluded — it never had an entry here either (matches
17
17
  * pre-existing behavior: `getDefaultModel(AUTO)` returns `undefined`).
18
+ *
19
+ * This table is not what a call with no model uses. The runtime order is the
20
+ * explicit model, then the provider's environment variable, then the default in
21
+ * the dynamic model configuration, then the registry default
22
+ * (`providerRegistry.ts`); the OpenAI row mirrors that registry default. Only
23
+ * an `OpenAIProvider` constructed directly with none of those falls back to
24
+ * `gpt-5.4` (`openAI/client.ts`).
18
25
  */
19
26
  export declare const DEFAULT_MODELS: Record<Exclude<AIProviderName, CatalogProviderName | AIProviderName.AUTO>, string>;
20
27
  /**
@@ -56,6 +63,9 @@ export declare function getAllProviderChoices(): string[];
56
63
  /**
57
64
  * Get the default model for a provider
58
65
  *
66
+ * Reads `DEFAULT_MODELS`, which mirrors the registry default rather than the
67
+ * runtime resolution order (see the note on that table).
68
+ *
59
69
  * @param provider - The AI provider
60
70
  * @returns Default model string for the provider
61
71
  */
@@ -43,14 +43,19 @@ function catalogTopModels(entry) {
43
43
  */
44
44
  const TOP_MODELS_CONFIG = {
45
45
  [AIProviderName.OPENAI]: [
46
+ {
47
+ model: OpenAIModels.GPT_5_4,
48
+ description: "Recommended - Direct OpenAI GPT-5.4 model",
49
+ },
50
+ { model: OpenAIModels.GPT_5_4_MINI, description: "Cost-effective, fast" },
46
51
  {
47
52
  model: OpenAIModels.GPT_4O,
48
- description: "Recommended - Latest multimodal model",
53
+ description: "Previous generation multimodal model",
49
54
  },
50
55
  { model: OpenAIModels.GPT_4O_MINI, description: "Cost-effective, fast" },
51
56
  {
52
57
  model: OpenAIModels.GPT_5_2,
53
- description: "Latest flagship with deep reasoning",
58
+ description: "Previous flagship with deep reasoning",
54
59
  },
55
60
  { model: OpenAIModels.O3, description: "Advanced reasoning model" },
56
61
  { model: OpenAIModels.GPT_4_TURBO, description: "Previous generation" },
@@ -422,9 +427,16 @@ const TOP_MODELS_CONFIG = {
422
427
  *
423
428
  * AUTO is also excluded — it never had an entry here either (matches
424
429
  * pre-existing behavior: `getDefaultModel(AUTO)` returns `undefined`).
430
+ *
431
+ * This table is not what a call with no model uses. The runtime order is the
432
+ * explicit model, then the provider's environment variable, then the default in
433
+ * the dynamic model configuration, then the registry default
434
+ * (`providerRegistry.ts`); the OpenAI row mirrors that registry default. Only
435
+ * an `OpenAIProvider` constructed directly with none of those falls back to
436
+ * `gpt-5.4` (`openAI/client.ts`).
425
437
  */
426
438
  export const DEFAULT_MODELS = {
427
- [AIProviderName.OPENAI]: OpenAIModels.GPT_4O,
439
+ [AIProviderName.OPENAI]: OpenAIModels.GPT_4O_MINI,
428
440
  [AIProviderName.ANTHROPIC]: AnthropicModels.CLAUDE_SONNET_4_5,
429
441
  [AIProviderName.GOOGLE_AI]: GoogleAIModels.GEMINI_2_5_FLASH,
430
442
  [AIProviderName.VERTEX]: VertexModels.GEMINI_2_5_FLASH,
@@ -572,6 +584,9 @@ export function getAllProviderChoices() {
572
584
  /**
573
585
  * Get the default model for a provider
574
586
  *
587
+ * Reads `DEFAULT_MODELS`, which mirrors the registry default rather than the
588
+ * runtime resolution order (see the note on that table).
589
+ *
575
590
  * @param provider - The AI provider
576
591
  * @returns Default model string for the provider
577
592
  */
@@ -305,10 +305,17 @@ export class ProviderHealthChecker {
305
305
  }
306
306
  return;
307
307
  }
308
+ // A descriptor with no apiKey variable at all (LM Studio, llama.cpp) would
309
+ // otherwise read process.env[""] below and report a missing key named by
310
+ // nothing; hasProviderEnvVars already treats these as usable with defaults.
311
+ const providerDescriptor = ProviderFactory.getDescriptor(providerName);
308
312
  // Providers that don't use API keys directly
309
313
  if (providerName === AIProviderName.OLLAMA ||
310
314
  providerName === AIProviderName.BEDROCK ||
311
- providerName === AIProviderName.LITELLM) {
315
+ providerName === AIProviderName.LITELLM ||
316
+ providerDescriptor?.localRuntime === true ||
317
+ (providerDescriptor?.envVars.optional === true &&
318
+ !providerDescriptor.envVars.apiKey)) {
312
319
  healthStatus.hasApiKey = true;
313
320
  return;
314
321
  }
@@ -996,8 +1003,10 @@ export class ProviderHealthChecker {
996
1003
  ];
997
1004
  case AIProviderName.OPENAI:
998
1005
  return [
999
- OpenAIModels.GPT_4O,
1006
+ OpenAIModels.GPT_5_4,
1007
+ OpenAIModels.GPT_5_4_MINI,
1000
1008
  OpenAIModels.GPT_4O_MINI,
1009
+ OpenAIModels.GPT_4O,
1001
1010
  OpenAIModels.GPT_3_5_TURBO,
1002
1011
  ];
1003
1012
  case AIProviderName.GOOGLE_AI:
@@ -1541,8 +1550,18 @@ export class ProviderHealthChecker {
1541
1550
  return provider;
1542
1551
  }
1543
1552
  }
1544
- // Fallback to first healthy provider
1545
- const firstHealthyProvider = healthStatuses.find((h) => h.isHealthy);
1553
+ // Fallback to first healthy provider. A local runtime that nothing probes
1554
+ // (LM Studio, llama.cpp) is "healthy" only in that it needs no
1555
+ // configuration, which says nothing about whether it is running, so it
1556
+ // must not outrank a provider the caller actually configured.
1557
+ const firstHealthyProvider = healthStatuses.find((h) => {
1558
+ if (!h.isHealthy) {
1559
+ return false;
1560
+ }
1561
+ const descriptor = ProviderFactory.getDescriptor(h.provider);
1562
+ return !(descriptor?.localRuntime === true &&
1563
+ descriptor.healthCheck === "env-only");
1564
+ });
1546
1565
  if (firstHealthyProvider) {
1547
1566
  logger.info(`Using fallback healthy provider: ${firstHealthyProvider.provider}`);
1548
1567
  return firstHealthyProvider.provider;
@@ -86,6 +86,10 @@ export declare function getErrorStatusCode(error: unknown): number | undefined;
86
86
  * @param operation - The async operation to execute (should already use `maxRetries: 0`)
87
87
  * @param span - The OTel span to annotate with retry events and attributes
88
88
  * @param label - A human-readable label for log messages (e.g. "generateText", "streamText")
89
+ * @param sleep - Wait between attempts; receives the delay and the caller's abort signal
90
+ * @param abortSignal - The caller's cancellation. An abort during the wait ends the call with the
91
+ * signal's reason instead of waiting out the delay and running the operation
92
+ * again, and an already-aborted signal is checked before every attempt.
89
93
  * @returns The result of the operation
90
94
  */
91
- export declare function withProviderRetry<T>(operation: () => Promise<T>, span: Span | undefined, label: string, sleep?: (delayMs: number) => Promise<void>): Promise<T>;
95
+ export declare function withProviderRetry<T>(operation: () => Promise<T>, span: Span | undefined, label: string, sleep?: (delayMs: number, abortSignal?: AbortSignal) => Promise<void>, abortSignal?: AbortSignal): Promise<T>;
@@ -33,7 +33,21 @@ export const NO_HINT_FLOOR_MS = 10_000;
33
33
  * get a prompt rate-limit error rather than a silent multi-minute stall.
34
34
  */
35
35
  export const MAX_RETRY_AFTER_MS = 60_000;
36
- const sleepWithTimeout = (delayMs) => new Promise((resolve) => setTimeout(resolve, delayMs));
36
+ const sleepWithTimeout = (delayMs, abortSignal) => new Promise((resolve, reject) => {
37
+ if (abortSignal?.aborted) {
38
+ reject(abortSignal.reason);
39
+ return;
40
+ }
41
+ const onAbort = () => {
42
+ clearTimeout(timer);
43
+ reject(abortSignal?.reason);
44
+ };
45
+ const timer = setTimeout(() => {
46
+ abortSignal?.removeEventListener("abort", onAbort);
47
+ resolve();
48
+ }, delayMs);
49
+ abortSignal?.addEventListener("abort", onAbort, { once: true });
50
+ });
37
51
  /**
38
52
  * Check whether an error thrown by the AI SDK is retryable.
39
53
  *
@@ -244,10 +258,15 @@ function getRetryAfterMs(error) {
244
258
  * @param operation - The async operation to execute (should already use `maxRetries: 0`)
245
259
  * @param span - The OTel span to annotate with retry events and attributes
246
260
  * @param label - A human-readable label for log messages (e.g. "generateText", "streamText")
261
+ * @param sleep - Wait between attempts; receives the delay and the caller's abort signal
262
+ * @param abortSignal - The caller's cancellation. An abort during the wait ends the call with the
263
+ * signal's reason instead of waiting out the delay and running the operation
264
+ * again, and an already-aborted signal is checked before every attempt.
247
265
  * @returns The result of the operation
248
266
  */
249
- export async function withProviderRetry(operation, span, label, sleep = sleepWithTimeout) {
267
+ export async function withProviderRetry(operation, span, label, sleep = sleepWithTimeout, abortSignal) {
250
268
  for (let attempt = 0; attempt <= MAX_PROVIDER_RETRIES; attempt++) {
269
+ abortSignal?.throwIfAborted();
251
270
  try {
252
271
  const result = await operation();
253
272
  // Record how many attempts it took on the span
@@ -314,7 +333,10 @@ export async function withProviderRetry(operation, span, label, sleep = sleepWit
314
333
  statusCode,
315
334
  error: errorMessage,
316
335
  });
317
- await sleep(delay);
336
+ await sleep(delay, abortSignal);
337
+ // A custom sleep may ignore the signal, and a signal can abort in the
338
+ // same tick the timer fires.
339
+ abortSignal?.throwIfAborted();
318
340
  }
319
341
  }
320
342
  // This should never be reached due to the throw inside the loop,
@@ -52,16 +52,8 @@ export async function getBestProvider(requestedProvider) {
52
52
  // Fall through to cloud providers
53
53
  }
54
54
  }
55
- /**
56
- * Provider priority order rationale:
57
- * - LiteLLM and Ollama are prioritized first for local/self-hosted deployments,
58
- * avoiding unnecessary dependence on external providers during fallback scenarios.
59
- * - Vertex (Google Cloud AI) follows for enterprise-grade reliability.
60
- * - Google AI follows as second cloud priority for comprehensive Google AI ecosystem support.
61
- * - OpenAI maintains high priority due to its consistent reliability and broad model support.
62
- * - Other providers are ordered based on a combination of reliability, feature set, and historical performance.
63
- * Please update this comment if the order is changed in the future, and document the rationale for maintainability.
64
- */
55
+ // Order comes from ProviderDescriptor.autoSelectPriority (lower = tried
56
+ // first); see providerDescriptors.ts and the catalog JSON.
65
57
  const providers = PROVIDER_DESCRIPTORS.filter((d) => d.autoSelectPriority !== undefined)
66
58
  .sort((a, b) => (a.autoSelectPriority ?? 0) - (b.autoSelectPriority ?? 0))
67
59
  .map((d) => d.name);