@juspay/neurolink 12.47.2 → 12.47.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -2
- package/dist/browser/neurolink.min.js +354 -353
- package/dist/core/loopEngine.js +1 -1
- package/dist/processors/media/VideoProcessor.js +7 -2
- package/dist/providers/amazonSagemaker.js +1 -1
- package/dist/providers/anthropic/client.d.ts +10 -0
- package/dist/providers/anthropic/client.js +58 -38
- package/dist/providers/nvidiaNim/client.js +3 -1
- package/dist/providers/openaiChatCompletionsBase.js +2 -2
- package/dist/types/model.d.ts +7 -6
- package/dist/types/providers.d.ts +15 -1
- package/dist/utils/errorClassifier.d.ts +7 -4
- package/dist/utils/errorClassifier.js +18 -6
- package/dist/utils/messageBuilder.js +15 -1
- package/dist/utils/modelChoices.d.ts +10 -0
- package/dist/utils/modelChoices.js +18 -3
- package/dist/utils/providerHealth.js +23 -4
- package/dist/utils/providerRetry.d.ts +5 -1
- package/dist/utils/providerRetry.js +25 -3
- package/dist/utils/providerUtils.js +2 -10
- package/docs-site/static/search-index.json +10 -8
- package/package.json +2 -1
package/dist/core/loopEngine.js
CHANGED
|
@@ -366,7 +366,7 @@ export function runAgenticLoop(adapter, initialConversation, options) {
|
|
|
366
366
|
// The caller's span, when it passes one. withProviderRetry writes
|
|
367
367
|
// gen_ai.provider.total_attempts here, so a loop that threaded a
|
|
368
368
|
// span before it moved onto this engine keeps emitting it.
|
|
369
|
-
options.span, `${adapter.providerLabel}.step
|
|
369
|
+
options.span, `${adapter.providerLabel}.step`, undefined, internalAbort.signal);
|
|
370
370
|
}
|
|
371
371
|
catch (err) {
|
|
372
372
|
throw err instanceof PostEmissionStepError ? err.cause : err;
|
|
@@ -445,7 +445,10 @@ export class VideoProcessor extends BaseFileProcessor {
|
|
|
445
445
|
metadata = this.buildMetadata(probeResult.data, buffer.length);
|
|
446
446
|
}
|
|
447
447
|
}
|
|
448
|
-
|
|
448
|
+
// mediabunny reports a duration of 0 for a clip it can open but not
|
|
449
|
+
// time, and ffprobe's "N/A" parses to NaN: neither is a missing
|
|
450
|
+
// result, so both take the ffmpeg fallback.
|
|
451
|
+
if (!metadata || !(metadata.duration > 0)) {
|
|
449
452
|
// ffmpeg-static ships ffmpeg only, so a host that relies on it has
|
|
450
453
|
// no ffprobe. Without a duration no frame timestamps can be chosen.
|
|
451
454
|
const ffmpegProbe = await this.probeVideoWithFfmpeg(tempVideoPath);
|
|
@@ -456,7 +459,9 @@ export class VideoProcessor extends BaseFileProcessor {
|
|
|
456
459
|
// Nothing downstream reports this: an empty duration selects no
|
|
457
460
|
// frames, the request still succeeds, and the model is simply
|
|
458
461
|
// told nothing about the video.
|
|
459
|
-
logger.warn(
|
|
462
|
+
logger.warn(metadata
|
|
463
|
+
? `[NEUROLINK] No positive duration could be read for ${filename} (the first reader gave none and ffmpeg could not supply one), so frame times cannot be chosen: ${ffmpegProbe.error}`
|
|
464
|
+
: `[NEUROLINK] No metadata could be read for ${filename} (mediabunny, ffprobe and ffmpeg all failed), so no keyframes will be extracted: ${ffmpegProbe.error}`);
|
|
460
465
|
}
|
|
461
466
|
}
|
|
462
467
|
if (!metadata) {
|
|
@@ -242,7 +242,7 @@ export class AmazonSageMakerProvider extends BaseProvider {
|
|
|
242
242
|
...(options.toolTimeoutMs !== undefined
|
|
243
243
|
? { toolTimeoutMs: options.toolTimeoutMs }
|
|
244
244
|
: {}),
|
|
245
|
-
runStep: (call) => withProviderRetry(call, undefined, "sagemaker generate").catch((err) => {
|
|
245
|
+
runStep: (call) => withProviderRetry(call, undefined, "sagemaker generate", undefined, options.abortSignal).catch((err) => {
|
|
246
246
|
throw this.handleProviderError(err);
|
|
247
247
|
}),
|
|
248
248
|
}, toolExecutionSummaries);
|
|
@@ -11,6 +11,15 @@ export declare class AnthropicProvider extends BaseProvider {
|
|
|
11
11
|
private readonly authMethod;
|
|
12
12
|
private readonly subscriptionTier;
|
|
13
13
|
private readonly enableBetaFeatures;
|
|
14
|
+
/**
|
|
15
|
+
* Where requests go when not to `api.anthropic.com`: the credentials'
|
|
16
|
+
* `baseURL`, else the config's, else `ANTHROPIC_BASE_URL` — normalized once
|
|
17
|
+
* (no trailing slash, no `/vN` suffix: the SDK appends `/v1` itself).
|
|
18
|
+
* Undefined means the vendor's own endpoint; every "proxy in use" decision
|
|
19
|
+
* reads this, never the environment directly, so a per-instance gateway is
|
|
20
|
+
* honoured the same way the environment one always was.
|
|
21
|
+
*/
|
|
22
|
+
private readonly baseURL;
|
|
14
23
|
private oauthToken;
|
|
15
24
|
private lastResponseMetadata;
|
|
16
25
|
private usageInfo;
|
|
@@ -25,6 +34,7 @@ export declare class AnthropicProvider extends BaseProvider {
|
|
|
25
34
|
constructor(modelName?: string, sdk?: unknown, config?: AnthropicProviderConfig, credentials?: {
|
|
26
35
|
apiKey?: string;
|
|
27
36
|
oauthToken?: string;
|
|
37
|
+
baseURL?: string;
|
|
28
38
|
});
|
|
29
39
|
/**
|
|
30
40
|
* Get authentication headers based on current auth method and configuration.
|
|
@@ -58,6 +58,31 @@ const getAnthropicApiKey = () => {
|
|
|
58
58
|
const getDefaultAnthropicModel = () => {
|
|
59
59
|
return getProviderModel("ANTHROPIC_MODEL", AnthropicModels.CLAUDE_SONNET_4_6);
|
|
60
60
|
};
|
|
61
|
+
/**
|
|
62
|
+
* The official Anthropic SDK builds `${baseURL}/v1/messages` itself, so a
|
|
63
|
+
* version-suffixed base URL — the form the previous @ai-sdk/anthropic
|
|
64
|
+
* implementation REQUIRED (`https://api.anthropic.com/v1`) — would double up
|
|
65
|
+
* as `/v1/v1/messages`. Normalize the inverse way: strip trailing slashes and
|
|
66
|
+
* a trailing `/vN` segment, so both historical forms keep working whether the
|
|
67
|
+
* URL comes from the credentials, the config or `ANTHROPIC_BASE_URL`. Blank
|
|
68
|
+
* means "the vendor's endpoint".
|
|
69
|
+
*/
|
|
70
|
+
function normalizeAnthropicBaseURL(raw) {
|
|
71
|
+
const value = raw?.trim();
|
|
72
|
+
if (!value) {
|
|
73
|
+
return undefined;
|
|
74
|
+
}
|
|
75
|
+
const trimmed = value.replace(/\/+$/, "");
|
|
76
|
+
const stripped = trimmed.replace(/\/v\d+$/, "");
|
|
77
|
+
if (stripped !== trimmed) {
|
|
78
|
+
logger.debug("[AnthropicProvider] Stripping the version suffix from the base URL — " +
|
|
79
|
+
"the official Anthropic SDK appends /v1 to the base URL itself.", {
|
|
80
|
+
baseURL: redactUrlCredentials(value),
|
|
81
|
+
rewrittenTo: redactUrlCredentials(stripped),
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
return stripped;
|
|
85
|
+
}
|
|
61
86
|
const streamTracer = trace.getTracer("neurolink.provider.anthropic");
|
|
62
87
|
/**
|
|
63
88
|
* Get OAuth token from stored credentials file or environment.
|
|
@@ -476,6 +501,15 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
476
501
|
authMethod;
|
|
477
502
|
subscriptionTier;
|
|
478
503
|
enableBetaFeatures;
|
|
504
|
+
/**
|
|
505
|
+
* Where requests go when not to `api.anthropic.com`: the credentials'
|
|
506
|
+
* `baseURL`, else the config's, else `ANTHROPIC_BASE_URL` — normalized once
|
|
507
|
+
* (no trailing slash, no `/vN` suffix: the SDK appends `/v1` itself).
|
|
508
|
+
* Undefined means the vendor's own endpoint; every "proxy in use" decision
|
|
509
|
+
* reads this, never the environment directly, so a per-instance gateway is
|
|
510
|
+
* honoured the same way the environment one always was.
|
|
511
|
+
*/
|
|
512
|
+
baseURL;
|
|
479
513
|
oauthToken;
|
|
480
514
|
lastResponseMetadata = null;
|
|
481
515
|
usageInfo = null;
|
|
@@ -511,12 +545,16 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
511
545
|
const subscriptionTier = config?.subscriptionTier ??
|
|
512
546
|
(authMethod === "oauth" ? detectSubscriptionTier(oauthToken) : "api");
|
|
513
547
|
const targetModel = modelName || getDefaultAnthropicModel();
|
|
548
|
+
// Resolved before super(): the tier check below needs it, and the
|
|
549
|
+
// credentials win over the config, which wins over the environment —
|
|
550
|
+
// the same precedence `apiKey` has.
|
|
551
|
+
const baseURL = normalizeAnthropicBaseURL(credentials?.baseURL ?? config?.baseURL ?? process.env.ANTHROPIC_BASE_URL);
|
|
514
552
|
// Determine effective model based on tier access.
|
|
515
|
-
// Skip tier validation when a proxy is in use (
|
|
516
|
-
//
|
|
517
|
-
//
|
|
553
|
+
// Skip tier validation when a proxy is in use (a base URL is set) — the
|
|
554
|
+
// proxy handles model access and auth, so the SDK should pass the
|
|
555
|
+
// requested model through without downgrading.
|
|
518
556
|
let effectiveModel = targetModel;
|
|
519
|
-
const usingProxy =
|
|
557
|
+
const usingProxy = baseURL !== undefined;
|
|
520
558
|
if (!usingProxy &&
|
|
521
559
|
subscriptionTier !== "api" &&
|
|
522
560
|
!isModelAvailableForTier(targetModel, subscriptionTier)) {
|
|
@@ -533,6 +571,7 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
533
571
|
// Store computed values
|
|
534
572
|
this.oauthToken = oauthToken;
|
|
535
573
|
this.subscriptionTier = subscriptionTier;
|
|
574
|
+
this.baseURL = baseURL;
|
|
536
575
|
// Use the auth method already resolved above (before tier computation)
|
|
537
576
|
this.authMethod = authMethod;
|
|
538
577
|
// Build headers based on auth method and subscription tier
|
|
@@ -570,6 +609,9 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
570
609
|
// The claude-code-20250219 beta header triggers "credential only for Claude Code" error
|
|
571
610
|
client = new Anthropic({
|
|
572
611
|
apiKey: "oauth-authenticated", // Placeholder, actual auth is in fetch wrapper
|
|
612
|
+
// The same gateway the API-key branch honours: without it, OAuth
|
|
613
|
+
// credentials that also name a base URL would still reach the vendor.
|
|
614
|
+
...(this.baseURL && { baseURL: this.baseURL }),
|
|
573
615
|
// Note: No headers passed - fetch wrapper sets oauth-2025-04-20 beta header
|
|
574
616
|
// Limit capture wraps the OAuth fetch so subscription quota headers
|
|
575
617
|
// (anthropic-ratelimit-unified-*) are recorded on every request —
|
|
@@ -599,33 +641,10 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
599
641
|
else {
|
|
600
642
|
// Traditional API key authentication
|
|
601
643
|
const apiKeyToUse = credentials?.apiKey ?? config?.apiKey ?? getAnthropicApiKey();
|
|
602
|
-
// The official Anthropic SDK builds `${baseURL}/v1/messages` itself, so
|
|
603
|
-
// a version-suffixed base URL — the form the previous @ai-sdk/anthropic
|
|
604
|
-
// implementation REQUIRED (`https://api.anthropic.com/v1`) — would
|
|
605
|
-
// double up as `/v1/v1/messages`. Normalize the inverse way now: strip
|
|
606
|
-
// a trailing `/vN` segment when present so both historical forms of
|
|
607
|
-
// ANTHROPIC_BASE_URL keep working.
|
|
608
|
-
const normalizedBaseURL = (() => {
|
|
609
|
-
const raw = process.env.ANTHROPIC_BASE_URL;
|
|
610
|
-
if (!raw) {
|
|
611
|
-
return undefined;
|
|
612
|
-
}
|
|
613
|
-
const trimmed = raw.replace(/\/+$/, "");
|
|
614
|
-
const stripped = trimmed.replace(/\/v\d+$/, "");
|
|
615
|
-
if (stripped !== trimmed) {
|
|
616
|
-
logger.debug("[AnthropicProvider] Stripping the version suffix from " +
|
|
617
|
-
"ANTHROPIC_BASE_URL — the official Anthropic SDK appends /v1 " +
|
|
618
|
-
"to the base URL itself.", {
|
|
619
|
-
baseURL: redactUrlCredentials(raw),
|
|
620
|
-
rewrittenTo: redactUrlCredentials(stripped),
|
|
621
|
-
});
|
|
622
|
-
}
|
|
623
|
-
return stripped;
|
|
624
|
-
})();
|
|
625
644
|
client = new Anthropic({
|
|
626
645
|
apiKey: apiKeyToUse,
|
|
627
646
|
defaultHeaders: headers,
|
|
628
|
-
...(
|
|
647
|
+
...(this.baseURL && { baseURL: this.baseURL }),
|
|
629
648
|
// Same capture as the OAuth branch: works for direct API-key traffic
|
|
630
649
|
// (legacy requests/tokens counters) and for the NeuroLink Claude proxy
|
|
631
650
|
// (verbatim unified quota plus x-neurolink-* account/pool state).
|
|
@@ -676,10 +695,11 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
676
695
|
*/
|
|
677
696
|
getAuthHeaders() {
|
|
678
697
|
const headers = {};
|
|
679
|
-
// When routing through proxy (
|
|
680
|
-
// OAuth beta set so the proxy forwards
|
|
681
|
-
// Anthropic treats the request with tighter
|
|
682
|
-
|
|
698
|
+
// When routing through a proxy (a base URL is set — per instance or
|
|
699
|
+
// ANTHROPIC_BASE_URL), use the full OAuth beta set so the proxy forwards
|
|
700
|
+
// them upstream. Without these, Anthropic treats the request with tighter
|
|
701
|
+
// non-subscription rate limits.
|
|
702
|
+
const usingProxy = this.baseURL !== undefined;
|
|
683
703
|
if (this.enableBetaFeatures) {
|
|
684
704
|
if (usingProxy) {
|
|
685
705
|
// The 1M-context beta requires a plan upgrade on most accounts;
|
|
@@ -705,9 +725,9 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
705
725
|
}
|
|
706
726
|
}
|
|
707
727
|
if (usingProxy) {
|
|
708
|
-
// WAFs in front of
|
|
709
|
-
//
|
|
710
|
-
//
|
|
728
|
+
// WAFs in front of Anthropic proxies commonly block the bare SDK UA
|
|
729
|
+
// ("Anthropic/JS x.y.z"); send the claude-cli UA the OAuth path already
|
|
730
|
+
// uses. Direct-to-Anthropic traffic keeps the honest SDK UA.
|
|
711
731
|
headers["User-Agent"] = CLAUDE_CLI_USER_AGENT;
|
|
712
732
|
}
|
|
713
733
|
// Add subscription-specific headers if applicable
|
|
@@ -736,8 +756,8 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
736
756
|
// Proxy mode: bypass tier validation entirely — the proxy handles model
|
|
737
757
|
// access. Log at debug level so users can tell why an unknown model name
|
|
738
758
|
// "validated" when their proxy may not actually expose it.
|
|
739
|
-
if (
|
|
740
|
-
logger.debug("[validateModelAccess] Bypassing tier check (
|
|
759
|
+
if (this.baseURL !== undefined) {
|
|
760
|
+
logger.debug("[validateModelAccess] Bypassing tier check (a base URL is set — the proxy enforces access)", { model });
|
|
741
761
|
return true;
|
|
742
762
|
}
|
|
743
763
|
// API tier has access to all models
|
|
@@ -1603,7 +1623,7 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
1603
1623
|
...(options.toolTimeoutMs !== undefined
|
|
1604
1624
|
? { toolTimeoutMs: options.toolTimeoutMs }
|
|
1605
1625
|
: {}),
|
|
1606
|
-
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, "anthropic generate").catch((err) => {
|
|
1626
|
+
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, "anthropic generate", undefined, options.abortSignal).catch((err) => {
|
|
1607
1627
|
throw this.handleProviderError(err);
|
|
1608
1628
|
}),
|
|
1609
1629
|
}, toolExecutionSummaries);
|
|
@@ -270,7 +270,9 @@ export class NvidiaNimProvider extends OpenAIChatCompletionsProvider {
|
|
|
270
270
|
message: "NVIDIA NIM rate limit exceeded",
|
|
271
271
|
},
|
|
272
272
|
{
|
|
273
|
-
|
|
273
|
+
// NIM answers most of its roster with a 404 whose text ("Function …
|
|
274
|
+
// not found for account …") names no model, so the status decides.
|
|
275
|
+
match: (ctx) => ctx.statusCode === 404 || /404|model_not_found/.test(ctx.message),
|
|
274
276
|
errorClass: InvalidModelError,
|
|
275
277
|
message: () => `NVIDIA NIM model '${this.modelName}' not available. Browse the catalog at https://build.nvidia.com/models`,
|
|
276
278
|
},
|
|
@@ -989,7 +989,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
989
989
|
// supplied around each call: without them a 429 surfaces as a raw
|
|
990
990
|
// upstream string instead of a RateLimitError, and a throttle is
|
|
991
991
|
// never retried.
|
|
992
|
-
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, `${this.providerName} generate
|
|
992
|
+
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, `${this.providerName} generate`, undefined, options.abortSignal).catch((err) => {
|
|
993
993
|
throw this.handleProviderError(err);
|
|
994
994
|
}),
|
|
995
995
|
}, toolExecutionSummaries);
|
|
@@ -2188,7 +2188,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
2188
2188
|
};
|
|
2189
2189
|
let res;
|
|
2190
2190
|
try {
|
|
2191
|
-
res = await withProviderRetry(doFetch, trace.getActiveSpan() ?? undefined, `${this.providerName} stream
|
|
2191
|
+
res = await withProviderRetry(doFetch, trace.getActiveSpan() ?? undefined, `${this.providerName} stream`, undefined, args.abortSignal);
|
|
2192
2192
|
}
|
|
2193
2193
|
catch (err) {
|
|
2194
2194
|
// The one-shot 400 context-overflow fallback lives outside
|
package/dist/types/model.d.ts
CHANGED
|
@@ -272,9 +272,9 @@ export type ModelRoutingOptions = {
|
|
|
272
272
|
/**
|
|
273
273
|
* A single model's metadata inside a provider's manifest. This is the one
|
|
274
274
|
* canonical shape every model-metadata consumer (context windows, pricing,
|
|
275
|
-
* MODEL_REGISTRY, vision capability, output-token ceilings)
|
|
276
|
-
*
|
|
277
|
-
*
|
|
275
|
+
* MODEL_REGISTRY, vision capability, output-token ceilings) reads from —
|
|
276
|
+
* contextWindows.ts, pricing.ts, modelRegistry.ts, providerImageAdapter.ts and
|
|
277
|
+
* core/constants.ts.
|
|
278
278
|
*
|
|
279
279
|
* `pricingPerMTok` is optional by design: a model with no verified price
|
|
280
280
|
* (e.g. a just-announced model pricing.ts hasn't priced yet) must not report
|
|
@@ -311,9 +311,10 @@ export type ProviderModelManifestEntry = {
|
|
|
311
311
|
* forward verbatim for the ids that already had a MODEL_REGISTRY entry
|
|
312
312
|
* before this migration. Absent for every id that never had one — those
|
|
313
313
|
* get performance/useCases/category derived mechanically instead (see
|
|
314
|
-
*
|
|
315
|
-
* genuinely new model: mechanical derivation is the
|
|
316
|
-
* a fabricated "curated" value would be worse than an
|
|
314
|
+
* buildManifestDerivedEntries in src/lib/models/modelRegistry.ts). Never
|
|
315
|
+
* populate this for a genuinely new model: mechanical derivation is the
|
|
316
|
+
* correct default, and a fabricated "curated" value would be worse than an
|
|
317
|
+
* honestly-derived one.
|
|
317
318
|
*/
|
|
318
319
|
curated?: {
|
|
319
320
|
performance?: ModelPerformance;
|
|
@@ -105,9 +105,16 @@ export type NeurolinkCredentials = {
|
|
|
105
105
|
apiKey?: string;
|
|
106
106
|
baseURL?: string;
|
|
107
107
|
};
|
|
108
|
+
/**
|
|
109
|
+
* Anthropic. `baseURL` points the official SDK client at a gateway or proxy
|
|
110
|
+
* instead of `api.anthropic.com` (with or without a trailing `/v1`); it
|
|
111
|
+
* takes precedence over `ANTHROPIC_BASE_URL`, exactly as `apiKey` does over
|
|
112
|
+
* `ANTHROPIC_API_KEY`.
|
|
113
|
+
*/
|
|
108
114
|
anthropic?: {
|
|
109
115
|
apiKey?: string;
|
|
110
116
|
oauthToken?: string;
|
|
117
|
+
baseURL?: string;
|
|
111
118
|
};
|
|
112
119
|
googleAiStudio?: {
|
|
113
120
|
apiKey?: string;
|
|
@@ -2401,7 +2408,14 @@ export type ProviderDescriptor = {
|
|
|
2401
2408
|
* checks to its own server (TypeSafe). See {@link DecisionLimits}.
|
|
2402
2409
|
*/
|
|
2403
2410
|
decisionLimits?: DecisionLimits;
|
|
2404
|
-
/**
|
|
2411
|
+
/**
|
|
2412
|
+
* Ascending priority (1 = tried first) in the auto-select fallback chain
|
|
2413
|
+
* used by getBestProvider(). Undefined = not part of the auto-select chain.
|
|
2414
|
+
*
|
|
2415
|
+
* Priorities favor local and self-hosted deployments first to avoid an
|
|
2416
|
+
* external dependency during fallback, then cloud providers according to
|
|
2417
|
+
* reliability, feature set and model coverage.
|
|
2418
|
+
*/
|
|
2405
2419
|
autoSelectPriority?: number;
|
|
2406
2420
|
/** Format-validation regex sourced from providerConfig.ts's API_KEY_FORMATS, when one exists for this provider. */
|
|
2407
2421
|
apiKeyFormatPattern?: RegExp;
|
|
@@ -21,10 +21,13 @@ export declare function classifyProviderError(error: unknown, rules: ProviderErr
|
|
|
21
21
|
/**
|
|
22
22
|
* Generic fallback rule table covering the five categories every
|
|
23
23
|
* OpenAI-compatible provider already hand-rolled near-identically:
|
|
24
|
-
* auth (401), rate limit (429), model-not-found
|
|
25
|
-
* errors, and 5xx server errors.
|
|
26
|
-
*
|
|
27
|
-
*
|
|
24
|
+
* auth (401), rate limit (429), model-not-found, network/connection
|
|
25
|
+
* errors, and 5xx server errors. Model-not-found is a 404 whose text names
|
|
26
|
+
* the model or deployment as missing, or the old "model not found" message
|
|
27
|
+
* text at any status; any other 404 stays a plain `ProviderError` carrying
|
|
28
|
+
* the status and the vendor's message. Providers with a provider-specific
|
|
29
|
+
* auth message (naming the exact env var) prepend one override rule and
|
|
30
|
+
* spread this table after it — see errorClassifier usage in any migrated
|
|
28
31
|
* provider's formatProviderError for the pattern.
|
|
29
32
|
*/
|
|
30
33
|
export declare const DEFAULT_ERROR_RULES: ProviderErrorRule[];
|
|
@@ -113,13 +113,20 @@ export function classifyProviderError(error, rules, provider, modelName) {
|
|
|
113
113
|
const message = typeof rule.message === "function" ? rule.message(ctx) : rule.message;
|
|
114
114
|
return new rule.errorClass(message, provider);
|
|
115
115
|
}
|
|
116
|
+
// A 404 alone is a route answer (a wrong base URL gives the same reply): it only
|
|
117
|
+
// means "missing model" when the text names a model or deployment as absent.
|
|
118
|
+
// The gap is bounded rather than "no dot" because real model ids contain dots.
|
|
119
|
+
const MODEL_404_TEXT = /model[_ ]?not[_ ]?found|unknown model|no such model|invalid model|unsupported model|\b(?:model|deployment)\b.{0,120}\b(?:does not exist|not found|unavailable|not (?:available|supported))\b|\b(?:does not exist|not found)\b.{0,120}\b(?:model|deployment)\b|unable to access.{0,60}\bmodel\b/i;
|
|
116
120
|
/**
|
|
117
121
|
* Generic fallback rule table covering the five categories every
|
|
118
122
|
* OpenAI-compatible provider already hand-rolled near-identically:
|
|
119
|
-
* auth (401), rate limit (429), model-not-found
|
|
120
|
-
* errors, and 5xx server errors.
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
+
* auth (401), rate limit (429), model-not-found, network/connection
|
|
124
|
+
* errors, and 5xx server errors. Model-not-found is a 404 whose text names
|
|
125
|
+
* the model or deployment as missing, or the old "model not found" message
|
|
126
|
+
* text at any status; any other 404 stays a plain `ProviderError` carrying
|
|
127
|
+
* the status and the vendor's message. Providers with a provider-specific
|
|
128
|
+
* auth message (naming the exact env var) prepend one override rule and
|
|
129
|
+
* spread this table after it — see errorClassifier usage in any migrated
|
|
123
130
|
* provider's formatProviderError for the pattern.
|
|
124
131
|
*/
|
|
125
132
|
export const DEFAULT_ERROR_RULES = [
|
|
@@ -135,13 +142,18 @@ export const DEFAULT_ERROR_RULES = [
|
|
|
135
142
|
message: (ctx) => `${ctx.provider} rate limit exceeded. Please try again later.`,
|
|
136
143
|
},
|
|
137
144
|
{
|
|
138
|
-
match: (ctx) => ctx.
|
|
139
|
-
|
|
145
|
+
match: (ctx) => /model_not_found|model not found/i.test(ctx.message) ||
|
|
146
|
+
(ctx.statusCode === 404 && MODEL_404_TEXT.test(ctx.message)),
|
|
140
147
|
errorClass: InvalidModelError,
|
|
141
148
|
message: (ctx) => ctx.modelName
|
|
142
149
|
? `${ctx.provider} model '${ctx.modelName}' not found.`
|
|
143
150
|
: `${ctx.provider} model not found.`,
|
|
144
151
|
},
|
|
152
|
+
{
|
|
153
|
+
match: (ctx) => ctx.statusCode === 404,
|
|
154
|
+
errorClass: ProviderError,
|
|
155
|
+
message: (ctx) => `${ctx.provider} returned HTTP 404: ${ctx.message}`,
|
|
156
|
+
},
|
|
145
157
|
{
|
|
146
158
|
// Message regex covers providers/SDKs that surface a code as text
|
|
147
159
|
// (e.g. AWS SDK wrapping "ECONNRESET" into its own message). errorCode
|
|
@@ -2566,7 +2566,14 @@ mediaOptions = {}) {
|
|
|
2566
2566
|
});
|
|
2567
2567
|
try {
|
|
2568
2568
|
const extracted = await parser.getText();
|
|
2569
|
-
|
|
2569
|
+
// pdf-parse appends a "-- n of N --" marker after every page, so
|
|
2570
|
+
// extracted.text is never empty; a scan is detected from the
|
|
2571
|
+
// per-page text instead.
|
|
2572
|
+
const pageTexts = extracted?.pages?.map((page) => page.text) ?? [
|
|
2573
|
+
extracted?.text ?? "",
|
|
2574
|
+
];
|
|
2575
|
+
const hasTextLayer = pageTexts.some((text) => text.trim().length > 0);
|
|
2576
|
+
const pdfText = hasTextLayer ? (extracted?.text ?? "").trim() : "";
|
|
2570
2577
|
if (pdfText.length > 0) {
|
|
2571
2578
|
content.push({
|
|
2572
2579
|
type: "text",
|
|
@@ -2574,6 +2581,13 @@ mediaOptions = {}) {
|
|
|
2574
2581
|
});
|
|
2575
2582
|
logger.info(`[PDF→Text] ✅ Extracted text for non-vision provider ${provider}: ${name} (${pdfText.length} chars)`);
|
|
2576
2583
|
}
|
|
2584
|
+
else if (providerCanSeeImages) {
|
|
2585
|
+
content.push({
|
|
2586
|
+
type: "text",
|
|
2587
|
+
text: `\n[Attached PDF: ${name} — no extractable text layer (likely a scanned document); page images are attached below.]`,
|
|
2588
|
+
});
|
|
2589
|
+
logger.warn(`[PDF→Text] ${name} has no text layer; page images follow`);
|
|
2590
|
+
}
|
|
2577
2591
|
else {
|
|
2578
2592
|
content.push({
|
|
2579
2593
|
type: "text",
|
|
@@ -15,6 +15,13 @@ import type { CatalogProviderName, ModelChoice } from "../types/index.js";
|
|
|
15
15
|
*
|
|
16
16
|
* AUTO is also excluded — it never had an entry here either (matches
|
|
17
17
|
* pre-existing behavior: `getDefaultModel(AUTO)` returns `undefined`).
|
|
18
|
+
*
|
|
19
|
+
* This table is not what a call with no model uses. The runtime order is the
|
|
20
|
+
* explicit model, then the provider's environment variable, then the default in
|
|
21
|
+
* the dynamic model configuration, then the registry default
|
|
22
|
+
* (`providerRegistry.ts`); the OpenAI row mirrors that registry default. Only
|
|
23
|
+
* an `OpenAIProvider` constructed directly with none of those falls back to
|
|
24
|
+
* `gpt-5.4` (`openAI/client.ts`).
|
|
18
25
|
*/
|
|
19
26
|
export declare const DEFAULT_MODELS: Record<Exclude<AIProviderName, CatalogProviderName | AIProviderName.AUTO>, string>;
|
|
20
27
|
/**
|
|
@@ -56,6 +63,9 @@ export declare function getAllProviderChoices(): string[];
|
|
|
56
63
|
/**
|
|
57
64
|
* Get the default model for a provider
|
|
58
65
|
*
|
|
66
|
+
* Reads `DEFAULT_MODELS`, which mirrors the registry default rather than the
|
|
67
|
+
* runtime resolution order (see the note on that table).
|
|
68
|
+
*
|
|
59
69
|
* @param provider - The AI provider
|
|
60
70
|
* @returns Default model string for the provider
|
|
61
71
|
*/
|
|
@@ -43,14 +43,19 @@ function catalogTopModels(entry) {
|
|
|
43
43
|
*/
|
|
44
44
|
const TOP_MODELS_CONFIG = {
|
|
45
45
|
[AIProviderName.OPENAI]: [
|
|
46
|
+
{
|
|
47
|
+
model: OpenAIModels.GPT_5_4,
|
|
48
|
+
description: "Recommended - Direct OpenAI GPT-5.4 model",
|
|
49
|
+
},
|
|
50
|
+
{ model: OpenAIModels.GPT_5_4_MINI, description: "Cost-effective, fast" },
|
|
46
51
|
{
|
|
47
52
|
model: OpenAIModels.GPT_4O,
|
|
48
|
-
description: "
|
|
53
|
+
description: "Previous generation multimodal model",
|
|
49
54
|
},
|
|
50
55
|
{ model: OpenAIModels.GPT_4O_MINI, description: "Cost-effective, fast" },
|
|
51
56
|
{
|
|
52
57
|
model: OpenAIModels.GPT_5_2,
|
|
53
|
-
description: "
|
|
58
|
+
description: "Previous flagship with deep reasoning",
|
|
54
59
|
},
|
|
55
60
|
{ model: OpenAIModels.O3, description: "Advanced reasoning model" },
|
|
56
61
|
{ model: OpenAIModels.GPT_4_TURBO, description: "Previous generation" },
|
|
@@ -422,9 +427,16 @@ const TOP_MODELS_CONFIG = {
|
|
|
422
427
|
*
|
|
423
428
|
* AUTO is also excluded — it never had an entry here either (matches
|
|
424
429
|
* pre-existing behavior: `getDefaultModel(AUTO)` returns `undefined`).
|
|
430
|
+
*
|
|
431
|
+
* This table is not what a call with no model uses. The runtime order is the
|
|
432
|
+
* explicit model, then the provider's environment variable, then the default in
|
|
433
|
+
* the dynamic model configuration, then the registry default
|
|
434
|
+
* (`providerRegistry.ts`); the OpenAI row mirrors that registry default. Only
|
|
435
|
+
* an `OpenAIProvider` constructed directly with none of those falls back to
|
|
436
|
+
* `gpt-5.4` (`openAI/client.ts`).
|
|
425
437
|
*/
|
|
426
438
|
export const DEFAULT_MODELS = {
|
|
427
|
-
[AIProviderName.OPENAI]: OpenAIModels.
|
|
439
|
+
[AIProviderName.OPENAI]: OpenAIModels.GPT_4O_MINI,
|
|
428
440
|
[AIProviderName.ANTHROPIC]: AnthropicModels.CLAUDE_SONNET_4_5,
|
|
429
441
|
[AIProviderName.GOOGLE_AI]: GoogleAIModels.GEMINI_2_5_FLASH,
|
|
430
442
|
[AIProviderName.VERTEX]: VertexModels.GEMINI_2_5_FLASH,
|
|
@@ -572,6 +584,9 @@ export function getAllProviderChoices() {
|
|
|
572
584
|
/**
|
|
573
585
|
* Get the default model for a provider
|
|
574
586
|
*
|
|
587
|
+
* Reads `DEFAULT_MODELS`, which mirrors the registry default rather than the
|
|
588
|
+
* runtime resolution order (see the note on that table).
|
|
589
|
+
*
|
|
575
590
|
* @param provider - The AI provider
|
|
576
591
|
* @returns Default model string for the provider
|
|
577
592
|
*/
|
|
@@ -305,10 +305,17 @@ export class ProviderHealthChecker {
|
|
|
305
305
|
}
|
|
306
306
|
return;
|
|
307
307
|
}
|
|
308
|
+
// A descriptor with no apiKey variable at all (LM Studio, llama.cpp) would
|
|
309
|
+
// otherwise read process.env[""] below and report a missing key named by
|
|
310
|
+
// nothing; hasProviderEnvVars already treats these as usable with defaults.
|
|
311
|
+
const providerDescriptor = ProviderFactory.getDescriptor(providerName);
|
|
308
312
|
// Providers that don't use API keys directly
|
|
309
313
|
if (providerName === AIProviderName.OLLAMA ||
|
|
310
314
|
providerName === AIProviderName.BEDROCK ||
|
|
311
|
-
providerName === AIProviderName.LITELLM
|
|
315
|
+
providerName === AIProviderName.LITELLM ||
|
|
316
|
+
providerDescriptor?.localRuntime === true ||
|
|
317
|
+
(providerDescriptor?.envVars.optional === true &&
|
|
318
|
+
!providerDescriptor.envVars.apiKey)) {
|
|
312
319
|
healthStatus.hasApiKey = true;
|
|
313
320
|
return;
|
|
314
321
|
}
|
|
@@ -996,8 +1003,10 @@ export class ProviderHealthChecker {
|
|
|
996
1003
|
];
|
|
997
1004
|
case AIProviderName.OPENAI:
|
|
998
1005
|
return [
|
|
999
|
-
OpenAIModels.
|
|
1006
|
+
OpenAIModels.GPT_5_4,
|
|
1007
|
+
OpenAIModels.GPT_5_4_MINI,
|
|
1000
1008
|
OpenAIModels.GPT_4O_MINI,
|
|
1009
|
+
OpenAIModels.GPT_4O,
|
|
1001
1010
|
OpenAIModels.GPT_3_5_TURBO,
|
|
1002
1011
|
];
|
|
1003
1012
|
case AIProviderName.GOOGLE_AI:
|
|
@@ -1541,8 +1550,18 @@ export class ProviderHealthChecker {
|
|
|
1541
1550
|
return provider;
|
|
1542
1551
|
}
|
|
1543
1552
|
}
|
|
1544
|
-
// Fallback to first healthy provider
|
|
1545
|
-
|
|
1553
|
+
// Fallback to first healthy provider. A local runtime that nothing probes
|
|
1554
|
+
// (LM Studio, llama.cpp) is "healthy" only in that it needs no
|
|
1555
|
+
// configuration, which says nothing about whether it is running, so it
|
|
1556
|
+
// must not outrank a provider the caller actually configured.
|
|
1557
|
+
const firstHealthyProvider = healthStatuses.find((h) => {
|
|
1558
|
+
if (!h.isHealthy) {
|
|
1559
|
+
return false;
|
|
1560
|
+
}
|
|
1561
|
+
const descriptor = ProviderFactory.getDescriptor(h.provider);
|
|
1562
|
+
return !(descriptor?.localRuntime === true &&
|
|
1563
|
+
descriptor.healthCheck === "env-only");
|
|
1564
|
+
});
|
|
1546
1565
|
if (firstHealthyProvider) {
|
|
1547
1566
|
logger.info(`Using fallback healthy provider: ${firstHealthyProvider.provider}`);
|
|
1548
1567
|
return firstHealthyProvider.provider;
|
|
@@ -86,6 +86,10 @@ export declare function getErrorStatusCode(error: unknown): number | undefined;
|
|
|
86
86
|
* @param operation - The async operation to execute (should already use `maxRetries: 0`)
|
|
87
87
|
* @param span - The OTel span to annotate with retry events and attributes
|
|
88
88
|
* @param label - A human-readable label for log messages (e.g. "generateText", "streamText")
|
|
89
|
+
* @param sleep - Wait between attempts; receives the delay and the caller's abort signal
|
|
90
|
+
* @param abortSignal - The caller's cancellation. An abort during the wait ends the call with the
|
|
91
|
+
* signal's reason instead of waiting out the delay and running the operation
|
|
92
|
+
* again, and an already-aborted signal is checked before every attempt.
|
|
89
93
|
* @returns The result of the operation
|
|
90
94
|
*/
|
|
91
|
-
export declare function withProviderRetry<T>(operation: () => Promise<T>, span: Span | undefined, label: string, sleep?: (delayMs: number) => Promise<void
|
|
95
|
+
export declare function withProviderRetry<T>(operation: () => Promise<T>, span: Span | undefined, label: string, sleep?: (delayMs: number, abortSignal?: AbortSignal) => Promise<void>, abortSignal?: AbortSignal): Promise<T>;
|
|
@@ -33,7 +33,21 @@ export const NO_HINT_FLOOR_MS = 10_000;
|
|
|
33
33
|
* get a prompt rate-limit error rather than a silent multi-minute stall.
|
|
34
34
|
*/
|
|
35
35
|
export const MAX_RETRY_AFTER_MS = 60_000;
|
|
36
|
-
const sleepWithTimeout = (delayMs) => new Promise((resolve) =>
|
|
36
|
+
const sleepWithTimeout = (delayMs, abortSignal) => new Promise((resolve, reject) => {
|
|
37
|
+
if (abortSignal?.aborted) {
|
|
38
|
+
reject(abortSignal.reason);
|
|
39
|
+
return;
|
|
40
|
+
}
|
|
41
|
+
const onAbort = () => {
|
|
42
|
+
clearTimeout(timer);
|
|
43
|
+
reject(abortSignal?.reason);
|
|
44
|
+
};
|
|
45
|
+
const timer = setTimeout(() => {
|
|
46
|
+
abortSignal?.removeEventListener("abort", onAbort);
|
|
47
|
+
resolve();
|
|
48
|
+
}, delayMs);
|
|
49
|
+
abortSignal?.addEventListener("abort", onAbort, { once: true });
|
|
50
|
+
});
|
|
37
51
|
/**
|
|
38
52
|
* Check whether an error thrown by the AI SDK is retryable.
|
|
39
53
|
*
|
|
@@ -244,10 +258,15 @@ function getRetryAfterMs(error) {
|
|
|
244
258
|
* @param operation - The async operation to execute (should already use `maxRetries: 0`)
|
|
245
259
|
* @param span - The OTel span to annotate with retry events and attributes
|
|
246
260
|
* @param label - A human-readable label for log messages (e.g. "generateText", "streamText")
|
|
261
|
+
* @param sleep - Wait between attempts; receives the delay and the caller's abort signal
|
|
262
|
+
* @param abortSignal - The caller's cancellation. An abort during the wait ends the call with the
|
|
263
|
+
* signal's reason instead of waiting out the delay and running the operation
|
|
264
|
+
* again, and an already-aborted signal is checked before every attempt.
|
|
247
265
|
* @returns The result of the operation
|
|
248
266
|
*/
|
|
249
|
-
export async function withProviderRetry(operation, span, label, sleep = sleepWithTimeout) {
|
|
267
|
+
export async function withProviderRetry(operation, span, label, sleep = sleepWithTimeout, abortSignal) {
|
|
250
268
|
for (let attempt = 0; attempt <= MAX_PROVIDER_RETRIES; attempt++) {
|
|
269
|
+
abortSignal?.throwIfAborted();
|
|
251
270
|
try {
|
|
252
271
|
const result = await operation();
|
|
253
272
|
// Record how many attempts it took on the span
|
|
@@ -314,7 +333,10 @@ export async function withProviderRetry(operation, span, label, sleep = sleepWit
|
|
|
314
333
|
statusCode,
|
|
315
334
|
error: errorMessage,
|
|
316
335
|
});
|
|
317
|
-
await sleep(delay);
|
|
336
|
+
await sleep(delay, abortSignal);
|
|
337
|
+
// A custom sleep may ignore the signal, and a signal can abort in the
|
|
338
|
+
// same tick the timer fires.
|
|
339
|
+
abortSignal?.throwIfAborted();
|
|
318
340
|
}
|
|
319
341
|
}
|
|
320
342
|
// This should never be reached due to the throw inside the loop,
|
|
@@ -52,16 +52,8 @@ export async function getBestProvider(requestedProvider) {
|
|
|
52
52
|
// Fall through to cloud providers
|
|
53
53
|
}
|
|
54
54
|
}
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
* - LiteLLM and Ollama are prioritized first for local/self-hosted deployments,
|
|
58
|
-
* avoiding unnecessary dependence on external providers during fallback scenarios.
|
|
59
|
-
* - Vertex (Google Cloud AI) follows for enterprise-grade reliability.
|
|
60
|
-
* - Google AI follows as second cloud priority for comprehensive Google AI ecosystem support.
|
|
61
|
-
* - OpenAI maintains high priority due to its consistent reliability and broad model support.
|
|
62
|
-
* - Other providers are ordered based on a combination of reliability, feature set, and historical performance.
|
|
63
|
-
* Please update this comment if the order is changed in the future, and document the rationale for maintainability.
|
|
64
|
-
*/
|
|
55
|
+
// Order comes from ProviderDescriptor.autoSelectPriority (lower = tried
|
|
56
|
+
// first); see providerDescriptors.ts and the catalog JSON.
|
|
65
57
|
const providers = PROVIDER_DESCRIPTORS.filter((d) => d.autoSelectPriority !== undefined)
|
|
66
58
|
.sort((a, b) => (a.autoSelectPriority ?? 0) - (b.autoSelectPriority ?? 0))
|
|
67
59
|
.map((d) => d.name);
|