@juspay/neurolink 12.47.3 → 12.47.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -2
- package/dist/browser/neurolink.min.js +393 -392
- package/dist/core/loopEngine.js +1 -1
- package/dist/localUsage/copilotCliReader.js +1 -1
- package/dist/localUsage/cursorReader.js +1 -1
- package/dist/localUsage/hermesReader.js +1 -1
- package/dist/localUsage/openCodeReader.js +6 -5
- package/dist/neurolink.d.ts +13 -1
- package/dist/neurolink.js +141 -99
- package/dist/processors/media/VideoProcessor.js +7 -2
- package/dist/providers/amazonSagemaker.js +1 -1
- package/dist/providers/anthropic/client.js +1 -1
- package/dist/providers/nvidiaNim/client.js +3 -1
- package/dist/providers/openaiChatCompletionsBase.js +31 -12
- package/dist/proxy/proxyConfig.js +11 -0
- package/dist/types/model.d.ts +7 -6
- package/dist/types/providers.d.ts +8 -1
- package/dist/utils/errorClassifier.d.ts +12 -6
- package/dist/utils/errorClassifier.js +23 -8
- package/dist/utils/messageBuilder.js +15 -1
- package/dist/utils/modelChoices.d.ts +10 -0
- package/dist/utils/modelChoices.js +18 -3
- package/dist/utils/providerHealth.js +23 -4
- package/dist/utils/providerRetry.d.ts +5 -1
- package/dist/utils/providerRetry.js +25 -3
- package/dist/utils/providerUtils.js +2 -10
- package/docs-site/static/search-index.json +8 -6
- package/package.json +2 -1
package/dist/core/loopEngine.js
CHANGED
|
@@ -366,7 +366,7 @@ export function runAgenticLoop(adapter, initialConversation, options) {
|
|
|
366
366
|
// The caller's span, when it passes one. withProviderRetry writes
|
|
367
367
|
// gen_ai.provider.total_attempts here, so a loop that threaded a
|
|
368
368
|
// span before it moved onto this engine keeps emitting it.
|
|
369
|
-
options.span, `${adapter.providerLabel}.step
|
|
369
|
+
options.span, `${adapter.providerLabel}.step`, undefined, internalAbort.signal);
|
|
370
370
|
}
|
|
371
371
|
catch (err) {
|
|
372
372
|
throw err instanceof PostEmissionStepError ? err.cause : err;
|
|
@@ -125,7 +125,7 @@ export async function createCopilotCliReader() {
|
|
|
125
125
|
errors.push({
|
|
126
126
|
cliId: CLI_ID,
|
|
127
127
|
filePath: dbPath,
|
|
128
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
128
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
129
129
|
});
|
|
130
130
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
131
131
|
}
|
|
@@ -375,7 +375,7 @@ export async function createCursorReader() {
|
|
|
375
375
|
errors.push({
|
|
376
376
|
cliId: CLI_ID,
|
|
377
377
|
filePath: chatsRoot(),
|
|
378
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
378
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
379
379
|
});
|
|
380
380
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
381
381
|
}
|
|
@@ -318,7 +318,7 @@ export async function createHermesReader() {
|
|
|
318
318
|
errors.push({
|
|
319
319
|
cliId: CLI_ID,
|
|
320
320
|
filePath: hermesHome(),
|
|
321
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
321
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
322
322
|
});
|
|
323
323
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
324
324
|
}
|
|
@@ -80,10 +80,11 @@ export async function createOpenCodeReader() {
|
|
|
80
80
|
const errors = [];
|
|
81
81
|
const models = new Set();
|
|
82
82
|
const dbPath = databasePath();
|
|
83
|
-
// `node:sqlite`
|
|
84
|
-
//
|
|
85
|
-
//
|
|
86
|
-
//
|
|
83
|
+
// `node:sqlite` arrived in Node 22.5.0 behind `--experimental-sqlite`, was
|
|
84
|
+
// unflagged in 22.13.0 and is still marked experimental, so it can be
|
|
85
|
+
// absent or change shape. Imported lazily and behind a try/catch: a
|
|
86
|
+
// runtime without it must degrade to a reported failure for this one
|
|
87
|
+
// reader, not take down a scan of all the others.
|
|
87
88
|
let DatabaseSync;
|
|
88
89
|
try {
|
|
89
90
|
const sqlite = await import("node:sqlite");
|
|
@@ -102,7 +103,7 @@ export async function createOpenCodeReader() {
|
|
|
102
103
|
errors.push({
|
|
103
104
|
cliId: CLI_ID,
|
|
104
105
|
filePath: dbPath,
|
|
105
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
106
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
106
107
|
});
|
|
107
108
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
108
109
|
}
|
package/dist/neurolink.d.ts
CHANGED
|
@@ -129,7 +129,7 @@ export declare class NeuroLink {
|
|
|
129
129
|
* request dedup — BZ-664's actual goal — is untouched: the first
|
|
130
130
|
* occurrence in a request may still be served from cache. Request-scoped
|
|
131
131
|
* like _disableToolCacheForCurrentRequest above (assigned a fresh Set at
|
|
132
|
-
* request start so
|
|
132
|
+
* request start so `preservingTurnState`'s save/restore-by-reference works).
|
|
133
133
|
*/
|
|
134
134
|
private _toolCacheKeysServedThisRequest;
|
|
135
135
|
/** True only while a generate()/stream() turn is executing — the
|
|
@@ -1094,6 +1094,18 @@ export declare class NeuroLink {
|
|
|
1094
1094
|
*/
|
|
1095
1095
|
private streamWithIterationFallback;
|
|
1096
1096
|
private executeStreamRequest;
|
|
1097
|
+
/**
|
|
1098
|
+
* Runs an internal call that re-enters the public generate() (the tool-routing
|
|
1099
|
+
* router, the classifier router) without ending the outer turn's tool-cache
|
|
1100
|
+
* state. generate()'s own `finally` resets these fields, so the outer turn
|
|
1101
|
+
* would otherwise lose its repeat-call cache bypass for every later tool
|
|
1102
|
+
* call. Restored by reference: the nested call assigns a new Set rather than
|
|
1103
|
+
* mutating the outer one.
|
|
1104
|
+
*
|
|
1105
|
+
* Covers one turn re-entering itself. Two concurrent turns on one instance
|
|
1106
|
+
* still share these fields.
|
|
1107
|
+
*/
|
|
1108
|
+
private preservingTurnState;
|
|
1097
1109
|
/**
|
|
1098
1110
|
* Pre-call tool routing for both stream() and generate() turns: runs the
|
|
1099
1111
|
* router LLM once per turn and appends the unpicked servers' registered tool
|
package/dist/neurolink.js
CHANGED
|
@@ -433,7 +433,7 @@ export class NeuroLink {
|
|
|
433
433
|
* request dedup — BZ-664's actual goal — is untouched: the first
|
|
434
434
|
* occurrence in a request may still be served from cache. Request-scoped
|
|
435
435
|
* like _disableToolCacheForCurrentRequest above (assigned a fresh Set at
|
|
436
|
-
* request start so
|
|
436
|
+
* request start so `preservingTurnState`'s save/restore-by-reference works).
|
|
437
437
|
*/
|
|
438
438
|
_toolCacheKeysServedThisRequest = new Set();
|
|
439
439
|
/** True only while a generate()/stream() turn is executing — the
|
|
@@ -994,9 +994,9 @@ export class NeuroLink {
|
|
|
994
994
|
// generate() (marked so it never recursively re-routes). Fails open.
|
|
995
995
|
this.classifierRouter = config?.classifierRouter?.enabled
|
|
996
996
|
? new ClassifierRouter(config.classifierRouter, {
|
|
997
|
-
generate: (genOptions) => this.generate({
|
|
997
|
+
generate: (genOptions) => this.preservingTurnState(() => this.generate({
|
|
998
998
|
...genOptions,
|
|
999
|
-
}),
|
|
999
|
+
})),
|
|
1000
1000
|
// Fail-open by construction: tryDecide returns null rather than
|
|
1001
1001
|
// throwing, so an absent or broken decision provider leaves routing
|
|
1002
1002
|
// exactly as it was.
|
|
@@ -7265,6 +7265,30 @@ Current user's request: ${currentInput}`;
|
|
|
7265
7265
|
throw error;
|
|
7266
7266
|
}
|
|
7267
7267
|
}
|
|
7268
|
+
/**
|
|
7269
|
+
* Runs an internal call that re-enters the public generate() (the tool-routing
|
|
7270
|
+
* router, the classifier router) without ending the outer turn's tool-cache
|
|
7271
|
+
* state. generate()'s own `finally` resets these fields, so the outer turn
|
|
7272
|
+
* would otherwise lose its repeat-call cache bypass for every later tool
|
|
7273
|
+
* call. Restored by reference: the nested call assigns a new Set rather than
|
|
7274
|
+
* mutating the outer one.
|
|
7275
|
+
*
|
|
7276
|
+
* Covers one turn re-entering itself. Two concurrent turns on one instance
|
|
7277
|
+
* still share these fields.
|
|
7278
|
+
*/
|
|
7279
|
+
async preservingTurnState(run) {
|
|
7280
|
+
const disableToolCache = this._disableToolCacheForCurrentRequest;
|
|
7281
|
+
const keysServed = this._toolCacheKeysServedThisRequest;
|
|
7282
|
+
const turnActive = this._generationTurnActive;
|
|
7283
|
+
try {
|
|
7284
|
+
return await run();
|
|
7285
|
+
}
|
|
7286
|
+
finally {
|
|
7287
|
+
this._disableToolCacheForCurrentRequest = disableToolCache;
|
|
7288
|
+
this._toolCacheKeysServedThisRequest = keysServed;
|
|
7289
|
+
this._generationTurnActive = turnActive;
|
|
7290
|
+
}
|
|
7291
|
+
}
|
|
7268
7292
|
/**
|
|
7269
7293
|
* Pre-call tool routing for both stream() and generate() turns: runs the
|
|
7270
7294
|
* router LLM once per turn and appends the unpicked servers' registered tool
|
|
@@ -7420,104 +7444,92 @@ Current user's request: ${currentInput}`;
|
|
|
7420
7444
|
return;
|
|
7421
7445
|
}
|
|
7422
7446
|
}
|
|
7423
|
-
// The router call below re-enters the public generate(), whose finally
|
|
7424
|
-
// block resets _disableToolCacheForCurrentRequest to false. That flag is
|
|
7425
|
-
// turn-scoped (set at the top of this turn) and read by the main tool
|
|
7426
|
-
// execution path that runs after routing, so save it before the router
|
|
7427
|
-
// call and restore it afterward to keep the turn's cache setting intact.
|
|
7428
|
-
const cacheDisabledForCurrentRequest = this._disableToolCacheForCurrentRequest;
|
|
7429
7447
|
let routedExcludeTools;
|
|
7430
7448
|
let resolvedDecision;
|
|
7431
|
-
|
|
7432
|
-
|
|
7433
|
-
|
|
7434
|
-
|
|
7435
|
-
|
|
7436
|
-
|
|
7437
|
-
|
|
7438
|
-
|
|
7439
|
-
|
|
7440
|
-
|
|
7441
|
-
|
|
7442
|
-
|
|
7443
|
-
|
|
7444
|
-
|
|
7445
|
-
|
|
7446
|
-
|
|
7447
|
-
|
|
7448
|
-
|
|
7449
|
-
|
|
7450
|
-
|
|
7451
|
-
|
|
7452
|
-
|
|
7453
|
-
|
|
7454
|
-
|
|
7455
|
-
|
|
7456
|
-
|
|
7457
|
-
|
|
7458
|
-
|
|
7459
|
-
|
|
7460
|
-
|
|
7461
|
-
|
|
7462
|
-
|
|
7463
|
-
|
|
7464
|
-
|
|
7465
|
-
this.toolRoutingVectorCache = new Map();
|
|
7466
|
-
}
|
|
7449
|
+
// Intercept the decision so we can store it in the cache.
|
|
7450
|
+
const captureDecision = (decision) => {
|
|
7451
|
+
resolvedDecision = decision;
|
|
7452
|
+
emitDecision(decision);
|
|
7453
|
+
};
|
|
7454
|
+
// --- ITEM B: build the embedFn for the L2 embedding fast-path ---
|
|
7455
|
+
// The vector cache is persisted at the NeuroLink instance level so
|
|
7456
|
+
// tool embedding vectors are computed once and reused across turns
|
|
7457
|
+
// (Finding 1 fix). It is cleared by setToolRoutingServers() when the
|
|
7458
|
+
// catalog changes so stale vectors are never used after an update.
|
|
7459
|
+
let routingEmbedFn;
|
|
7460
|
+
const embeddingCfg = routingConfig.embedding;
|
|
7461
|
+
if (embeddingCfg?.enabled === true) {
|
|
7462
|
+
try {
|
|
7463
|
+
// Resolve the embedding provider: use the explicitly configured one
|
|
7464
|
+
// if present, otherwise fall back to the stream/generate call's
|
|
7465
|
+
// provider. The factory call is wrapped in try/catch so a provider
|
|
7466
|
+
// that doesn't support embedMany (it throws at call time, not
|
|
7467
|
+
// construction time) fails open when routingEmbedFn is invoked.
|
|
7468
|
+
const embProviderName = embeddingCfg.provider ??
|
|
7469
|
+
(options.provider && options.provider !== "auto"
|
|
7470
|
+
? options.provider
|
|
7471
|
+
: undefined) ??
|
|
7472
|
+
routingConfig.routerModel?.provider;
|
|
7473
|
+
if (embProviderName) {
|
|
7474
|
+
const embProvider = await AIProviderFactory.createProvider(embProviderName, embeddingCfg.model, true, this, undefined, this.resolveCredentials(options.credentials));
|
|
7475
|
+
// Bind embedMany with the configured model (may be undefined —
|
|
7476
|
+
// the provider uses its default embedding model in that case).
|
|
7477
|
+
routingEmbedFn = (texts) => withTimeout(embProvider.embedMany(texts, embeddingCfg.model), embeddingCfg.timeoutMs ?? 10000);
|
|
7478
|
+
// Lazy-init the persistent vector cache for this instance.
|
|
7479
|
+
// Subsequent turns reuse the same Map so text→vector lookups
|
|
7480
|
+
// already populated from earlier turns are served from memory.
|
|
7481
|
+
if (!this.toolRoutingVectorCache) {
|
|
7482
|
+
this.toolRoutingVectorCache = new Map();
|
|
7467
7483
|
}
|
|
7468
7484
|
}
|
|
7469
|
-
catch (embSetupError) {
|
|
7470
|
-
logger.debug("[ToolRouting] Embedding provider setup failed, L2 path disabled for this turn", {
|
|
7471
|
-
error: embSetupError instanceof Error
|
|
7472
|
-
? embSetupError.message
|
|
7473
|
-
: String(embSetupError),
|
|
7474
|
-
});
|
|
7475
|
-
// routingEmbedFn remains undefined — fast-path is skipped.
|
|
7476
|
-
}
|
|
7477
7485
|
}
|
|
7478
|
-
|
|
7479
|
-
|
|
7480
|
-
|
|
7481
|
-
|
|
7482
|
-
|
|
7483
|
-
|
|
7484
|
-
|
|
7485
|
-
|
|
7486
|
-
|
|
7487
|
-
|
|
7488
|
-
|
|
7489
|
-
|
|
7490
|
-
|
|
7491
|
-
|
|
7492
|
-
|
|
7493
|
-
|
|
7494
|
-
|
|
7495
|
-
|
|
7496
|
-
|
|
7497
|
-
|
|
7498
|
-
|
|
7499
|
-
|
|
7500
|
-
|
|
7501
|
-
|
|
7502
|
-
|
|
7503
|
-
|
|
7504
|
-
|
|
7505
|
-
|
|
7506
|
-
|
|
7507
|
-
|
|
7508
|
-
|
|
7509
|
-
|
|
7510
|
-
|
|
7511
|
-
|
|
7512
|
-
|
|
7513
|
-
|
|
7514
|
-
|
|
7515
|
-
|
|
7516
|
-
|
|
7517
|
-
|
|
7518
|
-
|
|
7519
|
-
|
|
7520
|
-
|
|
7486
|
+
catch (embSetupError) {
|
|
7487
|
+
logger.debug("[ToolRouting] Embedding provider setup failed, L2 path disabled for this turn", {
|
|
7488
|
+
error: embSetupError instanceof Error
|
|
7489
|
+
? embSetupError.message
|
|
7490
|
+
: String(embSetupError),
|
|
7491
|
+
});
|
|
7492
|
+
// routingEmbedFn remains undefined — fast-path is skipped.
|
|
7493
|
+
}
|
|
7494
|
+
}
|
|
7495
|
+
routedExcludeTools = await resolveToolRoutingExclusions({
|
|
7496
|
+
catalog,
|
|
7497
|
+
alwaysIncludeServerIds: routingConfig.alwaysIncludeServerIds ?? [],
|
|
7498
|
+
userQuery: routingQuery,
|
|
7499
|
+
routerPromptPrefix: routingConfig.routerPromptPrefix,
|
|
7500
|
+
routerModel: {
|
|
7501
|
+
provider: routingConfig.routerModel?.provider ??
|
|
7502
|
+
options.provider,
|
|
7503
|
+
model: routingConfig.routerModel?.model ?? options.model,
|
|
7504
|
+
region: routingConfig.routerModel?.region ?? options.region,
|
|
7505
|
+
temperature: routingConfig.routerModel?.temperature,
|
|
7506
|
+
},
|
|
7507
|
+
timeoutMs: routingConfig.timeoutMs ?? DEFAULT_TOOL_ROUTING_TIMEOUT_MS,
|
|
7508
|
+
// Forward the abort signal so a cancelled turn aborts the router
|
|
7509
|
+
// call promptly instead of waiting out the routing timeout.
|
|
7510
|
+
generateFn: (generateOptions) => this.preservingTurnState(() => this.generate({
|
|
7511
|
+
...generateOptions,
|
|
7512
|
+
abortSignal: options.abortSignal,
|
|
7513
|
+
})),
|
|
7514
|
+
// Calibrated per-server routing when a decision provider is
|
|
7515
|
+
// configured. tryDecide returns null without one, so the resolver
|
|
7516
|
+
// falls straight through to the generative router as before.
|
|
7517
|
+
decideFn: (decisionOptions) => this.tryDecide({
|
|
7518
|
+
...decisionOptions,
|
|
7519
|
+
signal: options.abortSignal,
|
|
7520
|
+
}),
|
|
7521
|
+
decisionMinDropConfidence: routingConfig.minDropConfidence,
|
|
7522
|
+
emitDecision: captureDecision,
|
|
7523
|
+
// L2 / ITEM D — only populated when embedding is configured.
|
|
7524
|
+
embedFn: routingEmbedFn,
|
|
7525
|
+
embeddingConfig: embeddingCfg,
|
|
7526
|
+
granularity: routingConfig.granularity ?? "server",
|
|
7527
|
+
// Pass the persistent vector cache so tool embeddings are reused
|
|
7528
|
+
// across turns (Finding 1).
|
|
7529
|
+
embeddingVectorCache: routingEmbedFn !== undefined
|
|
7530
|
+
? this.toolRoutingVectorCache
|
|
7531
|
+
: undefined,
|
|
7532
|
+
});
|
|
7521
7533
|
// Aborted during the router call — skip applying now-stale exclusions;
|
|
7522
7534
|
// the main generation path enforces the abort itself.
|
|
7523
7535
|
if (options.abortSignal?.aborted) {
|
|
@@ -7986,6 +7998,20 @@ Current user's request: ${currentInput}`;
|
|
|
7986
7998
|
yield* incrementalFallback;
|
|
7987
7999
|
}
|
|
7988
8000
|
ttsResolver?.(streamedTTSResult);
|
|
8001
|
+
// `streamState.finishReason` starts as a "stop" placeholder, before a
|
|
8002
|
+
// single chunk exists. A provider that records how the turn really
|
|
8003
|
+
// ended in metadata (a token-limit cut, a content filter, a step cap
|
|
8004
|
+
// that left tool calls pending) has that adopted here, once the drain
|
|
8005
|
+
// is done. Only the three unified values count: a raw vendor string
|
|
8006
|
+
// or "stop" leaves the graded reason alone, and a fallback's own
|
|
8007
|
+
// value is never overwritten.
|
|
8008
|
+
const drainedFinishReason = mcpStreamOutcome.metadata?.finishReason;
|
|
8009
|
+
if (!metadata.fallbackAttempted &&
|
|
8010
|
+
(drainedFinishReason === "length" ||
|
|
8011
|
+
drainedFinishReason === "tool-calls" ||
|
|
8012
|
+
drainedFinishReason === "content-filter")) {
|
|
8013
|
+
streamState.finishReason = drainedFinishReason;
|
|
8014
|
+
}
|
|
7989
8015
|
resolvedUsage = mcpStreamOutcome.usage;
|
|
7990
8016
|
if (!resolvedUsage && mcpStreamOutcome.analytics) {
|
|
7991
8017
|
try {
|
|
@@ -8176,8 +8202,19 @@ Current user's request: ${currentInput}`;
|
|
|
8176
8202
|
}
|
|
8177
8203
|
})();
|
|
8178
8204
|
const streamResult = await this.processStreamResult(processedStream, enhancedOptions, factoryResult);
|
|
8179
|
-
|
|
8205
|
+
streamState.finishReason =
|
|
8180
8206
|
streamState.finishReason || streamResult.finishReason;
|
|
8207
|
+
// Live, like toolCalls/toolResults below: the reason a stream ends with is
|
|
8208
|
+
// only known after it drains, so a plain copy would stay the creation-time
|
|
8209
|
+
// placeholder for every truncated or capped turn.
|
|
8210
|
+
Object.defineProperty(streamResult, "finishReason", {
|
|
8211
|
+
enumerable: true,
|
|
8212
|
+
configurable: true,
|
|
8213
|
+
get: () => streamState.finishReason,
|
|
8214
|
+
set: (value) => {
|
|
8215
|
+
streamState.finishReason = value;
|
|
8216
|
+
},
|
|
8217
|
+
});
|
|
8181
8218
|
// #1819 / E1b: a top-level cross-provider fallback (handleStreamFallback,
|
|
8182
8219
|
// inside `processedStream` above) only reassigns
|
|
8183
8220
|
// `streamState.toolCalls`/`toolResults` once the caller starts draining
|
|
@@ -9207,13 +9244,18 @@ Current user's request: ${currentInput}`;
|
|
|
9207
9244
|
if (toolResultsDescriptor) {
|
|
9208
9245
|
Object.defineProperty(response, "toolResults", toolResultsDescriptor);
|
|
9209
9246
|
}
|
|
9247
|
+
const finishReasonDescriptor = Object.getOwnPropertyDescriptor(streamResult, "finishReason");
|
|
9248
|
+
if (finishReasonDescriptor) {
|
|
9249
|
+
Object.defineProperty(response, "finishReason", finishReasonDescriptor);
|
|
9250
|
+
}
|
|
9210
9251
|
if (!source) {
|
|
9211
9252
|
return response;
|
|
9212
9253
|
}
|
|
9213
9254
|
// NeuroLink grades the turn's terminal state itself (a normalized
|
|
9214
9255
|
// finishReason, stopReason for aborts and time limits), so those fields
|
|
9215
|
-
//
|
|
9216
|
-
//
|
|
9256
|
+
// keep NeuroLink's own value here (finishReason through the descriptor
|
|
9257
|
+
// copied above). Copying the provider's getter for them would report the
|
|
9258
|
+
// raw vendor reason, e.g. Anthropic's end_turn, instead.
|
|
9217
9259
|
//
|
|
9218
9260
|
// toolCalls/toolResults are excluded for a different reason: `source`
|
|
9219
9261
|
// (the PRIMARY provider's own result) may itself define them as live
|
|
@@ -445,7 +445,10 @@ export class VideoProcessor extends BaseFileProcessor {
|
|
|
445
445
|
metadata = this.buildMetadata(probeResult.data, buffer.length);
|
|
446
446
|
}
|
|
447
447
|
}
|
|
448
|
-
|
|
448
|
+
// mediabunny reports a duration of 0 for a clip it can open but not
|
|
449
|
+
// time, and ffprobe's "N/A" parses to NaN: neither is a missing
|
|
450
|
+
// result, so both take the ffmpeg fallback.
|
|
451
|
+
if (!metadata || !(metadata.duration > 0)) {
|
|
449
452
|
// ffmpeg-static ships ffmpeg only, so a host that relies on it has
|
|
450
453
|
// no ffprobe. Without a duration no frame timestamps can be chosen.
|
|
451
454
|
const ffmpegProbe = await this.probeVideoWithFfmpeg(tempVideoPath);
|
|
@@ -456,7 +459,9 @@ export class VideoProcessor extends BaseFileProcessor {
|
|
|
456
459
|
// Nothing downstream reports this: an empty duration selects no
|
|
457
460
|
// frames, the request still succeeds, and the model is simply
|
|
458
461
|
// told nothing about the video.
|
|
459
|
-
logger.warn(
|
|
462
|
+
logger.warn(metadata
|
|
463
|
+
? `[NEUROLINK] No positive duration could be read for ${filename} (the first reader gave none and ffmpeg could not supply one), so frame times cannot be chosen: ${ffmpegProbe.error}`
|
|
464
|
+
: `[NEUROLINK] No metadata could be read for ${filename} (mediabunny, ffprobe and ffmpeg all failed), so no keyframes will be extracted: ${ffmpegProbe.error}`);
|
|
460
465
|
}
|
|
461
466
|
}
|
|
462
467
|
if (!metadata) {
|
|
@@ -242,7 +242,7 @@ export class AmazonSageMakerProvider extends BaseProvider {
|
|
|
242
242
|
...(options.toolTimeoutMs !== undefined
|
|
243
243
|
? { toolTimeoutMs: options.toolTimeoutMs }
|
|
244
244
|
: {}),
|
|
245
|
-
runStep: (call) => withProviderRetry(call, undefined, "sagemaker generate").catch((err) => {
|
|
245
|
+
runStep: (call) => withProviderRetry(call, undefined, "sagemaker generate", undefined, options.abortSignal).catch((err) => {
|
|
246
246
|
throw this.handleProviderError(err);
|
|
247
247
|
}),
|
|
248
248
|
}, toolExecutionSummaries);
|
|
@@ -1623,7 +1623,7 @@ export class AnthropicProvider extends BaseProvider {
|
|
|
1623
1623
|
...(options.toolTimeoutMs !== undefined
|
|
1624
1624
|
? { toolTimeoutMs: options.toolTimeoutMs }
|
|
1625
1625
|
: {}),
|
|
1626
|
-
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, "anthropic generate").catch((err) => {
|
|
1626
|
+
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, "anthropic generate", undefined, options.abortSignal).catch((err) => {
|
|
1627
1627
|
throw this.handleProviderError(err);
|
|
1628
1628
|
}),
|
|
1629
1629
|
}, toolExecutionSummaries);
|
|
@@ -270,7 +270,9 @@ export class NvidiaNimProvider extends OpenAIChatCompletionsProvider {
|
|
|
270
270
|
message: "NVIDIA NIM rate limit exceeded",
|
|
271
271
|
},
|
|
272
272
|
{
|
|
273
|
-
|
|
273
|
+
// NIM answers most of its roster with a 404 whose text ("Function …
|
|
274
|
+
// not found for account …") names no model, so the status decides.
|
|
275
|
+
match: (ctx) => ctx.statusCode === 404 || /404|model_not_found/.test(ctx.message),
|
|
274
276
|
errorClass: InvalidModelError,
|
|
275
277
|
message: () => `NVIDIA NIM model '${this.modelName}' not available. Browse the catalog at https://build.nvidia.com/models`,
|
|
276
278
|
},
|
|
@@ -99,6 +99,28 @@ const resolveAgainstSchema = (text, schema) => {
|
|
|
99
99
|
const scalar = recoverScalarRoot(text, schema);
|
|
100
100
|
return scalar.kind === "accepted" ? scalar.value : undefined;
|
|
101
101
|
};
|
|
102
|
+
/**
|
|
103
|
+
* The one place a finish reason becomes the unified spelling. Accepts the wire's
|
|
104
|
+
* `finish_reason` (`tool_calls`, `content_filter`) and the unified spelling a
|
|
105
|
+
* middleware's own finish part already carries (`tool-calls`, `content-filter`),
|
|
106
|
+
* since both reach the stream path. Anything else, including `stop`, `error` and
|
|
107
|
+
* an absent value, reads as `stop`.
|
|
108
|
+
*/
|
|
109
|
+
const toUnifiedFinishReason = (reason) => {
|
|
110
|
+
switch (reason) {
|
|
111
|
+
case "length":
|
|
112
|
+
return "length";
|
|
113
|
+
case "tool_calls":
|
|
114
|
+
case "function_call":
|
|
115
|
+
case "tool-calls":
|
|
116
|
+
return "tool-calls";
|
|
117
|
+
case "content_filter":
|
|
118
|
+
case "content-filter":
|
|
119
|
+
return "content-filter";
|
|
120
|
+
default:
|
|
121
|
+
return "stop";
|
|
122
|
+
}
|
|
123
|
+
};
|
|
102
124
|
// Pull one native chunk at a time and forward cancellation to its iterator.
|
|
103
125
|
const chunksToV3Stream = (source, completion, cancel) => {
|
|
104
126
|
const iterator = source[Symbol.asyncIterator]();
|
|
@@ -774,13 +796,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
774
796
|
});
|
|
775
797
|
}
|
|
776
798
|
const rawFinish = choice?.finish_reason;
|
|
777
|
-
const unified = rawFinish
|
|
778
|
-
? "length"
|
|
779
|
-
: rawFinish === "tool_calls" || rawFinish === "function_call"
|
|
780
|
-
? "tool-calls"
|
|
781
|
-
: rawFinish === "content_filter"
|
|
782
|
-
? "content-filter"
|
|
783
|
-
: "stop";
|
|
799
|
+
const unified = toUnifiedFinishReason(rawFinish);
|
|
784
800
|
return {
|
|
785
801
|
content,
|
|
786
802
|
finishReason: { unified, raw: rawFinish ?? "stop" },
|
|
@@ -989,7 +1005,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
989
1005
|
// supplied around each call: without them a 429 surfaces as a raw
|
|
990
1006
|
// upstream string instead of a RateLimitError, and a throttle is
|
|
991
1007
|
// never retried.
|
|
992
|
-
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, `${this.providerName} generate
|
|
1008
|
+
runStep: (call) => withProviderRetry(call, trace.getActiveSpan() ?? undefined, `${this.providerName} generate`, undefined, options.abortSignal).catch((err) => {
|
|
993
1009
|
throw this.handleProviderError(err);
|
|
994
1010
|
}),
|
|
995
1011
|
}, toolExecutionSummaries);
|
|
@@ -1839,7 +1855,8 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1839
1855
|
if (!loopPromise) {
|
|
1840
1856
|
resolveFinish("stop");
|
|
1841
1857
|
}
|
|
1842
|
-
|
|
1858
|
+
const rawFinishReason = await finishPromise;
|
|
1859
|
+
streamMetadata.rawFinishReason = rawFinishReason;
|
|
1843
1860
|
// Structured output for a `stream({ schema })` turn. Runs HERE —
|
|
1844
1861
|
// after the stream is fully drained — so a tool-free re-ask (when the
|
|
1845
1862
|
// streamed answer isn't already schema-valid) never reaches
|
|
@@ -1884,9 +1901,11 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1884
1901
|
// Only reached once the structured-output re-ask above (when there
|
|
1885
1902
|
// was one) has resolved without throwing. A caller abort during that
|
|
1886
1903
|
// re-ask rejects `resolveStreamStructuredData` and skips this line
|
|
1887
|
-
// entirely, so `metadata.finishReason` never claims
|
|
1904
|
+
// entirely, so `metadata.finishReason` never claims a finish for a
|
|
1888
1905
|
// turn that actually failed after the wire stream itself finished.
|
|
1889
|
-
|
|
1906
|
+
// The reason is the turn's own (a token-limit cut reads "length"), not
|
|
1907
|
+
// a blanket "stop" for every stream that drained without throwing.
|
|
1908
|
+
streamMetadata.finishReason = toUnifiedFinishReason(rawFinishReason);
|
|
1890
1909
|
// No-output path: stream completed normally but yielded zero text.
|
|
1891
1910
|
// Build an enriched sentinel + stamp the active OTel span so
|
|
1892
1911
|
// Pipeline B (ContextEnricher) surfaces a WARNING-level Langfuse
|
|
@@ -2188,7 +2207,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
2188
2207
|
};
|
|
2189
2208
|
let res;
|
|
2190
2209
|
try {
|
|
2191
|
-
res = await withProviderRetry(doFetch, trace.getActiveSpan() ?? undefined, `${this.providerName} stream
|
|
2210
|
+
res = await withProviderRetry(doFetch, trace.getActiveSpan() ?? undefined, `${this.providerName} stream`, undefined, args.abortSignal);
|
|
2192
2211
|
}
|
|
2193
2212
|
catch (err) {
|
|
2194
2213
|
// The one-shot 400 context-overflow fallback lives outside
|
|
@@ -241,6 +241,10 @@ function kebabToCamelRoutingKey(kebabKey) {
|
|
|
241
241
|
* *because* one of the spellings was explicitly `null` — never merely
|
|
242
242
|
* because both were absent — so callers can warn once per null key without
|
|
243
243
|
* warning on ordinary omission.
|
|
244
|
+
*
|
|
245
|
+
* `account-allowlist` is the one key whose `null` `validateProxyConfig`
|
|
246
|
+
* rejects before this runs, so for it the `null` branch is only reachable
|
|
247
|
+
* through a direct `parseRoutingConfig` call.
|
|
244
248
|
*/
|
|
245
249
|
function readLegacyRoutingKey(routing, kebabKey) {
|
|
246
250
|
const camelKey = kebabToCamelRoutingKey(kebabKey);
|
|
@@ -313,6 +317,13 @@ export function validateProxyConfig(config) {
|
|
|
313
317
|
}
|
|
314
318
|
});
|
|
315
319
|
}
|
|
320
|
+
// A `null` allowlist must not read as "unset": a reload that loses a
|
|
321
|
+
// restriction this way would silently open the proxy to every account.
|
|
322
|
+
// Checked per spelling, before the legacy read coalesces them.
|
|
323
|
+
if (routing["account-allowlist"] === null ||
|
|
324
|
+
routing[kebabToCamelRoutingKey("account-allowlist")] === null) {
|
|
325
|
+
errors.push("routing.account-allowlist must be an array of non-empty strings (null is not allowed; omit the key to remove the restriction)");
|
|
326
|
+
}
|
|
316
327
|
const rawAccountAllowlist = readLegacyRoutingKey(routing, "account-allowlist").value;
|
|
317
328
|
if (rawAccountAllowlist !== undefined) {
|
|
318
329
|
if (!Array.isArray(rawAccountAllowlist)) {
|
package/dist/types/model.d.ts
CHANGED
|
@@ -272,9 +272,9 @@ export type ModelRoutingOptions = {
|
|
|
272
272
|
/**
|
|
273
273
|
* A single model's metadata inside a provider's manifest. This is the one
|
|
274
274
|
* canonical shape every model-metadata consumer (context windows, pricing,
|
|
275
|
-
* MODEL_REGISTRY, vision capability, output-token ceilings)
|
|
276
|
-
*
|
|
277
|
-
*
|
|
275
|
+
* MODEL_REGISTRY, vision capability, output-token ceilings) reads from —
|
|
276
|
+
* contextWindows.ts, pricing.ts, modelRegistry.ts, providerImageAdapter.ts and
|
|
277
|
+
* core/constants.ts.
|
|
278
278
|
*
|
|
279
279
|
* `pricingPerMTok` is optional by design: a model with no verified price
|
|
280
280
|
* (e.g. a just-announced model pricing.ts hasn't priced yet) must not report
|
|
@@ -311,9 +311,10 @@ export type ProviderModelManifestEntry = {
|
|
|
311
311
|
* forward verbatim for the ids that already had a MODEL_REGISTRY entry
|
|
312
312
|
* before this migration. Absent for every id that never had one — those
|
|
313
313
|
* get performance/useCases/category derived mechanically instead (see
|
|
314
|
-
*
|
|
315
|
-
* genuinely new model: mechanical derivation is the
|
|
316
|
-
* a fabricated "curated" value would be worse than an
|
|
314
|
+
* buildManifestDerivedEntries in src/lib/models/modelRegistry.ts). Never
|
|
315
|
+
* populate this for a genuinely new model: mechanical derivation is the
|
|
316
|
+
* correct default, and a fabricated "curated" value would be worse than an
|
|
317
|
+
* honestly-derived one.
|
|
317
318
|
*/
|
|
318
319
|
curated?: {
|
|
319
320
|
performance?: ModelPerformance;
|
|
@@ -2408,7 +2408,14 @@ export type ProviderDescriptor = {
|
|
|
2408
2408
|
* checks to its own server (TypeSafe). See {@link DecisionLimits}.
|
|
2409
2409
|
*/
|
|
2410
2410
|
decisionLimits?: DecisionLimits;
|
|
2411
|
-
/**
|
|
2411
|
+
/**
|
|
2412
|
+
* Ascending priority (1 = tried first) in the auto-select fallback chain
|
|
2413
|
+
* used by getBestProvider(). Undefined = not part of the auto-select chain.
|
|
2414
|
+
*
|
|
2415
|
+
* Priorities favor local and self-hosted deployments first to avoid an
|
|
2416
|
+
* external dependency during fallback, then cloud providers according to
|
|
2417
|
+
* reliability, feature set and model coverage.
|
|
2418
|
+
*/
|
|
2412
2419
|
autoSelectPriority?: number;
|
|
2413
2420
|
/** Format-validation regex sourced from providerConfig.ts's API_KEY_FORMATS, when one exists for this provider. */
|
|
2414
2421
|
apiKeyFormatPattern?: RegExp;
|