@juspay/neurolink 12.47.4 → 12.47.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -2
- package/dist/browser/neurolink.min.js +399 -399
- package/dist/localUsage/copilotCliReader.js +1 -1
- package/dist/localUsage/cursorReader.js +1 -1
- package/dist/localUsage/hermesReader.js +1 -1
- package/dist/localUsage/openCodeReader.js +6 -5
- package/dist/neurolink.d.ts +13 -1
- package/dist/neurolink.js +141 -99
- package/dist/providers/openaiChatCompletionsBase.js +29 -10
- package/dist/proxy/proxyConfig.js +11 -0
- package/dist/utils/errorClassifier.d.ts +5 -2
- package/dist/utils/errorClassifier.js +13 -3
- package/docs-site/static/search-index.json +4 -4
- package/package.json +1 -1
|
@@ -125,7 +125,7 @@ export async function createCopilotCliReader() {
|
|
|
125
125
|
errors.push({
|
|
126
126
|
cliId: CLI_ID,
|
|
127
127
|
filePath: dbPath,
|
|
128
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
128
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
129
129
|
});
|
|
130
130
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
131
131
|
}
|
|
@@ -375,7 +375,7 @@ export async function createCursorReader() {
|
|
|
375
375
|
errors.push({
|
|
376
376
|
cliId: CLI_ID,
|
|
377
377
|
filePath: chatsRoot(),
|
|
378
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
378
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
379
379
|
});
|
|
380
380
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
381
381
|
}
|
|
@@ -318,7 +318,7 @@ export async function createHermesReader() {
|
|
|
318
318
|
errors.push({
|
|
319
319
|
cliId: CLI_ID,
|
|
320
320
|
filePath: hermesHome(),
|
|
321
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
321
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
322
322
|
});
|
|
323
323
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
324
324
|
}
|
|
@@ -80,10 +80,11 @@ export async function createOpenCodeReader() {
|
|
|
80
80
|
const errors = [];
|
|
81
81
|
const models = new Set();
|
|
82
82
|
const dbPath = databasePath();
|
|
83
|
-
// `node:sqlite`
|
|
84
|
-
//
|
|
85
|
-
//
|
|
86
|
-
//
|
|
83
|
+
// `node:sqlite` arrived in Node 22.5.0 behind `--experimental-sqlite`, was
|
|
84
|
+
// unflagged in 22.13.0 and is still marked experimental, so it can be
|
|
85
|
+
// absent or change shape. Imported lazily and behind a try/catch: a
|
|
86
|
+
// runtime without it must degrade to a reported failure for this one
|
|
87
|
+
// reader, not take down a scan of all the others.
|
|
87
88
|
let DatabaseSync;
|
|
88
89
|
try {
|
|
89
90
|
const sqlite = await import("node:sqlite");
|
|
@@ -102,7 +103,7 @@ export async function createOpenCodeReader() {
|
|
|
102
103
|
errors.push({
|
|
103
104
|
cliId: CLI_ID,
|
|
104
105
|
filePath: dbPath,
|
|
105
|
-
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)}`,
|
|
106
|
+
message: `node:sqlite unavailable on this runtime: ${error instanceof Error ? error.message : String(error)} (needs Node >=22.13.0; on 22.5-22.12 pass --experimental-sqlite)`,
|
|
106
107
|
});
|
|
107
108
|
return { cliId: CLI_ID, totals, filesScanned: 0, errors };
|
|
108
109
|
}
|
package/dist/neurolink.d.ts
CHANGED
|
@@ -129,7 +129,7 @@ export declare class NeuroLink {
|
|
|
129
129
|
* request dedup — BZ-664's actual goal — is untouched: the first
|
|
130
130
|
* occurrence in a request may still be served from cache. Request-scoped
|
|
131
131
|
* like _disableToolCacheForCurrentRequest above (assigned a fresh Set at
|
|
132
|
-
* request start so
|
|
132
|
+
* request start so `preservingTurnState`'s save/restore-by-reference works).
|
|
133
133
|
*/
|
|
134
134
|
private _toolCacheKeysServedThisRequest;
|
|
135
135
|
/** True only while a generate()/stream() turn is executing — the
|
|
@@ -1094,6 +1094,18 @@ export declare class NeuroLink {
|
|
|
1094
1094
|
*/
|
|
1095
1095
|
private streamWithIterationFallback;
|
|
1096
1096
|
private executeStreamRequest;
|
|
1097
|
+
/**
|
|
1098
|
+
* Runs an internal call that re-enters the public generate() (the tool-routing
|
|
1099
|
+
* router, the classifier router) without ending the outer turn's tool-cache
|
|
1100
|
+
* state. generate()'s own `finally` resets these fields, so the outer turn
|
|
1101
|
+
* would otherwise lose its repeat-call cache bypass for every later tool
|
|
1102
|
+
* call. Restored by reference: the nested call assigns a new Set rather than
|
|
1103
|
+
* mutating the outer one.
|
|
1104
|
+
*
|
|
1105
|
+
* Covers one turn re-entering itself. Two concurrent turns on one instance
|
|
1106
|
+
* still share these fields.
|
|
1107
|
+
*/
|
|
1108
|
+
private preservingTurnState;
|
|
1097
1109
|
/**
|
|
1098
1110
|
* Pre-call tool routing for both stream() and generate() turns: runs the
|
|
1099
1111
|
* router LLM once per turn and appends the unpicked servers' registered tool
|
package/dist/neurolink.js
CHANGED
|
@@ -433,7 +433,7 @@ export class NeuroLink {
|
|
|
433
433
|
* request dedup — BZ-664's actual goal — is untouched: the first
|
|
434
434
|
* occurrence in a request may still be served from cache. Request-scoped
|
|
435
435
|
* like _disableToolCacheForCurrentRequest above (assigned a fresh Set at
|
|
436
|
-
* request start so
|
|
436
|
+
* request start so `preservingTurnState`'s save/restore-by-reference works).
|
|
437
437
|
*/
|
|
438
438
|
_toolCacheKeysServedThisRequest = new Set();
|
|
439
439
|
/** True only while a generate()/stream() turn is executing — the
|
|
@@ -994,9 +994,9 @@ export class NeuroLink {
|
|
|
994
994
|
// generate() (marked so it never recursively re-routes). Fails open.
|
|
995
995
|
this.classifierRouter = config?.classifierRouter?.enabled
|
|
996
996
|
? new ClassifierRouter(config.classifierRouter, {
|
|
997
|
-
generate: (genOptions) => this.generate({
|
|
997
|
+
generate: (genOptions) => this.preservingTurnState(() => this.generate({
|
|
998
998
|
...genOptions,
|
|
999
|
-
}),
|
|
999
|
+
})),
|
|
1000
1000
|
// Fail-open by construction: tryDecide returns null rather than
|
|
1001
1001
|
// throwing, so an absent or broken decision provider leaves routing
|
|
1002
1002
|
// exactly as it was.
|
|
@@ -7265,6 +7265,30 @@ Current user's request: ${currentInput}`;
|
|
|
7265
7265
|
throw error;
|
|
7266
7266
|
}
|
|
7267
7267
|
}
|
|
7268
|
+
/**
|
|
7269
|
+
* Runs an internal call that re-enters the public generate() (the tool-routing
|
|
7270
|
+
* router, the classifier router) without ending the outer turn's tool-cache
|
|
7271
|
+
* state. generate()'s own `finally` resets these fields, so the outer turn
|
|
7272
|
+
* would otherwise lose its repeat-call cache bypass for every later tool
|
|
7273
|
+
* call. Restored by reference: the nested call assigns a new Set rather than
|
|
7274
|
+
* mutating the outer one.
|
|
7275
|
+
*
|
|
7276
|
+
* Covers one turn re-entering itself. Two concurrent turns on one instance
|
|
7277
|
+
* still share these fields.
|
|
7278
|
+
*/
|
|
7279
|
+
async preservingTurnState(run) {
|
|
7280
|
+
const disableToolCache = this._disableToolCacheForCurrentRequest;
|
|
7281
|
+
const keysServed = this._toolCacheKeysServedThisRequest;
|
|
7282
|
+
const turnActive = this._generationTurnActive;
|
|
7283
|
+
try {
|
|
7284
|
+
return await run();
|
|
7285
|
+
}
|
|
7286
|
+
finally {
|
|
7287
|
+
this._disableToolCacheForCurrentRequest = disableToolCache;
|
|
7288
|
+
this._toolCacheKeysServedThisRequest = keysServed;
|
|
7289
|
+
this._generationTurnActive = turnActive;
|
|
7290
|
+
}
|
|
7291
|
+
}
|
|
7268
7292
|
/**
|
|
7269
7293
|
* Pre-call tool routing for both stream() and generate() turns: runs the
|
|
7270
7294
|
* router LLM once per turn and appends the unpicked servers' registered tool
|
|
@@ -7420,104 +7444,92 @@ Current user's request: ${currentInput}`;
|
|
|
7420
7444
|
return;
|
|
7421
7445
|
}
|
|
7422
7446
|
}
|
|
7423
|
-
// The router call below re-enters the public generate(), whose finally
|
|
7424
|
-
// block resets _disableToolCacheForCurrentRequest to false. That flag is
|
|
7425
|
-
// turn-scoped (set at the top of this turn) and read by the main tool
|
|
7426
|
-
// execution path that runs after routing, so save it before the router
|
|
7427
|
-
// call and restore it afterward to keep the turn's cache setting intact.
|
|
7428
|
-
const cacheDisabledForCurrentRequest = this._disableToolCacheForCurrentRequest;
|
|
7429
7447
|
let routedExcludeTools;
|
|
7430
7448
|
let resolvedDecision;
|
|
7431
|
-
|
|
7432
|
-
|
|
7433
|
-
|
|
7434
|
-
|
|
7435
|
-
|
|
7436
|
-
|
|
7437
|
-
|
|
7438
|
-
|
|
7439
|
-
|
|
7440
|
-
|
|
7441
|
-
|
|
7442
|
-
|
|
7443
|
-
|
|
7444
|
-
|
|
7445
|
-
|
|
7446
|
-
|
|
7447
|
-
|
|
7448
|
-
|
|
7449
|
-
|
|
7450
|
-
|
|
7451
|
-
|
|
7452
|
-
|
|
7453
|
-
|
|
7454
|
-
|
|
7455
|
-
|
|
7456
|
-
|
|
7457
|
-
|
|
7458
|
-
|
|
7459
|
-
|
|
7460
|
-
|
|
7461
|
-
|
|
7462
|
-
|
|
7463
|
-
|
|
7464
|
-
|
|
7465
|
-
this.toolRoutingVectorCache = new Map();
|
|
7466
|
-
}
|
|
7449
|
+
// Intercept the decision so we can store it in the cache.
|
|
7450
|
+
const captureDecision = (decision) => {
|
|
7451
|
+
resolvedDecision = decision;
|
|
7452
|
+
emitDecision(decision);
|
|
7453
|
+
};
|
|
7454
|
+
// --- ITEM B: build the embedFn for the L2 embedding fast-path ---
|
|
7455
|
+
// The vector cache is persisted at the NeuroLink instance level so
|
|
7456
|
+
// tool embedding vectors are computed once and reused across turns
|
|
7457
|
+
// (Finding 1 fix). It is cleared by setToolRoutingServers() when the
|
|
7458
|
+
// catalog changes so stale vectors are never used after an update.
|
|
7459
|
+
let routingEmbedFn;
|
|
7460
|
+
const embeddingCfg = routingConfig.embedding;
|
|
7461
|
+
if (embeddingCfg?.enabled === true) {
|
|
7462
|
+
try {
|
|
7463
|
+
// Resolve the embedding provider: use the explicitly configured one
|
|
7464
|
+
// if present, otherwise fall back to the stream/generate call's
|
|
7465
|
+
// provider. The factory call is wrapped in try/catch so a provider
|
|
7466
|
+
// that doesn't support embedMany (it throws at call time, not
|
|
7467
|
+
// construction time) fails open when routingEmbedFn is invoked.
|
|
7468
|
+
const embProviderName = embeddingCfg.provider ??
|
|
7469
|
+
(options.provider && options.provider !== "auto"
|
|
7470
|
+
? options.provider
|
|
7471
|
+
: undefined) ??
|
|
7472
|
+
routingConfig.routerModel?.provider;
|
|
7473
|
+
if (embProviderName) {
|
|
7474
|
+
const embProvider = await AIProviderFactory.createProvider(embProviderName, embeddingCfg.model, true, this, undefined, this.resolveCredentials(options.credentials));
|
|
7475
|
+
// Bind embedMany with the configured model (may be undefined —
|
|
7476
|
+
// the provider uses its default embedding model in that case).
|
|
7477
|
+
routingEmbedFn = (texts) => withTimeout(embProvider.embedMany(texts, embeddingCfg.model), embeddingCfg.timeoutMs ?? 10000);
|
|
7478
|
+
// Lazy-init the persistent vector cache for this instance.
|
|
7479
|
+
// Subsequent turns reuse the same Map so text→vector lookups
|
|
7480
|
+
// already populated from earlier turns are served from memory.
|
|
7481
|
+
if (!this.toolRoutingVectorCache) {
|
|
7482
|
+
this.toolRoutingVectorCache = new Map();
|
|
7467
7483
|
}
|
|
7468
7484
|
}
|
|
7469
|
-
catch (embSetupError) {
|
|
7470
|
-
logger.debug("[ToolRouting] Embedding provider setup failed, L2 path disabled for this turn", {
|
|
7471
|
-
error: embSetupError instanceof Error
|
|
7472
|
-
? embSetupError.message
|
|
7473
|
-
: String(embSetupError),
|
|
7474
|
-
});
|
|
7475
|
-
// routingEmbedFn remains undefined — fast-path is skipped.
|
|
7476
|
-
}
|
|
7477
7485
|
}
|
|
7478
|
-
|
|
7479
|
-
|
|
7480
|
-
|
|
7481
|
-
|
|
7482
|
-
|
|
7483
|
-
|
|
7484
|
-
|
|
7485
|
-
|
|
7486
|
-
|
|
7487
|
-
|
|
7488
|
-
|
|
7489
|
-
|
|
7490
|
-
|
|
7491
|
-
|
|
7492
|
-
|
|
7493
|
-
|
|
7494
|
-
|
|
7495
|
-
|
|
7496
|
-
|
|
7497
|
-
|
|
7498
|
-
|
|
7499
|
-
|
|
7500
|
-
|
|
7501
|
-
|
|
7502
|
-
|
|
7503
|
-
|
|
7504
|
-
|
|
7505
|
-
|
|
7506
|
-
|
|
7507
|
-
|
|
7508
|
-
|
|
7509
|
-
|
|
7510
|
-
|
|
7511
|
-
|
|
7512
|
-
|
|
7513
|
-
|
|
7514
|
-
|
|
7515
|
-
|
|
7516
|
-
|
|
7517
|
-
|
|
7518
|
-
|
|
7519
|
-
|
|
7520
|
-
|
|
7486
|
+
catch (embSetupError) {
|
|
7487
|
+
logger.debug("[ToolRouting] Embedding provider setup failed, L2 path disabled for this turn", {
|
|
7488
|
+
error: embSetupError instanceof Error
|
|
7489
|
+
? embSetupError.message
|
|
7490
|
+
: String(embSetupError),
|
|
7491
|
+
});
|
|
7492
|
+
// routingEmbedFn remains undefined — fast-path is skipped.
|
|
7493
|
+
}
|
|
7494
|
+
}
|
|
7495
|
+
routedExcludeTools = await resolveToolRoutingExclusions({
|
|
7496
|
+
catalog,
|
|
7497
|
+
alwaysIncludeServerIds: routingConfig.alwaysIncludeServerIds ?? [],
|
|
7498
|
+
userQuery: routingQuery,
|
|
7499
|
+
routerPromptPrefix: routingConfig.routerPromptPrefix,
|
|
7500
|
+
routerModel: {
|
|
7501
|
+
provider: routingConfig.routerModel?.provider ??
|
|
7502
|
+
options.provider,
|
|
7503
|
+
model: routingConfig.routerModel?.model ?? options.model,
|
|
7504
|
+
region: routingConfig.routerModel?.region ?? options.region,
|
|
7505
|
+
temperature: routingConfig.routerModel?.temperature,
|
|
7506
|
+
},
|
|
7507
|
+
timeoutMs: routingConfig.timeoutMs ?? DEFAULT_TOOL_ROUTING_TIMEOUT_MS,
|
|
7508
|
+
// Forward the abort signal so a cancelled turn aborts the router
|
|
7509
|
+
// call promptly instead of waiting out the routing timeout.
|
|
7510
|
+
generateFn: (generateOptions) => this.preservingTurnState(() => this.generate({
|
|
7511
|
+
...generateOptions,
|
|
7512
|
+
abortSignal: options.abortSignal,
|
|
7513
|
+
})),
|
|
7514
|
+
// Calibrated per-server routing when a decision provider is
|
|
7515
|
+
// configured. tryDecide returns null without one, so the resolver
|
|
7516
|
+
// falls straight through to the generative router as before.
|
|
7517
|
+
decideFn: (decisionOptions) => this.tryDecide({
|
|
7518
|
+
...decisionOptions,
|
|
7519
|
+
signal: options.abortSignal,
|
|
7520
|
+
}),
|
|
7521
|
+
decisionMinDropConfidence: routingConfig.minDropConfidence,
|
|
7522
|
+
emitDecision: captureDecision,
|
|
7523
|
+
// L2 / ITEM D — only populated when embedding is configured.
|
|
7524
|
+
embedFn: routingEmbedFn,
|
|
7525
|
+
embeddingConfig: embeddingCfg,
|
|
7526
|
+
granularity: routingConfig.granularity ?? "server",
|
|
7527
|
+
// Pass the persistent vector cache so tool embeddings are reused
|
|
7528
|
+
// across turns (Finding 1).
|
|
7529
|
+
embeddingVectorCache: routingEmbedFn !== undefined
|
|
7530
|
+
? this.toolRoutingVectorCache
|
|
7531
|
+
: undefined,
|
|
7532
|
+
});
|
|
7521
7533
|
// Aborted during the router call — skip applying now-stale exclusions;
|
|
7522
7534
|
// the main generation path enforces the abort itself.
|
|
7523
7535
|
if (options.abortSignal?.aborted) {
|
|
@@ -7986,6 +7998,20 @@ Current user's request: ${currentInput}`;
|
|
|
7986
7998
|
yield* incrementalFallback;
|
|
7987
7999
|
}
|
|
7988
8000
|
ttsResolver?.(streamedTTSResult);
|
|
8001
|
+
// `streamState.finishReason` starts as a "stop" placeholder, before a
|
|
8002
|
+
// single chunk exists. A provider that records how the turn really
|
|
8003
|
+
// ended in metadata (a token-limit cut, a content filter, a step cap
|
|
8004
|
+
// that left tool calls pending) has that adopted here, once the drain
|
|
8005
|
+
// is done. Only the three unified values count: a raw vendor string
|
|
8006
|
+
// or "stop" leaves the graded reason alone, and a fallback's own
|
|
8007
|
+
// value is never overwritten.
|
|
8008
|
+
const drainedFinishReason = mcpStreamOutcome.metadata?.finishReason;
|
|
8009
|
+
if (!metadata.fallbackAttempted &&
|
|
8010
|
+
(drainedFinishReason === "length" ||
|
|
8011
|
+
drainedFinishReason === "tool-calls" ||
|
|
8012
|
+
drainedFinishReason === "content-filter")) {
|
|
8013
|
+
streamState.finishReason = drainedFinishReason;
|
|
8014
|
+
}
|
|
7989
8015
|
resolvedUsage = mcpStreamOutcome.usage;
|
|
7990
8016
|
if (!resolvedUsage && mcpStreamOutcome.analytics) {
|
|
7991
8017
|
try {
|
|
@@ -8176,8 +8202,19 @@ Current user's request: ${currentInput}`;
|
|
|
8176
8202
|
}
|
|
8177
8203
|
})();
|
|
8178
8204
|
const streamResult = await this.processStreamResult(processedStream, enhancedOptions, factoryResult);
|
|
8179
|
-
|
|
8205
|
+
streamState.finishReason =
|
|
8180
8206
|
streamState.finishReason || streamResult.finishReason;
|
|
8207
|
+
// Live, like toolCalls/toolResults below: the reason a stream ends with is
|
|
8208
|
+
// only known after it drains, so a plain copy would stay the creation-time
|
|
8209
|
+
// placeholder for every truncated or capped turn.
|
|
8210
|
+
Object.defineProperty(streamResult, "finishReason", {
|
|
8211
|
+
enumerable: true,
|
|
8212
|
+
configurable: true,
|
|
8213
|
+
get: () => streamState.finishReason,
|
|
8214
|
+
set: (value) => {
|
|
8215
|
+
streamState.finishReason = value;
|
|
8216
|
+
},
|
|
8217
|
+
});
|
|
8181
8218
|
// #1819 / E1b: a top-level cross-provider fallback (handleStreamFallback,
|
|
8182
8219
|
// inside `processedStream` above) only reassigns
|
|
8183
8220
|
// `streamState.toolCalls`/`toolResults` once the caller starts draining
|
|
@@ -9207,13 +9244,18 @@ Current user's request: ${currentInput}`;
|
|
|
9207
9244
|
if (toolResultsDescriptor) {
|
|
9208
9245
|
Object.defineProperty(response, "toolResults", toolResultsDescriptor);
|
|
9209
9246
|
}
|
|
9247
|
+
const finishReasonDescriptor = Object.getOwnPropertyDescriptor(streamResult, "finishReason");
|
|
9248
|
+
if (finishReasonDescriptor) {
|
|
9249
|
+
Object.defineProperty(response, "finishReason", finishReasonDescriptor);
|
|
9250
|
+
}
|
|
9210
9251
|
if (!source) {
|
|
9211
9252
|
return response;
|
|
9212
9253
|
}
|
|
9213
9254
|
// NeuroLink grades the turn's terminal state itself (a normalized
|
|
9214
9255
|
// finishReason, stopReason for aborts and time limits), so those fields
|
|
9215
|
-
//
|
|
9216
|
-
//
|
|
9256
|
+
// keep NeuroLink's own value here (finishReason through the descriptor
|
|
9257
|
+
// copied above). Copying the provider's getter for them would report the
|
|
9258
|
+
// raw vendor reason, e.g. Anthropic's end_turn, instead.
|
|
9217
9259
|
//
|
|
9218
9260
|
// toolCalls/toolResults are excluded for a different reason: `source`
|
|
9219
9261
|
// (the PRIMARY provider's own result) may itself define them as live
|
|
@@ -99,6 +99,28 @@ const resolveAgainstSchema = (text, schema) => {
|
|
|
99
99
|
const scalar = recoverScalarRoot(text, schema);
|
|
100
100
|
return scalar.kind === "accepted" ? scalar.value : undefined;
|
|
101
101
|
};
|
|
102
|
+
/**
|
|
103
|
+
* The one place a finish reason becomes the unified spelling. Accepts the wire's
|
|
104
|
+
* `finish_reason` (`tool_calls`, `content_filter`) and the unified spelling a
|
|
105
|
+
* middleware's own finish part already carries (`tool-calls`, `content-filter`),
|
|
106
|
+
* since both reach the stream path. Anything else, including `stop`, `error` and
|
|
107
|
+
* an absent value, reads as `stop`.
|
|
108
|
+
*/
|
|
109
|
+
const toUnifiedFinishReason = (reason) => {
|
|
110
|
+
switch (reason) {
|
|
111
|
+
case "length":
|
|
112
|
+
return "length";
|
|
113
|
+
case "tool_calls":
|
|
114
|
+
case "function_call":
|
|
115
|
+
case "tool-calls":
|
|
116
|
+
return "tool-calls";
|
|
117
|
+
case "content_filter":
|
|
118
|
+
case "content-filter":
|
|
119
|
+
return "content-filter";
|
|
120
|
+
default:
|
|
121
|
+
return "stop";
|
|
122
|
+
}
|
|
123
|
+
};
|
|
102
124
|
// Pull one native chunk at a time and forward cancellation to its iterator.
|
|
103
125
|
const chunksToV3Stream = (source, completion, cancel) => {
|
|
104
126
|
const iterator = source[Symbol.asyncIterator]();
|
|
@@ -774,13 +796,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
774
796
|
});
|
|
775
797
|
}
|
|
776
798
|
const rawFinish = choice?.finish_reason;
|
|
777
|
-
const unified = rawFinish
|
|
778
|
-
? "length"
|
|
779
|
-
: rawFinish === "tool_calls" || rawFinish === "function_call"
|
|
780
|
-
? "tool-calls"
|
|
781
|
-
: rawFinish === "content_filter"
|
|
782
|
-
? "content-filter"
|
|
783
|
-
: "stop";
|
|
799
|
+
const unified = toUnifiedFinishReason(rawFinish);
|
|
784
800
|
return {
|
|
785
801
|
content,
|
|
786
802
|
finishReason: { unified, raw: rawFinish ?? "stop" },
|
|
@@ -1839,7 +1855,8 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1839
1855
|
if (!loopPromise) {
|
|
1840
1856
|
resolveFinish("stop");
|
|
1841
1857
|
}
|
|
1842
|
-
|
|
1858
|
+
const rawFinishReason = await finishPromise;
|
|
1859
|
+
streamMetadata.rawFinishReason = rawFinishReason;
|
|
1843
1860
|
// Structured output for a `stream({ schema })` turn. Runs HERE —
|
|
1844
1861
|
// after the stream is fully drained — so a tool-free re-ask (when the
|
|
1845
1862
|
// streamed answer isn't already schema-valid) never reaches
|
|
@@ -1884,9 +1901,11 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1884
1901
|
// Only reached once the structured-output re-ask above (when there
|
|
1885
1902
|
// was one) has resolved without throwing. A caller abort during that
|
|
1886
1903
|
// re-ask rejects `resolveStreamStructuredData` and skips this line
|
|
1887
|
-
// entirely, so `metadata.finishReason` never claims
|
|
1904
|
+
// entirely, so `metadata.finishReason` never claims a finish for a
|
|
1888
1905
|
// turn that actually failed after the wire stream itself finished.
|
|
1889
|
-
|
|
1906
|
+
// The reason is the turn's own (a token-limit cut reads "length"), not
|
|
1907
|
+
// a blanket "stop" for every stream that drained without throwing.
|
|
1908
|
+
streamMetadata.finishReason = toUnifiedFinishReason(rawFinishReason);
|
|
1890
1909
|
// No-output path: stream completed normally but yielded zero text.
|
|
1891
1910
|
// Build an enriched sentinel + stamp the active OTel span so
|
|
1892
1911
|
// Pipeline B (ContextEnricher) surfaces a WARNING-level Langfuse
|
|
@@ -241,6 +241,10 @@ function kebabToCamelRoutingKey(kebabKey) {
|
|
|
241
241
|
* *because* one of the spellings was explicitly `null` — never merely
|
|
242
242
|
* because both were absent — so callers can warn once per null key without
|
|
243
243
|
* warning on ordinary omission.
|
|
244
|
+
*
|
|
245
|
+
* `account-allowlist` is the one key whose `null` `validateProxyConfig`
|
|
246
|
+
* rejects before this runs, so for it the `null` branch is only reachable
|
|
247
|
+
* through a direct `parseRoutingConfig` call.
|
|
244
248
|
*/
|
|
245
249
|
function readLegacyRoutingKey(routing, kebabKey) {
|
|
246
250
|
const camelKey = kebabToCamelRoutingKey(kebabKey);
|
|
@@ -313,6 +317,13 @@ export function validateProxyConfig(config) {
|
|
|
313
317
|
}
|
|
314
318
|
});
|
|
315
319
|
}
|
|
320
|
+
// A `null` allowlist must not read as "unset": a reload that loses a
|
|
321
|
+
// restriction this way would silently open the proxy to every account.
|
|
322
|
+
// Checked per spelling, before the legacy read coalesces them.
|
|
323
|
+
if (routing["account-allowlist"] === null ||
|
|
324
|
+
routing[kebabToCamelRoutingKey("account-allowlist")] === null) {
|
|
325
|
+
errors.push("routing.account-allowlist must be an array of non-empty strings (null is not allowed; omit the key to remove the restriction)");
|
|
326
|
+
}
|
|
316
327
|
const rawAccountAllowlist = readLegacyRoutingKey(routing, "account-allowlist").value;
|
|
317
328
|
if (rawAccountAllowlist !== undefined) {
|
|
318
329
|
if (!Array.isArray(rawAccountAllowlist)) {
|
|
@@ -1,7 +1,10 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Shared provider-error classification.
|
|
3
|
-
* `formatProviderError(error)`
|
|
2
|
+
* Shared provider-error classification. Migrated providers'
|
|
3
|
+
* `formatProviderError(error)` delegate here instead of hand-rolling their
|
|
4
4
|
* own TimeoutError-check → .includes()-chain → `new XError(...)` ladder.
|
|
5
|
+
* Not yet migrated (they do not call it): Google AI Studio, SageMaker, the
|
|
6
|
+
* media and embedding providers (Ideogram, Recraft, Stability, Replicate,
|
|
7
|
+
* Jina, Voyage) and the System One decision provider.
|
|
5
8
|
*
|
|
6
9
|
* `classifyProviderError` picks the Error subclass + message; it does NOT
|
|
7
10
|
* stamp statusCode/isRetryable/retryAfterMs onto the result — that
|
|
@@ -1,7 +1,10 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Shared provider-error classification.
|
|
3
|
-
* `formatProviderError(error)`
|
|
2
|
+
* Shared provider-error classification. Migrated providers'
|
|
3
|
+
* `formatProviderError(error)` delegate here instead of hand-rolling their
|
|
4
4
|
* own TimeoutError-check → .includes()-chain → `new XError(...)` ladder.
|
|
5
|
+
* Not yet migrated (they do not call it): Google AI Studio, SageMaker, the
|
|
6
|
+
* media and embedding providers (Ideogram, Recraft, Stability, Replicate,
|
|
7
|
+
* Jina, Voyage) and the System One decision provider.
|
|
5
8
|
*
|
|
6
9
|
* `classifyProviderError` picks the Error subclass + message; it does NOT
|
|
7
10
|
* stamp statusCode/isRetryable/retryAfterMs onto the result — that
|
|
@@ -116,7 +119,14 @@ export function classifyProviderError(error, rules, provider, modelName) {
|
|
|
116
119
|
// A 404 alone is a route answer (a wrong base URL gives the same reply): it only
|
|
117
120
|
// means "missing model" when the text names a model or deployment as absent.
|
|
118
121
|
// The gap is bounded rather than "no dot" because real model ids contain dots.
|
|
119
|
-
|
|
122
|
+
// "model" or "deployment" followed by a route noun ("model gateway route") is a
|
|
123
|
+
// modifier of that route, not the subject of the 404, in either word order, and
|
|
124
|
+
// that holds for the named phrases ("unknown model gateway route") as well.
|
|
125
|
+
const NOT_A_ROUTE_MODIFIER = "(?![\\s-]+(?:route|gateway|endpoint|url|path|proxy|server|service|api|host)s?(?![\\w-]))";
|
|
126
|
+
const MODEL_WORD = `\\bmodel\\b${NOT_A_ROUTE_MODIFIER}`;
|
|
127
|
+
const MODEL_OR_DEPLOYMENT = `\\b(?:model|deployment)\\b${NOT_A_ROUTE_MODIFIER}`;
|
|
128
|
+
const NAMED_MODEL_ERROR = `(?:unknown|no such|invalid|unsupported) model${NOT_A_ROUTE_MODIFIER}`;
|
|
129
|
+
const MODEL_404_TEXT = new RegExp(`model[_ ]?not[_ ]?found|${NAMED_MODEL_ERROR}|${MODEL_OR_DEPLOYMENT}.{0,120}\\b(?:does not exist|not found|unavailable|not (?:available|supported))\\b|\\b(?:does not exist|not found)\\b.{0,120}${MODEL_OR_DEPLOYMENT}|unable to access.{0,60}${MODEL_WORD}`, "i");
|
|
120
130
|
/**
|
|
121
131
|
* Generic fallback rule table covering the five categories every
|
|
122
132
|
* OpenAI-compatible provider already hand-rolled near-identically:
|