@juspay/neurolink 12.46.0 → 12.47.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -2
- package/README.md +75 -50
- package/dist/browser/neurolink.min.js +426 -426
- package/dist/cli/commands/decide.js +2 -2
- package/dist/cli/commands/setup.js +2 -1
- package/dist/cli/factories/commandFactory.js +1 -1
- package/dist/constants/enums.d.ts +16 -0
- package/dist/constants/enums.js +17 -0
- package/dist/factories/providerDescriptors.js +84 -5
- package/dist/factories/providerRegistry.js +10 -1
- package/dist/models/manifestRegistry.js +2 -0
- package/dist/models/manifests/cloudflareClef.d.ts +16 -0
- package/dist/models/manifests/cloudflareClef.js +42 -0
- package/dist/providers/anthropic/client.js +6 -1
- package/dist/providers/catalog/huggingface.json +2 -1
- package/dist/providers/catalog/loader.js +5 -1
- package/dist/providers/catalog/schema.d.ts +1 -0
- package/dist/providers/catalog/schema.js +8 -0
- package/dist/providers/cloudflareClef.d.ts +52 -0
- package/dist/providers/cloudflareClef.js +331 -0
- package/dist/providers/ideogram.js +9 -30
- package/dist/providers/llamaCpp.js +2 -1
- package/dist/providers/openAI/client.js +4 -5
- package/dist/providers/openaiChatCompletionsBase.js +12 -7
- package/dist/providers/openaiChatCompletionsClient.js +25 -7
- package/dist/providers/recraft.js +9 -28
- package/dist/providers/systemOneDecision.d.ts +12 -1
- package/dist/providers/systemOneDecision.js +70 -18
- package/dist/types/decision.d.ts +22 -0
- package/dist/types/providerCatalog.d.ts +2 -0
- package/dist/types/providers.d.ts +15 -0
- package/dist/utils/modelChoices.js +13 -1
- package/dist/utils/pricing.js +12 -0
- package/dist/utils/providerConfig.d.ts +7 -0
- package/dist/utils/providerConfig.js +19 -0
- package/dist/utils/providerRetry.d.ts +5 -0
- package/dist/utils/providerRetry.js +18 -12
- package/docs-site/static/search-index.json +580 -557
- package/package.json +2 -1
|
@@ -6,7 +6,7 @@ import { ProviderError } from "../types/index.js";
|
|
|
6
6
|
import { prepareDecisionMedia } from "../utils/decisionMedia.js";
|
|
7
7
|
import { logger } from "../utils/logger.js";
|
|
8
8
|
import { redactUrlsInText } from "../utils/logSanitize.js";
|
|
9
|
-
import { estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
|
|
9
|
+
import { CHARS_PER_TOKEN, estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
|
|
10
10
|
/**
|
|
11
11
|
* Generous enough for a cold start (measured at 2.0–2.7s after idle on Jev)
|
|
12
12
|
* while still bounded. Every internal caller is fail-open, so this only bites
|
|
@@ -177,15 +177,52 @@ const sleep = (ms, signal) => new Promise((resolve, reject) => {
|
|
|
177
177
|
* tokenizer, so when a provider declares a rate, non-ASCII characters are
|
|
178
178
|
* counted at it instead. ASCII text is estimated exactly as before.
|
|
179
179
|
*/
|
|
180
|
-
function estimateDecisionStateTokens(text, nonAsciiTokensPerChar) {
|
|
181
|
-
|
|
180
|
+
function estimateDecisionStateTokens(text, nonAsciiTokensPerChar, rates = {}) {
|
|
181
|
+
const { digit, symbol, astral } = rates;
|
|
182
|
+
if (nonAsciiTokensPerChar === undefined &&
|
|
183
|
+
digit === undefined &&
|
|
184
|
+
symbol === undefined &&
|
|
185
|
+
astral === undefined) {
|
|
182
186
|
return estimateTokens(text);
|
|
183
187
|
}
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
188
|
+
// Each class of character is counted once, at its own rate. A class with no
|
|
189
|
+
// declared rate stays in the ordinary four-characters-a-token estimate (or,
|
|
190
|
+
// for non-ASCII, at that estimate's per-character rate), so a provider that
|
|
191
|
+
// declares none of the extra rates is estimated exactly as before.
|
|
192
|
+
const isDigit = (c) => c >= "0" && c <= "9";
|
|
193
|
+
const isSymbol = (c) => /[!-/:-@[-`{-~]/.test(c);
|
|
194
|
+
let digits = 0;
|
|
195
|
+
let symbols = 0;
|
|
196
|
+
let astrals = 0;
|
|
197
|
+
let nonAscii = 0;
|
|
198
|
+
const rest = [];
|
|
199
|
+
for (const c of text) {
|
|
200
|
+
const code = c.codePointAt(0) ?? 0;
|
|
201
|
+
if (code > 0xffff && astral !== undefined) {
|
|
202
|
+
astrals += 1;
|
|
203
|
+
}
|
|
204
|
+
else if (code > 0x7f) {
|
|
205
|
+
nonAscii += 1;
|
|
206
|
+
}
|
|
207
|
+
else if (digit !== undefined && isDigit(c)) {
|
|
208
|
+
digits += 1;
|
|
209
|
+
}
|
|
210
|
+
else if (symbol !== undefined && isSymbol(c)) {
|
|
211
|
+
symbols += 1;
|
|
212
|
+
}
|
|
213
|
+
else {
|
|
214
|
+
rest.push(c);
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
// Digits, punctuation and astral characters (emoji) are charged separately
|
|
218
|
+
// only when the provider declares a rate for them: a tokenizer that reads each
|
|
219
|
+
// digit, and most punctuation, on its own makes a number-heavy or JSON-heavy
|
|
220
|
+
// state several times longer than four characters a token suggests.
|
|
221
|
+
return (estimateTokens(rest.join("")) +
|
|
222
|
+
Math.ceil(digits * (digit ?? 0)) +
|
|
223
|
+
Math.ceil(symbols * (symbol ?? 0)) +
|
|
224
|
+
Math.ceil(astrals * (astral ?? 0)) +
|
|
225
|
+
Math.ceil(nonAscii * (nonAsciiTokensPerChar ?? 1 / CHARS_PER_TOKEN)));
|
|
189
226
|
}
|
|
190
227
|
/**
|
|
191
228
|
* The shared half of every "System One" decision provider — a model that takes
|
|
@@ -251,6 +288,15 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
251
288
|
defaultTimeoutMs(_questionCount) {
|
|
252
289
|
return this.getDescriptorDecideMs();
|
|
253
290
|
}
|
|
291
|
+
/**
|
|
292
|
+
* The part of a successful response that holds `answers`, `usage` and
|
|
293
|
+
* `model`. The default is the response itself; a vendor that wraps every
|
|
294
|
+
* answer in an envelope returns what is inside it. Errors are not passed
|
|
295
|
+
* through here: they are read from the whole response.
|
|
296
|
+
*/
|
|
297
|
+
readDecisionPayload(payload) {
|
|
298
|
+
return payload;
|
|
299
|
+
}
|
|
254
300
|
/** Per-question confidence a transport reports outside the answer objects. */
|
|
255
301
|
reportedConfidence(_payload) {
|
|
256
302
|
return {};
|
|
@@ -354,7 +400,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
354
400
|
const signal = request.signal
|
|
355
401
|
? AbortSignal.any([request.signal, timeout])
|
|
356
402
|
: timeout;
|
|
357
|
-
const response = await this.proxyFetch(this.decisionEndpoint(), {
|
|
403
|
+
const response = await this.proxyFetch(this.decisionEndpoint(resolvedModel), {
|
|
358
404
|
method: "POST",
|
|
359
405
|
headers: this.decisionHeaders(),
|
|
360
406
|
body,
|
|
@@ -393,7 +439,8 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
393
439
|
await sleep(retryAfterMs ?? 2 ** attempt * 250 + Math.random() * 250, request.signal);
|
|
394
440
|
continue;
|
|
395
441
|
}
|
|
396
|
-
|
|
442
|
+
const decoded = this.readDecisionPayload(payload);
|
|
443
|
+
if (!isRecord(decoded) || !isRecord(decoded.answers)) {
|
|
397
444
|
throw this.decisionError({
|
|
398
445
|
kind: "server",
|
|
399
446
|
message: `${label} returned a response without an answers map.`,
|
|
@@ -402,9 +449,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
402
449
|
retryable: false,
|
|
403
450
|
});
|
|
404
451
|
}
|
|
405
|
-
const reported = this.reportedConfidence(
|
|
452
|
+
const reported = this.reportedConfidence(decoded);
|
|
406
453
|
const answers = {};
|
|
407
|
-
for (const [id, raw] of Object.entries(
|
|
454
|
+
for (const [id, raw] of Object.entries(decoded.answers)) {
|
|
408
455
|
const parsed = parseDecisionAnswer(raw, reported[id]);
|
|
409
456
|
if (parsed) {
|
|
410
457
|
answers[id] = parsed;
|
|
@@ -415,13 +462,13 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
415
462
|
});
|
|
416
463
|
}
|
|
417
464
|
}
|
|
418
|
-
const usage = isRecord(
|
|
465
|
+
const usage = isRecord(decoded.usage) ? decoded.usage : {};
|
|
419
466
|
return {
|
|
420
467
|
// `resolvedModel`, not `this.modelName`: when the caller pinned a
|
|
421
468
|
// model for this one request and the response omits its own, the
|
|
422
469
|
// instance default would be reported instead of the model actually
|
|
423
470
|
// asked for.
|
|
424
|
-
model: this.resolveResponseModel(
|
|
471
|
+
model: this.resolveResponseModel(decoded, resolvedModel),
|
|
425
472
|
provider: this.providerName,
|
|
426
473
|
answers,
|
|
427
474
|
// Two spellings for one field. The System One wire sends
|
|
@@ -485,8 +532,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
485
532
|
* applies to.
|
|
486
533
|
*/
|
|
487
534
|
assertWithinRequestBytes(body) {
|
|
488
|
-
const
|
|
489
|
-
?.decisionLimits?.media
|
|
535
|
+
const media = PROVIDER_DESCRIPTORS_BY_NAME.get(this.providerName)
|
|
536
|
+
?.decisionLimits?.media;
|
|
537
|
+
const limit = media?.maxRequestBytes;
|
|
490
538
|
if (limit === undefined) {
|
|
491
539
|
return;
|
|
492
540
|
}
|
|
@@ -494,7 +542,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
494
542
|
if (bytes > limit) {
|
|
495
543
|
throw this.decisionError({
|
|
496
544
|
kind: "invalid_request",
|
|
497
|
-
message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images, or a shorter video.`,
|
|
545
|
+
message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images${media?.video ? ", or a shorter video" : ""}.`,
|
|
498
546
|
retryable: false,
|
|
499
547
|
});
|
|
500
548
|
}
|
|
@@ -520,7 +568,11 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
520
568
|
}
|
|
521
569
|
const modelLimits = limits.models?.[model];
|
|
522
570
|
const maxStateTokens = modelLimits?.maxStateTokens ?? limits.maxStateTokens;
|
|
523
|
-
const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar
|
|
571
|
+
const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar, {
|
|
572
|
+
digit: limits.digitTokensPerChar,
|
|
573
|
+
symbol: limits.symbolTokensPerChar,
|
|
574
|
+
astral: limits.astralTokensPerChar,
|
|
575
|
+
});
|
|
524
576
|
if (stateTokens > maxStateTokens) {
|
|
525
577
|
throw this.decisionError({
|
|
526
578
|
kind: "max_tokens_exceeded",
|
package/dist/types/decision.d.ts
CHANGED
|
@@ -153,6 +153,28 @@ export type DecisionLimits = {
|
|
|
153
153
|
* estimate for every character.
|
|
154
154
|
*/
|
|
155
155
|
nonAsciiTokensPerChar?: number;
|
|
156
|
+
/**
|
|
157
|
+
* Tokens charged per ASCII digit. A tokenizer that reads every digit as its
|
|
158
|
+
* own token makes numbers, ids and timestamps several times longer than the
|
|
159
|
+
* default estimate of four characters per token. Absent = digits are
|
|
160
|
+
* estimated like any other ASCII character.
|
|
161
|
+
*/
|
|
162
|
+
digitTokensPerChar?: number;
|
|
163
|
+
/**
|
|
164
|
+
* Tokens charged per ASCII punctuation or symbol character (`,` `.` `{` `"`
|
|
165
|
+
* `:` and the like). The same tokenizers that read each digit alone read most
|
|
166
|
+
* punctuation alone too, so JSON, logs and lists of numbers run far above four
|
|
167
|
+
* characters a token. Absent = punctuation is estimated like any other ASCII
|
|
168
|
+
* character.
|
|
169
|
+
*/
|
|
170
|
+
symbolTokensPerChar?: number;
|
|
171
|
+
/**
|
|
172
|
+
* Tokens charged per character outside the Basic Multilingual Plane (emoji
|
|
173
|
+
* and the like), which is counted separately from `nonAsciiTokensPerChar`
|
|
174
|
+
* because it costs about twice as much. Absent = charged at
|
|
175
|
+
* `nonAsciiTokensPerChar`.
|
|
176
|
+
*/
|
|
177
|
+
astralTokensPerChar?: number;
|
|
156
178
|
/** Per-model limits, keyed by model id; each field overrides the one above. */
|
|
157
179
|
models?: Readonly<Record<string, {
|
|
158
180
|
maxStateTokens: number;
|
|
@@ -56,6 +56,8 @@ export type CatalogErrorRuleClass = "authentication" | "rate-limit" | "invalid-m
|
|
|
56
56
|
export type CatalogErrorRuleJson = {
|
|
57
57
|
status?: number;
|
|
58
58
|
pattern?: string;
|
|
59
|
+
/** Restrict the pattern to these HTTP statuses; status matching stays independent. */
|
|
60
|
+
patternStatuses?: number[];
|
|
59
61
|
class: CatalogErrorRuleClass;
|
|
60
62
|
message: string;
|
|
61
63
|
};
|
|
@@ -583,6 +583,19 @@ export type NeurolinkCredentials = {
|
|
|
583
583
|
apiKey?: string;
|
|
584
584
|
baseURL?: string;
|
|
585
585
|
};
|
|
586
|
+
/**
|
|
587
|
+
* Cloudflare Clef — the `decide` inference type, reached through Workers AI.
|
|
588
|
+
* The token and account id are the ones the `cloudflare` text provider reads
|
|
589
|
+
* (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), but this slice is separate
|
|
590
|
+
* so the two cannot be mixed up. Both are required. `baseURL` is optional and
|
|
591
|
+
* defaults to `https://api.cloudflare.com/client/v4`; requests go to
|
|
592
|
+
* `<baseURL>/accounts/<accountId>/ai/run/@cf/cloudflare/<model>`.
|
|
593
|
+
*/
|
|
594
|
+
cloudflareClef?: {
|
|
595
|
+
apiKey?: string;
|
|
596
|
+
accountId?: string;
|
|
597
|
+
baseURL?: string;
|
|
598
|
+
};
|
|
586
599
|
};
|
|
587
600
|
/**
|
|
588
601
|
* Voyage AI /embeddings response shape.
|
|
@@ -2325,6 +2338,8 @@ export type ProviderDescriptor = {
|
|
|
2325
2338
|
extraRequired?: readonly string[];
|
|
2326
2339
|
/** Alternate ways to satisfy extraRequired when it isn't a plain env-var list (e.g. Vertex's file-path-OR-individual-fields auth). Each entry is either a single env var name (satisfied alone) or a nested array of names that must ALL be present together (e.g. Vertex's GOOGLE_AUTH_CLIENT_EMAIL + GOOGLE_AUTH_PRIVATE_KEY pair, which is only valid as a pair). Evaluate with `satisfiesFallbacks()` (providerConfig.ts) rather than re-deriving this logic at each call site. */
|
|
2327
2340
|
extraRequiredFallbacks?: readonly (string | readonly string[])[];
|
|
2341
|
+
/** For an `extraRequired` name that can also be given in `credentials.<credentialsKey>`: the field of that slice that stands for it (e.g. `{ CLOUDFLARE_ACCOUNT_ID: "accountId" }`). A base URL needs no entry; it is matched through `baseURL` above. */
|
|
2342
|
+
extraRequiredCredentialFields?: Readonly<Record<string, string>>;
|
|
2328
2343
|
/** True when the provider is usable with zero configuration (local runtime with a documented default URL, or a documented non-secret default like LiteLLM's "sk-anything"). */
|
|
2329
2344
|
optional?: boolean;
|
|
2330
2345
|
};
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* Centralized model choices for CLI commands
|
|
3
3
|
* Derives choices from model enums to ensure consistency
|
|
4
4
|
*/
|
|
5
|
-
import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, ReplicateModels, } from "../constants/enums.js";
|
|
5
|
+
import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, CloudflareClefModels, ReplicateModels, } from "../constants/enums.js";
|
|
6
6
|
import { getCatalogJsonEntries } from "../providers/catalog/loader.js";
|
|
7
7
|
/**
|
|
8
8
|
* Looks up the JSON catalog entry for a provider, if it is one of the 11
|
|
@@ -399,6 +399,16 @@ const TOP_MODELS_CONFIG = {
|
|
|
399
399
|
description: "Recommended - Perplexity decision model that also reads images",
|
|
400
400
|
},
|
|
401
401
|
],
|
|
402
|
+
[AIProviderName.CLOUDFLARE_CLEF]: [
|
|
403
|
+
{
|
|
404
|
+
model: CloudflareClefModels.CLEF,
|
|
405
|
+
description: "Recommended - Cloudflare Clef 27B decision model that also reads images (state limited to ~2,000 tokens)",
|
|
406
|
+
},
|
|
407
|
+
{
|
|
408
|
+
model: CloudflareClefModels.CLEF_FLASH,
|
|
409
|
+
description: "Clef-flash 9B - the faster, cheaper Cloudflare decision model (same ~2,000-token state limit)",
|
|
410
|
+
},
|
|
411
|
+
],
|
|
402
412
|
[AIProviderName.AUTO]: [],
|
|
403
413
|
};
|
|
404
414
|
/**
|
|
@@ -441,6 +451,7 @@ export const DEFAULT_MODELS = {
|
|
|
441
451
|
[AIProviderName.LAYA]: LayaModels.TYPED_DECISIONS,
|
|
442
452
|
[AIProviderName.XOR]: XorModels.XOR_1_1,
|
|
443
453
|
[AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels.PPLX_DECIDER_V1_27B,
|
|
454
|
+
[AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels.CLEF,
|
|
444
455
|
};
|
|
445
456
|
/**
|
|
446
457
|
* Model enum mappings for getAllModels.
|
|
@@ -476,6 +487,7 @@ const MODEL_ENUMS = {
|
|
|
476
487
|
[AIProviderName.LAYA]: LayaModels,
|
|
477
488
|
[AIProviderName.XOR]: XorModels,
|
|
478
489
|
[AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels,
|
|
490
|
+
[AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels,
|
|
479
491
|
[AIProviderName.AUTO]: null,
|
|
480
492
|
};
|
|
481
493
|
/**
|
package/dist/utils/pricing.js
CHANGED
|
@@ -788,6 +788,18 @@ const PRICING = {
|
|
|
788
788
|
_default: { input: 0.04 / 1_000_000, output: 0 },
|
|
789
789
|
"pplx-decider-v1-27b": { input: 0.04 / 1_000_000, output: 0 },
|
|
790
790
|
},
|
|
791
|
+
"cloudflare-clef": {
|
|
792
|
+
// Workers AI bills Clef per million INPUT tokens ($0.24 for clef, $0.09 for
|
|
793
|
+
// clef-flash); no output price is listed. `output: 0` is a price, not a token
|
|
794
|
+
// count, so do not "correct" it from `usage.output_tokens`. The `cf-ai-neurons`
|
|
795
|
+
// response header was checked against the published neuron rates on
|
|
796
|
+
// 2026-10-03 and matched them exactly (21,818 and 8,182 neurons per million
|
|
797
|
+
// input tokens). Rates from Cloudflare's Workers AI pricing page. An unknown
|
|
798
|
+
// model name is priced as clef, the dearer of the two.
|
|
799
|
+
_default: { input: 0.24 / 1_000_000, output: 0 },
|
|
800
|
+
clef: { input: 0.24 / 1_000_000, output: 0 },
|
|
801
|
+
"clef-flash": { input: 0.09 / 1_000_000, output: 0 },
|
|
802
|
+
},
|
|
791
803
|
stability: {
|
|
792
804
|
// Stability AI bills per image; symbolic per-token rate.
|
|
793
805
|
_default: { input: 0, output: 0.04 / 1_000 },
|
|
@@ -416,3 +416,10 @@ export declare function createXorConfig(): ProviderConfigOptions;
|
|
|
416
416
|
* provider and has a public endpoint, so the key alone configures it.
|
|
417
417
|
*/
|
|
418
418
|
export declare function createPerplexityDeciderConfig(): ProviderConfigOptions;
|
|
419
|
+
/**
|
|
420
|
+
* Cloudflare Clef — the `decide` inference type, reached through Workers AI,
|
|
421
|
+
* not the Workers AI text models. It reads the same CLOUDFLARE_API_KEY and
|
|
422
|
+
* CLOUDFLARE_ACCOUNT_ID as the `cloudflare` text provider; the endpoint is
|
|
423
|
+
* Cloudflare's own, so the two together configure it.
|
|
424
|
+
*/
|
|
425
|
+
export declare function createCloudflareClefConfig(): ProviderConfigOptions;
|
|
@@ -1376,3 +1376,22 @@ export function createPerplexityDeciderConfig() {
|
|
|
1376
1376
|
],
|
|
1377
1377
|
};
|
|
1378
1378
|
}
|
|
1379
|
+
/**
|
|
1380
|
+
* Cloudflare Clef — the `decide` inference type, reached through Workers AI,
|
|
1381
|
+
* not the Workers AI text models. It reads the same CLOUDFLARE_API_KEY and
|
|
1382
|
+
* CLOUDFLARE_ACCOUNT_ID as the `cloudflare` text provider; the endpoint is
|
|
1383
|
+
* Cloudflare's own, so the two together configure it.
|
|
1384
|
+
*/
|
|
1385
|
+
export function createCloudflareClefConfig() {
|
|
1386
|
+
return {
|
|
1387
|
+
providerName: "Cloudflare Clef",
|
|
1388
|
+
envVarName: "CLOUDFLARE_API_KEY",
|
|
1389
|
+
setupUrl: "https://dash.cloudflare.com/profile/api-tokens",
|
|
1390
|
+
description: "API token (Workers AI scope) for Cloudflare's Clef decision models",
|
|
1391
|
+
instructions: [
|
|
1392
|
+
"1. Create an API token with the 'Workers AI: Read + Write' permission (https://dash.cloudflare.com/profile/api-tokens)",
|
|
1393
|
+
"2. Set CLOUDFLARE_API_KEY to it, and CLOUDFLARE_ACCOUNT_ID to your account id (in the dashboard URL, or under 'Account ID'); they are the same settings the Cloudflare text provider reads, so setting them also lets decide() use Clef when no other decision provider is configured",
|
|
1394
|
+
"3. The Clef endpoint ignores state text past about 2,048 tokens, so use it for short decisions",
|
|
1395
|
+
],
|
|
1396
|
+
};
|
|
1397
|
+
}
|
|
@@ -52,6 +52,11 @@ export declare function duckTypedStatusCode(error: unknown): number | undefined;
|
|
|
52
52
|
* hand-rolled OpenAI-compatible client's record). Shared with baseProvider.
|
|
53
53
|
*/
|
|
54
54
|
export declare function extractRetryAfterMsFromError(error: unknown): number | undefined;
|
|
55
|
+
/**
|
|
56
|
+
* OpenAI's chat transport keeps the type inside the JSON response body,
|
|
57
|
+
* rather than stamping it onto the raw error.
|
|
58
|
+
*/
|
|
59
|
+
export declare function readOpenAIBodyErrorType(responseBody: unknown): string | undefined;
|
|
55
60
|
/**
|
|
56
61
|
* An OpenAI-wire `insufficient_quota` error: the account is out of credit or
|
|
57
62
|
* has hit a spend cap.
|
|
@@ -105,6 +105,23 @@ export function extractRetryAfterMsFromError(error) {
|
|
|
105
105
|
}
|
|
106
106
|
return undefined;
|
|
107
107
|
}
|
|
108
|
+
/**
|
|
109
|
+
* OpenAI's chat transport keeps the type inside the JSON response body,
|
|
110
|
+
* rather than stamping it onto the raw error.
|
|
111
|
+
*/
|
|
112
|
+
export function readOpenAIBodyErrorType(responseBody) {
|
|
113
|
+
if (typeof responseBody !== "string") {
|
|
114
|
+
return undefined;
|
|
115
|
+
}
|
|
116
|
+
try {
|
|
117
|
+
const parsed = JSON.parse(responseBody);
|
|
118
|
+
const inner = parsed?.error;
|
|
119
|
+
return typeof inner?.type === "string" ? inner.type : undefined;
|
|
120
|
+
}
|
|
121
|
+
catch {
|
|
122
|
+
return undefined;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
108
125
|
/**
|
|
109
126
|
* An OpenAI-wire `insufficient_quota` error: the account is out of credit or
|
|
110
127
|
* has hit a spend cap.
|
|
@@ -130,18 +147,7 @@ export function isOpenAIQuotaExhaustedError(error) {
|
|
|
130
147
|
if (err.type === "insufficient_quota") {
|
|
131
148
|
return true;
|
|
132
149
|
}
|
|
133
|
-
|
|
134
|
-
if (typeof err.responseBody === "string") {
|
|
135
|
-
try {
|
|
136
|
-
const parsed = JSON.parse(err.responseBody);
|
|
137
|
-
const inner = parsed?.error;
|
|
138
|
-
return inner?.type === "insufficient_quota";
|
|
139
|
-
}
|
|
140
|
-
catch {
|
|
141
|
-
return false;
|
|
142
|
-
}
|
|
143
|
-
}
|
|
144
|
-
return false;
|
|
150
|
+
return readOpenAIBodyErrorType(err.responseBody) === "insufficient_quota";
|
|
145
151
|
}
|
|
146
152
|
/**
|
|
147
153
|
* True when `error` is an abort, or wraps one at any depth via `cause`.
|