@juspay/neurolink 12.46.0 → 12.47.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +2 -2
  2. package/README.md +75 -50
  3. package/dist/browser/neurolink.min.js +426 -426
  4. package/dist/cli/commands/decide.js +2 -2
  5. package/dist/cli/commands/setup.js +2 -1
  6. package/dist/cli/factories/commandFactory.js +1 -1
  7. package/dist/constants/enums.d.ts +16 -0
  8. package/dist/constants/enums.js +17 -0
  9. package/dist/factories/providerDescriptors.js +84 -5
  10. package/dist/factories/providerRegistry.js +10 -1
  11. package/dist/models/manifestRegistry.js +2 -0
  12. package/dist/models/manifests/cloudflareClef.d.ts +16 -0
  13. package/dist/models/manifests/cloudflareClef.js +42 -0
  14. package/dist/providers/anthropic/client.js +6 -1
  15. package/dist/providers/catalog/huggingface.json +2 -1
  16. package/dist/providers/catalog/loader.js +5 -1
  17. package/dist/providers/catalog/schema.d.ts +1 -0
  18. package/dist/providers/catalog/schema.js +8 -0
  19. package/dist/providers/cloudflareClef.d.ts +52 -0
  20. package/dist/providers/cloudflareClef.js +331 -0
  21. package/dist/providers/ideogram.js +9 -30
  22. package/dist/providers/llamaCpp.js +2 -1
  23. package/dist/providers/openAI/client.js +4 -5
  24. package/dist/providers/openaiChatCompletionsBase.js +12 -7
  25. package/dist/providers/openaiChatCompletionsClient.js +25 -7
  26. package/dist/providers/recraft.js +9 -28
  27. package/dist/providers/systemOneDecision.d.ts +12 -1
  28. package/dist/providers/systemOneDecision.js +70 -18
  29. package/dist/types/decision.d.ts +22 -0
  30. package/dist/types/providerCatalog.d.ts +2 -0
  31. package/dist/types/providers.d.ts +15 -0
  32. package/dist/utils/modelChoices.js +13 -1
  33. package/dist/utils/pricing.js +12 -0
  34. package/dist/utils/providerConfig.d.ts +7 -0
  35. package/dist/utils/providerConfig.js +19 -0
  36. package/dist/utils/providerRetry.d.ts +5 -0
  37. package/dist/utils/providerRetry.js +18 -12
  38. package/docs-site/static/search-index.json +580 -557
  39. package/package.json +2 -1
@@ -6,7 +6,7 @@ import { ProviderError } from "../types/index.js";
6
6
  import { prepareDecisionMedia } from "../utils/decisionMedia.js";
7
7
  import { logger } from "../utils/logger.js";
8
8
  import { redactUrlsInText } from "../utils/logSanitize.js";
9
- import { estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
9
+ import { CHARS_PER_TOKEN, estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
10
10
  /**
11
11
  * Generous enough for a cold start (measured at 2.0–2.7s after idle on Jev)
12
12
  * while still bounded. Every internal caller is fail-open, so this only bites
@@ -177,15 +177,52 @@ const sleep = (ms, signal) => new Promise((resolve, reject) => {
177
177
  * tokenizer, so when a provider declares a rate, non-ASCII characters are
178
178
  * counted at it instead. ASCII text is estimated exactly as before.
179
179
  */
180
- function estimateDecisionStateTokens(text, nonAsciiTokensPerChar) {
181
- if (nonAsciiTokensPerChar === undefined) {
180
+ function estimateDecisionStateTokens(text, nonAsciiTokensPerChar, rates = {}) {
181
+ const { digit, symbol, astral } = rates;
182
+ if (nonAsciiTokensPerChar === undefined &&
183
+ digit === undefined &&
184
+ symbol === undefined &&
185
+ astral === undefined) {
182
186
  return estimateTokens(text);
183
187
  }
184
- const chars = [...text];
185
- const isAscii = (c) => (c.codePointAt(0) ?? 0) <= 0x7f;
186
- const ascii = chars.filter(isAscii).join("");
187
- const nonAsciiCount = chars.length - ascii.length;
188
- return (estimateTokens(ascii) + Math.ceil(nonAsciiCount * nonAsciiTokensPerChar));
188
+ // Each class of character is counted once, at its own rate. A class with no
189
+ // declared rate stays in the ordinary four-characters-a-token estimate (or,
190
+ // for non-ASCII, at that estimate's per-character rate), so a provider that
191
+ // declares none of the extra rates is estimated exactly as before.
192
+ const isDigit = (c) => c >= "0" && c <= "9";
193
+ const isSymbol = (c) => /[!-/:-@[-`{-~]/.test(c);
194
+ let digits = 0;
195
+ let symbols = 0;
196
+ let astrals = 0;
197
+ let nonAscii = 0;
198
+ const rest = [];
199
+ for (const c of text) {
200
+ const code = c.codePointAt(0) ?? 0;
201
+ if (code > 0xffff && astral !== undefined) {
202
+ astrals += 1;
203
+ }
204
+ else if (code > 0x7f) {
205
+ nonAscii += 1;
206
+ }
207
+ else if (digit !== undefined && isDigit(c)) {
208
+ digits += 1;
209
+ }
210
+ else if (symbol !== undefined && isSymbol(c)) {
211
+ symbols += 1;
212
+ }
213
+ else {
214
+ rest.push(c);
215
+ }
216
+ }
217
+ // Digits, punctuation and astral characters (emoji) are charged separately
218
+ // only when the provider declares a rate for them: a tokenizer that reads each
219
+ // digit, and most punctuation, on its own makes a number-heavy or JSON-heavy
220
+ // state several times longer than four characters a token suggests.
221
+ return (estimateTokens(rest.join("")) +
222
+ Math.ceil(digits * (digit ?? 0)) +
223
+ Math.ceil(symbols * (symbol ?? 0)) +
224
+ Math.ceil(astrals * (astral ?? 0)) +
225
+ Math.ceil(nonAscii * (nonAsciiTokensPerChar ?? 1 / CHARS_PER_TOKEN)));
189
226
  }
190
227
  /**
191
228
  * The shared half of every "System One" decision provider — a model that takes
@@ -251,6 +288,15 @@ export class SystemOneDecisionProvider extends BaseProvider {
251
288
  defaultTimeoutMs(_questionCount) {
252
289
  return this.getDescriptorDecideMs();
253
290
  }
291
+ /**
292
+ * The part of a successful response that holds `answers`, `usage` and
293
+ * `model`. The default is the response itself; a vendor that wraps every
294
+ * answer in an envelope returns what is inside it. Errors are not passed
295
+ * through here: they are read from the whole response.
296
+ */
297
+ readDecisionPayload(payload) {
298
+ return payload;
299
+ }
254
300
  /** Per-question confidence a transport reports outside the answer objects. */
255
301
  reportedConfidence(_payload) {
256
302
  return {};
@@ -354,7 +400,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
354
400
  const signal = request.signal
355
401
  ? AbortSignal.any([request.signal, timeout])
356
402
  : timeout;
357
- const response = await this.proxyFetch(this.decisionEndpoint(), {
403
+ const response = await this.proxyFetch(this.decisionEndpoint(resolvedModel), {
358
404
  method: "POST",
359
405
  headers: this.decisionHeaders(),
360
406
  body,
@@ -393,7 +439,8 @@ export class SystemOneDecisionProvider extends BaseProvider {
393
439
  await sleep(retryAfterMs ?? 2 ** attempt * 250 + Math.random() * 250, request.signal);
394
440
  continue;
395
441
  }
396
- if (!isRecord(payload) || !isRecord(payload.answers)) {
442
+ const decoded = this.readDecisionPayload(payload);
443
+ if (!isRecord(decoded) || !isRecord(decoded.answers)) {
397
444
  throw this.decisionError({
398
445
  kind: "server",
399
446
  message: `${label} returned a response without an answers map.`,
@@ -402,9 +449,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
402
449
  retryable: false,
403
450
  });
404
451
  }
405
- const reported = this.reportedConfidence(payload);
452
+ const reported = this.reportedConfidence(decoded);
406
453
  const answers = {};
407
- for (const [id, raw] of Object.entries(payload.answers)) {
454
+ for (const [id, raw] of Object.entries(decoded.answers)) {
408
455
  const parsed = parseDecisionAnswer(raw, reported[id]);
409
456
  if (parsed) {
410
457
  answers[id] = parsed;
@@ -415,13 +462,13 @@ export class SystemOneDecisionProvider extends BaseProvider {
415
462
  });
416
463
  }
417
464
  }
418
- const usage = isRecord(payload.usage) ? payload.usage : {};
465
+ const usage = isRecord(decoded.usage) ? decoded.usage : {};
419
466
  return {
420
467
  // `resolvedModel`, not `this.modelName`: when the caller pinned a
421
468
  // model for this one request and the response omits its own, the
422
469
  // instance default would be reported instead of the model actually
423
470
  // asked for.
424
- model: this.resolveResponseModel(payload, resolvedModel),
471
+ model: this.resolveResponseModel(decoded, resolvedModel),
425
472
  provider: this.providerName,
426
473
  answers,
427
474
  // Two spellings for one field. The System One wire sends
@@ -485,8 +532,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
485
532
  * applies to.
486
533
  */
487
534
  assertWithinRequestBytes(body) {
488
- const limit = PROVIDER_DESCRIPTORS_BY_NAME.get(this.providerName)
489
- ?.decisionLimits?.media?.maxRequestBytes;
535
+ const media = PROVIDER_DESCRIPTORS_BY_NAME.get(this.providerName)
536
+ ?.decisionLimits?.media;
537
+ const limit = media?.maxRequestBytes;
490
538
  if (limit === undefined) {
491
539
  return;
492
540
  }
@@ -494,7 +542,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
494
542
  if (bytes > limit) {
495
543
  throw this.decisionError({
496
544
  kind: "invalid_request",
497
- message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images, or a shorter video.`,
545
+ message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images${media?.video ? ", or a shorter video" : ""}.`,
498
546
  retryable: false,
499
547
  });
500
548
  }
@@ -520,7 +568,11 @@ export class SystemOneDecisionProvider extends BaseProvider {
520
568
  }
521
569
  const modelLimits = limits.models?.[model];
522
570
  const maxStateTokens = modelLimits?.maxStateTokens ?? limits.maxStateTokens;
523
- const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar);
571
+ const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar, {
572
+ digit: limits.digitTokensPerChar,
573
+ symbol: limits.symbolTokensPerChar,
574
+ astral: limits.astralTokensPerChar,
575
+ });
524
576
  if (stateTokens > maxStateTokens) {
525
577
  throw this.decisionError({
526
578
  kind: "max_tokens_exceeded",
@@ -153,6 +153,28 @@ export type DecisionLimits = {
153
153
  * estimate for every character.
154
154
  */
155
155
  nonAsciiTokensPerChar?: number;
156
+ /**
157
+ * Tokens charged per ASCII digit. A tokenizer that reads every digit as its
158
+ * own token makes numbers, ids and timestamps several times longer than the
159
+ * default estimate of four characters per token. Absent = digits are
160
+ * estimated like any other ASCII character.
161
+ */
162
+ digitTokensPerChar?: number;
163
+ /**
164
+ * Tokens charged per ASCII punctuation or symbol character (`,` `.` `{` `"`
165
+ * `:` and the like). The same tokenizers that read each digit alone read most
166
+ * punctuation alone too, so JSON, logs and lists of numbers run far above four
167
+ * characters a token. Absent = punctuation is estimated like any other ASCII
168
+ * character.
169
+ */
170
+ symbolTokensPerChar?: number;
171
+ /**
172
+ * Tokens charged per character outside the Basic Multilingual Plane (emoji
173
+ * and the like), which is counted separately from `nonAsciiTokensPerChar`
174
+ * because it costs about twice as much. Absent = charged at
175
+ * `nonAsciiTokensPerChar`.
176
+ */
177
+ astralTokensPerChar?: number;
156
178
  /** Per-model limits, keyed by model id; each field overrides the one above. */
157
179
  models?: Readonly<Record<string, {
158
180
  maxStateTokens: number;
@@ -56,6 +56,8 @@ export type CatalogErrorRuleClass = "authentication" | "rate-limit" | "invalid-m
56
56
  export type CatalogErrorRuleJson = {
57
57
  status?: number;
58
58
  pattern?: string;
59
+ /** Restrict the pattern to these HTTP statuses; status matching stays independent. */
60
+ patternStatuses?: number[];
59
61
  class: CatalogErrorRuleClass;
60
62
  message: string;
61
63
  };
@@ -583,6 +583,19 @@ export type NeurolinkCredentials = {
583
583
  apiKey?: string;
584
584
  baseURL?: string;
585
585
  };
586
+ /**
587
+ * Cloudflare Clef — the `decide` inference type, reached through Workers AI.
588
+ * The token and account id are the ones the `cloudflare` text provider reads
589
+ * (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), but this slice is separate
590
+ * so the two cannot be mixed up. Both are required. `baseURL` is optional and
591
+ * defaults to `https://api.cloudflare.com/client/v4`; requests go to
592
+ * `<baseURL>/accounts/<accountId>/ai/run/@cf/cloudflare/<model>`.
593
+ */
594
+ cloudflareClef?: {
595
+ apiKey?: string;
596
+ accountId?: string;
597
+ baseURL?: string;
598
+ };
586
599
  };
587
600
  /**
588
601
  * Voyage AI /embeddings response shape.
@@ -2325,6 +2338,8 @@ export type ProviderDescriptor = {
2325
2338
  extraRequired?: readonly string[];
2326
2339
  /** Alternate ways to satisfy extraRequired when it isn't a plain env-var list (e.g. Vertex's file-path-OR-individual-fields auth). Each entry is either a single env var name (satisfied alone) or a nested array of names that must ALL be present together (e.g. Vertex's GOOGLE_AUTH_CLIENT_EMAIL + GOOGLE_AUTH_PRIVATE_KEY pair, which is only valid as a pair). Evaluate with `satisfiesFallbacks()` (providerConfig.ts) rather than re-deriving this logic at each call site. */
2327
2340
  extraRequiredFallbacks?: readonly (string | readonly string[])[];
2341
+ /** For an `extraRequired` name that can also be given in `credentials.<credentialsKey>`: the field of that slice that stands for it (e.g. `{ CLOUDFLARE_ACCOUNT_ID: "accountId" }`). A base URL needs no entry; it is matched through `baseURL` above. */
2342
+ extraRequiredCredentialFields?: Readonly<Record<string, string>>;
2328
2343
  /** True when the provider is usable with zero configuration (local runtime with a documented default URL, or a documented non-secret default like LiteLLM's "sk-anything"). */
2329
2344
  optional?: boolean;
2330
2345
  };
@@ -2,7 +2,7 @@
2
2
  * Centralized model choices for CLI commands
3
3
  * Derives choices from model enums to ensure consistency
4
4
  */
5
- import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, ReplicateModels, } from "../constants/enums.js";
5
+ import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, CloudflareClefModels, ReplicateModels, } from "../constants/enums.js";
6
6
  import { getCatalogJsonEntries } from "../providers/catalog/loader.js";
7
7
  /**
8
8
  * Looks up the JSON catalog entry for a provider, if it is one of the 11
@@ -399,6 +399,16 @@ const TOP_MODELS_CONFIG = {
399
399
  description: "Recommended - Perplexity decision model that also reads images",
400
400
  },
401
401
  ],
402
+ [AIProviderName.CLOUDFLARE_CLEF]: [
403
+ {
404
+ model: CloudflareClefModels.CLEF,
405
+ description: "Recommended - Cloudflare Clef 27B decision model that also reads images (state limited to ~2,000 tokens)",
406
+ },
407
+ {
408
+ model: CloudflareClefModels.CLEF_FLASH,
409
+ description: "Clef-flash 9B - the faster, cheaper Cloudflare decision model (same ~2,000-token state limit)",
410
+ },
411
+ ],
402
412
  [AIProviderName.AUTO]: [],
403
413
  };
404
414
  /**
@@ -441,6 +451,7 @@ export const DEFAULT_MODELS = {
441
451
  [AIProviderName.LAYA]: LayaModels.TYPED_DECISIONS,
442
452
  [AIProviderName.XOR]: XorModels.XOR_1_1,
443
453
  [AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels.PPLX_DECIDER_V1_27B,
454
+ [AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels.CLEF,
444
455
  };
445
456
  /**
446
457
  * Model enum mappings for getAllModels.
@@ -476,6 +487,7 @@ const MODEL_ENUMS = {
476
487
  [AIProviderName.LAYA]: LayaModels,
477
488
  [AIProviderName.XOR]: XorModels,
478
489
  [AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels,
490
+ [AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels,
479
491
  [AIProviderName.AUTO]: null,
480
492
  };
481
493
  /**
@@ -788,6 +788,18 @@ const PRICING = {
788
788
  _default: { input: 0.04 / 1_000_000, output: 0 },
789
789
  "pplx-decider-v1-27b": { input: 0.04 / 1_000_000, output: 0 },
790
790
  },
791
+ "cloudflare-clef": {
792
+ // Workers AI bills Clef per million INPUT tokens ($0.24 for clef, $0.09 for
793
+ // clef-flash); no output price is listed. `output: 0` is a price, not a token
794
+ // count, so do not "correct" it from `usage.output_tokens`. The `cf-ai-neurons`
795
+ // response header was checked against the published neuron rates on
796
+ // 2026-10-03 and matched them exactly (21,818 and 8,182 neurons per million
797
+ // input tokens). Rates from Cloudflare's Workers AI pricing page. An unknown
798
+ // model name is priced as clef, the dearer of the two.
799
+ _default: { input: 0.24 / 1_000_000, output: 0 },
800
+ clef: { input: 0.24 / 1_000_000, output: 0 },
801
+ "clef-flash": { input: 0.09 / 1_000_000, output: 0 },
802
+ },
791
803
  stability: {
792
804
  // Stability AI bills per image; symbolic per-token rate.
793
805
  _default: { input: 0, output: 0.04 / 1_000 },
@@ -416,3 +416,10 @@ export declare function createXorConfig(): ProviderConfigOptions;
416
416
  * provider and has a public endpoint, so the key alone configures it.
417
417
  */
418
418
  export declare function createPerplexityDeciderConfig(): ProviderConfigOptions;
419
+ /**
420
+ * Cloudflare Clef — the `decide` inference type, reached through Workers AI,
421
+ * not the Workers AI text models. It reads the same CLOUDFLARE_API_KEY and
422
+ * CLOUDFLARE_ACCOUNT_ID as the `cloudflare` text provider; the endpoint is
423
+ * Cloudflare's own, so the two together configure it.
424
+ */
425
+ export declare function createCloudflareClefConfig(): ProviderConfigOptions;
@@ -1376,3 +1376,22 @@ export function createPerplexityDeciderConfig() {
1376
1376
  ],
1377
1377
  };
1378
1378
  }
1379
+ /**
1380
+ * Cloudflare Clef — the `decide` inference type, reached through Workers AI,
1381
+ * not the Workers AI text models. It reads the same CLOUDFLARE_API_KEY and
1382
+ * CLOUDFLARE_ACCOUNT_ID as the `cloudflare` text provider; the endpoint is
1383
+ * Cloudflare's own, so the two together configure it.
1384
+ */
1385
+ export function createCloudflareClefConfig() {
1386
+ return {
1387
+ providerName: "Cloudflare Clef",
1388
+ envVarName: "CLOUDFLARE_API_KEY",
1389
+ setupUrl: "https://dash.cloudflare.com/profile/api-tokens",
1390
+ description: "API token (Workers AI scope) for Cloudflare's Clef decision models",
1391
+ instructions: [
1392
+ "1. Create an API token with the 'Workers AI: Read + Write' permission (https://dash.cloudflare.com/profile/api-tokens)",
1393
+ "2. Set CLOUDFLARE_API_KEY to it, and CLOUDFLARE_ACCOUNT_ID to your account id (in the dashboard URL, or under 'Account ID'); they are the same settings the Cloudflare text provider reads, so setting them also lets decide() use Clef when no other decision provider is configured",
1394
+ "3. The Clef endpoint ignores state text past about 2,048 tokens, so use it for short decisions",
1395
+ ],
1396
+ };
1397
+ }
@@ -52,6 +52,11 @@ export declare function duckTypedStatusCode(error: unknown): number | undefined;
52
52
  * hand-rolled OpenAI-compatible client's record). Shared with baseProvider.
53
53
  */
54
54
  export declare function extractRetryAfterMsFromError(error: unknown): number | undefined;
55
+ /**
56
+ * OpenAI's chat transport keeps the type inside the JSON response body,
57
+ * rather than stamping it onto the raw error.
58
+ */
59
+ export declare function readOpenAIBodyErrorType(responseBody: unknown): string | undefined;
55
60
  /**
56
61
  * An OpenAI-wire `insufficient_quota` error: the account is out of credit or
57
62
  * has hit a spend cap.
@@ -105,6 +105,23 @@ export function extractRetryAfterMsFromError(error) {
105
105
  }
106
106
  return undefined;
107
107
  }
108
+ /**
109
+ * OpenAI's chat transport keeps the type inside the JSON response body,
110
+ * rather than stamping it onto the raw error.
111
+ */
112
+ export function readOpenAIBodyErrorType(responseBody) {
113
+ if (typeof responseBody !== "string") {
114
+ return undefined;
115
+ }
116
+ try {
117
+ const parsed = JSON.parse(responseBody);
118
+ const inner = parsed?.error;
119
+ return typeof inner?.type === "string" ? inner.type : undefined;
120
+ }
121
+ catch {
122
+ return undefined;
123
+ }
124
+ }
108
125
  /**
109
126
  * An OpenAI-wire `insufficient_quota` error: the account is out of credit or
110
127
  * has hit a spend cap.
@@ -130,18 +147,7 @@ export function isOpenAIQuotaExhaustedError(error) {
130
147
  if (err.type === "insufficient_quota") {
131
148
  return true;
132
149
  }
133
- // The SDK keeps the raw payload as a string; the type lives inside it.
134
- if (typeof err.responseBody === "string") {
135
- try {
136
- const parsed = JSON.parse(err.responseBody);
137
- const inner = parsed?.error;
138
- return inner?.type === "insufficient_quota";
139
- }
140
- catch {
141
- return false;
142
- }
143
- }
144
- return false;
150
+ return readOpenAIBodyErrorType(err.responseBody) === "insufficient_quota";
145
151
  }
146
152
  /**
147
153
  * True when `error` is an abort, or wraps one at any depth via `cause`.