@juspay/neurolink 12.46.1 → 12.47.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,331 @@
1
+ import { CloudflareClefModels } from "../constants/enums.js";
2
+ import { logger } from "../utils/logger.js";
3
+ import { redactUrlForError } from "../utils/logSanitize.js";
4
+ import { getProviderModel } from "../utils/providerConfig.js";
5
+ import { isRecord, redactCredentials, SystemOneDecisionProvider, } from "./systemOneDecision.js";
6
+ const CLOUDFLARE_DEFAULT_BASE_URL = "https://api.cloudflare.com/client/v4";
7
+ const MODEL_PREFIX = "@cf/cloudflare/";
8
+ /**
9
+ * Cloudflare account ids are 32 hex digits. The pattern is wider so that a
10
+ * change of format needs no release, but it still keeps `/`, `.`, `?` and
11
+ * whitespace out of the URL path the id is placed in.
12
+ */
13
+ const ACCOUNT_ID_PATTERN = /^[A-Za-z0-9_-]{1,64}$/;
14
+ /** The only image formats the API reads; anything else is a 422. */
15
+ const SUPPORTED_IMAGE_TYPES = new Set([
16
+ "image/png",
17
+ "image/jpeg",
18
+ "image/webp",
19
+ ]);
20
+ /**
21
+ * The uuid Workers AI appends to an `AiError` message, in parentheses. It starts
22
+ * at a literal `(`, with no leading `\s*`: that one backtracks quadratically on
23
+ * a long run of spaces, and the message is whatever the server sent.
24
+ */
25
+ const TRAILING_REQUEST_ID = /\(([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})\)\s*$/i;
26
+ const MAX_ERROR_MESSAGE_CHARS = 500;
27
+ /**
28
+ * An error message is cut to this before any pattern runs on it, so a hostile or
29
+ * garbled body cannot hold the event loop. Cloudflare's own messages are far
30
+ * shorter (the longest seen was about 350 characters).
31
+ */
32
+ const MAX_PARSED_MESSAGE_CHARS = 4096;
33
+ function normalizeBaseURL(raw) {
34
+ return raw.replace(/\/+$/, "");
35
+ }
36
+ /**
37
+ * A base URL that cannot work is refused up front, and its text is never
38
+ * repeated: `fetch` rejects a URL with userinfo and echoes it in the error, the
39
+ * route is appended after a query string or fragment, and any of them can hold
40
+ * a credential.
41
+ */
42
+ function baseURLProblem(baseURL) {
43
+ const fix = `Set CLOUDFLARE_CLEF_BASE_URL or pass credentials.cloudflareClef.baseURL to a base URL, or leave both unset to use ${CLOUDFLARE_DEFAULT_BASE_URL}.`;
44
+ try {
45
+ const url = new URL(baseURL);
46
+ if (url.protocol !== "https:" && url.protocol !== "http:") {
47
+ return `The Cloudflare base URL must start with https:// or http://. ${fix}`;
48
+ }
49
+ // `url.search` and `url.hash` are empty for a bare trailing `?` or `#`, which
50
+ // would still put the route into the query or the fragment, so the text is
51
+ // checked as well.
52
+ return url.username ||
53
+ url.password ||
54
+ url.search ||
55
+ url.hash ||
56
+ /[?#]/.test(baseURL)
57
+ ? `The Cloudflare base URL must not carry credentials, a query string or a fragment. ${fix}`
58
+ : undefined;
59
+ }
60
+ catch {
61
+ return `The Cloudflare base URL is not a valid absolute URL. ${fix}`;
62
+ }
63
+ }
64
+ /**
65
+ * The name Cloudflare wants in the path and the body: `clef` or `clef-flash`,
66
+ * given either bare or as `@cf/cloudflare/clef`. Anything with a `/` in it is
67
+ * refused, so a model name cannot reach a different route.
68
+ */
69
+ function wireModel(model) {
70
+ const bare = model.trim().replace(/^@cf\/cloudflare\//i, "");
71
+ return /^[a-z0-9][a-z0-9._-]*$/i.test(bare) ? bare.toLowerCase() : undefined;
72
+ }
73
+ /**
74
+ * Why this image cannot be sent, or undefined when it can. The data URL is
75
+ * never repeated: it is megabytes of base64. Size limits are left to the
76
+ * server, which refuses an oversized image at once with a clear 422.
77
+ */
78
+ function imageProblem(dataUrl, position) {
79
+ const type = /^data:([a-z]+\/[a-z0-9.+-]+);base64,/i
80
+ .exec(dataUrl)?.[1]
81
+ ?.toLowerCase();
82
+ return type && SUPPORTED_IMAGE_TYPES.has(type)
83
+ ? undefined
84
+ : `Image ${position} is not a PNG, JPEG or WebP image, the only formats Cloudflare Clef reads.`;
85
+ }
86
+ /** The API takes a string or structured data as `state`; a number or a boolean is sent as its text. */
87
+ function wireState(state) {
88
+ return typeof state === "number" || typeof state === "boolean"
89
+ ? String(state)
90
+ : state;
91
+ }
92
+ /**
93
+ * `{"success":false,"errors":[{"code","message"}],...}` is Cloudflare's own
94
+ * envelope. A model-side refusal nests a second one inside the message:
95
+ * `AiError: AiError: {"error":{"message","details":{"fieldErrors":{...}}}} (<uuid>)`.
96
+ * Both are flattened to one readable line, with the uuid taken out as the
97
+ * request id. A body that is not this envelope reads as undefined and the
98
+ * caller falls back to the status.
99
+ */
100
+ function readCloudflareError(body) {
101
+ if (!isRecord(body) || !Array.isArray(body.errors)) {
102
+ return undefined;
103
+ }
104
+ const first = body.errors.find(isRecord);
105
+ if (!first || typeof first.message !== "string" || first.message === "") {
106
+ return undefined;
107
+ }
108
+ let text = first.message.slice(0, MAX_PARSED_MESSAGE_CHARS).trimEnd();
109
+ let requestId;
110
+ const trailing = TRAILING_REQUEST_ID.exec(text);
111
+ if (trailing) {
112
+ requestId = trailing[1];
113
+ text = text.slice(0, trailing.index).trimEnd();
114
+ }
115
+ text = text.replace(/^(?:AiError:\s*|Ai:\s*)+/, "").trim();
116
+ if (text.startsWith("{")) {
117
+ try {
118
+ const inner = JSON.parse(text);
119
+ if (isRecord(inner) && isRecord(inner.error)) {
120
+ const details = isRecord(inner.error.details)
121
+ ? inner.error.details
122
+ : undefined;
123
+ const fields = isRecord(details?.fieldErrors)
124
+ ? Object.entries(details.fieldErrors).flatMap(([field, problems]) => Array.isArray(problems)
125
+ ? problems
126
+ .filter((p) => typeof p === "string")
127
+ .map((p) => `${field}: ${p}`)
128
+ : [])
129
+ : [];
130
+ const head = typeof inner.error.message === "string" ? inner.error.message : text;
131
+ text = fields.length > 0 ? `${head}: ${fields.join("; ")}` : head;
132
+ }
133
+ }
134
+ catch {
135
+ // not JSON after all; keep the text as it is
136
+ }
137
+ }
138
+ return {
139
+ message: text,
140
+ code: typeof first.code === "number" ? first.code : undefined,
141
+ requestId,
142
+ };
143
+ }
144
+ /**
145
+ * Only 401 is `authentication`, and so only 401 trips the breaker: Cloudflare
146
+ * answered a token it does not know with a 401 and code 10000 (seen live). A 403
147
+ * was never seen, and no such token was available to provoke one; if Cloudflare
148
+ * uses it for a token that lacks Workers AI permission, that is fixed in the
149
+ * dashboard, and a breaker that latched on it would keep this instance disabled
150
+ * until the process restarted, even after the permission was granted. So a 403
151
+ * is a non-retried `invalid_request` that carries Cloudflare's own message. A
152
+ * request over the context window is a 413, code 5021, and a 429 is "Capacity
153
+ * temporarily exceeded" (code 3040, no Retry-After), which is retryable.
154
+ * Validation refusals are 422, and a malformed body or a wrong route is a 400;
155
+ * none of those is retried.
156
+ */
157
+ function cloudflareErrorKind(status, code) {
158
+ if (status === 401) {
159
+ return "authentication";
160
+ }
161
+ if (status === 413 || code === 5021) {
162
+ return "max_tokens_exceeded";
163
+ }
164
+ if (status === 429) {
165
+ return "rate_limit";
166
+ }
167
+ if (status === 503) {
168
+ return "overloaded";
169
+ }
170
+ return status >= 500 ? "server" : "invalid_request";
171
+ }
172
+ /**
173
+ * Cloudflare Clef — the `decide` inference type only.
174
+ *
175
+ * `@cf/cloudflare/clef` (27B) and `@cf/cloudflare/clef-flash` (9B) answer the
176
+ * same typed `noul` / `choice` / `score` questions as the other decision
177
+ * providers, follow the System One wire, and read images. They are reached
178
+ * through the Workers AI REST API, which wraps every answer in Cloudflare's own
179
+ * `{ result, success, errors }` envelope and puts the model in the URL path.
180
+ *
181
+ * The token and account id are the ones the Cloudflare Workers AI text provider
182
+ * reads (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), so a host that has
183
+ * configured that provider has also configured this one.
184
+ *
185
+ * The Workers AI endpoint ignores text past about 2,048 tokens, far below
186
+ * the documented 64K (hosted service or model: unknown). See `decisionLimits`
187
+ * on the descriptor for the measured figures.
188
+ *
189
+ * @see https://developers.cloudflare.com/workers-ai/models/clef/
190
+ */
191
+ export class CloudflareClefProvider extends SystemOneDecisionProvider {
192
+ apiKey;
193
+ accountId;
194
+ baseURL;
195
+ constructor(modelName, sdk, _region, credentials) {
196
+ super(modelName, "cloudflare-clef", sdk);
197
+ // A slice that names its own base URL must bring its own token: the shared
198
+ // CLOUDFLARE_API_KEY also runs the host's Workers AI text provider, and it
199
+ // must never be sent as a bearer token to an endpoint a caller chose.
200
+ const ownEndpoint = Boolean(credentials?.baseURL?.trim());
201
+ this.apiKey =
202
+ credentials?.apiKey?.trim() ||
203
+ (ownEndpoint ? "" : (process.env.CLOUDFLARE_API_KEY?.trim() ?? ""));
204
+ this.accountId =
205
+ credentials?.accountId?.trim() ||
206
+ (process.env.CLOUDFLARE_ACCOUNT_ID?.trim() ?? "");
207
+ // `||`, not `??`, so a blank override counts as unset.
208
+ this.baseURL = normalizeBaseURL(credentials?.baseURL?.trim() ||
209
+ process.env.CLOUDFLARE_CLEF_BASE_URL?.trim() ||
210
+ CLOUDFLARE_DEFAULT_BASE_URL);
211
+ // A base URL that `baseURLProblem` refuses is not logged at all: it can
212
+ // carry `user:pass@` or a `?token=`.
213
+ logger.debug("Cloudflare Clef decision provider initialized (decide only)", {
214
+ modelName: this.modelName,
215
+ baseURL: baseURLProblem(this.baseURL)
216
+ ? "(invalid)"
217
+ : redactUrlForError(this.baseURL),
218
+ accountConfigured: this.accountId !== "",
219
+ });
220
+ }
221
+ getDefaultModel() {
222
+ return getProviderModel("CLOUDFLARE_CLEF_MODEL", CloudflareClefModels.CLEF);
223
+ }
224
+ vendorLabel() {
225
+ return "Cloudflare";
226
+ }
227
+ vendorDisplayName() {
228
+ return "Cloudflare Clef";
229
+ }
230
+ decisionApiKey() {
231
+ return this.apiKey;
232
+ }
233
+ missingKeyMessage() {
234
+ return "Cloudflare Clef requires an API token with Workers AI permission. Set CLOUDFLARE_API_KEY or pass credentials.cloudflareClef.apiKey.";
235
+ }
236
+ missingConfigMessage() {
237
+ if (this.accountId === "") {
238
+ return "Cloudflare Clef requires the account id. Set CLOUDFLARE_ACCOUNT_ID or pass credentials.cloudflareClef.accountId.";
239
+ }
240
+ if (!ACCOUNT_ID_PATTERN.test(this.accountId)) {
241
+ return "The Cloudflare account id may contain only letters, digits, '-' and '_'. Copy it from the Cloudflare dashboard.";
242
+ }
243
+ return baseURLProblem(this.baseURL);
244
+ }
245
+ /** The model is part of the path, so the endpoint depends on the model asked for. */
246
+ decisionEndpoint(model) {
247
+ return `${this.baseURL}/accounts/${encodeURIComponent(this.accountId)}/ai/run/${MODEL_PREFIX}${encodeURIComponent(wireModel(model) ?? model)}`;
248
+ }
249
+ decisionHeaders() {
250
+ return {
251
+ Authorization: `Bearer ${this.apiKey}`,
252
+ "Content-Type": "application/json",
253
+ };
254
+ }
255
+ /**
256
+ * Images travel in their own `images` array, as `data:` URLs, placed before
257
+ * the state by the server. `model` is sent although the path already names
258
+ * it: the API documents it as required, and refuses a body whose `model`
259
+ * differs from the path.
260
+ */
261
+ buildDecisionBody(state, questions, model, media) {
262
+ const wire = wireModel(model);
263
+ if (!wire) {
264
+ throw this.decisionError({
265
+ kind: "invalid_request",
266
+ message: `"${model.slice(0, 40)}" is not a Cloudflare model name. Use "clef" or "clef-flash".`,
267
+ retryable: false,
268
+ });
269
+ }
270
+ // Workers AI also accepts image/jpg; use the canonical JPEG MIME type.
271
+ const images = (media?.images ?? []).map((dataUrl) => dataUrl.replace(/^data:image\/jpg;base64,/i, "data:image/jpeg;base64,"));
272
+ images.forEach((dataUrl, index) => {
273
+ const problem = imageProblem(dataUrl, index + 1);
274
+ if (problem) {
275
+ throw this.decisionError({
276
+ kind: "invalid_request",
277
+ message: problem,
278
+ retryable: false,
279
+ });
280
+ }
281
+ });
282
+ // The canonical JPEG prefix is one byte longer than image/jpg. Keep the
283
+ // prepared media metadata aligned with the URLs actually sent.
284
+ if (media) {
285
+ media.images = images;
286
+ media.bytes = images.reduce((total, image) => total + Buffer.byteLength(image), 0);
287
+ }
288
+ return {
289
+ model: wire,
290
+ state: wireState(state),
291
+ questions,
292
+ ...(images.length > 0 ? { images } : {}),
293
+ };
294
+ }
295
+ /** A success is `{ result: { model, answers, usage }, success: true }`. */
296
+ readDecisionPayload(payload) {
297
+ return isRecord(payload) && isRecord(payload.result)
298
+ ? payload.result
299
+ : payload;
300
+ }
301
+ parseDecisionError(status, payload, requestId) {
302
+ const fallback = `Cloudflare request failed with HTTP ${status}`;
303
+ const parsed = readCloudflareError(payload);
304
+ const kind = cloudflareErrorKind(status, parsed?.code);
305
+ let message = redactCredentials(parsed?.message ?? fallback, this.apiKey)
306
+ .trim()
307
+ .slice(0, MAX_ERROR_MESSAGE_CHARS) || fallback;
308
+ if (kind === "max_tokens_exceeded") {
309
+ message +=
310
+ " Cloudflare counts the whole request, image data included, at about four characters per token.";
311
+ }
312
+ else if (parsed?.code === 7000) {
313
+ message += `. Check the model name (Clef is served as "clef" and "clef-flash") and the base URL, which must end in /client/v4 and name your account.`;
314
+ }
315
+ return {
316
+ kind,
317
+ message,
318
+ status,
319
+ requestId: requestId ?? parsed?.requestId,
320
+ retryable: kind === "rate_limit" || kind === "overloaded" || kind === "server",
321
+ };
322
+ }
323
+ /**
324
+ * `cf-ai-req-id` is on every answer that reached the model, success or
325
+ * refusal. A 401 and a wrong-path 400 never reach it, so they carry only the
326
+ * edge's `cf-ray`.
327
+ */
328
+ readRequestId(headers) {
329
+ return headers.get("cf-ai-req-id") ?? headers.get("cf-ray") ?? undefined;
330
+ }
331
+ }
@@ -59,7 +59,11 @@ export declare abstract class SystemOneDecisionProvider extends BaseProvider {
59
59
  * nothing is missing.
60
60
  */
61
61
  protected missingConfigMessage(): string | undefined;
62
- protected abstract decisionEndpoint(): string;
62
+ /**
63
+ * Where the request goes. The model asked for is passed in because one
64
+ * vendor names it in the URL path; every other provider ignores it.
65
+ */
66
+ protected abstract decisionEndpoint(model: string): string;
63
67
  protected abstract decisionHeaders(): Record<string, string>;
64
68
  protected abstract buildDecisionBody(state: DecisionState, questions: Record<string, Record<string, unknown>>, model: string, media?: DecisionPreparedMedia): Record<string, unknown>;
65
69
  protected abstract parseDecisionError(status: number, payload: unknown, requestId: string | undefined): DecisionError;
@@ -81,6 +85,13 @@ export declare abstract class SystemOneDecisionProvider extends BaseProvider {
81
85
  * grows with the number of questions scales the descriptor's allowance here.
82
86
  */
83
87
  protected defaultTimeoutMs(_questionCount: number): number | undefined;
88
+ /**
89
+ * The part of a successful response that holds `answers`, `usage` and
90
+ * `model`. The default is the response itself; a vendor that wraps every
91
+ * answer in an envelope returns what is inside it. Errors are not passed
92
+ * through here: they are read from the whole response.
93
+ */
94
+ protected readDecisionPayload(payload: unknown): unknown;
84
95
  /** Per-question confidence a transport reports outside the answer objects. */
85
96
  protected reportedConfidence(_payload: unknown): Record<string, number>;
86
97
  protected resolveResponseModel(payload: Record<string, unknown>, requestedModel: string): string;
@@ -6,7 +6,7 @@ import { ProviderError } from "../types/index.js";
6
6
  import { prepareDecisionMedia } from "../utils/decisionMedia.js";
7
7
  import { logger } from "../utils/logger.js";
8
8
  import { redactUrlsInText } from "../utils/logSanitize.js";
9
- import { estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
9
+ import { CHARS_PER_TOKEN, estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
10
10
  /**
11
11
  * Generous enough for a cold start (measured at 2.0–2.7s after idle on Jev)
12
12
  * while still bounded. Every internal caller is fail-open, so this only bites
@@ -177,15 +177,52 @@ const sleep = (ms, signal) => new Promise((resolve, reject) => {
177
177
  * tokenizer, so when a provider declares a rate, non-ASCII characters are
178
178
  * counted at it instead. ASCII text is estimated exactly as before.
179
179
  */
180
- function estimateDecisionStateTokens(text, nonAsciiTokensPerChar) {
181
- if (nonAsciiTokensPerChar === undefined) {
180
+ function estimateDecisionStateTokens(text, nonAsciiTokensPerChar, rates = {}) {
181
+ const { digit, symbol, astral } = rates;
182
+ if (nonAsciiTokensPerChar === undefined &&
183
+ digit === undefined &&
184
+ symbol === undefined &&
185
+ astral === undefined) {
182
186
  return estimateTokens(text);
183
187
  }
184
- const chars = [...text];
185
- const isAscii = (c) => (c.codePointAt(0) ?? 0) <= 0x7f;
186
- const ascii = chars.filter(isAscii).join("");
187
- const nonAsciiCount = chars.length - ascii.length;
188
- return (estimateTokens(ascii) + Math.ceil(nonAsciiCount * nonAsciiTokensPerChar));
188
+ // Each class of character is counted once, at its own rate. A class with no
189
+ // declared rate stays in the ordinary four-characters-a-token estimate (or,
190
+ // for non-ASCII, at that estimate's per-character rate), so a provider that
191
+ // declares none of the extra rates is estimated exactly as before.
192
+ const isDigit = (c) => c >= "0" && c <= "9";
193
+ const isSymbol = (c) => /[!-/:-@[-`{-~]/.test(c);
194
+ let digits = 0;
195
+ let symbols = 0;
196
+ let astrals = 0;
197
+ let nonAscii = 0;
198
+ const rest = [];
199
+ for (const c of text) {
200
+ const code = c.codePointAt(0) ?? 0;
201
+ if (code > 0xffff && astral !== undefined) {
202
+ astrals += 1;
203
+ }
204
+ else if (code > 0x7f) {
205
+ nonAscii += 1;
206
+ }
207
+ else if (digit !== undefined && isDigit(c)) {
208
+ digits += 1;
209
+ }
210
+ else if (symbol !== undefined && isSymbol(c)) {
211
+ symbols += 1;
212
+ }
213
+ else {
214
+ rest.push(c);
215
+ }
216
+ }
217
+ // Digits, punctuation and astral characters (emoji) are charged separately
218
+ // only when the provider declares a rate for them: a tokenizer that reads each
219
+ // digit, and most punctuation, on its own makes a number-heavy or JSON-heavy
220
+ // state several times longer than four characters a token suggests.
221
+ return (estimateTokens(rest.join("")) +
222
+ Math.ceil(digits * (digit ?? 0)) +
223
+ Math.ceil(symbols * (symbol ?? 0)) +
224
+ Math.ceil(astrals * (astral ?? 0)) +
225
+ Math.ceil(nonAscii * (nonAsciiTokensPerChar ?? 1 / CHARS_PER_TOKEN)));
189
226
  }
190
227
  /**
191
228
  * The shared half of every "System One" decision provider — a model that takes
@@ -251,6 +288,15 @@ export class SystemOneDecisionProvider extends BaseProvider {
251
288
  defaultTimeoutMs(_questionCount) {
252
289
  return this.getDescriptorDecideMs();
253
290
  }
291
+ /**
292
+ * The part of a successful response that holds `answers`, `usage` and
293
+ * `model`. The default is the response itself; a vendor that wraps every
294
+ * answer in an envelope returns what is inside it. Errors are not passed
295
+ * through here: they are read from the whole response.
296
+ */
297
+ readDecisionPayload(payload) {
298
+ return payload;
299
+ }
254
300
  /** Per-question confidence a transport reports outside the answer objects. */
255
301
  reportedConfidence(_payload) {
256
302
  return {};
@@ -354,7 +400,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
354
400
  const signal = request.signal
355
401
  ? AbortSignal.any([request.signal, timeout])
356
402
  : timeout;
357
- const response = await this.proxyFetch(this.decisionEndpoint(), {
403
+ const response = await this.proxyFetch(this.decisionEndpoint(resolvedModel), {
358
404
  method: "POST",
359
405
  headers: this.decisionHeaders(),
360
406
  body,
@@ -393,7 +439,8 @@ export class SystemOneDecisionProvider extends BaseProvider {
393
439
  await sleep(retryAfterMs ?? 2 ** attempt * 250 + Math.random() * 250, request.signal);
394
440
  continue;
395
441
  }
396
- if (!isRecord(payload) || !isRecord(payload.answers)) {
442
+ const decoded = this.readDecisionPayload(payload);
443
+ if (!isRecord(decoded) || !isRecord(decoded.answers)) {
397
444
  throw this.decisionError({
398
445
  kind: "server",
399
446
  message: `${label} returned a response without an answers map.`,
@@ -402,9 +449,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
402
449
  retryable: false,
403
450
  });
404
451
  }
405
- const reported = this.reportedConfidence(payload);
452
+ const reported = this.reportedConfidence(decoded);
406
453
  const answers = {};
407
- for (const [id, raw] of Object.entries(payload.answers)) {
454
+ for (const [id, raw] of Object.entries(decoded.answers)) {
408
455
  const parsed = parseDecisionAnswer(raw, reported[id]);
409
456
  if (parsed) {
410
457
  answers[id] = parsed;
@@ -415,13 +462,13 @@ export class SystemOneDecisionProvider extends BaseProvider {
415
462
  });
416
463
  }
417
464
  }
418
- const usage = isRecord(payload.usage) ? payload.usage : {};
465
+ const usage = isRecord(decoded.usage) ? decoded.usage : {};
419
466
  return {
420
467
  // `resolvedModel`, not `this.modelName`: when the caller pinned a
421
468
  // model for this one request and the response omits its own, the
422
469
  // instance default would be reported instead of the model actually
423
470
  // asked for.
424
- model: this.resolveResponseModel(payload, resolvedModel),
471
+ model: this.resolveResponseModel(decoded, resolvedModel),
425
472
  provider: this.providerName,
426
473
  answers,
427
474
  // Two spellings for one field. The System One wire sends
@@ -485,8 +532,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
485
532
  * applies to.
486
533
  */
487
534
  assertWithinRequestBytes(body) {
488
- const limit = PROVIDER_DESCRIPTORS_BY_NAME.get(this.providerName)
489
- ?.decisionLimits?.media?.maxRequestBytes;
535
+ const media = PROVIDER_DESCRIPTORS_BY_NAME.get(this.providerName)
536
+ ?.decisionLimits?.media;
537
+ const limit = media?.maxRequestBytes;
490
538
  if (limit === undefined) {
491
539
  return;
492
540
  }
@@ -494,7 +542,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
494
542
  if (bytes > limit) {
495
543
  throw this.decisionError({
496
544
  kind: "invalid_request",
497
- message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images, or a shorter video.`,
545
+ message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images${media?.video ? ", or a shorter video" : ""}.`,
498
546
  retryable: false,
499
547
  });
500
548
  }
@@ -520,7 +568,11 @@ export class SystemOneDecisionProvider extends BaseProvider {
520
568
  }
521
569
  const modelLimits = limits.models?.[model];
522
570
  const maxStateTokens = modelLimits?.maxStateTokens ?? limits.maxStateTokens;
523
- const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar);
571
+ const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar, {
572
+ digit: limits.digitTokensPerChar,
573
+ symbol: limits.symbolTokensPerChar,
574
+ astral: limits.astralTokensPerChar,
575
+ });
524
576
  if (stateTokens > maxStateTokens) {
525
577
  throw this.decisionError({
526
578
  kind: "max_tokens_exceeded",
@@ -153,6 +153,28 @@ export type DecisionLimits = {
153
153
  * estimate for every character.
154
154
  */
155
155
  nonAsciiTokensPerChar?: number;
156
+ /**
157
+ * Tokens charged per ASCII digit. A tokenizer that reads every digit as its
158
+ * own token makes numbers, ids and timestamps several times longer than the
159
+ * default estimate of four characters per token. Absent = digits are
160
+ * estimated like any other ASCII character.
161
+ */
162
+ digitTokensPerChar?: number;
163
+ /**
164
+ * Tokens charged per ASCII punctuation or symbol character (`,` `.` `{` `"`
165
+ * `:` and the like). The same tokenizers that read each digit alone read most
166
+ * punctuation alone too, so JSON, logs and lists of numbers run far above four
167
+ * characters a token. Absent = punctuation is estimated like any other ASCII
168
+ * character.
169
+ */
170
+ symbolTokensPerChar?: number;
171
+ /**
172
+ * Tokens charged per character outside the Basic Multilingual Plane (emoji
173
+ * and the like), which is counted separately from `nonAsciiTokensPerChar`
174
+ * because it costs about twice as much. Absent = charged at
175
+ * `nonAsciiTokensPerChar`.
176
+ */
177
+ astralTokensPerChar?: number;
156
178
  /** Per-model limits, keyed by model id; each field overrides the one above. */
157
179
  models?: Readonly<Record<string, {
158
180
  maxStateTokens: number;
@@ -583,6 +583,19 @@ export type NeurolinkCredentials = {
583
583
  apiKey?: string;
584
584
  baseURL?: string;
585
585
  };
586
+ /**
587
+ * Cloudflare Clef — the `decide` inference type, reached through Workers AI.
588
+ * The token and account id are the ones the `cloudflare` text provider reads
589
+ * (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), but this slice is separate
590
+ * so the two cannot be mixed up. Both are required. `baseURL` is optional and
591
+ * defaults to `https://api.cloudflare.com/client/v4`; requests go to
592
+ * `<baseURL>/accounts/<accountId>/ai/run/@cf/cloudflare/<model>`.
593
+ */
594
+ cloudflareClef?: {
595
+ apiKey?: string;
596
+ accountId?: string;
597
+ baseURL?: string;
598
+ };
586
599
  };
587
600
  /**
588
601
  * Voyage AI /embeddings response shape.
@@ -2325,6 +2338,8 @@ export type ProviderDescriptor = {
2325
2338
  extraRequired?: readonly string[];
2326
2339
  /** Alternate ways to satisfy extraRequired when it isn't a plain env-var list (e.g. Vertex's file-path-OR-individual-fields auth). Each entry is either a single env var name (satisfied alone) or a nested array of names that must ALL be present together (e.g. Vertex's GOOGLE_AUTH_CLIENT_EMAIL + GOOGLE_AUTH_PRIVATE_KEY pair, which is only valid as a pair). Evaluate with `satisfiesFallbacks()` (providerConfig.ts) rather than re-deriving this logic at each call site. */
2327
2340
  extraRequiredFallbacks?: readonly (string | readonly string[])[];
2341
+ /** For an `extraRequired` name that can also be given in `credentials.<credentialsKey>`: the field of that slice that stands for it (e.g. `{ CLOUDFLARE_ACCOUNT_ID: "accountId" }`). A base URL needs no entry; it is matched through `baseURL` above. */
2342
+ extraRequiredCredentialFields?: Readonly<Record<string, string>>;
2328
2343
  /** True when the provider is usable with zero configuration (local runtime with a documented default URL, or a documented non-secret default like LiteLLM's "sk-anything"). */
2329
2344
  optional?: boolean;
2330
2345
  };
@@ -2,7 +2,7 @@
2
2
  * Centralized model choices for CLI commands
3
3
  * Derives choices from model enums to ensure consistency
4
4
  */
5
- import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, ReplicateModels, } from "../constants/enums.js";
5
+ import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, CloudflareClefModels, ReplicateModels, } from "../constants/enums.js";
6
6
  import { getCatalogJsonEntries } from "../providers/catalog/loader.js";
7
7
  /**
8
8
  * Looks up the JSON catalog entry for a provider, if it is one of the 11
@@ -399,6 +399,16 @@ const TOP_MODELS_CONFIG = {
399
399
  description: "Recommended - Perplexity decision model that also reads images",
400
400
  },
401
401
  ],
402
+ [AIProviderName.CLOUDFLARE_CLEF]: [
403
+ {
404
+ model: CloudflareClefModels.CLEF,
405
+ description: "Recommended - Cloudflare Clef 27B decision model that also reads images (state limited to ~2,000 tokens)",
406
+ },
407
+ {
408
+ model: CloudflareClefModels.CLEF_FLASH,
409
+ description: "Clef-flash 9B - the faster, cheaper Cloudflare decision model (same ~2,000-token state limit)",
410
+ },
411
+ ],
402
412
  [AIProviderName.AUTO]: [],
403
413
  };
404
414
  /**
@@ -441,6 +451,7 @@ export const DEFAULT_MODELS = {
441
451
  [AIProviderName.LAYA]: LayaModels.TYPED_DECISIONS,
442
452
  [AIProviderName.XOR]: XorModels.XOR_1_1,
443
453
  [AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels.PPLX_DECIDER_V1_27B,
454
+ [AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels.CLEF,
444
455
  };
445
456
  /**
446
457
  * Model enum mappings for getAllModels.
@@ -476,6 +487,7 @@ const MODEL_ENUMS = {
476
487
  [AIProviderName.LAYA]: LayaModels,
477
488
  [AIProviderName.XOR]: XorModels,
478
489
  [AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels,
490
+ [AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels,
479
491
  [AIProviderName.AUTO]: null,
480
492
  };
481
493
  /**
@@ -788,6 +788,18 @@ const PRICING = {
788
788
  _default: { input: 0.04 / 1_000_000, output: 0 },
789
789
  "pplx-decider-v1-27b": { input: 0.04 / 1_000_000, output: 0 },
790
790
  },
791
+ "cloudflare-clef": {
792
+ // Workers AI bills Clef per million INPUT tokens ($0.24 for clef, $0.09 for
793
+ // clef-flash); no output price is listed. `output: 0` is a price, not a token
794
+ // count, so do not "correct" it from `usage.output_tokens`. The `cf-ai-neurons`
795
+ // response header was checked against the published neuron rates on
796
+ // 2026-10-03 and matched them exactly (21,818 and 8,182 neurons per million
797
+ // input tokens). Rates from Cloudflare's Workers AI pricing page. An unknown
798
+ // model name is priced as clef, the dearer of the two.
799
+ _default: { input: 0.24 / 1_000_000, output: 0 },
800
+ clef: { input: 0.24 / 1_000_000, output: 0 },
801
+ "clef-flash": { input: 0.09 / 1_000_000, output: 0 },
802
+ },
791
803
  stability: {
792
804
  // Stability AI bills per image; symbolic per-token rate.
793
805
  _default: { input: 0, output: 0.04 / 1_000 },
@@ -416,3 +416,10 @@ export declare function createXorConfig(): ProviderConfigOptions;
416
416
  * provider and has a public endpoint, so the key alone configures it.
417
417
  */
418
418
  export declare function createPerplexityDeciderConfig(): ProviderConfigOptions;
419
+ /**
420
+ * Cloudflare Clef — the `decide` inference type, reached through Workers AI,
421
+ * not the Workers AI text models. It reads the same CLOUDFLARE_API_KEY and
422
+ * CLOUDFLARE_ACCOUNT_ID as the `cloudflare` text provider; the endpoint is
423
+ * Cloudflare's own, so the two together configure it.
424
+ */
425
+ export declare function createCloudflareClefConfig(): ProviderConfigOptions;