@juspay/neurolink 12.46.1 → 12.47.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +3 -3
- package/README.md +75 -50
- package/dist/browser/neurolink.min.js +409 -409
- package/dist/cli/commands/decide.js +2 -2
- package/dist/cli/commands/setup.js +2 -1
- package/dist/cli/factories/commandFactory.js +1 -1
- package/dist/constants/enums.d.ts +16 -0
- package/dist/constants/enums.js +17 -0
- package/dist/factories/providerDescriptors.js +84 -5
- package/dist/factories/providerRegistry.js +10 -1
- package/dist/models/manifestRegistry.js +2 -0
- package/dist/models/manifests/cloudflareClef.d.ts +16 -0
- package/dist/models/manifests/cloudflareClef.js +42 -0
- package/dist/providers/cloudflareClef.d.ts +52 -0
- package/dist/providers/cloudflareClef.js +331 -0
- package/dist/providers/systemOneDecision.d.ts +12 -1
- package/dist/providers/systemOneDecision.js +70 -18
- package/dist/types/decision.d.ts +22 -0
- package/dist/types/providers.d.ts +15 -0
- package/dist/utils/modelChoices.js +13 -1
- package/dist/utils/pricing.js +12 -0
- package/dist/utils/providerConfig.d.ts +7 -0
- package/dist/utils/providerConfig.js +19 -0
- package/docs-site/static/search-index.json +78 -56
- package/package.json +2 -1
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
import { CloudflareClefModels } from "../constants/enums.js";
|
|
2
|
+
import { logger } from "../utils/logger.js";
|
|
3
|
+
import { redactUrlForError } from "../utils/logSanitize.js";
|
|
4
|
+
import { getProviderModel } from "../utils/providerConfig.js";
|
|
5
|
+
import { isRecord, redactCredentials, SystemOneDecisionProvider, } from "./systemOneDecision.js";
|
|
6
|
+
const CLOUDFLARE_DEFAULT_BASE_URL = "https://api.cloudflare.com/client/v4";
|
|
7
|
+
const MODEL_PREFIX = "@cf/cloudflare/";
|
|
8
|
+
/**
|
|
9
|
+
* Cloudflare account ids are 32 hex digits. The pattern is wider so that a
|
|
10
|
+
* change of format needs no release, but it still keeps `/`, `.`, `?` and
|
|
11
|
+
* whitespace out of the URL path the id is placed in.
|
|
12
|
+
*/
|
|
13
|
+
const ACCOUNT_ID_PATTERN = /^[A-Za-z0-9_-]{1,64}$/;
|
|
14
|
+
/** The only image formats the API reads; anything else is a 422. */
|
|
15
|
+
const SUPPORTED_IMAGE_TYPES = new Set([
|
|
16
|
+
"image/png",
|
|
17
|
+
"image/jpeg",
|
|
18
|
+
"image/webp",
|
|
19
|
+
]);
|
|
20
|
+
/**
|
|
21
|
+
* The uuid Workers AI appends to an `AiError` message, in parentheses. It starts
|
|
22
|
+
* at a literal `(`, with no leading `\s*`: that one backtracks quadratically on
|
|
23
|
+
* a long run of spaces, and the message is whatever the server sent.
|
|
24
|
+
*/
|
|
25
|
+
const TRAILING_REQUEST_ID = /\(([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})\)\s*$/i;
|
|
26
|
+
const MAX_ERROR_MESSAGE_CHARS = 500;
|
|
27
|
+
/**
|
|
28
|
+
* An error message is cut to this before any pattern runs on it, so a hostile or
|
|
29
|
+
* garbled body cannot hold the event loop. Cloudflare's own messages are far
|
|
30
|
+
* shorter (the longest seen was about 350 characters).
|
|
31
|
+
*/
|
|
32
|
+
const MAX_PARSED_MESSAGE_CHARS = 4096;
|
|
33
|
+
function normalizeBaseURL(raw) {
|
|
34
|
+
return raw.replace(/\/+$/, "");
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* A base URL that cannot work is refused up front, and its text is never
|
|
38
|
+
* repeated: `fetch` rejects a URL with userinfo and echoes it in the error, the
|
|
39
|
+
* route is appended after a query string or fragment, and any of them can hold
|
|
40
|
+
* a credential.
|
|
41
|
+
*/
|
|
42
|
+
function baseURLProblem(baseURL) {
|
|
43
|
+
const fix = `Set CLOUDFLARE_CLEF_BASE_URL or pass credentials.cloudflareClef.baseURL to a base URL, or leave both unset to use ${CLOUDFLARE_DEFAULT_BASE_URL}.`;
|
|
44
|
+
try {
|
|
45
|
+
const url = new URL(baseURL);
|
|
46
|
+
if (url.protocol !== "https:" && url.protocol !== "http:") {
|
|
47
|
+
return `The Cloudflare base URL must start with https:// or http://. ${fix}`;
|
|
48
|
+
}
|
|
49
|
+
// `url.search` and `url.hash` are empty for a bare trailing `?` or `#`, which
|
|
50
|
+
// would still put the route into the query or the fragment, so the text is
|
|
51
|
+
// checked as well.
|
|
52
|
+
return url.username ||
|
|
53
|
+
url.password ||
|
|
54
|
+
url.search ||
|
|
55
|
+
url.hash ||
|
|
56
|
+
/[?#]/.test(baseURL)
|
|
57
|
+
? `The Cloudflare base URL must not carry credentials, a query string or a fragment. ${fix}`
|
|
58
|
+
: undefined;
|
|
59
|
+
}
|
|
60
|
+
catch {
|
|
61
|
+
return `The Cloudflare base URL is not a valid absolute URL. ${fix}`;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* The name Cloudflare wants in the path and the body: `clef` or `clef-flash`,
|
|
66
|
+
* given either bare or as `@cf/cloudflare/clef`. Anything with a `/` in it is
|
|
67
|
+
* refused, so a model name cannot reach a different route.
|
|
68
|
+
*/
|
|
69
|
+
function wireModel(model) {
|
|
70
|
+
const bare = model.trim().replace(/^@cf\/cloudflare\//i, "");
|
|
71
|
+
return /^[a-z0-9][a-z0-9._-]*$/i.test(bare) ? bare.toLowerCase() : undefined;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Why this image cannot be sent, or undefined when it can. The data URL is
|
|
75
|
+
* never repeated: it is megabytes of base64. Size limits are left to the
|
|
76
|
+
* server, which refuses an oversized image at once with a clear 422.
|
|
77
|
+
*/
|
|
78
|
+
function imageProblem(dataUrl, position) {
|
|
79
|
+
const type = /^data:([a-z]+\/[a-z0-9.+-]+);base64,/i
|
|
80
|
+
.exec(dataUrl)?.[1]
|
|
81
|
+
?.toLowerCase();
|
|
82
|
+
return type && SUPPORTED_IMAGE_TYPES.has(type)
|
|
83
|
+
? undefined
|
|
84
|
+
: `Image ${position} is not a PNG, JPEG or WebP image, the only formats Cloudflare Clef reads.`;
|
|
85
|
+
}
|
|
86
|
+
/** The API takes a string or structured data as `state`; a number or a boolean is sent as its text. */
|
|
87
|
+
function wireState(state) {
|
|
88
|
+
return typeof state === "number" || typeof state === "boolean"
|
|
89
|
+
? String(state)
|
|
90
|
+
: state;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* `{"success":false,"errors":[{"code","message"}],...}` is Cloudflare's own
|
|
94
|
+
* envelope. A model-side refusal nests a second one inside the message:
|
|
95
|
+
* `AiError: AiError: {"error":{"message","details":{"fieldErrors":{...}}}} (<uuid>)`.
|
|
96
|
+
* Both are flattened to one readable line, with the uuid taken out as the
|
|
97
|
+
* request id. A body that is not this envelope reads as undefined and the
|
|
98
|
+
* caller falls back to the status.
|
|
99
|
+
*/
|
|
100
|
+
function readCloudflareError(body) {
|
|
101
|
+
if (!isRecord(body) || !Array.isArray(body.errors)) {
|
|
102
|
+
return undefined;
|
|
103
|
+
}
|
|
104
|
+
const first = body.errors.find(isRecord);
|
|
105
|
+
if (!first || typeof first.message !== "string" || first.message === "") {
|
|
106
|
+
return undefined;
|
|
107
|
+
}
|
|
108
|
+
let text = first.message.slice(0, MAX_PARSED_MESSAGE_CHARS).trimEnd();
|
|
109
|
+
let requestId;
|
|
110
|
+
const trailing = TRAILING_REQUEST_ID.exec(text);
|
|
111
|
+
if (trailing) {
|
|
112
|
+
requestId = trailing[1];
|
|
113
|
+
text = text.slice(0, trailing.index).trimEnd();
|
|
114
|
+
}
|
|
115
|
+
text = text.replace(/^(?:AiError:\s*|Ai:\s*)+/, "").trim();
|
|
116
|
+
if (text.startsWith("{")) {
|
|
117
|
+
try {
|
|
118
|
+
const inner = JSON.parse(text);
|
|
119
|
+
if (isRecord(inner) && isRecord(inner.error)) {
|
|
120
|
+
const details = isRecord(inner.error.details)
|
|
121
|
+
? inner.error.details
|
|
122
|
+
: undefined;
|
|
123
|
+
const fields = isRecord(details?.fieldErrors)
|
|
124
|
+
? Object.entries(details.fieldErrors).flatMap(([field, problems]) => Array.isArray(problems)
|
|
125
|
+
? problems
|
|
126
|
+
.filter((p) => typeof p === "string")
|
|
127
|
+
.map((p) => `${field}: ${p}`)
|
|
128
|
+
: [])
|
|
129
|
+
: [];
|
|
130
|
+
const head = typeof inner.error.message === "string" ? inner.error.message : text;
|
|
131
|
+
text = fields.length > 0 ? `${head}: ${fields.join("; ")}` : head;
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
catch {
|
|
135
|
+
// not JSON after all; keep the text as it is
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return {
|
|
139
|
+
message: text,
|
|
140
|
+
code: typeof first.code === "number" ? first.code : undefined,
|
|
141
|
+
requestId,
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Only 401 is `authentication`, and so only 401 trips the breaker: Cloudflare
|
|
146
|
+
* answered a token it does not know with a 401 and code 10000 (seen live). A 403
|
|
147
|
+
* was never seen, and no such token was available to provoke one; if Cloudflare
|
|
148
|
+
* uses it for a token that lacks Workers AI permission, that is fixed in the
|
|
149
|
+
* dashboard, and a breaker that latched on it would keep this instance disabled
|
|
150
|
+
* until the process restarted, even after the permission was granted. So a 403
|
|
151
|
+
* is a non-retried `invalid_request` that carries Cloudflare's own message. A
|
|
152
|
+
* request over the context window is a 413, code 5021, and a 429 is "Capacity
|
|
153
|
+
* temporarily exceeded" (code 3040, no Retry-After), which is retryable.
|
|
154
|
+
* Validation refusals are 422, and a malformed body or a wrong route is a 400;
|
|
155
|
+
* none of those is retried.
|
|
156
|
+
*/
|
|
157
|
+
function cloudflareErrorKind(status, code) {
|
|
158
|
+
if (status === 401) {
|
|
159
|
+
return "authentication";
|
|
160
|
+
}
|
|
161
|
+
if (status === 413 || code === 5021) {
|
|
162
|
+
return "max_tokens_exceeded";
|
|
163
|
+
}
|
|
164
|
+
if (status === 429) {
|
|
165
|
+
return "rate_limit";
|
|
166
|
+
}
|
|
167
|
+
if (status === 503) {
|
|
168
|
+
return "overloaded";
|
|
169
|
+
}
|
|
170
|
+
return status >= 500 ? "server" : "invalid_request";
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* Cloudflare Clef — the `decide` inference type only.
|
|
174
|
+
*
|
|
175
|
+
* `@cf/cloudflare/clef` (27B) and `@cf/cloudflare/clef-flash` (9B) answer the
|
|
176
|
+
* same typed `noul` / `choice` / `score` questions as the other decision
|
|
177
|
+
* providers, follow the System One wire, and read images. They are reached
|
|
178
|
+
* through the Workers AI REST API, which wraps every answer in Cloudflare's own
|
|
179
|
+
* `{ result, success, errors }` envelope and puts the model in the URL path.
|
|
180
|
+
*
|
|
181
|
+
* The token and account id are the ones the Cloudflare Workers AI text provider
|
|
182
|
+
* reads (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), so a host that has
|
|
183
|
+
* configured that provider has also configured this one.
|
|
184
|
+
*
|
|
185
|
+
* The Workers AI endpoint ignores text past about 2,048 tokens, far below
|
|
186
|
+
* the documented 64K (hosted service or model: unknown). See `decisionLimits`
|
|
187
|
+
* on the descriptor for the measured figures.
|
|
188
|
+
*
|
|
189
|
+
* @see https://developers.cloudflare.com/workers-ai/models/clef/
|
|
190
|
+
*/
|
|
191
|
+
export class CloudflareClefProvider extends SystemOneDecisionProvider {
|
|
192
|
+
apiKey;
|
|
193
|
+
accountId;
|
|
194
|
+
baseURL;
|
|
195
|
+
constructor(modelName, sdk, _region, credentials) {
|
|
196
|
+
super(modelName, "cloudflare-clef", sdk);
|
|
197
|
+
// A slice that names its own base URL must bring its own token: the shared
|
|
198
|
+
// CLOUDFLARE_API_KEY also runs the host's Workers AI text provider, and it
|
|
199
|
+
// must never be sent as a bearer token to an endpoint a caller chose.
|
|
200
|
+
const ownEndpoint = Boolean(credentials?.baseURL?.trim());
|
|
201
|
+
this.apiKey =
|
|
202
|
+
credentials?.apiKey?.trim() ||
|
|
203
|
+
(ownEndpoint ? "" : (process.env.CLOUDFLARE_API_KEY?.trim() ?? ""));
|
|
204
|
+
this.accountId =
|
|
205
|
+
credentials?.accountId?.trim() ||
|
|
206
|
+
(process.env.CLOUDFLARE_ACCOUNT_ID?.trim() ?? "");
|
|
207
|
+
// `||`, not `??`, so a blank override counts as unset.
|
|
208
|
+
this.baseURL = normalizeBaseURL(credentials?.baseURL?.trim() ||
|
|
209
|
+
process.env.CLOUDFLARE_CLEF_BASE_URL?.trim() ||
|
|
210
|
+
CLOUDFLARE_DEFAULT_BASE_URL);
|
|
211
|
+
// A base URL that `baseURLProblem` refuses is not logged at all: it can
|
|
212
|
+
// carry `user:pass@` or a `?token=`.
|
|
213
|
+
logger.debug("Cloudflare Clef decision provider initialized (decide only)", {
|
|
214
|
+
modelName: this.modelName,
|
|
215
|
+
baseURL: baseURLProblem(this.baseURL)
|
|
216
|
+
? "(invalid)"
|
|
217
|
+
: redactUrlForError(this.baseURL),
|
|
218
|
+
accountConfigured: this.accountId !== "",
|
|
219
|
+
});
|
|
220
|
+
}
|
|
221
|
+
getDefaultModel() {
|
|
222
|
+
return getProviderModel("CLOUDFLARE_CLEF_MODEL", CloudflareClefModels.CLEF);
|
|
223
|
+
}
|
|
224
|
+
vendorLabel() {
|
|
225
|
+
return "Cloudflare";
|
|
226
|
+
}
|
|
227
|
+
vendorDisplayName() {
|
|
228
|
+
return "Cloudflare Clef";
|
|
229
|
+
}
|
|
230
|
+
decisionApiKey() {
|
|
231
|
+
return this.apiKey;
|
|
232
|
+
}
|
|
233
|
+
missingKeyMessage() {
|
|
234
|
+
return "Cloudflare Clef requires an API token with Workers AI permission. Set CLOUDFLARE_API_KEY or pass credentials.cloudflareClef.apiKey.";
|
|
235
|
+
}
|
|
236
|
+
missingConfigMessage() {
|
|
237
|
+
if (this.accountId === "") {
|
|
238
|
+
return "Cloudflare Clef requires the account id. Set CLOUDFLARE_ACCOUNT_ID or pass credentials.cloudflareClef.accountId.";
|
|
239
|
+
}
|
|
240
|
+
if (!ACCOUNT_ID_PATTERN.test(this.accountId)) {
|
|
241
|
+
return "The Cloudflare account id may contain only letters, digits, '-' and '_'. Copy it from the Cloudflare dashboard.";
|
|
242
|
+
}
|
|
243
|
+
return baseURLProblem(this.baseURL);
|
|
244
|
+
}
|
|
245
|
+
/** The model is part of the path, so the endpoint depends on the model asked for. */
|
|
246
|
+
decisionEndpoint(model) {
|
|
247
|
+
return `${this.baseURL}/accounts/${encodeURIComponent(this.accountId)}/ai/run/${MODEL_PREFIX}${encodeURIComponent(wireModel(model) ?? model)}`;
|
|
248
|
+
}
|
|
249
|
+
decisionHeaders() {
|
|
250
|
+
return {
|
|
251
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
252
|
+
"Content-Type": "application/json",
|
|
253
|
+
};
|
|
254
|
+
}
|
|
255
|
+
/**
|
|
256
|
+
* Images travel in their own `images` array, as `data:` URLs, placed before
|
|
257
|
+
* the state by the server. `model` is sent although the path already names
|
|
258
|
+
* it: the API documents it as required, and refuses a body whose `model`
|
|
259
|
+
* differs from the path.
|
|
260
|
+
*/
|
|
261
|
+
buildDecisionBody(state, questions, model, media) {
|
|
262
|
+
const wire = wireModel(model);
|
|
263
|
+
if (!wire) {
|
|
264
|
+
throw this.decisionError({
|
|
265
|
+
kind: "invalid_request",
|
|
266
|
+
message: `"${model.slice(0, 40)}" is not a Cloudflare model name. Use "clef" or "clef-flash".`,
|
|
267
|
+
retryable: false,
|
|
268
|
+
});
|
|
269
|
+
}
|
|
270
|
+
// Workers AI also accepts image/jpg; use the canonical JPEG MIME type.
|
|
271
|
+
const images = (media?.images ?? []).map((dataUrl) => dataUrl.replace(/^data:image\/jpg;base64,/i, "data:image/jpeg;base64,"));
|
|
272
|
+
images.forEach((dataUrl, index) => {
|
|
273
|
+
const problem = imageProblem(dataUrl, index + 1);
|
|
274
|
+
if (problem) {
|
|
275
|
+
throw this.decisionError({
|
|
276
|
+
kind: "invalid_request",
|
|
277
|
+
message: problem,
|
|
278
|
+
retryable: false,
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
});
|
|
282
|
+
// The canonical JPEG prefix is one byte longer than image/jpg. Keep the
|
|
283
|
+
// prepared media metadata aligned with the URLs actually sent.
|
|
284
|
+
if (media) {
|
|
285
|
+
media.images = images;
|
|
286
|
+
media.bytes = images.reduce((total, image) => total + Buffer.byteLength(image), 0);
|
|
287
|
+
}
|
|
288
|
+
return {
|
|
289
|
+
model: wire,
|
|
290
|
+
state: wireState(state),
|
|
291
|
+
questions,
|
|
292
|
+
...(images.length > 0 ? { images } : {}),
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
/** A success is `{ result: { model, answers, usage }, success: true }`. */
|
|
296
|
+
readDecisionPayload(payload) {
|
|
297
|
+
return isRecord(payload) && isRecord(payload.result)
|
|
298
|
+
? payload.result
|
|
299
|
+
: payload;
|
|
300
|
+
}
|
|
301
|
+
parseDecisionError(status, payload, requestId) {
|
|
302
|
+
const fallback = `Cloudflare request failed with HTTP ${status}`;
|
|
303
|
+
const parsed = readCloudflareError(payload);
|
|
304
|
+
const kind = cloudflareErrorKind(status, parsed?.code);
|
|
305
|
+
let message = redactCredentials(parsed?.message ?? fallback, this.apiKey)
|
|
306
|
+
.trim()
|
|
307
|
+
.slice(0, MAX_ERROR_MESSAGE_CHARS) || fallback;
|
|
308
|
+
if (kind === "max_tokens_exceeded") {
|
|
309
|
+
message +=
|
|
310
|
+
" Cloudflare counts the whole request, image data included, at about four characters per token.";
|
|
311
|
+
}
|
|
312
|
+
else if (parsed?.code === 7000) {
|
|
313
|
+
message += `. Check the model name (Clef is served as "clef" and "clef-flash") and the base URL, which must end in /client/v4 and name your account.`;
|
|
314
|
+
}
|
|
315
|
+
return {
|
|
316
|
+
kind,
|
|
317
|
+
message,
|
|
318
|
+
status,
|
|
319
|
+
requestId: requestId ?? parsed?.requestId,
|
|
320
|
+
retryable: kind === "rate_limit" || kind === "overloaded" || kind === "server",
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
/**
|
|
324
|
+
* `cf-ai-req-id` is on every answer that reached the model, success or
|
|
325
|
+
* refusal. A 401 and a wrong-path 400 never reach it, so they carry only the
|
|
326
|
+
* edge's `cf-ray`.
|
|
327
|
+
*/
|
|
328
|
+
readRequestId(headers) {
|
|
329
|
+
return headers.get("cf-ai-req-id") ?? headers.get("cf-ray") ?? undefined;
|
|
330
|
+
}
|
|
331
|
+
}
|
|
@@ -59,7 +59,11 @@ export declare abstract class SystemOneDecisionProvider extends BaseProvider {
|
|
|
59
59
|
* nothing is missing.
|
|
60
60
|
*/
|
|
61
61
|
protected missingConfigMessage(): string | undefined;
|
|
62
|
-
|
|
62
|
+
/**
|
|
63
|
+
* Where the request goes. The model asked for is passed in because one
|
|
64
|
+
* vendor names it in the URL path; every other provider ignores it.
|
|
65
|
+
*/
|
|
66
|
+
protected abstract decisionEndpoint(model: string): string;
|
|
63
67
|
protected abstract decisionHeaders(): Record<string, string>;
|
|
64
68
|
protected abstract buildDecisionBody(state: DecisionState, questions: Record<string, Record<string, unknown>>, model: string, media?: DecisionPreparedMedia): Record<string, unknown>;
|
|
65
69
|
protected abstract parseDecisionError(status: number, payload: unknown, requestId: string | undefined): DecisionError;
|
|
@@ -81,6 +85,13 @@ export declare abstract class SystemOneDecisionProvider extends BaseProvider {
|
|
|
81
85
|
* grows with the number of questions scales the descriptor's allowance here.
|
|
82
86
|
*/
|
|
83
87
|
protected defaultTimeoutMs(_questionCount: number): number | undefined;
|
|
88
|
+
/**
|
|
89
|
+
* The part of a successful response that holds `answers`, `usage` and
|
|
90
|
+
* `model`. The default is the response itself; a vendor that wraps every
|
|
91
|
+
* answer in an envelope returns what is inside it. Errors are not passed
|
|
92
|
+
* through here: they are read from the whole response.
|
|
93
|
+
*/
|
|
94
|
+
protected readDecisionPayload(payload: unknown): unknown;
|
|
84
95
|
/** Per-question confidence a transport reports outside the answer objects. */
|
|
85
96
|
protected reportedConfidence(_payload: unknown): Record<string, number>;
|
|
86
97
|
protected resolveResponseModel(payload: Record<string, unknown>, requestedModel: string): string;
|
|
@@ -6,7 +6,7 @@ import { ProviderError } from "../types/index.js";
|
|
|
6
6
|
import { prepareDecisionMedia } from "../utils/decisionMedia.js";
|
|
7
7
|
import { logger } from "../utils/logger.js";
|
|
8
8
|
import { redactUrlsInText } from "../utils/logSanitize.js";
|
|
9
|
-
import { estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
|
|
9
|
+
import { CHARS_PER_TOKEN, estimateTokens, serializeForEstimate, } from "../utils/tokenEstimation.js";
|
|
10
10
|
/**
|
|
11
11
|
* Generous enough for a cold start (measured at 2.0–2.7s after idle on Jev)
|
|
12
12
|
* while still bounded. Every internal caller is fail-open, so this only bites
|
|
@@ -177,15 +177,52 @@ const sleep = (ms, signal) => new Promise((resolve, reject) => {
|
|
|
177
177
|
* tokenizer, so when a provider declares a rate, non-ASCII characters are
|
|
178
178
|
* counted at it instead. ASCII text is estimated exactly as before.
|
|
179
179
|
*/
|
|
180
|
-
function estimateDecisionStateTokens(text, nonAsciiTokensPerChar) {
|
|
181
|
-
|
|
180
|
+
function estimateDecisionStateTokens(text, nonAsciiTokensPerChar, rates = {}) {
|
|
181
|
+
const { digit, symbol, astral } = rates;
|
|
182
|
+
if (nonAsciiTokensPerChar === undefined &&
|
|
183
|
+
digit === undefined &&
|
|
184
|
+
symbol === undefined &&
|
|
185
|
+
astral === undefined) {
|
|
182
186
|
return estimateTokens(text);
|
|
183
187
|
}
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
188
|
+
// Each class of character is counted once, at its own rate. A class with no
|
|
189
|
+
// declared rate stays in the ordinary four-characters-a-token estimate (or,
|
|
190
|
+
// for non-ASCII, at that estimate's per-character rate), so a provider that
|
|
191
|
+
// declares none of the extra rates is estimated exactly as before.
|
|
192
|
+
const isDigit = (c) => c >= "0" && c <= "9";
|
|
193
|
+
const isSymbol = (c) => /[!-/:-@[-`{-~]/.test(c);
|
|
194
|
+
let digits = 0;
|
|
195
|
+
let symbols = 0;
|
|
196
|
+
let astrals = 0;
|
|
197
|
+
let nonAscii = 0;
|
|
198
|
+
const rest = [];
|
|
199
|
+
for (const c of text) {
|
|
200
|
+
const code = c.codePointAt(0) ?? 0;
|
|
201
|
+
if (code > 0xffff && astral !== undefined) {
|
|
202
|
+
astrals += 1;
|
|
203
|
+
}
|
|
204
|
+
else if (code > 0x7f) {
|
|
205
|
+
nonAscii += 1;
|
|
206
|
+
}
|
|
207
|
+
else if (digit !== undefined && isDigit(c)) {
|
|
208
|
+
digits += 1;
|
|
209
|
+
}
|
|
210
|
+
else if (symbol !== undefined && isSymbol(c)) {
|
|
211
|
+
symbols += 1;
|
|
212
|
+
}
|
|
213
|
+
else {
|
|
214
|
+
rest.push(c);
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
// Digits, punctuation and astral characters (emoji) are charged separately
|
|
218
|
+
// only when the provider declares a rate for them: a tokenizer that reads each
|
|
219
|
+
// digit, and most punctuation, on its own makes a number-heavy or JSON-heavy
|
|
220
|
+
// state several times longer than four characters a token suggests.
|
|
221
|
+
return (estimateTokens(rest.join("")) +
|
|
222
|
+
Math.ceil(digits * (digit ?? 0)) +
|
|
223
|
+
Math.ceil(symbols * (symbol ?? 0)) +
|
|
224
|
+
Math.ceil(astrals * (astral ?? 0)) +
|
|
225
|
+
Math.ceil(nonAscii * (nonAsciiTokensPerChar ?? 1 / CHARS_PER_TOKEN)));
|
|
189
226
|
}
|
|
190
227
|
/**
|
|
191
228
|
* The shared half of every "System One" decision provider — a model that takes
|
|
@@ -251,6 +288,15 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
251
288
|
defaultTimeoutMs(_questionCount) {
|
|
252
289
|
return this.getDescriptorDecideMs();
|
|
253
290
|
}
|
|
291
|
+
/**
|
|
292
|
+
* The part of a successful response that holds `answers`, `usage` and
|
|
293
|
+
* `model`. The default is the response itself; a vendor that wraps every
|
|
294
|
+
* answer in an envelope returns what is inside it. Errors are not passed
|
|
295
|
+
* through here: they are read from the whole response.
|
|
296
|
+
*/
|
|
297
|
+
readDecisionPayload(payload) {
|
|
298
|
+
return payload;
|
|
299
|
+
}
|
|
254
300
|
/** Per-question confidence a transport reports outside the answer objects. */
|
|
255
301
|
reportedConfidence(_payload) {
|
|
256
302
|
return {};
|
|
@@ -354,7 +400,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
354
400
|
const signal = request.signal
|
|
355
401
|
? AbortSignal.any([request.signal, timeout])
|
|
356
402
|
: timeout;
|
|
357
|
-
const response = await this.proxyFetch(this.decisionEndpoint(), {
|
|
403
|
+
const response = await this.proxyFetch(this.decisionEndpoint(resolvedModel), {
|
|
358
404
|
method: "POST",
|
|
359
405
|
headers: this.decisionHeaders(),
|
|
360
406
|
body,
|
|
@@ -393,7 +439,8 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
393
439
|
await sleep(retryAfterMs ?? 2 ** attempt * 250 + Math.random() * 250, request.signal);
|
|
394
440
|
continue;
|
|
395
441
|
}
|
|
396
|
-
|
|
442
|
+
const decoded = this.readDecisionPayload(payload);
|
|
443
|
+
if (!isRecord(decoded) || !isRecord(decoded.answers)) {
|
|
397
444
|
throw this.decisionError({
|
|
398
445
|
kind: "server",
|
|
399
446
|
message: `${label} returned a response without an answers map.`,
|
|
@@ -402,9 +449,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
402
449
|
retryable: false,
|
|
403
450
|
});
|
|
404
451
|
}
|
|
405
|
-
const reported = this.reportedConfidence(
|
|
452
|
+
const reported = this.reportedConfidence(decoded);
|
|
406
453
|
const answers = {};
|
|
407
|
-
for (const [id, raw] of Object.entries(
|
|
454
|
+
for (const [id, raw] of Object.entries(decoded.answers)) {
|
|
408
455
|
const parsed = parseDecisionAnswer(raw, reported[id]);
|
|
409
456
|
if (parsed) {
|
|
410
457
|
answers[id] = parsed;
|
|
@@ -415,13 +462,13 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
415
462
|
});
|
|
416
463
|
}
|
|
417
464
|
}
|
|
418
|
-
const usage = isRecord(
|
|
465
|
+
const usage = isRecord(decoded.usage) ? decoded.usage : {};
|
|
419
466
|
return {
|
|
420
467
|
// `resolvedModel`, not `this.modelName`: when the caller pinned a
|
|
421
468
|
// model for this one request and the response omits its own, the
|
|
422
469
|
// instance default would be reported instead of the model actually
|
|
423
470
|
// asked for.
|
|
424
|
-
model: this.resolveResponseModel(
|
|
471
|
+
model: this.resolveResponseModel(decoded, resolvedModel),
|
|
425
472
|
provider: this.providerName,
|
|
426
473
|
answers,
|
|
427
474
|
// Two spellings for one field. The System One wire sends
|
|
@@ -485,8 +532,9 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
485
532
|
* applies to.
|
|
486
533
|
*/
|
|
487
534
|
assertWithinRequestBytes(body) {
|
|
488
|
-
const
|
|
489
|
-
?.decisionLimits?.media
|
|
535
|
+
const media = PROVIDER_DESCRIPTORS_BY_NAME.get(this.providerName)
|
|
536
|
+
?.decisionLimits?.media;
|
|
537
|
+
const limit = media?.maxRequestBytes;
|
|
490
538
|
if (limit === undefined) {
|
|
491
539
|
return;
|
|
492
540
|
}
|
|
@@ -494,7 +542,7 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
494
542
|
if (bytes > limit) {
|
|
495
543
|
throw this.decisionError({
|
|
496
544
|
kind: "invalid_request",
|
|
497
|
-
message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images, or a shorter video.`,
|
|
545
|
+
message: `The request is ${bytes} bytes; ${this.vendorLabel()} accepts at most ${limit}. Send fewer or smaller images${media?.video ? ", or a shorter video" : ""}.`,
|
|
498
546
|
retryable: false,
|
|
499
547
|
});
|
|
500
548
|
}
|
|
@@ -520,7 +568,11 @@ export class SystemOneDecisionProvider extends BaseProvider {
|
|
|
520
568
|
}
|
|
521
569
|
const modelLimits = limits.models?.[model];
|
|
522
570
|
const maxStateTokens = modelLimits?.maxStateTokens ?? limits.maxStateTokens;
|
|
523
|
-
const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar
|
|
571
|
+
const stateTokens = estimateDecisionStateTokens(serializeForEstimate(request.state), modelLimits?.nonAsciiTokensPerChar ?? limits.nonAsciiTokensPerChar, {
|
|
572
|
+
digit: limits.digitTokensPerChar,
|
|
573
|
+
symbol: limits.symbolTokensPerChar,
|
|
574
|
+
astral: limits.astralTokensPerChar,
|
|
575
|
+
});
|
|
524
576
|
if (stateTokens > maxStateTokens) {
|
|
525
577
|
throw this.decisionError({
|
|
526
578
|
kind: "max_tokens_exceeded",
|
package/dist/types/decision.d.ts
CHANGED
|
@@ -153,6 +153,28 @@ export type DecisionLimits = {
|
|
|
153
153
|
* estimate for every character.
|
|
154
154
|
*/
|
|
155
155
|
nonAsciiTokensPerChar?: number;
|
|
156
|
+
/**
|
|
157
|
+
* Tokens charged per ASCII digit. A tokenizer that reads every digit as its
|
|
158
|
+
* own token makes numbers, ids and timestamps several times longer than the
|
|
159
|
+
* default estimate of four characters per token. Absent = digits are
|
|
160
|
+
* estimated like any other ASCII character.
|
|
161
|
+
*/
|
|
162
|
+
digitTokensPerChar?: number;
|
|
163
|
+
/**
|
|
164
|
+
* Tokens charged per ASCII punctuation or symbol character (`,` `.` `{` `"`
|
|
165
|
+
* `:` and the like). The same tokenizers that read each digit alone read most
|
|
166
|
+
* punctuation alone too, so JSON, logs and lists of numbers run far above four
|
|
167
|
+
* characters a token. Absent = punctuation is estimated like any other ASCII
|
|
168
|
+
* character.
|
|
169
|
+
*/
|
|
170
|
+
symbolTokensPerChar?: number;
|
|
171
|
+
/**
|
|
172
|
+
* Tokens charged per character outside the Basic Multilingual Plane (emoji
|
|
173
|
+
* and the like), which is counted separately from `nonAsciiTokensPerChar`
|
|
174
|
+
* because it costs about twice as much. Absent = charged at
|
|
175
|
+
* `nonAsciiTokensPerChar`.
|
|
176
|
+
*/
|
|
177
|
+
astralTokensPerChar?: number;
|
|
156
178
|
/** Per-model limits, keyed by model id; each field overrides the one above. */
|
|
157
179
|
models?: Readonly<Record<string, {
|
|
158
180
|
maxStateTokens: number;
|
|
@@ -583,6 +583,19 @@ export type NeurolinkCredentials = {
|
|
|
583
583
|
apiKey?: string;
|
|
584
584
|
baseURL?: string;
|
|
585
585
|
};
|
|
586
|
+
/**
|
|
587
|
+
* Cloudflare Clef — the `decide` inference type, reached through Workers AI.
|
|
588
|
+
* The token and account id are the ones the `cloudflare` text provider reads
|
|
589
|
+
* (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), but this slice is separate
|
|
590
|
+
* so the two cannot be mixed up. Both are required. `baseURL` is optional and
|
|
591
|
+
* defaults to `https://api.cloudflare.com/client/v4`; requests go to
|
|
592
|
+
* `<baseURL>/accounts/<accountId>/ai/run/@cf/cloudflare/<model>`.
|
|
593
|
+
*/
|
|
594
|
+
cloudflareClef?: {
|
|
595
|
+
apiKey?: string;
|
|
596
|
+
accountId?: string;
|
|
597
|
+
baseURL?: string;
|
|
598
|
+
};
|
|
586
599
|
};
|
|
587
600
|
/**
|
|
588
601
|
* Voyage AI /embeddings response shape.
|
|
@@ -2325,6 +2338,8 @@ export type ProviderDescriptor = {
|
|
|
2325
2338
|
extraRequired?: readonly string[];
|
|
2326
2339
|
/** Alternate ways to satisfy extraRequired when it isn't a plain env-var list (e.g. Vertex's file-path-OR-individual-fields auth). Each entry is either a single env var name (satisfied alone) or a nested array of names that must ALL be present together (e.g. Vertex's GOOGLE_AUTH_CLIENT_EMAIL + GOOGLE_AUTH_PRIVATE_KEY pair, which is only valid as a pair). Evaluate with `satisfiesFallbacks()` (providerConfig.ts) rather than re-deriving this logic at each call site. */
|
|
2327
2340
|
extraRequiredFallbacks?: readonly (string | readonly string[])[];
|
|
2341
|
+
/** For an `extraRequired` name that can also be given in `credentials.<credentialsKey>`: the field of that slice that stands for it (e.g. `{ CLOUDFLARE_ACCOUNT_ID: "accountId" }`). A base URL needs no entry; it is matched through `baseURL` above. */
|
|
2342
|
+
extraRequiredCredentialFields?: Readonly<Record<string, string>>;
|
|
2328
2343
|
/** True when the provider is usable with zero configuration (local runtime with a documented default URL, or a documented non-secret default like LiteLLM's "sk-anything"). */
|
|
2329
2344
|
optional?: boolean;
|
|
2330
2345
|
};
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* Centralized model choices for CLI commands
|
|
3
3
|
* Derives choices from model enums to ensure consistency
|
|
4
4
|
*/
|
|
5
|
-
import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, ReplicateModels, } from "../constants/enums.js";
|
|
5
|
+
import { AIProviderName, OpenAIModels, AnthropicModels, GoogleAIModels, BedrockModels, VertexModels, OllamaModels, AzureOpenAIModels, LiteLLMModels, SageMakerModels, OpenRouterModels, NvidiaNimModels, CohereModels, VoyageModels, JinaModels, StabilityModels, IdeogramModels, RecraftModels, TypeSafeModels, LayaModels, XorModels, PerplexityDeciderModels, CloudflareClefModels, ReplicateModels, } from "../constants/enums.js";
|
|
6
6
|
import { getCatalogJsonEntries } from "../providers/catalog/loader.js";
|
|
7
7
|
/**
|
|
8
8
|
* Looks up the JSON catalog entry for a provider, if it is one of the 11
|
|
@@ -399,6 +399,16 @@ const TOP_MODELS_CONFIG = {
|
|
|
399
399
|
description: "Recommended - Perplexity decision model that also reads images",
|
|
400
400
|
},
|
|
401
401
|
],
|
|
402
|
+
[AIProviderName.CLOUDFLARE_CLEF]: [
|
|
403
|
+
{
|
|
404
|
+
model: CloudflareClefModels.CLEF,
|
|
405
|
+
description: "Recommended - Cloudflare Clef 27B decision model that also reads images (state limited to ~2,000 tokens)",
|
|
406
|
+
},
|
|
407
|
+
{
|
|
408
|
+
model: CloudflareClefModels.CLEF_FLASH,
|
|
409
|
+
description: "Clef-flash 9B - the faster, cheaper Cloudflare decision model (same ~2,000-token state limit)",
|
|
410
|
+
},
|
|
411
|
+
],
|
|
402
412
|
[AIProviderName.AUTO]: [],
|
|
403
413
|
};
|
|
404
414
|
/**
|
|
@@ -441,6 +451,7 @@ export const DEFAULT_MODELS = {
|
|
|
441
451
|
[AIProviderName.LAYA]: LayaModels.TYPED_DECISIONS,
|
|
442
452
|
[AIProviderName.XOR]: XorModels.XOR_1_1,
|
|
443
453
|
[AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels.PPLX_DECIDER_V1_27B,
|
|
454
|
+
[AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels.CLEF,
|
|
444
455
|
};
|
|
445
456
|
/**
|
|
446
457
|
* Model enum mappings for getAllModels.
|
|
@@ -476,6 +487,7 @@ const MODEL_ENUMS = {
|
|
|
476
487
|
[AIProviderName.LAYA]: LayaModels,
|
|
477
488
|
[AIProviderName.XOR]: XorModels,
|
|
478
489
|
[AIProviderName.PERPLEXITY_DECIDER]: PerplexityDeciderModels,
|
|
490
|
+
[AIProviderName.CLOUDFLARE_CLEF]: CloudflareClefModels,
|
|
479
491
|
[AIProviderName.AUTO]: null,
|
|
480
492
|
};
|
|
481
493
|
/**
|
package/dist/utils/pricing.js
CHANGED
|
@@ -788,6 +788,18 @@ const PRICING = {
|
|
|
788
788
|
_default: { input: 0.04 / 1_000_000, output: 0 },
|
|
789
789
|
"pplx-decider-v1-27b": { input: 0.04 / 1_000_000, output: 0 },
|
|
790
790
|
},
|
|
791
|
+
"cloudflare-clef": {
|
|
792
|
+
// Workers AI bills Clef per million INPUT tokens ($0.24 for clef, $0.09 for
|
|
793
|
+
// clef-flash); no output price is listed. `output: 0` is a price, not a token
|
|
794
|
+
// count, so do not "correct" it from `usage.output_tokens`. The `cf-ai-neurons`
|
|
795
|
+
// response header was checked against the published neuron rates on
|
|
796
|
+
// 2026-10-03 and matched them exactly (21,818 and 8,182 neurons per million
|
|
797
|
+
// input tokens). Rates from Cloudflare's Workers AI pricing page. An unknown
|
|
798
|
+
// model name is priced as clef, the dearer of the two.
|
|
799
|
+
_default: { input: 0.24 / 1_000_000, output: 0 },
|
|
800
|
+
clef: { input: 0.24 / 1_000_000, output: 0 },
|
|
801
|
+
"clef-flash": { input: 0.09 / 1_000_000, output: 0 },
|
|
802
|
+
},
|
|
791
803
|
stability: {
|
|
792
804
|
// Stability AI bills per image; symbolic per-token rate.
|
|
793
805
|
_default: { input: 0, output: 0.04 / 1_000 },
|
|
@@ -416,3 +416,10 @@ export declare function createXorConfig(): ProviderConfigOptions;
|
|
|
416
416
|
* provider and has a public endpoint, so the key alone configures it.
|
|
417
417
|
*/
|
|
418
418
|
export declare function createPerplexityDeciderConfig(): ProviderConfigOptions;
|
|
419
|
+
/**
|
|
420
|
+
* Cloudflare Clef — the `decide` inference type, reached through Workers AI,
|
|
421
|
+
* not the Workers AI text models. It reads the same CLOUDFLARE_API_KEY and
|
|
422
|
+
* CLOUDFLARE_ACCOUNT_ID as the `cloudflare` text provider; the endpoint is
|
|
423
|
+
* Cloudflare's own, so the two together configure it.
|
|
424
|
+
*/
|
|
425
|
+
export declare function createCloudflareClefConfig(): ProviderConfigOptions;
|