@juspay/neurolink 12.46.0 → 12.47.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -2
- package/README.md +75 -50
- package/dist/browser/neurolink.min.js +426 -426
- package/dist/cli/commands/decide.js +2 -2
- package/dist/cli/commands/setup.js +2 -1
- package/dist/cli/factories/commandFactory.js +1 -1
- package/dist/constants/enums.d.ts +16 -0
- package/dist/constants/enums.js +17 -0
- package/dist/factories/providerDescriptors.js +84 -5
- package/dist/factories/providerRegistry.js +10 -1
- package/dist/models/manifestRegistry.js +2 -0
- package/dist/models/manifests/cloudflareClef.d.ts +16 -0
- package/dist/models/manifests/cloudflareClef.js +42 -0
- package/dist/providers/anthropic/client.js +6 -1
- package/dist/providers/catalog/huggingface.json +2 -1
- package/dist/providers/catalog/loader.js +5 -1
- package/dist/providers/catalog/schema.d.ts +1 -0
- package/dist/providers/catalog/schema.js +8 -0
- package/dist/providers/cloudflareClef.d.ts +52 -0
- package/dist/providers/cloudflareClef.js +331 -0
- package/dist/providers/ideogram.js +9 -30
- package/dist/providers/llamaCpp.js +2 -1
- package/dist/providers/openAI/client.js +4 -5
- package/dist/providers/openaiChatCompletionsBase.js +12 -7
- package/dist/providers/openaiChatCompletionsClient.js +25 -7
- package/dist/providers/recraft.js +9 -28
- package/dist/providers/systemOneDecision.d.ts +12 -1
- package/dist/providers/systemOneDecision.js +70 -18
- package/dist/types/decision.d.ts +22 -0
- package/dist/types/providerCatalog.d.ts +2 -0
- package/dist/types/providers.d.ts +15 -0
- package/dist/utils/modelChoices.js +13 -1
- package/dist/utils/pricing.js +12 -0
- package/dist/utils/providerConfig.d.ts +7 -0
- package/dist/utils/providerConfig.js +19 -0
- package/dist/utils/providerRetry.d.ts +5 -0
- package/dist/utils/providerRetry.js +18 -12
- package/docs-site/static/search-index.json +580 -557
- package/package.json +2 -1
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
import { CloudflareClefModels } from "../constants/enums.js";
|
|
2
|
+
import { logger } from "../utils/logger.js";
|
|
3
|
+
import { redactUrlForError } from "../utils/logSanitize.js";
|
|
4
|
+
import { getProviderModel } from "../utils/providerConfig.js";
|
|
5
|
+
import { isRecord, redactCredentials, SystemOneDecisionProvider, } from "./systemOneDecision.js";
|
|
6
|
+
const CLOUDFLARE_DEFAULT_BASE_URL = "https://api.cloudflare.com/client/v4";
|
|
7
|
+
const MODEL_PREFIX = "@cf/cloudflare/";
|
|
8
|
+
/**
|
|
9
|
+
* Cloudflare account ids are 32 hex digits. The pattern is wider so that a
|
|
10
|
+
* change of format needs no release, but it still keeps `/`, `.`, `?` and
|
|
11
|
+
* whitespace out of the URL path the id is placed in.
|
|
12
|
+
*/
|
|
13
|
+
const ACCOUNT_ID_PATTERN = /^[A-Za-z0-9_-]{1,64}$/;
|
|
14
|
+
/** The only image formats the API reads; anything else is a 422. */
|
|
15
|
+
const SUPPORTED_IMAGE_TYPES = new Set([
|
|
16
|
+
"image/png",
|
|
17
|
+
"image/jpeg",
|
|
18
|
+
"image/webp",
|
|
19
|
+
]);
|
|
20
|
+
/**
|
|
21
|
+
* The uuid Workers AI appends to an `AiError` message, in parentheses. It starts
|
|
22
|
+
* at a literal `(`, with no leading `\s*`: that one backtracks quadratically on
|
|
23
|
+
* a long run of spaces, and the message is whatever the server sent.
|
|
24
|
+
*/
|
|
25
|
+
const TRAILING_REQUEST_ID = /\(([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})\)\s*$/i;
|
|
26
|
+
const MAX_ERROR_MESSAGE_CHARS = 500;
|
|
27
|
+
/**
|
|
28
|
+
* An error message is cut to this before any pattern runs on it, so a hostile or
|
|
29
|
+
* garbled body cannot hold the event loop. Cloudflare's own messages are far
|
|
30
|
+
* shorter (the longest seen was about 350 characters).
|
|
31
|
+
*/
|
|
32
|
+
const MAX_PARSED_MESSAGE_CHARS = 4096;
|
|
33
|
+
function normalizeBaseURL(raw) {
|
|
34
|
+
return raw.replace(/\/+$/, "");
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* A base URL that cannot work is refused up front, and its text is never
|
|
38
|
+
* repeated: `fetch` rejects a URL with userinfo and echoes it in the error, the
|
|
39
|
+
* route is appended after a query string or fragment, and any of them can hold
|
|
40
|
+
* a credential.
|
|
41
|
+
*/
|
|
42
|
+
function baseURLProblem(baseURL) {
|
|
43
|
+
const fix = `Set CLOUDFLARE_CLEF_BASE_URL or pass credentials.cloudflareClef.baseURL to a base URL, or leave both unset to use ${CLOUDFLARE_DEFAULT_BASE_URL}.`;
|
|
44
|
+
try {
|
|
45
|
+
const url = new URL(baseURL);
|
|
46
|
+
if (url.protocol !== "https:" && url.protocol !== "http:") {
|
|
47
|
+
return `The Cloudflare base URL must start with https:// or http://. ${fix}`;
|
|
48
|
+
}
|
|
49
|
+
// `url.search` and `url.hash` are empty for a bare trailing `?` or `#`, which
|
|
50
|
+
// would still put the route into the query or the fragment, so the text is
|
|
51
|
+
// checked as well.
|
|
52
|
+
return url.username ||
|
|
53
|
+
url.password ||
|
|
54
|
+
url.search ||
|
|
55
|
+
url.hash ||
|
|
56
|
+
/[?#]/.test(baseURL)
|
|
57
|
+
? `The Cloudflare base URL must not carry credentials, a query string or a fragment. ${fix}`
|
|
58
|
+
: undefined;
|
|
59
|
+
}
|
|
60
|
+
catch {
|
|
61
|
+
return `The Cloudflare base URL is not a valid absolute URL. ${fix}`;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* The name Cloudflare wants in the path and the body: `clef` or `clef-flash`,
|
|
66
|
+
* given either bare or as `@cf/cloudflare/clef`. Anything with a `/` in it is
|
|
67
|
+
* refused, so a model name cannot reach a different route.
|
|
68
|
+
*/
|
|
69
|
+
function wireModel(model) {
|
|
70
|
+
const bare = model.trim().replace(/^@cf\/cloudflare\//i, "");
|
|
71
|
+
return /^[a-z0-9][a-z0-9._-]*$/i.test(bare) ? bare.toLowerCase() : undefined;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Why this image cannot be sent, or undefined when it can. The data URL is
|
|
75
|
+
* never repeated: it is megabytes of base64. Size limits are left to the
|
|
76
|
+
* server, which refuses an oversized image at once with a clear 422.
|
|
77
|
+
*/
|
|
78
|
+
function imageProblem(dataUrl, position) {
|
|
79
|
+
const type = /^data:([a-z]+\/[a-z0-9.+-]+);base64,/i
|
|
80
|
+
.exec(dataUrl)?.[1]
|
|
81
|
+
?.toLowerCase();
|
|
82
|
+
return type && SUPPORTED_IMAGE_TYPES.has(type)
|
|
83
|
+
? undefined
|
|
84
|
+
: `Image ${position} is not a PNG, JPEG or WebP image, the only formats Cloudflare Clef reads.`;
|
|
85
|
+
}
|
|
86
|
+
/** The API takes a string or structured data as `state`; a number or a boolean is sent as its text. */
|
|
87
|
+
function wireState(state) {
|
|
88
|
+
return typeof state === "number" || typeof state === "boolean"
|
|
89
|
+
? String(state)
|
|
90
|
+
: state;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* `{"success":false,"errors":[{"code","message"}],...}` is Cloudflare's own
|
|
94
|
+
* envelope. A model-side refusal nests a second one inside the message:
|
|
95
|
+
* `AiError: AiError: {"error":{"message","details":{"fieldErrors":{...}}}} (<uuid>)`.
|
|
96
|
+
* Both are flattened to one readable line, with the uuid taken out as the
|
|
97
|
+
* request id. A body that is not this envelope reads as undefined and the
|
|
98
|
+
* caller falls back to the status.
|
|
99
|
+
*/
|
|
100
|
+
function readCloudflareError(body) {
|
|
101
|
+
if (!isRecord(body) || !Array.isArray(body.errors)) {
|
|
102
|
+
return undefined;
|
|
103
|
+
}
|
|
104
|
+
const first = body.errors.find(isRecord);
|
|
105
|
+
if (!first || typeof first.message !== "string" || first.message === "") {
|
|
106
|
+
return undefined;
|
|
107
|
+
}
|
|
108
|
+
let text = first.message.slice(0, MAX_PARSED_MESSAGE_CHARS).trimEnd();
|
|
109
|
+
let requestId;
|
|
110
|
+
const trailing = TRAILING_REQUEST_ID.exec(text);
|
|
111
|
+
if (trailing) {
|
|
112
|
+
requestId = trailing[1];
|
|
113
|
+
text = text.slice(0, trailing.index).trimEnd();
|
|
114
|
+
}
|
|
115
|
+
text = text.replace(/^(?:AiError:\s*|Ai:\s*)+/, "").trim();
|
|
116
|
+
if (text.startsWith("{")) {
|
|
117
|
+
try {
|
|
118
|
+
const inner = JSON.parse(text);
|
|
119
|
+
if (isRecord(inner) && isRecord(inner.error)) {
|
|
120
|
+
const details = isRecord(inner.error.details)
|
|
121
|
+
? inner.error.details
|
|
122
|
+
: undefined;
|
|
123
|
+
const fields = isRecord(details?.fieldErrors)
|
|
124
|
+
? Object.entries(details.fieldErrors).flatMap(([field, problems]) => Array.isArray(problems)
|
|
125
|
+
? problems
|
|
126
|
+
.filter((p) => typeof p === "string")
|
|
127
|
+
.map((p) => `${field}: ${p}`)
|
|
128
|
+
: [])
|
|
129
|
+
: [];
|
|
130
|
+
const head = typeof inner.error.message === "string" ? inner.error.message : text;
|
|
131
|
+
text = fields.length > 0 ? `${head}: ${fields.join("; ")}` : head;
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
catch {
|
|
135
|
+
// not JSON after all; keep the text as it is
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return {
|
|
139
|
+
message: text,
|
|
140
|
+
code: typeof first.code === "number" ? first.code : undefined,
|
|
141
|
+
requestId,
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Only 401 is `authentication`, and so only 401 trips the breaker: Cloudflare
|
|
146
|
+
* answered a token it does not know with a 401 and code 10000 (seen live). A 403
|
|
147
|
+
* was never seen, and no such token was available to provoke one; if Cloudflare
|
|
148
|
+
* uses it for a token that lacks Workers AI permission, that is fixed in the
|
|
149
|
+
* dashboard, and a breaker that latched on it would keep this instance disabled
|
|
150
|
+
* until the process restarted, even after the permission was granted. So a 403
|
|
151
|
+
* is a non-retried `invalid_request` that carries Cloudflare's own message. A
|
|
152
|
+
* request over the context window is a 413, code 5021, and a 429 is "Capacity
|
|
153
|
+
* temporarily exceeded" (code 3040, no Retry-After), which is retryable.
|
|
154
|
+
* Validation refusals are 422, and a malformed body or a wrong route is a 400;
|
|
155
|
+
* none of those is retried.
|
|
156
|
+
*/
|
|
157
|
+
function cloudflareErrorKind(status, code) {
|
|
158
|
+
if (status === 401) {
|
|
159
|
+
return "authentication";
|
|
160
|
+
}
|
|
161
|
+
if (status === 413 || code === 5021) {
|
|
162
|
+
return "max_tokens_exceeded";
|
|
163
|
+
}
|
|
164
|
+
if (status === 429) {
|
|
165
|
+
return "rate_limit";
|
|
166
|
+
}
|
|
167
|
+
if (status === 503) {
|
|
168
|
+
return "overloaded";
|
|
169
|
+
}
|
|
170
|
+
return status >= 500 ? "server" : "invalid_request";
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* Cloudflare Clef — the `decide` inference type only.
|
|
174
|
+
*
|
|
175
|
+
* `@cf/cloudflare/clef` (27B) and `@cf/cloudflare/clef-flash` (9B) answer the
|
|
176
|
+
* same typed `noul` / `choice` / `score` questions as the other decision
|
|
177
|
+
* providers, follow the System One wire, and read images. They are reached
|
|
178
|
+
* through the Workers AI REST API, which wraps every answer in Cloudflare's own
|
|
179
|
+
* `{ result, success, errors }` envelope and puts the model in the URL path.
|
|
180
|
+
*
|
|
181
|
+
* The token and account id are the ones the Cloudflare Workers AI text provider
|
|
182
|
+
* reads (`CLOUDFLARE_API_KEY`, `CLOUDFLARE_ACCOUNT_ID`), so a host that has
|
|
183
|
+
* configured that provider has also configured this one.
|
|
184
|
+
*
|
|
185
|
+
* The Workers AI endpoint ignores text past about 2,048 tokens, far below
|
|
186
|
+
* the documented 64K (hosted service or model: unknown). See `decisionLimits`
|
|
187
|
+
* on the descriptor for the measured figures.
|
|
188
|
+
*
|
|
189
|
+
* @see https://developers.cloudflare.com/workers-ai/models/clef/
|
|
190
|
+
*/
|
|
191
|
+
export class CloudflareClefProvider extends SystemOneDecisionProvider {
|
|
192
|
+
apiKey;
|
|
193
|
+
accountId;
|
|
194
|
+
baseURL;
|
|
195
|
+
constructor(modelName, sdk, _region, credentials) {
|
|
196
|
+
super(modelName, "cloudflare-clef", sdk);
|
|
197
|
+
// A slice that names its own base URL must bring its own token: the shared
|
|
198
|
+
// CLOUDFLARE_API_KEY also runs the host's Workers AI text provider, and it
|
|
199
|
+
// must never be sent as a bearer token to an endpoint a caller chose.
|
|
200
|
+
const ownEndpoint = Boolean(credentials?.baseURL?.trim());
|
|
201
|
+
this.apiKey =
|
|
202
|
+
credentials?.apiKey?.trim() ||
|
|
203
|
+
(ownEndpoint ? "" : (process.env.CLOUDFLARE_API_KEY?.trim() ?? ""));
|
|
204
|
+
this.accountId =
|
|
205
|
+
credentials?.accountId?.trim() ||
|
|
206
|
+
(process.env.CLOUDFLARE_ACCOUNT_ID?.trim() ?? "");
|
|
207
|
+
// `||`, not `??`, so a blank override counts as unset.
|
|
208
|
+
this.baseURL = normalizeBaseURL(credentials?.baseURL?.trim() ||
|
|
209
|
+
process.env.CLOUDFLARE_CLEF_BASE_URL?.trim() ||
|
|
210
|
+
CLOUDFLARE_DEFAULT_BASE_URL);
|
|
211
|
+
// A base URL that `baseURLProblem` refuses is not logged at all: it can
|
|
212
|
+
// carry `user:pass@` or a `?token=`.
|
|
213
|
+
logger.debug("Cloudflare Clef decision provider initialized (decide only)", {
|
|
214
|
+
modelName: this.modelName,
|
|
215
|
+
baseURL: baseURLProblem(this.baseURL)
|
|
216
|
+
? "(invalid)"
|
|
217
|
+
: redactUrlForError(this.baseURL),
|
|
218
|
+
accountConfigured: this.accountId !== "",
|
|
219
|
+
});
|
|
220
|
+
}
|
|
221
|
+
getDefaultModel() {
|
|
222
|
+
return getProviderModel("CLOUDFLARE_CLEF_MODEL", CloudflareClefModels.CLEF);
|
|
223
|
+
}
|
|
224
|
+
vendorLabel() {
|
|
225
|
+
return "Cloudflare";
|
|
226
|
+
}
|
|
227
|
+
vendorDisplayName() {
|
|
228
|
+
return "Cloudflare Clef";
|
|
229
|
+
}
|
|
230
|
+
decisionApiKey() {
|
|
231
|
+
return this.apiKey;
|
|
232
|
+
}
|
|
233
|
+
missingKeyMessage() {
|
|
234
|
+
return "Cloudflare Clef requires an API token with Workers AI permission. Set CLOUDFLARE_API_KEY or pass credentials.cloudflareClef.apiKey.";
|
|
235
|
+
}
|
|
236
|
+
missingConfigMessage() {
|
|
237
|
+
if (this.accountId === "") {
|
|
238
|
+
return "Cloudflare Clef requires the account id. Set CLOUDFLARE_ACCOUNT_ID or pass credentials.cloudflareClef.accountId.";
|
|
239
|
+
}
|
|
240
|
+
if (!ACCOUNT_ID_PATTERN.test(this.accountId)) {
|
|
241
|
+
return "The Cloudflare account id may contain only letters, digits, '-' and '_'. Copy it from the Cloudflare dashboard.";
|
|
242
|
+
}
|
|
243
|
+
return baseURLProblem(this.baseURL);
|
|
244
|
+
}
|
|
245
|
+
/** The model is part of the path, so the endpoint depends on the model asked for. */
|
|
246
|
+
decisionEndpoint(model) {
|
|
247
|
+
return `${this.baseURL}/accounts/${encodeURIComponent(this.accountId)}/ai/run/${MODEL_PREFIX}${encodeURIComponent(wireModel(model) ?? model)}`;
|
|
248
|
+
}
|
|
249
|
+
decisionHeaders() {
|
|
250
|
+
return {
|
|
251
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
252
|
+
"Content-Type": "application/json",
|
|
253
|
+
};
|
|
254
|
+
}
|
|
255
|
+
/**
|
|
256
|
+
* Images travel in their own `images` array, as `data:` URLs, placed before
|
|
257
|
+
* the state by the server. `model` is sent although the path already names
|
|
258
|
+
* it: the API documents it as required, and refuses a body whose `model`
|
|
259
|
+
* differs from the path.
|
|
260
|
+
*/
|
|
261
|
+
buildDecisionBody(state, questions, model, media) {
|
|
262
|
+
const wire = wireModel(model);
|
|
263
|
+
if (!wire) {
|
|
264
|
+
throw this.decisionError({
|
|
265
|
+
kind: "invalid_request",
|
|
266
|
+
message: `"${model.slice(0, 40)}" is not a Cloudflare model name. Use "clef" or "clef-flash".`,
|
|
267
|
+
retryable: false,
|
|
268
|
+
});
|
|
269
|
+
}
|
|
270
|
+
// Workers AI also accepts image/jpg; use the canonical JPEG MIME type.
|
|
271
|
+
const images = (media?.images ?? []).map((dataUrl) => dataUrl.replace(/^data:image\/jpg;base64,/i, "data:image/jpeg;base64,"));
|
|
272
|
+
images.forEach((dataUrl, index) => {
|
|
273
|
+
const problem = imageProblem(dataUrl, index + 1);
|
|
274
|
+
if (problem) {
|
|
275
|
+
throw this.decisionError({
|
|
276
|
+
kind: "invalid_request",
|
|
277
|
+
message: problem,
|
|
278
|
+
retryable: false,
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
});
|
|
282
|
+
// The canonical JPEG prefix is one byte longer than image/jpg. Keep the
|
|
283
|
+
// prepared media metadata aligned with the URLs actually sent.
|
|
284
|
+
if (media) {
|
|
285
|
+
media.images = images;
|
|
286
|
+
media.bytes = images.reduce((total, image) => total + Buffer.byteLength(image), 0);
|
|
287
|
+
}
|
|
288
|
+
return {
|
|
289
|
+
model: wire,
|
|
290
|
+
state: wireState(state),
|
|
291
|
+
questions,
|
|
292
|
+
...(images.length > 0 ? { images } : {}),
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
/** A success is `{ result: { model, answers, usage }, success: true }`. */
|
|
296
|
+
readDecisionPayload(payload) {
|
|
297
|
+
return isRecord(payload) && isRecord(payload.result)
|
|
298
|
+
? payload.result
|
|
299
|
+
: payload;
|
|
300
|
+
}
|
|
301
|
+
parseDecisionError(status, payload, requestId) {
|
|
302
|
+
const fallback = `Cloudflare request failed with HTTP ${status}`;
|
|
303
|
+
const parsed = readCloudflareError(payload);
|
|
304
|
+
const kind = cloudflareErrorKind(status, parsed?.code);
|
|
305
|
+
let message = redactCredentials(parsed?.message ?? fallback, this.apiKey)
|
|
306
|
+
.trim()
|
|
307
|
+
.slice(0, MAX_ERROR_MESSAGE_CHARS) || fallback;
|
|
308
|
+
if (kind === "max_tokens_exceeded") {
|
|
309
|
+
message +=
|
|
310
|
+
" Cloudflare counts the whole request, image data included, at about four characters per token.";
|
|
311
|
+
}
|
|
312
|
+
else if (parsed?.code === 7000) {
|
|
313
|
+
message += `. Check the model name (Clef is served as "clef" and "clef-flash") and the base URL, which must end in /client/v4 and name your account.`;
|
|
314
|
+
}
|
|
315
|
+
return {
|
|
316
|
+
kind,
|
|
317
|
+
message,
|
|
318
|
+
status,
|
|
319
|
+
requestId: requestId ?? parsed?.requestId,
|
|
320
|
+
retryable: kind === "rate_limit" || kind === "overloaded" || kind === "server",
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
/**
|
|
324
|
+
* `cf-ai-req-id` is on every answer that reached the model, success or
|
|
325
|
+
* refusal. A 401 and a wrong-path 400 never reach it, so they carry only the
|
|
326
|
+
* edge's `cf-ray`.
|
|
327
|
+
*/
|
|
328
|
+
readRequestId(headers) {
|
|
329
|
+
return headers.get("cf-ai-req-id") ?? headers.get("cf-ray") ?? undefined;
|
|
330
|
+
}
|
|
331
|
+
}
|
|
@@ -5,8 +5,8 @@ import { createProxyFetch } from "../proxy/proxyFetch.js";
|
|
|
5
5
|
import { AuthenticationError, ProviderError, RateLimitError, } from "../types/index.js";
|
|
6
6
|
import { logger } from "../utils/logger.js";
|
|
7
7
|
import { createIdeogramConfig, getProviderModel, validateApiKey, } from "../utils/providerConfig.js";
|
|
8
|
-
import { MAX_IMAGE_BYTES
|
|
9
|
-
import {
|
|
8
|
+
import { MAX_IMAGE_BYTES } from "../utils/sizeGuard.js";
|
|
9
|
+
import { safeDownload } from "../utils/safeFetch.js";
|
|
10
10
|
const IDEOGRAM_DEFAULT_BASE_URL = "https://api.ideogram.ai";
|
|
11
11
|
const REQUEST_TIMEOUT_MS = 120_000;
|
|
12
12
|
const getIdeogramApiKey = () => validateApiKey(createIdeogramConfig());
|
|
@@ -134,34 +134,13 @@ export class IdeogramProvider extends BaseProvider {
|
|
|
134
134
|
if (!url) {
|
|
135
135
|
throw new Error("Ideogram returned no image URL");
|
|
136
136
|
}
|
|
137
|
-
//
|
|
138
|
-
//
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
const dlController = new AbortController();
|
|
145
|
-
const dlTimeoutId = setTimeout(() => dlController.abort(), 60_000);
|
|
146
|
-
let dl;
|
|
147
|
-
try {
|
|
148
|
-
dl = await this.proxyFetch(url, { signal: dlController.signal });
|
|
149
|
-
}
|
|
150
|
-
catch (err) {
|
|
151
|
-
if (err instanceof Error && err.name === "AbortError") {
|
|
152
|
-
throw new Error("Ideogram image download timed out after 60s", {
|
|
153
|
-
cause: err,
|
|
154
|
-
});
|
|
155
|
-
}
|
|
156
|
-
throw err;
|
|
157
|
-
}
|
|
158
|
-
finally {
|
|
159
|
-
clearTimeout(dlTimeoutId);
|
|
160
|
-
}
|
|
161
|
-
if (!dl.ok) {
|
|
162
|
-
throw new Error(`Failed to download Ideogram image: ${dl.status}`);
|
|
163
|
-
}
|
|
164
|
-
const buffer = await readBoundedBuffer(dl, MAX_IMAGE_BYTES, "Ideogram image");
|
|
137
|
+
// Resolve and validate once, dial only those addresses, and refuse
|
|
138
|
+
// redirects so a changing DNS answer cannot steer this download.
|
|
139
|
+
const buffer = await safeDownload(url, {
|
|
140
|
+
label: "Ideogram image",
|
|
141
|
+
maxBytes: MAX_IMAGE_BYTES,
|
|
142
|
+
timeoutMs: 60_000,
|
|
143
|
+
});
|
|
165
144
|
const base64 = buffer.toString("base64");
|
|
166
145
|
const generationTimeMs = Date.now() - startTime;
|
|
167
146
|
logger.info(`[IdeogramProvider] Generated image (${buffer.length} bytes) in ${generationTimeMs}ms — model ${this.modelName}`);
|
|
@@ -57,7 +57,8 @@ export class LlamaCppProvider extends OpenAIChatCompletionsProvider {
|
|
|
57
57
|
buildLocalUnreachableErrorRule(error, () => `llama.cpp server not reachable at ${redactUrlCredentials(this.config.baseURL)}. ` +
|
|
58
58
|
"Start it with: ./llama-server -m model.gguf --port 8080"),
|
|
59
59
|
{
|
|
60
|
-
match: (ctx) =>
|
|
60
|
+
match: (ctx) => ctx.statusCode === 400 ||
|
|
61
|
+
(ctx.statusCode === undefined && /\b400\b/.test(ctx.message)),
|
|
61
62
|
errorClass: ProviderError,
|
|
62
63
|
message: "llama.cpp rejected the request. Common cause: model doesn't support tools (start llama-server with --jinja for tool support).",
|
|
63
64
|
},
|
|
@@ -12,7 +12,7 @@ import { assertSafeUrl } from "../../utils/ssrfGuard.js";
|
|
|
12
12
|
import { createTimeoutController } from "../../utils/timeout.js";
|
|
13
13
|
import { stripTrailingSlash } from "../openaiChatCompletionsClient.js";
|
|
14
14
|
import { OpenAIChatCompletionsProvider } from "../openaiChatCompletionsBase.js";
|
|
15
|
-
import { isOpenAIQuotaExhaustedError } from "../../utils/providerRetry.js";
|
|
15
|
+
import { isOpenAIQuotaExhaustedError, readOpenAIBodyErrorType, } from "../../utils/providerRetry.js";
|
|
16
16
|
const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
|
|
17
17
|
/**
|
|
18
18
|
* Resolve the effective OpenAI base URL from optional credential / env
|
|
@@ -91,13 +91,12 @@ export class OpenAIProvider extends OpenAIChatCompletionsProvider {
|
|
|
91
91
|
return getOpenAIModel();
|
|
92
92
|
}
|
|
93
93
|
formatProviderError(error) {
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
// captured by the rule closures below.
|
|
94
|
+
// Chat errors retain the JSON in responseBody; embedding errors stamp
|
|
95
|
+
// type onto the raw error. Keep both paths available to the rule closures.
|
|
97
96
|
const errorObj = error;
|
|
98
97
|
const errorType = errorObj?.type && typeof errorObj.type === "string"
|
|
99
98
|
? errorObj.type
|
|
100
|
-
:
|
|
99
|
+
: readOpenAIBodyErrorType(errorObj?.responseBody);
|
|
101
100
|
const rules = [
|
|
102
101
|
// Curator P1-1 / Reviewer Finding #4: only the explicit auth markers
|
|
103
102
|
// map to AuthenticationError. Earlier we treated every
|
|
@@ -1005,6 +1005,7 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1005
1005
|
// endpoint, the one it was written for, is a Tier-2 catalog provider on
|
|
1006
1006
|
// this very base class, so without this the recovery would simply not
|
|
1007
1007
|
// happen for it.
|
|
1008
|
+
let promptSideFallbackRan = false;
|
|
1008
1009
|
let loop;
|
|
1009
1010
|
try {
|
|
1010
1011
|
loop = await runLoop(conversation, responseFormat);
|
|
@@ -1018,6 +1019,8 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1018
1019
|
throw error;
|
|
1019
1020
|
}
|
|
1020
1021
|
logger.warn(`[${this.providerName}] provider rejected response_format — retrying with the schema in the system prompt`, { provider: this.providerName, model: modelId });
|
|
1022
|
+
// Re-asking after this fallback would bill the same request again.
|
|
1023
|
+
promptSideFallbackRan = true;
|
|
1021
1024
|
loop = await runLoop(appendJsonSchemaInstruction(conversation, responseFormat.schema), undefined);
|
|
1022
1025
|
}
|
|
1023
1026
|
// The vendor can also IGNORE `response_format` and answer in prose without
|
|
@@ -1027,7 +1030,8 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1027
1030
|
// above was reached; the native loop has no such parser, so the silent
|
|
1028
1031
|
// case sailed through and handed the caller prose. Same recovery, keyed on
|
|
1029
1032
|
// the result rather than on an exception.
|
|
1030
|
-
if (
|
|
1033
|
+
if (!promptSideFallbackRan &&
|
|
1034
|
+
responseFormat !== undefined &&
|
|
1031
1035
|
options.schema !== undefined &&
|
|
1032
1036
|
!yieldsSchemaValidObject(loop.text, options.schema)) {
|
|
1033
1037
|
logger.warn(`[${this.providerName}] response_format did not yield a schema-valid object — retrying with the schema in the system prompt`, { provider: this.providerName, model: modelId });
|
|
@@ -1821,12 +1825,10 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1821
1825
|
// `loopPromise` is undefined when a middleware blocked the request
|
|
1822
1826
|
// before `doStream` ran, in which case there is no loop to surface.
|
|
1823
1827
|
await loopPromise;
|
|
1824
|
-
//
|
|
1825
|
-
//
|
|
1826
|
-
//
|
|
1827
|
-
//
|
|
1828
|
-
// normally and already called `resolveFinish` with the wire value —
|
|
1829
|
-
// `finishPromise` is settled, not pending.
|
|
1828
|
+
// Native loops settle the finish promise before returning. A
|
|
1829
|
+
// middleware can close its synthetic stream without calling doStream,
|
|
1830
|
+
// leaving no loop to settle it. A promise settles once, so an explicit
|
|
1831
|
+
// finish part already observed still wins.
|
|
1830
1832
|
//
|
|
1831
1833
|
// `rawFinishReason` is the verbatim vendor value (e.g. "stop",
|
|
1832
1834
|
// "length", "tool_calls", "content_filter") and is accurate the
|
|
@@ -1834,6 +1836,9 @@ export class OpenAIChatCompletionsProvider extends BaseProvider {
|
|
|
1834
1836
|
// post-stream work follows. The graded `finishReason` is set later,
|
|
1835
1837
|
// once that post-stream work (the structured-output re-ask below)
|
|
1836
1838
|
// has actually succeeded — see the assignment after that block.
|
|
1839
|
+
if (!loopPromise) {
|
|
1840
|
+
resolveFinish("stop");
|
|
1841
|
+
}
|
|
1837
1842
|
streamMetadata.rawFinishReason = await finishPromise;
|
|
1838
1843
|
// Structured output for a `stream({ schema })` turn. Runs HERE —
|
|
1839
1844
|
// after the stream is fully drained — so a tool-free re-ask (when the
|
|
@@ -454,11 +454,23 @@ const SCHEMA_NODES = [
|
|
|
454
454
|
"then",
|
|
455
455
|
"else",
|
|
456
456
|
];
|
|
457
|
-
// OpenAI's strict mode does not accept these
|
|
457
|
+
// OpenAI's strict mode does not accept these keywords. A schema
|
|
458
458
|
// carrying one cannot be sent with `strict: true` at all, so it is not merely
|
|
459
459
|
// "not yet normalised" — it must drop to non-strict, where the schema is
|
|
460
460
|
// honoured as written.
|
|
461
|
-
const STRICT_UNSUPPORTED = [
|
|
461
|
+
const STRICT_UNSUPPORTED = [
|
|
462
|
+
"allOf",
|
|
463
|
+
"oneOf",
|
|
464
|
+
"not",
|
|
465
|
+
"if",
|
|
466
|
+
"then",
|
|
467
|
+
"else",
|
|
468
|
+
"dependentRequired",
|
|
469
|
+
"dependentSchemas",
|
|
470
|
+
"patternProperties",
|
|
471
|
+
"uniqueItems",
|
|
472
|
+
"prefixItems",
|
|
473
|
+
];
|
|
462
474
|
const mapValues = (obj, fn) => Object.fromEntries(Object.entries((obj ?? {})).map(([k, v]) => [
|
|
463
475
|
k,
|
|
464
476
|
fn(v),
|
|
@@ -500,7 +512,7 @@ const withClosedObjects = (node) => {
|
|
|
500
512
|
/**
|
|
501
513
|
* True when the schema can legally be sent with `strict: true`: every object
|
|
502
514
|
* node lists all of its properties as required, every object is closed, and
|
|
503
|
-
* no
|
|
515
|
+
* no unsupported keyword or tuple-valued items appears anywhere.
|
|
504
516
|
*
|
|
505
517
|
* Deliberately conservative — a false negative costs only the stronger
|
|
506
518
|
* guarantee, while a false positive costs the whole request.
|
|
@@ -516,6 +528,10 @@ const satisfiesStrictRequired = (node) => {
|
|
|
516
528
|
if (STRICT_UNSUPPORTED.some((k) => k in rec)) {
|
|
517
529
|
return false;
|
|
518
530
|
}
|
|
531
|
+
// Draft-07 tuples have an items array; strict mode accepts one items schema.
|
|
532
|
+
if (Array.isArray(rec.items)) {
|
|
533
|
+
return false;
|
|
534
|
+
}
|
|
519
535
|
if (rec.type === "object") {
|
|
520
536
|
if (rec.additionalProperties !== false) {
|
|
521
537
|
return false;
|
|
@@ -532,6 +548,8 @@ const satisfiesStrictRequired = (node) => {
|
|
|
532
548
|
const nodesOk = SCHEMA_NODES.filter((k) => k in rec).every((k) => satisfiesStrictRequired(rec[k]));
|
|
533
549
|
return mapsOk && nodesOk;
|
|
534
550
|
};
|
|
551
|
+
const hasObjectRoot = (schema) => schema.type === "object" && !("anyOf" in schema) && !("oneOf" in schema);
|
|
552
|
+
const isStrictLegal = (schema) => hasObjectRoot(schema) && satisfiesStrictRequired(schema);
|
|
535
553
|
export const v3ResponseFormatToOpenAI = (rf) => {
|
|
536
554
|
if (rf.type === "text") {
|
|
537
555
|
return { type: "text" };
|
|
@@ -539,7 +557,7 @@ export const v3ResponseFormatToOpenAI = (rf) => {
|
|
|
539
557
|
if (!rf.schema) {
|
|
540
558
|
return { type: "json_object" };
|
|
541
559
|
}
|
|
542
|
-
// Mutate as little as possible,
|
|
560
|
+
// Mutate as little as possible (root must be an object, without anyOf/oneOf):
|
|
543
561
|
//
|
|
544
562
|
// 1. already strict-legal -> send it UNTOUCHED with strict: true
|
|
545
563
|
// 2. legal once closed -> send the closed copy with strict: true
|
|
@@ -559,9 +577,9 @@ export const v3ResponseFormatToOpenAI = (rf) => {
|
|
|
559
577
|
// what it did before this change.
|
|
560
578
|
const original = rf.schema;
|
|
561
579
|
const closed = withClosedObjects(original);
|
|
562
|
-
const schema =
|
|
580
|
+
const schema = isStrictLegal(original)
|
|
563
581
|
? original
|
|
564
|
-
:
|
|
582
|
+
: isStrictLegal(closed)
|
|
565
583
|
? closed
|
|
566
584
|
: original;
|
|
567
585
|
return {
|
|
@@ -570,7 +588,7 @@ export const v3ResponseFormatToOpenAI = (rf) => {
|
|
|
570
588
|
name: rf.name ?? "response",
|
|
571
589
|
schema: schema,
|
|
572
590
|
...(rf.description ? { description: rf.description } : {}),
|
|
573
|
-
strict:
|
|
591
|
+
strict: isStrictLegal(schema),
|
|
574
592
|
},
|
|
575
593
|
};
|
|
576
594
|
};
|
|
@@ -5,8 +5,8 @@ import { createProxyFetch } from "../proxy/proxyFetch.js";
|
|
|
5
5
|
import { AuthenticationError, InvalidModelError, ProviderError, RateLimitError, } from "../types/index.js";
|
|
6
6
|
import { logger } from "../utils/logger.js";
|
|
7
7
|
import { createRecraftConfig, getProviderModel, validateApiKey, } from "../utils/providerConfig.js";
|
|
8
|
-
import { MAX_IMAGE_BYTES
|
|
9
|
-
import {
|
|
8
|
+
import { MAX_IMAGE_BYTES } from "../utils/sizeGuard.js";
|
|
9
|
+
import { safeDownload } from "../utils/safeFetch.js";
|
|
10
10
|
import { detectImageMimeType } from "../utils/imageDetection.js";
|
|
11
11
|
const RECRAFT_DEFAULT_BASE_URL = "https://external.api.recraft.ai/v1";
|
|
12
12
|
const REQUEST_TIMEOUT_MS = 120_000;
|
|
@@ -141,32 +141,13 @@ export class RecraftProvider extends BaseProvider {
|
|
|
141
141
|
base64 = entry.b64_json;
|
|
142
142
|
}
|
|
143
143
|
else if (entry.url) {
|
|
144
|
-
//
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
try {
|
|
152
|
-
dl = await this.proxyFetch(entry.url, { signal: dlController.signal });
|
|
153
|
-
}
|
|
154
|
-
catch (err) {
|
|
155
|
-
if (err instanceof Error && err.name === "AbortError") {
|
|
156
|
-
throw new Error("Recraft image download timed out after 60s", {
|
|
157
|
-
cause: err,
|
|
158
|
-
});
|
|
159
|
-
}
|
|
160
|
-
throw err;
|
|
161
|
-
}
|
|
162
|
-
finally {
|
|
163
|
-
clearTimeout(dlTimeoutId);
|
|
164
|
-
}
|
|
165
|
-
if (!dl.ok) {
|
|
166
|
-
throw new Error(`Failed to download Recraft image: ${dl.status}`);
|
|
167
|
-
}
|
|
168
|
-
const dlBuf = await readBoundedBuffer(dl, MAX_IMAGE_BYTES, "Recraft image");
|
|
169
|
-
base64 = dlBuf.toString("base64");
|
|
144
|
+
// See ideogram.ts: pin validated addresses and refuse redirects.
|
|
145
|
+
const buffer = await safeDownload(entry.url, {
|
|
146
|
+
label: "Recraft image",
|
|
147
|
+
maxBytes: MAX_IMAGE_BYTES,
|
|
148
|
+
timeoutMs: 60_000,
|
|
149
|
+
});
|
|
150
|
+
base64 = buffer.toString("base64");
|
|
170
151
|
}
|
|
171
152
|
else {
|
|
172
153
|
throw new Error("Recraft response missing both b64_json and url");
|
|
@@ -59,7 +59,11 @@ export declare abstract class SystemOneDecisionProvider extends BaseProvider {
|
|
|
59
59
|
* nothing is missing.
|
|
60
60
|
*/
|
|
61
61
|
protected missingConfigMessage(): string | undefined;
|
|
62
|
-
|
|
62
|
+
/**
|
|
63
|
+
* Where the request goes. The model asked for is passed in because one
|
|
64
|
+
* vendor names it in the URL path; every other provider ignores it.
|
|
65
|
+
*/
|
|
66
|
+
protected abstract decisionEndpoint(model: string): string;
|
|
63
67
|
protected abstract decisionHeaders(): Record<string, string>;
|
|
64
68
|
protected abstract buildDecisionBody(state: DecisionState, questions: Record<string, Record<string, unknown>>, model: string, media?: DecisionPreparedMedia): Record<string, unknown>;
|
|
65
69
|
protected abstract parseDecisionError(status: number, payload: unknown, requestId: string | undefined): DecisionError;
|
|
@@ -81,6 +85,13 @@ export declare abstract class SystemOneDecisionProvider extends BaseProvider {
|
|
|
81
85
|
* grows with the number of questions scales the descriptor's allowance here.
|
|
82
86
|
*/
|
|
83
87
|
protected defaultTimeoutMs(_questionCount: number): number | undefined;
|
|
88
|
+
/**
|
|
89
|
+
* The part of a successful response that holds `answers`, `usage` and
|
|
90
|
+
* `model`. The default is the response itself; a vendor that wraps every
|
|
91
|
+
* answer in an envelope returns what is inside it. Errors are not passed
|
|
92
|
+
* through here: they are read from the whole response.
|
|
93
|
+
*/
|
|
94
|
+
protected readDecisionPayload(payload: unknown): unknown;
|
|
84
95
|
/** Per-question confidence a transport reports outside the answer objects. */
|
|
85
96
|
protected reportedConfidence(_payload: unknown): Record<string, number>;
|
|
86
97
|
protected resolveResponseModel(payload: Record<string, unknown>, requestedModel: string): string;
|