visual-ai-assertions 0.23.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +140 -83
- package/dist/index.cjs +600 -190
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +151 -30
- package/dist/index.d.ts +151 -30
- package/dist/index.js +600 -190
- package/dist/index.js.map +1 -1
- package/package.json +8 -9
package/dist/index.cjs
CHANGED
|
@@ -67,135 +67,6 @@ __export(index_exports, {
|
|
|
67
67
|
});
|
|
68
68
|
module.exports = __toCommonJS(index_exports);
|
|
69
69
|
|
|
70
|
-
// src/constants.ts
|
|
71
|
-
var ReasoningEffort = {
|
|
72
|
-
LOW: "low",
|
|
73
|
-
MEDIUM: "medium",
|
|
74
|
-
HIGH: "high",
|
|
75
|
-
XHIGH: "xhigh"
|
|
76
|
-
};
|
|
77
|
-
var ImageDetail = {
|
|
78
|
-
AUTO: "auto",
|
|
79
|
-
LOW: "low",
|
|
80
|
-
HIGH: "high"
|
|
81
|
-
};
|
|
82
|
-
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
83
|
-
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
84
|
-
var Provider = {
|
|
85
|
-
ANTHROPIC: "anthropic",
|
|
86
|
-
OPENAI: "openai",
|
|
87
|
-
GOOGLE: "google",
|
|
88
|
-
OPENROUTER: "openrouter"
|
|
89
|
-
};
|
|
90
|
-
var Model = {
|
|
91
|
-
Anthropic: {
|
|
92
|
-
FABLE_5_1: "claude-fable-5-1",
|
|
93
|
-
FABLE_5: "claude-fable-5",
|
|
94
|
-
OPUS_5: "claude-opus-5",
|
|
95
|
-
OPUS_4_8: "claude-opus-4-8",
|
|
96
|
-
OPUS_4_7: "claude-opus-4-7",
|
|
97
|
-
OPUS_4_6: "claude-opus-4-6",
|
|
98
|
-
SONNET_5: "claude-sonnet-5",
|
|
99
|
-
SONNET_4_6: "claude-sonnet-4-6",
|
|
100
|
-
HAIKU_4_5: "claude-haiku-4-5"
|
|
101
|
-
},
|
|
102
|
-
OpenAI: {
|
|
103
|
-
GPT_6_ASTRA: "gpt-6-astra",
|
|
104
|
-
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
105
|
-
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
106
|
-
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
107
|
-
GPT_5_5: "gpt-5.5",
|
|
108
|
-
GPT_5_4: "gpt-5.4",
|
|
109
|
-
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
110
|
-
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
111
|
-
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
112
|
-
GPT_5_2: "gpt-5.2",
|
|
113
|
-
GPT_5_MINI: "gpt-5-mini"
|
|
114
|
-
},
|
|
115
|
-
Google: {
|
|
116
|
-
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
117
|
-
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
118
|
-
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
119
|
-
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
120
|
-
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
121
|
-
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
122
|
-
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
123
|
-
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
124
|
-
},
|
|
125
|
-
/**
|
|
126
|
-
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
127
|
-
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
128
|
-
* recognizes them. All listed models accept image input.
|
|
129
|
-
*/
|
|
130
|
-
OpenRouter: {
|
|
131
|
-
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
132
|
-
GROK_4_6: "x-ai/grok-4.6",
|
|
133
|
-
GROK_4_5: "x-ai/grok-4.5",
|
|
134
|
-
KIMI_K3: "moonshotai/kimi-k3",
|
|
135
|
-
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
136
|
-
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
137
|
-
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
138
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
139
|
-
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
140
|
-
}
|
|
141
|
-
};
|
|
142
|
-
var DEFAULT_MODELS = {
|
|
143
|
-
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
144
|
-
[Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
|
|
145
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
146
|
-
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
147
|
-
};
|
|
148
|
-
var DEFAULT_MAX_TOKENS = 4096;
|
|
149
|
-
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
150
|
-
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
151
|
-
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
152
|
-
Model.OpenAI.GPT_6_ASTRA
|
|
153
|
-
]);
|
|
154
|
-
var MODEL_TO_PROVIDER = new Map([
|
|
155
|
-
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
156
|
-
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
157
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
158
|
-
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
159
|
-
]);
|
|
160
|
-
var VALID_PROVIDERS = Object.values(Provider);
|
|
161
|
-
var PROVIDER_DEFAULT_REASONING = {
|
|
162
|
-
openai: "medium",
|
|
163
|
-
anthropic: "off",
|
|
164
|
-
google: "off",
|
|
165
|
-
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
166
|
-
openrouter: "off"
|
|
167
|
-
};
|
|
168
|
-
var Content = {
|
|
169
|
-
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
170
|
-
PLACEHOLDER_TEXT: "placeholder-text",
|
|
171
|
-
/** Detects error messages, banners, stack traces, or error codes */
|
|
172
|
-
ERROR_MESSAGES: "error-messages",
|
|
173
|
-
/** Detects broken image icons or failed-to-load image indicators */
|
|
174
|
-
BROKEN_IMAGES: "broken-images",
|
|
175
|
-
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
176
|
-
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
177
|
-
};
|
|
178
|
-
var Layout = {
|
|
179
|
-
/** Detects elements that unintentionally overlap each other */
|
|
180
|
-
OVERLAP: "overlap",
|
|
181
|
-
/** Detects content cut off or extending beyond container boundaries */
|
|
182
|
-
OVERFLOW: "overflow",
|
|
183
|
-
/** Detects inconsistent alignment of text, images, and UI components */
|
|
184
|
-
ALIGNMENT: "alignment"
|
|
185
|
-
};
|
|
186
|
-
var Accessibility = {
|
|
187
|
-
/** Detects insufficient color contrast between text and backgrounds */
|
|
188
|
-
CONTRAST: "contrast",
|
|
189
|
-
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
190
|
-
READABILITY: "readability",
|
|
191
|
-
/** Detects interactive elements that are not visually distinct */
|
|
192
|
-
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
193
|
-
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
194
|
-
COLOR_BLINDNESS: "color-blindness",
|
|
195
|
-
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
196
|
-
COLOR_ALONE: "color-alone"
|
|
197
|
-
};
|
|
198
|
-
|
|
199
70
|
// src/errors.ts
|
|
200
71
|
var VisualAIError = class extends Error {
|
|
201
72
|
code;
|
|
@@ -327,10 +198,15 @@ Example for a failing check:
|
|
|
327
198
|
]
|
|
328
199
|
}
|
|
329
200
|
${JSON_INSTRUCTIONS}`;
|
|
330
|
-
|
|
201
|
+
function buildCheckOutputSchemaVideo(unit) {
|
|
202
|
+
const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
|
|
203
|
+
const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
|
|
204
|
+
const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
|
|
205
|
+
const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
|
|
206
|
+
return `IMPORTANT: Follow this evaluation order:
|
|
331
207
|
1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
|
|
332
|
-
2. A statement passes if it is true at
|
|
333
|
-
3. For each statement that passes, set "timestampSeconds" to
|
|
208
|
+
2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
|
|
209
|
+
3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
|
|
334
210
|
4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
|
|
335
211
|
5. Write "reasoning" as a brief overall summary of the evaluation
|
|
336
212
|
6. Include "issues" only for statements that failed
|
|
@@ -344,7 +220,7 @@ Respond with a JSON object matching this exact structure:
|
|
|
344
220
|
{
|
|
345
221
|
"statement": string, // the original statement text
|
|
346
222
|
"pass": boolean, // whether this statement is true at any point in the timeline
|
|
347
|
-
"reasoning": string, // explanation for this statement, citing
|
|
223
|
+
"reasoning": string, // explanation for this statement, ${citing} where relevant
|
|
348
224
|
"confidence": "high" | "medium" | "low",
|
|
349
225
|
"timestampSeconds": number | null
|
|
350
226
|
// seconds from the start of the clip where the statement is most clearly true,
|
|
@@ -362,10 +238,13 @@ Example for a passing video check:
|
|
|
362
238
|
"reasoning": "The success toast appeared briefly around 3.5s.",
|
|
363
239
|
"issues": [],
|
|
364
240
|
"statements": [
|
|
365
|
-
{ "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right
|
|
241
|
+
{ "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
|
|
366
242
|
]
|
|
367
243
|
}
|
|
368
244
|
${JSON_INSTRUCTIONS}`;
|
|
245
|
+
}
|
|
246
|
+
var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
|
|
247
|
+
var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
|
|
369
248
|
var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
|
|
370
249
|
{
|
|
371
250
|
"summary": string, // high-level analysis summary
|
|
@@ -398,6 +277,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
|
|
|
398
277
|
Prioritize issues by severity (critical / major / minor) as for image input.
|
|
399
278
|
Cite frame indices in "frameReferences" so the user can locate the moments you describe.
|
|
400
279
|
${JSON_INSTRUCTIONS}`;
|
|
280
|
+
var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
|
|
281
|
+
{
|
|
282
|
+
"summary": string, // high-level summary of what happens across the video
|
|
283
|
+
"issues": [...], // list of issues/findings, can be empty
|
|
284
|
+
"timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
|
|
285
|
+
}
|
|
286
|
+
${ISSUE_SCHEMA_INSTRUCTIONS}
|
|
287
|
+
|
|
288
|
+
Prioritize issues by severity (critical / major / minor) as for image input.
|
|
289
|
+
Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
|
|
290
|
+
${JSON_INSTRUCTIONS}`;
|
|
401
291
|
var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
|
|
402
292
|
{
|
|
403
293
|
"pass": boolean, // true if no critical or major changes found
|
|
@@ -419,15 +309,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
|
|
|
419
309
|
var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
|
|
420
310
|
var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
|
|
421
311
|
var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
|
|
422
|
-
|
|
312
|
+
var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
|
|
313
|
+
var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
|
|
314
|
+
function buildNativeVideoSection(durationSeconds) {
|
|
315
|
+
return `Video recording:
|
|
316
|
+
- Total duration: ${durationSeconds.toFixed(2)}s
|
|
317
|
+
|
|
318
|
+
The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
|
|
319
|
+
}
|
|
320
|
+
function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
|
|
423
321
|
const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
|
|
322
|
+
const attached = frameTimestamps.length;
|
|
323
|
+
const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
|
|
324
|
+
const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
|
|
424
325
|
return `Video timeline:
|
|
425
326
|
- Total duration: ${durationSeconds.toFixed(2)}s
|
|
426
|
-
|
|
327
|
+
${sampledLine}
|
|
427
328
|
- Frame index \u2192 timestamp:
|
|
428
329
|
${formatted}
|
|
429
330
|
|
|
430
|
-
Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful
|
|
331
|
+
Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
|
|
431
332
|
}
|
|
432
333
|
var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
|
|
433
334
|
var COMPARE_EDGE_RULES = [
|
|
@@ -442,30 +343,52 @@ function buildCheckPrompt(statements, options) {
|
|
|
442
343
|
const stmts = Array.isArray(statements) ? statements : [statements];
|
|
443
344
|
const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
|
|
444
345
|
const media = options?.media;
|
|
445
|
-
const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
|
|
346
|
+
const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
|
|
446
347
|
const sections = [options?.role ?? defaultRole];
|
|
447
348
|
if (media?.kind === "video") {
|
|
448
|
-
sections.push(
|
|
349
|
+
sections.push(
|
|
350
|
+
buildVideoTimelineSection(
|
|
351
|
+
media.frameTimestamps,
|
|
352
|
+
media.durationSeconds,
|
|
353
|
+
media.droppedUnchanged
|
|
354
|
+
)
|
|
355
|
+
);
|
|
356
|
+
} else if (media?.kind === "native-video") {
|
|
357
|
+
sections.push(buildNativeVideoSection(media.durationSeconds));
|
|
449
358
|
}
|
|
450
359
|
if (options?.instructions && options.instructions.length > 0) {
|
|
451
360
|
sections.push(buildInstructionsSection(options.instructions));
|
|
452
361
|
}
|
|
453
362
|
sections.push(`Statements to evaluate:
|
|
454
363
|
${statementsBlock}`);
|
|
455
|
-
sections.push(
|
|
364
|
+
sections.push(
|
|
365
|
+
media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
|
|
366
|
+
);
|
|
456
367
|
return sections.join("\n\n");
|
|
457
368
|
}
|
|
458
369
|
function buildAskPrompt(userPrompt, options) {
|
|
459
370
|
const media = options?.media;
|
|
460
|
-
const sections = [
|
|
371
|
+
const sections = [
|
|
372
|
+
media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
|
|
373
|
+
];
|
|
461
374
|
if (media?.kind === "video") {
|
|
462
|
-
sections.push(
|
|
375
|
+
sections.push(
|
|
376
|
+
buildVideoTimelineSection(
|
|
377
|
+
media.frameTimestamps,
|
|
378
|
+
media.durationSeconds,
|
|
379
|
+
media.droppedUnchanged
|
|
380
|
+
)
|
|
381
|
+
);
|
|
382
|
+
} else if (media?.kind === "native-video") {
|
|
383
|
+
sections.push(buildNativeVideoSection(media.durationSeconds));
|
|
463
384
|
}
|
|
464
385
|
if (options?.instructions && options.instructions.length > 0) {
|
|
465
386
|
sections.push(buildInstructionsSection(options.instructions));
|
|
466
387
|
}
|
|
467
388
|
sections.push(`User request: ${userPrompt}`);
|
|
468
|
-
sections.push(
|
|
389
|
+
sections.push(
|
|
390
|
+
media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
|
|
391
|
+
);
|
|
469
392
|
return sections.join("\n\n");
|
|
470
393
|
}
|
|
471
394
|
function buildAiDiffPrompt() {
|
|
@@ -510,7 +433,7 @@ function visibleRole(finalState, requireCorrectRendering) {
|
|
|
510
433
|
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
511
434
|
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
512
435
|
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
513
|
-
"
|
|
436
|
+
"The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
|
|
514
437
|
];
|
|
515
438
|
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
516
439
|
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
@@ -541,6 +464,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
|
541
464
|
return buildCheckPrompt(statements, { role, instructions });
|
|
542
465
|
}
|
|
543
466
|
|
|
467
|
+
// src/constants.ts
|
|
468
|
+
var ReasoningEffort = {
|
|
469
|
+
MINIMAL: "minimal",
|
|
470
|
+
LOW: "low",
|
|
471
|
+
MEDIUM: "medium",
|
|
472
|
+
HIGH: "high",
|
|
473
|
+
XHIGH: "xhigh"
|
|
474
|
+
};
|
|
475
|
+
var ImageDetail = {
|
|
476
|
+
AUTO: "auto",
|
|
477
|
+
LOW: "low",
|
|
478
|
+
HIGH: "high"
|
|
479
|
+
};
|
|
480
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
481
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
482
|
+
var Provider = {
|
|
483
|
+
ANTHROPIC: "anthropic",
|
|
484
|
+
OPENAI: "openai",
|
|
485
|
+
GOOGLE: "google",
|
|
486
|
+
OPENROUTER: "openrouter"
|
|
487
|
+
};
|
|
488
|
+
var Model = {
|
|
489
|
+
Anthropic: {
|
|
490
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
491
|
+
FABLE_5: "claude-fable-5",
|
|
492
|
+
OPUS_5_5: "claude-opus-5-5",
|
|
493
|
+
OPUS_5: "claude-opus-5",
|
|
494
|
+
OPUS_4_8: "claude-opus-4-8",
|
|
495
|
+
OPUS_4_7: "claude-opus-4-7",
|
|
496
|
+
OPUS_4_6: "claude-opus-4-6",
|
|
497
|
+
SONNET_5_5: "claude-sonnet-5-5",
|
|
498
|
+
SONNET_5: "claude-sonnet-5",
|
|
499
|
+
SONNET_4_6: "claude-sonnet-4-6",
|
|
500
|
+
HAIKU_4_5: "claude-haiku-4-5"
|
|
501
|
+
},
|
|
502
|
+
OpenAI: {
|
|
503
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
504
|
+
GPT_6_1_SOL: "gpt-6.1-sol",
|
|
505
|
+
GPT_6_SOL: "gpt-6-sol",
|
|
506
|
+
GPT_6_LUNA: "gpt-6-luna",
|
|
507
|
+
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
508
|
+
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
509
|
+
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
510
|
+
GPT_5_5: "gpt-5.5",
|
|
511
|
+
GPT_5_4: "gpt-5.4",
|
|
512
|
+
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
513
|
+
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
514
|
+
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
515
|
+
GPT_5_2: "gpt-5.2",
|
|
516
|
+
GPT_5_MINI: "gpt-5-mini"
|
|
517
|
+
},
|
|
518
|
+
Google: {
|
|
519
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
520
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
521
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
522
|
+
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
523
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
524
|
+
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
525
|
+
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
526
|
+
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
527
|
+
},
|
|
528
|
+
/**
|
|
529
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
530
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
531
|
+
* recognizes them. All listed models accept image input.
|
|
532
|
+
*/
|
|
533
|
+
OpenRouter: {
|
|
534
|
+
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
535
|
+
GROK_4_7: "x-ai/grok-4.7",
|
|
536
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
537
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
538
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
539
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
540
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
541
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
542
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
543
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
|
|
544
|
+
MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
|
|
545
|
+
}
|
|
546
|
+
};
|
|
547
|
+
var DEFAULT_MODELS = {
|
|
548
|
+
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
|
|
549
|
+
[Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
|
|
550
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
|
|
551
|
+
[Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
|
|
552
|
+
};
|
|
553
|
+
var DEFAULT_MAX_TOKENS = 4096;
|
|
554
|
+
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
555
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
556
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
557
|
+
Model.OpenRouter.QWEN_3_8_MAX,
|
|
558
|
+
Model.OpenRouter.QWEN_3_7_PLUS
|
|
559
|
+
]);
|
|
560
|
+
var MODEL_TO_PROVIDER = new Map([
|
|
561
|
+
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
562
|
+
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
563
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
564
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
565
|
+
]);
|
|
566
|
+
var VALID_PROVIDERS = Object.values(Provider);
|
|
567
|
+
var PROVIDER_DEFAULT_REASONING = {
|
|
568
|
+
openai: "medium",
|
|
569
|
+
anthropic: "off",
|
|
570
|
+
google: "off",
|
|
571
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
572
|
+
openrouter: "off"
|
|
573
|
+
};
|
|
574
|
+
var Content = {
|
|
575
|
+
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
576
|
+
PLACEHOLDER_TEXT: "placeholder-text",
|
|
577
|
+
/** Detects error messages, banners, stack traces, or error codes */
|
|
578
|
+
ERROR_MESSAGES: "error-messages",
|
|
579
|
+
/** Detects broken image icons or failed-to-load image indicators */
|
|
580
|
+
BROKEN_IMAGES: "broken-images",
|
|
581
|
+
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
582
|
+
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
583
|
+
};
|
|
584
|
+
var Layout = {
|
|
585
|
+
/** Detects elements that unintentionally overlap each other */
|
|
586
|
+
OVERLAP: "overlap",
|
|
587
|
+
/** Detects content cut off or extending beyond container boundaries */
|
|
588
|
+
OVERFLOW: "overflow",
|
|
589
|
+
/** Detects inconsistent alignment of text, images, and UI components */
|
|
590
|
+
ALIGNMENT: "alignment"
|
|
591
|
+
};
|
|
592
|
+
var Accessibility = {
|
|
593
|
+
/** Detects insufficient color contrast between text and backgrounds */
|
|
594
|
+
CONTRAST: "contrast",
|
|
595
|
+
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
596
|
+
READABILITY: "readability",
|
|
597
|
+
/** Detects interactive elements that are not visually distinct */
|
|
598
|
+
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
599
|
+
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
600
|
+
COLOR_BLINDNESS: "color-blindness",
|
|
601
|
+
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
602
|
+
COLOR_ALONE: "color-alone"
|
|
603
|
+
};
|
|
604
|
+
|
|
544
605
|
// src/templates/accessibility.ts
|
|
545
606
|
var ALL_CHECKS = Object.values(Accessibility);
|
|
546
607
|
var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
|
|
@@ -650,17 +711,22 @@ function parseRetryAfter(value) {
|
|
|
650
711
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
651
712
|
Model.Anthropic.FABLE_5_1,
|
|
652
713
|
Model.Anthropic.FABLE_5,
|
|
714
|
+
Model.Anthropic.OPUS_5_5,
|
|
653
715
|
Model.Anthropic.OPUS_5,
|
|
654
716
|
Model.Anthropic.OPUS_4_8,
|
|
655
717
|
Model.Anthropic.OPUS_4_7,
|
|
718
|
+
Model.Anthropic.SONNET_5_5,
|
|
656
719
|
Model.Anthropic.SONNET_5
|
|
657
720
|
]);
|
|
658
721
|
function mapEffort(level, model) {
|
|
722
|
+
if (level === "minimal") return "low";
|
|
659
723
|
if (level !== "xhigh") return level;
|
|
660
724
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
661
725
|
}
|
|
662
726
|
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
663
727
|
var EFFORT_TO_BUDGET_TOKENS = {
|
|
728
|
+
// 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
|
|
729
|
+
minimal: 1024,
|
|
664
730
|
low: 1024,
|
|
665
731
|
medium: 4096,
|
|
666
732
|
high: 8192,
|
|
@@ -766,11 +832,20 @@ var AnthropicDriver = class {
|
|
|
766
832
|
|
|
767
833
|
// src/providers/google.ts
|
|
768
834
|
var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
|
|
835
|
+
var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
|
|
836
|
+
var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
|
|
837
|
+
var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
|
|
769
838
|
function needsCodeExecution(model) {
|
|
770
839
|
const match = model.match(/^gemini-(\d+)/);
|
|
771
840
|
return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
|
|
772
841
|
}
|
|
842
|
+
function sleep(ms) {
|
|
843
|
+
return new Promise((resolve2) => setTimeout(resolve2, ms));
|
|
844
|
+
}
|
|
773
845
|
var GOOGLE_THINKING_LEVEL = {
|
|
846
|
+
// Gemini does define a "minimal" thinking level, but some models reject it
|
|
847
|
+
// (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
|
|
848
|
+
minimal: "low",
|
|
774
849
|
low: "low",
|
|
775
850
|
medium: "medium",
|
|
776
851
|
high: "high",
|
|
@@ -837,24 +912,28 @@ var GoogleDriver = class {
|
|
|
837
912
|
});
|
|
838
913
|
return this.client;
|
|
839
914
|
}
|
|
840
|
-
|
|
841
|
-
|
|
915
|
+
/** Request config shared by image and video messages. */
|
|
916
|
+
generationConfig() {
|
|
917
|
+
return {
|
|
918
|
+
responseMimeType: "application/json",
|
|
919
|
+
maxOutputTokens: this.maxTokens,
|
|
920
|
+
...this.reasoningEffort && {
|
|
921
|
+
thinkingConfig: {
|
|
922
|
+
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
923
|
+
}
|
|
924
|
+
},
|
|
925
|
+
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
926
|
+
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
927
|
+
}
|
|
928
|
+
};
|
|
929
|
+
}
|
|
930
|
+
/** Runs one generateContent call and normalizes finish reasons, text, and usage. */
|
|
931
|
+
async generate(client, contents) {
|
|
842
932
|
try {
|
|
843
933
|
const response = await client.models.generateContent({
|
|
844
934
|
model: this.model,
|
|
845
|
-
contents
|
|
846
|
-
config:
|
|
847
|
-
responseMimeType: "application/json",
|
|
848
|
-
maxOutputTokens: this.maxTokens,
|
|
849
|
-
...this.reasoningEffort && {
|
|
850
|
-
thinkingConfig: {
|
|
851
|
-
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
852
|
-
}
|
|
853
|
-
},
|
|
854
|
-
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
855
|
-
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
856
|
-
}
|
|
857
|
-
}
|
|
935
|
+
contents,
|
|
936
|
+
config: this.generationConfig()
|
|
858
937
|
});
|
|
859
938
|
const finishReason = response.candidates?.[0]?.finishReason;
|
|
860
939
|
if (finishReason === "MAX_TOKENS") {
|
|
@@ -869,9 +948,8 @@ var GoogleDriver = class {
|
|
|
869
948
|
`Response blocked: Google returned finishReason "${finishReason}".`
|
|
870
949
|
);
|
|
871
950
|
}
|
|
872
|
-
const text = response.text ?? "";
|
|
873
951
|
return {
|
|
874
|
-
text,
|
|
952
|
+
text: response.text ?? "",
|
|
875
953
|
usage: toGeminiUsage(response.usageMetadata)
|
|
876
954
|
};
|
|
877
955
|
} catch (err) {
|
|
@@ -879,6 +957,80 @@ var GoogleDriver = class {
|
|
|
879
957
|
throw mapProviderError(err);
|
|
880
958
|
}
|
|
881
959
|
}
|
|
960
|
+
async sendMessage(images, prompt, _options) {
|
|
961
|
+
const client = await this.getClient();
|
|
962
|
+
return this.generate(client, [...this.toGeminiParts(images), prompt]);
|
|
963
|
+
}
|
|
964
|
+
/**
|
|
965
|
+
* Uploads a video through the Files API and waits until Gemini has finished
|
|
966
|
+
* processing it. Returns the ACTIVE file record.
|
|
967
|
+
*/
|
|
968
|
+
async uploadVideo(client, video) {
|
|
969
|
+
try {
|
|
970
|
+
const uploaded = await client.files.upload({
|
|
971
|
+
file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
|
|
972
|
+
config: { mimeType: video.mimeType }
|
|
973
|
+
});
|
|
974
|
+
const name = uploaded.name;
|
|
975
|
+
if (!name) {
|
|
976
|
+
throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
|
|
977
|
+
}
|
|
978
|
+
const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
|
|
979
|
+
let current = uploaded;
|
|
980
|
+
while (current.state === "PROCESSING") {
|
|
981
|
+
if (Date.now() > deadline) {
|
|
982
|
+
throw new VisualAIProviderError(
|
|
983
|
+
`Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
|
|
984
|
+
);
|
|
985
|
+
}
|
|
986
|
+
await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
|
|
987
|
+
current = await client.files.get({ name });
|
|
988
|
+
}
|
|
989
|
+
if (current.state !== "ACTIVE") {
|
|
990
|
+
throw new VisualAIProviderError(
|
|
991
|
+
`Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
|
|
992
|
+
);
|
|
993
|
+
}
|
|
994
|
+
if (!current.uri) {
|
|
995
|
+
throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
|
|
996
|
+
}
|
|
997
|
+
return current;
|
|
998
|
+
} catch (err) {
|
|
999
|
+
if (err instanceof VisualAIProviderError) throw err;
|
|
1000
|
+
throw mapProviderError(err);
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
/**
|
|
1004
|
+
* Sends the video bytes themselves. Gemini samples the clip server-side at
|
|
1005
|
+
* `video.fps` and, unlike sampled frames, also hears the audio track. Small
|
|
1006
|
+
* videos go inline; larger ones are uploaded via the Files API and deleted
|
|
1007
|
+
* again afterwards (they would expire on their own after 48 h).
|
|
1008
|
+
*/
|
|
1009
|
+
async sendVideoMessage(video, prompt, _options) {
|
|
1010
|
+
const client = await this.getClient();
|
|
1011
|
+
const videoMetadata = { fps: video.fps };
|
|
1012
|
+
if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
|
|
1013
|
+
const part = {
|
|
1014
|
+
inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
|
|
1015
|
+
videoMetadata
|
|
1016
|
+
};
|
|
1017
|
+
const response = await this.generate(client, [part, prompt]);
|
|
1018
|
+
return { ...response, delivery: "inline" };
|
|
1019
|
+
}
|
|
1020
|
+
const file = await this.uploadVideo(client, video);
|
|
1021
|
+
try {
|
|
1022
|
+
const part = {
|
|
1023
|
+
fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
|
|
1024
|
+
videoMetadata
|
|
1025
|
+
};
|
|
1026
|
+
const response = await this.generate(client, [part, prompt]);
|
|
1027
|
+
return { ...response, delivery: "file" };
|
|
1028
|
+
} finally {
|
|
1029
|
+
if (file.name) {
|
|
1030
|
+
await client.files.delete({ name: file.name }).catch(() => void 0);
|
|
1031
|
+
}
|
|
1032
|
+
}
|
|
1033
|
+
}
|
|
882
1034
|
async generateImage(images, prompt, options) {
|
|
883
1035
|
const client = await this.getClient();
|
|
884
1036
|
const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
|
|
@@ -1011,6 +1163,9 @@ var OpenAIDriver = class {
|
|
|
1011
1163
|
// src/providers/openrouter.ts
|
|
1012
1164
|
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
1013
1165
|
var OPENROUTER_REASONING_EFFORT = {
|
|
1166
|
+
// OpenRouter normalizes upstream vendors to low/medium/high only, so
|
|
1167
|
+
// "minimal" has no native equivalent and clamps to the floor.
|
|
1168
|
+
minimal: "low",
|
|
1014
1169
|
low: "low",
|
|
1015
1170
|
medium: "medium",
|
|
1016
1171
|
high: "high",
|
|
@@ -1170,10 +1325,20 @@ function parseBooleanEnv(envName, value) {
|
|
|
1170
1325
|
`Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
|
|
1171
1326
|
);
|
|
1172
1327
|
}
|
|
1328
|
+
function parseReasoningEffortEnv(envName, value) {
|
|
1329
|
+
if (value === void 0 || value === "") return void 0;
|
|
1330
|
+
const levels = Object.values(ReasoningEffort);
|
|
1331
|
+
const lower = value.toLowerCase();
|
|
1332
|
+
if (levels.includes(lower)) return lower;
|
|
1333
|
+
throw new VisualAIConfigError(
|
|
1334
|
+
`Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
|
|
1335
|
+
);
|
|
1336
|
+
}
|
|
1173
1337
|
var debugDeprecationWarned = false;
|
|
1174
1338
|
function resolveConfig(config) {
|
|
1175
1339
|
const provider = resolveProvider(config);
|
|
1176
1340
|
const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
|
|
1341
|
+
const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
|
|
1177
1342
|
const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
|
|
1178
1343
|
const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
|
|
1179
1344
|
const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
|
|
@@ -1191,12 +1356,12 @@ function resolveConfig(config) {
|
|
|
1191
1356
|
}
|
|
1192
1357
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1193
1358
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1194
|
-
const effortNeedsLargeBudget =
|
|
1359
|
+
const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
|
|
1195
1360
|
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1196
1361
|
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1197
1362
|
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1198
1363
|
if (debug) {
|
|
1199
|
-
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${
|
|
1364
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
|
|
1200
1365
|
process.stderr.write(
|
|
1201
1366
|
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1202
1367
|
`
|
|
@@ -1208,7 +1373,7 @@ function resolveConfig(config) {
|
|
|
1208
1373
|
apiKey: config.apiKey,
|
|
1209
1374
|
model,
|
|
1210
1375
|
maxTokens,
|
|
1211
|
-
reasoningEffort
|
|
1376
|
+
reasoningEffort,
|
|
1212
1377
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1213
1378
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1214
1379
|
timeout: config.timeout,
|
|
@@ -1231,6 +1396,10 @@ var PRICING_TABLE = {
|
|
|
1231
1396
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1232
1397
|
outputPricePerToken: 50 / PER_MILLION
|
|
1233
1398
|
},
|
|
1399
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
|
|
1400
|
+
inputPricePerToken: 4 / PER_MILLION,
|
|
1401
|
+
outputPricePerToken: 20 / PER_MILLION
|
|
1402
|
+
},
|
|
1234
1403
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1235
1404
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1236
1405
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1239,6 +1408,10 @@ var PRICING_TABLE = {
|
|
|
1239
1408
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1240
1409
|
outputPricePerToken: 25 / PER_MILLION
|
|
1241
1410
|
},
|
|
1411
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
|
|
1412
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1413
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1414
|
+
},
|
|
1242
1415
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
|
|
1243
1416
|
inputPricePerToken: 3 / PER_MILLION,
|
|
1244
1417
|
outputPricePerToken: 15 / PER_MILLION
|
|
@@ -1265,6 +1438,22 @@ var PRICING_TABLE = {
|
|
|
1265
1438
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1266
1439
|
outputPricePerToken: 50 / PER_MILLION
|
|
1267
1440
|
},
|
|
1441
|
+
// Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
|
|
1442
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
|
|
1443
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1444
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1445
|
+
},
|
|
1446
|
+
// Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
|
|
1447
|
+
// bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
|
|
1448
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
|
|
1449
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1450
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1451
|
+
},
|
|
1452
|
+
// Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
|
|
1453
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
|
|
1454
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1455
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1456
|
+
},
|
|
1268
1457
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1269
1458
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1270
1459
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1361,6 +1550,10 @@ var PRICING_TABLE = {
|
|
|
1361
1550
|
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1362
1551
|
outputPricePerToken: 0.2 / PER_MILLION
|
|
1363
1552
|
},
|
|
1553
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
|
|
1554
|
+
inputPricePerToken: 1.6 / PER_MILLION,
|
|
1555
|
+
outputPricePerToken: 4.8 / PER_MILLION
|
|
1556
|
+
},
|
|
1364
1557
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1365
1558
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1366
1559
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1394,6 +1587,13 @@ var PRICING_TABLE = {
|
|
|
1394
1587
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1395
1588
|
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1396
1589
|
outputPricePerToken: 0.5 / PER_MILLION
|
|
1590
|
+
},
|
|
1591
|
+
// Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
|
|
1592
|
+
// upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
|
|
1593
|
+
// is $0.0036/MTok, not modelled (no provider gets a cache discount here).
|
|
1594
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
|
|
1595
|
+
inputPricePerToken: 0.435 / PER_MILLION,
|
|
1596
|
+
outputPricePerToken: 0.87 / PER_MILLION
|
|
1397
1597
|
}
|
|
1398
1598
|
};
|
|
1399
1599
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -1471,6 +1671,15 @@ async function timedSendMessage(driver, images, prompt, options) {
|
|
|
1471
1671
|
const durationSeconds = (performance.now() - start) / 1e3;
|
|
1472
1672
|
return { ...response, durationSeconds };
|
|
1473
1673
|
}
|
|
1674
|
+
async function timedSendVideoMessage(driver, video, prompt, options) {
|
|
1675
|
+
if (!driver.sendVideoMessage) {
|
|
1676
|
+
throw new VisualAIError("Provider driver does not support native video delivery");
|
|
1677
|
+
}
|
|
1678
|
+
const start = performance.now();
|
|
1679
|
+
const response = await driver.sendVideoMessage(video, prompt, options);
|
|
1680
|
+
const durationSeconds = (performance.now() - start) / 1e3;
|
|
1681
|
+
return { ...response, durationSeconds };
|
|
1682
|
+
}
|
|
1474
1683
|
|
|
1475
1684
|
// src/core/diff.ts
|
|
1476
1685
|
var import_sharp = __toESM(require("sharp"), 1);
|
|
@@ -1710,6 +1919,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
|
|
|
1710
1919
|
};
|
|
1711
1920
|
}
|
|
1712
1921
|
|
|
1922
|
+
// src/core/media.ts
|
|
1923
|
+
var import_promises4 = require("fs/promises");
|
|
1924
|
+
|
|
1713
1925
|
// src/core/debug-frames.ts
|
|
1714
1926
|
var import_node_crypto = require("crypto");
|
|
1715
1927
|
var import_promises2 = require("fs/promises");
|
|
@@ -1765,6 +1977,92 @@ async function saveDebugFrames(frames, env = process.env) {
|
|
|
1765
1977
|
return runDir;
|
|
1766
1978
|
}
|
|
1767
1979
|
|
|
1980
|
+
// src/core/frame-dedupe.ts
|
|
1981
|
+
var import_sharp3 = __toESM(require("sharp"), 1);
|
|
1982
|
+
var DEDUPE_THUMBNAIL_EDGE = 256;
|
|
1983
|
+
var DEDUPE_PIXEL_TOLERANCE = 24;
|
|
1984
|
+
var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
|
|
1985
|
+
function resolveDedupeOptions(raw) {
|
|
1986
|
+
if (raw === void 0 || raw === true) {
|
|
1987
|
+
return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
|
|
1988
|
+
}
|
|
1989
|
+
if (raw === false) {
|
|
1990
|
+
return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
|
|
1991
|
+
}
|
|
1992
|
+
const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
|
|
1993
|
+
if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
|
|
1994
|
+
throw new VisualAIVideoError(
|
|
1995
|
+
`Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
|
|
1996
|
+
);
|
|
1997
|
+
}
|
|
1998
|
+
return { enabled: true, threshold };
|
|
1999
|
+
}
|
|
2000
|
+
async function frameSignature(frame) {
|
|
2001
|
+
try {
|
|
2002
|
+
const { data, info } = await (0, import_sharp3.default)(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
|
|
2003
|
+
fit: "inside",
|
|
2004
|
+
withoutEnlargement: true
|
|
2005
|
+
}).raw().toBuffer({ resolveWithObject: true });
|
|
2006
|
+
return { width: info.width, height: info.height, pixels: data };
|
|
2007
|
+
} catch (err) {
|
|
2008
|
+
const reason = err instanceof Error ? err.message : String(err);
|
|
2009
|
+
throw new VisualAIVideoError(
|
|
2010
|
+
`Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
|
|
2011
|
+
);
|
|
2012
|
+
}
|
|
2013
|
+
}
|
|
2014
|
+
function changedFraction(a, b) {
|
|
2015
|
+
if (a.width !== b.width || a.height !== b.height) {
|
|
2016
|
+
return 1;
|
|
2017
|
+
}
|
|
2018
|
+
let changed = 0;
|
|
2019
|
+
for (let i = 0; i < a.pixels.length; i++) {
|
|
2020
|
+
if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
|
|
2021
|
+
changed++;
|
|
2022
|
+
}
|
|
2023
|
+
}
|
|
2024
|
+
return changed / a.pixels.length;
|
|
2025
|
+
}
|
|
2026
|
+
function reindex(frame, index) {
|
|
2027
|
+
if (frame.index === index) {
|
|
2028
|
+
return frame;
|
|
2029
|
+
}
|
|
2030
|
+
return {
|
|
2031
|
+
data: frame.data,
|
|
2032
|
+
mimeType: frame.mimeType,
|
|
2033
|
+
get base64() {
|
|
2034
|
+
return frame.base64;
|
|
2035
|
+
},
|
|
2036
|
+
timestampSeconds: frame.timestampSeconds,
|
|
2037
|
+
index
|
|
2038
|
+
};
|
|
2039
|
+
}
|
|
2040
|
+
async function dedupeFrames(frames, options) {
|
|
2041
|
+
const { enabled, threshold } = resolveDedupeOptions(options);
|
|
2042
|
+
if (!enabled || frames.length < 2) {
|
|
2043
|
+
return { frames: [...frames], dropped: 0 };
|
|
2044
|
+
}
|
|
2045
|
+
const signed = await Promise.all(
|
|
2046
|
+
frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
|
|
2047
|
+
);
|
|
2048
|
+
const [first, ...rest] = signed;
|
|
2049
|
+
if (first === void 0) {
|
|
2050
|
+
return { frames: [], dropped: 0 };
|
|
2051
|
+
}
|
|
2052
|
+
const kept = [first.frame];
|
|
2053
|
+
let lastKept = first.signature;
|
|
2054
|
+
for (const { frame, signature } of rest) {
|
|
2055
|
+
if (changedFraction(lastKept, signature) >= threshold) {
|
|
2056
|
+
kept.push(frame);
|
|
2057
|
+
lastKept = signature;
|
|
2058
|
+
}
|
|
2059
|
+
}
|
|
2060
|
+
return {
|
|
2061
|
+
frames: kept.map((frame, index) => reindex(frame, index)),
|
|
2062
|
+
dropped: frames.length - kept.length
|
|
2063
|
+
};
|
|
2064
|
+
}
|
|
2065
|
+
|
|
1768
2066
|
// src/core/video.ts
|
|
1769
2067
|
var import_promises3 = require("fs/promises");
|
|
1770
2068
|
var import_node_os = require("os");
|
|
@@ -2000,7 +2298,7 @@ async function probeDurationSeconds(videoPath) {
|
|
|
2000
2298
|
});
|
|
2001
2299
|
});
|
|
2002
2300
|
}
|
|
2003
|
-
|
|
2301
|
+
function resolveVideoSamplingOptions(options = {}) {
|
|
2004
2302
|
const fps = options.fps ?? DEFAULT_FPS;
|
|
2005
2303
|
const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
|
|
2006
2304
|
const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
|
|
@@ -2020,13 +2318,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
|
|
|
2020
2318
|
`Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
|
|
2021
2319
|
);
|
|
2022
2320
|
}
|
|
2023
|
-
|
|
2024
|
-
|
|
2321
|
+
return { fps, maxFrames, maxDurationSeconds };
|
|
2322
|
+
}
|
|
2323
|
+
function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
|
|
2025
2324
|
if (durationSeconds > maxDurationSeconds) {
|
|
2026
2325
|
throw new VisualAIVideoError(
|
|
2027
2326
|
`Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
|
|
2028
2327
|
);
|
|
2029
2328
|
}
|
|
2329
|
+
}
|
|
2330
|
+
async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
|
|
2331
|
+
const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
|
|
2332
|
+
const ffmpeg = await loadFfmpegFactory();
|
|
2333
|
+
const durationSeconds = await probeDurationSeconds(videoPath);
|
|
2334
|
+
assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
|
|
2030
2335
|
const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
|
|
2031
2336
|
try {
|
|
2032
2337
|
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
|
|
@@ -2130,6 +2435,7 @@ function isTimestampedFrameInput(frame) {
|
|
|
2130
2435
|
async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
2131
2436
|
const rawFrames = input.frames;
|
|
2132
2437
|
const fps = input.fps ?? DEFAULT_FPS;
|
|
2438
|
+
resolveDedupeOptions(input.dedupe);
|
|
2133
2439
|
if (rawFrames.length === 0) {
|
|
2134
2440
|
throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
|
|
2135
2441
|
}
|
|
@@ -2141,7 +2447,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
|
|
|
2141
2447
|
if (!Number.isFinite(fps) || fps <= 0) {
|
|
2142
2448
|
throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
|
|
2143
2449
|
}
|
|
2144
|
-
const
|
|
2450
|
+
const sampled = await Promise.all(
|
|
2145
2451
|
rawFrames.map(async (raw, index) => {
|
|
2146
2452
|
const timestamped = isTimestampedFrameInput(raw);
|
|
2147
2453
|
const imageInput = timestamped ? raw.image : raw;
|
|
@@ -2164,20 +2470,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
|
|
|
2164
2470
|
};
|
|
2165
2471
|
})
|
|
2166
2472
|
);
|
|
2167
|
-
const durationSeconds =
|
|
2473
|
+
const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
|
|
2474
|
+
const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
|
|
2168
2475
|
await saveDebugFrames(frames);
|
|
2169
|
-
return { kind: "video", frames, durationSeconds };
|
|
2476
|
+
return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
|
|
2477
|
+
}
|
|
2478
|
+
async function normalizeNativeVideo(input, videoOptions) {
|
|
2479
|
+
const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
|
|
2480
|
+
const { path, mimeType, cleanup } = await resolveVideoToPath(input);
|
|
2481
|
+
try {
|
|
2482
|
+
const durationSeconds = await probeDurationSeconds(path);
|
|
2483
|
+
assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
|
|
2484
|
+
const data = await (0, import_promises4.readFile)(path);
|
|
2485
|
+
return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
|
|
2486
|
+
} finally {
|
|
2487
|
+
try {
|
|
2488
|
+
await cleanup();
|
|
2489
|
+
} catch {
|
|
2490
|
+
}
|
|
2491
|
+
}
|
|
2170
2492
|
}
|
|
2171
|
-
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
2493
|
+
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
|
|
2172
2494
|
if (isFramesInput(input)) {
|
|
2173
2495
|
return normalizeFrames(input, maxDimension);
|
|
2174
2496
|
}
|
|
2175
2497
|
if (isVideoInput(input)) {
|
|
2498
|
+
if (nativeVideo) {
|
|
2499
|
+
return normalizeNativeVideo(input, videoOptions);
|
|
2500
|
+
}
|
|
2501
|
+
resolveDedupeOptions(videoOptions?.dedupe);
|
|
2176
2502
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
2177
2503
|
try {
|
|
2178
|
-
const { frames, durationSeconds } = await extractFrames(
|
|
2504
|
+
const { frames: sampled, durationSeconds } = await extractFrames(
|
|
2505
|
+
path,
|
|
2506
|
+
videoOptions,
|
|
2507
|
+
maxDimension
|
|
2508
|
+
);
|
|
2509
|
+
const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
|
|
2179
2510
|
await saveDebugFrames(frames);
|
|
2180
|
-
return { kind: "video", frames, durationSeconds };
|
|
2511
|
+
return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
|
|
2181
2512
|
} finally {
|
|
2182
2513
|
try {
|
|
2183
2514
|
await cleanup();
|
|
@@ -2267,6 +2598,12 @@ var AskResultSchema = import_zod.z.object({
|
|
|
2267
2598
|
* omitting the key, even for image inputs that were never asked to populate it.
|
|
2268
2599
|
*/
|
|
2269
2600
|
frameReferences: import_zod.z.array(import_zod.z.number().int().nonnegative()).nullable().optional(),
|
|
2601
|
+
/**
|
|
2602
|
+
* For natively delivered video, the timestamps (seconds from the start of
|
|
2603
|
+
* the clip) the model relied on to answer. The native counterpart of
|
|
2604
|
+
* `frameReferences`. Nullable for the same strict-schema reason.
|
|
2605
|
+
*/
|
|
2606
|
+
timestampReferences: import_zod.z.array(import_zod.z.number().nonnegative()).nullable().optional(),
|
|
2270
2607
|
usage: UsageInfoSchema.optional()
|
|
2271
2608
|
});
|
|
2272
2609
|
|
|
@@ -2278,6 +2615,12 @@ function stripCodeFences(text) {
|
|
|
2278
2615
|
}
|
|
2279
2616
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2280
2617
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2618
|
+
var AskImageResponseSchema = AskResponseSchema.omit({
|
|
2619
|
+
frameReferences: true,
|
|
2620
|
+
timestampReferences: true
|
|
2621
|
+
});
|
|
2622
|
+
var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
|
|
2623
|
+
var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
|
|
2281
2624
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2282
2625
|
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2283
2626
|
function parseJson(text) {
|
|
@@ -2345,7 +2688,8 @@ function parseAskResponse(raw) {
|
|
|
2345
2688
|
const result = parseResponse(raw, AskResponseSchema);
|
|
2346
2689
|
return {
|
|
2347
2690
|
...result,
|
|
2348
|
-
frameReferences: result.frameReferences ?? void 0
|
|
2691
|
+
frameReferences: result.frameReferences ?? void 0,
|
|
2692
|
+
timestampReferences: result.timestampReferences ?? void 0
|
|
2349
2693
|
};
|
|
2350
2694
|
}
|
|
2351
2695
|
function parseCompareResponse(raw) {
|
|
@@ -2369,31 +2713,75 @@ function createDriver(provider, config) {
|
|
|
2369
2713
|
return PROVIDER_REGISTRY[provider](config);
|
|
2370
2714
|
}
|
|
2371
2715
|
var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
|
|
2372
|
-
var
|
|
2716
|
+
var askSchemaOptionsByMedia = {
|
|
2717
|
+
image: toSchemaOptions(AskImageResponseSchema),
|
|
2718
|
+
video: toSchemaOptions(AskFramesResponseSchema),
|
|
2719
|
+
"native-video": toSchemaOptions(AskNativeVideoResponseSchema)
|
|
2720
|
+
};
|
|
2373
2721
|
var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
|
|
2374
2722
|
function mediaToProviderInputs(media) {
|
|
2375
2723
|
if (media.kind === "image") {
|
|
2376
2724
|
return {
|
|
2725
|
+
kind: "images",
|
|
2377
2726
|
images: [media.image],
|
|
2378
2727
|
mediaContext: { kind: "image" },
|
|
2379
|
-
|
|
2728
|
+
frames: void 0
|
|
2729
|
+
};
|
|
2730
|
+
}
|
|
2731
|
+
if (media.kind === "native-video") {
|
|
2732
|
+
return {
|
|
2733
|
+
kind: "native-video",
|
|
2734
|
+
video: media.video,
|
|
2735
|
+
mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
|
|
2380
2736
|
};
|
|
2381
2737
|
}
|
|
2382
2738
|
const timestamps = media.frames.map((f) => f.timestampSeconds);
|
|
2383
2739
|
return {
|
|
2740
|
+
kind: "images",
|
|
2384
2741
|
images: media.frames,
|
|
2385
2742
|
mediaContext: {
|
|
2386
2743
|
kind: "video",
|
|
2387
2744
|
frameTimestamps: timestamps,
|
|
2388
|
-
durationSeconds: media.durationSeconds
|
|
2745
|
+
durationSeconds: media.durationSeconds,
|
|
2746
|
+
droppedUnchanged: media.droppedUnchanged
|
|
2389
2747
|
},
|
|
2390
|
-
|
|
2748
|
+
frames: {
|
|
2391
2749
|
count: media.frames.length,
|
|
2392
2750
|
timestampsSeconds: timestamps,
|
|
2393
|
-
durationSeconds: media.durationSeconds
|
|
2751
|
+
durationSeconds: media.durationSeconds,
|
|
2752
|
+
droppedUnchanged: media.droppedUnchanged
|
|
2394
2753
|
}
|
|
2395
2754
|
};
|
|
2396
2755
|
}
|
|
2756
|
+
async function sendMedia(driver, dispatch, prompt, options) {
|
|
2757
|
+
if (dispatch.kind === "native-video") {
|
|
2758
|
+
const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
|
|
2759
|
+
const { durationSeconds, fps, mimeType } = dispatch.video;
|
|
2760
|
+
return {
|
|
2761
|
+
response: response2,
|
|
2762
|
+
metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
|
|
2763
|
+
};
|
|
2764
|
+
}
|
|
2765
|
+
const response = await timedSendMessage(driver, dispatch.images, prompt, options);
|
|
2766
|
+
return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
|
|
2767
|
+
}
|
|
2768
|
+
function resolveNativeVideo(input, videoOptions, driver, provider) {
|
|
2769
|
+
const mode = videoOptions?.mode ?? "auto";
|
|
2770
|
+
if (mode !== "auto" && mode !== "native" && mode !== "frames") {
|
|
2771
|
+
throw new VisualAIConfigError(
|
|
2772
|
+
`Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
|
|
2773
|
+
);
|
|
2774
|
+
}
|
|
2775
|
+
const supported = typeof driver.sendVideoMessage === "function";
|
|
2776
|
+
if (mode === "frames") return false;
|
|
2777
|
+
if (mode === "auto") return supported;
|
|
2778
|
+
if (!supported && !isFramesInput(input) && isVideoInput(input)) {
|
|
2779
|
+
throw new VisualAIConfigError(
|
|
2780
|
+
`Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
|
|
2781
|
+
);
|
|
2782
|
+
}
|
|
2783
|
+
return true;
|
|
2784
|
+
}
|
|
2397
2785
|
function visualAI(config = {}) {
|
|
2398
2786
|
const resolvedConfig = resolveConfig(config);
|
|
2399
2787
|
const driverConfig = {
|
|
@@ -2431,38 +2819,60 @@ function visualAI(config = {}) {
|
|
|
2431
2819
|
throw new VisualAIConfigError("At least one statement is required for check()");
|
|
2432
2820
|
}
|
|
2433
2821
|
return withErrorDebug(resolvedConfig, "check", async () => {
|
|
2434
|
-
const
|
|
2435
|
-
|
|
2822
|
+
const nativeVideo = resolveNativeVideo(
|
|
2823
|
+
input,
|
|
2824
|
+
options?.video,
|
|
2825
|
+
driver,
|
|
2826
|
+
resolvedConfig.provider
|
|
2827
|
+
);
|
|
2828
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
|
|
2829
|
+
const dispatch = mediaToProviderInputs(media);
|
|
2436
2830
|
const prompt = buildCheckPrompt(stmts, {
|
|
2437
2831
|
instructions: options?.instructions,
|
|
2438
|
-
media: mediaContext
|
|
2832
|
+
media: dispatch.mediaContext
|
|
2439
2833
|
});
|
|
2440
2834
|
debugLog(resolvedConfig, "check prompt", prompt, "prompt");
|
|
2441
|
-
const response = await
|
|
2835
|
+
const { response, metadata } = await sendMedia(
|
|
2836
|
+
driver,
|
|
2837
|
+
dispatch,
|
|
2838
|
+
prompt,
|
|
2839
|
+
checkSchemaOptions
|
|
2840
|
+
);
|
|
2442
2841
|
debugLog(resolvedConfig, "check response", response.text, "response");
|
|
2443
2842
|
const result = parseCheckResponse(response.text);
|
|
2444
2843
|
return {
|
|
2445
2844
|
...result,
|
|
2446
|
-
...
|
|
2845
|
+
...metadata,
|
|
2447
2846
|
usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
|
|
2448
2847
|
};
|
|
2449
2848
|
});
|
|
2450
2849
|
},
|
|
2451
2850
|
async ask(input, userPrompt, options) {
|
|
2452
2851
|
return withErrorDebug(resolvedConfig, "ask", async () => {
|
|
2453
|
-
const
|
|
2454
|
-
|
|
2852
|
+
const nativeVideo = resolveNativeVideo(
|
|
2853
|
+
input,
|
|
2854
|
+
options?.video,
|
|
2855
|
+
driver,
|
|
2856
|
+
resolvedConfig.provider
|
|
2857
|
+
);
|
|
2858
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
|
|
2859
|
+
const dispatch = mediaToProviderInputs(media);
|
|
2455
2860
|
const prompt = buildAskPrompt(userPrompt, {
|
|
2456
2861
|
instructions: options?.instructions,
|
|
2457
|
-
media: mediaContext
|
|
2862
|
+
media: dispatch.mediaContext
|
|
2458
2863
|
});
|
|
2459
2864
|
debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
|
|
2460
|
-
const response = await
|
|
2865
|
+
const { response, metadata } = await sendMedia(
|
|
2866
|
+
driver,
|
|
2867
|
+
dispatch,
|
|
2868
|
+
prompt,
|
|
2869
|
+
askSchemaOptionsByMedia[dispatch.mediaContext.kind]
|
|
2870
|
+
);
|
|
2461
2871
|
debugLog(resolvedConfig, "ask response", response.text, "response");
|
|
2462
2872
|
const result = parseAskResponse(response.text);
|
|
2463
2873
|
return {
|
|
2464
2874
|
...result,
|
|
2465
|
-
...
|
|
2875
|
+
...metadata,
|
|
2466
2876
|
usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
|
|
2467
2877
|
};
|
|
2468
2878
|
});
|
|
@@ -2480,7 +2890,7 @@ function visualAI(config = {}) {
|
|
|
2480
2890
|
debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
|
|
2481
2891
|
const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
|
|
2482
2892
|
debugLog(resolvedConfig, "compare response", response.text, "response");
|
|
2483
|
-
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model
|
|
2893
|
+
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
|
|
2484
2894
|
const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
|
|
2485
2895
|
let diffImage;
|
|
2486
2896
|
if (effectiveDiffImage) {
|