visual-ai-assertions 0.23.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +140 -83
- package/dist/index.cjs +600 -190
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +151 -30
- package/dist/index.d.ts +151 -30
- package/dist/index.js +600 -190
- package/dist/index.js.map +1 -1
- package/package.json +8 -9
package/dist/index.js
CHANGED
|
@@ -1,132 +1,3 @@
|
|
|
1
|
-
// src/constants.ts
|
|
2
|
-
var ReasoningEffort = {
|
|
3
|
-
LOW: "low",
|
|
4
|
-
MEDIUM: "medium",
|
|
5
|
-
HIGH: "high",
|
|
6
|
-
XHIGH: "xhigh"
|
|
7
|
-
};
|
|
8
|
-
var ImageDetail = {
|
|
9
|
-
AUTO: "auto",
|
|
10
|
-
LOW: "low",
|
|
11
|
-
HIGH: "high"
|
|
12
|
-
};
|
|
13
|
-
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
14
|
-
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
15
|
-
var Provider = {
|
|
16
|
-
ANTHROPIC: "anthropic",
|
|
17
|
-
OPENAI: "openai",
|
|
18
|
-
GOOGLE: "google",
|
|
19
|
-
OPENROUTER: "openrouter"
|
|
20
|
-
};
|
|
21
|
-
var Model = {
|
|
22
|
-
Anthropic: {
|
|
23
|
-
FABLE_5_1: "claude-fable-5-1",
|
|
24
|
-
FABLE_5: "claude-fable-5",
|
|
25
|
-
OPUS_5: "claude-opus-5",
|
|
26
|
-
OPUS_4_8: "claude-opus-4-8",
|
|
27
|
-
OPUS_4_7: "claude-opus-4-7",
|
|
28
|
-
OPUS_4_6: "claude-opus-4-6",
|
|
29
|
-
SONNET_5: "claude-sonnet-5",
|
|
30
|
-
SONNET_4_6: "claude-sonnet-4-6",
|
|
31
|
-
HAIKU_4_5: "claude-haiku-4-5"
|
|
32
|
-
},
|
|
33
|
-
OpenAI: {
|
|
34
|
-
GPT_6_ASTRA: "gpt-6-astra",
|
|
35
|
-
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
36
|
-
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
37
|
-
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
38
|
-
GPT_5_5: "gpt-5.5",
|
|
39
|
-
GPT_5_4: "gpt-5.4",
|
|
40
|
-
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
41
|
-
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
42
|
-
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
43
|
-
GPT_5_2: "gpt-5.2",
|
|
44
|
-
GPT_5_MINI: "gpt-5-mini"
|
|
45
|
-
},
|
|
46
|
-
Google: {
|
|
47
|
-
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
48
|
-
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
49
|
-
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
50
|
-
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
51
|
-
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
52
|
-
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
53
|
-
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
54
|
-
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
55
|
-
},
|
|
56
|
-
/**
|
|
57
|
-
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
58
|
-
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
59
|
-
* recognizes them. All listed models accept image input.
|
|
60
|
-
*/
|
|
61
|
-
OpenRouter: {
|
|
62
|
-
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
63
|
-
GROK_4_6: "x-ai/grok-4.6",
|
|
64
|
-
GROK_4_5: "x-ai/grok-4.5",
|
|
65
|
-
KIMI_K3: "moonshotai/kimi-k3",
|
|
66
|
-
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
67
|
-
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
68
|
-
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
69
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
70
|
-
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
71
|
-
}
|
|
72
|
-
};
|
|
73
|
-
var DEFAULT_MODELS = {
|
|
74
|
-
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
75
|
-
[Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
|
|
76
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
77
|
-
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
78
|
-
};
|
|
79
|
-
var DEFAULT_MAX_TOKENS = 4096;
|
|
80
|
-
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
81
|
-
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
82
|
-
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
83
|
-
Model.OpenAI.GPT_6_ASTRA
|
|
84
|
-
]);
|
|
85
|
-
var MODEL_TO_PROVIDER = new Map([
|
|
86
|
-
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
87
|
-
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
88
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
89
|
-
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
90
|
-
]);
|
|
91
|
-
var VALID_PROVIDERS = Object.values(Provider);
|
|
92
|
-
var PROVIDER_DEFAULT_REASONING = {
|
|
93
|
-
openai: "medium",
|
|
94
|
-
anthropic: "off",
|
|
95
|
-
google: "off",
|
|
96
|
-
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
97
|
-
openrouter: "off"
|
|
98
|
-
};
|
|
99
|
-
var Content = {
|
|
100
|
-
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
101
|
-
PLACEHOLDER_TEXT: "placeholder-text",
|
|
102
|
-
/** Detects error messages, banners, stack traces, or error codes */
|
|
103
|
-
ERROR_MESSAGES: "error-messages",
|
|
104
|
-
/** Detects broken image icons or failed-to-load image indicators */
|
|
105
|
-
BROKEN_IMAGES: "broken-images",
|
|
106
|
-
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
107
|
-
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
108
|
-
};
|
|
109
|
-
var Layout = {
|
|
110
|
-
/** Detects elements that unintentionally overlap each other */
|
|
111
|
-
OVERLAP: "overlap",
|
|
112
|
-
/** Detects content cut off or extending beyond container boundaries */
|
|
113
|
-
OVERFLOW: "overflow",
|
|
114
|
-
/** Detects inconsistent alignment of text, images, and UI components */
|
|
115
|
-
ALIGNMENT: "alignment"
|
|
116
|
-
};
|
|
117
|
-
var Accessibility = {
|
|
118
|
-
/** Detects insufficient color contrast between text and backgrounds */
|
|
119
|
-
CONTRAST: "contrast",
|
|
120
|
-
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
121
|
-
READABILITY: "readability",
|
|
122
|
-
/** Detects interactive elements that are not visually distinct */
|
|
123
|
-
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
124
|
-
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
125
|
-
COLOR_BLINDNESS: "color-blindness",
|
|
126
|
-
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
127
|
-
COLOR_ALONE: "color-alone"
|
|
128
|
-
};
|
|
129
|
-
|
|
130
1
|
// src/errors.ts
|
|
131
2
|
var VisualAIError = class extends Error {
|
|
132
3
|
code;
|
|
@@ -258,10 +129,15 @@ Example for a failing check:
|
|
|
258
129
|
]
|
|
259
130
|
}
|
|
260
131
|
${JSON_INSTRUCTIONS}`;
|
|
261
|
-
|
|
132
|
+
function buildCheckOutputSchemaVideo(unit) {
|
|
133
|
+
const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
|
|
134
|
+
const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
|
|
135
|
+
const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
|
|
136
|
+
const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
|
|
137
|
+
return `IMPORTANT: Follow this evaluation order:
|
|
262
138
|
1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
|
|
263
|
-
2. A statement passes if it is true at
|
|
264
|
-
3. For each statement that passes, set "timestampSeconds" to
|
|
139
|
+
2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
|
|
140
|
+
3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
|
|
265
141
|
4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
|
|
266
142
|
5. Write "reasoning" as a brief overall summary of the evaluation
|
|
267
143
|
6. Include "issues" only for statements that failed
|
|
@@ -275,7 +151,7 @@ Respond with a JSON object matching this exact structure:
|
|
|
275
151
|
{
|
|
276
152
|
"statement": string, // the original statement text
|
|
277
153
|
"pass": boolean, // whether this statement is true at any point in the timeline
|
|
278
|
-
"reasoning": string, // explanation for this statement, citing
|
|
154
|
+
"reasoning": string, // explanation for this statement, ${citing} where relevant
|
|
279
155
|
"confidence": "high" | "medium" | "low",
|
|
280
156
|
"timestampSeconds": number | null
|
|
281
157
|
// seconds from the start of the clip where the statement is most clearly true,
|
|
@@ -293,10 +169,13 @@ Example for a passing video check:
|
|
|
293
169
|
"reasoning": "The success toast appeared briefly around 3.5s.",
|
|
294
170
|
"issues": [],
|
|
295
171
|
"statements": [
|
|
296
|
-
{ "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right
|
|
172
|
+
{ "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
|
|
297
173
|
]
|
|
298
174
|
}
|
|
299
175
|
${JSON_INSTRUCTIONS}`;
|
|
176
|
+
}
|
|
177
|
+
var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
|
|
178
|
+
var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
|
|
300
179
|
var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
|
|
301
180
|
{
|
|
302
181
|
"summary": string, // high-level analysis summary
|
|
@@ -329,6 +208,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
|
|
|
329
208
|
Prioritize issues by severity (critical / major / minor) as for image input.
|
|
330
209
|
Cite frame indices in "frameReferences" so the user can locate the moments you describe.
|
|
331
210
|
${JSON_INSTRUCTIONS}`;
|
|
211
|
+
var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
|
|
212
|
+
{
|
|
213
|
+
"summary": string, // high-level summary of what happens across the video
|
|
214
|
+
"issues": [...], // list of issues/findings, can be empty
|
|
215
|
+
"timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
|
|
216
|
+
}
|
|
217
|
+
${ISSUE_SCHEMA_INSTRUCTIONS}
|
|
218
|
+
|
|
219
|
+
Prioritize issues by severity (critical / major / minor) as for image input.
|
|
220
|
+
Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
|
|
221
|
+
${JSON_INSTRUCTIONS}`;
|
|
332
222
|
var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
|
|
333
223
|
{
|
|
334
224
|
"pass": boolean, // true if no critical or major changes found
|
|
@@ -350,15 +240,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
|
|
|
350
240
|
var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
|
|
351
241
|
var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
|
|
352
242
|
var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
|
|
353
|
-
|
|
243
|
+
var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
|
|
244
|
+
var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
|
|
245
|
+
function buildNativeVideoSection(durationSeconds) {
|
|
246
|
+
return `Video recording:
|
|
247
|
+
- Total duration: ${durationSeconds.toFixed(2)}s
|
|
248
|
+
|
|
249
|
+
The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
|
|
250
|
+
}
|
|
251
|
+
function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
|
|
354
252
|
const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
|
|
253
|
+
const attached = frameTimestamps.length;
|
|
254
|
+
const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
|
|
255
|
+
const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
|
|
355
256
|
return `Video timeline:
|
|
356
257
|
- Total duration: ${durationSeconds.toFixed(2)}s
|
|
357
|
-
|
|
258
|
+
${sampledLine}
|
|
358
259
|
- Frame index \u2192 timestamp:
|
|
359
260
|
${formatted}
|
|
360
261
|
|
|
361
|
-
Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful
|
|
262
|
+
Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
|
|
362
263
|
}
|
|
363
264
|
var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
|
|
364
265
|
var COMPARE_EDGE_RULES = [
|
|
@@ -373,30 +274,52 @@ function buildCheckPrompt(statements, options) {
|
|
|
373
274
|
const stmts = Array.isArray(statements) ? statements : [statements];
|
|
374
275
|
const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
|
|
375
276
|
const media = options?.media;
|
|
376
|
-
const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
|
|
277
|
+
const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
|
|
377
278
|
const sections = [options?.role ?? defaultRole];
|
|
378
279
|
if (media?.kind === "video") {
|
|
379
|
-
sections.push(
|
|
280
|
+
sections.push(
|
|
281
|
+
buildVideoTimelineSection(
|
|
282
|
+
media.frameTimestamps,
|
|
283
|
+
media.durationSeconds,
|
|
284
|
+
media.droppedUnchanged
|
|
285
|
+
)
|
|
286
|
+
);
|
|
287
|
+
} else if (media?.kind === "native-video") {
|
|
288
|
+
sections.push(buildNativeVideoSection(media.durationSeconds));
|
|
380
289
|
}
|
|
381
290
|
if (options?.instructions && options.instructions.length > 0) {
|
|
382
291
|
sections.push(buildInstructionsSection(options.instructions));
|
|
383
292
|
}
|
|
384
293
|
sections.push(`Statements to evaluate:
|
|
385
294
|
${statementsBlock}`);
|
|
386
|
-
sections.push(
|
|
295
|
+
sections.push(
|
|
296
|
+
media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
|
|
297
|
+
);
|
|
387
298
|
return sections.join("\n\n");
|
|
388
299
|
}
|
|
389
300
|
function buildAskPrompt(userPrompt, options) {
|
|
390
301
|
const media = options?.media;
|
|
391
|
-
const sections = [
|
|
302
|
+
const sections = [
|
|
303
|
+
media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
|
|
304
|
+
];
|
|
392
305
|
if (media?.kind === "video") {
|
|
393
|
-
sections.push(
|
|
306
|
+
sections.push(
|
|
307
|
+
buildVideoTimelineSection(
|
|
308
|
+
media.frameTimestamps,
|
|
309
|
+
media.durationSeconds,
|
|
310
|
+
media.droppedUnchanged
|
|
311
|
+
)
|
|
312
|
+
);
|
|
313
|
+
} else if (media?.kind === "native-video") {
|
|
314
|
+
sections.push(buildNativeVideoSection(media.durationSeconds));
|
|
394
315
|
}
|
|
395
316
|
if (options?.instructions && options.instructions.length > 0) {
|
|
396
317
|
sections.push(buildInstructionsSection(options.instructions));
|
|
397
318
|
}
|
|
398
319
|
sections.push(`User request: ${userPrompt}`);
|
|
399
|
-
sections.push(
|
|
320
|
+
sections.push(
|
|
321
|
+
media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
|
|
322
|
+
);
|
|
400
323
|
return sections.join("\n\n");
|
|
401
324
|
}
|
|
402
325
|
function buildAiDiffPrompt() {
|
|
@@ -441,7 +364,7 @@ function visibleRole(finalState, requireCorrectRendering) {
|
|
|
441
364
|
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
442
365
|
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
443
366
|
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
444
|
-
"
|
|
367
|
+
"The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
|
|
445
368
|
];
|
|
446
369
|
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
447
370
|
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
@@ -472,6 +395,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
|
472
395
|
return buildCheckPrompt(statements, { role, instructions });
|
|
473
396
|
}
|
|
474
397
|
|
|
398
|
+
// src/constants.ts
|
|
399
|
+
var ReasoningEffort = {
|
|
400
|
+
MINIMAL: "minimal",
|
|
401
|
+
LOW: "low",
|
|
402
|
+
MEDIUM: "medium",
|
|
403
|
+
HIGH: "high",
|
|
404
|
+
XHIGH: "xhigh"
|
|
405
|
+
};
|
|
406
|
+
var ImageDetail = {
|
|
407
|
+
AUTO: "auto",
|
|
408
|
+
LOW: "low",
|
|
409
|
+
HIGH: "high"
|
|
410
|
+
};
|
|
411
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
412
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
413
|
+
var Provider = {
|
|
414
|
+
ANTHROPIC: "anthropic",
|
|
415
|
+
OPENAI: "openai",
|
|
416
|
+
GOOGLE: "google",
|
|
417
|
+
OPENROUTER: "openrouter"
|
|
418
|
+
};
|
|
419
|
+
var Model = {
|
|
420
|
+
Anthropic: {
|
|
421
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
422
|
+
FABLE_5: "claude-fable-5",
|
|
423
|
+
OPUS_5_5: "claude-opus-5-5",
|
|
424
|
+
OPUS_5: "claude-opus-5",
|
|
425
|
+
OPUS_4_8: "claude-opus-4-8",
|
|
426
|
+
OPUS_4_7: "claude-opus-4-7",
|
|
427
|
+
OPUS_4_6: "claude-opus-4-6",
|
|
428
|
+
SONNET_5_5: "claude-sonnet-5-5",
|
|
429
|
+
SONNET_5: "claude-sonnet-5",
|
|
430
|
+
SONNET_4_6: "claude-sonnet-4-6",
|
|
431
|
+
HAIKU_4_5: "claude-haiku-4-5"
|
|
432
|
+
},
|
|
433
|
+
OpenAI: {
|
|
434
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
435
|
+
GPT_6_1_SOL: "gpt-6.1-sol",
|
|
436
|
+
GPT_6_SOL: "gpt-6-sol",
|
|
437
|
+
GPT_6_LUNA: "gpt-6-luna",
|
|
438
|
+
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
439
|
+
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
440
|
+
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
441
|
+
GPT_5_5: "gpt-5.5",
|
|
442
|
+
GPT_5_4: "gpt-5.4",
|
|
443
|
+
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
444
|
+
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
445
|
+
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
446
|
+
GPT_5_2: "gpt-5.2",
|
|
447
|
+
GPT_5_MINI: "gpt-5-mini"
|
|
448
|
+
},
|
|
449
|
+
Google: {
|
|
450
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
451
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
452
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
453
|
+
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
454
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
455
|
+
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
456
|
+
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
457
|
+
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
458
|
+
},
|
|
459
|
+
/**
|
|
460
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
461
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
462
|
+
* recognizes them. All listed models accept image input.
|
|
463
|
+
*/
|
|
464
|
+
OpenRouter: {
|
|
465
|
+
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
466
|
+
GROK_4_7: "x-ai/grok-4.7",
|
|
467
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
468
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
469
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
470
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
471
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
472
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
473
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
474
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
|
|
475
|
+
MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
|
|
476
|
+
}
|
|
477
|
+
};
|
|
478
|
+
var DEFAULT_MODELS = {
|
|
479
|
+
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
|
|
480
|
+
[Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
|
|
481
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
|
|
482
|
+
[Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
|
|
483
|
+
};
|
|
484
|
+
var DEFAULT_MAX_TOKENS = 4096;
|
|
485
|
+
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
486
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
487
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
488
|
+
Model.OpenRouter.QWEN_3_8_MAX,
|
|
489
|
+
Model.OpenRouter.QWEN_3_7_PLUS
|
|
490
|
+
]);
|
|
491
|
+
var MODEL_TO_PROVIDER = new Map([
|
|
492
|
+
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
493
|
+
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
494
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
495
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
496
|
+
]);
|
|
497
|
+
var VALID_PROVIDERS = Object.values(Provider);
|
|
498
|
+
var PROVIDER_DEFAULT_REASONING = {
|
|
499
|
+
openai: "medium",
|
|
500
|
+
anthropic: "off",
|
|
501
|
+
google: "off",
|
|
502
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
503
|
+
openrouter: "off"
|
|
504
|
+
};
|
|
505
|
+
var Content = {
|
|
506
|
+
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
507
|
+
PLACEHOLDER_TEXT: "placeholder-text",
|
|
508
|
+
/** Detects error messages, banners, stack traces, or error codes */
|
|
509
|
+
ERROR_MESSAGES: "error-messages",
|
|
510
|
+
/** Detects broken image icons or failed-to-load image indicators */
|
|
511
|
+
BROKEN_IMAGES: "broken-images",
|
|
512
|
+
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
513
|
+
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
514
|
+
};
|
|
515
|
+
var Layout = {
|
|
516
|
+
/** Detects elements that unintentionally overlap each other */
|
|
517
|
+
OVERLAP: "overlap",
|
|
518
|
+
/** Detects content cut off or extending beyond container boundaries */
|
|
519
|
+
OVERFLOW: "overflow",
|
|
520
|
+
/** Detects inconsistent alignment of text, images, and UI components */
|
|
521
|
+
ALIGNMENT: "alignment"
|
|
522
|
+
};
|
|
523
|
+
var Accessibility = {
|
|
524
|
+
/** Detects insufficient color contrast between text and backgrounds */
|
|
525
|
+
CONTRAST: "contrast",
|
|
526
|
+
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
527
|
+
READABILITY: "readability",
|
|
528
|
+
/** Detects interactive elements that are not visually distinct */
|
|
529
|
+
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
530
|
+
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
531
|
+
COLOR_BLINDNESS: "color-blindness",
|
|
532
|
+
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
533
|
+
COLOR_ALONE: "color-alone"
|
|
534
|
+
};
|
|
535
|
+
|
|
475
536
|
// src/templates/accessibility.ts
|
|
476
537
|
var ALL_CHECKS = Object.values(Accessibility);
|
|
477
538
|
var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
|
|
@@ -581,17 +642,22 @@ function parseRetryAfter(value) {
|
|
|
581
642
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
582
643
|
Model.Anthropic.FABLE_5_1,
|
|
583
644
|
Model.Anthropic.FABLE_5,
|
|
645
|
+
Model.Anthropic.OPUS_5_5,
|
|
584
646
|
Model.Anthropic.OPUS_5,
|
|
585
647
|
Model.Anthropic.OPUS_4_8,
|
|
586
648
|
Model.Anthropic.OPUS_4_7,
|
|
649
|
+
Model.Anthropic.SONNET_5_5,
|
|
587
650
|
Model.Anthropic.SONNET_5
|
|
588
651
|
]);
|
|
589
652
|
function mapEffort(level, model) {
|
|
653
|
+
if (level === "minimal") return "low";
|
|
590
654
|
if (level !== "xhigh") return level;
|
|
591
655
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
592
656
|
}
|
|
593
657
|
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
594
658
|
var EFFORT_TO_BUDGET_TOKENS = {
|
|
659
|
+
// 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
|
|
660
|
+
minimal: 1024,
|
|
595
661
|
low: 1024,
|
|
596
662
|
medium: 4096,
|
|
597
663
|
high: 8192,
|
|
@@ -697,11 +763,20 @@ var AnthropicDriver = class {
|
|
|
697
763
|
|
|
698
764
|
// src/providers/google.ts
|
|
699
765
|
var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
|
|
766
|
+
var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
|
|
767
|
+
var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
|
|
768
|
+
var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
|
|
700
769
|
function needsCodeExecution(model) {
|
|
701
770
|
const match = model.match(/^gemini-(\d+)/);
|
|
702
771
|
return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
|
|
703
772
|
}
|
|
773
|
+
function sleep(ms) {
|
|
774
|
+
return new Promise((resolve2) => setTimeout(resolve2, ms));
|
|
775
|
+
}
|
|
704
776
|
var GOOGLE_THINKING_LEVEL = {
|
|
777
|
+
// Gemini does define a "minimal" thinking level, but some models reject it
|
|
778
|
+
// (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
|
|
779
|
+
minimal: "low",
|
|
705
780
|
low: "low",
|
|
706
781
|
medium: "medium",
|
|
707
782
|
high: "high",
|
|
@@ -768,24 +843,28 @@ var GoogleDriver = class {
|
|
|
768
843
|
});
|
|
769
844
|
return this.client;
|
|
770
845
|
}
|
|
771
|
-
|
|
772
|
-
|
|
846
|
+
/** Request config shared by image and video messages. */
|
|
847
|
+
generationConfig() {
|
|
848
|
+
return {
|
|
849
|
+
responseMimeType: "application/json",
|
|
850
|
+
maxOutputTokens: this.maxTokens,
|
|
851
|
+
...this.reasoningEffort && {
|
|
852
|
+
thinkingConfig: {
|
|
853
|
+
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
854
|
+
}
|
|
855
|
+
},
|
|
856
|
+
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
857
|
+
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
858
|
+
}
|
|
859
|
+
};
|
|
860
|
+
}
|
|
861
|
+
/** Runs one generateContent call and normalizes finish reasons, text, and usage. */
|
|
862
|
+
async generate(client, contents) {
|
|
773
863
|
try {
|
|
774
864
|
const response = await client.models.generateContent({
|
|
775
865
|
model: this.model,
|
|
776
|
-
contents
|
|
777
|
-
config:
|
|
778
|
-
responseMimeType: "application/json",
|
|
779
|
-
maxOutputTokens: this.maxTokens,
|
|
780
|
-
...this.reasoningEffort && {
|
|
781
|
-
thinkingConfig: {
|
|
782
|
-
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
783
|
-
}
|
|
784
|
-
},
|
|
785
|
-
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
786
|
-
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
787
|
-
}
|
|
788
|
-
}
|
|
866
|
+
contents,
|
|
867
|
+
config: this.generationConfig()
|
|
789
868
|
});
|
|
790
869
|
const finishReason = response.candidates?.[0]?.finishReason;
|
|
791
870
|
if (finishReason === "MAX_TOKENS") {
|
|
@@ -800,9 +879,8 @@ var GoogleDriver = class {
|
|
|
800
879
|
`Response blocked: Google returned finishReason "${finishReason}".`
|
|
801
880
|
);
|
|
802
881
|
}
|
|
803
|
-
const text = response.text ?? "";
|
|
804
882
|
return {
|
|
805
|
-
text,
|
|
883
|
+
text: response.text ?? "",
|
|
806
884
|
usage: toGeminiUsage(response.usageMetadata)
|
|
807
885
|
};
|
|
808
886
|
} catch (err) {
|
|
@@ -810,6 +888,80 @@ var GoogleDriver = class {
|
|
|
810
888
|
throw mapProviderError(err);
|
|
811
889
|
}
|
|
812
890
|
}
|
|
891
|
+
async sendMessage(images, prompt, _options) {
|
|
892
|
+
const client = await this.getClient();
|
|
893
|
+
return this.generate(client, [...this.toGeminiParts(images), prompt]);
|
|
894
|
+
}
|
|
895
|
+
/**
|
|
896
|
+
* Uploads a video through the Files API and waits until Gemini has finished
|
|
897
|
+
* processing it. Returns the ACTIVE file record.
|
|
898
|
+
*/
|
|
899
|
+
async uploadVideo(client, video) {
|
|
900
|
+
try {
|
|
901
|
+
const uploaded = await client.files.upload({
|
|
902
|
+
file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
|
|
903
|
+
config: { mimeType: video.mimeType }
|
|
904
|
+
});
|
|
905
|
+
const name = uploaded.name;
|
|
906
|
+
if (!name) {
|
|
907
|
+
throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
|
|
908
|
+
}
|
|
909
|
+
const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
|
|
910
|
+
let current = uploaded;
|
|
911
|
+
while (current.state === "PROCESSING") {
|
|
912
|
+
if (Date.now() > deadline) {
|
|
913
|
+
throw new VisualAIProviderError(
|
|
914
|
+
`Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
|
|
915
|
+
);
|
|
916
|
+
}
|
|
917
|
+
await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
|
|
918
|
+
current = await client.files.get({ name });
|
|
919
|
+
}
|
|
920
|
+
if (current.state !== "ACTIVE") {
|
|
921
|
+
throw new VisualAIProviderError(
|
|
922
|
+
`Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
|
|
923
|
+
);
|
|
924
|
+
}
|
|
925
|
+
if (!current.uri) {
|
|
926
|
+
throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
|
|
927
|
+
}
|
|
928
|
+
return current;
|
|
929
|
+
} catch (err) {
|
|
930
|
+
if (err instanceof VisualAIProviderError) throw err;
|
|
931
|
+
throw mapProviderError(err);
|
|
932
|
+
}
|
|
933
|
+
}
|
|
934
|
+
/**
|
|
935
|
+
* Sends the video bytes themselves. Gemini samples the clip server-side at
|
|
936
|
+
* `video.fps` and, unlike sampled frames, also hears the audio track. Small
|
|
937
|
+
* videos go inline; larger ones are uploaded via the Files API and deleted
|
|
938
|
+
* again afterwards (they would expire on their own after 48 h).
|
|
939
|
+
*/
|
|
940
|
+
async sendVideoMessage(video, prompt, _options) {
|
|
941
|
+
const client = await this.getClient();
|
|
942
|
+
const videoMetadata = { fps: video.fps };
|
|
943
|
+
if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
|
|
944
|
+
const part = {
|
|
945
|
+
inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
|
|
946
|
+
videoMetadata
|
|
947
|
+
};
|
|
948
|
+
const response = await this.generate(client, [part, prompt]);
|
|
949
|
+
return { ...response, delivery: "inline" };
|
|
950
|
+
}
|
|
951
|
+
const file = await this.uploadVideo(client, video);
|
|
952
|
+
try {
|
|
953
|
+
const part = {
|
|
954
|
+
fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
|
|
955
|
+
videoMetadata
|
|
956
|
+
};
|
|
957
|
+
const response = await this.generate(client, [part, prompt]);
|
|
958
|
+
return { ...response, delivery: "file" };
|
|
959
|
+
} finally {
|
|
960
|
+
if (file.name) {
|
|
961
|
+
await client.files.delete({ name: file.name }).catch(() => void 0);
|
|
962
|
+
}
|
|
963
|
+
}
|
|
964
|
+
}
|
|
813
965
|
async generateImage(images, prompt, options) {
|
|
814
966
|
const client = await this.getClient();
|
|
815
967
|
const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
|
|
@@ -942,6 +1094,9 @@ var OpenAIDriver = class {
|
|
|
942
1094
|
// src/providers/openrouter.ts
|
|
943
1095
|
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
944
1096
|
var OPENROUTER_REASONING_EFFORT = {
|
|
1097
|
+
// OpenRouter normalizes upstream vendors to low/medium/high only, so
|
|
1098
|
+
// "minimal" has no native equivalent and clamps to the floor.
|
|
1099
|
+
minimal: "low",
|
|
945
1100
|
low: "low",
|
|
946
1101
|
medium: "medium",
|
|
947
1102
|
high: "high",
|
|
@@ -1101,10 +1256,20 @@ function parseBooleanEnv(envName, value) {
|
|
|
1101
1256
|
`Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
|
|
1102
1257
|
);
|
|
1103
1258
|
}
|
|
1259
|
+
function parseReasoningEffortEnv(envName, value) {
|
|
1260
|
+
if (value === void 0 || value === "") return void 0;
|
|
1261
|
+
const levels = Object.values(ReasoningEffort);
|
|
1262
|
+
const lower = value.toLowerCase();
|
|
1263
|
+
if (levels.includes(lower)) return lower;
|
|
1264
|
+
throw new VisualAIConfigError(
|
|
1265
|
+
`Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
|
|
1266
|
+
);
|
|
1267
|
+
}
|
|
1104
1268
|
var debugDeprecationWarned = false;
|
|
1105
1269
|
function resolveConfig(config) {
|
|
1106
1270
|
const provider = resolveProvider(config);
|
|
1107
1271
|
const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
|
|
1272
|
+
const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
|
|
1108
1273
|
const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
|
|
1109
1274
|
const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
|
|
1110
1275
|
const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
|
|
@@ -1122,12 +1287,12 @@ function resolveConfig(config) {
|
|
|
1122
1287
|
}
|
|
1123
1288
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1124
1289
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1125
|
-
const effortNeedsLargeBudget =
|
|
1290
|
+
const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
|
|
1126
1291
|
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1127
1292
|
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1128
1293
|
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1129
1294
|
if (debug) {
|
|
1130
|
-
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${
|
|
1295
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
|
|
1131
1296
|
process.stderr.write(
|
|
1132
1297
|
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1133
1298
|
`
|
|
@@ -1139,7 +1304,7 @@ function resolveConfig(config) {
|
|
|
1139
1304
|
apiKey: config.apiKey,
|
|
1140
1305
|
model,
|
|
1141
1306
|
maxTokens,
|
|
1142
|
-
reasoningEffort
|
|
1307
|
+
reasoningEffort,
|
|
1143
1308
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1144
1309
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1145
1310
|
timeout: config.timeout,
|
|
@@ -1162,6 +1327,10 @@ var PRICING_TABLE = {
|
|
|
1162
1327
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1163
1328
|
outputPricePerToken: 50 / PER_MILLION
|
|
1164
1329
|
},
|
|
1330
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
|
|
1331
|
+
inputPricePerToken: 4 / PER_MILLION,
|
|
1332
|
+
outputPricePerToken: 20 / PER_MILLION
|
|
1333
|
+
},
|
|
1165
1334
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1166
1335
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1167
1336
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1170,6 +1339,10 @@ var PRICING_TABLE = {
|
|
|
1170
1339
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1171
1340
|
outputPricePerToken: 25 / PER_MILLION
|
|
1172
1341
|
},
|
|
1342
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
|
|
1343
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1344
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1345
|
+
},
|
|
1173
1346
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
|
|
1174
1347
|
inputPricePerToken: 3 / PER_MILLION,
|
|
1175
1348
|
outputPricePerToken: 15 / PER_MILLION
|
|
@@ -1196,6 +1369,22 @@ var PRICING_TABLE = {
|
|
|
1196
1369
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1197
1370
|
outputPricePerToken: 50 / PER_MILLION
|
|
1198
1371
|
},
|
|
1372
|
+
// Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
|
|
1373
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
|
|
1374
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1375
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1376
|
+
},
|
|
1377
|
+
// Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
|
|
1378
|
+
// bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
|
|
1379
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
|
|
1380
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1381
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1382
|
+
},
|
|
1383
|
+
// Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
|
|
1384
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
|
|
1385
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1386
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1387
|
+
},
|
|
1199
1388
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1200
1389
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1201
1390
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1292,6 +1481,10 @@ var PRICING_TABLE = {
|
|
|
1292
1481
|
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1293
1482
|
outputPricePerToken: 0.2 / PER_MILLION
|
|
1294
1483
|
},
|
|
1484
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
|
|
1485
|
+
inputPricePerToken: 1.6 / PER_MILLION,
|
|
1486
|
+
outputPricePerToken: 4.8 / PER_MILLION
|
|
1487
|
+
},
|
|
1295
1488
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1296
1489
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1297
1490
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1325,6 +1518,13 @@ var PRICING_TABLE = {
|
|
|
1325
1518
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1326
1519
|
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1327
1520
|
outputPricePerToken: 0.5 / PER_MILLION
|
|
1521
|
+
},
|
|
1522
|
+
// Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
|
|
1523
|
+
// upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
|
|
1524
|
+
// is $0.0036/MTok, not modelled (no provider gets a cache discount here).
|
|
1525
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
|
|
1526
|
+
inputPricePerToken: 0.435 / PER_MILLION,
|
|
1527
|
+
outputPricePerToken: 0.87 / PER_MILLION
|
|
1328
1528
|
}
|
|
1329
1529
|
};
|
|
1330
1530
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -1402,6 +1602,15 @@ async function timedSendMessage(driver, images, prompt, options) {
|
|
|
1402
1602
|
const durationSeconds = (performance.now() - start) / 1e3;
|
|
1403
1603
|
return { ...response, durationSeconds };
|
|
1404
1604
|
}
|
|
1605
|
+
async function timedSendVideoMessage(driver, video, prompt, options) {
|
|
1606
|
+
if (!driver.sendVideoMessage) {
|
|
1607
|
+
throw new VisualAIError("Provider driver does not support native video delivery");
|
|
1608
|
+
}
|
|
1609
|
+
const start = performance.now();
|
|
1610
|
+
const response = await driver.sendVideoMessage(video, prompt, options);
|
|
1611
|
+
const durationSeconds = (performance.now() - start) / 1e3;
|
|
1612
|
+
return { ...response, durationSeconds };
|
|
1613
|
+
}
|
|
1405
1614
|
|
|
1406
1615
|
// src/core/diff.ts
|
|
1407
1616
|
import sharp from "sharp";
|
|
@@ -1641,6 +1850,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
|
|
|
1641
1850
|
};
|
|
1642
1851
|
}
|
|
1643
1852
|
|
|
1853
|
+
// src/core/media.ts
|
|
1854
|
+
import { readFile as readFile3 } from "fs/promises";
|
|
1855
|
+
|
|
1644
1856
|
// src/core/debug-frames.ts
|
|
1645
1857
|
import { randomBytes } from "crypto";
|
|
1646
1858
|
import { mkdir, writeFile } from "fs/promises";
|
|
@@ -1696,6 +1908,92 @@ async function saveDebugFrames(frames, env = process.env) {
|
|
|
1696
1908
|
return runDir;
|
|
1697
1909
|
}
|
|
1698
1910
|
|
|
1911
|
+
// src/core/frame-dedupe.ts
|
|
1912
|
+
import sharp3 from "sharp";
|
|
1913
|
+
var DEDUPE_THUMBNAIL_EDGE = 256;
|
|
1914
|
+
var DEDUPE_PIXEL_TOLERANCE = 24;
|
|
1915
|
+
var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
|
|
1916
|
+
function resolveDedupeOptions(raw) {
|
|
1917
|
+
if (raw === void 0 || raw === true) {
|
|
1918
|
+
return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
|
|
1919
|
+
}
|
|
1920
|
+
if (raw === false) {
|
|
1921
|
+
return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
|
|
1922
|
+
}
|
|
1923
|
+
const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
|
|
1924
|
+
if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
|
|
1925
|
+
throw new VisualAIVideoError(
|
|
1926
|
+
`Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
|
|
1927
|
+
);
|
|
1928
|
+
}
|
|
1929
|
+
return { enabled: true, threshold };
|
|
1930
|
+
}
|
|
1931
|
+
async function frameSignature(frame) {
|
|
1932
|
+
try {
|
|
1933
|
+
const { data, info } = await sharp3(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
|
|
1934
|
+
fit: "inside",
|
|
1935
|
+
withoutEnlargement: true
|
|
1936
|
+
}).raw().toBuffer({ resolveWithObject: true });
|
|
1937
|
+
return { width: info.width, height: info.height, pixels: data };
|
|
1938
|
+
} catch (err) {
|
|
1939
|
+
const reason = err instanceof Error ? err.message : String(err);
|
|
1940
|
+
throw new VisualAIVideoError(
|
|
1941
|
+
`Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
|
|
1942
|
+
);
|
|
1943
|
+
}
|
|
1944
|
+
}
|
|
1945
|
+
function changedFraction(a, b) {
|
|
1946
|
+
if (a.width !== b.width || a.height !== b.height) {
|
|
1947
|
+
return 1;
|
|
1948
|
+
}
|
|
1949
|
+
let changed = 0;
|
|
1950
|
+
for (let i = 0; i < a.pixels.length; i++) {
|
|
1951
|
+
if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
|
|
1952
|
+
changed++;
|
|
1953
|
+
}
|
|
1954
|
+
}
|
|
1955
|
+
return changed / a.pixels.length;
|
|
1956
|
+
}
|
|
1957
|
+
function reindex(frame, index) {
|
|
1958
|
+
if (frame.index === index) {
|
|
1959
|
+
return frame;
|
|
1960
|
+
}
|
|
1961
|
+
return {
|
|
1962
|
+
data: frame.data,
|
|
1963
|
+
mimeType: frame.mimeType,
|
|
1964
|
+
get base64() {
|
|
1965
|
+
return frame.base64;
|
|
1966
|
+
},
|
|
1967
|
+
timestampSeconds: frame.timestampSeconds,
|
|
1968
|
+
index
|
|
1969
|
+
};
|
|
1970
|
+
}
|
|
1971
|
+
async function dedupeFrames(frames, options) {
|
|
1972
|
+
const { enabled, threshold } = resolveDedupeOptions(options);
|
|
1973
|
+
if (!enabled || frames.length < 2) {
|
|
1974
|
+
return { frames: [...frames], dropped: 0 };
|
|
1975
|
+
}
|
|
1976
|
+
const signed = await Promise.all(
|
|
1977
|
+
frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
|
|
1978
|
+
);
|
|
1979
|
+
const [first, ...rest] = signed;
|
|
1980
|
+
if (first === void 0) {
|
|
1981
|
+
return { frames: [], dropped: 0 };
|
|
1982
|
+
}
|
|
1983
|
+
const kept = [first.frame];
|
|
1984
|
+
let lastKept = first.signature;
|
|
1985
|
+
for (const { frame, signature } of rest) {
|
|
1986
|
+
if (changedFraction(lastKept, signature) >= threshold) {
|
|
1987
|
+
kept.push(frame);
|
|
1988
|
+
lastKept = signature;
|
|
1989
|
+
}
|
|
1990
|
+
}
|
|
1991
|
+
return {
|
|
1992
|
+
frames: kept.map((frame, index) => reindex(frame, index)),
|
|
1993
|
+
dropped: frames.length - kept.length
|
|
1994
|
+
};
|
|
1995
|
+
}
|
|
1996
|
+
|
|
1699
1997
|
// src/core/video.ts
|
|
1700
1998
|
import { mkdtemp, readFile as readFile2, readdir, rm, writeFile as writeFile2 } from "fs/promises";
|
|
1701
1999
|
import { tmpdir } from "os";
|
|
@@ -1931,7 +2229,7 @@ async function probeDurationSeconds(videoPath) {
|
|
|
1931
2229
|
});
|
|
1932
2230
|
});
|
|
1933
2231
|
}
|
|
1934
|
-
|
|
2232
|
+
function resolveVideoSamplingOptions(options = {}) {
|
|
1935
2233
|
const fps = options.fps ?? DEFAULT_FPS;
|
|
1936
2234
|
const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
|
|
1937
2235
|
const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
|
|
@@ -1951,13 +2249,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
|
|
|
1951
2249
|
`Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
|
|
1952
2250
|
);
|
|
1953
2251
|
}
|
|
1954
|
-
|
|
1955
|
-
|
|
2252
|
+
return { fps, maxFrames, maxDurationSeconds };
|
|
2253
|
+
}
|
|
2254
|
+
function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
|
|
1956
2255
|
if (durationSeconds > maxDurationSeconds) {
|
|
1957
2256
|
throw new VisualAIVideoError(
|
|
1958
2257
|
`Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
|
|
1959
2258
|
);
|
|
1960
2259
|
}
|
|
2260
|
+
}
|
|
2261
|
+
async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
|
|
2262
|
+
const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
|
|
2263
|
+
const ffmpeg = await loadFfmpegFactory();
|
|
2264
|
+
const durationSeconds = await probeDurationSeconds(videoPath);
|
|
2265
|
+
assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
|
|
1961
2266
|
const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
|
|
1962
2267
|
try {
|
|
1963
2268
|
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
|
|
@@ -2061,6 +2366,7 @@ function isTimestampedFrameInput(frame) {
|
|
|
2061
2366
|
async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
2062
2367
|
const rawFrames = input.frames;
|
|
2063
2368
|
const fps = input.fps ?? DEFAULT_FPS;
|
|
2369
|
+
resolveDedupeOptions(input.dedupe);
|
|
2064
2370
|
if (rawFrames.length === 0) {
|
|
2065
2371
|
throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
|
|
2066
2372
|
}
|
|
@@ -2072,7 +2378,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
|
|
|
2072
2378
|
if (!Number.isFinite(fps) || fps <= 0) {
|
|
2073
2379
|
throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
|
|
2074
2380
|
}
|
|
2075
|
-
const
|
|
2381
|
+
const sampled = await Promise.all(
|
|
2076
2382
|
rawFrames.map(async (raw, index) => {
|
|
2077
2383
|
const timestamped = isTimestampedFrameInput(raw);
|
|
2078
2384
|
const imageInput = timestamped ? raw.image : raw;
|
|
@@ -2095,20 +2401,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
|
|
|
2095
2401
|
};
|
|
2096
2402
|
})
|
|
2097
2403
|
);
|
|
2098
|
-
const durationSeconds =
|
|
2404
|
+
const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
|
|
2405
|
+
const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
|
|
2099
2406
|
await saveDebugFrames(frames);
|
|
2100
|
-
return { kind: "video", frames, durationSeconds };
|
|
2407
|
+
return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
|
|
2408
|
+
}
|
|
2409
|
+
async function normalizeNativeVideo(input, videoOptions) {
|
|
2410
|
+
const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
|
|
2411
|
+
const { path, mimeType, cleanup } = await resolveVideoToPath(input);
|
|
2412
|
+
try {
|
|
2413
|
+
const durationSeconds = await probeDurationSeconds(path);
|
|
2414
|
+
assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
|
|
2415
|
+
const data = await readFile3(path);
|
|
2416
|
+
return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
|
|
2417
|
+
} finally {
|
|
2418
|
+
try {
|
|
2419
|
+
await cleanup();
|
|
2420
|
+
} catch {
|
|
2421
|
+
}
|
|
2422
|
+
}
|
|
2101
2423
|
}
|
|
2102
|
-
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
2424
|
+
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
|
|
2103
2425
|
if (isFramesInput(input)) {
|
|
2104
2426
|
return normalizeFrames(input, maxDimension);
|
|
2105
2427
|
}
|
|
2106
2428
|
if (isVideoInput(input)) {
|
|
2429
|
+
if (nativeVideo) {
|
|
2430
|
+
return normalizeNativeVideo(input, videoOptions);
|
|
2431
|
+
}
|
|
2432
|
+
resolveDedupeOptions(videoOptions?.dedupe);
|
|
2107
2433
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
2108
2434
|
try {
|
|
2109
|
-
const { frames, durationSeconds } = await extractFrames(
|
|
2435
|
+
const { frames: sampled, durationSeconds } = await extractFrames(
|
|
2436
|
+
path,
|
|
2437
|
+
videoOptions,
|
|
2438
|
+
maxDimension
|
|
2439
|
+
);
|
|
2440
|
+
const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
|
|
2110
2441
|
await saveDebugFrames(frames);
|
|
2111
|
-
return { kind: "video", frames, durationSeconds };
|
|
2442
|
+
return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
|
|
2112
2443
|
} finally {
|
|
2113
2444
|
try {
|
|
2114
2445
|
await cleanup();
|
|
@@ -2198,6 +2529,12 @@ var AskResultSchema = z.object({
|
|
|
2198
2529
|
* omitting the key, even for image inputs that were never asked to populate it.
|
|
2199
2530
|
*/
|
|
2200
2531
|
frameReferences: z.array(z.number().int().nonnegative()).nullable().optional(),
|
|
2532
|
+
/**
|
|
2533
|
+
* For natively delivered video, the timestamps (seconds from the start of
|
|
2534
|
+
* the clip) the model relied on to answer. The native counterpart of
|
|
2535
|
+
* `frameReferences`. Nullable for the same strict-schema reason.
|
|
2536
|
+
*/
|
|
2537
|
+
timestampReferences: z.array(z.number().nonnegative()).nullable().optional(),
|
|
2201
2538
|
usage: UsageInfoSchema.optional()
|
|
2202
2539
|
});
|
|
2203
2540
|
|
|
@@ -2209,6 +2546,12 @@ function stripCodeFences(text) {
|
|
|
2209
2546
|
}
|
|
2210
2547
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2211
2548
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2549
|
+
var AskImageResponseSchema = AskResponseSchema.omit({
|
|
2550
|
+
frameReferences: true,
|
|
2551
|
+
timestampReferences: true
|
|
2552
|
+
});
|
|
2553
|
+
var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
|
|
2554
|
+
var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
|
|
2212
2555
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2213
2556
|
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2214
2557
|
function parseJson(text) {
|
|
@@ -2276,7 +2619,8 @@ function parseAskResponse(raw) {
|
|
|
2276
2619
|
const result = parseResponse(raw, AskResponseSchema);
|
|
2277
2620
|
return {
|
|
2278
2621
|
...result,
|
|
2279
|
-
frameReferences: result.frameReferences ?? void 0
|
|
2622
|
+
frameReferences: result.frameReferences ?? void 0,
|
|
2623
|
+
timestampReferences: result.timestampReferences ?? void 0
|
|
2280
2624
|
};
|
|
2281
2625
|
}
|
|
2282
2626
|
function parseCompareResponse(raw) {
|
|
@@ -2300,31 +2644,75 @@ function createDriver(provider, config) {
|
|
|
2300
2644
|
return PROVIDER_REGISTRY[provider](config);
|
|
2301
2645
|
}
|
|
2302
2646
|
var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
|
|
2303
|
-
var
|
|
2647
|
+
var askSchemaOptionsByMedia = {
|
|
2648
|
+
image: toSchemaOptions(AskImageResponseSchema),
|
|
2649
|
+
video: toSchemaOptions(AskFramesResponseSchema),
|
|
2650
|
+
"native-video": toSchemaOptions(AskNativeVideoResponseSchema)
|
|
2651
|
+
};
|
|
2304
2652
|
var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
|
|
2305
2653
|
function mediaToProviderInputs(media) {
|
|
2306
2654
|
if (media.kind === "image") {
|
|
2307
2655
|
return {
|
|
2656
|
+
kind: "images",
|
|
2308
2657
|
images: [media.image],
|
|
2309
2658
|
mediaContext: { kind: "image" },
|
|
2310
|
-
|
|
2659
|
+
frames: void 0
|
|
2660
|
+
};
|
|
2661
|
+
}
|
|
2662
|
+
if (media.kind === "native-video") {
|
|
2663
|
+
return {
|
|
2664
|
+
kind: "native-video",
|
|
2665
|
+
video: media.video,
|
|
2666
|
+
mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
|
|
2311
2667
|
};
|
|
2312
2668
|
}
|
|
2313
2669
|
const timestamps = media.frames.map((f) => f.timestampSeconds);
|
|
2314
2670
|
return {
|
|
2671
|
+
kind: "images",
|
|
2315
2672
|
images: media.frames,
|
|
2316
2673
|
mediaContext: {
|
|
2317
2674
|
kind: "video",
|
|
2318
2675
|
frameTimestamps: timestamps,
|
|
2319
|
-
durationSeconds: media.durationSeconds
|
|
2676
|
+
durationSeconds: media.durationSeconds,
|
|
2677
|
+
droppedUnchanged: media.droppedUnchanged
|
|
2320
2678
|
},
|
|
2321
|
-
|
|
2679
|
+
frames: {
|
|
2322
2680
|
count: media.frames.length,
|
|
2323
2681
|
timestampsSeconds: timestamps,
|
|
2324
|
-
durationSeconds: media.durationSeconds
|
|
2682
|
+
durationSeconds: media.durationSeconds,
|
|
2683
|
+
droppedUnchanged: media.droppedUnchanged
|
|
2325
2684
|
}
|
|
2326
2685
|
};
|
|
2327
2686
|
}
|
|
2687
|
+
async function sendMedia(driver, dispatch, prompt, options) {
|
|
2688
|
+
if (dispatch.kind === "native-video") {
|
|
2689
|
+
const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
|
|
2690
|
+
const { durationSeconds, fps, mimeType } = dispatch.video;
|
|
2691
|
+
return {
|
|
2692
|
+
response: response2,
|
|
2693
|
+
metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
|
|
2694
|
+
};
|
|
2695
|
+
}
|
|
2696
|
+
const response = await timedSendMessage(driver, dispatch.images, prompt, options);
|
|
2697
|
+
return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
|
|
2698
|
+
}
|
|
2699
|
+
function resolveNativeVideo(input, videoOptions, driver, provider) {
|
|
2700
|
+
const mode = videoOptions?.mode ?? "auto";
|
|
2701
|
+
if (mode !== "auto" && mode !== "native" && mode !== "frames") {
|
|
2702
|
+
throw new VisualAIConfigError(
|
|
2703
|
+
`Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
|
|
2704
|
+
);
|
|
2705
|
+
}
|
|
2706
|
+
const supported = typeof driver.sendVideoMessage === "function";
|
|
2707
|
+
if (mode === "frames") return false;
|
|
2708
|
+
if (mode === "auto") return supported;
|
|
2709
|
+
if (!supported && !isFramesInput(input) && isVideoInput(input)) {
|
|
2710
|
+
throw new VisualAIConfigError(
|
|
2711
|
+
`Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
|
|
2712
|
+
);
|
|
2713
|
+
}
|
|
2714
|
+
return true;
|
|
2715
|
+
}
|
|
2328
2716
|
function visualAI(config = {}) {
|
|
2329
2717
|
const resolvedConfig = resolveConfig(config);
|
|
2330
2718
|
const driverConfig = {
|
|
@@ -2362,38 +2750,60 @@ function visualAI(config = {}) {
|
|
|
2362
2750
|
throw new VisualAIConfigError("At least one statement is required for check()");
|
|
2363
2751
|
}
|
|
2364
2752
|
return withErrorDebug(resolvedConfig, "check", async () => {
|
|
2365
|
-
const
|
|
2366
|
-
|
|
2753
|
+
const nativeVideo = resolveNativeVideo(
|
|
2754
|
+
input,
|
|
2755
|
+
options?.video,
|
|
2756
|
+
driver,
|
|
2757
|
+
resolvedConfig.provider
|
|
2758
|
+
);
|
|
2759
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
|
|
2760
|
+
const dispatch = mediaToProviderInputs(media);
|
|
2367
2761
|
const prompt = buildCheckPrompt(stmts, {
|
|
2368
2762
|
instructions: options?.instructions,
|
|
2369
|
-
media: mediaContext
|
|
2763
|
+
media: dispatch.mediaContext
|
|
2370
2764
|
});
|
|
2371
2765
|
debugLog(resolvedConfig, "check prompt", prompt, "prompt");
|
|
2372
|
-
const response = await
|
|
2766
|
+
const { response, metadata } = await sendMedia(
|
|
2767
|
+
driver,
|
|
2768
|
+
dispatch,
|
|
2769
|
+
prompt,
|
|
2770
|
+
checkSchemaOptions
|
|
2771
|
+
);
|
|
2373
2772
|
debugLog(resolvedConfig, "check response", response.text, "response");
|
|
2374
2773
|
const result = parseCheckResponse(response.text);
|
|
2375
2774
|
return {
|
|
2376
2775
|
...result,
|
|
2377
|
-
...
|
|
2776
|
+
...metadata,
|
|
2378
2777
|
usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
|
|
2379
2778
|
};
|
|
2380
2779
|
});
|
|
2381
2780
|
},
|
|
2382
2781
|
async ask(input, userPrompt, options) {
|
|
2383
2782
|
return withErrorDebug(resolvedConfig, "ask", async () => {
|
|
2384
|
-
const
|
|
2385
|
-
|
|
2783
|
+
const nativeVideo = resolveNativeVideo(
|
|
2784
|
+
input,
|
|
2785
|
+
options?.video,
|
|
2786
|
+
driver,
|
|
2787
|
+
resolvedConfig.provider
|
|
2788
|
+
);
|
|
2789
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
|
|
2790
|
+
const dispatch = mediaToProviderInputs(media);
|
|
2386
2791
|
const prompt = buildAskPrompt(userPrompt, {
|
|
2387
2792
|
instructions: options?.instructions,
|
|
2388
|
-
media: mediaContext
|
|
2793
|
+
media: dispatch.mediaContext
|
|
2389
2794
|
});
|
|
2390
2795
|
debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
|
|
2391
|
-
const response = await
|
|
2796
|
+
const { response, metadata } = await sendMedia(
|
|
2797
|
+
driver,
|
|
2798
|
+
dispatch,
|
|
2799
|
+
prompt,
|
|
2800
|
+
askSchemaOptionsByMedia[dispatch.mediaContext.kind]
|
|
2801
|
+
);
|
|
2392
2802
|
debugLog(resolvedConfig, "ask response", response.text, "response");
|
|
2393
2803
|
const result = parseAskResponse(response.text);
|
|
2394
2804
|
return {
|
|
2395
2805
|
...result,
|
|
2396
|
-
...
|
|
2806
|
+
...metadata,
|
|
2397
2807
|
usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
|
|
2398
2808
|
};
|
|
2399
2809
|
});
|
|
@@ -2411,7 +2821,7 @@ function visualAI(config = {}) {
|
|
|
2411
2821
|
debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
|
|
2412
2822
|
const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
|
|
2413
2823
|
debugLog(resolvedConfig, "compare response", response.text, "response");
|
|
2414
|
-
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model
|
|
2824
|
+
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
|
|
2415
2825
|
const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
|
|
2416
2826
|
let diffImage;
|
|
2417
2827
|
if (effectiveDiffImage) {
|