visual-ai-assertions 0.23.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,132 +1,3 @@
1
- // src/constants.ts
2
- var ReasoningEffort = {
3
- LOW: "low",
4
- MEDIUM: "medium",
5
- HIGH: "high",
6
- XHIGH: "xhigh"
7
- };
8
- var ImageDetail = {
9
- AUTO: "auto",
10
- LOW: "low",
11
- HIGH: "high"
12
- };
13
- var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
14
- var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
15
- var Provider = {
16
- ANTHROPIC: "anthropic",
17
- OPENAI: "openai",
18
- GOOGLE: "google",
19
- OPENROUTER: "openrouter"
20
- };
21
- var Model = {
22
- Anthropic: {
23
- FABLE_5_1: "claude-fable-5-1",
24
- FABLE_5: "claude-fable-5",
25
- OPUS_5: "claude-opus-5",
26
- OPUS_4_8: "claude-opus-4-8",
27
- OPUS_4_7: "claude-opus-4-7",
28
- OPUS_4_6: "claude-opus-4-6",
29
- SONNET_5: "claude-sonnet-5",
30
- SONNET_4_6: "claude-sonnet-4-6",
31
- HAIKU_4_5: "claude-haiku-4-5"
32
- },
33
- OpenAI: {
34
- GPT_6_ASTRA: "gpt-6-astra",
35
- GPT_5_6_SOL: "gpt-5.6-sol",
36
- GPT_5_6_TERRA: "gpt-5.6-terra",
37
- GPT_5_6_LUNA: "gpt-5.6-luna",
38
- GPT_5_5: "gpt-5.5",
39
- GPT_5_4: "gpt-5.4",
40
- GPT_5_4_PRO: "gpt-5.4-pro",
41
- GPT_5_4_MINI: "gpt-5.4-mini",
42
- GPT_5_4_NANO: "gpt-5.4-nano",
43
- GPT_5_2: "gpt-5.2",
44
- GPT_5_MINI: "gpt-5-mini"
45
- },
46
- Google: {
47
- GEMINI_3_8_FLASH: "gemini-3.8-flash",
48
- GEMINI_3_7_FLASH: "gemini-3.7-flash",
49
- GEMINI_3_6_FLASH: "gemini-3.6-flash",
50
- GEMINI_3_5_FLASH: "gemini-3.5-flash",
51
- GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
52
- GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
53
- GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
54
- GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
55
- },
56
- /**
57
- * Models routed through OpenRouter (https://openrouter.ai). Slugs always
58
- * carry a vendor prefix (`vendor/model`), which is how provider inference
59
- * recognizes them. All listed models accept image input.
60
- */
61
- OpenRouter: {
62
- MUSE_SPARK_1_3: "meta/muse-spark-1.3",
63
- GROK_4_6: "x-ai/grok-4.6",
64
- GROK_4_5: "x-ai/grok-4.5",
65
- KIMI_K3: "moonshotai/kimi-k3",
66
- KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
67
- QWEN_3_8_MAX: "qwen/qwen3.8-max",
68
- QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
69
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
70
- GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
71
- }
72
- };
73
- var DEFAULT_MODELS = {
74
- [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
75
- [Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
76
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
77
- [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
78
- };
79
- var DEFAULT_MAX_TOKENS = 4096;
80
- var OPENAI_REASONING_MAX_TOKENS = 16384;
81
- var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
82
- var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
83
- Model.OpenAI.GPT_6_ASTRA
84
- ]);
85
- var MODEL_TO_PROVIDER = new Map([
86
- ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
87
- ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
88
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
89
- ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
90
- ]);
91
- var VALID_PROVIDERS = Object.values(Provider);
92
- var PROVIDER_DEFAULT_REASONING = {
93
- openai: "medium",
94
- anthropic: "off",
95
- google: "off",
96
- // Varies by upstream model; the driver sends no reasoning field unless configured.
97
- openrouter: "off"
98
- };
99
- var Content = {
100
- /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
101
- PLACEHOLDER_TEXT: "placeholder-text",
102
- /** Detects error messages, banners, stack traces, or error codes */
103
- ERROR_MESSAGES: "error-messages",
104
- /** Detects broken image icons or failed-to-load image indicators */
105
- BROKEN_IMAGES: "broken-images",
106
- /** Detects UI elements that unintentionally overlap and obscure content */
107
- OVERLAPPING_ELEMENTS: "overlapping-elements"
108
- };
109
- var Layout = {
110
- /** Detects elements that unintentionally overlap each other */
111
- OVERLAP: "overlap",
112
- /** Detects content cut off or extending beyond container boundaries */
113
- OVERFLOW: "overflow",
114
- /** Detects inconsistent alignment of text, images, and UI components */
115
- ALIGNMENT: "alignment"
116
- };
117
- var Accessibility = {
118
- /** Detects insufficient color contrast between text and backgrounds */
119
- CONTRAST: "contrast",
120
- /** Detects text that is cut off, overlapping, too small, or obscured */
121
- READABILITY: "readability",
122
- /** Detects interactive elements that are not visually distinct */
123
- INTERACTIVE_VISIBILITY: "interactive-visibility",
124
- /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
125
- COLOR_BLINDNESS: "color-blindness",
126
- /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
127
- COLOR_ALONE: "color-alone"
128
- };
129
-
130
1
  // src/errors.ts
131
2
  var VisualAIError = class extends Error {
132
3
  code;
@@ -258,10 +129,15 @@ Example for a failing check:
258
129
  ]
259
130
  }
260
131
  ${JSON_INSTRUCTIONS}`;
261
- var CHECK_OUTPUT_SCHEMA_VIDEO = `IMPORTANT: Follow this evaluation order:
132
+ function buildCheckOutputSchemaVideo(unit) {
133
+ const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
134
+ const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
135
+ const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
136
+ const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
137
+ return `IMPORTANT: Follow this evaluation order:
262
138
  1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
263
- 2. A statement passes if it is true at ANY frame of the timeline, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
264
- 3. For each statement that passes, set "timestampSeconds" to the timestamp of the frame that most clearly demonstrates it (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
139
+ 2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
140
+ 3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
265
141
  4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
266
142
  5. Write "reasoning" as a brief overall summary of the evaluation
267
143
  6. Include "issues" only for statements that failed
@@ -275,7 +151,7 @@ Respond with a JSON object matching this exact structure:
275
151
  {
276
152
  "statement": string, // the original statement text
277
153
  "pass": boolean, // whether this statement is true at any point in the timeline
278
- "reasoning": string, // explanation for this statement, citing frame timestamps where relevant
154
+ "reasoning": string, // explanation for this statement, ${citing} where relevant
279
155
  "confidence": "high" | "medium" | "low",
280
156
  "timestampSeconds": number | null
281
157
  // seconds from the start of the clip where the statement is most clearly true,
@@ -293,10 +169,13 @@ Example for a passing video check:
293
169
  "reasoning": "The success toast appeared briefly around 3.5s.",
294
170
  "issues": [],
295
171
  "statements": [
296
- { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right at the 3.5s frame", "confidence": "high", "timestampSeconds": 3.5 }
172
+ { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
297
173
  ]
298
174
  }
299
175
  ${JSON_INSTRUCTIONS}`;
176
+ }
177
+ var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
178
+ var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
300
179
  var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
301
180
  {
302
181
  "summary": string, // high-level analysis summary
@@ -329,6 +208,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
329
208
  Prioritize issues by severity (critical / major / minor) as for image input.
330
209
  Cite frame indices in "frameReferences" so the user can locate the moments you describe.
331
210
  ${JSON_INSTRUCTIONS}`;
211
+ var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
212
+ {
213
+ "summary": string, // high-level summary of what happens across the video
214
+ "issues": [...], // list of issues/findings, can be empty
215
+ "timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
216
+ }
217
+ ${ISSUE_SCHEMA_INSTRUCTIONS}
218
+
219
+ Prioritize issues by severity (critical / major / minor) as for image input.
220
+ Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
221
+ ${JSON_INSTRUCTIONS}`;
332
222
  var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
333
223
  {
334
224
  "pass": boolean, // true if no critical or major changes found
@@ -350,15 +240,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
350
240
  var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
351
241
  var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
352
242
  var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
353
- function buildVideoTimelineSection(frameTimestamps, durationSeconds) {
243
+ var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
244
+ var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
245
+ function buildNativeVideoSection(durationSeconds) {
246
+ return `Video recording:
247
+ - Total duration: ${durationSeconds.toFixed(2)}s
248
+
249
+ The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
250
+ }
251
+ function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
354
252
  const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
253
+ const attached = frameTimestamps.length;
254
+ const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
255
+ const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
355
256
  return `Video timeline:
356
257
  - Total duration: ${durationSeconds.toFixed(2)}s
357
- - ${frameTimestamps.length} frames sampled (in chronological order)
258
+ ${sampledLine}
358
259
  - Frame index \u2192 timestamp:
359
260
  ${formatted}
360
261
 
361
- Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.`;
262
+ Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
362
263
  }
363
264
  var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
364
265
  var COMPARE_EDGE_RULES = [
@@ -373,30 +274,52 @@ function buildCheckPrompt(statements, options) {
373
274
  const stmts = Array.isArray(statements) ? statements : [statements];
374
275
  const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
375
276
  const media = options?.media;
376
- const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
277
+ const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
377
278
  const sections = [options?.role ?? defaultRole];
378
279
  if (media?.kind === "video") {
379
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
280
+ sections.push(
281
+ buildVideoTimelineSection(
282
+ media.frameTimestamps,
283
+ media.durationSeconds,
284
+ media.droppedUnchanged
285
+ )
286
+ );
287
+ } else if (media?.kind === "native-video") {
288
+ sections.push(buildNativeVideoSection(media.durationSeconds));
380
289
  }
381
290
  if (options?.instructions && options.instructions.length > 0) {
382
291
  sections.push(buildInstructionsSection(options.instructions));
383
292
  }
384
293
  sections.push(`Statements to evaluate:
385
294
  ${statementsBlock}`);
386
- sections.push(media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE);
295
+ sections.push(
296
+ media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
297
+ );
387
298
  return sections.join("\n\n");
388
299
  }
389
300
  function buildAskPrompt(userPrompt, options) {
390
301
  const media = options?.media;
391
- const sections = [media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : DEFAULT_ASK_ROLE];
302
+ const sections = [
303
+ media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
304
+ ];
392
305
  if (media?.kind === "video") {
393
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
306
+ sections.push(
307
+ buildVideoTimelineSection(
308
+ media.frameTimestamps,
309
+ media.durationSeconds,
310
+ media.droppedUnchanged
311
+ )
312
+ );
313
+ } else if (media?.kind === "native-video") {
314
+ sections.push(buildNativeVideoSection(media.durationSeconds));
394
315
  }
395
316
  if (options?.instructions && options.instructions.length > 0) {
396
317
  sections.push(buildInstructionsSection(options.instructions));
397
318
  }
398
319
  sections.push(`User request: ${userPrompt}`);
399
- sections.push(media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE);
320
+ sections.push(
321
+ media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
322
+ );
400
323
  return sections.join("\n\n");
401
324
  }
402
325
  function buildAiDiffPrompt() {
@@ -441,7 +364,7 @@ function visibleRole(finalState, requireCorrectRendering) {
441
364
  var ELEMENTS_VISIBLE_CLIPPING_RULES = [
442
365
  "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
443
366
  "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
444
- "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
367
+ "The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
445
368
  ];
446
369
  var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
447
370
  var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
@@ -472,6 +395,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
472
395
  return buildCheckPrompt(statements, { role, instructions });
473
396
  }
474
397
 
398
+ // src/constants.ts
399
+ var ReasoningEffort = {
400
+ MINIMAL: "minimal",
401
+ LOW: "low",
402
+ MEDIUM: "medium",
403
+ HIGH: "high",
404
+ XHIGH: "xhigh"
405
+ };
406
+ var ImageDetail = {
407
+ AUTO: "auto",
408
+ LOW: "low",
409
+ HIGH: "high"
410
+ };
411
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
412
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
413
+ var Provider = {
414
+ ANTHROPIC: "anthropic",
415
+ OPENAI: "openai",
416
+ GOOGLE: "google",
417
+ OPENROUTER: "openrouter"
418
+ };
419
+ var Model = {
420
+ Anthropic: {
421
+ FABLE_5_1: "claude-fable-5-1",
422
+ FABLE_5: "claude-fable-5",
423
+ OPUS_5_5: "claude-opus-5-5",
424
+ OPUS_5: "claude-opus-5",
425
+ OPUS_4_8: "claude-opus-4-8",
426
+ OPUS_4_7: "claude-opus-4-7",
427
+ OPUS_4_6: "claude-opus-4-6",
428
+ SONNET_5_5: "claude-sonnet-5-5",
429
+ SONNET_5: "claude-sonnet-5",
430
+ SONNET_4_6: "claude-sonnet-4-6",
431
+ HAIKU_4_5: "claude-haiku-4-5"
432
+ },
433
+ OpenAI: {
434
+ GPT_6_ASTRA: "gpt-6-astra",
435
+ GPT_6_1_SOL: "gpt-6.1-sol",
436
+ GPT_6_SOL: "gpt-6-sol",
437
+ GPT_6_LUNA: "gpt-6-luna",
438
+ GPT_5_6_SOL: "gpt-5.6-sol",
439
+ GPT_5_6_TERRA: "gpt-5.6-terra",
440
+ GPT_5_6_LUNA: "gpt-5.6-luna",
441
+ GPT_5_5: "gpt-5.5",
442
+ GPT_5_4: "gpt-5.4",
443
+ GPT_5_4_PRO: "gpt-5.4-pro",
444
+ GPT_5_4_MINI: "gpt-5.4-mini",
445
+ GPT_5_4_NANO: "gpt-5.4-nano",
446
+ GPT_5_2: "gpt-5.2",
447
+ GPT_5_MINI: "gpt-5-mini"
448
+ },
449
+ Google: {
450
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
451
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
452
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
453
+ GEMINI_3_5_FLASH: "gemini-3.5-flash",
454
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
455
+ GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
456
+ GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
457
+ GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
458
+ },
459
+ /**
460
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
461
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
462
+ * recognizes them. All listed models accept image input.
463
+ */
464
+ OpenRouter: {
465
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
466
+ GROK_4_7: "x-ai/grok-4.7",
467
+ GROK_4_6: "x-ai/grok-4.6",
468
+ GROK_4_5: "x-ai/grok-4.5",
469
+ KIMI_K3: "moonshotai/kimi-k3",
470
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
471
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
472
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
473
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
474
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
475
+ MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
476
+ }
477
+ };
478
+ var DEFAULT_MODELS = {
479
+ [Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
480
+ [Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
481
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
482
+ [Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
483
+ };
484
+ var DEFAULT_MAX_TOKENS = 4096;
485
+ var OPENAI_REASONING_MAX_TOKENS = 16384;
486
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
487
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
488
+ Model.OpenRouter.QWEN_3_8_MAX,
489
+ Model.OpenRouter.QWEN_3_7_PLUS
490
+ ]);
491
+ var MODEL_TO_PROVIDER = new Map([
492
+ ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
493
+ ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
494
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
495
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
496
+ ]);
497
+ var VALID_PROVIDERS = Object.values(Provider);
498
+ var PROVIDER_DEFAULT_REASONING = {
499
+ openai: "medium",
500
+ anthropic: "off",
501
+ google: "off",
502
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
503
+ openrouter: "off"
504
+ };
505
+ var Content = {
506
+ /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
507
+ PLACEHOLDER_TEXT: "placeholder-text",
508
+ /** Detects error messages, banners, stack traces, or error codes */
509
+ ERROR_MESSAGES: "error-messages",
510
+ /** Detects broken image icons or failed-to-load image indicators */
511
+ BROKEN_IMAGES: "broken-images",
512
+ /** Detects UI elements that unintentionally overlap and obscure content */
513
+ OVERLAPPING_ELEMENTS: "overlapping-elements"
514
+ };
515
+ var Layout = {
516
+ /** Detects elements that unintentionally overlap each other */
517
+ OVERLAP: "overlap",
518
+ /** Detects content cut off or extending beyond container boundaries */
519
+ OVERFLOW: "overflow",
520
+ /** Detects inconsistent alignment of text, images, and UI components */
521
+ ALIGNMENT: "alignment"
522
+ };
523
+ var Accessibility = {
524
+ /** Detects insufficient color contrast between text and backgrounds */
525
+ CONTRAST: "contrast",
526
+ /** Detects text that is cut off, overlapping, too small, or obscured */
527
+ READABILITY: "readability",
528
+ /** Detects interactive elements that are not visually distinct */
529
+ INTERACTIVE_VISIBILITY: "interactive-visibility",
530
+ /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
531
+ COLOR_BLINDNESS: "color-blindness",
532
+ /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
533
+ COLOR_ALONE: "color-alone"
534
+ };
535
+
475
536
  // src/templates/accessibility.ts
476
537
  var ALL_CHECKS = Object.values(Accessibility);
477
538
  var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
@@ -581,17 +642,22 @@ function parseRetryAfter(value) {
581
642
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
582
643
  Model.Anthropic.FABLE_5_1,
583
644
  Model.Anthropic.FABLE_5,
645
+ Model.Anthropic.OPUS_5_5,
584
646
  Model.Anthropic.OPUS_5,
585
647
  Model.Anthropic.OPUS_4_8,
586
648
  Model.Anthropic.OPUS_4_7,
649
+ Model.Anthropic.SONNET_5_5,
587
650
  Model.Anthropic.SONNET_5
588
651
  ]);
589
652
  function mapEffort(level, model) {
653
+ if (level === "minimal") return "low";
590
654
  if (level !== "xhigh") return level;
591
655
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
592
656
  }
593
657
  var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
594
658
  var EFFORT_TO_BUDGET_TOKENS = {
659
+ // 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
660
+ minimal: 1024,
595
661
  low: 1024,
596
662
  medium: 4096,
597
663
  high: 8192,
@@ -697,11 +763,20 @@ var AnthropicDriver = class {
697
763
 
698
764
  // src/providers/google.ts
699
765
  var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
766
+ var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
767
+ var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
768
+ var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
700
769
  function needsCodeExecution(model) {
701
770
  const match = model.match(/^gemini-(\d+)/);
702
771
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
703
772
  }
773
+ function sleep(ms) {
774
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
775
+ }
704
776
  var GOOGLE_THINKING_LEVEL = {
777
+ // Gemini does define a "minimal" thinking level, but some models reject it
778
+ // (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
779
+ minimal: "low",
705
780
  low: "low",
706
781
  medium: "medium",
707
782
  high: "high",
@@ -768,24 +843,28 @@ var GoogleDriver = class {
768
843
  });
769
844
  return this.client;
770
845
  }
771
- async sendMessage(images, prompt, _options) {
772
- const client = await this.getClient();
846
+ /** Request config shared by image and video messages. */
847
+ generationConfig() {
848
+ return {
849
+ responseMimeType: "application/json",
850
+ maxOutputTokens: this.maxTokens,
851
+ ...this.reasoningEffort && {
852
+ thinkingConfig: {
853
+ thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
854
+ }
855
+ },
856
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
857
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
858
+ }
859
+ };
860
+ }
861
+ /** Runs one generateContent call and normalizes finish reasons, text, and usage. */
862
+ async generate(client, contents) {
773
863
  try {
774
864
  const response = await client.models.generateContent({
775
865
  model: this.model,
776
- contents: [...this.toGeminiParts(images), prompt],
777
- config: {
778
- responseMimeType: "application/json",
779
- maxOutputTokens: this.maxTokens,
780
- ...this.reasoningEffort && {
781
- thinkingConfig: {
782
- thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
783
- }
784
- },
785
- ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
786
- mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
787
- }
788
- }
866
+ contents,
867
+ config: this.generationConfig()
789
868
  });
790
869
  const finishReason = response.candidates?.[0]?.finishReason;
791
870
  if (finishReason === "MAX_TOKENS") {
@@ -800,9 +879,8 @@ var GoogleDriver = class {
800
879
  `Response blocked: Google returned finishReason "${finishReason}".`
801
880
  );
802
881
  }
803
- const text = response.text ?? "";
804
882
  return {
805
- text,
883
+ text: response.text ?? "",
806
884
  usage: toGeminiUsage(response.usageMetadata)
807
885
  };
808
886
  } catch (err) {
@@ -810,6 +888,80 @@ var GoogleDriver = class {
810
888
  throw mapProviderError(err);
811
889
  }
812
890
  }
891
+ async sendMessage(images, prompt, _options) {
892
+ const client = await this.getClient();
893
+ return this.generate(client, [...this.toGeminiParts(images), prompt]);
894
+ }
895
+ /**
896
+ * Uploads a video through the Files API and waits until Gemini has finished
897
+ * processing it. Returns the ACTIVE file record.
898
+ */
899
+ async uploadVideo(client, video) {
900
+ try {
901
+ const uploaded = await client.files.upload({
902
+ file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
903
+ config: { mimeType: video.mimeType }
904
+ });
905
+ const name = uploaded.name;
906
+ if (!name) {
907
+ throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
908
+ }
909
+ const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
910
+ let current = uploaded;
911
+ while (current.state === "PROCESSING") {
912
+ if (Date.now() > deadline) {
913
+ throw new VisualAIProviderError(
914
+ `Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
915
+ );
916
+ }
917
+ await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
918
+ current = await client.files.get({ name });
919
+ }
920
+ if (current.state !== "ACTIVE") {
921
+ throw new VisualAIProviderError(
922
+ `Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
923
+ );
924
+ }
925
+ if (!current.uri) {
926
+ throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
927
+ }
928
+ return current;
929
+ } catch (err) {
930
+ if (err instanceof VisualAIProviderError) throw err;
931
+ throw mapProviderError(err);
932
+ }
933
+ }
934
+ /**
935
+ * Sends the video bytes themselves. Gemini samples the clip server-side at
936
+ * `video.fps` and, unlike sampled frames, also hears the audio track. Small
937
+ * videos go inline; larger ones are uploaded via the Files API and deleted
938
+ * again afterwards (they would expire on their own after 48 h).
939
+ */
940
+ async sendVideoMessage(video, prompt, _options) {
941
+ const client = await this.getClient();
942
+ const videoMetadata = { fps: video.fps };
943
+ if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
944
+ const part = {
945
+ inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
946
+ videoMetadata
947
+ };
948
+ const response = await this.generate(client, [part, prompt]);
949
+ return { ...response, delivery: "inline" };
950
+ }
951
+ const file = await this.uploadVideo(client, video);
952
+ try {
953
+ const part = {
954
+ fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
955
+ videoMetadata
956
+ };
957
+ const response = await this.generate(client, [part, prompt]);
958
+ return { ...response, delivery: "file" };
959
+ } finally {
960
+ if (file.name) {
961
+ await client.files.delete({ name: file.name }).catch(() => void 0);
962
+ }
963
+ }
964
+ }
813
965
  async generateImage(images, prompt, options) {
814
966
  const client = await this.getClient();
815
967
  const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
@@ -942,6 +1094,9 @@ var OpenAIDriver = class {
942
1094
  // src/providers/openrouter.ts
943
1095
  var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
944
1096
  var OPENROUTER_REASONING_EFFORT = {
1097
+ // OpenRouter normalizes upstream vendors to low/medium/high only, so
1098
+ // "minimal" has no native equivalent and clamps to the floor.
1099
+ minimal: "low",
945
1100
  low: "low",
946
1101
  medium: "medium",
947
1102
  high: "high",
@@ -1101,10 +1256,20 @@ function parseBooleanEnv(envName, value) {
1101
1256
  `Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
1102
1257
  );
1103
1258
  }
1259
+ function parseReasoningEffortEnv(envName, value) {
1260
+ if (value === void 0 || value === "") return void 0;
1261
+ const levels = Object.values(ReasoningEffort);
1262
+ const lower = value.toLowerCase();
1263
+ if (levels.includes(lower)) return lower;
1264
+ throw new VisualAIConfigError(
1265
+ `Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
1266
+ );
1267
+ }
1104
1268
  var debugDeprecationWarned = false;
1105
1269
  function resolveConfig(config) {
1106
1270
  const provider = resolveProvider(config);
1107
1271
  const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
1272
+ const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
1108
1273
  const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
1109
1274
  const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
1110
1275
  const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
@@ -1122,12 +1287,12 @@ function resolveConfig(config) {
1122
1287
  }
1123
1288
  const userSetMaxTokens = config.maxTokens !== void 0;
1124
1289
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1125
- const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1290
+ const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
1126
1291
  const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1127
1292
  if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1128
1293
  maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1129
1294
  if (debug) {
1130
- const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1295
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
1131
1296
  process.stderr.write(
1132
1297
  `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1133
1298
  `
@@ -1139,7 +1304,7 @@ function resolveConfig(config) {
1139
1304
  apiKey: config.apiKey,
1140
1305
  model,
1141
1306
  maxTokens,
1142
- reasoningEffort: config.reasoningEffort,
1307
+ reasoningEffort,
1143
1308
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1144
1309
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1145
1310
  timeout: config.timeout,
@@ -1162,6 +1327,10 @@ var PRICING_TABLE = {
1162
1327
  inputPricePerToken: 10 / PER_MILLION,
1163
1328
  outputPricePerToken: 50 / PER_MILLION
1164
1329
  },
1330
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
1331
+ inputPricePerToken: 4 / PER_MILLION,
1332
+ outputPricePerToken: 20 / PER_MILLION
1333
+ },
1165
1334
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1166
1335
  inputPricePerToken: 5 / PER_MILLION,
1167
1336
  outputPricePerToken: 25 / PER_MILLION
@@ -1170,6 +1339,10 @@ var PRICING_TABLE = {
1170
1339
  inputPricePerToken: 5 / PER_MILLION,
1171
1340
  outputPricePerToken: 25 / PER_MILLION
1172
1341
  },
1342
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
1343
+ inputPricePerToken: 2 / PER_MILLION,
1344
+ outputPricePerToken: 10 / PER_MILLION
1345
+ },
1173
1346
  [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
1174
1347
  inputPricePerToken: 3 / PER_MILLION,
1175
1348
  outputPricePerToken: 15 / PER_MILLION
@@ -1196,6 +1369,22 @@ var PRICING_TABLE = {
1196
1369
  inputPricePerToken: 10 / PER_MILLION,
1197
1370
  outputPricePerToken: 50 / PER_MILLION
1198
1371
  },
1372
+ // Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
1373
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
1374
+ inputPricePerToken: 2 / PER_MILLION,
1375
+ outputPricePerToken: 10 / PER_MILLION
1376
+ },
1377
+ // Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
1378
+ // bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
1379
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
1380
+ inputPricePerToken: 2 / PER_MILLION,
1381
+ outputPricePerToken: 10 / PER_MILLION
1382
+ },
1383
+ // Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
1384
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
1385
+ inputPricePerToken: 0.1 / PER_MILLION,
1386
+ outputPricePerToken: 0.5 / PER_MILLION
1387
+ },
1199
1388
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1200
1389
  inputPricePerToken: 5 / PER_MILLION,
1201
1390
  outputPricePerToken: 30 / PER_MILLION
@@ -1292,6 +1481,10 @@ var PRICING_TABLE = {
1292
1481
  inputPricePerToken: 0.1 / PER_MILLION,
1293
1482
  outputPricePerToken: 0.2 / PER_MILLION
1294
1483
  },
1484
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
1485
+ inputPricePerToken: 1.6 / PER_MILLION,
1486
+ outputPricePerToken: 4.8 / PER_MILLION
1487
+ },
1295
1488
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1296
1489
  inputPricePerToken: 2 / PER_MILLION,
1297
1490
  outputPricePerToken: 6 / PER_MILLION
@@ -1325,6 +1518,13 @@ var PRICING_TABLE = {
1325
1518
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1326
1519
  inputPricePerToken: 0.15 / PER_MILLION,
1327
1520
  outputPricePerToken: 0.5 / PER_MILLION
1521
+ },
1522
+ // Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
1523
+ // upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
1524
+ // is $0.0036/MTok, not modelled (no provider gets a cache discount here).
1525
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
1526
+ inputPricePerToken: 0.435 / PER_MILLION,
1527
+ outputPricePerToken: 0.87 / PER_MILLION
1328
1528
  }
1329
1529
  };
1330
1530
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1402,6 +1602,15 @@ async function timedSendMessage(driver, images, prompt, options) {
1402
1602
  const durationSeconds = (performance.now() - start) / 1e3;
1403
1603
  return { ...response, durationSeconds };
1404
1604
  }
1605
+ async function timedSendVideoMessage(driver, video, prompt, options) {
1606
+ if (!driver.sendVideoMessage) {
1607
+ throw new VisualAIError("Provider driver does not support native video delivery");
1608
+ }
1609
+ const start = performance.now();
1610
+ const response = await driver.sendVideoMessage(video, prompt, options);
1611
+ const durationSeconds = (performance.now() - start) / 1e3;
1612
+ return { ...response, durationSeconds };
1613
+ }
1405
1614
 
1406
1615
  // src/core/diff.ts
1407
1616
  import sharp from "sharp";
@@ -1641,6 +1850,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
1641
1850
  };
1642
1851
  }
1643
1852
 
1853
+ // src/core/media.ts
1854
+ import { readFile as readFile3 } from "fs/promises";
1855
+
1644
1856
  // src/core/debug-frames.ts
1645
1857
  import { randomBytes } from "crypto";
1646
1858
  import { mkdir, writeFile } from "fs/promises";
@@ -1696,6 +1908,92 @@ async function saveDebugFrames(frames, env = process.env) {
1696
1908
  return runDir;
1697
1909
  }
1698
1910
 
1911
+ // src/core/frame-dedupe.ts
1912
+ import sharp3 from "sharp";
1913
+ var DEDUPE_THUMBNAIL_EDGE = 256;
1914
+ var DEDUPE_PIXEL_TOLERANCE = 24;
1915
+ var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
1916
+ function resolveDedupeOptions(raw) {
1917
+ if (raw === void 0 || raw === true) {
1918
+ return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
1919
+ }
1920
+ if (raw === false) {
1921
+ return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
1922
+ }
1923
+ const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
1924
+ if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
1925
+ throw new VisualAIVideoError(
1926
+ `Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
1927
+ );
1928
+ }
1929
+ return { enabled: true, threshold };
1930
+ }
1931
+ async function frameSignature(frame) {
1932
+ try {
1933
+ const { data, info } = await sharp3(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
1934
+ fit: "inside",
1935
+ withoutEnlargement: true
1936
+ }).raw().toBuffer({ resolveWithObject: true });
1937
+ return { width: info.width, height: info.height, pixels: data };
1938
+ } catch (err) {
1939
+ const reason = err instanceof Error ? err.message : String(err);
1940
+ throw new VisualAIVideoError(
1941
+ `Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
1942
+ );
1943
+ }
1944
+ }
1945
+ function changedFraction(a, b) {
1946
+ if (a.width !== b.width || a.height !== b.height) {
1947
+ return 1;
1948
+ }
1949
+ let changed = 0;
1950
+ for (let i = 0; i < a.pixels.length; i++) {
1951
+ if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
1952
+ changed++;
1953
+ }
1954
+ }
1955
+ return changed / a.pixels.length;
1956
+ }
1957
+ function reindex(frame, index) {
1958
+ if (frame.index === index) {
1959
+ return frame;
1960
+ }
1961
+ return {
1962
+ data: frame.data,
1963
+ mimeType: frame.mimeType,
1964
+ get base64() {
1965
+ return frame.base64;
1966
+ },
1967
+ timestampSeconds: frame.timestampSeconds,
1968
+ index
1969
+ };
1970
+ }
1971
+ async function dedupeFrames(frames, options) {
1972
+ const { enabled, threshold } = resolveDedupeOptions(options);
1973
+ if (!enabled || frames.length < 2) {
1974
+ return { frames: [...frames], dropped: 0 };
1975
+ }
1976
+ const signed = await Promise.all(
1977
+ frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
1978
+ );
1979
+ const [first, ...rest] = signed;
1980
+ if (first === void 0) {
1981
+ return { frames: [], dropped: 0 };
1982
+ }
1983
+ const kept = [first.frame];
1984
+ let lastKept = first.signature;
1985
+ for (const { frame, signature } of rest) {
1986
+ if (changedFraction(lastKept, signature) >= threshold) {
1987
+ kept.push(frame);
1988
+ lastKept = signature;
1989
+ }
1990
+ }
1991
+ return {
1992
+ frames: kept.map((frame, index) => reindex(frame, index)),
1993
+ dropped: frames.length - kept.length
1994
+ };
1995
+ }
1996
+
1699
1997
  // src/core/video.ts
1700
1998
  import { mkdtemp, readFile as readFile2, readdir, rm, writeFile as writeFile2 } from "fs/promises";
1701
1999
  import { tmpdir } from "os";
@@ -1931,7 +2229,7 @@ async function probeDurationSeconds(videoPath) {
1931
2229
  });
1932
2230
  });
1933
2231
  }
1934
- async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2232
+ function resolveVideoSamplingOptions(options = {}) {
1935
2233
  const fps = options.fps ?? DEFAULT_FPS;
1936
2234
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1937
2235
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1951,13 +2249,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
1951
2249
  `Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
1952
2250
  );
1953
2251
  }
1954
- const ffmpeg = await loadFfmpegFactory();
1955
- const durationSeconds = await probeDurationSeconds(videoPath);
2252
+ return { fps, maxFrames, maxDurationSeconds };
2253
+ }
2254
+ function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
1956
2255
  if (durationSeconds > maxDurationSeconds) {
1957
2256
  throw new VisualAIVideoError(
1958
2257
  `Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
1959
2258
  );
1960
2259
  }
2260
+ }
2261
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2262
+ const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
2263
+ const ffmpeg = await loadFfmpegFactory();
2264
+ const durationSeconds = await probeDurationSeconds(videoPath);
2265
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
1961
2266
  const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
1962
2267
  try {
1963
2268
  const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
@@ -2061,6 +2366,7 @@ function isTimestampedFrameInput(frame) {
2061
2366
  async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2062
2367
  const rawFrames = input.frames;
2063
2368
  const fps = input.fps ?? DEFAULT_FPS;
2369
+ resolveDedupeOptions(input.dedupe);
2064
2370
  if (rawFrames.length === 0) {
2065
2371
  throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
2066
2372
  }
@@ -2072,7 +2378,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2072
2378
  if (!Number.isFinite(fps) || fps <= 0) {
2073
2379
  throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
2074
2380
  }
2075
- const frames = await Promise.all(
2381
+ const sampled = await Promise.all(
2076
2382
  rawFrames.map(async (raw, index) => {
2077
2383
  const timestamped = isTimestampedFrameInput(raw);
2078
2384
  const imageInput = timestamped ? raw.image : raw;
@@ -2095,20 +2401,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2095
2401
  };
2096
2402
  })
2097
2403
  );
2098
- const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2404
+ const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2405
+ const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
2099
2406
  await saveDebugFrames(frames);
2100
- return { kind: "video", frames, durationSeconds };
2407
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2408
+ }
2409
+ async function normalizeNativeVideo(input, videoOptions) {
2410
+ const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
2411
+ const { path, mimeType, cleanup } = await resolveVideoToPath(input);
2412
+ try {
2413
+ const durationSeconds = await probeDurationSeconds(path);
2414
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2415
+ const data = await readFile3(path);
2416
+ return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
2417
+ } finally {
2418
+ try {
2419
+ await cleanup();
2420
+ } catch {
2421
+ }
2422
+ }
2101
2423
  }
2102
- async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2424
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
2103
2425
  if (isFramesInput(input)) {
2104
2426
  return normalizeFrames(input, maxDimension);
2105
2427
  }
2106
2428
  if (isVideoInput(input)) {
2429
+ if (nativeVideo) {
2430
+ return normalizeNativeVideo(input, videoOptions);
2431
+ }
2432
+ resolveDedupeOptions(videoOptions?.dedupe);
2107
2433
  const { path, cleanup } = await resolveVideoToPath(input);
2108
2434
  try {
2109
- const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
2435
+ const { frames: sampled, durationSeconds } = await extractFrames(
2436
+ path,
2437
+ videoOptions,
2438
+ maxDimension
2439
+ );
2440
+ const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
2110
2441
  await saveDebugFrames(frames);
2111
- return { kind: "video", frames, durationSeconds };
2442
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2112
2443
  } finally {
2113
2444
  try {
2114
2445
  await cleanup();
@@ -2198,6 +2529,12 @@ var AskResultSchema = z.object({
2198
2529
  * omitting the key, even for image inputs that were never asked to populate it.
2199
2530
  */
2200
2531
  frameReferences: z.array(z.number().int().nonnegative()).nullable().optional(),
2532
+ /**
2533
+ * For natively delivered video, the timestamps (seconds from the start of
2534
+ * the clip) the model relied on to answer. The native counterpart of
2535
+ * `frameReferences`. Nullable for the same strict-schema reason.
2536
+ */
2537
+ timestampReferences: z.array(z.number().nonnegative()).nullable().optional(),
2201
2538
  usage: UsageInfoSchema.optional()
2202
2539
  });
2203
2540
 
@@ -2209,6 +2546,12 @@ function stripCodeFences(text) {
2209
2546
  }
2210
2547
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2211
2548
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2549
+ var AskImageResponseSchema = AskResponseSchema.omit({
2550
+ frameReferences: true,
2551
+ timestampReferences: true
2552
+ });
2553
+ var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
2554
+ var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
2212
2555
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2213
2556
  var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2214
2557
  function parseJson(text) {
@@ -2276,7 +2619,8 @@ function parseAskResponse(raw) {
2276
2619
  const result = parseResponse(raw, AskResponseSchema);
2277
2620
  return {
2278
2621
  ...result,
2279
- frameReferences: result.frameReferences ?? void 0
2622
+ frameReferences: result.frameReferences ?? void 0,
2623
+ timestampReferences: result.timestampReferences ?? void 0
2280
2624
  };
2281
2625
  }
2282
2626
  function parseCompareResponse(raw) {
@@ -2300,31 +2644,75 @@ function createDriver(provider, config) {
2300
2644
  return PROVIDER_REGISTRY[provider](config);
2301
2645
  }
2302
2646
  var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
2303
- var askSchemaOptions = toSchemaOptions(AskResponseSchema);
2647
+ var askSchemaOptionsByMedia = {
2648
+ image: toSchemaOptions(AskImageResponseSchema),
2649
+ video: toSchemaOptions(AskFramesResponseSchema),
2650
+ "native-video": toSchemaOptions(AskNativeVideoResponseSchema)
2651
+ };
2304
2652
  var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2305
2653
  function mediaToProviderInputs(media) {
2306
2654
  if (media.kind === "image") {
2307
2655
  return {
2656
+ kind: "images",
2308
2657
  images: [media.image],
2309
2658
  mediaContext: { kind: "image" },
2310
- framesMetadata: void 0
2659
+ frames: void 0
2660
+ };
2661
+ }
2662
+ if (media.kind === "native-video") {
2663
+ return {
2664
+ kind: "native-video",
2665
+ video: media.video,
2666
+ mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
2311
2667
  };
2312
2668
  }
2313
2669
  const timestamps = media.frames.map((f) => f.timestampSeconds);
2314
2670
  return {
2671
+ kind: "images",
2315
2672
  images: media.frames,
2316
2673
  mediaContext: {
2317
2674
  kind: "video",
2318
2675
  frameTimestamps: timestamps,
2319
- durationSeconds: media.durationSeconds
2676
+ durationSeconds: media.durationSeconds,
2677
+ droppedUnchanged: media.droppedUnchanged
2320
2678
  },
2321
- framesMetadata: {
2679
+ frames: {
2322
2680
  count: media.frames.length,
2323
2681
  timestampsSeconds: timestamps,
2324
- durationSeconds: media.durationSeconds
2682
+ durationSeconds: media.durationSeconds,
2683
+ droppedUnchanged: media.droppedUnchanged
2325
2684
  }
2326
2685
  };
2327
2686
  }
2687
+ async function sendMedia(driver, dispatch, prompt, options) {
2688
+ if (dispatch.kind === "native-video") {
2689
+ const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
2690
+ const { durationSeconds, fps, mimeType } = dispatch.video;
2691
+ return {
2692
+ response: response2,
2693
+ metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
2694
+ };
2695
+ }
2696
+ const response = await timedSendMessage(driver, dispatch.images, prompt, options);
2697
+ return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
2698
+ }
2699
+ function resolveNativeVideo(input, videoOptions, driver, provider) {
2700
+ const mode = videoOptions?.mode ?? "auto";
2701
+ if (mode !== "auto" && mode !== "native" && mode !== "frames") {
2702
+ throw new VisualAIConfigError(
2703
+ `Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
2704
+ );
2705
+ }
2706
+ const supported = typeof driver.sendVideoMessage === "function";
2707
+ if (mode === "frames") return false;
2708
+ if (mode === "auto") return supported;
2709
+ if (!supported && !isFramesInput(input) && isVideoInput(input)) {
2710
+ throw new VisualAIConfigError(
2711
+ `Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
2712
+ );
2713
+ }
2714
+ return true;
2715
+ }
2328
2716
  function visualAI(config = {}) {
2329
2717
  const resolvedConfig = resolveConfig(config);
2330
2718
  const driverConfig = {
@@ -2362,38 +2750,60 @@ function visualAI(config = {}) {
2362
2750
  throw new VisualAIConfigError("At least one statement is required for check()");
2363
2751
  }
2364
2752
  return withErrorDebug(resolvedConfig, "check", async () => {
2365
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2366
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2753
+ const nativeVideo = resolveNativeVideo(
2754
+ input,
2755
+ options?.video,
2756
+ driver,
2757
+ resolvedConfig.provider
2758
+ );
2759
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2760
+ const dispatch = mediaToProviderInputs(media);
2367
2761
  const prompt = buildCheckPrompt(stmts, {
2368
2762
  instructions: options?.instructions,
2369
- media: mediaContext
2763
+ media: dispatch.mediaContext
2370
2764
  });
2371
2765
  debugLog(resolvedConfig, "check prompt", prompt, "prompt");
2372
- const response = await timedSendMessage(driver, images, prompt, checkSchemaOptions);
2766
+ const { response, metadata } = await sendMedia(
2767
+ driver,
2768
+ dispatch,
2769
+ prompt,
2770
+ checkSchemaOptions
2771
+ );
2373
2772
  debugLog(resolvedConfig, "check response", response.text, "response");
2374
2773
  const result = parseCheckResponse(response.text);
2375
2774
  return {
2376
2775
  ...result,
2377
- ...framesMetadata ? { frames: framesMetadata } : {},
2776
+ ...metadata,
2378
2777
  usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
2379
2778
  };
2380
2779
  });
2381
2780
  },
2382
2781
  async ask(input, userPrompt, options) {
2383
2782
  return withErrorDebug(resolvedConfig, "ask", async () => {
2384
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2385
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2783
+ const nativeVideo = resolveNativeVideo(
2784
+ input,
2785
+ options?.video,
2786
+ driver,
2787
+ resolvedConfig.provider
2788
+ );
2789
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2790
+ const dispatch = mediaToProviderInputs(media);
2386
2791
  const prompt = buildAskPrompt(userPrompt, {
2387
2792
  instructions: options?.instructions,
2388
- media: mediaContext
2793
+ media: dispatch.mediaContext
2389
2794
  });
2390
2795
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2391
- const response = await timedSendMessage(driver, images, prompt, askSchemaOptions);
2796
+ const { response, metadata } = await sendMedia(
2797
+ driver,
2798
+ dispatch,
2799
+ prompt,
2800
+ askSchemaOptionsByMedia[dispatch.mediaContext.kind]
2801
+ );
2392
2802
  debugLog(resolvedConfig, "ask response", response.text, "response");
2393
2803
  const result = parseAskResponse(response.text);
2394
2804
  return {
2395
2805
  ...result,
2396
- ...framesMetadata ? { frames: framesMetadata } : {},
2806
+ ...metadata,
2397
2807
  usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
2398
2808
  };
2399
2809
  });
@@ -2411,7 +2821,7 @@ function visualAI(config = {}) {
2411
2821
  debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
2412
2822
  const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
2413
2823
  debugLog(resolvedConfig, "compare response", response.text, "response");
2414
- const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model === Model.Google.GEMINI_3_FLASH_PREVIEW;
2824
+ const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
2415
2825
  const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
2416
2826
  let diffImage;
2417
2827
  if (effectiveDiffImage) {