visual-ai-assertions 0.23.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -67,135 +67,6 @@ __export(index_exports, {
67
67
  });
68
68
  module.exports = __toCommonJS(index_exports);
69
69
 
70
- // src/constants.ts
71
- var ReasoningEffort = {
72
- LOW: "low",
73
- MEDIUM: "medium",
74
- HIGH: "high",
75
- XHIGH: "xhigh"
76
- };
77
- var ImageDetail = {
78
- AUTO: "auto",
79
- LOW: "low",
80
- HIGH: "high"
81
- };
82
- var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
83
- var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
84
- var Provider = {
85
- ANTHROPIC: "anthropic",
86
- OPENAI: "openai",
87
- GOOGLE: "google",
88
- OPENROUTER: "openrouter"
89
- };
90
- var Model = {
91
- Anthropic: {
92
- FABLE_5_1: "claude-fable-5-1",
93
- FABLE_5: "claude-fable-5",
94
- OPUS_5: "claude-opus-5",
95
- OPUS_4_8: "claude-opus-4-8",
96
- OPUS_4_7: "claude-opus-4-7",
97
- OPUS_4_6: "claude-opus-4-6",
98
- SONNET_5: "claude-sonnet-5",
99
- SONNET_4_6: "claude-sonnet-4-6",
100
- HAIKU_4_5: "claude-haiku-4-5"
101
- },
102
- OpenAI: {
103
- GPT_6_ASTRA: "gpt-6-astra",
104
- GPT_5_6_SOL: "gpt-5.6-sol",
105
- GPT_5_6_TERRA: "gpt-5.6-terra",
106
- GPT_5_6_LUNA: "gpt-5.6-luna",
107
- GPT_5_5: "gpt-5.5",
108
- GPT_5_4: "gpt-5.4",
109
- GPT_5_4_PRO: "gpt-5.4-pro",
110
- GPT_5_4_MINI: "gpt-5.4-mini",
111
- GPT_5_4_NANO: "gpt-5.4-nano",
112
- GPT_5_2: "gpt-5.2",
113
- GPT_5_MINI: "gpt-5-mini"
114
- },
115
- Google: {
116
- GEMINI_3_8_FLASH: "gemini-3.8-flash",
117
- GEMINI_3_7_FLASH: "gemini-3.7-flash",
118
- GEMINI_3_6_FLASH: "gemini-3.6-flash",
119
- GEMINI_3_5_FLASH: "gemini-3.5-flash",
120
- GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
121
- GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
122
- GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
123
- GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
124
- },
125
- /**
126
- * Models routed through OpenRouter (https://openrouter.ai). Slugs always
127
- * carry a vendor prefix (`vendor/model`), which is how provider inference
128
- * recognizes them. All listed models accept image input.
129
- */
130
- OpenRouter: {
131
- MUSE_SPARK_1_3: "meta/muse-spark-1.3",
132
- GROK_4_6: "x-ai/grok-4.6",
133
- GROK_4_5: "x-ai/grok-4.5",
134
- KIMI_K3: "moonshotai/kimi-k3",
135
- KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
136
- QWEN_3_8_MAX: "qwen/qwen3.8-max",
137
- QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
138
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
139
- GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
140
- }
141
- };
142
- var DEFAULT_MODELS = {
143
- [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
144
- [Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
145
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
146
- [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
147
- };
148
- var DEFAULT_MAX_TOKENS = 4096;
149
- var OPENAI_REASONING_MAX_TOKENS = 16384;
150
- var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
151
- var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
152
- Model.OpenAI.GPT_6_ASTRA
153
- ]);
154
- var MODEL_TO_PROVIDER = new Map([
155
- ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
156
- ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
157
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
158
- ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
159
- ]);
160
- var VALID_PROVIDERS = Object.values(Provider);
161
- var PROVIDER_DEFAULT_REASONING = {
162
- openai: "medium",
163
- anthropic: "off",
164
- google: "off",
165
- // Varies by upstream model; the driver sends no reasoning field unless configured.
166
- openrouter: "off"
167
- };
168
- var Content = {
169
- /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
170
- PLACEHOLDER_TEXT: "placeholder-text",
171
- /** Detects error messages, banners, stack traces, or error codes */
172
- ERROR_MESSAGES: "error-messages",
173
- /** Detects broken image icons or failed-to-load image indicators */
174
- BROKEN_IMAGES: "broken-images",
175
- /** Detects UI elements that unintentionally overlap and obscure content */
176
- OVERLAPPING_ELEMENTS: "overlapping-elements"
177
- };
178
- var Layout = {
179
- /** Detects elements that unintentionally overlap each other */
180
- OVERLAP: "overlap",
181
- /** Detects content cut off or extending beyond container boundaries */
182
- OVERFLOW: "overflow",
183
- /** Detects inconsistent alignment of text, images, and UI components */
184
- ALIGNMENT: "alignment"
185
- };
186
- var Accessibility = {
187
- /** Detects insufficient color contrast between text and backgrounds */
188
- CONTRAST: "contrast",
189
- /** Detects text that is cut off, overlapping, too small, or obscured */
190
- READABILITY: "readability",
191
- /** Detects interactive elements that are not visually distinct */
192
- INTERACTIVE_VISIBILITY: "interactive-visibility",
193
- /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
194
- COLOR_BLINDNESS: "color-blindness",
195
- /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
196
- COLOR_ALONE: "color-alone"
197
- };
198
-
199
70
  // src/errors.ts
200
71
  var VisualAIError = class extends Error {
201
72
  code;
@@ -327,10 +198,15 @@ Example for a failing check:
327
198
  ]
328
199
  }
329
200
  ${JSON_INSTRUCTIONS}`;
330
- var CHECK_OUTPUT_SCHEMA_VIDEO = `IMPORTANT: Follow this evaluation order:
201
+ function buildCheckOutputSchemaVideo(unit) {
202
+ const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
203
+ const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
204
+ const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
205
+ const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
206
+ return `IMPORTANT: Follow this evaluation order:
331
207
  1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
332
- 2. A statement passes if it is true at ANY frame of the timeline, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
333
- 3. For each statement that passes, set "timestampSeconds" to the timestamp of the frame that most clearly demonstrates it (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
208
+ 2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
209
+ 3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
334
210
  4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
335
211
  5. Write "reasoning" as a brief overall summary of the evaluation
336
212
  6. Include "issues" only for statements that failed
@@ -344,7 +220,7 @@ Respond with a JSON object matching this exact structure:
344
220
  {
345
221
  "statement": string, // the original statement text
346
222
  "pass": boolean, // whether this statement is true at any point in the timeline
347
- "reasoning": string, // explanation for this statement, citing frame timestamps where relevant
223
+ "reasoning": string, // explanation for this statement, ${citing} where relevant
348
224
  "confidence": "high" | "medium" | "low",
349
225
  "timestampSeconds": number | null
350
226
  // seconds from the start of the clip where the statement is most clearly true,
@@ -362,10 +238,13 @@ Example for a passing video check:
362
238
  "reasoning": "The success toast appeared briefly around 3.5s.",
363
239
  "issues": [],
364
240
  "statements": [
365
- { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right at the 3.5s frame", "confidence": "high", "timestampSeconds": 3.5 }
241
+ { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
366
242
  ]
367
243
  }
368
244
  ${JSON_INSTRUCTIONS}`;
245
+ }
246
+ var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
247
+ var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
369
248
  var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
370
249
  {
371
250
  "summary": string, // high-level analysis summary
@@ -398,6 +277,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
398
277
  Prioritize issues by severity (critical / major / minor) as for image input.
399
278
  Cite frame indices in "frameReferences" so the user can locate the moments you describe.
400
279
  ${JSON_INSTRUCTIONS}`;
280
+ var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
281
+ {
282
+ "summary": string, // high-level summary of what happens across the video
283
+ "issues": [...], // list of issues/findings, can be empty
284
+ "timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
285
+ }
286
+ ${ISSUE_SCHEMA_INSTRUCTIONS}
287
+
288
+ Prioritize issues by severity (critical / major / minor) as for image input.
289
+ Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
290
+ ${JSON_INSTRUCTIONS}`;
401
291
  var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
402
292
  {
403
293
  "pass": boolean, // true if no critical or major changes found
@@ -419,15 +309,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
419
309
  var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
420
310
  var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
421
311
  var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
422
- function buildVideoTimelineSection(frameTimestamps, durationSeconds) {
312
+ var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
313
+ var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
314
+ function buildNativeVideoSection(durationSeconds) {
315
+ return `Video recording:
316
+ - Total duration: ${durationSeconds.toFixed(2)}s
317
+
318
+ The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
319
+ }
320
+ function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
423
321
  const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
322
+ const attached = frameTimestamps.length;
323
+ const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
324
+ const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
424
325
  return `Video timeline:
425
326
  - Total duration: ${durationSeconds.toFixed(2)}s
426
- - ${frameTimestamps.length} frames sampled (in chronological order)
327
+ ${sampledLine}
427
328
  - Frame index \u2192 timestamp:
428
329
  ${formatted}
429
330
 
430
- Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.`;
331
+ Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
431
332
  }
432
333
  var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
433
334
  var COMPARE_EDGE_RULES = [
@@ -442,30 +343,52 @@ function buildCheckPrompt(statements, options) {
442
343
  const stmts = Array.isArray(statements) ? statements : [statements];
443
344
  const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
444
345
  const media = options?.media;
445
- const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
346
+ const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
446
347
  const sections = [options?.role ?? defaultRole];
447
348
  if (media?.kind === "video") {
448
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
349
+ sections.push(
350
+ buildVideoTimelineSection(
351
+ media.frameTimestamps,
352
+ media.durationSeconds,
353
+ media.droppedUnchanged
354
+ )
355
+ );
356
+ } else if (media?.kind === "native-video") {
357
+ sections.push(buildNativeVideoSection(media.durationSeconds));
449
358
  }
450
359
  if (options?.instructions && options.instructions.length > 0) {
451
360
  sections.push(buildInstructionsSection(options.instructions));
452
361
  }
453
362
  sections.push(`Statements to evaluate:
454
363
  ${statementsBlock}`);
455
- sections.push(media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE);
364
+ sections.push(
365
+ media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
366
+ );
456
367
  return sections.join("\n\n");
457
368
  }
458
369
  function buildAskPrompt(userPrompt, options) {
459
370
  const media = options?.media;
460
- const sections = [media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : DEFAULT_ASK_ROLE];
371
+ const sections = [
372
+ media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
373
+ ];
461
374
  if (media?.kind === "video") {
462
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
375
+ sections.push(
376
+ buildVideoTimelineSection(
377
+ media.frameTimestamps,
378
+ media.durationSeconds,
379
+ media.droppedUnchanged
380
+ )
381
+ );
382
+ } else if (media?.kind === "native-video") {
383
+ sections.push(buildNativeVideoSection(media.durationSeconds));
463
384
  }
464
385
  if (options?.instructions && options.instructions.length > 0) {
465
386
  sections.push(buildInstructionsSection(options.instructions));
466
387
  }
467
388
  sections.push(`User request: ${userPrompt}`);
468
- sections.push(media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE);
389
+ sections.push(
390
+ media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
391
+ );
469
392
  return sections.join("\n\n");
470
393
  }
471
394
  function buildAiDiffPrompt() {
@@ -510,7 +433,7 @@ function visibleRole(finalState, requireCorrectRendering) {
510
433
  var ELEMENTS_VISIBLE_CLIPPING_RULES = [
511
434
  "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
512
435
  "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
513
- "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
436
+ "The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
514
437
  ];
515
438
  var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
516
439
  var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
@@ -541,6 +464,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
541
464
  return buildCheckPrompt(statements, { role, instructions });
542
465
  }
543
466
 
467
+ // src/constants.ts
468
+ var ReasoningEffort = {
469
+ MINIMAL: "minimal",
470
+ LOW: "low",
471
+ MEDIUM: "medium",
472
+ HIGH: "high",
473
+ XHIGH: "xhigh"
474
+ };
475
+ var ImageDetail = {
476
+ AUTO: "auto",
477
+ LOW: "low",
478
+ HIGH: "high"
479
+ };
480
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
481
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
482
+ var Provider = {
483
+ ANTHROPIC: "anthropic",
484
+ OPENAI: "openai",
485
+ GOOGLE: "google",
486
+ OPENROUTER: "openrouter"
487
+ };
488
+ var Model = {
489
+ Anthropic: {
490
+ FABLE_5_1: "claude-fable-5-1",
491
+ FABLE_5: "claude-fable-5",
492
+ OPUS_5_5: "claude-opus-5-5",
493
+ OPUS_5: "claude-opus-5",
494
+ OPUS_4_8: "claude-opus-4-8",
495
+ OPUS_4_7: "claude-opus-4-7",
496
+ OPUS_4_6: "claude-opus-4-6",
497
+ SONNET_5_5: "claude-sonnet-5-5",
498
+ SONNET_5: "claude-sonnet-5",
499
+ SONNET_4_6: "claude-sonnet-4-6",
500
+ HAIKU_4_5: "claude-haiku-4-5"
501
+ },
502
+ OpenAI: {
503
+ GPT_6_ASTRA: "gpt-6-astra",
504
+ GPT_6_1_SOL: "gpt-6.1-sol",
505
+ GPT_6_SOL: "gpt-6-sol",
506
+ GPT_6_LUNA: "gpt-6-luna",
507
+ GPT_5_6_SOL: "gpt-5.6-sol",
508
+ GPT_5_6_TERRA: "gpt-5.6-terra",
509
+ GPT_5_6_LUNA: "gpt-5.6-luna",
510
+ GPT_5_5: "gpt-5.5",
511
+ GPT_5_4: "gpt-5.4",
512
+ GPT_5_4_PRO: "gpt-5.4-pro",
513
+ GPT_5_4_MINI: "gpt-5.4-mini",
514
+ GPT_5_4_NANO: "gpt-5.4-nano",
515
+ GPT_5_2: "gpt-5.2",
516
+ GPT_5_MINI: "gpt-5-mini"
517
+ },
518
+ Google: {
519
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
520
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
521
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
522
+ GEMINI_3_5_FLASH: "gemini-3.5-flash",
523
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
524
+ GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
525
+ GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
526
+ GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
527
+ },
528
+ /**
529
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
530
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
531
+ * recognizes them. All listed models accept image input.
532
+ */
533
+ OpenRouter: {
534
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
535
+ GROK_4_7: "x-ai/grok-4.7",
536
+ GROK_4_6: "x-ai/grok-4.6",
537
+ GROK_4_5: "x-ai/grok-4.5",
538
+ KIMI_K3: "moonshotai/kimi-k3",
539
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
540
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
541
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
542
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
543
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
544
+ MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
545
+ }
546
+ };
547
+ var DEFAULT_MODELS = {
548
+ [Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
549
+ [Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
550
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
551
+ [Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
552
+ };
553
+ var DEFAULT_MAX_TOKENS = 4096;
554
+ var OPENAI_REASONING_MAX_TOKENS = 16384;
555
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
556
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
557
+ Model.OpenRouter.QWEN_3_8_MAX,
558
+ Model.OpenRouter.QWEN_3_7_PLUS
559
+ ]);
560
+ var MODEL_TO_PROVIDER = new Map([
561
+ ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
562
+ ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
563
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
564
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
565
+ ]);
566
+ var VALID_PROVIDERS = Object.values(Provider);
567
+ var PROVIDER_DEFAULT_REASONING = {
568
+ openai: "medium",
569
+ anthropic: "off",
570
+ google: "off",
571
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
572
+ openrouter: "off"
573
+ };
574
+ var Content = {
575
+ /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
576
+ PLACEHOLDER_TEXT: "placeholder-text",
577
+ /** Detects error messages, banners, stack traces, or error codes */
578
+ ERROR_MESSAGES: "error-messages",
579
+ /** Detects broken image icons or failed-to-load image indicators */
580
+ BROKEN_IMAGES: "broken-images",
581
+ /** Detects UI elements that unintentionally overlap and obscure content */
582
+ OVERLAPPING_ELEMENTS: "overlapping-elements"
583
+ };
584
+ var Layout = {
585
+ /** Detects elements that unintentionally overlap each other */
586
+ OVERLAP: "overlap",
587
+ /** Detects content cut off or extending beyond container boundaries */
588
+ OVERFLOW: "overflow",
589
+ /** Detects inconsistent alignment of text, images, and UI components */
590
+ ALIGNMENT: "alignment"
591
+ };
592
+ var Accessibility = {
593
+ /** Detects insufficient color contrast between text and backgrounds */
594
+ CONTRAST: "contrast",
595
+ /** Detects text that is cut off, overlapping, too small, or obscured */
596
+ READABILITY: "readability",
597
+ /** Detects interactive elements that are not visually distinct */
598
+ INTERACTIVE_VISIBILITY: "interactive-visibility",
599
+ /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
600
+ COLOR_BLINDNESS: "color-blindness",
601
+ /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
602
+ COLOR_ALONE: "color-alone"
603
+ };
604
+
544
605
  // src/templates/accessibility.ts
545
606
  var ALL_CHECKS = Object.values(Accessibility);
546
607
  var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
@@ -650,17 +711,22 @@ function parseRetryAfter(value) {
650
711
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
651
712
  Model.Anthropic.FABLE_5_1,
652
713
  Model.Anthropic.FABLE_5,
714
+ Model.Anthropic.OPUS_5_5,
653
715
  Model.Anthropic.OPUS_5,
654
716
  Model.Anthropic.OPUS_4_8,
655
717
  Model.Anthropic.OPUS_4_7,
718
+ Model.Anthropic.SONNET_5_5,
656
719
  Model.Anthropic.SONNET_5
657
720
  ]);
658
721
  function mapEffort(level, model) {
722
+ if (level === "minimal") return "low";
659
723
  if (level !== "xhigh") return level;
660
724
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
661
725
  }
662
726
  var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
663
727
  var EFFORT_TO_BUDGET_TOKENS = {
728
+ // 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
729
+ minimal: 1024,
664
730
  low: 1024,
665
731
  medium: 4096,
666
732
  high: 8192,
@@ -766,11 +832,20 @@ var AnthropicDriver = class {
766
832
 
767
833
  // src/providers/google.ts
768
834
  var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
835
+ var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
836
+ var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
837
+ var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
769
838
  function needsCodeExecution(model) {
770
839
  const match = model.match(/^gemini-(\d+)/);
771
840
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
772
841
  }
842
+ function sleep(ms) {
843
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
844
+ }
773
845
  var GOOGLE_THINKING_LEVEL = {
846
+ // Gemini does define a "minimal" thinking level, but some models reject it
847
+ // (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
848
+ minimal: "low",
774
849
  low: "low",
775
850
  medium: "medium",
776
851
  high: "high",
@@ -837,24 +912,28 @@ var GoogleDriver = class {
837
912
  });
838
913
  return this.client;
839
914
  }
840
- async sendMessage(images, prompt, _options) {
841
- const client = await this.getClient();
915
+ /** Request config shared by image and video messages. */
916
+ generationConfig() {
917
+ return {
918
+ responseMimeType: "application/json",
919
+ maxOutputTokens: this.maxTokens,
920
+ ...this.reasoningEffort && {
921
+ thinkingConfig: {
922
+ thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
923
+ }
924
+ },
925
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
926
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
927
+ }
928
+ };
929
+ }
930
+ /** Runs one generateContent call and normalizes finish reasons, text, and usage. */
931
+ async generate(client, contents) {
842
932
  try {
843
933
  const response = await client.models.generateContent({
844
934
  model: this.model,
845
- contents: [...this.toGeminiParts(images), prompt],
846
- config: {
847
- responseMimeType: "application/json",
848
- maxOutputTokens: this.maxTokens,
849
- ...this.reasoningEffort && {
850
- thinkingConfig: {
851
- thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
852
- }
853
- },
854
- ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
855
- mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
856
- }
857
- }
935
+ contents,
936
+ config: this.generationConfig()
858
937
  });
859
938
  const finishReason = response.candidates?.[0]?.finishReason;
860
939
  if (finishReason === "MAX_TOKENS") {
@@ -869,9 +948,8 @@ var GoogleDriver = class {
869
948
  `Response blocked: Google returned finishReason "${finishReason}".`
870
949
  );
871
950
  }
872
- const text = response.text ?? "";
873
951
  return {
874
- text,
952
+ text: response.text ?? "",
875
953
  usage: toGeminiUsage(response.usageMetadata)
876
954
  };
877
955
  } catch (err) {
@@ -879,6 +957,80 @@ var GoogleDriver = class {
879
957
  throw mapProviderError(err);
880
958
  }
881
959
  }
960
+ async sendMessage(images, prompt, _options) {
961
+ const client = await this.getClient();
962
+ return this.generate(client, [...this.toGeminiParts(images), prompt]);
963
+ }
964
+ /**
965
+ * Uploads a video through the Files API and waits until Gemini has finished
966
+ * processing it. Returns the ACTIVE file record.
967
+ */
968
+ async uploadVideo(client, video) {
969
+ try {
970
+ const uploaded = await client.files.upload({
971
+ file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
972
+ config: { mimeType: video.mimeType }
973
+ });
974
+ const name = uploaded.name;
975
+ if (!name) {
976
+ throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
977
+ }
978
+ const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
979
+ let current = uploaded;
980
+ while (current.state === "PROCESSING") {
981
+ if (Date.now() > deadline) {
982
+ throw new VisualAIProviderError(
983
+ `Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
984
+ );
985
+ }
986
+ await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
987
+ current = await client.files.get({ name });
988
+ }
989
+ if (current.state !== "ACTIVE") {
990
+ throw new VisualAIProviderError(
991
+ `Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
992
+ );
993
+ }
994
+ if (!current.uri) {
995
+ throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
996
+ }
997
+ return current;
998
+ } catch (err) {
999
+ if (err instanceof VisualAIProviderError) throw err;
1000
+ throw mapProviderError(err);
1001
+ }
1002
+ }
1003
+ /**
1004
+ * Sends the video bytes themselves. Gemini samples the clip server-side at
1005
+ * `video.fps` and, unlike sampled frames, also hears the audio track. Small
1006
+ * videos go inline; larger ones are uploaded via the Files API and deleted
1007
+ * again afterwards (they would expire on their own after 48 h).
1008
+ */
1009
+ async sendVideoMessage(video, prompt, _options) {
1010
+ const client = await this.getClient();
1011
+ const videoMetadata = { fps: video.fps };
1012
+ if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
1013
+ const part = {
1014
+ inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
1015
+ videoMetadata
1016
+ };
1017
+ const response = await this.generate(client, [part, prompt]);
1018
+ return { ...response, delivery: "inline" };
1019
+ }
1020
+ const file = await this.uploadVideo(client, video);
1021
+ try {
1022
+ const part = {
1023
+ fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
1024
+ videoMetadata
1025
+ };
1026
+ const response = await this.generate(client, [part, prompt]);
1027
+ return { ...response, delivery: "file" };
1028
+ } finally {
1029
+ if (file.name) {
1030
+ await client.files.delete({ name: file.name }).catch(() => void 0);
1031
+ }
1032
+ }
1033
+ }
882
1034
  async generateImage(images, prompt, options) {
883
1035
  const client = await this.getClient();
884
1036
  const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
@@ -1011,6 +1163,9 @@ var OpenAIDriver = class {
1011
1163
  // src/providers/openrouter.ts
1012
1164
  var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
1013
1165
  var OPENROUTER_REASONING_EFFORT = {
1166
+ // OpenRouter normalizes upstream vendors to low/medium/high only, so
1167
+ // "minimal" has no native equivalent and clamps to the floor.
1168
+ minimal: "low",
1014
1169
  low: "low",
1015
1170
  medium: "medium",
1016
1171
  high: "high",
@@ -1170,10 +1325,20 @@ function parseBooleanEnv(envName, value) {
1170
1325
  `Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
1171
1326
  );
1172
1327
  }
1328
+ function parseReasoningEffortEnv(envName, value) {
1329
+ if (value === void 0 || value === "") return void 0;
1330
+ const levels = Object.values(ReasoningEffort);
1331
+ const lower = value.toLowerCase();
1332
+ if (levels.includes(lower)) return lower;
1333
+ throw new VisualAIConfigError(
1334
+ `Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
1335
+ );
1336
+ }
1173
1337
  var debugDeprecationWarned = false;
1174
1338
  function resolveConfig(config) {
1175
1339
  const provider = resolveProvider(config);
1176
1340
  const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
1341
+ const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
1177
1342
  const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
1178
1343
  const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
1179
1344
  const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
@@ -1191,12 +1356,12 @@ function resolveConfig(config) {
1191
1356
  }
1192
1357
  const userSetMaxTokens = config.maxTokens !== void 0;
1193
1358
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1194
- const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1359
+ const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
1195
1360
  const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1196
1361
  if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1197
1362
  maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1198
1363
  if (debug) {
1199
- const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1364
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
1200
1365
  process.stderr.write(
1201
1366
  `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1202
1367
  `
@@ -1208,7 +1373,7 @@ function resolveConfig(config) {
1208
1373
  apiKey: config.apiKey,
1209
1374
  model,
1210
1375
  maxTokens,
1211
- reasoningEffort: config.reasoningEffort,
1376
+ reasoningEffort,
1212
1377
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1213
1378
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1214
1379
  timeout: config.timeout,
@@ -1231,6 +1396,10 @@ var PRICING_TABLE = {
1231
1396
  inputPricePerToken: 10 / PER_MILLION,
1232
1397
  outputPricePerToken: 50 / PER_MILLION
1233
1398
  },
1399
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
1400
+ inputPricePerToken: 4 / PER_MILLION,
1401
+ outputPricePerToken: 20 / PER_MILLION
1402
+ },
1234
1403
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1235
1404
  inputPricePerToken: 5 / PER_MILLION,
1236
1405
  outputPricePerToken: 25 / PER_MILLION
@@ -1239,6 +1408,10 @@ var PRICING_TABLE = {
1239
1408
  inputPricePerToken: 5 / PER_MILLION,
1240
1409
  outputPricePerToken: 25 / PER_MILLION
1241
1410
  },
1411
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
1412
+ inputPricePerToken: 2 / PER_MILLION,
1413
+ outputPricePerToken: 10 / PER_MILLION
1414
+ },
1242
1415
  [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
1243
1416
  inputPricePerToken: 3 / PER_MILLION,
1244
1417
  outputPricePerToken: 15 / PER_MILLION
@@ -1265,6 +1438,22 @@ var PRICING_TABLE = {
1265
1438
  inputPricePerToken: 10 / PER_MILLION,
1266
1439
  outputPricePerToken: 50 / PER_MILLION
1267
1440
  },
1441
+ // Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
1442
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
1443
+ inputPricePerToken: 2 / PER_MILLION,
1444
+ outputPricePerToken: 10 / PER_MILLION
1445
+ },
1446
+ // Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
1447
+ // bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
1448
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
1449
+ inputPricePerToken: 2 / PER_MILLION,
1450
+ outputPricePerToken: 10 / PER_MILLION
1451
+ },
1452
+ // Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
1453
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
1454
+ inputPricePerToken: 0.1 / PER_MILLION,
1455
+ outputPricePerToken: 0.5 / PER_MILLION
1456
+ },
1268
1457
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1269
1458
  inputPricePerToken: 5 / PER_MILLION,
1270
1459
  outputPricePerToken: 30 / PER_MILLION
@@ -1361,6 +1550,10 @@ var PRICING_TABLE = {
1361
1550
  inputPricePerToken: 0.1 / PER_MILLION,
1362
1551
  outputPricePerToken: 0.2 / PER_MILLION
1363
1552
  },
1553
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
1554
+ inputPricePerToken: 1.6 / PER_MILLION,
1555
+ outputPricePerToken: 4.8 / PER_MILLION
1556
+ },
1364
1557
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1365
1558
  inputPricePerToken: 2 / PER_MILLION,
1366
1559
  outputPricePerToken: 6 / PER_MILLION
@@ -1394,6 +1587,13 @@ var PRICING_TABLE = {
1394
1587
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1395
1588
  inputPricePerToken: 0.15 / PER_MILLION,
1396
1589
  outputPricePerToken: 0.5 / PER_MILLION
1590
+ },
1591
+ // Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
1592
+ // upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
1593
+ // is $0.0036/MTok, not modelled (no provider gets a cache discount here).
1594
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
1595
+ inputPricePerToken: 0.435 / PER_MILLION,
1596
+ outputPricePerToken: 0.87 / PER_MILLION
1397
1597
  }
1398
1598
  };
1399
1599
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1471,6 +1671,15 @@ async function timedSendMessage(driver, images, prompt, options) {
1471
1671
  const durationSeconds = (performance.now() - start) / 1e3;
1472
1672
  return { ...response, durationSeconds };
1473
1673
  }
1674
+ async function timedSendVideoMessage(driver, video, prompt, options) {
1675
+ if (!driver.sendVideoMessage) {
1676
+ throw new VisualAIError("Provider driver does not support native video delivery");
1677
+ }
1678
+ const start = performance.now();
1679
+ const response = await driver.sendVideoMessage(video, prompt, options);
1680
+ const durationSeconds = (performance.now() - start) / 1e3;
1681
+ return { ...response, durationSeconds };
1682
+ }
1474
1683
 
1475
1684
  // src/core/diff.ts
1476
1685
  var import_sharp = __toESM(require("sharp"), 1);
@@ -1710,6 +1919,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
1710
1919
  };
1711
1920
  }
1712
1921
 
1922
+ // src/core/media.ts
1923
+ var import_promises4 = require("fs/promises");
1924
+
1713
1925
  // src/core/debug-frames.ts
1714
1926
  var import_node_crypto = require("crypto");
1715
1927
  var import_promises2 = require("fs/promises");
@@ -1765,6 +1977,92 @@ async function saveDebugFrames(frames, env = process.env) {
1765
1977
  return runDir;
1766
1978
  }
1767
1979
 
1980
+ // src/core/frame-dedupe.ts
1981
+ var import_sharp3 = __toESM(require("sharp"), 1);
1982
+ var DEDUPE_THUMBNAIL_EDGE = 256;
1983
+ var DEDUPE_PIXEL_TOLERANCE = 24;
1984
+ var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
1985
+ function resolveDedupeOptions(raw) {
1986
+ if (raw === void 0 || raw === true) {
1987
+ return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
1988
+ }
1989
+ if (raw === false) {
1990
+ return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
1991
+ }
1992
+ const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
1993
+ if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
1994
+ throw new VisualAIVideoError(
1995
+ `Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
1996
+ );
1997
+ }
1998
+ return { enabled: true, threshold };
1999
+ }
2000
+ async function frameSignature(frame) {
2001
+ try {
2002
+ const { data, info } = await (0, import_sharp3.default)(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
2003
+ fit: "inside",
2004
+ withoutEnlargement: true
2005
+ }).raw().toBuffer({ resolveWithObject: true });
2006
+ return { width: info.width, height: info.height, pixels: data };
2007
+ } catch (err) {
2008
+ const reason = err instanceof Error ? err.message : String(err);
2009
+ throw new VisualAIVideoError(
2010
+ `Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
2011
+ );
2012
+ }
2013
+ }
2014
+ function changedFraction(a, b) {
2015
+ if (a.width !== b.width || a.height !== b.height) {
2016
+ return 1;
2017
+ }
2018
+ let changed = 0;
2019
+ for (let i = 0; i < a.pixels.length; i++) {
2020
+ if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
2021
+ changed++;
2022
+ }
2023
+ }
2024
+ return changed / a.pixels.length;
2025
+ }
2026
+ function reindex(frame, index) {
2027
+ if (frame.index === index) {
2028
+ return frame;
2029
+ }
2030
+ return {
2031
+ data: frame.data,
2032
+ mimeType: frame.mimeType,
2033
+ get base64() {
2034
+ return frame.base64;
2035
+ },
2036
+ timestampSeconds: frame.timestampSeconds,
2037
+ index
2038
+ };
2039
+ }
2040
+ async function dedupeFrames(frames, options) {
2041
+ const { enabled, threshold } = resolveDedupeOptions(options);
2042
+ if (!enabled || frames.length < 2) {
2043
+ return { frames: [...frames], dropped: 0 };
2044
+ }
2045
+ const signed = await Promise.all(
2046
+ frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
2047
+ );
2048
+ const [first, ...rest] = signed;
2049
+ if (first === void 0) {
2050
+ return { frames: [], dropped: 0 };
2051
+ }
2052
+ const kept = [first.frame];
2053
+ let lastKept = first.signature;
2054
+ for (const { frame, signature } of rest) {
2055
+ if (changedFraction(lastKept, signature) >= threshold) {
2056
+ kept.push(frame);
2057
+ lastKept = signature;
2058
+ }
2059
+ }
2060
+ return {
2061
+ frames: kept.map((frame, index) => reindex(frame, index)),
2062
+ dropped: frames.length - kept.length
2063
+ };
2064
+ }
2065
+
1768
2066
  // src/core/video.ts
1769
2067
  var import_promises3 = require("fs/promises");
1770
2068
  var import_node_os = require("os");
@@ -2000,7 +2298,7 @@ async function probeDurationSeconds(videoPath) {
2000
2298
  });
2001
2299
  });
2002
2300
  }
2003
- async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2301
+ function resolveVideoSamplingOptions(options = {}) {
2004
2302
  const fps = options.fps ?? DEFAULT_FPS;
2005
2303
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
2006
2304
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -2020,13 +2318,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
2020
2318
  `Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
2021
2319
  );
2022
2320
  }
2023
- const ffmpeg = await loadFfmpegFactory();
2024
- const durationSeconds = await probeDurationSeconds(videoPath);
2321
+ return { fps, maxFrames, maxDurationSeconds };
2322
+ }
2323
+ function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
2025
2324
  if (durationSeconds > maxDurationSeconds) {
2026
2325
  throw new VisualAIVideoError(
2027
2326
  `Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
2028
2327
  );
2029
2328
  }
2329
+ }
2330
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2331
+ const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
2332
+ const ffmpeg = await loadFfmpegFactory();
2333
+ const durationSeconds = await probeDurationSeconds(videoPath);
2334
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2030
2335
  const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
2031
2336
  try {
2032
2337
  const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
@@ -2130,6 +2435,7 @@ function isTimestampedFrameInput(frame) {
2130
2435
  async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2131
2436
  const rawFrames = input.frames;
2132
2437
  const fps = input.fps ?? DEFAULT_FPS;
2438
+ resolveDedupeOptions(input.dedupe);
2133
2439
  if (rawFrames.length === 0) {
2134
2440
  throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
2135
2441
  }
@@ -2141,7 +2447,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2141
2447
  if (!Number.isFinite(fps) || fps <= 0) {
2142
2448
  throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
2143
2449
  }
2144
- const frames = await Promise.all(
2450
+ const sampled = await Promise.all(
2145
2451
  rawFrames.map(async (raw, index) => {
2146
2452
  const timestamped = isTimestampedFrameInput(raw);
2147
2453
  const imageInput = timestamped ? raw.image : raw;
@@ -2164,20 +2470,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2164
2470
  };
2165
2471
  })
2166
2472
  );
2167
- const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2473
+ const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2474
+ const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
2168
2475
  await saveDebugFrames(frames);
2169
- return { kind: "video", frames, durationSeconds };
2476
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2477
+ }
2478
+ async function normalizeNativeVideo(input, videoOptions) {
2479
+ const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
2480
+ const { path, mimeType, cleanup } = await resolveVideoToPath(input);
2481
+ try {
2482
+ const durationSeconds = await probeDurationSeconds(path);
2483
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2484
+ const data = await (0, import_promises4.readFile)(path);
2485
+ return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
2486
+ } finally {
2487
+ try {
2488
+ await cleanup();
2489
+ } catch {
2490
+ }
2491
+ }
2170
2492
  }
2171
- async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2493
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
2172
2494
  if (isFramesInput(input)) {
2173
2495
  return normalizeFrames(input, maxDimension);
2174
2496
  }
2175
2497
  if (isVideoInput(input)) {
2498
+ if (nativeVideo) {
2499
+ return normalizeNativeVideo(input, videoOptions);
2500
+ }
2501
+ resolveDedupeOptions(videoOptions?.dedupe);
2176
2502
  const { path, cleanup } = await resolveVideoToPath(input);
2177
2503
  try {
2178
- const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
2504
+ const { frames: sampled, durationSeconds } = await extractFrames(
2505
+ path,
2506
+ videoOptions,
2507
+ maxDimension
2508
+ );
2509
+ const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
2179
2510
  await saveDebugFrames(frames);
2180
- return { kind: "video", frames, durationSeconds };
2511
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2181
2512
  } finally {
2182
2513
  try {
2183
2514
  await cleanup();
@@ -2267,6 +2598,12 @@ var AskResultSchema = import_zod.z.object({
2267
2598
  * omitting the key, even for image inputs that were never asked to populate it.
2268
2599
  */
2269
2600
  frameReferences: import_zod.z.array(import_zod.z.number().int().nonnegative()).nullable().optional(),
2601
+ /**
2602
+ * For natively delivered video, the timestamps (seconds from the start of
2603
+ * the clip) the model relied on to answer. The native counterpart of
2604
+ * `frameReferences`. Nullable for the same strict-schema reason.
2605
+ */
2606
+ timestampReferences: import_zod.z.array(import_zod.z.number().nonnegative()).nullable().optional(),
2270
2607
  usage: UsageInfoSchema.optional()
2271
2608
  });
2272
2609
 
@@ -2278,6 +2615,12 @@ function stripCodeFences(text) {
2278
2615
  }
2279
2616
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2280
2617
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2618
+ var AskImageResponseSchema = AskResponseSchema.omit({
2619
+ frameReferences: true,
2620
+ timestampReferences: true
2621
+ });
2622
+ var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
2623
+ var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
2281
2624
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2282
2625
  var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2283
2626
  function parseJson(text) {
@@ -2345,7 +2688,8 @@ function parseAskResponse(raw) {
2345
2688
  const result = parseResponse(raw, AskResponseSchema);
2346
2689
  return {
2347
2690
  ...result,
2348
- frameReferences: result.frameReferences ?? void 0
2691
+ frameReferences: result.frameReferences ?? void 0,
2692
+ timestampReferences: result.timestampReferences ?? void 0
2349
2693
  };
2350
2694
  }
2351
2695
  function parseCompareResponse(raw) {
@@ -2369,31 +2713,75 @@ function createDriver(provider, config) {
2369
2713
  return PROVIDER_REGISTRY[provider](config);
2370
2714
  }
2371
2715
  var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
2372
- var askSchemaOptions = toSchemaOptions(AskResponseSchema);
2716
+ var askSchemaOptionsByMedia = {
2717
+ image: toSchemaOptions(AskImageResponseSchema),
2718
+ video: toSchemaOptions(AskFramesResponseSchema),
2719
+ "native-video": toSchemaOptions(AskNativeVideoResponseSchema)
2720
+ };
2373
2721
  var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2374
2722
  function mediaToProviderInputs(media) {
2375
2723
  if (media.kind === "image") {
2376
2724
  return {
2725
+ kind: "images",
2377
2726
  images: [media.image],
2378
2727
  mediaContext: { kind: "image" },
2379
- framesMetadata: void 0
2728
+ frames: void 0
2729
+ };
2730
+ }
2731
+ if (media.kind === "native-video") {
2732
+ return {
2733
+ kind: "native-video",
2734
+ video: media.video,
2735
+ mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
2380
2736
  };
2381
2737
  }
2382
2738
  const timestamps = media.frames.map((f) => f.timestampSeconds);
2383
2739
  return {
2740
+ kind: "images",
2384
2741
  images: media.frames,
2385
2742
  mediaContext: {
2386
2743
  kind: "video",
2387
2744
  frameTimestamps: timestamps,
2388
- durationSeconds: media.durationSeconds
2745
+ durationSeconds: media.durationSeconds,
2746
+ droppedUnchanged: media.droppedUnchanged
2389
2747
  },
2390
- framesMetadata: {
2748
+ frames: {
2391
2749
  count: media.frames.length,
2392
2750
  timestampsSeconds: timestamps,
2393
- durationSeconds: media.durationSeconds
2751
+ durationSeconds: media.durationSeconds,
2752
+ droppedUnchanged: media.droppedUnchanged
2394
2753
  }
2395
2754
  };
2396
2755
  }
2756
+ async function sendMedia(driver, dispatch, prompt, options) {
2757
+ if (dispatch.kind === "native-video") {
2758
+ const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
2759
+ const { durationSeconds, fps, mimeType } = dispatch.video;
2760
+ return {
2761
+ response: response2,
2762
+ metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
2763
+ };
2764
+ }
2765
+ const response = await timedSendMessage(driver, dispatch.images, prompt, options);
2766
+ return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
2767
+ }
2768
+ function resolveNativeVideo(input, videoOptions, driver, provider) {
2769
+ const mode = videoOptions?.mode ?? "auto";
2770
+ if (mode !== "auto" && mode !== "native" && mode !== "frames") {
2771
+ throw new VisualAIConfigError(
2772
+ `Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
2773
+ );
2774
+ }
2775
+ const supported = typeof driver.sendVideoMessage === "function";
2776
+ if (mode === "frames") return false;
2777
+ if (mode === "auto") return supported;
2778
+ if (!supported && !isFramesInput(input) && isVideoInput(input)) {
2779
+ throw new VisualAIConfigError(
2780
+ `Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
2781
+ );
2782
+ }
2783
+ return true;
2784
+ }
2397
2785
  function visualAI(config = {}) {
2398
2786
  const resolvedConfig = resolveConfig(config);
2399
2787
  const driverConfig = {
@@ -2431,38 +2819,60 @@ function visualAI(config = {}) {
2431
2819
  throw new VisualAIConfigError("At least one statement is required for check()");
2432
2820
  }
2433
2821
  return withErrorDebug(resolvedConfig, "check", async () => {
2434
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2435
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2822
+ const nativeVideo = resolveNativeVideo(
2823
+ input,
2824
+ options?.video,
2825
+ driver,
2826
+ resolvedConfig.provider
2827
+ );
2828
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2829
+ const dispatch = mediaToProviderInputs(media);
2436
2830
  const prompt = buildCheckPrompt(stmts, {
2437
2831
  instructions: options?.instructions,
2438
- media: mediaContext
2832
+ media: dispatch.mediaContext
2439
2833
  });
2440
2834
  debugLog(resolvedConfig, "check prompt", prompt, "prompt");
2441
- const response = await timedSendMessage(driver, images, prompt, checkSchemaOptions);
2835
+ const { response, metadata } = await sendMedia(
2836
+ driver,
2837
+ dispatch,
2838
+ prompt,
2839
+ checkSchemaOptions
2840
+ );
2442
2841
  debugLog(resolvedConfig, "check response", response.text, "response");
2443
2842
  const result = parseCheckResponse(response.text);
2444
2843
  return {
2445
2844
  ...result,
2446
- ...framesMetadata ? { frames: framesMetadata } : {},
2845
+ ...metadata,
2447
2846
  usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
2448
2847
  };
2449
2848
  });
2450
2849
  },
2451
2850
  async ask(input, userPrompt, options) {
2452
2851
  return withErrorDebug(resolvedConfig, "ask", async () => {
2453
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2454
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2852
+ const nativeVideo = resolveNativeVideo(
2853
+ input,
2854
+ options?.video,
2855
+ driver,
2856
+ resolvedConfig.provider
2857
+ );
2858
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2859
+ const dispatch = mediaToProviderInputs(media);
2455
2860
  const prompt = buildAskPrompt(userPrompt, {
2456
2861
  instructions: options?.instructions,
2457
- media: mediaContext
2862
+ media: dispatch.mediaContext
2458
2863
  });
2459
2864
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2460
- const response = await timedSendMessage(driver, images, prompt, askSchemaOptions);
2865
+ const { response, metadata } = await sendMedia(
2866
+ driver,
2867
+ dispatch,
2868
+ prompt,
2869
+ askSchemaOptionsByMedia[dispatch.mediaContext.kind]
2870
+ );
2461
2871
  debugLog(resolvedConfig, "ask response", response.text, "response");
2462
2872
  const result = parseAskResponse(response.text);
2463
2873
  return {
2464
2874
  ...result,
2465
- ...framesMetadata ? { frames: framesMetadata } : {},
2875
+ ...metadata,
2466
2876
  usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
2467
2877
  };
2468
2878
  });
@@ -2480,7 +2890,7 @@ function visualAI(config = {}) {
2480
2890
  debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
2481
2891
  const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
2482
2892
  debugLog(resolvedConfig, "compare response", response.text, "response");
2483
- const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model === Model.Google.GEMINI_3_FLASH_PREVIEW;
2893
+ const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
2484
2894
  const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
2485
2895
  let diffImage;
2486
2896
  if (effectiveDiffImage) {