visual-ai-assertions 0.23.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -258,10 +258,15 @@ Example for a failing check:
258
258
  ]
259
259
  }
260
260
  ${JSON_INSTRUCTIONS}`;
261
- var CHECK_OUTPUT_SCHEMA_VIDEO = `IMPORTANT: Follow this evaluation order:
261
+ function buildCheckOutputSchemaVideo(unit) {
262
+ const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
263
+ const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
264
+ const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
265
+ const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
266
+ return `IMPORTANT: Follow this evaluation order:
262
267
  1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
263
- 2. A statement passes if it is true at ANY frame of the timeline, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
264
- 3. For each statement that passes, set "timestampSeconds" to the timestamp of the frame that most clearly demonstrates it (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
268
+ 2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
269
+ 3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
265
270
  4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
266
271
  5. Write "reasoning" as a brief overall summary of the evaluation
267
272
  6. Include "issues" only for statements that failed
@@ -275,7 +280,7 @@ Respond with a JSON object matching this exact structure:
275
280
  {
276
281
  "statement": string, // the original statement text
277
282
  "pass": boolean, // whether this statement is true at any point in the timeline
278
- "reasoning": string, // explanation for this statement, citing frame timestamps where relevant
283
+ "reasoning": string, // explanation for this statement, ${citing} where relevant
279
284
  "confidence": "high" | "medium" | "low",
280
285
  "timestampSeconds": number | null
281
286
  // seconds from the start of the clip where the statement is most clearly true,
@@ -293,10 +298,13 @@ Example for a passing video check:
293
298
  "reasoning": "The success toast appeared briefly around 3.5s.",
294
299
  "issues": [],
295
300
  "statements": [
296
- { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right at the 3.5s frame", "confidence": "high", "timestampSeconds": 3.5 }
301
+ { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
297
302
  ]
298
303
  }
299
304
  ${JSON_INSTRUCTIONS}`;
305
+ }
306
+ var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
307
+ var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
300
308
  var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
301
309
  {
302
310
  "summary": string, // high-level analysis summary
@@ -329,6 +337,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
329
337
  Prioritize issues by severity (critical / major / minor) as for image input.
330
338
  Cite frame indices in "frameReferences" so the user can locate the moments you describe.
331
339
  ${JSON_INSTRUCTIONS}`;
340
+ var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
341
+ {
342
+ "summary": string, // high-level summary of what happens across the video
343
+ "issues": [...], // list of issues/findings, can be empty
344
+ "timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
345
+ }
346
+ ${ISSUE_SCHEMA_INSTRUCTIONS}
347
+
348
+ Prioritize issues by severity (critical / major / minor) as for image input.
349
+ Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
350
+ ${JSON_INSTRUCTIONS}`;
332
351
  var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
333
352
  {
334
353
  "pass": boolean, // true if no critical or major changes found
@@ -350,15 +369,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
350
369
  var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
351
370
  var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
352
371
  var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
353
- function buildVideoTimelineSection(frameTimestamps, durationSeconds) {
372
+ var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
373
+ var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
374
+ function buildNativeVideoSection(durationSeconds) {
375
+ return `Video recording:
376
+ - Total duration: ${durationSeconds.toFixed(2)}s
377
+
378
+ The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
379
+ }
380
+ function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
354
381
  const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
382
+ const attached = frameTimestamps.length;
383
+ const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
384
+ const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
355
385
  return `Video timeline:
356
386
  - Total duration: ${durationSeconds.toFixed(2)}s
357
- - ${frameTimestamps.length} frames sampled (in chronological order)
387
+ ${sampledLine}
358
388
  - Frame index \u2192 timestamp:
359
389
  ${formatted}
360
390
 
361
- Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.`;
391
+ Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
362
392
  }
363
393
  var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
364
394
  var COMPARE_EDGE_RULES = [
@@ -373,30 +403,52 @@ function buildCheckPrompt(statements, options) {
373
403
  const stmts = Array.isArray(statements) ? statements : [statements];
374
404
  const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
375
405
  const media = options?.media;
376
- const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
406
+ const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
377
407
  const sections = [options?.role ?? defaultRole];
378
408
  if (media?.kind === "video") {
379
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
409
+ sections.push(
410
+ buildVideoTimelineSection(
411
+ media.frameTimestamps,
412
+ media.durationSeconds,
413
+ media.droppedUnchanged
414
+ )
415
+ );
416
+ } else if (media?.kind === "native-video") {
417
+ sections.push(buildNativeVideoSection(media.durationSeconds));
380
418
  }
381
419
  if (options?.instructions && options.instructions.length > 0) {
382
420
  sections.push(buildInstructionsSection(options.instructions));
383
421
  }
384
422
  sections.push(`Statements to evaluate:
385
423
  ${statementsBlock}`);
386
- sections.push(media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE);
424
+ sections.push(
425
+ media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
426
+ );
387
427
  return sections.join("\n\n");
388
428
  }
389
429
  function buildAskPrompt(userPrompt, options) {
390
430
  const media = options?.media;
391
- const sections = [media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : DEFAULT_ASK_ROLE];
431
+ const sections = [
432
+ media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
433
+ ];
392
434
  if (media?.kind === "video") {
393
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
435
+ sections.push(
436
+ buildVideoTimelineSection(
437
+ media.frameTimestamps,
438
+ media.durationSeconds,
439
+ media.droppedUnchanged
440
+ )
441
+ );
442
+ } else if (media?.kind === "native-video") {
443
+ sections.push(buildNativeVideoSection(media.durationSeconds));
394
444
  }
395
445
  if (options?.instructions && options.instructions.length > 0) {
396
446
  sections.push(buildInstructionsSection(options.instructions));
397
447
  }
398
448
  sections.push(`User request: ${userPrompt}`);
399
- sections.push(media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE);
449
+ sections.push(
450
+ media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
451
+ );
400
452
  return sections.join("\n\n");
401
453
  }
402
454
  function buildAiDiffPrompt() {
@@ -697,10 +749,16 @@ var AnthropicDriver = class {
697
749
 
698
750
  // src/providers/google.ts
699
751
  var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
752
+ var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
753
+ var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
754
+ var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
700
755
  function needsCodeExecution(model) {
701
756
  const match = model.match(/^gemini-(\d+)/);
702
757
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
703
758
  }
759
+ function sleep(ms) {
760
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
761
+ }
704
762
  var GOOGLE_THINKING_LEVEL = {
705
763
  low: "low",
706
764
  medium: "medium",
@@ -768,24 +826,28 @@ var GoogleDriver = class {
768
826
  });
769
827
  return this.client;
770
828
  }
771
- async sendMessage(images, prompt, _options) {
772
- const client = await this.getClient();
829
+ /** Request config shared by image and video messages. */
830
+ generationConfig() {
831
+ return {
832
+ responseMimeType: "application/json",
833
+ maxOutputTokens: this.maxTokens,
834
+ ...this.reasoningEffort && {
835
+ thinkingConfig: {
836
+ thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
837
+ }
838
+ },
839
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
840
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
841
+ }
842
+ };
843
+ }
844
+ /** Runs one generateContent call and normalizes finish reasons, text, and usage. */
845
+ async generate(client, contents) {
773
846
  try {
774
847
  const response = await client.models.generateContent({
775
848
  model: this.model,
776
- contents: [...this.toGeminiParts(images), prompt],
777
- config: {
778
- responseMimeType: "application/json",
779
- maxOutputTokens: this.maxTokens,
780
- ...this.reasoningEffort && {
781
- thinkingConfig: {
782
- thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
783
- }
784
- },
785
- ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
786
- mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
787
- }
788
- }
849
+ contents,
850
+ config: this.generationConfig()
789
851
  });
790
852
  const finishReason = response.candidates?.[0]?.finishReason;
791
853
  if (finishReason === "MAX_TOKENS") {
@@ -800,9 +862,8 @@ var GoogleDriver = class {
800
862
  `Response blocked: Google returned finishReason "${finishReason}".`
801
863
  );
802
864
  }
803
- const text = response.text ?? "";
804
865
  return {
805
- text,
866
+ text: response.text ?? "",
806
867
  usage: toGeminiUsage(response.usageMetadata)
807
868
  };
808
869
  } catch (err) {
@@ -810,6 +871,80 @@ var GoogleDriver = class {
810
871
  throw mapProviderError(err);
811
872
  }
812
873
  }
874
+ async sendMessage(images, prompt, _options) {
875
+ const client = await this.getClient();
876
+ return this.generate(client, [...this.toGeminiParts(images), prompt]);
877
+ }
878
+ /**
879
+ * Uploads a video through the Files API and waits until Gemini has finished
880
+ * processing it. Returns the ACTIVE file record.
881
+ */
882
+ async uploadVideo(client, video) {
883
+ try {
884
+ const uploaded = await client.files.upload({
885
+ file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
886
+ config: { mimeType: video.mimeType }
887
+ });
888
+ const name = uploaded.name;
889
+ if (!name) {
890
+ throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
891
+ }
892
+ const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
893
+ let current = uploaded;
894
+ while (current.state === "PROCESSING") {
895
+ if (Date.now() > deadline) {
896
+ throw new VisualAIProviderError(
897
+ `Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
898
+ );
899
+ }
900
+ await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
901
+ current = await client.files.get({ name });
902
+ }
903
+ if (current.state !== "ACTIVE") {
904
+ throw new VisualAIProviderError(
905
+ `Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
906
+ );
907
+ }
908
+ if (!current.uri) {
909
+ throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
910
+ }
911
+ return current;
912
+ } catch (err) {
913
+ if (err instanceof VisualAIProviderError) throw err;
914
+ throw mapProviderError(err);
915
+ }
916
+ }
917
+ /**
918
+ * Sends the video bytes themselves. Gemini samples the clip server-side at
919
+ * `video.fps` and, unlike sampled frames, also hears the audio track. Small
920
+ * videos go inline; larger ones are uploaded via the Files API and deleted
921
+ * again afterwards (they would expire on their own after 48 h).
922
+ */
923
+ async sendVideoMessage(video, prompt, _options) {
924
+ const client = await this.getClient();
925
+ const videoMetadata = { fps: video.fps };
926
+ if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
927
+ const part = {
928
+ inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
929
+ videoMetadata
930
+ };
931
+ const response = await this.generate(client, [part, prompt]);
932
+ return { ...response, delivery: "inline" };
933
+ }
934
+ const file = await this.uploadVideo(client, video);
935
+ try {
936
+ const part = {
937
+ fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
938
+ videoMetadata
939
+ };
940
+ const response = await this.generate(client, [part, prompt]);
941
+ return { ...response, delivery: "file" };
942
+ } finally {
943
+ if (file.name) {
944
+ await client.files.delete({ name: file.name }).catch(() => void 0);
945
+ }
946
+ }
947
+ }
813
948
  async generateImage(images, prompt, options) {
814
949
  const client = await this.getClient();
815
950
  const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
@@ -1402,6 +1537,15 @@ async function timedSendMessage(driver, images, prompt, options) {
1402
1537
  const durationSeconds = (performance.now() - start) / 1e3;
1403
1538
  return { ...response, durationSeconds };
1404
1539
  }
1540
+ async function timedSendVideoMessage(driver, video, prompt, options) {
1541
+ if (!driver.sendVideoMessage) {
1542
+ throw new VisualAIError("Provider driver does not support native video delivery");
1543
+ }
1544
+ const start = performance.now();
1545
+ const response = await driver.sendVideoMessage(video, prompt, options);
1546
+ const durationSeconds = (performance.now() - start) / 1e3;
1547
+ return { ...response, durationSeconds };
1548
+ }
1405
1549
 
1406
1550
  // src/core/diff.ts
1407
1551
  import sharp from "sharp";
@@ -1641,6 +1785,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
1641
1785
  };
1642
1786
  }
1643
1787
 
1788
+ // src/core/media.ts
1789
+ import { readFile as readFile3 } from "fs/promises";
1790
+
1644
1791
  // src/core/debug-frames.ts
1645
1792
  import { randomBytes } from "crypto";
1646
1793
  import { mkdir, writeFile } from "fs/promises";
@@ -1696,6 +1843,92 @@ async function saveDebugFrames(frames, env = process.env) {
1696
1843
  return runDir;
1697
1844
  }
1698
1845
 
1846
+ // src/core/frame-dedupe.ts
1847
+ import sharp3 from "sharp";
1848
+ var DEDUPE_THUMBNAIL_EDGE = 256;
1849
+ var DEDUPE_PIXEL_TOLERANCE = 24;
1850
+ var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
1851
+ function resolveDedupeOptions(raw) {
1852
+ if (raw === void 0 || raw === true) {
1853
+ return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
1854
+ }
1855
+ if (raw === false) {
1856
+ return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
1857
+ }
1858
+ const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
1859
+ if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
1860
+ throw new VisualAIVideoError(
1861
+ `Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
1862
+ );
1863
+ }
1864
+ return { enabled: true, threshold };
1865
+ }
1866
+ async function frameSignature(frame) {
1867
+ try {
1868
+ const { data, info } = await sharp3(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
1869
+ fit: "inside",
1870
+ withoutEnlargement: true
1871
+ }).raw().toBuffer({ resolveWithObject: true });
1872
+ return { width: info.width, height: info.height, pixels: data };
1873
+ } catch (err) {
1874
+ const reason = err instanceof Error ? err.message : String(err);
1875
+ throw new VisualAIVideoError(
1876
+ `Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
1877
+ );
1878
+ }
1879
+ }
1880
+ function changedFraction(a, b) {
1881
+ if (a.width !== b.width || a.height !== b.height) {
1882
+ return 1;
1883
+ }
1884
+ let changed = 0;
1885
+ for (let i = 0; i < a.pixels.length; i++) {
1886
+ if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
1887
+ changed++;
1888
+ }
1889
+ }
1890
+ return changed / a.pixels.length;
1891
+ }
1892
+ function reindex(frame, index) {
1893
+ if (frame.index === index) {
1894
+ return frame;
1895
+ }
1896
+ return {
1897
+ data: frame.data,
1898
+ mimeType: frame.mimeType,
1899
+ get base64() {
1900
+ return frame.base64;
1901
+ },
1902
+ timestampSeconds: frame.timestampSeconds,
1903
+ index
1904
+ };
1905
+ }
1906
+ async function dedupeFrames(frames, options) {
1907
+ const { enabled, threshold } = resolveDedupeOptions(options);
1908
+ if (!enabled || frames.length < 2) {
1909
+ return { frames: [...frames], dropped: 0 };
1910
+ }
1911
+ const signed = await Promise.all(
1912
+ frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
1913
+ );
1914
+ const [first, ...rest] = signed;
1915
+ if (first === void 0) {
1916
+ return { frames: [], dropped: 0 };
1917
+ }
1918
+ const kept = [first.frame];
1919
+ let lastKept = first.signature;
1920
+ for (const { frame, signature } of rest) {
1921
+ if (changedFraction(lastKept, signature) >= threshold) {
1922
+ kept.push(frame);
1923
+ lastKept = signature;
1924
+ }
1925
+ }
1926
+ return {
1927
+ frames: kept.map((frame, index) => reindex(frame, index)),
1928
+ dropped: frames.length - kept.length
1929
+ };
1930
+ }
1931
+
1699
1932
  // src/core/video.ts
1700
1933
  import { mkdtemp, readFile as readFile2, readdir, rm, writeFile as writeFile2 } from "fs/promises";
1701
1934
  import { tmpdir } from "os";
@@ -1931,7 +2164,7 @@ async function probeDurationSeconds(videoPath) {
1931
2164
  });
1932
2165
  });
1933
2166
  }
1934
- async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2167
+ function resolveVideoSamplingOptions(options = {}) {
1935
2168
  const fps = options.fps ?? DEFAULT_FPS;
1936
2169
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1937
2170
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1951,13 +2184,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
1951
2184
  `Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
1952
2185
  );
1953
2186
  }
1954
- const ffmpeg = await loadFfmpegFactory();
1955
- const durationSeconds = await probeDurationSeconds(videoPath);
2187
+ return { fps, maxFrames, maxDurationSeconds };
2188
+ }
2189
+ function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
1956
2190
  if (durationSeconds > maxDurationSeconds) {
1957
2191
  throw new VisualAIVideoError(
1958
2192
  `Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
1959
2193
  );
1960
2194
  }
2195
+ }
2196
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2197
+ const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
2198
+ const ffmpeg = await loadFfmpegFactory();
2199
+ const durationSeconds = await probeDurationSeconds(videoPath);
2200
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
1961
2201
  const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
1962
2202
  try {
1963
2203
  const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
@@ -2061,6 +2301,7 @@ function isTimestampedFrameInput(frame) {
2061
2301
  async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2062
2302
  const rawFrames = input.frames;
2063
2303
  const fps = input.fps ?? DEFAULT_FPS;
2304
+ resolveDedupeOptions(input.dedupe);
2064
2305
  if (rawFrames.length === 0) {
2065
2306
  throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
2066
2307
  }
@@ -2072,7 +2313,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2072
2313
  if (!Number.isFinite(fps) || fps <= 0) {
2073
2314
  throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
2074
2315
  }
2075
- const frames = await Promise.all(
2316
+ const sampled = await Promise.all(
2076
2317
  rawFrames.map(async (raw, index) => {
2077
2318
  const timestamped = isTimestampedFrameInput(raw);
2078
2319
  const imageInput = timestamped ? raw.image : raw;
@@ -2095,20 +2336,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2095
2336
  };
2096
2337
  })
2097
2338
  );
2098
- const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2339
+ const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2340
+ const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
2099
2341
  await saveDebugFrames(frames);
2100
- return { kind: "video", frames, durationSeconds };
2342
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2343
+ }
2344
+ async function normalizeNativeVideo(input, videoOptions) {
2345
+ const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
2346
+ const { path, mimeType, cleanup } = await resolveVideoToPath(input);
2347
+ try {
2348
+ const durationSeconds = await probeDurationSeconds(path);
2349
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2350
+ const data = await readFile3(path);
2351
+ return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
2352
+ } finally {
2353
+ try {
2354
+ await cleanup();
2355
+ } catch {
2356
+ }
2357
+ }
2101
2358
  }
2102
- async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2359
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
2103
2360
  if (isFramesInput(input)) {
2104
2361
  return normalizeFrames(input, maxDimension);
2105
2362
  }
2106
2363
  if (isVideoInput(input)) {
2364
+ if (nativeVideo) {
2365
+ return normalizeNativeVideo(input, videoOptions);
2366
+ }
2367
+ resolveDedupeOptions(videoOptions?.dedupe);
2107
2368
  const { path, cleanup } = await resolveVideoToPath(input);
2108
2369
  try {
2109
- const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
2370
+ const { frames: sampled, durationSeconds } = await extractFrames(
2371
+ path,
2372
+ videoOptions,
2373
+ maxDimension
2374
+ );
2375
+ const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
2110
2376
  await saveDebugFrames(frames);
2111
- return { kind: "video", frames, durationSeconds };
2377
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2112
2378
  } finally {
2113
2379
  try {
2114
2380
  await cleanup();
@@ -2198,6 +2464,12 @@ var AskResultSchema = z.object({
2198
2464
  * omitting the key, even for image inputs that were never asked to populate it.
2199
2465
  */
2200
2466
  frameReferences: z.array(z.number().int().nonnegative()).nullable().optional(),
2467
+ /**
2468
+ * For natively delivered video, the timestamps (seconds from the start of
2469
+ * the clip) the model relied on to answer. The native counterpart of
2470
+ * `frameReferences`. Nullable for the same strict-schema reason.
2471
+ */
2472
+ timestampReferences: z.array(z.number().nonnegative()).nullable().optional(),
2201
2473
  usage: UsageInfoSchema.optional()
2202
2474
  });
2203
2475
 
@@ -2276,7 +2548,8 @@ function parseAskResponse(raw) {
2276
2548
  const result = parseResponse(raw, AskResponseSchema);
2277
2549
  return {
2278
2550
  ...result,
2279
- frameReferences: result.frameReferences ?? void 0
2551
+ frameReferences: result.frameReferences ?? void 0,
2552
+ timestampReferences: result.timestampReferences ?? void 0
2280
2553
  };
2281
2554
  }
2282
2555
  function parseCompareResponse(raw) {
@@ -2305,26 +2578,66 @@ var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2305
2578
  function mediaToProviderInputs(media) {
2306
2579
  if (media.kind === "image") {
2307
2580
  return {
2581
+ kind: "images",
2308
2582
  images: [media.image],
2309
2583
  mediaContext: { kind: "image" },
2310
- framesMetadata: void 0
2584
+ frames: void 0
2585
+ };
2586
+ }
2587
+ if (media.kind === "native-video") {
2588
+ return {
2589
+ kind: "native-video",
2590
+ video: media.video,
2591
+ mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
2311
2592
  };
2312
2593
  }
2313
2594
  const timestamps = media.frames.map((f) => f.timestampSeconds);
2314
2595
  return {
2596
+ kind: "images",
2315
2597
  images: media.frames,
2316
2598
  mediaContext: {
2317
2599
  kind: "video",
2318
2600
  frameTimestamps: timestamps,
2319
- durationSeconds: media.durationSeconds
2601
+ durationSeconds: media.durationSeconds,
2602
+ droppedUnchanged: media.droppedUnchanged
2320
2603
  },
2321
- framesMetadata: {
2604
+ frames: {
2322
2605
  count: media.frames.length,
2323
2606
  timestampsSeconds: timestamps,
2324
- durationSeconds: media.durationSeconds
2607
+ durationSeconds: media.durationSeconds,
2608
+ droppedUnchanged: media.droppedUnchanged
2325
2609
  }
2326
2610
  };
2327
2611
  }
2612
+ async function sendMedia(driver, dispatch, prompt, options) {
2613
+ if (dispatch.kind === "native-video") {
2614
+ const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
2615
+ const { durationSeconds, fps, mimeType } = dispatch.video;
2616
+ return {
2617
+ response: response2,
2618
+ metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
2619
+ };
2620
+ }
2621
+ const response = await timedSendMessage(driver, dispatch.images, prompt, options);
2622
+ return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
2623
+ }
2624
+ function resolveNativeVideo(input, videoOptions, driver, provider) {
2625
+ const mode = videoOptions?.mode ?? "auto";
2626
+ if (mode !== "auto" && mode !== "native" && mode !== "frames") {
2627
+ throw new VisualAIConfigError(
2628
+ `Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
2629
+ );
2630
+ }
2631
+ const supported = typeof driver.sendVideoMessage === "function";
2632
+ if (mode === "frames") return false;
2633
+ if (mode === "auto") return supported;
2634
+ if (!supported && !isFramesInput(input) && isVideoInput(input)) {
2635
+ throw new VisualAIConfigError(
2636
+ `Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
2637
+ );
2638
+ }
2639
+ return true;
2640
+ }
2328
2641
  function visualAI(config = {}) {
2329
2642
  const resolvedConfig = resolveConfig(config);
2330
2643
  const driverConfig = {
@@ -2362,38 +2675,55 @@ function visualAI(config = {}) {
2362
2675
  throw new VisualAIConfigError("At least one statement is required for check()");
2363
2676
  }
2364
2677
  return withErrorDebug(resolvedConfig, "check", async () => {
2365
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2366
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2678
+ const nativeVideo = resolveNativeVideo(
2679
+ input,
2680
+ options?.video,
2681
+ driver,
2682
+ resolvedConfig.provider
2683
+ );
2684
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2685
+ const dispatch = mediaToProviderInputs(media);
2367
2686
  const prompt = buildCheckPrompt(stmts, {
2368
2687
  instructions: options?.instructions,
2369
- media: mediaContext
2688
+ media: dispatch.mediaContext
2370
2689
  });
2371
2690
  debugLog(resolvedConfig, "check prompt", prompt, "prompt");
2372
- const response = await timedSendMessage(driver, images, prompt, checkSchemaOptions);
2691
+ const { response, metadata } = await sendMedia(
2692
+ driver,
2693
+ dispatch,
2694
+ prompt,
2695
+ checkSchemaOptions
2696
+ );
2373
2697
  debugLog(resolvedConfig, "check response", response.text, "response");
2374
2698
  const result = parseCheckResponse(response.text);
2375
2699
  return {
2376
2700
  ...result,
2377
- ...framesMetadata ? { frames: framesMetadata } : {},
2701
+ ...metadata,
2378
2702
  usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
2379
2703
  };
2380
2704
  });
2381
2705
  },
2382
2706
  async ask(input, userPrompt, options) {
2383
2707
  return withErrorDebug(resolvedConfig, "ask", async () => {
2384
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2385
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2708
+ const nativeVideo = resolveNativeVideo(
2709
+ input,
2710
+ options?.video,
2711
+ driver,
2712
+ resolvedConfig.provider
2713
+ );
2714
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2715
+ const dispatch = mediaToProviderInputs(media);
2386
2716
  const prompt = buildAskPrompt(userPrompt, {
2387
2717
  instructions: options?.instructions,
2388
- media: mediaContext
2718
+ media: dispatch.mediaContext
2389
2719
  });
2390
2720
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2391
- const response = await timedSendMessage(driver, images, prompt, askSchemaOptions);
2721
+ const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2392
2722
  debugLog(resolvedConfig, "ask response", response.text, "response");
2393
2723
  const result = parseAskResponse(response.text);
2394
2724
  return {
2395
2725
  ...result,
2396
- ...framesMetadata ? { frames: framesMetadata } : {},
2726
+ ...metadata,
2397
2727
  usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
2398
2728
  };
2399
2729
  });