visual-ai-assertions 0.23.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -327,10 +327,15 @@ Example for a failing check:
327
327
  ]
328
328
  }
329
329
  ${JSON_INSTRUCTIONS}`;
330
- var CHECK_OUTPUT_SCHEMA_VIDEO = `IMPORTANT: Follow this evaluation order:
330
+ function buildCheckOutputSchemaVideo(unit) {
331
+ const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
332
+ const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
333
+ const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
334
+ const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
335
+ return `IMPORTANT: Follow this evaluation order:
331
336
  1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
332
- 2. A statement passes if it is true at ANY frame of the timeline, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
333
- 3. For each statement that passes, set "timestampSeconds" to the timestamp of the frame that most clearly demonstrates it (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
337
+ 2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
338
+ 3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
334
339
  4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
335
340
  5. Write "reasoning" as a brief overall summary of the evaluation
336
341
  6. Include "issues" only for statements that failed
@@ -344,7 +349,7 @@ Respond with a JSON object matching this exact structure:
344
349
  {
345
350
  "statement": string, // the original statement text
346
351
  "pass": boolean, // whether this statement is true at any point in the timeline
347
- "reasoning": string, // explanation for this statement, citing frame timestamps where relevant
352
+ "reasoning": string, // explanation for this statement, ${citing} where relevant
348
353
  "confidence": "high" | "medium" | "low",
349
354
  "timestampSeconds": number | null
350
355
  // seconds from the start of the clip where the statement is most clearly true,
@@ -362,10 +367,13 @@ Example for a passing video check:
362
367
  "reasoning": "The success toast appeared briefly around 3.5s.",
363
368
  "issues": [],
364
369
  "statements": [
365
- { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right at the 3.5s frame", "confidence": "high", "timestampSeconds": 3.5 }
370
+ { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
366
371
  ]
367
372
  }
368
373
  ${JSON_INSTRUCTIONS}`;
374
+ }
375
+ var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
376
+ var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
369
377
  var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
370
378
  {
371
379
  "summary": string, // high-level analysis summary
@@ -398,6 +406,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
398
406
  Prioritize issues by severity (critical / major / minor) as for image input.
399
407
  Cite frame indices in "frameReferences" so the user can locate the moments you describe.
400
408
  ${JSON_INSTRUCTIONS}`;
409
+ var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
410
+ {
411
+ "summary": string, // high-level summary of what happens across the video
412
+ "issues": [...], // list of issues/findings, can be empty
413
+ "timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
414
+ }
415
+ ${ISSUE_SCHEMA_INSTRUCTIONS}
416
+
417
+ Prioritize issues by severity (critical / major / minor) as for image input.
418
+ Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
419
+ ${JSON_INSTRUCTIONS}`;
401
420
  var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
402
421
  {
403
422
  "pass": boolean, // true if no critical or major changes found
@@ -419,15 +438,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
419
438
  var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
420
439
  var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
421
440
  var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
422
- function buildVideoTimelineSection(frameTimestamps, durationSeconds) {
441
+ var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
442
+ var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
443
+ function buildNativeVideoSection(durationSeconds) {
444
+ return `Video recording:
445
+ - Total duration: ${durationSeconds.toFixed(2)}s
446
+
447
+ The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
448
+ }
449
+ function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
423
450
  const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
451
+ const attached = frameTimestamps.length;
452
+ const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
453
+ const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
424
454
  return `Video timeline:
425
455
  - Total duration: ${durationSeconds.toFixed(2)}s
426
- - ${frameTimestamps.length} frames sampled (in chronological order)
456
+ ${sampledLine}
427
457
  - Frame index \u2192 timestamp:
428
458
  ${formatted}
429
459
 
430
- Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.`;
460
+ Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
431
461
  }
432
462
  var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
433
463
  var COMPARE_EDGE_RULES = [
@@ -442,30 +472,52 @@ function buildCheckPrompt(statements, options) {
442
472
  const stmts = Array.isArray(statements) ? statements : [statements];
443
473
  const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
444
474
  const media = options?.media;
445
- const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
475
+ const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
446
476
  const sections = [options?.role ?? defaultRole];
447
477
  if (media?.kind === "video") {
448
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
478
+ sections.push(
479
+ buildVideoTimelineSection(
480
+ media.frameTimestamps,
481
+ media.durationSeconds,
482
+ media.droppedUnchanged
483
+ )
484
+ );
485
+ } else if (media?.kind === "native-video") {
486
+ sections.push(buildNativeVideoSection(media.durationSeconds));
449
487
  }
450
488
  if (options?.instructions && options.instructions.length > 0) {
451
489
  sections.push(buildInstructionsSection(options.instructions));
452
490
  }
453
491
  sections.push(`Statements to evaluate:
454
492
  ${statementsBlock}`);
455
- sections.push(media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE);
493
+ sections.push(
494
+ media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
495
+ );
456
496
  return sections.join("\n\n");
457
497
  }
458
498
  function buildAskPrompt(userPrompt, options) {
459
499
  const media = options?.media;
460
- const sections = [media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : DEFAULT_ASK_ROLE];
500
+ const sections = [
501
+ media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
502
+ ];
461
503
  if (media?.kind === "video") {
462
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
504
+ sections.push(
505
+ buildVideoTimelineSection(
506
+ media.frameTimestamps,
507
+ media.durationSeconds,
508
+ media.droppedUnchanged
509
+ )
510
+ );
511
+ } else if (media?.kind === "native-video") {
512
+ sections.push(buildNativeVideoSection(media.durationSeconds));
463
513
  }
464
514
  if (options?.instructions && options.instructions.length > 0) {
465
515
  sections.push(buildInstructionsSection(options.instructions));
466
516
  }
467
517
  sections.push(`User request: ${userPrompt}`);
468
- sections.push(media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE);
518
+ sections.push(
519
+ media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
520
+ );
469
521
  return sections.join("\n\n");
470
522
  }
471
523
  function buildAiDiffPrompt() {
@@ -766,10 +818,16 @@ var AnthropicDriver = class {
766
818
 
767
819
  // src/providers/google.ts
768
820
  var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
821
+ var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
822
+ var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
823
+ var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
769
824
  function needsCodeExecution(model) {
770
825
  const match = model.match(/^gemini-(\d+)/);
771
826
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
772
827
  }
828
+ function sleep(ms) {
829
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
830
+ }
773
831
  var GOOGLE_THINKING_LEVEL = {
774
832
  low: "low",
775
833
  medium: "medium",
@@ -837,24 +895,28 @@ var GoogleDriver = class {
837
895
  });
838
896
  return this.client;
839
897
  }
840
- async sendMessage(images, prompt, _options) {
841
- const client = await this.getClient();
898
+ /** Request config shared by image and video messages. */
899
+ generationConfig() {
900
+ return {
901
+ responseMimeType: "application/json",
902
+ maxOutputTokens: this.maxTokens,
903
+ ...this.reasoningEffort && {
904
+ thinkingConfig: {
905
+ thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
906
+ }
907
+ },
908
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
909
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
910
+ }
911
+ };
912
+ }
913
+ /** Runs one generateContent call and normalizes finish reasons, text, and usage. */
914
+ async generate(client, contents) {
842
915
  try {
843
916
  const response = await client.models.generateContent({
844
917
  model: this.model,
845
- contents: [...this.toGeminiParts(images), prompt],
846
- config: {
847
- responseMimeType: "application/json",
848
- maxOutputTokens: this.maxTokens,
849
- ...this.reasoningEffort && {
850
- thinkingConfig: {
851
- thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
852
- }
853
- },
854
- ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
855
- mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
856
- }
857
- }
918
+ contents,
919
+ config: this.generationConfig()
858
920
  });
859
921
  const finishReason = response.candidates?.[0]?.finishReason;
860
922
  if (finishReason === "MAX_TOKENS") {
@@ -869,9 +931,8 @@ var GoogleDriver = class {
869
931
  `Response blocked: Google returned finishReason "${finishReason}".`
870
932
  );
871
933
  }
872
- const text = response.text ?? "";
873
934
  return {
874
- text,
935
+ text: response.text ?? "",
875
936
  usage: toGeminiUsage(response.usageMetadata)
876
937
  };
877
938
  } catch (err) {
@@ -879,6 +940,80 @@ var GoogleDriver = class {
879
940
  throw mapProviderError(err);
880
941
  }
881
942
  }
943
+ async sendMessage(images, prompt, _options) {
944
+ const client = await this.getClient();
945
+ return this.generate(client, [...this.toGeminiParts(images), prompt]);
946
+ }
947
+ /**
948
+ * Uploads a video through the Files API and waits until Gemini has finished
949
+ * processing it. Returns the ACTIVE file record.
950
+ */
951
+ async uploadVideo(client, video) {
952
+ try {
953
+ const uploaded = await client.files.upload({
954
+ file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
955
+ config: { mimeType: video.mimeType }
956
+ });
957
+ const name = uploaded.name;
958
+ if (!name) {
959
+ throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
960
+ }
961
+ const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
962
+ let current = uploaded;
963
+ while (current.state === "PROCESSING") {
964
+ if (Date.now() > deadline) {
965
+ throw new VisualAIProviderError(
966
+ `Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
967
+ );
968
+ }
969
+ await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
970
+ current = await client.files.get({ name });
971
+ }
972
+ if (current.state !== "ACTIVE") {
973
+ throw new VisualAIProviderError(
974
+ `Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
975
+ );
976
+ }
977
+ if (!current.uri) {
978
+ throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
979
+ }
980
+ return current;
981
+ } catch (err) {
982
+ if (err instanceof VisualAIProviderError) throw err;
983
+ throw mapProviderError(err);
984
+ }
985
+ }
986
+ /**
987
+ * Sends the video bytes themselves. Gemini samples the clip server-side at
988
+ * `video.fps` and, unlike sampled frames, also hears the audio track. Small
989
+ * videos go inline; larger ones are uploaded via the Files API and deleted
990
+ * again afterwards (they would expire on their own after 48 h).
991
+ */
992
+ async sendVideoMessage(video, prompt, _options) {
993
+ const client = await this.getClient();
994
+ const videoMetadata = { fps: video.fps };
995
+ if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
996
+ const part = {
997
+ inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
998
+ videoMetadata
999
+ };
1000
+ const response = await this.generate(client, [part, prompt]);
1001
+ return { ...response, delivery: "inline" };
1002
+ }
1003
+ const file = await this.uploadVideo(client, video);
1004
+ try {
1005
+ const part = {
1006
+ fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
1007
+ videoMetadata
1008
+ };
1009
+ const response = await this.generate(client, [part, prompt]);
1010
+ return { ...response, delivery: "file" };
1011
+ } finally {
1012
+ if (file.name) {
1013
+ await client.files.delete({ name: file.name }).catch(() => void 0);
1014
+ }
1015
+ }
1016
+ }
882
1017
  async generateImage(images, prompt, options) {
883
1018
  const client = await this.getClient();
884
1019
  const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
@@ -1471,6 +1606,15 @@ async function timedSendMessage(driver, images, prompt, options) {
1471
1606
  const durationSeconds = (performance.now() - start) / 1e3;
1472
1607
  return { ...response, durationSeconds };
1473
1608
  }
1609
+ async function timedSendVideoMessage(driver, video, prompt, options) {
1610
+ if (!driver.sendVideoMessage) {
1611
+ throw new VisualAIError("Provider driver does not support native video delivery");
1612
+ }
1613
+ const start = performance.now();
1614
+ const response = await driver.sendVideoMessage(video, prompt, options);
1615
+ const durationSeconds = (performance.now() - start) / 1e3;
1616
+ return { ...response, durationSeconds };
1617
+ }
1474
1618
 
1475
1619
  // src/core/diff.ts
1476
1620
  var import_sharp = __toESM(require("sharp"), 1);
@@ -1710,6 +1854,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
1710
1854
  };
1711
1855
  }
1712
1856
 
1857
+ // src/core/media.ts
1858
+ var import_promises4 = require("fs/promises");
1859
+
1713
1860
  // src/core/debug-frames.ts
1714
1861
  var import_node_crypto = require("crypto");
1715
1862
  var import_promises2 = require("fs/promises");
@@ -1765,6 +1912,92 @@ async function saveDebugFrames(frames, env = process.env) {
1765
1912
  return runDir;
1766
1913
  }
1767
1914
 
1915
+ // src/core/frame-dedupe.ts
1916
+ var import_sharp3 = __toESM(require("sharp"), 1);
1917
+ var DEDUPE_THUMBNAIL_EDGE = 256;
1918
+ var DEDUPE_PIXEL_TOLERANCE = 24;
1919
+ var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
1920
+ function resolveDedupeOptions(raw) {
1921
+ if (raw === void 0 || raw === true) {
1922
+ return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
1923
+ }
1924
+ if (raw === false) {
1925
+ return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
1926
+ }
1927
+ const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
1928
+ if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
1929
+ throw new VisualAIVideoError(
1930
+ `Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
1931
+ );
1932
+ }
1933
+ return { enabled: true, threshold };
1934
+ }
1935
+ async function frameSignature(frame) {
1936
+ try {
1937
+ const { data, info } = await (0, import_sharp3.default)(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
1938
+ fit: "inside",
1939
+ withoutEnlargement: true
1940
+ }).raw().toBuffer({ resolveWithObject: true });
1941
+ return { width: info.width, height: info.height, pixels: data };
1942
+ } catch (err) {
1943
+ const reason = err instanceof Error ? err.message : String(err);
1944
+ throw new VisualAIVideoError(
1945
+ `Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
1946
+ );
1947
+ }
1948
+ }
1949
+ function changedFraction(a, b) {
1950
+ if (a.width !== b.width || a.height !== b.height) {
1951
+ return 1;
1952
+ }
1953
+ let changed = 0;
1954
+ for (let i = 0; i < a.pixels.length; i++) {
1955
+ if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
1956
+ changed++;
1957
+ }
1958
+ }
1959
+ return changed / a.pixels.length;
1960
+ }
1961
+ function reindex(frame, index) {
1962
+ if (frame.index === index) {
1963
+ return frame;
1964
+ }
1965
+ return {
1966
+ data: frame.data,
1967
+ mimeType: frame.mimeType,
1968
+ get base64() {
1969
+ return frame.base64;
1970
+ },
1971
+ timestampSeconds: frame.timestampSeconds,
1972
+ index
1973
+ };
1974
+ }
1975
+ async function dedupeFrames(frames, options) {
1976
+ const { enabled, threshold } = resolveDedupeOptions(options);
1977
+ if (!enabled || frames.length < 2) {
1978
+ return { frames: [...frames], dropped: 0 };
1979
+ }
1980
+ const signed = await Promise.all(
1981
+ frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
1982
+ );
1983
+ const [first, ...rest] = signed;
1984
+ if (first === void 0) {
1985
+ return { frames: [], dropped: 0 };
1986
+ }
1987
+ const kept = [first.frame];
1988
+ let lastKept = first.signature;
1989
+ for (const { frame, signature } of rest) {
1990
+ if (changedFraction(lastKept, signature) >= threshold) {
1991
+ kept.push(frame);
1992
+ lastKept = signature;
1993
+ }
1994
+ }
1995
+ return {
1996
+ frames: kept.map((frame, index) => reindex(frame, index)),
1997
+ dropped: frames.length - kept.length
1998
+ };
1999
+ }
2000
+
1768
2001
  // src/core/video.ts
1769
2002
  var import_promises3 = require("fs/promises");
1770
2003
  var import_node_os = require("os");
@@ -2000,7 +2233,7 @@ async function probeDurationSeconds(videoPath) {
2000
2233
  });
2001
2234
  });
2002
2235
  }
2003
- async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2236
+ function resolveVideoSamplingOptions(options = {}) {
2004
2237
  const fps = options.fps ?? DEFAULT_FPS;
2005
2238
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
2006
2239
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -2020,13 +2253,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
2020
2253
  `Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
2021
2254
  );
2022
2255
  }
2023
- const ffmpeg = await loadFfmpegFactory();
2024
- const durationSeconds = await probeDurationSeconds(videoPath);
2256
+ return { fps, maxFrames, maxDurationSeconds };
2257
+ }
2258
+ function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
2025
2259
  if (durationSeconds > maxDurationSeconds) {
2026
2260
  throw new VisualAIVideoError(
2027
2261
  `Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
2028
2262
  );
2029
2263
  }
2264
+ }
2265
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2266
+ const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
2267
+ const ffmpeg = await loadFfmpegFactory();
2268
+ const durationSeconds = await probeDurationSeconds(videoPath);
2269
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2030
2270
  const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
2031
2271
  try {
2032
2272
  const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
@@ -2130,6 +2370,7 @@ function isTimestampedFrameInput(frame) {
2130
2370
  async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2131
2371
  const rawFrames = input.frames;
2132
2372
  const fps = input.fps ?? DEFAULT_FPS;
2373
+ resolveDedupeOptions(input.dedupe);
2133
2374
  if (rawFrames.length === 0) {
2134
2375
  throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
2135
2376
  }
@@ -2141,7 +2382,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2141
2382
  if (!Number.isFinite(fps) || fps <= 0) {
2142
2383
  throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
2143
2384
  }
2144
- const frames = await Promise.all(
2385
+ const sampled = await Promise.all(
2145
2386
  rawFrames.map(async (raw, index) => {
2146
2387
  const timestamped = isTimestampedFrameInput(raw);
2147
2388
  const imageInput = timestamped ? raw.image : raw;
@@ -2164,20 +2405,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2164
2405
  };
2165
2406
  })
2166
2407
  );
2167
- const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2408
+ const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2409
+ const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
2168
2410
  await saveDebugFrames(frames);
2169
- return { kind: "video", frames, durationSeconds };
2411
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2412
+ }
2413
+ async function normalizeNativeVideo(input, videoOptions) {
2414
+ const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
2415
+ const { path, mimeType, cleanup } = await resolveVideoToPath(input);
2416
+ try {
2417
+ const durationSeconds = await probeDurationSeconds(path);
2418
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2419
+ const data = await (0, import_promises4.readFile)(path);
2420
+ return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
2421
+ } finally {
2422
+ try {
2423
+ await cleanup();
2424
+ } catch {
2425
+ }
2426
+ }
2170
2427
  }
2171
- async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2428
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
2172
2429
  if (isFramesInput(input)) {
2173
2430
  return normalizeFrames(input, maxDimension);
2174
2431
  }
2175
2432
  if (isVideoInput(input)) {
2433
+ if (nativeVideo) {
2434
+ return normalizeNativeVideo(input, videoOptions);
2435
+ }
2436
+ resolveDedupeOptions(videoOptions?.dedupe);
2176
2437
  const { path, cleanup } = await resolveVideoToPath(input);
2177
2438
  try {
2178
- const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
2439
+ const { frames: sampled, durationSeconds } = await extractFrames(
2440
+ path,
2441
+ videoOptions,
2442
+ maxDimension
2443
+ );
2444
+ const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
2179
2445
  await saveDebugFrames(frames);
2180
- return { kind: "video", frames, durationSeconds };
2446
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2181
2447
  } finally {
2182
2448
  try {
2183
2449
  await cleanup();
@@ -2267,6 +2533,12 @@ var AskResultSchema = import_zod.z.object({
2267
2533
  * omitting the key, even for image inputs that were never asked to populate it.
2268
2534
  */
2269
2535
  frameReferences: import_zod.z.array(import_zod.z.number().int().nonnegative()).nullable().optional(),
2536
+ /**
2537
+ * For natively delivered video, the timestamps (seconds from the start of
2538
+ * the clip) the model relied on to answer. The native counterpart of
2539
+ * `frameReferences`. Nullable for the same strict-schema reason.
2540
+ */
2541
+ timestampReferences: import_zod.z.array(import_zod.z.number().nonnegative()).nullable().optional(),
2270
2542
  usage: UsageInfoSchema.optional()
2271
2543
  });
2272
2544
 
@@ -2345,7 +2617,8 @@ function parseAskResponse(raw) {
2345
2617
  const result = parseResponse(raw, AskResponseSchema);
2346
2618
  return {
2347
2619
  ...result,
2348
- frameReferences: result.frameReferences ?? void 0
2620
+ frameReferences: result.frameReferences ?? void 0,
2621
+ timestampReferences: result.timestampReferences ?? void 0
2349
2622
  };
2350
2623
  }
2351
2624
  function parseCompareResponse(raw) {
@@ -2374,26 +2647,66 @@ var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2374
2647
  function mediaToProviderInputs(media) {
2375
2648
  if (media.kind === "image") {
2376
2649
  return {
2650
+ kind: "images",
2377
2651
  images: [media.image],
2378
2652
  mediaContext: { kind: "image" },
2379
- framesMetadata: void 0
2653
+ frames: void 0
2654
+ };
2655
+ }
2656
+ if (media.kind === "native-video") {
2657
+ return {
2658
+ kind: "native-video",
2659
+ video: media.video,
2660
+ mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
2380
2661
  };
2381
2662
  }
2382
2663
  const timestamps = media.frames.map((f) => f.timestampSeconds);
2383
2664
  return {
2665
+ kind: "images",
2384
2666
  images: media.frames,
2385
2667
  mediaContext: {
2386
2668
  kind: "video",
2387
2669
  frameTimestamps: timestamps,
2388
- durationSeconds: media.durationSeconds
2670
+ durationSeconds: media.durationSeconds,
2671
+ droppedUnchanged: media.droppedUnchanged
2389
2672
  },
2390
- framesMetadata: {
2673
+ frames: {
2391
2674
  count: media.frames.length,
2392
2675
  timestampsSeconds: timestamps,
2393
- durationSeconds: media.durationSeconds
2676
+ durationSeconds: media.durationSeconds,
2677
+ droppedUnchanged: media.droppedUnchanged
2394
2678
  }
2395
2679
  };
2396
2680
  }
2681
+ async function sendMedia(driver, dispatch, prompt, options) {
2682
+ if (dispatch.kind === "native-video") {
2683
+ const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
2684
+ const { durationSeconds, fps, mimeType } = dispatch.video;
2685
+ return {
2686
+ response: response2,
2687
+ metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
2688
+ };
2689
+ }
2690
+ const response = await timedSendMessage(driver, dispatch.images, prompt, options);
2691
+ return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
2692
+ }
2693
+ function resolveNativeVideo(input, videoOptions, driver, provider) {
2694
+ const mode = videoOptions?.mode ?? "auto";
2695
+ if (mode !== "auto" && mode !== "native" && mode !== "frames") {
2696
+ throw new VisualAIConfigError(
2697
+ `Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
2698
+ );
2699
+ }
2700
+ const supported = typeof driver.sendVideoMessage === "function";
2701
+ if (mode === "frames") return false;
2702
+ if (mode === "auto") return supported;
2703
+ if (!supported && !isFramesInput(input) && isVideoInput(input)) {
2704
+ throw new VisualAIConfigError(
2705
+ `Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
2706
+ );
2707
+ }
2708
+ return true;
2709
+ }
2397
2710
  function visualAI(config = {}) {
2398
2711
  const resolvedConfig = resolveConfig(config);
2399
2712
  const driverConfig = {
@@ -2431,38 +2744,55 @@ function visualAI(config = {}) {
2431
2744
  throw new VisualAIConfigError("At least one statement is required for check()");
2432
2745
  }
2433
2746
  return withErrorDebug(resolvedConfig, "check", async () => {
2434
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2435
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2747
+ const nativeVideo = resolveNativeVideo(
2748
+ input,
2749
+ options?.video,
2750
+ driver,
2751
+ resolvedConfig.provider
2752
+ );
2753
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2754
+ const dispatch = mediaToProviderInputs(media);
2436
2755
  const prompt = buildCheckPrompt(stmts, {
2437
2756
  instructions: options?.instructions,
2438
- media: mediaContext
2757
+ media: dispatch.mediaContext
2439
2758
  });
2440
2759
  debugLog(resolvedConfig, "check prompt", prompt, "prompt");
2441
- const response = await timedSendMessage(driver, images, prompt, checkSchemaOptions);
2760
+ const { response, metadata } = await sendMedia(
2761
+ driver,
2762
+ dispatch,
2763
+ prompt,
2764
+ checkSchemaOptions
2765
+ );
2442
2766
  debugLog(resolvedConfig, "check response", response.text, "response");
2443
2767
  const result = parseCheckResponse(response.text);
2444
2768
  return {
2445
2769
  ...result,
2446
- ...framesMetadata ? { frames: framesMetadata } : {},
2770
+ ...metadata,
2447
2771
  usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
2448
2772
  };
2449
2773
  });
2450
2774
  },
2451
2775
  async ask(input, userPrompt, options) {
2452
2776
  return withErrorDebug(resolvedConfig, "ask", async () => {
2453
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2454
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2777
+ const nativeVideo = resolveNativeVideo(
2778
+ input,
2779
+ options?.video,
2780
+ driver,
2781
+ resolvedConfig.provider
2782
+ );
2783
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2784
+ const dispatch = mediaToProviderInputs(media);
2455
2785
  const prompt = buildAskPrompt(userPrompt, {
2456
2786
  instructions: options?.instructions,
2457
- media: mediaContext
2787
+ media: dispatch.mediaContext
2458
2788
  });
2459
2789
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2460
- const response = await timedSendMessage(driver, images, prompt, askSchemaOptions);
2790
+ const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2461
2791
  debugLog(resolvedConfig, "ask response", response.text, "response");
2462
2792
  const result = parseAskResponse(response.text);
2463
2793
  return {
2464
2794
  ...result,
2465
- ...framesMetadata ? { frames: framesMetadata } : {},
2795
+ ...metadata,
2466
2796
  usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
2467
2797
  };
2468
2798
  });