visual-ai-assertions 0.22.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -135,7 +135,8 @@ var Model = {
135
135
  KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
136
136
  QWEN_3_8_MAX: "qwen/qwen3.8-max",
137
137
  QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
138
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
138
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
139
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
139
140
  }
140
141
  };
141
142
  var DEFAULT_MODELS = {
@@ -326,10 +327,15 @@ Example for a failing check:
326
327
  ]
327
328
  }
328
329
  ${JSON_INSTRUCTIONS}`;
329
- var CHECK_OUTPUT_SCHEMA_VIDEO = `IMPORTANT: Follow this evaluation order:
330
+ function buildCheckOutputSchemaVideo(unit) {
331
+ const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
332
+ const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
333
+ const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
334
+ const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
335
+ return `IMPORTANT: Follow this evaluation order:
330
336
  1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
331
- 2. A statement passes if it is true at ANY frame of the timeline, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
332
- 3. For each statement that passes, set "timestampSeconds" to the timestamp of the frame that most clearly demonstrates it (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
337
+ 2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
338
+ 3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
333
339
  4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
334
340
  5. Write "reasoning" as a brief overall summary of the evaluation
335
341
  6. Include "issues" only for statements that failed
@@ -343,7 +349,7 @@ Respond with a JSON object matching this exact structure:
343
349
  {
344
350
  "statement": string, // the original statement text
345
351
  "pass": boolean, // whether this statement is true at any point in the timeline
346
- "reasoning": string, // explanation for this statement, citing frame timestamps where relevant
352
+ "reasoning": string, // explanation for this statement, ${citing} where relevant
347
353
  "confidence": "high" | "medium" | "low",
348
354
  "timestampSeconds": number | null
349
355
  // seconds from the start of the clip where the statement is most clearly true,
@@ -361,10 +367,13 @@ Example for a passing video check:
361
367
  "reasoning": "The success toast appeared briefly around 3.5s.",
362
368
  "issues": [],
363
369
  "statements": [
364
- { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right at the 3.5s frame", "confidence": "high", "timestampSeconds": 3.5 }
370
+ { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
365
371
  ]
366
372
  }
367
373
  ${JSON_INSTRUCTIONS}`;
374
+ }
375
+ var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
376
+ var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
368
377
  var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
369
378
  {
370
379
  "summary": string, // high-level analysis summary
@@ -397,6 +406,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
397
406
  Prioritize issues by severity (critical / major / minor) as for image input.
398
407
  Cite frame indices in "frameReferences" so the user can locate the moments you describe.
399
408
  ${JSON_INSTRUCTIONS}`;
409
+ var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
410
+ {
411
+ "summary": string, // high-level summary of what happens across the video
412
+ "issues": [...], // list of issues/findings, can be empty
413
+ "timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
414
+ }
415
+ ${ISSUE_SCHEMA_INSTRUCTIONS}
416
+
417
+ Prioritize issues by severity (critical / major / minor) as for image input.
418
+ Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
419
+ ${JSON_INSTRUCTIONS}`;
400
420
  var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
401
421
  {
402
422
  "pass": boolean, // true if no critical or major changes found
@@ -418,15 +438,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
418
438
  var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
419
439
  var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
420
440
  var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
421
- function buildVideoTimelineSection(frameTimestamps, durationSeconds) {
441
+ var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
442
+ var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
443
+ function buildNativeVideoSection(durationSeconds) {
444
+ return `Video recording:
445
+ - Total duration: ${durationSeconds.toFixed(2)}s
446
+
447
+ The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
448
+ }
449
+ function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
422
450
  const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
451
+ const attached = frameTimestamps.length;
452
+ const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
453
+ const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
423
454
  return `Video timeline:
424
455
  - Total duration: ${durationSeconds.toFixed(2)}s
425
- - ${frameTimestamps.length} frames sampled (in chronological order)
456
+ ${sampledLine}
426
457
  - Frame index \u2192 timestamp:
427
458
  ${formatted}
428
459
 
429
- Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.`;
460
+ Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
430
461
  }
431
462
  var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
432
463
  var COMPARE_EDGE_RULES = [
@@ -441,30 +472,52 @@ function buildCheckPrompt(statements, options) {
441
472
  const stmts = Array.isArray(statements) ? statements : [statements];
442
473
  const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
443
474
  const media = options?.media;
444
- const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
475
+ const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
445
476
  const sections = [options?.role ?? defaultRole];
446
477
  if (media?.kind === "video") {
447
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
478
+ sections.push(
479
+ buildVideoTimelineSection(
480
+ media.frameTimestamps,
481
+ media.durationSeconds,
482
+ media.droppedUnchanged
483
+ )
484
+ );
485
+ } else if (media?.kind === "native-video") {
486
+ sections.push(buildNativeVideoSection(media.durationSeconds));
448
487
  }
449
488
  if (options?.instructions && options.instructions.length > 0) {
450
489
  sections.push(buildInstructionsSection(options.instructions));
451
490
  }
452
491
  sections.push(`Statements to evaluate:
453
492
  ${statementsBlock}`);
454
- sections.push(media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE);
493
+ sections.push(
494
+ media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
495
+ );
455
496
  return sections.join("\n\n");
456
497
  }
457
498
  function buildAskPrompt(userPrompt, options) {
458
499
  const media = options?.media;
459
- const sections = [media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : DEFAULT_ASK_ROLE];
500
+ const sections = [
501
+ media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
502
+ ];
460
503
  if (media?.kind === "video") {
461
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
504
+ sections.push(
505
+ buildVideoTimelineSection(
506
+ media.frameTimestamps,
507
+ media.durationSeconds,
508
+ media.droppedUnchanged
509
+ )
510
+ );
511
+ } else if (media?.kind === "native-video") {
512
+ sections.push(buildNativeVideoSection(media.durationSeconds));
462
513
  }
463
514
  if (options?.instructions && options.instructions.length > 0) {
464
515
  sections.push(buildInstructionsSection(options.instructions));
465
516
  }
466
517
  sections.push(`User request: ${userPrompt}`);
467
- sections.push(media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE);
518
+ sections.push(
519
+ media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
520
+ );
468
521
  return sections.join("\n\n");
469
522
  }
470
523
  function buildAiDiffPrompt() {
@@ -765,10 +818,16 @@ var AnthropicDriver = class {
765
818
 
766
819
  // src/providers/google.ts
767
820
  var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
821
+ var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
822
+ var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
823
+ var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
768
824
  function needsCodeExecution(model) {
769
825
  const match = model.match(/^gemini-(\d+)/);
770
826
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
771
827
  }
828
+ function sleep(ms) {
829
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
830
+ }
772
831
  var GOOGLE_THINKING_LEVEL = {
773
832
  low: "low",
774
833
  medium: "medium",
@@ -836,24 +895,28 @@ var GoogleDriver = class {
836
895
  });
837
896
  return this.client;
838
897
  }
839
- async sendMessage(images, prompt, _options) {
840
- const client = await this.getClient();
898
+ /** Request config shared by image and video messages. */
899
+ generationConfig() {
900
+ return {
901
+ responseMimeType: "application/json",
902
+ maxOutputTokens: this.maxTokens,
903
+ ...this.reasoningEffort && {
904
+ thinkingConfig: {
905
+ thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
906
+ }
907
+ },
908
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
909
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
910
+ }
911
+ };
912
+ }
913
+ /** Runs one generateContent call and normalizes finish reasons, text, and usage. */
914
+ async generate(client, contents) {
841
915
  try {
842
916
  const response = await client.models.generateContent({
843
917
  model: this.model,
844
- contents: [...this.toGeminiParts(images), prompt],
845
- config: {
846
- responseMimeType: "application/json",
847
- maxOutputTokens: this.maxTokens,
848
- ...this.reasoningEffort && {
849
- thinkingConfig: {
850
- thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
851
- }
852
- },
853
- ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
854
- mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
855
- }
856
- }
918
+ contents,
919
+ config: this.generationConfig()
857
920
  });
858
921
  const finishReason = response.candidates?.[0]?.finishReason;
859
922
  if (finishReason === "MAX_TOKENS") {
@@ -868,9 +931,8 @@ var GoogleDriver = class {
868
931
  `Response blocked: Google returned finishReason "${finishReason}".`
869
932
  );
870
933
  }
871
- const text = response.text ?? "";
872
934
  return {
873
- text,
935
+ text: response.text ?? "",
874
936
  usage: toGeminiUsage(response.usageMetadata)
875
937
  };
876
938
  } catch (err) {
@@ -878,6 +940,80 @@ var GoogleDriver = class {
878
940
  throw mapProviderError(err);
879
941
  }
880
942
  }
943
+ async sendMessage(images, prompt, _options) {
944
+ const client = await this.getClient();
945
+ return this.generate(client, [...this.toGeminiParts(images), prompt]);
946
+ }
947
+ /**
948
+ * Uploads a video through the Files API and waits until Gemini has finished
949
+ * processing it. Returns the ACTIVE file record.
950
+ */
951
+ async uploadVideo(client, video) {
952
+ try {
953
+ const uploaded = await client.files.upload({
954
+ file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
955
+ config: { mimeType: video.mimeType }
956
+ });
957
+ const name = uploaded.name;
958
+ if (!name) {
959
+ throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
960
+ }
961
+ const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
962
+ let current = uploaded;
963
+ while (current.state === "PROCESSING") {
964
+ if (Date.now() > deadline) {
965
+ throw new VisualAIProviderError(
966
+ `Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
967
+ );
968
+ }
969
+ await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
970
+ current = await client.files.get({ name });
971
+ }
972
+ if (current.state !== "ACTIVE") {
973
+ throw new VisualAIProviderError(
974
+ `Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
975
+ );
976
+ }
977
+ if (!current.uri) {
978
+ throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
979
+ }
980
+ return current;
981
+ } catch (err) {
982
+ if (err instanceof VisualAIProviderError) throw err;
983
+ throw mapProviderError(err);
984
+ }
985
+ }
986
+ /**
987
+ * Sends the video bytes themselves. Gemini samples the clip server-side at
988
+ * `video.fps` and, unlike sampled frames, also hears the audio track. Small
989
+ * videos go inline; larger ones are uploaded via the Files API and deleted
990
+ * again afterwards (they would expire on their own after 48 h).
991
+ */
992
+ async sendVideoMessage(video, prompt, _options) {
993
+ const client = await this.getClient();
994
+ const videoMetadata = { fps: video.fps };
995
+ if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
996
+ const part = {
997
+ inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
998
+ videoMetadata
999
+ };
1000
+ const response = await this.generate(client, [part, prompt]);
1001
+ return { ...response, delivery: "inline" };
1002
+ }
1003
+ const file = await this.uploadVideo(client, video);
1004
+ try {
1005
+ const part = {
1006
+ fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
1007
+ videoMetadata
1008
+ };
1009
+ const response = await this.generate(client, [part, prompt]);
1010
+ return { ...response, delivery: "file" };
1011
+ } finally {
1012
+ if (file.name) {
1013
+ await client.files.delete({ name: file.name }).catch(() => void 0);
1014
+ }
1015
+ }
1016
+ }
881
1017
  async generateImage(images, prompt, options) {
882
1018
  const client = await this.getClient();
883
1019
  const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
@@ -1387,6 +1523,12 @@ var PRICING_TABLE = {
1387
1523
  [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1388
1524
  inputPricePerToken: 0.1875 / PER_MILLION,
1389
1525
  outputPricePerToken: 1.125 / PER_MILLION
1526
+ },
1527
+ // Verified 2026-09-10 against https://openrouter.ai/api/v1/models. Cached
1528
+ // input is $0.03/MTok, not modelled (no provider gets a cache discount here).
1529
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1530
+ inputPricePerToken: 0.15 / PER_MILLION,
1531
+ outputPricePerToken: 0.5 / PER_MILLION
1390
1532
  }
1391
1533
  };
1392
1534
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1464,6 +1606,15 @@ async function timedSendMessage(driver, images, prompt, options) {
1464
1606
  const durationSeconds = (performance.now() - start) / 1e3;
1465
1607
  return { ...response, durationSeconds };
1466
1608
  }
1609
+ async function timedSendVideoMessage(driver, video, prompt, options) {
1610
+ if (!driver.sendVideoMessage) {
1611
+ throw new VisualAIError("Provider driver does not support native video delivery");
1612
+ }
1613
+ const start = performance.now();
1614
+ const response = await driver.sendVideoMessage(video, prompt, options);
1615
+ const durationSeconds = (performance.now() - start) / 1e3;
1616
+ return { ...response, durationSeconds };
1617
+ }
1467
1618
 
1468
1619
  // src/core/diff.ts
1469
1620
  var import_sharp = __toESM(require("sharp"), 1);
@@ -1703,6 +1854,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
1703
1854
  };
1704
1855
  }
1705
1856
 
1857
+ // src/core/media.ts
1858
+ var import_promises4 = require("fs/promises");
1859
+
1706
1860
  // src/core/debug-frames.ts
1707
1861
  var import_node_crypto = require("crypto");
1708
1862
  var import_promises2 = require("fs/promises");
@@ -1758,6 +1912,92 @@ async function saveDebugFrames(frames, env = process.env) {
1758
1912
  return runDir;
1759
1913
  }
1760
1914
 
1915
+ // src/core/frame-dedupe.ts
1916
+ var import_sharp3 = __toESM(require("sharp"), 1);
1917
+ var DEDUPE_THUMBNAIL_EDGE = 256;
1918
+ var DEDUPE_PIXEL_TOLERANCE = 24;
1919
+ var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
1920
+ function resolveDedupeOptions(raw) {
1921
+ if (raw === void 0 || raw === true) {
1922
+ return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
1923
+ }
1924
+ if (raw === false) {
1925
+ return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
1926
+ }
1927
+ const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
1928
+ if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
1929
+ throw new VisualAIVideoError(
1930
+ `Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
1931
+ );
1932
+ }
1933
+ return { enabled: true, threshold };
1934
+ }
1935
+ async function frameSignature(frame) {
1936
+ try {
1937
+ const { data, info } = await (0, import_sharp3.default)(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
1938
+ fit: "inside",
1939
+ withoutEnlargement: true
1940
+ }).raw().toBuffer({ resolveWithObject: true });
1941
+ return { width: info.width, height: info.height, pixels: data };
1942
+ } catch (err) {
1943
+ const reason = err instanceof Error ? err.message : String(err);
1944
+ throw new VisualAIVideoError(
1945
+ `Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
1946
+ );
1947
+ }
1948
+ }
1949
+ function changedFraction(a, b) {
1950
+ if (a.width !== b.width || a.height !== b.height) {
1951
+ return 1;
1952
+ }
1953
+ let changed = 0;
1954
+ for (let i = 0; i < a.pixels.length; i++) {
1955
+ if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
1956
+ changed++;
1957
+ }
1958
+ }
1959
+ return changed / a.pixels.length;
1960
+ }
1961
+ function reindex(frame, index) {
1962
+ if (frame.index === index) {
1963
+ return frame;
1964
+ }
1965
+ return {
1966
+ data: frame.data,
1967
+ mimeType: frame.mimeType,
1968
+ get base64() {
1969
+ return frame.base64;
1970
+ },
1971
+ timestampSeconds: frame.timestampSeconds,
1972
+ index
1973
+ };
1974
+ }
1975
+ async function dedupeFrames(frames, options) {
1976
+ const { enabled, threshold } = resolveDedupeOptions(options);
1977
+ if (!enabled || frames.length < 2) {
1978
+ return { frames: [...frames], dropped: 0 };
1979
+ }
1980
+ const signed = await Promise.all(
1981
+ frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
1982
+ );
1983
+ const [first, ...rest] = signed;
1984
+ if (first === void 0) {
1985
+ return { frames: [], dropped: 0 };
1986
+ }
1987
+ const kept = [first.frame];
1988
+ let lastKept = first.signature;
1989
+ for (const { frame, signature } of rest) {
1990
+ if (changedFraction(lastKept, signature) >= threshold) {
1991
+ kept.push(frame);
1992
+ lastKept = signature;
1993
+ }
1994
+ }
1995
+ return {
1996
+ frames: kept.map((frame, index) => reindex(frame, index)),
1997
+ dropped: frames.length - kept.length
1998
+ };
1999
+ }
2000
+
1761
2001
  // src/core/video.ts
1762
2002
  var import_promises3 = require("fs/promises");
1763
2003
  var import_node_os = require("os");
@@ -1993,7 +2233,7 @@ async function probeDurationSeconds(videoPath) {
1993
2233
  });
1994
2234
  });
1995
2235
  }
1996
- async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2236
+ function resolveVideoSamplingOptions(options = {}) {
1997
2237
  const fps = options.fps ?? DEFAULT_FPS;
1998
2238
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1999
2239
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -2013,13 +2253,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
2013
2253
  `Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
2014
2254
  );
2015
2255
  }
2016
- const ffmpeg = await loadFfmpegFactory();
2017
- const durationSeconds = await probeDurationSeconds(videoPath);
2256
+ return { fps, maxFrames, maxDurationSeconds };
2257
+ }
2258
+ function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
2018
2259
  if (durationSeconds > maxDurationSeconds) {
2019
2260
  throw new VisualAIVideoError(
2020
2261
  `Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
2021
2262
  );
2022
2263
  }
2264
+ }
2265
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2266
+ const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
2267
+ const ffmpeg = await loadFfmpegFactory();
2268
+ const durationSeconds = await probeDurationSeconds(videoPath);
2269
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2023
2270
  const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
2024
2271
  try {
2025
2272
  const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
@@ -2123,6 +2370,7 @@ function isTimestampedFrameInput(frame) {
2123
2370
  async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2124
2371
  const rawFrames = input.frames;
2125
2372
  const fps = input.fps ?? DEFAULT_FPS;
2373
+ resolveDedupeOptions(input.dedupe);
2126
2374
  if (rawFrames.length === 0) {
2127
2375
  throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
2128
2376
  }
@@ -2134,7 +2382,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2134
2382
  if (!Number.isFinite(fps) || fps <= 0) {
2135
2383
  throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
2136
2384
  }
2137
- const frames = await Promise.all(
2385
+ const sampled = await Promise.all(
2138
2386
  rawFrames.map(async (raw, index) => {
2139
2387
  const timestamped = isTimestampedFrameInput(raw);
2140
2388
  const imageInput = timestamped ? raw.image : raw;
@@ -2157,20 +2405,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2157
2405
  };
2158
2406
  })
2159
2407
  );
2160
- const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2408
+ const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2409
+ const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
2161
2410
  await saveDebugFrames(frames);
2162
- return { kind: "video", frames, durationSeconds };
2411
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2163
2412
  }
2164
- async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2413
+ async function normalizeNativeVideo(input, videoOptions) {
2414
+ const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
2415
+ const { path, mimeType, cleanup } = await resolveVideoToPath(input);
2416
+ try {
2417
+ const durationSeconds = await probeDurationSeconds(path);
2418
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2419
+ const data = await (0, import_promises4.readFile)(path);
2420
+ return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
2421
+ } finally {
2422
+ try {
2423
+ await cleanup();
2424
+ } catch {
2425
+ }
2426
+ }
2427
+ }
2428
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
2165
2429
  if (isFramesInput(input)) {
2166
2430
  return normalizeFrames(input, maxDimension);
2167
2431
  }
2168
2432
  if (isVideoInput(input)) {
2433
+ if (nativeVideo) {
2434
+ return normalizeNativeVideo(input, videoOptions);
2435
+ }
2436
+ resolveDedupeOptions(videoOptions?.dedupe);
2169
2437
  const { path, cleanup } = await resolveVideoToPath(input);
2170
2438
  try {
2171
- const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
2439
+ const { frames: sampled, durationSeconds } = await extractFrames(
2440
+ path,
2441
+ videoOptions,
2442
+ maxDimension
2443
+ );
2444
+ const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
2172
2445
  await saveDebugFrames(frames);
2173
- return { kind: "video", frames, durationSeconds };
2446
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2174
2447
  } finally {
2175
2448
  try {
2176
2449
  await cleanup();
@@ -2260,6 +2533,12 @@ var AskResultSchema = import_zod.z.object({
2260
2533
  * omitting the key, even for image inputs that were never asked to populate it.
2261
2534
  */
2262
2535
  frameReferences: import_zod.z.array(import_zod.z.number().int().nonnegative()).nullable().optional(),
2536
+ /**
2537
+ * For natively delivered video, the timestamps (seconds from the start of
2538
+ * the clip) the model relied on to answer. The native counterpart of
2539
+ * `frameReferences`. Nullable for the same strict-schema reason.
2540
+ */
2541
+ timestampReferences: import_zod.z.array(import_zod.z.number().nonnegative()).nullable().optional(),
2263
2542
  usage: UsageInfoSchema.optional()
2264
2543
  });
2265
2544
 
@@ -2338,7 +2617,8 @@ function parseAskResponse(raw) {
2338
2617
  const result = parseResponse(raw, AskResponseSchema);
2339
2618
  return {
2340
2619
  ...result,
2341
- frameReferences: result.frameReferences ?? void 0
2620
+ frameReferences: result.frameReferences ?? void 0,
2621
+ timestampReferences: result.timestampReferences ?? void 0
2342
2622
  };
2343
2623
  }
2344
2624
  function parseCompareResponse(raw) {
@@ -2367,26 +2647,66 @@ var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2367
2647
  function mediaToProviderInputs(media) {
2368
2648
  if (media.kind === "image") {
2369
2649
  return {
2650
+ kind: "images",
2370
2651
  images: [media.image],
2371
2652
  mediaContext: { kind: "image" },
2372
- framesMetadata: void 0
2653
+ frames: void 0
2654
+ };
2655
+ }
2656
+ if (media.kind === "native-video") {
2657
+ return {
2658
+ kind: "native-video",
2659
+ video: media.video,
2660
+ mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
2373
2661
  };
2374
2662
  }
2375
2663
  const timestamps = media.frames.map((f) => f.timestampSeconds);
2376
2664
  return {
2665
+ kind: "images",
2377
2666
  images: media.frames,
2378
2667
  mediaContext: {
2379
2668
  kind: "video",
2380
2669
  frameTimestamps: timestamps,
2381
- durationSeconds: media.durationSeconds
2670
+ durationSeconds: media.durationSeconds,
2671
+ droppedUnchanged: media.droppedUnchanged
2382
2672
  },
2383
- framesMetadata: {
2673
+ frames: {
2384
2674
  count: media.frames.length,
2385
2675
  timestampsSeconds: timestamps,
2386
- durationSeconds: media.durationSeconds
2676
+ durationSeconds: media.durationSeconds,
2677
+ droppedUnchanged: media.droppedUnchanged
2387
2678
  }
2388
2679
  };
2389
2680
  }
2681
+ async function sendMedia(driver, dispatch, prompt, options) {
2682
+ if (dispatch.kind === "native-video") {
2683
+ const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
2684
+ const { durationSeconds, fps, mimeType } = dispatch.video;
2685
+ return {
2686
+ response: response2,
2687
+ metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
2688
+ };
2689
+ }
2690
+ const response = await timedSendMessage(driver, dispatch.images, prompt, options);
2691
+ return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
2692
+ }
2693
+ function resolveNativeVideo(input, videoOptions, driver, provider) {
2694
+ const mode = videoOptions?.mode ?? "auto";
2695
+ if (mode !== "auto" && mode !== "native" && mode !== "frames") {
2696
+ throw new VisualAIConfigError(
2697
+ `Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
2698
+ );
2699
+ }
2700
+ const supported = typeof driver.sendVideoMessage === "function";
2701
+ if (mode === "frames") return false;
2702
+ if (mode === "auto") return supported;
2703
+ if (!supported && !isFramesInput(input) && isVideoInput(input)) {
2704
+ throw new VisualAIConfigError(
2705
+ `Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
2706
+ );
2707
+ }
2708
+ return true;
2709
+ }
2390
2710
  function visualAI(config = {}) {
2391
2711
  const resolvedConfig = resolveConfig(config);
2392
2712
  const driverConfig = {
@@ -2424,38 +2744,55 @@ function visualAI(config = {}) {
2424
2744
  throw new VisualAIConfigError("At least one statement is required for check()");
2425
2745
  }
2426
2746
  return withErrorDebug(resolvedConfig, "check", async () => {
2427
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2428
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2747
+ const nativeVideo = resolveNativeVideo(
2748
+ input,
2749
+ options?.video,
2750
+ driver,
2751
+ resolvedConfig.provider
2752
+ );
2753
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2754
+ const dispatch = mediaToProviderInputs(media);
2429
2755
  const prompt = buildCheckPrompt(stmts, {
2430
2756
  instructions: options?.instructions,
2431
- media: mediaContext
2757
+ media: dispatch.mediaContext
2432
2758
  });
2433
2759
  debugLog(resolvedConfig, "check prompt", prompt, "prompt");
2434
- const response = await timedSendMessage(driver, images, prompt, checkSchemaOptions);
2760
+ const { response, metadata } = await sendMedia(
2761
+ driver,
2762
+ dispatch,
2763
+ prompt,
2764
+ checkSchemaOptions
2765
+ );
2435
2766
  debugLog(resolvedConfig, "check response", response.text, "response");
2436
2767
  const result = parseCheckResponse(response.text);
2437
2768
  return {
2438
2769
  ...result,
2439
- ...framesMetadata ? { frames: framesMetadata } : {},
2770
+ ...metadata,
2440
2771
  usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
2441
2772
  };
2442
2773
  });
2443
2774
  },
2444
2775
  async ask(input, userPrompt, options) {
2445
2776
  return withErrorDebug(resolvedConfig, "ask", async () => {
2446
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2447
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2777
+ const nativeVideo = resolveNativeVideo(
2778
+ input,
2779
+ options?.video,
2780
+ driver,
2781
+ resolvedConfig.provider
2782
+ );
2783
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2784
+ const dispatch = mediaToProviderInputs(media);
2448
2785
  const prompt = buildAskPrompt(userPrompt, {
2449
2786
  instructions: options?.instructions,
2450
- media: mediaContext
2787
+ media: dispatch.mediaContext
2451
2788
  });
2452
2789
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2453
- const response = await timedSendMessage(driver, images, prompt, askSchemaOptions);
2790
+ const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2454
2791
  debugLog(resolvedConfig, "ask response", response.text, "response");
2455
2792
  const result = parseAskResponse(response.text);
2456
2793
  return {
2457
2794
  ...result,
2458
- ...framesMetadata ? { frames: framesMetadata } : {},
2795
+ ...metadata,
2459
2796
  usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
2460
2797
  };
2461
2798
  });