visual-ai-assertions 0.22.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -66,7 +66,8 @@ var Model = {
66
66
  KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
67
67
  QWEN_3_8_MAX: "qwen/qwen3.8-max",
68
68
  QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
69
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
69
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
70
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
70
71
  }
71
72
  };
72
73
  var DEFAULT_MODELS = {
@@ -257,10 +258,15 @@ Example for a failing check:
257
258
  ]
258
259
  }
259
260
  ${JSON_INSTRUCTIONS}`;
260
- var CHECK_OUTPUT_SCHEMA_VIDEO = `IMPORTANT: Follow this evaluation order:
261
+ function buildCheckOutputSchemaVideo(unit) {
262
+ const anyPoint = unit === "frame" ? "ANY frame of the timeline" : "ANY moment of the video";
263
+ const bestPoint = unit === "frame" ? "the timestamp of the frame that most clearly demonstrates it" : "the timestamp of the moment that most clearly demonstrates it";
264
+ const citing = unit === "frame" ? "citing frame timestamps" : "citing timestamps";
265
+ const exampleWhere = unit === "frame" ? "at the 3.5s frame" : "at 3.5s";
266
+ return `IMPORTANT: Follow this evaluation order:
261
267
  1. First, evaluate EACH statement independently across the entire timeline and populate the "statements" array
262
- 2. A statement passes if it is true at ANY frame of the timeline, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
263
- 3. For each statement that passes, set "timestampSeconds" to the timestamp of the frame that most clearly demonstrates it (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
268
+ 2. A statement passes if it is true at ${anyPoint}, unless the wording explicitly says otherwise (e.g. "throughout", "at all times")
269
+ 3. For each statement that passes, set "timestampSeconds" to ${bestPoint} (or where it first becomes true). Use null when the statement fails or applies across the whole clip.
264
270
  4. Then, set "pass" to true ONLY if every statement passed (logical AND of all statement results)
265
271
  5. Write "reasoning" as a brief overall summary of the evaluation
266
272
  6. Include "issues" only for statements that failed
@@ -274,7 +280,7 @@ Respond with a JSON object matching this exact structure:
274
280
  {
275
281
  "statement": string, // the original statement text
276
282
  "pass": boolean, // whether this statement is true at any point in the timeline
277
- "reasoning": string, // explanation for this statement, citing frame timestamps where relevant
283
+ "reasoning": string, // explanation for this statement, ${citing} where relevant
278
284
  "confidence": "high" | "medium" | "low",
279
285
  "timestampSeconds": number | null
280
286
  // seconds from the start of the clip where the statement is most clearly true,
@@ -292,10 +298,13 @@ Example for a passing video check:
292
298
  "reasoning": "The success toast appeared briefly around 3.5s.",
293
299
  "issues": [],
294
300
  "statements": [
295
- { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right at the 3.5s frame", "confidence": "high", "timestampSeconds": 3.5 }
301
+ { "statement": "A success toast with text 'Saved' appears", "pass": true, "reasoning": "A green toast labeled 'Saved' is visible in the bottom-right ${exampleWhere}", "confidence": "high", "timestampSeconds": 3.5 }
296
302
  ]
297
303
  }
298
304
  ${JSON_INSTRUCTIONS}`;
305
+ }
306
+ var CHECK_OUTPUT_SCHEMA_VIDEO = buildCheckOutputSchemaVideo("frame");
307
+ var CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO = buildCheckOutputSchemaVideo("moment");
299
308
  var ASK_OUTPUT_SCHEMA_IMAGE = `Respond with a JSON object matching this exact structure:
300
309
  {
301
310
  "summary": string, // high-level analysis summary
@@ -328,6 +337,17 @@ ${ISSUE_SCHEMA_INSTRUCTIONS}
328
337
  Prioritize issues by severity (critical / major / minor) as for image input.
329
338
  Cite frame indices in "frameReferences" so the user can locate the moments you describe.
330
339
  ${JSON_INSTRUCTIONS}`;
340
+ var ASK_OUTPUT_SCHEMA_NATIVE_VIDEO = `Respond with a JSON object matching this exact structure:
341
+ {
342
+ "summary": string, // high-level summary of what happens across the video
343
+ "issues": [...], // list of issues/findings, can be empty
344
+ "timestampReferences": number[] // seconds from the start of the clip of the moments the answer relies on (in order)
345
+ }
346
+ ${ISSUE_SCHEMA_INSTRUCTIONS}
347
+
348
+ Prioritize issues by severity (critical / major / minor) as for image input.
349
+ Cite timestamps in "timestampReferences" so the user can locate the moments you describe.
350
+ ${JSON_INSTRUCTIONS}`;
331
351
  var COMPARE_OUTPUT_SCHEMA = `Respond with a JSON object matching this exact structure:
332
352
  {
333
353
  "pass": boolean, // true if no critical or major changes found
@@ -349,15 +369,26 @@ var DEFAULT_CHECK_ROLE = "You are a visual QA assistant. Evaluate the provided i
349
369
  var DEFAULT_CHECK_ROLE_VIDEO = "You are a visual QA assistant. Evaluate the provided sequence of video frames precisely and objectively, treating them as a chronological timeline.";
350
370
  var DEFAULT_ASK_ROLE = "You are a visual QA assistant. Analyze the provided image based on the user's request.";
351
371
  var DEFAULT_ASK_ROLE_VIDEO = "You are a visual QA assistant. Analyze the provided sequence of video frames as a chronological timeline based on the user's request.";
352
- function buildVideoTimelineSection(frameTimestamps, durationSeconds) {
372
+ var DEFAULT_CHECK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Evaluate the provided video recording precisely and objectively, treating it as a chronological timeline.";
373
+ var DEFAULT_ASK_ROLE_NATIVE_VIDEO = "You are a visual QA assistant. Analyze the provided video recording as a chronological timeline based on the user's request.";
374
+ function buildNativeVideoSection(durationSeconds) {
375
+ return `Video recording:
376
+ - Total duration: ${durationSeconds.toFixed(2)}s
377
+
378
+ The attached file is the complete video recording. Treat it as a chronological timeline and refer to moments by timestamp (seconds from the start of the clip) where helpful.`;
379
+ }
380
+ function buildVideoTimelineSection(frameTimestamps, durationSeconds, droppedUnchanged = 0) {
353
381
  const formatted = frameTimestamps.map((t, i) => ` ${i}: ${t.toFixed(2)}s`).join("\n");
382
+ const attached = frameTimestamps.length;
383
+ const sampledLine = droppedUnchanged > 0 ? `- ${attached + droppedUnchanged} frames sampled (in chronological order); ${droppedUnchanged} ${droppedUnchanged === 1 ? "was" : "were"} dropped because ${droppedUnchanged === 1 ? "it" : "they"} did not visibly change from the preceding kept frame, so ${attached} ${attached === 1 ? "image is" : "images are"} attached` : `- ${attached} frames sampled (in chronological order)`;
384
+ const droppedGuidance = droppedUnchanged > 0 ? ` Frames sampled between two consecutive listed timestamps looked the same as the earlier listed frame, and frames sampled after the last listed timestamp looked the same as the last attached image until the clip ended at ${durationSeconds.toFixed(2)}s.` : "";
354
385
  return `Video timeline:
355
386
  - Total duration: ${durationSeconds.toFixed(2)}s
356
- - ${frameTimestamps.length} frames sampled (in chronological order)
387
+ ${sampledLine}
357
388
  - Frame index \u2192 timestamp:
358
389
  ${formatted}
359
390
 
360
- Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.`;
391
+ Treat the attached images as a chronological timeline. The first image is the earliest frame, the last is the latest. Refer to frames by timestamp where helpful.${droppedGuidance}`;
361
392
  }
362
393
  var COMPARE_ROLE = "You are performing a visual regression test. Compare the BEFORE image (baseline) to the AFTER image (current) and identify all visual differences. Flag changes that appear unintentional or problematic.";
363
394
  var COMPARE_EDGE_RULES = [
@@ -372,30 +403,52 @@ function buildCheckPrompt(statements, options) {
372
403
  const stmts = Array.isArray(statements) ? statements : [statements];
373
404
  const statementsBlock = stmts.map((s, i) => `${i + 1}. "${s}"`).join("\n");
374
405
  const media = options?.media;
375
- const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : DEFAULT_CHECK_ROLE;
406
+ const defaultRole = media?.kind === "video" ? DEFAULT_CHECK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_CHECK_ROLE_NATIVE_VIDEO : DEFAULT_CHECK_ROLE;
376
407
  const sections = [options?.role ?? defaultRole];
377
408
  if (media?.kind === "video") {
378
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
409
+ sections.push(
410
+ buildVideoTimelineSection(
411
+ media.frameTimestamps,
412
+ media.durationSeconds,
413
+ media.droppedUnchanged
414
+ )
415
+ );
416
+ } else if (media?.kind === "native-video") {
417
+ sections.push(buildNativeVideoSection(media.durationSeconds));
379
418
  }
380
419
  if (options?.instructions && options.instructions.length > 0) {
381
420
  sections.push(buildInstructionsSection(options.instructions));
382
421
  }
383
422
  sections.push(`Statements to evaluate:
384
423
  ${statementsBlock}`);
385
- sections.push(media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE);
424
+ sections.push(
425
+ media?.kind === "video" ? CHECK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? CHECK_OUTPUT_SCHEMA_NATIVE_VIDEO : CHECK_OUTPUT_SCHEMA_IMAGE
426
+ );
386
427
  return sections.join("\n\n");
387
428
  }
388
429
  function buildAskPrompt(userPrompt, options) {
389
430
  const media = options?.media;
390
- const sections = [media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : DEFAULT_ASK_ROLE];
431
+ const sections = [
432
+ media?.kind === "video" ? DEFAULT_ASK_ROLE_VIDEO : media?.kind === "native-video" ? DEFAULT_ASK_ROLE_NATIVE_VIDEO : DEFAULT_ASK_ROLE
433
+ ];
391
434
  if (media?.kind === "video") {
392
- sections.push(buildVideoTimelineSection(media.frameTimestamps, media.durationSeconds));
435
+ sections.push(
436
+ buildVideoTimelineSection(
437
+ media.frameTimestamps,
438
+ media.durationSeconds,
439
+ media.droppedUnchanged
440
+ )
441
+ );
442
+ } else if (media?.kind === "native-video") {
443
+ sections.push(buildNativeVideoSection(media.durationSeconds));
393
444
  }
394
445
  if (options?.instructions && options.instructions.length > 0) {
395
446
  sections.push(buildInstructionsSection(options.instructions));
396
447
  }
397
448
  sections.push(`User request: ${userPrompt}`);
398
- sections.push(media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE);
449
+ sections.push(
450
+ media?.kind === "video" ? ASK_OUTPUT_SCHEMA_VIDEO : media?.kind === "native-video" ? ASK_OUTPUT_SCHEMA_NATIVE_VIDEO : ASK_OUTPUT_SCHEMA_IMAGE
451
+ );
399
452
  return sections.join("\n\n");
400
453
  }
401
454
  function buildAiDiffPrompt() {
@@ -696,10 +749,16 @@ var AnthropicDriver = class {
696
749
 
697
750
  // src/providers/google.ts
698
751
  var DEFAULT_IMAGE_GEN_MODEL = "gemini-2.5-flash-image";
752
+ var GEMINI_INLINE_VIDEO_LIMIT_BYTES = 19 * 1024 * 1024;
753
+ var GEMINI_FILE_POLL_INTERVAL_MS = 2e3;
754
+ var GEMINI_FILE_POLL_TIMEOUT_MS = 5 * 6e4;
699
755
  function needsCodeExecution(model) {
700
756
  const match = model.match(/^gemini-(\d+)/);
701
757
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
702
758
  }
759
+ function sleep(ms) {
760
+ return new Promise((resolve2) => setTimeout(resolve2, ms));
761
+ }
703
762
  var GOOGLE_THINKING_LEVEL = {
704
763
  low: "low",
705
764
  medium: "medium",
@@ -767,24 +826,28 @@ var GoogleDriver = class {
767
826
  });
768
827
  return this.client;
769
828
  }
770
- async sendMessage(images, prompt, _options) {
771
- const client = await this.getClient();
829
+ /** Request config shared by image and video messages. */
830
+ generationConfig() {
831
+ return {
832
+ responseMimeType: "application/json",
833
+ maxOutputTokens: this.maxTokens,
834
+ ...this.reasoningEffort && {
835
+ thinkingConfig: {
836
+ thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
837
+ }
838
+ },
839
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
840
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
841
+ }
842
+ };
843
+ }
844
+ /** Runs one generateContent call and normalizes finish reasons, text, and usage. */
845
+ async generate(client, contents) {
772
846
  try {
773
847
  const response = await client.models.generateContent({
774
848
  model: this.model,
775
- contents: [...this.toGeminiParts(images), prompt],
776
- config: {
777
- responseMimeType: "application/json",
778
- maxOutputTokens: this.maxTokens,
779
- ...this.reasoningEffort && {
780
- thinkingConfig: {
781
- thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
782
- }
783
- },
784
- ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
785
- mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
786
- }
787
- }
849
+ contents,
850
+ config: this.generationConfig()
788
851
  });
789
852
  const finishReason = response.candidates?.[0]?.finishReason;
790
853
  if (finishReason === "MAX_TOKENS") {
@@ -799,9 +862,8 @@ var GoogleDriver = class {
799
862
  `Response blocked: Google returned finishReason "${finishReason}".`
800
863
  );
801
864
  }
802
- const text = response.text ?? "";
803
865
  return {
804
- text,
866
+ text: response.text ?? "",
805
867
  usage: toGeminiUsage(response.usageMetadata)
806
868
  };
807
869
  } catch (err) {
@@ -809,6 +871,80 @@ var GoogleDriver = class {
809
871
  throw mapProviderError(err);
810
872
  }
811
873
  }
874
+ async sendMessage(images, prompt, _options) {
875
+ const client = await this.getClient();
876
+ return this.generate(client, [...this.toGeminiParts(images), prompt]);
877
+ }
878
+ /**
879
+ * Uploads a video through the Files API and waits until Gemini has finished
880
+ * processing it. Returns the ACTIVE file record.
881
+ */
882
+ async uploadVideo(client, video) {
883
+ try {
884
+ const uploaded = await client.files.upload({
885
+ file: new Blob([new Uint8Array(video.data)], { type: video.mimeType }),
886
+ config: { mimeType: video.mimeType }
887
+ });
888
+ const name = uploaded.name;
889
+ if (!name) {
890
+ throw new VisualAIProviderError("Gemini Files API returned a file without a name.");
891
+ }
892
+ const deadline = Date.now() + GEMINI_FILE_POLL_TIMEOUT_MS;
893
+ let current = uploaded;
894
+ while (current.state === "PROCESSING") {
895
+ if (Date.now() > deadline) {
896
+ throw new VisualAIProviderError(
897
+ `Gemini file ${name} was still processing after ${GEMINI_FILE_POLL_TIMEOUT_MS}ms.`
898
+ );
899
+ }
900
+ await sleep(GEMINI_FILE_POLL_INTERVAL_MS);
901
+ current = await client.files.get({ name });
902
+ }
903
+ if (current.state !== "ACTIVE") {
904
+ throw new VisualAIProviderError(
905
+ `Gemini file ${name} ended in state ${current.state ?? "unknown"}: ${current.error?.message ?? "no error message"}`
906
+ );
907
+ }
908
+ if (!current.uri) {
909
+ throw new VisualAIProviderError("Gemini Files API returned an ACTIVE file without a URI.");
910
+ }
911
+ return current;
912
+ } catch (err) {
913
+ if (err instanceof VisualAIProviderError) throw err;
914
+ throw mapProviderError(err);
915
+ }
916
+ }
917
+ /**
918
+ * Sends the video bytes themselves. Gemini samples the clip server-side at
919
+ * `video.fps` and, unlike sampled frames, also hears the audio track. Small
920
+ * videos go inline; larger ones are uploaded via the Files API and deleted
921
+ * again afterwards (they would expire on their own after 48 h).
922
+ */
923
+ async sendVideoMessage(video, prompt, _options) {
924
+ const client = await this.getClient();
925
+ const videoMetadata = { fps: video.fps };
926
+ if (video.data.byteLength <= GEMINI_INLINE_VIDEO_LIMIT_BYTES) {
927
+ const part = {
928
+ inlineData: { data: video.data.toString("base64"), mimeType: video.mimeType },
929
+ videoMetadata
930
+ };
931
+ const response = await this.generate(client, [part, prompt]);
932
+ return { ...response, delivery: "inline" };
933
+ }
934
+ const file = await this.uploadVideo(client, video);
935
+ try {
936
+ const part = {
937
+ fileData: { fileUri: file.uri, mimeType: file.mimeType ?? video.mimeType },
938
+ videoMetadata
939
+ };
940
+ const response = await this.generate(client, [part, prompt]);
941
+ return { ...response, delivery: "file" };
942
+ } finally {
943
+ if (file.name) {
944
+ await client.files.delete({ name: file.name }).catch(() => void 0);
945
+ }
946
+ }
947
+ }
812
948
  async generateImage(images, prompt, options) {
813
949
  const client = await this.getClient();
814
950
  const imageModel = options?.model ?? DEFAULT_IMAGE_GEN_MODEL;
@@ -1318,6 +1454,12 @@ var PRICING_TABLE = {
1318
1454
  [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1319
1455
  inputPricePerToken: 0.1875 / PER_MILLION,
1320
1456
  outputPricePerToken: 1.125 / PER_MILLION
1457
+ },
1458
+ // Verified 2026-09-10 against https://openrouter.ai/api/v1/models. Cached
1459
+ // input is $0.03/MTok, not modelled (no provider gets a cache discount here).
1460
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1461
+ inputPricePerToken: 0.15 / PER_MILLION,
1462
+ outputPricePerToken: 0.5 / PER_MILLION
1321
1463
  }
1322
1464
  };
1323
1465
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1395,6 +1537,15 @@ async function timedSendMessage(driver, images, prompt, options) {
1395
1537
  const durationSeconds = (performance.now() - start) / 1e3;
1396
1538
  return { ...response, durationSeconds };
1397
1539
  }
1540
+ async function timedSendVideoMessage(driver, video, prompt, options) {
1541
+ if (!driver.sendVideoMessage) {
1542
+ throw new VisualAIError("Provider driver does not support native video delivery");
1543
+ }
1544
+ const start = performance.now();
1545
+ const response = await driver.sendVideoMessage(video, prompt, options);
1546
+ const durationSeconds = (performance.now() - start) / 1e3;
1547
+ return { ...response, durationSeconds };
1548
+ }
1398
1549
 
1399
1550
  // src/core/diff.ts
1400
1551
  import sharp from "sharp";
@@ -1634,6 +1785,9 @@ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION)
1634
1785
  };
1635
1786
  }
1636
1787
 
1788
+ // src/core/media.ts
1789
+ import { readFile as readFile3 } from "fs/promises";
1790
+
1637
1791
  // src/core/debug-frames.ts
1638
1792
  import { randomBytes } from "crypto";
1639
1793
  import { mkdir, writeFile } from "fs/promises";
@@ -1689,6 +1843,92 @@ async function saveDebugFrames(frames, env = process.env) {
1689
1843
  return runDir;
1690
1844
  }
1691
1845
 
1846
+ // src/core/frame-dedupe.ts
1847
+ import sharp3 from "sharp";
1848
+ var DEDUPE_THUMBNAIL_EDGE = 256;
1849
+ var DEDUPE_PIXEL_TOLERANCE = 24;
1850
+ var DEFAULT_DEDUPE_THRESHOLD = 1e-3;
1851
+ function resolveDedupeOptions(raw) {
1852
+ if (raw === void 0 || raw === true) {
1853
+ return { enabled: true, threshold: DEFAULT_DEDUPE_THRESHOLD };
1854
+ }
1855
+ if (raw === false) {
1856
+ return { enabled: false, threshold: DEFAULT_DEDUPE_THRESHOLD };
1857
+ }
1858
+ const threshold = raw.threshold ?? DEFAULT_DEDUPE_THRESHOLD;
1859
+ if (!Number.isFinite(threshold) || threshold <= 0 || threshold > 1) {
1860
+ throw new VisualAIVideoError(
1861
+ `Invalid dedupe threshold: ${String(threshold)}. Must be a finite number in (0, 1].`
1862
+ );
1863
+ }
1864
+ return { enabled: true, threshold };
1865
+ }
1866
+ async function frameSignature(frame) {
1867
+ try {
1868
+ const { data, info } = await sharp3(frame.data).flatten({ background: { r: 255, g: 255, b: 255 } }).greyscale().resize(DEDUPE_THUMBNAIL_EDGE, DEDUPE_THUMBNAIL_EDGE, {
1869
+ fit: "inside",
1870
+ withoutEnlargement: true
1871
+ }).raw().toBuffer({ resolveWithObject: true });
1872
+ return { width: info.width, height: info.height, pixels: data };
1873
+ } catch (err) {
1874
+ const reason = err instanceof Error ? err.message : String(err);
1875
+ throw new VisualAIVideoError(
1876
+ `Failed to decode frame ${frame.index} (${frame.timestampSeconds.toFixed(2)}s) for change detection: ${reason}`
1877
+ );
1878
+ }
1879
+ }
1880
+ function changedFraction(a, b) {
1881
+ if (a.width !== b.width || a.height !== b.height) {
1882
+ return 1;
1883
+ }
1884
+ let changed = 0;
1885
+ for (let i = 0; i < a.pixels.length; i++) {
1886
+ if (Math.abs((a.pixels[i] ?? 0) - (b.pixels[i] ?? 0)) > DEDUPE_PIXEL_TOLERANCE) {
1887
+ changed++;
1888
+ }
1889
+ }
1890
+ return changed / a.pixels.length;
1891
+ }
1892
+ function reindex(frame, index) {
1893
+ if (frame.index === index) {
1894
+ return frame;
1895
+ }
1896
+ return {
1897
+ data: frame.data,
1898
+ mimeType: frame.mimeType,
1899
+ get base64() {
1900
+ return frame.base64;
1901
+ },
1902
+ timestampSeconds: frame.timestampSeconds,
1903
+ index
1904
+ };
1905
+ }
1906
+ async function dedupeFrames(frames, options) {
1907
+ const { enabled, threshold } = resolveDedupeOptions(options);
1908
+ if (!enabled || frames.length < 2) {
1909
+ return { frames: [...frames], dropped: 0 };
1910
+ }
1911
+ const signed = await Promise.all(
1912
+ frames.map(async (frame) => ({ frame, signature: await frameSignature(frame) }))
1913
+ );
1914
+ const [first, ...rest] = signed;
1915
+ if (first === void 0) {
1916
+ return { frames: [], dropped: 0 };
1917
+ }
1918
+ const kept = [first.frame];
1919
+ let lastKept = first.signature;
1920
+ for (const { frame, signature } of rest) {
1921
+ if (changedFraction(lastKept, signature) >= threshold) {
1922
+ kept.push(frame);
1923
+ lastKept = signature;
1924
+ }
1925
+ }
1926
+ return {
1927
+ frames: kept.map((frame, index) => reindex(frame, index)),
1928
+ dropped: frames.length - kept.length
1929
+ };
1930
+ }
1931
+
1692
1932
  // src/core/video.ts
1693
1933
  import { mkdtemp, readFile as readFile2, readdir, rm, writeFile as writeFile2 } from "fs/promises";
1694
1934
  import { tmpdir } from "os";
@@ -1924,7 +2164,7 @@ async function probeDurationSeconds(videoPath) {
1924
2164
  });
1925
2165
  });
1926
2166
  }
1927
- async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2167
+ function resolveVideoSamplingOptions(options = {}) {
1928
2168
  const fps = options.fps ?? DEFAULT_FPS;
1929
2169
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1930
2170
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1944,13 +2184,20 @@ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_D
1944
2184
  `Invalid maxDurationSeconds: ${maxDurationSeconds}. Must be a finite number > 0.`
1945
2185
  );
1946
2186
  }
1947
- const ffmpeg = await loadFfmpegFactory();
1948
- const durationSeconds = await probeDurationSeconds(videoPath);
2187
+ return { fps, maxFrames, maxDurationSeconds };
2188
+ }
2189
+ function assertDurationWithinLimit(durationSeconds, maxDurationSeconds) {
1949
2190
  if (durationSeconds > maxDurationSeconds) {
1950
2191
  throw new VisualAIVideoError(
1951
2192
  `Video duration ${durationSeconds.toFixed(2)}s exceeds limit of ${maxDurationSeconds}s. Pass { maxDurationSeconds: N } to override, or trim the source video.`
1952
2193
  );
1953
2194
  }
2195
+ }
2196
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
2197
+ const { fps, maxFrames, maxDurationSeconds } = resolveVideoSamplingOptions(options);
2198
+ const ffmpeg = await loadFfmpegFactory();
2199
+ const durationSeconds = await probeDurationSeconds(videoPath);
2200
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
1954
2201
  const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
1955
2202
  try {
1956
2203
  const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
@@ -2054,6 +2301,7 @@ function isTimestampedFrameInput(frame) {
2054
2301
  async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2055
2302
  const rawFrames = input.frames;
2056
2303
  const fps = input.fps ?? DEFAULT_FPS;
2304
+ resolveDedupeOptions(input.dedupe);
2057
2305
  if (rawFrames.length === 0) {
2058
2306
  throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
2059
2307
  }
@@ -2065,7 +2313,7 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2065
2313
  if (!Number.isFinite(fps) || fps <= 0) {
2066
2314
  throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
2067
2315
  }
2068
- const frames = await Promise.all(
2316
+ const sampled = await Promise.all(
2069
2317
  rawFrames.map(async (raw, index) => {
2070
2318
  const timestamped = isTimestampedFrameInput(raw);
2071
2319
  const imageInput = timestamped ? raw.image : raw;
@@ -2088,20 +2336,45 @@ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION
2088
2336
  };
2089
2337
  })
2090
2338
  );
2091
- const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2339
+ const durationSeconds = sampled.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
2340
+ const { frames, dropped } = await dedupeFrames(sampled, input.dedupe);
2092
2341
  await saveDebugFrames(frames);
2093
- return { kind: "video", frames, durationSeconds };
2342
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2094
2343
  }
2095
- async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
2344
+ async function normalizeNativeVideo(input, videoOptions) {
2345
+ const { fps, maxDurationSeconds } = resolveVideoSamplingOptions(videoOptions);
2346
+ const { path, mimeType, cleanup } = await resolveVideoToPath(input);
2347
+ try {
2348
+ const durationSeconds = await probeDurationSeconds(path);
2349
+ assertDurationWithinLimit(durationSeconds, maxDurationSeconds);
2350
+ const data = await readFile3(path);
2351
+ return { kind: "native-video", video: { data, mimeType, durationSeconds, fps } };
2352
+ } finally {
2353
+ try {
2354
+ await cleanup();
2355
+ } catch {
2356
+ }
2357
+ }
2358
+ }
2359
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION, nativeVideo = false) {
2096
2360
  if (isFramesInput(input)) {
2097
2361
  return normalizeFrames(input, maxDimension);
2098
2362
  }
2099
2363
  if (isVideoInput(input)) {
2364
+ if (nativeVideo) {
2365
+ return normalizeNativeVideo(input, videoOptions);
2366
+ }
2367
+ resolveDedupeOptions(videoOptions?.dedupe);
2100
2368
  const { path, cleanup } = await resolveVideoToPath(input);
2101
2369
  try {
2102
- const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
2370
+ const { frames: sampled, durationSeconds } = await extractFrames(
2371
+ path,
2372
+ videoOptions,
2373
+ maxDimension
2374
+ );
2375
+ const { frames, dropped } = await dedupeFrames(sampled, videoOptions?.dedupe);
2103
2376
  await saveDebugFrames(frames);
2104
- return { kind: "video", frames, durationSeconds };
2377
+ return { kind: "video", frames, durationSeconds, droppedUnchanged: dropped };
2105
2378
  } finally {
2106
2379
  try {
2107
2380
  await cleanup();
@@ -2191,6 +2464,12 @@ var AskResultSchema = z.object({
2191
2464
  * omitting the key, even for image inputs that were never asked to populate it.
2192
2465
  */
2193
2466
  frameReferences: z.array(z.number().int().nonnegative()).nullable().optional(),
2467
+ /**
2468
+ * For natively delivered video, the timestamps (seconds from the start of
2469
+ * the clip) the model relied on to answer. The native counterpart of
2470
+ * `frameReferences`. Nullable for the same strict-schema reason.
2471
+ */
2472
+ timestampReferences: z.array(z.number().nonnegative()).nullable().optional(),
2194
2473
  usage: UsageInfoSchema.optional()
2195
2474
  });
2196
2475
 
@@ -2269,7 +2548,8 @@ function parseAskResponse(raw) {
2269
2548
  const result = parseResponse(raw, AskResponseSchema);
2270
2549
  return {
2271
2550
  ...result,
2272
- frameReferences: result.frameReferences ?? void 0
2551
+ frameReferences: result.frameReferences ?? void 0,
2552
+ timestampReferences: result.timestampReferences ?? void 0
2273
2553
  };
2274
2554
  }
2275
2555
  function parseCompareResponse(raw) {
@@ -2298,26 +2578,66 @@ var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2298
2578
  function mediaToProviderInputs(media) {
2299
2579
  if (media.kind === "image") {
2300
2580
  return {
2581
+ kind: "images",
2301
2582
  images: [media.image],
2302
2583
  mediaContext: { kind: "image" },
2303
- framesMetadata: void 0
2584
+ frames: void 0
2585
+ };
2586
+ }
2587
+ if (media.kind === "native-video") {
2588
+ return {
2589
+ kind: "native-video",
2590
+ video: media.video,
2591
+ mediaContext: { kind: "native-video", durationSeconds: media.video.durationSeconds }
2304
2592
  };
2305
2593
  }
2306
2594
  const timestamps = media.frames.map((f) => f.timestampSeconds);
2307
2595
  return {
2596
+ kind: "images",
2308
2597
  images: media.frames,
2309
2598
  mediaContext: {
2310
2599
  kind: "video",
2311
2600
  frameTimestamps: timestamps,
2312
- durationSeconds: media.durationSeconds
2601
+ durationSeconds: media.durationSeconds,
2602
+ droppedUnchanged: media.droppedUnchanged
2313
2603
  },
2314
- framesMetadata: {
2604
+ frames: {
2315
2605
  count: media.frames.length,
2316
2606
  timestampsSeconds: timestamps,
2317
- durationSeconds: media.durationSeconds
2607
+ durationSeconds: media.durationSeconds,
2608
+ droppedUnchanged: media.droppedUnchanged
2318
2609
  }
2319
2610
  };
2320
2611
  }
2612
+ async function sendMedia(driver, dispatch, prompt, options) {
2613
+ if (dispatch.kind === "native-video") {
2614
+ const response2 = await timedSendVideoMessage(driver, dispatch.video, prompt, options);
2615
+ const { durationSeconds, fps, mimeType } = dispatch.video;
2616
+ return {
2617
+ response: response2,
2618
+ metadata: { video: { durationSeconds, fps, mimeType, delivery: response2.delivery } }
2619
+ };
2620
+ }
2621
+ const response = await timedSendMessage(driver, dispatch.images, prompt, options);
2622
+ return { response, metadata: dispatch.frames ? { frames: dispatch.frames } : {} };
2623
+ }
2624
+ function resolveNativeVideo(input, videoOptions, driver, provider) {
2625
+ const mode = videoOptions?.mode ?? "auto";
2626
+ if (mode !== "auto" && mode !== "native" && mode !== "frames") {
2627
+ throw new VisualAIConfigError(
2628
+ `Invalid video mode: ${mode}. Expected "auto", "native", or "frames".`
2629
+ );
2630
+ }
2631
+ const supported = typeof driver.sendVideoMessage === "function";
2632
+ if (mode === "frames") return false;
2633
+ if (mode === "auto") return supported;
2634
+ if (!supported && !isFramesInput(input) && isVideoInput(input)) {
2635
+ throw new VisualAIConfigError(
2636
+ `Native video delivery is not supported by the "${provider}" provider. Use a Google model, or set video: { mode: "frames" } to sample frames instead.`
2637
+ );
2638
+ }
2639
+ return true;
2640
+ }
2321
2641
  function visualAI(config = {}) {
2322
2642
  const resolvedConfig = resolveConfig(config);
2323
2643
  const driverConfig = {
@@ -2355,38 +2675,55 @@ function visualAI(config = {}) {
2355
2675
  throw new VisualAIConfigError("At least one statement is required for check()");
2356
2676
  }
2357
2677
  return withErrorDebug(resolvedConfig, "check", async () => {
2358
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2359
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2678
+ const nativeVideo = resolveNativeVideo(
2679
+ input,
2680
+ options?.video,
2681
+ driver,
2682
+ resolvedConfig.provider
2683
+ );
2684
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2685
+ const dispatch = mediaToProviderInputs(media);
2360
2686
  const prompt = buildCheckPrompt(stmts, {
2361
2687
  instructions: options?.instructions,
2362
- media: mediaContext
2688
+ media: dispatch.mediaContext
2363
2689
  });
2364
2690
  debugLog(resolvedConfig, "check prompt", prompt, "prompt");
2365
- const response = await timedSendMessage(driver, images, prompt, checkSchemaOptions);
2691
+ const { response, metadata } = await sendMedia(
2692
+ driver,
2693
+ dispatch,
2694
+ prompt,
2695
+ checkSchemaOptions
2696
+ );
2366
2697
  debugLog(resolvedConfig, "check response", response.text, "response");
2367
2698
  const result = parseCheckResponse(response.text);
2368
2699
  return {
2369
2700
  ...result,
2370
- ...framesMetadata ? { frames: framesMetadata } : {},
2701
+ ...metadata,
2371
2702
  usage: processUsage("check", response.usage, response.durationSeconds, resolvedConfig)
2372
2703
  };
2373
2704
  });
2374
2705
  },
2375
2706
  async ask(input, userPrompt, options) {
2376
2707
  return withErrorDebug(resolvedConfig, "ask", async () => {
2377
- const media = await normalizeMedia(input, options?.video, maxImageDimension);
2378
- const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2708
+ const nativeVideo = resolveNativeVideo(
2709
+ input,
2710
+ options?.video,
2711
+ driver,
2712
+ resolvedConfig.provider
2713
+ );
2714
+ const media = await normalizeMedia(input, options?.video, maxImageDimension, nativeVideo);
2715
+ const dispatch = mediaToProviderInputs(media);
2379
2716
  const prompt = buildAskPrompt(userPrompt, {
2380
2717
  instructions: options?.instructions,
2381
- media: mediaContext
2718
+ media: dispatch.mediaContext
2382
2719
  });
2383
2720
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2384
- const response = await timedSendMessage(driver, images, prompt, askSchemaOptions);
2721
+ const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2385
2722
  debugLog(resolvedConfig, "ask response", response.text, "response");
2386
2723
  const result = parseAskResponse(response.text);
2387
2724
  return {
2388
2725
  ...result,
2389
- ...framesMetadata ? { frames: framesMetadata } : {},
2726
+ ...metadata,
2390
2727
  usage: processUsage("ask", response.usage, response.durationSeconds, resolvedConfig)
2391
2728
  };
2392
2729
  });