visual-ai-assertions 0.16.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5,6 +5,13 @@ var ReasoningEffort = {
5
5
  HIGH: "high",
6
6
  XHIGH: "xhigh"
7
7
  };
8
+ var ImageDetail = {
9
+ AUTO: "auto",
10
+ LOW: "low",
11
+ HIGH: "high"
12
+ };
13
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
14
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
8
15
  var Provider = {
9
16
  ANTHROPIC: "anthropic",
10
17
  OPENAI: "openai",
@@ -14,6 +21,7 @@ var Provider = {
14
21
  var Model = {
15
22
  Anthropic: {
16
23
  FABLE_5: "claude-fable-5",
24
+ OPUS_5: "claude-opus-5",
17
25
  OPUS_4_8: "claude-opus-4-8",
18
26
  OPUS_4_7: "claude-opus-4-7",
19
27
  OPUS_4_6: "claude-opus-4-6",
@@ -34,6 +42,8 @@ var Model = {
34
42
  GPT_5_MINI: "gpt-5-mini"
35
43
  },
36
44
  Google: {
45
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
46
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
37
47
  GEMINI_3_6_FLASH: "gemini-3.6-flash",
38
48
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
39
49
  GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
@@ -47,9 +57,11 @@ var Model = {
47
57
  * recognizes them. All listed models accept image input.
48
58
  */
49
59
  OpenRouter: {
60
+ GROK_4_6: "x-ai/grok-4.6",
50
61
  GROK_4_5: "x-ai/grok-4.5",
51
62
  KIMI_K3: "moonshotai/kimi-k3",
52
63
  KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
64
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
53
65
  QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
54
66
  QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
55
67
  }
@@ -532,6 +544,7 @@ function parseRetryAfter(value) {
532
544
  // src/providers/anthropic.ts
533
545
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
534
546
  Model.Anthropic.FABLE_5,
547
+ Model.Anthropic.OPUS_5,
535
548
  Model.Anthropic.OPUS_4_8,
536
549
  Model.Anthropic.OPUS_4_7,
537
550
  Model.Anthropic.SONNET_5
@@ -627,7 +640,10 @@ var AnthropicDriver = class {
627
640
  text,
628
641
  usage: {
629
642
  inputTokens: message.usage.input_tokens,
630
- outputTokens: message.usage.output_tokens
643
+ outputTokens: message.usage.output_tokens,
644
+ ...message.usage.cache_read_input_tokens !== void 0 && {
645
+ cachedInputTokens: message.usage.cache_read_input_tokens
646
+ }
631
647
  }
632
648
  };
633
649
  } catch (err) {
@@ -644,23 +660,41 @@ function needsCodeExecution(model) {
644
660
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
645
661
  }
646
662
  var GOOGLE_THINKING_LEVEL = {
647
- low: "minimal",
648
- medium: "low",
649
- high: "medium",
663
+ low: "low",
664
+ medium: "medium",
665
+ high: "high",
650
666
  xhigh: "high"
651
667
  };
668
+ var GOOGLE_MEDIA_RESOLUTION = {
669
+ low: "MEDIA_RESOLUTION_LOW",
670
+ high: "MEDIA_RESOLUTION_HIGH"
671
+ };
672
+ function toGeminiUsage(um) {
673
+ if (!um) return void 0;
674
+ const thoughts = um.thoughtsTokenCount ?? 0;
675
+ return {
676
+ inputTokens: um.promptTokenCount ?? 0,
677
+ outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
678
+ ...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
679
+ ...um.cachedContentTokenCount !== void 0 && {
680
+ cachedInputTokens: um.cachedContentTokenCount
681
+ }
682
+ };
683
+ }
652
684
  var GoogleDriver = class {
653
685
  client;
654
686
  model;
655
687
  maxTokens;
656
688
  apiKeyOrEnv;
657
689
  reasoningEffort;
690
+ imageDetail;
658
691
  constructor(config) {
659
692
  this.model = config.model;
660
693
  this.maxTokens = config.maxTokens;
661
694
  this.client = null;
662
695
  this.apiKeyOrEnv = config.apiKey;
663
696
  this.reasoningEffort = config.reasoningEffort;
697
+ this.imageDetail = config.imageDetail;
664
698
  }
665
699
  toGeminiParts(images) {
666
700
  return images.map((img) => ({
@@ -700,6 +734,9 @@ var GoogleDriver = class {
700
734
  thinkingConfig: {
701
735
  thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
702
736
  }
737
+ },
738
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
739
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
703
740
  }
704
741
  }
705
742
  });
@@ -717,14 +754,9 @@ var GoogleDriver = class {
717
754
  );
718
755
  }
719
756
  const text = response.text ?? "";
720
- const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
721
757
  return {
722
758
  text,
723
- usage: response.usageMetadata ? {
724
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
725
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
726
- ...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
727
- } : void 0
759
+ usage: toGeminiUsage(response.usageMetadata)
728
760
  };
729
761
  } catch (err) {
730
762
  if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
@@ -755,10 +787,7 @@ var GoogleDriver = class {
755
787
  return {
756
788
  imageData: Buffer.from(imagePart.inlineData.data, "base64"),
757
789
  mimeType: imagePart.inlineData.mimeType,
758
- usage: response.usageMetadata ? {
759
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
760
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
761
- } : void 0
790
+ usage: toGeminiUsage(response.usageMetadata)
762
791
  };
763
792
  } catch (err) {
764
793
  if (err instanceof VisualAIProviderError) throw err;
@@ -774,12 +803,14 @@ var OpenAIDriver = class {
774
803
  maxTokens;
775
804
  apiKeyOrEnv;
776
805
  reasoningEffort;
806
+ imageDetail;
777
807
  constructor(config) {
778
808
  this.model = config.model;
779
809
  this.maxTokens = config.maxTokens;
780
810
  this.client = null;
781
811
  this.apiKeyOrEnv = config.apiKey;
782
812
  this.reasoningEffort = config.reasoningEffort;
813
+ this.imageDetail = config.imageDetail;
783
814
  }
784
815
  async getClient() {
785
816
  if (this.client) return this.client;
@@ -801,9 +832,11 @@ var OpenAIDriver = class {
801
832
  }
802
833
  async sendMessage(images, prompt, options) {
803
834
  const client = await this.getClient();
835
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
804
836
  const imageBlocks = images.map((img) => ({
805
837
  type: "input_image",
806
- image_url: `data:${img.mimeType};base64,${img.base64}`
838
+ image_url: `data:${img.mimeType};base64,${img.base64}`,
839
+ ...detail ? { detail } : {}
807
840
  }));
808
841
  try {
809
842
  const format = options?.responseSchema ? {
@@ -828,21 +861,23 @@ var OpenAIDriver = class {
828
861
  }
829
862
  const response = await client.responses.create(requestParams);
830
863
  if (response.status && response.status !== "completed") {
831
- const detail = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
864
+ const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
832
865
  throw new VisualAITruncationError(
833
- `Response truncated: OpenAI returned status "${response.status}"${detail}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
866
+ `Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
834
867
  response.output_text ?? "",
835
868
  this.maxTokens
836
869
  );
837
870
  }
838
871
  const text = response.output_text ?? "";
839
872
  const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
873
+ const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
840
874
  return {
841
875
  text,
842
876
  usage: response.usage ? {
843
877
  inputTokens: response.usage.input_tokens,
844
878
  outputTokens: response.usage.output_tokens,
845
- ...reasoningTokens !== void 0 && { reasoningTokens }
879
+ ...reasoningTokens !== void 0 && { reasoningTokens },
880
+ ...cachedInputTokens !== void 0 && { cachedInputTokens }
846
881
  } : void 0
847
882
  };
848
883
  } catch (err) {
@@ -866,12 +901,14 @@ var OpenRouterDriver = class {
866
901
  maxTokens;
867
902
  apiKeyOrEnv;
868
903
  reasoningEffort;
904
+ imageDetail;
869
905
  constructor(config) {
870
906
  this.model = config.model;
871
907
  this.maxTokens = config.maxTokens;
872
908
  this.client = null;
873
909
  this.apiKeyOrEnv = config.apiKey;
874
910
  this.reasoningEffort = config.reasoningEffort;
911
+ this.imageDetail = config.imageDetail;
875
912
  }
876
913
  async getClient() {
877
914
  if (this.client) return this.client;
@@ -895,9 +932,13 @@ var OpenRouterDriver = class {
895
932
  }
896
933
  async sendMessage(images, prompt, options) {
897
934
  const client = await this.getClient();
935
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
898
936
  const imageParts = images.map((img) => ({
899
937
  type: "image_url",
900
- image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
938
+ image_url: {
939
+ url: `data:${img.mimeType};base64,${img.base64}`,
940
+ ...detail ? { detail } : {}
941
+ }
901
942
  }));
902
943
  try {
903
944
  const responseFormat = options?.responseSchema ? {
@@ -938,12 +979,16 @@ var OpenRouterDriver = class {
938
979
  );
939
980
  }
940
981
  const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
982
+ const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
983
+ const cost = response.usage?.cost;
941
984
  return {
942
985
  text,
943
986
  usage: response.usage ? {
944
987
  inputTokens: response.usage.prompt_tokens,
945
988
  outputTokens: response.usage.completion_tokens,
946
- ...reasoningTokens !== void 0 && { reasoningTokens }
989
+ ...reasoningTokens !== void 0 && { reasoningTokens },
990
+ ...cachedInputTokens !== void 0 && { cachedInputTokens },
991
+ ...cost !== void 0 && { cost }
947
992
  } : void 0
948
993
  };
949
994
  } catch (err) {
@@ -1029,6 +1074,8 @@ function resolveConfig(config) {
1029
1074
  model,
1030
1075
  maxTokens,
1031
1076
  reasoningEffort: config.reasoningEffort,
1077
+ maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1078
+ imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1032
1079
  debug,
1033
1080
  debugPrompt,
1034
1081
  debugResponse,
@@ -1043,6 +1090,10 @@ var PRICING_TABLE = {
1043
1090
  inputPricePerToken: 10 / PER_MILLION,
1044
1091
  outputPricePerToken: 50 / PER_MILLION
1045
1092
  },
1093
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1094
+ inputPricePerToken: 5 / PER_MILLION,
1095
+ outputPricePerToken: 25 / PER_MILLION
1096
+ },
1046
1097
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
1047
1098
  inputPricePerToken: 5 / PER_MILLION,
1048
1099
  outputPricePerToken: 25 / PER_MILLION
@@ -1072,12 +1123,12 @@ var PRICING_TABLE = {
1072
1123
  outputPricePerToken: 30 / PER_MILLION
1073
1124
  },
1074
1125
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
1075
- inputPricePerToken: 2.5 / PER_MILLION,
1076
- outputPricePerToken: 15 / PER_MILLION
1126
+ inputPricePerToken: 2 / PER_MILLION,
1127
+ outputPricePerToken: 12 / PER_MILLION
1077
1128
  },
1078
1129
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
1079
- inputPricePerToken: 1 / PER_MILLION,
1080
- outputPricePerToken: 6 / PER_MILLION
1130
+ inputPricePerToken: 0.2 / PER_MILLION,
1131
+ outputPricePerToken: 1.2 / PER_MILLION
1081
1132
  },
1082
1133
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
1083
1134
  inputPricePerToken: 5 / PER_MILLION,
@@ -1107,6 +1158,18 @@ var PRICING_TABLE = {
1107
1158
  inputPricePerToken: 0.25 / PER_MILLION,
1108
1159
  outputPricePerToken: 2 / PER_MILLION
1109
1160
  },
1161
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1162
+ // on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
1163
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
1164
+ inputPricePerToken: 0.75 / PER_MILLION,
1165
+ outputPricePerToken: 3.75 / PER_MILLION
1166
+ },
1167
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1168
+ // on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
1169
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
1170
+ inputPricePerToken: 0.75 / PER_MILLION,
1171
+ outputPricePerToken: 3.75 / PER_MILLION
1172
+ },
1110
1173
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1111
1174
  inputPricePerToken: 1.5 / PER_MILLION,
1112
1175
  outputPricePerToken: 7.5 / PER_MILLION
@@ -1133,6 +1196,10 @@ var PRICING_TABLE = {
1133
1196
  },
1134
1197
  // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1135
1198
  // against https://openrouter.ai/api/v1/models).
1199
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1200
+ inputPricePerToken: 2 / PER_MILLION,
1201
+ outputPricePerToken: 6 / PER_MILLION
1202
+ },
1136
1203
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1137
1204
  inputPricePerToken: 2 / PER_MILLION,
1138
1205
  outputPricePerToken: 6 / PER_MILLION
@@ -1145,6 +1212,10 @@ var PRICING_TABLE = {
1145
1212
  inputPricePerToken: 0.82 / PER_MILLION,
1146
1213
  outputPricePerToken: 3.75 / PER_MILLION
1147
1214
  },
1215
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
1216
+ inputPricePerToken: 2 / PER_MILLION,
1217
+ outputPricePerToken: 6 / PER_MILLION
1218
+ },
1148
1219
  [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1149
1220
  inputPricePerToken: 0.32 / PER_MILLION,
1150
1221
  outputPricePerToken: 1.28 / PER_MILLION
@@ -1174,8 +1245,9 @@ function usageLog(config, method, usage) {
1174
1245
  const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
1175
1246
  const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
1176
1247
  const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
1248
+ const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
1177
1249
  process.stderr.write(
1178
- `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1250
+ `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1179
1251
  `
1180
1252
  );
1181
1253
  }
@@ -1186,7 +1258,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
1186
1258
  inputTokens,
1187
1259
  outputTokens,
1188
1260
  ...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
1261
+ ...rawUsage?.cachedInputTokens !== void 0 && {
1262
+ cachedInputTokens: rawUsage.cachedInputTokens
1263
+ },
1189
1264
  estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
1265
+ ...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
1190
1266
  durationSeconds
1191
1267
  };
1192
1268
  usageLog(config, method, usage);
@@ -1230,7 +1306,9 @@ import sharp from "sharp";
1230
1306
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1231
1307
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1232
1308
  Model.Google.GEMINI_3_5_FLASH,
1233
- Model.Google.GEMINI_3_6_FLASH
1309
+ Model.Google.GEMINI_3_6_FLASH,
1310
+ Model.Google.GEMINI_3_7_FLASH,
1311
+ Model.Google.GEMINI_3_8_FLASH
1234
1312
  ]);
1235
1313
  async function generateAiDiff(imgA, imgB, model, driver) {
1236
1314
  if (!driver.generateImage) {
@@ -1320,7 +1398,6 @@ var EXTENSION_TO_MIME = {
1320
1398
  ".webp": "image/webp",
1321
1399
  ".gif": "image/gif"
1322
1400
  };
1323
- var MAX_DIMENSION = 1568;
1324
1401
  var URL_FETCH_TIMEOUT_MS = 1e4;
1325
1402
  function isSupportedMimeType(value) {
1326
1403
  return SUPPORTED_FORMATS.has(value);
@@ -1344,14 +1421,14 @@ function detectMimeType(data) {
1344
1421
  }
1345
1422
  throw new VisualAIImageError("Unable to detect image format from file content");
1346
1423
  }
1347
- async function resizeIfNeeded(data, mimeType) {
1424
+ async function resizeIfNeeded(data, mimeType, maxDimension) {
1348
1425
  if (mimeType === "image/gif") {
1349
1426
  return data;
1350
1427
  }
1351
1428
  if (mimeType === "image/png" && data.length >= 24) {
1352
1429
  const width2 = data.readUInt32BE(16);
1353
1430
  const height2 = data.readUInt32BE(20);
1354
- if (width2 <= MAX_DIMENSION && height2 <= MAX_DIMENSION) {
1431
+ if (width2 <= maxDimension && height2 <= maxDimension) {
1355
1432
  return data;
1356
1433
  }
1357
1434
  }
@@ -1359,12 +1436,12 @@ async function resizeIfNeeded(data, mimeType) {
1359
1436
  const metadata = await pipeline.metadata();
1360
1437
  const width = metadata.width ?? 0;
1361
1438
  const height = metadata.height ?? 0;
1362
- if (width <= MAX_DIMENSION && height <= MAX_DIMENSION) {
1439
+ if (width <= maxDimension && height <= maxDimension) {
1363
1440
  return data;
1364
1441
  }
1365
1442
  return pipeline.resize({
1366
- width: MAX_DIMENSION,
1367
- height: MAX_DIMENSION,
1443
+ width: maxDimension,
1444
+ height: maxDimension,
1368
1445
  fit: "inside",
1369
1446
  withoutEnlargement: true
1370
1447
  }).toBuffer();
@@ -1419,7 +1496,7 @@ function loadFromBase64(input) {
1419
1496
  }
1420
1497
  return { data, mimeType: mimeType ?? detectMimeType(data) };
1421
1498
  }
1422
- async function normalizeImage(input) {
1499
+ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1423
1500
  let data;
1424
1501
  let mimeType;
1425
1502
  if (Buffer.isBuffer(input)) {
@@ -1448,7 +1525,7 @@ async function normalizeImage(input) {
1448
1525
  "Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
1449
1526
  );
1450
1527
  }
1451
- data = await resizeIfNeeded(data, mimeType);
1528
+ data = await resizeIfNeeded(data, mimeType, maxDimension);
1452
1529
  let cachedBase64;
1453
1530
  return {
1454
1531
  data,
@@ -1752,7 +1829,7 @@ async function probeDurationSeconds(videoPath) {
1752
1829
  });
1753
1830
  });
1754
1831
  }
1755
- async function extractFrames(videoPath, options = {}) {
1832
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
1756
1833
  const fps = options.fps ?? DEFAULT_FPS;
1757
1834
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1758
1835
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1781,7 +1858,7 @@ async function extractFrames(videoPath, options = {}) {
1781
1858
  }
1782
1859
  const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
1783
1860
  try {
1784
- const filter = `fps=${fps},scale='if(gt(iw,ih),min(${FRAME_MAX_DIMENSION},iw),-2)':'if(gt(iw,ih),-2,min(${FRAME_MAX_DIMENSION},ih))':flags=area`;
1861
+ const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
1785
1862
  await new Promise((resolve2, reject) => {
1786
1863
  let settled = false;
1787
1864
  const cmd = ffmpeg(videoPath);
@@ -1879,7 +1956,7 @@ function isFramesInput(input) {
1879
1956
  function isTimestampedFrameInput(frame) {
1880
1957
  return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
1881
1958
  }
1882
- async function normalizeFrames(input) {
1959
+ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1883
1960
  const rawFrames = input.frames;
1884
1961
  const fps = input.fps ?? DEFAULT_FPS;
1885
1962
  if (rawFrames.length === 0) {
@@ -1904,7 +1981,7 @@ async function normalizeFrames(input) {
1904
1981
  `Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
1905
1982
  );
1906
1983
  }
1907
- const image = await normalizeImage(imageInput);
1984
+ const image = await normalizeImage(imageInput, maxDimension);
1908
1985
  return {
1909
1986
  data: image.data,
1910
1987
  mimeType: image.mimeType,
@@ -1920,14 +1997,14 @@ async function normalizeFrames(input) {
1920
1997
  await saveDebugFrames(frames);
1921
1998
  return { kind: "video", frames, durationSeconds };
1922
1999
  }
1923
- async function normalizeMedia(input, videoOptions) {
2000
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1924
2001
  if (isFramesInput(input)) {
1925
- return normalizeFrames(input);
2002
+ return normalizeFrames(input, maxDimension);
1926
2003
  }
1927
2004
  if (isVideoInput(input)) {
1928
2005
  const { path, cleanup } = await resolveVideoToPath(input);
1929
2006
  try {
1930
- const { frames, durationSeconds } = await extractFrames(path, videoOptions);
2007
+ const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
1931
2008
  await saveDebugFrames(frames);
1932
2009
  return { kind: "video", frames, durationSeconds };
1933
2010
  } finally {
@@ -1937,7 +2014,7 @@ async function normalizeMedia(input, videoOptions) {
1937
2014
  }
1938
2015
  }
1939
2016
  }
1940
- const image = await normalizeImage(input);
2017
+ const image = await normalizeImage(input, maxDimension);
1941
2018
  return { kind: "image", image };
1942
2019
  }
1943
2020
 
@@ -1978,7 +2055,17 @@ var UsageInfoSchema = z.object({
1978
2055
  outputTokens: z.number(),
1979
2056
  /** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
1980
2057
  reasoningTokens: z.number().optional(),
2058
+ /**
2059
+ * Prompt tokens served from the provider's cache, when reported. Informational
2060
+ * only — `estimatedCost` does not apply a cache discount, because providers
2061
+ * differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
2062
+ * Google) or billed as a separate bucket alongside it (Anthropic).
2063
+ */
2064
+ cachedInputTokens: z.number().optional(),
2065
+ /** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
1981
2066
  estimatedCost: z.number().optional(),
2067
+ /** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
2068
+ reportedCost: z.number().optional(),
1982
2069
  durationSeconds: z.number().nonnegative().optional()
1983
2070
  });
1984
2071
  var BaseResultSchema = z.object({
@@ -2118,16 +2205,18 @@ function visualAI(config = {}) {
2118
2205
  apiKey: resolvedConfig.apiKey,
2119
2206
  model: resolvedConfig.model,
2120
2207
  maxTokens: resolvedConfig.maxTokens,
2121
- reasoningEffort: resolvedConfig.reasoningEffort
2208
+ reasoningEffort: resolvedConfig.reasoningEffort,
2209
+ imageDetail: resolvedConfig.imageDetail
2122
2210
  };
2123
2211
  const driver = createDriver(resolvedConfig.provider, driverConfig);
2212
+ const maxImageDimension = resolvedConfig.maxImageDimension;
2124
2213
  async function checkElementsVisibility(image, elements, visible, options) {
2125
2214
  const methodName = visible ? "elementsVisible" : "elementsHidden";
2126
2215
  if (elements.length === 0) {
2127
2216
  throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
2128
2217
  }
2129
2218
  return withErrorDebug(resolvedConfig, methodName, async () => {
2130
- const img = await normalizeImage(image);
2219
+ const img = await normalizeImage(image, maxImageDimension);
2131
2220
  const prompt = buildElementsVisibilityPrompt(elements, visible, options);
2132
2221
  debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
2133
2222
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2146,7 +2235,7 @@ function visualAI(config = {}) {
2146
2235
  throw new VisualAIConfigError("At least one statement is required for check()");
2147
2236
  }
2148
2237
  return withErrorDebug(resolvedConfig, "check", async () => {
2149
- const media = await normalizeMedia(input, options?.video);
2238
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2150
2239
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2151
2240
  const prompt = buildCheckPrompt(stmts, {
2152
2241
  instructions: options?.instructions,
@@ -2165,7 +2254,7 @@ function visualAI(config = {}) {
2165
2254
  },
2166
2255
  async ask(input, userPrompt, options) {
2167
2256
  return withErrorDebug(resolvedConfig, "ask", async () => {
2168
- const media = await normalizeMedia(input, options?.video);
2257
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2169
2258
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2170
2259
  const prompt = buildAskPrompt(userPrompt, {
2171
2260
  instructions: options?.instructions,
@@ -2184,7 +2273,10 @@ function visualAI(config = {}) {
2184
2273
  },
2185
2274
  async compare(imageA, imageB, options) {
2186
2275
  return withErrorDebug(resolvedConfig, "compare", async () => {
2187
- const [imgA, imgB] = await Promise.all([normalizeImage(imageA), normalizeImage(imageB)]);
2276
+ const [imgA, imgB] = await Promise.all([
2277
+ normalizeImage(imageA, maxImageDimension),
2278
+ normalizeImage(imageB, maxImageDimension)
2279
+ ]);
2188
2280
  const prompt = buildComparePrompt({
2189
2281
  userPrompt: options?.prompt,
2190
2282
  instructions: options?.instructions
@@ -2222,7 +2314,7 @@ function visualAI(config = {}) {
2222
2314
  },
2223
2315
  async accessibility(image, options) {
2224
2316
  return withErrorDebug(resolvedConfig, "accessibility", async () => {
2225
- const img = await normalizeImage(image);
2317
+ const img = await normalizeImage(image, maxImageDimension);
2226
2318
  const prompt = buildAccessibilityPrompt(options);
2227
2319
  debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
2228
2320
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2241,7 +2333,7 @@ function visualAI(config = {}) {
2241
2333
  },
2242
2334
  async layout(image, options) {
2243
2335
  return withErrorDebug(resolvedConfig, "layout", async () => {
2244
- const img = await normalizeImage(image);
2336
+ const img = await normalizeImage(image, maxImageDimension);
2245
2337
  const prompt = buildLayoutPrompt(options);
2246
2338
  debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
2247
2339
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2255,7 +2347,7 @@ function visualAI(config = {}) {
2255
2347
  },
2256
2348
  async pageLoad(image, options) {
2257
2349
  return withErrorDebug(resolvedConfig, "pageLoad", async () => {
2258
- const img = await normalizeImage(image);
2350
+ const img = await normalizeImage(image, maxImageDimension);
2259
2351
  const prompt = buildPageLoadPrompt(options);
2260
2352
  debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
2261
2353
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2269,7 +2361,7 @@ function visualAI(config = {}) {
2269
2361
  },
2270
2362
  async content(image, options) {
2271
2363
  return withErrorDebug(resolvedConfig, "content", async () => {
2272
- const img = await normalizeImage(image);
2364
+ const img = await normalizeImage(image, maxImageDimension);
2273
2365
  const prompt = buildContentPrompt(options);
2274
2366
  debugLog(resolvedConfig, "content prompt", prompt, "prompt");
2275
2367
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2347,6 +2439,7 @@ export {
2347
2439
  ConfidenceSchema,
2348
2440
  Content,
2349
2441
  DEFAULT_MODELS,
2442
+ ImageDetail,
2350
2443
  IssueCategorySchema,
2351
2444
  IssuePrioritySchema,
2352
2445
  IssueSchema,