visual-ai-assertions 0.16.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5,6 +5,13 @@ var ReasoningEffort = {
5
5
  HIGH: "high",
6
6
  XHIGH: "xhigh"
7
7
  };
8
+ var ImageDetail = {
9
+ AUTO: "auto",
10
+ LOW: "low",
11
+ HIGH: "high"
12
+ };
13
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
14
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
8
15
  var Provider = {
9
16
  ANTHROPIC: "anthropic",
10
17
  OPENAI: "openai",
@@ -14,6 +21,7 @@ var Provider = {
14
21
  var Model = {
15
22
  Anthropic: {
16
23
  FABLE_5: "claude-fable-5",
24
+ OPUS_5: "claude-opus-5",
17
25
  OPUS_4_8: "claude-opus-4-8",
18
26
  OPUS_4_7: "claude-opus-4-7",
19
27
  OPUS_4_6: "claude-opus-4-6",
@@ -34,6 +42,8 @@ var Model = {
34
42
  GPT_5_MINI: "gpt-5-mini"
35
43
  },
36
44
  Google: {
45
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
46
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
37
47
  GEMINI_3_6_FLASH: "gemini-3.6-flash",
38
48
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
39
49
  GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
@@ -47,9 +57,12 @@ var Model = {
47
57
  * recognizes them. All listed models accept image input.
48
58
  */
49
59
  OpenRouter: {
60
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
61
+ GROK_4_6: "x-ai/grok-4.6",
50
62
  GROK_4_5: "x-ai/grok-4.5",
51
63
  KIMI_K3: "moonshotai/kimi-k3",
52
64
  KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
65
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
53
66
  QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
54
67
  QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
55
68
  }
@@ -532,6 +545,7 @@ function parseRetryAfter(value) {
532
545
  // src/providers/anthropic.ts
533
546
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
534
547
  Model.Anthropic.FABLE_5,
548
+ Model.Anthropic.OPUS_5,
535
549
  Model.Anthropic.OPUS_4_8,
536
550
  Model.Anthropic.OPUS_4_7,
537
551
  Model.Anthropic.SONNET_5
@@ -627,7 +641,10 @@ var AnthropicDriver = class {
627
641
  text,
628
642
  usage: {
629
643
  inputTokens: message.usage.input_tokens,
630
- outputTokens: message.usage.output_tokens
644
+ outputTokens: message.usage.output_tokens,
645
+ ...message.usage.cache_read_input_tokens !== void 0 && {
646
+ cachedInputTokens: message.usage.cache_read_input_tokens
647
+ }
631
648
  }
632
649
  };
633
650
  } catch (err) {
@@ -644,23 +661,41 @@ function needsCodeExecution(model) {
644
661
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
645
662
  }
646
663
  var GOOGLE_THINKING_LEVEL = {
647
- low: "minimal",
648
- medium: "low",
649
- high: "medium",
664
+ low: "low",
665
+ medium: "medium",
666
+ high: "high",
650
667
  xhigh: "high"
651
668
  };
669
+ var GOOGLE_MEDIA_RESOLUTION = {
670
+ low: "MEDIA_RESOLUTION_LOW",
671
+ high: "MEDIA_RESOLUTION_HIGH"
672
+ };
673
+ function toGeminiUsage(um) {
674
+ if (!um) return void 0;
675
+ const thoughts = um.thoughtsTokenCount ?? 0;
676
+ return {
677
+ inputTokens: um.promptTokenCount ?? 0,
678
+ outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
679
+ ...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
680
+ ...um.cachedContentTokenCount !== void 0 && {
681
+ cachedInputTokens: um.cachedContentTokenCount
682
+ }
683
+ };
684
+ }
652
685
  var GoogleDriver = class {
653
686
  client;
654
687
  model;
655
688
  maxTokens;
656
689
  apiKeyOrEnv;
657
690
  reasoningEffort;
691
+ imageDetail;
658
692
  constructor(config) {
659
693
  this.model = config.model;
660
694
  this.maxTokens = config.maxTokens;
661
695
  this.client = null;
662
696
  this.apiKeyOrEnv = config.apiKey;
663
697
  this.reasoningEffort = config.reasoningEffort;
698
+ this.imageDetail = config.imageDetail;
664
699
  }
665
700
  toGeminiParts(images) {
666
701
  return images.map((img) => ({
@@ -700,6 +735,9 @@ var GoogleDriver = class {
700
735
  thinkingConfig: {
701
736
  thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
702
737
  }
738
+ },
739
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
740
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
703
741
  }
704
742
  }
705
743
  });
@@ -717,14 +755,9 @@ var GoogleDriver = class {
717
755
  );
718
756
  }
719
757
  const text = response.text ?? "";
720
- const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
721
758
  return {
722
759
  text,
723
- usage: response.usageMetadata ? {
724
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
725
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
726
- ...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
727
- } : void 0
760
+ usage: toGeminiUsage(response.usageMetadata)
728
761
  };
729
762
  } catch (err) {
730
763
  if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
@@ -755,10 +788,7 @@ var GoogleDriver = class {
755
788
  return {
756
789
  imageData: Buffer.from(imagePart.inlineData.data, "base64"),
757
790
  mimeType: imagePart.inlineData.mimeType,
758
- usage: response.usageMetadata ? {
759
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
760
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
761
- } : void 0
791
+ usage: toGeminiUsage(response.usageMetadata)
762
792
  };
763
793
  } catch (err) {
764
794
  if (err instanceof VisualAIProviderError) throw err;
@@ -774,12 +804,14 @@ var OpenAIDriver = class {
774
804
  maxTokens;
775
805
  apiKeyOrEnv;
776
806
  reasoningEffort;
807
+ imageDetail;
777
808
  constructor(config) {
778
809
  this.model = config.model;
779
810
  this.maxTokens = config.maxTokens;
780
811
  this.client = null;
781
812
  this.apiKeyOrEnv = config.apiKey;
782
813
  this.reasoningEffort = config.reasoningEffort;
814
+ this.imageDetail = config.imageDetail;
783
815
  }
784
816
  async getClient() {
785
817
  if (this.client) return this.client;
@@ -801,9 +833,11 @@ var OpenAIDriver = class {
801
833
  }
802
834
  async sendMessage(images, prompt, options) {
803
835
  const client = await this.getClient();
836
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
804
837
  const imageBlocks = images.map((img) => ({
805
838
  type: "input_image",
806
- image_url: `data:${img.mimeType};base64,${img.base64}`
839
+ image_url: `data:${img.mimeType};base64,${img.base64}`,
840
+ ...detail ? { detail } : {}
807
841
  }));
808
842
  try {
809
843
  const format = options?.responseSchema ? {
@@ -828,21 +862,23 @@ var OpenAIDriver = class {
828
862
  }
829
863
  const response = await client.responses.create(requestParams);
830
864
  if (response.status && response.status !== "completed") {
831
- const detail = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
865
+ const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
832
866
  throw new VisualAITruncationError(
833
- `Response truncated: OpenAI returned status "${response.status}"${detail}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
867
+ `Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
834
868
  response.output_text ?? "",
835
869
  this.maxTokens
836
870
  );
837
871
  }
838
872
  const text = response.output_text ?? "";
839
873
  const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
874
+ const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
840
875
  return {
841
876
  text,
842
877
  usage: response.usage ? {
843
878
  inputTokens: response.usage.input_tokens,
844
879
  outputTokens: response.usage.output_tokens,
845
- ...reasoningTokens !== void 0 && { reasoningTokens }
880
+ ...reasoningTokens !== void 0 && { reasoningTokens },
881
+ ...cachedInputTokens !== void 0 && { cachedInputTokens }
846
882
  } : void 0
847
883
  };
848
884
  } catch (err) {
@@ -866,12 +902,14 @@ var OpenRouterDriver = class {
866
902
  maxTokens;
867
903
  apiKeyOrEnv;
868
904
  reasoningEffort;
905
+ imageDetail;
869
906
  constructor(config) {
870
907
  this.model = config.model;
871
908
  this.maxTokens = config.maxTokens;
872
909
  this.client = null;
873
910
  this.apiKeyOrEnv = config.apiKey;
874
911
  this.reasoningEffort = config.reasoningEffort;
912
+ this.imageDetail = config.imageDetail;
875
913
  }
876
914
  async getClient() {
877
915
  if (this.client) return this.client;
@@ -895,9 +933,13 @@ var OpenRouterDriver = class {
895
933
  }
896
934
  async sendMessage(images, prompt, options) {
897
935
  const client = await this.getClient();
936
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
898
937
  const imageParts = images.map((img) => ({
899
938
  type: "image_url",
900
- image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
939
+ image_url: {
940
+ url: `data:${img.mimeType};base64,${img.base64}`,
941
+ ...detail ? { detail } : {}
942
+ }
901
943
  }));
902
944
  try {
903
945
  const responseFormat = options?.responseSchema ? {
@@ -938,12 +980,16 @@ var OpenRouterDriver = class {
938
980
  );
939
981
  }
940
982
  const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
983
+ const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
984
+ const cost = response.usage?.cost;
941
985
  return {
942
986
  text,
943
987
  usage: response.usage ? {
944
988
  inputTokens: response.usage.prompt_tokens,
945
989
  outputTokens: response.usage.completion_tokens,
946
- ...reasoningTokens !== void 0 && { reasoningTokens }
990
+ ...reasoningTokens !== void 0 && { reasoningTokens },
991
+ ...cachedInputTokens !== void 0 && { cachedInputTokens },
992
+ ...cost !== void 0 && { cost }
947
993
  } : void 0
948
994
  };
949
995
  } catch (err) {
@@ -1029,6 +1075,8 @@ function resolveConfig(config) {
1029
1075
  model,
1030
1076
  maxTokens,
1031
1077
  reasoningEffort: config.reasoningEffort,
1078
+ maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1079
+ imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1032
1080
  debug,
1033
1081
  debugPrompt,
1034
1082
  debugResponse,
@@ -1043,6 +1091,10 @@ var PRICING_TABLE = {
1043
1091
  inputPricePerToken: 10 / PER_MILLION,
1044
1092
  outputPricePerToken: 50 / PER_MILLION
1045
1093
  },
1094
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1095
+ inputPricePerToken: 5 / PER_MILLION,
1096
+ outputPricePerToken: 25 / PER_MILLION
1097
+ },
1046
1098
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
1047
1099
  inputPricePerToken: 5 / PER_MILLION,
1048
1100
  outputPricePerToken: 25 / PER_MILLION
@@ -1072,12 +1124,12 @@ var PRICING_TABLE = {
1072
1124
  outputPricePerToken: 30 / PER_MILLION
1073
1125
  },
1074
1126
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
1075
- inputPricePerToken: 2.5 / PER_MILLION,
1076
- outputPricePerToken: 15 / PER_MILLION
1127
+ inputPricePerToken: 2 / PER_MILLION,
1128
+ outputPricePerToken: 12 / PER_MILLION
1077
1129
  },
1078
1130
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
1079
- inputPricePerToken: 1 / PER_MILLION,
1080
- outputPricePerToken: 6 / PER_MILLION
1131
+ inputPricePerToken: 0.2 / PER_MILLION,
1132
+ outputPricePerToken: 1.2 / PER_MILLION
1081
1133
  },
1082
1134
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
1083
1135
  inputPricePerToken: 5 / PER_MILLION,
@@ -1107,6 +1159,18 @@ var PRICING_TABLE = {
1107
1159
  inputPricePerToken: 0.25 / PER_MILLION,
1108
1160
  outputPricePerToken: 2 / PER_MILLION
1109
1161
  },
1162
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1163
+ // on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
1164
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
1165
+ inputPricePerToken: 0.75 / PER_MILLION,
1166
+ outputPricePerToken: 3.75 / PER_MILLION
1167
+ },
1168
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1169
+ // on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
1170
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
1171
+ inputPricePerToken: 0.75 / PER_MILLION,
1172
+ outputPricePerToken: 3.75 / PER_MILLION
1173
+ },
1110
1174
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1111
1175
  inputPricePerToken: 1.5 / PER_MILLION,
1112
1176
  outputPricePerToken: 7.5 / PER_MILLION
@@ -1131,8 +1195,30 @@ var PRICING_TABLE = {
1131
1195
  inputPricePerToken: 0.5 / PER_MILLION,
1132
1196
  outputPricePerToken: 3 / PER_MILLION
1133
1197
  },
1134
- // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1198
+ // OpenRouter passes through upstream per-model pricing (verified 2026-09-03
1135
1199
  // against https://openrouter.ai/api/v1/models).
1200
+ // Meta's own listed rates ($1.25 / $4.25, cached input $0.15) match
1201
+ // OpenRouter's pass-through exactly. Cached input is not modelled here:
1202
+ // `calculateCost` applies no cache discount on any provider.
1203
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MUSE_SPARK_1_3}`]: {
1204
+ inputPricePerToken: 1.25 / PER_MILLION,
1205
+ outputPricePerToken: 4.25 / PER_MILLION
1206
+ },
1207
+ // Muse Spark 1.3's data-sharing tier: same model, ~12x cheaper, because
1208
+ // Meta trains on everything submitted through it. Deliberately keyed by the
1209
+ // literal slug rather than a `Model.OpenRouter` entry — it must never be
1210
+ // reachable via autocomplete or default selection. Pass the string yourself
1211
+ // (`model: "meta/muse-spark-1.3-contributor"`) to opt in; cost is still
1212
+ // tracked correctly once you do. Cached input is $0.002/MTok, not modelled
1213
+ // (no provider gets a cache discount here).
1214
+ [`${Provider.OPENROUTER}:meta/muse-spark-1.3-contributor`]: {
1215
+ inputPricePerToken: 0.1 / PER_MILLION,
1216
+ outputPricePerToken: 0.2 / PER_MILLION
1217
+ },
1218
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1219
+ inputPricePerToken: 2 / PER_MILLION,
1220
+ outputPricePerToken: 6 / PER_MILLION
1221
+ },
1136
1222
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1137
1223
  inputPricePerToken: 2 / PER_MILLION,
1138
1224
  outputPricePerToken: 6 / PER_MILLION
@@ -1145,6 +1231,10 @@ var PRICING_TABLE = {
1145
1231
  inputPricePerToken: 0.82 / PER_MILLION,
1146
1232
  outputPricePerToken: 3.75 / PER_MILLION
1147
1233
  },
1234
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
1235
+ inputPricePerToken: 2 / PER_MILLION,
1236
+ outputPricePerToken: 6 / PER_MILLION
1237
+ },
1148
1238
  [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1149
1239
  inputPricePerToken: 0.32 / PER_MILLION,
1150
1240
  outputPricePerToken: 1.28 / PER_MILLION
@@ -1174,8 +1264,9 @@ function usageLog(config, method, usage) {
1174
1264
  const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
1175
1265
  const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
1176
1266
  const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
1267
+ const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
1177
1268
  process.stderr.write(
1178
- `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1269
+ `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1179
1270
  `
1180
1271
  );
1181
1272
  }
@@ -1186,7 +1277,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
1186
1277
  inputTokens,
1187
1278
  outputTokens,
1188
1279
  ...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
1280
+ ...rawUsage?.cachedInputTokens !== void 0 && {
1281
+ cachedInputTokens: rawUsage.cachedInputTokens
1282
+ },
1189
1283
  estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
1284
+ ...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
1190
1285
  durationSeconds
1191
1286
  };
1192
1287
  usageLog(config, method, usage);
@@ -1230,7 +1325,9 @@ import sharp from "sharp";
1230
1325
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1231
1326
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1232
1327
  Model.Google.GEMINI_3_5_FLASH,
1233
- Model.Google.GEMINI_3_6_FLASH
1328
+ Model.Google.GEMINI_3_6_FLASH,
1329
+ Model.Google.GEMINI_3_7_FLASH,
1330
+ Model.Google.GEMINI_3_8_FLASH
1234
1331
  ]);
1235
1332
  async function generateAiDiff(imgA, imgB, model, driver) {
1236
1333
  if (!driver.generateImage) {
@@ -1320,7 +1417,6 @@ var EXTENSION_TO_MIME = {
1320
1417
  ".webp": "image/webp",
1321
1418
  ".gif": "image/gif"
1322
1419
  };
1323
- var MAX_DIMENSION = 1568;
1324
1420
  var URL_FETCH_TIMEOUT_MS = 1e4;
1325
1421
  function isSupportedMimeType(value) {
1326
1422
  return SUPPORTED_FORMATS.has(value);
@@ -1344,14 +1440,14 @@ function detectMimeType(data) {
1344
1440
  }
1345
1441
  throw new VisualAIImageError("Unable to detect image format from file content");
1346
1442
  }
1347
- async function resizeIfNeeded(data, mimeType) {
1443
+ async function resizeIfNeeded(data, mimeType, maxDimension) {
1348
1444
  if (mimeType === "image/gif") {
1349
1445
  return data;
1350
1446
  }
1351
1447
  if (mimeType === "image/png" && data.length >= 24) {
1352
1448
  const width2 = data.readUInt32BE(16);
1353
1449
  const height2 = data.readUInt32BE(20);
1354
- if (width2 <= MAX_DIMENSION && height2 <= MAX_DIMENSION) {
1450
+ if (width2 <= maxDimension && height2 <= maxDimension) {
1355
1451
  return data;
1356
1452
  }
1357
1453
  }
@@ -1359,12 +1455,12 @@ async function resizeIfNeeded(data, mimeType) {
1359
1455
  const metadata = await pipeline.metadata();
1360
1456
  const width = metadata.width ?? 0;
1361
1457
  const height = metadata.height ?? 0;
1362
- if (width <= MAX_DIMENSION && height <= MAX_DIMENSION) {
1458
+ if (width <= maxDimension && height <= maxDimension) {
1363
1459
  return data;
1364
1460
  }
1365
1461
  return pipeline.resize({
1366
- width: MAX_DIMENSION,
1367
- height: MAX_DIMENSION,
1462
+ width: maxDimension,
1463
+ height: maxDimension,
1368
1464
  fit: "inside",
1369
1465
  withoutEnlargement: true
1370
1466
  }).toBuffer();
@@ -1419,7 +1515,7 @@ function loadFromBase64(input) {
1419
1515
  }
1420
1516
  return { data, mimeType: mimeType ?? detectMimeType(data) };
1421
1517
  }
1422
- async function normalizeImage(input) {
1518
+ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1423
1519
  let data;
1424
1520
  let mimeType;
1425
1521
  if (Buffer.isBuffer(input)) {
@@ -1448,7 +1544,7 @@ async function normalizeImage(input) {
1448
1544
  "Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
1449
1545
  );
1450
1546
  }
1451
- data = await resizeIfNeeded(data, mimeType);
1547
+ data = await resizeIfNeeded(data, mimeType, maxDimension);
1452
1548
  let cachedBase64;
1453
1549
  return {
1454
1550
  data,
@@ -1752,7 +1848,7 @@ async function probeDurationSeconds(videoPath) {
1752
1848
  });
1753
1849
  });
1754
1850
  }
1755
- async function extractFrames(videoPath, options = {}) {
1851
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
1756
1852
  const fps = options.fps ?? DEFAULT_FPS;
1757
1853
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1758
1854
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1781,7 +1877,7 @@ async function extractFrames(videoPath, options = {}) {
1781
1877
  }
1782
1878
  const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
1783
1879
  try {
1784
- const filter = `fps=${fps},scale='if(gt(iw,ih),min(${FRAME_MAX_DIMENSION},iw),-2)':'if(gt(iw,ih),-2,min(${FRAME_MAX_DIMENSION},ih))':flags=area`;
1880
+ const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
1785
1881
  await new Promise((resolve2, reject) => {
1786
1882
  let settled = false;
1787
1883
  const cmd = ffmpeg(videoPath);
@@ -1879,7 +1975,7 @@ function isFramesInput(input) {
1879
1975
  function isTimestampedFrameInput(frame) {
1880
1976
  return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
1881
1977
  }
1882
- async function normalizeFrames(input) {
1978
+ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1883
1979
  const rawFrames = input.frames;
1884
1980
  const fps = input.fps ?? DEFAULT_FPS;
1885
1981
  if (rawFrames.length === 0) {
@@ -1904,7 +2000,7 @@ async function normalizeFrames(input) {
1904
2000
  `Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
1905
2001
  );
1906
2002
  }
1907
- const image = await normalizeImage(imageInput);
2003
+ const image = await normalizeImage(imageInput, maxDimension);
1908
2004
  return {
1909
2005
  data: image.data,
1910
2006
  mimeType: image.mimeType,
@@ -1920,14 +2016,14 @@ async function normalizeFrames(input) {
1920
2016
  await saveDebugFrames(frames);
1921
2017
  return { kind: "video", frames, durationSeconds };
1922
2018
  }
1923
- async function normalizeMedia(input, videoOptions) {
2019
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1924
2020
  if (isFramesInput(input)) {
1925
- return normalizeFrames(input);
2021
+ return normalizeFrames(input, maxDimension);
1926
2022
  }
1927
2023
  if (isVideoInput(input)) {
1928
2024
  const { path, cleanup } = await resolveVideoToPath(input);
1929
2025
  try {
1930
- const { frames, durationSeconds } = await extractFrames(path, videoOptions);
2026
+ const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
1931
2027
  await saveDebugFrames(frames);
1932
2028
  return { kind: "video", frames, durationSeconds };
1933
2029
  } finally {
@@ -1937,7 +2033,7 @@ async function normalizeMedia(input, videoOptions) {
1937
2033
  }
1938
2034
  }
1939
2035
  }
1940
- const image = await normalizeImage(input);
2036
+ const image = await normalizeImage(input, maxDimension);
1941
2037
  return { kind: "image", image };
1942
2038
  }
1943
2039
 
@@ -1978,7 +2074,17 @@ var UsageInfoSchema = z.object({
1978
2074
  outputTokens: z.number(),
1979
2075
  /** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
1980
2076
  reasoningTokens: z.number().optional(),
2077
+ /**
2078
+ * Prompt tokens served from the provider's cache, when reported. Informational
2079
+ * only — `estimatedCost` does not apply a cache discount, because providers
2080
+ * differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
2081
+ * Google) or billed as a separate bucket alongside it (Anthropic).
2082
+ */
2083
+ cachedInputTokens: z.number().optional(),
2084
+ /** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
1981
2085
  estimatedCost: z.number().optional(),
2086
+ /** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
2087
+ reportedCost: z.number().optional(),
1982
2088
  durationSeconds: z.number().nonnegative().optional()
1983
2089
  });
1984
2090
  var BaseResultSchema = z.object({
@@ -2118,16 +2224,18 @@ function visualAI(config = {}) {
2118
2224
  apiKey: resolvedConfig.apiKey,
2119
2225
  model: resolvedConfig.model,
2120
2226
  maxTokens: resolvedConfig.maxTokens,
2121
- reasoningEffort: resolvedConfig.reasoningEffort
2227
+ reasoningEffort: resolvedConfig.reasoningEffort,
2228
+ imageDetail: resolvedConfig.imageDetail
2122
2229
  };
2123
2230
  const driver = createDriver(resolvedConfig.provider, driverConfig);
2231
+ const maxImageDimension = resolvedConfig.maxImageDimension;
2124
2232
  async function checkElementsVisibility(image, elements, visible, options) {
2125
2233
  const methodName = visible ? "elementsVisible" : "elementsHidden";
2126
2234
  if (elements.length === 0) {
2127
2235
  throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
2128
2236
  }
2129
2237
  return withErrorDebug(resolvedConfig, methodName, async () => {
2130
- const img = await normalizeImage(image);
2238
+ const img = await normalizeImage(image, maxImageDimension);
2131
2239
  const prompt = buildElementsVisibilityPrompt(elements, visible, options);
2132
2240
  debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
2133
2241
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2146,7 +2254,7 @@ function visualAI(config = {}) {
2146
2254
  throw new VisualAIConfigError("At least one statement is required for check()");
2147
2255
  }
2148
2256
  return withErrorDebug(resolvedConfig, "check", async () => {
2149
- const media = await normalizeMedia(input, options?.video);
2257
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2150
2258
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2151
2259
  const prompt = buildCheckPrompt(stmts, {
2152
2260
  instructions: options?.instructions,
@@ -2165,7 +2273,7 @@ function visualAI(config = {}) {
2165
2273
  },
2166
2274
  async ask(input, userPrompt, options) {
2167
2275
  return withErrorDebug(resolvedConfig, "ask", async () => {
2168
- const media = await normalizeMedia(input, options?.video);
2276
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2169
2277
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2170
2278
  const prompt = buildAskPrompt(userPrompt, {
2171
2279
  instructions: options?.instructions,
@@ -2184,7 +2292,10 @@ function visualAI(config = {}) {
2184
2292
  },
2185
2293
  async compare(imageA, imageB, options) {
2186
2294
  return withErrorDebug(resolvedConfig, "compare", async () => {
2187
- const [imgA, imgB] = await Promise.all([normalizeImage(imageA), normalizeImage(imageB)]);
2295
+ const [imgA, imgB] = await Promise.all([
2296
+ normalizeImage(imageA, maxImageDimension),
2297
+ normalizeImage(imageB, maxImageDimension)
2298
+ ]);
2188
2299
  const prompt = buildComparePrompt({
2189
2300
  userPrompt: options?.prompt,
2190
2301
  instructions: options?.instructions
@@ -2222,7 +2333,7 @@ function visualAI(config = {}) {
2222
2333
  },
2223
2334
  async accessibility(image, options) {
2224
2335
  return withErrorDebug(resolvedConfig, "accessibility", async () => {
2225
- const img = await normalizeImage(image);
2336
+ const img = await normalizeImage(image, maxImageDimension);
2226
2337
  const prompt = buildAccessibilityPrompt(options);
2227
2338
  debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
2228
2339
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2241,7 +2352,7 @@ function visualAI(config = {}) {
2241
2352
  },
2242
2353
  async layout(image, options) {
2243
2354
  return withErrorDebug(resolvedConfig, "layout", async () => {
2244
- const img = await normalizeImage(image);
2355
+ const img = await normalizeImage(image, maxImageDimension);
2245
2356
  const prompt = buildLayoutPrompt(options);
2246
2357
  debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
2247
2358
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2255,7 +2366,7 @@ function visualAI(config = {}) {
2255
2366
  },
2256
2367
  async pageLoad(image, options) {
2257
2368
  return withErrorDebug(resolvedConfig, "pageLoad", async () => {
2258
- const img = await normalizeImage(image);
2369
+ const img = await normalizeImage(image, maxImageDimension);
2259
2370
  const prompt = buildPageLoadPrompt(options);
2260
2371
  debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
2261
2372
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2269,7 +2380,7 @@ function visualAI(config = {}) {
2269
2380
  },
2270
2381
  async content(image, options) {
2271
2382
  return withErrorDebug(resolvedConfig, "content", async () => {
2272
- const img = await normalizeImage(image);
2383
+ const img = await normalizeImage(image, maxImageDimension);
2273
2384
  const prompt = buildContentPrompt(options);
2274
2385
  debugLog(resolvedConfig, "content prompt", prompt, "prompt");
2275
2386
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2347,6 +2458,7 @@ export {
2347
2458
  ConfidenceSchema,
2348
2459
  Content,
2349
2460
  DEFAULT_MODELS,
2461
+ ImageDetail,
2350
2462
  IssueCategorySchema,
2351
2463
  IssuePrioritySchema,
2352
2464
  IssueSchema,