visual-ai-assertions 0.16.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -524,7 +524,7 @@ When omitted, each provider uses its default behavior. The `"xhigh"` level enabl
524
524
  | Anthropic (Fable 5/Opus 4.8/4.7/Sonnet 5) | `thinking.type: "adaptive"` + `output_config.effort` | `effort: "xhigh"` |
525
525
  | Anthropic (other) | `thinking.type: "adaptive"` + `output_config.effort` | `effort: "max"` |
526
526
  | OpenAI | `reasoning.effort` (Responses API) | `effort: "xhigh"` |
527
- | Google | `thinkingConfig.thinkingBudget` (1024 / 8192 / 24576) | `24576` (max budget) |
527
+ | Google | `thinkingConfig.thinkingLevel` (1:1: low/medium/high) | `"high"` (max level) |
528
528
  | OpenRouter | `reasoning.effort` (normalized low/medium/high) | `effort: "high"` |
529
529
 
530
530
  ## Supported Models
@@ -548,8 +548,8 @@ All listed models support image/vision input. Pass any model ID to the `model` c
548
548
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
549
549
  | ------------- | --------------- | ------------ | ------------- | --------------------------------- |
550
550
  | GPT-5.6 Sol | `gpt-5.6-sol` | $5 | $30 | Newest flagship, frontier tier |
551
- | GPT-5.6 Terra | `gpt-5.6-terra` | $2.50 | $15 | Newest balanced, everyday tier |
552
- | GPT-5.6 Luna | `gpt-5.6-luna` | $1 | $6 | Newest, fastest/cheapest tier |
551
+ | GPT-5.6 Terra | `gpt-5.6-terra` | $2 | $12 | Newest balanced, everyday tier |
552
+ | GPT-5.6 Luna | `gpt-5.6-luna` | $0.20 | $1.20 | Newest, fastest/cheapest tier |
553
553
  | GPT-5.5 | `gpt-5.5` | $5 | $30 | Previous flagship, 1M context |
554
554
  | GPT-5.4 Pro | `gpt-5.4-pro` | $30 | $180 | Most capable, extended context |
555
555
  | GPT-5.4 | `gpt-5.4` | $2.50 | $15 | Best vision quality |
@@ -562,26 +562,32 @@ All listed models support image/vision input. Pass any model ID to the `model` c
562
562
 
563
563
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
564
564
  | --------------------- | ------------------------ | ------------ | ------------- | --------------------------------- |
565
- | Gemini 3.6 Flash | `gemini-3.6-flash` | $1.50 | $7.50 | Newest GA flash; fewer out-tokens |
565
+ | Gemini 3.8 Flash | `gemini-3.8-flash` | $0.75 | $3.75 | Newest GA flash; intro pricing¹ |
566
+ | Gemini 3.7 Flash | `gemini-3.7-flash` | $0.75 | $3.75 | Prior GA flash; intro pricing¹ |
567
+ | Gemini 3.6 Flash | `gemini-3.6-flash` | $1.50 | $7.50 | Prior GA flash; fewer out-tokens |
566
568
  | Gemini 3.5 Flash | `gemini-3.5-flash` | $1.50 | $9 | Strongest agentic & coding model |
567
569
  | Gemini 3.5 Flash Lite | `gemini-3.5-flash-lite` | $0.30 | $2.50 | GA — fast, cheap, agentic tier |
568
570
  | Gemini 3.1 Pro | `gemini-3.1-pro-preview` | $2 | $12 | Preview — most advanced reasoning |
569
571
  | Gemini 3.1 Flash Lite | `gemini-3.1-flash-lite` | $0.25 | $1.50 | GA — lightweight and cheap |
570
572
  | Gemini 3 Flash | `gemini-3-flash-preview` | $0.50 | $3 | **Default** — fast and capable |
571
573
 
574
+ ¹ Gemini 3.8 Flash and 3.7 Flash introductory pricing runs through 2026-12-31; both revert to $1.50 / $7.50 per MTok on 2027-01-01.
575
+
572
576
  ### OpenRouter
573
577
 
574
578
  Any [OpenRouter](https://openrouter.ai/models) model slug (always `vendor/model`) is accepted — the vendor prefix is how the library recognizes an OpenRouter model. The models below are tested and have pricing built in. Note that OpenRouter may route a request to different upstream hosts with different quantizations; keep that in mind when comparing benchmark numbers.
575
579
 
576
580
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
577
581
  | -------------- | --------------------------- | ------------ | ------------- | ------------------------------------- |
578
- | Grok 4.5 | `x-ai/grok-4.5` | $2 | $6 | xAI flagship, 500K context |
582
+ | Grok 4.6 | `x-ai/grok-4.6` | $2 | $6 | Newest xAI flagship, 500K context |
583
+ | Grok 4.5 | `x-ai/grok-4.5` | $2 | $6 | Prior xAI flagship, 500K context |
579
584
  | Kimi K3 | `moonshotai/kimi-k3` | $3 | $15 | Moonshot flagship, 1M context |
580
585
  | Kimi K2.7 Code | `moonshotai/kimi-k2.7-code` | $0.82 | $3.75 | Agentic/coding tier with vision |
586
+ | Qwen3.8 Max | `qwen/qwen3.8-max` | $2 | $6 | First Max tier with image input |
581
587
  | Qwen3.7 Plus | `qwen/qwen3.7-plus` | $0.32 | $1.28 | Cost-effective, GUI/screen-reading |
582
588
  | Qwen3.6 Flash | `qwen/qwen3.6-flash` | $0.19 | $1.13 | **Default** — cheap flash vision tier |
583
589
 
584
- `qwen/qwen3.7-max` is not listed because it accepts no image input on OpenRouter.
590
+ `qwen/qwen3.7-max` and the DeepSeek V4 family (`deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`, and dated variants such as `deepseek/deepseek-v4-pro-0813`) are not listed because they accept no image input on OpenRouter.
585
591
 
586
592
  ## License
587
593
 
package/dist/index.cjs CHANGED
@@ -38,6 +38,7 @@ __export(index_exports, {
38
38
  ConfidenceSchema: () => ConfidenceSchema,
39
39
  Content: () => Content,
40
40
  DEFAULT_MODELS: () => DEFAULT_MODELS,
41
+ ImageDetail: () => ImageDetail,
41
42
  IssueCategorySchema: () => IssueCategorySchema,
42
43
  IssuePrioritySchema: () => IssuePrioritySchema,
43
44
  IssueSchema: () => IssueSchema,
@@ -73,6 +74,13 @@ var ReasoningEffort = {
73
74
  HIGH: "high",
74
75
  XHIGH: "xhigh"
75
76
  };
77
+ var ImageDetail = {
78
+ AUTO: "auto",
79
+ LOW: "low",
80
+ HIGH: "high"
81
+ };
82
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
83
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
76
84
  var Provider = {
77
85
  ANTHROPIC: "anthropic",
78
86
  OPENAI: "openai",
@@ -82,6 +90,7 @@ var Provider = {
82
90
  var Model = {
83
91
  Anthropic: {
84
92
  FABLE_5: "claude-fable-5",
93
+ OPUS_5: "claude-opus-5",
85
94
  OPUS_4_8: "claude-opus-4-8",
86
95
  OPUS_4_7: "claude-opus-4-7",
87
96
  OPUS_4_6: "claude-opus-4-6",
@@ -102,6 +111,8 @@ var Model = {
102
111
  GPT_5_MINI: "gpt-5-mini"
103
112
  },
104
113
  Google: {
114
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
115
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
105
116
  GEMINI_3_6_FLASH: "gemini-3.6-flash",
106
117
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
107
118
  GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
@@ -115,9 +126,11 @@ var Model = {
115
126
  * recognizes them. All listed models accept image input.
116
127
  */
117
128
  OpenRouter: {
129
+ GROK_4_6: "x-ai/grok-4.6",
118
130
  GROK_4_5: "x-ai/grok-4.5",
119
131
  KIMI_K3: "moonshotai/kimi-k3",
120
132
  KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
133
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
121
134
  QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
122
135
  QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
123
136
  }
@@ -600,6 +613,7 @@ function parseRetryAfter(value) {
600
613
  // src/providers/anthropic.ts
601
614
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
602
615
  Model.Anthropic.FABLE_5,
616
+ Model.Anthropic.OPUS_5,
603
617
  Model.Anthropic.OPUS_4_8,
604
618
  Model.Anthropic.OPUS_4_7,
605
619
  Model.Anthropic.SONNET_5
@@ -695,7 +709,10 @@ var AnthropicDriver = class {
695
709
  text,
696
710
  usage: {
697
711
  inputTokens: message.usage.input_tokens,
698
- outputTokens: message.usage.output_tokens
712
+ outputTokens: message.usage.output_tokens,
713
+ ...message.usage.cache_read_input_tokens !== void 0 && {
714
+ cachedInputTokens: message.usage.cache_read_input_tokens
715
+ }
699
716
  }
700
717
  };
701
718
  } catch (err) {
@@ -712,23 +729,41 @@ function needsCodeExecution(model) {
712
729
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
713
730
  }
714
731
  var GOOGLE_THINKING_LEVEL = {
715
- low: "minimal",
716
- medium: "low",
717
- high: "medium",
732
+ low: "low",
733
+ medium: "medium",
734
+ high: "high",
718
735
  xhigh: "high"
719
736
  };
737
+ var GOOGLE_MEDIA_RESOLUTION = {
738
+ low: "MEDIA_RESOLUTION_LOW",
739
+ high: "MEDIA_RESOLUTION_HIGH"
740
+ };
741
+ function toGeminiUsage(um) {
742
+ if (!um) return void 0;
743
+ const thoughts = um.thoughtsTokenCount ?? 0;
744
+ return {
745
+ inputTokens: um.promptTokenCount ?? 0,
746
+ outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
747
+ ...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
748
+ ...um.cachedContentTokenCount !== void 0 && {
749
+ cachedInputTokens: um.cachedContentTokenCount
750
+ }
751
+ };
752
+ }
720
753
  var GoogleDriver = class {
721
754
  client;
722
755
  model;
723
756
  maxTokens;
724
757
  apiKeyOrEnv;
725
758
  reasoningEffort;
759
+ imageDetail;
726
760
  constructor(config) {
727
761
  this.model = config.model;
728
762
  this.maxTokens = config.maxTokens;
729
763
  this.client = null;
730
764
  this.apiKeyOrEnv = config.apiKey;
731
765
  this.reasoningEffort = config.reasoningEffort;
766
+ this.imageDetail = config.imageDetail;
732
767
  }
733
768
  toGeminiParts(images) {
734
769
  return images.map((img) => ({
@@ -768,6 +803,9 @@ var GoogleDriver = class {
768
803
  thinkingConfig: {
769
804
  thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
770
805
  }
806
+ },
807
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
808
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
771
809
  }
772
810
  }
773
811
  });
@@ -785,14 +823,9 @@ var GoogleDriver = class {
785
823
  );
786
824
  }
787
825
  const text = response.text ?? "";
788
- const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
789
826
  return {
790
827
  text,
791
- usage: response.usageMetadata ? {
792
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
793
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
794
- ...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
795
- } : void 0
828
+ usage: toGeminiUsage(response.usageMetadata)
796
829
  };
797
830
  } catch (err) {
798
831
  if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
@@ -823,10 +856,7 @@ var GoogleDriver = class {
823
856
  return {
824
857
  imageData: Buffer.from(imagePart.inlineData.data, "base64"),
825
858
  mimeType: imagePart.inlineData.mimeType,
826
- usage: response.usageMetadata ? {
827
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
828
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
829
- } : void 0
859
+ usage: toGeminiUsage(response.usageMetadata)
830
860
  };
831
861
  } catch (err) {
832
862
  if (err instanceof VisualAIProviderError) throw err;
@@ -842,12 +872,14 @@ var OpenAIDriver = class {
842
872
  maxTokens;
843
873
  apiKeyOrEnv;
844
874
  reasoningEffort;
875
+ imageDetail;
845
876
  constructor(config) {
846
877
  this.model = config.model;
847
878
  this.maxTokens = config.maxTokens;
848
879
  this.client = null;
849
880
  this.apiKeyOrEnv = config.apiKey;
850
881
  this.reasoningEffort = config.reasoningEffort;
882
+ this.imageDetail = config.imageDetail;
851
883
  }
852
884
  async getClient() {
853
885
  if (this.client) return this.client;
@@ -869,9 +901,11 @@ var OpenAIDriver = class {
869
901
  }
870
902
  async sendMessage(images, prompt, options) {
871
903
  const client = await this.getClient();
904
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
872
905
  const imageBlocks = images.map((img) => ({
873
906
  type: "input_image",
874
- image_url: `data:${img.mimeType};base64,${img.base64}`
907
+ image_url: `data:${img.mimeType};base64,${img.base64}`,
908
+ ...detail ? { detail } : {}
875
909
  }));
876
910
  try {
877
911
  const format = options?.responseSchema ? {
@@ -896,21 +930,23 @@ var OpenAIDriver = class {
896
930
  }
897
931
  const response = await client.responses.create(requestParams);
898
932
  if (response.status && response.status !== "completed") {
899
- const detail = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
933
+ const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
900
934
  throw new VisualAITruncationError(
901
- `Response truncated: OpenAI returned status "${response.status}"${detail}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
935
+ `Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
902
936
  response.output_text ?? "",
903
937
  this.maxTokens
904
938
  );
905
939
  }
906
940
  const text = response.output_text ?? "";
907
941
  const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
942
+ const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
908
943
  return {
909
944
  text,
910
945
  usage: response.usage ? {
911
946
  inputTokens: response.usage.input_tokens,
912
947
  outputTokens: response.usage.output_tokens,
913
- ...reasoningTokens !== void 0 && { reasoningTokens }
948
+ ...reasoningTokens !== void 0 && { reasoningTokens },
949
+ ...cachedInputTokens !== void 0 && { cachedInputTokens }
914
950
  } : void 0
915
951
  };
916
952
  } catch (err) {
@@ -934,12 +970,14 @@ var OpenRouterDriver = class {
934
970
  maxTokens;
935
971
  apiKeyOrEnv;
936
972
  reasoningEffort;
973
+ imageDetail;
937
974
  constructor(config) {
938
975
  this.model = config.model;
939
976
  this.maxTokens = config.maxTokens;
940
977
  this.client = null;
941
978
  this.apiKeyOrEnv = config.apiKey;
942
979
  this.reasoningEffort = config.reasoningEffort;
980
+ this.imageDetail = config.imageDetail;
943
981
  }
944
982
  async getClient() {
945
983
  if (this.client) return this.client;
@@ -963,9 +1001,13 @@ var OpenRouterDriver = class {
963
1001
  }
964
1002
  async sendMessage(images, prompt, options) {
965
1003
  const client = await this.getClient();
1004
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
966
1005
  const imageParts = images.map((img) => ({
967
1006
  type: "image_url",
968
- image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
1007
+ image_url: {
1008
+ url: `data:${img.mimeType};base64,${img.base64}`,
1009
+ ...detail ? { detail } : {}
1010
+ }
969
1011
  }));
970
1012
  try {
971
1013
  const responseFormat = options?.responseSchema ? {
@@ -1006,12 +1048,16 @@ var OpenRouterDriver = class {
1006
1048
  );
1007
1049
  }
1008
1050
  const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
1051
+ const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
1052
+ const cost = response.usage?.cost;
1009
1053
  return {
1010
1054
  text,
1011
1055
  usage: response.usage ? {
1012
1056
  inputTokens: response.usage.prompt_tokens,
1013
1057
  outputTokens: response.usage.completion_tokens,
1014
- ...reasoningTokens !== void 0 && { reasoningTokens }
1058
+ ...reasoningTokens !== void 0 && { reasoningTokens },
1059
+ ...cachedInputTokens !== void 0 && { cachedInputTokens },
1060
+ ...cost !== void 0 && { cost }
1015
1061
  } : void 0
1016
1062
  };
1017
1063
  } catch (err) {
@@ -1097,6 +1143,8 @@ function resolveConfig(config) {
1097
1143
  model,
1098
1144
  maxTokens,
1099
1145
  reasoningEffort: config.reasoningEffort,
1146
+ maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1147
+ imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1100
1148
  debug,
1101
1149
  debugPrompt,
1102
1150
  debugResponse,
@@ -1111,6 +1159,10 @@ var PRICING_TABLE = {
1111
1159
  inputPricePerToken: 10 / PER_MILLION,
1112
1160
  outputPricePerToken: 50 / PER_MILLION
1113
1161
  },
1162
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1163
+ inputPricePerToken: 5 / PER_MILLION,
1164
+ outputPricePerToken: 25 / PER_MILLION
1165
+ },
1114
1166
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
1115
1167
  inputPricePerToken: 5 / PER_MILLION,
1116
1168
  outputPricePerToken: 25 / PER_MILLION
@@ -1140,12 +1192,12 @@ var PRICING_TABLE = {
1140
1192
  outputPricePerToken: 30 / PER_MILLION
1141
1193
  },
1142
1194
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
1143
- inputPricePerToken: 2.5 / PER_MILLION,
1144
- outputPricePerToken: 15 / PER_MILLION
1195
+ inputPricePerToken: 2 / PER_MILLION,
1196
+ outputPricePerToken: 12 / PER_MILLION
1145
1197
  },
1146
1198
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
1147
- inputPricePerToken: 1 / PER_MILLION,
1148
- outputPricePerToken: 6 / PER_MILLION
1199
+ inputPricePerToken: 0.2 / PER_MILLION,
1200
+ outputPricePerToken: 1.2 / PER_MILLION
1149
1201
  },
1150
1202
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
1151
1203
  inputPricePerToken: 5 / PER_MILLION,
@@ -1175,6 +1227,18 @@ var PRICING_TABLE = {
1175
1227
  inputPricePerToken: 0.25 / PER_MILLION,
1176
1228
  outputPricePerToken: 2 / PER_MILLION
1177
1229
  },
1230
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1231
+ // on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
1232
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
1233
+ inputPricePerToken: 0.75 / PER_MILLION,
1234
+ outputPricePerToken: 3.75 / PER_MILLION
1235
+ },
1236
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1237
+ // on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
1238
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
1239
+ inputPricePerToken: 0.75 / PER_MILLION,
1240
+ outputPricePerToken: 3.75 / PER_MILLION
1241
+ },
1178
1242
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1179
1243
  inputPricePerToken: 1.5 / PER_MILLION,
1180
1244
  outputPricePerToken: 7.5 / PER_MILLION
@@ -1201,6 +1265,10 @@ var PRICING_TABLE = {
1201
1265
  },
1202
1266
  // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1203
1267
  // against https://openrouter.ai/api/v1/models).
1268
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1269
+ inputPricePerToken: 2 / PER_MILLION,
1270
+ outputPricePerToken: 6 / PER_MILLION
1271
+ },
1204
1272
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1205
1273
  inputPricePerToken: 2 / PER_MILLION,
1206
1274
  outputPricePerToken: 6 / PER_MILLION
@@ -1213,6 +1281,10 @@ var PRICING_TABLE = {
1213
1281
  inputPricePerToken: 0.82 / PER_MILLION,
1214
1282
  outputPricePerToken: 3.75 / PER_MILLION
1215
1283
  },
1284
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
1285
+ inputPricePerToken: 2 / PER_MILLION,
1286
+ outputPricePerToken: 6 / PER_MILLION
1287
+ },
1216
1288
  [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1217
1289
  inputPricePerToken: 0.32 / PER_MILLION,
1218
1290
  outputPricePerToken: 1.28 / PER_MILLION
@@ -1242,8 +1314,9 @@ function usageLog(config, method, usage) {
1242
1314
  const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
1243
1315
  const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
1244
1316
  const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
1317
+ const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
1245
1318
  process.stderr.write(
1246
- `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1319
+ `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1247
1320
  `
1248
1321
  );
1249
1322
  }
@@ -1254,7 +1327,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
1254
1327
  inputTokens,
1255
1328
  outputTokens,
1256
1329
  ...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
1330
+ ...rawUsage?.cachedInputTokens !== void 0 && {
1331
+ cachedInputTokens: rawUsage.cachedInputTokens
1332
+ },
1257
1333
  estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
1334
+ ...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
1258
1335
  durationSeconds
1259
1336
  };
1260
1337
  usageLog(config, method, usage);
@@ -1298,7 +1375,9 @@ var import_sharp = __toESM(require("sharp"), 1);
1298
1375
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1299
1376
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1300
1377
  Model.Google.GEMINI_3_5_FLASH,
1301
- Model.Google.GEMINI_3_6_FLASH
1378
+ Model.Google.GEMINI_3_6_FLASH,
1379
+ Model.Google.GEMINI_3_7_FLASH,
1380
+ Model.Google.GEMINI_3_8_FLASH
1302
1381
  ]);
1303
1382
  async function generateAiDiff(imgA, imgB, model, driver) {
1304
1383
  if (!driver.generateImage) {
@@ -1388,7 +1467,6 @@ var EXTENSION_TO_MIME = {
1388
1467
  ".webp": "image/webp",
1389
1468
  ".gif": "image/gif"
1390
1469
  };
1391
- var MAX_DIMENSION = 1568;
1392
1470
  var URL_FETCH_TIMEOUT_MS = 1e4;
1393
1471
  function isSupportedMimeType(value) {
1394
1472
  return SUPPORTED_FORMATS.has(value);
@@ -1412,14 +1490,14 @@ function detectMimeType(data) {
1412
1490
  }
1413
1491
  throw new VisualAIImageError("Unable to detect image format from file content");
1414
1492
  }
1415
- async function resizeIfNeeded(data, mimeType) {
1493
+ async function resizeIfNeeded(data, mimeType, maxDimension) {
1416
1494
  if (mimeType === "image/gif") {
1417
1495
  return data;
1418
1496
  }
1419
1497
  if (mimeType === "image/png" && data.length >= 24) {
1420
1498
  const width2 = data.readUInt32BE(16);
1421
1499
  const height2 = data.readUInt32BE(20);
1422
- if (width2 <= MAX_DIMENSION && height2 <= MAX_DIMENSION) {
1500
+ if (width2 <= maxDimension && height2 <= maxDimension) {
1423
1501
  return data;
1424
1502
  }
1425
1503
  }
@@ -1427,12 +1505,12 @@ async function resizeIfNeeded(data, mimeType) {
1427
1505
  const metadata = await pipeline.metadata();
1428
1506
  const width = metadata.width ?? 0;
1429
1507
  const height = metadata.height ?? 0;
1430
- if (width <= MAX_DIMENSION && height <= MAX_DIMENSION) {
1508
+ if (width <= maxDimension && height <= maxDimension) {
1431
1509
  return data;
1432
1510
  }
1433
1511
  return pipeline.resize({
1434
- width: MAX_DIMENSION,
1435
- height: MAX_DIMENSION,
1512
+ width: maxDimension,
1513
+ height: maxDimension,
1436
1514
  fit: "inside",
1437
1515
  withoutEnlargement: true
1438
1516
  }).toBuffer();
@@ -1487,7 +1565,7 @@ function loadFromBase64(input) {
1487
1565
  }
1488
1566
  return { data, mimeType: mimeType ?? detectMimeType(data) };
1489
1567
  }
1490
- async function normalizeImage(input) {
1568
+ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1491
1569
  let data;
1492
1570
  let mimeType;
1493
1571
  if (Buffer.isBuffer(input)) {
@@ -1516,7 +1594,7 @@ async function normalizeImage(input) {
1516
1594
  "Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
1517
1595
  );
1518
1596
  }
1519
- data = await resizeIfNeeded(data, mimeType);
1597
+ data = await resizeIfNeeded(data, mimeType, maxDimension);
1520
1598
  let cachedBase64;
1521
1599
  return {
1522
1600
  data,
@@ -1820,7 +1898,7 @@ async function probeDurationSeconds(videoPath) {
1820
1898
  });
1821
1899
  });
1822
1900
  }
1823
- async function extractFrames(videoPath, options = {}) {
1901
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
1824
1902
  const fps = options.fps ?? DEFAULT_FPS;
1825
1903
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1826
1904
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1849,7 +1927,7 @@ async function extractFrames(videoPath, options = {}) {
1849
1927
  }
1850
1928
  const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
1851
1929
  try {
1852
- const filter = `fps=${fps},scale='if(gt(iw,ih),min(${FRAME_MAX_DIMENSION},iw),-2)':'if(gt(iw,ih),-2,min(${FRAME_MAX_DIMENSION},ih))':flags=area`;
1930
+ const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
1853
1931
  await new Promise((resolve2, reject) => {
1854
1932
  let settled = false;
1855
1933
  const cmd = ffmpeg(videoPath);
@@ -1947,7 +2025,7 @@ function isFramesInput(input) {
1947
2025
  function isTimestampedFrameInput(frame) {
1948
2026
  return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
1949
2027
  }
1950
- async function normalizeFrames(input) {
2028
+ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1951
2029
  const rawFrames = input.frames;
1952
2030
  const fps = input.fps ?? DEFAULT_FPS;
1953
2031
  if (rawFrames.length === 0) {
@@ -1972,7 +2050,7 @@ async function normalizeFrames(input) {
1972
2050
  `Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
1973
2051
  );
1974
2052
  }
1975
- const image = await normalizeImage(imageInput);
2053
+ const image = await normalizeImage(imageInput, maxDimension);
1976
2054
  return {
1977
2055
  data: image.data,
1978
2056
  mimeType: image.mimeType,
@@ -1988,14 +2066,14 @@ async function normalizeFrames(input) {
1988
2066
  await saveDebugFrames(frames);
1989
2067
  return { kind: "video", frames, durationSeconds };
1990
2068
  }
1991
- async function normalizeMedia(input, videoOptions) {
2069
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1992
2070
  if (isFramesInput(input)) {
1993
- return normalizeFrames(input);
2071
+ return normalizeFrames(input, maxDimension);
1994
2072
  }
1995
2073
  if (isVideoInput(input)) {
1996
2074
  const { path, cleanup } = await resolveVideoToPath(input);
1997
2075
  try {
1998
- const { frames, durationSeconds } = await extractFrames(path, videoOptions);
2076
+ const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
1999
2077
  await saveDebugFrames(frames);
2000
2078
  return { kind: "video", frames, durationSeconds };
2001
2079
  } finally {
@@ -2005,7 +2083,7 @@ async function normalizeMedia(input, videoOptions) {
2005
2083
  }
2006
2084
  }
2007
2085
  }
2008
- const image = await normalizeImage(input);
2086
+ const image = await normalizeImage(input, maxDimension);
2009
2087
  return { kind: "image", image };
2010
2088
  }
2011
2089
 
@@ -2046,7 +2124,17 @@ var UsageInfoSchema = import_zod.z.object({
2046
2124
  outputTokens: import_zod.z.number(),
2047
2125
  /** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
2048
2126
  reasoningTokens: import_zod.z.number().optional(),
2127
+ /**
2128
+ * Prompt tokens served from the provider's cache, when reported. Informational
2129
+ * only — `estimatedCost` does not apply a cache discount, because providers
2130
+ * differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
2131
+ * Google) or billed as a separate bucket alongside it (Anthropic).
2132
+ */
2133
+ cachedInputTokens: import_zod.z.number().optional(),
2134
+ /** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
2049
2135
  estimatedCost: import_zod.z.number().optional(),
2136
+ /** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
2137
+ reportedCost: import_zod.z.number().optional(),
2050
2138
  durationSeconds: import_zod.z.number().nonnegative().optional()
2051
2139
  });
2052
2140
  var BaseResultSchema = import_zod.z.object({
@@ -2186,16 +2274,18 @@ function visualAI(config = {}) {
2186
2274
  apiKey: resolvedConfig.apiKey,
2187
2275
  model: resolvedConfig.model,
2188
2276
  maxTokens: resolvedConfig.maxTokens,
2189
- reasoningEffort: resolvedConfig.reasoningEffort
2277
+ reasoningEffort: resolvedConfig.reasoningEffort,
2278
+ imageDetail: resolvedConfig.imageDetail
2190
2279
  };
2191
2280
  const driver = createDriver(resolvedConfig.provider, driverConfig);
2281
+ const maxImageDimension = resolvedConfig.maxImageDimension;
2192
2282
  async function checkElementsVisibility(image, elements, visible, options) {
2193
2283
  const methodName = visible ? "elementsVisible" : "elementsHidden";
2194
2284
  if (elements.length === 0) {
2195
2285
  throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
2196
2286
  }
2197
2287
  return withErrorDebug(resolvedConfig, methodName, async () => {
2198
- const img = await normalizeImage(image);
2288
+ const img = await normalizeImage(image, maxImageDimension);
2199
2289
  const prompt = buildElementsVisibilityPrompt(elements, visible, options);
2200
2290
  debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
2201
2291
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2214,7 +2304,7 @@ function visualAI(config = {}) {
2214
2304
  throw new VisualAIConfigError("At least one statement is required for check()");
2215
2305
  }
2216
2306
  return withErrorDebug(resolvedConfig, "check", async () => {
2217
- const media = await normalizeMedia(input, options?.video);
2307
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2218
2308
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2219
2309
  const prompt = buildCheckPrompt(stmts, {
2220
2310
  instructions: options?.instructions,
@@ -2233,7 +2323,7 @@ function visualAI(config = {}) {
2233
2323
  },
2234
2324
  async ask(input, userPrompt, options) {
2235
2325
  return withErrorDebug(resolvedConfig, "ask", async () => {
2236
- const media = await normalizeMedia(input, options?.video);
2326
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2237
2327
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2238
2328
  const prompt = buildAskPrompt(userPrompt, {
2239
2329
  instructions: options?.instructions,
@@ -2252,7 +2342,10 @@ function visualAI(config = {}) {
2252
2342
  },
2253
2343
  async compare(imageA, imageB, options) {
2254
2344
  return withErrorDebug(resolvedConfig, "compare", async () => {
2255
- const [imgA, imgB] = await Promise.all([normalizeImage(imageA), normalizeImage(imageB)]);
2345
+ const [imgA, imgB] = await Promise.all([
2346
+ normalizeImage(imageA, maxImageDimension),
2347
+ normalizeImage(imageB, maxImageDimension)
2348
+ ]);
2256
2349
  const prompt = buildComparePrompt({
2257
2350
  userPrompt: options?.prompt,
2258
2351
  instructions: options?.instructions
@@ -2290,7 +2383,7 @@ function visualAI(config = {}) {
2290
2383
  },
2291
2384
  async accessibility(image, options) {
2292
2385
  return withErrorDebug(resolvedConfig, "accessibility", async () => {
2293
- const img = await normalizeImage(image);
2386
+ const img = await normalizeImage(image, maxImageDimension);
2294
2387
  const prompt = buildAccessibilityPrompt(options);
2295
2388
  debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
2296
2389
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2309,7 +2402,7 @@ function visualAI(config = {}) {
2309
2402
  },
2310
2403
  async layout(image, options) {
2311
2404
  return withErrorDebug(resolvedConfig, "layout", async () => {
2312
- const img = await normalizeImage(image);
2405
+ const img = await normalizeImage(image, maxImageDimension);
2313
2406
  const prompt = buildLayoutPrompt(options);
2314
2407
  debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
2315
2408
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2323,7 +2416,7 @@ function visualAI(config = {}) {
2323
2416
  },
2324
2417
  async pageLoad(image, options) {
2325
2418
  return withErrorDebug(resolvedConfig, "pageLoad", async () => {
2326
- const img = await normalizeImage(image);
2419
+ const img = await normalizeImage(image, maxImageDimension);
2327
2420
  const prompt = buildPageLoadPrompt(options);
2328
2421
  debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
2329
2422
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2337,7 +2430,7 @@ function visualAI(config = {}) {
2337
2430
  },
2338
2431
  async content(image, options) {
2339
2432
  return withErrorDebug(resolvedConfig, "content", async () => {
2340
- const img = await normalizeImage(image);
2433
+ const img = await normalizeImage(image, maxImageDimension);
2341
2434
  const prompt = buildContentPrompt(options);
2342
2435
  debugLog(resolvedConfig, "content prompt", prompt, "prompt");
2343
2436
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2416,6 +2509,7 @@ function assertVisualCompareResult(result, label) {
2416
2509
  ConfidenceSchema,
2417
2510
  Content,
2418
2511
  DEFAULT_MODELS,
2512
+ ImageDetail,
2419
2513
  IssueCategorySchema,
2420
2514
  IssuePrioritySchema,
2421
2515
  IssueSchema,