visual-ai-assertions 0.14.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -38,6 +38,7 @@ __export(index_exports, {
38
38
  ConfidenceSchema: () => ConfidenceSchema,
39
39
  Content: () => Content,
40
40
  DEFAULT_MODELS: () => DEFAULT_MODELS,
41
+ ImageDetail: () => ImageDetail,
41
42
  IssueCategorySchema: () => IssueCategorySchema,
42
43
  IssuePrioritySchema: () => IssuePrioritySchema,
43
44
  IssueSchema: () => IssueSchema,
@@ -73,14 +74,23 @@ var ReasoningEffort = {
73
74
  HIGH: "high",
74
75
  XHIGH: "xhigh"
75
76
  };
77
+ var ImageDetail = {
78
+ AUTO: "auto",
79
+ LOW: "low",
80
+ HIGH: "high"
81
+ };
82
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
83
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
76
84
  var Provider = {
77
85
  ANTHROPIC: "anthropic",
78
86
  OPENAI: "openai",
79
- GOOGLE: "google"
87
+ GOOGLE: "google",
88
+ OPENROUTER: "openrouter"
80
89
  };
81
90
  var Model = {
82
91
  Anthropic: {
83
92
  FABLE_5: "claude-fable-5",
93
+ OPUS_5: "claude-opus-5",
84
94
  OPUS_4_8: "claude-opus-4-8",
85
95
  OPUS_4_7: "claude-opus-4-7",
86
96
  OPUS_4_6: "claude-opus-4-6",
@@ -101,29 +111,51 @@ var Model = {
101
111
  GPT_5_MINI: "gpt-5-mini"
102
112
  },
103
113
  Google: {
114
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
115
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
116
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
104
117
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
118
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
105
119
  GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
106
120
  GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
107
121
  GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
122
+ },
123
+ /**
124
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
125
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
126
+ * recognizes them. All listed models accept image input.
127
+ */
128
+ OpenRouter: {
129
+ GROK_4_6: "x-ai/grok-4.6",
130
+ GROK_4_5: "x-ai/grok-4.5",
131
+ KIMI_K3: "moonshotai/kimi-k3",
132
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
133
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
134
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
135
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
108
136
  }
109
137
  };
110
138
  var DEFAULT_MODELS = {
111
139
  [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
112
140
  [Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
113
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
141
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
142
+ [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
114
143
  };
115
144
  var DEFAULT_MAX_TOKENS = 4096;
116
145
  var OPENAI_REASONING_MAX_TOKENS = 16384;
117
146
  var MODEL_TO_PROVIDER = new Map([
118
147
  ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
119
148
  ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
120
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
149
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
150
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
121
151
  ]);
122
152
  var VALID_PROVIDERS = Object.values(Provider);
123
153
  var PROVIDER_DEFAULT_REASONING = {
124
154
  openai: "medium",
125
155
  anthropic: "off",
126
- google: "off"
156
+ google: "off",
157
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
158
+ openrouter: "off"
127
159
  };
128
160
  var Content = {
129
161
  /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
@@ -581,6 +613,7 @@ function parseRetryAfter(value) {
581
613
  // src/providers/anthropic.ts
582
614
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
583
615
  Model.Anthropic.FABLE_5,
616
+ Model.Anthropic.OPUS_5,
584
617
  Model.Anthropic.OPUS_4_8,
585
618
  Model.Anthropic.OPUS_4_7,
586
619
  Model.Anthropic.SONNET_5
@@ -589,6 +622,13 @@ function mapEffort(level, model) {
589
622
  if (level !== "xhigh") return level;
590
623
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
591
624
  }
625
+ var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
626
+ var EFFORT_TO_BUDGET_TOKENS = {
627
+ low: 1024,
628
+ medium: 4096,
629
+ high: 8192,
630
+ xhigh: 16384
631
+ };
592
632
  var AnthropicDriver = class {
593
633
  client;
594
634
  model;
@@ -644,10 +684,16 @@ var AnthropicDriver = class {
644
684
  ]
645
685
  };
646
686
  if (this.reasoningEffort) {
647
- requestParams.thinking = { type: "adaptive" };
648
- requestParams.output_config = {
649
- effort: mapEffort(this.reasoningEffort, this.model)
650
- };
687
+ if (BUDGET_THINKING_MODELS.has(this.model)) {
688
+ const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
689
+ requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
690
+ requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
691
+ } else {
692
+ requestParams.thinking = { type: "adaptive" };
693
+ requestParams.output_config = {
694
+ effort: mapEffort(this.reasoningEffort, this.model)
695
+ };
696
+ }
651
697
  }
652
698
  const message = await client.messages.create(requestParams);
653
699
  const textBlock = message.content.find((block) => block.type === "text");
@@ -663,7 +709,10 @@ var AnthropicDriver = class {
663
709
  text,
664
710
  usage: {
665
711
  inputTokens: message.usage.input_tokens,
666
- outputTokens: message.usage.output_tokens
712
+ outputTokens: message.usage.output_tokens,
713
+ ...message.usage.cache_read_input_tokens !== void 0 && {
714
+ cachedInputTokens: message.usage.cache_read_input_tokens
715
+ }
667
716
  }
668
717
  };
669
718
  } catch (err) {
@@ -680,23 +729,41 @@ function needsCodeExecution(model) {
680
729
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
681
730
  }
682
731
  var GOOGLE_THINKING_LEVEL = {
683
- low: "minimal",
684
- medium: "low",
685
- high: "medium",
732
+ low: "low",
733
+ medium: "medium",
734
+ high: "high",
686
735
  xhigh: "high"
687
736
  };
737
+ var GOOGLE_MEDIA_RESOLUTION = {
738
+ low: "MEDIA_RESOLUTION_LOW",
739
+ high: "MEDIA_RESOLUTION_HIGH"
740
+ };
741
+ function toGeminiUsage(um) {
742
+ if (!um) return void 0;
743
+ const thoughts = um.thoughtsTokenCount ?? 0;
744
+ return {
745
+ inputTokens: um.promptTokenCount ?? 0,
746
+ outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
747
+ ...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
748
+ ...um.cachedContentTokenCount !== void 0 && {
749
+ cachedInputTokens: um.cachedContentTokenCount
750
+ }
751
+ };
752
+ }
688
753
  var GoogleDriver = class {
689
754
  client;
690
755
  model;
691
756
  maxTokens;
692
757
  apiKeyOrEnv;
693
758
  reasoningEffort;
759
+ imageDetail;
694
760
  constructor(config) {
695
761
  this.model = config.model;
696
762
  this.maxTokens = config.maxTokens;
697
763
  this.client = null;
698
764
  this.apiKeyOrEnv = config.apiKey;
699
765
  this.reasoningEffort = config.reasoningEffort;
766
+ this.imageDetail = config.imageDetail;
700
767
  }
701
768
  toGeminiParts(images) {
702
769
  return images.map((img) => ({
@@ -736,6 +803,9 @@ var GoogleDriver = class {
736
803
  thinkingConfig: {
737
804
  thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
738
805
  }
806
+ },
807
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
808
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
739
809
  }
740
810
  }
741
811
  });
@@ -753,14 +823,9 @@ var GoogleDriver = class {
753
823
  );
754
824
  }
755
825
  const text = response.text ?? "";
756
- const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
757
826
  return {
758
827
  text,
759
- usage: response.usageMetadata ? {
760
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
761
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
762
- ...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
763
- } : void 0
828
+ usage: toGeminiUsage(response.usageMetadata)
764
829
  };
765
830
  } catch (err) {
766
831
  if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
@@ -791,10 +856,7 @@ var GoogleDriver = class {
791
856
  return {
792
857
  imageData: Buffer.from(imagePart.inlineData.data, "base64"),
793
858
  mimeType: imagePart.inlineData.mimeType,
794
- usage: response.usageMetadata ? {
795
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
796
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
797
- } : void 0
859
+ usage: toGeminiUsage(response.usageMetadata)
798
860
  };
799
861
  } catch (err) {
800
862
  if (err instanceof VisualAIProviderError) throw err;
@@ -810,12 +872,14 @@ var OpenAIDriver = class {
810
872
  maxTokens;
811
873
  apiKeyOrEnv;
812
874
  reasoningEffort;
875
+ imageDetail;
813
876
  constructor(config) {
814
877
  this.model = config.model;
815
878
  this.maxTokens = config.maxTokens;
816
879
  this.client = null;
817
880
  this.apiKeyOrEnv = config.apiKey;
818
881
  this.reasoningEffort = config.reasoningEffort;
882
+ this.imageDetail = config.imageDetail;
819
883
  }
820
884
  async getClient() {
821
885
  if (this.client) return this.client;
@@ -837,9 +901,11 @@ var OpenAIDriver = class {
837
901
  }
838
902
  async sendMessage(images, prompt, options) {
839
903
  const client = await this.getClient();
904
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
840
905
  const imageBlocks = images.map((img) => ({
841
906
  type: "input_image",
842
- image_url: `data:${img.mimeType};base64,${img.base64}`
907
+ image_url: `data:${img.mimeType};base64,${img.base64}`,
908
+ ...detail ? { detail } : {}
843
909
  }));
844
910
  try {
845
911
  const format = options?.responseSchema ? {
@@ -864,21 +930,23 @@ var OpenAIDriver = class {
864
930
  }
865
931
  const response = await client.responses.create(requestParams);
866
932
  if (response.status && response.status !== "completed") {
867
- const detail = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
933
+ const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
868
934
  throw new VisualAITruncationError(
869
- `Response truncated: OpenAI returned status "${response.status}"${detail}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
935
+ `Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
870
936
  response.output_text ?? "",
871
937
  this.maxTokens
872
938
  );
873
939
  }
874
940
  const text = response.output_text ?? "";
875
941
  const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
942
+ const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
876
943
  return {
877
944
  text,
878
945
  usage: response.usage ? {
879
946
  inputTokens: response.usage.input_tokens,
880
947
  outputTokens: response.usage.output_tokens,
881
- ...reasoningTokens !== void 0 && { reasoningTokens }
948
+ ...reasoningTokens !== void 0 && { reasoningTokens },
949
+ ...cachedInputTokens !== void 0 && { cachedInputTokens }
882
950
  } : void 0
883
951
  };
884
952
  } catch (err) {
@@ -888,6 +956,119 @@ var OpenAIDriver = class {
888
956
  }
889
957
  };
890
958
 
959
+ // src/providers/openrouter.ts
960
+ var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
961
+ var OPENROUTER_REASONING_EFFORT = {
962
+ low: "low",
963
+ medium: "medium",
964
+ high: "high",
965
+ xhigh: "high"
966
+ };
967
+ var OpenRouterDriver = class {
968
+ client;
969
+ model;
970
+ maxTokens;
971
+ apiKeyOrEnv;
972
+ reasoningEffort;
973
+ imageDetail;
974
+ constructor(config) {
975
+ this.model = config.model;
976
+ this.maxTokens = config.maxTokens;
977
+ this.client = null;
978
+ this.apiKeyOrEnv = config.apiKey;
979
+ this.reasoningEffort = config.reasoningEffort;
980
+ this.imageDetail = config.imageDetail;
981
+ }
982
+ async getClient() {
983
+ if (this.client) return this.client;
984
+ let OpenAI;
985
+ try {
986
+ const mod = await import("openai");
987
+ OpenAI = mod.default;
988
+ } catch {
989
+ throw new VisualAIConfigError(
990
+ "OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
991
+ );
992
+ }
993
+ const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
994
+ if (!apiKey) {
995
+ throw new VisualAIAuthError(
996
+ "OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
997
+ );
998
+ }
999
+ this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
1000
+ return this.client;
1001
+ }
1002
+ async sendMessage(images, prompt, options) {
1003
+ const client = await this.getClient();
1004
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
1005
+ const imageParts = images.map((img) => ({
1006
+ type: "image_url",
1007
+ image_url: {
1008
+ url: `data:${img.mimeType};base64,${img.base64}`,
1009
+ ...detail ? { detail } : {}
1010
+ }
1011
+ }));
1012
+ try {
1013
+ const responseFormat = options?.responseSchema ? {
1014
+ type: "json_schema",
1015
+ json_schema: {
1016
+ name: "visual_ai_response",
1017
+ strict: true,
1018
+ schema: options.responseSchema
1019
+ }
1020
+ } : { type: "json_object" };
1021
+ const requestParams = {
1022
+ model: this.model,
1023
+ max_tokens: this.maxTokens,
1024
+ response_format: responseFormat,
1025
+ messages: [
1026
+ {
1027
+ role: "user",
1028
+ content: [...imageParts, { type: "text", text: prompt }]
1029
+ }
1030
+ ],
1031
+ // OpenRouter-specific: include token accounting in the response.
1032
+ usage: { include: true }
1033
+ };
1034
+ if (this.reasoningEffort) {
1035
+ requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
1036
+ }
1037
+ const response = await client.chat.completions.create(requestParams);
1038
+ const choice = response.choices?.[0];
1039
+ if (!choice?.message) {
1040
+ throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
1041
+ }
1042
+ const text = choice.message.content ?? "";
1043
+ if (choice.finish_reason === "length") {
1044
+ throw new VisualAITruncationError(
1045
+ `Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
1046
+ text,
1047
+ this.maxTokens
1048
+ );
1049
+ }
1050
+ const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
1051
+ const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
1052
+ const cost = response.usage?.cost;
1053
+ return {
1054
+ text,
1055
+ usage: response.usage ? {
1056
+ inputTokens: response.usage.prompt_tokens,
1057
+ outputTokens: response.usage.completion_tokens,
1058
+ ...reasoningTokens !== void 0 && { reasoningTokens },
1059
+ ...cachedInputTokens !== void 0 && { cachedInputTokens },
1060
+ ...cost !== void 0 && { cost }
1061
+ } : void 0
1062
+ };
1063
+ } catch (err) {
1064
+ if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
1065
+ throw err;
1066
+ }
1067
+ throw mapProviderError(err);
1068
+ }
1069
+ }
1070
+ };
1071
+
891
1072
  // src/core/config.ts
892
1073
  var MODEL_PREFIX_TO_PROVIDER = [
893
1074
  ["claude-", "anthropic"],
@@ -900,6 +1081,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
900
1081
  function inferProviderFromModel(model) {
901
1082
  const known = MODEL_TO_PROVIDER.get(model);
902
1083
  if (known) return known;
1084
+ if (model.includes("/")) return "openrouter";
903
1085
  const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
904
1086
  return prefixMatch?.[1];
905
1087
  }
@@ -912,12 +1094,13 @@ function resolveProvider(config) {
912
1094
  const apiKeyProviderMap = [
913
1095
  ["ANTHROPIC_API_KEY", "anthropic"],
914
1096
  ["OPENAI_API_KEY", "openai"],
915
- ["GOOGLE_API_KEY", "google"]
1097
+ ["GOOGLE_API_KEY", "google"],
1098
+ ["OPENROUTER_API_KEY", "openrouter"]
916
1099
  ];
917
1100
  const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
918
1101
  if (detected) return detected[1];
919
1102
  throw new VisualAIConfigError(
920
- "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
1103
+ "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
921
1104
  );
922
1105
  }
923
1106
  function parseBooleanEnv(envName, value) {
@@ -945,11 +1128,11 @@ function resolveConfig(config) {
945
1128
  }
946
1129
  const userSetMaxTokens = config.maxTokens !== void 0;
947
1130
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
948
- if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
1131
+ if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
949
1132
  maxTokens = OPENAI_REASONING_MAX_TOKENS;
950
1133
  if (debug) {
951
1134
  process.stderr.write(
952
- `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for OpenAI with reasoningEffort "${config.reasoningEffort}".
1135
+ `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
953
1136
  `
954
1137
  );
955
1138
  }
@@ -960,6 +1143,8 @@ function resolveConfig(config) {
960
1143
  model,
961
1144
  maxTokens,
962
1145
  reasoningEffort: config.reasoningEffort,
1146
+ maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1147
+ imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
963
1148
  debug,
964
1149
  debugPrompt,
965
1150
  debugResponse,
@@ -974,6 +1159,10 @@ var PRICING_TABLE = {
974
1159
  inputPricePerToken: 10 / PER_MILLION,
975
1160
  outputPricePerToken: 50 / PER_MILLION
976
1161
  },
1162
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1163
+ inputPricePerToken: 5 / PER_MILLION,
1164
+ outputPricePerToken: 25 / PER_MILLION
1165
+ },
977
1166
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
978
1167
  inputPricePerToken: 5 / PER_MILLION,
979
1168
  outputPricePerToken: 25 / PER_MILLION
@@ -1003,12 +1192,12 @@ var PRICING_TABLE = {
1003
1192
  outputPricePerToken: 30 / PER_MILLION
1004
1193
  },
1005
1194
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
1006
- inputPricePerToken: 2.5 / PER_MILLION,
1007
- outputPricePerToken: 15 / PER_MILLION
1195
+ inputPricePerToken: 2 / PER_MILLION,
1196
+ outputPricePerToken: 12 / PER_MILLION
1008
1197
  },
1009
1198
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
1010
- inputPricePerToken: 1 / PER_MILLION,
1011
- outputPricePerToken: 6 / PER_MILLION
1199
+ inputPricePerToken: 0.2 / PER_MILLION,
1200
+ outputPricePerToken: 1.2 / PER_MILLION
1012
1201
  },
1013
1202
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
1014
1203
  inputPricePerToken: 5 / PER_MILLION,
@@ -1038,10 +1227,30 @@ var PRICING_TABLE = {
1038
1227
  inputPricePerToken: 0.25 / PER_MILLION,
1039
1228
  outputPricePerToken: 2 / PER_MILLION
1040
1229
  },
1230
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1231
+ // on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
1232
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
1233
+ inputPricePerToken: 0.75 / PER_MILLION,
1234
+ outputPricePerToken: 3.75 / PER_MILLION
1235
+ },
1236
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1237
+ // on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
1238
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
1239
+ inputPricePerToken: 0.75 / PER_MILLION,
1240
+ outputPricePerToken: 3.75 / PER_MILLION
1241
+ },
1242
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1243
+ inputPricePerToken: 1.5 / PER_MILLION,
1244
+ outputPricePerToken: 7.5 / PER_MILLION
1245
+ },
1041
1246
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
1042
1247
  inputPricePerToken: 1.5 / PER_MILLION,
1043
1248
  outputPricePerToken: 9 / PER_MILLION
1044
1249
  },
1250
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
1251
+ inputPricePerToken: 0.3 / PER_MILLION,
1252
+ outputPricePerToken: 2.5 / PER_MILLION
1253
+ },
1045
1254
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
1046
1255
  inputPricePerToken: 2 / PER_MILLION,
1047
1256
  outputPricePerToken: 12 / PER_MILLION
@@ -1053,6 +1262,36 @@ var PRICING_TABLE = {
1053
1262
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
1054
1263
  inputPricePerToken: 0.5 / PER_MILLION,
1055
1264
  outputPricePerToken: 3 / PER_MILLION
1265
+ },
1266
+ // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1267
+ // against https://openrouter.ai/api/v1/models).
1268
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1269
+ inputPricePerToken: 2 / PER_MILLION,
1270
+ outputPricePerToken: 6 / PER_MILLION
1271
+ },
1272
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1273
+ inputPricePerToken: 2 / PER_MILLION,
1274
+ outputPricePerToken: 6 / PER_MILLION
1275
+ },
1276
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
1277
+ inputPricePerToken: 3 / PER_MILLION,
1278
+ outputPricePerToken: 15 / PER_MILLION
1279
+ },
1280
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
1281
+ inputPricePerToken: 0.82 / PER_MILLION,
1282
+ outputPricePerToken: 3.75 / PER_MILLION
1283
+ },
1284
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
1285
+ inputPricePerToken: 2 / PER_MILLION,
1286
+ outputPricePerToken: 6 / PER_MILLION
1287
+ },
1288
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1289
+ inputPricePerToken: 0.32 / PER_MILLION,
1290
+ outputPricePerToken: 1.28 / PER_MILLION
1291
+ },
1292
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1293
+ inputPricePerToken: 0.1875 / PER_MILLION,
1294
+ outputPricePerToken: 1.125 / PER_MILLION
1056
1295
  }
1057
1296
  };
1058
1297
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1075,8 +1314,9 @@ function usageLog(config, method, usage) {
1075
1314
  const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
1076
1315
  const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
1077
1316
  const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
1317
+ const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
1078
1318
  process.stderr.write(
1079
- `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1319
+ `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1080
1320
  `
1081
1321
  );
1082
1322
  }
@@ -1087,7 +1327,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
1087
1327
  inputTokens,
1088
1328
  outputTokens,
1089
1329
  ...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
1330
+ ...rawUsage?.cachedInputTokens !== void 0 && {
1331
+ cachedInputTokens: rawUsage.cachedInputTokens
1332
+ },
1090
1333
  estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
1334
+ ...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
1091
1335
  durationSeconds
1092
1336
  };
1093
1337
  usageLog(config, method, usage);
@@ -1130,7 +1374,10 @@ async function timedSendMessage(driver, images, prompt, options) {
1130
1374
  var import_sharp = __toESM(require("sharp"), 1);
1131
1375
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1132
1376
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1133
- Model.Google.GEMINI_3_5_FLASH
1377
+ Model.Google.GEMINI_3_5_FLASH,
1378
+ Model.Google.GEMINI_3_6_FLASH,
1379
+ Model.Google.GEMINI_3_7_FLASH,
1380
+ Model.Google.GEMINI_3_8_FLASH
1134
1381
  ]);
1135
1382
  async function generateAiDiff(imgA, imgB, model, driver) {
1136
1383
  if (!driver.generateImage) {
@@ -1220,7 +1467,6 @@ var EXTENSION_TO_MIME = {
1220
1467
  ".webp": "image/webp",
1221
1468
  ".gif": "image/gif"
1222
1469
  };
1223
- var MAX_DIMENSION = 1568;
1224
1470
  var URL_FETCH_TIMEOUT_MS = 1e4;
1225
1471
  function isSupportedMimeType(value) {
1226
1472
  return SUPPORTED_FORMATS.has(value);
@@ -1244,14 +1490,14 @@ function detectMimeType(data) {
1244
1490
  }
1245
1491
  throw new VisualAIImageError("Unable to detect image format from file content");
1246
1492
  }
1247
- async function resizeIfNeeded(data, mimeType) {
1493
+ async function resizeIfNeeded(data, mimeType, maxDimension) {
1248
1494
  if (mimeType === "image/gif") {
1249
1495
  return data;
1250
1496
  }
1251
1497
  if (mimeType === "image/png" && data.length >= 24) {
1252
1498
  const width2 = data.readUInt32BE(16);
1253
1499
  const height2 = data.readUInt32BE(20);
1254
- if (width2 <= MAX_DIMENSION && height2 <= MAX_DIMENSION) {
1500
+ if (width2 <= maxDimension && height2 <= maxDimension) {
1255
1501
  return data;
1256
1502
  }
1257
1503
  }
@@ -1259,12 +1505,12 @@ async function resizeIfNeeded(data, mimeType) {
1259
1505
  const metadata = await pipeline.metadata();
1260
1506
  const width = metadata.width ?? 0;
1261
1507
  const height = metadata.height ?? 0;
1262
- if (width <= MAX_DIMENSION && height <= MAX_DIMENSION) {
1508
+ if (width <= maxDimension && height <= maxDimension) {
1263
1509
  return data;
1264
1510
  }
1265
1511
  return pipeline.resize({
1266
- width: MAX_DIMENSION,
1267
- height: MAX_DIMENSION,
1512
+ width: maxDimension,
1513
+ height: maxDimension,
1268
1514
  fit: "inside",
1269
1515
  withoutEnlargement: true
1270
1516
  }).toBuffer();
@@ -1319,7 +1565,7 @@ function loadFromBase64(input) {
1319
1565
  }
1320
1566
  return { data, mimeType: mimeType ?? detectMimeType(data) };
1321
1567
  }
1322
- async function normalizeImage(input) {
1568
+ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1323
1569
  let data;
1324
1570
  let mimeType;
1325
1571
  if (Buffer.isBuffer(input)) {
@@ -1348,7 +1594,7 @@ async function normalizeImage(input) {
1348
1594
  "Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
1349
1595
  );
1350
1596
  }
1351
- data = await resizeIfNeeded(data, mimeType);
1597
+ data = await resizeIfNeeded(data, mimeType, maxDimension);
1352
1598
  let cachedBase64;
1353
1599
  return {
1354
1600
  data,
@@ -1652,7 +1898,7 @@ async function probeDurationSeconds(videoPath) {
1652
1898
  });
1653
1899
  });
1654
1900
  }
1655
- async function extractFrames(videoPath, options = {}) {
1901
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
1656
1902
  const fps = options.fps ?? DEFAULT_FPS;
1657
1903
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1658
1904
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1681,7 +1927,7 @@ async function extractFrames(videoPath, options = {}) {
1681
1927
  }
1682
1928
  const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
1683
1929
  try {
1684
- const filter = `fps=${fps},scale='if(gt(iw,ih),min(${FRAME_MAX_DIMENSION},iw),-2)':'if(gt(iw,ih),-2,min(${FRAME_MAX_DIMENSION},ih))':flags=area`;
1930
+ const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
1685
1931
  await new Promise((resolve2, reject) => {
1686
1932
  let settled = false;
1687
1933
  const cmd = ffmpeg(videoPath);
@@ -1779,7 +2025,7 @@ function isFramesInput(input) {
1779
2025
  function isTimestampedFrameInput(frame) {
1780
2026
  return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
1781
2027
  }
1782
- async function normalizeFrames(input) {
2028
+ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1783
2029
  const rawFrames = input.frames;
1784
2030
  const fps = input.fps ?? DEFAULT_FPS;
1785
2031
  if (rawFrames.length === 0) {
@@ -1804,7 +2050,7 @@ async function normalizeFrames(input) {
1804
2050
  `Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
1805
2051
  );
1806
2052
  }
1807
- const image = await normalizeImage(imageInput);
2053
+ const image = await normalizeImage(imageInput, maxDimension);
1808
2054
  return {
1809
2055
  data: image.data,
1810
2056
  mimeType: image.mimeType,
@@ -1820,14 +2066,14 @@ async function normalizeFrames(input) {
1820
2066
  await saveDebugFrames(frames);
1821
2067
  return { kind: "video", frames, durationSeconds };
1822
2068
  }
1823
- async function normalizeMedia(input, videoOptions) {
2069
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1824
2070
  if (isFramesInput(input)) {
1825
- return normalizeFrames(input);
2071
+ return normalizeFrames(input, maxDimension);
1826
2072
  }
1827
2073
  if (isVideoInput(input)) {
1828
2074
  const { path, cleanup } = await resolveVideoToPath(input);
1829
2075
  try {
1830
- const { frames, durationSeconds } = await extractFrames(path, videoOptions);
2076
+ const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
1831
2077
  await saveDebugFrames(frames);
1832
2078
  return { kind: "video", frames, durationSeconds };
1833
2079
  } finally {
@@ -1837,7 +2083,7 @@ async function normalizeMedia(input, videoOptions) {
1837
2083
  }
1838
2084
  }
1839
2085
  }
1840
- const image = await normalizeImage(input);
2086
+ const image = await normalizeImage(input, maxDimension);
1841
2087
  return { kind: "image", image };
1842
2088
  }
1843
2089
 
@@ -1878,7 +2124,17 @@ var UsageInfoSchema = import_zod.z.object({
1878
2124
  outputTokens: import_zod.z.number(),
1879
2125
  /** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
1880
2126
  reasoningTokens: import_zod.z.number().optional(),
2127
+ /**
2128
+ * Prompt tokens served from the provider's cache, when reported. Informational
2129
+ * only — `estimatedCost` does not apply a cache discount, because providers
2130
+ * differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
2131
+ * Google) or billed as a separate bucket alongside it (Anthropic).
2132
+ */
2133
+ cachedInputTokens: import_zod.z.number().optional(),
2134
+ /** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
1881
2135
  estimatedCost: import_zod.z.number().optional(),
2136
+ /** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
2137
+ reportedCost: import_zod.z.number().optional(),
1882
2138
  durationSeconds: import_zod.z.number().nonnegative().optional()
1883
2139
  });
1884
2140
  var BaseResultSchema = import_zod.z.object({
@@ -1980,7 +2236,8 @@ function toSchemaOptions(schema) {
1980
2236
  var PROVIDER_REGISTRY = {
1981
2237
  anthropic: (config) => new AnthropicDriver(config),
1982
2238
  openai: (config) => new OpenAIDriver(config),
1983
- google: (config) => new GoogleDriver(config)
2239
+ google: (config) => new GoogleDriver(config),
2240
+ openrouter: (config) => new OpenRouterDriver(config)
1984
2241
  };
1985
2242
  function createDriver(provider, config) {
1986
2243
  return PROVIDER_REGISTRY[provider](config);
@@ -2017,16 +2274,18 @@ function visualAI(config = {}) {
2017
2274
  apiKey: resolvedConfig.apiKey,
2018
2275
  model: resolvedConfig.model,
2019
2276
  maxTokens: resolvedConfig.maxTokens,
2020
- reasoningEffort: resolvedConfig.reasoningEffort
2277
+ reasoningEffort: resolvedConfig.reasoningEffort,
2278
+ imageDetail: resolvedConfig.imageDetail
2021
2279
  };
2022
2280
  const driver = createDriver(resolvedConfig.provider, driverConfig);
2281
+ const maxImageDimension = resolvedConfig.maxImageDimension;
2023
2282
  async function checkElementsVisibility(image, elements, visible, options) {
2024
2283
  const methodName = visible ? "elementsVisible" : "elementsHidden";
2025
2284
  if (elements.length === 0) {
2026
2285
  throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
2027
2286
  }
2028
2287
  return withErrorDebug(resolvedConfig, methodName, async () => {
2029
- const img = await normalizeImage(image);
2288
+ const img = await normalizeImage(image, maxImageDimension);
2030
2289
  const prompt = buildElementsVisibilityPrompt(elements, visible, options);
2031
2290
  debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
2032
2291
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2045,7 +2304,7 @@ function visualAI(config = {}) {
2045
2304
  throw new VisualAIConfigError("At least one statement is required for check()");
2046
2305
  }
2047
2306
  return withErrorDebug(resolvedConfig, "check", async () => {
2048
- const media = await normalizeMedia(input, options?.video);
2307
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2049
2308
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2050
2309
  const prompt = buildCheckPrompt(stmts, {
2051
2310
  instructions: options?.instructions,
@@ -2064,7 +2323,7 @@ function visualAI(config = {}) {
2064
2323
  },
2065
2324
  async ask(input, userPrompt, options) {
2066
2325
  return withErrorDebug(resolvedConfig, "ask", async () => {
2067
- const media = await normalizeMedia(input, options?.video);
2326
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2068
2327
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2069
2328
  const prompt = buildAskPrompt(userPrompt, {
2070
2329
  instructions: options?.instructions,
@@ -2083,7 +2342,10 @@ function visualAI(config = {}) {
2083
2342
  },
2084
2343
  async compare(imageA, imageB, options) {
2085
2344
  return withErrorDebug(resolvedConfig, "compare", async () => {
2086
- const [imgA, imgB] = await Promise.all([normalizeImage(imageA), normalizeImage(imageB)]);
2345
+ const [imgA, imgB] = await Promise.all([
2346
+ normalizeImage(imageA, maxImageDimension),
2347
+ normalizeImage(imageB, maxImageDimension)
2348
+ ]);
2087
2349
  const prompt = buildComparePrompt({
2088
2350
  userPrompt: options?.prompt,
2089
2351
  instructions: options?.instructions
@@ -2121,7 +2383,7 @@ function visualAI(config = {}) {
2121
2383
  },
2122
2384
  async accessibility(image, options) {
2123
2385
  return withErrorDebug(resolvedConfig, "accessibility", async () => {
2124
- const img = await normalizeImage(image);
2386
+ const img = await normalizeImage(image, maxImageDimension);
2125
2387
  const prompt = buildAccessibilityPrompt(options);
2126
2388
  debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
2127
2389
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2140,7 +2402,7 @@ function visualAI(config = {}) {
2140
2402
  },
2141
2403
  async layout(image, options) {
2142
2404
  return withErrorDebug(resolvedConfig, "layout", async () => {
2143
- const img = await normalizeImage(image);
2405
+ const img = await normalizeImage(image, maxImageDimension);
2144
2406
  const prompt = buildLayoutPrompt(options);
2145
2407
  debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
2146
2408
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2154,7 +2416,7 @@ function visualAI(config = {}) {
2154
2416
  },
2155
2417
  async pageLoad(image, options) {
2156
2418
  return withErrorDebug(resolvedConfig, "pageLoad", async () => {
2157
- const img = await normalizeImage(image);
2419
+ const img = await normalizeImage(image, maxImageDimension);
2158
2420
  const prompt = buildPageLoadPrompt(options);
2159
2421
  debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
2160
2422
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2168,7 +2430,7 @@ function visualAI(config = {}) {
2168
2430
  },
2169
2431
  async content(image, options) {
2170
2432
  return withErrorDebug(resolvedConfig, "content", async () => {
2171
- const img = await normalizeImage(image);
2433
+ const img = await normalizeImage(image, maxImageDimension);
2172
2434
  const prompt = buildContentPrompt(options);
2173
2435
  debugLog(resolvedConfig, "content prompt", prompt, "prompt");
2174
2436
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2247,6 +2509,7 @@ function assertVisualCompareResult(result, label) {
2247
2509
  ConfidenceSchema,
2248
2510
  Content,
2249
2511
  DEFAULT_MODELS,
2512
+ ImageDetail,
2250
2513
  IssueCategorySchema,
2251
2514
  IssuePrioritySchema,
2252
2515
  IssueSchema,