visual-ai-assertions 0.14.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -5,14 +5,23 @@ var ReasoningEffort = {
5
5
  HIGH: "high",
6
6
  XHIGH: "xhigh"
7
7
  };
8
+ var ImageDetail = {
9
+ AUTO: "auto",
10
+ LOW: "low",
11
+ HIGH: "high"
12
+ };
13
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
14
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
8
15
  var Provider = {
9
16
  ANTHROPIC: "anthropic",
10
17
  OPENAI: "openai",
11
- GOOGLE: "google"
18
+ GOOGLE: "google",
19
+ OPENROUTER: "openrouter"
12
20
  };
13
21
  var Model = {
14
22
  Anthropic: {
15
23
  FABLE_5: "claude-fable-5",
24
+ OPUS_5: "claude-opus-5",
16
25
  OPUS_4_8: "claude-opus-4-8",
17
26
  OPUS_4_7: "claude-opus-4-7",
18
27
  OPUS_4_6: "claude-opus-4-6",
@@ -33,29 +42,51 @@ var Model = {
33
42
  GPT_5_MINI: "gpt-5-mini"
34
43
  },
35
44
  Google: {
45
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
46
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
47
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
36
48
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
49
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
37
50
  GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
38
51
  GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
39
52
  GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
53
+ },
54
+ /**
55
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
56
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
57
+ * recognizes them. All listed models accept image input.
58
+ */
59
+ OpenRouter: {
60
+ GROK_4_6: "x-ai/grok-4.6",
61
+ GROK_4_5: "x-ai/grok-4.5",
62
+ KIMI_K3: "moonshotai/kimi-k3",
63
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
64
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
65
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
66
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
40
67
  }
41
68
  };
42
69
  var DEFAULT_MODELS = {
43
70
  [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
44
71
  [Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
45
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
72
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
73
+ [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
46
74
  };
47
75
  var DEFAULT_MAX_TOKENS = 4096;
48
76
  var OPENAI_REASONING_MAX_TOKENS = 16384;
49
77
  var MODEL_TO_PROVIDER = new Map([
50
78
  ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
51
79
  ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
52
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
80
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
81
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
53
82
  ]);
54
83
  var VALID_PROVIDERS = Object.values(Provider);
55
84
  var PROVIDER_DEFAULT_REASONING = {
56
85
  openai: "medium",
57
86
  anthropic: "off",
58
- google: "off"
87
+ google: "off",
88
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
89
+ openrouter: "off"
59
90
  };
60
91
  var Content = {
61
92
  /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
@@ -513,6 +544,7 @@ function parseRetryAfter(value) {
513
544
  // src/providers/anthropic.ts
514
545
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
515
546
  Model.Anthropic.FABLE_5,
547
+ Model.Anthropic.OPUS_5,
516
548
  Model.Anthropic.OPUS_4_8,
517
549
  Model.Anthropic.OPUS_4_7,
518
550
  Model.Anthropic.SONNET_5
@@ -521,6 +553,13 @@ function mapEffort(level, model) {
521
553
  if (level !== "xhigh") return level;
522
554
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
523
555
  }
556
+ var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
557
+ var EFFORT_TO_BUDGET_TOKENS = {
558
+ low: 1024,
559
+ medium: 4096,
560
+ high: 8192,
561
+ xhigh: 16384
562
+ };
524
563
  var AnthropicDriver = class {
525
564
  client;
526
565
  model;
@@ -576,10 +615,16 @@ var AnthropicDriver = class {
576
615
  ]
577
616
  };
578
617
  if (this.reasoningEffort) {
579
- requestParams.thinking = { type: "adaptive" };
580
- requestParams.output_config = {
581
- effort: mapEffort(this.reasoningEffort, this.model)
582
- };
618
+ if (BUDGET_THINKING_MODELS.has(this.model)) {
619
+ const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
620
+ requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
621
+ requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
622
+ } else {
623
+ requestParams.thinking = { type: "adaptive" };
624
+ requestParams.output_config = {
625
+ effort: mapEffort(this.reasoningEffort, this.model)
626
+ };
627
+ }
583
628
  }
584
629
  const message = await client.messages.create(requestParams);
585
630
  const textBlock = message.content.find((block) => block.type === "text");
@@ -595,7 +640,10 @@ var AnthropicDriver = class {
595
640
  text,
596
641
  usage: {
597
642
  inputTokens: message.usage.input_tokens,
598
- outputTokens: message.usage.output_tokens
643
+ outputTokens: message.usage.output_tokens,
644
+ ...message.usage.cache_read_input_tokens !== void 0 && {
645
+ cachedInputTokens: message.usage.cache_read_input_tokens
646
+ }
599
647
  }
600
648
  };
601
649
  } catch (err) {
@@ -612,23 +660,41 @@ function needsCodeExecution(model) {
612
660
  return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
613
661
  }
614
662
  var GOOGLE_THINKING_LEVEL = {
615
- low: "minimal",
616
- medium: "low",
617
- high: "medium",
663
+ low: "low",
664
+ medium: "medium",
665
+ high: "high",
618
666
  xhigh: "high"
619
667
  };
668
+ var GOOGLE_MEDIA_RESOLUTION = {
669
+ low: "MEDIA_RESOLUTION_LOW",
670
+ high: "MEDIA_RESOLUTION_HIGH"
671
+ };
672
+ function toGeminiUsage(um) {
673
+ if (!um) return void 0;
674
+ const thoughts = um.thoughtsTokenCount ?? 0;
675
+ return {
676
+ inputTokens: um.promptTokenCount ?? 0,
677
+ outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
678
+ ...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
679
+ ...um.cachedContentTokenCount !== void 0 && {
680
+ cachedInputTokens: um.cachedContentTokenCount
681
+ }
682
+ };
683
+ }
620
684
  var GoogleDriver = class {
621
685
  client;
622
686
  model;
623
687
  maxTokens;
624
688
  apiKeyOrEnv;
625
689
  reasoningEffort;
690
+ imageDetail;
626
691
  constructor(config) {
627
692
  this.model = config.model;
628
693
  this.maxTokens = config.maxTokens;
629
694
  this.client = null;
630
695
  this.apiKeyOrEnv = config.apiKey;
631
696
  this.reasoningEffort = config.reasoningEffort;
697
+ this.imageDetail = config.imageDetail;
632
698
  }
633
699
  toGeminiParts(images) {
634
700
  return images.map((img) => ({
@@ -668,6 +734,9 @@ var GoogleDriver = class {
668
734
  thinkingConfig: {
669
735
  thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
670
736
  }
737
+ },
738
+ ...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
739
+ mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
671
740
  }
672
741
  }
673
742
  });
@@ -685,14 +754,9 @@ var GoogleDriver = class {
685
754
  );
686
755
  }
687
756
  const text = response.text ?? "";
688
- const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
689
757
  return {
690
758
  text,
691
- usage: response.usageMetadata ? {
692
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
693
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
694
- ...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
695
- } : void 0
759
+ usage: toGeminiUsage(response.usageMetadata)
696
760
  };
697
761
  } catch (err) {
698
762
  if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
@@ -723,10 +787,7 @@ var GoogleDriver = class {
723
787
  return {
724
788
  imageData: Buffer.from(imagePart.inlineData.data, "base64"),
725
789
  mimeType: imagePart.inlineData.mimeType,
726
- usage: response.usageMetadata ? {
727
- inputTokens: response.usageMetadata.promptTokenCount ?? 0,
728
- outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
729
- } : void 0
790
+ usage: toGeminiUsage(response.usageMetadata)
730
791
  };
731
792
  } catch (err) {
732
793
  if (err instanceof VisualAIProviderError) throw err;
@@ -742,12 +803,14 @@ var OpenAIDriver = class {
742
803
  maxTokens;
743
804
  apiKeyOrEnv;
744
805
  reasoningEffort;
806
+ imageDetail;
745
807
  constructor(config) {
746
808
  this.model = config.model;
747
809
  this.maxTokens = config.maxTokens;
748
810
  this.client = null;
749
811
  this.apiKeyOrEnv = config.apiKey;
750
812
  this.reasoningEffort = config.reasoningEffort;
813
+ this.imageDetail = config.imageDetail;
751
814
  }
752
815
  async getClient() {
753
816
  if (this.client) return this.client;
@@ -769,9 +832,11 @@ var OpenAIDriver = class {
769
832
  }
770
833
  async sendMessage(images, prompt, options) {
771
834
  const client = await this.getClient();
835
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
772
836
  const imageBlocks = images.map((img) => ({
773
837
  type: "input_image",
774
- image_url: `data:${img.mimeType};base64,${img.base64}`
838
+ image_url: `data:${img.mimeType};base64,${img.base64}`,
839
+ ...detail ? { detail } : {}
775
840
  }));
776
841
  try {
777
842
  const format = options?.responseSchema ? {
@@ -796,21 +861,23 @@ var OpenAIDriver = class {
796
861
  }
797
862
  const response = await client.responses.create(requestParams);
798
863
  if (response.status && response.status !== "completed") {
799
- const detail = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
864
+ const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
800
865
  throw new VisualAITruncationError(
801
- `Response truncated: OpenAI returned status "${response.status}"${detail}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
866
+ `Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
802
867
  response.output_text ?? "",
803
868
  this.maxTokens
804
869
  );
805
870
  }
806
871
  const text = response.output_text ?? "";
807
872
  const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
873
+ const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
808
874
  return {
809
875
  text,
810
876
  usage: response.usage ? {
811
877
  inputTokens: response.usage.input_tokens,
812
878
  outputTokens: response.usage.output_tokens,
813
- ...reasoningTokens !== void 0 && { reasoningTokens }
879
+ ...reasoningTokens !== void 0 && { reasoningTokens },
880
+ ...cachedInputTokens !== void 0 && { cachedInputTokens }
814
881
  } : void 0
815
882
  };
816
883
  } catch (err) {
@@ -820,6 +887,119 @@ var OpenAIDriver = class {
820
887
  }
821
888
  };
822
889
 
890
+ // src/providers/openrouter.ts
891
+ var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
892
+ var OPENROUTER_REASONING_EFFORT = {
893
+ low: "low",
894
+ medium: "medium",
895
+ high: "high",
896
+ xhigh: "high"
897
+ };
898
+ var OpenRouterDriver = class {
899
+ client;
900
+ model;
901
+ maxTokens;
902
+ apiKeyOrEnv;
903
+ reasoningEffort;
904
+ imageDetail;
905
+ constructor(config) {
906
+ this.model = config.model;
907
+ this.maxTokens = config.maxTokens;
908
+ this.client = null;
909
+ this.apiKeyOrEnv = config.apiKey;
910
+ this.reasoningEffort = config.reasoningEffort;
911
+ this.imageDetail = config.imageDetail;
912
+ }
913
+ async getClient() {
914
+ if (this.client) return this.client;
915
+ let OpenAI;
916
+ try {
917
+ const mod = await import("openai");
918
+ OpenAI = mod.default;
919
+ } catch {
920
+ throw new VisualAIConfigError(
921
+ "OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
922
+ );
923
+ }
924
+ const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
925
+ if (!apiKey) {
926
+ throw new VisualAIAuthError(
927
+ "OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
928
+ );
929
+ }
930
+ this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
931
+ return this.client;
932
+ }
933
+ async sendMessage(images, prompt, options) {
934
+ const client = await this.getClient();
935
+ const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
936
+ const imageParts = images.map((img) => ({
937
+ type: "image_url",
938
+ image_url: {
939
+ url: `data:${img.mimeType};base64,${img.base64}`,
940
+ ...detail ? { detail } : {}
941
+ }
942
+ }));
943
+ try {
944
+ const responseFormat = options?.responseSchema ? {
945
+ type: "json_schema",
946
+ json_schema: {
947
+ name: "visual_ai_response",
948
+ strict: true,
949
+ schema: options.responseSchema
950
+ }
951
+ } : { type: "json_object" };
952
+ const requestParams = {
953
+ model: this.model,
954
+ max_tokens: this.maxTokens,
955
+ response_format: responseFormat,
956
+ messages: [
957
+ {
958
+ role: "user",
959
+ content: [...imageParts, { type: "text", text: prompt }]
960
+ }
961
+ ],
962
+ // OpenRouter-specific: include token accounting in the response.
963
+ usage: { include: true }
964
+ };
965
+ if (this.reasoningEffort) {
966
+ requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
967
+ }
968
+ const response = await client.chat.completions.create(requestParams);
969
+ const choice = response.choices?.[0];
970
+ if (!choice?.message) {
971
+ throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
972
+ }
973
+ const text = choice.message.content ?? "";
974
+ if (choice.finish_reason === "length") {
975
+ throw new VisualAITruncationError(
976
+ `Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
977
+ text,
978
+ this.maxTokens
979
+ );
980
+ }
981
+ const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
982
+ const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
983
+ const cost = response.usage?.cost;
984
+ return {
985
+ text,
986
+ usage: response.usage ? {
987
+ inputTokens: response.usage.prompt_tokens,
988
+ outputTokens: response.usage.completion_tokens,
989
+ ...reasoningTokens !== void 0 && { reasoningTokens },
990
+ ...cachedInputTokens !== void 0 && { cachedInputTokens },
991
+ ...cost !== void 0 && { cost }
992
+ } : void 0
993
+ };
994
+ } catch (err) {
995
+ if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
996
+ throw err;
997
+ }
998
+ throw mapProviderError(err);
999
+ }
1000
+ }
1001
+ };
1002
+
823
1003
  // src/core/config.ts
824
1004
  var MODEL_PREFIX_TO_PROVIDER = [
825
1005
  ["claude-", "anthropic"],
@@ -832,6 +1012,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
832
1012
  function inferProviderFromModel(model) {
833
1013
  const known = MODEL_TO_PROVIDER.get(model);
834
1014
  if (known) return known;
1015
+ if (model.includes("/")) return "openrouter";
835
1016
  const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
836
1017
  return prefixMatch?.[1];
837
1018
  }
@@ -844,12 +1025,13 @@ function resolveProvider(config) {
844
1025
  const apiKeyProviderMap = [
845
1026
  ["ANTHROPIC_API_KEY", "anthropic"],
846
1027
  ["OPENAI_API_KEY", "openai"],
847
- ["GOOGLE_API_KEY", "google"]
1028
+ ["GOOGLE_API_KEY", "google"],
1029
+ ["OPENROUTER_API_KEY", "openrouter"]
848
1030
  ];
849
1031
  const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
850
1032
  if (detected) return detected[1];
851
1033
  throw new VisualAIConfigError(
852
- "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
1034
+ "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
853
1035
  );
854
1036
  }
855
1037
  function parseBooleanEnv(envName, value) {
@@ -877,11 +1059,11 @@ function resolveConfig(config) {
877
1059
  }
878
1060
  const userSetMaxTokens = config.maxTokens !== void 0;
879
1061
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
880
- if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
1062
+ if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
881
1063
  maxTokens = OPENAI_REASONING_MAX_TOKENS;
882
1064
  if (debug) {
883
1065
  process.stderr.write(
884
- `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for OpenAI with reasoningEffort "${config.reasoningEffort}".
1066
+ `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
885
1067
  `
886
1068
  );
887
1069
  }
@@ -892,6 +1074,8 @@ function resolveConfig(config) {
892
1074
  model,
893
1075
  maxTokens,
894
1076
  reasoningEffort: config.reasoningEffort,
1077
+ maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1078
+ imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
895
1079
  debug,
896
1080
  debugPrompt,
897
1081
  debugResponse,
@@ -906,6 +1090,10 @@ var PRICING_TABLE = {
906
1090
  inputPricePerToken: 10 / PER_MILLION,
907
1091
  outputPricePerToken: 50 / PER_MILLION
908
1092
  },
1093
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1094
+ inputPricePerToken: 5 / PER_MILLION,
1095
+ outputPricePerToken: 25 / PER_MILLION
1096
+ },
909
1097
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
910
1098
  inputPricePerToken: 5 / PER_MILLION,
911
1099
  outputPricePerToken: 25 / PER_MILLION
@@ -935,12 +1123,12 @@ var PRICING_TABLE = {
935
1123
  outputPricePerToken: 30 / PER_MILLION
936
1124
  },
937
1125
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
938
- inputPricePerToken: 2.5 / PER_MILLION,
939
- outputPricePerToken: 15 / PER_MILLION
1126
+ inputPricePerToken: 2 / PER_MILLION,
1127
+ outputPricePerToken: 12 / PER_MILLION
940
1128
  },
941
1129
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
942
- inputPricePerToken: 1 / PER_MILLION,
943
- outputPricePerToken: 6 / PER_MILLION
1130
+ inputPricePerToken: 0.2 / PER_MILLION,
1131
+ outputPricePerToken: 1.2 / PER_MILLION
944
1132
  },
945
1133
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
946
1134
  inputPricePerToken: 5 / PER_MILLION,
@@ -970,10 +1158,30 @@ var PRICING_TABLE = {
970
1158
  inputPricePerToken: 0.25 / PER_MILLION,
971
1159
  outputPricePerToken: 2 / PER_MILLION
972
1160
  },
1161
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1162
+ // on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
1163
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
1164
+ inputPricePerToken: 0.75 / PER_MILLION,
1165
+ outputPricePerToken: 3.75 / PER_MILLION
1166
+ },
1167
+ // Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
1168
+ // on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
1169
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
1170
+ inputPricePerToken: 0.75 / PER_MILLION,
1171
+ outputPricePerToken: 3.75 / PER_MILLION
1172
+ },
1173
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1174
+ inputPricePerToken: 1.5 / PER_MILLION,
1175
+ outputPricePerToken: 7.5 / PER_MILLION
1176
+ },
973
1177
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
974
1178
  inputPricePerToken: 1.5 / PER_MILLION,
975
1179
  outputPricePerToken: 9 / PER_MILLION
976
1180
  },
1181
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
1182
+ inputPricePerToken: 0.3 / PER_MILLION,
1183
+ outputPricePerToken: 2.5 / PER_MILLION
1184
+ },
977
1185
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
978
1186
  inputPricePerToken: 2 / PER_MILLION,
979
1187
  outputPricePerToken: 12 / PER_MILLION
@@ -985,6 +1193,36 @@ var PRICING_TABLE = {
985
1193
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
986
1194
  inputPricePerToken: 0.5 / PER_MILLION,
987
1195
  outputPricePerToken: 3 / PER_MILLION
1196
+ },
1197
+ // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1198
+ // against https://openrouter.ai/api/v1/models).
1199
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1200
+ inputPricePerToken: 2 / PER_MILLION,
1201
+ outputPricePerToken: 6 / PER_MILLION
1202
+ },
1203
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1204
+ inputPricePerToken: 2 / PER_MILLION,
1205
+ outputPricePerToken: 6 / PER_MILLION
1206
+ },
1207
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
1208
+ inputPricePerToken: 3 / PER_MILLION,
1209
+ outputPricePerToken: 15 / PER_MILLION
1210
+ },
1211
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
1212
+ inputPricePerToken: 0.82 / PER_MILLION,
1213
+ outputPricePerToken: 3.75 / PER_MILLION
1214
+ },
1215
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
1216
+ inputPricePerToken: 2 / PER_MILLION,
1217
+ outputPricePerToken: 6 / PER_MILLION
1218
+ },
1219
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1220
+ inputPricePerToken: 0.32 / PER_MILLION,
1221
+ outputPricePerToken: 1.28 / PER_MILLION
1222
+ },
1223
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1224
+ inputPricePerToken: 0.1875 / PER_MILLION,
1225
+ outputPricePerToken: 1.125 / PER_MILLION
988
1226
  }
989
1227
  };
990
1228
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1007,8 +1245,9 @@ function usageLog(config, method, usage) {
1007
1245
  const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
1008
1246
  const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
1009
1247
  const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
1248
+ const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
1010
1249
  process.stderr.write(
1011
- `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1250
+ `[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
1012
1251
  `
1013
1252
  );
1014
1253
  }
@@ -1019,7 +1258,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
1019
1258
  inputTokens,
1020
1259
  outputTokens,
1021
1260
  ...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
1261
+ ...rawUsage?.cachedInputTokens !== void 0 && {
1262
+ cachedInputTokens: rawUsage.cachedInputTokens
1263
+ },
1022
1264
  estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
1265
+ ...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
1023
1266
  durationSeconds
1024
1267
  };
1025
1268
  usageLog(config, method, usage);
@@ -1062,7 +1305,10 @@ async function timedSendMessage(driver, images, prompt, options) {
1062
1305
  import sharp from "sharp";
1063
1306
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1064
1307
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1065
- Model.Google.GEMINI_3_5_FLASH
1308
+ Model.Google.GEMINI_3_5_FLASH,
1309
+ Model.Google.GEMINI_3_6_FLASH,
1310
+ Model.Google.GEMINI_3_7_FLASH,
1311
+ Model.Google.GEMINI_3_8_FLASH
1066
1312
  ]);
1067
1313
  async function generateAiDiff(imgA, imgB, model, driver) {
1068
1314
  if (!driver.generateImage) {
@@ -1152,7 +1398,6 @@ var EXTENSION_TO_MIME = {
1152
1398
  ".webp": "image/webp",
1153
1399
  ".gif": "image/gif"
1154
1400
  };
1155
- var MAX_DIMENSION = 1568;
1156
1401
  var URL_FETCH_TIMEOUT_MS = 1e4;
1157
1402
  function isSupportedMimeType(value) {
1158
1403
  return SUPPORTED_FORMATS.has(value);
@@ -1176,14 +1421,14 @@ function detectMimeType(data) {
1176
1421
  }
1177
1422
  throw new VisualAIImageError("Unable to detect image format from file content");
1178
1423
  }
1179
- async function resizeIfNeeded(data, mimeType) {
1424
+ async function resizeIfNeeded(data, mimeType, maxDimension) {
1180
1425
  if (mimeType === "image/gif") {
1181
1426
  return data;
1182
1427
  }
1183
1428
  if (mimeType === "image/png" && data.length >= 24) {
1184
1429
  const width2 = data.readUInt32BE(16);
1185
1430
  const height2 = data.readUInt32BE(20);
1186
- if (width2 <= MAX_DIMENSION && height2 <= MAX_DIMENSION) {
1431
+ if (width2 <= maxDimension && height2 <= maxDimension) {
1187
1432
  return data;
1188
1433
  }
1189
1434
  }
@@ -1191,12 +1436,12 @@ async function resizeIfNeeded(data, mimeType) {
1191
1436
  const metadata = await pipeline.metadata();
1192
1437
  const width = metadata.width ?? 0;
1193
1438
  const height = metadata.height ?? 0;
1194
- if (width <= MAX_DIMENSION && height <= MAX_DIMENSION) {
1439
+ if (width <= maxDimension && height <= maxDimension) {
1195
1440
  return data;
1196
1441
  }
1197
1442
  return pipeline.resize({
1198
- width: MAX_DIMENSION,
1199
- height: MAX_DIMENSION,
1443
+ width: maxDimension,
1444
+ height: maxDimension,
1200
1445
  fit: "inside",
1201
1446
  withoutEnlargement: true
1202
1447
  }).toBuffer();
@@ -1251,7 +1496,7 @@ function loadFromBase64(input) {
1251
1496
  }
1252
1497
  return { data, mimeType: mimeType ?? detectMimeType(data) };
1253
1498
  }
1254
- async function normalizeImage(input) {
1499
+ async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1255
1500
  let data;
1256
1501
  let mimeType;
1257
1502
  if (Buffer.isBuffer(input)) {
@@ -1280,7 +1525,7 @@ async function normalizeImage(input) {
1280
1525
  "Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
1281
1526
  );
1282
1527
  }
1283
- data = await resizeIfNeeded(data, mimeType);
1528
+ data = await resizeIfNeeded(data, mimeType, maxDimension);
1284
1529
  let cachedBase64;
1285
1530
  return {
1286
1531
  data,
@@ -1584,7 +1829,7 @@ async function probeDurationSeconds(videoPath) {
1584
1829
  });
1585
1830
  });
1586
1831
  }
1587
- async function extractFrames(videoPath, options = {}) {
1832
+ async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
1588
1833
  const fps = options.fps ?? DEFAULT_FPS;
1589
1834
  const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
1590
1835
  const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
@@ -1613,7 +1858,7 @@ async function extractFrames(videoPath, options = {}) {
1613
1858
  }
1614
1859
  const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
1615
1860
  try {
1616
- const filter = `fps=${fps},scale='if(gt(iw,ih),min(${FRAME_MAX_DIMENSION},iw),-2)':'if(gt(iw,ih),-2,min(${FRAME_MAX_DIMENSION},ih))':flags=area`;
1861
+ const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
1617
1862
  await new Promise((resolve2, reject) => {
1618
1863
  let settled = false;
1619
1864
  const cmd = ffmpeg(videoPath);
@@ -1711,7 +1956,7 @@ function isFramesInput(input) {
1711
1956
  function isTimestampedFrameInput(frame) {
1712
1957
  return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
1713
1958
  }
1714
- async function normalizeFrames(input) {
1959
+ async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1715
1960
  const rawFrames = input.frames;
1716
1961
  const fps = input.fps ?? DEFAULT_FPS;
1717
1962
  if (rawFrames.length === 0) {
@@ -1736,7 +1981,7 @@ async function normalizeFrames(input) {
1736
1981
  `Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
1737
1982
  );
1738
1983
  }
1739
- const image = await normalizeImage(imageInput);
1984
+ const image = await normalizeImage(imageInput, maxDimension);
1740
1985
  return {
1741
1986
  data: image.data,
1742
1987
  mimeType: image.mimeType,
@@ -1752,14 +1997,14 @@ async function normalizeFrames(input) {
1752
1997
  await saveDebugFrames(frames);
1753
1998
  return { kind: "video", frames, durationSeconds };
1754
1999
  }
1755
- async function normalizeMedia(input, videoOptions) {
2000
+ async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
1756
2001
  if (isFramesInput(input)) {
1757
- return normalizeFrames(input);
2002
+ return normalizeFrames(input, maxDimension);
1758
2003
  }
1759
2004
  if (isVideoInput(input)) {
1760
2005
  const { path, cleanup } = await resolveVideoToPath(input);
1761
2006
  try {
1762
- const { frames, durationSeconds } = await extractFrames(path, videoOptions);
2007
+ const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
1763
2008
  await saveDebugFrames(frames);
1764
2009
  return { kind: "video", frames, durationSeconds };
1765
2010
  } finally {
@@ -1769,7 +2014,7 @@ async function normalizeMedia(input, videoOptions) {
1769
2014
  }
1770
2015
  }
1771
2016
  }
1772
- const image = await normalizeImage(input);
2017
+ const image = await normalizeImage(input, maxDimension);
1773
2018
  return { kind: "image", image };
1774
2019
  }
1775
2020
 
@@ -1810,7 +2055,17 @@ var UsageInfoSchema = z.object({
1810
2055
  outputTokens: z.number(),
1811
2056
  /** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
1812
2057
  reasoningTokens: z.number().optional(),
2058
+ /**
2059
+ * Prompt tokens served from the provider's cache, when reported. Informational
2060
+ * only — `estimatedCost` does not apply a cache discount, because providers
2061
+ * differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
2062
+ * Google) or billed as a separate bucket alongside it (Anthropic).
2063
+ */
2064
+ cachedInputTokens: z.number().optional(),
2065
+ /** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
1813
2066
  estimatedCost: z.number().optional(),
2067
+ /** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
2068
+ reportedCost: z.number().optional(),
1814
2069
  durationSeconds: z.number().nonnegative().optional()
1815
2070
  });
1816
2071
  var BaseResultSchema = z.object({
@@ -1912,7 +2167,8 @@ function toSchemaOptions(schema) {
1912
2167
  var PROVIDER_REGISTRY = {
1913
2168
  anthropic: (config) => new AnthropicDriver(config),
1914
2169
  openai: (config) => new OpenAIDriver(config),
1915
- google: (config) => new GoogleDriver(config)
2170
+ google: (config) => new GoogleDriver(config),
2171
+ openrouter: (config) => new OpenRouterDriver(config)
1916
2172
  };
1917
2173
  function createDriver(provider, config) {
1918
2174
  return PROVIDER_REGISTRY[provider](config);
@@ -1949,16 +2205,18 @@ function visualAI(config = {}) {
1949
2205
  apiKey: resolvedConfig.apiKey,
1950
2206
  model: resolvedConfig.model,
1951
2207
  maxTokens: resolvedConfig.maxTokens,
1952
- reasoningEffort: resolvedConfig.reasoningEffort
2208
+ reasoningEffort: resolvedConfig.reasoningEffort,
2209
+ imageDetail: resolvedConfig.imageDetail
1953
2210
  };
1954
2211
  const driver = createDriver(resolvedConfig.provider, driverConfig);
2212
+ const maxImageDimension = resolvedConfig.maxImageDimension;
1955
2213
  async function checkElementsVisibility(image, elements, visible, options) {
1956
2214
  const methodName = visible ? "elementsVisible" : "elementsHidden";
1957
2215
  if (elements.length === 0) {
1958
2216
  throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
1959
2217
  }
1960
2218
  return withErrorDebug(resolvedConfig, methodName, async () => {
1961
- const img = await normalizeImage(image);
2219
+ const img = await normalizeImage(image, maxImageDimension);
1962
2220
  const prompt = buildElementsVisibilityPrompt(elements, visible, options);
1963
2221
  debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
1964
2222
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -1977,7 +2235,7 @@ function visualAI(config = {}) {
1977
2235
  throw new VisualAIConfigError("At least one statement is required for check()");
1978
2236
  }
1979
2237
  return withErrorDebug(resolvedConfig, "check", async () => {
1980
- const media = await normalizeMedia(input, options?.video);
2238
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
1981
2239
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
1982
2240
  const prompt = buildCheckPrompt(stmts, {
1983
2241
  instructions: options?.instructions,
@@ -1996,7 +2254,7 @@ function visualAI(config = {}) {
1996
2254
  },
1997
2255
  async ask(input, userPrompt, options) {
1998
2256
  return withErrorDebug(resolvedConfig, "ask", async () => {
1999
- const media = await normalizeMedia(input, options?.video);
2257
+ const media = await normalizeMedia(input, options?.video, maxImageDimension);
2000
2258
  const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
2001
2259
  const prompt = buildAskPrompt(userPrompt, {
2002
2260
  instructions: options?.instructions,
@@ -2015,7 +2273,10 @@ function visualAI(config = {}) {
2015
2273
  },
2016
2274
  async compare(imageA, imageB, options) {
2017
2275
  return withErrorDebug(resolvedConfig, "compare", async () => {
2018
- const [imgA, imgB] = await Promise.all([normalizeImage(imageA), normalizeImage(imageB)]);
2276
+ const [imgA, imgB] = await Promise.all([
2277
+ normalizeImage(imageA, maxImageDimension),
2278
+ normalizeImage(imageB, maxImageDimension)
2279
+ ]);
2019
2280
  const prompt = buildComparePrompt({
2020
2281
  userPrompt: options?.prompt,
2021
2282
  instructions: options?.instructions
@@ -2053,7 +2314,7 @@ function visualAI(config = {}) {
2053
2314
  },
2054
2315
  async accessibility(image, options) {
2055
2316
  return withErrorDebug(resolvedConfig, "accessibility", async () => {
2056
- const img = await normalizeImage(image);
2317
+ const img = await normalizeImage(image, maxImageDimension);
2057
2318
  const prompt = buildAccessibilityPrompt(options);
2058
2319
  debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
2059
2320
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2072,7 +2333,7 @@ function visualAI(config = {}) {
2072
2333
  },
2073
2334
  async layout(image, options) {
2074
2335
  return withErrorDebug(resolvedConfig, "layout", async () => {
2075
- const img = await normalizeImage(image);
2336
+ const img = await normalizeImage(image, maxImageDimension);
2076
2337
  const prompt = buildLayoutPrompt(options);
2077
2338
  debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
2078
2339
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2086,7 +2347,7 @@ function visualAI(config = {}) {
2086
2347
  },
2087
2348
  async pageLoad(image, options) {
2088
2349
  return withErrorDebug(resolvedConfig, "pageLoad", async () => {
2089
- const img = await normalizeImage(image);
2350
+ const img = await normalizeImage(image, maxImageDimension);
2090
2351
  const prompt = buildPageLoadPrompt(options);
2091
2352
  debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
2092
2353
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2100,7 +2361,7 @@ function visualAI(config = {}) {
2100
2361
  },
2101
2362
  async content(image, options) {
2102
2363
  return withErrorDebug(resolvedConfig, "content", async () => {
2103
- const img = await normalizeImage(image);
2364
+ const img = await normalizeImage(image, maxImageDimension);
2104
2365
  const prompt = buildContentPrompt(options);
2105
2366
  debugLog(resolvedConfig, "content prompt", prompt, "prompt");
2106
2367
  const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
@@ -2178,6 +2439,7 @@ export {
2178
2439
  ConfidenceSchema,
2179
2440
  Content,
2180
2441
  DEFAULT_MODELS,
2442
+ ImageDetail,
2181
2443
  IssueCategorySchema,
2182
2444
  IssuePrioritySchema,
2183
2445
  IssueSchema,