visual-ai-assertions 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # visual-ai-assertions
2
2
 
3
- AI-powered visual assertions for E2E tests. Send screenshots — or short video recordings — to Claude, GPT, or Gemini and get structured, typed results.
3
+ AI-powered visual assertions for E2E tests. Send screenshots — or short video recordings — to Claude, GPT, Gemini — or Grok, Kimi, and Qwen via OpenRouter — and get structured, typed results.
4
4
 
5
5
  ## Installation
6
6
 
@@ -11,6 +11,7 @@ npm install visual-ai-assertions
11
11
  # Optional: install additional provider SDKs
12
12
  npm install @anthropic-ai/sdk # for Claude
13
13
  npm install @google/genai # for Gemini
14
+ # OpenRouter (Grok, Kimi, Qwen, ...) uses the OpenAI SDK — no extra install
14
15
 
15
16
  # Zod is a peer dependency
16
17
  npm install zod
@@ -441,11 +442,12 @@ The `VisualAIKnownError` union and `isVisualAIKnownError()` helper are useful wh
441
442
 
442
443
  ### API Keys
443
444
 
444
- | Provider | Environment Variable |
445
- | --------- | -------------------- |
446
- | Anthropic | `ANTHROPIC_API_KEY` |
447
- | OpenAI | `OPENAI_API_KEY` |
448
- | Google | `GOOGLE_API_KEY` |
445
+ | Provider | Environment Variable |
446
+ | ---------- | -------------------- |
447
+ | Anthropic | `ANTHROPIC_API_KEY` |
448
+ | OpenAI | `OPENAI_API_KEY` |
449
+ | Google | `GOOGLE_API_KEY` |
450
+ | OpenRouter | `OPENROUTER_API_KEY` |
449
451
 
450
452
  ### Optional Configuration
451
453
 
@@ -498,11 +500,12 @@ type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif"
498
500
 
499
501
  **Default models:**
500
502
 
501
- | Provider | Default Model |
502
- | --------- | ------------------------ |
503
- | Anthropic | `claude-sonnet-4-6` |
504
- | OpenAI | `gpt-5-mini` |
505
- | Google | `gemini-3-flash-preview` |
503
+ | Provider | Default Model |
504
+ | ---------- | ------------------------ |
505
+ | Anthropic | `claude-sonnet-4-6` |
506
+ | OpenAI | `gpt-5.4-mini` |
507
+ | Google | `gemini-3-flash-preview` |
508
+ | OpenRouter | `qwen/qwen3.6-flash` |
506
509
 
507
510
  ## Reasoning Effort
508
511
 
@@ -522,6 +525,7 @@ When omitted, each provider uses its default behavior. The `"xhigh"` level enabl
522
525
  | Anthropic (other) | `thinking.type: "adaptive"` + `output_config.effort` | `effort: "max"` |
523
526
  | OpenAI | `reasoning.effort` (Responses API) | `effort: "xhigh"` |
524
527
  | Google | `thinkingConfig.thinkingBudget` (1024 / 8192 / 24576) | `24576` (max budget) |
528
+ | OpenRouter | `reasoning.effort` (normalized low/medium/high) | `effort: "high"` |
525
529
 
526
530
  ## Supported Models
527
531
 
@@ -558,11 +562,27 @@ All listed models support image/vision input. Pass any model ID to the `model` c
558
562
 
559
563
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
560
564
  | --------------------- | ------------------------ | ------------ | ------------- | --------------------------------- |
565
+ | Gemini 3.6 Flash | `gemini-3.6-flash` | $1.50 | $7.50 | Newest GA flash; fewer out-tokens |
561
566
  | Gemini 3.5 Flash | `gemini-3.5-flash` | $1.50 | $9 | Strongest agentic & coding model |
567
+ | Gemini 3.5 Flash Lite | `gemini-3.5-flash-lite` | $0.30 | $2.50 | GA — fast, cheap, agentic tier |
562
568
  | Gemini 3.1 Pro | `gemini-3.1-pro-preview` | $2 | $12 | Preview — most advanced reasoning |
563
569
  | Gemini 3.1 Flash Lite | `gemini-3.1-flash-lite` | $0.25 | $1.50 | GA — lightweight and cheap |
564
570
  | Gemini 3 Flash | `gemini-3-flash-preview` | $0.50 | $3 | **Default** — fast and capable |
565
571
 
572
+ ### OpenRouter
573
+
574
+ Any [OpenRouter](https://openrouter.ai/models) model slug (always `vendor/model`) is accepted — the vendor prefix is how the library recognizes an OpenRouter model. The models below are tested and have pricing built in. Note that OpenRouter may route a request to different upstream hosts with different quantizations; keep that in mind when comparing benchmark numbers.
575
+
576
+ | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
577
+ | -------------- | --------------------------- | ------------ | ------------- | ------------------------------------- |
578
+ | Grok 4.5 | `x-ai/grok-4.5` | $2 | $6 | xAI flagship, 500K context |
579
+ | Kimi K3 | `moonshotai/kimi-k3` | $3 | $15 | Moonshot flagship, 1M context |
580
+ | Kimi K2.7 Code | `moonshotai/kimi-k2.7-code` | $0.82 | $3.75 | Agentic/coding tier with vision |
581
+ | Qwen3.7 Plus | `qwen/qwen3.7-plus` | $0.32 | $1.28 | Cost-effective, GUI/screen-reading |
582
+ | Qwen3.6 Flash | `qwen/qwen3.6-flash` | $0.19 | $1.13 | **Default** — cheap flash vision tier |
583
+
584
+ `qwen/qwen3.7-max` is not listed because it accepts no image input on OpenRouter.
585
+
566
586
  ## License
567
587
 
568
588
  MIT
package/dist/index.cjs CHANGED
@@ -76,7 +76,8 @@ var ReasoningEffort = {
76
76
  var Provider = {
77
77
  ANTHROPIC: "anthropic",
78
78
  OPENAI: "openai",
79
- GOOGLE: "google"
79
+ GOOGLE: "google",
80
+ OPENROUTER: "openrouter"
80
81
  };
81
82
  var Model = {
82
83
  Anthropic: {
@@ -101,29 +102,47 @@ var Model = {
101
102
  GPT_5_MINI: "gpt-5-mini"
102
103
  },
103
104
  Google: {
105
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
104
106
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
107
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
105
108
  GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
106
109
  GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
107
110
  GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
111
+ },
112
+ /**
113
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
114
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
115
+ * recognizes them. All listed models accept image input.
116
+ */
117
+ OpenRouter: {
118
+ GROK_4_5: "x-ai/grok-4.5",
119
+ KIMI_K3: "moonshotai/kimi-k3",
120
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
121
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
122
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
108
123
  }
109
124
  };
110
125
  var DEFAULT_MODELS = {
111
126
  [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
112
127
  [Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
113
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
128
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
129
+ [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
114
130
  };
115
131
  var DEFAULT_MAX_TOKENS = 4096;
116
132
  var OPENAI_REASONING_MAX_TOKENS = 16384;
117
133
  var MODEL_TO_PROVIDER = new Map([
118
134
  ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
119
135
  ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
120
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
136
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
137
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
121
138
  ]);
122
139
  var VALID_PROVIDERS = Object.values(Provider);
123
140
  var PROVIDER_DEFAULT_REASONING = {
124
141
  openai: "medium",
125
142
  anthropic: "off",
126
- google: "off"
143
+ google: "off",
144
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
145
+ openrouter: "off"
127
146
  };
128
147
  var Content = {
129
148
  /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
@@ -589,6 +608,13 @@ function mapEffort(level, model) {
589
608
  if (level !== "xhigh") return level;
590
609
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
591
610
  }
611
+ var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
612
+ var EFFORT_TO_BUDGET_TOKENS = {
613
+ low: 1024,
614
+ medium: 4096,
615
+ high: 8192,
616
+ xhigh: 16384
617
+ };
592
618
  var AnthropicDriver = class {
593
619
  client;
594
620
  model;
@@ -644,10 +670,16 @@ var AnthropicDriver = class {
644
670
  ]
645
671
  };
646
672
  if (this.reasoningEffort) {
647
- requestParams.thinking = { type: "adaptive" };
648
- requestParams.output_config = {
649
- effort: mapEffort(this.reasoningEffort, this.model)
650
- };
673
+ if (BUDGET_THINKING_MODELS.has(this.model)) {
674
+ const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
675
+ requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
676
+ requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
677
+ } else {
678
+ requestParams.thinking = { type: "adaptive" };
679
+ requestParams.output_config = {
680
+ effort: mapEffort(this.reasoningEffort, this.model)
681
+ };
682
+ }
651
683
  }
652
684
  const message = await client.messages.create(requestParams);
653
685
  const textBlock = message.content.find((block) => block.type === "text");
@@ -888,6 +920,109 @@ var OpenAIDriver = class {
888
920
  }
889
921
  };
890
922
 
923
+ // src/providers/openrouter.ts
924
+ var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
925
+ var OPENROUTER_REASONING_EFFORT = {
926
+ low: "low",
927
+ medium: "medium",
928
+ high: "high",
929
+ xhigh: "high"
930
+ };
931
+ var OpenRouterDriver = class {
932
+ client;
933
+ model;
934
+ maxTokens;
935
+ apiKeyOrEnv;
936
+ reasoningEffort;
937
+ constructor(config) {
938
+ this.model = config.model;
939
+ this.maxTokens = config.maxTokens;
940
+ this.client = null;
941
+ this.apiKeyOrEnv = config.apiKey;
942
+ this.reasoningEffort = config.reasoningEffort;
943
+ }
944
+ async getClient() {
945
+ if (this.client) return this.client;
946
+ let OpenAI;
947
+ try {
948
+ const mod = await import("openai");
949
+ OpenAI = mod.default;
950
+ } catch {
951
+ throw new VisualAIConfigError(
952
+ "OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
953
+ );
954
+ }
955
+ const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
956
+ if (!apiKey) {
957
+ throw new VisualAIAuthError(
958
+ "OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
959
+ );
960
+ }
961
+ this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
962
+ return this.client;
963
+ }
964
+ async sendMessage(images, prompt, options) {
965
+ const client = await this.getClient();
966
+ const imageParts = images.map((img) => ({
967
+ type: "image_url",
968
+ image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
969
+ }));
970
+ try {
971
+ const responseFormat = options?.responseSchema ? {
972
+ type: "json_schema",
973
+ json_schema: {
974
+ name: "visual_ai_response",
975
+ strict: true,
976
+ schema: options.responseSchema
977
+ }
978
+ } : { type: "json_object" };
979
+ const requestParams = {
980
+ model: this.model,
981
+ max_tokens: this.maxTokens,
982
+ response_format: responseFormat,
983
+ messages: [
984
+ {
985
+ role: "user",
986
+ content: [...imageParts, { type: "text", text: prompt }]
987
+ }
988
+ ],
989
+ // OpenRouter-specific: include token accounting in the response.
990
+ usage: { include: true }
991
+ };
992
+ if (this.reasoningEffort) {
993
+ requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
994
+ }
995
+ const response = await client.chat.completions.create(requestParams);
996
+ const choice = response.choices?.[0];
997
+ if (!choice?.message) {
998
+ throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
999
+ }
1000
+ const text = choice.message.content ?? "";
1001
+ if (choice.finish_reason === "length") {
1002
+ throw new VisualAITruncationError(
1003
+ `Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
1004
+ text,
1005
+ this.maxTokens
1006
+ );
1007
+ }
1008
+ const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
1009
+ return {
1010
+ text,
1011
+ usage: response.usage ? {
1012
+ inputTokens: response.usage.prompt_tokens,
1013
+ outputTokens: response.usage.completion_tokens,
1014
+ ...reasoningTokens !== void 0 && { reasoningTokens }
1015
+ } : void 0
1016
+ };
1017
+ } catch (err) {
1018
+ if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
1019
+ throw err;
1020
+ }
1021
+ throw mapProviderError(err);
1022
+ }
1023
+ }
1024
+ };
1025
+
891
1026
  // src/core/config.ts
892
1027
  var MODEL_PREFIX_TO_PROVIDER = [
893
1028
  ["claude-", "anthropic"],
@@ -900,6 +1035,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
900
1035
  function inferProviderFromModel(model) {
901
1036
  const known = MODEL_TO_PROVIDER.get(model);
902
1037
  if (known) return known;
1038
+ if (model.includes("/")) return "openrouter";
903
1039
  const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
904
1040
  return prefixMatch?.[1];
905
1041
  }
@@ -912,12 +1048,13 @@ function resolveProvider(config) {
912
1048
  const apiKeyProviderMap = [
913
1049
  ["ANTHROPIC_API_KEY", "anthropic"],
914
1050
  ["OPENAI_API_KEY", "openai"],
915
- ["GOOGLE_API_KEY", "google"]
1051
+ ["GOOGLE_API_KEY", "google"],
1052
+ ["OPENROUTER_API_KEY", "openrouter"]
916
1053
  ];
917
1054
  const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
918
1055
  if (detected) return detected[1];
919
1056
  throw new VisualAIConfigError(
920
- "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
1057
+ "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
921
1058
  );
922
1059
  }
923
1060
  function parseBooleanEnv(envName, value) {
@@ -945,11 +1082,11 @@ function resolveConfig(config) {
945
1082
  }
946
1083
  const userSetMaxTokens = config.maxTokens !== void 0;
947
1084
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
948
- if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
1085
+ if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
949
1086
  maxTokens = OPENAI_REASONING_MAX_TOKENS;
950
1087
  if (debug) {
951
1088
  process.stderr.write(
952
- `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for OpenAI with reasoningEffort "${config.reasoningEffort}".
1089
+ `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
953
1090
  `
954
1091
  );
955
1092
  }
@@ -1038,10 +1175,18 @@ var PRICING_TABLE = {
1038
1175
  inputPricePerToken: 0.25 / PER_MILLION,
1039
1176
  outputPricePerToken: 2 / PER_MILLION
1040
1177
  },
1178
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1179
+ inputPricePerToken: 1.5 / PER_MILLION,
1180
+ outputPricePerToken: 7.5 / PER_MILLION
1181
+ },
1041
1182
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
1042
1183
  inputPricePerToken: 1.5 / PER_MILLION,
1043
1184
  outputPricePerToken: 9 / PER_MILLION
1044
1185
  },
1186
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
1187
+ inputPricePerToken: 0.3 / PER_MILLION,
1188
+ outputPricePerToken: 2.5 / PER_MILLION
1189
+ },
1045
1190
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
1046
1191
  inputPricePerToken: 2 / PER_MILLION,
1047
1192
  outputPricePerToken: 12 / PER_MILLION
@@ -1053,6 +1198,28 @@ var PRICING_TABLE = {
1053
1198
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
1054
1199
  inputPricePerToken: 0.5 / PER_MILLION,
1055
1200
  outputPricePerToken: 3 / PER_MILLION
1201
+ },
1202
+ // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1203
+ // against https://openrouter.ai/api/v1/models).
1204
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1205
+ inputPricePerToken: 2 / PER_MILLION,
1206
+ outputPricePerToken: 6 / PER_MILLION
1207
+ },
1208
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
1209
+ inputPricePerToken: 3 / PER_MILLION,
1210
+ outputPricePerToken: 15 / PER_MILLION
1211
+ },
1212
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
1213
+ inputPricePerToken: 0.82 / PER_MILLION,
1214
+ outputPricePerToken: 3.75 / PER_MILLION
1215
+ },
1216
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1217
+ inputPricePerToken: 0.32 / PER_MILLION,
1218
+ outputPricePerToken: 1.28 / PER_MILLION
1219
+ },
1220
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1221
+ inputPricePerToken: 0.1875 / PER_MILLION,
1222
+ outputPricePerToken: 1.125 / PER_MILLION
1056
1223
  }
1057
1224
  };
1058
1225
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1130,7 +1297,8 @@ async function timedSendMessage(driver, images, prompt, options) {
1130
1297
  var import_sharp = __toESM(require("sharp"), 1);
1131
1298
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1132
1299
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1133
- Model.Google.GEMINI_3_5_FLASH
1300
+ Model.Google.GEMINI_3_5_FLASH,
1301
+ Model.Google.GEMINI_3_6_FLASH
1134
1302
  ]);
1135
1303
  async function generateAiDiff(imgA, imgB, model, driver) {
1136
1304
  if (!driver.generateImage) {
@@ -1980,7 +2148,8 @@ function toSchemaOptions(schema) {
1980
2148
  var PROVIDER_REGISTRY = {
1981
2149
  anthropic: (config) => new AnthropicDriver(config),
1982
2150
  openai: (config) => new OpenAIDriver(config),
1983
- google: (config) => new GoogleDriver(config)
2151
+ google: (config) => new GoogleDriver(config),
2152
+ openrouter: (config) => new OpenRouterDriver(config)
1984
2153
  };
1985
2154
  function createDriver(provider, config) {
1986
2155
  return PROVIDER_REGISTRY[provider](config);