visual-ai-assertions 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -14,6 +14,7 @@ declare const Provider: {
14
14
  readonly ANTHROPIC: "anthropic";
15
15
  readonly OPENAI: "openai";
16
16
  readonly GOOGLE: "google";
17
+ readonly OPENROUTER: "openrouter";
17
18
  };
18
19
  /** Known model names grouped by provider. */
19
20
  declare const Model: {
@@ -39,19 +40,34 @@ declare const Model: {
39
40
  readonly GPT_5_MINI: "gpt-5-mini";
40
41
  };
41
42
  readonly Google: {
43
+ readonly GEMINI_3_6_FLASH: "gemini-3.6-flash";
42
44
  readonly GEMINI_3_5_FLASH: "gemini-3.5-flash";
45
+ readonly GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite";
43
46
  readonly GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview";
44
47
  readonly GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite";
45
48
  readonly GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview";
46
49
  };
50
+ /**
51
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
52
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
53
+ * recognizes them. All listed models accept image input.
54
+ */
55
+ readonly OpenRouter: {
56
+ readonly GROK_4_5: "x-ai/grok-4.5";
57
+ readonly KIMI_K3: "moonshotai/kimi-k3";
58
+ readonly KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code";
59
+ readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
60
+ readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
61
+ };
47
62
  };
48
63
  /** Union of all built-in model name literals exposed by `Model`. */
49
- type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google];
64
+ type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
50
65
  /** Default model selection used when a caller omits `config.model`. */
51
66
  declare const DEFAULT_MODELS: {
52
67
  readonly anthropic: "claude-sonnet-4-6";
53
68
  readonly openai: "gpt-5.4-mini";
54
69
  readonly google: "gemini-3-flash-preview";
70
+ readonly openrouter: "qwen/qwen3.6-flash";
55
71
  };
56
72
  /** Built-in content checks available through `client.content()`. */
57
73
  declare const Content: {
@@ -528,7 +544,7 @@ type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif"
528
544
  /** Supported video MIME types the client can accept and sample frames from. */
529
545
  type SupportedVideoMimeType = "video/mp4" | "video/webm" | "video/quicktime" | "video/x-matroska";
530
546
  /** Supported provider identifiers. */
531
- type ProviderName = "anthropic" | "openai" | "google";
547
+ type ProviderName = "anthropic" | "openai" | "google" | "openrouter";
532
548
  /**
533
549
  * Configuration for creating a visual AI client.
534
550
  *
package/dist/index.d.ts CHANGED
@@ -14,6 +14,7 @@ declare const Provider: {
14
14
  readonly ANTHROPIC: "anthropic";
15
15
  readonly OPENAI: "openai";
16
16
  readonly GOOGLE: "google";
17
+ readonly OPENROUTER: "openrouter";
17
18
  };
18
19
  /** Known model names grouped by provider. */
19
20
  declare const Model: {
@@ -39,19 +40,34 @@ declare const Model: {
39
40
  readonly GPT_5_MINI: "gpt-5-mini";
40
41
  };
41
42
  readonly Google: {
43
+ readonly GEMINI_3_6_FLASH: "gemini-3.6-flash";
42
44
  readonly GEMINI_3_5_FLASH: "gemini-3.5-flash";
45
+ readonly GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite";
43
46
  readonly GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview";
44
47
  readonly GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite";
45
48
  readonly GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview";
46
49
  };
50
+ /**
51
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
52
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
53
+ * recognizes them. All listed models accept image input.
54
+ */
55
+ readonly OpenRouter: {
56
+ readonly GROK_4_5: "x-ai/grok-4.5";
57
+ readonly KIMI_K3: "moonshotai/kimi-k3";
58
+ readonly KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code";
59
+ readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
60
+ readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
61
+ };
47
62
  };
48
63
  /** Union of all built-in model name literals exposed by `Model`. */
49
- type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google];
64
+ type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
50
65
  /** Default model selection used when a caller omits `config.model`. */
51
66
  declare const DEFAULT_MODELS: {
52
67
  readonly anthropic: "claude-sonnet-4-6";
53
68
  readonly openai: "gpt-5.4-mini";
54
69
  readonly google: "gemini-3-flash-preview";
70
+ readonly openrouter: "qwen/qwen3.6-flash";
55
71
  };
56
72
  /** Built-in content checks available through `client.content()`. */
57
73
  declare const Content: {
@@ -528,7 +544,7 @@ type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif"
528
544
  /** Supported video MIME types the client can accept and sample frames from. */
529
545
  type SupportedVideoMimeType = "video/mp4" | "video/webm" | "video/quicktime" | "video/x-matroska";
530
546
  /** Supported provider identifiers. */
531
- type ProviderName = "anthropic" | "openai" | "google";
547
+ type ProviderName = "anthropic" | "openai" | "google" | "openrouter";
532
548
  /**
533
549
  * Configuration for creating a visual AI client.
534
550
  *
package/dist/index.js CHANGED
@@ -8,7 +8,8 @@ var ReasoningEffort = {
8
8
  var Provider = {
9
9
  ANTHROPIC: "anthropic",
10
10
  OPENAI: "openai",
11
- GOOGLE: "google"
11
+ GOOGLE: "google",
12
+ OPENROUTER: "openrouter"
12
13
  };
13
14
  var Model = {
14
15
  Anthropic: {
@@ -33,29 +34,47 @@ var Model = {
33
34
  GPT_5_MINI: "gpt-5-mini"
34
35
  },
35
36
  Google: {
37
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
36
38
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
39
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
37
40
  GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
38
41
  GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
39
42
  GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
43
+ },
44
+ /**
45
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
46
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
47
+ * recognizes them. All listed models accept image input.
48
+ */
49
+ OpenRouter: {
50
+ GROK_4_5: "x-ai/grok-4.5",
51
+ KIMI_K3: "moonshotai/kimi-k3",
52
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
53
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
54
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
40
55
  }
41
56
  };
42
57
  var DEFAULT_MODELS = {
43
58
  [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
44
59
  [Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
45
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
60
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
61
+ [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
46
62
  };
47
63
  var DEFAULT_MAX_TOKENS = 4096;
48
64
  var OPENAI_REASONING_MAX_TOKENS = 16384;
49
65
  var MODEL_TO_PROVIDER = new Map([
50
66
  ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
51
67
  ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
52
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
68
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
69
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
53
70
  ]);
54
71
  var VALID_PROVIDERS = Object.values(Provider);
55
72
  var PROVIDER_DEFAULT_REASONING = {
56
73
  openai: "medium",
57
74
  anthropic: "off",
58
- google: "off"
75
+ google: "off",
76
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
77
+ openrouter: "off"
59
78
  };
60
79
  var Content = {
61
80
  /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
@@ -521,6 +540,13 @@ function mapEffort(level, model) {
521
540
  if (level !== "xhigh") return level;
522
541
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
523
542
  }
543
+ var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
544
+ var EFFORT_TO_BUDGET_TOKENS = {
545
+ low: 1024,
546
+ medium: 4096,
547
+ high: 8192,
548
+ xhigh: 16384
549
+ };
524
550
  var AnthropicDriver = class {
525
551
  client;
526
552
  model;
@@ -576,10 +602,16 @@ var AnthropicDriver = class {
576
602
  ]
577
603
  };
578
604
  if (this.reasoningEffort) {
579
- requestParams.thinking = { type: "adaptive" };
580
- requestParams.output_config = {
581
- effort: mapEffort(this.reasoningEffort, this.model)
582
- };
605
+ if (BUDGET_THINKING_MODELS.has(this.model)) {
606
+ const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
607
+ requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
608
+ requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
609
+ } else {
610
+ requestParams.thinking = { type: "adaptive" };
611
+ requestParams.output_config = {
612
+ effort: mapEffort(this.reasoningEffort, this.model)
613
+ };
614
+ }
583
615
  }
584
616
  const message = await client.messages.create(requestParams);
585
617
  const textBlock = message.content.find((block) => block.type === "text");
@@ -820,6 +852,109 @@ var OpenAIDriver = class {
820
852
  }
821
853
  };
822
854
 
855
+ // src/providers/openrouter.ts
856
+ var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
857
+ var OPENROUTER_REASONING_EFFORT = {
858
+ low: "low",
859
+ medium: "medium",
860
+ high: "high",
861
+ xhigh: "high"
862
+ };
863
+ var OpenRouterDriver = class {
864
+ client;
865
+ model;
866
+ maxTokens;
867
+ apiKeyOrEnv;
868
+ reasoningEffort;
869
+ constructor(config) {
870
+ this.model = config.model;
871
+ this.maxTokens = config.maxTokens;
872
+ this.client = null;
873
+ this.apiKeyOrEnv = config.apiKey;
874
+ this.reasoningEffort = config.reasoningEffort;
875
+ }
876
+ async getClient() {
877
+ if (this.client) return this.client;
878
+ let OpenAI;
879
+ try {
880
+ const mod = await import("openai");
881
+ OpenAI = mod.default;
882
+ } catch {
883
+ throw new VisualAIConfigError(
884
+ "OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
885
+ );
886
+ }
887
+ const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
888
+ if (!apiKey) {
889
+ throw new VisualAIAuthError(
890
+ "OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
891
+ );
892
+ }
893
+ this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
894
+ return this.client;
895
+ }
896
+ async sendMessage(images, prompt, options) {
897
+ const client = await this.getClient();
898
+ const imageParts = images.map((img) => ({
899
+ type: "image_url",
900
+ image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
901
+ }));
902
+ try {
903
+ const responseFormat = options?.responseSchema ? {
904
+ type: "json_schema",
905
+ json_schema: {
906
+ name: "visual_ai_response",
907
+ strict: true,
908
+ schema: options.responseSchema
909
+ }
910
+ } : { type: "json_object" };
911
+ const requestParams = {
912
+ model: this.model,
913
+ max_tokens: this.maxTokens,
914
+ response_format: responseFormat,
915
+ messages: [
916
+ {
917
+ role: "user",
918
+ content: [...imageParts, { type: "text", text: prompt }]
919
+ }
920
+ ],
921
+ // OpenRouter-specific: include token accounting in the response.
922
+ usage: { include: true }
923
+ };
924
+ if (this.reasoningEffort) {
925
+ requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
926
+ }
927
+ const response = await client.chat.completions.create(requestParams);
928
+ const choice = response.choices?.[0];
929
+ if (!choice?.message) {
930
+ throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
931
+ }
932
+ const text = choice.message.content ?? "";
933
+ if (choice.finish_reason === "length") {
934
+ throw new VisualAITruncationError(
935
+ `Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
936
+ text,
937
+ this.maxTokens
938
+ );
939
+ }
940
+ const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
941
+ return {
942
+ text,
943
+ usage: response.usage ? {
944
+ inputTokens: response.usage.prompt_tokens,
945
+ outputTokens: response.usage.completion_tokens,
946
+ ...reasoningTokens !== void 0 && { reasoningTokens }
947
+ } : void 0
948
+ };
949
+ } catch (err) {
950
+ if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
951
+ throw err;
952
+ }
953
+ throw mapProviderError(err);
954
+ }
955
+ }
956
+ };
957
+
823
958
  // src/core/config.ts
824
959
  var MODEL_PREFIX_TO_PROVIDER = [
825
960
  ["claude-", "anthropic"],
@@ -832,6 +967,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
832
967
  function inferProviderFromModel(model) {
833
968
  const known = MODEL_TO_PROVIDER.get(model);
834
969
  if (known) return known;
970
+ if (model.includes("/")) return "openrouter";
835
971
  const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
836
972
  return prefixMatch?.[1];
837
973
  }
@@ -844,12 +980,13 @@ function resolveProvider(config) {
844
980
  const apiKeyProviderMap = [
845
981
  ["ANTHROPIC_API_KEY", "anthropic"],
846
982
  ["OPENAI_API_KEY", "openai"],
847
- ["GOOGLE_API_KEY", "google"]
983
+ ["GOOGLE_API_KEY", "google"],
984
+ ["OPENROUTER_API_KEY", "openrouter"]
848
985
  ];
849
986
  const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
850
987
  if (detected) return detected[1];
851
988
  throw new VisualAIConfigError(
852
- "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
989
+ "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
853
990
  );
854
991
  }
855
992
  function parseBooleanEnv(envName, value) {
@@ -877,11 +1014,11 @@ function resolveConfig(config) {
877
1014
  }
878
1015
  const userSetMaxTokens = config.maxTokens !== void 0;
879
1016
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
880
- if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
1017
+ if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
881
1018
  maxTokens = OPENAI_REASONING_MAX_TOKENS;
882
1019
  if (debug) {
883
1020
  process.stderr.write(
884
- `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for OpenAI with reasoningEffort "${config.reasoningEffort}".
1021
+ `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
885
1022
  `
886
1023
  );
887
1024
  }
@@ -970,10 +1107,18 @@ var PRICING_TABLE = {
970
1107
  inputPricePerToken: 0.25 / PER_MILLION,
971
1108
  outputPricePerToken: 2 / PER_MILLION
972
1109
  },
1110
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1111
+ inputPricePerToken: 1.5 / PER_MILLION,
1112
+ outputPricePerToken: 7.5 / PER_MILLION
1113
+ },
973
1114
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
974
1115
  inputPricePerToken: 1.5 / PER_MILLION,
975
1116
  outputPricePerToken: 9 / PER_MILLION
976
1117
  },
1118
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
1119
+ inputPricePerToken: 0.3 / PER_MILLION,
1120
+ outputPricePerToken: 2.5 / PER_MILLION
1121
+ },
977
1122
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
978
1123
  inputPricePerToken: 2 / PER_MILLION,
979
1124
  outputPricePerToken: 12 / PER_MILLION
@@ -985,6 +1130,28 @@ var PRICING_TABLE = {
985
1130
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
986
1131
  inputPricePerToken: 0.5 / PER_MILLION,
987
1132
  outputPricePerToken: 3 / PER_MILLION
1133
+ },
1134
+ // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1135
+ // against https://openrouter.ai/api/v1/models).
1136
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1137
+ inputPricePerToken: 2 / PER_MILLION,
1138
+ outputPricePerToken: 6 / PER_MILLION
1139
+ },
1140
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
1141
+ inputPricePerToken: 3 / PER_MILLION,
1142
+ outputPricePerToken: 15 / PER_MILLION
1143
+ },
1144
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
1145
+ inputPricePerToken: 0.82 / PER_MILLION,
1146
+ outputPricePerToken: 3.75 / PER_MILLION
1147
+ },
1148
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1149
+ inputPricePerToken: 0.32 / PER_MILLION,
1150
+ outputPricePerToken: 1.28 / PER_MILLION
1151
+ },
1152
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1153
+ inputPricePerToken: 0.1875 / PER_MILLION,
1154
+ outputPricePerToken: 1.125 / PER_MILLION
988
1155
  }
989
1156
  };
990
1157
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1062,7 +1229,8 @@ async function timedSendMessage(driver, images, prompt, options) {
1062
1229
  import sharp from "sharp";
1063
1230
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1064
1231
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1065
- Model.Google.GEMINI_3_5_FLASH
1232
+ Model.Google.GEMINI_3_5_FLASH,
1233
+ Model.Google.GEMINI_3_6_FLASH
1066
1234
  ]);
1067
1235
  async function generateAiDiff(imgA, imgB, model, driver) {
1068
1236
  if (!driver.generateImage) {
@@ -1912,7 +2080,8 @@ function toSchemaOptions(schema) {
1912
2080
  var PROVIDER_REGISTRY = {
1913
2081
  anthropic: (config) => new AnthropicDriver(config),
1914
2082
  openai: (config) => new OpenAIDriver(config),
1915
- google: (config) => new GoogleDriver(config)
2083
+ google: (config) => new GoogleDriver(config),
2084
+ openrouter: (config) => new OpenRouterDriver(config)
1916
2085
  };
1917
2086
  function createDriver(provider, config) {
1918
2087
  return PROVIDER_REGISTRY[provider](config);