visual-ai-assertions 0.16.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -6
- package/dist/index.cjs +165 -52
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +101 -1
- package/dist/index.d.ts +101 -1
- package/dist/index.js +164 -52
- package/dist/index.js.map +1 -1
- package/package.json +6 -2
package/README.md
CHANGED
|
@@ -524,7 +524,7 @@ When omitted, each provider uses its default behavior. The `"xhigh"` level enabl
|
|
|
524
524
|
| Anthropic (Fable 5/Opus 4.8/4.7/Sonnet 5) | `thinking.type: "adaptive"` + `output_config.effort` | `effort: "xhigh"` |
|
|
525
525
|
| Anthropic (other) | `thinking.type: "adaptive"` + `output_config.effort` | `effort: "max"` |
|
|
526
526
|
| OpenAI | `reasoning.effort` (Responses API) | `effort: "xhigh"` |
|
|
527
|
-
| Google | `thinkingConfig.
|
|
527
|
+
| Google | `thinkingConfig.thinkingLevel` (1:1: low/medium/high) | `"high"` (max level) |
|
|
528
528
|
| OpenRouter | `reasoning.effort` (normalized low/medium/high) | `effort: "high"` |
|
|
529
529
|
|
|
530
530
|
## Supported Models
|
|
@@ -548,8 +548,8 @@ All listed models support image/vision input. Pass any model ID to the `model` c
|
|
|
548
548
|
| Model | Model ID | Input $/MTok | Output $/MTok | Notes |
|
|
549
549
|
| ------------- | --------------- | ------------ | ------------- | --------------------------------- |
|
|
550
550
|
| GPT-5.6 Sol | `gpt-5.6-sol` | $5 | $30 | Newest flagship, frontier tier |
|
|
551
|
-
| GPT-5.6 Terra | `gpt-5.6-terra` | $2
|
|
552
|
-
| GPT-5.6 Luna | `gpt-5.6-luna` | $
|
|
551
|
+
| GPT-5.6 Terra | `gpt-5.6-terra` | $2 | $12 | Newest balanced, everyday tier |
|
|
552
|
+
| GPT-5.6 Luna | `gpt-5.6-luna` | $0.20 | $1.20 | Newest, fastest/cheapest tier |
|
|
553
553
|
| GPT-5.5 | `gpt-5.5` | $5 | $30 | Previous flagship, 1M context |
|
|
554
554
|
| GPT-5.4 Pro | `gpt-5.4-pro` | $30 | $180 | Most capable, extended context |
|
|
555
555
|
| GPT-5.4 | `gpt-5.4` | $2.50 | $15 | Best vision quality |
|
|
@@ -562,26 +562,37 @@ All listed models support image/vision input. Pass any model ID to the `model` c
|
|
|
562
562
|
|
|
563
563
|
| Model | Model ID | Input $/MTok | Output $/MTok | Notes |
|
|
564
564
|
| --------------------- | ------------------------ | ------------ | ------------- | --------------------------------- |
|
|
565
|
-
| Gemini 3.
|
|
565
|
+
| Gemini 3.8 Flash | `gemini-3.8-flash` | $0.75 | $3.75 | Newest GA flash; intro pricing¹ |
|
|
566
|
+
| Gemini 3.7 Flash | `gemini-3.7-flash` | $0.75 | $3.75 | Prior GA flash; intro pricing¹ |
|
|
567
|
+
| Gemini 3.6 Flash | `gemini-3.6-flash` | $1.50 | $7.50 | Prior GA flash; fewer out-tokens |
|
|
566
568
|
| Gemini 3.5 Flash | `gemini-3.5-flash` | $1.50 | $9 | Strongest agentic & coding model |
|
|
567
569
|
| Gemini 3.5 Flash Lite | `gemini-3.5-flash-lite` | $0.30 | $2.50 | GA — fast, cheap, agentic tier |
|
|
568
570
|
| Gemini 3.1 Pro | `gemini-3.1-pro-preview` | $2 | $12 | Preview — most advanced reasoning |
|
|
569
571
|
| Gemini 3.1 Flash Lite | `gemini-3.1-flash-lite` | $0.25 | $1.50 | GA — lightweight and cheap |
|
|
570
572
|
| Gemini 3 Flash | `gemini-3-flash-preview` | $0.50 | $3 | **Default** — fast and capable |
|
|
571
573
|
|
|
574
|
+
¹ Gemini 3.8 Flash and 3.7 Flash introductory pricing runs through 2026-12-31; both revert to $1.50 / $7.50 per MTok on 2027-01-01.
|
|
575
|
+
|
|
572
576
|
### OpenRouter
|
|
573
577
|
|
|
574
578
|
Any [OpenRouter](https://openrouter.ai/models) model slug (always `vendor/model`) is accepted — the vendor prefix is how the library recognizes an OpenRouter model. The models below are tested and have pricing built in. Note that OpenRouter may route a request to different upstream hosts with different quantizations; keep that in mind when comparing benchmark numbers.
|
|
575
579
|
|
|
576
580
|
| Model | Model ID | Input $/MTok | Output $/MTok | Notes |
|
|
577
581
|
| -------------- | --------------------------- | ------------ | ------------- | ------------------------------------- |
|
|
578
|
-
|
|
|
582
|
+
| Muse Spark 1.3 | `meta/muse-spark-1.3` | $1.25 | $4.25 | Meta flagship, 1M context; gated¹ |
|
|
583
|
+
| Grok 4.6 | `x-ai/grok-4.6` | $2 | $6 | Newest xAI flagship, 500K context |
|
|
584
|
+
| Grok 4.5 | `x-ai/grok-4.5` | $2 | $6 | Prior xAI flagship, 500K context |
|
|
579
585
|
| Kimi K3 | `moonshotai/kimi-k3` | $3 | $15 | Moonshot flagship, 1M context |
|
|
580
586
|
| Kimi K2.7 Code | `moonshotai/kimi-k2.7-code` | $0.82 | $3.75 | Agentic/coding tier with vision |
|
|
587
|
+
| Qwen3.8 Max | `qwen/qwen3.8-max` | $2 | $6 | First Max tier with image input |
|
|
581
588
|
| Qwen3.7 Plus | `qwen/qwen3.7-plus` | $0.32 | $1.28 | Cost-effective, GUI/screen-reading |
|
|
582
589
|
| Qwen3.6 Flash | `qwen/qwen3.6-flash` | $0.19 | $1.13 | **Default** — cheap flash vision tier |
|
|
583
590
|
|
|
584
|
-
|
|
591
|
+
¹ Muse Spark 1.3 is age-gated by OpenRouter: calls return HTTP 403 (`VisualAIAuthError`) until the account completes the 18+ confirmation at [openrouter.ai/settings/preferences](https://openrouter.ai/settings/preferences). It also reasons by default — expect several hundred reasoning tokens per call even with no `reasoningEffort` set.
|
|
592
|
+
|
|
593
|
+
Meta also publishes `meta/muse-spark-1.3-contributor`, the same model at $0.10 / $0.20 per MTok — about 12x cheaper — because Meta uses everything submitted through it for product improvement. It has **no named constant** (`Model.OpenRouter` does not expose it) and never appears by default anywhere in this library, so using it takes a deliberate, explicit choice: pass the slug directly as a plain string, `visualAI({ model: "meta/muse-spark-1.3-contributor" })`. Any OpenRouter slug works this way — see the note above the table — and cost tracking works correctly once you opt in. OpenRouter itself blocks it with HTTP 404 (`paid-model-training-violation-by-account`) until the account's privacy settings allow training endpoints, at [openrouter.ai/settings/privacy](https://openrouter.ai/settings/privacy). Only use it if sending your screenshots to Meta for training is a trade you've deliberately made.
|
|
594
|
+
|
|
595
|
+
`qwen/qwen3.7-max` and the DeepSeek V4 family (`deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`, and dated variants such as `deepseek/deepseek-v4-pro-0813`) are not listed because they accept no image input on OpenRouter.
|
|
585
596
|
|
|
586
597
|
## License
|
|
587
598
|
|
package/dist/index.cjs
CHANGED
|
@@ -38,6 +38,7 @@ __export(index_exports, {
|
|
|
38
38
|
ConfidenceSchema: () => ConfidenceSchema,
|
|
39
39
|
Content: () => Content,
|
|
40
40
|
DEFAULT_MODELS: () => DEFAULT_MODELS,
|
|
41
|
+
ImageDetail: () => ImageDetail,
|
|
41
42
|
IssueCategorySchema: () => IssueCategorySchema,
|
|
42
43
|
IssuePrioritySchema: () => IssuePrioritySchema,
|
|
43
44
|
IssueSchema: () => IssueSchema,
|
|
@@ -73,6 +74,13 @@ var ReasoningEffort = {
|
|
|
73
74
|
HIGH: "high",
|
|
74
75
|
XHIGH: "xhigh"
|
|
75
76
|
};
|
|
77
|
+
var ImageDetail = {
|
|
78
|
+
AUTO: "auto",
|
|
79
|
+
LOW: "low",
|
|
80
|
+
HIGH: "high"
|
|
81
|
+
};
|
|
82
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
83
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
76
84
|
var Provider = {
|
|
77
85
|
ANTHROPIC: "anthropic",
|
|
78
86
|
OPENAI: "openai",
|
|
@@ -82,6 +90,7 @@ var Provider = {
|
|
|
82
90
|
var Model = {
|
|
83
91
|
Anthropic: {
|
|
84
92
|
FABLE_5: "claude-fable-5",
|
|
93
|
+
OPUS_5: "claude-opus-5",
|
|
85
94
|
OPUS_4_8: "claude-opus-4-8",
|
|
86
95
|
OPUS_4_7: "claude-opus-4-7",
|
|
87
96
|
OPUS_4_6: "claude-opus-4-6",
|
|
@@ -102,6 +111,8 @@ var Model = {
|
|
|
102
111
|
GPT_5_MINI: "gpt-5-mini"
|
|
103
112
|
},
|
|
104
113
|
Google: {
|
|
114
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
115
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
105
116
|
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
106
117
|
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
107
118
|
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
@@ -115,9 +126,12 @@ var Model = {
|
|
|
115
126
|
* recognizes them. All listed models accept image input.
|
|
116
127
|
*/
|
|
117
128
|
OpenRouter: {
|
|
129
|
+
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
130
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
118
131
|
GROK_4_5: "x-ai/grok-4.5",
|
|
119
132
|
KIMI_K3: "moonshotai/kimi-k3",
|
|
120
133
|
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
134
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
121
135
|
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
122
136
|
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
123
137
|
}
|
|
@@ -600,6 +614,7 @@ function parseRetryAfter(value) {
|
|
|
600
614
|
// src/providers/anthropic.ts
|
|
601
615
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
602
616
|
Model.Anthropic.FABLE_5,
|
|
617
|
+
Model.Anthropic.OPUS_5,
|
|
603
618
|
Model.Anthropic.OPUS_4_8,
|
|
604
619
|
Model.Anthropic.OPUS_4_7,
|
|
605
620
|
Model.Anthropic.SONNET_5
|
|
@@ -695,7 +710,10 @@ var AnthropicDriver = class {
|
|
|
695
710
|
text,
|
|
696
711
|
usage: {
|
|
697
712
|
inputTokens: message.usage.input_tokens,
|
|
698
|
-
outputTokens: message.usage.output_tokens
|
|
713
|
+
outputTokens: message.usage.output_tokens,
|
|
714
|
+
...message.usage.cache_read_input_tokens !== void 0 && {
|
|
715
|
+
cachedInputTokens: message.usage.cache_read_input_tokens
|
|
716
|
+
}
|
|
699
717
|
}
|
|
700
718
|
};
|
|
701
719
|
} catch (err) {
|
|
@@ -712,23 +730,41 @@ function needsCodeExecution(model) {
|
|
|
712
730
|
return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
|
|
713
731
|
}
|
|
714
732
|
var GOOGLE_THINKING_LEVEL = {
|
|
715
|
-
low: "
|
|
716
|
-
medium: "
|
|
717
|
-
high: "
|
|
733
|
+
low: "low",
|
|
734
|
+
medium: "medium",
|
|
735
|
+
high: "high",
|
|
718
736
|
xhigh: "high"
|
|
719
737
|
};
|
|
738
|
+
var GOOGLE_MEDIA_RESOLUTION = {
|
|
739
|
+
low: "MEDIA_RESOLUTION_LOW",
|
|
740
|
+
high: "MEDIA_RESOLUTION_HIGH"
|
|
741
|
+
};
|
|
742
|
+
function toGeminiUsage(um) {
|
|
743
|
+
if (!um) return void 0;
|
|
744
|
+
const thoughts = um.thoughtsTokenCount ?? 0;
|
|
745
|
+
return {
|
|
746
|
+
inputTokens: um.promptTokenCount ?? 0,
|
|
747
|
+
outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
|
|
748
|
+
...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
|
|
749
|
+
...um.cachedContentTokenCount !== void 0 && {
|
|
750
|
+
cachedInputTokens: um.cachedContentTokenCount
|
|
751
|
+
}
|
|
752
|
+
};
|
|
753
|
+
}
|
|
720
754
|
var GoogleDriver = class {
|
|
721
755
|
client;
|
|
722
756
|
model;
|
|
723
757
|
maxTokens;
|
|
724
758
|
apiKeyOrEnv;
|
|
725
759
|
reasoningEffort;
|
|
760
|
+
imageDetail;
|
|
726
761
|
constructor(config) {
|
|
727
762
|
this.model = config.model;
|
|
728
763
|
this.maxTokens = config.maxTokens;
|
|
729
764
|
this.client = null;
|
|
730
765
|
this.apiKeyOrEnv = config.apiKey;
|
|
731
766
|
this.reasoningEffort = config.reasoningEffort;
|
|
767
|
+
this.imageDetail = config.imageDetail;
|
|
732
768
|
}
|
|
733
769
|
toGeminiParts(images) {
|
|
734
770
|
return images.map((img) => ({
|
|
@@ -768,6 +804,9 @@ var GoogleDriver = class {
|
|
|
768
804
|
thinkingConfig: {
|
|
769
805
|
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
770
806
|
}
|
|
807
|
+
},
|
|
808
|
+
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
809
|
+
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
771
810
|
}
|
|
772
811
|
}
|
|
773
812
|
});
|
|
@@ -785,14 +824,9 @@ var GoogleDriver = class {
|
|
|
785
824
|
);
|
|
786
825
|
}
|
|
787
826
|
const text = response.text ?? "";
|
|
788
|
-
const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
|
|
789
827
|
return {
|
|
790
828
|
text,
|
|
791
|
-
usage: response.usageMetadata
|
|
792
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
793
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
|
|
794
|
-
...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
|
|
795
|
-
} : void 0
|
|
829
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
796
830
|
};
|
|
797
831
|
} catch (err) {
|
|
798
832
|
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
|
|
@@ -823,10 +857,7 @@ var GoogleDriver = class {
|
|
|
823
857
|
return {
|
|
824
858
|
imageData: Buffer.from(imagePart.inlineData.data, "base64"),
|
|
825
859
|
mimeType: imagePart.inlineData.mimeType,
|
|
826
|
-
usage: response.usageMetadata
|
|
827
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
828
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
|
|
829
|
-
} : void 0
|
|
860
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
830
861
|
};
|
|
831
862
|
} catch (err) {
|
|
832
863
|
if (err instanceof VisualAIProviderError) throw err;
|
|
@@ -842,12 +873,14 @@ var OpenAIDriver = class {
|
|
|
842
873
|
maxTokens;
|
|
843
874
|
apiKeyOrEnv;
|
|
844
875
|
reasoningEffort;
|
|
876
|
+
imageDetail;
|
|
845
877
|
constructor(config) {
|
|
846
878
|
this.model = config.model;
|
|
847
879
|
this.maxTokens = config.maxTokens;
|
|
848
880
|
this.client = null;
|
|
849
881
|
this.apiKeyOrEnv = config.apiKey;
|
|
850
882
|
this.reasoningEffort = config.reasoningEffort;
|
|
883
|
+
this.imageDetail = config.imageDetail;
|
|
851
884
|
}
|
|
852
885
|
async getClient() {
|
|
853
886
|
if (this.client) return this.client;
|
|
@@ -869,9 +902,11 @@ var OpenAIDriver = class {
|
|
|
869
902
|
}
|
|
870
903
|
async sendMessage(images, prompt, options) {
|
|
871
904
|
const client = await this.getClient();
|
|
905
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
872
906
|
const imageBlocks = images.map((img) => ({
|
|
873
907
|
type: "input_image",
|
|
874
|
-
image_url: `data:${img.mimeType};base64,${img.base64}
|
|
908
|
+
image_url: `data:${img.mimeType};base64,${img.base64}`,
|
|
909
|
+
...detail ? { detail } : {}
|
|
875
910
|
}));
|
|
876
911
|
try {
|
|
877
912
|
const format = options?.responseSchema ? {
|
|
@@ -896,21 +931,23 @@ var OpenAIDriver = class {
|
|
|
896
931
|
}
|
|
897
932
|
const response = await client.responses.create(requestParams);
|
|
898
933
|
if (response.status && response.status !== "completed") {
|
|
899
|
-
const
|
|
934
|
+
const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
|
|
900
935
|
throw new VisualAITruncationError(
|
|
901
|
-
`Response truncated: OpenAI returned status "${response.status}"${
|
|
936
|
+
`Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
902
937
|
response.output_text ?? "",
|
|
903
938
|
this.maxTokens
|
|
904
939
|
);
|
|
905
940
|
}
|
|
906
941
|
const text = response.output_text ?? "";
|
|
907
942
|
const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
|
|
943
|
+
const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
|
|
908
944
|
return {
|
|
909
945
|
text,
|
|
910
946
|
usage: response.usage ? {
|
|
911
947
|
inputTokens: response.usage.input_tokens,
|
|
912
948
|
outputTokens: response.usage.output_tokens,
|
|
913
|
-
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
949
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
950
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens }
|
|
914
951
|
} : void 0
|
|
915
952
|
};
|
|
916
953
|
} catch (err) {
|
|
@@ -934,12 +971,14 @@ var OpenRouterDriver = class {
|
|
|
934
971
|
maxTokens;
|
|
935
972
|
apiKeyOrEnv;
|
|
936
973
|
reasoningEffort;
|
|
974
|
+
imageDetail;
|
|
937
975
|
constructor(config) {
|
|
938
976
|
this.model = config.model;
|
|
939
977
|
this.maxTokens = config.maxTokens;
|
|
940
978
|
this.client = null;
|
|
941
979
|
this.apiKeyOrEnv = config.apiKey;
|
|
942
980
|
this.reasoningEffort = config.reasoningEffort;
|
|
981
|
+
this.imageDetail = config.imageDetail;
|
|
943
982
|
}
|
|
944
983
|
async getClient() {
|
|
945
984
|
if (this.client) return this.client;
|
|
@@ -963,9 +1002,13 @@ var OpenRouterDriver = class {
|
|
|
963
1002
|
}
|
|
964
1003
|
async sendMessage(images, prompt, options) {
|
|
965
1004
|
const client = await this.getClient();
|
|
1005
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
966
1006
|
const imageParts = images.map((img) => ({
|
|
967
1007
|
type: "image_url",
|
|
968
|
-
image_url: {
|
|
1008
|
+
image_url: {
|
|
1009
|
+
url: `data:${img.mimeType};base64,${img.base64}`,
|
|
1010
|
+
...detail ? { detail } : {}
|
|
1011
|
+
}
|
|
969
1012
|
}));
|
|
970
1013
|
try {
|
|
971
1014
|
const responseFormat = options?.responseSchema ? {
|
|
@@ -1006,12 +1049,16 @@ var OpenRouterDriver = class {
|
|
|
1006
1049
|
);
|
|
1007
1050
|
}
|
|
1008
1051
|
const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
|
|
1052
|
+
const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
|
|
1053
|
+
const cost = response.usage?.cost;
|
|
1009
1054
|
return {
|
|
1010
1055
|
text,
|
|
1011
1056
|
usage: response.usage ? {
|
|
1012
1057
|
inputTokens: response.usage.prompt_tokens,
|
|
1013
1058
|
outputTokens: response.usage.completion_tokens,
|
|
1014
|
-
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
1059
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
1060
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens },
|
|
1061
|
+
...cost !== void 0 && { cost }
|
|
1015
1062
|
} : void 0
|
|
1016
1063
|
};
|
|
1017
1064
|
} catch (err) {
|
|
@@ -1097,6 +1144,8 @@ function resolveConfig(config) {
|
|
|
1097
1144
|
model,
|
|
1098
1145
|
maxTokens,
|
|
1099
1146
|
reasoningEffort: config.reasoningEffort,
|
|
1147
|
+
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1148
|
+
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1100
1149
|
debug,
|
|
1101
1150
|
debugPrompt,
|
|
1102
1151
|
debugResponse,
|
|
@@ -1111,6 +1160,10 @@ var PRICING_TABLE = {
|
|
|
1111
1160
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1112
1161
|
outputPricePerToken: 50 / PER_MILLION
|
|
1113
1162
|
},
|
|
1163
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1164
|
+
inputPricePerToken: 5 / PER_MILLION,
|
|
1165
|
+
outputPricePerToken: 25 / PER_MILLION
|
|
1166
|
+
},
|
|
1114
1167
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
|
|
1115
1168
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1116
1169
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1140,12 +1193,12 @@ var PRICING_TABLE = {
|
|
|
1140
1193
|
outputPricePerToken: 30 / PER_MILLION
|
|
1141
1194
|
},
|
|
1142
1195
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
|
|
1143
|
-
inputPricePerToken: 2
|
|
1144
|
-
outputPricePerToken:
|
|
1196
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1197
|
+
outputPricePerToken: 12 / PER_MILLION
|
|
1145
1198
|
},
|
|
1146
1199
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
|
|
1147
|
-
inputPricePerToken:
|
|
1148
|
-
outputPricePerToken:
|
|
1200
|
+
inputPricePerToken: 0.2 / PER_MILLION,
|
|
1201
|
+
outputPricePerToken: 1.2 / PER_MILLION
|
|
1149
1202
|
},
|
|
1150
1203
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
|
|
1151
1204
|
inputPricePerToken: 5 / PER_MILLION,
|
|
@@ -1175,6 +1228,18 @@ var PRICING_TABLE = {
|
|
|
1175
1228
|
inputPricePerToken: 0.25 / PER_MILLION,
|
|
1176
1229
|
outputPricePerToken: 2 / PER_MILLION
|
|
1177
1230
|
},
|
|
1231
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1232
|
+
// on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
|
|
1233
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
|
|
1234
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1235
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1236
|
+
},
|
|
1237
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1238
|
+
// on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
|
|
1239
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
|
|
1240
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1241
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1242
|
+
},
|
|
1178
1243
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
|
|
1179
1244
|
inputPricePerToken: 1.5 / PER_MILLION,
|
|
1180
1245
|
outputPricePerToken: 7.5 / PER_MILLION
|
|
@@ -1199,8 +1264,30 @@ var PRICING_TABLE = {
|
|
|
1199
1264
|
inputPricePerToken: 0.5 / PER_MILLION,
|
|
1200
1265
|
outputPricePerToken: 3 / PER_MILLION
|
|
1201
1266
|
},
|
|
1202
|
-
// OpenRouter passes through upstream per-model pricing (verified 2026-
|
|
1267
|
+
// OpenRouter passes through upstream per-model pricing (verified 2026-09-03
|
|
1203
1268
|
// against https://openrouter.ai/api/v1/models).
|
|
1269
|
+
// Meta's own listed rates ($1.25 / $4.25, cached input $0.15) match
|
|
1270
|
+
// OpenRouter's pass-through exactly. Cached input is not modelled here:
|
|
1271
|
+
// `calculateCost` applies no cache discount on any provider.
|
|
1272
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.MUSE_SPARK_1_3}`]: {
|
|
1273
|
+
inputPricePerToken: 1.25 / PER_MILLION,
|
|
1274
|
+
outputPricePerToken: 4.25 / PER_MILLION
|
|
1275
|
+
},
|
|
1276
|
+
// Muse Spark 1.3's data-sharing tier: same model, ~12x cheaper, because
|
|
1277
|
+
// Meta trains on everything submitted through it. Deliberately keyed by the
|
|
1278
|
+
// literal slug rather than a `Model.OpenRouter` entry — it must never be
|
|
1279
|
+
// reachable via autocomplete or default selection. Pass the string yourself
|
|
1280
|
+
// (`model: "meta/muse-spark-1.3-contributor"`) to opt in; cost is still
|
|
1281
|
+
// tracked correctly once you do. Cached input is $0.002/MTok, not modelled
|
|
1282
|
+
// (no provider gets a cache discount here).
|
|
1283
|
+
[`${Provider.OPENROUTER}:meta/muse-spark-1.3-contributor`]: {
|
|
1284
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1285
|
+
outputPricePerToken: 0.2 / PER_MILLION
|
|
1286
|
+
},
|
|
1287
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1288
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1289
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1290
|
+
},
|
|
1204
1291
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
|
|
1205
1292
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1206
1293
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1213,6 +1300,10 @@ var PRICING_TABLE = {
|
|
|
1213
1300
|
inputPricePerToken: 0.82 / PER_MILLION,
|
|
1214
1301
|
outputPricePerToken: 3.75 / PER_MILLION
|
|
1215
1302
|
},
|
|
1303
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
|
|
1304
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1305
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1306
|
+
},
|
|
1216
1307
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
|
|
1217
1308
|
inputPricePerToken: 0.32 / PER_MILLION,
|
|
1218
1309
|
outputPricePerToken: 1.28 / PER_MILLION
|
|
@@ -1242,8 +1333,9 @@ function usageLog(config, method, usage) {
|
|
|
1242
1333
|
const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
|
|
1243
1334
|
const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
|
|
1244
1335
|
const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
|
|
1336
|
+
const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
|
|
1245
1337
|
process.stderr.write(
|
|
1246
|
-
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1338
|
+
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1247
1339
|
`
|
|
1248
1340
|
);
|
|
1249
1341
|
}
|
|
@@ -1254,7 +1346,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
|
|
|
1254
1346
|
inputTokens,
|
|
1255
1347
|
outputTokens,
|
|
1256
1348
|
...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
|
|
1349
|
+
...rawUsage?.cachedInputTokens !== void 0 && {
|
|
1350
|
+
cachedInputTokens: rawUsage.cachedInputTokens
|
|
1351
|
+
},
|
|
1257
1352
|
estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
|
|
1353
|
+
...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
|
|
1258
1354
|
durationSeconds
|
|
1259
1355
|
};
|
|
1260
1356
|
usageLog(config, method, usage);
|
|
@@ -1298,7 +1394,9 @@ var import_sharp = __toESM(require("sharp"), 1);
|
|
|
1298
1394
|
var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
|
|
1299
1395
|
Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
1300
1396
|
Model.Google.GEMINI_3_5_FLASH,
|
|
1301
|
-
Model.Google.GEMINI_3_6_FLASH
|
|
1397
|
+
Model.Google.GEMINI_3_6_FLASH,
|
|
1398
|
+
Model.Google.GEMINI_3_7_FLASH,
|
|
1399
|
+
Model.Google.GEMINI_3_8_FLASH
|
|
1302
1400
|
]);
|
|
1303
1401
|
async function generateAiDiff(imgA, imgB, model, driver) {
|
|
1304
1402
|
if (!driver.generateImage) {
|
|
@@ -1388,7 +1486,6 @@ var EXTENSION_TO_MIME = {
|
|
|
1388
1486
|
".webp": "image/webp",
|
|
1389
1487
|
".gif": "image/gif"
|
|
1390
1488
|
};
|
|
1391
|
-
var MAX_DIMENSION = 1568;
|
|
1392
1489
|
var URL_FETCH_TIMEOUT_MS = 1e4;
|
|
1393
1490
|
function isSupportedMimeType(value) {
|
|
1394
1491
|
return SUPPORTED_FORMATS.has(value);
|
|
@@ -1412,14 +1509,14 @@ function detectMimeType(data) {
|
|
|
1412
1509
|
}
|
|
1413
1510
|
throw new VisualAIImageError("Unable to detect image format from file content");
|
|
1414
1511
|
}
|
|
1415
|
-
async function resizeIfNeeded(data, mimeType) {
|
|
1512
|
+
async function resizeIfNeeded(data, mimeType, maxDimension) {
|
|
1416
1513
|
if (mimeType === "image/gif") {
|
|
1417
1514
|
return data;
|
|
1418
1515
|
}
|
|
1419
1516
|
if (mimeType === "image/png" && data.length >= 24) {
|
|
1420
1517
|
const width2 = data.readUInt32BE(16);
|
|
1421
1518
|
const height2 = data.readUInt32BE(20);
|
|
1422
|
-
if (width2 <=
|
|
1519
|
+
if (width2 <= maxDimension && height2 <= maxDimension) {
|
|
1423
1520
|
return data;
|
|
1424
1521
|
}
|
|
1425
1522
|
}
|
|
@@ -1427,12 +1524,12 @@ async function resizeIfNeeded(data, mimeType) {
|
|
|
1427
1524
|
const metadata = await pipeline.metadata();
|
|
1428
1525
|
const width = metadata.width ?? 0;
|
|
1429
1526
|
const height = metadata.height ?? 0;
|
|
1430
|
-
if (width <=
|
|
1527
|
+
if (width <= maxDimension && height <= maxDimension) {
|
|
1431
1528
|
return data;
|
|
1432
1529
|
}
|
|
1433
1530
|
return pipeline.resize({
|
|
1434
|
-
width:
|
|
1435
|
-
height:
|
|
1531
|
+
width: maxDimension,
|
|
1532
|
+
height: maxDimension,
|
|
1436
1533
|
fit: "inside",
|
|
1437
1534
|
withoutEnlargement: true
|
|
1438
1535
|
}).toBuffer();
|
|
@@ -1487,7 +1584,7 @@ function loadFromBase64(input) {
|
|
|
1487
1584
|
}
|
|
1488
1585
|
return { data, mimeType: mimeType ?? detectMimeType(data) };
|
|
1489
1586
|
}
|
|
1490
|
-
async function normalizeImage(input) {
|
|
1587
|
+
async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1491
1588
|
let data;
|
|
1492
1589
|
let mimeType;
|
|
1493
1590
|
if (Buffer.isBuffer(input)) {
|
|
@@ -1516,7 +1613,7 @@ async function normalizeImage(input) {
|
|
|
1516
1613
|
"Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
|
|
1517
1614
|
);
|
|
1518
1615
|
}
|
|
1519
|
-
data = await resizeIfNeeded(data, mimeType);
|
|
1616
|
+
data = await resizeIfNeeded(data, mimeType, maxDimension);
|
|
1520
1617
|
let cachedBase64;
|
|
1521
1618
|
return {
|
|
1522
1619
|
data,
|
|
@@ -1820,7 +1917,7 @@ async function probeDurationSeconds(videoPath) {
|
|
|
1820
1917
|
});
|
|
1821
1918
|
});
|
|
1822
1919
|
}
|
|
1823
|
-
async function extractFrames(videoPath, options = {}) {
|
|
1920
|
+
async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
|
|
1824
1921
|
const fps = options.fps ?? DEFAULT_FPS;
|
|
1825
1922
|
const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
|
|
1826
1923
|
const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
|
|
@@ -1849,7 +1946,7 @@ async function extractFrames(videoPath, options = {}) {
|
|
|
1849
1946
|
}
|
|
1850
1947
|
const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
|
|
1851
1948
|
try {
|
|
1852
|
-
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${
|
|
1949
|
+
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
|
|
1853
1950
|
await new Promise((resolve2, reject) => {
|
|
1854
1951
|
let settled = false;
|
|
1855
1952
|
const cmd = ffmpeg(videoPath);
|
|
@@ -1947,7 +2044,7 @@ function isFramesInput(input) {
|
|
|
1947
2044
|
function isTimestampedFrameInput(frame) {
|
|
1948
2045
|
return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
|
|
1949
2046
|
}
|
|
1950
|
-
async function normalizeFrames(input) {
|
|
2047
|
+
async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1951
2048
|
const rawFrames = input.frames;
|
|
1952
2049
|
const fps = input.fps ?? DEFAULT_FPS;
|
|
1953
2050
|
if (rawFrames.length === 0) {
|
|
@@ -1972,7 +2069,7 @@ async function normalizeFrames(input) {
|
|
|
1972
2069
|
`Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
|
|
1973
2070
|
);
|
|
1974
2071
|
}
|
|
1975
|
-
const image = await normalizeImage(imageInput);
|
|
2072
|
+
const image = await normalizeImage(imageInput, maxDimension);
|
|
1976
2073
|
return {
|
|
1977
2074
|
data: image.data,
|
|
1978
2075
|
mimeType: image.mimeType,
|
|
@@ -1988,14 +2085,14 @@ async function normalizeFrames(input) {
|
|
|
1988
2085
|
await saveDebugFrames(frames);
|
|
1989
2086
|
return { kind: "video", frames, durationSeconds };
|
|
1990
2087
|
}
|
|
1991
|
-
async function normalizeMedia(input, videoOptions) {
|
|
2088
|
+
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1992
2089
|
if (isFramesInput(input)) {
|
|
1993
|
-
return normalizeFrames(input);
|
|
2090
|
+
return normalizeFrames(input, maxDimension);
|
|
1994
2091
|
}
|
|
1995
2092
|
if (isVideoInput(input)) {
|
|
1996
2093
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
1997
2094
|
try {
|
|
1998
|
-
const { frames, durationSeconds } = await extractFrames(path, videoOptions);
|
|
2095
|
+
const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
|
|
1999
2096
|
await saveDebugFrames(frames);
|
|
2000
2097
|
return { kind: "video", frames, durationSeconds };
|
|
2001
2098
|
} finally {
|
|
@@ -2005,7 +2102,7 @@ async function normalizeMedia(input, videoOptions) {
|
|
|
2005
2102
|
}
|
|
2006
2103
|
}
|
|
2007
2104
|
}
|
|
2008
|
-
const image = await normalizeImage(input);
|
|
2105
|
+
const image = await normalizeImage(input, maxDimension);
|
|
2009
2106
|
return { kind: "image", image };
|
|
2010
2107
|
}
|
|
2011
2108
|
|
|
@@ -2046,7 +2143,17 @@ var UsageInfoSchema = import_zod.z.object({
|
|
|
2046
2143
|
outputTokens: import_zod.z.number(),
|
|
2047
2144
|
/** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
|
|
2048
2145
|
reasoningTokens: import_zod.z.number().optional(),
|
|
2146
|
+
/**
|
|
2147
|
+
* Prompt tokens served from the provider's cache, when reported. Informational
|
|
2148
|
+
* only — `estimatedCost` does not apply a cache discount, because providers
|
|
2149
|
+
* differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
|
|
2150
|
+
* Google) or billed as a separate bucket alongside it (Anthropic).
|
|
2151
|
+
*/
|
|
2152
|
+
cachedInputTokens: import_zod.z.number().optional(),
|
|
2153
|
+
/** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
|
|
2049
2154
|
estimatedCost: import_zod.z.number().optional(),
|
|
2155
|
+
/** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
|
|
2156
|
+
reportedCost: import_zod.z.number().optional(),
|
|
2050
2157
|
durationSeconds: import_zod.z.number().nonnegative().optional()
|
|
2051
2158
|
});
|
|
2052
2159
|
var BaseResultSchema = import_zod.z.object({
|
|
@@ -2186,16 +2293,18 @@ function visualAI(config = {}) {
|
|
|
2186
2293
|
apiKey: resolvedConfig.apiKey,
|
|
2187
2294
|
model: resolvedConfig.model,
|
|
2188
2295
|
maxTokens: resolvedConfig.maxTokens,
|
|
2189
|
-
reasoningEffort: resolvedConfig.reasoningEffort
|
|
2296
|
+
reasoningEffort: resolvedConfig.reasoningEffort,
|
|
2297
|
+
imageDetail: resolvedConfig.imageDetail
|
|
2190
2298
|
};
|
|
2191
2299
|
const driver = createDriver(resolvedConfig.provider, driverConfig);
|
|
2300
|
+
const maxImageDimension = resolvedConfig.maxImageDimension;
|
|
2192
2301
|
async function checkElementsVisibility(image, elements, visible, options) {
|
|
2193
2302
|
const methodName = visible ? "elementsVisible" : "elementsHidden";
|
|
2194
2303
|
if (elements.length === 0) {
|
|
2195
2304
|
throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
|
|
2196
2305
|
}
|
|
2197
2306
|
return withErrorDebug(resolvedConfig, methodName, async () => {
|
|
2198
|
-
const img = await normalizeImage(image);
|
|
2307
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2199
2308
|
const prompt = buildElementsVisibilityPrompt(elements, visible, options);
|
|
2200
2309
|
debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
|
|
2201
2310
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2214,7 +2323,7 @@ function visualAI(config = {}) {
|
|
|
2214
2323
|
throw new VisualAIConfigError("At least one statement is required for check()");
|
|
2215
2324
|
}
|
|
2216
2325
|
return withErrorDebug(resolvedConfig, "check", async () => {
|
|
2217
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2326
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2218
2327
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2219
2328
|
const prompt = buildCheckPrompt(stmts, {
|
|
2220
2329
|
instructions: options?.instructions,
|
|
@@ -2233,7 +2342,7 @@ function visualAI(config = {}) {
|
|
|
2233
2342
|
},
|
|
2234
2343
|
async ask(input, userPrompt, options) {
|
|
2235
2344
|
return withErrorDebug(resolvedConfig, "ask", async () => {
|
|
2236
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2345
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2237
2346
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2238
2347
|
const prompt = buildAskPrompt(userPrompt, {
|
|
2239
2348
|
instructions: options?.instructions,
|
|
@@ -2252,7 +2361,10 @@ function visualAI(config = {}) {
|
|
|
2252
2361
|
},
|
|
2253
2362
|
async compare(imageA, imageB, options) {
|
|
2254
2363
|
return withErrorDebug(resolvedConfig, "compare", async () => {
|
|
2255
|
-
const [imgA, imgB] = await Promise.all([
|
|
2364
|
+
const [imgA, imgB] = await Promise.all([
|
|
2365
|
+
normalizeImage(imageA, maxImageDimension),
|
|
2366
|
+
normalizeImage(imageB, maxImageDimension)
|
|
2367
|
+
]);
|
|
2256
2368
|
const prompt = buildComparePrompt({
|
|
2257
2369
|
userPrompt: options?.prompt,
|
|
2258
2370
|
instructions: options?.instructions
|
|
@@ -2290,7 +2402,7 @@ function visualAI(config = {}) {
|
|
|
2290
2402
|
},
|
|
2291
2403
|
async accessibility(image, options) {
|
|
2292
2404
|
return withErrorDebug(resolvedConfig, "accessibility", async () => {
|
|
2293
|
-
const img = await normalizeImage(image);
|
|
2405
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2294
2406
|
const prompt = buildAccessibilityPrompt(options);
|
|
2295
2407
|
debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
|
|
2296
2408
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2309,7 +2421,7 @@ function visualAI(config = {}) {
|
|
|
2309
2421
|
},
|
|
2310
2422
|
async layout(image, options) {
|
|
2311
2423
|
return withErrorDebug(resolvedConfig, "layout", async () => {
|
|
2312
|
-
const img = await normalizeImage(image);
|
|
2424
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2313
2425
|
const prompt = buildLayoutPrompt(options);
|
|
2314
2426
|
debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
|
|
2315
2427
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2323,7 +2435,7 @@ function visualAI(config = {}) {
|
|
|
2323
2435
|
},
|
|
2324
2436
|
async pageLoad(image, options) {
|
|
2325
2437
|
return withErrorDebug(resolvedConfig, "pageLoad", async () => {
|
|
2326
|
-
const img = await normalizeImage(image);
|
|
2438
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2327
2439
|
const prompt = buildPageLoadPrompt(options);
|
|
2328
2440
|
debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
|
|
2329
2441
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2337,7 +2449,7 @@ function visualAI(config = {}) {
|
|
|
2337
2449
|
},
|
|
2338
2450
|
async content(image, options) {
|
|
2339
2451
|
return withErrorDebug(resolvedConfig, "content", async () => {
|
|
2340
|
-
const img = await normalizeImage(image);
|
|
2452
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2341
2453
|
const prompt = buildContentPrompt(options);
|
|
2342
2454
|
debugLog(resolvedConfig, "content prompt", prompt, "prompt");
|
|
2343
2455
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2416,6 +2528,7 @@ function assertVisualCompareResult(result, label) {
|
|
|
2416
2528
|
ConfidenceSchema,
|
|
2417
2529
|
Content,
|
|
2418
2530
|
DEFAULT_MODELS,
|
|
2531
|
+
ImageDetail,
|
|
2419
2532
|
IssueCategorySchema,
|
|
2420
2533
|
IssuePrioritySchema,
|
|
2421
2534
|
IssueSchema,
|