visual-ai-assertions 0.14.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +40 -14
- package/dist/index.cjs +325 -62
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +118 -3
- package/dist/index.d.ts +118 -3
- package/dist/index.js +324 -62
- package/dist/index.js.map +1 -1
- package/package.json +8 -2
package/dist/index.cjs
CHANGED
|
@@ -38,6 +38,7 @@ __export(index_exports, {
|
|
|
38
38
|
ConfidenceSchema: () => ConfidenceSchema,
|
|
39
39
|
Content: () => Content,
|
|
40
40
|
DEFAULT_MODELS: () => DEFAULT_MODELS,
|
|
41
|
+
ImageDetail: () => ImageDetail,
|
|
41
42
|
IssueCategorySchema: () => IssueCategorySchema,
|
|
42
43
|
IssuePrioritySchema: () => IssuePrioritySchema,
|
|
43
44
|
IssueSchema: () => IssueSchema,
|
|
@@ -73,14 +74,23 @@ var ReasoningEffort = {
|
|
|
73
74
|
HIGH: "high",
|
|
74
75
|
XHIGH: "xhigh"
|
|
75
76
|
};
|
|
77
|
+
var ImageDetail = {
|
|
78
|
+
AUTO: "auto",
|
|
79
|
+
LOW: "low",
|
|
80
|
+
HIGH: "high"
|
|
81
|
+
};
|
|
82
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
83
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
76
84
|
var Provider = {
|
|
77
85
|
ANTHROPIC: "anthropic",
|
|
78
86
|
OPENAI: "openai",
|
|
79
|
-
GOOGLE: "google"
|
|
87
|
+
GOOGLE: "google",
|
|
88
|
+
OPENROUTER: "openrouter"
|
|
80
89
|
};
|
|
81
90
|
var Model = {
|
|
82
91
|
Anthropic: {
|
|
83
92
|
FABLE_5: "claude-fable-5",
|
|
93
|
+
OPUS_5: "claude-opus-5",
|
|
84
94
|
OPUS_4_8: "claude-opus-4-8",
|
|
85
95
|
OPUS_4_7: "claude-opus-4-7",
|
|
86
96
|
OPUS_4_6: "claude-opus-4-6",
|
|
@@ -101,29 +111,51 @@ var Model = {
|
|
|
101
111
|
GPT_5_MINI: "gpt-5-mini"
|
|
102
112
|
},
|
|
103
113
|
Google: {
|
|
114
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
115
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
116
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
104
117
|
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
118
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
105
119
|
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
106
120
|
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
107
121
|
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
122
|
+
},
|
|
123
|
+
/**
|
|
124
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
125
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
126
|
+
* recognizes them. All listed models accept image input.
|
|
127
|
+
*/
|
|
128
|
+
OpenRouter: {
|
|
129
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
130
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
131
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
132
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
133
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
134
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
135
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
108
136
|
}
|
|
109
137
|
};
|
|
110
138
|
var DEFAULT_MODELS = {
|
|
111
139
|
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
112
140
|
[Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
|
|
113
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
|
|
141
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
142
|
+
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
114
143
|
};
|
|
115
144
|
var DEFAULT_MAX_TOKENS = 4096;
|
|
116
145
|
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
117
146
|
var MODEL_TO_PROVIDER = new Map([
|
|
118
147
|
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
119
148
|
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
120
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
|
|
149
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
150
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
121
151
|
]);
|
|
122
152
|
var VALID_PROVIDERS = Object.values(Provider);
|
|
123
153
|
var PROVIDER_DEFAULT_REASONING = {
|
|
124
154
|
openai: "medium",
|
|
125
155
|
anthropic: "off",
|
|
126
|
-
google: "off"
|
|
156
|
+
google: "off",
|
|
157
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
158
|
+
openrouter: "off"
|
|
127
159
|
};
|
|
128
160
|
var Content = {
|
|
129
161
|
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
@@ -581,6 +613,7 @@ function parseRetryAfter(value) {
|
|
|
581
613
|
// src/providers/anthropic.ts
|
|
582
614
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
583
615
|
Model.Anthropic.FABLE_5,
|
|
616
|
+
Model.Anthropic.OPUS_5,
|
|
584
617
|
Model.Anthropic.OPUS_4_8,
|
|
585
618
|
Model.Anthropic.OPUS_4_7,
|
|
586
619
|
Model.Anthropic.SONNET_5
|
|
@@ -589,6 +622,13 @@ function mapEffort(level, model) {
|
|
|
589
622
|
if (level !== "xhigh") return level;
|
|
590
623
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
591
624
|
}
|
|
625
|
+
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
626
|
+
var EFFORT_TO_BUDGET_TOKENS = {
|
|
627
|
+
low: 1024,
|
|
628
|
+
medium: 4096,
|
|
629
|
+
high: 8192,
|
|
630
|
+
xhigh: 16384
|
|
631
|
+
};
|
|
592
632
|
var AnthropicDriver = class {
|
|
593
633
|
client;
|
|
594
634
|
model;
|
|
@@ -644,10 +684,16 @@ var AnthropicDriver = class {
|
|
|
644
684
|
]
|
|
645
685
|
};
|
|
646
686
|
if (this.reasoningEffort) {
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
687
|
+
if (BUDGET_THINKING_MODELS.has(this.model)) {
|
|
688
|
+
const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
|
|
689
|
+
requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
|
|
690
|
+
requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
|
|
691
|
+
} else {
|
|
692
|
+
requestParams.thinking = { type: "adaptive" };
|
|
693
|
+
requestParams.output_config = {
|
|
694
|
+
effort: mapEffort(this.reasoningEffort, this.model)
|
|
695
|
+
};
|
|
696
|
+
}
|
|
651
697
|
}
|
|
652
698
|
const message = await client.messages.create(requestParams);
|
|
653
699
|
const textBlock = message.content.find((block) => block.type === "text");
|
|
@@ -663,7 +709,10 @@ var AnthropicDriver = class {
|
|
|
663
709
|
text,
|
|
664
710
|
usage: {
|
|
665
711
|
inputTokens: message.usage.input_tokens,
|
|
666
|
-
outputTokens: message.usage.output_tokens
|
|
712
|
+
outputTokens: message.usage.output_tokens,
|
|
713
|
+
...message.usage.cache_read_input_tokens !== void 0 && {
|
|
714
|
+
cachedInputTokens: message.usage.cache_read_input_tokens
|
|
715
|
+
}
|
|
667
716
|
}
|
|
668
717
|
};
|
|
669
718
|
} catch (err) {
|
|
@@ -680,23 +729,41 @@ function needsCodeExecution(model) {
|
|
|
680
729
|
return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
|
|
681
730
|
}
|
|
682
731
|
var GOOGLE_THINKING_LEVEL = {
|
|
683
|
-
low: "
|
|
684
|
-
medium: "
|
|
685
|
-
high: "
|
|
732
|
+
low: "low",
|
|
733
|
+
medium: "medium",
|
|
734
|
+
high: "high",
|
|
686
735
|
xhigh: "high"
|
|
687
736
|
};
|
|
737
|
+
var GOOGLE_MEDIA_RESOLUTION = {
|
|
738
|
+
low: "MEDIA_RESOLUTION_LOW",
|
|
739
|
+
high: "MEDIA_RESOLUTION_HIGH"
|
|
740
|
+
};
|
|
741
|
+
function toGeminiUsage(um) {
|
|
742
|
+
if (!um) return void 0;
|
|
743
|
+
const thoughts = um.thoughtsTokenCount ?? 0;
|
|
744
|
+
return {
|
|
745
|
+
inputTokens: um.promptTokenCount ?? 0,
|
|
746
|
+
outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
|
|
747
|
+
...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
|
|
748
|
+
...um.cachedContentTokenCount !== void 0 && {
|
|
749
|
+
cachedInputTokens: um.cachedContentTokenCount
|
|
750
|
+
}
|
|
751
|
+
};
|
|
752
|
+
}
|
|
688
753
|
var GoogleDriver = class {
|
|
689
754
|
client;
|
|
690
755
|
model;
|
|
691
756
|
maxTokens;
|
|
692
757
|
apiKeyOrEnv;
|
|
693
758
|
reasoningEffort;
|
|
759
|
+
imageDetail;
|
|
694
760
|
constructor(config) {
|
|
695
761
|
this.model = config.model;
|
|
696
762
|
this.maxTokens = config.maxTokens;
|
|
697
763
|
this.client = null;
|
|
698
764
|
this.apiKeyOrEnv = config.apiKey;
|
|
699
765
|
this.reasoningEffort = config.reasoningEffort;
|
|
766
|
+
this.imageDetail = config.imageDetail;
|
|
700
767
|
}
|
|
701
768
|
toGeminiParts(images) {
|
|
702
769
|
return images.map((img) => ({
|
|
@@ -736,6 +803,9 @@ var GoogleDriver = class {
|
|
|
736
803
|
thinkingConfig: {
|
|
737
804
|
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
738
805
|
}
|
|
806
|
+
},
|
|
807
|
+
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
808
|
+
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
739
809
|
}
|
|
740
810
|
}
|
|
741
811
|
});
|
|
@@ -753,14 +823,9 @@ var GoogleDriver = class {
|
|
|
753
823
|
);
|
|
754
824
|
}
|
|
755
825
|
const text = response.text ?? "";
|
|
756
|
-
const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
|
|
757
826
|
return {
|
|
758
827
|
text,
|
|
759
|
-
usage: response.usageMetadata
|
|
760
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
761
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
|
|
762
|
-
...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
|
|
763
|
-
} : void 0
|
|
828
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
764
829
|
};
|
|
765
830
|
} catch (err) {
|
|
766
831
|
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
|
|
@@ -791,10 +856,7 @@ var GoogleDriver = class {
|
|
|
791
856
|
return {
|
|
792
857
|
imageData: Buffer.from(imagePart.inlineData.data, "base64"),
|
|
793
858
|
mimeType: imagePart.inlineData.mimeType,
|
|
794
|
-
usage: response.usageMetadata
|
|
795
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
796
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
|
|
797
|
-
} : void 0
|
|
859
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
798
860
|
};
|
|
799
861
|
} catch (err) {
|
|
800
862
|
if (err instanceof VisualAIProviderError) throw err;
|
|
@@ -810,12 +872,14 @@ var OpenAIDriver = class {
|
|
|
810
872
|
maxTokens;
|
|
811
873
|
apiKeyOrEnv;
|
|
812
874
|
reasoningEffort;
|
|
875
|
+
imageDetail;
|
|
813
876
|
constructor(config) {
|
|
814
877
|
this.model = config.model;
|
|
815
878
|
this.maxTokens = config.maxTokens;
|
|
816
879
|
this.client = null;
|
|
817
880
|
this.apiKeyOrEnv = config.apiKey;
|
|
818
881
|
this.reasoningEffort = config.reasoningEffort;
|
|
882
|
+
this.imageDetail = config.imageDetail;
|
|
819
883
|
}
|
|
820
884
|
async getClient() {
|
|
821
885
|
if (this.client) return this.client;
|
|
@@ -837,9 +901,11 @@ var OpenAIDriver = class {
|
|
|
837
901
|
}
|
|
838
902
|
async sendMessage(images, prompt, options) {
|
|
839
903
|
const client = await this.getClient();
|
|
904
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
840
905
|
const imageBlocks = images.map((img) => ({
|
|
841
906
|
type: "input_image",
|
|
842
|
-
image_url: `data:${img.mimeType};base64,${img.base64}
|
|
907
|
+
image_url: `data:${img.mimeType};base64,${img.base64}`,
|
|
908
|
+
...detail ? { detail } : {}
|
|
843
909
|
}));
|
|
844
910
|
try {
|
|
845
911
|
const format = options?.responseSchema ? {
|
|
@@ -864,21 +930,23 @@ var OpenAIDriver = class {
|
|
|
864
930
|
}
|
|
865
931
|
const response = await client.responses.create(requestParams);
|
|
866
932
|
if (response.status && response.status !== "completed") {
|
|
867
|
-
const
|
|
933
|
+
const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
|
|
868
934
|
throw new VisualAITruncationError(
|
|
869
|
-
`Response truncated: OpenAI returned status "${response.status}"${
|
|
935
|
+
`Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
870
936
|
response.output_text ?? "",
|
|
871
937
|
this.maxTokens
|
|
872
938
|
);
|
|
873
939
|
}
|
|
874
940
|
const text = response.output_text ?? "";
|
|
875
941
|
const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
|
|
942
|
+
const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
|
|
876
943
|
return {
|
|
877
944
|
text,
|
|
878
945
|
usage: response.usage ? {
|
|
879
946
|
inputTokens: response.usage.input_tokens,
|
|
880
947
|
outputTokens: response.usage.output_tokens,
|
|
881
|
-
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
948
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
949
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens }
|
|
882
950
|
} : void 0
|
|
883
951
|
};
|
|
884
952
|
} catch (err) {
|
|
@@ -888,6 +956,119 @@ var OpenAIDriver = class {
|
|
|
888
956
|
}
|
|
889
957
|
};
|
|
890
958
|
|
|
959
|
+
// src/providers/openrouter.ts
|
|
960
|
+
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
961
|
+
var OPENROUTER_REASONING_EFFORT = {
|
|
962
|
+
low: "low",
|
|
963
|
+
medium: "medium",
|
|
964
|
+
high: "high",
|
|
965
|
+
xhigh: "high"
|
|
966
|
+
};
|
|
967
|
+
var OpenRouterDriver = class {
|
|
968
|
+
client;
|
|
969
|
+
model;
|
|
970
|
+
maxTokens;
|
|
971
|
+
apiKeyOrEnv;
|
|
972
|
+
reasoningEffort;
|
|
973
|
+
imageDetail;
|
|
974
|
+
constructor(config) {
|
|
975
|
+
this.model = config.model;
|
|
976
|
+
this.maxTokens = config.maxTokens;
|
|
977
|
+
this.client = null;
|
|
978
|
+
this.apiKeyOrEnv = config.apiKey;
|
|
979
|
+
this.reasoningEffort = config.reasoningEffort;
|
|
980
|
+
this.imageDetail = config.imageDetail;
|
|
981
|
+
}
|
|
982
|
+
async getClient() {
|
|
983
|
+
if (this.client) return this.client;
|
|
984
|
+
let OpenAI;
|
|
985
|
+
try {
|
|
986
|
+
const mod = await import("openai");
|
|
987
|
+
OpenAI = mod.default;
|
|
988
|
+
} catch {
|
|
989
|
+
throw new VisualAIConfigError(
|
|
990
|
+
"OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
|
|
991
|
+
);
|
|
992
|
+
}
|
|
993
|
+
const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
|
|
994
|
+
if (!apiKey) {
|
|
995
|
+
throw new VisualAIAuthError(
|
|
996
|
+
"OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
|
|
997
|
+
);
|
|
998
|
+
}
|
|
999
|
+
this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
|
|
1000
|
+
return this.client;
|
|
1001
|
+
}
|
|
1002
|
+
async sendMessage(images, prompt, options) {
|
|
1003
|
+
const client = await this.getClient();
|
|
1004
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
1005
|
+
const imageParts = images.map((img) => ({
|
|
1006
|
+
type: "image_url",
|
|
1007
|
+
image_url: {
|
|
1008
|
+
url: `data:${img.mimeType};base64,${img.base64}`,
|
|
1009
|
+
...detail ? { detail } : {}
|
|
1010
|
+
}
|
|
1011
|
+
}));
|
|
1012
|
+
try {
|
|
1013
|
+
const responseFormat = options?.responseSchema ? {
|
|
1014
|
+
type: "json_schema",
|
|
1015
|
+
json_schema: {
|
|
1016
|
+
name: "visual_ai_response",
|
|
1017
|
+
strict: true,
|
|
1018
|
+
schema: options.responseSchema
|
|
1019
|
+
}
|
|
1020
|
+
} : { type: "json_object" };
|
|
1021
|
+
const requestParams = {
|
|
1022
|
+
model: this.model,
|
|
1023
|
+
max_tokens: this.maxTokens,
|
|
1024
|
+
response_format: responseFormat,
|
|
1025
|
+
messages: [
|
|
1026
|
+
{
|
|
1027
|
+
role: "user",
|
|
1028
|
+
content: [...imageParts, { type: "text", text: prompt }]
|
|
1029
|
+
}
|
|
1030
|
+
],
|
|
1031
|
+
// OpenRouter-specific: include token accounting in the response.
|
|
1032
|
+
usage: { include: true }
|
|
1033
|
+
};
|
|
1034
|
+
if (this.reasoningEffort) {
|
|
1035
|
+
requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
|
|
1036
|
+
}
|
|
1037
|
+
const response = await client.chat.completions.create(requestParams);
|
|
1038
|
+
const choice = response.choices?.[0];
|
|
1039
|
+
if (!choice?.message) {
|
|
1040
|
+
throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
|
|
1041
|
+
}
|
|
1042
|
+
const text = choice.message.content ?? "";
|
|
1043
|
+
if (choice.finish_reason === "length") {
|
|
1044
|
+
throw new VisualAITruncationError(
|
|
1045
|
+
`Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
1046
|
+
text,
|
|
1047
|
+
this.maxTokens
|
|
1048
|
+
);
|
|
1049
|
+
}
|
|
1050
|
+
const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
|
|
1051
|
+
const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
|
|
1052
|
+
const cost = response.usage?.cost;
|
|
1053
|
+
return {
|
|
1054
|
+
text,
|
|
1055
|
+
usage: response.usage ? {
|
|
1056
|
+
inputTokens: response.usage.prompt_tokens,
|
|
1057
|
+
outputTokens: response.usage.completion_tokens,
|
|
1058
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
1059
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens },
|
|
1060
|
+
...cost !== void 0 && { cost }
|
|
1061
|
+
} : void 0
|
|
1062
|
+
};
|
|
1063
|
+
} catch (err) {
|
|
1064
|
+
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
|
|
1065
|
+
throw err;
|
|
1066
|
+
}
|
|
1067
|
+
throw mapProviderError(err);
|
|
1068
|
+
}
|
|
1069
|
+
}
|
|
1070
|
+
};
|
|
1071
|
+
|
|
891
1072
|
// src/core/config.ts
|
|
892
1073
|
var MODEL_PREFIX_TO_PROVIDER = [
|
|
893
1074
|
["claude-", "anthropic"],
|
|
@@ -900,6 +1081,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
|
|
|
900
1081
|
function inferProviderFromModel(model) {
|
|
901
1082
|
const known = MODEL_TO_PROVIDER.get(model);
|
|
902
1083
|
if (known) return known;
|
|
1084
|
+
if (model.includes("/")) return "openrouter";
|
|
903
1085
|
const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
|
|
904
1086
|
return prefixMatch?.[1];
|
|
905
1087
|
}
|
|
@@ -912,12 +1094,13 @@ function resolveProvider(config) {
|
|
|
912
1094
|
const apiKeyProviderMap = [
|
|
913
1095
|
["ANTHROPIC_API_KEY", "anthropic"],
|
|
914
1096
|
["OPENAI_API_KEY", "openai"],
|
|
915
|
-
["GOOGLE_API_KEY", "google"]
|
|
1097
|
+
["GOOGLE_API_KEY", "google"],
|
|
1098
|
+
["OPENROUTER_API_KEY", "openrouter"]
|
|
916
1099
|
];
|
|
917
1100
|
const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
|
|
918
1101
|
if (detected) return detected[1];
|
|
919
1102
|
throw new VisualAIConfigError(
|
|
920
|
-
"Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
|
|
1103
|
+
"Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
|
|
921
1104
|
);
|
|
922
1105
|
}
|
|
923
1106
|
function parseBooleanEnv(envName, value) {
|
|
@@ -945,11 +1128,11 @@ function resolveConfig(config) {
|
|
|
945
1128
|
}
|
|
946
1129
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
947
1130
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
948
|
-
if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
|
|
1131
|
+
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
|
|
949
1132
|
maxTokens = OPENAI_REASONING_MAX_TOKENS;
|
|
950
1133
|
if (debug) {
|
|
951
1134
|
process.stderr.write(
|
|
952
|
-
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for
|
|
1135
|
+
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
|
|
953
1136
|
`
|
|
954
1137
|
);
|
|
955
1138
|
}
|
|
@@ -960,6 +1143,8 @@ function resolveConfig(config) {
|
|
|
960
1143
|
model,
|
|
961
1144
|
maxTokens,
|
|
962
1145
|
reasoningEffort: config.reasoningEffort,
|
|
1146
|
+
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1147
|
+
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
963
1148
|
debug,
|
|
964
1149
|
debugPrompt,
|
|
965
1150
|
debugResponse,
|
|
@@ -974,6 +1159,10 @@ var PRICING_TABLE = {
|
|
|
974
1159
|
inputPricePerToken: 10 / PER_MILLION,
|
|
975
1160
|
outputPricePerToken: 50 / PER_MILLION
|
|
976
1161
|
},
|
|
1162
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1163
|
+
inputPricePerToken: 5 / PER_MILLION,
|
|
1164
|
+
outputPricePerToken: 25 / PER_MILLION
|
|
1165
|
+
},
|
|
977
1166
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
|
|
978
1167
|
inputPricePerToken: 5 / PER_MILLION,
|
|
979
1168
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1003,12 +1192,12 @@ var PRICING_TABLE = {
|
|
|
1003
1192
|
outputPricePerToken: 30 / PER_MILLION
|
|
1004
1193
|
},
|
|
1005
1194
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
|
|
1006
|
-
inputPricePerToken: 2
|
|
1007
|
-
outputPricePerToken:
|
|
1195
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1196
|
+
outputPricePerToken: 12 / PER_MILLION
|
|
1008
1197
|
},
|
|
1009
1198
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
|
|
1010
|
-
inputPricePerToken:
|
|
1011
|
-
outputPricePerToken:
|
|
1199
|
+
inputPricePerToken: 0.2 / PER_MILLION,
|
|
1200
|
+
outputPricePerToken: 1.2 / PER_MILLION
|
|
1012
1201
|
},
|
|
1013
1202
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
|
|
1014
1203
|
inputPricePerToken: 5 / PER_MILLION,
|
|
@@ -1038,10 +1227,30 @@ var PRICING_TABLE = {
|
|
|
1038
1227
|
inputPricePerToken: 0.25 / PER_MILLION,
|
|
1039
1228
|
outputPricePerToken: 2 / PER_MILLION
|
|
1040
1229
|
},
|
|
1230
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1231
|
+
// on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
|
|
1232
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
|
|
1233
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1234
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1235
|
+
},
|
|
1236
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1237
|
+
// on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
|
|
1238
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
|
|
1239
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1240
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1241
|
+
},
|
|
1242
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
|
|
1243
|
+
inputPricePerToken: 1.5 / PER_MILLION,
|
|
1244
|
+
outputPricePerToken: 7.5 / PER_MILLION
|
|
1245
|
+
},
|
|
1041
1246
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
|
|
1042
1247
|
inputPricePerToken: 1.5 / PER_MILLION,
|
|
1043
1248
|
outputPricePerToken: 9 / PER_MILLION
|
|
1044
1249
|
},
|
|
1250
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
|
|
1251
|
+
inputPricePerToken: 0.3 / PER_MILLION,
|
|
1252
|
+
outputPricePerToken: 2.5 / PER_MILLION
|
|
1253
|
+
},
|
|
1045
1254
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
|
|
1046
1255
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1047
1256
|
outputPricePerToken: 12 / PER_MILLION
|
|
@@ -1053,6 +1262,36 @@ var PRICING_TABLE = {
|
|
|
1053
1262
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
|
|
1054
1263
|
inputPricePerToken: 0.5 / PER_MILLION,
|
|
1055
1264
|
outputPricePerToken: 3 / PER_MILLION
|
|
1265
|
+
},
|
|
1266
|
+
// OpenRouter passes through upstream per-model pricing (verified 2026-07-22
|
|
1267
|
+
// against https://openrouter.ai/api/v1/models).
|
|
1268
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1269
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1270
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1271
|
+
},
|
|
1272
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
|
|
1273
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1274
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1275
|
+
},
|
|
1276
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
|
|
1277
|
+
inputPricePerToken: 3 / PER_MILLION,
|
|
1278
|
+
outputPricePerToken: 15 / PER_MILLION
|
|
1279
|
+
},
|
|
1280
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
|
|
1281
|
+
inputPricePerToken: 0.82 / PER_MILLION,
|
|
1282
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1283
|
+
},
|
|
1284
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
|
|
1285
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1286
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1287
|
+
},
|
|
1288
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
|
|
1289
|
+
inputPricePerToken: 0.32 / PER_MILLION,
|
|
1290
|
+
outputPricePerToken: 1.28 / PER_MILLION
|
|
1291
|
+
},
|
|
1292
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
|
|
1293
|
+
inputPricePerToken: 0.1875 / PER_MILLION,
|
|
1294
|
+
outputPricePerToken: 1.125 / PER_MILLION
|
|
1056
1295
|
}
|
|
1057
1296
|
};
|
|
1058
1297
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -1075,8 +1314,9 @@ function usageLog(config, method, usage) {
|
|
|
1075
1314
|
const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
|
|
1076
1315
|
const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
|
|
1077
1316
|
const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
|
|
1317
|
+
const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
|
|
1078
1318
|
process.stderr.write(
|
|
1079
|
-
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1319
|
+
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1080
1320
|
`
|
|
1081
1321
|
);
|
|
1082
1322
|
}
|
|
@@ -1087,7 +1327,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
|
|
|
1087
1327
|
inputTokens,
|
|
1088
1328
|
outputTokens,
|
|
1089
1329
|
...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
|
|
1330
|
+
...rawUsage?.cachedInputTokens !== void 0 && {
|
|
1331
|
+
cachedInputTokens: rawUsage.cachedInputTokens
|
|
1332
|
+
},
|
|
1090
1333
|
estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
|
|
1334
|
+
...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
|
|
1091
1335
|
durationSeconds
|
|
1092
1336
|
};
|
|
1093
1337
|
usageLog(config, method, usage);
|
|
@@ -1130,7 +1374,10 @@ async function timedSendMessage(driver, images, prompt, options) {
|
|
|
1130
1374
|
var import_sharp = __toESM(require("sharp"), 1);
|
|
1131
1375
|
var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
|
|
1132
1376
|
Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
1133
|
-
Model.Google.GEMINI_3_5_FLASH
|
|
1377
|
+
Model.Google.GEMINI_3_5_FLASH,
|
|
1378
|
+
Model.Google.GEMINI_3_6_FLASH,
|
|
1379
|
+
Model.Google.GEMINI_3_7_FLASH,
|
|
1380
|
+
Model.Google.GEMINI_3_8_FLASH
|
|
1134
1381
|
]);
|
|
1135
1382
|
async function generateAiDiff(imgA, imgB, model, driver) {
|
|
1136
1383
|
if (!driver.generateImage) {
|
|
@@ -1220,7 +1467,6 @@ var EXTENSION_TO_MIME = {
|
|
|
1220
1467
|
".webp": "image/webp",
|
|
1221
1468
|
".gif": "image/gif"
|
|
1222
1469
|
};
|
|
1223
|
-
var MAX_DIMENSION = 1568;
|
|
1224
1470
|
var URL_FETCH_TIMEOUT_MS = 1e4;
|
|
1225
1471
|
function isSupportedMimeType(value) {
|
|
1226
1472
|
return SUPPORTED_FORMATS.has(value);
|
|
@@ -1244,14 +1490,14 @@ function detectMimeType(data) {
|
|
|
1244
1490
|
}
|
|
1245
1491
|
throw new VisualAIImageError("Unable to detect image format from file content");
|
|
1246
1492
|
}
|
|
1247
|
-
async function resizeIfNeeded(data, mimeType) {
|
|
1493
|
+
async function resizeIfNeeded(data, mimeType, maxDimension) {
|
|
1248
1494
|
if (mimeType === "image/gif") {
|
|
1249
1495
|
return data;
|
|
1250
1496
|
}
|
|
1251
1497
|
if (mimeType === "image/png" && data.length >= 24) {
|
|
1252
1498
|
const width2 = data.readUInt32BE(16);
|
|
1253
1499
|
const height2 = data.readUInt32BE(20);
|
|
1254
|
-
if (width2 <=
|
|
1500
|
+
if (width2 <= maxDimension && height2 <= maxDimension) {
|
|
1255
1501
|
return data;
|
|
1256
1502
|
}
|
|
1257
1503
|
}
|
|
@@ -1259,12 +1505,12 @@ async function resizeIfNeeded(data, mimeType) {
|
|
|
1259
1505
|
const metadata = await pipeline.metadata();
|
|
1260
1506
|
const width = metadata.width ?? 0;
|
|
1261
1507
|
const height = metadata.height ?? 0;
|
|
1262
|
-
if (width <=
|
|
1508
|
+
if (width <= maxDimension && height <= maxDimension) {
|
|
1263
1509
|
return data;
|
|
1264
1510
|
}
|
|
1265
1511
|
return pipeline.resize({
|
|
1266
|
-
width:
|
|
1267
|
-
height:
|
|
1512
|
+
width: maxDimension,
|
|
1513
|
+
height: maxDimension,
|
|
1268
1514
|
fit: "inside",
|
|
1269
1515
|
withoutEnlargement: true
|
|
1270
1516
|
}).toBuffer();
|
|
@@ -1319,7 +1565,7 @@ function loadFromBase64(input) {
|
|
|
1319
1565
|
}
|
|
1320
1566
|
return { data, mimeType: mimeType ?? detectMimeType(data) };
|
|
1321
1567
|
}
|
|
1322
|
-
async function normalizeImage(input) {
|
|
1568
|
+
async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1323
1569
|
let data;
|
|
1324
1570
|
let mimeType;
|
|
1325
1571
|
if (Buffer.isBuffer(input)) {
|
|
@@ -1348,7 +1594,7 @@ async function normalizeImage(input) {
|
|
|
1348
1594
|
"Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
|
|
1349
1595
|
);
|
|
1350
1596
|
}
|
|
1351
|
-
data = await resizeIfNeeded(data, mimeType);
|
|
1597
|
+
data = await resizeIfNeeded(data, mimeType, maxDimension);
|
|
1352
1598
|
let cachedBase64;
|
|
1353
1599
|
return {
|
|
1354
1600
|
data,
|
|
@@ -1652,7 +1898,7 @@ async function probeDurationSeconds(videoPath) {
|
|
|
1652
1898
|
});
|
|
1653
1899
|
});
|
|
1654
1900
|
}
|
|
1655
|
-
async function extractFrames(videoPath, options = {}) {
|
|
1901
|
+
async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
|
|
1656
1902
|
const fps = options.fps ?? DEFAULT_FPS;
|
|
1657
1903
|
const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
|
|
1658
1904
|
const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
|
|
@@ -1681,7 +1927,7 @@ async function extractFrames(videoPath, options = {}) {
|
|
|
1681
1927
|
}
|
|
1682
1928
|
const outputDir = await (0, import_promises3.mkdtemp)((0, import_node_path3.join)((0, import_node_os.tmpdir)(), "visual-ai-frames-"));
|
|
1683
1929
|
try {
|
|
1684
|
-
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${
|
|
1930
|
+
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
|
|
1685
1931
|
await new Promise((resolve2, reject) => {
|
|
1686
1932
|
let settled = false;
|
|
1687
1933
|
const cmd = ffmpeg(videoPath);
|
|
@@ -1779,7 +2025,7 @@ function isFramesInput(input) {
|
|
|
1779
2025
|
function isTimestampedFrameInput(frame) {
|
|
1780
2026
|
return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
|
|
1781
2027
|
}
|
|
1782
|
-
async function normalizeFrames(input) {
|
|
2028
|
+
async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1783
2029
|
const rawFrames = input.frames;
|
|
1784
2030
|
const fps = input.fps ?? DEFAULT_FPS;
|
|
1785
2031
|
if (rawFrames.length === 0) {
|
|
@@ -1804,7 +2050,7 @@ async function normalizeFrames(input) {
|
|
|
1804
2050
|
`Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
|
|
1805
2051
|
);
|
|
1806
2052
|
}
|
|
1807
|
-
const image = await normalizeImage(imageInput);
|
|
2053
|
+
const image = await normalizeImage(imageInput, maxDimension);
|
|
1808
2054
|
return {
|
|
1809
2055
|
data: image.data,
|
|
1810
2056
|
mimeType: image.mimeType,
|
|
@@ -1820,14 +2066,14 @@ async function normalizeFrames(input) {
|
|
|
1820
2066
|
await saveDebugFrames(frames);
|
|
1821
2067
|
return { kind: "video", frames, durationSeconds };
|
|
1822
2068
|
}
|
|
1823
|
-
async function normalizeMedia(input, videoOptions) {
|
|
2069
|
+
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1824
2070
|
if (isFramesInput(input)) {
|
|
1825
|
-
return normalizeFrames(input);
|
|
2071
|
+
return normalizeFrames(input, maxDimension);
|
|
1826
2072
|
}
|
|
1827
2073
|
if (isVideoInput(input)) {
|
|
1828
2074
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
1829
2075
|
try {
|
|
1830
|
-
const { frames, durationSeconds } = await extractFrames(path, videoOptions);
|
|
2076
|
+
const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
|
|
1831
2077
|
await saveDebugFrames(frames);
|
|
1832
2078
|
return { kind: "video", frames, durationSeconds };
|
|
1833
2079
|
} finally {
|
|
@@ -1837,7 +2083,7 @@ async function normalizeMedia(input, videoOptions) {
|
|
|
1837
2083
|
}
|
|
1838
2084
|
}
|
|
1839
2085
|
}
|
|
1840
|
-
const image = await normalizeImage(input);
|
|
2086
|
+
const image = await normalizeImage(input, maxDimension);
|
|
1841
2087
|
return { kind: "image", image };
|
|
1842
2088
|
}
|
|
1843
2089
|
|
|
@@ -1878,7 +2124,17 @@ var UsageInfoSchema = import_zod.z.object({
|
|
|
1878
2124
|
outputTokens: import_zod.z.number(),
|
|
1879
2125
|
/** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
|
|
1880
2126
|
reasoningTokens: import_zod.z.number().optional(),
|
|
2127
|
+
/**
|
|
2128
|
+
* Prompt tokens served from the provider's cache, when reported. Informational
|
|
2129
|
+
* only — `estimatedCost` does not apply a cache discount, because providers
|
|
2130
|
+
* differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
|
|
2131
|
+
* Google) or billed as a separate bucket alongside it (Anthropic).
|
|
2132
|
+
*/
|
|
2133
|
+
cachedInputTokens: import_zod.z.number().optional(),
|
|
2134
|
+
/** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
|
|
1881
2135
|
estimatedCost: import_zod.z.number().optional(),
|
|
2136
|
+
/** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
|
|
2137
|
+
reportedCost: import_zod.z.number().optional(),
|
|
1882
2138
|
durationSeconds: import_zod.z.number().nonnegative().optional()
|
|
1883
2139
|
});
|
|
1884
2140
|
var BaseResultSchema = import_zod.z.object({
|
|
@@ -1980,7 +2236,8 @@ function toSchemaOptions(schema) {
|
|
|
1980
2236
|
var PROVIDER_REGISTRY = {
|
|
1981
2237
|
anthropic: (config) => new AnthropicDriver(config),
|
|
1982
2238
|
openai: (config) => new OpenAIDriver(config),
|
|
1983
|
-
google: (config) => new GoogleDriver(config)
|
|
2239
|
+
google: (config) => new GoogleDriver(config),
|
|
2240
|
+
openrouter: (config) => new OpenRouterDriver(config)
|
|
1984
2241
|
};
|
|
1985
2242
|
function createDriver(provider, config) {
|
|
1986
2243
|
return PROVIDER_REGISTRY[provider](config);
|
|
@@ -2017,16 +2274,18 @@ function visualAI(config = {}) {
|
|
|
2017
2274
|
apiKey: resolvedConfig.apiKey,
|
|
2018
2275
|
model: resolvedConfig.model,
|
|
2019
2276
|
maxTokens: resolvedConfig.maxTokens,
|
|
2020
|
-
reasoningEffort: resolvedConfig.reasoningEffort
|
|
2277
|
+
reasoningEffort: resolvedConfig.reasoningEffort,
|
|
2278
|
+
imageDetail: resolvedConfig.imageDetail
|
|
2021
2279
|
};
|
|
2022
2280
|
const driver = createDriver(resolvedConfig.provider, driverConfig);
|
|
2281
|
+
const maxImageDimension = resolvedConfig.maxImageDimension;
|
|
2023
2282
|
async function checkElementsVisibility(image, elements, visible, options) {
|
|
2024
2283
|
const methodName = visible ? "elementsVisible" : "elementsHidden";
|
|
2025
2284
|
if (elements.length === 0) {
|
|
2026
2285
|
throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
|
|
2027
2286
|
}
|
|
2028
2287
|
return withErrorDebug(resolvedConfig, methodName, async () => {
|
|
2029
|
-
const img = await normalizeImage(image);
|
|
2288
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2030
2289
|
const prompt = buildElementsVisibilityPrompt(elements, visible, options);
|
|
2031
2290
|
debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
|
|
2032
2291
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2045,7 +2304,7 @@ function visualAI(config = {}) {
|
|
|
2045
2304
|
throw new VisualAIConfigError("At least one statement is required for check()");
|
|
2046
2305
|
}
|
|
2047
2306
|
return withErrorDebug(resolvedConfig, "check", async () => {
|
|
2048
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2307
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2049
2308
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2050
2309
|
const prompt = buildCheckPrompt(stmts, {
|
|
2051
2310
|
instructions: options?.instructions,
|
|
@@ -2064,7 +2323,7 @@ function visualAI(config = {}) {
|
|
|
2064
2323
|
},
|
|
2065
2324
|
async ask(input, userPrompt, options) {
|
|
2066
2325
|
return withErrorDebug(resolvedConfig, "ask", async () => {
|
|
2067
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2326
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2068
2327
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2069
2328
|
const prompt = buildAskPrompt(userPrompt, {
|
|
2070
2329
|
instructions: options?.instructions,
|
|
@@ -2083,7 +2342,10 @@ function visualAI(config = {}) {
|
|
|
2083
2342
|
},
|
|
2084
2343
|
async compare(imageA, imageB, options) {
|
|
2085
2344
|
return withErrorDebug(resolvedConfig, "compare", async () => {
|
|
2086
|
-
const [imgA, imgB] = await Promise.all([
|
|
2345
|
+
const [imgA, imgB] = await Promise.all([
|
|
2346
|
+
normalizeImage(imageA, maxImageDimension),
|
|
2347
|
+
normalizeImage(imageB, maxImageDimension)
|
|
2348
|
+
]);
|
|
2087
2349
|
const prompt = buildComparePrompt({
|
|
2088
2350
|
userPrompt: options?.prompt,
|
|
2089
2351
|
instructions: options?.instructions
|
|
@@ -2121,7 +2383,7 @@ function visualAI(config = {}) {
|
|
|
2121
2383
|
},
|
|
2122
2384
|
async accessibility(image, options) {
|
|
2123
2385
|
return withErrorDebug(resolvedConfig, "accessibility", async () => {
|
|
2124
|
-
const img = await normalizeImage(image);
|
|
2386
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2125
2387
|
const prompt = buildAccessibilityPrompt(options);
|
|
2126
2388
|
debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
|
|
2127
2389
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2140,7 +2402,7 @@ function visualAI(config = {}) {
|
|
|
2140
2402
|
},
|
|
2141
2403
|
async layout(image, options) {
|
|
2142
2404
|
return withErrorDebug(resolvedConfig, "layout", async () => {
|
|
2143
|
-
const img = await normalizeImage(image);
|
|
2405
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2144
2406
|
const prompt = buildLayoutPrompt(options);
|
|
2145
2407
|
debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
|
|
2146
2408
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2154,7 +2416,7 @@ function visualAI(config = {}) {
|
|
|
2154
2416
|
},
|
|
2155
2417
|
async pageLoad(image, options) {
|
|
2156
2418
|
return withErrorDebug(resolvedConfig, "pageLoad", async () => {
|
|
2157
|
-
const img = await normalizeImage(image);
|
|
2419
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2158
2420
|
const prompt = buildPageLoadPrompt(options);
|
|
2159
2421
|
debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
|
|
2160
2422
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2168,7 +2430,7 @@ function visualAI(config = {}) {
|
|
|
2168
2430
|
},
|
|
2169
2431
|
async content(image, options) {
|
|
2170
2432
|
return withErrorDebug(resolvedConfig, "content", async () => {
|
|
2171
|
-
const img = await normalizeImage(image);
|
|
2433
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2172
2434
|
const prompt = buildContentPrompt(options);
|
|
2173
2435
|
debugLog(resolvedConfig, "content prompt", prompt, "prompt");
|
|
2174
2436
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2247,6 +2509,7 @@ function assertVisualCompareResult(result, label) {
|
|
|
2247
2509
|
ConfidenceSchema,
|
|
2248
2510
|
Content,
|
|
2249
2511
|
DEFAULT_MODELS,
|
|
2512
|
+
ImageDetail,
|
|
2250
2513
|
IssueCategorySchema,
|
|
2251
2514
|
IssuePrioritySchema,
|
|
2252
2515
|
IssueSchema,
|