visual-ai-assertions 0.14.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +40 -14
- package/dist/index.cjs +325 -62
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +118 -3
- package/dist/index.d.ts +118 -3
- package/dist/index.js +324 -62
- package/dist/index.js.map +1 -1
- package/package.json +8 -2
package/dist/index.js
CHANGED
|
@@ -5,14 +5,23 @@ var ReasoningEffort = {
|
|
|
5
5
|
HIGH: "high",
|
|
6
6
|
XHIGH: "xhigh"
|
|
7
7
|
};
|
|
8
|
+
var ImageDetail = {
|
|
9
|
+
AUTO: "auto",
|
|
10
|
+
LOW: "low",
|
|
11
|
+
HIGH: "high"
|
|
12
|
+
};
|
|
13
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
14
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
8
15
|
var Provider = {
|
|
9
16
|
ANTHROPIC: "anthropic",
|
|
10
17
|
OPENAI: "openai",
|
|
11
|
-
GOOGLE: "google"
|
|
18
|
+
GOOGLE: "google",
|
|
19
|
+
OPENROUTER: "openrouter"
|
|
12
20
|
};
|
|
13
21
|
var Model = {
|
|
14
22
|
Anthropic: {
|
|
15
23
|
FABLE_5: "claude-fable-5",
|
|
24
|
+
OPUS_5: "claude-opus-5",
|
|
16
25
|
OPUS_4_8: "claude-opus-4-8",
|
|
17
26
|
OPUS_4_7: "claude-opus-4-7",
|
|
18
27
|
OPUS_4_6: "claude-opus-4-6",
|
|
@@ -33,29 +42,51 @@ var Model = {
|
|
|
33
42
|
GPT_5_MINI: "gpt-5-mini"
|
|
34
43
|
},
|
|
35
44
|
Google: {
|
|
45
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
46
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
47
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
36
48
|
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
49
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
37
50
|
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
38
51
|
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
39
52
|
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
53
|
+
},
|
|
54
|
+
/**
|
|
55
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
56
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
57
|
+
* recognizes them. All listed models accept image input.
|
|
58
|
+
*/
|
|
59
|
+
OpenRouter: {
|
|
60
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
61
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
62
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
63
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
64
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
65
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
66
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
40
67
|
}
|
|
41
68
|
};
|
|
42
69
|
var DEFAULT_MODELS = {
|
|
43
70
|
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
44
71
|
[Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
|
|
45
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
|
|
72
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
73
|
+
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
46
74
|
};
|
|
47
75
|
var DEFAULT_MAX_TOKENS = 4096;
|
|
48
76
|
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
49
77
|
var MODEL_TO_PROVIDER = new Map([
|
|
50
78
|
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
51
79
|
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
52
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
|
|
80
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
81
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
53
82
|
]);
|
|
54
83
|
var VALID_PROVIDERS = Object.values(Provider);
|
|
55
84
|
var PROVIDER_DEFAULT_REASONING = {
|
|
56
85
|
openai: "medium",
|
|
57
86
|
anthropic: "off",
|
|
58
|
-
google: "off"
|
|
87
|
+
google: "off",
|
|
88
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
89
|
+
openrouter: "off"
|
|
59
90
|
};
|
|
60
91
|
var Content = {
|
|
61
92
|
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
@@ -513,6 +544,7 @@ function parseRetryAfter(value) {
|
|
|
513
544
|
// src/providers/anthropic.ts
|
|
514
545
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
515
546
|
Model.Anthropic.FABLE_5,
|
|
547
|
+
Model.Anthropic.OPUS_5,
|
|
516
548
|
Model.Anthropic.OPUS_4_8,
|
|
517
549
|
Model.Anthropic.OPUS_4_7,
|
|
518
550
|
Model.Anthropic.SONNET_5
|
|
@@ -521,6 +553,13 @@ function mapEffort(level, model) {
|
|
|
521
553
|
if (level !== "xhigh") return level;
|
|
522
554
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
523
555
|
}
|
|
556
|
+
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
557
|
+
var EFFORT_TO_BUDGET_TOKENS = {
|
|
558
|
+
low: 1024,
|
|
559
|
+
medium: 4096,
|
|
560
|
+
high: 8192,
|
|
561
|
+
xhigh: 16384
|
|
562
|
+
};
|
|
524
563
|
var AnthropicDriver = class {
|
|
525
564
|
client;
|
|
526
565
|
model;
|
|
@@ -576,10 +615,16 @@ var AnthropicDriver = class {
|
|
|
576
615
|
]
|
|
577
616
|
};
|
|
578
617
|
if (this.reasoningEffort) {
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
618
|
+
if (BUDGET_THINKING_MODELS.has(this.model)) {
|
|
619
|
+
const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
|
|
620
|
+
requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
|
|
621
|
+
requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
|
|
622
|
+
} else {
|
|
623
|
+
requestParams.thinking = { type: "adaptive" };
|
|
624
|
+
requestParams.output_config = {
|
|
625
|
+
effort: mapEffort(this.reasoningEffort, this.model)
|
|
626
|
+
};
|
|
627
|
+
}
|
|
583
628
|
}
|
|
584
629
|
const message = await client.messages.create(requestParams);
|
|
585
630
|
const textBlock = message.content.find((block) => block.type === "text");
|
|
@@ -595,7 +640,10 @@ var AnthropicDriver = class {
|
|
|
595
640
|
text,
|
|
596
641
|
usage: {
|
|
597
642
|
inputTokens: message.usage.input_tokens,
|
|
598
|
-
outputTokens: message.usage.output_tokens
|
|
643
|
+
outputTokens: message.usage.output_tokens,
|
|
644
|
+
...message.usage.cache_read_input_tokens !== void 0 && {
|
|
645
|
+
cachedInputTokens: message.usage.cache_read_input_tokens
|
|
646
|
+
}
|
|
599
647
|
}
|
|
600
648
|
};
|
|
601
649
|
} catch (err) {
|
|
@@ -612,23 +660,41 @@ function needsCodeExecution(model) {
|
|
|
612
660
|
return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
|
|
613
661
|
}
|
|
614
662
|
var GOOGLE_THINKING_LEVEL = {
|
|
615
|
-
low: "
|
|
616
|
-
medium: "
|
|
617
|
-
high: "
|
|
663
|
+
low: "low",
|
|
664
|
+
medium: "medium",
|
|
665
|
+
high: "high",
|
|
618
666
|
xhigh: "high"
|
|
619
667
|
};
|
|
668
|
+
var GOOGLE_MEDIA_RESOLUTION = {
|
|
669
|
+
low: "MEDIA_RESOLUTION_LOW",
|
|
670
|
+
high: "MEDIA_RESOLUTION_HIGH"
|
|
671
|
+
};
|
|
672
|
+
function toGeminiUsage(um) {
|
|
673
|
+
if (!um) return void 0;
|
|
674
|
+
const thoughts = um.thoughtsTokenCount ?? 0;
|
|
675
|
+
return {
|
|
676
|
+
inputTokens: um.promptTokenCount ?? 0,
|
|
677
|
+
outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
|
|
678
|
+
...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
|
|
679
|
+
...um.cachedContentTokenCount !== void 0 && {
|
|
680
|
+
cachedInputTokens: um.cachedContentTokenCount
|
|
681
|
+
}
|
|
682
|
+
};
|
|
683
|
+
}
|
|
620
684
|
var GoogleDriver = class {
|
|
621
685
|
client;
|
|
622
686
|
model;
|
|
623
687
|
maxTokens;
|
|
624
688
|
apiKeyOrEnv;
|
|
625
689
|
reasoningEffort;
|
|
690
|
+
imageDetail;
|
|
626
691
|
constructor(config) {
|
|
627
692
|
this.model = config.model;
|
|
628
693
|
this.maxTokens = config.maxTokens;
|
|
629
694
|
this.client = null;
|
|
630
695
|
this.apiKeyOrEnv = config.apiKey;
|
|
631
696
|
this.reasoningEffort = config.reasoningEffort;
|
|
697
|
+
this.imageDetail = config.imageDetail;
|
|
632
698
|
}
|
|
633
699
|
toGeminiParts(images) {
|
|
634
700
|
return images.map((img) => ({
|
|
@@ -668,6 +734,9 @@ var GoogleDriver = class {
|
|
|
668
734
|
thinkingConfig: {
|
|
669
735
|
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
670
736
|
}
|
|
737
|
+
},
|
|
738
|
+
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
739
|
+
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
671
740
|
}
|
|
672
741
|
}
|
|
673
742
|
});
|
|
@@ -685,14 +754,9 @@ var GoogleDriver = class {
|
|
|
685
754
|
);
|
|
686
755
|
}
|
|
687
756
|
const text = response.text ?? "";
|
|
688
|
-
const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
|
|
689
757
|
return {
|
|
690
758
|
text,
|
|
691
|
-
usage: response.usageMetadata
|
|
692
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
693
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
|
|
694
|
-
...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
|
|
695
|
-
} : void 0
|
|
759
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
696
760
|
};
|
|
697
761
|
} catch (err) {
|
|
698
762
|
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
|
|
@@ -723,10 +787,7 @@ var GoogleDriver = class {
|
|
|
723
787
|
return {
|
|
724
788
|
imageData: Buffer.from(imagePart.inlineData.data, "base64"),
|
|
725
789
|
mimeType: imagePart.inlineData.mimeType,
|
|
726
|
-
usage: response.usageMetadata
|
|
727
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
728
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
|
|
729
|
-
} : void 0
|
|
790
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
730
791
|
};
|
|
731
792
|
} catch (err) {
|
|
732
793
|
if (err instanceof VisualAIProviderError) throw err;
|
|
@@ -742,12 +803,14 @@ var OpenAIDriver = class {
|
|
|
742
803
|
maxTokens;
|
|
743
804
|
apiKeyOrEnv;
|
|
744
805
|
reasoningEffort;
|
|
806
|
+
imageDetail;
|
|
745
807
|
constructor(config) {
|
|
746
808
|
this.model = config.model;
|
|
747
809
|
this.maxTokens = config.maxTokens;
|
|
748
810
|
this.client = null;
|
|
749
811
|
this.apiKeyOrEnv = config.apiKey;
|
|
750
812
|
this.reasoningEffort = config.reasoningEffort;
|
|
813
|
+
this.imageDetail = config.imageDetail;
|
|
751
814
|
}
|
|
752
815
|
async getClient() {
|
|
753
816
|
if (this.client) return this.client;
|
|
@@ -769,9 +832,11 @@ var OpenAIDriver = class {
|
|
|
769
832
|
}
|
|
770
833
|
async sendMessage(images, prompt, options) {
|
|
771
834
|
const client = await this.getClient();
|
|
835
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
772
836
|
const imageBlocks = images.map((img) => ({
|
|
773
837
|
type: "input_image",
|
|
774
|
-
image_url: `data:${img.mimeType};base64,${img.base64}
|
|
838
|
+
image_url: `data:${img.mimeType};base64,${img.base64}`,
|
|
839
|
+
...detail ? { detail } : {}
|
|
775
840
|
}));
|
|
776
841
|
try {
|
|
777
842
|
const format = options?.responseSchema ? {
|
|
@@ -796,21 +861,23 @@ var OpenAIDriver = class {
|
|
|
796
861
|
}
|
|
797
862
|
const response = await client.responses.create(requestParams);
|
|
798
863
|
if (response.status && response.status !== "completed") {
|
|
799
|
-
const
|
|
864
|
+
const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
|
|
800
865
|
throw new VisualAITruncationError(
|
|
801
|
-
`Response truncated: OpenAI returned status "${response.status}"${
|
|
866
|
+
`Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
802
867
|
response.output_text ?? "",
|
|
803
868
|
this.maxTokens
|
|
804
869
|
);
|
|
805
870
|
}
|
|
806
871
|
const text = response.output_text ?? "";
|
|
807
872
|
const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
|
|
873
|
+
const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
|
|
808
874
|
return {
|
|
809
875
|
text,
|
|
810
876
|
usage: response.usage ? {
|
|
811
877
|
inputTokens: response.usage.input_tokens,
|
|
812
878
|
outputTokens: response.usage.output_tokens,
|
|
813
|
-
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
879
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
880
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens }
|
|
814
881
|
} : void 0
|
|
815
882
|
};
|
|
816
883
|
} catch (err) {
|
|
@@ -820,6 +887,119 @@ var OpenAIDriver = class {
|
|
|
820
887
|
}
|
|
821
888
|
};
|
|
822
889
|
|
|
890
|
+
// src/providers/openrouter.ts
|
|
891
|
+
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
892
|
+
var OPENROUTER_REASONING_EFFORT = {
|
|
893
|
+
low: "low",
|
|
894
|
+
medium: "medium",
|
|
895
|
+
high: "high",
|
|
896
|
+
xhigh: "high"
|
|
897
|
+
};
|
|
898
|
+
var OpenRouterDriver = class {
|
|
899
|
+
client;
|
|
900
|
+
model;
|
|
901
|
+
maxTokens;
|
|
902
|
+
apiKeyOrEnv;
|
|
903
|
+
reasoningEffort;
|
|
904
|
+
imageDetail;
|
|
905
|
+
constructor(config) {
|
|
906
|
+
this.model = config.model;
|
|
907
|
+
this.maxTokens = config.maxTokens;
|
|
908
|
+
this.client = null;
|
|
909
|
+
this.apiKeyOrEnv = config.apiKey;
|
|
910
|
+
this.reasoningEffort = config.reasoningEffort;
|
|
911
|
+
this.imageDetail = config.imageDetail;
|
|
912
|
+
}
|
|
913
|
+
async getClient() {
|
|
914
|
+
if (this.client) return this.client;
|
|
915
|
+
let OpenAI;
|
|
916
|
+
try {
|
|
917
|
+
const mod = await import("openai");
|
|
918
|
+
OpenAI = mod.default;
|
|
919
|
+
} catch {
|
|
920
|
+
throw new VisualAIConfigError(
|
|
921
|
+
"OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
|
|
922
|
+
);
|
|
923
|
+
}
|
|
924
|
+
const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
|
|
925
|
+
if (!apiKey) {
|
|
926
|
+
throw new VisualAIAuthError(
|
|
927
|
+
"OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
|
|
928
|
+
);
|
|
929
|
+
}
|
|
930
|
+
this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
|
|
931
|
+
return this.client;
|
|
932
|
+
}
|
|
933
|
+
async sendMessage(images, prompt, options) {
|
|
934
|
+
const client = await this.getClient();
|
|
935
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
936
|
+
const imageParts = images.map((img) => ({
|
|
937
|
+
type: "image_url",
|
|
938
|
+
image_url: {
|
|
939
|
+
url: `data:${img.mimeType};base64,${img.base64}`,
|
|
940
|
+
...detail ? { detail } : {}
|
|
941
|
+
}
|
|
942
|
+
}));
|
|
943
|
+
try {
|
|
944
|
+
const responseFormat = options?.responseSchema ? {
|
|
945
|
+
type: "json_schema",
|
|
946
|
+
json_schema: {
|
|
947
|
+
name: "visual_ai_response",
|
|
948
|
+
strict: true,
|
|
949
|
+
schema: options.responseSchema
|
|
950
|
+
}
|
|
951
|
+
} : { type: "json_object" };
|
|
952
|
+
const requestParams = {
|
|
953
|
+
model: this.model,
|
|
954
|
+
max_tokens: this.maxTokens,
|
|
955
|
+
response_format: responseFormat,
|
|
956
|
+
messages: [
|
|
957
|
+
{
|
|
958
|
+
role: "user",
|
|
959
|
+
content: [...imageParts, { type: "text", text: prompt }]
|
|
960
|
+
}
|
|
961
|
+
],
|
|
962
|
+
// OpenRouter-specific: include token accounting in the response.
|
|
963
|
+
usage: { include: true }
|
|
964
|
+
};
|
|
965
|
+
if (this.reasoningEffort) {
|
|
966
|
+
requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
|
|
967
|
+
}
|
|
968
|
+
const response = await client.chat.completions.create(requestParams);
|
|
969
|
+
const choice = response.choices?.[0];
|
|
970
|
+
if (!choice?.message) {
|
|
971
|
+
throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
|
|
972
|
+
}
|
|
973
|
+
const text = choice.message.content ?? "";
|
|
974
|
+
if (choice.finish_reason === "length") {
|
|
975
|
+
throw new VisualAITruncationError(
|
|
976
|
+
`Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
977
|
+
text,
|
|
978
|
+
this.maxTokens
|
|
979
|
+
);
|
|
980
|
+
}
|
|
981
|
+
const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
|
|
982
|
+
const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
|
|
983
|
+
const cost = response.usage?.cost;
|
|
984
|
+
return {
|
|
985
|
+
text,
|
|
986
|
+
usage: response.usage ? {
|
|
987
|
+
inputTokens: response.usage.prompt_tokens,
|
|
988
|
+
outputTokens: response.usage.completion_tokens,
|
|
989
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
990
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens },
|
|
991
|
+
...cost !== void 0 && { cost }
|
|
992
|
+
} : void 0
|
|
993
|
+
};
|
|
994
|
+
} catch (err) {
|
|
995
|
+
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
|
|
996
|
+
throw err;
|
|
997
|
+
}
|
|
998
|
+
throw mapProviderError(err);
|
|
999
|
+
}
|
|
1000
|
+
}
|
|
1001
|
+
};
|
|
1002
|
+
|
|
823
1003
|
// src/core/config.ts
|
|
824
1004
|
var MODEL_PREFIX_TO_PROVIDER = [
|
|
825
1005
|
["claude-", "anthropic"],
|
|
@@ -832,6 +1012,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
|
|
|
832
1012
|
function inferProviderFromModel(model) {
|
|
833
1013
|
const known = MODEL_TO_PROVIDER.get(model);
|
|
834
1014
|
if (known) return known;
|
|
1015
|
+
if (model.includes("/")) return "openrouter";
|
|
835
1016
|
const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
|
|
836
1017
|
return prefixMatch?.[1];
|
|
837
1018
|
}
|
|
@@ -844,12 +1025,13 @@ function resolveProvider(config) {
|
|
|
844
1025
|
const apiKeyProviderMap = [
|
|
845
1026
|
["ANTHROPIC_API_KEY", "anthropic"],
|
|
846
1027
|
["OPENAI_API_KEY", "openai"],
|
|
847
|
-
["GOOGLE_API_KEY", "google"]
|
|
1028
|
+
["GOOGLE_API_KEY", "google"],
|
|
1029
|
+
["OPENROUTER_API_KEY", "openrouter"]
|
|
848
1030
|
];
|
|
849
1031
|
const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
|
|
850
1032
|
if (detected) return detected[1];
|
|
851
1033
|
throw new VisualAIConfigError(
|
|
852
|
-
"Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
|
|
1034
|
+
"Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
|
|
853
1035
|
);
|
|
854
1036
|
}
|
|
855
1037
|
function parseBooleanEnv(envName, value) {
|
|
@@ -877,11 +1059,11 @@ function resolveConfig(config) {
|
|
|
877
1059
|
}
|
|
878
1060
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
879
1061
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
880
|
-
if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
|
|
1062
|
+
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
|
|
881
1063
|
maxTokens = OPENAI_REASONING_MAX_TOKENS;
|
|
882
1064
|
if (debug) {
|
|
883
1065
|
process.stderr.write(
|
|
884
|
-
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for
|
|
1066
|
+
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
|
|
885
1067
|
`
|
|
886
1068
|
);
|
|
887
1069
|
}
|
|
@@ -892,6 +1074,8 @@ function resolveConfig(config) {
|
|
|
892
1074
|
model,
|
|
893
1075
|
maxTokens,
|
|
894
1076
|
reasoningEffort: config.reasoningEffort,
|
|
1077
|
+
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1078
|
+
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
895
1079
|
debug,
|
|
896
1080
|
debugPrompt,
|
|
897
1081
|
debugResponse,
|
|
@@ -906,6 +1090,10 @@ var PRICING_TABLE = {
|
|
|
906
1090
|
inputPricePerToken: 10 / PER_MILLION,
|
|
907
1091
|
outputPricePerToken: 50 / PER_MILLION
|
|
908
1092
|
},
|
|
1093
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1094
|
+
inputPricePerToken: 5 / PER_MILLION,
|
|
1095
|
+
outputPricePerToken: 25 / PER_MILLION
|
|
1096
|
+
},
|
|
909
1097
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
|
|
910
1098
|
inputPricePerToken: 5 / PER_MILLION,
|
|
911
1099
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -935,12 +1123,12 @@ var PRICING_TABLE = {
|
|
|
935
1123
|
outputPricePerToken: 30 / PER_MILLION
|
|
936
1124
|
},
|
|
937
1125
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
|
|
938
|
-
inputPricePerToken: 2
|
|
939
|
-
outputPricePerToken:
|
|
1126
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1127
|
+
outputPricePerToken: 12 / PER_MILLION
|
|
940
1128
|
},
|
|
941
1129
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
|
|
942
|
-
inputPricePerToken:
|
|
943
|
-
outputPricePerToken:
|
|
1130
|
+
inputPricePerToken: 0.2 / PER_MILLION,
|
|
1131
|
+
outputPricePerToken: 1.2 / PER_MILLION
|
|
944
1132
|
},
|
|
945
1133
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
|
|
946
1134
|
inputPricePerToken: 5 / PER_MILLION,
|
|
@@ -970,10 +1158,30 @@ var PRICING_TABLE = {
|
|
|
970
1158
|
inputPricePerToken: 0.25 / PER_MILLION,
|
|
971
1159
|
outputPricePerToken: 2 / PER_MILLION
|
|
972
1160
|
},
|
|
1161
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1162
|
+
// on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
|
|
1163
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
|
|
1164
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1165
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1166
|
+
},
|
|
1167
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1168
|
+
// on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
|
|
1169
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
|
|
1170
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1171
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1172
|
+
},
|
|
1173
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
|
|
1174
|
+
inputPricePerToken: 1.5 / PER_MILLION,
|
|
1175
|
+
outputPricePerToken: 7.5 / PER_MILLION
|
|
1176
|
+
},
|
|
973
1177
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
|
|
974
1178
|
inputPricePerToken: 1.5 / PER_MILLION,
|
|
975
1179
|
outputPricePerToken: 9 / PER_MILLION
|
|
976
1180
|
},
|
|
1181
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
|
|
1182
|
+
inputPricePerToken: 0.3 / PER_MILLION,
|
|
1183
|
+
outputPricePerToken: 2.5 / PER_MILLION
|
|
1184
|
+
},
|
|
977
1185
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
|
|
978
1186
|
inputPricePerToken: 2 / PER_MILLION,
|
|
979
1187
|
outputPricePerToken: 12 / PER_MILLION
|
|
@@ -985,6 +1193,36 @@ var PRICING_TABLE = {
|
|
|
985
1193
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
|
|
986
1194
|
inputPricePerToken: 0.5 / PER_MILLION,
|
|
987
1195
|
outputPricePerToken: 3 / PER_MILLION
|
|
1196
|
+
},
|
|
1197
|
+
// OpenRouter passes through upstream per-model pricing (verified 2026-07-22
|
|
1198
|
+
// against https://openrouter.ai/api/v1/models).
|
|
1199
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1200
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1201
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1202
|
+
},
|
|
1203
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
|
|
1204
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1205
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1206
|
+
},
|
|
1207
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
|
|
1208
|
+
inputPricePerToken: 3 / PER_MILLION,
|
|
1209
|
+
outputPricePerToken: 15 / PER_MILLION
|
|
1210
|
+
},
|
|
1211
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
|
|
1212
|
+
inputPricePerToken: 0.82 / PER_MILLION,
|
|
1213
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1214
|
+
},
|
|
1215
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
|
|
1216
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1217
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1218
|
+
},
|
|
1219
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
|
|
1220
|
+
inputPricePerToken: 0.32 / PER_MILLION,
|
|
1221
|
+
outputPricePerToken: 1.28 / PER_MILLION
|
|
1222
|
+
},
|
|
1223
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
|
|
1224
|
+
inputPricePerToken: 0.1875 / PER_MILLION,
|
|
1225
|
+
outputPricePerToken: 1.125 / PER_MILLION
|
|
988
1226
|
}
|
|
989
1227
|
};
|
|
990
1228
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -1007,8 +1245,9 @@ function usageLog(config, method, usage) {
|
|
|
1007
1245
|
const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
|
|
1008
1246
|
const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
|
|
1009
1247
|
const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
|
|
1248
|
+
const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
|
|
1010
1249
|
process.stderr.write(
|
|
1011
|
-
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1250
|
+
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1012
1251
|
`
|
|
1013
1252
|
);
|
|
1014
1253
|
}
|
|
@@ -1019,7 +1258,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
|
|
|
1019
1258
|
inputTokens,
|
|
1020
1259
|
outputTokens,
|
|
1021
1260
|
...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
|
|
1261
|
+
...rawUsage?.cachedInputTokens !== void 0 && {
|
|
1262
|
+
cachedInputTokens: rawUsage.cachedInputTokens
|
|
1263
|
+
},
|
|
1022
1264
|
estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
|
|
1265
|
+
...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
|
|
1023
1266
|
durationSeconds
|
|
1024
1267
|
};
|
|
1025
1268
|
usageLog(config, method, usage);
|
|
@@ -1062,7 +1305,10 @@ async function timedSendMessage(driver, images, prompt, options) {
|
|
|
1062
1305
|
import sharp from "sharp";
|
|
1063
1306
|
var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
|
|
1064
1307
|
Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
1065
|
-
Model.Google.GEMINI_3_5_FLASH
|
|
1308
|
+
Model.Google.GEMINI_3_5_FLASH,
|
|
1309
|
+
Model.Google.GEMINI_3_6_FLASH,
|
|
1310
|
+
Model.Google.GEMINI_3_7_FLASH,
|
|
1311
|
+
Model.Google.GEMINI_3_8_FLASH
|
|
1066
1312
|
]);
|
|
1067
1313
|
async function generateAiDiff(imgA, imgB, model, driver) {
|
|
1068
1314
|
if (!driver.generateImage) {
|
|
@@ -1152,7 +1398,6 @@ var EXTENSION_TO_MIME = {
|
|
|
1152
1398
|
".webp": "image/webp",
|
|
1153
1399
|
".gif": "image/gif"
|
|
1154
1400
|
};
|
|
1155
|
-
var MAX_DIMENSION = 1568;
|
|
1156
1401
|
var URL_FETCH_TIMEOUT_MS = 1e4;
|
|
1157
1402
|
function isSupportedMimeType(value) {
|
|
1158
1403
|
return SUPPORTED_FORMATS.has(value);
|
|
@@ -1176,14 +1421,14 @@ function detectMimeType(data) {
|
|
|
1176
1421
|
}
|
|
1177
1422
|
throw new VisualAIImageError("Unable to detect image format from file content");
|
|
1178
1423
|
}
|
|
1179
|
-
async function resizeIfNeeded(data, mimeType) {
|
|
1424
|
+
async function resizeIfNeeded(data, mimeType, maxDimension) {
|
|
1180
1425
|
if (mimeType === "image/gif") {
|
|
1181
1426
|
return data;
|
|
1182
1427
|
}
|
|
1183
1428
|
if (mimeType === "image/png" && data.length >= 24) {
|
|
1184
1429
|
const width2 = data.readUInt32BE(16);
|
|
1185
1430
|
const height2 = data.readUInt32BE(20);
|
|
1186
|
-
if (width2 <=
|
|
1431
|
+
if (width2 <= maxDimension && height2 <= maxDimension) {
|
|
1187
1432
|
return data;
|
|
1188
1433
|
}
|
|
1189
1434
|
}
|
|
@@ -1191,12 +1436,12 @@ async function resizeIfNeeded(data, mimeType) {
|
|
|
1191
1436
|
const metadata = await pipeline.metadata();
|
|
1192
1437
|
const width = metadata.width ?? 0;
|
|
1193
1438
|
const height = metadata.height ?? 0;
|
|
1194
|
-
if (width <=
|
|
1439
|
+
if (width <= maxDimension && height <= maxDimension) {
|
|
1195
1440
|
return data;
|
|
1196
1441
|
}
|
|
1197
1442
|
return pipeline.resize({
|
|
1198
|
-
width:
|
|
1199
|
-
height:
|
|
1443
|
+
width: maxDimension,
|
|
1444
|
+
height: maxDimension,
|
|
1200
1445
|
fit: "inside",
|
|
1201
1446
|
withoutEnlargement: true
|
|
1202
1447
|
}).toBuffer();
|
|
@@ -1251,7 +1496,7 @@ function loadFromBase64(input) {
|
|
|
1251
1496
|
}
|
|
1252
1497
|
return { data, mimeType: mimeType ?? detectMimeType(data) };
|
|
1253
1498
|
}
|
|
1254
|
-
async function normalizeImage(input) {
|
|
1499
|
+
async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1255
1500
|
let data;
|
|
1256
1501
|
let mimeType;
|
|
1257
1502
|
if (Buffer.isBuffer(input)) {
|
|
@@ -1280,7 +1525,7 @@ async function normalizeImage(input) {
|
|
|
1280
1525
|
"Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
|
|
1281
1526
|
);
|
|
1282
1527
|
}
|
|
1283
|
-
data = await resizeIfNeeded(data, mimeType);
|
|
1528
|
+
data = await resizeIfNeeded(data, mimeType, maxDimension);
|
|
1284
1529
|
let cachedBase64;
|
|
1285
1530
|
return {
|
|
1286
1531
|
data,
|
|
@@ -1584,7 +1829,7 @@ async function probeDurationSeconds(videoPath) {
|
|
|
1584
1829
|
});
|
|
1585
1830
|
});
|
|
1586
1831
|
}
|
|
1587
|
-
async function extractFrames(videoPath, options = {}) {
|
|
1832
|
+
async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
|
|
1588
1833
|
const fps = options.fps ?? DEFAULT_FPS;
|
|
1589
1834
|
const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
|
|
1590
1835
|
const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
|
|
@@ -1613,7 +1858,7 @@ async function extractFrames(videoPath, options = {}) {
|
|
|
1613
1858
|
}
|
|
1614
1859
|
const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
|
|
1615
1860
|
try {
|
|
1616
|
-
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${
|
|
1861
|
+
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
|
|
1617
1862
|
await new Promise((resolve2, reject) => {
|
|
1618
1863
|
let settled = false;
|
|
1619
1864
|
const cmd = ffmpeg(videoPath);
|
|
@@ -1711,7 +1956,7 @@ function isFramesInput(input) {
|
|
|
1711
1956
|
function isTimestampedFrameInput(frame) {
|
|
1712
1957
|
return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
|
|
1713
1958
|
}
|
|
1714
|
-
async function normalizeFrames(input) {
|
|
1959
|
+
async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1715
1960
|
const rawFrames = input.frames;
|
|
1716
1961
|
const fps = input.fps ?? DEFAULT_FPS;
|
|
1717
1962
|
if (rawFrames.length === 0) {
|
|
@@ -1736,7 +1981,7 @@ async function normalizeFrames(input) {
|
|
|
1736
1981
|
`Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
|
|
1737
1982
|
);
|
|
1738
1983
|
}
|
|
1739
|
-
const image = await normalizeImage(imageInput);
|
|
1984
|
+
const image = await normalizeImage(imageInput, maxDimension);
|
|
1740
1985
|
return {
|
|
1741
1986
|
data: image.data,
|
|
1742
1987
|
mimeType: image.mimeType,
|
|
@@ -1752,14 +1997,14 @@ async function normalizeFrames(input) {
|
|
|
1752
1997
|
await saveDebugFrames(frames);
|
|
1753
1998
|
return { kind: "video", frames, durationSeconds };
|
|
1754
1999
|
}
|
|
1755
|
-
async function normalizeMedia(input, videoOptions) {
|
|
2000
|
+
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1756
2001
|
if (isFramesInput(input)) {
|
|
1757
|
-
return normalizeFrames(input);
|
|
2002
|
+
return normalizeFrames(input, maxDimension);
|
|
1758
2003
|
}
|
|
1759
2004
|
if (isVideoInput(input)) {
|
|
1760
2005
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
1761
2006
|
try {
|
|
1762
|
-
const { frames, durationSeconds } = await extractFrames(path, videoOptions);
|
|
2007
|
+
const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
|
|
1763
2008
|
await saveDebugFrames(frames);
|
|
1764
2009
|
return { kind: "video", frames, durationSeconds };
|
|
1765
2010
|
} finally {
|
|
@@ -1769,7 +2014,7 @@ async function normalizeMedia(input, videoOptions) {
|
|
|
1769
2014
|
}
|
|
1770
2015
|
}
|
|
1771
2016
|
}
|
|
1772
|
-
const image = await normalizeImage(input);
|
|
2017
|
+
const image = await normalizeImage(input, maxDimension);
|
|
1773
2018
|
return { kind: "image", image };
|
|
1774
2019
|
}
|
|
1775
2020
|
|
|
@@ -1810,7 +2055,17 @@ var UsageInfoSchema = z.object({
|
|
|
1810
2055
|
outputTokens: z.number(),
|
|
1811
2056
|
/** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
|
|
1812
2057
|
reasoningTokens: z.number().optional(),
|
|
2058
|
+
/**
|
|
2059
|
+
* Prompt tokens served from the provider's cache, when reported. Informational
|
|
2060
|
+
* only — `estimatedCost` does not apply a cache discount, because providers
|
|
2061
|
+
* differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
|
|
2062
|
+
* Google) or billed as a separate bucket alongside it (Anthropic).
|
|
2063
|
+
*/
|
|
2064
|
+
cachedInputTokens: z.number().optional(),
|
|
2065
|
+
/** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
|
|
1813
2066
|
estimatedCost: z.number().optional(),
|
|
2067
|
+
/** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
|
|
2068
|
+
reportedCost: z.number().optional(),
|
|
1814
2069
|
durationSeconds: z.number().nonnegative().optional()
|
|
1815
2070
|
});
|
|
1816
2071
|
var BaseResultSchema = z.object({
|
|
@@ -1912,7 +2167,8 @@ function toSchemaOptions(schema) {
|
|
|
1912
2167
|
var PROVIDER_REGISTRY = {
|
|
1913
2168
|
anthropic: (config) => new AnthropicDriver(config),
|
|
1914
2169
|
openai: (config) => new OpenAIDriver(config),
|
|
1915
|
-
google: (config) => new GoogleDriver(config)
|
|
2170
|
+
google: (config) => new GoogleDriver(config),
|
|
2171
|
+
openrouter: (config) => new OpenRouterDriver(config)
|
|
1916
2172
|
};
|
|
1917
2173
|
function createDriver(provider, config) {
|
|
1918
2174
|
return PROVIDER_REGISTRY[provider](config);
|
|
@@ -1949,16 +2205,18 @@ function visualAI(config = {}) {
|
|
|
1949
2205
|
apiKey: resolvedConfig.apiKey,
|
|
1950
2206
|
model: resolvedConfig.model,
|
|
1951
2207
|
maxTokens: resolvedConfig.maxTokens,
|
|
1952
|
-
reasoningEffort: resolvedConfig.reasoningEffort
|
|
2208
|
+
reasoningEffort: resolvedConfig.reasoningEffort,
|
|
2209
|
+
imageDetail: resolvedConfig.imageDetail
|
|
1953
2210
|
};
|
|
1954
2211
|
const driver = createDriver(resolvedConfig.provider, driverConfig);
|
|
2212
|
+
const maxImageDimension = resolvedConfig.maxImageDimension;
|
|
1955
2213
|
async function checkElementsVisibility(image, elements, visible, options) {
|
|
1956
2214
|
const methodName = visible ? "elementsVisible" : "elementsHidden";
|
|
1957
2215
|
if (elements.length === 0) {
|
|
1958
2216
|
throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
|
|
1959
2217
|
}
|
|
1960
2218
|
return withErrorDebug(resolvedConfig, methodName, async () => {
|
|
1961
|
-
const img = await normalizeImage(image);
|
|
2219
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
1962
2220
|
const prompt = buildElementsVisibilityPrompt(elements, visible, options);
|
|
1963
2221
|
debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
|
|
1964
2222
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -1977,7 +2235,7 @@ function visualAI(config = {}) {
|
|
|
1977
2235
|
throw new VisualAIConfigError("At least one statement is required for check()");
|
|
1978
2236
|
}
|
|
1979
2237
|
return withErrorDebug(resolvedConfig, "check", async () => {
|
|
1980
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2238
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
1981
2239
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
1982
2240
|
const prompt = buildCheckPrompt(stmts, {
|
|
1983
2241
|
instructions: options?.instructions,
|
|
@@ -1996,7 +2254,7 @@ function visualAI(config = {}) {
|
|
|
1996
2254
|
},
|
|
1997
2255
|
async ask(input, userPrompt, options) {
|
|
1998
2256
|
return withErrorDebug(resolvedConfig, "ask", async () => {
|
|
1999
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2257
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2000
2258
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2001
2259
|
const prompt = buildAskPrompt(userPrompt, {
|
|
2002
2260
|
instructions: options?.instructions,
|
|
@@ -2015,7 +2273,10 @@ function visualAI(config = {}) {
|
|
|
2015
2273
|
},
|
|
2016
2274
|
async compare(imageA, imageB, options) {
|
|
2017
2275
|
return withErrorDebug(resolvedConfig, "compare", async () => {
|
|
2018
|
-
const [imgA, imgB] = await Promise.all([
|
|
2276
|
+
const [imgA, imgB] = await Promise.all([
|
|
2277
|
+
normalizeImage(imageA, maxImageDimension),
|
|
2278
|
+
normalizeImage(imageB, maxImageDimension)
|
|
2279
|
+
]);
|
|
2019
2280
|
const prompt = buildComparePrompt({
|
|
2020
2281
|
userPrompt: options?.prompt,
|
|
2021
2282
|
instructions: options?.instructions
|
|
@@ -2053,7 +2314,7 @@ function visualAI(config = {}) {
|
|
|
2053
2314
|
},
|
|
2054
2315
|
async accessibility(image, options) {
|
|
2055
2316
|
return withErrorDebug(resolvedConfig, "accessibility", async () => {
|
|
2056
|
-
const img = await normalizeImage(image);
|
|
2317
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2057
2318
|
const prompt = buildAccessibilityPrompt(options);
|
|
2058
2319
|
debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
|
|
2059
2320
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2072,7 +2333,7 @@ function visualAI(config = {}) {
|
|
|
2072
2333
|
},
|
|
2073
2334
|
async layout(image, options) {
|
|
2074
2335
|
return withErrorDebug(resolvedConfig, "layout", async () => {
|
|
2075
|
-
const img = await normalizeImage(image);
|
|
2336
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2076
2337
|
const prompt = buildLayoutPrompt(options);
|
|
2077
2338
|
debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
|
|
2078
2339
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2086,7 +2347,7 @@ function visualAI(config = {}) {
|
|
|
2086
2347
|
},
|
|
2087
2348
|
async pageLoad(image, options) {
|
|
2088
2349
|
return withErrorDebug(resolvedConfig, "pageLoad", async () => {
|
|
2089
|
-
const img = await normalizeImage(image);
|
|
2350
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2090
2351
|
const prompt = buildPageLoadPrompt(options);
|
|
2091
2352
|
debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
|
|
2092
2353
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2100,7 +2361,7 @@ function visualAI(config = {}) {
|
|
|
2100
2361
|
},
|
|
2101
2362
|
async content(image, options) {
|
|
2102
2363
|
return withErrorDebug(resolvedConfig, "content", async () => {
|
|
2103
|
-
const img = await normalizeImage(image);
|
|
2364
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2104
2365
|
const prompt = buildContentPrompt(options);
|
|
2105
2366
|
debugLog(resolvedConfig, "content prompt", prompt, "prompt");
|
|
2106
2367
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2178,6 +2439,7 @@ export {
|
|
|
2178
2439
|
ConfidenceSchema,
|
|
2179
2440
|
Content,
|
|
2180
2441
|
DEFAULT_MODELS,
|
|
2442
|
+
ImageDetail,
|
|
2181
2443
|
IssueCategorySchema,
|
|
2182
2444
|
IssuePrioritySchema,
|
|
2183
2445
|
IssueSchema,
|